70#define DEBUG_TYPE "openmp-ir-builder"
77 cl::desc(
"Use optimistic attributes describing "
78 "'as-if' properties of runtime calls."),
82 "openmp-ir-builder-unroll-threshold-factor",
cl::Hidden,
83 cl::desc(
"Factor for the unroll threshold to account for code "
84 "simplifications still taking place"),
88 "openmp-ir-builder-use-default-max-threads",
cl::Hidden,
99 if (!IP1.isSet() || !IP2.isSet())
101 return IP1.getBlock() == IP2.getBlock() && IP1.getPoint() == IP2.getPoint();
106 switch (SchedType & ~OMPScheduleType::MonotonicityMask) {
107 case OMPScheduleType::UnorderedStaticChunked:
108 case OMPScheduleType::UnorderedStatic:
109 case OMPScheduleType::UnorderedDynamicChunked:
110 case OMPScheduleType::UnorderedGuidedChunked:
111 case OMPScheduleType::UnorderedRuntime:
112 case OMPScheduleType::UnorderedAuto:
113 case OMPScheduleType::UnorderedTrapezoidal:
114 case OMPScheduleType::UnorderedGreedy:
115 case OMPScheduleType::UnorderedBalanced:
116 case OMPScheduleType::UnorderedGuidedIterativeChunked:
117 case OMPScheduleType::UnorderedGuidedAnalyticalChunked:
118 case OMPScheduleType::UnorderedSteal:
119 case OMPScheduleType::UnorderedStaticBalancedChunked:
120 case OMPScheduleType::UnorderedGuidedSimd:
121 case OMPScheduleType::UnorderedRuntimeSimd:
122 case OMPScheduleType::OrderedStaticChunked:
123 case OMPScheduleType::OrderedStatic:
124 case OMPScheduleType::OrderedDynamicChunked:
125 case OMPScheduleType::OrderedGuidedChunked:
126 case OMPScheduleType::OrderedRuntime:
127 case OMPScheduleType::OrderedAuto:
128 case OMPScheduleType::OrderdTrapezoidal:
129 case OMPScheduleType::NomergeUnorderedStaticChunked:
130 case OMPScheduleType::NomergeUnorderedStatic:
131 case OMPScheduleType::NomergeUnorderedDynamicChunked:
132 case OMPScheduleType::NomergeUnorderedGuidedChunked:
133 case OMPScheduleType::NomergeUnorderedRuntime:
134 case OMPScheduleType::NomergeUnorderedAuto:
135 case OMPScheduleType::NomergeUnorderedTrapezoidal:
136 case OMPScheduleType::NomergeUnorderedGreedy:
137 case OMPScheduleType::NomergeUnorderedBalanced:
138 case OMPScheduleType::NomergeUnorderedGuidedIterativeChunked:
139 case OMPScheduleType::NomergeUnorderedGuidedAnalyticalChunked:
140 case OMPScheduleType::NomergeUnorderedSteal:
141 case OMPScheduleType::NomergeOrderedStaticChunked:
142 case OMPScheduleType::NomergeOrderedStatic:
143 case OMPScheduleType::NomergeOrderedDynamicChunked:
144 case OMPScheduleType::NomergeOrderedGuidedChunked:
145 case OMPScheduleType::NomergeOrderedRuntime:
146 case OMPScheduleType::NomergeOrderedAuto:
147 case OMPScheduleType::NomergeOrderedTrapezoidal:
148 case OMPScheduleType::OrderedDistributeChunked:
149 case OMPScheduleType::OrderedDistribute:
157 SchedType & OMPScheduleType::MonotonicityMask;
158 if (MonotonicityFlags == OMPScheduleType::MonotonicityMask)
172 Builder.restoreIP(IP);
176 if (Builder.GetInsertPoint() != BB->
end())
186 unsigned Line = FSP->getScopeLine() ? FSP->getScopeLine() : FSP->getLine();
187 Builder.SetCurrentDebugLocation(
193 return T.isAMDGPU() ||
T.isNVPTX() ||
T.isSPIRV();
199 Kernel->getFnAttribute(
"target-features").getValueAsString();
200 if (Features.
count(
"+wavefrontsize64"))
215 bool HasSimdModifier,
bool HasDistScheduleChunks) {
217 switch (ClauseKind) {
218 case OMP_SCHEDULE_Default:
219 case OMP_SCHEDULE_Static:
220 return HasChunks ? OMPScheduleType::BaseStaticChunked
221 : OMPScheduleType::BaseStatic;
222 case OMP_SCHEDULE_Dynamic:
223 return OMPScheduleType::BaseDynamicChunked;
224 case OMP_SCHEDULE_Guided:
225 return HasSimdModifier ? OMPScheduleType::BaseGuidedSimd
226 : OMPScheduleType::BaseGuidedChunked;
227 case OMP_SCHEDULE_Auto:
229 case OMP_SCHEDULE_Runtime:
230 return HasSimdModifier ? OMPScheduleType::BaseRuntimeSimd
231 : OMPScheduleType::BaseRuntime;
232 case OMP_SCHEDULE_Distribute:
233 return HasDistScheduleChunks ? OMPScheduleType::BaseDistributeChunked
234 : OMPScheduleType::BaseDistribute;
242 bool HasOrderedClause) {
243 assert((BaseScheduleType & OMPScheduleType::ModifierMask) ==
244 OMPScheduleType::None &&
245 "Must not have ordering nor monotonicity flags already set");
248 ? OMPScheduleType::ModifierOrdered
249 : OMPScheduleType::ModifierUnordered;
250 OMPScheduleType OrderingScheduleType = BaseScheduleType | OrderingModifier;
253 if (OrderingScheduleType ==
254 (OMPScheduleType::BaseGuidedSimd | OMPScheduleType::ModifierOrdered))
255 return OMPScheduleType::OrderedGuidedChunked;
256 else if (OrderingScheduleType == (OMPScheduleType::BaseRuntimeSimd |
257 OMPScheduleType::ModifierOrdered))
258 return OMPScheduleType::OrderedRuntime;
260 return OrderingScheduleType;
266 bool HasSimdModifier,
bool HasMonotonic,
267 bool HasNonmonotonic,
bool HasOrderedClause) {
268 assert((ScheduleType & OMPScheduleType::MonotonicityMask) ==
269 OMPScheduleType::None &&
270 "Must not have monotonicity flags already set");
271 assert((!HasMonotonic || !HasNonmonotonic) &&
272 "Monotonic and Nonmonotonic are contradicting each other");
275 return ScheduleType | OMPScheduleType::ModifierMonotonic;
276 }
else if (HasNonmonotonic) {
277 return ScheduleType | OMPScheduleType::ModifierNonmonotonic;
287 if ((BaseScheduleType == OMPScheduleType::BaseStatic) ||
288 (BaseScheduleType == OMPScheduleType::BaseStaticChunked) ||
294 return ScheduleType | OMPScheduleType::ModifierNonmonotonic;
302 bool HasSimdModifier,
bool HasMonotonicModifier,
303 bool HasNonmonotonicModifier,
bool HasOrderedClause,
304 bool HasDistScheduleChunks) {
306 ClauseKind, HasChunks, HasSimdModifier, HasDistScheduleChunks);
310 OrderedSchedule, HasSimdModifier, HasMonotonicModifier,
311 HasNonmonotonicModifier, HasOrderedClause);
319static std::optional<omp::OMPTgtExecModeFlags>
324 if (
Call->getCalledFunction()->getName() ==
"__kmpc_target_init") {
325 TargetInitCall =
Call;
350 std::optional<omp::OMPTgtExecModeFlags> ExecMode =
362 if (
Instruction *Term = Source->getTerminatorOrNull()) {
371 NewBr->setDebugLoc(
DL);
376 assert(New->getFirstInsertionPt() == New->begin() &&
377 "Target BB must not have PHI nodes");
393 New->splice(New->begin(), Old, IP.
getPoint(), Old->
end());
397 NewBr->setDebugLoc(
DL);
409 Builder.SetInsertPoint(Old);
413 Builder.SetCurrentDebugLocation(
DebugLoc);
423 New->replaceSuccessorsPhiUsesWith(Old, New);
432 Builder.SetInsertPoint(Builder.GetInsertBlock()->getTerminator());
434 Builder.SetInsertPoint(Builder.GetInsertBlock());
437 Builder.SetCurrentDebugLocation(
DebugLoc);
446 Builder.SetInsertPoint(Builder.GetInsertBlock()->getTerminator());
448 Builder.SetInsertPoint(Builder.GetInsertBlock());
451 Builder.SetCurrentDebugLocation(
DebugLoc);
468 const Twine &Name =
"",
bool AsPtr =
true,
469 bool Is64Bit =
false) {
470 Builder.restoreIP(OuterAllocaIP);
474 Builder.CreateAlloca(IntTy,
nullptr, Name +
".addr");
478 FakeVal = FakeValAddr;
480 FakeVal = Builder.CreateLoad(IntTy, FakeValAddr, Name +
".val");
485 Builder.restoreIP(InnerAllocaIP);
488 UseFakeVal = Builder.CreateLoad(IntTy, FakeVal, Name +
".use");
491 FakeVal, Is64Bit ? Builder.getInt64(10) : Builder.getInt32(10)));
504enum OpenMPOffloadingRequiresDirFlags {
506 OMP_REQ_UNDEFINED = 0x000,
508 OMP_REQ_NONE = 0x001,
510 OMP_REQ_REVERSE_OFFLOAD = 0x002,
512 OMP_REQ_UNIFIED_ADDRESS = 0x004,
514 OMP_REQ_UNIFIED_SHARED_MEMORY = 0x008,
516 OMP_REQ_DYNAMIC_ALLOCATORS = 0x010,
523 DominatorTree *DT =
nullptr,
bool AggregateArgs =
false,
524 BlockFrequencyInfo *BFI =
nullptr,
525 BranchProbabilityInfo *BPI =
nullptr,
526 AssumptionCache *AC =
nullptr,
bool AllowVarArgs =
false,
527 bool AllowAlloca =
false,
528 BasicBlock *AllocationBlock =
nullptr,
530 std::string Suffix =
"",
bool ArgsInZeroAddressSpace =
false)
531 : CodeExtractor(BBs, DT, AggregateArgs, BFI, BPI, AC, AllowVarArgs,
532 AllowAlloca, AllocationBlock, DeallocationBlocks, Suffix,
533 ArgsInZeroAddressSpace),
534 OMPBuilder(OMPBuilder) {}
536 virtual ~OMPCodeExtractor() =
default;
539 OpenMPIRBuilder &OMPBuilder;
542class DeviceSharedMemCodeExtractor :
public OMPCodeExtractor {
544 using OMPCodeExtractor::OMPCodeExtractor;
545 virtual ~DeviceSharedMemCodeExtractor() =
default;
549 allocateVar(IRBuilder<>::InsertPoint AllocaIP,
Type *VarType,
550 const Twine &Name = Twine(
""),
551 AddrSpaceCastInst **CastedAlloc =
nullptr)
override {
552 return OMPBuilder.createOMPAllocShared(AllocaIP, VarType, Name);
555 virtual Instruction *deallocateVar(IRBuilder<>::InsertPoint DeallocIP,
557 return OMPBuilder.createOMPFreeShared(DeallocIP, Var, VarType);
564 OpenMPIRBuilder &OMPBuilder;
566 DeviceSharedMemOutlineInfo(OpenMPIRBuilder &OMPBuilder)
567 : OMPBuilder(OMPBuilder) {}
568 virtual ~DeviceSharedMemOutlineInfo() =
default;
570 virtual std::unique_ptr<CodeExtractor>
572 bool ArgsInZeroAddressSpace,
573 Twine Suffix = Twine(
""))
override;
579 : RequiresFlags(OMP_REQ_UNDEFINED) {}
583 bool HasRequiresReverseOffload,
bool HasRequiresUnifiedAddress,
584 bool HasRequiresUnifiedSharedMemory,
bool HasRequiresDynamicAllocators)
587 RequiresFlags(OMP_REQ_UNDEFINED) {
588 if (HasRequiresReverseOffload)
589 RequiresFlags |= OMP_REQ_REVERSE_OFFLOAD;
590 if (HasRequiresUnifiedAddress)
591 RequiresFlags |= OMP_REQ_UNIFIED_ADDRESS;
592 if (HasRequiresUnifiedSharedMemory)
593 RequiresFlags |= OMP_REQ_UNIFIED_SHARED_MEMORY;
594 if (HasRequiresDynamicAllocators)
595 RequiresFlags |= OMP_REQ_DYNAMIC_ALLOCATORS;
599 return RequiresFlags & OMP_REQ_REVERSE_OFFLOAD;
603 return RequiresFlags & OMP_REQ_UNIFIED_ADDRESS;
607 return RequiresFlags & OMP_REQ_UNIFIED_SHARED_MEMORY;
611 return RequiresFlags & OMP_REQ_DYNAMIC_ALLOCATORS;
616 :
static_cast<int64_t
>(OMP_REQ_NONE);
621 RequiresFlags |= OMP_REQ_REVERSE_OFFLOAD;
623 RequiresFlags &= ~OMP_REQ_REVERSE_OFFLOAD;
628 RequiresFlags |= OMP_REQ_UNIFIED_ADDRESS;
630 RequiresFlags &= ~OMP_REQ_UNIFIED_ADDRESS;
635 RequiresFlags |= OMP_REQ_UNIFIED_SHARED_MEMORY;
637 RequiresFlags &= ~OMP_REQ_UNIFIED_SHARED_MEMORY;
642 RequiresFlags |= OMP_REQ_DYNAMIC_ALLOCATORS;
644 RequiresFlags &= ~OMP_REQ_DYNAMIC_ALLOCATORS;
657 constexpr size_t MaxDim = 3;
662 Value *DynCGroupMemFallbackFlag =
664 DynCGroupMemFallbackFlag =
Builder.CreateShl(DynCGroupMemFallbackFlag, 2);
669 StrictBlocksFlag =
Builder.CreateShl(StrictBlocksFlag, 6);
670 StrictThreadsFlag =
Builder.CreateShl(StrictThreadsFlag, 7);
672 Value *Flags =
Builder.CreateOr(HasNoWaitFlag, DynCGroupMemFallbackFlag);
673 Flags =
Builder.CreateOr(Flags, StrictBlocksFlag);
674 Flags =
Builder.CreateOr(Flags, StrictThreadsFlag);
680 Value *NumThreads3D =
711 auto FnAttrs = Attrs.getFnAttrs();
712 auto RetAttrs = Attrs.getRetAttrs();
714 for (
size_t ArgNo = 0; ArgNo < Fn.
arg_size(); ++ArgNo)
719 bool Param =
true) ->
void {
720 bool HasSignExt = AS.hasAttribute(Attribute::SExt);
721 bool HasZeroExt = AS.hasAttribute(Attribute::ZExt);
722 if (HasSignExt || HasZeroExt) {
723 assert(AS.getNumAttributes() == 1 &&
724 "Currently not handling extension attr combined with others.");
726 if (
auto AK = TargetLibraryInfo::getExtAttrForI32Param(
T, HasSignExt))
729 TargetLibraryInfo::getExtAttrForI32Return(
T, HasSignExt))
736#define OMP_ATTRS_SET(VarName, AttrSet) AttributeSet VarName = AttrSet;
737#include "llvm/Frontend/OpenMP/OMPKinds.def"
741#define OMP_RTL_ATTRS(Enum, FnAttrSet, RetAttrSet, ArgAttrSets) \
743 FnAttrs = FnAttrs.addAttributes(Ctx, FnAttrSet); \
744 addAttrSet(RetAttrs, RetAttrSet, false); \
745 for (size_t ArgNo = 0; ArgNo < ArgAttrSets.size(); ++ArgNo) \
746 addAttrSet(ArgAttrs[ArgNo], ArgAttrSets[ArgNo]); \
747 Fn.setAttributes(AttributeList::get(Ctx, FnAttrs, RetAttrs, ArgAttrs)); \
749#include "llvm/Frontend/OpenMP/OMPKinds.def"
763#define OMP_RTL(Enum, Str, IsVarArg, ReturnType, ...) \
765 FnTy = FunctionType::get(ReturnType, ArrayRef<Type *>{__VA_ARGS__}, \
767 Fn = M.getFunction(Str); \
769#include "llvm/Frontend/OpenMP/OMPKinds.def"
775#define OMP_RTL(Enum, Str, ...) \
777 Fn = Function::Create(FnTy, GlobalValue::ExternalLinkage, Str, M); \
779#include "llvm/Frontend/OpenMP/OMPKinds.def"
783 if (FnID == OMPRTL___kmpc_fork_call || FnID == OMPRTL___kmpc_fork_teams) {
793 LLVMContext::MD_callback,
795 2, {-1, -1},
true)}));
808 assert(Fn &&
"Failed to create OpenMP runtime function");
819 Builder.SetInsertPoint(FiniBB);
831 FiniBB = OtherFiniBB;
833 Builder.SetInsertPoint(FiniBB->getFirstNonPHIIt());
841 auto EndIt = FiniBB->end();
842 if (FiniBB->size() >= 1)
843 if (
auto Prev = std::prev(EndIt); Prev->isTerminator())
848 FiniBB->replaceAllUsesWith(OtherFiniBB);
849 FiniBB->eraseFromParent();
850 FiniBB = OtherFiniBB;
857 assert(Fn &&
"Failed to create OpenMP runtime function pointer");
880 for (
auto Inst =
Block->getReverseIterator()->begin();
881 Inst !=
Block->getReverseIterator()->end();) {
910 Block.getParent()->getEntryBlock().getTerminator()->getIterator();
931 DeferredOutlines.
push_back(std::move(OI));
935 ParallelRegionBlockSet.
clear();
937 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
947 bool ArgsInZeroAddressSpace =
Config.isTargetDevice();
948 std::unique_ptr<CodeExtractor> Extractor =
949 OI->createCodeExtractor(Blocks, ArgsInZeroAddressSpace,
".omp_par");
953 <<
" Exit: " << OI->ExitBB->getName() <<
"\n");
954 assert(Extractor->isEligible() &&
955 "Expected OpenMP outlining to be possible!");
957 for (
auto *V : OI->ExcludeArgsFromAggregate)
958 Extractor->excludeArgFromAggregate(V);
961 Extractor->extractCodeRegion(CEAC, OI->Inputs, OI->Outputs);
965 if (TargetCpuAttr.isStringAttribute())
968 auto TargetFeaturesAttr = OuterFn->
getFnAttribute(
"target-features");
969 if (TargetFeaturesAttr.isStringAttribute())
970 OutlinedFn->
addFnAttr(TargetFeaturesAttr);
973 LLVM_DEBUG(
dbgs() <<
" Outlined function: " << *OutlinedFn <<
"\n");
975 "OpenMP outlined functions should not return a value!");
980 M.getFunctionList().insertAfter(OuterFn->
getIterator(), OutlinedFn);
987 assert(OI->EntryBB->getUniquePredecessor() == &ArtificialEntry);
994 "Expected instructions to add in the outlined region entry");
996 End = ArtificialEntry.
rend();
1001 if (
I.isTerminator()) {
1003 if (
Instruction *TI = OI->EntryBB->getTerminatorOrNull())
1004 TI->adoptDbgRecords(&ArtificialEntry,
I.getIterator(),
false);
1008 I.moveBeforePreserving(*OI->EntryBB,
1009 OI->EntryBB->getFirstInsertionPt());
1012 OI->EntryBB->moveBefore(&ArtificialEntry);
1019 if (OI->PostOutlineCB)
1020 OI->PostOutlineCB(*OutlinedFn);
1022 if (OI->FixUpNonEntryAllocas)
1054 errs() <<
"Error of kind: " << Kind
1055 <<
" when emitting offload entries and metadata during "
1056 "OMPIRBuilder finalization \n";
1064 if (
Config.isTargetDevice())
1065 applyDeclareTargetGlobalReplacements();
1067 if (
Config.EmitLLVMUsedMetaInfo.value_or(
false)) {
1068 std::vector<WeakTrackingVH> LLVMCompilerUsed = {
1069 M.getGlobalVariable(
"__openmp_nvptx_data_transfer_temporary_storage")};
1070 emitUsed(
"llvm.compiler.used", LLVMCompilerUsed);
1080 assert(Original && Replacement &&
1081 "Null values provided to registerDeclareTargetGlobalReplacement");
1085void OpenMPIRBuilder::applyDeclareTargetGlobalReplacements() {
1091 "A null value was inserted into DeclareTargetGlobalReplacements");
1095 if (!OldGV || !NewGV)
1129 for (
unsigned I = 0, E =
PHI->getNumIncomingValues();
I < E; ++
I) {
1130 if (
PHI->getIncomingValue(
I) != OldGV)
1135 Builder.SetCurrentDebugLocation(
PHI->getDebugLoc());
1137 PHI->setIncomingValue(
I, EdgeLoad);
1143 Builder.SetCurrentDebugLocation(Insn->getDebugLoc());
1159 "Non-default address space declare target global");
1161 unsigned DestAS = ASC->getType()->getPointerAddressSpace();
1162 if (DestAS == 0 && NewGVAS != OldGVAS) {
1163 ASC->replaceAllUsesWith(
Load);
1164 ASC->eraseFromParent();
1169 Insn->replaceUsesOfWith(OldGV,
Load);
1185 ConstantInt::get(I32Ty,
Value), Name);
1198 for (
unsigned I = 0, E =
List.size();
I != E; ++
I)
1202 if (UsedArray.
empty())
1209 GV->setSection(
"llvm.metadata");
1215 auto *Int8Ty =
Builder.getInt8Ty();
1218 ConstantInt::get(Int8Ty, Mode),
Twine(KernelName,
"_exec_mode"));
1226 unsigned Reserve2Flags) {
1228 LocFlags |= OMP_IDENT_FLAG_KMPC;
1235 ConstantInt::get(Int32,
uint32_t(LocFlags)),
1236 ConstantInt::get(Int32, Reserve2Flags),
1237 ConstantInt::get(Int32, SrcLocStrSize), SrcLocStr};
1239 size_t SrcLocStrArgIdx = 4;
1240 if (OpenMPIRBuilder::Ident->getElementType(SrcLocStrArgIdx)
1244 SrcLocStr, OpenMPIRBuilder::Ident->getElementType(SrcLocStrArgIdx));
1251 if (
GV.getValueType() == OpenMPIRBuilder::Ident &&
GV.hasInitializer())
1252 if (
GV.getInitializer() == Initializer)
1257 M, OpenMPIRBuilder::Ident,
1260 M.getDataLayout().getDefaultGlobalsAddressSpace());
1272 SrcLocStrSize = LocStr.
size();
1281 if (
GV.isConstant() &&
GV.hasInitializer() &&
1282 GV.getInitializer() == Initializer)
1285 SrcLocStr =
Builder.CreateGlobalString(
1286 LocStr,
"",
M.getDataLayout().getDefaultGlobalsAddressSpace(),
1294 unsigned Line,
unsigned Column,
1300 Buffer.
append(FunctionName);
1302 Buffer.
append(std::to_string(Line));
1304 Buffer.
append(std::to_string(Column));
1312 StringRef UnknownLoc =
";unknown;unknown;0;0;;";
1323 !DIL->getFilename().empty() ? DIL->getFilename() :
M.getName();
1328 DIL->getColumn(), SrcLocStrSize);
1334 Loc.IP.getBlock()->getParent());
1340 "omp_global_thread_num");
1348 "expected one result pointer type per in_reduction item");
1351 if (OrigPtrs.
empty())
1352 return Builder.saveIP();
1371 for (
unsigned Idx = 0; Idx < OrigPtrs.
size(); ++Idx) {
1374 Value *OrigPtr = OrigPtrs[Idx];
1376 OrigPtrTy && OrigPtrTy->getAddressSpace() != 0)
1377 OrigPtr = Builder.CreateAddrSpaceCast(OrigPtr, PtrTy);
1379 Value *
Priv = Builder.CreateCall(GetThData, {Gtid, NullDesc, OrigPtr},
1385 ResPtrTy && ResPtrTy->getAddressSpace() != 0)
1386 Priv = Builder.CreateAddrSpaceCast(
Priv, ResultPtrTys[Idx]);
1388 MapPrivateCB(Idx,
Priv);
1395 bool ForceSimpleCall,
bool CheckCancelFlag) {
1405 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_FOR;
1408 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_SECTIONS;
1411 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL_SINGLE;
1414 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_EXPL;
1417 BarrierLocFlags = OMP_IDENT_FLAG_BARRIER_IMPL;
1430 bool UseCancelBarrier =
1435 ? OMPRTL___kmpc_cancel_barrier
1436 : OMPRTL___kmpc_barrier),
1439 if (UseCancelBarrier && CheckCancelFlag)
1449 omp::Directive CanceledDirective) {
1454 auto *UI =
Builder.CreateUnreachable();
1462 Builder.SetInsertPoint(ElseTI);
1463 auto ElseIP =
Builder.saveIP();
1471 Builder.SetInsertPoint(ThenTI);
1473 Value *CancelKind =
nullptr;
1474 switch (CanceledDirective) {
1475#define OMP_CANCEL_KIND(Enum, Str, DirectiveEnum, Value) \
1476 case DirectiveEnum: \
1477 CancelKind = Builder.getInt32(Value); \
1479#include "llvm/Frontend/OpenMP/OMPKinds.def"
1496 Builder.SetInsertPoint(UI->getParent());
1497 UI->eraseFromParent();
1504 omp::Directive CanceledDirective) {
1509 auto *UI =
Builder.CreateUnreachable();
1512 Value *CancelKind =
nullptr;
1513 switch (CanceledDirective) {
1514#define OMP_CANCEL_KIND(Enum, Str, DirectiveEnum, Value) \
1515 case DirectiveEnum: \
1516 CancelKind = Builder.getInt32(Value); \
1518#include "llvm/Frontend/OpenMP/OMPKinds.def"
1535 Builder.SetInsertPoint(UI->getParent());
1536 UI->eraseFromParent();
1549 auto *KernelArgsPtr =
1550 Builder.CreateAlloca(OpenMPIRBuilder::KernelArgs,
nullptr,
"kernel_args");
1555 Builder.CreateStructGEP(OpenMPIRBuilder::KernelArgs, KernelArgsPtr,
I);
1558 M.getDataLayout().getPrefTypeAlign(KernelArgs[
I]->getType()));
1562 NumThreads, HostPtr, KernelArgsPtr};
1589 assert(OutlinedFnID &&
"Invalid outlined function ID!");
1593 Value *Return =
nullptr;
1613 Builder, AllocaIP, Return, RTLoc, DeviceID, Args.NumTeams.front(),
1614 Args.NumThreads.front(), OutlinedFnID, ArgsVector));
1621 Builder.CreateCondBr(
Failed, OffloadFailedBlock, OffloadContBlock);
1623 auto CurFn =
Builder.GetInsertBlock()->getParent();
1630 emitBlock(OffloadContBlock, CurFn,
true);
1635 Value *CancelFlag, omp::Directive CanceledDirective) {
1637 "Unexpected cancellation!");
1657 Builder.CreateCondBr(Cmp, NonCancellationBlock, CancellationBlock,
1666 Builder.SetInsertPoint(CancellationBlock);
1667 Builder.CreateBr(*FiniBBOrErr);
1670 Builder.SetInsertPoint(NonCancellationBlock, NonCancellationBlock->
begin());
1682 size_t NumArgs = OutlinedFn.
arg_size();
1683 assert((NumArgs == 2 || NumArgs == 3) &&
1684 "expected a 2-3 argument parallel outlined function");
1685 bool UseArgStruct = NumArgs == 3;
1690 {Builder.getInt16Ty(), Builder.getInt32Ty()},
1694 OutlinedFn.
getName() +
".wrapper", OMPIRBuilder->
M);
1696 WrapperFn->addParamAttr(0, Attribute::NoUndef);
1697 WrapperFn->addParamAttr(0, Attribute::ZExt);
1698 WrapperFn->addParamAttr(1, Attribute::NoUndef);
1702 Builder.SetInsertPoint(EntryBB);
1705 Value *AddrAlloca = Builder.CreateAlloca(Builder.getInt32Ty(),
1707 AddrAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1708 AddrAlloca, Builder.getPtrTy(0),
1709 AddrAlloca->
getName() +
".ascast");
1711 Value *ZeroAlloca = Builder.CreateAlloca(Builder.getInt32Ty(),
1713 ZeroAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1714 ZeroAlloca, Builder.getPtrTy(0),
1715 ZeroAlloca->
getName() +
".ascast");
1717 Value *ArgsAlloca =
nullptr;
1719 ArgsAlloca = Builder.CreateAlloca(Builder.getPtrTy(),
1720 nullptr,
"global_args");
1721 ArgsAlloca = Builder.CreatePointerBitCastOrAddrSpaceCast(
1722 ArgsAlloca, Builder.getPtrTy(0),
1723 ArgsAlloca->
getName() +
".ascast");
1727 Builder.CreateStore(WrapperFn->getArg(1), AddrAlloca);
1728 Builder.CreateStore(Builder.getInt32(0), ZeroAlloca);
1732 llvm::omp::RuntimeFunction::OMPRTL___kmpc_get_shared_variables),
1740 Value *StructArg = Builder.CreateLoad(Builder.getPtrTy(), ArgsAlloca);
1741 StructArg = Builder.CreateInBoundsGEP(Builder.getPtrTy(), StructArg,
1742 {Builder.getInt64(0)});
1743 StructArg = Builder.CreateLoad(Builder.getPtrTy(), StructArg,
"structArg");
1744 Args.push_back(StructArg);
1748 Builder.CreateCall(&OutlinedFn, Args);
1749 Builder.CreateRetVoid();
1764 "Expected at least tid and bounded tid as arguments");
1765 unsigned NumCapturedVars = OutlinedFn.
arg_size() - 2;
1773 OutlinedFn.
addFnAttr(Attribute::NoUnwind);
1776 assert(CI &&
"Expected call instruction to outlined function");
1777 CI->
getParent()->setName(
"omp_parallel");
1779 Builder.SetInsertPoint(CI);
1780 Type *PtrTy = OMPIRBuilder->VoidPtr;
1783 OpenMPIRBuilder ::InsertPointTy CurrentIP = Builder.saveIP();
1787 Value *Args = ArgsAlloca;
1791 Args = Builder.CreatePointerCast(ArgsAlloca, PtrTy);
1792 Builder.restoreIP(CurrentIP);
1795 for (
unsigned Idx = 0; Idx < NumCapturedVars; Idx++) {
1797 Value *StoreAddress = Builder.CreateConstInBoundsGEP2_64(
1799 Builder.CreateStore(V, StoreAddress);
1803 IfCondition ? Builder.CreateSExtOrTrunc(IfCondition, OMPIRBuilder->Int32)
1804 : Builder.getInt32(1);
1805 Value *NumThreadsArg =
1806 NumThreads ? Builder.CreateZExtOrTrunc(NumThreads, OMPIRBuilder->Int32)
1807 : Builder.getInt32(-1);
1817 Value *Parallel60CallArgs[] = {
1822 Builder.getInt32(-1),
1826 Builder.getInt64(NumCapturedVars),
1827 Builder.getInt32(0)};
1835 << *Builder.GetInsertBlock()->getParent() <<
"\n");
1838 Builder.SetInsertPoint(PrivTID);
1840 Builder.CreateStore(Builder.CreateLoad(OMPIRBuilder->Int32, OutlinedAI),
1847 I->eraseFromParent();
1870 if (!
F->hasMetadata(LLVMContext::MD_callback)) {
1878 F->addMetadata(LLVMContext::MD_callback,
1887 OutlinedFn.
addFnAttr(Attribute::NoUnwind);
1890 "Expected at least tid and bounded tid as arguments");
1891 unsigned NumCapturedVars = OutlinedFn.
arg_size() - 2;
1894 CI->
getParent()->setName(
"omp_parallel");
1895 Builder.SetInsertPoint(CI);
1898 Value *ForkCallArgs[] = {Ident, Builder.getInt32(NumCapturedVars),
1902 RealArgs.
append(std::begin(ForkCallArgs), std::end(ForkCallArgs));
1904 Value *
Cond = Builder.CreateSExtOrTrunc(IfCondition, OMPIRBuilder->Int32);
1911 auto PtrTy = OMPIRBuilder->VoidPtr;
1912 if (IfCondition && NumCapturedVars == 0) {
1920 << *Builder.GetInsertBlock()->getParent() <<
"\n");
1923 Builder.SetInsertPoint(PrivTID);
1925 Builder.CreateStore(Builder.CreateLoad(OMPIRBuilder->Int32, OutlinedAI),
1932 I->eraseFromParent();
1940 Value *NumThreads, omp::ProcBindKind ProcBind,
bool IsCancellable) {
1949 const bool NeedThreadID = NumThreads ||
Config.isTargetDevice() ||
1950 (ProcBind != OMP_PROC_BIND_default);
1957 bool ArgsInZeroAddressSpace =
Config.isTargetDevice();
1961 if (NumThreads && !
Config.isTargetDevice()) {
1964 Builder.CreateIntCast(NumThreads, Int32,
false)};
1969 if (ProcBind != OMP_PROC_BIND_default) {
1973 ConstantInt::get(Int32,
unsigned(ProcBind),
true)};
1995 Builder.CreateAlloca(Int32,
nullptr,
"zero.addr");
1998 if (ArgsInZeroAddressSpace &&
M.getDataLayout().getAllocaAddrSpace() != 0) {
2001 TIDAddrAlloca, PointerType ::get(
M.getContext(), 0),
"tid.addr.ascast");
2005 PointerType ::get(
M.getContext(), 0),
2006 "zero.addr.ascast");
2030 if (IP.getBlock()->end() == IP.getPoint()) {
2036 assert(IP.getBlock()->getTerminator()->getNumSuccessors() == 1 &&
2037 IP.getBlock()->getTerminator()->getSuccessor(0) == PRegExitBB &&
2038 "Unexpected insertion point for finalization call!");
2050 Builder.CreateAlloca(Int32,
nullptr,
"tid.addr.local");
2056 Builder.CreateLoad(Int32, ZeroAddr,
"zero.addr.use");
2074 LLVM_DEBUG(
dbgs() <<
"Before body codegen: " << *OuterFn <<
"\n");
2077 assert(BodyGenCB &&
"Expected body generation callback!");
2079 if (
Error Err = BodyGenCB(InnerAllocaIP, CodeGenIP, PRegExitBB))
2082 LLVM_DEBUG(
dbgs() <<
"After body codegen: " << *OuterFn <<
"\n");
2086 bool UsesDeviceSharedMemory =
2088 std::unique_ptr<OutlineInfo> OI =
2089 UsesDeviceSharedMemory
2090 ? std::make_unique<DeviceSharedMemOutlineInfo>(*
this)
2091 : std::make_unique<OutlineInfo>();
2093 if (
Config.isTargetDevice()) {
2095 OI->PostOutlineCB = [=, ToBeDeletedVec =
2096 std::move(ToBeDeleted)](
Function &OutlinedFn) {
2098 IfCondition, NumThreads, PrivTID, PrivTIDAddr,
2099 ThreadID, ToBeDeletedVec);
2103 OI->PostOutlineCB = [=, ToBeDeletedVec =
2104 std::move(ToBeDeleted)](
Function &OutlinedFn) {
2106 PrivTID, PrivTIDAddr, ToBeDeletedVec);
2110 OI->FixUpNonEntryAllocas =
true;
2111 OI->OuterAllocBB = OuterAllocaBlock;
2112 OI->EntryBB = PRegEntryBB;
2113 OI->ExitBB = PRegExitBB;
2114 OI->OuterDeallocBBs.reserve(OuterDeallocBlocks.
size());
2115 copy(OuterDeallocBlocks, OI->OuterDeallocBBs.
end());
2119 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
2131 ".omp_par", ArgsInZeroAddressSpace);
2136 Extractor.findAllocas(CEAC, SinkingCands, HoistingCands, CommonExit);
2138 Extractor.findInputsOutputs(Inputs, Outputs, SinkingCands,
2143 return GV->getValueType() == OpenMPIRBuilder::Ident;
2148 LLVM_DEBUG(
dbgs() <<
"Before privatization: " << *OuterFn <<
"\n");
2154 if (&V == TIDAddr || &V == ZeroAddr) {
2155 OI->ExcludeArgsFromAggregate.push_back(&V);
2160 for (
Use &U : V.uses())
2162 if (ParallelRegionBlockSet.
count(UserI->getParent()))
2172 if (!V.getType()->isPointerTy()) {
2176 Builder.restoreIP(OuterAllocIP);
2178 if (UsesDeviceSharedMemory) {
2181 V.getName() +
".reloaded");
2182 for (
BasicBlock *DeallocBlock : OuterDeallocBlocks)
2184 InsertPointTy(DeallocBlock, DeallocBlock->getFirstInsertionPt()),
2187 Ptr =
Builder.CreateAlloca(V.getType(),
nullptr,
2188 V.getName() +
".reloaded");
2193 Builder.SetInsertPoint(InsertBB,
2198 Builder.restoreIP(InnerAllocaIP);
2199 Inner =
Builder.CreateLoad(V.getType(), Ptr);
2202 Value *ReplacementValue =
nullptr;
2205 ReplacementValue = PrivTID;
2208 PrivCB(InnerAllocaIP,
Builder.saveIP(), V, *Inner, ReplacementValue);
2216 assert(ReplacementValue &&
2217 "Expected copy/create callback to set replacement value!");
2218 if (ReplacementValue == &V)
2223 UPtr->set(ReplacementValue);
2248 for (
Value *Output : Outputs)
2252 "OpenMP outlining should not produce live-out values!");
2254 LLVM_DEBUG(
dbgs() <<
"After privatization: " << *OuterFn <<
"\n");
2256 for (
auto *BB : Blocks)
2257 dbgs() <<
" PBR: " << BB->getName() <<
"\n";
2265 assert(FiniInfo.DK == OMPD_parallel &&
2266 "Unexpected finalization stack state!");
2277 Builder.CreateBr(*FiniBBOrErr);
2281 Term->eraseFromParent();
2287 InsertPointTy AfterIP(UI->getParent(), UI->getParent()->end());
2288 UI->eraseFromParent();
2320 Value *Severity = ConstantInt::get(Int32, IsFatal ? 2 : 1);
2322 Value *Args[] = {Ident, Severity, MessageArg};
2351 static_cast<unsigned int>(RTLDependInfoFields::BaseAddr));
2353 Builder.CreateStore(DepValPtr, Addr);
2356 DependInfo, Entry,
static_cast<unsigned int>(RTLDependInfoFields::Len));
2358 ConstantInt::get(SizeTy,
2363 DependInfo, Entry,
static_cast<unsigned int>(RTLDependInfoFields::Flags));
2365 static_cast<unsigned int>(Dep.
DepKind)),
2378 if (Dependencies.
empty())
2398 Type *DependInfo = OMPBuilder.DependInfo;
2400 Value *DepArray =
nullptr;
2402 Builder.SetInsertPoint(
2406 DepArray = Builder.CreateAlloca(DepArrayTy,
nullptr,
".dep.arr.addr");
2408 Builder.restoreIP(OldIP);
2410 for (
const auto &[DepIdx, Dep] :
enumerate(Dependencies)) {
2412 Builder.CreateConstInBoundsGEP2_64(DepArrayTy, DepArray, 0, DepIdx);
2436 Value *DepArray =
nullptr;
2437 Type *DepArrayTy =
nullptr;
2438 Value *NumDeps =
nullptr;
2441 NumDeps = Dependencies.
NumDeps;
2442 }
else if (!Dependencies.
Deps.empty()) {
2445 Builder.GetInsertBlock()->getParent()->getEntryBlock();
2449 DepArray =
Builder.CreateAlloca(DepArrayTy,
nullptr,
".dep.arr.addr");
2450 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
2453 for (
const auto &[DepIdx, Dep] :
enumerate(Dependencies.
Deps)) {
2455 Builder.CreateConstInBoundsGEP2_64(DepArrayTy, DepArray, 0, DepIdx);
2469 ConstantInt::get(
Builder.getInt32Ty(), 0),
2471 ConstantInt::get(
Builder.getInt32Ty(),
false)};
2474 omp::RuntimeFunction::OMPRTL___kmpc_omp_taskwait_deps_51),
2484 unsigned ProgramAddressSpace = M.getDataLayout().getProgramAddressSpace();
2496 auto *VoidPtrTy =
PointerType::get(Builder.getContext(), ProgramAddressSpace);
2499 Builder.getVoidTy(), {VoidPtrTy, VoidPtrTy, Builder.getInt32Ty()},
2503 "omp_taskloop_dup", M);
2506 Value *LastprivateFlagArg = DupFunction->
getArg(2);
2507 DestTaskArg->
setName(
"dest_task");
2508 SrcTaskArg->
setName(
"src_task");
2509 LastprivateFlagArg->
setName(
"lastprivate_flag");
2512 Builder.SetInsertPoint(
2515 auto GetTaskContextPtrFromArg = [&](
Value *Arg) ->
Value * {
2516 Type *TaskWithPrivatesTy =
2518 Value *TaskPrivates = Builder.CreateGEP(
2519 TaskWithPrivatesTy, Arg, {Builder.getInt32(0), Builder.getInt32(1)});
2520 Value *ContextPtr = Builder.CreateGEP(
2521 PrivatesTy, TaskPrivates,
2522 {Builder.getInt32(0), Builder.getInt32(PrivatesIndex)});
2526 Value *DestTaskContextPtr = GetTaskContextPtrFromArg(DestTaskArg);
2527 Value *SrcTaskContextPtr = GetTaskContextPtrFromArg(SrcTaskArg);
2529 DestTaskContextPtr->
setName(
"destPtr");
2530 SrcTaskContextPtr->
setName(
"srcPtr");
2535 Expected<IRBuilderBase::InsertPoint> AfterIPOrError =
2536 DupCB(AllocaIP, CodeGenIP, DestTaskContextPtr, SrcTaskContextPtr);
2537 if (!AfterIPOrError)
2539 Builder.restoreIP(*AfterIPOrError);
2549 llvm::function_ref<llvm::Expected<llvm::CanonicalLoopInfo *>()> LoopInfo,
2551 Value *GrainSize,
bool NoGroup,
int Sched,
Value *Final,
bool Mergeable,
2553 Value *TaskContextStructPtrVal) {
2558 uint32_t SrcLocStrSize;
2574 if (
Error Err = BodyGenCB(TaskloopAllocaIP, TaskloopBodyIP, TaskloopExitBB))
2577 llvm::Expected<llvm::CanonicalLoopInfo *> result = LoopInfo();
2582 llvm::CanonicalLoopInfo *CLI = result.
get();
2583 auto OI = std::make_unique<OutlineInfo>();
2584 OI->EntryBB = TaskloopAllocaBB;
2585 OI->OuterAllocBB = AllocaIP.getBlock();
2586 OI->ExitBB = TaskloopExitBB;
2587 OI->OuterDeallocBBs.reserve(DeallocBlocks.
size());
2588 copy(DeallocBlocks, OI->OuterDeallocBBs.end());
2591 SmallVector<Instruction *> ToBeDeleted;
2594 Builder, AllocaIP, ToBeDeleted, TaskloopAllocaIP,
"global.tid",
false));
2596 TaskloopAllocaIP,
"lb",
false,
true);
2598 TaskloopAllocaIP,
"ub",
false,
true);
2600 TaskloopAllocaIP,
"step",
false,
true);
2603 OI->Inputs.insert(FakeLB);
2604 OI->Inputs.insert(FakeUB);
2605 OI->Inputs.insert(FakeStep);
2606 if (TaskContextStructPtrVal)
2607 OI->Inputs.insert(TaskContextStructPtrVal);
2608 assert(((TaskContextStructPtrVal && DupCB) ||
2609 (!TaskContextStructPtrVal && !DupCB)) &&
2610 "Task context struct ptr and duplication callback must be both set "
2616 unsigned ProgramAddressSpace =
M.getDataLayout().getProgramAddressSpace();
2620 {FakeLB->getType(), FakeUB->getType(), FakeStep->getType(), PointerTy});
2621 Expected<Value *> TaskDupFnOrErr = createTaskDuplicationFunction(
2624 if (!TaskDupFnOrErr) {
2627 Value *TaskDupFn = *TaskDupFnOrErr;
2629 OI->PostOutlineCB = [
this, Ident, LBVal, UBVal, StepVal, Untied,
2630 TaskloopAllocaBB, CLI, TaskDupFn, ToBeDeleted, IfCond,
2631 GrainSize, NoGroup, Sched, FakeLB, FakeUB, FakeStep,
2632 FakeSharedsTy, Final, Mergeable, Priority,
2633 NumOfCollapseLoops](
Function &OutlinedFn)
mutable {
2635 assert(OutlinedFn.hasOneUse() &&
2636 "there must be a single user for the outlined function");
2643 Value *CastedLBVal =
2644 Builder.CreateIntCast(LBVal,
Builder.getInt64Ty(),
true,
"lb64");
2645 Value *CastedUBVal =
2646 Builder.CreateIntCast(UBVal,
Builder.getInt64Ty(),
true,
"ub64");
2647 Value *CastedStepVal =
2648 Builder.CreateIntCast(StepVal,
Builder.getInt64Ty(),
true,
"step64");
2650 Builder.SetInsertPoint(StaleCI);
2663 Builder.CreateCall(TaskgroupFn, {Ident, ThreadID});
2684 divideCeil(
M.getDataLayout().getTypeSizeInBits(Task), 8));
2686 AllocaInst *ArgStructAlloca =
2688 assert(ArgStructAlloca &&
2689 "Unable to find the alloca instruction corresponding to arguments "
2690 "for extracted function");
2691 std::optional<TypeSize> ArgAllocSize =
2694 "Unable to determine size of arguments for extracted function");
2695 Value *SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
2700 CallInst *TaskData =
Builder.CreateCall(
2701 TaskAllocFn, {Ident, ThreadID,
Flags,
2702 TaskSize, SharedsSize,
2707 Value *TaskShareds =
Builder.CreateLoad(VoidPtr, TaskData);
2708 Builder.CreateMemCpy(TaskShareds, Alignment, Shareds, Alignment,
2713 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(0)});
2716 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(1)});
2719 FakeSharedsTy, TaskShareds, {
Builder.getInt32(0),
Builder.getInt32(2)});
2725 IfCond ?
Builder.CreateIntCast(IfCond,
Builder.getInt32Ty(),
true)
2731 Value *GrainSizeVal =
2732 GrainSize ?
Builder.CreateIntCast(GrainSize,
Builder.getInt64Ty(),
true)
2734 Value *TaskDup = TaskDupFn;
2736 Value *
Args[] = {Ident, ThreadID, TaskData, IfCondVal, Lb, Ub,
2737 Loadstep, NoGroupVal, SchedVal, GrainSizeVal, TaskDup};
2742 Builder.CreateCall(TaskloopFn, Args);
2749 Builder.CreateCall(EndTaskgroupFn, {Ident, ThreadID});
2754 Builder.SetInsertPoint(TaskloopAllocaBB, TaskloopAllocaBB->begin());
2756 LoadInst *SharedsOutlined =
2757 Builder.CreateLoad(VoidPtr, OutlinedFn.getArg(1));
2758 OutlinedFn.getArg(1)->replaceUsesWithIf(
2760 [SharedsOutlined](Use &U) {
return U.getUser() != SharedsOutlined; });
2763 Type *IVTy =
IV->getType();
2769 Value *TaskLB =
nullptr;
2770 Value *TaskUB =
nullptr;
2771 Value *TaskStep =
nullptr;
2772 Value *LoadTaskLB =
nullptr;
2773 Value *LoadTaskUB =
nullptr;
2774 Value *LoadTaskStep =
nullptr;
2775 for (Instruction &
I : *TaskloopAllocaBB) {
2776 if (
I.getOpcode() == Instruction::GetElementPtr) {
2779 switch (CI->getZExtValue()) {
2791 }
else if (
I.getOpcode() == Instruction::Load) {
2793 if (
Load.getPointerOperand() == TaskLB) {
2794 assert(TaskLB !=
nullptr &&
"Expected value for TaskLB");
2796 }
else if (
Load.getPointerOperand() == TaskUB) {
2797 assert(TaskUB !=
nullptr &&
"Expected value for TaskUB");
2799 }
else if (
Load.getPointerOperand() == TaskStep) {
2800 assert(TaskStep !=
nullptr &&
"Expected value for TaskStep");
2806 Builder.SetInsertPoint(CLI->getPreheader()->getTerminator());
2808 assert(LoadTaskLB !=
nullptr &&
"Expected value for LoadTaskLB");
2809 assert(LoadTaskUB !=
nullptr &&
"Expected value for LoadTaskUB");
2810 assert(LoadTaskStep !=
nullptr &&
"Expected value for LoadTaskStep");
2812 Builder.CreateSub(LoadTaskUB, LoadTaskLB), LoadTaskStep);
2813 Value *TripCount =
Builder.CreateAdd(TripCountMinusOne, One,
"trip_cnt");
2814 Value *CastedTripCount =
Builder.CreateIntCast(TripCount, IVTy,
true);
2815 Value *CastedTaskLB =
Builder.CreateIntCast(LoadTaskLB, IVTy,
true);
2817 CLI->setTripCount(CastedTripCount);
2819 Builder.SetInsertPoint(CLI->getBody(),
2820 CLI->getBody()->getFirstInsertionPt());
2822 if (NumOfCollapseLoops > 1) {
2828 Builder.CreateSub(CastedTaskLB, ConstantInt::get(IVTy, 1)));
2831 for (
auto IVUse = CLI->getIndVar()->uses().begin();
2832 IVUse != CLI->getIndVar()->uses().end(); IVUse++) {
2833 User *IVUser = IVUse->getUser();
2835 if (
Op->getOpcode() == Instruction::URem ||
2836 Op->getOpcode() == Instruction::UDiv) {
2841 for (User *User : UsersToReplace) {
2842 User->replaceUsesOfWith(CLI->getIndVar(), IVPlusTaskLB);
2859 assert(CLI->getIndVar()->getNumUses() == 3 &&
2860 "Canonical loop should have exactly three uses of the ind var");
2861 for (User *IVUser : CLI->getIndVar()->users()) {
2863 if (
Mul->getOpcode() == Instruction::Mul) {
2864 for (User *MulUser :
Mul->users()) {
2866 if (
Add->getOpcode() == Instruction::Add) {
2867 Add->setOperand(1, CastedTaskLB);
2876 FakeLB->replaceAllUsesWith(CastedLBVal);
2877 FakeUB->replaceAllUsesWith(CastedUBVal);
2878 FakeStep->replaceAllUsesWith(CastedStepVal);
2880 I->eraseFromParent();
2885 Builder.SetInsertPoint(TaskloopExitBB, TaskloopExitBB->
begin());
2891 M.getContext(),
M.getDataLayout().getPointerSizeInBits());
2901 bool Mergeable,
Value *EventHandle,
Value *Priority) {
2933 if (
Error Err = BodyGenCB(TaskAllocaIP, TaskBodyIP, TaskExitBB))
2936 auto OI = std::make_unique<OutlineInfo>();
2937 OI->EntryBB = TaskAllocaBB;
2938 OI->OuterAllocBB = AllocaIP.
getBlock();
2939 OI->ExitBB = TaskExitBB;
2940 OI->OuterDeallocBBs.reserve(DeallocBlocks.
size());
2941 copy(DeallocBlocks, OI->OuterDeallocBBs.
end());
2946 Builder, AllocaIP, ToBeDeleted, TaskAllocaIP,
"global.tid",
false));
2948 OI->PostOutlineCB = [
this, Ident, Tied, Final, IfCondition, Dependencies,
2949 Affinities, Mergeable, Priority, EventHandle,
2951 ToBeDeleted](
Function &OutlinedFn)
mutable {
2953 assert(OutlinedFn.hasOneUse() &&
2954 "there must be a single user for the outlined function");
2959 bool HasShareds = StaleCI->
arg_size() > 1;
2960 Builder.SetInsertPoint(StaleCI);
2985 bool UseMergedIf0Path = ConstIfCondition && ConstIfCondition->isZero();
2989 Flags =
Builder.CreateOr(FinalFlag, Flags);
2992 if (Mergeable || UseMergedIf0Path)
3004 divideCeil(
M.getDataLayout().getTypeSizeInBits(Task), 8));
3013 assert(ArgStructAlloca &&
3014 "Unable to find the alloca instruction corresponding to arguments "
3015 "for extracted function");
3016 std::optional<TypeSize> ArgAllocSize =
3019 "Unable to determine size of arguments for extracted function");
3020 SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
3026 TaskAllocFn, {Ident, ThreadID, Flags,
3027 TaskSize, SharedsSize,
3030 if (Affinities.
Count && Affinities.
Info) {
3032 OMPRTL___kmpc_omp_reg_task_with_affinity);
3043 OMPRTL___kmpc_task_allow_completion_event);
3047 Builder.CreatePointerBitCastOrAddrSpaceCast(EventHandle,
3049 EventVal =
Builder.CreatePtrToInt(EventVal,
Builder.getInt64Ty());
3050 Builder.CreateStore(EventVal, EventHandleAddr);
3056 Value *TaskShareds =
Builder.CreateLoad(VoidPtr, TaskData);
3057 Builder.CreateMemCpy(TaskShareds, Alignment, Shareds, Alignment,
3071 Constant *Zero = ConstantInt::get(Int32Ty, 0);
3075 Builder.CreateInBoundsGEP(TaskPtr, TaskData, {Zero, Zero});
3078 VoidPtr, VoidPtr,
Builder.getInt32Ty(), VoidPtr, VoidPtr);
3080 TaskStructType, TaskGEP, {Zero, ConstantInt::get(Int32Ty, 4)});
3083 Value *CmplrData =
Builder.CreateInBoundsGEP(CmplrStructType,
3084 PriorityData, {Zero, Zero});
3085 Builder.CreateStore(Priority, CmplrData);
3088 Value *DepArray =
nullptr;
3089 Value *NumDeps =
nullptr;
3092 NumDeps = Dependencies.
NumDeps;
3093 }
else if (!Dependencies.
Deps.empty()) {
3095 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
3115 if (IfCondition && !UseMergedIf0Path) {
3120 Builder.GetInsertPoint()->getParent()->getTerminator();
3121 Instruction *ThenTI = IfTerminator, *ElseTI =
nullptr;
3122 Builder.SetInsertPoint(IfTerminator);
3125 Builder.SetInsertPoint(ElseTI);
3132 {Ident, ThreadID, NumDeps, DepArray,
3133 ConstantInt::get(
Builder.getInt32Ty(), 0),
3148 Builder.SetInsertPoint(ThenTI);
3156 {Ident, ThreadID, TaskData, NumDeps, DepArray,
3157 ConstantInt::get(
Builder.getInt32Ty(), 0),
3168 Builder.SetInsertPoint(TaskAllocaBB, TaskAllocaBB->
begin());
3170 LoadInst *Shareds =
Builder.CreateLoad(VoidPtr, OutlinedFn.getArg(1));
3171 OutlinedFn.getArg(1)->replaceUsesWithIf(
3172 Shareds, [Shareds](
Use &U) {
return U.getUser() != Shareds; });
3178 Builder.ClearInsertionPoint();
3180 I->eraseFromParent();
3184 Builder.SetInsertPoint(TaskExitBB, TaskExitBB->
begin());
3206 if (
Error Err = BodyGenCB(AllocaIP,
Builder.saveIP(), DeallocBlocks))
3209 Builder.SetInsertPoint(TaskgroupExitBB);
3252 unsigned CaseNumber = 0;
3253 for (
auto SectionCB : SectionCBs) {
3255 M.getContext(),
"omp_section_loop.body.case", CurFn,
Continue);
3257 Builder.SetInsertPoint(CaseBB);
3272 Value *LB = ConstantInt::get(I32Ty, 0);
3273 Value *UB = ConstantInt::get(I32Ty, SectionCBs.
size());
3274 Value *ST = ConstantInt::get(I32Ty, 1);
3276 Loc, LoopBodyGenCB, LB, UB, ST,
true,
false, AllocaIP,
"section_loop");
3281 applyStaticWorkshareLoop(
Loc.DL, *
LoopInfo, AllocaIP,
3282 WorksharingLoopType::ForStaticLoop, !IsNowait);
3288 assert(LoopFini &&
"Bad structure of static workshare loop finalization");
3292 assert(FiniInfo.DK == OMPD_sections &&
3293 "Unexpected finalization stack state!");
3294 if (
Error Err = FiniInfo.mergeFiniBB(
Builder, LoopFini))
3308 if (IP.getBlock()->end() != IP.getPoint())
3319 auto *CaseBB =
Loc.IP.getBlock();
3320 auto *CondBB = CaseBB->getSinglePredecessor()->getSinglePredecessor();
3321 auto *ExitBB = CondBB->getTerminator()->getSuccessor(1);
3327 Directive OMPD = Directive::OMPD_sections;
3330 return EmitOMPInlinedRegion(OMPD,
nullptr,
nullptr, BodyGenCB, FiniCBWrapper,
3341Value *OpenMPIRBuilder::getGPUThreadID() {
3344 OMPRTL___kmpc_get_hardware_thread_id_in_block),
3348Value *OpenMPIRBuilder::getGPUWarpSize() {
3353Value *OpenMPIRBuilder::getNVPTXWarpID() {
3354 unsigned LaneIDBits =
Log2_32(
Config.getGridValue().GV_Warp_Size);
3355 return Builder.CreateAShr(getGPUThreadID(), LaneIDBits,
"nvptx_warp_id");
3358Value *OpenMPIRBuilder::getNVPTXLaneID() {
3359 unsigned LaneIDBits =
Log2_32(
Config.getGridValue().GV_Warp_Size);
3360 assert(LaneIDBits < 32 &&
"Invalid LaneIDBits size in NVPTX device.");
3361 unsigned LaneIDMask = ~0
u >> (32u - LaneIDBits);
3362 return Builder.CreateAnd(getGPUThreadID(),
Builder.getInt32(LaneIDMask),
3369 uint64_t FromSize =
M.getDataLayout().getTypeStoreSize(FromType);
3370 uint64_t ToSize =
M.getDataLayout().getTypeStoreSize(ToType);
3371 assert(FromSize > 0 &&
"From size must be greater than zero");
3372 assert(ToSize > 0 &&
"To size must be greater than zero");
3373 if (FromType == ToType)
3375 if (FromSize == ToSize)
3376 return Builder.CreateBitCast(From, ToType);
3378 return Builder.CreateIntCast(From, ToType,
true);
3384 Value *ValCastItem =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3385 CastItem,
Builder.getPtrTy(0));
3386 Builder.CreateStore(From, ValCastItem);
3387 return Builder.CreateLoad(ToType, CastItem);
3394 uint64_t Size =
M.getDataLayout().getTypeStoreSize(ElementType);
3395 assert(
Size <= 8 &&
"Unsupported bitwidth in shuffle instruction");
3399 Value *ElemCast = castValueToType(AllocaIP, Element, CastTy);
3401 Builder.CreateIntCast(getGPUWarpSize(),
Builder.getInt16Ty(),
true);
3403 Size <= 4 ? RuntimeFunction::OMPRTL___kmpc_shuffle_int32
3404 : RuntimeFunction::OMPRTL___kmpc_shuffle_int64);
3405 Value *WarpSizeCast =
3407 Value *ShuffleCall =
3412 return castValueToType(AllocaIP, ShuffleCall, ElementType);
3419 uint64_t Size =
M.getDataLayout().getTypeStoreSize(ElemType);
3431 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3432 Value *ElemPtr = DstAddr;
3433 Value *Ptr = SrcAddr;
3434 for (
unsigned IntSize = 8; IntSize >= 1; IntSize /= 2) {
3438 Ptr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3441 Builder.CreateGEP(ElemType, SrcAddr, {ConstantInt::get(IndexTy, 1)});
3442 ElemPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3446 if ((
Size / IntSize) > 1) {
3447 Value *PtrEnd =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3448 SrcAddrGEP,
Builder.getPtrTy());
3465 Builder.CreatePointerBitCastOrAddrSpaceCast(Ptr,
Builder.getPtrTy()));
3467 Builder.CreateICmpSGT(PtrDiff,
Builder.getInt64(IntSize - 1)), ThenBB,
3470 Value *Res = createRuntimeShuffleFunction(
3473 IntType, Ptr,
M.getDataLayout().getPrefTypeAlign(ElemType)),
3475 Builder.CreateAlignedStore(Res, ElemPtr,
3476 M.getDataLayout().getPrefTypeAlign(ElemType));
3478 Builder.CreateGEP(IntType, Ptr, {ConstantInt::get(IndexTy, 1)});
3479 Value *LocalElemPtr =
3480 Builder.CreateGEP(IntType, ElemPtr, {ConstantInt::get(IndexTy, 1)});
3488 Value *Res = createRuntimeShuffleFunction(
3489 AllocaIP,
Builder.CreateLoad(IntType, Ptr), IntType,
Offset);
3490 Builder.CreateStore(Res, ElemPtr);
3491 Ptr =
Builder.CreateGEP(IntType, Ptr, {ConstantInt::get(IndexTy, 1)});
3493 Builder.CreateGEP(IntType, ElemPtr, {ConstantInt::get(IndexTy, 1)});
3499Error OpenMPIRBuilder::emitReductionListCopy(
3504 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3505 Value *RemoteLaneOffset = CopyOptions.RemoteLaneOffset;
3509 for (
auto En :
enumerate(ReductionInfos)) {
3511 Value *SrcElementAddr =
nullptr;
3512 AllocaInst *DestAlloca =
nullptr;
3513 Value *DestElementAddr =
nullptr;
3514 Value *DestElementPtrAddr =
nullptr;
3516 bool ShuffleInElement =
false;
3519 bool UpdateDestListPtr =
false;
3523 ReductionArrayTy, SrcBase,
3524 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
3525 SrcElementAddr =
Builder.CreateLoad(
Builder.getPtrTy(), SrcElementPtrAddr);
3529 DestElementPtrAddr =
Builder.CreateInBoundsGEP(
3530 ReductionArrayTy, DestBase,
3531 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
3532 bool IsByRefElem = (!IsByRef.
empty() && IsByRef[En.index()]);
3538 Type *DestAllocaType =
3539 IsByRefElem ? RI.ByRefAllocatedType : RI.ElementType;
3540 DestAlloca =
Builder.CreateAlloca(DestAllocaType,
nullptr,
3541 ".omp.reduction.element");
3543 M.getDataLayout().getPrefTypeAlign(DestAllocaType));
3544 DestElementAddr = DestAlloca;
3547 DestElementAddr->
getName() +
".ascast");
3549 ShuffleInElement =
true;
3550 UpdateDestListPtr =
true;
3562 if (ShuffleInElement) {
3563 Type *ShuffleType = RI.ElementType;
3564 Value *ShuffleSrcAddr = SrcElementAddr;
3565 Value *ShuffleDestAddr = DestElementAddr;
3566 AllocaInst *LocalStorage =
nullptr;
3569 assert(RI.ByRefElementType &&
"Expected by-ref element type to be set");
3570 assert(RI.ByRefAllocatedType &&
3571 "Expected by-ref allocated type to be set");
3576 ShuffleType = RI.ByRefElementType;
3578 if (RI.DataPtrPtrGen) {
3581 Builder.saveIP(), ShuffleSrcAddr, ShuffleSrcAddr);
3584 return GenResult.takeError();
3593 LocalStorage =
Builder.CreateAlloca(ShuffleType);
3595 ShuffleDestAddr = LocalStorage;
3600 ShuffleDestAddr = DestElementAddr;
3604 shuffleAndStore(AllocaIP, ShuffleSrcAddr, ShuffleDestAddr, ShuffleType,
3605 RemoteLaneOffset, ReductionArrayTy, IsByRefElem);
3607 if (IsByRefElem && RI.DataPtrPtrGen) {
3609 Value *DestDescriptorAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3610 DestAlloca,
Builder.getPtrTy(),
".ascast");
3613 DestDescriptorAddr, LocalStorage, SrcElementAddr,
3614 RI.ByRefAllocatedType, RI.DataPtrPtrGen);
3617 return GenResult.takeError();
3620 switch (RI.EvaluationKind) {
3622 Value *Elem =
Builder.CreateLoad(RI.ElementType, SrcElementAddr);
3624 Builder.CreateStore(Elem, DestElementAddr);
3628 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
3629 RI.ElementType, SrcElementAddr, 0, 0,
".realp");
3631 RI.ElementType->getStructElementType(0), SrcRealPtr,
".real");
3633 RI.ElementType, SrcElementAddr, 0, 1,
".imagp");
3635 RI.ElementType->getStructElementType(1), SrcImgPtr,
".imag");
3637 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
3638 RI.ElementType, DestElementAddr, 0, 0,
".realp");
3639 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
3640 RI.ElementType, DestElementAddr, 0, 1,
".imagp");
3641 Builder.CreateStore(SrcReal, DestRealPtr);
3642 Builder.CreateStore(SrcImg, DestImgPtr);
3647 M.getDataLayout().getTypeStoreSize(RI.ElementType));
3649 DestElementAddr,
M.getDataLayout().getPrefTypeAlign(RI.ElementType),
3650 SrcElementAddr,
M.getDataLayout().getPrefTypeAlign(RI.ElementType),
3662 if (UpdateDestListPtr) {
3663 Value *CastDestAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3664 DestElementAddr,
Builder.getPtrTy(),
3665 DestElementAddr->
getName() +
".ascast");
3666 Builder.CreateStore(CastDestAddr, DestElementPtrAddr);
3673Expected<Function *> OpenMPIRBuilder::emitInterWarpCopyFunction(
3676 IRBuilder<>::InsertPointGuard IPG(
Builder);
3677 LLVMContext &Ctx =
M.getContext();
3679 Builder.getVoidTy(), {Builder.getPtrTy(), Builder.getInt32Ty()},
3683 "_omp_reduction_inter_warp_copy_func", &
M);
3689 Builder.SetInsertPoint(EntryBB);
3707 StringRef TransferMediumName =
3708 "__openmp_nvptx_data_transfer_temporary_storage";
3709 GlobalVariable *TransferMedium =
M.getGlobalVariable(TransferMediumName);
3710 unsigned WarpSize =
Config.getGridValue().GV_Warp_Size;
3712 if (!TransferMedium) {
3713 TransferMedium =
new GlobalVariable(
3721 Value *GPUThreadID = getGPUThreadID();
3723 Value *LaneID = getNVPTXLaneID();
3725 Value *WarpID = getNVPTXWarpID();
3729 Builder.GetInsertBlock()->getFirstInsertionPt());
3733 AllocaInst *ReduceListAlloca =
Builder.CreateAlloca(
3734 Arg0Type,
nullptr, ReduceListArg->
getName() +
".addr");
3735 AllocaInst *NumWarpsAlloca =
3736 Builder.CreateAlloca(Arg1Type,
nullptr, NumWarpsArg->
getName() +
".addr");
3737 Value *ReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3738 ReduceListAlloca, Arg0Type, ReduceListAlloca->
getName() +
".ascast");
3739 Value *NumWarpsAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3740 NumWarpsAlloca,
Builder.getPtrTy(0),
3741 NumWarpsAlloca->
getName() +
".ascast");
3742 Builder.CreateStore(ReduceListArg, ReduceListAddrCast);
3743 Builder.CreateStore(NumWarpsArg, NumWarpsAddrCast);
3752 for (
auto En :
enumerate(ReductionInfos)) {
3758 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
3759 unsigned RealTySize =
M.getDataLayout().getTypeAllocSize(
3760 IsByRefElem ? RI.ByRefElementType : RI.ElementType);
3761 for (
unsigned TySize = 4; TySize > 0 && RealTySize > 0; TySize /= 2) {
3764 unsigned NumIters = RealTySize / TySize;
3767 Value *Cnt =
nullptr;
3768 Value *CntAddr =
nullptr;
3775 Builder.CreateAlloca(
Builder.getInt32Ty(),
nullptr,
".cnt.addr");
3777 CntAddr =
Builder.CreateAddrSpaceCast(CntAddr,
Builder.getPtrTy(),
3778 CntAddr->
getName() +
".ascast");
3790 Cnt, ConstantInt::get(
Builder.getInt32Ty(), NumIters));
3791 Builder.CreateCondBr(Cmp, BodyBB, ExitBB);
3798 omp::Directive::OMPD_unknown,
3802 return BarrierIP1.takeError();
3808 Value *IsWarpMaster =
Builder.CreateIsNull(LaneID,
"warp_master");
3809 Builder.CreateCondBr(IsWarpMaster, ThenBB, ElseBB);
3813 auto *RedListArrayTy =
3816 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
3818 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
3819 {ConstantInt::get(IndexTy, 0),
3820 ConstantInt::get(IndexTy, En.index())});
3824 if (IsByRefElem && RI.DataPtrPtrGen) {
3826 RI.DataPtrPtrGen(
Builder.saveIP(), ElemPtr, ElemPtr);
3829 return GenRes.takeError();
3840 ArrayTy, TransferMedium, {
Builder.getInt64(0), WarpID});
3845 Builder.CreateStore(Elem, MediumPtr,
3857 omp::Directive::OMPD_unknown,
3861 return BarrierIP2.takeError();
3868 Value *NumWarpsVal =
3871 Value *IsActiveThread =
3872 Builder.CreateICmpULT(GPUThreadID, NumWarpsVal,
"is_active_thread");
3873 Builder.CreateCondBr(IsActiveThread, W0ThenBB, W0ElseBB);
3880 ArrayTy, TransferMedium, {
Builder.getInt64(0), GPUThreadID});
3882 Value *TargetElemPtrPtr =
3883 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
3884 {ConstantInt::get(IndexTy, 0),
3885 ConstantInt::get(IndexTy, En.index())});
3886 Value *TargetElemPtrVal =
3888 Value *TargetElemPtr = TargetElemPtrVal;
3890 if (IsByRefElem && RI.DataPtrPtrGen) {
3892 RI.DataPtrPtrGen(
Builder.saveIP(), TargetElemPtr, TargetElemPtr);
3895 return GenRes.takeError();
3897 TargetElemPtr =
Builder.CreateLoad(
Builder.getPtrTy(), TargetElemPtr);
3905 Value *SrcMediumValue =
3906 Builder.CreateLoad(CType, SrcMediumPtrVal,
true);
3907 Builder.CreateStore(SrcMediumValue, TargetElemPtr);
3917 Cnt, ConstantInt::get(
Builder.getInt32Ty(), 1));
3918 Builder.CreateStore(Cnt, CntAddr,
false);
3920 auto *CurFn =
Builder.GetInsertBlock()->getParent();
3924 RealTySize %= TySize;
3933Expected<Function *> OpenMPIRBuilder::emitShuffleAndReduceFunction(
3936 LLVMContext &Ctx =
M.getContext();
3937 IRBuilder<>::InsertPointGuard IPG(
Builder);
3938 FunctionType *FuncTy =
3940 {Builder.getPtrTy(), Builder.getInt16Ty(),
3941 Builder.getInt16Ty(), Builder.getInt16Ty()},
3945 "_omp_reduction_shuffle_and_reduce_func", &
M);
3956 Builder.SetInsertPoint(EntryBB);
3968 Type *ReduceListArgType = ReduceListArg->
getType();
3972 ReduceListArgType,
nullptr, ReduceListArg->
getName() +
".addr");
3973 Value *LaneIdAlloca =
Builder.CreateAlloca(LaneIDArgType,
nullptr,
3974 LaneIDArg->
getName() +
".addr");
3976 LaneIDArgType,
nullptr, RemoteLaneOffsetArg->
getName() +
".addr");
3977 Value *AlgoVerAlloca =
Builder.CreateAlloca(LaneIDArgType,
nullptr,
3978 AlgoVerArg->
getName() +
".addr");
3985 RedListArrayTy,
nullptr,
".omp.reduction.remote_reduce_list");
3987 Value *ReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3988 ReduceListAlloca, ReduceListArgType,
3989 ReduceListAlloca->
getName() +
".ascast");
3990 Value *LaneIdAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3991 LaneIdAlloca, LaneIDArgPtrType, LaneIdAlloca->
getName() +
".ascast");
3992 Value *RemoteLaneOffsetAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3993 RemoteLaneOffsetAlloca, LaneIDArgPtrType,
3994 RemoteLaneOffsetAlloca->
getName() +
".ascast");
3995 Value *AlgoVerAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3996 AlgoVerAlloca, LaneIDArgPtrType, AlgoVerAlloca->
getName() +
".ascast");
3997 Value *RemoteListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
3998 RemoteReductionListAlloca,
Builder.getPtrTy(),
3999 RemoteReductionListAlloca->
getName() +
".ascast");
4001 Builder.CreateStore(ReduceListArg, ReduceListAddrCast);
4002 Builder.CreateStore(LaneIDArg, LaneIdAddrCast);
4003 Builder.CreateStore(RemoteLaneOffsetArg, RemoteLaneOffsetAddrCast);
4004 Builder.CreateStore(AlgoVerArg, AlgoVerAddrCast);
4006 Value *ReduceList =
Builder.CreateLoad(ReduceListArgType, ReduceListAddrCast);
4007 Value *LaneId =
Builder.CreateLoad(LaneIDArgType, LaneIdAddrCast);
4008 Value *RemoteLaneOffset =
4009 Builder.CreateLoad(LaneIDArgType, RemoteLaneOffsetAddrCast);
4010 Value *AlgoVer =
Builder.CreateLoad(LaneIDArgType, AlgoVerAddrCast);
4017 Error EmitRedLsCpRes = emitReductionListCopy(
4019 ReduceList, RemoteListAddrCast, IsByRef,
4020 {RemoteLaneOffset,
nullptr,
nullptr});
4023 return EmitRedLsCpRes;
4048 Value *LaneComp =
Builder.CreateICmpULT(LaneId, RemoteLaneOffset);
4053 Value *Algo2AndLaneIdComp =
Builder.CreateAnd(Algo2, LaneIdComp);
4054 Value *RemoteOffsetComp =
4056 Value *CondAlgo2 =
Builder.CreateAnd(Algo2AndLaneIdComp, RemoteOffsetComp);
4057 Value *CA0OrCA1 =
Builder.CreateOr(CondAlgo0, CondAlgo1);
4058 Value *CondReduce =
Builder.CreateOr(CA0OrCA1, CondAlgo2);
4064 Builder.CreateCondBr(CondReduce, ThenBB, ElseBB);
4066 Value *LocalReduceListPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4067 ReduceList,
Builder.getPtrTy());
4068 Value *RemoteReduceListPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4069 RemoteListAddrCast,
Builder.getPtrTy());
4071 ->addFnAttr(Attribute::NoUnwind);
4082 Value *LaneIdGtOffset =
Builder.CreateICmpUGE(LaneId, RemoteLaneOffset);
4083 Value *CondCopy =
Builder.CreateAnd(Algo1, LaneIdGtOffset);
4088 Builder.CreateCondBr(CondCopy, CpyThenBB, CpyElseBB);
4092 EmitRedLsCpRes = emitReductionListCopy(
4094 RemoteListAddrCast, ReduceList, IsByRef);
4097 return EmitRedLsCpRes;
4112OpenMPIRBuilder::generateReductionDescriptor(
4114 Type *DescriptorType,
4120 Value *DescriptorSize =
4121 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(DescriptorType));
4123 DescriptorAddr,
M.getDataLayout().getPrefTypeAlign(DescriptorType),
4124 SrcDescriptorAddr,
M.getDataLayout().getPrefTypeAlign(DescriptorType),
4128 Value *DataPtrField;
4130 DataPtrPtrGen(
Builder.saveIP(), DescriptorAddr, DataPtrField);
4133 return GenResult.takeError();
4136 DataPtr,
Builder.getPtrTy(),
".ascast"),
4142Expected<Value *> OpenMPIRBuilder::createReductionDescriptorCopy(
4144 Value *SrcDescriptorAddr,
Type *DescriptorPtrTy,
const Twine &Name) {
4148 AllocaInst *DescriptorAlloca =
4149 Builder.CreateAlloca(RI.ByRefAllocatedType,
nullptr, Name);
4151 M.getDataLayout().getPrefTypeAlign(RI.ByRefAllocatedType));
4152 Value *DescriptorAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4153 DescriptorAlloca, DescriptorPtrTy,
4154 DescriptorAlloca->
getName() +
".ascast");
4159 generateReductionDescriptor(DescriptorAddr, DataPtr, SrcDescriptorAddr,
4160 RI.ByRefAllocatedType, RI.DataPtrPtrGen);
4162 return GenResult.takeError();
4164 return DescriptorAddr;
4167Expected<Function *> OpenMPIRBuilder::emitListToGlobalCopyFunction(
4170 IRBuilder<>::InsertPointGuard IPG(
Builder);
4171 LLVMContext &Ctx =
M.getContext();
4174 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4178 "_omp_reduction_list_to_global_copy_func", &
M);
4185 Builder.SetInsertPoint(EntryBlock);
4196 BufferArg->
getName() +
".addr");
4200 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4201 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4202 BufferArgAlloca,
Builder.getPtrTy(),
4203 BufferArgAlloca->
getName() +
".ascast");
4204 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4205 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4206 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4207 ReduceListArgAlloca,
Builder.getPtrTy(),
4208 ReduceListArgAlloca->
getName() +
".ascast");
4210 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4211 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4212 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4214 Value *LocalReduceList =
4216 Value *BufferArgVal =
4220 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4221 for (
auto En :
enumerate(ReductionInfos)) {
4223 auto *RedListArrayTy =
4227 RedListArrayTy, LocalReduceList,
4228 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4234 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferArgVal, Idxs);
4236 ReductionsBufferTy, BufferVD, 0, En.index());
4238 switch (RI.EvaluationKind) {
4240 Value *TargetElement;
4242 if (IsByRef.
empty() || !IsByRef[En.index()]) {
4243 TargetElement =
Builder.CreateLoad(RI.ElementType, ElemPtr);
4245 if (RI.DataPtrPtrGen) {
4247 RI.DataPtrPtrGen(
Builder.saveIP(), ElemPtr, ElemPtr);
4250 return GenResult.takeError();
4254 TargetElement =
Builder.CreateLoad(RI.ByRefElementType, ElemPtr);
4257 Builder.CreateStore(TargetElement, GlobVal);
4261 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4262 RI.ElementType, ElemPtr, 0, 0,
".realp");
4264 RI.ElementType->getStructElementType(0), SrcRealPtr,
".real");
4266 RI.ElementType, ElemPtr, 0, 1,
".imagp");
4268 RI.ElementType->getStructElementType(1), SrcImgPtr,
".imag");
4270 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4271 RI.ElementType, GlobVal, 0, 0,
".realp");
4272 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
4273 RI.ElementType, GlobVal, 0, 1,
".imagp");
4274 Builder.CreateStore(SrcReal, DestRealPtr);
4275 Builder.CreateStore(SrcImg, DestImgPtr);
4280 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(RI.ElementType));
4282 GlobVal,
M.getDataLayout().getPrefTypeAlign(RI.ElementType), ElemPtr,
4283 M.getDataLayout().getPrefTypeAlign(RI.ElementType), SizeVal,
false);
4293Expected<Function *> OpenMPIRBuilder::emitListToGlobalReduceFunction(
4296 IRBuilder<>::InsertPointGuard IPG(
Builder);
4297 LLVMContext &Ctx =
M.getContext();
4300 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4304 "_omp_reduction_list_to_global_reduce_func", &
M);
4311 Builder.SetInsertPoint(EntryBlock);
4322 BufferArg->
getName() +
".addr");
4326 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4327 auto *RedListArrayTy =
4332 Value *LocalReduceList =
4333 Builder.CreateAlloca(RedListArrayTy,
nullptr,
".omp.reduction.red_list");
4337 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4338 BufferArgAlloca,
Builder.getPtrTy(),
4339 BufferArgAlloca->
getName() +
".ascast");
4340 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4341 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4342 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4343 ReduceListArgAlloca,
Builder.getPtrTy(),
4344 ReduceListArgAlloca->
getName() +
".ascast");
4345 Value *LocalReduceListAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4346 LocalReduceList,
Builder.getPtrTy(),
4347 LocalReduceList->
getName() +
".ascast");
4349 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4350 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4351 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4356 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4357 for (
auto En :
enumerate(ReductionInfos)) {
4360 Value *TargetElementPtrPtr =
Builder.CreateInBoundsGEP(
4361 RedListArrayTy, LocalReduceListAddrCast,
4362 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4364 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4366 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4367 ReductionsBufferTy, BufferVD, 0, En.index());
4369 if (!IsByRef.
empty() && IsByRef[En.index()] && RI.DataPtrPtrGen) {
4373 Value *SrcElementPtrPtr =
4374 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceList,
4375 {ConstantInt::get(IndexTy, 0),
4376 ConstantInt::get(IndexTy, En.index())});
4377 Value *SrcDescriptorAddr =
4381 Expected<Value *> ByRefAlloc = createReductionDescriptorCopy(
4382 AllocaIP, RI, GlobValPtr, SrcDescriptorAddr,
Builder.getPtrTy());
4386 Builder.CreateStore(*ByRefAlloc, TargetElementPtrPtr);
4388 Builder.CreateStore(GlobValPtr, TargetElementPtrPtr);
4396 ->addFnAttr(Attribute::NoUnwind);
4401Expected<Function *> OpenMPIRBuilder::emitGlobalToListCopyFunction(
4404 IRBuilder<>::InsertPointGuard IPG(
Builder);
4405 LLVMContext &Ctx =
M.getContext();
4408 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4412 "_omp_reduction_global_to_list_copy_func", &
M);
4419 Builder.SetInsertPoint(EntryBlock);
4430 BufferArg->
getName() +
".addr");
4434 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4435 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4436 BufferArgAlloca,
Builder.getPtrTy(),
4437 BufferArgAlloca->
getName() +
".ascast");
4438 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4439 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4440 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4441 ReduceListArgAlloca,
Builder.getPtrTy(),
4442 ReduceListArgAlloca->
getName() +
".ascast");
4443 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4444 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4445 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4447 Value *LocalReduceList =
4452 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4453 for (
auto En :
enumerate(ReductionInfos)) {
4454 const OpenMPIRBuilder::ReductionInfo &RI = En.value();
4455 auto *RedListArrayTy =
4459 RedListArrayTy, LocalReduceList,
4460 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4465 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4466 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4467 ReductionsBufferTy, BufferVD, 0, En.index());
4473 if (!IsByRef.
empty() && IsByRef[En.index()]) {
4480 return GenResult.takeError();
4486 Value *TargetElement =
Builder.CreateLoad(ElemType, GlobValPtr);
4487 Builder.CreateStore(TargetElement, ElemPtr);
4491 Value *SrcRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4500 Value *DestRealPtr =
Builder.CreateConstInBoundsGEP2_32(
4502 Value *DestImgPtr =
Builder.CreateConstInBoundsGEP2_32(
4504 Builder.CreateStore(SrcReal, DestRealPtr);
4505 Builder.CreateStore(SrcImg, DestImgPtr);
4512 ElemPtr,
M.getDataLayout().getPrefTypeAlign(RI.
ElementType),
4513 GlobValPtr,
M.getDataLayout().getPrefTypeAlign(RI.
ElementType),
4524Expected<Function *> OpenMPIRBuilder::emitGlobalToListReduceFunction(
4527 IRBuilder<>::InsertPointGuard IPG(
Builder);
4528 LLVMContext &Ctx =
M.getContext();
4531 {Builder.getPtrTy(), Builder.getInt32Ty(), Builder.getPtrTy()},
4535 "_omp_reduction_global_to_list_reduce_func", &
M);
4542 Builder.SetInsertPoint(EntryBlock);
4553 BufferArg->
getName() +
".addr");
4557 Builder.getPtrTy(),
nullptr, ReduceListArg->
getName() +
".addr");
4563 Value *LocalReduceList =
4564 Builder.CreateAlloca(RedListArrayTy,
nullptr,
".omp.reduction.red_list");
4568 Value *BufferArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4569 BufferArgAlloca,
Builder.getPtrTy(),
4570 BufferArgAlloca->
getName() +
".ascast");
4571 Value *IdxArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4572 IdxArgAlloca,
Builder.getPtrTy(), IdxArgAlloca->
getName() +
".ascast");
4573 Value *ReduceListArgAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4574 ReduceListArgAlloca,
Builder.getPtrTy(),
4575 ReduceListArgAlloca->
getName() +
".ascast");
4576 Value *ReductionList =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4577 LocalReduceList,
Builder.getPtrTy(),
4578 LocalReduceList->
getName() +
".ascast");
4580 Builder.CreateStore(BufferArg, BufferArgAddrCast);
4581 Builder.CreateStore(IdxArg, IdxArgAddrCast);
4582 Builder.CreateStore(ReduceListArg, ReduceListArgAddrCast);
4587 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4588 for (
auto En :
enumerate(ReductionInfos)) {
4591 Value *TargetElementPtrPtr =
Builder.CreateInBoundsGEP(
4592 RedListArrayTy, ReductionList,
4593 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4596 Builder.CreateInBoundsGEP(ReductionsBufferTy, BufferVal, Idxs);
4597 Value *GlobValPtr =
Builder.CreateConstInBoundsGEP2_32(
4598 ReductionsBufferTy, BufferVD, 0, En.index());
4600 if (!IsByRef.
empty() && IsByRef[En.index()] && RI.DataPtrPtrGen) {
4602 Value *ReduceListVal =
4604 Value *SrcElementPtrPtr =
4605 Builder.CreateInBoundsGEP(RedListArrayTy, ReduceListVal,
4606 {ConstantInt::get(IndexTy, 0),
4607 ConstantInt::get(IndexTy, En.index())});
4608 Value *SrcDescriptorAddr =
4612 Expected<Value *> ByRefAlloc = createReductionDescriptorCopy(
4613 AllocaIP, RI, GlobValPtr, SrcDescriptorAddr,
Builder.getPtrTy());
4617 Builder.CreateStore(*ByRefAlloc, TargetElementPtrPtr);
4619 Builder.CreateStore(GlobValPtr, TargetElementPtrPtr);
4627 ->addFnAttr(Attribute::NoUnwind);
4632std::string OpenMPIRBuilder::getReductionFuncName(StringRef Name)
const {
4633 std::string Suffix =
4635 return (Name + Suffix).str();
4638Expected<Function *> OpenMPIRBuilder::createReductionFunction(
4641 AttributeList FuncAttrs) {
4642 IRBuilder<>::InsertPointGuard IPG(
Builder);
4644 {Builder.getPtrTy(), Builder.getPtrTy()},
4646 std::string
Name = getReductionFuncName(ReducerName);
4655 Builder.SetInsertPoint(EntryBB);
4660 Value *LHSArrayPtr =
nullptr;
4661 Value *RHSArrayPtr =
nullptr;
4668 Builder.CreateAlloca(Arg0Type,
nullptr, Arg0->
getName() +
".addr");
4670 Builder.CreateAlloca(Arg1Type,
nullptr, Arg1->
getName() +
".addr");
4671 Value *LHSAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4672 LHSAlloca, Arg0Type, LHSAlloca->
getName() +
".ascast");
4673 Value *RHSAddrCast =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4674 RHSAlloca, Arg1Type, RHSAlloca->
getName() +
".ascast");
4675 Builder.CreateStore(Arg0, LHSAddrCast);
4676 Builder.CreateStore(Arg1, RHSAddrCast);
4677 LHSArrayPtr =
Builder.CreateLoad(Arg0Type, LHSAddrCast);
4678 RHSArrayPtr =
Builder.CreateLoad(Arg1Type, RHSAddrCast);
4682 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4684 for (
auto En :
enumerate(ReductionInfos)) {
4687 RedArrayTy, RHSArrayPtr,
4688 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4690 Value *RHSPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4691 RHSI8Ptr, RI.PrivateVariable->getType(),
4692 RHSI8Ptr->
getName() +
".ascast");
4695 RedArrayTy, LHSArrayPtr,
4696 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4698 Value *LHSPtr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4699 LHSI8Ptr, RI.Variable->getType(), LHSI8Ptr->
getName() +
".ascast");
4708 if (!IsByRef.
empty() && !IsByRef[En.index()]) {
4709 LHS =
Builder.CreateLoad(RI.ElementType, LHSPtr);
4710 RHS =
Builder.CreateLoad(RI.ElementType, RHSPtr);
4717 return AfterIP.takeError();
4718 if (!
Builder.GetInsertBlock())
4719 return ReductionFunc;
4723 if (!IsByRef.
empty() && !IsByRef[En.index()])
4724 Builder.CreateStore(Reduced, LHSPtr);
4729 for (
auto En :
enumerate(ReductionInfos)) {
4730 unsigned Index = En.index();
4732 Value *LHSFixupPtr, *RHSFixupPtr;
4733 Builder.restoreIP(RI.ReductionGenClang(
4734 Builder.saveIP(), Index, &LHSFixupPtr, &RHSFixupPtr, ReductionFunc));
4739 LHSPtrs[Index], [ReductionFunc](
const Use &U) {
4744 RHSPtrs[Index], [ReductionFunc](
const Use &U) {
4758 return ReductionFunc;
4766 assert(RI.Variable &&
"expected non-null variable");
4767 assert(RI.PrivateVariable &&
"expected non-null private variable");
4768 assert((RI.ReductionGen || RI.ReductionGenClang) &&
4769 "expected non-null reduction generator callback");
4772 RI.Variable->getType() == RI.PrivateVariable->getType() &&
4773 "expected variables and their private equivalents to have the same "
4776 assert(RI.Variable->getType()->isPointerTy() &&
4777 "expected variables to be pointers");
4794 ArrayRef<bool> IsByRef,
bool IsNoWait,
bool IsTeamsReduction,
bool IsSPMD,
4796 Value *SrcLocInfo) {
4810 if (ReductionInfos.
size() == 0)
4820 Builder.SetInsertPoint(InsertBlock, InsertBlock->
end());
4824 AttributeList FuncAttrs;
4825 AttrBuilder AttrBldr(Ctx);
4827 AttrBldr.addAttribute(Attr);
4828 AttrBldr.removeAttribute(Attribute::OptimizeNone);
4829 FuncAttrs = FuncAttrs.addFnAttributes(Ctx, AttrBldr);
4833 Builder.GetInsertBlock()->getParent()->getName(), ReductionInfos, IsByRef,
4835 if (!ReductionResult)
4837 Function *ReductionFunc = *ReductionResult;
4841 if (GridValue.has_value())
4842 Config.setGridValue(GridValue.value());
4857 Builder.getPtrTy(
M.getDataLayout().getProgramAddressSpace());
4861 Value *ReductionListAlloca =
4862 Builder.CreateAlloca(RedArrayTy,
nullptr,
".omp.reduction.red_list");
4863 Value *ReductionList =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4864 ReductionListAlloca, PtrTy, ReductionListAlloca->
getName() +
".ascast");
4867 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
4868 for (
auto En :
enumerate(ReductionInfos)) {
4871 RedArrayTy, ReductionList,
4872 {ConstantInt::get(IndexTy, 0), ConstantInt::get(IndexTy, En.index())});
4875 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
4880 Builder.CreatePointerBitCastOrAddrSpaceCast(PrivateVar, PtrTy);
4881 Builder.CreateStore(CastElem, ElemPtr);
4885 ReductionInfos, ReductionFunc, FuncAttrs, IsByRef);
4891 emitInterWarpCopyFunction(
Loc, ReductionInfos, FuncAttrs, IsByRef);
4897 Value *RL =
Builder.CreatePointerBitCastOrAddrSpaceCast(ReductionList, PtrTy);
4906 unsigned MaxDataSize = 0;
4908 for (
auto En :
enumerate(ReductionInfos)) {
4912 Type *RedTypeArg = (!IsByRef.
empty() && IsByRef[En.index()])
4913 ? En.value().ByRefElementType
4914 : En.value().ElementType;
4915 auto Size =
M.getDataLayout().getTypeStoreSize(RedTypeArg);
4916 if (
Size > MaxDataSize)
4920 Value *ReductionDataSize =
4921 Builder.getInt64(MaxDataSize * ReductionInfos.
size());
4925 Function *CopyScratchToListFunc =
nullptr;
4927 Value *ScratchForCopyBack =
nullptr;
4930 Value *RLForCopyBack = RL;
4932 bool IsAtomicReduction =
4935 if (!IsTeamsReduction) {
4936 Value *SarFuncCast =
4937 Builder.CreatePointerBitCastOrAddrSpaceCast(*SarFunc, FuncPtrTy);
4939 Builder.CreatePointerBitCastOrAddrSpaceCast(WcFunc, FuncPtrTy);
4940 Value *Args[] = {SrcLocInfo, ReductionDataSize, RL, SarFuncCast,
4943 RuntimeFunction::OMPRTL___kmpc_nvptx_parallel_reduce_nowait_v2);
4945 }
else if (IsAtomicReduction) {
4949 RuntimeFunction::OMPRTL___kmpc_is_team_main_thread);
4954 Ctx, ReductionTypeArgs,
"struct._globalized_locals_ty");
4957 ReductionInfos, ReductionsBufferTy, FuncAttrs, IsByRef);
4962 ReductionInfos, ReductionsBufferTy, FuncAttrs, IsByRef);
4967 ReductionInfos, ReductionFunc, ReductionsBufferTy, FuncAttrs, IsByRef);
4990 Value *RuntimeRL = RL;
4997 ReductionsBufferTy,
nullptr,
".omp.reduction.scratch");
4998 Value *PerThreadScratch =
Builder.CreatePointerBitCastOrAddrSpaceCast(
4999 PerThreadScratchAlloca, PtrTy,
5000 PerThreadScratchAlloca->
getName() +
".ascast");
5003 Value *PerThreadRedListAlloca =
5004 Builder.CreateAlloca(RedArrayTy,
nullptr,
5005 ".omp.reduction.per_thread_red_list");
5006 RuntimeRL =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5007 PerThreadRedListAlloca, PtrTy,
5008 PerThreadRedListAlloca->
getName() +
".ascast");
5013 for (
auto En :
enumerate(ReductionInfos)) {
5015 bool IsByRefElem = !IsByRef.
empty() && IsByRef[En.index()];
5018 ReductionsBufferTy, PerThreadScratch, 0, En.index());
5019 Value *Slot =
Builder.CreateConstInBoundsGEP2_32(RedArrayTy, RuntimeRL,
5022 Value *RuntimeListEntry = FieldPtr;
5024 Value *SrcDescriptor =
5027 AllocaIP, RI, FieldPtr, SrcDescriptor, PtrTy);
5030 RuntimeListEntry = *Descriptor;
5032 Builder.CreateStore(RuntimeListEntry, Slot);
5038 Type *CopyArg0Ty = (*LtGCFunc)->getFunctionType()->getParamType(0);
5039 Type *CopyArg2Ty = (*LtGCFunc)->getFunctionType()->getParamType(2);
5040 ScratchForCopyBack =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5041 PerThreadScratch, CopyArg0Ty);
5043 Builder.CreatePointerBitCastOrAddrSpaceCast(RL, CopyArg2Ty);
5051 *LtGCFunc, {ScratchForCopyBack,
Builder.getInt32(0), RLForCopyBack});
5052 CopyScratchToListFunc = *GtLCFunc;
5055 Value *Args3[] = {SrcLocInfo, RuntimeRL, *SarFunc, WcFunc,
5056 *LtGCFunc, *GtLCFunc, *GtLRFunc};
5059 RuntimeFunction::OMPRTL___kmpc_gpu_xteam_reduce_nowait);
5079 if (ScratchForCopyBack) {
5082 CopyScratchToListFunc,
5083 {ScratchForCopyBack,
Builder.getInt32(0), RLForCopyBack});
5087 for (
auto En :
enumerate(ReductionInfos)) {
5093 if (IsAtomicReduction) {
5109 Value *LHSPtr, *RHSPtr;
5111 &LHSPtr, &RHSPtr, CurFunc));
5117 RedValue =
Builder.CreatePointerBitCastOrAddrSpaceCast(
5119 if (RHSPtr->
getType() != RHS->getType())
5121 Builder.CreatePointerBitCastOrAddrSpaceCast(RHS, RHSPtr->
getType());
5132 if (IsByRef.
empty() || !IsByRef[En.index()]) {
5134 "red.value." +
Twine(En.index()));
5145 if (!IsByRef.
empty() && !IsByRef[En.index()])
5150 if (ContinuationBlock) {
5151 Builder.CreateBr(ContinuationBlock);
5152 Builder.SetInsertPoint(ContinuationBlock);
5154 Config.setEmitLLVMUsed();
5165 ".omp.reduction.func", &M);
5176 Builder.SetInsertPoint(ReductionFuncBlock);
5178 Value *LHSArrayPtr =
nullptr;
5179 Value *RHSArrayPtr =
nullptr;
5190 Builder.CreateAlloca(Arg0Type,
nullptr, Arg0->
getName() +
".addr");
5192 Builder.CreateAlloca(Arg1Type,
nullptr, Arg1->
getName() +
".addr");
5193 Value *LHSAddrCast =
5194 Builder.CreatePointerBitCastOrAddrSpaceCast(LHSAlloca, Arg0Type);
5195 Value *RHSAddrCast =
5196 Builder.CreatePointerBitCastOrAddrSpaceCast(RHSAlloca, Arg1Type);
5197 Builder.CreateStore(Arg0, LHSAddrCast);
5198 Builder.CreateStore(Arg1, RHSAddrCast);
5199 LHSArrayPtr = Builder.CreateLoad(Arg0Type, LHSAddrCast);
5200 RHSArrayPtr = Builder.CreateLoad(Arg1Type, RHSAddrCast);
5202 LHSArrayPtr = ReductionFunc->
getArg(0);
5203 RHSArrayPtr = ReductionFunc->
getArg(1);
5206 unsigned NumReductions = ReductionInfos.
size();
5209 for (
auto En :
enumerate(ReductionInfos)) {
5211 Value *LHSI8PtrPtr = Builder.CreateConstInBoundsGEP2_64(
5212 RedArrayTy, LHSArrayPtr, 0, En.index());
5213 Value *LHSI8Ptr = Builder.CreateLoad(Builder.getPtrTy(), LHSI8PtrPtr);
5214 Value *LHSPtr = Builder.CreatePointerBitCastOrAddrSpaceCast(
5217 Value *RHSI8PtrPtr = Builder.CreateConstInBoundsGEP2_64(
5218 RedArrayTy, RHSArrayPtr, 0, En.index());
5219 Value *RHSI8Ptr = Builder.CreateLoad(Builder.getPtrTy(), RHSI8PtrPtr);
5220 Value *RHSPtr = Builder.CreatePointerBitCastOrAddrSpaceCast(
5229 Builder.restoreIP(*AfterIP);
5231 if (!Builder.GetInsertBlock())
5235 if (!IsByRef[En.index()])
5236 Builder.CreateStore(Reduced, LHSPtr);
5238 Builder.CreateRetVoid();
5245 bool IsNoWait,
bool IsTeamsReduction) {
5249 IsByRef, IsNoWait, IsTeamsReduction);
5256 if (ReductionInfos.
size() == 0)
5266 unsigned NumReductions = ReductionInfos.
size();
5269 Value *RedArray =
Builder.CreateAlloca(RedArrayTy,
nullptr,
"red.array");
5271 Builder.SetInsertPoint(InsertBlock, InsertBlock->
end());
5273 for (
auto En :
enumerate(ReductionInfos)) {
5274 unsigned Index = En.index();
5276 Value *RedArrayElemPtr =
Builder.CreateConstInBoundsGEP2_64(
5277 RedArrayTy, RedArray, 0, Index,
"red.array.elem." +
Twine(Index));
5284 M.getDataLayout(),
M.getDataLayout().getDefaultGlobalsAddressSpace());
5294 ? IdentFlag::OMP_IDENT_FLAG_ATOMIC_REDUCE
5299 unsigned RedArrayByteSize =
DL.getTypeStoreSize(RedArrayTy);
5300 Constant *RedArraySize = ConstantInt::get(IndexTy, RedArrayByteSize);
5302 Value *Lock = getOMPCriticalRegionLock(
".reduction");
5304 IsNoWait ? RuntimeFunction::OMPRTL___kmpc_reduce_nowait
5305 : RuntimeFunction::OMPRTL___kmpc_reduce);
5308 {Ident, ThreadId, NumVariables, RedArraySize,
5309 RedArray, ReductionFunc, Lock},
5320 Builder.CreateSwitch(ReduceCall, ContinuationBlock, 2);
5321 Switch->addCase(
Builder.getInt32(1), NonAtomicRedBlock);
5322 Switch->addCase(
Builder.getInt32(2), AtomicRedBlock);
5327 Builder.SetInsertPoint(NonAtomicRedBlock);
5328 for (
auto En :
enumerate(ReductionInfos)) {
5334 if (!IsByRef[En.index()]) {
5336 "red.value." +
Twine(En.index()));
5338 Value *PrivateRedValue =
5340 "red.private.value." +
Twine(En.index()));
5348 if (!
Builder.GetInsertBlock())
5351 if (!IsByRef[En.index()])
5355 IsNoWait ? RuntimeFunction::OMPRTL___kmpc_end_reduce_nowait
5356 : RuntimeFunction::OMPRTL___kmpc_end_reduce);
5358 Builder.CreateBr(ContinuationBlock);
5363 Builder.SetInsertPoint(AtomicRedBlock);
5364 if (CanGenerateAtomic &&
llvm::none_of(IsByRef, [](
bool P) {
return P; })) {
5371 if (!
Builder.GetInsertBlock())
5374 Builder.CreateBr(ContinuationBlock);
5387 if (!
Builder.GetInsertBlock())
5390 Builder.SetInsertPoint(ContinuationBlock);
5401 Directive OMPD = Directive::OMPD_master;
5406 Value *Args[] = {Ident, ThreadId};
5414 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
5425 Directive OMPD = Directive::OMPD_masked;
5431 Value *ArgsEnd[] = {Ident, ThreadId};
5439 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
5449 Call->setDoesNotThrow();
5464 bool IsInclusive,
ScanInfo *ScanRedInfo) {
5466 llvm::Error Err = emitScanBasedDirectiveDeclsIR(AllocaIP, ScanVars,
5467 ScanVarsType, ScanRedInfo);
5478 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5481 Type *DestTy = ScanVarsType[i];
5482 Value *Val =
Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5485 Builder.CreateStore(Src, Val);
5490 Builder.GetInsertBlock()->getParent());
5493 IV = ScanRedInfo->
IV;
5496 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5499 Type *DestTy = ScanVarsType[i];
5501 Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5503 Builder.CreateStore(Src, ScanVars[i]);
5517 Builder.GetInsertBlock()->getParent());
5522Error OpenMPIRBuilder::emitScanBasedDirectiveDeclsIR(
5526 Builder.restoreIP(AllocaIP);
5528 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5530 Builder.CreateAlloca(Builder.getPtrTy(),
nullptr,
"vla");
5537 Builder.restoreIP(CodeGenIP);
5539 Builder.CreateAdd(ScanRedInfo->
Span, Builder.getInt32(1));
5540 for (
size_t i = 0; i < ScanVars.
size(); i++) {
5544 Value *Buff = Builder.CreateMalloc(
IntPtrTy, ScanVarsType[i], Allocsize,
5545 AllocSpan,
nullptr,
"arr");
5546 Builder.CreateStore(Buff, (*(ScanRedInfo->
ScanBuffPtrs))[ScanVars[i]]);
5564 Builder.SetInsertPoint(
Builder.GetInsertBlock()->getTerminator());
5573Error OpenMPIRBuilder::emitScanBasedDirectiveFinalsIR(
5579 Value *PrivateVar = RedInfo.PrivateVariable;
5580 Value *OrigVar = RedInfo.Variable;
5584 Type *SrcTy = RedInfo.ElementType;
5589 Builder.CreateStore(Src, OrigVar);
5612 Builder.SetInsertPoint(
Builder.GetInsertBlock()->getTerminator());
5637 Builder.GetInsertBlock()->getModule(),
5644 Builder.GetInsertBlock()->getModule(),
5650 llvm::ConstantInt::get(ScanRedInfo->
Span->
getType(), 1));
5651 Builder.SetInsertPoint(InputBB);
5654 Builder.SetInsertPoint(LoopBB);
5670 Builder.CreateCondBr(CmpI, InnerLoopBB, InnerExitBB);
5672 Builder.SetInsertPoint(InnerLoopBB);
5676 Value *ReductionVal = RedInfo.PrivateVariable;
5679 Type *DestTy = RedInfo.ElementType;
5682 Builder.CreateInBoundsGEP(DestTy, Buff,
IV,
"arrayOffset");
5685 Builder.CreateInBoundsGEP(DestTy, Buff, OffsetIval,
"arrayOffset");
5690 RedInfo.ReductionGen(
Builder.saveIP(), LHS, RHS, Result);
5693 Builder.CreateStore(Result, LHSPtr);
5696 IVal, llvm::ConstantInt::get(
Builder.getInt32Ty(), 1));
5698 CmpI =
Builder.CreateICmpUGE(NextIVal, Pow2K);
5699 Builder.CreateCondBr(CmpI, InnerLoopBB, InnerExitBB);
5702 Counter, llvm::ConstantInt::get(Counter->
getType(), 1));
5708 Builder.CreateCondBr(Cmp, LoopBB, ExitBB);
5729 Error Err = emitScanBasedDirectiveFinalsIR(ReductionInfos, ScanRedInfo);
5736Error OpenMPIRBuilder::emitScanBasedDirectiveIR(
5748 Error Err = InputLoopGen();
5759 Error Err = ScanLoopGen(Builder.saveIP());
5766void OpenMPIRBuilder::createScanBBs(ScanInfo *ScanRedInfo) {
5803 Builder.SetInsertPoint(Preheader);
5806 Builder.SetInsertPoint(Header);
5807 PHINode *IndVarPHI =
Builder.CreatePHI(IndVarTy, 2,
"omp_" + Name +
".iv");
5808 IndVarPHI->
addIncoming(ConstantInt::get(IndVarTy, 0), Preheader);
5813 Builder.CreateICmpULT(IndVarPHI, TripCount,
"omp_" + Name +
".cmp");
5814 Builder.CreateCondBr(Cmp, Body, Exit);
5819 Builder.SetInsertPoint(Latch);
5829 bool HasNSW =
Config.hasNoSignedWrap();
5832 unsigned BitWidth = CI->getType()->getIntegerBitWidth();
5834 if (CI->getValue().ugt(SignedMax))
5836 }
else if (IsCollapsed) {
5841 Builder.CreateAdd(IndVarPHI, ConstantInt::get(IndVarTy, 1),
5842 "omp_" + Name +
".next",
true, HasNSW);
5853 CL->Header = Header;
5872 NextBB, NextBB, Name);
5904 Value *Start,
Value *Stop,
Value *Step,
bool IsSigned,
bool InclusiveStop,
5913 ComputeLoc, Start, Stop, Step, IsSigned, InclusiveStop, Name);
5914 ScanRedInfo->
Span = TripCount;
5920 ScanRedInfo->
IV =
IV;
5921 createScanBBs(ScanRedInfo);
5924 assert(Terminator->getNumSuccessors() == 1);
5925 BasicBlock *ContinueBlock = Terminator->getSuccessor(0);
5928 Builder.GetInsertBlock()->getParent());
5931 Builder.GetInsertBlock()->getParent());
5932 Builder.CreateBr(ContinueBlock);
5938 const auto &&InputLoopGen = [&]() ->
Error {
5940 Builder.saveIP(), BodyGen, Start, Stop, Step, IsSigned, InclusiveStop,
5941 ComputeIP, Name,
true, ScanRedInfo);
5945 Builder.restoreIP((*LoopInfo)->getAfterIP());
5951 InclusiveStop, ComputeIP, Name,
true, ScanRedInfo);
5955 Builder.restoreIP((*LoopInfo)->getAfterIP());
5959 Error Err = emitScanBasedDirectiveIR(InputLoopGen, ScanLoopGen, ScanRedInfo);
5967 bool IsSigned,
bool InclusiveStop,
const Twine &Name) {
5977 assert(IndVarTy == Stop->
getType() &&
"Stop type mismatch");
5978 assert(IndVarTy == Step->
getType() &&
"Step type mismatch");
5982 ConstantInt *Zero = ConstantInt::get(IndVarTy, 0);
5998 Incr =
Builder.CreateSelect(IsNeg,
Builder.CreateNeg(Step), Step);
6001 Span =
Builder.CreateSub(UB, LB,
"",
false,
true);
6005 Span =
Builder.CreateSub(Stop, Start,
"",
true);
6010 Value *CountIfLooping;
6011 if (InclusiveStop) {
6012 CountIfLooping =
Builder.CreateAdd(
Builder.CreateUDiv(Span, Incr), One);
6018 CountIfLooping =
Builder.CreateSelect(OneCmp, One, CountIfTwo);
6021 return Builder.CreateSelect(ZeroCmp, Zero, CountIfLooping,
6022 "omp_" + Name +
".tripcount");
6027 Value *Start,
Value *Stop,
Value *Step,
bool IsSigned,
bool InclusiveStop,
6034 ComputeLoc, Start, Stop, Step, IsSigned, InclusiveStop, Name);
6039 Config.hasNoSignedWrap());
6040 Value *IndVar =
Builder.CreateAdd(Span, Start,
"",
false,
6041 Config.hasNoSignedWrap());
6043 ScanRedInfo->
IV = IndVar;
6044 return BodyGenCB(
Builder.saveIP(), IndVar);
6050 Builder.getCurrentDebugLocation());
6061 unsigned Bitwidth = Ty->getIntegerBitWidth();
6064 M, omp::RuntimeFunction::OMPRTL___kmpc_dist_for_static_init_4u);
6067 M, omp::RuntimeFunction::OMPRTL___kmpc_dist_for_static_init_8u);
6077 unsigned Bitwidth = Ty->getIntegerBitWidth();
6080 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_init_4u);
6083 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_init_8u);
6091 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6093 "Require dedicated allocate IP");
6099 uint32_t SrcLocStrSize;
6103 case WorksharingLoopType::ForStaticLoop:
6104 Flag = OMP_IDENT_FLAG_WORK_LOOP;
6106 case WorksharingLoopType::DistributeStaticLoop:
6107 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6109 case WorksharingLoopType::DistributeForStaticLoop:
6110 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE | OMP_IDENT_FLAG_WORK_LOOP;
6117 Type *IVTy =
IV->getType();
6118 FunctionCallee StaticInit =
6119 LoopType == WorksharingLoopType::DistributeForStaticLoop
6122 FunctionCallee StaticFini =
6126 Builder.SetInsertPoint(AllocaIP.getBlock()->getFirstNonPHIOrDbgOrAlloca());
6129 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6130 Value *PLowerBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.lowerbound");
6131 Value *PUpperBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.upperbound");
6132 Value *PStride =
Builder.CreateAlloca(IVTy,
nullptr,
"p.stride");
6141 Constant *One = ConstantInt::get(IVTy, 1);
6142 Builder.CreateStore(Zero, PLowerBound);
6144 Builder.CreateStore(UpperBound, PUpperBound);
6145 Builder.CreateStore(One, PStride);
6151 (LoopType == WorksharingLoopType::DistributeStaticLoop)
6152 ? OMPScheduleType::OrderedDistribute
6155 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6159 auto BuildInitCall = [LoopType, SrcLoc, ThreadNum, PLastIter, PLowerBound,
6160 PUpperBound, IVTy, PStride, One,
Zero, StaticInit,
6163 PLowerBound, PUpperBound});
6164 if (LoopType == WorksharingLoopType::DistributeForStaticLoop) {
6165 Value *PDistUpperBound =
6166 Builder.CreateAlloca(IVTy,
nullptr,
"p.distupperbound");
6167 Args.push_back(PDistUpperBound);
6172 BuildInitCall(SchedulingType,
Builder);
6173 if (HasDistSchedule &&
6174 LoopType != WorksharingLoopType::DistributeStaticLoop) {
6175 Constant *DistScheduleSchedType = ConstantInt::get(
6180 BuildInitCall(DistScheduleSchedType,
Builder);
6182 Value *LowerBound =
Builder.CreateLoad(IVTy, PLowerBound);
6183 Value *InclusiveUpperBound =
Builder.CreateLoad(IVTy, PUpperBound);
6184 Value *TripCountMinusOne =
Builder.CreateSub(InclusiveUpperBound, LowerBound);
6185 Value *TripCount =
Builder.CreateAdd(TripCountMinusOne, One);
6186 CLI->setTripCount(TripCount);
6192 CLI->mapIndVar([&](Instruction *OldIV) ->
Value * {
6196 return Builder.CreateAdd(OldIV, LowerBound,
"",
false,
6197 Config.hasNoSignedWrap());
6209 omp::Directive::OMPD_for,
false,
6212 return BarrierIP.takeError();
6239 Reachable.insert(
Block);
6249 Ctx, {
MDString::get(Ctx,
"llvm.loop.parallel_accesses"), AccessGroup}));
6253OpenMPIRBuilder::applyStaticChunkedWorkshareLoop(
6257 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6258 assert((ChunkSize || DistScheduleChunkSize) &&
"Chunk size is required");
6263 Type *IVTy =
IV->getType();
6265 "Max supported tripcount bitwidth is 64 bits");
6267 :
Type::getInt64Ty(Ctx);
6270 Constant *One = ConstantInt::get(InternalIVTy, 1);
6275 SmallVector<Instruction *> UIs;
6276 for (BasicBlock &BB : *
F)
6277 if (!BB.hasTerminator())
6278 UIs.
push_back(
new UnreachableInst(
F->getContext(), &BB));
6283 LoopInfo &&LI = LIA.
run(*
F,
FAM);
6284 for (Instruction *
I : UIs)
6285 I->eraseFromParent();
6288 if (ChunkSize || DistScheduleChunkSize)
6293 FunctionCallee StaticInit =
6295 FunctionCallee StaticFini =
6301 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6302 Value *PLowerBound =
6303 Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.lowerbound");
6304 Value *PUpperBound =
6305 Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.upperbound");
6306 Value *PStride =
Builder.CreateAlloca(InternalIVTy,
nullptr,
"p.stride");
6315 ChunkSize ? ChunkSize : Zero, InternalIVTy,
"chunksize");
6316 Value *CastedDistScheduleChunkSize =
Builder.CreateZExtOrTrunc(
6317 DistScheduleChunkSize ? DistScheduleChunkSize : Zero, InternalIVTy,
6318 "distschedulechunksize");
6319 Value *CastedTripCount =
6320 Builder.CreateZExt(OrigTripCount, InternalIVTy,
"tripcount");
6323 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6325 ConstantInt::get(I32Type,
static_cast<int>(DistScheduleSchedType));
6326 Builder.CreateStore(Zero, PLowerBound);
6327 Value *OrigUpperBound =
Builder.CreateSub(CastedTripCount, One);
6328 Value *IsTripCountZero =
Builder.CreateICmpEQ(CastedTripCount, Zero);
6330 Builder.CreateSelect(IsTripCountZero, Zero, OrigUpperBound);
6331 Builder.CreateStore(UpperBound, PUpperBound);
6332 Builder.CreateStore(One, PStride);
6336 uint32_t SrcLocStrSize;
6339 if (DistScheduleSchedType != OMPScheduleType::None) {
6340 Flag |= OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6345 auto BuildInitCall = [StaticInit, SrcLoc, ThreadNum, PLastIter, PLowerBound,
6346 PUpperBound, PStride, One,
6347 this](
Value *SchedulingType,
Value *ChunkSize,
6350 StaticInit, {SrcLoc, ThreadNum,
6351 SchedulingType, PLastIter,
6352 PLowerBound, PUpperBound,
6356 BuildInitCall(SchedulingType, CastedChunkSize,
Builder);
6357 if (DistScheduleSchedType != OMPScheduleType::None &&
6358 SchedType != OMPScheduleType::OrderedDistributeChunked &&
6359 SchedType != OMPScheduleType::OrderedDistribute) {
6363 BuildInitCall(DistSchedulingType, CastedDistScheduleChunkSize,
Builder);
6367 Value *FirstChunkStart =
6368 Builder.CreateLoad(InternalIVTy, PLowerBound,
"omp_firstchunk.lb");
6369 Value *FirstChunkStop =
6370 Builder.CreateLoad(InternalIVTy, PUpperBound,
"omp_firstchunk.ub");
6371 Value *FirstChunkEnd =
Builder.CreateAdd(FirstChunkStop, One);
6373 Builder.CreateSub(FirstChunkEnd, FirstChunkStart,
"omp_chunk.range");
6374 Value *NextChunkStride =
6375 Builder.CreateLoad(InternalIVTy, PStride,
"omp_dispatch.stride");
6379 Value *DispatchCounter;
6387 DispatchCounter = Counter;
6390 FirstChunkStart, CastedTripCount, NextChunkStride,
6413 Value *ChunkEnd =
Builder.CreateAdd(DispatchCounter, ChunkRange);
6414 Value *IsLastChunk =
6415 Builder.CreateICmpUGE(ChunkEnd, CastedTripCount,
"omp_chunk.is_last");
6416 Value *CountUntilOrigTripCount =
6417 Builder.CreateSub(CastedTripCount, DispatchCounter);
6419 IsLastChunk, CountUntilOrigTripCount, ChunkRange,
"omp_chunk.tripcount");
6420 Value *BackcastedChunkTC =
6421 Builder.CreateTrunc(ChunkTripCount, IVTy,
"omp_chunk.tripcount.trunc");
6422 CLI->setTripCount(BackcastedChunkTC);
6427 Value *BackcastedDispatchCounter =
6428 Builder.CreateTrunc(DispatchCounter, IVTy,
"omp_dispatch.iv.trunc");
6429 CLI->mapIndVar([&](Instruction *) ->
Value * {
6431 return Builder.CreateAdd(
IV, BackcastedDispatchCounter);
6444 return AfterIP.takeError();
6459static FunctionCallee
6462 unsigned Bitwidth = Ty->getIntegerBitWidth();
6465 case WorksharingLoopType::ForStaticLoop:
6468 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_loop_4u);
6471 M, omp::RuntimeFunction::OMPRTL___kmpc_for_static_loop_8u);
6473 case WorksharingLoopType::DistributeStaticLoop:
6476 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_static_loop_4u);
6479 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_static_loop_8u);
6481 case WorksharingLoopType::DistributeForStaticLoop:
6484 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_for_static_loop_4u);
6487 M, omp::RuntimeFunction::OMPRTL___kmpc_distribute_for_static_loop_8u);
6490 if (Bitwidth != 32 && Bitwidth != 64) {
6502 Function &LoopBodyFn,
bool NoLoop) {
6513 if (LoopType == WorksharingLoopType::DistributeStaticLoop) {
6514 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6515 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), 0));
6516 Builder.restoreIP({InsertBlock, std::prev(InsertBlock->
end())});
6521 M, omp::RuntimeFunction::OMPRTL_omp_get_num_threads);
6522 Builder.restoreIP({InsertBlock, std::prev(InsertBlock->
end())});
6526 Builder.CreateZExtOrTrunc(NumThreads, TripCountTy,
"num.threads.cast"));
6527 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6528 if (LoopType == WorksharingLoopType::DistributeForStaticLoop) {
6529 RealArgs.
push_back(ConstantInt::get(TripCountTy, 0));
6530 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), NoLoop));
6532 RealArgs.
push_back(ConstantInt::get(Builder.getInt8Ty(), 0));
6556 Builder.restoreIP({Preheader, Preheader->
end()});
6559 Builder.CreateBr(CLI->
getExit());
6567 CleanUpInfo.
collectBlocks(RegionBlockSet, BlocksToBeRemoved);
6575 "Expected unique undroppable user of outlined function");
6577 assert(OutlinedFnCallInstruction &&
"Expected outlined function call");
6579 "Expected outlined function call to be located in loop preheader");
6581 if (OutlinedFnCallInstruction->
arg_size() > 1)
6588 LoopBodyArg, TripCount, OutlinedFn, NoLoop);
6590 for (
auto &ToBeDeletedItem : ToBeDeleted)
6591 ToBeDeletedItem->eraseFromParent();
6598 uint32_t SrcLocStrSize;
6602 case WorksharingLoopType::ForStaticLoop:
6603 Flag = OMP_IDENT_FLAG_WORK_LOOP;
6605 case WorksharingLoopType::DistributeStaticLoop:
6606 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE;
6608 case WorksharingLoopType::DistributeForStaticLoop:
6609 Flag = OMP_IDENT_FLAG_WORK_DISTRIBUTE | OMP_IDENT_FLAG_WORK_LOOP;
6614 auto OI = std::make_unique<OutlineInfo>();
6619 SmallVector<Instruction *, 4> ToBeDeleted;
6621 OI->OuterAllocBB = AllocaIP.getBlock();
6644 SmallPtrSet<BasicBlock *, 32> ParallelRegionBlockSet;
6646 OI->collectBlocks(ParallelRegionBlockSet, Blocks);
6648 CodeExtractorAnalysisCache CEAC(*OuterFn);
6649 CodeExtractor Extractor(Blocks,
6663 SetVector<Value *> SinkingCands, HoistingCands;
6667 Extractor.findAllocas(CEAC, SinkingCands, HoistingCands, CommonExit);
6674 for (
auto Use :
Users) {
6676 if (ParallelRegionBlockSet.
count(Inst->getParent())) {
6677 Inst->replaceUsesOfWith(CLI->
getIndVar(), NewLoopCntLoad);
6683 OI->ExcludeArgsFromAggregate.push_back(NewLoopCntLoad);
6690 OI->PostOutlineCB = [=, ToBeDeletedVec =
6691 std::move(ToBeDeleted)](
Function &OutlinedFn) {
6701 bool NeedsBarrier, omp::ScheduleKind SchedKind,
Value *ChunkSize,
6702 bool HasSimdModifier,
bool HasMonotonicModifier,
6703 bool HasNonmonotonicModifier,
bool HasOrderedClause,
6705 Value *DistScheduleChunkSize) {
6706 if (
Config.isTargetDevice())
6707 return applyWorkshareLoopTarget(
DL, CLI, AllocaIP, LoopType, NoLoop);
6709 SchedKind, ChunkSize, HasSimdModifier, HasMonotonicModifier,
6710 HasNonmonotonicModifier, HasOrderedClause, DistScheduleChunkSize);
6712 bool IsOrdered = (EffectiveScheduleType & OMPScheduleType::ModifierOrdered) ==
6713 OMPScheduleType::ModifierOrdered;
6715 if (HasDistSchedule) {
6716 DistScheduleSchedType = DistScheduleChunkSize
6717 ? OMPScheduleType::OrderedDistributeChunked
6718 : OMPScheduleType::OrderedDistribute;
6720 switch (EffectiveScheduleType & ~OMPScheduleType::ModifierMask) {
6721 case OMPScheduleType::BaseStatic:
6722 case OMPScheduleType::BaseDistribute:
6723 assert((!ChunkSize || !DistScheduleChunkSize) &&
6724 "No chunk size with static-chunked schedule");
6725 if (IsOrdered && !HasDistSchedule)
6726 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6727 NeedsBarrier, ChunkSize);
6729 if (DistScheduleChunkSize)
6730 return applyStaticChunkedWorkshareLoop(
6731 DL, CLI, AllocaIP, NeedsBarrier, ChunkSize, EffectiveScheduleType,
6732 DistScheduleChunkSize, DistScheduleSchedType);
6733 return applyStaticWorkshareLoop(
DL, CLI, AllocaIP, LoopType, NeedsBarrier,
6736 case OMPScheduleType::BaseStaticChunked:
6737 case OMPScheduleType::BaseDistributeChunked:
6738 if (IsOrdered && !HasDistSchedule)
6739 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6740 NeedsBarrier, ChunkSize);
6742 return applyStaticChunkedWorkshareLoop(
6743 DL, CLI, AllocaIP, NeedsBarrier, ChunkSize, EffectiveScheduleType,
6744 DistScheduleChunkSize, DistScheduleSchedType);
6746 case OMPScheduleType::BaseRuntime:
6747 case OMPScheduleType::BaseAuto:
6748 case OMPScheduleType::BaseGreedy:
6749 case OMPScheduleType::BaseBalanced:
6750 case OMPScheduleType::BaseSteal:
6751 case OMPScheduleType::BaseRuntimeSimd:
6753 "schedule type does not support user-defined chunk sizes");
6755 case OMPScheduleType::BaseGuidedSimd:
6756 case OMPScheduleType::BaseDynamicChunked:
6757 case OMPScheduleType::BaseGuidedChunked:
6758 case OMPScheduleType::BaseGuidedIterativeChunked:
6759 case OMPScheduleType::BaseGuidedAnalyticalChunked:
6760 case OMPScheduleType::BaseStaticBalancedChunked:
6761 return applyDynamicWorkshareLoop(
DL, CLI, AllocaIP, EffectiveScheduleType,
6762 NeedsBarrier, ChunkSize);
6775 unsigned Bitwidth = Ty->getIntegerBitWidth();
6778 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_init_4u);
6781 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_init_8u);
6789static FunctionCallee
6791 unsigned Bitwidth = Ty->getIntegerBitWidth();
6794 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_next_4u);
6797 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_next_8u);
6804static FunctionCallee
6806 unsigned Bitwidth = Ty->getIntegerBitWidth();
6809 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_fini_4u);
6812 M, omp::RuntimeFunction::OMPRTL___kmpc_dispatch_fini_8u);
6817OpenMPIRBuilder::applyDynamicWorkshareLoop(
DebugLoc DL, CanonicalLoopInfo *CLI,
6820 bool NeedsBarrier,
Value *Chunk) {
6821 assert(CLI->
isValid() &&
"Requires a valid canonical loop");
6823 "Require dedicated allocate IP");
6825 "Require valid schedule type");
6827 bool Ordered = (SchedType & OMPScheduleType::ModifierOrdered) ==
6828 OMPScheduleType::ModifierOrdered;
6833 uint32_t SrcLocStrSize;
6840 Type *IVTy =
IV->getType();
6845 Builder.SetInsertPoint(AllocaIP.getBlock()->getFirstNonPHIOrDbgOrAlloca());
6847 Value *PLastIter =
Builder.CreateAlloca(I32Type,
nullptr,
"p.lastiter");
6848 Value *PLowerBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.lowerbound");
6849 Value *PUpperBound =
Builder.CreateAlloca(IVTy,
nullptr,
"p.upperbound");
6850 Value *PStride =
Builder.CreateAlloca(IVTy,
nullptr,
"p.stride");
6859 Constant *One = ConstantInt::get(IVTy, 1);
6860 Builder.CreateStore(One, PLowerBound);
6862 Builder.CreateStore(UpperBound, PUpperBound);
6863 Builder.CreateStore(One, PStride);
6881 ConstantInt::get(I32Type,
static_cast<int>(SchedType));
6893 Builder.SetInsertPoint(OuterCond, OuterCond->getFirstInsertionPt());
6896 {SrcLoc, ThreadNum, PLastIter, PLowerBound, PUpperBound, PStride});
6897 Constant *Zero32 = ConstantInt::get(I32Type, 0);
6900 Builder.CreateSub(
Builder.CreateLoad(IVTy, PLowerBound), One,
"lb");
6901 Builder.CreateCondBr(MoreWork, Header, Exit);
6907 PI->setIncomingBlock(0, OuterCond);
6908 PI->setIncomingValue(0, LowerBound);
6913 Br->setSuccessor(OuterCond);
6919 UpperBound =
Builder.CreateLoad(IVTy, PUpperBound,
"ub");
6922 CI->setOperand(1, UpperBound);
6926 assert(BI->getSuccessor(1) == Exit);
6927 BI->setSuccessor(1, OuterCond);
6941 omp::Directive::OMPD_for,
false,
6944 return BarrierIP.takeError();
6996 assert(
Loops.size() >= 1 &&
"At least one loop required");
6997 size_t NumLoops =
Loops.size();
7001 return Loops.front();
7013 Loop->collectControlBlocks(OldControlBBs);
7017 if (ComputeIP.
isSet())
7024 Value *CollapsedTripCount =
nullptr;
7027 "All loops to collapse must be valid canonical loops");
7028 Value *OrigTripCount = L->getTripCount();
7029 if (!CollapsedTripCount) {
7030 CollapsedTripCount = OrigTripCount;
7035 CollapsedTripCount =
7036 Builder.CreateNUWMul(CollapsedTripCount, OrigTripCount);
7042 OrigPreheader->
getNextNode(), OrigAfter,
"collapsed",
7049 Builder.restoreIP(Result->getBodyIP());
7051 Value *Leftover = Result->getIndVar();
7053 NewIndVars.
resize(NumLoops);
7054 for (
int i = NumLoops - 1; i >= 1; --i) {
7055 Value *OrigTripCount =
Loops[i]->getTripCount();
7057 Value *NewIndVar =
Builder.CreateURem(Leftover, OrigTripCount);
7058 NewIndVars[i] = NewIndVar;
7060 Leftover =
Builder.CreateUDiv(Leftover, OrigTripCount);
7063 NewIndVars[0] = Leftover;
7072 BasicBlock *ContinueBlock = Result->getBody();
7074 auto ContinueWith = [&ContinueBlock, &ContinuePred,
DL](
BasicBlock *Dest,
7081 ContinueBlock =
nullptr;
7082 ContinuePred = NextSrc;
7089 for (
size_t i = 0; i < NumLoops - 1; ++i)
7090 ContinueWith(
Loops[i]->getBody(),
Loops[i + 1]->getHeader());
7096 for (
size_t i = NumLoops - 1; i > 0; --i)
7097 ContinueWith(
Loops[i]->getAfter(),
Loops[i - 1]->getLatch());
7100 ContinueWith(Result->getLatch(),
nullptr);
7107 for (
size_t i = 0; i < NumLoops; ++i)
7108 Loops[i]->getIndVar()->replaceAllUsesWith(NewIndVars[i]);
7122std::vector<CanonicalLoopInfo *>
7126 "Must pass as many tile sizes as there are loops");
7127 int NumLoops =
Loops.size();
7128 assert(NumLoops >= 1 &&
"At least one loop to tile required");
7140 Loop->collectControlBlocks(OldControlBBs);
7148 assert(L->isValid() &&
"All input loops must be valid canonical loops");
7149 OrigTripCounts.
push_back(L->getTripCount());
7160 for (
int i = 0; i < NumLoops - 1; ++i) {
7173 for (
int i = 0; i < NumLoops; ++i) {
7175 Value *OrigTripCount = OrigTripCounts[i];
7188 Value *FloorTripOverflow =
7189 Builder.CreateICmpNE(FloorTripRem, ConstantInt::get(IVType, 0));
7191 FloorTripOverflow =
Builder.CreateZExt(FloorTripOverflow, IVType);
7192 Value *FloorTripCount =
7193 Builder.CreateAdd(FloorCompleteTripCount, FloorTripOverflow,
7194 "omp_floor" +
Twine(i) +
".tripcount",
true);
7197 FloorCompleteCount.
push_back(FloorCompleteTripCount);
7203 std::vector<CanonicalLoopInfo *> Result;
7204 Result.reserve(NumLoops * 2);
7217 auto EmbeddNewLoop =
7218 [
this,
DL,
F, InnerEnter, &Enter, &
Continue, &OutroInsertBefore](
7221 DL, TripCount,
F, InnerEnter, OutroInsertBefore, Name);
7226 Enter = EmbeddedLoop->
getBody();
7228 OutroInsertBefore = EmbeddedLoop->
getLatch();
7229 return EmbeddedLoop;
7233 const Twine &NameBase) {
7236 EmbeddNewLoop(
P.value(), NameBase +
Twine(
P.index()));
7237 Result.push_back(EmbeddedLoop);
7241 EmbeddNewLoops(FloorCount,
"floor");
7247 for (
int i = 0; i < NumLoops; ++i) {
7251 Value *FloorIsEpilogue =
7253 Value *TileTripCount =
7260 EmbeddNewLoops(TileCounts,
"tile");
7265 for (std::pair<BasicBlock *, BasicBlock *>
P : InbetweenCode) {
7274 BodyEnter =
nullptr;
7275 BodyEntered = ExitBB;
7287 Builder.restoreIP(Result.back()->getBodyIP());
7288 for (
int i = 0; i < NumLoops; ++i) {
7291 Value *OrigIndVar = OrigIndVars[i];
7319 if (Properties.
empty())
7342 assert(
Loop->isValid() &&
"Expecting a valid CanonicalLoopInfo");
7346 assert(Latch &&
"A valid CanonicalLoopInfo must have a unique latch");
7354 if (
I.mayReadOrWriteMemory()) {
7358 I.setMetadata(LLVMContext::MD_access_group, AccessGroup);
7372 Loop->collectControlBlocks(oldControlBBs);
7377 assert(L->isValid() &&
"All input loops must be valid canonical loops");
7378 origTripCounts.
push_back(L->getTripCount());
7387 Builder.SetInsertPoint(TCBlock);
7388 Value *fusedTripCount =
nullptr;
7390 assert(L->isValid() &&
"All loops to fuse must be valid canonical loops");
7391 Value *origTripCount = L->getTripCount();
7392 if (!fusedTripCount) {
7393 fusedTripCount = origTripCount;
7396 Value *condTP =
Builder.CreateICmpSGT(fusedTripCount, origTripCount);
7397 fusedTripCount =
Builder.CreateSelect(condTP, fusedTripCount, origTripCount,
7411 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7412 Loops[i]->getPreheader()->moveBefore(TCBlock);
7413 Loops[i]->getAfter()->moveBefore(TCBlock);
7417 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7429 for (
size_t i = 0; i <
Loops.size(); ++i) {
7431 F->getContext(),
"omp.fused.inner.cond",
F,
Loops[i]->getBody());
7432 Builder.SetInsertPoint(condBlock);
7440 for (
size_t i = 0; i <
Loops.size() - 1; ++i) {
7441 Builder.SetInsertPoint(condBBs[i]);
7442 Builder.CreateCondBr(condValues[i],
Loops[i]->getBody(), condBBs[i + 1]);
7458 "omp.fused.pre_latch");
7491 const Twine &NamePrefix) {
7520 C, NamePrefix +
".if.then",
Cond->getParent(),
Cond->getNextNode());
7522 C, NamePrefix +
".if.else",
Cond->getParent(), CanonicalLoop->
getExit());
7525 Builder.SetInsertPoint(SplitBeforeIt);
7527 Builder.CreateCondBr(IfCond, ThenBlock, ElseBlock);
7530 spliceBB(IP, ThenBlock,
false, Builder.getCurrentDebugLocation());
7533 Builder.SetInsertPoint(ElseBlock);
7539 ExistingBlocks.
reserve(L->getNumBlocks() + 1);
7541 ExistingBlocks.
append(L->block_begin(), L->block_end());
7547 assert(LoopCond && LoopHeader &&
"Invalid loop structure");
7549 if (
Block == L->getLoopPreheader() ||
Block == L->getLoopLatch() ||
7556 if (
Block == ThenBlock)
7557 NewBB->
setName(NamePrefix +
".if.else");
7560 VMap[
Block] = NewBB;
7568 L->getLoopLatch()->splitBasicBlockBefore(
L->getLoopLatch()->begin(),
7569 NamePrefix +
".pre_latch");
7573 L->addBasicBlockToLoop(ThenBlock, LI);
7579 if (TargetTriple.
isX86()) {
7580 if (Features.
lookup(
"avx512f"))
7582 else if (Features.
lookup(
"avx"))
7586 if (TargetTriple.
isPPC())
7588 if (TargetTriple.
isWasm())
7595 Value *IfCond, OrderKind Order,
7605 if (!BB.hasTerminator())
7621 I->eraseFromParent();
7624 if (AlignedVars.
size()) {
7626 for (
auto &AlignedItem : AlignedVars) {
7627 Value *AlignedPtr = AlignedItem.first;
7628 Value *Alignment = AlignedItem.second;
7631 Builder.CreateAlignmentAssumption(
F->getDataLayout(), AlignedPtr,
7639 createIfVersion(CanonicalLoop, IfCond, VMap, LIA, LI, L,
"simd");
7652 Reachable.insert(
Block);
7662 if ((Safelen ==
nullptr) || (Order == OrderKind::OMP_ORDER_concurrent))
7678 if (Simdlen || Safelen) {
7682 ConstantInt *VectorizeWidth = Simdlen ==
nullptr ? Safelen : Simdlen;
7708static std::unique_ptr<TargetMachine>
7712 StringRef CPU =
F->getFnAttribute(
"target-cpu").getValueAsString();
7713 StringRef Features =
F->getFnAttribute(
"target-features").getValueAsString();
7724 std::nullopt, OptLevel));
7742 if (!BB.hasTerminator())
7755 [&](
const Function &
F) {
return TM->getTargetTransformInfo(
F); });
7756 FAM.registerPass([&]() {
return TIRA; });
7770 I->eraseFromParent();
7773 assert(L &&
"Expecting CanonicalLoopInfo to be recognized as a loop");
7778 nullptr, ORE,
static_cast<int>(OptLevel),
7798 <<
" Threshold=" << UP.
Threshold <<
"\n"
7801 <<
" PartialOptSizeThreshold="
7821 Ptr =
Load->getPointerOperand();
7823 Ptr =
Store->getPointerOperand();
7830 if (Alloca->getParent() == &
F->getEntryBlock())
7850 int MaxTripCount = 0;
7851 bool MaxOrZero =
false;
7852 unsigned TripMultiple = 0;
7856 MaxTripCount, MaxOrZero, TripMultiple, UCE, UP, PP);
7857 LLVM_DEBUG(
dbgs() <<
"Suggesting unroll factor of " << Factor <<
"\n");
7868 assert(Factor >= 0 &&
"Unroll factor must not be negative");
7884 Ctx, {
MDString::get(Ctx,
"llvm.loop.unroll.count"), FactorConst}));
7897 *UnrolledCLI =
Loop;
7902 "unrolling only makes sense with a factor of 2 or larger");
7904 Type *IndVarTy =
Loop->getIndVarType();
7911 std::vector<CanonicalLoopInfo *>
LoopNest =
7926 Ctx, {
MDString::get(Ctx,
"llvm.loop.unroll.count"), FactorConst})});
7929 (*UnrolledCLI)->assertOK();
7947 Value *Args[] = {Ident, ThreadId, BufSize, CpyBuf, CpyFn, DidItLD};
7966 if (!CPVars.
empty()) {
7971 Directive OMPD = Directive::OMPD_single;
7976 Value *Args[] = {Ident, ThreadId};
7985 if (
Error Err = FiniCB(IP))
8006 EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCBWrapper,
8013 for (
size_t I = 0, E = CPVars.
size();
I < E; ++
I)
8016 ConstantInt::get(Int64, 0), CPVars[
I],
8019 }
else if (!IsNowait) {
8022 omp::Directive::OMPD_unknown,
false,
8040 Directive::OMPD_scope,
nullptr,
nullptr,
8041 BodyGenCB, FiniCB,
false,
true,
8049 omp::Directive::OMPD_unknown,
8065 Directive OMPD = Directive::OMPD_critical;
8070 Value *LockVar = getOMPCriticalRegionLock(CriticalName);
8071 Value *Args[] = {Ident, ThreadId, LockVar};
8088 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
8096 const Twine &Name,
bool IsDependSource) {
8100 "OpenMP runtime requires depend vec with i64 type");
8113 for (
unsigned I = 0;
I < NumLoops; ++
I) {
8127 Value *Args[] = {Ident, ThreadId, DependBaseAddrGEP};
8145 Directive OMPD = Directive::OMPD_ordered_blockassoc;
8154 Value *Args[] = {Ident, ThreadId};
8164 return EmitOMPInlinedRegion(OMPD, EntryCall, ExitCall, BodyGenCB, FiniCB,
8171 bool HasFinalize,
bool IsCancellable) {
8178 BasicBlock *EntryBB = Builder.GetInsertBlock();
8187 emitCommonDirectiveEntry(OMPD, EntryCall, ExitBB, Conditional);
8199 "Unexpected control flow graph state!!");
8201 emitCommonDirectiveExit(OMPD, FinIP, ExitCall, HasFinalize);
8203 return AfterIP.takeError();
8208 "Unexpected Insertion point location!");
8211 auto InsertBB = merged ? ExitPredBB : ExitBB;
8214 Builder.SetInsertPoint(InsertBB);
8216 return Builder.saveIP();
8220 Directive OMPD,
Value *EntryCall, BasicBlock *ExitBB,
bool Conditional) {
8222 if (!Conditional || !EntryCall)
8228 auto *UI =
new UnreachableInst(
Builder.getContext(), ThenBB);
8238 Builder.CreateCondBr(CallBool, ThenBB, ExitBB);
8242 UI->eraseFromParent();
8250 omp::Directive OMPD,
InsertPointTy FinIP, Instruction *ExitCall,
8258 "Unexpected finalization stack state!");
8261 assert(Fi.DK == OMPD &&
"Unexpected Directive for Finalization call!");
8263 if (
Error Err = Fi.mergeFiniBB(
Builder, FinIP.getBlock()))
8264 return std::move(Err);
8268 Builder.SetInsertPoint(FinIP.getBlock()->getTerminator());
8278 return IRBuilder<>::InsertPoint(ExitCall->
getParent(),
8312 "copyin.not.master.end");
8319 Builder.SetInsertPoint(OMP_Entry);
8322 Value *cmp =
Builder.CreateICmpNE(MasterPtr, PrivatePtr);
8323 Builder.CreateCondBr(cmp, CopyBegin, CopyEnd);
8325 Builder.SetInsertPoint(CopyBegin);
8343 Value *Args[] = {ThreadId,
Size, Allocator};
8366 return Builder.CreateCall(Fn, Args, Name);
8380 Value *Args[] = {ThreadId, Addr, Allocator};
8387 const Twine &Name) {
8395 M.getContext(),
M.getDataLayout().getPrefTypeAlign(Int64)));
8401 const Twine &Name) {
8403 Loc,
Builder.getInt64(
M.getDataLayout().getTypeAllocSize(VarType)), Name);
8408 const Twine &Name) {
8414 return Builder.CreateCall(Fn, Args, Name);
8419 const Twine &Name) {
8421 Loc, Addr,
Builder.getInt64(
M.getDataLayout().getTypeAllocSize(VarType)),
8428 Value *DependenceAddress,
bool HaveNowaitClause) {
8438 else if (
Device->getType() != Int32)
8440 Constant *InteropTypeVal = ConstantInt::get(Int32, (
int)InteropType);
8441 if (NumDependences ==
nullptr) {
8442 NumDependences = ConstantInt::get(Int32, 0);
8446 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8448 Ident, ThreadId, InteropVar, InteropTypeVal,
8449 Device, NumDependences, DependenceAddress, HaveNowaitClauseVal};
8458 Value *NumDependences,
Value *DependenceAddress,
bool HaveNowaitClause) {
8468 else if (
Device->getType() != Int32)
8470 if (NumDependences ==
nullptr) {
8471 NumDependences = ConstantInt::get(Int32, 0);
8475 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8477 Ident, ThreadId, InteropVar,
Device,
8478 NumDependences, DependenceAddress, HaveNowaitClauseVal};
8487 Value *NumDependences,
8488 Value *DependenceAddress,
8489 bool HaveNowaitClause) {
8498 else if (
Device->getType() != Int32)
8500 if (NumDependences ==
nullptr) {
8501 NumDependences = ConstantInt::get(Int32, 0);
8505 Value *HaveNowaitClauseVal = ConstantInt::get(Int32, HaveNowaitClause);
8507 Ident, ThreadId, InteropVar,
Device,
8508 NumDependences, DependenceAddress, HaveNowaitClauseVal};
8538 assert(!Attrs.MaxThreads.empty() && !Attrs.MaxTeams.empty() &&
8539 "expected num_threads and num_teams to be specified");
8559 const std::string DebugPrefix =
"_debug__";
8560 if (KernelName.
ends_with(DebugPrefix)) {
8561 KernelName = KernelName.
drop_back(DebugPrefix.length());
8562 Kernel =
M.getFunction(KernelName);
8568 if (Attrs.MinTeams.front() > 1 || Attrs.MaxTeams.front() > 0)
8570 Attrs.MaxTeams.front());
8574 int32_t MaxThreadsVal = Attrs.MaxThreads.front();
8579 Attrs.MinThreads.front());
8581 MaxThreadsVal = Attrs.MinThreads.front();
8585 if (MaxThreadsVal > 0)
8598 omp::RuntimeFunction::OMPRTL___kmpc_target_init);
8601 Twine DynamicEnvironmentName = KernelName +
"_dynamic_environment";
8602 Constant *DynamicEnvironmentInitializer =
8606 DynamicEnvironmentInitializer, DynamicEnvironmentName,
8608 DL.getDefaultGlobalsAddressSpace());
8612 DynamicEnvironmentGV->
getType() == DynamicEnvironmentPtr
8613 ? DynamicEnvironmentGV
8615 DynamicEnvironmentPtr);
8618 ConfigurationEnvironment, {
8619 UseGenericStateMachineVal,
8620 MayUseNestedParallelismVal,
8629 KernelEnvironment, {
8630 ConfigurationEnvironmentInitializer,
8634 std::string KernelEnvironmentName =
8635 (KernelName +
"_kernel_environment").str();
8638 KernelEnvironmentInitializer, KernelEnvironmentName,
8640 DL.getDefaultGlobalsAddressSpace());
8644 KernelEnvironmentGV->
getType() == KernelEnvironmentPtr
8645 ? KernelEnvironmentGV
8647 KernelEnvironmentPtr);
8648 Value *KernelLaunchEnvironment =
8651 KernelLaunchEnvironment =
8652 KernelLaunchEnvironment->
getType() == KernelLaunchEnvParamTy
8653 ? KernelLaunchEnvironment
8654 :
Builder.CreateAddrSpaceCast(KernelLaunchEnvironment,
8655 KernelLaunchEnvParamTy);
8657 Fn, {KernelEnvironment, KernelLaunchEnvironment});
8669 auto *UI =
Builder.CreateUnreachable();
8675 Builder.SetInsertPoint(WorkerExitBB);
8679 Builder.SetInsertPoint(CheckBBTI);
8680 Builder.CreateCondBr(ExecUserCode, UI->getParent(), WorkerExitBB);
8682 CheckBBTI->eraseFromParent();
8683 UI->eraseFromParent();
8691 int32_t TeamsReductionDataSize) {
8696 omp::RuntimeFunction::OMPRTL___kmpc_target_deinit);
8700 if (!TeamsReductionDataSize)
8706 const std::string DebugPrefix =
"_debug__";
8708 KernelName = KernelName.
drop_back(DebugPrefix.length());
8709 auto *KernelEnvironmentGV =
8710 M.getNamedGlobal((KernelName +
"_kernel_environment").str());
8711 assert(KernelEnvironmentGV &&
"Expected kernel environment global\n");
8712 auto *KernelEnvironmentInitializer = KernelEnvironmentGV->getInitializer();
8714 KernelEnvironmentInitializer,
8715 ConstantInt::get(Int32, TeamsReductionDataSize), {0, 7});
8716 KernelEnvironmentGV->setInitializer(NewInitializer);
8721 if (
Kernel.hasFnAttribute(Name)) {
8722 int32_t OldLimit =
Kernel.getFnAttributeAsParsedInteger(Name);
8728std::pair<int32_t, int32_t>
8730 int32_t ThreadLimit =
8731 Kernel.getFnAttributeAsParsedInteger(
"omp_target_thread_limit");
8734 const auto &Attr =
Kernel.getFnAttribute(
"amdgpu-flat-work-group-size");
8735 if (!Attr.isValid() || !Attr.isStringAttribute())
8736 return {0, ThreadLimit};
8737 auto [LBStr, UBStr] = Attr.getValueAsString().split(
',');
8740 return {0, ThreadLimit};
8741 UB = ThreadLimit ? std::min(ThreadLimit, UB) : UB;
8749 return {0, ThreadLimit ? std::min(ThreadLimit, UB) : UB};
8751 return {0, ThreadLimit};
8757 Kernel.addFnAttr(
"omp_target_thread_limit", std::to_string(UB));
8760 Kernel.addFnAttr(
"amdgpu-flat-work-group-size",
8768std::pair<int32_t, int32_t>
8771 return {0,
Kernel.getFnAttributeAsParsedInteger(
"omp_target_num_teams")};
8775 int32_t LB, int32_t UB) {
8783 Kernel.addFnAttr(
"omp_target_num_teams", std::to_string(LB));
8786void OpenMPIRBuilder::setOutlinedTargetRegionFunctionAttributes(
8795 else if (
T.isNVPTX())
8797 else if (
T.isSPIRV())
8803 StringRef EntryFnIDName) {
8804 if (
Config.isTargetDevice()) {
8805 assert(OutlinedFn &&
"The outlined function must exist if embedded");
8809 return new GlobalVariable(
8814Constant *OpenMPIRBuilder::createTargetRegionEntryAddr(
Function *OutlinedFn,
8815 StringRef EntryFnName) {
8819 assert(!
M.getGlobalVariable(EntryFnName,
true) &&
8820 "Named kernel already exists?");
8821 return new GlobalVariable(
8834 if (
Config.isTargetDevice() || !
Config.openMPOffloadMandatory()) {
8838 OutlinedFn = *CBResult;
8840 OutlinedFn =
nullptr;
8846 if (!IsOffloadEntry)
8849 std::string EntryFnIDName =
8851 ? std::string(EntryFnName)
8855 EntryFnName, EntryFnIDName);
8863 setOutlinedTargetRegionFunctionAttributes(OutlinedFn);
8864 auto OutlinedFnID = createOutlinedFunctionID(OutlinedFn, EntryFnIDName);
8865 auto EntryAddr = createTargetRegionEntryAddr(OutlinedFn, EntryFnName);
8867 EntryInfo, EntryAddr, OutlinedFnID,
8869 return OutlinedFnID;
8887 bool IsStandAlone = !BodyGenCB;
8894 MapInfo = &GenMapInfoCB(
Builder.saveIP());
8896 AllocaIP,
Builder.saveIP(), *MapInfo, Info, CustomMapperCB,
8897 true, DeviceAddrCB))
8904 Value *PointerNum =
Builder.getInt32(Info.NumberOfPtrs);
8914 SrcLocInfo, DeviceID,
8921 assert(MapperFunc &&
"MapperFunc missing for standalone target data");
8925 if (Info.HasNoWait) {
8935 if (Info.HasNoWait) {
8939 emitBlock(OffloadContBlock, CurFn,
true);
8945 bool RequiresOuterTargetTask = Info.HasNoWait;
8946 if (!RequiresOuterTargetTask)
8947 cantFail(TaskBodyCB(
nullptr,
nullptr,
8951 {}, RTArgs, Info.HasNoWait));
8954 omp::OMPRTL___tgt_target_data_begin_mapper);
8958 for (
auto DeviceMap : Info.DevicePtrInfoMap) {
8962 Builder.CreateStore(LI, DeviceMap.second.second);
8999 Value *PointerNum =
Builder.getInt32(Info.NumberOfPtrs);
9008 Value *OffloadingArgs[] = {SrcLocInfo, DeviceID,
9031 return emitIfClause(IfCond, BeginThenGen, BeginElseGen, AllocaIP);
9032 return BeginThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9047 return emitIfClause(IfCond, EndThenGen, EndElseGen, AllocaIP);
9048 return EndThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9051 return emitIfClause(IfCond, BeginThenGen, EndElseGen, AllocaIP);
9052 return BeginThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
9063 bool IsGPUDistribute) {
9064 assert((IVSize == 32 || IVSize == 64) &&
9065 "IV size is not compatible with the omp runtime");
9067 if (IsGPUDistribute)
9069 ? (IVSigned ? omp::OMPRTL___kmpc_distribute_static_init_4
9070 : omp::OMPRTL___kmpc_distribute_static_init_4u)
9071 : (IVSigned ? omp::OMPRTL___kmpc_distribute_static_init_8
9072 : omp::OMPRTL___kmpc_distribute_static_init_8u);
9074 Name = IVSize == 32 ? (IVSigned ? omp::OMPRTL___kmpc_for_static_init_4
9075 : omp::OMPRTL___kmpc_for_static_init_4u)
9076 : (IVSigned ? omp::OMPRTL___kmpc_for_static_init_8
9077 : omp::OMPRTL___kmpc_for_static_init_8u);
9084 assert((IVSize == 32 || IVSize == 64) &&
9085 "IV size is not compatible with the omp runtime");
9087 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_init_4
9088 : omp::OMPRTL___kmpc_dispatch_init_4u)
9089 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_init_8
9090 : omp::OMPRTL___kmpc_dispatch_init_8u);
9097 assert((IVSize == 32 || IVSize == 64) &&
9098 "IV size is not compatible with the omp runtime");
9100 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_next_4
9101 : omp::OMPRTL___kmpc_dispatch_next_4u)
9102 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_next_8
9103 : omp::OMPRTL___kmpc_dispatch_next_8u);
9110 assert((IVSize == 32 || IVSize == 64) &&
9111 "IV size is not compatible with the omp runtime");
9113 ? (IVSigned ? omp::OMPRTL___kmpc_dispatch_fini_4
9114 : omp::OMPRTL___kmpc_dispatch_fini_4u)
9115 : (IVSigned ? omp::OMPRTL___kmpc_dispatch_fini_8
9116 : omp::OMPRTL___kmpc_dispatch_fini_8u);
9127 DenseMap<
Value *, std::tuple<Value *, unsigned>> &ValueReplacementMap) {
9135 auto GetUpdatedDIVariable = [&](
DILocalVariable *OldVar,
unsigned arg) {
9139 if (NewVar && (arg == NewVar->
getArg()))
9149 auto UpdateDebugRecord = [&](
auto *DR) {
9152 for (
auto Loc : DR->location_ops()) {
9153 auto Iter = ValueReplacementMap.find(
Loc);
9154 if (Iter != ValueReplacementMap.end()) {
9155 DR->replaceVariableLocationOp(
Loc, std::get<0>(Iter->second));
9156 ArgNo = std::get<1>(Iter->second) + 1;
9160 DR->setVariable(GetUpdatedDIVariable(OldVar, ArgNo));
9165 if (DVR->getNumVariableLocationOps() != 1u) {
9166 DVR->setKillLocation();
9169 Value *
Loc = DVR->getVariableLocationOp(0u);
9176 RequiredBB = &DVR->getFunction()->getEntryBlock();
9178 if (RequiredBB && RequiredBB != CurBB) {
9190 "Unexpected debug intrinsic");
9192 UpdateDebugRecord(&DVR);
9193 MoveDebugRecordToCorrectBlock(&DVR);
9196 for (
auto *DVR : DVRsToDelete)
9197 DVR->getMarker()->MarkedInstr->dropOneDbgRecord(DVR);
9201 Module *M = Func->getParent();
9204 DB.createQualifiedType(dwarf::DW_TAG_pointer_type,
nullptr);
9205 unsigned ArgNo = Func->arg_size();
9207 NewSP,
"dyn_ptr", ArgNo, NewSP->
getFile(), 0, VoidPtrTy,
9208 false, DINode::DIFlags::FlagArtificial);
9210 Argument *LastArg = Func->getArg(Func->arg_size() - 1);
9211 DB.insertDeclare(LastArg, Var, DB.createExpression(),
Loc,
9232 for (
auto &Arg : Inputs)
9233 ParameterTypes.
push_back(Arg->getType()->isPointerTy()
9237 for (
auto &Arg : Inputs)
9238 ParameterTypes.
push_back(Arg->getType());
9246 auto BB = Builder.GetInsertBlock();
9247 auto M = BB->getModule();
9258 if (TargetCpuAttr.isStringAttribute())
9259 Func->addFnAttr(TargetCpuAttr);
9261 auto TargetFeaturesAttr = ParentFn->
getFnAttribute(
"target-features");
9262 if (TargetFeaturesAttr.isStringAttribute())
9263 Func->addFnAttr(TargetFeaturesAttr);
9268 OMPBuilder.
emitUsed(
"llvm.compiler.used", {ExecMode});
9279 Builder.SetInsertPoint(EntryBB);
9285 BasicBlock *UserCodeEntryBB = Builder.GetInsertBlock();
9295 splitBB(Builder,
true,
"outlined.body");
9302 Builder.SetInsertPoint(ExitBB);
9309 Builder.CreateRetVoid();
9313 auto AllocaIP = Builder.saveIP();
9318 const auto &ArgRange =
make_range(Func->arg_begin(), Func->arg_end() - 1);
9350 if (Instr->getFunction() == Func)
9351 Instr->replaceUsesOfWith(
Input, InputCopy);
9357 for (
auto InArg :
zip(Inputs, ArgRange)) {
9359 Argument &Arg = std::get<1>(InArg);
9360 Value *InputCopy =
nullptr;
9363 Arg,
Input, InputCopy, AllocaIP, Builder.saveIP(),
9367 Builder.restoreIP(*AfterIP);
9368 ValueReplacementMap[
Input] = std::make_tuple(InputCopy, Arg.
getArgNo());
9388 DeferredReplacement.push_back(std::make_pair(
Input, InputCopy));
9395 ReplaceValue(
Input, InputCopy, Func);
9399 for (
auto Deferred : DeferredReplacement)
9400 ReplaceValue(std::get<0>(Deferred), std::get<1>(Deferred), Func);
9403 ValueReplacementMap);
9411 Value *TaskWithPrivates,
9412 Type *TaskWithPrivatesTy) {
9414 Type *TaskTy = OMPIRBuilder.Task;
9417 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskWithPrivates, 0);
9418 Value *Shareds = TaskT;
9428 if (TaskWithPrivatesTy != TaskTy)
9429 Shareds = Builder.CreateStructGEP(TaskTy, TaskT, 0);
9446 const size_t NumOffloadingArrays,
const int SharedArgsOperandNo) {
9451 assert((!NumOffloadingArrays || PrivatesTy) &&
9452 "PrivatesTy cannot be nullptr when there are offloadingArrays"
9485 Type *TaskPtrTy = OMPBuilder.TaskPtr;
9486 [[maybe_unused]]
Type *TaskTy = OMPBuilder.Task;
9492 ".omp_target_task_proxy_func", M);
9493 Value *ThreadId = ProxyFn->getArg(0);
9494 Value *TaskWithPrivates = ProxyFn->getArg(1);
9495 ThreadId->
setName(
"thread.id");
9496 TaskWithPrivates->
setName(
"task");
9498 bool HasShareds = SharedArgsOperandNo > 0;
9499 bool HasOffloadingArrays = NumOffloadingArrays > 0;
9502 Builder.SetInsertPoint(EntryBB);
9508 if (HasOffloadingArrays) {
9509 assert(TaskTy != TaskWithPrivatesTy &&
9510 "If there are offloading arrays to pass to the target"
9511 "TaskTy cannot be the same as TaskWithPrivatesTy");
9514 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskWithPrivates, 1);
9515 for (
unsigned int i = 0; i < NumOffloadingArrays; ++i)
9517 Builder.CreateStructGEP(PrivatesTy, Privates, i));
9521 auto *ArgStructAlloca =
9523 assert(ArgStructAlloca &&
9524 "Unable to find the alloca instruction corresponding to arguments "
9525 "for extracted function");
9527 std::optional<TypeSize> ArgAllocSize =
9529 assert(ArgStructType && ArgAllocSize &&
9530 "Unable to determine size of arguments for extracted function");
9531 uint64_t StructSize = ArgAllocSize->getFixedValue();
9534 Builder.CreateAlloca(ArgStructType,
nullptr,
"structArg");
9536 Value *SharedsSize = Builder.getInt64(StructSize);
9539 OMPBuilder, Builder, TaskWithPrivates, TaskWithPrivatesTy);
9541 Builder.CreateMemCpy(
9542 NewArgStructAlloca, NewArgStructAlloca->
getAlign(), LoadShared,
9544 KernelLaunchArgs.
push_back(NewArgStructAlloca);
9547 Builder.CreateRetVoid();
9553 return GEP->getSourceElementType();
9555 return Alloca->getAllocatedType();
9578 if (OffloadingArraysToPrivatize.
empty())
9579 return OMPIRBuilder.Task;
9582 for (
Value *V : OffloadingArraysToPrivatize) {
9583 assert(V->getType()->isPointerTy() &&
9584 "Expected pointer to array to privatize. Got a non-pointer value "
9587 assert(ArrayTy &&
"ArrayType cannot be nullptr");
9593 "struct.task_with_privates");
9607 EntryFnName, Inputs, CBFunc,
9612 EntryInfo, GenerateOutlinedFunction, IsOffloadEntry, OutlinedFn,
9749 TargetTaskAllocaBB->
begin());
9752 auto OI = std::make_unique<OutlineInfo>();
9753 OI->EntryBB = TargetTaskAllocaBB;
9754 OI->OuterAllocBB = AllocaIP.
getBlock();
9759 Builder, AllocaIP, ToBeDeleted, TargetTaskAllocaIP,
"global.tid",
false));
9762 Builder.restoreIP(TargetTaskBodyIP);
9763 if (
Error Err = TaskBodyCB(DeviceID, RTLoc, TargetTaskAllocaIP))
9781 bool NeedsTargetTask = HasNoWait && DeviceID;
9782 if (NeedsTargetTask) {
9788 OffloadingArraysToPrivatize.
push_back(V);
9789 OI->ExcludeArgsFromAggregate.push_back(V);
9793 OI->PostOutlineCB = [
this, ToBeDeleted, Dependencies, NeedsTargetTask,
9794 DeviceID, OffloadingArraysToPrivatize](
9797 "there must be a single user for the outlined function");
9811 const unsigned int NumStaleCIArgs = StaleCI->
arg_size();
9812 bool HasShareds = NumStaleCIArgs > OffloadingArraysToPrivatize.
size() + 1;
9814 NumStaleCIArgs == (OffloadingArraysToPrivatize.
size() + 2)) &&
9815 "Wrong number of arguments for StaleCI when shareds are present");
9816 int SharedArgOperandNo =
9817 HasShareds ? OffloadingArraysToPrivatize.
size() + 1 : 0;
9823 if (!OffloadingArraysToPrivatize.
empty())
9828 *
this,
Builder, StaleCI, PrivatesTy, TaskWithPrivatesTy,
9829 OffloadingArraysToPrivatize.
size(), SharedArgOperandNo);
9831 LLVM_DEBUG(
dbgs() <<
"Proxy task entry function created: " << *ProxyFn
9834 Builder.SetInsertPoint(StaleCI);
9851 OMPRTL___kmpc_omp_target_task_alloc);
9863 M.getDataLayout().getTypeStoreSize(TaskWithPrivatesTy));
9870 auto *ArgStructAlloca =
9872 assert(ArgStructAlloca &&
9873 "Unable to find the alloca instruction corresponding to arguments "
9874 "for extracted function");
9875 std::optional<TypeSize> ArgAllocSize =
9878 "Unable to determine size of arguments for extracted function");
9879 SharedsSize =
Builder.getInt64(ArgAllocSize->getFixedValue());
9898 TaskSize, SharedsSize,
9901 if (NeedsTargetTask) {
9902 assert(DeviceID &&
"Expected non-empty device ID.");
9912 *
this,
Builder, TaskData, TaskWithPrivatesTy);
9913 Builder.CreateMemCpy(TaskShareds, Alignment, Shareds, Alignment,
9916 if (!OffloadingArraysToPrivatize.
empty()) {
9918 Builder.CreateStructGEP(TaskWithPrivatesTy, TaskData, 1);
9919 for (
unsigned int i = 0; i < OffloadingArraysToPrivatize.
size(); ++i) {
9920 Value *PtrToPrivatize = OffloadingArraysToPrivatize[i];
9927 "ElementType should match ArrayType");
9930 Value *Dst =
Builder.CreateStructGEP(PrivatesTy, Privates, i);
9932 Dst, Alignment, PtrToPrivatize, Alignment,
9933 Builder.getInt64(
M.getDataLayout().getTypeStoreSize(ElementType)));
9937 Value *DepArray =
nullptr;
9938 Value *NumDeps =
nullptr;
9941 NumDeps = Dependencies.
NumDeps;
9942 }
else if (!Dependencies.
Deps.empty()) {
9944 NumDeps =
Builder.getInt32(Dependencies.
Deps.size());
9955 if (!NeedsTargetTask) {
9964 ConstantInt::get(
Builder.getInt32Ty(), 0),
9977 }
else if (DepArray) {
9985 {Ident, ThreadID, TaskData, NumDeps, DepArray,
9986 ConstantInt::get(
Builder.getInt32Ty(), 0),
9994 Builder.ClearInsertionPoint();
9997 I->eraseFromParent();
10002 << *(
Builder.GetInsertBlock()) <<
"\n");
10004 << *(
Builder.GetInsertBlock()->getParent()->getParent())
10016 CustomMapperCB, IsNonContiguous, DeviceAddrCB))
10039 Builder.restoreIP(IP);
10045 return Builder.saveIP();
10048 bool HasDependencies = !Dependencies.
empty();
10049 bool RequiresOuterTargetTask = HasNoWait || HasDependencies;
10066 if (OutlinedFnID && DeviceID)
10068 EmitTargetCallFallbackCB, KArgs,
10069 DeviceID, RTLoc, TargetTaskAllocaIP);
10077 return EmitTargetCallFallbackCB(OMPBuilder.
Builder.
saveIP());
10084 auto &&EmitTargetCallElse =
10091 if (RequiresOuterTargetTask) {
10098 Dependencies, EmptyRTArgs, HasNoWait);
10100 return EmitTargetCallFallbackCB(Builder.saveIP());
10103 Builder.restoreIP(AfterIP);
10107 auto &&EmitTargetCallThen =
10111 Info.HasNoWait = HasNoWait;
10116 AllocaIP, Builder.saveIP(), Info, RTArgs, MapInfo, CustomMapperCB,
10122 for (
auto [DefaultVal, RuntimeVal] :
10124 NumTeamsC.
push_back(RuntimeVal ? RuntimeVal
10125 : Builder.getInt32(DefaultVal));
10129 auto InitMaxThreadsClause = [&Builder](
Value *
Clause) {
10131 Clause = Builder.CreateIntCast(
Clause, Builder.getInt32Ty(),
10135 auto CombineMaxThreadsClauses = [&Builder](
Value *
Clause,
Value *&Result) {
10138 Result ? Builder.CreateSelect(Builder.CreateICmpULT(Result,
Clause),
10146 Value *MaxThreadsClause =
10148 ? InitMaxThreadsClause(RuntimeAttrs.
MaxThreads.front())
10151 for (
auto [TeamsVal, TargetVal] :
zip_equal(
10153 Value *TeamsThreadLimitClause = InitMaxThreadsClause(TeamsVal);
10154 Value *NumThreads = InitMaxThreadsClause(TargetVal);
10156 CombineMaxThreadsClauses(TeamsThreadLimitClause, NumThreads);
10157 CombineMaxThreadsClauses(MaxThreadsClause, NumThreads);
10159 NumThreadsC.
push_back(NumThreads ? NumThreads : Builder.getInt32(0));
10162 unsigned NumTargetItems = Info.NumberOfPtrs;
10170 Builder.getInt64Ty(),
10172 : Builder.getInt64(0);
10176 DynCGroupMem = Builder.getInt32(0);
10179 NumTargetItems, RTArgs, TripCount, NumTeamsC, NumThreadsC, DynCGroupMem,
10180 HasNoWait,
false,
false,
10181 DynCGroupMemFallback);
10188 if (RequiresOuterTargetTask)
10190 RTLoc, AllocaIP, Dependencies,
10191 KArgs.
RTArgs, Info.HasNoWait);
10194 Builder, OutlinedFnID, EmitTargetCallFallbackCB, KArgs,
10195 RuntimeAttrs.
DeviceID, RTLoc, AllocaIP);
10198 Builder.restoreIP(AfterIP);
10205 if (!OutlinedFnID) {
10206 cantFail(EmitTargetCallElse(AllocaIP, Builder.saveIP(), DeallocBlocks));
10212 cantFail(EmitTargetCallThen(AllocaIP, Builder.saveIP(), DeallocBlocks));
10217 EmitTargetCallElse, AllocaIP));
10230 bool HasNowait,
Value *DynCGroupMem,
10236 Builder.restoreIP(CodeGenIP);
10244 *
this,
Builder, IsOffloadEntry, EntryInfo, DefaultAttrs, OutlinedFn,
10245 OutlinedFnID, Inputs, CBFunc, ArgAccessorFuncCB))
10251 if (!
Config.isTargetDevice())
10253 RuntimeAttrs, IfCond, OutlinedFn, OutlinedFnID, Inputs,
10254 GenMapInfoCB, CustomMapperCB, Dependencies, HasNowait,
10255 DynCGroupMem, DynCGroupMemFallback);
10269 return OS.
str().str();
10274 return OpenMPIRBuilder::getNameWithSeparators(Parts,
Config.firstSeparator(),
10280 auto &Elem = *
InternalVars.try_emplace(Name,
nullptr).first;
10282 assert(Elem.second->getValueType() == Ty &&
10283 "OMP internal variable has different type than requested");
10296 :
M.getTargetTriple().isAMDGPU()
10298 :
DL.getDefaultGlobalsAddressSpace();
10299 auto Linkage = this->
M.getTargetTriple().isWasm()
10307 const llvm::Align PtrAlign =
DL.getPointerABIAlignment(AddressSpaceVal);
10308 GV->setAlignment(std::max(TypeAlign, PtrAlign));
10312 return Elem.second;
10315Value *OpenMPIRBuilder::getOMPCriticalRegionLock(
StringRef CriticalName) {
10316 std::string Prefix =
Twine(
"gomp_critical_user_", CriticalName).
str();
10317 std::string Name = getNameWithSeparators({Prefix,
"var"},
".",
".");
10328 return SizePtrToInt;
10333 std::string VarName) {
10341 return MaptypesArrayGlobal;
10346 unsigned NumOperands,
10355 ArrI8PtrTy,
nullptr,
".offload_baseptrs");
10359 ArrI64Ty,
nullptr,
".offload_sizes");
10370 int64_t DeviceID,
unsigned NumOperands) {
10376 Value *ArgsBaseGEP =
10378 {Builder.getInt32(0), Builder.getInt32(0)});
10381 {Builder.getInt32(0), Builder.getInt32(0)});
10382 Value *ArgSizesGEP =
10384 {Builder.getInt32(0), Builder.getInt32(0)});
10388 Builder.getInt32(NumOperands),
10389 ArgsBaseGEP, ArgsGEP, ArgSizesGEP,
10390 MaptypesArg, MapnamesArg, NullPtr});
10397 assert((!ForEndCall || Info.separateBeginEndCalls()) &&
10398 "expected region end call to runtime only when end call is separate");
10400 auto VoidPtrTy = UnqualPtrTy;
10401 auto VoidPtrPtrTy = UnqualPtrTy;
10403 auto Int64PtrTy = UnqualPtrTy;
10405 if (!Info.NumberOfPtrs) {
10417 Info.RTArgs.BasePointersArray,
10420 ArrayType::get(VoidPtrTy, Info.NumberOfPtrs), Info.RTArgs.PointersArray,
10424 ArrayType::get(Int64Ty, Info.NumberOfPtrs), Info.RTArgs.SizesArray,
10428 ForEndCall && Info.RTArgs.MapTypesArrayEnd ? Info.RTArgs.MapTypesArrayEnd
10429 : Info.RTArgs.MapTypesArray,
10435 if (!Info.EmitDebug)
10439 ArrayType::get(VoidPtrTy, Info.NumberOfPtrs), Info.RTArgs.MapNamesArray,
10444 if (!Info.HasMapper)
10448 Builder.CreatePointerCast(Info.RTArgs.MappersArray, VoidPtrPtrTy);
10469 "struct.descriptor_dim");
10471 enum { OffsetFD = 0, CountFD, StrideFD };
10475 for (
unsigned I = 0, L = 0, E = NonContigInfo.
Dims.
size();
I < E; ++
I) {
10478 if (NonContigInfo.
Dims[
I] == 1)
10483 Builder.CreateAlloca(ArrayTy,
nullptr,
"dims");
10484 Builder.restoreIP(CodeGenIP);
10485 for (
unsigned II = 0, EE = NonContigInfo.
Dims[
I];
II < EE; ++
II) {
10486 unsigned RevIdx = EE -
II - 1;
10490 Value *OffsetLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, OffsetFD);
10492 NonContigInfo.
Offsets[L][RevIdx], OffsetLVal,
10493 M.getDataLayout().getPrefTypeAlign(OffsetLVal->
getType()));
10495 Value *CountLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, CountFD);
10497 NonContigInfo.
Counts[L][RevIdx], CountLVal,
10498 M.getDataLayout().getPrefTypeAlign(CountLVal->
getType()));
10500 Value *StrideLVal =
Builder.CreateStructGEP(DimTy, DimsLVal, StrideFD);
10502 NonContigInfo.
Strides[L][RevIdx], StrideLVal,
10503 M.getDataLayout().getPrefTypeAlign(CountLVal->
getType()));
10506 Builder.restoreIP(CodeGenIP);
10507 Value *DAddr =
Builder.CreatePointerBitCastOrAddrSpaceCast(
10508 DimsAddr,
Builder.getPtrTy());
10511 Info.RTArgs.PointersArray, 0,
I);
10513 DAddr,
P,
M.getDataLayout().getPrefTypeAlign(
Builder.getPtrTy()));
10518void OpenMPIRBuilder::emitUDMapperArrayInitOrDel(
10522 StringRef Prefix = IsInit ?
".init" :
".del";
10528 Builder.CreateICmpSGT(
Size, Builder.getInt64(1),
"omp.arrayinit.isarray");
10529 Value *DeleteBit = Builder.CreateAnd(
10532 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10533 OpenMPOffloadMappingFlags::OMP_MAP_DELETE)));
10538 Value *BaseIsBegin = Builder.CreateICmpNE(
Base, Begin);
10539 Cond = Builder.CreateOr(IsArray, BaseIsBegin);
10540 DeleteCond = Builder.CreateIsNull(
10545 DeleteCond =
Builder.CreateIsNotNull(
10561 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10562 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10563 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10564 MapTypeArg =
Builder.CreateOr(
10567 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10568 OpenMPOffloadMappingFlags::OMP_MAP_IMPLICIT)));
10572 Value *OffloadingArgs[] = {MapperHandle,
Base, Begin,
10573 ArraySize, MapTypeArg, MapName};
10584 bool PreserveMemberOfFlags,
bool PropagatePresentToPointee) {
10600 MapperFn->
addFnAttr(Attribute::NoInline);
10601 MapperFn->
addFnAttr(Attribute::NoUnwind);
10612 Builder.SetInsertPoint(EntryBB);
10625 TypeSize ElementSize =
M.getDataLayout().getTypeStoreSize(ElemTy);
10627 Value *PtrBegin = BeginIn;
10633 emitUDMapperArrayInitOrDel(MapperFn, MapperHandle, BaseIn, BeginIn,
Size,
10634 MapType, MapName, ElementSize, HeadBB,
10645 Builder.CreateICmpEQ(PtrBegin, PtrEnd,
"omp.arraymap.isempty");
10646 Builder.CreateCondBr(IsEmpty, DoneBB, BodyBB);
10652 Builder.CreatePHI(PtrBegin->
getType(), 2,
"omp.arraymap.ptrcurrent");
10653 PtrPHI->addIncoming(PtrBegin, HeadBB);
10658 return Info.takeError();
10662 Value *OffloadingArgs[] = {MapperHandle};
10666 Value *ShiftedPreviousSize =
10670 for (
unsigned I = 0;
I < Info->BasePointers.size(); ++
I) {
10671 Value *CurBaseArg = Info->BasePointers[
I];
10672 Value *CurBeginArg = Info->Pointers[
I];
10673 Value *CurSizeArg = Info->Sizes[
I];
10674 Value *CurNameArg = Info->Names.size()
10679 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10682 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10684 constexpr uint64_t MemberOfMask =
10685 static_cast<uint64_t
>(OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF);
10686 constexpr uint64_t AttachBit =
10687 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10688 OpenMPOffloadMappingFlags::OMP_MAP_ATTACH);
10746 Value *MemberMapType;
10747 if (PreserveMemberOfFlags || (RawType & AttachBit) ||
10748 Info->HasAttachPtr[
I]) {
10749 if (RawType & MemberOfMask)
10750 MemberMapType =
Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10752 MemberMapType = OriMapType;
10754 MemberMapType =
Builder.CreateNUWAdd(OriMapType, ShiftedPreviousSize);
10772 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10773 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10774 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10784 Builder.CreateCondBr(IsAlloc, AllocBB, AllocElseBB);
10790 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10791 OpenMPOffloadMappingFlags::OMP_MAP_TO |
10792 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10798 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10799 OpenMPOffloadMappingFlags::OMP_MAP_TO)));
10800 Builder.CreateCondBr(IsTo, ToBB, ToElseBB);
10806 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10807 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10813 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10814 OpenMPOffloadMappingFlags::OMP_MAP_FROM)));
10815 Builder.CreateCondBr(IsFrom, FromBB, EndBB);
10821 ~
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10822 OpenMPOffloadMappingFlags::OMP_MAP_TO)));
10831 CurMapType->
addIncoming(MemberMapType, ToElseBB);
10868 uint64_t ModifierBits =
10869 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10870 OpenMPOffloadMappingFlags::OMP_MAP_ALWAYS |
10871 OpenMPOffloadMappingFlags::OMP_MAP_DELETE |
10872 OpenMPOffloadMappingFlags::OMP_MAP_CLOSE);
10873 if (PropagatePresentToPointee && Info->HasAttachPtr[
I])
10875 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10876 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
10877 Value *ImportedModifierBits =
10880 CurMapType, ImportedModifierBits,
"omp.maptype.with.modifiers");
10885 Value *FinalMapType =
10886 (RawType & AttachBit) ? CurMapType : CurMapTypeWithModifiers;
10888 Value *OffloadingArgs[] = {MapperHandle, CurBaseArg, CurBeginArg,
10889 CurSizeArg, FinalMapType, CurNameArg};
10891 auto ChildMapperFn = CustomMapperCB(
I);
10892 if (!ChildMapperFn)
10893 return ChildMapperFn.takeError();
10894 if (*ChildMapperFn) {
10909 Value *PtrNext =
Builder.CreateConstGEP1_32(ElemTy, PtrPHI, 1,
10910 "omp.arraymap.next");
10911 PtrPHI->addIncoming(PtrNext, LastBB);
10912 Value *IsDone =
Builder.CreateICmpEQ(PtrNext, PtrEnd,
"omp.arraymap.isdone");
10914 Builder.CreateCondBr(IsDone, ExitBB, BodyBB);
10919 emitUDMapperArrayInitOrDel(MapperFn, MapperHandle, BaseIn, BeginIn,
Size,
10920 MapType, MapName, ElementSize, DoneBB,
10933 bool IsNonContiguous,
10937 Info.clearArrayInfo();
10940 if (Info.NumberOfPtrs == 0)
10949 Info.RTArgs.BasePointersArray =
Builder.CreateAlloca(
10950 PointerArrayType,
nullptr,
".offload_baseptrs");
10952 Info.RTArgs.PointersArray =
Builder.CreateAlloca(
10953 PointerArrayType,
nullptr,
".offload_ptrs");
10955 PointerArrayType,
nullptr,
".offload_mappers");
10956 Info.RTArgs.MappersArray = MappersArray;
10963 ConstantInt::get(Int64Ty, 0));
10965 for (
unsigned I = 0, E = CombinedInfo.
Sizes.
size();
I < E; ++
I) {
10966 bool IsNonContigEntry =
10968 (
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
10970 OpenMPOffloadMappingFlags::OMP_MAP_NON_CONTIG) != 0);
10973 if (IsNonContigEntry) {
10975 "Index must be in-bounds for NON_CONTIG Dims array");
10977 assert(DimCount > 0 &&
"NON_CONTIG DimCount must be > 0");
10978 ConstSizes[
I] = ConstantInt::get(Int64Ty, DimCount);
10983 ConstSizes[
I] = CI;
10987 RuntimeSizes.
set(
I);
10990 if (RuntimeSizes.
all()) {
10992 Info.RTArgs.SizesArray =
Builder.CreateAlloca(
10993 SizeArrayType,
nullptr,
".offload_sizes");
10999 auto *SizesArrayGbl =
11004 if (!RuntimeSizes.
any()) {
11005 Info.RTArgs.SizesArray = SizesArrayGbl;
11007 unsigned IndexSize =
M.getDataLayout().getIndexSizeInBits(0);
11008 Align OffloadSizeAlign =
M.getDataLayout().getABIIntegerTypeAlignment(64);
11011 SizeArrayType,
nullptr,
".offload_sizes");
11015 Buffer,
M.getDataLayout().getPrefTypeAlign(Buffer->
getType()),
11016 SizesArrayGbl, OffloadSizeAlign,
11021 Info.RTArgs.SizesArray = Buffer;
11029 for (
auto mapFlag : CombinedInfo.
Types)
11031 static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
11035 Info.RTArgs.MapTypesArray = MapTypesArrayGbl;
11041 Info.RTArgs.MapNamesArray = MapNamesArrayGbl;
11042 Info.EmitDebug =
true;
11044 Info.RTArgs.MapNamesArray =
11046 Info.EmitDebug =
false;
11051 if (Info.separateBeginEndCalls()) {
11052 bool EndMapTypesDiffer =
false;
11053 for (uint64_t &
Type : Mapping) {
11054 if (
Type &
static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>
>(
11055 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT)) {
11056 Type &= ~static_cast<std::underlying_type_t<OpenMPOffloadMappingFlags>>(
11057 OpenMPOffloadMappingFlags::OMP_MAP_PRESENT);
11058 EndMapTypesDiffer =
true;
11061 if (EndMapTypesDiffer) {
11063 Info.RTArgs.MapTypesArrayEnd = MapTypesArrayGbl;
11068 for (
unsigned I = 0;
I < Info.NumberOfPtrs; ++
I) {
11071 ArrayType::get(PtrTy, Info.NumberOfPtrs), Info.RTArgs.BasePointersArray,
11073 Builder.CreateAlignedStore(BPVal, BP,
11074 M.getDataLayout().getPrefTypeAlign(PtrTy));
11076 if (Info.requiresDevicePointerInfo()) {
11078 CodeGenIP =
Builder.saveIP();
11080 Info.DevicePtrInfoMap[BPVal] = {BP,
Builder.CreateAlloca(PtrTy)};
11083 DeviceAddrCB(
I, Info.DevicePtrInfoMap[BPVal].second);
11085 Info.DevicePtrInfoMap[BPVal] = {BP, BP};
11087 DeviceAddrCB(
I, BP);
11093 ArrayType::get(PtrTy, Info.NumberOfPtrs), Info.RTArgs.PointersArray, 0,
11096 Builder.CreateAlignedStore(PVal,
P,
11097 M.getDataLayout().getPrefTypeAlign(PtrTy));
11099 if (RuntimeSizes.
test(
I)) {
11101 ArrayType::get(Int64Ty, Info.NumberOfPtrs), Info.RTArgs.SizesArray,
11107 S,
M.getDataLayout().getPrefTypeAlign(PtrTy));
11110 unsigned IndexSize =
M.getDataLayout().getIndexSizeInBits(0);
11113 auto CustomMFunc = CustomMapperCB(
I);
11115 return CustomMFunc.takeError();
11117 MFunc =
Builder.CreatePointerCast(*CustomMFunc, PtrTy);
11120 PointerArrayType, MappersArray,
11123 MFunc, MAddr,
M.getDataLayout().getPrefTypeAlign(MAddr->
getType()));
11127 Info.NumberOfPtrs == 0)
11144 Builder.ClearInsertionPoint();
11175 auto CondConstant = CI->getSExtValue();
11177 return ThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
11179 return ElseGen(AllocaIP,
Builder.saveIP(), DeallocBlocks);
11189 Builder.CreateCondBr(
Cond, ThenBlock, ElseBlock);
11192 if (
Error Err = ThenGen(AllocaIP,
Builder.saveIP(), DeallocBlocks))
11198 if (
Error Err = ElseGen(AllocaIP,
Builder.saveIP(), DeallocBlocks))
11207bool OpenMPIRBuilder::checkAndEmitFlushAfterAtomic(
11211 "Unexpected Atomic Ordering.");
11213 bool Flush =
false;
11275 assert(
X.Var->getType()->isPointerTy() &&
11276 "OMP Atomic expects a pointer to target memory");
11277 Type *XElemTy =
X.ElemTy;
11280 "OMP atomic read expected a scalar type");
11282 Value *XRead =
nullptr;
11286 Builder.CreateLoad(XElemTy,
X.Var,
X.IsVolatile,
"omp.atomic.read");
11295 unsigned LoadSize =
DL.getTypeStoreSize(XElemTy);
11298 OldVal->
getAlign(),
true , AllocaIP,
X.Var);
11300 XRead = AtomicLoadRes.first;
11307 Builder.CreateLoad(IntCastTy,
X.Var,
X.IsVolatile,
"omp.atomic.load");
11310 XRead =
Builder.CreateBitCast(XLoad, XElemTy,
"atomic.flt.cast");
11312 XRead =
Builder.CreateIntToPtr(XLoad, XElemTy,
"atomic.ptr.cast");
11315 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Read);
11316 Builder.CreateStore(XRead, V.Var, V.IsVolatile);
11327 assert(
X.Var->getType()->isPointerTy() &&
11328 "OMP Atomic expects a pointer to target memory");
11329 Type *XElemTy =
X.ElemTy;
11332 "OMP atomic write expected a scalar type");
11340 unsigned LoadSize =
DL.getTypeStoreSize(XElemTy);
11343 OldVal->
getAlign(),
true , AllocaIP,
X.Var);
11351 Builder.CreateBitCast(Expr, IntCastTy,
"atomic.src.int.cast");
11356 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Write);
11363 AtomicUpdateCallbackTy &UpdateOp,
bool IsXBinopExpr,
11364 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11370 Type *XTy =
X.Var->getType();
11372 "OMP Atomic expects a pointer to target memory");
11373 Type *XElemTy =
X.ElemTy;
11376 "OMP atomic update expected a scalar or struct type");
11379 "OpenMP atomic does not support LT or GT operations");
11383 AllocaIP,
X.Var,
X.ElemTy, Expr, AO, RMWOp, UpdateOp,
X.IsVolatile,
11384 IsXBinopExpr, IsIgnoreDenormalMode, IsFineGrainedMemory, IsRemoteMemory);
11386 return AtomicResult.takeError();
11387 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Update);
11392Value *OpenMPIRBuilder::emitRMWOpAsInstruction(
Value *Src1,
Value *Src2,
11396 return Builder.CreateAdd(Src1, Src2);
11398 return Builder.CreateSub(Src1, Src2);
11400 return Builder.CreateAnd(Src1, Src2);
11402 return Builder.CreateNeg(Builder.CreateAnd(Src1, Src2));
11404 return Builder.CreateOr(Src1, Src2);
11406 return Builder.CreateXor(Src1, Src2);
11445Expected<std::pair<Value *, Value *>> OpenMPIRBuilder::emitAtomicUpdate(
11448 AtomicUpdateCallbackTy &UpdateOp,
bool VolatileX,
bool IsXBinopExpr,
11449 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11451 bool emitRMWOp =
false;
11459 emitRMWOp = XElemTy;
11462 emitRMWOp = (IsXBinopExpr && XElemTy);
11469 std::pair<Value *, Value *> Res;
11471 AtomicRMWInst *RMWInst =
11472 Builder.CreateAtomicRMW(RMWOp,
X, Expr, llvm::MaybeAlign(), AO);
11473 if (
T.isAMDGPU()) {
11474 if (IsIgnoreDenormalMode)
11475 RMWInst->
setMetadata(
"amdgpu.ignore.denormal.mode",
11477 if (!IsFineGrainedMemory)
11478 RMWInst->
setMetadata(
"amdgpu.no.fine.grained.memory",
11480 if (!IsRemoteMemory)
11484 Res.first = RMWInst;
11489 Res.second = Res.first;
11491 Res.second = emitRMWOpAsInstruction(Res.first, Expr, RMWOp);
11494 Builder.CreateLoad(XElemTy,
X,
X->getName() +
".atomic.load");
11500 OpenMPIRBuilder::AtomicInfo atomicInfo(
11502 OldVal->
getAlign(),
true , AllocaIP,
X);
11503 auto AtomicLoadRes = atomicInfo.EmitAtomicLoadLibcall(AO);
11506 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11513 AllocaInst *NewAtomicAddr =
Builder.CreateAlloca(XElemTy);
11514 NewAtomicAddr->
setName(
X->getName() +
"x.new.val");
11515 Builder.SetInsertPoint(ContBB);
11517 PHI->addIncoming(AtomicLoadRes.first, CurBB);
11519 Expected<Value *> CBResult = UpdateOp(OldExprVal,
Builder);
11522 Value *Upd = *CBResult;
11523 Builder.CreateStore(Upd, NewAtomicAddr);
11526 auto Result = atomicInfo.EmitAtomicCompareExchangeLibcall(
11527 AtomicLoadRes.second, NewAtomicAddr, AO, Failure);
11528 LoadInst *PHILoad =
Builder.CreateLoad(XElemTy,
Result.first);
11529 PHI->addIncoming(PHILoad,
Builder.GetInsertBlock());
11532 Res.first = OldExprVal;
11535 if (UnreachableInst *ExitTI =
11538 Builder.SetInsertPoint(ExitBB);
11540 Builder.SetInsertPoint(ExitTI);
11543 IntegerType *IntCastTy =
11546 Builder.CreateLoad(IntCastTy,
X,
X->getName() +
".atomic.load");
11556 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11563 AllocaInst *NewAtomicAddr =
Builder.CreateAlloca(XElemTy);
11564 NewAtomicAddr->
setName(
X->getName() +
"x.new.val");
11565 Builder.SetInsertPoint(ContBB);
11567 PHI->addIncoming(OldVal, CurBB);
11572 OldExprVal =
Builder.CreateBitCast(
PHI, XElemTy,
11573 X->getName() +
".atomic.fltCast");
11575 OldExprVal =
Builder.CreateIntToPtr(
PHI, XElemTy,
11576 X->getName() +
".atomic.ptrCast");
11580 Expected<Value *> CBResult = UpdateOp(OldExprVal,
Builder);
11583 Value *Upd = *CBResult;
11584 Builder.CreateStore(Upd, NewAtomicAddr);
11585 LoadInst *DesiredVal =
Builder.CreateLoad(IntCastTy, NewAtomicAddr);
11589 X,
PHI, DesiredVal, llvm::MaybeAlign(), AO, Failure);
11590 Result->setVolatile(VolatileX);
11591 Value *PreviousVal =
Builder.CreateExtractValue(Result, 0);
11592 Value *SuccessFailureVal =
Builder.CreateExtractValue(Result, 1);
11593 PHI->addIncoming(PreviousVal,
Builder.GetInsertBlock());
11594 Builder.CreateCondBr(SuccessFailureVal, ExitBB, ContBB);
11596 Res.first = OldExprVal;
11600 if (UnreachableInst *ExitTI =
11603 Builder.SetInsertPoint(ExitBB);
11605 Builder.SetInsertPoint(ExitTI);
11616 bool UpdateExpr,
bool IsPostfixUpdate,
bool IsXBinopExpr,
11617 bool IsIgnoreDenormalMode,
bool IsFineGrainedMemory,
bool IsRemoteMemory) {
11622 Type *XTy =
X.Var->getType();
11624 "OMP Atomic expects a pointer to target memory");
11625 Type *XElemTy =
X.ElemTy;
11628 "OMP atomic capture expected a scalar or struct type");
11630 "OpenMP atomic does not support LT or GT operations");
11637 AllocaIP,
X.Var,
X.ElemTy, Expr, AO, AtomicOp, UpdateOp,
X.IsVolatile,
11638 IsXBinopExpr, IsIgnoreDenormalMode, IsFineGrainedMemory, IsRemoteMemory);
11641 Value *CapturedVal =
11642 (IsPostfixUpdate ? AtomicResult->first : AtomicResult->second);
11643 Builder.CreateStore(CapturedVal, V.Var, V.IsVolatile);
11645 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Capture);
11653 bool IsFailOnly,
bool IsWeak) {
11657 IsPostfixUpdate, IsFailOnly, Failure, IsWeak);
11669 assert(
X.Var->getType()->isPointerTy() &&
11670 "OMP atomic expects a pointer to target memory");
11673 assert(V.Var->getType()->isPointerTy() &&
"v.var must be of pointer type");
11674 assert(V.ElemTy ==
X.ElemTy &&
"x and v must be of same type");
11677 bool IsInteger = E->getType()->isIntegerTy();
11679 if (
Op == OMPAtomicCompareOp::EQ) {
11682 Value *OldValue =
nullptr;
11683 Value *SuccessOrFail =
nullptr;
11721 X.Var->getName() +
".atomic.load");
11727 Value *EIsNaN =
Builder.CreateFCmpUNO(E, E,
"atomic.e.isnan");
11728 Value *XIsNaN =
Builder.CreateFCmpUNO(XFP, XFP,
"atomic.x.isnan");
11729 Value *EitherNaN =
Builder.CreateOr(EIsNaN, XIsNaN,
"atomic.either.nan");
11734 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11738 M.getContext(),
X.Var->getName() +
".atomic.nan",
F, ExitBB);
11740 M.getContext(),
X.Var->getName() +
".atomic.notnan",
F, ExitBB);
11742 M.getContext(),
X.Var->getName() +
".atomic.zero",
F, ExitBB);
11744 M.getContext(),
X.Var->getName() +
".atomic.normal",
F, ExitBB);
11748 Builder.SetInsertPoint(CurBB);
11749 Builder.CreateCondBr(EitherNaN, NaNBB, NotNaNBB);
11752 Builder.SetInsertPoint(NaNBB);
11756 Builder.SetInsertPoint(NotNaNBB);
11759 X.Var->getName() +
".atomic.xiszero");
11761 "atomic.e.iszero");
11762 Value *BothZero =
Builder.CreateAnd(XIsZero, EIsZero,
"atomic.both.zero");
11763 Builder.CreateCondBr(BothZero, ZeroBB, NormalBB);
11766 Builder.SetInsertPoint(ZeroBB);
11768 X.Var, XCurr, DBCast,
MaybeAlign(), AO, Failure);
11770 Value *OldZero =
Builder.CreateExtractValue(ResZero, 0);
11771 Value *OkZero =
Builder.CreateExtractValue(ResZero, 1);
11775 Builder.SetInsertPoint(NormalBB);
11777 X.Var, EBCast, DBCast,
MaybeAlign(), AO, Failure);
11779 Value *OldNormal =
Builder.CreateExtractValue(ResNormal, 0);
11780 Value *OkNormal =
Builder.CreateExtractValue(ResNormal, 1);
11786 Builder.CreatePHI(IntCastTy, 3,
X.Var->getName() +
".atomic.old");
11791 X.Var->getName() +
".atomic.ok");
11798 Builder.SetInsertPoint(ExitBB);
11803 OldValue =
Builder.CreateBitCast(OldIntPHI,
X.ElemTy,
11804 X.Var->getName() +
".atomic.old.fp");
11805 SuccessOrFail = SuccessPHI;
11813 Result =
Builder.CreateAtomicCmpXchg(
X.Var, EBCast, DBCast,
11819 Result->setWeak(IsWeak);
11822 OldValue =
Builder.CreateExtractValue(Result, 0);
11824 OldValue =
Builder.CreateBitCast(OldValue,
X.ElemTy);
11826 "OldValue and V must be of same type");
11827 if (IsPostfixUpdate) {
11828 Builder.CreateStore(OldValue, V.Var, V.IsVolatile);
11830 SuccessOrFail =
Builder.CreateExtractValue(Result, 1);
11834 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11836 CurBBTI,
X.Var->getName() +
".atomic.exit");
11842 Builder.CreateCondBr(SuccessOrFail, ExitBB, ContBB);
11844 Builder.SetInsertPoint(ContBB);
11845 Builder.CreateStore(OldValue, V.Var);
11851 Builder.SetInsertPoint(ExitBB);
11853 Builder.SetInsertPoint(ExitTI);
11856 Value *CapturedValue =
11857 Builder.CreateSelect(SuccessOrFail, E, OldValue);
11858 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11864 assert(R.Var->getType()->isPointerTy() &&
11865 "r.var must be of pointer type");
11866 assert(R.ElemTy->isIntegerTy() &&
"r must be of integral type");
11868 Value *SuccessFailureVal =
11869 Builder.CreateExtractValue(Result, 1);
11870 Value *ResultCast =
11871 R.IsSigned ?
Builder.CreateSExt(SuccessFailureVal, R.ElemTy)
11872 :
Builder.CreateZExt(SuccessFailureVal, R.ElemTy);
11873 Builder.CreateStore(ResultCast, R.Var, R.IsVolatile);
11882 "OldValue and V must be of same type");
11883 if (IsPostfixUpdate) {
11884 Builder.CreateStore(OldValue, V.Var, V.IsVolatile);
11889 CurBBTI = CurBBTI ? CurBBTI :
Builder.CreateUnreachable();
11891 CurBBTI,
X.Var->getName() +
".atomic.exit");
11897 Builder.CreateCondBr(SuccessOrFail, ExitBB, ContBB);
11899 Builder.SetInsertPoint(ContBB);
11900 Builder.CreateStore(OldValue, V.Var);
11906 Builder.SetInsertPoint(ExitBB);
11908 Builder.SetInsertPoint(ExitTI);
11911 Value *CapturedValue =
11912 Builder.CreateSelect(SuccessOrFail, E, OldValue);
11913 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
11919 assert(R.Var->getType()->isPointerTy() &&
11920 "r.var must be of pointer type");
11921 assert(R.ElemTy->isIntegerTy() &&
"r must be of integral type");
11923 Value *ResultCast = R.IsSigned
11924 ?
Builder.CreateSExt(SuccessOrFail, R.ElemTy)
11925 :
Builder.CreateZExt(SuccessOrFail, R.ElemTy);
11926 Builder.CreateStore(ResultCast, R.Var, R.IsVolatile);
11930 assert((
Op == OMPAtomicCompareOp::MAX ||
Op == OMPAtomicCompareOp::MIN) &&
11931 "Op should be either max or min at this point");
11932 assert(!IsFailOnly &&
"IsFailOnly is only valid when the comparison is ==");
11943 if (IsXBinopExpr) {
11972 Value *CapturedValue =
nullptr;
11973 if (IsPostfixUpdate) {
11974 CapturedValue = OldValue;
11999 Value *NonAtomicCmp =
Builder.CreateCmp(Pred, OldValue, E);
12000 CapturedValue =
Builder.CreateSelect(NonAtomicCmp, E, OldValue);
12002 Builder.CreateStore(CapturedValue, V.Var, V.IsVolatile);
12006 checkAndEmitFlushAfterAtomic(
Loc, AO, AtomicKind::Compare);
12026 if (&OuterAllocaBB ==
Builder.GetInsertBlock()) {
12053 bool SubClausesPresent =
12054 (NumTeamsLower || NumTeamsUpper || ThreadLimit || IfExpr);
12056 if (!
Config.isTargetDevice() && SubClausesPresent) {
12057 assert((NumTeamsLower ==
nullptr || NumTeamsUpper !=
nullptr) &&
12058 "if lowerbound is non-null, then upperbound must also be non-null "
12059 "for bounds on num_teams");
12061 if (NumTeamsUpper ==
nullptr)
12062 NumTeamsUpper =
Builder.getInt32(0);
12064 if (NumTeamsLower ==
nullptr)
12065 NumTeamsLower = NumTeamsUpper;
12069 "argument to if clause must be an integer value");
12073 IfExpr =
Builder.CreateICmpNE(IfExpr,
12074 ConstantInt::get(IfExpr->
getType(), 0));
12075 NumTeamsUpper =
Builder.CreateSelect(
12076 IfExpr, NumTeamsUpper,
Builder.getInt32(1),
"numTeamsUpper");
12079 NumTeamsLower =
Builder.CreateSelect(
12080 IfExpr, NumTeamsLower,
Builder.getInt32(1),
"numTeamsLower");
12083 if (ThreadLimit ==
nullptr)
12084 ThreadLimit =
Builder.getInt32(0);
12088 Value *NumTeamsLowerInt32 =
12090 Value *NumTeamsUpperInt32 =
12092 Value *ThreadLimitInt32 =
12099 {Ident, ThreadNum, NumTeamsLowerInt32, NumTeamsUpperInt32,
12100 ThreadLimitInt32});
12105 if (
Error Err = BodyGenCB(AllocaIP, CodeGenIP, ExitBB))
12108 auto OI = std::make_unique<OutlineInfo>();
12109 OI->EntryBB = AllocaBB;
12110 OI->ExitBB = ExitBB;
12111 OI->OuterAllocBB = &OuterAllocaBB;
12117 Builder, OuterAllocaIP, ToBeDeleted, AllocaIP,
"gid",
true));
12119 Builder, OuterAllocaIP, ToBeDeleted, AllocaIP,
"tid",
true));
12121 auto HostPostOutlineCB = [
this, Ident,
12122 ToBeDeleted](
Function &OutlinedFn)
mutable {
12127 "there must be a single user for the outlined function");
12132 "Outlined function must have two or three arguments only");
12134 bool HasShared = OutlinedFn.
arg_size() == 3;
12142 assert(StaleCI &&
"Error while outlining - no CallInst user found for the "
12143 "outlined function.");
12144 Builder.SetInsertPoint(StaleCI);
12151 omp::RuntimeFunction::OMPRTL___kmpc_fork_teams),
12154 Builder.ClearInsertionPoint();
12156 I->eraseFromParent();
12159 if (!
Config.isTargetDevice())
12160 OI->PostOutlineCB = HostPostOutlineCB;
12164 Builder.SetInsertPoint(ExitBB);
12177 if (OuterAllocaBB ==
Builder.GetInsertBlock()) {
12192 if (
Error Err = BodyGenCB(AllocaIP, CodeGenIP, ExitBB))
12197 if (
Config.isTargetDevice()) {
12198 auto OI = std::make_unique<OutlineInfo>();
12199 OI->OuterAllocBB = OuterAllocIP.
getBlock();
12200 OI->EntryBB = AllocaBB;
12201 OI->ExitBB = ExitBB;
12202 OI->OuterDeallocBBs.reserve(OuterDeallocBlocks.
size());
12203 copy(OuterDeallocBlocks, OI->OuterDeallocBBs.
end());
12207 Builder.SetInsertPoint(ExitBB);
12214 std::string VarName) {
12223 return MapNamesArrayGlobal;
12228void OpenMPIRBuilder::initializeTypes(
Module &M) {
12232 unsigned ProgramAS = M.getDataLayout().getProgramAddressSpace();
12233#define OMP_TYPE(VarName, InitValue) VarName = InitValue;
12234#define OMP_ARRAY_TYPE(VarName, ElemTy, ArraySize) \
12235 VarName##Ty = ArrayType::get(ElemTy, ArraySize); \
12236 VarName##PtrTy = PointerType::get(Ctx, DefaultTargetAS);
12237#define OMP_FUNCTION_TYPE(VarName, IsVarArg, ReturnType, ...) \
12238 VarName = FunctionType::get(ReturnType, {__VA_ARGS__}, IsVarArg); \
12239 VarName##Ptr = PointerType::get(Ctx, ProgramAS);
12240#define OMP_STRUCT_TYPE(VarName, StructName, Packed, ...) \
12241 T = StructType::getTypeByName(Ctx, StructName); \
12243 T = StructType::create(Ctx, {__VA_ARGS__}, StructName, Packed); \
12245 VarName##Ptr = PointerType::get(Ctx, DefaultTargetAS);
12246#include "llvm/Frontend/OpenMP/OMPKinds.def"
12257 while (!Worklist.
empty()) {
12261 if (
BlockSet.insert(SuccBB).second)
12266std::unique_ptr<CodeExtractor>
12268 bool ArgsInZeroAddressSpace,
12270 return std::make_unique<CodeExtractor>(
12280 Suffix.
str(), ArgsInZeroAddressSpace);
12283std::unique_ptr<CodeExtractor> DeviceSharedMemOutlineInfo::createCodeExtractor(
12285 return std::make_unique<DeviceSharedMemCodeExtractor>(
12286 OMPBuilder, Blocks,
nullptr,
12294 OuterDeallocBBs.empty()
12297 Suffix.
str(), ArgsInZeroAddressSpace);
12301 uint64_t
Size, int32_t Flags,
12307 Name.empty() ? Addr->
getName() : Name,
Size, Flags, 0);
12319 Fn->
addFnAttr(
"uniform-work-group-size");
12320 Fn->
addFnAttr(Attribute::MustProgress);
12338 auto &&GetMDInt = [
this](
unsigned V) {
12345 NamedMDNode *MD =
M.getOrInsertNamedMetadata(
"omp_offload.info");
12346 auto &&TargetRegionMetadataEmitter =
12347 [&
C, MD, &OrderedEntries, &GetMDInt, &GetMDString](
12362 GetMDInt(E.getKind()), GetMDInt(EntryInfo.DeviceID),
12363 GetMDInt(EntryInfo.FileID), GetMDString(EntryInfo.ParentName),
12364 GetMDInt(EntryInfo.Line), GetMDInt(EntryInfo.Count),
12365 GetMDInt(E.getOrder())};
12368 OrderedEntries[E.getOrder()] = std::make_pair(&E, EntryInfo);
12377 auto &&DeviceGlobalVarMetadataEmitter =
12378 [&
C, &OrderedEntries, &GetMDInt, &GetMDString, MD](
12388 Metadata *
Ops[] = {GetMDInt(E.getKind()), GetMDString(MangledName),
12389 GetMDInt(E.getFlags()), GetMDInt(E.getOrder())};
12393 OrderedEntries[E.getOrder()] = std::make_pair(&E, varInfo);
12400 DeviceGlobalVarMetadataEmitter);
12402 for (
const auto &E : OrderedEntries) {
12403 assert(E.first &&
"All ordered entries must exist!");
12404 if (
const auto *CE =
12407 if (!CE->getID() || !CE->getAddress()) {
12411 if (!
M.getNamedValue(FnName))
12419 }
else if (
const auto *CE =
dyn_cast<
12428 if (
Config.isTargetDevice() &&
Config.hasRequiresUnifiedSharedMemory())
12430 if (!CE->getAddress()) {
12435 if (CE->getVarSize() == 0)
12439 assert(((
Config.isTargetDevice() && !CE->getAddress()) ||
12440 (!
Config.isTargetDevice() && CE->getAddress())) &&
12441 "Declaret target link address is set.");
12442 if (
Config.isTargetDevice())
12444 if (!CE->getAddress()) {
12451 if (!CE->getAddress()) {
12464 if ((
GV->hasLocalLinkage() ||
GV->hasHiddenVisibility()) &&
12468 OMPTargetGlobalVarEntryIndirectVTable))
12477 Flags, CE->getLinkage(), CE->getVarName());
12480 Flags, CE->getLinkage());
12491 if (
Config.hasRequiresFlags() && !
Config.isTargetDevice())
12497 Config.getRequiresFlags());
12507 OS <<
"_" <<
Count;
12512 unsigned NewCount = getTargetRegionEntryInfoCount(EntryInfo);
12515 EntryInfo.
Line, NewCount);
12523 auto FileIDInfo = CallBack();
12524 uint64_t FileID = 0;
12526 ID =
Status->getUniqueID();
12527 FileID =
Status->getUniqueID().getFile();
12531 FileID =
hash_value(std::get<0>(FileIDInfo));
12535 std::get<1>(FileIDInfo));
12540 for (uint64_t Remain =
12541 static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12543 !(Remain & 1); Remain = Remain >> 1)
12561 if (
static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12563 static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12570 if (
static_cast<std::underlying_type_t<omp::OpenMPOffloadMappingFlags>
>(
12576 Flags &=
~omp::OpenMPOffloadMappingFlags::OMP_MAP_MEMBER_OF;
12577 Flags |= MemberOfFlag;
12583 bool IsDeclaration,
bool IsExternallyVisible,
12585 std::vector<GlobalVariable *> &GeneratedRefs,
bool OpenMPSIMD,
12586 std::vector<Triple> TargetTriple,
Type *LlvmPtrTy,
12587 std::function<
Constant *()> GlobalInitializer,
12598 Config.hasRequiresUnifiedSharedMemory())) {
12603 if (!IsExternallyVisible)
12605 OS <<
"_decl_tgt_ref_ptr";
12608 Value *Ptr =
M.getNamedValue(PtrName);
12617 if (!
Config.isTargetDevice()) {
12618 if (GlobalInitializer)
12619 GV->setInitializer(GlobalInitializer());
12625 CaptureClause, DeviceClause, IsDeclaration, IsExternallyVisible,
12626 EntryInfo, MangledName, GeneratedRefs, OpenMPSIMD, TargetTriple,
12627 GlobalInitializer, VariableLinkage, LlvmPtrTy,
cast<Constant>(Ptr));
12639 bool IsDeclaration,
bool IsExternallyVisible,
12641 std::vector<GlobalVariable *> &GeneratedRefs,
bool OpenMPSIMD,
12642 std::vector<Triple> TargetTriple,
12643 std::function<
Constant *()> GlobalInitializer,
12647 (TargetTriple.empty() && !
Config.isTargetDevice()))
12658 !
Config.hasRequiresUnifiedSharedMemory()) {
12660 VarName = MangledName;
12663 if (!IsDeclaration)
12665 M.getDataLayout().getTypeSizeInBits(LlvmVal->
getValueType()), 8);
12668 Linkage = (VariableLinkage) ? VariableLinkage() : LlvmVal->
getLinkage();
12672 if (
Config.isTargetDevice() &&
12681 if (!
M.getNamedValue(RefName)) {
12685 GvAddrRef->setConstant(
true);
12687 GvAddrRef->setInitializer(Addr);
12688 GeneratedRefs.push_back(GvAddrRef);
12697 if (
Config.isTargetDevice()) {
12698 VarName = (Addr) ? Addr->
getName() :
"";
12702 CaptureClause, DeviceClause, IsDeclaration, IsExternallyVisible,
12703 EntryInfo, MangledName, GeneratedRefs, OpenMPSIMD, TargetTriple,
12704 LlvmPtrTy, GlobalInitializer, VariableLinkage);
12705 VarName = (Addr) ? Addr->
getName() :
"";
12707 VarSize =
M.getDataLayout().getPointerSize();
12726 auto &&GetMDInt = [MN](
unsigned Idx) {
12731 auto &&GetMDString = [MN](
unsigned Idx) {
12733 return V->getString();
12736 switch (GetMDInt(0)) {
12740 case OffloadEntriesInfoManager::OffloadEntryInfo::
12741 OffloadingEntryInfoTargetRegion: {
12751 case OffloadEntriesInfoManager::OffloadEntryInfo::
12752 OffloadingEntryInfoDeviceGlobalVar:
12765 if (HostFilePath.
empty())
12769 if (std::error_code Err = Buf.getError()) {
12771 "OpenMPIRBuilder: " +
12779 if (std::error_code Err =
M.getError()) {
12781 (
"error parsing host file inside of OpenMPIRBuilder: " + Err.message())
12795 "expected a valid insertion block for creating an iterator loop");
12805 Builder.getCurrentDebugLocation(),
"omp.it.cont");
12817 T->eraseFromParent();
12826 if (!BodyBr || BodyBr->getSuccessor() != CLI->
getLatch()) {
12828 "iterator bodygen must terminate the canonical body with an "
12829 "unconditional branch to the loop latch",
12853 for (
const auto &
ParamAttr : ParamAttrs) {
12896 return std::string(Out.
str());
12904 unsigned VecRegSize;
12906 ISADataTy ISAData[] = {
12925 for (
char Mask :
Masked) {
12926 for (
const ISADataTy &
Data : ISAData) {
12929 Out <<
"_ZGV" <<
Data.ISA << Mask;
12931 assert(NumElts &&
"Non-zero simdlen/cdtsize expected");
12945template <
typename T>
12948 StringRef MangledName,
bool OutputBecomesInput,
12952 Out << Prefix << ISA << LMask << VLEN;
12953 if (OutputBecomesInput)
12955 Out << ParSeq <<
'_' << MangledName;
12964 bool OutputBecomesInput,
12969 OutputBecomesInput, Fn);
12971 OutputBecomesInput, Fn);
12975 OutputBecomesInput, Fn);
12977 OutputBecomesInput, Fn);
12981 OutputBecomesInput, Fn);
12983 OutputBecomesInput, Fn);
12988 OutputBecomesInput, Fn);
12999 char ISA,
unsigned NarrowestDataSize,
bool OutputBecomesInput) {
13000 assert((ISA ==
'n' || ISA ==
's') &&
"Expected ISA either 's' or 'n'.");
13012 OutputBecomesInput, Fn);
13019 OutputBecomesInput, Fn);
13021 OutputBecomesInput, Fn);
13025 OutputBecomesInput, Fn);
13029 OutputBecomesInput, Fn);
13038 OutputBecomesInput, Fn);
13045 MangledName, OutputBecomesInput, Fn);
13047 MangledName, OutputBecomesInput, Fn);
13051 MangledName, OutputBecomesInput, Fn);
13055 MangledName, OutputBecomesInput, Fn);
13065 return OffloadEntriesTargetRegion.empty() &&
13066 OffloadEntriesDeviceGlobalVar.empty();
13069unsigned OffloadEntriesInfoManager::getTargetRegionEntryInfoCount(
13071 auto It = OffloadEntriesTargetRegionCount.find(
13072 getTargetRegionEntryCountKey(EntryInfo));
13073 if (It == OffloadEntriesTargetRegionCount.end())
13078void OffloadEntriesInfoManager::incrementTargetRegionEntryInfoCount(
13080 OffloadEntriesTargetRegionCount[getTargetRegionEntryCountKey(EntryInfo)] =
13081 EntryInfo.
Count + 1;
13087 OffloadEntriesTargetRegion[EntryInfo] =
13090 ++OffloadingEntriesNum;
13096 assert(EntryInfo.
Count == 0 &&
"expected default EntryInfo");
13099 EntryInfo.
Count = getTargetRegionEntryInfoCount(EntryInfo);
13103 if (OMPBuilder->Config.isTargetDevice()) {
13108 auto &Entry = OffloadEntriesTargetRegion[EntryInfo];
13109 Entry.setAddress(Addr);
13111 Entry.setFlags(Flags);
13117 "Target region entry already registered!");
13119 OffloadEntriesTargetRegion[EntryInfo] = Entry;
13120 ++OffloadingEntriesNum;
13122 incrementTargetRegionEntryInfoCount(EntryInfo);
13129 EntryInfo.
Count = getTargetRegionEntryInfoCount(EntryInfo);
13131 auto It = OffloadEntriesTargetRegion.find(EntryInfo);
13132 if (It == OffloadEntriesTargetRegion.end()) {
13136 if (!IgnoreAddressId && (It->second.getAddress() || It->second.getID()))
13144 for (
const auto &It : OffloadEntriesTargetRegion) {
13145 Action(It.first, It.second);
13151 OffloadEntriesDeviceGlobalVar.try_emplace(Name, Order, Flags);
13152 ++OffloadingEntriesNum;
13158 if (OMPBuilder->Config.isTargetDevice()) {
13162 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
13164 if (Entry.getVarSize() == 0) {
13165 Entry.setVarSize(VarSize);
13166 Entry.setLinkage(Linkage);
13170 Entry.setVarSize(VarSize);
13171 Entry.setLinkage(Linkage);
13172 Entry.setAddress(Addr);
13175 auto &Entry = OffloadEntriesDeviceGlobalVar[VarName];
13176 assert(Entry.isValid() && Entry.getFlags() == Flags &&
13177 "Entry not initialized!");
13178 if (Entry.getVarSize() == 0) {
13179 Entry.setVarSize(VarSize);
13180 Entry.setLinkage(Linkage);
13187 OffloadEntriesDeviceGlobalVar.try_emplace(VarName, OffloadingEntriesNum,
13188 Addr, VarSize, Flags, Linkage,
13191 OffloadEntriesDeviceGlobalVar.try_emplace(
13192 VarName, OffloadingEntriesNum, Addr, VarSize, Flags, Linkage,
"");
13193 ++OffloadingEntriesNum;
13200 for (
const auto &E : OffloadEntriesDeviceGlobalVar)
13201 Action(E.getKey(), E.getValue());
13208void CanonicalLoopInfo::collectControlBlocks(
13215 BBs.
append({getPreheader(), Header,
Cond, Latch, Exit, getAfter()});
13227void CanonicalLoopInfo::setTripCount(
Value *TripCount) {
13239void CanonicalLoopInfo::mapIndVar(
13249 for (
Use &U : OldIV->
uses()) {
13253 if (
User->getParent() == getCond())
13255 if (
User->getParent() == getLatch())
13261 Value *NewIV = Updater(OldIV);
13264 for (Use *U : ReplacableUses)
13285 "Preheader must terminate with unconditional branch");
13287 "Preheader must jump to header");
13291 "Header must terminate with unconditional branch");
13292 assert(Header->getSingleSuccessor() == Cond &&
13293 "Header must jump to exiting block");
13296 assert(Cond->getSinglePredecessor() == Header &&
13297 "Exiting block only reachable from header");
13300 "Exiting block must terminate with conditional branch");
13302 "Exiting block's first successor jump to the body");
13304 "Exiting block's second successor must exit the loop");
13308 "Body only reachable from exiting block");
13313 "Latch must terminate with unconditional branch");
13314 assert(Latch->getSingleSuccessor() == Header &&
"Latch must jump to header");
13317 assert(Latch->getSinglePredecessor() !=
nullptr);
13322 "Exit block must terminate with unconditional branch");
13323 assert(Exit->getSingleSuccessor() == After &&
13324 "Exit block must jump to after block");
13328 "After block only reachable from exit block");
13332 assert(IndVar &&
"Canonical induction variable not found?");
13334 "Induction variable must be an integer");
13336 "Induction variable must be a PHI in the loop header");
13342 auto *NextIndVar =
cast<PHINode>(IndVar)->getIncomingValue(1);
13350 assert(TripCount &&
"Loop trip count not found?");
13352 "Trip count and induction variable must have the same type");
13356 "Exit condition must be a signed less-than comparison");
13358 "Exit condition must compare the induction variable");
13360 "Exit condition must compare with the trip count");
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Expand Atomic instructions
This file contains the simple types necessary to represent the attributes associated with functions a...
static const Function * getParent(const Value *V)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file provides various utilities for inspecting and working with the control flow graph in LLVM I...
This header defines various interfaces for pass management in LLVM.
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
static cl::opt< unsigned > TileSize("fuse-matrix-tile-size", cl::init(4), cl::Hidden, cl::desc("Tile size for matrix instruction fusion using square-shaped tiles."))
uint64_t IntrinsicInst * II
#define OMP_KERNEL_ARG_VERSION
Provides definitions for Target specific Grid Values.
static Value * removeASCastIfPresent(Value *V)
static void createTargetLoopWorkshareCall(OpenMPIRBuilder *OMPBuilder, WorksharingLoopType LoopType, BasicBlock *InsertBlock, Value *Ident, Value *LoopBodyArg, Value *TripCount, Function &LoopBodyFn, bool NoLoop)
Value * createFakeIntVal(IRBuilderBase &Builder, OpenMPIRBuilder::InsertPointTy OuterAllocaIP, llvm::SmallVectorImpl< Instruction * > &ToBeDeleted, OpenMPIRBuilder::InsertPointTy InnerAllocaIP, const Twine &Name="", bool AsPtr=true, bool Is64Bit=false)
static Function * createTargetParallelWrapper(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn)
Create wrapper function used to gather the outlined function's argument structure from a shared buffe...
static void redirectTo(BasicBlock *Source, BasicBlock *Target, DebugLoc DL)
Make Source branch to Target.
static FunctionCallee getKmpcDistForStaticInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
static void applyParallelAccessesMetadata(CanonicalLoopInfo *CLI, LLVMContext &Ctx, Loop *Loop, LoopInfo &LoopInfo, SmallVector< Metadata * > &LoopMDList)
static void addAArch64VectorName(T VLEN, StringRef LMask, StringRef Prefix, char ISA, StringRef ParSeq, StringRef MangledName, bool OutputBecomesInput, llvm::Function *Fn)
static FunctionCallee getKmpcForDynamicFiniForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for finalizing the dynamic loop using depending on type.
static Expected< Function * > createOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, StringRef FuncName, SmallVectorImpl< Value * > &Inputs, OpenMPIRBuilder::TargetBodyGenCallbackTy &CBFunc, OpenMPIRBuilder::TargetGenArgAccessorsCallbackTy &ArgAccessorFuncCB)
static void FixupDebugInfoForOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, Function *Func, DenseMap< Value *, std::tuple< Value *, unsigned > > &ValueReplacementMap)
static OMPScheduleType getOpenMPOrderingScheduleType(OMPScheduleType BaseScheduleType, bool HasOrderedClause)
Adds ordering modifier flags to schedule type.
static OMPScheduleType getOpenMPMonotonicityScheduleType(OMPScheduleType ScheduleType, bool HasSimdModifier, bool HasMonotonic, bool HasNonmonotonic, bool HasOrderedClause)
Adds monotonicity modifier flags to schedule type.
static std::string mangleVectorParameters(ArrayRef< llvm::OpenMPIRBuilder::DeclareSimdAttrTy > ParamAttrs)
Mangle the parameter part of the vector function name according to their OpenMP classification.
static bool isGenericKernel(Function &Fn)
static void workshareLoopTargetCallback(OpenMPIRBuilder *OMPIRBuilder, CanonicalLoopInfo *CLI, Value *Ident, Function &OutlinedFn, const SmallVector< Instruction *, 4 > &ToBeDeleted, WorksharingLoopType LoopType, bool NoLoop)
static bool isValidWorkshareLoopScheduleType(OMPScheduleType SchedType)
static bool isAtomicableReductionSet(ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos)
static llvm::CallInst * emitNoUnwindRuntimeCall(IRBuilder<> &Builder, llvm::FunctionCallee Callee, ArrayRef< llvm::Value * > Args, const llvm::Twine &Name)
static Error populateReductionFunction(Function *ReductionFunc, ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos, IRBuilder<> &Builder, ArrayRef< bool > IsByRef, bool IsGPU)
static Function * getFreshReductionFunc(Module &M)
static void raiseUserConstantDataAllocasToEntryBlock(IRBuilderBase &Builder, Function *Function)
static FunctionCallee getKmpcForDynamicNextForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for updating the next loop using OpenMP dynamic scheduling depending...
static bool isConflictIP(IRBuilder<>::InsertPoint IP1, IRBuilder<>::InsertPoint IP2)
Return whether IP1 and IP2 are ambiguous, i.e.
static void checkReductionInfos(ArrayRef< OpenMPIRBuilder::ReductionInfo > ReductionInfos, bool IsGPU)
static Type * getOffloadingArrayType(Value *V)
static OMPScheduleType getOpenMPBaseScheduleType(llvm::omp::ScheduleKind ClauseKind, bool HasChunks, bool HasSimdModifier, bool HasDistScheduleChunks)
Determine which scheduling algorithm to use, determined from schedule clause arguments.
static OMPScheduleType computeOpenMPScheduleType(ScheduleKind ClauseKind, bool HasChunks, bool HasSimdModifier, bool HasMonotonicModifier, bool HasNonmonotonicModifier, bool HasOrderedClause, bool HasDistScheduleChunks)
Determine the schedule type using schedule and ordering clause arguments.
static FunctionCallee getKmpcForDynamicInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
Returns an LLVM function to call for initializing loop bounds using OpenMP dynamic scheduling dependi...
static std::optional< omp::OMPTgtExecModeFlags > getTargetKernelExecMode(Function &Kernel)
Given a function, if it represents the entry point of a target kernel, this returns the execution mod...
static StructType * createTaskWithPrivatesTy(OpenMPIRBuilder &OMPIRBuilder, ArrayRef< Value * > OffloadingArraysToPrivatize)
static cl::opt< double > UnrollThresholdFactor("openmp-ir-builder-unroll-threshold-factor", cl::Hidden, cl::desc("Factor for the unroll threshold to account for code " "simplifications still taking place"), cl::init(1.5))
static cl::opt< bool > UseDefaultMaxThreads("openmp-ir-builder-use-default-max-threads", cl::Hidden, cl::desc("Use a default max threads if none is provided."), cl::init(true))
static int32_t computeHeuristicUnrollFactor(CanonicalLoopInfo *CLI)
Heuristically determine the best-performant unroll factor for CLI.
static void emitTargetCall(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, OpenMPIRBuilder::InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, OpenMPIRBuilder::TargetDataInfo &Info, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, const OpenMPIRBuilder::TargetKernelRuntimeAttrs &RuntimeAttrs, Value *IfCond, Function *OutlinedFn, Constant *OutlinedFnID, SmallVectorImpl< Value * > &Args, OpenMPIRBuilder::GenMapInfoCallbackTy GenMapInfoCB, OpenMPIRBuilder::CustomMapperCallbackTy CustomMapperCB, const OpenMPIRBuilder::DependenciesInfo &Dependencies, bool HasNoWait, Value *DynCGroupMem, OMPDynGroupprivateFallbackType DynCGroupMemFallback)
static Value * emitTaskDependencies(OpenMPIRBuilder &OMPBuilder, const SmallVectorImpl< OpenMPIRBuilder::DependData > &Dependencies)
static Error emitTargetOutlinedFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, bool IsOffloadEntry, TargetRegionEntryInfo &EntryInfo, const OpenMPIRBuilder::TargetKernelDefaultAttrs &DefaultAttrs, Function *&OutlinedFn, Constant *&OutlinedFnID, SmallVectorImpl< Value * > &Inputs, OpenMPIRBuilder::TargetBodyGenCallbackTy &CBFunc, OpenMPIRBuilder::TargetGenArgAccessorsCallbackTy &ArgAccessorFuncCB)
static void updateNVPTXAttr(Function &Kernel, StringRef Name, int32_t Value, bool Min)
static OpenMPIRBuilder::InsertPointTy getInsertPointAfterInstr(Instruction *I)
static void redirectAllPredecessorsTo(BasicBlock *OldTarget, BasicBlock *NewTarget, DebugLoc DL)
Redirect all edges that branch to OldTarget to NewTarget.
static void hoistNonEntryAllocasToEntryBlock(llvm::BasicBlock &Block)
static std::unique_ptr< TargetMachine > createTargetMachine(Function *F, CodeGenOptLevel OptLevel)
Create the TargetMachine object to query the backend for optimization preferences.
static FunctionCallee getKmpcForStaticInitForType(Type *Ty, Module &M, OpenMPIRBuilder &OMPBuilder)
static void addAccessGroupMetadata(BasicBlock *Block, MDNode *AccessGroup, LoopInfo &LI)
Attach llvm.access.group metadata to the memref instructions of Block.
static void addBasicBlockMetadata(BasicBlock *BB, ArrayRef< Metadata * > Properties)
Attach metadata Properties to the basic block described by BB.
static void restoreIPandDebugLoc(llvm::IRBuilderBase &Builder, llvm::IRBuilderBase::InsertPoint IP)
This is a wrapper over IRBuilderBase::restoreIP that also restores a current debug location when the ...
static LoadInst * loadSharedDataFromTaskDescriptor(OpenMPIRBuilder &OMPIRBuilder, IRBuilderBase &Builder, Value *TaskWithPrivates, Type *TaskWithPrivatesTy)
Given a task descriptor, TaskWithPrivates, return the pointer to the block of pointers containing sha...
static cl::opt< bool > OptimisticAttributes("openmp-ir-builder-optimistic-attributes", cl::Hidden, cl::desc("Use optimistic attributes describing " "'as-if' properties of runtime calls."), cl::init(false))
static bool hasGridValue(const Triple &T)
static FunctionCallee getKmpcForStaticLoopForType(Type *Ty, OpenMPIRBuilder *OMPBuilder, WorksharingLoopType LoopType)
static const omp::GV & getGridValue(const Triple &T, Function *Kernel)
static void addAArch64AdvSIMDNDSNames(unsigned NDS, StringRef Mask, StringRef Prefix, char ISA, StringRef ParSeq, StringRef MangledName, bool OutputBecomesInput, llvm::Function *Fn)
static Function * emitTargetTaskProxyFunction(OpenMPIRBuilder &OMPBuilder, IRBuilderBase &Builder, CallInst *StaleCI, StructType *PrivatesTy, StructType *TaskWithPrivatesTy, const size_t NumOffloadingArrays, const int SharedArgsOperandNo)
Create an entry point for a target task with the following.
static void addLoopMetadata(CanonicalLoopInfo *Loop, ArrayRef< Metadata * > Properties)
Attach loop metadata Properties to the loop described by Loop.
static AtomicOrdering TransformReleaseAcquireRelease(AtomicOrdering AO)
static void removeUnusedBlocksFromParent(ArrayRef< BasicBlock * > BBs)
static void targetParallelCallback(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn, Function *OuterFn, BasicBlock *OuterAllocaBB, Value *Ident, Value *IfCondition, Value *NumThreads, Instruction *PrivTID, AllocaInst *PrivTIDAddr, Value *ThreadID, const SmallVector< Instruction *, 4 > &ToBeDeleted)
static void hostParallelCallback(OpenMPIRBuilder *OMPIRBuilder, Function &OutlinedFn, Function *OuterFn, Value *Ident, Value *IfCondition, Instruction *PrivTID, AllocaInst *PrivTIDAddr, const SmallVector< Instruction *, 4 > &ToBeDeleted)
FunctionAnalysisManager FAM
This file defines the Pass Instrumentation classes that provide instrumentation points into the pass ...
const SmallVectorImpl< MachineOperand > & Cond
Remove Loads Into Fake Uses
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
SmallPtrSet< BasicBlock *, 0 > BlockSet
This file implements the SmallBitVector class.
This file defines the SmallSet class.
static SymbolRef::Type getType(const Symbol *Sym)
Defines the virtual file system interface vfs::FileSystem.
static cl::opt< unsigned > MaxThreads("xcore-max-threads", cl::Optional, cl::desc("Maximum number of threads (for emulation thread-local storage)"), cl::Hidden, cl::value_desc("number"), cl::init(8))
static const uint32_t IV[8]
Class for arbitrary precision integers.
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
An arbitrary precision integer that knows its signedness.
static APSInt getUnsigned(uint64_t X)
This class represents a conversion between pointers from one address space to another.
an instruction to allocate memory on the stack
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
PointerType * getType() const
Overload to return most specific pointer type.
Type * getAllocatedType() const
Return the type that is being allocated by the instruction.
unsigned getAddressSpace() const
Return the address space for the allocation.
LLVM_ABI std::optional< TypeSize > getAllocationSize(const DataLayout &DL) const
Get allocation size in bytes.
LLVM_ABI bool isArrayAllocation() const
Return true if there is an allocation size parameter to the allocation instruction that is not 1.
void setAlignment(Align Align)
const Value * getArraySize() const
Get the number of elements allocated.
bool registerPass(PassBuilderT &&PassBuilder)
Register an analysis pass with the manager.
This class represents an incoming formal argument to a Function.
unsigned getArgNo() const
Return the index of this formal argument in its containing function.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
Class to represent array types.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
A function analysis which provides an AssumptionCache.
LLVM_ABI AssumptionCache run(Function &F, FunctionAnalysisManager &)
A cache of @llvm.assume calls within a function.
An instruction that atomically checks whether a specified value is in a memory location,...
void setWeak(bool IsWeak)
static AtomicOrdering getStrongestFailureOrdering(AtomicOrdering SuccessOrdering)
Returns the strongest permitted ordering on failure, given the desired ordering on success.
LLVM_ABI std::pair< LoadInst *, AllocaInst * > EmitAtomicLoadLibcall(AtomicOrdering AO)
LLVM_ABI void EmitAtomicStoreLibcall(AtomicOrdering AO, Value *Source)
an instruction that atomically reads a memory location, combines it with another value,...
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
This class holds the attributes for a particular argument, parameter, function, or return value.
LLVM_ABI AttributeSet addAttributes(LLVMContext &C, AttributeSet AS) const
Add attributes to the attribute set.
LLVM_ABI AttributeSet addAttribute(LLVMContext &C, Attribute::AttrKind Kind) const
Add an argument attribute.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
LLVM_ABI void replaceSuccessorsPhiUsesWith(BasicBlock *Old, BasicBlock *New)
Update all phi nodes in this basic block's successors to refer to basic block New instead of basic bl...
iterator begin()
Instruction iterator methods.
LLVM_ABI const_iterator getFirstInsertionPt() const
Returns an iterator to the first instruction in this block that is suitable for inserting a non-PHI i...
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
reverse_iterator rbegin()
bool hasTerminator() const LLVM_READONLY
Returns whether the block has a terminator.
const Instruction & back() const
LLVM_ABI BasicBlock * splitBasicBlockBefore(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction and insert the new basic blo...
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
LLVM_ABI void insertDbgRecordBefore(DbgRecord *DR, InstListType::iterator Here)
Insert a DbgRecord into a block at the position given by Here.
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
LLVM_ABI InstListType::const_iterator getFirstNonPHIOrDbg(bool SkipPseudoOp=true) const
Returns a pointer to the first instruction in this block that is not a PHINode or a debug intrinsic,...
LLVM_ABI const BasicBlock * getUniqueSuccessor() const
Return the successor of this block if it has a unique successor.
LLVM_ABI const BasicBlock * getSinglePredecessor() const
Return the predecessor of this block if it has a single predecessor block.
const Instruction & front() const
InstListType::reverse_iterator reverse_iterator
LLVM_ABI const BasicBlock * getUniquePredecessor() const
Return the predecessor of this block if it has a unique predecessor block.
const Instruction * getTerminatorOrNull() const LLVM_READONLY
Returns the terminator instruction if the block is well formed or null if the block is not well forme...
LLVM_ABI const BasicBlock * getSingleSuccessor() const
Return the successor of this block if it has a single successor.
LLVM_ABI SymbolTableList< BasicBlock >::iterator eraseFromParent()
Unlink 'this' from the containing function and delete it.
InstListType::iterator iterator
Instruction iterators...
LLVM_ABI LLVMContext & getContext() const
Get the context in which this basic block lives.
void moveBefore(BasicBlock *MovePos)
Unlink this basic block from its current function and insert it into the function that MovePos lives ...
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
void splice(BasicBlock::iterator ToIt, BasicBlock *FromBB)
Transfer all instructions from FromBB to this basic block at ToIt.
LLVM_ABI void removePredecessor(BasicBlock *Pred, bool KeepOneInputPHIs=false)
Update PHI nodes in this BasicBlock before removal of predecessor Pred.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
Class to represented the control flow structure of an OpenMP canonical loop.
Value * getTripCount() const
Returns the llvm::Value containing the number of loop iterations.
BasicBlock * getHeader() const
The header is the entry for each iteration.
LLVM_ABI void assertOK() const
Consistency self-check.
Type * getIndVarType() const
Return the type of the induction variable (and the trip count).
BasicBlock * getBody() const
The body block is the single entry for a loop iteration and not controlled by CanonicalLoopInfo.
bool isValid() const
Returns whether this object currently represents the IR of a loop.
void setLastIter(Value *IterVar)
Sets the last iteration variable for this loop.
OpenMPIRBuilder::InsertPointTy getAfterIP() const
Return the insertion point for user code after the loop.
OpenMPIRBuilder::InsertPointTy getBodyIP() const
Return the insertion point for user code in the body.
BasicBlock * getAfter() const
The after block is intended for clean-up code such as lifetime end markers.
Function * getFunction() const
LLVM_ABI void invalidate()
Invalidate this loop.
BasicBlock * getLatch() const
Reaching the latch indicates the end of the loop body code.
OpenMPIRBuilder::InsertPointTy getPreheaderIP() const
Return the insertion point for user code before the loop.
BasicBlock * getCond() const
The condition block computes whether there is another loop iteration.
BasicBlock * getExit() const
Reaching the exit indicates no more iterations are being executed.
LLVM_ABI BasicBlock * getPreheader() const
The preheader ensures that there is only a single edge entering the loop.
Instruction * getIndVar() const
Returns the instruction representing the current logical induction variable.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ ICMP_SLT
signed less than
@ ICMP_SLE
signed less or equal
@ FCMP_OLT
0 1 0 0 True if ordered and less than
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
static LLVM_ABI Constant * getString(LLVMContext &Context, StringRef Initializer, bool AddNull=true, bool ByteString=false)
This method constructs a CDS and initializes it with a text string.
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getTruncOrBitCast(Constant *C, Type *Ty)
static LLVM_ABI Constant * getPointerBitCastOrAddrSpaceCast(Constant *C, Type *Ty)
Create a BitCast or AddrSpaceCast for a pointer type depending on the address space.
static LLVM_ABI Constant * getSizeOf(Type *Ty)
getSizeOf constant expr - computes the (alloc) size of a type (in address-units, not bits) in a targe...
static LLVM_ABI Constant * getAddrSpaceCast(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
DILocalScope * getScope() const
Get the local scope for this variable.
DINodeArray getAnnotations() const
Subprogram description. Uses SubclassData1.
uint32_t getAlignInBits() const
StringRef getName() const
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeStoreSize(Type *Ty) const
Returns the maximum number of bytes that may be overwritten by storing the specified type.
Record of a variable value-assignment, aka a non instruction representation of the dbg....
Analysis pass which computes a DominatorTree.
LLVM_ABI DominatorTree run(Function &F, FunctionAnalysisManager &)
Run the analysis pass over a function and produce a dominator tree.
bool properlyDominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
properlyDominates - Returns true iff A dominates B and A != B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Represents either an error or a value T.
Lightweight error class with error context and mandatory checking.
static ErrorSuccess success()
Create a success value.
Tagged union holding either a T or a Error.
Error takeError()
Take ownership of the stored error.
reference get()
Returns a reference to the stored T value.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
void addFnAttr(Attribute::AttrKind Kind)
Add function attributes to this function.
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
const BasicBlock & getEntryBlock() const
FunctionType * getFunctionType() const
Returns the FunctionType for me.
void removeFromParent()
removeFromParent - This method unlinks 'this' from the containing module, but does not delete it.
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
DISubprogram * getSubprogram() const
Get the attached subprogram.
AttributeList getAttributes() const
Return the attribute list for this Function.
const Function & getFunction() const
void setAttributes(AttributeList Attrs)
Set the attribute list for this Function.
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
adds the attribute to the list of attributes for the given arg.
Function::iterator insert(Function::iterator Position, BasicBlock *BB)
Insert BB in the basic block list at Position.
Type * getReturnType() const
Returns the type of the ret val.
void setCallingConv(CallingConv::ID CC)
Argument * getArg(unsigned i) const
bool hasMetadata() const
Return true if this GlobalObject has any metadata attached to it.
LLVM_ABI void addMetadata(unsigned KindID, MDNode &MD)
Add a metadata attachment.
LinkageTypes getLinkage() const
void setLinkage(LinkageTypes LT)
Module * getParent()
Get the module that this global value is contained inside of...
void setDSOLocal(bool Local)
PointerType * getType() const
Global values are always pointers.
@ HiddenVisibility
The GV is hidden.
@ ProtectedVisibility
The GV is protected.
void setVisibility(VisibilityTypes V)
LinkageTypes
An enumeration for the kinds of linkage for global values.
@ PrivateLinkage
Like Internal, but omit from symbol table.
@ CommonLinkage
Tentative definitions.
@ InternalLinkage
Rename collisions when linking (static functions).
@ WeakODRLinkage
Same, but only replaced by something equivalent.
@ WeakAnyLinkage
Keep one copy of named function when linking (weak)
@ AppendingLinkage
Special purpose, only applies to global arrays.
@ LinkOnceODRLinkage
Same, but only replaced by something equivalent.
Type * getValueType() const
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
InsertPoint - A saved insertion point.
BasicBlock * getBlock() const
bool isSet() const
Returns true if this insert point is set.
BasicBlock::iterator getPoint() const
Common base class shared among various IRBuilders.
InsertPoint saveIP() const
Returns the current insert point.
void restoreIP(InsertPoint IP)
Sets the current insert point to a previously-saved location.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
LLVM_ABI const DebugLoc & getStableDebugLoc() const
Fetch the debug location for this node, unless this is a debug intrinsic, in which case fetch the deb...
LLVM_ABI void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
LLVM_ABI unsigned getNumSuccessors() const LLVM_READONLY
Return the number of successors that this instruction has.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
LLVM_ABI BasicBlock * getSuccessor(unsigned Idx) const LLVM_READONLY
Return the specified successor. This instruction must be a terminator.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI void moveBeforePreserving(InstListType::iterator MovePos)
Perform a moveBefore operation, while signalling that the caller intends to preserve the original ord...
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void insertAfter(Instruction *InsertPos)
Insert an unlinked instruction into a basic block immediately after the specified instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
Analysis pass that exposes the LoopInfo for a function.
LLVM_ABI LoopInfo run(Function &F, FunctionAnalysisManager &AM)
ArrayRef< BlockT * > getBlocks() const
Get a list of the basic blocks which make up this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
This class represents a loop nest and can be used to query its properties.
Represents a single loop in the control flow graph.
LLVM_ABI MDNode * createCallbackEncoding(unsigned CalleeArgNo, ArrayRef< int > Arguments, bool VarArgsArePassed)
Return metadata describing a callback (see llvm::AbstractCallSite).
LLVM_ABI void replaceOperandWith(unsigned I, Metadata *New)
Replace a specific operand.
static MDTuple * getDistinct(LLVMContext &Context, ArrayRef< Metadata * > MDs)
ArrayRef< MDOperand > operands() const
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
This class implements a map that also provides access to all stored values in a deterministic order.
A Module instance is used to store all the information related to an LLVM module.
LLVMContext & getContext() const
Get the global data context.
const DataLayout & getDataLayout() const
Get the data layout for the module's target platform.
iterator_range< op_iterator > operands()
LLVM_ABI void addOperand(MDNode *M)
Device global variable entries info.
Target region entries info.
Base class of the entries info.
Class that manages information about offload code regions and data.
function_ref< void(StringRef, const OffloadEntryInfoDeviceGlobalVar &)> OffloadDeviceGlobalVarEntryInfoActTy
Applies action Action on all registered entries.
OMPTargetDeviceClauseKind
Kind of device clause for declare target variables and functions NOTE: Currently not used as a part o...
@ OMPTargetDeviceClauseAny
The target is marked for all devices.
LLVM_ABI void registerDeviceGlobalVarEntryInfo(StringRef VarName, Constant *Addr, int64_t VarSize, OMPTargetGlobalVarEntryKind Flags, GlobalValue::LinkageTypes Linkage)
Register device global variable entry.
LLVM_ABI void initializeDeviceGlobalVarEntryInfo(StringRef Name, OMPTargetGlobalVarEntryKind Flags, unsigned Order)
Initialize device global variable entry.
LLVM_ABI void actOnDeviceGlobalVarEntriesInfo(const OffloadDeviceGlobalVarEntryInfoActTy &Action)
OMPTargetRegionEntryKind
Kind of the target registry entry.
@ OMPTargetRegionEntryTargetRegion
Mark the entry as target region.
LLVM_ABI void getTargetRegionEntryFnName(SmallVectorImpl< char > &Name, const TargetRegionEntryInfo &EntryInfo)
LLVM_ABI bool hasTargetRegionEntryInfo(TargetRegionEntryInfo EntryInfo, bool IgnoreAddressId=false) const
Return true if a target region entry with the provided information exists.
LLVM_ABI void registerTargetRegionEntryInfo(TargetRegionEntryInfo EntryInfo, Constant *Addr, Constant *ID, OMPTargetRegionEntryKind Flags)
Register target region entry.
LLVM_ABI void actOnTargetRegionEntriesInfo(const OffloadTargetRegionEntryInfoActTy &Action)
LLVM_ABI void initializeTargetRegionEntryInfo(const TargetRegionEntryInfo &EntryInfo, unsigned Order)
Initialize target region entry.
OMPTargetGlobalVarEntryKind
Kind of the global variable entry..
@ OMPTargetGlobalVarEntryEnter
Mark the entry as a declare target enter.
@ OMPTargetGlobalRegisterRequires
Mark the entry as a register requires global.
@ OMPTargetGlobalVarEntryIndirect
Mark the entry as a declare target indirect global.
@ OMPTargetGlobalVarEntryLink
Mark the entry as a to declare target link.
@ OMPTargetGlobalVarEntryTo
Mark the entry as a to declare target.
@ OMPTargetGlobalVarEntryIndirectVTable
Mark the entry as a declare target indirect vtable.
function_ref< void(const TargetRegionEntryInfo &EntryInfo, const OffloadEntryInfoTargetRegion &)> OffloadTargetRegionEntryInfoActTy
brief Applies action Action on all registered entries.
bool hasDeviceGlobalVarEntryInfo(StringRef VarName) const
Checks if the variable with the given name has been registered already.
LLVM_ABI bool empty() const
Return true if a there are no entries defined.
std::optional< bool > IsTargetDevice
Flag to define whether to generate code for the role of the OpenMP host (if set to false) or device (...
std::optional< bool > IsGPU
Flag for specifying if the compilation is done for an accelerator.
LLVM_ABI int64_t getRequiresFlags() const
Returns requires directive clauses as flags compatible with those expected by libomptarget.
std::optional< bool > OpenMPOffloadMandatory
Flag for specifying if offloading is mandatory.
LLVM_ABI void setHasRequiresReverseOffload(bool Value)
LLVM_ABI OpenMPIRBuilderConfig()
LLVM_ABI bool hasRequiresUnifiedSharedMemory() const
LLVM_ABI void setHasRequiresUnifiedSharedMemory(bool Value)
unsigned getDefaultTargetAS() const
LLVM_ABI bool hasRequiresDynamicAllocators() const
LLVM_ABI void setHasRequiresUnifiedAddress(bool Value)
bool isTargetDevice() const
LLVM_ABI void setHasRequiresDynamicAllocators(bool Value)
LLVM_ABI bool hasRequiresReverseOffload() const
bool hasRequiresFlags() const
LLVM_ABI bool hasRequiresUnifiedAddress() const
Struct that keeps the information that should be kept throughout a 'target data' region.
An interface to create LLVM-IR for OpenMP directives.
LLVM_ABI InsertPointOrErrorTy createOrderedThreadsSimd(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsThreads)
Generator for 'omp ordered [threads | simd]'.
LLVM_ABI void emitAArch64DeclareSimdFunction(llvm::Function *Fn, unsigned VLENVal, llvm::ArrayRef< DeclareSimdAttrTy > ParamAttrs, DeclareSimdBranch Branch, char ISA, unsigned NarrowestDataSize, bool OutputBecomesInput)
Emit AArch64 vector-function ABI attributes for a declare simd function.
LLVM_ABI Constant * getOrCreateIdent(Constant *SrcLocStr, uint32_t SrcLocStrSize, omp::IdentFlag Flags=omp::IdentFlag(0), unsigned Reserve2Flags=0)
Return an ident_t* encoding the source location SrcLocStr and Flags.
LLVM_ABI void registerDeclareTargetGlobalReplacement(GlobalValue *Original, GlobalValue *Replacement)
Register a module-scope replacement of a declare target global variable.
LLVM_ABI FunctionCallee getOrCreateRuntimeFunction(Module &M, omp::RuntimeFunction FnID)
Return the function declaration for the runtime function with FnID.
LLVM_ABI InsertPointOrErrorTy createCancel(const LocationDescription &Loc, Value *IfCondition, omp::Directive CanceledDirective)
Generator for 'omp cancel'.
std::function< Expected< Function * >(StringRef FunctionName)> FunctionGenCallback
Functions used to generate a function with the given name.
LLVM_ABI CallInst * createOMPAllocShared(const LocationDescription &Loc, Value *Size, const Twine &Name=Twine(""))
Create a runtime call for kmpc_alloc_shared.
ReductionGenCBKind
Enum class for the RedctionGen CallBack type to be used.
LLVM_ABI CanonicalLoopInfo * collapseLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops, InsertPointTy ComputeIP)
Collapse a loop nest into a single loop.
LLVM_ABI void createTaskyield(const LocationDescription &Loc)
Generator for 'omp taskyield'.
std::function< Error(InsertPointTy CodeGenIP)> FinalizeCallbackTy
Callback type for variable finalization (think destructors).
LLVM_ABI void emitBranch(BasicBlock *Target)
LLVM_ABI Error emitCancelationCheckImpl(Value *CancelFlag, omp::Directive CanceledDirective)
Generate control flow and cleanup for cancellation.
static LLVM_ABI void writeThreadBoundsForKernel(const Triple &T, Function &Kernel, int32_t LB, int32_t UB)
LLVM_ABI void emitTaskwaitImpl(const LocationDescription &Loc)
Generate a taskwait runtime call.
LLVM_ABI Constant * registerTargetRegionFunction(TargetRegionEntryInfo &EntryInfo, Function *OutlinedFunction, StringRef EntryFnName, StringRef EntryFnIDName)
Registers the given function and sets up the attribtues of the function Returns the FunctionID.
LLVM_ABI GlobalVariable * emitKernelExecutionMode(StringRef KernelName, omp::OMPTgtExecModeFlags Mode)
Emit the kernel execution mode.
LLVM_ABI void initialize()
Initialize the internal state, this will put structures types and potentially other helpers into the ...
LLVM_ABI InsertPointTy createAtomicCompare(const LocationDescription &Loc, AtomicOpValue &X, AtomicOpValue &V, AtomicOpValue &R, Value *E, Value *D, AtomicOrdering AO, omp::OMPAtomicCompareOp Op, bool IsXBinopExpr, bool IsPostfixUpdate, bool IsFailOnly, bool IsWeak=false)
LLVM_ABI InsertPointTy createAtomicWrite(const LocationDescription &Loc, AtomicOpValue &X, Value *Expr, AtomicOrdering AO, InsertPointTy AllocaIP)
Emit atomic write for : X = Expr — Only Scalar data types.
LLVM_ABI void loadOffloadInfoMetadata(Module &M)
Loads all the offload entries information from the host IR metadata.
function_ref< MapInfosTy &(InsertPointTy CodeGenIP)> GenMapInfoCallbackTy
Callback type for creating the map infos for the kernel parameters.
LLVM_ABI Error emitOffloadingArrays(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, MapInfosTy &CombinedInfo, TargetDataInfo &Info, CustomMapperCallbackTy CustomMapperCB, bool IsNonContiguous=false, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr)
Emit the arrays used to pass the captures and map information to the offloading runtime library.
LLVM_ABI void unrollLoopFull(DebugLoc DL, CanonicalLoopInfo *Loop)
Fully unroll a loop.
function_ref< Error(InsertPointTy CodeGenIP, Value *IndVar)> LoopBodyGenCallbackTy
Callback type for loop body code generation.
LLVM_ABI InsertPointOrErrorTy emitScanReduction(const LocationDescription &Loc, ArrayRef< llvm::OpenMPIRBuilder::ReductionInfo > ReductionInfos, ScanInfo *ScanRedInfo)
This function performs the scan reduction of the values updated in the input phase.
LLVM_ABI void emitFlush(const LocationDescription &Loc)
Generate a flush runtime call.
LLVM_ABI InsertPointOrErrorTy createScope(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsNowait)
Generator for 'omp scope'.
static LLVM_ABI std::pair< int32_t, int32_t > readThreadBoundsForKernel(const Triple &T, Function &Kernel)
}
OpenMPIRBuilderConfig Config
The OpenMPIRBuilder Configuration.
LLVM_ABI CallInst * createOMPInteropDestroy(const LocationDescription &Loc, Value *InteropVar, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_destroy.
LLVM_ABI void emitUsed(StringRef Name, ArrayRef< llvm::WeakTrackingVH > List)
Emit the llvm.used metadata.
LLVM_ABI InsertPointOrErrorTy createSingle(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, bool IsNowait, ArrayRef< llvm::Value * > CPVars={}, ArrayRef< llvm::Function * > CPFuncs={})
Generator for 'omp single'.
LLVM_ABI InsertPointOrErrorTy createTeams(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, Value *NumTeamsLower=nullptr, Value *NumTeamsUpper=nullptr, Value *ThreadLimit=nullptr, Value *IfExpr=nullptr)
Generator for #omp teams
std::forward_list< CanonicalLoopInfo > LoopInfos
Collection of owned canonical loop objects that eventually need to be free'd.
LLVM_ABI llvm::StructType * getKmpTaskAffinityInfoTy()
Return the LLVM struct type matching runtime kmp_task_affinity_info_t.
LLVM_ABI std::string createPlatformSpecificName(ArrayRef< StringRef > Parts) const
Get the create a name using the platform specific separators.
LLVM_ABI FunctionCallee createDispatchNextFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_next_* runtime function for the specified size IVSize and sign IVSigned.
static LLVM_ABI void getKernelArgsVector(TargetKernelArgs &KernelArgs, IRBuilderBase &Builder, SmallVector< Value * > &ArgsVector)
Create the kernel args vector used by emitTargetKernel.
LLVM_ABI InsertPointOrErrorTy createTarget(const LocationDescription &Loc, bool IsOffloadEntry, OpenMPIRBuilder::InsertPointTy AllocaIP, OpenMPIRBuilder::InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks, TargetDataInfo &Info, TargetRegionEntryInfo &EntryInfo, const TargetKernelDefaultAttrs &DefaultAttrs, const TargetKernelRuntimeAttrs &RuntimeAttrs, Value *IfCond, SmallVectorImpl< Value * > &Inputs, GenMapInfoCallbackTy GenMapInfoCB, TargetBodyGenCallbackTy BodyGenCB, TargetGenArgAccessorsCallbackTy ArgAccessorFuncCB, CustomMapperCallbackTy CustomMapperCB, const DependenciesInfo &Dependencies={}, bool HasNowait=false, Value *DynCGroupMem=nullptr, omp::OMPDynGroupprivateFallbackType DynCGroupMemFallback=omp::OMPDynGroupprivateFallbackType::Abort)
Generator for 'omp target'.
LLVM_ABI void unrollLoopHeuristic(DebugLoc DL, CanonicalLoopInfo *Loop)
Fully or partially unroll a loop.
LLVM_ABI omp::OpenMPOffloadMappingFlags getMemberOfFlag(unsigned Position)
Get OMP_MAP_MEMBER_OF flag with extra bits reserved based on the position given.
LLVM_ABI void addAttributes(omp::RuntimeFunction FnID, Function &Fn)
Add attributes known for FnID to Fn.
Module & M
The underlying LLVM-IR module.
StringMap< Constant * > SrcLocStrMap
Map to remember source location strings.
LLVM_ABI void createMapperAllocas(const LocationDescription &Loc, InsertPointTy AllocaIP, unsigned NumOperands, struct MapperAllocas &MapperAllocas)
Create the allocas instruction used in call to mapper functions.
SmallVector< DeclareTargetGlobalReplacement, 8 > DeclareTargetGlobalReplacements
Collection of declare target globals to rewrite uses of during device module finalizaiton.
LLVM_ABI Constant * getOrCreateSrcLocStr(StringRef LocStr, uint32_t &SrcLocStrSize)
Return the (LLVM-IR) string describing the source location LocStr.
LLVM_ABI Error emitTargetRegionFunction(TargetRegionEntryInfo &EntryInfo, FunctionGenCallback &GenerateFunctionCallback, bool IsOffloadEntry, Function *&OutlinedFn, Constant *&OutlinedFnID)
Create a unique name for the entry function using the source location information of the current targ...
LLVM_ABI InsertPointOrErrorTy createIteratorLoop(LocationDescription Loc, llvm::Value *TripCount, IteratorBodyGenTy BodyGen, llvm::StringRef Name="iterator")
Create a canonical iterator loop at the current insertion point.
LLVM_ABI Expected< SmallVector< llvm::CanonicalLoopInfo * > > createCanonicalScanLoops(const LocationDescription &Loc, LoopBodyGenCallbackTy BodyGenCB, Value *Start, Value *Stop, Value *Step, bool IsSigned, bool InclusiveStop, InsertPointTy ComputeIP, const Twine &Name, ScanInfo *ScanRedInfo)
Generator for the control flow structure of an OpenMP canonical loops if the parent directive has an ...
LLVM_ABI FunctionCallee createDispatchFiniFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_fini_* runtime function for the specified size IVSize and sign IVSigned.
function_ref< InsertPointOrErrorTy( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks)> TargetBodyGenCallbackTy
LLVM_ABI void unrollLoopPartial(DebugLoc DL, CanonicalLoopInfo *Loop, int32_t Factor, CanonicalLoopInfo **UnrolledCLI)
Partially unroll a loop.
function_ref< Error(Value *DeviceID, Value *RTLoc, IRBuilderBase::InsertPoint TargetTaskAllocaIP)> TargetTaskBodyCallbackTy
Callback type for generating the bodies of device directives that require outer target tasks (e....
Expected< MapInfosTy & > MapInfosOrErrorTy
bool HandleFPNegZero
Emit atomic compare for constructs: — Only scalar data types cond-expr-stmt: x = x ordop expr ?
LLVM_ABI void emitTaskyieldImpl(const LocationDescription &Loc)
Generate a taskyield runtime call.
LLVM_ABI void emitMapperCall(const LocationDescription &Loc, Function *MapperFunc, Value *SrcLocInfo, Value *MaptypesArg, Value *MapnamesArg, struct MapperAllocas &MapperAllocas, int64_t DeviceID, unsigned NumOperands)
Create the call for the target mapper function.
LLVM_ABI InsertPointOrErrorTy createDistribute(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB)
Generator for #omp distribute
LLVM_ABI InsertPointOrErrorTy createTask(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB, bool Tied=true, Value *Final=nullptr, Value *IfCondition=nullptr, const DependenciesInfo &Dependencies={}, const AffinityData &Affinities={}, bool Mergeable=false, Value *EventHandle=nullptr, Value *Priority=nullptr)
Generator for #omp taskloop
function_ref< Expected< Function * >(unsigned int)> CustomMapperCallbackTy
LLVM_ABI InsertPointTy createOrderedDepend(const LocationDescription &Loc, InsertPointTy AllocaIP, unsigned NumLoops, ArrayRef< llvm::Value * > StoreValues, const Twine &Name, bool IsDependSource)
Generator for 'omp ordered depend (source | sink)'.
LLVM_ABI InsertPointTy createCopyinClauseBlocks(InsertPointTy IP, Value *MasterAddr, Value *PrivateAddr, llvm::IntegerType *IntPtrTy, bool BranchtoEnd=true)
Generate conditional branch and relevant BasicBlocks through which private threads copy the 'copyin' ...
function_ref< InsertPointOrErrorTy( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, Value &Original, Value &Inner, Value *&ReplVal)> PrivatizeCallbackTy
Callback type for variable privatization (think copy & default constructor).
LLVM_ABI bool isFinalized()
Check whether the finalize function has already run.
SmallVector< FinalizationInfo, 8 > FinalizationStack
The finalization stack made up of finalize callbacks currently in-flight, wrapped into FinalizationIn...
LLVM_ABI std::vector< CanonicalLoopInfo * > tileLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops, ArrayRef< Value * > TileSizes)
Tile a loop nest.
LLVM_ABI CallInst * createOMPInteropInit(const LocationDescription &Loc, Value *InteropVar, omp::OMPInteropType InteropType, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_init.
LLVM_ABI Error emitIfClause(Value *Cond, BodyGenCallbackTy ThenGen, BodyGenCallbackTy ElseGen, InsertPointTy AllocaIP={}, ArrayRef< BasicBlock * > DeallocBlocks={})
Emits code for OpenMP 'if' clause using specified BodyGenCallbackTy Here is the logic: if (Cond) { Th...
LLVM_ABI void finalize(Function *Fn=nullptr)
Finalize the underlying module, e.g., by outlining regions.
LLVM_ABI Function * getOrCreateRuntimeFunctionPtr(omp::RuntimeFunction FnID)
void addOutlineInfo(std::unique_ptr< OutlineInfo > &&OI)
Add a new region that will be outlined later.
LLVM_ABI InsertPointTy createTargetInit(const LocationDescription &Loc, const llvm::OpenMPIRBuilder::TargetKernelDefaultAttrs &Attrs)
The omp target interface.
LLVM_ABI InsertPointOrErrorTy createReductions(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< ReductionInfo > ReductionInfos, ArrayRef< bool > IsByRef, bool IsNoWait=false, bool IsTeamsReduction=false)
Generator for 'omp reduction'.
const Triple T
The target triple of the underlying module.
DenseMap< std::pair< Constant *, uint64_t >, Constant * > IdentMap
Map to remember existing ident_t*.
LLVM_ABI CallInst * createOMPFree(const LocationDescription &Loc, Value *Addr, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_free.
LLVM_ABI InsertPointOrErrorTy createReductionsGPU(const LocationDescription &Loc, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< ReductionInfo > ReductionInfos, ArrayRef< bool > IsByRef, bool IsNoWait=false, bool IsTeamsReduction=false, bool IsSPMD=false, ReductionGenCBKind ReductionGenCBKind=ReductionGenCBKind::MLIR, std::optional< omp::GV > GridValue={}, Value *SrcLocInfo=nullptr)
Design of OpenMP reductions on the GPU.
LLVM_ABI FunctionCallee createForStaticInitFunction(unsigned IVSize, bool IVSigned, bool IsGPUDistribute)
Returns __kmpc_for_static_init_* runtime function for the specified size IVSize and sign IVSigned.
LLVM_ABI CallInst * createOMPAlloc(const LocationDescription &Loc, Value *Size, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_alloc.
LLVM_ABI void emitNonContiguousDescriptor(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, MapInfosTy &CombinedInfo, TargetDataInfo &Info)
Emit an array of struct descriptors to be assigned to the offload args.
LLVM_ABI InsertPointOrErrorTy createSection(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB)
Generator for 'omp section'.
LLVM_ABI InsertPointOrErrorTy createTaskgroup(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB)
Generator for the taskgroup construct.
LLVM_ABI InsertPointOrErrorTy createParallel(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< BasicBlock * > DeallocBlocks, BodyGenCallbackTy BodyGenCB, PrivatizeCallbackTy PrivCB, FinalizeCallbackTy FiniCB, Value *IfCondition, Value *NumThreads, omp::ProcBindKind ProcBind, bool IsCancellable)
Generator for 'omp parallel'.
function_ref< InsertPointOrErrorTy(InsertPointTy)> EmitFallbackCallbackTy
Callback function type for functions emitting the host fallback code that is executed when the kernel...
static LLVM_ABI TargetRegionEntryInfo getTargetEntryUniqueInfo(FileIdentifierInfoCallbackTy CallBack, vfs::FileSystem &VFS, StringRef ParentName="")
Creates a unique info for a target entry when provided a filename and line number from.
LLVM_ABI void emitTaskDependency(IRBuilderBase &Builder, Value *Entry, const DependData &Dep)
Store one kmp_depend_info entry at the given Entry pointer.
LLVM_ABI void emitBlock(BasicBlock *BB, Function *CurFn, bool IsFinished=false)
LLVM_ABI Value * getOrCreateThreadID(Value *Ident)
Return the current thread ID.
LLVM_ABI InsertPointOrErrorTy createMaster(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB)
Generator for 'omp master'.
LLVM_ABI InsertPointOrErrorTy createTargetData(const LocationDescription &Loc, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks, Value *DeviceID, Value *IfCond, TargetDataInfo &Info, GenMapInfoCallbackTy GenMapInfoCB, CustomMapperCallbackTy CustomMapperCB, omp::RuntimeFunction *MapperFunc=nullptr, function_ref< InsertPointOrErrorTy(InsertPointTy CodeGenIP, BodyGenTy BodyGenType)> BodyGenCB=nullptr, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr, Value *SrcLocInfo=nullptr)
Generator for 'omp target data'.
LLVM_ABI CallInst * createRuntimeFunctionCall(FunctionCallee Callee, ArrayRef< Value * > Args, StringRef Name="")
LLVM_ABI InsertPointOrErrorTy emitKernelLaunch(const LocationDescription &Loc, Value *OutlinedFnID, EmitFallbackCallbackTy EmitTargetCallFallbackCB, TargetKernelArgs &Args, Value *DeviceID, Value *RTLoc, InsertPointTy AllocaIP)
Generate a target region entry call and host fallback call.
StringMap< GlobalVariable *, BumpPtrAllocator > InternalVars
An ordered map of auto-generated variables to their unique names.
LLVM_ABI InsertPointOrErrorTy createCancellationPoint(const LocationDescription &Loc, omp::Directive CanceledDirective)
Generator for 'omp cancellation point'.
LLVM_ABI CallInst * createOMPAlignedAlloc(const LocationDescription &Loc, Value *Align, Value *Size, Value *Allocator, std::string Name="")
Create a runtime call for kmpc_align_alloc.
LLVM_ABI FunctionCallee createDispatchInitFunction(unsigned IVSize, bool IVSigned)
Returns __kmpc_dispatch_init_* runtime function for the specified size IVSize and sign IVSigned.
LLVM_ABI InsertPointOrErrorTy createScan(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< llvm::Value * > ScanVars, ArrayRef< llvm::Type * > ScanVarsType, bool IsInclusive, ScanInfo *ScanRedInfo)
This directive split and directs the control flow to input phase blocks or scan phase blocks based on...
LLVM_ABI CallInst * createOMPFreeShared(const LocationDescription &Loc, Value *Addr, Value *Size, const Twine &Name=Twine(""))
Create a runtime call for kmpc_free_shared.
LLVM_ABI CallInst * createOMPInteropUse(const LocationDescription &Loc, Value *InteropVar, Value *Device, Value *NumDependences, Value *DependenceAddress, bool HaveNowaitClause)
Create a runtime call for __tgt_interop_use.
IRBuilder<>::InsertPoint InsertPointTy
Type used throughout for insertion points.
LLVM_ABI GlobalVariable * getOrCreateInternalVariable(Type *Ty, const StringRef &Name, std::optional< unsigned > AddressSpace={})
Gets (if variable with the given name already exist) or creates internal global variable with the spe...
LLVM_ABI GlobalVariable * createOffloadMapnames(SmallVectorImpl< llvm::Constant * > &Names, std::string VarName)
Create the global variable holding the offload names information.
std::forward_list< ScanInfo > ScanInfos
Collection of owned ScanInfo objects that eventually need to be free'd.
static LLVM_ABI void writeTeamsForKernel(const Triple &T, Function &Kernel, int32_t LB, int32_t UB)
LLVM_ABI Value * calculateCanonicalLoopTripCount(const LocationDescription &Loc, Value *Start, Value *Stop, Value *Step, bool IsSigned, bool InclusiveStop, const Twine &Name="loop")
Calculate the trip count of a canonical loop.
LLVM_ABI InsertPointOrErrorTy createBarrier(const LocationDescription &Loc, omp::Directive Kind, bool ForceSimpleCall=false, bool CheckCancelFlag=true)
Emitter methods for OpenMP directives.
LLVM_ABI void setCorrectMemberOfFlag(omp::OpenMPOffloadMappingFlags &Flags, omp::OpenMPOffloadMappingFlags MemberOfFlag)
Given an initial flag set, this function modifies it to contain the passed in MemberOfFlag generated ...
LLVM_ABI Error emitOffloadingArraysAndArgs(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, TargetDataInfo &Info, TargetDataRTArgs &RTArgs, MapInfosTy &CombinedInfo, CustomMapperCallbackTy CustomMapperCB, bool IsNonContiguous=false, bool ForEndCall=false, function_ref< void(unsigned int, Value *)> DeviceAddrCB=nullptr)
Allocates memory for and populates the arrays required for offloading (offload_{baseptrs|ptrs|mappers...
LLVM_ABI Constant * getOrCreateDefaultSrcLocStr(uint32_t &SrcLocStrSize)
Return the (LLVM-IR) string describing the default source location.
LLVM_ABI InsertPointOrErrorTy createCritical(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, StringRef CriticalName, Value *HintInst)
Generator for 'omp critical'.
LLVM_ABI void createError(const LocationDescription &Loc, bool IsFatal, Value *Message)
Generate a call to the runtime to emit the diagnostic of an OpenMP error directive with at(execution)...
LLVM_ABI void createOffloadEntry(Constant *ID, Constant *Addr, uint64_t Size, int32_t Flags, GlobalValue::LinkageTypes, StringRef Name="")
Creates offloading entry for the provided entry ID ID, address Addr, size Size, and flags Flags.
static LLVM_ABI unsigned getOpenMPDefaultSimdAlign(const Triple &TargetTriple, const StringMap< bool > &Features)
Get the default alignment value for given target.
LLVM_ABI unsigned getFlagMemberOffset()
Get the offset of the OMP_MAP_MEMBER_OF field.
LLVM_ABI InsertPointOrErrorTy applyWorkshareLoop(DebugLoc DL, CanonicalLoopInfo *CLI, InsertPointTy AllocaIP, bool NeedsBarrier, llvm::omp::ScheduleKind SchedKind=llvm::omp::OMP_SCHEDULE_Default, Value *ChunkSize=nullptr, bool HasSimdModifier=false, bool HasMonotonicModifier=false, bool HasNonmonotonicModifier=false, bool HasOrderedClause=false, omp::WorksharingLoopType LoopType=omp::WorksharingLoopType::ForStaticLoop, bool NoLoop=false, bool HasDistSchedule=false, Value *DistScheduleChunkSize=nullptr)
Modifies the canonical loop to be a workshare loop.
LLVM_ABI InsertPointOrErrorTy createAtomicCapture(const LocationDescription &Loc, InsertPointTy AllocaIP, AtomicOpValue &X, AtomicOpValue &V, Value *Expr, AtomicOrdering AO, AtomicRMWInst::BinOp RMWOp, AtomicUpdateCallbackTy &UpdateOp, bool UpdateExpr, bool IsPostfixUpdate, bool IsXBinopExpr, bool IsIgnoreDenormalMode=false, bool IsFineGrainedMemory=false, bool IsRemoteMemory=false)
Emit atomic update for constructs: — Only Scalar data types V = X; X = X BinOp Expr ,...
LLVM_ABI CanonicalLoopInfo * createLoopSkeleton(DebugLoc DL, Value *TripCount, Function *F, BasicBlock *PreInsertBefore, BasicBlock *PostInsertBefore, const Twine &Name={}, bool IsCollapsed=false)
Create the control flow structure of a canonical OpenMP loop.
LLVM_ABI void createOffloadEntriesAndInfoMetadata(EmitMetadataErrorReportFunctionTy &ErrorReportFunction)
LLVM_ABI void applySimd(CanonicalLoopInfo *Loop, MapVector< Value *, Value * > AlignedVars, Value *IfCond, omp::OrderKind Order, ConstantInt *Simdlen, ConstantInt *Safelen)
Add metadata to simd-ize a loop.
SmallVector< std::unique_ptr< OutlineInfo >, 16 > OutlineInfos
Collection of regions that need to be outlined during finalization.
LLVM_ABI InsertPointOrErrorTy createAtomicUpdate(const LocationDescription &Loc, InsertPointTy AllocaIP, AtomicOpValue &X, Value *Expr, AtomicOrdering AO, AtomicRMWInst::BinOp RMWOp, AtomicUpdateCallbackTy &UpdateOp, bool IsXBinopExpr, bool IsIgnoreDenormalMode=false, bool IsFineGrainedMemory=false, bool IsRemoteMemory=false)
Emit atomic update for constructs: X = X BinOp Expr ,or X = Expr BinOp X For complex Operations: X = ...
std::function< std::tuple< std::string, uint64_t >()> FileIdentifierInfoCallbackTy
bool isLastFinalizationInfoCancellable(omp::Directive DK)
Return true if the last entry in the finalization stack is of kind DK and cancellable.
LLVM_ABI InsertPointTy emitTargetKernel(const LocationDescription &Loc, InsertPointTy AllocaIP, Value *&Return, Value *Ident, Value *DeviceID, Value *NumTeams, Value *NumThreads, Value *HostPtr, ArrayRef< Value * > KernelArgs)
Generate a target region entry call.
LLVM_ABI GlobalVariable * createOffloadMaptypes(SmallVectorImpl< uint64_t > &Mappings, std::string VarName)
Create the global variable holding the offload mappings information.
LLVM_ABI ~OpenMPIRBuilder()
LLVM_ABI Expected< Function * > emitUserDefinedMapper(function_ref< MapInfosOrErrorTy(InsertPointTy CodeGenIP, llvm::Value *PtrPHI, llvm::Value *BeginArg)> PrivAndGenMapInfoCB, llvm::Type *ElemTy, StringRef FuncName, CustomMapperCallbackTy CustomMapperCB, bool PreserveMemberOfFlags=false, bool PropagatePresentToPointee=false)
Emit the user-defined mapper function.
LLVM_ABI CallInst * createCachedThreadPrivate(const LocationDescription &Loc, llvm::Value *Pointer, llvm::ConstantInt *Size, const llvm::Twine &Name=Twine(""))
Create a runtime call for kmpc_threadprivate_cached.
IRBuilder Builder
The LLVM-IR Builder used to create IR.
LLVM_ABI GlobalValue * createGlobalFlag(unsigned Value, StringRef Name)
Create a hidden global flag Name in the module with initial value Value.
LLVM_ABI void emitOffloadingArraysArgument(IRBuilderBase &Builder, OpenMPIRBuilder::TargetDataRTArgs &RTArgs, OpenMPIRBuilder::TargetDataInfo &Info, bool ForEndCall=false)
Emit the arguments to be passed to the runtime library based on the arrays of base pointers,...
LLVM_ABI InsertPointOrErrorTy createMasked(const LocationDescription &Loc, BodyGenCallbackTy BodyGenCB, FinalizeCallbackTy FiniCB, Value *Filter)
Generator for 'omp masked'.
LLVM_ABI Expected< CanonicalLoopInfo * > createCanonicalLoop(const LocationDescription &Loc, LoopBodyGenCallbackTy BodyGenCB, Value *TripCount, const Twine &Name="loop")
Generator for the control flow structure of an OpenMP canonical loop.
function_ref< Expected< InsertPointTy >( InsertPointTy AllocaIP, InsertPointTy CodeGenIP, Value *DestPtr, Value *SrcPtr)> TaskDupCallbackTy
Callback type for task duplication function code generation.
LLVM_ABI Value * getSizeInBytes(Value *BasePtr)
Computes the size of type in bytes.
llvm::function_ref< llvm::Error( InsertPointTy BodyIP, llvm::Value *LinearIV)> IteratorBodyGenTy
LLVM_ABI FunctionCallee createDispatchDeinitFunction()
Returns __kmpc_dispatch_deinit runtime function.
LLVM_ABI void registerTargetGlobalVariable(OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind CaptureClause, OffloadEntriesInfoManager::OMPTargetDeviceClauseKind DeviceClause, bool IsDeclaration, bool IsExternallyVisible, TargetRegionEntryInfo EntryInfo, StringRef MangledName, std::vector< GlobalVariable * > &GeneratedRefs, bool OpenMPSIMD, std::vector< Triple > TargetTriple, std::function< Constant *()> GlobalInitializer, std::function< GlobalValue::LinkageTypes()> VariableLinkage, Type *LlvmPtrTy, Constant *Addr)
Registers a target variable for device or host.
LLVM_ABI void createTargetDeinit(const LocationDescription &Loc, int32_t TeamsReductionDataSize=0)
Create a runtime call for kmpc_target_deinit.
BodyGenTy
Type of BodyGen to use for region codegen.
LLVM_ABI CanonicalLoopInfo * fuseLoops(DebugLoc DL, ArrayRef< CanonicalLoopInfo * > Loops)
Fuse a sequence of loops.
LLVM_ABI void emitX86DeclareSimdFunction(llvm::Function *Fn, unsigned NumElements, const llvm::APSInt &VLENVal, llvm::ArrayRef< DeclareSimdAttrTy > ParamAttrs, DeclareSimdBranch Branch)
Emit x86 vector-function ABI attributes for a declare simd function.
SmallVector< llvm::Function *, 16 > ConstantAllocaRaiseCandidates
A collection of candidate target functions that's constant allocas will attempt to be raised on a cal...
OffloadEntriesInfoManager OffloadInfoManager
Info manager to keep track of target regions.
static LLVM_ABI std::pair< int32_t, int32_t > readTeamBoundsForKernel(const Triple &T, Function &Kernel)
Read/write a bounds on teams for Kernel.
const std::string ompOffloadInfoName
OMP Offload Info Metadata name string.
Expected< InsertPointTy > InsertPointOrErrorTy
Type used to represent an insertion point or an error value.
LLVM_ABI InsertPointTy createCopyPrivate(const LocationDescription &Loc, llvm::Value *BufSize, llvm::Value *CpyBuf, llvm::Value *CpyFn, llvm::Value *DidIt)
Generator for __kmpc_copyprivate.
LLVM_ABI InsertPointOrErrorTy createSections(const LocationDescription &Loc, InsertPointTy AllocaIP, ArrayRef< StorableBodyGenCallbackTy > SectionCBs, PrivatizeCallbackTy PrivCB, FinalizeCallbackTy FiniCB, bool IsCancellable, bool IsNowait)
Generator for 'omp sections'.
std::function< void(EmitMetadataErrorKind, TargetRegionEntryInfo)> EmitMetadataErrorReportFunctionTy
Callback function type.
function_ref< InsertPointOrErrorTy( Argument &Arg, Value *Input, Value *&RetVal, InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< InsertPointTy > DeallocIPs)> TargetGenArgAccessorsCallbackTy
LLVM_ABI Expected< ScanInfo * > scanInfoInitialize()
Creates a ScanInfo object, allocates and returns the pointer.
LLVM_ABI InsertPointOrErrorTy emitTargetTask(TargetTaskBodyCallbackTy TaskBodyCB, Value *DeviceID, Value *RTLoc, OpenMPIRBuilder::InsertPointTy AllocaIP, const DependenciesInfo &Dependencies, const TargetDataRTArgs &RTArgs, bool HasNoWait)
Generate a target-task for the target construct.
LLVM_ABI InsertPointTy createAtomicRead(const LocationDescription &Loc, AtomicOpValue &X, AtomicOpValue &V, AtomicOrdering AO, InsertPointTy AllocaIP)
Emit atomic Read for : V = X — Only Scalar data types.
function_ref< Error(InsertPointTy AllocaIP, InsertPointTy CodeGenIP, ArrayRef< BasicBlock * > DeallocBlocks)> BodyGenCallbackTy
Callback type for body (=inner region) code generation.
bool updateToLocation(const LocationDescription &Loc)
Update the internal location to Loc.
LLVM_ABI void createFlush(const LocationDescription &Loc)
Generator for 'omp flush'.
LLVM_ABI void createTaskwait(const LocationDescription &Loc, DependenciesInfo Dependencies={})
Generator for 'omp taskwait'.
LLVM_ABI Constant * getAddrOfDeclareTargetVar(OffloadEntriesInfoManager::OMPTargetGlobalVarEntryKind CaptureClause, OffloadEntriesInfoManager::OMPTargetDeviceClauseKind DeviceClause, bool IsDeclaration, bool IsExternallyVisible, TargetRegionEntryInfo EntryInfo, StringRef MangledName, std::vector< GlobalVariable * > &GeneratedRefs, bool OpenMPSIMD, std::vector< Triple > TargetTriple, Type *LlvmPtrTy, std::function< Constant *()> GlobalInitializer, std::function< GlobalValue::LinkageTypes()> VariableLinkage)
Retrieve (or create if non-existent) the address of a declare target variable, used in conjunction wi...
origPtr *with the address space normalization required by the runtime entry point *The NULL descriptor makes the runtime walk the enclosing taskgroups to *find the matching task_reduction registration for the item The lookups *are emitted at p Loc
EmitMetadataErrorKind
The kind of errors that can occur when emitting the offload entries and metadata.
@ EMIT_MD_DECLARE_TARGET_ERROR
@ EMIT_MD_GLOBAL_VAR_INDIRECT_ERROR
@ EMIT_MD_GLOBAL_VAR_LINK_ERROR
@ EMIT_MD_TARGET_REGION_ERROR
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
Pseudo-analysis pass that exposes the PassInstrumentation to pass managers.
Class to represent pointers.
static PointerType * getUnqual(LLVMContext &C)
This constructs an opaque pointer to an object in the default address space (address space zero).
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
PostDominatorTree Class - Concrete subclass of DominatorTree that is used to compute the post-dominat...
Analysis pass that exposes the ScalarEvolution for a function.
LLVM_ABI ScalarEvolution run(Function &F, FunctionAnalysisManager &AM)
The main scalar evolution driver.
ScanInfo holds the information to assist in lowering of Scan reduction.
llvm::SmallDenseMap< llvm::Value *, llvm::Value * > * ScanBuffPtrs
Maps the private reduction variable to the pointer of the temporary buffer.
llvm::BasicBlock * OMPScanLoopExit
Exit block of loop body.
llvm::Value * IV
Keeps track of value of iteration variable for input/scan loop to be used for Scan directive lowering...
llvm::BasicBlock * OMPAfterScanBlock
Dominates the body of the loop before scan directive.
llvm::BasicBlock * OMPScanInit
Block before loop body where scan initializations are done.
llvm::BasicBlock * OMPBeforeScanBlock
Dominates the body of the loop before scan directive.
llvm::BasicBlock * OMPScanFinish
Block after loop body where scan finalizations are done.
llvm::Value * Span
Stores the span of canonical loop being lowered to be used for temporary buffer allocation or Finaliz...
bool OMPFirstScanLoop
If true, it indicates Input phase is lowered; else it indicates ScanPhase is lowered.
llvm::BasicBlock * OMPScanDispatch
Controls the flow to before or after scan blocks.
A vector that has set insertion semantics.
bool remove_if(UnaryPredicate P)
Remove items from the set vector based on a predicate function.
bool empty() const
Determine if the SetVector is empty or not.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
bool any() const
Returns true if any bit is set.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
void append(StringRef RHS)
Append from a StringRef.
StringRef str() const
Explicit conversion to StringRef.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setAlignment(Align Align)
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this store instruction.
StringMap - This is an unconventional map that is specialized for handling keys that are "strings",...
ValueTy lookup(StringRef Key) const
lookup - Return the entry for the specified key, or a default constructed value if no such entry exis...
Represent a constant reference to a string, i.e.
std::string str() const
Get the contents as an std::string.
constexpr bool empty() const
Check if the string is empty.
constexpr size_t size() const
Get the string size.
size_t count(char C) const
Return the number of occurrences of C in the string.
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
StringRef drop_back(size_t N=1) const
Return a StringRef equal to 'this' but with the last N elements dropped.
Class to represent struct types.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
static LLVM_ABI StructType * create(LLVMContext &Context, StringRef Name)
This creates an identified struct.
Type * getElementType(unsigned N) const
LLVM_ABI void addCase(ConstantInt *OnVal, BasicBlock *Dest)
Add an entry to the switch instruction.
Analysis pass providing the TargetTransformInfo.
LLVM_ABI Result run(const Function &F, FunctionAnalysisManager &)
TargetTransformInfo Result
Analysis pass providing the TargetLibraryInfo.
Target - Wrapper for Target specific information.
TargetMachine * createTargetMachine(const Triple &TT, StringRef CPU, StringRef Features, const TargetOptions &Options, std::optional< Reloc::Model > RM, std::optional< CodeModel::Model > CM=std::nullopt, CodeGenOptLevel OL=CodeGenOptLevel::Default, bool JIT=false) const
createTargetMachine - Create a target specific machine implementation for the specified Triple.
Triple - Helper class for working with autoconf configuration names.
bool isPPC() const
Tests whether the target is PowerPC (32- or 64-bit LE or BE).
bool isX86() const
Tests whether the target is x86 (32- or 64-bit).
bool isWasm() const
Tests whether the target is wasm (32- and 64-bit).
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
LLVM_ABI unsigned getIntegerBitWidth() const
LLVM_ABI Type * getStructElementType(unsigned N) const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
bool isStructTy() const
True if this is an instance of StructType.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
bool isVoidTy() const
Return true if this is 'void'.
Unconditional Branch instruction.
static UncondBrInst * Create(BasicBlock *Target, InsertPosition InsertBefore=nullptr)
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
This function has undefined behavior.
Produce an estimate of the unrolled cost of the specified loop.
LLVM_ABI bool canUnroll(OptimizationRemarkEmitter *ORE=nullptr, const Loop *L=nullptr) const
Whether it is legal to unroll this loop.
uint64_t getRolledLoopSize() const
A Use represents the edge between a Value definition and its users.
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
user_iterator user_begin()
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
bool hasOneUse() const
Return true if there is exactly one use of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
iterator_range< user_iterator > users()
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
LLVM_ABI bool hasNUses(unsigned N) const
Return true if this Value has exactly N uses.
LLVM_ABI User * getUniqueUndroppableUser()
Return true if there is exactly one unique user of this value that cannot be dropped (that user can h...
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
LLVM_ABI bool replaceUsesWithIf(Value *New, llvm::function_ref< bool(Use &U)> ShouldReplace)
Go through the uses list for this definition and make each use point to "V" if the callback ShouldRep...
iterator_range< use_iterator > uses()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
self_iterator getIterator()
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
A raw_ostream that writes to an SmallVector or SmallString.
StringRef str() const
Return a StringRef for the vector contents.
The virtual file system interface.
llvm::ErrorOr< std::unique_ptr< llvm::MemoryBuffer > > getBufferForFile(const Twine &Name, int64_t FileSize=-1, bool RequiresNullTerminator=true, bool IsVolatile=false, bool IsText=true)
This is a convenience method that opens a file, gets its content and then closes the file.
virtual llvm::ErrorOr< Status > status(const Twine &Path)=0
Get the status of the entry at Path, if one exists.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ PTX_Kernel
Call to a PTX kernel. Passes all arguments in parameter space.
@ BasicBlock
Various leaf nodes.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
Flag
These should be considered private to the implementation of the MCInstrDesc class.
constexpr StringLiteral MaxNTID("nvvm.maxntid")
constexpr StringLiteral MaxClusterRank("nvvm.maxclusterrank")
initializer< Ty > init(const Ty &Val)
@ User
could "use" a pointer
LLVM_ABI GlobalVariable * emitOffloadingEntry(Module &M, object::OffloadKind Kind, Constant *Addr, StringRef Name, uint64_t Size, uint32_t Flags, uint64_t Data, Constant *AuxAddr=nullptr)
OpenMPOffloadMappingFlags
Values for bit flags used to specify the mapping type for offloading.
@ OMP_MAP_PTR_AND_OBJ
The element being mapped is a pointer-pointee pair; both the pointer and the pointee should be mapped...
@ OMP_MAP_MEMBER_OF
The 16 MSBs of the flags indicate whether the entry is member of some struct/class.
IdentFlag
IDs for all omp runtime library ident_t flag encodings (see their defintion in openmp/runtime/src/kmp...
RuntimeFunction
IDs for all omp runtime library (RTL) functions.
constexpr const GV & getAMDGPUGridValues()
static constexpr GV SPIRVGridValues
For generic SPIR-V GPUs.
OMPDynGroupprivateFallbackType
The fallback types for the dyn_groupprivate clause.
static constexpr GV NVPTXGridValues
For Nvidia GPUs.
@ OMP_TGT_EXEC_MODE_SPMD_NO_LOOP
@ OMP_TGT_EXEC_MODE_GENERIC
Function * Kernel
Summary of a kernel (=entry point for target offloading).
WorksharingLoopType
A type of worksharing loop construct.
OMPAtomicCompareOp
Atomic compare operations. Currently OpenMP only supports ==, >, and <.
NodeAddr< PhiNode * > Phi
friend class Instruction
Iterator for Instructions in a `BasicBlock.
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
LLVM_ABI BasicBlock * splitBBWithSuffix(IRBuilderBase &Builder, bool CreateBranch, llvm::Twine Suffix=".split")
Like splitBB, but reuses the current block's name for the new name.
detail::zippy< detail::zip_shortest, T, U, Args... > zip(T &&t, U &&u, Args &&...args)
zip iterator for two or more iteratable types.
LLVM_ABI unsigned computeUnrollCount(Loop *L, const TargetTransformInfo &TTI, DominatorTree &DT, LoopInfo *LI, AssumptionCache *AC, ScalarEvolution &SE, const SmallPtrSetImpl< const Value * > &EphValues, OptimizationRemarkEmitter *ORE, unsigned TripCount, unsigned MaxTripCount, bool MaxOrZero, unsigned TripMultiple, const UnrollCostEstimator &UCE, TargetTransformInfo::UnrollingPreferences &UP, TargetTransformInfo::PeelingPreferences &PP)
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
hash_code hash_value(const FixedPointSemantics &Val)
LLVM_ABI Expected< std::unique_ptr< Module > > parseBitcodeFile(MemoryBufferRef Buffer, LLVMContext &Context, ParserCallbacks Callbacks={})
Read the specified bitcode file, returning the module.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
@ LLVM_MARK_AS_BITMASK_ENUM
LLVM_ABI BasicBlock * CloneBasicBlock(const BasicBlock *BB, ValueToValueMapTy &VMap, const Twine &NameSuffix="", Function *F=nullptr, ClonedCodeInfo *CodeInfo=nullptr, bool MapAtoms=true)
Return a copy of the specified basic block, but without embedding the block into a particular functio...
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
unsigned getPointerAddressSpace(const Type *T)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
auto successors(const MachineBasicBlock *BB)
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
LLVM_ABI std::error_code inconvertibleErrorCode()
The value returned by this function can be returned from convertToErrorCode for Error values where no...
testing::Matcher< const detail::ErrorHolder & > Failed()
constexpr from_range_t from_range
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_ENABLE_BITMASK_ENUMS_IN_NAMESPACE()
LLVM_ABI BasicBlock * splitBB(IRBuilderBase::InsertPoint IP, bool CreateBranch, DebugLoc DL, llvm::Twine Name={})
Split a BasicBlock at an InsertPoint, even if the block is degenerate (missing the terminator).
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
LLVM_ABI TargetTransformInfo::UnrollingPreferences gatherUnrollingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, BlockFrequencyInfo *BFI, ProfileSummaryInfo *PSI, llvm::OptimizationRemarkEmitter &ORE, int OptLevel, std::optional< unsigned > UserThreshold, std::optional< bool > UserAllowPartial, std::optional< bool > UserRuntime, std::optional< bool > UserUpperBound, std::optional< unsigned > UserFullUnrollMaxCount)
Gather the various unrolling parameters based on the defaults, compiler flags, TTI overrides and user...
std::string utostr(uint64_t X, bool isNeg=false)
ErrorOr< T > expectedToErrorOrAndEmitErrors(LLVMContext &Ctx, Expected< T > Val)
bool isa_and_nonnull(const Y &Val)
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
auto dyn_cast_or_null(const Y &Val)
LLVM_ABI bool convertUsersOfConstantsToInstructions(ArrayRef< Constant * > Consts, Function *RestrictToFunc=nullptr, bool RemoveDeadConstants=true, bool IncludeSelf=false)
Replace constant expressions users of the given constants with instructions.
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
auto reverse(ContainerTy &&C)
LLVM_ABI TargetTransformInfo::PeelingPreferences gatherPeelingPreferences(Loop *L, ScalarEvolution &SE, const TargetTransformInfo &TTI, std::optional< bool > UserAllowPeeling, std::optional< bool > UserAllowProfileBasedPeeling, bool UnrollingSpecficValues=false)
LLVM_ABI void SplitBlockAndInsertIfThenElse(Value *Cond, BasicBlock::iterator SplitBefore, Instruction **ThenTerm, Instruction **ElseTerm, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr)
SplitBlockAndInsertIfThenElse is similar to SplitBlockAndInsertIfThen, but also creates the ElseBlock...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
CodeGenOptLevel
Code generation optimization level.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
format_object< Ts... > format(const char *Fmt, const Ts &... Vals)
These are helper functions used to produce formatted output.
Error make_error(ArgTs &&... Args)
Make a Error instance representing failure using the given error info type.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
void cantFail(Error Err, const char *Msg=nullptr)
Report a fatal error if Err is a failure value.
LLVM_ABI bool MergeBlockIntoPredecessor(BasicBlock *BB, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, MemoryDependenceResults *MemDep=nullptr, bool PredecessorWithTwoSuccessors=false, DominatorTree *DT=nullptr)
Attempts to merge a block into its predecessor, if possible.
@ Mul
Product of integers.
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
DWARFExpression::Operation Op
LLVM_ABI void remapInstructionsInBlocks(ArrayRef< BasicBlock * > Blocks, ValueToValueMapTy &VMap)
Remaps instructions in Blocks using the mapping in VMap.
ArrayRef(const T &OneElt) -> ArrayRef< T >
OutputIt copy(R &&Range, OutputIt Out)
constexpr unsigned BitWidth
ValueMap< const Value *, WeakTrackingVH > ValueToValueMapTy
LLVM_ABI void spliceBB(IRBuilderBase::InsertPoint IP, BasicBlock *New, bool CreateBranch, DebugLoc DL)
Move the instruction after an InsertPoint to the beginning of another BasicBlock.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
auto predecessors(const MachineBasicBlock *BB)
auto filter_to_vector(ContainerTy &&C, PredicateFn &&Pred)
Filter a range to a SmallVector with the element types deduced.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
LLVM_ABI Constant * ConstantFoldInsertValueInstruction(Constant *Agg, Constant *Val, ArrayRef< unsigned > Idxs)
Attempt to constant fold an insertvalue instruction with the specified operands and indices.
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI void DeleteDeadBlocks(ArrayRef< BasicBlock * > BBs, DomTreeUpdater *DTU=nullptr, bool KeepOneInputPHIs=false)
Delete the specified blocks from BB.
bool to_integer(StringRef S, N &Num, unsigned Base=0)
Convert the string S to an integer of the specified type using the radix Base. If Base is 0,...
static auto filterDbgVars(iterator_range< simple_ilist< DbgRecord >::iterator > R)
Filter the DbgRecord range to DbgVariableRecord types only and downcast.
This struct is a compact representation of a valid (non-zero power of two) alignment.
static LLVM_ABI void collectEphemeralValues(const Loop *L, AssumptionCache *AC, SmallPtrSetImpl< const Value * > &EphValues)
Collect a loop's ephemeral values (those used only by an assume or similar intrinsics in the loop).
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
A struct to pack the relevant information for an OpenMP affinity clause.
a struct to pack relevant information while generating atomic Ops
A struct to pack the relevant information for an OpenMP depend clause.
omp::RTLDependenceKindTy DepKind
A struct to pack static and dynamic dependency information for a task.
SmallVector< DependData > Deps
LLVM_ABI Error mergeFiniBB(IRBuilderBase &Builder, BasicBlock *ExistingFiniBB)
For cases where there is an unavoidable existing finalization block (e.g.
LLVM_ABI Expected< BasicBlock * > getFiniBB(IRBuilderBase &Builder)
The basic block to which control should be transferred to implement the FiniCB.
Description of a LLVM-IR insertion point (IP) and a debug/source location (filename,...
MapNonContiguousArrayTy Offsets
MapNonContiguousArrayTy Counts
MapNonContiguousArrayTy Strides
This structure contains combined information generated for mappable clauses, including base pointers,...
MapDeviceInfoArrayTy DevicePointers
MapValuesArrayTy BasePointers
MapValuesArrayTy Pointers
StructNonContiguousInfo NonContigInfo
Helper that contains information about regions we need to outline during finalization.
void collectBlocks(SmallPtrSetImpl< BasicBlock * > &BlockSet, SmallVectorImpl< BasicBlock * > &BlockVector)
Collect all blocks in between EntryBB and ExitBB in both the given vector and set.
BasicBlock * OuterAllocBB
virtual std::unique_ptr< CodeExtractor > createCodeExtractor(ArrayRef< BasicBlock * > Blocks, bool ArgsInZeroAddressSpace, Twine Suffix=Twine(""))
Create a CodeExtractor instance based on the information stored in this structure,...
Information about an OpenMP reduction.
EvalKind EvaluationKind
Reduction evaluation kind - scalar, complex or aggregate.
ReductionGenAtomicCBTy AtomicReductionGen
Callback for generating the atomic reduction body, may be null.
ReductionGenCBTy ReductionGen
Callback for generating the reduction body.
Value * Variable
Reduction variable of pointer type.
Value * PrivateVariable
Thread-private partial reduction variable.
ReductionGenClangCBTy ReductionGenClang
Clang callback for generating the reduction body.
Type * ElementType
Reduction element type, must match pointee type of variable.
ReductionGenDataPtrPtrCBTy DataPtrPtrGen
Container for the arguments used to pass data to the runtime library.
Value * SizesArray
The array of sizes passed to the runtime library.
Value * PointersArray
The array of section pointers passed to the runtime library.
Value * MappersArray
The array of user-defined mappers passed to the runtime library.
Value * MapTypesArrayEnd
The array of map types passed to the runtime library for the end of the region, or nullptr if there a...
Value * BasePointersArray
The array of base pointer passed to the runtime library.
Value * MapTypesArray
The array of map types passed to the runtime library for the beginning of the region or for the entir...
Value * MapNamesArray
The array of original declaration names of mapped pointers sent to the runtime library for debugging.
Data structure that contains the needed information to construct the kernel args vector.
bool StrictBlocks
True if the kernel strictly requires the number of blocks and threads above to run.
ArrayRef< Value * > NumThreads
The number of threads.
TargetDataRTArgs RTArgs
Arguments passed to the runtime library.
Value * NumIterations
The number of iterations.
Value * DynCGroupMem
The size of the dynamic shared memory.
unsigned NumTargetItems
Number of arguments passed to the runtime library.
bool HasNoWait
True if the kernel has 'no wait' clause.
ArrayRef< Value * > NumTeams
The number of teams.
omp::OMPDynGroupprivateFallbackType DynCGroupMemFallback
The fallback mechanism for the shared memory.
Container to pass the default attributes with which a kernel must be launched, used to set kernel att...
omp::OMPTgtExecModeFlags ExecFlags
SmallVector< int32_t, 3 > MaxTeams
Container to pass LLVM IR runtime values or constants related to the number of teams and threads with...
Value * DeviceID
Device ID value used in the kernel launch.
SmallVector< Value *, 3 > MaxTeams
Value * LoopTripCount
Total number of iterations of the SPMD or Generic-SPMD kernel or null if it is a generic kernel.
SmallVector< Value *, 3 > TargetThreadLimit
SmallVector< Value *, 3 > TeamsThreadLimit
SmallVector< Value * > MaxThreads
'parallel' construct 'num_threads' clause value, if present and it is an SPMD kernel.
Data structure to contain the information needed to uniquely identify a target entry.
static LLVM_ABI void getTargetRegionEntryFnName(SmallVectorImpl< char > &Name, StringRef ParentName, unsigned DeviceID, unsigned FileID, unsigned Line, unsigned Count)
static constexpr const char * KernelNamePrefix
The prefix used for kernel names.
static LLVM_ABI const Target * lookupTarget(const Triple &TheTriple, std::string &Error)
lookupTarget - Lookup a target based on a target triple.
Defines various target-specific GPU grid values that must be consistent between host RTL (plugin),...