LLVM 24.0.0git
AtomicExpandPass.cpp
Go to the documentation of this file.
1//===- AtomicExpandPass.cpp - Expand atomic instructions ------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains a pass (at IR level) to replace atomic instructions with
10// __atomic_* library calls, or target specific instruction which implement the
11// same semantics in a way which better fits the target backend. This can
12// include the use of (intrinsic-based) load-linked/store-conditional loops,
13// AtomicCmpXchg, or type coercions.
14//
15//===----------------------------------------------------------------------===//
16
17#include "llvm/ADT/ArrayRef.h"
28#include "llvm/IR/Attributes.h"
29#include "llvm/IR/BasicBlock.h"
30#include "llvm/IR/Constant.h"
31#include "llvm/IR/Constants.h"
32#include "llvm/IR/DataLayout.h"
34#include "llvm/IR/Function.h"
35#include "llvm/IR/IRBuilder.h"
36#include "llvm/IR/Instruction.h"
38#include "llvm/IR/MDBuilder.h"
40#include "llvm/IR/Module.h"
42#include "llvm/IR/Type.h"
43#include "llvm/IR/User.h"
44#include "llvm/IR/Value.h"
46#include "llvm/Pass.h"
49#include "llvm/Support/Debug.h"
54#include <cassert>
55#include <cstdint>
56#include <iterator>
57
58using namespace llvm;
59
60#define DEBUG_TYPE "atomic-expand"
61
62namespace {
63
64class AtomicExpandImpl {
65 const TargetLowering *TLI = nullptr;
66 const LibcallLoweringInfo *LibcallLowering = nullptr;
67 const DataLayout *DL = nullptr;
68 bool SingleThreaded = false;
69
70private:
71 /// Callback type for emitting a cmpxchg instruction during RMW expansion.
72 /// Parameters: (Builder, Addr, Loaded, NewVal, AddrAlign, MemOpOrder,
73 /// SSID, IsVolatile, /* OUT */ Success, /* OUT */ NewLoaded,
74 /// MetadataSrc)
75 using CreateCmpXchgInstFun = function_ref<void(
77 SyncScope::ID, bool, Value *&, Value *&, Instruction *)>;
78
79 void handleFailure(Instruction &FailedInst, const Twine &Msg,
80 Instruction *DiagnosticInst = nullptr) const {
81 LLVMContext &Ctx = FailedInst.getContext();
82
83 // TODO: Do not use generic error type.
84 Ctx.emitError(DiagnosticInst ? DiagnosticInst : &FailedInst, Msg);
85
86 if (!FailedInst.getType()->isVoidTy())
87 FailedInst.replaceAllUsesWith(PoisonValue::get(FailedInst.getType()));
88 FailedInst.eraseFromParent();
89 }
90
91 template <typename Inst>
92 void handleUnsupportedAtomicSize(Inst *I, const Twine &AtomicOpName,
93 Instruction *DiagnosticInst = nullptr) const;
94
95 bool bracketInstWithFences(Instruction *I, AtomicOrdering Order);
96 bool tryInsertTrailingSeqCstFence(Instruction *AtomicI);
97 template <typename AtomicInst>
98 bool tryInsertFencesForAtomic(AtomicInst *AtomicI, bool OrderingRequiresFence,
99 AtomicOrdering NewOrdering);
100 IntegerType *getCorrespondingIntegerType(Type *T, const DataLayout &DL);
101 LoadInst *convertAtomicLoadToIntegerType(LoadInst *LI);
102 bool tryExpandAtomicLoad(LoadInst *LI);
103 bool expandAtomicLoadToLL(LoadInst *LI);
104 bool expandAtomicLoadToCmpXchg(LoadInst *LI);
105 StoreInst *convertAtomicStoreToIntegerType(StoreInst *SI);
106 bool tryExpandAtomicStore(StoreInst *SI);
107 void expandAtomicStoreToXChg(StoreInst *SI);
108 bool tryExpandAtomicRMW(AtomicRMWInst *AI);
109 void expandAtomicSubToAdd(AtomicRMWInst *AI);
110 AtomicRMWInst *convertAtomicXchgToIntegerType(AtomicRMWInst *RMWI);
111 Value *
112 insertRMWLLSCLoop(IRBuilderBase &Builder, Type *ResultTy, Value *Addr,
113 Align AddrAlign, AtomicOrdering MemOpOrder,
114 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp);
115 void expandAtomicOpToLLSC(
116 Instruction *I, Type *ResultTy, Value *Addr, Align AddrAlign,
117 AtomicOrdering MemOpOrder,
118 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp);
119 void expandPartwordAtomicRMW(
121 AtomicRMWInst *widenPartwordAtomicRMW(AtomicRMWInst *AI);
122 bool expandPartwordCmpXchg(AtomicCmpXchgInst *I);
123 void expandAtomicRMWToMaskedIntrinsic(AtomicRMWInst *AI);
124 void expandAtomicCmpXchgToMaskedIntrinsic(AtomicCmpXchgInst *CI);
125
126 AtomicCmpXchgInst *convertCmpXchgToIntegerType(AtomicCmpXchgInst *CI);
127 Value *insertRMWCmpXchgLoop(
128 IRBuilderBase &Builder, Type *ResultType, Value *Addr, Align AddrAlign,
129 AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile,
130 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp,
131 CreateCmpXchgInstFun CreateCmpXchg, Instruction *MetadataSrc);
132 bool tryExpandAtomicCmpXchg(AtomicCmpXchgInst *CI);
133
134 bool expandAtomicCmpXchg(AtomicCmpXchgInst *CI);
135 bool isIdempotentRMW(AtomicRMWInst *RMWI);
136 bool simplifyIdempotentRMW(AtomicRMWInst *RMWI);
137
138 bool expandAtomicOpToLibcall(Instruction *I, unsigned Size, Align Alignment,
139 Value *PointerOperand, Value *ValueOperand,
140 Value *CASExpected, AtomicOrdering Ordering,
141 AtomicOrdering Ordering2,
142 ArrayRef<RTLIB::Libcall> Libcalls);
143 void expandAtomicLoadToLibcall(LoadInst *LI);
144 void expandAtomicStoreToLibcall(StoreInst *LI);
145 void expandAtomicRMWToLibcall(AtomicRMWInst *I);
146 void expandAtomicCASToLibcall(AtomicCmpXchgInst *I,
147 const Twine &AtomicOpName = "cmpxchg",
148 Instruction *DiagnosticInst = nullptr);
149
150 bool expandAtomicRMWToCmpXchg(AtomicRMWInst *AI,
151 CreateCmpXchgInstFun CreateCmpXchg);
152
153 bool lowerToNonAtomic(Instruction *I);
154 bool processAtomicInstr(Instruction *I);
155
156public:
157 bool run(Function &F, const ModuleLibcallLoweringInfo &LibcallResult,
158 const TargetMachine *TM);
159};
160
161class AtomicExpandLegacy : public FunctionPass {
162public:
163 static char ID; // Pass identification, replacement for typeid
164
165 AtomicExpandLegacy() : FunctionPass(ID) {}
166
167 void getAnalysisUsage(AnalysisUsage &AU) const override {
170 }
171
172 bool runOnFunction(Function &F) override;
173};
174
175// IRBuilder to be used for replacement atomic instructions.
176struct ReplacementIRBuilder
177 : IRBuilder<InstSimplifyFolder, IRBuilderCallbackInserter> {
178 MDNode *MMRAMD = nullptr;
179 MDNode *PCSectionsMD = nullptr;
180
181 // Preserves the DebugLoc from I, and preserves still valid metadata.
182 // Enable StrictFP builder mode when appropriate.
183 explicit ReplacementIRBuilder(Instruction *I, const DataLayout &DL)
184 : IRBuilder(
185 I->getContext(), InstSimplifyFolder(DL),
186 IRBuilderCallbackInserter([this](Instruction *I) { addMD(I); })) {
187 SetInsertPoint(I);
188 if (BB->getParent()->getAttributes().hasFnAttr(Attribute::StrictFP))
189 this->setIsFPConstrained(true);
190
191 MMRAMD = I->getMetadata(LLVMContext::MD_mmra);
192 PCSectionsMD = I->getMetadata(LLVMContext::MD_pcsections);
193 }
194
195 void addMD(Instruction *I) {
197 I->setMetadata(LLVMContext::MD_mmra, MMRAMD);
198 I->setMetadata(LLVMContext::MD_pcsections, PCSectionsMD);
199 }
200};
201
202} // end anonymous namespace
203
204char AtomicExpandLegacy::ID = 0;
205
206char &llvm::AtomicExpandID = AtomicExpandLegacy::ID;
207
209 "Expand Atomic instructions", false, false)
212INITIALIZE_PASS_END(AtomicExpandLegacy, DEBUG_TYPE,
213 "Expand Atomic instructions", false, false)
214
215// Helper functions to retrieve the size of atomic instructions.
216static unsigned getAtomicOpSize(LoadInst *LI) {
217 const DataLayout &DL = LI->getDataLayout();
218 return DL.getTypeStoreSize(LI->getType());
219}
220
221static unsigned getAtomicOpSize(StoreInst *SI) {
222 const DataLayout &DL = SI->getDataLayout();
223 return DL.getTypeStoreSize(SI->getValueOperand()->getType());
224}
225
226static unsigned getAtomicOpSize(AtomicRMWInst *RMWI) {
227 const DataLayout &DL = RMWI->getDataLayout();
228 return DL.getTypeStoreSize(RMWI->getValOperand()->getType());
229}
230
231static unsigned getAtomicOpSize(AtomicCmpXchgInst *CASI) {
232 const DataLayout &DL = CASI->getDataLayout();
233 return DL.getTypeStoreSize(CASI->getCompareOperand()->getType());
234}
235
236/// Copy metadata that's safe to preserve when widening atomics.
238 const Instruction &Source) {
240 Source.getAllMetadata(MD);
241 LLVMContext &Ctx = Dest.getContext();
242 MDBuilder MDB(Ctx);
243
244 for (auto [ID, N] : MD) {
245 switch (ID) {
246 case LLVMContext::MD_dbg:
247 case LLVMContext::MD_tbaa:
248 case LLVMContext::MD_tbaa_struct:
249 case LLVMContext::MD_alias_scope:
250 case LLVMContext::MD_mem_cache_hint:
251 case LLVMContext::MD_noalias:
252 case LLVMContext::MD_noalias_addrspace:
253 case LLVMContext::MD_access_group:
254 case LLVMContext::MD_mmra:
255 Dest.setMetadata(ID, N);
256 break;
257 default:
258 if (ID == Ctx.getMDKindID("amdgpu.no.remote.memory"))
259 Dest.setMetadata(ID, N);
260 else if (ID == Ctx.getMDKindID("amdgpu.no.fine.grained.memory"))
261 Dest.setMetadata(ID, N);
262
263 // Losing atomic.ignore.denormal.mode, but it doesn't matter for current
264 // uses.
265 break;
266 }
267 }
268}
269
270template <typename Inst>
271static bool atomicSizeSupported(const TargetLowering *TLI, Inst *I) {
272 unsigned Size = getAtomicOpSize(I);
273 Align Alignment = I->getAlign();
274 unsigned MaxSize = TLI->getMaxAtomicSizeInBitsSupported() / 8;
275 return Alignment >= Size && Size <= MaxSize;
276}
277
278template <typename Inst>
280 raw_ostream &OS) {
281 unsigned Size = getAtomicOpSize(I);
282 Align Alignment = I->getAlign();
283 bool NeedSeparator = false;
284
285 if (Alignment < Size) {
286 OS << "instruction alignment " << Alignment.value()
287 << " is smaller than the required " << Size
288 << "-byte alignment for this atomic operation";
289 NeedSeparator = true;
290 }
291
292 unsigned MaxSize = TLI->getMaxAtomicSizeInBitsSupported() / 8;
293 if (Size > MaxSize) {
294 if (NeedSeparator)
295 OS << "; ";
296 OS << "target supports atomics up to " << MaxSize
297 << " bytes, but this atomic accesses " << Size << " bytes";
298 }
299}
300
301template <typename Inst>
302void AtomicExpandImpl::handleUnsupportedAtomicSize(
303 Inst *I, const Twine &AtomicOpName, Instruction *DiagnosticInst) const {
304 assert(!atomicSizeSupported(TLI, I) && "expected unsupported atomic size");
305 SmallString<128> FailureReason;
306 raw_svector_ostream OS(FailureReason);
308 handleFailure(*I, Twine("unsupported ") + AtomicOpName + ": " + FailureReason,
309 DiagnosticInst);
310}
311
312bool AtomicExpandImpl::tryInsertTrailingSeqCstFence(Instruction *AtomicI) {
314 return false;
315
316 IRBuilder Builder(AtomicI);
317 if (auto *TrailingFence = TLI->emitTrailingFence(
318 Builder, AtomicI, AtomicOrdering::SequentiallyConsistent)) {
319 TrailingFence->moveAfter(AtomicI);
320 return true;
321 }
322 return false;
323}
324
325template <typename AtomicInst>
326bool AtomicExpandImpl::tryInsertFencesForAtomic(AtomicInst *AtomicI,
327 bool OrderingRequiresFence,
328 AtomicOrdering NewOrdering) {
329 bool ShouldInsertFences = TLI->shouldInsertFencesForAtomic(AtomicI);
330 if (OrderingRequiresFence && ShouldInsertFences) {
331 AtomicOrdering FenceOrdering = AtomicI->getOrdering();
332 AtomicI->setOrdering(NewOrdering);
333 return bracketInstWithFences(AtomicI, FenceOrdering);
334 }
335 if (!ShouldInsertFences)
336 return tryInsertTrailingSeqCstFence(AtomicI);
337 return false;
338}
339
340/// In a single-threaded environment, atomic operations can be lowered to their
341/// non-atomic equivalents: fences are removed, and atomic loads, stores, RMW,
342/// and cmpxchg become plain memory operations.
343bool AtomicExpandImpl::lowerToNonAtomic(Instruction *I) {
344 if (auto *FI = dyn_cast<FenceInst>(I)) {
345 FI->eraseFromParent();
346 return true;
347 }
348
349 if (auto *CXI = dyn_cast<AtomicCmpXchgInst>(I))
350 return lowerAtomicCmpXchgInst(CXI);
351
352 if (auto *RMWI = dyn_cast<AtomicRMWInst>(I))
353 return lowerAtomicRMWInst(RMWI);
354
355 if (auto *LI = dyn_cast<LoadInst>(I)) {
356 if (LI->isAtomic()) {
357 LI->setAtomic(AtomicOrdering::NotAtomic);
358 LI->setElementwise(false);
359 return true;
360 }
361
362 return false;
363 }
364
365 if (auto *SI = dyn_cast<StoreInst>(I)) {
366 if (SI->isAtomic()) {
367 SI->setAtomic(AtomicOrdering::NotAtomic);
368 SI->setElementwise(false);
369 return true;
370 }
371
372 return false;
373 }
374
375 return false;
376}
377
378bool AtomicExpandImpl::processAtomicInstr(Instruction *I) {
379 if (SingleThreaded)
380 return lowerToNonAtomic(I);
381
382 if (auto *LI = dyn_cast<LoadInst>(I)) {
383 if (!LI->isAtomic())
384 return false;
385
386 if (!atomicSizeSupported(TLI, LI)) {
387 expandAtomicLoadToLibcall(LI);
388 return true;
389 }
390
391 bool MadeChange = false;
392 if (TLI->shouldCastAtomicLoadInIR(LI) ==
393 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
394 LI = convertAtomicLoadToIntegerType(LI);
395 MadeChange = true;
396 }
397
398 MadeChange |= tryInsertFencesForAtomic(
399 LI, isAcquireOrStronger(LI->getOrdering()), AtomicOrdering::Monotonic);
400
401 MadeChange |= tryExpandAtomicLoad(LI);
402 return MadeChange;
403 }
404
405 if (auto *SI = dyn_cast<StoreInst>(I)) {
406 if (!SI->isAtomic())
407 return false;
408
409 if (!atomicSizeSupported(TLI, SI)) {
410 expandAtomicStoreToLibcall(SI);
411 return true;
412 }
413
414 bool MadeChange = false;
415 if (TLI->shouldCastAtomicStoreInIR(SI) ==
416 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
417 SI = convertAtomicStoreToIntegerType(SI);
418 MadeChange = true;
419 }
420
421 MadeChange |= tryInsertFencesForAtomic(
422 SI, isReleaseOrStronger(SI->getOrdering()), AtomicOrdering::Monotonic);
423
424 MadeChange |= tryExpandAtomicStore(SI);
425 return MadeChange;
426 }
427
428 if (auto *RMWI = dyn_cast<AtomicRMWInst>(I)) {
429 if (!atomicSizeSupported(TLI, RMWI)) {
430 expandAtomicRMWToLibcall(RMWI);
431 return true;
432 }
433
434 bool MadeChange = false;
435 if (TLI->shouldCastAtomicRMWIInIR(RMWI) ==
436 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
437 RMWI = convertAtomicXchgToIntegerType(RMWI);
438 MadeChange = true;
439 }
440
441 MadeChange |= tryInsertFencesForAtomic(
442 RMWI,
443 isReleaseOrStronger(RMWI->getOrdering()) ||
444 isAcquireOrStronger(RMWI->getOrdering()),
446
447 // There are two different ways of expanding RMW instructions:
448 // - into a load if it is idempotent
449 // - into a Cmpxchg/LL-SC loop otherwise
450 // we try them in that order.
451 MadeChange |= (isIdempotentRMW(RMWI) && simplifyIdempotentRMW(RMWI)) ||
452 tryExpandAtomicRMW(RMWI);
453 return MadeChange;
454 }
455
456 if (auto *CASI = dyn_cast<AtomicCmpXchgInst>(I)) {
457 if (!atomicSizeSupported(TLI, CASI)) {
458 expandAtomicCASToLibcall(CASI);
459 return true;
460 }
461
462 // TODO: when we're ready to make the change at the IR level, we can
463 // extend convertCmpXchgToInteger for floating point too.
464 bool MadeChange = false;
465 if (CASI->getCompareOperand()->getType()->isPointerTy()) {
466 // TODO: add a TLI hook to control this so that each target can
467 // convert to lowering the original type one at a time.
468 CASI = convertCmpXchgToIntegerType(CASI);
469 MadeChange = true;
470 }
471
472 auto CmpXchgExpansion = TLI->shouldExpandAtomicCmpXchgInIR(CASI);
473 if (TLI->shouldInsertFencesForAtomic(CASI)) {
474 if (CmpXchgExpansion == TargetLoweringBase::AtomicExpansionKind::None &&
475 (isReleaseOrStronger(CASI->getSuccessOrdering()) ||
476 isAcquireOrStronger(CASI->getSuccessOrdering()) ||
477 isAcquireOrStronger(CASI->getFailureOrdering()))) {
478 // If a compare and swap is lowered to LL/SC, we can do smarter fence
479 // insertion, with a stronger one on the success path than on the
480 // failure path. As a result, fence insertion is directly done by
481 // expandAtomicCmpXchg in that case.
482 AtomicOrdering FenceOrdering = CASI->getMergedOrdering();
483 AtomicOrdering CASOrdering =
485 CASI->setSuccessOrdering(CASOrdering);
486 CASI->setFailureOrdering(CASOrdering);
487 MadeChange |= bracketInstWithFences(CASI, FenceOrdering);
488 }
489 } else if (CmpXchgExpansion !=
490 TargetLoweringBase::AtomicExpansionKind::LLSC) {
491 // CmpXchg LLSC is handled in expandAtomicCmpXchg().
492 MadeChange |= tryInsertTrailingSeqCstFence(CASI);
493 }
494
495 MadeChange |= tryExpandAtomicCmpXchg(CASI);
496 return MadeChange;
497 }
498
499 return false;
500}
501
502bool AtomicExpandImpl::run(Function &F,
503 const ModuleLibcallLoweringInfo &LibcallResult,
504 const TargetMachine *TM) {
505 SingleThreaded = F.getParent()->getThreadModel() == ThreadModel::Single;
506
507 const auto *Subtarget = TM->getSubtargetImpl(F);
508 // In a single-threaded environment atomics are lowered to non-atomic form
509 if (!SingleThreaded && !Subtarget->enableAtomicExpand())
510 return false;
511 TLI = Subtarget->getTargetLowering();
512 LibcallLowering = &getLibcallLowering(LibcallResult, *Subtarget);
513 DL = &F.getDataLayout();
514
515 bool MadeChange = false;
516
517 for (Function::iterator BBI = F.begin(), BBE = F.end(); BBI != BBE; ++BBI) {
518 BasicBlock *BB = &*BBI;
519
521
522 for (BasicBlock::reverse_iterator I = BB->rbegin(), E = BB->rend(); I != E;
523 I = Next) {
524 Instruction &Inst = *I;
525 Next = std::next(I);
526
527 if (processAtomicInstr(&Inst)) {
528 MadeChange = true;
529
530 // New blocks may have been inserted.
531 BBE = F.end();
532 }
533 }
534 }
535
536 return MadeChange;
537}
538
539bool AtomicExpandLegacy::runOnFunction(Function &F) {
540
541 auto *TPC = getAnalysisIfAvailable<TargetPassConfig>();
542 if (!TPC)
543 return false;
544 auto *TM = &TPC->getTM<TargetMachine>();
545
546 const ModuleLibcallLoweringInfo &LibcallResult =
547 getAnalysis<LibcallLoweringInfoWrapper>().getResult(*F.getParent());
548 AtomicExpandImpl AE;
549 return AE.run(F, LibcallResult, TM);
550}
551
553 return new AtomicExpandLegacy();
554}
555
558 auto &MAMProxy = FAM.getResult<ModuleAnalysisManagerFunctionProxy>(F);
559
560 const ModuleLibcallLoweringInfo *LibcallResult =
561 MAMProxy.getCachedResult<LibcallLoweringModuleAnalysis>(*F.getParent());
562
563 if (!LibcallResult) {
564 F.getContext().emitError("'" + LibcallLoweringModuleAnalysis::name() +
565 "' analysis required");
566 return PreservedAnalyses::all();
567 }
568
569 AtomicExpandImpl AE;
570
571 bool Changed = AE.run(F, *LibcallResult, TM);
572 if (!Changed)
573 return PreservedAnalyses::all();
574
576}
577
578bool AtomicExpandImpl::bracketInstWithFences(Instruction *I,
579 AtomicOrdering Order) {
580 ReplacementIRBuilder Builder(I, *DL);
581
582 auto LeadingFence = TLI->emitLeadingFence(Builder, I, Order);
583
584 auto TrailingFence = TLI->emitTrailingFence(Builder, I, Order);
585 // We have a guard here because not every atomic operation generates a
586 // trailing fence.
587 if (TrailingFence)
588 TrailingFence->moveAfter(I);
589
590 return (LeadingFence || TrailingFence);
591}
592
593/// Get the iX type with the same bitwidth as T.
595AtomicExpandImpl::getCorrespondingIntegerType(Type *T, const DataLayout &DL) {
596 EVT VT = TLI->getMemValueType(DL, T);
597 unsigned BitWidth = VT.getStoreSizeInBits();
598 assert(BitWidth == VT.getSizeInBits() && "must be a power of two");
599 return IntegerType::get(T->getContext(), BitWidth);
600}
601
602/// Convert an atomic load of a non-integral type to an integer load of the
603/// equivalent bitwidth. See the function comment on
604/// convertAtomicStoreToIntegerType for background.
605LoadInst *AtomicExpandImpl::convertAtomicLoadToIntegerType(LoadInst *LI) {
606 auto *M = LI->getModule();
607 Type *NewTy = getCorrespondingIntegerType(LI->getType(), M->getDataLayout());
608
609 ReplacementIRBuilder Builder(LI, *DL);
610
611 Value *Addr = LI->getPointerOperand();
612
613 auto *NewLI = Builder.CreateLoad(NewTy, Addr, LI->getProperties());
614 LLVM_DEBUG(dbgs() << "Replaced " << *LI << " with " << *NewLI << "\n");
615
616 Value *NewVal = LI->getType()->isPtrOrPtrVectorTy()
617 ? Builder.CreateIntToPtr(NewLI, LI->getType())
618 : Builder.CreateBitCast(NewLI, LI->getType());
619 LI->replaceAllUsesWith(NewVal);
620 LI->eraseFromParent();
621 return NewLI;
622}
623
624AtomicRMWInst *
625AtomicExpandImpl::convertAtomicXchgToIntegerType(AtomicRMWInst *RMWI) {
627
628 auto *M = RMWI->getModule();
629 Type *NewTy =
630 getCorrespondingIntegerType(RMWI->getType(), M->getDataLayout());
631
632 ReplacementIRBuilder Builder(RMWI, *DL);
633
634 Value *Addr = RMWI->getPointerOperand();
635 Value *Val = RMWI->getValOperand();
636 Value *NewVal = Builder.CreateBitPreservingCastChain(*DL, Val, NewTy);
637
638 auto *NewRMWI = Builder.CreateAtomicRMW(AtomicRMWInst::Xchg, Addr, NewVal,
639 RMWI->getAlign(), RMWI->getOrdering(),
640 RMWI->getSyncScopeID());
641 NewRMWI->setVolatile(RMWI->isVolatile());
642 copyMetadataForAtomic(*NewRMWI, *RMWI);
643 LLVM_DEBUG(dbgs() << "Replaced " << *RMWI << " with " << *NewRMWI << "\n");
644
645 Value *NewRVal =
646 Builder.CreateBitPreservingCastChain(*DL, NewRMWI, RMWI->getType());
647 RMWI->replaceAllUsesWith(NewRVal);
648 RMWI->eraseFromParent();
649 return NewRMWI;
650}
651
652bool AtomicExpandImpl::tryExpandAtomicLoad(LoadInst *LI) {
653 switch (TLI->shouldExpandAtomicLoadInIR(LI)) {
654 case TargetLoweringBase::AtomicExpansionKind::None:
655 return false;
656 case TargetLoweringBase::AtomicExpansionKind::LLSC:
657 expandAtomicOpToLLSC(
658 LI, LI->getType(), LI->getPointerOperand(), LI->getAlign(),
659 LI->getOrdering(),
660 [](IRBuilderBase &Builder, Value *Loaded) { return Loaded; });
661 return true;
662 case TargetLoweringBase::AtomicExpansionKind::LLOnly:
663 return expandAtomicLoadToLL(LI);
664 case TargetLoweringBase::AtomicExpansionKind::CmpXChg:
665 return expandAtomicLoadToCmpXchg(LI);
666 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
667 LI->setAtomic(AtomicOrdering::NotAtomic);
668 return true;
669 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
670 TLI->emitExpandAtomicLoad(LI);
671 return true;
672 default:
673 llvm_unreachable("Unhandled case in tryExpandAtomicLoad");
674 }
675}
676
677bool AtomicExpandImpl::tryExpandAtomicStore(StoreInst *SI) {
678 switch (TLI->shouldExpandAtomicStoreInIR(SI)) {
679 case TargetLoweringBase::AtomicExpansionKind::None:
680 return false;
681 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
682 TLI->emitExpandAtomicStore(SI);
683 return true;
684 case TargetLoweringBase::AtomicExpansionKind::Expand:
685 expandAtomicStoreToXChg(SI);
686 return true;
687 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
688 SI->setAtomic(AtomicOrdering::NotAtomic);
689 return true;
690 default:
691 llvm_unreachable("Unhandled case in tryExpandAtomicStore");
692 }
693}
694
695bool AtomicExpandImpl::expandAtomicLoadToLL(LoadInst *LI) {
696 ReplacementIRBuilder Builder(LI, *DL);
697
698 // On some architectures, load-linked instructions are atomic for larger
699 // sizes than normal loads. For example, the only 64-bit load guaranteed
700 // to be single-copy atomic by ARM is an ldrexd (A3.5.3).
701 Value *Val = TLI->emitLoadLinked(Builder, LI->getType(),
702 LI->getPointerOperand(), LI->getOrdering());
704
705 LI->replaceAllUsesWith(Val);
706 LI->eraseFromParent();
707
708 return true;
709}
710
711bool AtomicExpandImpl::expandAtomicLoadToCmpXchg(LoadInst *LI) {
712 ReplacementIRBuilder Builder(LI, *DL);
713 AtomicOrdering Order = LI->getOrdering();
714 if (Order == AtomicOrdering::Unordered)
715 Order = AtomicOrdering::Monotonic;
716
717 Value *Addr = LI->getPointerOperand();
718 Type *Ty = LI->getType();
719
720 // cmpxchg supports only integer and pointer operands. If the load type is
721 // FP or vector, run the cmpxchg on the same-sized integer and bitcast the
722 // result back; mirrors createCmpXchgInstFun.
723 bool NeedBitcast = Ty->isFloatingPointTy() || Ty->isVectorTy();
724 Type *CmpXchgTy = Ty;
725 if (NeedBitcast)
726 CmpXchgTy = Builder.getIntNTy(Ty->getPrimitiveSizeInBits());
727 Constant *DummyVal = Constant::getNullValue(CmpXchgTy);
728
729 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
730 Addr, DummyVal, DummyVal, LI->getAlign(), Order,
732 LI->getSyncScopeID());
733 Pair->setVolatile(LI->isVolatile());
734 Value *Loaded = Builder.CreateExtractValue(Pair, 0, "loaded");
735 if (NeedBitcast)
736 Loaded = Builder.CreateBitCast(Loaded, Ty);
737
738 LI->replaceAllUsesWith(Loaded);
739 LI->eraseFromParent();
740
741 return true;
742}
743
744/// Convert an atomic store of a non-integral type to an integer store of the
745/// equivalent bitwidth. We used to not support floating point or vector
746/// atomics in the IR at all. The backends learned to deal with the bitcast
747/// idiom because that was the only way of expressing the notion of a atomic
748/// float or vector store. The long term plan is to teach each backend to
749/// instruction select from the original atomic store, but as a migration
750/// mechanism, we convert back to the old format which the backends understand.
751/// Each backend will need individual work to recognize the new format.
752StoreInst *AtomicExpandImpl::convertAtomicStoreToIntegerType(StoreInst *SI) {
753 ReplacementIRBuilder Builder(SI, *DL);
754 auto *M = SI->getModule();
755 Type *NewTy = getCorrespondingIntegerType(SI->getValueOperand()->getType(),
756 M->getDataLayout());
757 Value *NewVal = SI->getValueOperand()->getType()->isPtrOrPtrVectorTy()
758 ? Builder.CreatePtrToInt(SI->getValueOperand(), NewTy)
759 : Builder.CreateBitCast(SI->getValueOperand(), NewTy);
760
761 Value *Addr = SI->getPointerOperand();
762
763 StoreInst *NewSI = Builder.CreateStore(NewVal, Addr, SI->getProperties());
764 copyMetadataForAtomic(*NewSI, *SI);
765 LLVM_DEBUG(dbgs() << "Replaced " << *SI << " with " << *NewSI << "\n");
766 SI->eraseFromParent();
767 return NewSI;
768}
769
770void AtomicExpandImpl::expandAtomicStoreToXChg(StoreInst *SI) {
771 // This function is only called on atomic stores that are too large to be
772 // atomic if implemented as a native store. So we replace them by an
773 // atomic swap, that can be implemented for example as a ldrex/strex on ARM
774 // or lock cmpxchg8/16b on X86, as these are atomic for larger sizes.
775 // It is the responsibility of the target to only signal expansion via
776 // shouldExpandAtomicRMW in cases where this is required and possible.
777 ReplacementIRBuilder Builder(SI, *DL);
778 AtomicOrdering Ordering = SI->getOrdering();
779 assert(Ordering != AtomicOrdering::NotAtomic);
780 AtomicOrdering RMWOrdering = Ordering == AtomicOrdering::Unordered
781 ? AtomicOrdering::Monotonic
782 : Ordering;
783 AtomicRMWInst *AI = Builder.CreateAtomicRMW(
784 AtomicRMWInst::Xchg, SI->getPointerOperand(), SI->getValueOperand(),
785 SI->getAlign(), RMWOrdering, SI->getSyncScopeID());
786 AI->setVolatile(SI->isVolatile());
787 SI->eraseFromParent();
788
789 // Now we have an appropriate swap instruction, lower it as usual.
790 tryExpandAtomicRMW(AI);
791}
792
793static void createCmpXchgInstFun(IRBuilderBase &Builder, Value *Addr,
794 Value *Loaded, Value *NewVal, Align AddrAlign,
795 AtomicOrdering MemOpOrder, SyncScope::ID SSID,
796 bool IsVolatile, Value *&Success,
797 Value *&NewLoaded, Instruction *MetadataSrc) {
798 Type *OrigTy = NewVal->getType();
799
800 // This code can go away when cmpxchg supports FP and vector types.
801 assert(!OrigTy->isPointerTy());
802 bool NeedBitcast = OrigTy->isFloatingPointTy() || OrigTy->isVectorTy();
803 if (NeedBitcast) {
804 IntegerType *IntTy = Builder.getIntNTy(OrigTy->getPrimitiveSizeInBits());
805 NewVal = Builder.CreateBitCast(NewVal, IntTy);
806 Loaded = Builder.CreateBitCast(Loaded, IntTy);
807 }
808
809 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
810 Addr, Loaded, NewVal, AddrAlign, MemOpOrder,
812 Pair->setVolatile(IsVolatile);
813 if (MetadataSrc)
814 copyMetadataForAtomic(*Pair, *MetadataSrc);
815
816 Success = Builder.CreateExtractValue(Pair, 1, "success");
817 NewLoaded = Builder.CreateExtractValue(Pair, 0, "newloaded");
818
819 if (NeedBitcast)
820 NewLoaded = Builder.CreateBitCast(NewLoaded, OrigTy);
821}
822
823void AtomicExpandImpl::expandAtomicSubToAdd(AtomicRMWInst *AI) {
824 ReplacementIRBuilder Builder(AI, *DL);
826 Value *NewVal;
827
828 switch (AI->getOperation()) {
830 NewOp = AtomicRMWInst::Add;
831 NewVal = Builder.CreateNeg(AI->getValOperand(), "neg");
832 break;
834 NewOp = AtomicRMWInst::FAdd;
835 NewVal = Builder.CreateFNeg(AI->getValOperand(), "fneg");
836 break;
837 default:
838 llvm_unreachable("unsupported atomicrmw expansion");
839 }
840
841 AI->setOperation(NewOp);
842 AI->setOperand(1, NewVal);
843}
844
845bool AtomicExpandImpl::tryExpandAtomicRMW(AtomicRMWInst *AI) {
846 LLVMContext &Ctx = AI->getModule()->getContext();
847 TargetLowering::AtomicExpansionKind Kind = TLI->shouldExpandAtomicRMWInIR(AI);
848 switch (Kind) {
849 case TargetLoweringBase::AtomicExpansionKind::None:
850 return false;
851 case TargetLoweringBase::AtomicExpansionKind::LLSC: {
852 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
853 unsigned ValueSize = getAtomicOpSize(AI);
854 if (ValueSize < MinCASSize) {
855 expandPartwordAtomicRMW(AI,
856 TargetLoweringBase::AtomicExpansionKind::LLSC);
857 } else {
858 auto PerformOp = [&](IRBuilderBase &Builder, Value *Loaded) {
859 return buildAtomicRMWValue(AI->getOperation(), Builder, Loaded,
860 AI->getValOperand());
861 };
862 expandAtomicOpToLLSC(AI, AI->getType(), AI->getPointerOperand(),
863 AI->getAlign(), AI->getOrdering(), PerformOp);
864 }
865 return true;
866 }
867 case TargetLoweringBase::AtomicExpansionKind::CmpXChg: {
868 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
869 unsigned ValueSize = getAtomicOpSize(AI);
870 if (ValueSize < MinCASSize) {
871 expandPartwordAtomicRMW(AI,
872 TargetLoweringBase::AtomicExpansionKind::CmpXChg);
873 } else {
875 Ctx.getSyncScopeNames(SSNs);
876 auto MemScope = SSNs[AI->getSyncScopeID()].empty()
877 ? "system"
878 : SSNs[AI->getSyncScopeID()];
879 OptimizationRemarkEmitter ORE(AI->getFunction());
880 ORE.emit([&]() {
881 return OptimizationRemark(DEBUG_TYPE, "Passed", AI)
882 << "A compare and swap loop was generated for an atomic "
883 << AI->getOperationName(AI->getOperation()) << " operation at "
884 << MemScope << " memory scope";
885 });
886 expandAtomicRMWToCmpXchg(AI, createCmpXchgInstFun);
887 }
888 return true;
889 }
890 case TargetLoweringBase::AtomicExpansionKind::MaskedIntrinsic: {
891 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
892 unsigned ValueSize = getAtomicOpSize(AI);
893 if (ValueSize < MinCASSize) {
895 // Widen And/Or/Xor and give the target another chance at expanding it.
898 tryExpandAtomicRMW(widenPartwordAtomicRMW(AI));
899 return true;
900 }
901 }
902 expandAtomicRMWToMaskedIntrinsic(AI);
903 return true;
904 }
905 case TargetLoweringBase::AtomicExpansionKind::BitTestIntrinsic: {
907 return true;
908 }
909 case TargetLoweringBase::AtomicExpansionKind::CmpArithIntrinsic: {
911 return true;
912 }
913 case TargetLoweringBase::AtomicExpansionKind::Expand:
914 expandAtomicSubToAdd(AI);
915 return true;
916 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
917 return lowerAtomicRMWInst(AI);
918 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
919 TLI->emitExpandAtomicRMW(AI);
920 return true;
921 default:
922 llvm_unreachable("Unhandled case in tryExpandAtomicRMW");
923 }
924}
925
926namespace {
927
928struct PartwordMaskValues {
929 // These three fields are guaranteed to be set by createMaskInstrs.
930 Type *WordType = nullptr;
931 Type *ValueType = nullptr;
932 Type *IntValueType = nullptr;
933 Value *AlignedAddr = nullptr;
934 Align AlignedAddrAlignment;
935 // The remaining fields can be null.
936 Value *ShiftAmt = nullptr;
937 Value *Mask = nullptr;
938 Value *Inv_Mask = nullptr;
939};
940
941[[maybe_unused]]
942raw_ostream &operator<<(raw_ostream &O, const PartwordMaskValues &PMV) {
943 auto PrintObj = [&O](auto *V) {
944 if (V)
945 O << *V;
946 else
947 O << "nullptr";
948 O << '\n';
949 };
950 O << "PartwordMaskValues {\n";
951 O << " WordType: ";
952 PrintObj(PMV.WordType);
953 O << " ValueType: ";
954 PrintObj(PMV.ValueType);
955 O << " AlignedAddr: ";
956 PrintObj(PMV.AlignedAddr);
957 O << " AlignedAddrAlignment: " << PMV.AlignedAddrAlignment.value() << '\n';
958 O << " ShiftAmt: ";
959 PrintObj(PMV.ShiftAmt);
960 O << " Mask: ";
961 PrintObj(PMV.Mask);
962 O << " Inv_Mask: ";
963 PrintObj(PMV.Inv_Mask);
964 O << "}\n";
965 return O;
966}
967
968} // end anonymous namespace
969
970/// This is a helper function which builds instructions to provide
971/// values necessary for partword atomic operations. It takes an
972/// incoming address, Addr, and ValueType, and constructs the address,
973/// shift-amounts and masks needed to work with a larger value of size
974/// WordSize.
975///
976/// AlignedAddr: Addr rounded down to a multiple of WordSize
977///
978/// ShiftAmt: Number of bits to right-shift a WordSize value loaded
979/// from AlignAddr for it to have the same value as if
980/// ValueType was loaded from Addr.
981///
982/// Mask: Value to mask with the value loaded from AlignAddr to
983/// include only the part that would've been loaded from Addr.
984///
985/// Inv_Mask: The inverse of Mask.
986static PartwordMaskValues createMaskInstrs(IRBuilderBase &Builder,
988 Value *Addr, Align AddrAlign,
989 unsigned MinWordSize) {
990 PartwordMaskValues PMV;
991
992 Module *M = I->getModule();
993 LLVMContext &Ctx = M->getContext();
994 const DataLayout &DL = M->getDataLayout();
995 unsigned ValueSize = DL.getTypeStoreSize(ValueType);
996
997 PMV.ValueType = PMV.IntValueType = ValueType;
998 if (PMV.ValueType->isFloatingPointTy() || PMV.ValueType->isVectorTy())
999 PMV.IntValueType =
1000 Type::getIntNTy(Ctx, ValueType->getPrimitiveSizeInBits());
1001
1002 PMV.WordType = MinWordSize > ValueSize ? Type::getIntNTy(Ctx, MinWordSize * 8)
1003 : ValueType;
1004 if (PMV.ValueType == PMV.WordType) {
1005 PMV.AlignedAddr = Addr;
1006 PMV.AlignedAddrAlignment = AddrAlign;
1007 PMV.ShiftAmt = ConstantInt::get(PMV.ValueType, 0);
1008 PMV.Mask = ConstantInt::get(PMV.ValueType, ~0, /*isSigned*/ true);
1009 return PMV;
1010 }
1011
1012 PMV.AlignedAddrAlignment = Align(MinWordSize);
1013
1014 assert(ValueSize < MinWordSize);
1015
1016 PointerType *PtrTy = cast<PointerType>(Addr->getType());
1017 IntegerType *IntTy = DL.getIndexType(Ctx, PtrTy->getAddressSpace());
1018 Value *PtrLSB;
1019
1020 if (AddrAlign < MinWordSize) {
1021 PMV.AlignedAddr = Builder.CreateIntrinsic(
1022 Intrinsic::ptrmask, {PtrTy, IntTy},
1023 {Addr, ConstantInt::getSigned(IntTy, ~(uint64_t)(MinWordSize - 1))},
1024 nullptr, "AlignedAddr");
1025
1026 Value *AddrInt = Builder.CreatePtrToInt(Addr, IntTy);
1027 PtrLSB = Builder.CreateAnd(AddrInt, MinWordSize - 1, "PtrLSB");
1028 } else {
1029 // If the alignment is high enough, the LSB are known 0.
1030 PMV.AlignedAddr = Addr;
1031 PtrLSB = ConstantInt::getNullValue(IntTy);
1032 }
1033
1034 if (DL.isLittleEndian()) {
1035 // turn bytes into bits
1036 PMV.ShiftAmt = Builder.CreateShl(PtrLSB, 3);
1037 } else {
1038 // turn bytes into bits, and count from the other side.
1039 PMV.ShiftAmt = Builder.CreateShl(
1040 Builder.CreateXor(PtrLSB, MinWordSize - ValueSize), 3);
1041 }
1042
1043 PMV.ShiftAmt = Builder.CreateTrunc(PMV.ShiftAmt, PMV.WordType, "ShiftAmt");
1044 PMV.Mask = Builder.CreateShl(
1045 ConstantInt::get(PMV.WordType, (1 << (ValueSize * 8)) - 1), PMV.ShiftAmt,
1046 "Mask");
1047
1048 PMV.Inv_Mask = Builder.CreateNot(PMV.Mask, "Inv_Mask");
1049
1050 return PMV;
1051}
1052
1053static Value *extractMaskedValue(IRBuilderBase &Builder, Value *WideWord,
1054 const PartwordMaskValues &PMV) {
1055 assert(WideWord->getType() == PMV.WordType && "Widened type mismatch");
1056 if (PMV.WordType == PMV.ValueType)
1057 return WideWord;
1058
1059 Value *Shift = Builder.CreateLShr(WideWord, PMV.ShiftAmt, "shifted");
1060 Value *Trunc = Builder.CreateTrunc(Shift, PMV.IntValueType, "extracted");
1061 return Builder.CreateBitCast(Trunc, PMV.ValueType);
1062}
1063
1064static Value *insertMaskedValue(IRBuilderBase &Builder, Value *WideWord,
1065 Value *Updated, const PartwordMaskValues &PMV) {
1066 assert(WideWord->getType() == PMV.WordType && "Widened type mismatch");
1067 assert(Updated->getType() == PMV.ValueType && "Value type mismatch");
1068 if (PMV.WordType == PMV.ValueType)
1069 return Updated;
1070
1071 Updated = Builder.CreateBitCast(Updated, PMV.IntValueType);
1072
1073 Value *ZExt = Builder.CreateZExt(Updated, PMV.WordType, "extended");
1074 Value *Shift =
1075 Builder.CreateShl(ZExt, PMV.ShiftAmt, "shifted", /*HasNUW*/ true);
1076 Value *And = Builder.CreateAnd(WideWord, PMV.Inv_Mask, "unmasked");
1077 Value *Or = Builder.CreateOr(And, Shift, "inserted");
1078 return Or;
1079}
1080
1081/// Emit IR to implement a masked version of a given atomicrmw
1082/// operation. (That is, only the bits under the Mask should be
1083/// affected by the operation)
1085 IRBuilderBase &Builder, Value *Loaded,
1086 Value *ValOperand_Shifted, Value *Inc,
1087 const PartwordMaskValues &PMV) {
1088 // TODO: update to use
1089 // https://graphics.stanford.edu/~seander/bithacks.html#MaskedMerge in order
1090 // to merge bits from two values without requiring PMV.Inv_Mask.
1091
1094 "Or/Xor/And handled by widenPartwordAtomicRMW");
1095
1096 if (Op == AtomicRMWInst::Xchg) {
1097 // Clear all the bits we are exchanging out. These are the bits under the
1098 // mask. We can clear them with an `and` of the inverse mask.
1099 Value *Loaded_MaskOut = Builder.CreateAnd(Loaded, PMV.Inv_Mask);
1100 // Now that the prevous bits are cleared, we can swap in the new value with
1101 // an `or`.
1102 Value *FinalVal = Builder.CreateOr(Loaded_MaskOut, ValOperand_Shifted);
1103 return FinalVal;
1104 }
1105
1106 if (Op == AtomicRMWInst::Nand ||
1107 (!PMV.ValueType->isVectorTy() &&
1109 // For `Nand` and non-vector `Add` and `Sub`, we can perform the operation
1110 // on the entire word because the extra bits in the unmasked region don't
1111 // affect the computation in the masked region. The operation might still
1112 // overwrite the unmasked region (e.g. from integer overflow or underflow),
1113 // so we have to reapply the unmasked region afterwards.
1114 //
1115 // This trick doesn't work for vector `Add` and `Sub` because we use a
1116 // scalar operation on the entire word. Scalarizing vector `Add` and `Sub`
1117 // isn't legal because the vector versions may have element-wise overflows.
1118 // TODO: For these, can we use a wider vector op with additional lanes?
1119
1120 // Atomic operation across the entire word.
1121 Value *NewVal =
1122 buildAtomicRMWValue(Op, Builder, Loaded, ValOperand_Shifted);
1123 // Reapply the bits in the unmasked region.
1124 Value *NewVal_Masked = Builder.CreateAnd(NewVal, PMV.Mask);
1125 Value *Loaded_MaskOut = Builder.CreateAnd(Loaded, PMV.Inv_Mask);
1126 Value *FinalVal = Builder.CreateOr(Loaded_MaskOut, NewVal_Masked);
1127 return FinalVal;
1128 }
1129
1130 // All other ops operate on the sub-word size. Truncate down to the
1131 // original size, and expand out again after doing the operation. Bitcasts
1132 // will be inserted for FP values.
1133 assert(!ValOperand_Shifted);
1134 Value *Loaded_Extract = extractMaskedValue(Builder, Loaded, PMV);
1135 Value *NewVal = buildAtomicRMWValue(Op, Builder, Loaded_Extract, Inc);
1136 Value *FinalVal = insertMaskedValue(Builder, Loaded, NewVal, PMV);
1137 return FinalVal;
1138}
1139
1140/// Expand a sub-word atomicrmw operation into an appropriate
1141/// word-sized operation.
1142///
1143/// It will create an LL/SC or cmpxchg loop, as appropriate, the same
1144/// way as a typical atomicrmw expansion. The only difference here is
1145/// that the operation inside of the loop may operate upon only a
1146/// part of the value.
1147void AtomicExpandImpl::expandPartwordAtomicRMW(
1148 AtomicRMWInst *AI, TargetLoweringBase::AtomicExpansionKind ExpansionKind) {
1149 // Widen And/Or/Xor and give the target another chance at expanding it.
1153 tryExpandAtomicRMW(widenPartwordAtomicRMW(AI));
1154 return;
1155 }
1156 AtomicOrdering MemOpOrder = AI->getOrdering();
1157 SyncScope::ID SSID = AI->getSyncScopeID();
1158
1159 ReplacementIRBuilder Builder(AI, *DL);
1160
1161 PartwordMaskValues PMV =
1162 createMaskInstrs(Builder, AI, AI->getType(), AI->getPointerOperand(),
1163 AI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1164
1165 Value *ValOperand_Shifted = nullptr;
1166 bool NeedsShiftedOperand =
1168 (!PMV.ValueType->isVectorTy() &&
1170
1171 if (NeedsShiftedOperand) {
1172 Value *ValOp = Builder.CreateBitCast(AI->getValOperand(), PMV.IntValueType);
1173 ValOperand_Shifted =
1174 Builder.CreateShl(Builder.CreateZExt(ValOp, PMV.WordType), PMV.ShiftAmt,
1175 "ValOperand_Shifted");
1176 }
1177
1178 auto PerformPartwordOp = [&](IRBuilderBase &Builder, Value *Loaded) {
1179 return performMaskedAtomicOp(Op, Builder, Loaded, ValOperand_Shifted,
1180 AI->getValOperand(), PMV);
1181 };
1182
1183 Value *OldResult;
1184 if (ExpansionKind == TargetLoweringBase::AtomicExpansionKind::CmpXChg) {
1185 OldResult = insertRMWCmpXchgLoop(Builder, PMV.WordType, PMV.AlignedAddr,
1186 PMV.AlignedAddrAlignment, MemOpOrder, SSID,
1187 AI->isVolatile(), PerformPartwordOp,
1189 } else {
1190 assert(ExpansionKind == TargetLoweringBase::AtomicExpansionKind::LLSC);
1191 OldResult = insertRMWLLSCLoop(Builder, PMV.WordType, PMV.AlignedAddr,
1192 PMV.AlignedAddrAlignment, MemOpOrder,
1193 PerformPartwordOp);
1194 }
1195
1196 Value *FinalOldResult = extractMaskedValue(Builder, OldResult, PMV);
1197 AI->replaceAllUsesWith(FinalOldResult);
1198 AI->eraseFromParent();
1199}
1200
1201// Widen the bitwise atomicrmw (or/xor/and) to the minimum supported width.
1202AtomicRMWInst *AtomicExpandImpl::widenPartwordAtomicRMW(AtomicRMWInst *AI) {
1203 ReplacementIRBuilder Builder(AI, *DL);
1205
1207 Op == AtomicRMWInst::And) &&
1208 "Unable to widen operation");
1209
1210 PartwordMaskValues PMV =
1211 createMaskInstrs(Builder, AI, AI->getType(), AI->getPointerOperand(),
1212 AI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1213
1214 Value *ValOp = AI->getValOperand();
1215 if (ValOp->getType()->isVectorTy())
1216 // For vectors, bitcast to the integer type before extending. Note that
1217 // or/xor/and on vectors are equivalent to the same operation on an integer
1218 // that spans the vector, so we can use the integer type for the operation.
1219 ValOp = Builder.CreateBitCast(ValOp, PMV.IntValueType);
1220 Value *ValOperand_Shifted =
1221 Builder.CreateShl(Builder.CreateZExt(ValOp, PMV.WordType), PMV.ShiftAmt,
1222 "ValOperand_Shifted");
1223
1224 Value *NewOperand;
1225
1226 if (Op == AtomicRMWInst::And)
1227 NewOperand =
1228 Builder.CreateOr(ValOperand_Shifted, PMV.Inv_Mask, "AndOperand");
1229 else
1230 NewOperand = ValOperand_Shifted;
1231
1232 AtomicRMWInst *NewAI = Builder.CreateAtomicRMW(
1233 Op, PMV.AlignedAddr, NewOperand, PMV.AlignedAddrAlignment,
1234 AI->getOrdering(), AI->getSyncScopeID());
1235
1236 NewAI->setVolatile(AI->isVolatile());
1237 copyMetadataForAtomic(*NewAI, *AI);
1238
1239 Value *FinalOldResult = extractMaskedValue(Builder, NewAI, PMV);
1240 AI->replaceAllUsesWith(FinalOldResult);
1241 AI->eraseFromParent();
1242 return NewAI;
1243}
1244
1245bool AtomicExpandImpl::expandPartwordCmpXchg(AtomicCmpXchgInst *CI) {
1246 // The basic idea here is that we're expanding a cmpxchg of a
1247 // smaller memory size up to a word-sized cmpxchg. To do this, we
1248 // need to add a retry-loop for strong cmpxchg, so that
1249 // modifications to other parts of the word don't cause a spurious
1250 // failure.
1251
1252 // This generates code like the following:
1253 // [[Setup mask values PMV.*]]
1254 // %NewVal_Shifted = shl i32 %NewVal, %PMV.ShiftAmt
1255 // %Cmp_Shifted = shl i32 %Cmp, %PMV.ShiftAmt
1256 // %InitLoaded = load i32* %addr
1257 // %InitLoaded_MaskOut = and i32 %InitLoaded, %PMV.Inv_Mask
1258 // br partword.cmpxchg.loop
1259 // partword.cmpxchg.loop:
1260 // %Loaded_MaskOut = phi i32 [ %InitLoaded_MaskOut, %entry ],
1261 // [ %OldVal_MaskOut, %partword.cmpxchg.failure ]
1262 // %FullWord_NewVal = or i32 %Loaded_MaskOut, %NewVal_Shifted
1263 // %FullWord_Cmp = or i32 %Loaded_MaskOut, %Cmp_Shifted
1264 // %NewCI = cmpxchg i32* %PMV.AlignedAddr, i32 %FullWord_Cmp,
1265 // i32 %FullWord_NewVal success_ordering failure_ordering
1266 // %OldVal = extractvalue { i32, i1 } %NewCI, 0
1267 // %Success = extractvalue { i32, i1 } %NewCI, 1
1268 // br i1 %Success, label %partword.cmpxchg.end,
1269 // label %partword.cmpxchg.failure
1270 // partword.cmpxchg.failure:
1271 // %OldVal_MaskOut = and i32 %OldVal, %PMV.Inv_Mask
1272 // %ShouldContinue = icmp ne i32 %Loaded_MaskOut, %OldVal_MaskOut
1273 // br i1 %ShouldContinue, label %partword.cmpxchg.loop,
1274 // label %partword.cmpxchg.end
1275 // partword.cmpxchg.end:
1276 // %tmp1 = lshr i32 %OldVal, %PMV.ShiftAmt
1277 // %FinalOldVal = trunc i32 %tmp1 to i8
1278 // %tmp2 = insertvalue { i8, i1 } undef, i8 %FinalOldVal, 0
1279 // %Res = insertvalue { i8, i1 } %25, i1 %Success, 1
1280
1281 Value *Addr = CI->getPointerOperand();
1282 Value *Cmp = CI->getCompareOperand();
1283 Value *NewVal = CI->getNewValOperand();
1284
1285 BasicBlock *BB = CI->getParent();
1286 Function *F = BB->getParent();
1287 ReplacementIRBuilder Builder(CI, *DL);
1288 LLVMContext &Ctx = Builder.getContext();
1289
1290 BasicBlock *EndBB =
1291 BB->splitBasicBlock(CI->getIterator(), "partword.cmpxchg.end");
1292 auto FailureBB =
1293 BasicBlock::Create(Ctx, "partword.cmpxchg.failure", F, EndBB);
1294 auto LoopBB = BasicBlock::Create(Ctx, "partword.cmpxchg.loop", F, FailureBB);
1295
1296 // The split call above "helpfully" added a branch at the end of BB
1297 // (to the wrong place).
1298 std::prev(BB->end())->eraseFromParent();
1299 Builder.SetInsertPoint(BB);
1300
1301 PartwordMaskValues PMV =
1302 createMaskInstrs(Builder, CI, CI->getCompareOperand()->getType(), Addr,
1303 CI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1304
1305 // Shift the incoming values over, into the right location in the word.
1306 Value *NewVal_Shifted =
1307 Builder.CreateShl(Builder.CreateZExt(NewVal, PMV.WordType), PMV.ShiftAmt);
1308 Value *Cmp_Shifted =
1309 Builder.CreateShl(Builder.CreateZExt(Cmp, PMV.WordType), PMV.ShiftAmt);
1310
1311 // Load the entire current word, and mask into place the expected and new
1312 // values
1313 LoadInst *InitLoaded = Builder.CreateLoad(PMV.WordType, PMV.AlignedAddr);
1314 Value *InitLoaded_MaskOut = Builder.CreateAnd(InitLoaded, PMV.Inv_Mask);
1315 Builder.CreateBr(LoopBB);
1316
1317 // partword.cmpxchg.loop:
1318 Builder.SetInsertPoint(LoopBB);
1319 PHINode *Loaded_MaskOut = Builder.CreatePHI(PMV.WordType, 2);
1320 Loaded_MaskOut->addIncoming(InitLoaded_MaskOut, BB);
1321
1322 // The initial load must be atomic with the same synchronization scope
1323 // to avoid a data race with concurrent stores. If the instruction being
1324 // emulated is volatile, issue a volatile load.
1325 // addIncoming is done first so that any replaceAllUsesWith calls during
1326 // normalization correctly update the PHI incoming value.
1327 InitLoaded->setVolatile(CI->isVolatile());
1329 InitLoaded->setAtomic(AtomicOrdering::Monotonic, CI->getSyncScopeID());
1330 // The newly created load might need to be lowered further. Because it is
1331 // created in the same block as the atomicrmw, the AtomicExpand loop will
1332 // not process it again.
1333 processAtomicInstr(InitLoaded);
1334 }
1335
1336 // Mask/Or the expected and new values into place in the loaded word.
1337 Value *FullWord_NewVal = Builder.CreateOr(Loaded_MaskOut, NewVal_Shifted);
1338 Value *FullWord_Cmp = Builder.CreateOr(Loaded_MaskOut, Cmp_Shifted);
1339 AtomicCmpXchgInst *NewCI = Builder.CreateAtomicCmpXchg(
1340 PMV.AlignedAddr, FullWord_Cmp, FullWord_NewVal, PMV.AlignedAddrAlignment,
1342 NewCI->setVolatile(CI->isVolatile());
1343 // When we're building a strong cmpxchg, we need a loop, so you
1344 // might think we could use a weak cmpxchg inside. But, using strong
1345 // allows the below comparison for ShouldContinue, and we're
1346 // expecting the underlying cmpxchg to be a machine instruction,
1347 // which is strong anyways.
1348 NewCI->setWeak(CI->isWeak());
1349
1350 Value *OldVal = Builder.CreateExtractValue(NewCI, 0);
1351 Value *Success = Builder.CreateExtractValue(NewCI, 1);
1352
1353 if (CI->isWeak())
1354 Builder.CreateBr(EndBB);
1355 else
1356 Builder.CreateCondBr(Success, EndBB, FailureBB);
1357
1358 // partword.cmpxchg.failure:
1359 Builder.SetInsertPoint(FailureBB);
1360 // Upon failure, verify that the masked-out part of the loaded value
1361 // has been modified. If it didn't, abort the cmpxchg, since the
1362 // masked-in part must've.
1363 Value *OldVal_MaskOut = Builder.CreateAnd(OldVal, PMV.Inv_Mask);
1364 Value *ShouldContinue = Builder.CreateICmpNE(Loaded_MaskOut, OldVal_MaskOut);
1365 Builder.CreateCondBr(ShouldContinue, LoopBB, EndBB);
1366
1367 // Add the second value to the phi from above
1368 Loaded_MaskOut->addIncoming(OldVal_MaskOut, FailureBB);
1369
1370 // partword.cmpxchg.end:
1371 Builder.SetInsertPoint(CI);
1372
1373 Value *FinalOldVal = extractMaskedValue(Builder, OldVal, PMV);
1374 Value *Res = PoisonValue::get(CI->getType());
1375 Res = Builder.CreateInsertValue(Res, FinalOldVal, 0);
1376 Res = Builder.CreateInsertValue(Res, Success, 1);
1377
1378 CI->replaceAllUsesWith(Res);
1379 CI->eraseFromParent();
1380 return true;
1381}
1382
1383void AtomicExpandImpl::expandAtomicOpToLLSC(
1384 Instruction *I, Type *ResultType, Value *Addr, Align AddrAlign,
1385 AtomicOrdering MemOpOrder,
1386 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp) {
1387 ReplacementIRBuilder Builder(I, *DL);
1388 Value *Loaded = insertRMWLLSCLoop(Builder, ResultType, Addr, AddrAlign,
1389 MemOpOrder, PerformOp);
1390
1391 I->replaceAllUsesWith(Loaded);
1392 I->eraseFromParent();
1393}
1394
1395void AtomicExpandImpl::expandAtomicRMWToMaskedIntrinsic(AtomicRMWInst *AI) {
1396 ReplacementIRBuilder Builder(AI, *DL);
1397
1398 PartwordMaskValues PMV =
1399 createMaskInstrs(Builder, AI, AI->getType(), AI->getPointerOperand(),
1400 AI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1401
1402 // The value operand must be sign-extended for signed min/max so that the
1403 // target's signed comparison instructions can be used. Otherwise, just
1404 // zero-ext.
1405 Instruction::CastOps CastOp = Instruction::ZExt;
1406 AtomicRMWInst::BinOp RMWOp = AI->getOperation();
1407 if (RMWOp == AtomicRMWInst::Max || RMWOp == AtomicRMWInst::Min)
1408 CastOp = Instruction::SExt;
1409
1410 Value *ValOperand_Shifted = Builder.CreateShl(
1411 Builder.CreateCast(CastOp, AI->getValOperand(), PMV.WordType),
1412 PMV.ShiftAmt, "ValOperand_Shifted");
1413 Value *OldResult = TLI->emitMaskedAtomicRMWIntrinsic(
1414 Builder, AI, PMV.AlignedAddr, ValOperand_Shifted, PMV.Mask, PMV.ShiftAmt,
1415 AI->getOrdering());
1416 Value *FinalOldResult = extractMaskedValue(Builder, OldResult, PMV);
1417 AI->replaceAllUsesWith(FinalOldResult);
1418 AI->eraseFromParent();
1419}
1420
1421void AtomicExpandImpl::expandAtomicCmpXchgToMaskedIntrinsic(
1422 AtomicCmpXchgInst *CI) {
1423 ReplacementIRBuilder Builder(CI, *DL);
1424
1425 PartwordMaskValues PMV = createMaskInstrs(
1426 Builder, CI, CI->getCompareOperand()->getType(), CI->getPointerOperand(),
1427 CI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1428
1429 Value *CmpVal_Shifted = Builder.CreateShl(
1430 Builder.CreateZExt(CI->getCompareOperand(), PMV.WordType), PMV.ShiftAmt,
1431 "CmpVal_Shifted");
1432 Value *NewVal_Shifted = Builder.CreateShl(
1433 Builder.CreateZExt(CI->getNewValOperand(), PMV.WordType), PMV.ShiftAmt,
1434 "NewVal_Shifted");
1436 Builder, CI, PMV.AlignedAddr, CmpVal_Shifted, NewVal_Shifted, PMV.Mask,
1437 CI->getMergedOrdering());
1438 Value *FinalOldVal = extractMaskedValue(Builder, OldVal, PMV);
1439 Value *Res = PoisonValue::get(CI->getType());
1440 Res = Builder.CreateInsertValue(Res, FinalOldVal, 0);
1441 Value *Success = Builder.CreateICmpEQ(
1442 CmpVal_Shifted, Builder.CreateAnd(OldVal, PMV.Mask), "Success");
1443 Res = Builder.CreateInsertValue(Res, Success, 1);
1444
1445 CI->replaceAllUsesWith(Res);
1446 CI->eraseFromParent();
1447}
1448
1449Value *AtomicExpandImpl::insertRMWLLSCLoop(
1450 IRBuilderBase &Builder, Type *ResultTy, Value *Addr, Align AddrAlign,
1451 AtomicOrdering MemOpOrder,
1452 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp) {
1453 LLVMContext &Ctx = Builder.getContext();
1454 BasicBlock *BB = Builder.GetInsertBlock();
1455 Function *F = BB->getParent();
1456
1457 assert(AddrAlign >= F->getDataLayout().getTypeStoreSize(ResultTy) &&
1458 "Expected at least natural alignment at this point.");
1459
1460 // Given: atomicrmw some_op iN* %addr, iN %incr ordering
1461 //
1462 // The standard expansion we produce is:
1463 // [...]
1464 // atomicrmw.start:
1465 // %loaded = @load.linked(%addr)
1466 // %new = some_op iN %loaded, %incr
1467 // %stored = @store_conditional(%new, %addr)
1468 // %try_again = icmp i32 ne %stored, 0
1469 // br i1 %try_again, label %loop, label %atomicrmw.end
1470 // atomicrmw.end:
1471 // [...]
1472 BasicBlock *ExitBB =
1473 BB->splitBasicBlock(Builder.GetInsertPoint(), "atomicrmw.end");
1474 BasicBlock *LoopBB = BasicBlock::Create(Ctx, "atomicrmw.start", F, ExitBB);
1475
1476 // The split call above "helpfully" added a branch at the end of BB (to the
1477 // wrong place).
1478 std::prev(BB->end())->eraseFromParent();
1479 Builder.SetInsertPoint(BB);
1480 Builder.CreateBr(LoopBB);
1481
1482 // Start the main loop block now that we've taken care of the preliminaries.
1483 Builder.SetInsertPoint(LoopBB);
1484 Value *Loaded = TLI->emitLoadLinked(Builder, ResultTy, Addr, MemOpOrder);
1485
1486 Value *NewVal = PerformOp(Builder, Loaded);
1487
1488 Value *StoreSuccess =
1489 TLI->emitStoreConditional(Builder, NewVal, Addr, MemOpOrder);
1490 Value *TryAgain = Builder.CreateICmpNE(
1491 StoreSuccess, ConstantInt::get(IntegerType::get(Ctx, 32), 0), "tryagain");
1492
1493 Instruction *CondBr = Builder.CreateCondBr(TryAgain, LoopBB, ExitBB);
1494
1495 // Atomic RMW expands to a Load-linked / Store-Conditional loop, because it is
1496 // hard to predict precise branch weigths we mark the branch as "unknown"
1497 // (50/50) to prevent misleading optimizations.
1499
1500 Builder.SetInsertPoint(ExitBB, ExitBB->begin());
1501 return Loaded;
1502}
1503
1504/// Convert an atomic cmpxchg of a non-integral type to an integer cmpxchg of
1505/// the equivalent bitwidth. We used to not support pointer cmpxchg in the
1506/// IR. As a migration step, we convert back to what use to be the standard
1507/// way to represent a pointer cmpxchg so that we can update backends one by
1508/// one.
1509AtomicCmpXchgInst *
1510AtomicExpandImpl::convertCmpXchgToIntegerType(AtomicCmpXchgInst *CI) {
1511 auto *M = CI->getModule();
1512 Type *NewTy = getCorrespondingIntegerType(CI->getCompareOperand()->getType(),
1513 M->getDataLayout());
1514
1515 ReplacementIRBuilder Builder(CI, *DL);
1516
1517 Value *Addr = CI->getPointerOperand();
1518
1519 Value *NewCmp = Builder.CreatePtrToInt(CI->getCompareOperand(), NewTy);
1520 Value *NewNewVal = Builder.CreatePtrToInt(CI->getNewValOperand(), NewTy);
1521
1522 auto *NewCI = Builder.CreateAtomicCmpXchg(
1523 Addr, NewCmp, NewNewVal, CI->getAlign(), CI->getSuccessOrdering(),
1524 CI->getFailureOrdering(), CI->getSyncScopeID());
1525 NewCI->setVolatile(CI->isVolatile());
1526 NewCI->setWeak(CI->isWeak());
1527 LLVM_DEBUG(dbgs() << "Replaced " << *CI << " with " << *NewCI << "\n");
1528
1529 Value *OldVal = Builder.CreateExtractValue(NewCI, 0);
1530 Value *Succ = Builder.CreateExtractValue(NewCI, 1);
1531
1532 OldVal = Builder.CreateIntToPtr(OldVal, CI->getCompareOperand()->getType());
1533
1534 Value *Res = PoisonValue::get(CI->getType());
1535 Res = Builder.CreateInsertValue(Res, OldVal, 0);
1536 Res = Builder.CreateInsertValue(Res, Succ, 1);
1537
1538 CI->replaceAllUsesWith(Res);
1539 CI->eraseFromParent();
1540 return NewCI;
1541}
1542
1543bool AtomicExpandImpl::expandAtomicCmpXchg(AtomicCmpXchgInst *CI) {
1544 AtomicOrdering SuccessOrder = CI->getSuccessOrdering();
1545 AtomicOrdering FailureOrder = CI->getFailureOrdering();
1546 Value *Addr = CI->getPointerOperand();
1547 BasicBlock *BB = CI->getParent();
1548 Function *F = BB->getParent();
1549 LLVMContext &Ctx = F->getContext();
1550 // If shouldInsertFencesForAtomic() returns true, then the target does not
1551 // want to deal with memory orders, and emitLeading/TrailingFence should take
1552 // care of everything. Otherwise, emitLeading/TrailingFence are no-op and we
1553 // should preserve the ordering.
1554 bool ShouldInsertFencesForAtomic = TLI->shouldInsertFencesForAtomic(CI);
1555 AtomicOrdering MemOpOrder = ShouldInsertFencesForAtomic
1556 ? AtomicOrdering::Monotonic
1557 : CI->getMergedOrdering();
1558
1559 // In implementations which use a barrier to achieve release semantics, we can
1560 // delay emitting this barrier until we know a store is actually going to be
1561 // attempted. The cost of this delay is that we need 2 copies of the block
1562 // emitting the load-linked, affecting code size.
1563 //
1564 // Ideally, this logic would be unconditional except for the minsize check
1565 // since in other cases the extra blocks naturally collapse down to the
1566 // minimal loop. Unfortunately, this puts too much stress on later
1567 // optimisations so we avoid emitting the extra logic in those cases too.
1568 bool HasReleasedLoadBB = !CI->isWeak() && ShouldInsertFencesForAtomic &&
1569 SuccessOrder != AtomicOrdering::Monotonic &&
1570 SuccessOrder != AtomicOrdering::Acquire &&
1571 !F->hasMinSize();
1572
1573 // There's no overhead for sinking the release barrier in a weak cmpxchg, so
1574 // do it even on minsize.
1575 bool UseUnconditionalReleaseBarrier = F->hasMinSize() && !CI->isWeak();
1576
1577 // Given: cmpxchg some_op iN* %addr, iN %desired, iN %new success_ord fail_ord
1578 //
1579 // The full expansion we produce is:
1580 // [...]
1581 // %aligned.addr = ...
1582 // cmpxchg.start:
1583 // %unreleasedload = @load.linked(%aligned.addr)
1584 // %unreleasedload.extract = extract value from %unreleasedload
1585 // %should_store = icmp eq %unreleasedload.extract, %desired
1586 // br i1 %should_store, label %cmpxchg.releasingstore,
1587 // label %cmpxchg.nostore
1588 // cmpxchg.releasingstore:
1589 // fence?
1590 // br label cmpxchg.trystore
1591 // cmpxchg.trystore:
1592 // %loaded.trystore = phi [%unreleasedload, %cmpxchg.releasingstore],
1593 // [%releasedload, %cmpxchg.releasedload]
1594 // %updated.new = insert %new into %loaded.trystore
1595 // %stored = @store_conditional(%updated.new, %aligned.addr)
1596 // %success = icmp eq i32 %stored, 0
1597 // br i1 %success, label %cmpxchg.success,
1598 // label %cmpxchg.releasedload/%cmpxchg.failure
1599 // cmpxchg.releasedload:
1600 // %releasedload = @load.linked(%aligned.addr)
1601 // %releasedload.extract = extract value from %releasedload
1602 // %should_store = icmp eq %releasedload.extract, %desired
1603 // br i1 %should_store, label %cmpxchg.trystore,
1604 // label %cmpxchg.failure
1605 // cmpxchg.success:
1606 // fence?
1607 // br label %cmpxchg.end
1608 // cmpxchg.nostore:
1609 // %loaded.nostore = phi [%unreleasedload, %cmpxchg.start],
1610 // [%releasedload,
1611 // %cmpxchg.releasedload/%cmpxchg.trystore]
1612 // @load_linked_fail_balance()?
1613 // br label %cmpxchg.failure
1614 // cmpxchg.failure:
1615 // fence?
1616 // br label %cmpxchg.end
1617 // cmpxchg.end:
1618 // %loaded.exit = phi [%loaded.nostore, %cmpxchg.failure],
1619 // [%loaded.trystore, %cmpxchg.trystore]
1620 // %success = phi i1 [true, %cmpxchg.success], [false, %cmpxchg.failure]
1621 // %loaded = extract value from %loaded.exit
1622 // %restmp = insertvalue { iN, i1 } undef, iN %loaded, 0
1623 // %res = insertvalue { iN, i1 } %restmp, i1 %success, 1
1624 // [...]
1625 BasicBlock *ExitBB = BB->splitBasicBlock(CI->getIterator(), "cmpxchg.end");
1626 auto FailureBB = BasicBlock::Create(Ctx, "cmpxchg.failure", F, ExitBB);
1627 auto NoStoreBB = BasicBlock::Create(Ctx, "cmpxchg.nostore", F, FailureBB);
1628 auto SuccessBB = BasicBlock::Create(Ctx, "cmpxchg.success", F, NoStoreBB);
1629 auto ReleasedLoadBB =
1630 BasicBlock::Create(Ctx, "cmpxchg.releasedload", F, SuccessBB);
1631 auto TryStoreBB =
1632 BasicBlock::Create(Ctx, "cmpxchg.trystore", F, ReleasedLoadBB);
1633 auto ReleasingStoreBB =
1634 BasicBlock::Create(Ctx, "cmpxchg.fencedstore", F, TryStoreBB);
1635 auto StartBB = BasicBlock::Create(Ctx, "cmpxchg.start", F, ReleasingStoreBB);
1636
1637 ReplacementIRBuilder Builder(CI, *DL);
1638
1639 // The split call above "helpfully" added a branch at the end of BB (to the
1640 // wrong place), but we might want a fence too. It's easiest to just remove
1641 // the branch entirely.
1642 std::prev(BB->end())->eraseFromParent();
1643 Builder.SetInsertPoint(BB);
1644 if (ShouldInsertFencesForAtomic && UseUnconditionalReleaseBarrier)
1645 TLI->emitLeadingFence(Builder, CI, SuccessOrder);
1646
1647 PartwordMaskValues PMV =
1648 createMaskInstrs(Builder, CI, CI->getCompareOperand()->getType(), Addr,
1649 CI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1650 Builder.CreateBr(StartBB);
1651
1652 // Start the main loop block now that we've taken care of the preliminaries.
1653 Builder.SetInsertPoint(StartBB);
1654 Value *UnreleasedLoad =
1655 TLI->emitLoadLinked(Builder, PMV.WordType, PMV.AlignedAddr, MemOpOrder);
1656 Value *UnreleasedLoadExtract =
1657 extractMaskedValue(Builder, UnreleasedLoad, PMV);
1658 Value *ShouldStore = Builder.CreateICmpEQ(
1659 UnreleasedLoadExtract, CI->getCompareOperand(), "should_store");
1660
1661 // If the cmpxchg doesn't actually need any ordering when it fails, we can
1662 // jump straight past that fence instruction (if it exists).
1663 Builder.CreateCondBr(ShouldStore, ReleasingStoreBB, NoStoreBB,
1664 MDBuilder(F->getContext()).createLikelyBranchWeights());
1665
1666 Builder.SetInsertPoint(ReleasingStoreBB);
1667 if (ShouldInsertFencesForAtomic && !UseUnconditionalReleaseBarrier)
1668 TLI->emitLeadingFence(Builder, CI, SuccessOrder);
1669 Builder.CreateBr(TryStoreBB);
1670
1671 Builder.SetInsertPoint(TryStoreBB);
1672 PHINode *LoadedTryStore =
1673 Builder.CreatePHI(PMV.WordType, 2, "loaded.trystore");
1674 LoadedTryStore->addIncoming(UnreleasedLoad, ReleasingStoreBB);
1675 Value *NewValueInsert =
1676 insertMaskedValue(Builder, LoadedTryStore, CI->getNewValOperand(), PMV);
1677 Value *StoreSuccess = TLI->emitStoreConditional(Builder, NewValueInsert,
1678 PMV.AlignedAddr, MemOpOrder);
1679 StoreSuccess = Builder.CreateICmpEQ(
1680 StoreSuccess, ConstantInt::get(Type::getInt32Ty(Ctx), 0), "success");
1681 BasicBlock *RetryBB = HasReleasedLoadBB ? ReleasedLoadBB : StartBB;
1682 Builder.CreateCondBr(StoreSuccess, SuccessBB,
1683 CI->isWeak() ? FailureBB : RetryBB,
1684 MDBuilder(F->getContext()).createLikelyBranchWeights());
1685
1686 Builder.SetInsertPoint(ReleasedLoadBB);
1687 Value *SecondLoad;
1688 if (HasReleasedLoadBB) {
1689 SecondLoad =
1690 TLI->emitLoadLinked(Builder, PMV.WordType, PMV.AlignedAddr, MemOpOrder);
1691 Value *SecondLoadExtract = extractMaskedValue(Builder, SecondLoad, PMV);
1692 ShouldStore = Builder.CreateICmpEQ(SecondLoadExtract,
1693 CI->getCompareOperand(), "should_store");
1694
1695 // If the cmpxchg doesn't actually need any ordering when it fails, we can
1696 // jump straight past that fence instruction (if it exists).
1697 Builder.CreateCondBr(
1698 ShouldStore, TryStoreBB, NoStoreBB,
1699 MDBuilder(F->getContext()).createLikelyBranchWeights());
1700 // Update PHI node in TryStoreBB.
1701 LoadedTryStore->addIncoming(SecondLoad, ReleasedLoadBB);
1702 } else
1703 Builder.CreateUnreachable();
1704
1705 // Make sure later instructions don't get reordered with a fence if
1706 // necessary.
1707 Builder.SetInsertPoint(SuccessBB);
1708 if (ShouldInsertFencesForAtomic ||
1710 TLI->emitTrailingFence(Builder, CI, SuccessOrder);
1711 Builder.CreateBr(ExitBB);
1712
1713 Builder.SetInsertPoint(NoStoreBB);
1714 PHINode *LoadedNoStore =
1715 Builder.CreatePHI(UnreleasedLoad->getType(), 2, "loaded.nostore");
1716 LoadedNoStore->addIncoming(UnreleasedLoad, StartBB);
1717 if (HasReleasedLoadBB)
1718 LoadedNoStore->addIncoming(SecondLoad, ReleasedLoadBB);
1719
1720 // In the failing case, where we don't execute the store-conditional, the
1721 // target might want to balance out the load-linked with a dedicated
1722 // instruction (e.g., on ARM, clearing the exclusive monitor).
1724 Builder.CreateBr(FailureBB);
1725
1726 Builder.SetInsertPoint(FailureBB);
1727 PHINode *LoadedFailure =
1728 Builder.CreatePHI(UnreleasedLoad->getType(), 2, "loaded.failure");
1729 LoadedFailure->addIncoming(LoadedNoStore, NoStoreBB);
1730 if (CI->isWeak())
1731 LoadedFailure->addIncoming(LoadedTryStore, TryStoreBB);
1732 if (ShouldInsertFencesForAtomic)
1733 TLI->emitTrailingFence(Builder, CI, FailureOrder);
1734 Builder.CreateBr(ExitBB);
1735
1736 // Finally, we have control-flow based knowledge of whether the cmpxchg
1737 // succeeded or not. We expose this to later passes by converting any
1738 // subsequent "icmp eq/ne %loaded, %oldval" into a use of an appropriate
1739 // PHI.
1740 Builder.SetInsertPoint(ExitBB, ExitBB->begin());
1741 PHINode *LoadedExit =
1742 Builder.CreatePHI(UnreleasedLoad->getType(), 2, "loaded.exit");
1743 LoadedExit->addIncoming(LoadedTryStore, SuccessBB);
1744 LoadedExit->addIncoming(LoadedFailure, FailureBB);
1745 PHINode *Success = Builder.CreatePHI(Type::getInt1Ty(Ctx), 2, "success");
1746 Success->addIncoming(ConstantInt::getTrue(Ctx), SuccessBB);
1747 Success->addIncoming(ConstantInt::getFalse(Ctx), FailureBB);
1748
1749 // This is the "exit value" from the cmpxchg expansion. It may be of
1750 // a type wider than the one in the cmpxchg instruction.
1751 Value *LoadedFull = LoadedExit;
1752
1753 Builder.SetInsertPoint(ExitBB, std::next(Success->getIterator()));
1754 Value *Loaded = extractMaskedValue(Builder, LoadedFull, PMV);
1755
1756 // Look for any users of the cmpxchg that are just comparing the loaded value
1757 // against the desired one, and replace them with the CFG-derived version.
1759 for (auto *User : CI->users()) {
1760 ExtractValueInst *EV = dyn_cast<ExtractValueInst>(User);
1761 if (!EV)
1762 continue;
1763
1764 assert(EV->getNumIndices() == 1 && EV->getIndices()[0] <= 1 &&
1765 "weird extraction from { iN, i1 }");
1766
1767 if (EV->getIndices()[0] == 0)
1768 EV->replaceAllUsesWith(Loaded);
1769 else
1771
1772 PrunedInsts.push_back(EV);
1773 }
1774
1775 // We can remove the instructions now we're no longer iterating through them.
1776 for (auto *EV : PrunedInsts)
1777 EV->eraseFromParent();
1778
1779 if (!CI->use_empty()) {
1780 // Some use of the full struct return that we don't understand has happened,
1781 // so we've got to reconstruct it properly.
1782 Value *Res;
1783 Res = Builder.CreateInsertValue(PoisonValue::get(CI->getType()), Loaded, 0);
1784 Res = Builder.CreateInsertValue(Res, Success, 1);
1785
1786 CI->replaceAllUsesWith(Res);
1787 }
1788
1789 CI->eraseFromParent();
1790 return true;
1791}
1792
1793bool AtomicExpandImpl::isIdempotentRMW(AtomicRMWInst *RMWI) {
1794 if (RMWI->isVolatile())
1795 return false;
1796 // TODO: Add floating point support.
1797 auto C = dyn_cast<ConstantInt>(RMWI->getValOperand());
1798 if (!C)
1799 return false;
1800
1801 switch (RMWI->getOperation()) {
1802 case AtomicRMWInst::Add:
1803 case AtomicRMWInst::Sub:
1804 case AtomicRMWInst::Or:
1805 case AtomicRMWInst::Xor:
1806 return C->isZero();
1807 case AtomicRMWInst::And:
1808 return C->isMinusOne();
1809 case AtomicRMWInst::Min:
1810 return C->isMaxValue(true);
1811 case AtomicRMWInst::Max:
1812 return C->isMinValue(true);
1814 return C->isMaxValue(false);
1816 return C->isMinValue(false);
1817 default:
1818 return false;
1819 }
1820}
1821
1822bool AtomicExpandImpl::simplifyIdempotentRMW(AtomicRMWInst *RMWI) {
1823 if (auto ResultingLoad = TLI->lowerIdempotentRMWIntoFencedLoad(RMWI)) {
1824 tryExpandAtomicLoad(ResultingLoad);
1825 return true;
1826 }
1827 return false;
1828}
1829
1830Value *AtomicExpandImpl::insertRMWCmpXchgLoop(
1831 IRBuilderBase &Builder, Type *ResultTy, Value *Addr, Align AddrAlign,
1832 AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile,
1833 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp,
1834 CreateCmpXchgInstFun CreateCmpXchg, Instruction *MetadataSrc) {
1835 LLVMContext &Ctx = Builder.getContext();
1836 BasicBlock *BB = Builder.GetInsertBlock();
1837 Function *F = BB->getParent();
1838
1839 // Given: atomicrmw some_op iN* %addr, iN %incr ordering
1840 //
1841 // The standard expansion we produce is:
1842 // [...]
1843 // %init_loaded = load atomic iN* %addr
1844 // br label %loop
1845 // loop:
1846 // %loaded = phi iN [ %init_loaded, %entry ], [ %new_loaded, %loop ]
1847 // %new = some_op iN %loaded, %incr
1848 // %pair = cmpxchg iN* %addr, iN %loaded, iN %new
1849 // %new_loaded = extractvalue { iN, i1 } %pair, 0
1850 // %success = extractvalue { iN, i1 } %pair, 1
1851 // br i1 %success, label %atomicrmw.end, label %loop
1852 // atomicrmw.end:
1853 // [...]
1854 BasicBlock *ExitBB =
1855 BB->splitBasicBlock(Builder.GetInsertPoint(), "atomicrmw.end");
1856 BasicBlock *LoopBB = BasicBlock::Create(Ctx, "atomicrmw.start", F, ExitBB);
1857
1858 // The split call above "helpfully" added a branch at the end of BB (to the
1859 // wrong place), but we want a load. It's easiest to just remove
1860 // the branch entirely.
1861 std::prev(BB->end())->eraseFromParent();
1862 Builder.SetInsertPoint(BB);
1863 LoadInst *InitLoaded = Builder.CreateAlignedLoad(ResultTy, Addr, AddrAlign);
1864 Builder.CreateBr(LoopBB);
1865
1866 // Start the main loop block now that we've taken care of the preliminaries.
1867 Builder.SetInsertPoint(LoopBB);
1868 PHINode *Loaded = Builder.CreatePHI(ResultTy, 2, "loaded");
1869 Loaded->addIncoming(InitLoaded, BB);
1870
1871 // The initial load must be atomic with the same synchronization scope
1872 // to avoid a data race with concurrent stores. If the instruction being
1873 // emulated is volatile, issue a volatile load.
1874 // addIncoming is done first so that any replaceAllUsesWith calls during
1875 // normalization correctly update the PHI incoming value.
1876 InitLoaded->setVolatile(IsVolatile);
1878 InitLoaded->setAtomic(AtomicOrdering::Monotonic, SSID);
1879 // The newly created load might need to be lowered further. Because it is
1880 // created in the same block as the atomicrmw, the AtomicExpand loop will
1881 // not process it again.
1882 processAtomicInstr(InitLoaded);
1883 }
1884
1885 Value *NewVal = PerformOp(Builder, Loaded);
1886
1887 Value *NewLoaded = nullptr;
1888 Value *Success = nullptr;
1889
1890 CreateCmpXchg(Builder, Addr, Loaded, NewVal, AddrAlign,
1891 MemOpOrder == AtomicOrdering::Unordered
1892 ? AtomicOrdering::Monotonic
1893 : MemOpOrder,
1894 SSID, IsVolatile, Success, NewLoaded, MetadataSrc);
1895 assert(Success && NewLoaded);
1896
1897 Loaded->addIncoming(NewLoaded, LoopBB);
1898
1899 Instruction *CondBr = Builder.CreateCondBr(Success, ExitBB, LoopBB);
1900
1901 // Atomic RMW expands to a cmpxchg loop, Since precise branch weights
1902 // cannot be easily determined here, we mark the branch as "unknown" (50/50)
1903 // to prevent misleading optimizations.
1905
1906 Builder.SetInsertPoint(ExitBB, ExitBB->begin());
1907 return NewLoaded;
1908}
1909
1910bool AtomicExpandImpl::tryExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) {
1911 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
1912 unsigned ValueSize = getAtomicOpSize(CI);
1913
1914 switch (TLI->shouldExpandAtomicCmpXchgInIR(CI)) {
1915 default:
1916 llvm_unreachable("Unhandled case in tryExpandAtomicCmpXchg");
1917 case TargetLoweringBase::AtomicExpansionKind::None:
1918 if (ValueSize < MinCASSize)
1919 return expandPartwordCmpXchg(CI);
1920 return false;
1921 case TargetLoweringBase::AtomicExpansionKind::LLSC: {
1922 return expandAtomicCmpXchg(CI);
1923 }
1924 case TargetLoweringBase::AtomicExpansionKind::MaskedIntrinsic:
1925 expandAtomicCmpXchgToMaskedIntrinsic(CI);
1926 return true;
1927 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
1928 return lowerAtomicCmpXchgInst(CI);
1929 case TargetLoweringBase::AtomicExpansionKind::CustomExpand: {
1930 TLI->emitExpandAtomicCmpXchg(CI);
1931 return true;
1932 }
1933 }
1934}
1935
1936bool AtomicExpandImpl::expandAtomicRMWToCmpXchg(
1937 AtomicRMWInst *AI, CreateCmpXchgInstFun CreateCmpXchg) {
1938 ReplacementIRBuilder Builder(AI, AI->getDataLayout());
1939 Builder.setIsFPConstrained(
1940 AI->getFunction()->hasFnAttribute(Attribute::StrictFP));
1941
1942 // FIXME: If FP exceptions are observable, we should force them off for the
1943 // loop for the FP atomics.
1944 Value *Loaded = AtomicExpandImpl::insertRMWCmpXchgLoop(
1945 Builder, AI->getType(), AI->getPointerOperand(), AI->getAlign(),
1946 AI->getOrdering(), AI->getSyncScopeID(), AI->isVolatile(),
1947 [&](IRBuilderBase &Builder, Value *Loaded) {
1948 return buildAtomicRMWValue(AI->getOperation(), Builder, Loaded,
1949 AI->getValOperand());
1950 },
1951 CreateCmpXchg, /*MetadataSrc=*/AI);
1952
1953 AI->replaceAllUsesWith(Loaded);
1954 AI->eraseFromParent();
1955 return true;
1956}
1957
1958// In order to use one of the sized library calls such as
1959// __atomic_fetch_add_4, the alignment must be sufficient, the size
1960// must be one of the potentially-specialized sizes, and the value
1961// type must actually exist in C on the target (otherwise, the
1962// function wouldn't actually be defined.)
1963static bool canUseSizedAtomicCall(unsigned Size, Align Alignment,
1964 const DataLayout &DL) {
1965 // TODO: "LargestSize" is an approximation for "largest type that
1966 // you can express in C". It seems to be the case that int128 is
1967 // supported on all 64-bit platforms, otherwise only up to 64-bit
1968 // integers are supported. If we get this wrong, then we'll try to
1969 // call a sized libcall that doesn't actually exist. There should
1970 // really be some more reliable way in LLVM of determining integer
1971 // sizes which are valid in the target's C ABI...
1972 unsigned LargestSize = DL.getLargestLegalIntTypeSizeInBits() >= 64 ? 16 : 8;
1973 return Alignment >= Size &&
1974 (Size == 1 || Size == 2 || Size == 4 || Size == 8 || Size == 16) &&
1975 Size <= LargestSize;
1976}
1977
1978void AtomicExpandImpl::expandAtomicLoadToLibcall(LoadInst *I) {
1979 static const RTLIB::Libcall Libcalls[6] = {
1980 RTLIB::ATOMIC_LOAD, RTLIB::ATOMIC_LOAD_1, RTLIB::ATOMIC_LOAD_2,
1981 RTLIB::ATOMIC_LOAD_4, RTLIB::ATOMIC_LOAD_8, RTLIB::ATOMIC_LOAD_16};
1982 unsigned Size = getAtomicOpSize(I);
1983
1984 bool Expanded = expandAtomicOpToLibcall(
1985 I, Size, I->getAlign(), I->getPointerOperand(), nullptr, nullptr,
1986 I->getOrdering(), AtomicOrdering::NotAtomic, Libcalls);
1987 if (!Expanded)
1988 handleUnsupportedAtomicSize(I, "atomic load");
1989}
1990
1991void AtomicExpandImpl::expandAtomicStoreToLibcall(StoreInst *I) {
1992 static const RTLIB::Libcall Libcalls[6] = {
1993 RTLIB::ATOMIC_STORE, RTLIB::ATOMIC_STORE_1, RTLIB::ATOMIC_STORE_2,
1994 RTLIB::ATOMIC_STORE_4, RTLIB::ATOMIC_STORE_8, RTLIB::ATOMIC_STORE_16};
1995 unsigned Size = getAtomicOpSize(I);
1996
1997 bool Expanded = expandAtomicOpToLibcall(
1998 I, Size, I->getAlign(), I->getPointerOperand(), I->getValueOperand(),
1999 nullptr, I->getOrdering(), AtomicOrdering::NotAtomic, Libcalls);
2000 if (!Expanded)
2001 handleUnsupportedAtomicSize(I, "atomic store");
2002}
2003
2004void AtomicExpandImpl::expandAtomicCASToLibcall(AtomicCmpXchgInst *I,
2005 const Twine &AtomicOpName,
2006 Instruction *DiagnosticInst) {
2007 static const RTLIB::Libcall Libcalls[6] = {
2008 RTLIB::ATOMIC_COMPARE_EXCHANGE, RTLIB::ATOMIC_COMPARE_EXCHANGE_1,
2009 RTLIB::ATOMIC_COMPARE_EXCHANGE_2, RTLIB::ATOMIC_COMPARE_EXCHANGE_4,
2010 RTLIB::ATOMIC_COMPARE_EXCHANGE_8, RTLIB::ATOMIC_COMPARE_EXCHANGE_16};
2011 unsigned Size = getAtomicOpSize(I);
2012
2013 bool Expanded = expandAtomicOpToLibcall(
2014 I, Size, I->getAlign(), I->getPointerOperand(), I->getNewValOperand(),
2015 I->getCompareOperand(), I->getSuccessOrdering(), I->getFailureOrdering(),
2016 Libcalls);
2017 if (!Expanded)
2018 handleUnsupportedAtomicSize(I, AtomicOpName, DiagnosticInst);
2019}
2020
2022 static const RTLIB::Libcall LibcallsXchg[6] = {
2023 RTLIB::ATOMIC_EXCHANGE, RTLIB::ATOMIC_EXCHANGE_1,
2024 RTLIB::ATOMIC_EXCHANGE_2, RTLIB::ATOMIC_EXCHANGE_4,
2025 RTLIB::ATOMIC_EXCHANGE_8, RTLIB::ATOMIC_EXCHANGE_16};
2026 static const RTLIB::Libcall LibcallsAdd[6] = {
2027 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_ADD_1,
2028 RTLIB::ATOMIC_FETCH_ADD_2, RTLIB::ATOMIC_FETCH_ADD_4,
2029 RTLIB::ATOMIC_FETCH_ADD_8, RTLIB::ATOMIC_FETCH_ADD_16};
2030 static const RTLIB::Libcall LibcallsSub[6] = {
2031 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_SUB_1,
2032 RTLIB::ATOMIC_FETCH_SUB_2, RTLIB::ATOMIC_FETCH_SUB_4,
2033 RTLIB::ATOMIC_FETCH_SUB_8, RTLIB::ATOMIC_FETCH_SUB_16};
2034 static const RTLIB::Libcall LibcallsAnd[6] = {
2035 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_AND_1,
2036 RTLIB::ATOMIC_FETCH_AND_2, RTLIB::ATOMIC_FETCH_AND_4,
2037 RTLIB::ATOMIC_FETCH_AND_8, RTLIB::ATOMIC_FETCH_AND_16};
2038 static const RTLIB::Libcall LibcallsOr[6] = {
2039 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_OR_1,
2040 RTLIB::ATOMIC_FETCH_OR_2, RTLIB::ATOMIC_FETCH_OR_4,
2041 RTLIB::ATOMIC_FETCH_OR_8, RTLIB::ATOMIC_FETCH_OR_16};
2042 static const RTLIB::Libcall LibcallsXor[6] = {
2043 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_XOR_1,
2044 RTLIB::ATOMIC_FETCH_XOR_2, RTLIB::ATOMIC_FETCH_XOR_4,
2045 RTLIB::ATOMIC_FETCH_XOR_8, RTLIB::ATOMIC_FETCH_XOR_16};
2046 static const RTLIB::Libcall LibcallsNand[6] = {
2047 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_NAND_1,
2048 RTLIB::ATOMIC_FETCH_NAND_2, RTLIB::ATOMIC_FETCH_NAND_4,
2049 RTLIB::ATOMIC_FETCH_NAND_8, RTLIB::ATOMIC_FETCH_NAND_16};
2050
2051 switch (Op) {
2053 llvm_unreachable("Should not have BAD_BINOP.");
2055 return ArrayRef(LibcallsXchg);
2056 case AtomicRMWInst::Add:
2057 return ArrayRef(LibcallsAdd);
2058 case AtomicRMWInst::Sub:
2059 return ArrayRef(LibcallsSub);
2060 case AtomicRMWInst::And:
2061 return ArrayRef(LibcallsAnd);
2062 case AtomicRMWInst::Or:
2063 return ArrayRef(LibcallsOr);
2064 case AtomicRMWInst::Xor:
2065 return ArrayRef(LibcallsXor);
2067 return ArrayRef(LibcallsNand);
2068 case AtomicRMWInst::Max:
2069 case AtomicRMWInst::Min:
2084 // No atomic libcalls are available for these.
2085 return {};
2086 }
2087 llvm_unreachable("Unexpected AtomicRMW operation.");
2088}
2089
2090void AtomicExpandImpl::expandAtomicRMWToLibcall(AtomicRMWInst *I) {
2091 ArrayRef<RTLIB::Libcall> Libcalls = GetRMWLibcall(I->getOperation());
2092
2093 unsigned Size = getAtomicOpSize(I);
2094
2095 bool Success = false;
2096 if (!Libcalls.empty())
2097 Success = expandAtomicOpToLibcall(
2098 I, Size, I->getAlign(), I->getPointerOperand(), I->getValOperand(),
2099 nullptr, I->getOrdering(), AtomicOrdering::NotAtomic, Libcalls);
2100
2101 // The expansion failed: either there were no libcalls at all for
2102 // the operation (min/max), or there were only size-specialized
2103 // libcalls (add/sub/etc) and we needed a generic. So, expand to a
2104 // CAS libcall, via a CAS loop, instead.
2105 if (!Success) {
2106 expandAtomicRMWToCmpXchg(
2107 I, [this, I](IRBuilderBase &Builder, Value *Addr, Value *Loaded,
2108 Value *NewVal, Align Alignment, AtomicOrdering MemOpOrder,
2109 SyncScope::ID SSID, bool IsVolatile, Value *&Success,
2110 Value *&NewLoaded, Instruction *MetadataSrc) {
2111 // Create the CAS instruction normally...
2112 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
2113 Addr, Loaded, NewVal, Alignment, MemOpOrder,
2115 Pair->setVolatile(IsVolatile);
2116 if (MetadataSrc)
2117 copyMetadataForAtomic(*Pair, *MetadataSrc);
2118
2119 Success = Builder.CreateExtractValue(Pair, 1, "success");
2120 NewLoaded = Builder.CreateExtractValue(Pair, 0, "newloaded");
2121
2122 // ...and then expand the CAS into a libcall.
2123 expandAtomicCASToLibcall(
2124 Pair,
2125 "atomicrmw " + AtomicRMWInst::getOperationName(I->getOperation()),
2126 MetadataSrc);
2127 });
2128 }
2129}
2130
2131// A helper routine for the above expandAtomic*ToLibcall functions.
2132//
2133// 'Libcalls' contains an array of enum values for the particular
2134// ATOMIC libcalls to be emitted. All of the other arguments besides
2135// 'I' are extracted from the Instruction subclass by the
2136// caller. Depending on the particular call, some will be null.
2137bool AtomicExpandImpl::expandAtomicOpToLibcall(
2138 Instruction *I, unsigned Size, Align Alignment, Value *PointerOperand,
2139 Value *ValueOperand, Value *CASExpected, AtomicOrdering Ordering,
2140 AtomicOrdering Ordering2, ArrayRef<RTLIB::Libcall> Libcalls) {
2141 assert(Libcalls.size() == 6);
2142
2143 LLVMContext &Ctx = I->getContext();
2144 Module *M = I->getModule();
2145 const DataLayout &DL = M->getDataLayout();
2146 IRBuilder<> Builder(I);
2147 IRBuilder<> AllocaBuilder(&I->getFunction()->getEntryBlock().front());
2148
2149 bool UseSizedLibcall = canUseSizedAtomicCall(Size, Alignment, DL);
2150 Type *SizedIntTy = Type::getIntNTy(Ctx, Size * 8);
2151
2152 if (M->getTargetTriple().isOSWindows() && M->getTargetTriple().isX86_64() &&
2153 Size == 16) {
2154 // x86_64 Windows passes i128 as an XMM vector; on return, it is in
2155 // XMM0, and as a parameter, it is passed indirectly. The generic lowering
2156 // rules handles this correctly if we pass it as a v2i64 rather than
2157 // i128. This is what Clang does in the frontend for such types as well
2158 // (see WinX86_64ABIInfo::classify in Clang).
2159 SizedIntTy = FixedVectorType::get(Type::getInt64Ty(Ctx), 2);
2160 }
2161
2162 const Align AllocaAlignment = DL.getPrefTypeAlign(SizedIntTy);
2163
2164 // TODO: the "order" argument type is "int", not int32. So
2165 // getInt32Ty may be wrong if the arch uses e.g. 16-bit ints.
2166 assert(Ordering != AtomicOrdering::NotAtomic && "expect atomic MO");
2167 Constant *OrderingVal =
2168 ConstantInt::get(Type::getInt32Ty(Ctx), (int)toCABI(Ordering));
2169 Constant *Ordering2Val = nullptr;
2170 if (CASExpected) {
2171 assert(Ordering2 != AtomicOrdering::NotAtomic && "expect atomic MO");
2172 Ordering2Val =
2173 ConstantInt::get(Type::getInt32Ty(Ctx), (int)toCABI(Ordering2));
2174 }
2175 bool HasResult = I->getType() != Type::getVoidTy(Ctx);
2176
2177 RTLIB::Libcall RTLibType;
2178 if (UseSizedLibcall) {
2179 switch (Size) {
2180 case 1:
2181 RTLibType = Libcalls[1];
2182 break;
2183 case 2:
2184 RTLibType = Libcalls[2];
2185 break;
2186 case 4:
2187 RTLibType = Libcalls[3];
2188 break;
2189 case 8:
2190 RTLibType = Libcalls[4];
2191 break;
2192 case 16:
2193 RTLibType = Libcalls[5];
2194 break;
2195 }
2196 } else if (Libcalls[0] != RTLIB::UNKNOWN_LIBCALL) {
2197 RTLibType = Libcalls[0];
2198 } else {
2199 // Can't use sized function, and there's no generic for this
2200 // operation, so give up.
2201 return false;
2202 }
2203
2204 RTLIB::LibcallImpl LibcallImpl = LibcallLowering->getLibcallImpl(RTLibType);
2205 if (LibcallImpl == RTLIB::Unsupported) {
2206 // This target does not implement the requested atomic libcall so give up.
2207 return false;
2208 }
2209
2210 // Build up the function call. There's two kinds. First, the sized
2211 // variants. These calls are going to be one of the following (with
2212 // N=1,2,4,8,16):
2213 // iN __atomic_load_N(iN *ptr, int ordering)
2214 // void __atomic_store_N(iN *ptr, iN val, int ordering)
2215 // iN __atomic_{exchange|fetch_*}_N(iN *ptr, iN val, int ordering)
2216 // bool __atomic_compare_exchange_N(iN *ptr, iN *expected, iN desired,
2217 // int success_order, int failure_order)
2218 //
2219 // Note that these functions can be used for non-integer atomic
2220 // operations, the values just need to be bitcast to integers on the
2221 // way in and out.
2222 //
2223 // And, then, the generic variants. They look like the following:
2224 // void __atomic_load(size_t size, void *ptr, void *ret, int ordering)
2225 // void __atomic_store(size_t size, void *ptr, void *val, int ordering)
2226 // void __atomic_exchange(size_t size, void *ptr, void *val, void *ret,
2227 // int ordering)
2228 // bool __atomic_compare_exchange(size_t size, void *ptr, void *expected,
2229 // void *desired, int success_order,
2230 // int failure_order)
2231 //
2232 // The different signatures are built up depending on the
2233 // 'UseSizedLibcall', 'CASExpected', 'ValueOperand', and 'HasResult'
2234 // variables.
2235
2236 AllocaInst *AllocaCASExpected = nullptr;
2237 AllocaInst *AllocaValue = nullptr;
2238 AllocaInst *AllocaResult = nullptr;
2239
2240 Type *ResultTy;
2242 AttributeList Attr;
2243
2244 // 'size' argument.
2245 if (!UseSizedLibcall) {
2246 // Note, getIntPtrType is assumed equivalent to size_t.
2247 Args.push_back(ConstantInt::get(DL.getIntPtrType(Ctx), Size));
2248 }
2249
2250 // 'ptr' argument.
2251 // note: This assumes all address spaces share a common libfunc
2252 // implementation and that addresses are convertable. For systems without
2253 // that property, we'd need to extend this mechanism to support AS-specific
2254 // families of atomic intrinsics.
2255 Value *PtrVal = PointerOperand;
2256 PtrVal = Builder.CreateAddrSpaceCast(PtrVal, PointerType::getUnqual(Ctx));
2257 Args.push_back(PtrVal);
2258
2259 // 'expected' argument, if present.
2260 if (CASExpected) {
2261 AllocaCASExpected = AllocaBuilder.CreateAlloca(CASExpected->getType());
2262 AllocaCASExpected->setAlignment(AllocaAlignment);
2263 Builder.CreateLifetimeStart(AllocaCASExpected);
2264 Builder.CreateAlignedStore(CASExpected, AllocaCASExpected, AllocaAlignment);
2265 Args.push_back(AllocaCASExpected);
2266 }
2267
2268 // 'val' argument ('desired' for cas), if present.
2269 if (ValueOperand) {
2270 if (UseSizedLibcall) {
2271 Value *IntValue =
2272 Builder.CreateBitPreservingCastChain(DL, ValueOperand, SizedIntTy);
2273 Args.push_back(IntValue);
2274 } else {
2275 AllocaValue = AllocaBuilder.CreateAlloca(ValueOperand->getType());
2276 AllocaValue->setAlignment(AllocaAlignment);
2277 Builder.CreateLifetimeStart(AllocaValue);
2278 Builder.CreateAlignedStore(ValueOperand, AllocaValue, AllocaAlignment);
2279 Args.push_back(AllocaValue);
2280 }
2281 }
2282
2283 // 'ret' argument.
2284 if (!CASExpected && HasResult && !UseSizedLibcall) {
2285 AllocaResult = AllocaBuilder.CreateAlloca(I->getType());
2286 AllocaResult->setAlignment(AllocaAlignment);
2287 Builder.CreateLifetimeStart(AllocaResult);
2288 Args.push_back(AllocaResult);
2289 }
2290
2291 // 'ordering' ('success_order' for cas) argument.
2292 Args.push_back(OrderingVal);
2293
2294 // 'failure_order' argument, if present.
2295 if (Ordering2Val)
2296 Args.push_back(Ordering2Val);
2297
2298 // Now, the return type.
2299 if (CASExpected) {
2300 ResultTy = Type::getInt1Ty(Ctx);
2301 Attr = Attr.addRetAttribute(Ctx, Attribute::ZExt);
2302 } else if (HasResult && UseSizedLibcall)
2303 ResultTy = SizedIntTy;
2304 else
2305 ResultTy = Type::getVoidTy(Ctx);
2306
2307 // Done with setting up arguments and return types, create the call:
2309 for (Value *Arg : Args)
2310 ArgTys.push_back(Arg->getType());
2311 FunctionType *FnType = FunctionType::get(ResultTy, ArgTys, false);
2312 FunctionCallee LibcallFn = M->getOrInsertFunction(
2314 Attr);
2315 CallInst *Call = Builder.CreateCall(LibcallFn, Args);
2316 Call->setAttributes(Attr);
2317 Value *Result = Call;
2318
2319 // And then, extract the results...
2320 if (ValueOperand && !UseSizedLibcall)
2321 Builder.CreateLifetimeEnd(AllocaValue);
2322
2323 if (CASExpected) {
2324 // The final result from the CAS is {load of 'expected' alloca, bool result
2325 // from call}
2326 Type *FinalResultTy = I->getType();
2327 Value *V = PoisonValue::get(FinalResultTy);
2328 Value *ExpectedOut = Builder.CreateAlignedLoad(
2329 CASExpected->getType(), AllocaCASExpected, AllocaAlignment);
2330 Builder.CreateLifetimeEnd(AllocaCASExpected);
2331 V = Builder.CreateInsertValue(V, ExpectedOut, 0);
2332 V = Builder.CreateInsertValue(V, Result, 1);
2334 } else if (HasResult) {
2335 Value *V;
2336 if (UseSizedLibcall) {
2337 // Add bitcasts from Result's scalar type to I's <n x ptr> vector type
2338 auto *PtrTy = dyn_cast<PointerType>(I->getType()->getScalarType());
2339 auto *VTy = dyn_cast<VectorType>(I->getType());
2340 if (VTy && PtrTy && !Result->getType()->isVectorTy()) {
2341 unsigned AS = PtrTy->getAddressSpace();
2342 Value *BC = Builder.CreateBitCast(
2343 Result, VTy->getWithNewType(DL.getIntPtrType(Ctx, AS)));
2344 V = Builder.CreateIntToPtr(BC, I->getType());
2345 } else
2346 V = Builder.CreateBitOrPointerCast(Result, I->getType());
2347 } else {
2348 V = Builder.CreateAlignedLoad(I->getType(), AllocaResult,
2349 AllocaAlignment);
2350 Builder.CreateLifetimeEnd(AllocaResult);
2351 }
2352 I->replaceAllUsesWith(V);
2353 }
2354 I->eraseFromParent();
2355 return true;
2356}
#define Success
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static Value * performMaskedAtomicOp(AtomicRMWInst::BinOp Op, IRBuilderBase &Builder, Value *Loaded, Value *ValOperand_Shifted, Value *Inc, const PartwordMaskValues &PMV)
Emit IR to implement a masked version of a given atomicrmw operation.
static PartwordMaskValues createMaskInstrs(IRBuilderBase &Builder, Instruction *I, Type *ValueType, Value *Addr, Align AddrAlign, unsigned MinWordSize)
This is a helper function which builds instructions to provide values necessary for partword atomic o...
static bool canUseSizedAtomicCall(unsigned Size, Align Alignment, const DataLayout &DL)
static void createCmpXchgInstFun(IRBuilderBase &Builder, Value *Addr, Value *Loaded, Value *NewVal, Align AddrAlign, AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile, Value *&Success, Value *&NewLoaded, Instruction *MetadataSrc)
static Value * extractMaskedValue(IRBuilderBase &Builder, Value *WideWord, const PartwordMaskValues &PMV)
Expand Atomic static false unsigned getAtomicOpSize(LoadInst *LI)
static void writeUnsupportedAtomicSizeReason(const TargetLowering *TLI, Inst *I, raw_ostream &OS)
static bool atomicSizeSupported(const TargetLowering *TLI, Inst *I)
static Value * insertMaskedValue(IRBuilderBase &Builder, Value *WideWord, Value *Updated, const PartwordMaskValues &PMV)
static void copyMetadataForAtomic(Instruction &Dest, const Instruction &Source)
Copy metadata that's safe to preserve when widening atomics.
static ArrayRef< RTLIB::Libcall > GetRMWLibcall(AtomicRMWInst::BinOp Op)
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static bool runOnFunction(Function &F, bool PostInlining)
#define DEBUG_TYPE
Module.h This file contains the declarations for the Module class.
static bool isIdempotentRMW(AtomicRMWInst &RMWI)
Return true if and only if the given instruction does not modify the memory location referenced.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Machine Check Debug Module
This file provides utility for Memory Model Relaxation Annotations (MMRAs).
#define T
FunctionAnalysisManager FAM
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
This file contains the declarations for profiling metadata utility functions.
const char * Msg
This file defines the SmallString class.
This file defines the SmallVector class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
Target-Independent Code Generator Pass Configuration Options pass.
void setAlignment(Align Align)
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
An instruction that atomically checks whether a specified value is in a memory location,...
AtomicOrdering getMergedOrdering() const
Returns a single ordering which is at least as strong as both the success and failure orderings for t...
void setWeak(bool IsWeak)
bool isVolatile() const
Return true if this is a cmpxchg from a volatile memory location.
AtomicOrdering getFailureOrdering() const
Returns the failure ordering constraint of this cmpxchg instruction.
static AtomicOrdering getStrongestFailureOrdering(AtomicOrdering SuccessOrdering)
Returns the strongest permitted ordering on failure, given the desired ordering on success.
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
bool isWeak() const
Return true if this cmpxchg may spuriously fail.
void setVolatile(bool V)
Specify whether this is a volatile cmpxchg.
AtomicOrdering getSuccessOrdering() const
Returns the success ordering constraint of this cmpxchg instruction.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this cmpxchg instruction.
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
an instruction that atomically reads a memory location, combines it with another value,...
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
bool isVolatile() const
Return true if this is a RMW on a volatile memory location.
void setVolatile(bool V)
Specify whether this is a volatile RMW or not.
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Add
*p = old + v
@ FAdd
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ FSub
*p = old - v
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
@ Nand
*p = ~(old & v)
Value * getPointerOperand()
void setOperation(BinOp Operation)
BinOp getOperation() const
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
static LLVM_ABI StringRef getOperationName(BinOp Op)
AtomicOrdering getOrdering() const
Returns the ordering constraint of this rmw instruction.
iterator end()
Definition BasicBlock.h:459
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:446
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
reverse_iterator rbegin()
Definition BasicBlock.h:462
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
Definition BasicBlock.h:206
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
reverse_iterator rend()
Definition BasicBlock.h:464
void setAttributes(AttributeList A)
Set the attributes for this call.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
ArrayRef< unsigned > getIndices() const
unsigned getNumIndices() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
BasicBlockListType::iterator iterator
Definition Function.h:70
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:730
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNull=false)
Definition IRBuilder.h:2240
AtomicCmpXchgInst * CreateAtomicCmpXchg(Value *Ptr, Value *Cmp, Value *New, MaybeAlign Align, AtomicOrdering SuccessOrdering, AtomicOrdering FailureOrdering, SyncScope::ID SSID=SyncScope::System)
Definition IRBuilder.h:1960
Value * CreateInsertValue(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2715
LLVM_ABI CallInst * CreateLifetimeStart(Value *Ptr)
Create a lifetime.start intrinsic.
LLVM_ABI CallInst * CreateLifetimeEnd(Value *Ptr)
Create a lifetime.end intrinsic.
LoadInst * CreateAlignedLoad(Type *Ty, Value *Ptr, MaybeAlign Align, const char *Name)
Definition IRBuilder.h:1926
CondBrInst * CreateCondBr(Value *Cond, BasicBlock *True, BasicBlock *False, MDNode *BranchWeights=nullptr, MDNode *Unpredictable=nullptr)
Create a conditional 'br Cond, TrueDest, FalseDest' instruction.
Definition IRBuilder.h:1203
UnreachableInst * CreateUnreachable()
Definition IRBuilder.h:1345
Value * CreateExtractValue(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2708
BasicBlock::iterator GetInsertPoint() const
Definition IRBuilder.h:174
Value * CreateIntToPtr(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2230
Value * CreateCast(Instruction::CastOps Op, Value *V, Type *DestTy, const Twine &Name="", MDNode *FPMathTag=nullptr, FMFSource FMFSource={})
Definition IRBuilder.h:2276
BasicBlock * GetInsertBlock() const
Definition IRBuilder.h:173
LLVM_ABI Value * CreateBitPreservingCastChain(const DataLayout &DL, Value *V, Type *NewTy)
Create a chain of casts to convert V to NewTy, preserving the bit pattern of V.
Value * CreateICmpNE(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2378
UncondBrInst * CreateBr(BasicBlock *Dest)
Create an unconditional 'br label X' instruction.
Definition IRBuilder.h:1197
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2324
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Definition IRBuilder.h:2539
Value * CreateICmpEQ(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2374
void setIsFPConstrained(bool IsCon)
Enable/Disable use of constrained floating point math.
Definition IRBuilder.h:286
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2235
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Definition IRBuilder.h:1898
Value * CreateShl(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1498
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2113
LLVMContext & getContext() const
Definition IRBuilder.h:175
Value * CreateAnd(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:1557
Value * CreatePtrToInt(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2225
CallInst * CreateCall(FunctionType *FTy, Value *Callee, ArrayRef< Value * > Args={}, const Twine &Name="", MDNode *FPMathTag=nullptr)
Definition IRBuilder.h:2553
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Definition IRBuilder.h:179
StoreInst * CreateAlignedStore(Value *Val, Value *Ptr, MaybeAlign Align, bool isVolatile=false)
Definition IRBuilder.h:1945
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
Definition IRBuilder.h:1579
AtomicRMWInst * CreateAtomicRMW(AtomicRMWInst::BinOp Op, Value *Ptr, Value *Val, MaybeAlign Align, AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System, bool Elementwise=false)
Definition IRBuilder.h:1973
Provides an 'InsertHelper' that calls a user-provided callback after performing the default insertion...
Definition IRBuilder.h:75
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2901
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void moveAfter(Instruction *MovePos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
iterator_range< user_iterator > users()
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
LLVM_ABI void getSyncScopeNames(SmallVectorImpl< StringRef > &SSNs) const
getSyncScopeNames - Populates client supplied SmallVector with synchronization scope names registered...
Tracks which library functions to use for a particular subtarget or function.
RTLIB::LibcallImpl getLibcallImpl(RTLIB::Libcall Call) const
Return the lowering's selection of implementation call for Call.
An instruction for reading from memory.
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this load instruction.
void setVolatile(bool V)
Specify whether this is a volatile load or not.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this load instruction.
LoadStoreInstProperties getProperties() const
Returns the properties of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
Metadata node.
Definition Metadata.h:1081
Records a mapping from an opaque lowering context to its LibcallLoweringInfo.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
LLVMContext & getContext() const
Get the global data context.
Definition Module.h:332
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
virtual void getAnalysisUsage(AnalysisUsage &) const
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
Definition Pass.cpp:113
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses none()
Convenience factory function for the empty preserved set.
Definition Analysis.h:115
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual Value * emitStoreConditional(IRBuilderBase &Builder, Value *Val, Value *Addr, AtomicOrdering Ord) const
Perform a store-conditional operation to Addr.
EVT getMemValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
virtual void emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const
Perform a bit test atomicrmw using a target-specific intrinsic.
virtual AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
virtual bool shouldInsertFencesForAtomic(const Instruction *I) const
Whether AtomicExpandPass should automatically insert fences and reduce ordering for this atomic.
virtual AtomicOrdering atomicOperationOrderAfterFenceSplit(const Instruction *I) const
virtual void emitExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) const
Perform a cmpxchg expansion using a target-specific method.
unsigned getMinCmpXchgSizeInBits() const
Returns the size of the smallest cmpxchg or ll/sc instruction the backend supports.
virtual Value * emitMaskedAtomicRMWIntrinsic(IRBuilderBase &Builder, AtomicRMWInst *AI, Value *AlignedAddr, Value *Incr, Value *Mask, Value *ShiftAmt, AtomicOrdering Ord) const
Perform a masked atomicrmw using a target-specific intrinsic.
virtual AtomicExpansionKind shouldExpandAtomicCmpXchgInIR(const AtomicCmpXchgInst *AI) const
Returns how the given atomic cmpxchg should be expanded by the IR-level AtomicExpand pass.
virtual Value * emitLoadLinked(IRBuilderBase &Builder, Type *ValueTy, Value *Addr, AtomicOrdering Ord) const
Perform a load-linked operation on Addr, returning a "Value *" with the corresponding pointee type.
virtual void emitExpandAtomicRMW(AtomicRMWInst *AI) const
Perform a atomicrmw expansion using a target-specific way.
virtual void emitAtomicCmpXchgNoStoreLLBalance(IRBuilderBase &Builder) const
virtual void emitExpandAtomicStore(StoreInst *SI) const
Perform a atomic store using a target-specific way.
virtual AtomicExpansionKind shouldCastAtomicRMWIInIR(AtomicRMWInst *RMWI) const
Returns how the given atomic atomicrmw should be cast by the IR-level AtomicExpand pass.
virtual bool shouldInsertTrailingSeqCstFenceForAtomicStore(const Instruction *I) const
Whether AtomicExpandPass should automatically insert a seq_cst trailing fence without reducing the or...
virtual AtomicExpansionKind shouldExpandAtomicLoadInIR(LoadInst *LI) const
Returns how the given (atomic) load should be expanded by the IR-level AtomicExpand pass.
virtual Value * emitMaskedAtomicCmpXchgIntrinsic(IRBuilderBase &Builder, AtomicCmpXchgInst *CI, Value *AlignedAddr, Value *CmpVal, Value *NewVal, Value *Mask, AtomicOrdering Ord) const
Perform a masked cmpxchg using a target-specific intrinsic.
virtual bool shouldIssueAtomicLoadForAtomicEmulationLoop(void) const
unsigned getMaxAtomicSizeInBitsSupported() const
Returns the maximum atomic operation size (in bits) supported by the backend.
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
virtual void emitExpandAtomicLoad(LoadInst *LI) const
Perform a atomic load using a target-specific way.
virtual AtomicExpansionKind shouldExpandAtomicStoreInIR(StoreInst *SI) const
Returns how the given (atomic) store should be expanded by the IR-level AtomicExpand pass into.
virtual void emitCmpArithAtomicRMWIntrinsic(AtomicRMWInst *AI) const
Perform a atomicrmw which the result is only used by comparison, using a target-specific intrinsic.
virtual AtomicExpansionKind shouldCastAtomicStoreInIR(StoreInst *SI) const
Returns how the given (atomic) store should be cast by the IR-level AtomicExpand pass into.
virtual Instruction * emitTrailingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const
virtual AtomicExpansionKind shouldCastAtomicLoadInIR(LoadInst *LI) const
Returns how the given (atomic) load should be cast by the IR-level AtomicExpand pass.
virtual Instruction * emitLeadingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const
Inserts in the IR a target-specific intrinsic specifying a fence.
virtual LoadInst * lowerIdempotentRMWIntoFencedLoad(AtomicRMWInst *RMWI) const
On some platforms, an AtomicRMW that never actually modifies the value (such as fetch_add of 0) can b...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
Primary interface to the complete machine description for the target machine.
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
Target-Independent Code Generator Pass Configuration Options.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:280
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
void setOperand(unsigned i, Value *Val)
Definition User.h:212
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
bool use_empty() const
Definition Value.h:348
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI bool canInstructionHaveMMRAs(const Instruction &I)
LLVM_ABI void setExplicitlyUnknownBranchWeightsIfProfiled(Instruction &I, StringRef PassName, const Function *F=nullptr)
Like setExplicitlyUnknownBranchWeights(...), but only sets unknown branch weights in the new instruct...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
OuterAnalysisManagerProxy< ModuleAnalysisManager, Function > ModuleAnalysisManagerFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
bool isReleaseOrStronger(AtomicOrdering AO)
AtomicOrderingCABI toCABI(AtomicOrdering AO)
LLVM_ABI const LibcallLoweringInfo & getLibcallLowering(const ModuleLibcallLoweringInfo &ModuleInfo, const TargetSubtargetInfo &Subtarget)
Resolve the LibcallLoweringInfo for Subtarget from the module-level ModuleInfo, applying the subtarge...
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
IRBuilder(LLVMContext &, FolderTy, InserterTy) -> IRBuilder< FolderTy, InserterTy >
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI Value * buildAtomicRMWValue(AtomicRMWInst::BinOp Op, IRBuilderBase &Builder, Value *Loaded, Value *Val)
Emit IR to implement the given atomicrmw operation on values in registers, returning the new value.
AtomicOrdering
Atomic ordering for LLVM's memory model.
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
bool isAcquireOrStronger(AtomicOrdering AO)
constexpr unsigned BitWidth
LLVM_ABI bool lowerAtomicCmpXchgInst(AtomicCmpXchgInst *CXI)
Convert the given Cmpxchg into primitive load and compare.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool lowerAtomicRMWInst(AtomicRMWInst *RMWI)
Convert the given RMWI into primitive load and stores, assuming that doing so is legal.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI FunctionPass * createAtomicExpandLegacyPass()
AtomicExpandPass - At IR level this pass replace atomic instructions with __atomic_* library calls,...
LLVM_ABI char & AtomicExpandID
AtomicExpandID – Lowers atomic operations in terms of either cmpxchg load-linked/store-conditional lo...
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
Definition ValueTypes.h:435
Matching combinators.
static StringRef getLibcallImplName(RTLIB::LibcallImpl CallImpl)
Get the libcall routine name for the specified libcall implementation.