LLVM 24.0.0git
AtomicExpandPass.cpp
Go to the documentation of this file.
1//===- AtomicExpandPass.cpp - Expand atomic instructions ------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains a pass (at IR level) to replace atomic instructions with
10// __atomic_* library calls, or target specific instruction which implement the
11// same semantics in a way which better fits the target backend. This can
12// include the use of (intrinsic-based) load-linked/store-conditional loops,
13// AtomicCmpXchg, or type coercions.
14//
15//===----------------------------------------------------------------------===//
16
17#include "llvm/ADT/ArrayRef.h"
28#include "llvm/IR/Attributes.h"
29#include "llvm/IR/BasicBlock.h"
30#include "llvm/IR/Constant.h"
31#include "llvm/IR/Constants.h"
32#include "llvm/IR/DataLayout.h"
34#include "llvm/IR/Function.h"
35#include "llvm/IR/IRBuilder.h"
36#include "llvm/IR/Instruction.h"
38#include "llvm/IR/MDBuilder.h"
40#include "llvm/IR/Module.h"
42#include "llvm/IR/Type.h"
43#include "llvm/IR/User.h"
44#include "llvm/IR/Value.h"
46#include "llvm/Pass.h"
49#include "llvm/Support/Debug.h"
54#include <cassert>
55#include <cstdint>
56#include <iterator>
57
58using namespace llvm;
59
60#define DEBUG_TYPE "atomic-expand"
61
62namespace {
63
64class AtomicExpandImpl {
65 const TargetLowering *TLI = nullptr;
66 const LibcallLoweringInfo *LibcallLowering = nullptr;
67 const DataLayout *DL = nullptr;
68
69private:
70 /// Callback type for emitting a cmpxchg instruction during RMW expansion.
71 /// Parameters: (Builder, Addr, Loaded, NewVal, AddrAlign, MemOpOrder,
72 /// SSID, IsVolatile, /* OUT */ Success, /* OUT */ NewLoaded,
73 /// MetadataSrc)
74 using CreateCmpXchgInstFun = function_ref<void(
76 SyncScope::ID, bool, Value *&, Value *&, Instruction *)>;
77
78 void handleFailure(Instruction &FailedInst, const Twine &Msg,
79 Instruction *DiagnosticInst = nullptr) const {
80 LLVMContext &Ctx = FailedInst.getContext();
81
82 // TODO: Do not use generic error type.
83 Ctx.emitError(DiagnosticInst ? DiagnosticInst : &FailedInst, Msg);
84
85 if (!FailedInst.getType()->isVoidTy())
86 FailedInst.replaceAllUsesWith(PoisonValue::get(FailedInst.getType()));
87 FailedInst.eraseFromParent();
88 }
89
90 template <typename Inst>
91 void handleUnsupportedAtomicSize(Inst *I, const Twine &AtomicOpName,
92 Instruction *DiagnosticInst = nullptr) const;
93
94 bool bracketInstWithFences(Instruction *I, AtomicOrdering Order);
95 bool tryInsertTrailingSeqCstFence(Instruction *AtomicI);
96 template <typename AtomicInst>
97 bool tryInsertFencesForAtomic(AtomicInst *AtomicI, bool OrderingRequiresFence,
98 AtomicOrdering NewOrdering);
99 IntegerType *getCorrespondingIntegerType(Type *T, const DataLayout &DL);
100 LoadInst *convertAtomicLoadToIntegerType(LoadInst *LI);
101 bool tryExpandAtomicLoad(LoadInst *LI);
102 bool expandAtomicLoadToLL(LoadInst *LI);
103 bool expandAtomicLoadToCmpXchg(LoadInst *LI);
104 StoreInst *convertAtomicStoreToIntegerType(StoreInst *SI);
105 bool tryExpandAtomicStore(StoreInst *SI);
106 void expandAtomicStoreToXChg(StoreInst *SI);
107 bool tryExpandAtomicRMW(AtomicRMWInst *AI);
108 AtomicRMWInst *convertAtomicXchgToIntegerType(AtomicRMWInst *RMWI);
109 Value *
110 insertRMWLLSCLoop(IRBuilderBase &Builder, Type *ResultTy, Value *Addr,
111 Align AddrAlign, AtomicOrdering MemOpOrder,
112 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp);
113 void expandAtomicOpToLLSC(
114 Instruction *I, Type *ResultTy, Value *Addr, Align AddrAlign,
115 AtomicOrdering MemOpOrder,
116 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp);
117 void expandPartwordAtomicRMW(
119 AtomicRMWInst *widenPartwordAtomicRMW(AtomicRMWInst *AI);
120 bool expandPartwordCmpXchg(AtomicCmpXchgInst *I);
121 void expandAtomicRMWToMaskedIntrinsic(AtomicRMWInst *AI);
122 void expandAtomicCmpXchgToMaskedIntrinsic(AtomicCmpXchgInst *CI);
123
124 AtomicCmpXchgInst *convertCmpXchgToIntegerType(AtomicCmpXchgInst *CI);
125 Value *insertRMWCmpXchgLoop(
126 IRBuilderBase &Builder, Type *ResultType, Value *Addr, Align AddrAlign,
127 AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile,
128 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp,
129 CreateCmpXchgInstFun CreateCmpXchg, Instruction *MetadataSrc);
130 bool tryExpandAtomicCmpXchg(AtomicCmpXchgInst *CI);
131
132 bool expandAtomicCmpXchg(AtomicCmpXchgInst *CI);
133 bool isIdempotentRMW(AtomicRMWInst *RMWI);
134 bool simplifyIdempotentRMW(AtomicRMWInst *RMWI);
135
136 bool expandAtomicOpToLibcall(Instruction *I, unsigned Size, Align Alignment,
137 Value *PointerOperand, Value *ValueOperand,
138 Value *CASExpected, AtomicOrdering Ordering,
139 AtomicOrdering Ordering2,
140 ArrayRef<RTLIB::Libcall> Libcalls);
141 void expandAtomicLoadToLibcall(LoadInst *LI);
142 void expandAtomicStoreToLibcall(StoreInst *LI);
143 void expandAtomicRMWToLibcall(AtomicRMWInst *I);
144 void expandAtomicCASToLibcall(AtomicCmpXchgInst *I,
145 const Twine &AtomicOpName = "cmpxchg",
146 Instruction *DiagnosticInst = nullptr);
147
148 bool expandAtomicRMWToCmpXchg(AtomicRMWInst *AI,
149 CreateCmpXchgInstFun CreateCmpXchg);
150
151 bool processAtomicInstr(Instruction *I);
152
153public:
154 bool run(Function &F, const ModuleLibcallLoweringInfo &LibcallResult,
155 const TargetMachine *TM);
156};
157
158class AtomicExpandLegacy : public FunctionPass {
159public:
160 static char ID; // Pass identification, replacement for typeid
161
162 AtomicExpandLegacy() : FunctionPass(ID) {}
163
164 void getAnalysisUsage(AnalysisUsage &AU) const override {
167 }
168
169 bool runOnFunction(Function &F) override;
170};
171
172// IRBuilder to be used for replacement atomic instructions.
173struct ReplacementIRBuilder
174 : IRBuilder<InstSimplifyFolder, IRBuilderCallbackInserter> {
175 MDNode *MMRAMD = nullptr;
176 MDNode *PCSectionsMD = nullptr;
177
178 // Preserves the DebugLoc from I, and preserves still valid metadata.
179 // Enable StrictFP builder mode when appropriate.
180 explicit ReplacementIRBuilder(Instruction *I, const DataLayout &DL)
181 : IRBuilder(
182 I->getContext(), InstSimplifyFolder(DL),
183 IRBuilderCallbackInserter([this](Instruction *I) { addMD(I); })) {
184 SetInsertPoint(I);
185 if (BB->getParent()->getAttributes().hasFnAttr(Attribute::StrictFP))
186 this->setIsFPConstrained(true);
187
188 MMRAMD = I->getMetadata(LLVMContext::MD_mmra);
189 PCSectionsMD = I->getMetadata(LLVMContext::MD_pcsections);
190 }
191
192 void addMD(Instruction *I) {
194 I->setMetadata(LLVMContext::MD_mmra, MMRAMD);
195 I->setMetadata(LLVMContext::MD_pcsections, PCSectionsMD);
196 }
197};
198
199} // end anonymous namespace
200
201char AtomicExpandLegacy::ID = 0;
202
203char &llvm::AtomicExpandID = AtomicExpandLegacy::ID;
204
206 "Expand Atomic instructions", false, false)
209INITIALIZE_PASS_END(AtomicExpandLegacy, DEBUG_TYPE,
210 "Expand Atomic instructions", false, false)
211
212// Helper functions to retrieve the size of atomic instructions.
213static unsigned getAtomicOpSize(LoadInst *LI) {
214 const DataLayout &DL = LI->getDataLayout();
215 return DL.getTypeStoreSize(LI->getType());
216}
217
218static unsigned getAtomicOpSize(StoreInst *SI) {
219 const DataLayout &DL = SI->getDataLayout();
220 return DL.getTypeStoreSize(SI->getValueOperand()->getType());
221}
222
223static unsigned getAtomicOpSize(AtomicRMWInst *RMWI) {
224 const DataLayout &DL = RMWI->getDataLayout();
225 return DL.getTypeStoreSize(RMWI->getValOperand()->getType());
226}
227
228static unsigned getAtomicOpSize(AtomicCmpXchgInst *CASI) {
229 const DataLayout &DL = CASI->getDataLayout();
230 return DL.getTypeStoreSize(CASI->getCompareOperand()->getType());
231}
232
233/// Copy metadata that's safe to preserve when widening atomics.
235 const Instruction &Source) {
237 Source.getAllMetadata(MD);
238 LLVMContext &Ctx = Dest.getContext();
239 MDBuilder MDB(Ctx);
240
241 for (auto [ID, N] : MD) {
242 switch (ID) {
243 case LLVMContext::MD_dbg:
244 case LLVMContext::MD_tbaa:
245 case LLVMContext::MD_tbaa_struct:
246 case LLVMContext::MD_alias_scope:
247 case LLVMContext::MD_noalias:
248 case LLVMContext::MD_noalias_addrspace:
249 case LLVMContext::MD_access_group:
250 case LLVMContext::MD_mmra:
251 Dest.setMetadata(ID, N);
252 break;
253 default:
254 if (ID == Ctx.getMDKindID("amdgpu.no.remote.memory"))
255 Dest.setMetadata(ID, N);
256 else if (ID == Ctx.getMDKindID("amdgpu.no.fine.grained.memory"))
257 Dest.setMetadata(ID, N);
258
259 // Losing amdgpu.ignore.denormal.mode, but it doesn't matter for current
260 // uses.
261 break;
262 }
263 }
264}
265
266template <typename Inst>
267static bool atomicSizeSupported(const TargetLowering *TLI, Inst *I) {
268 unsigned Size = getAtomicOpSize(I);
269 Align Alignment = I->getAlign();
270 unsigned MaxSize = TLI->getMaxAtomicSizeInBitsSupported() / 8;
271 return Alignment >= Size && Size <= MaxSize;
272}
273
274template <typename Inst>
276 raw_ostream &OS) {
277 unsigned Size = getAtomicOpSize(I);
278 Align Alignment = I->getAlign();
279 bool NeedSeparator = false;
280
281 if (Alignment < Size) {
282 OS << "instruction alignment " << Alignment.value()
283 << " is smaller than the required " << Size
284 << "-byte alignment for this atomic operation";
285 NeedSeparator = true;
286 }
287
288 unsigned MaxSize = TLI->getMaxAtomicSizeInBitsSupported() / 8;
289 if (Size > MaxSize) {
290 if (NeedSeparator)
291 OS << "; ";
292 OS << "target supports atomics up to " << MaxSize
293 << " bytes, but this atomic accesses " << Size << " bytes";
294 }
295}
296
297template <typename Inst>
298void AtomicExpandImpl::handleUnsupportedAtomicSize(
299 Inst *I, const Twine &AtomicOpName, Instruction *DiagnosticInst) const {
300 assert(!atomicSizeSupported(TLI, I) && "expected unsupported atomic size");
301 SmallString<128> FailureReason;
302 raw_svector_ostream OS(FailureReason);
304 handleFailure(*I, Twine("unsupported ") + AtomicOpName + ": " + FailureReason,
305 DiagnosticInst);
306}
307
308bool AtomicExpandImpl::tryInsertTrailingSeqCstFence(Instruction *AtomicI) {
310 return false;
311
312 IRBuilder Builder(AtomicI);
313 if (auto *TrailingFence = TLI->emitTrailingFence(
314 Builder, AtomicI, AtomicOrdering::SequentiallyConsistent)) {
315 TrailingFence->moveAfter(AtomicI);
316 return true;
317 }
318 return false;
319}
320
321template <typename AtomicInst>
322bool AtomicExpandImpl::tryInsertFencesForAtomic(AtomicInst *AtomicI,
323 bool OrderingRequiresFence,
324 AtomicOrdering NewOrdering) {
325 bool ShouldInsertFences = TLI->shouldInsertFencesForAtomic(AtomicI);
326 if (OrderingRequiresFence && ShouldInsertFences) {
327 AtomicOrdering FenceOrdering = AtomicI->getOrdering();
328 AtomicI->setOrdering(NewOrdering);
329 return bracketInstWithFences(AtomicI, FenceOrdering);
330 }
331 if (!ShouldInsertFences)
332 return tryInsertTrailingSeqCstFence(AtomicI);
333 return false;
334}
335
336bool AtomicExpandImpl::processAtomicInstr(Instruction *I) {
337 if (auto *LI = dyn_cast<LoadInst>(I)) {
338 if (!LI->isAtomic())
339 return false;
340
341 if (!atomicSizeSupported(TLI, LI)) {
342 expandAtomicLoadToLibcall(LI);
343 return true;
344 }
345
346 bool MadeChange = false;
347 if (TLI->shouldCastAtomicLoadInIR(LI) ==
348 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
349 LI = convertAtomicLoadToIntegerType(LI);
350 MadeChange = true;
351 }
352
353 MadeChange |= tryInsertFencesForAtomic(
354 LI, isAcquireOrStronger(LI->getOrdering()), AtomicOrdering::Monotonic);
355
356 MadeChange |= tryExpandAtomicLoad(LI);
357 return MadeChange;
358 }
359
360 if (auto *SI = dyn_cast<StoreInst>(I)) {
361 if (!SI->isAtomic())
362 return false;
363
364 if (!atomicSizeSupported(TLI, SI)) {
365 expandAtomicStoreToLibcall(SI);
366 return true;
367 }
368
369 bool MadeChange = false;
370 if (TLI->shouldCastAtomicStoreInIR(SI) ==
371 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
372 SI = convertAtomicStoreToIntegerType(SI);
373 MadeChange = true;
374 }
375
376 MadeChange |= tryInsertFencesForAtomic(
377 SI, isReleaseOrStronger(SI->getOrdering()), AtomicOrdering::Monotonic);
378
379 MadeChange |= tryExpandAtomicStore(SI);
380 return MadeChange;
381 }
382
383 if (auto *RMWI = dyn_cast<AtomicRMWInst>(I)) {
384 if (!atomicSizeSupported(TLI, RMWI)) {
385 expandAtomicRMWToLibcall(RMWI);
386 return true;
387 }
388
389 bool MadeChange = false;
390 if (TLI->shouldCastAtomicRMWIInIR(RMWI) ==
391 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
392 RMWI = convertAtomicXchgToIntegerType(RMWI);
393 MadeChange = true;
394 }
395
396 MadeChange |= tryInsertFencesForAtomic(
397 RMWI,
398 isReleaseOrStronger(RMWI->getOrdering()) ||
399 isAcquireOrStronger(RMWI->getOrdering()),
401
402 // There are two different ways of expanding RMW instructions:
403 // - into a load if it is idempotent
404 // - into a Cmpxchg/LL-SC loop otherwise
405 // we try them in that order.
406 MadeChange |= (isIdempotentRMW(RMWI) && simplifyIdempotentRMW(RMWI)) ||
407 tryExpandAtomicRMW(RMWI);
408 return MadeChange;
409 }
410
411 if (auto *CASI = dyn_cast<AtomicCmpXchgInst>(I)) {
412 if (!atomicSizeSupported(TLI, CASI)) {
413 expandAtomicCASToLibcall(CASI);
414 return true;
415 }
416
417 // TODO: when we're ready to make the change at the IR level, we can
418 // extend convertCmpXchgToInteger for floating point too.
419 bool MadeChange = false;
420 if (CASI->getCompareOperand()->getType()->isPointerTy()) {
421 // TODO: add a TLI hook to control this so that each target can
422 // convert to lowering the original type one at a time.
423 CASI = convertCmpXchgToIntegerType(CASI);
424 MadeChange = true;
425 }
426
427 auto CmpXchgExpansion = TLI->shouldExpandAtomicCmpXchgInIR(CASI);
428 if (TLI->shouldInsertFencesForAtomic(CASI)) {
429 if (CmpXchgExpansion == TargetLoweringBase::AtomicExpansionKind::None &&
430 (isReleaseOrStronger(CASI->getSuccessOrdering()) ||
431 isAcquireOrStronger(CASI->getSuccessOrdering()) ||
432 isAcquireOrStronger(CASI->getFailureOrdering()))) {
433 // If a compare and swap is lowered to LL/SC, we can do smarter fence
434 // insertion, with a stronger one on the success path than on the
435 // failure path. As a result, fence insertion is directly done by
436 // expandAtomicCmpXchg in that case.
437 AtomicOrdering FenceOrdering = CASI->getMergedOrdering();
438 AtomicOrdering CASOrdering =
440 CASI->setSuccessOrdering(CASOrdering);
441 CASI->setFailureOrdering(CASOrdering);
442 MadeChange |= bracketInstWithFences(CASI, FenceOrdering);
443 }
444 } else if (CmpXchgExpansion !=
445 TargetLoweringBase::AtomicExpansionKind::LLSC) {
446 // CmpXchg LLSC is handled in expandAtomicCmpXchg().
447 MadeChange |= tryInsertTrailingSeqCstFence(CASI);
448 }
449
450 MadeChange |= tryExpandAtomicCmpXchg(CASI);
451 return MadeChange;
452 }
453
454 return false;
455}
456
457bool AtomicExpandImpl::run(Function &F,
458 const ModuleLibcallLoweringInfo &LibcallResult,
459 const TargetMachine *TM) {
460 const auto *Subtarget = TM->getSubtargetImpl(F);
461 if (!Subtarget->enableAtomicExpand())
462 return false;
463 TLI = Subtarget->getTargetLowering();
464 LibcallLowering = &LibcallResult.getLibcallLowering(*Subtarget);
465 DL = &F.getDataLayout();
466
467 bool MadeChange = false;
468
469 for (Function::iterator BBI = F.begin(), BBE = F.end(); BBI != BBE; ++BBI) {
470 BasicBlock *BB = &*BBI;
471
473
474 for (BasicBlock::reverse_iterator I = BB->rbegin(), E = BB->rend(); I != E;
475 I = Next) {
476 Instruction &Inst = *I;
477 Next = std::next(I);
478
479 if (processAtomicInstr(&Inst)) {
480 MadeChange = true;
481
482 // New blocks may have been inserted.
483 BBE = F.end();
484 }
485 }
486 }
487
488 return MadeChange;
489}
490
491bool AtomicExpandLegacy::runOnFunction(Function &F) {
492
493 auto *TPC = getAnalysisIfAvailable<TargetPassConfig>();
494 if (!TPC)
495 return false;
496 auto *TM = &TPC->getTM<TargetMachine>();
497
498 const ModuleLibcallLoweringInfo &LibcallResult =
499 getAnalysis<LibcallLoweringInfoWrapper>().getResult(*F.getParent());
500 AtomicExpandImpl AE;
501 return AE.run(F, LibcallResult, TM);
502}
503
505 return new AtomicExpandLegacy();
506}
507
510 auto &MAMProxy = FAM.getResult<ModuleAnalysisManagerFunctionProxy>(F);
511
512 const ModuleLibcallLoweringInfo *LibcallResult =
513 MAMProxy.getCachedResult<LibcallLoweringModuleAnalysis>(*F.getParent());
514
515 if (!LibcallResult) {
516 F.getContext().emitError("'" + LibcallLoweringModuleAnalysis::name() +
517 "' analysis required");
518 return PreservedAnalyses::all();
519 }
520
521 AtomicExpandImpl AE;
522
523 bool Changed = AE.run(F, *LibcallResult, TM);
524 if (!Changed)
525 return PreservedAnalyses::all();
526
528}
529
530bool AtomicExpandImpl::bracketInstWithFences(Instruction *I,
531 AtomicOrdering Order) {
532 ReplacementIRBuilder Builder(I, *DL);
533
534 auto LeadingFence = TLI->emitLeadingFence(Builder, I, Order);
535
536 auto TrailingFence = TLI->emitTrailingFence(Builder, I, Order);
537 // We have a guard here because not every atomic operation generates a
538 // trailing fence.
539 if (TrailingFence)
540 TrailingFence->moveAfter(I);
541
542 return (LeadingFence || TrailingFence);
543}
544
545/// Get the iX type with the same bitwidth as T.
547AtomicExpandImpl::getCorrespondingIntegerType(Type *T, const DataLayout &DL) {
548 EVT VT = TLI->getMemValueType(DL, T);
549 unsigned BitWidth = VT.getStoreSizeInBits();
550 assert(BitWidth == VT.getSizeInBits() && "must be a power of two");
551 return IntegerType::get(T->getContext(), BitWidth);
552}
553
554/// Convert an atomic load of a non-integral type to an integer load of the
555/// equivalent bitwidth. See the function comment on
556/// convertAtomicStoreToIntegerType for background.
557LoadInst *AtomicExpandImpl::convertAtomicLoadToIntegerType(LoadInst *LI) {
558 auto *M = LI->getModule();
559 Type *NewTy = getCorrespondingIntegerType(LI->getType(), M->getDataLayout());
560
561 ReplacementIRBuilder Builder(LI, *DL);
562
563 Value *Addr = LI->getPointerOperand();
564
565 auto *NewLI = Builder.CreateLoad(NewTy, Addr, LI->getProperties());
566 LLVM_DEBUG(dbgs() << "Replaced " << *LI << " with " << *NewLI << "\n");
567
568 Value *NewVal = LI->getType()->isPtrOrPtrVectorTy()
569 ? Builder.CreateIntToPtr(NewLI, LI->getType())
570 : Builder.CreateBitCast(NewLI, LI->getType());
571 LI->replaceAllUsesWith(NewVal);
572 LI->eraseFromParent();
573 return NewLI;
574}
575
576AtomicRMWInst *
577AtomicExpandImpl::convertAtomicXchgToIntegerType(AtomicRMWInst *RMWI) {
579
580 auto *M = RMWI->getModule();
581 Type *NewTy =
582 getCorrespondingIntegerType(RMWI->getType(), M->getDataLayout());
583
584 ReplacementIRBuilder Builder(RMWI, *DL);
585
586 Value *Addr = RMWI->getPointerOperand();
587 Value *Val = RMWI->getValOperand();
588 Value *NewVal = Val->getType()->isPointerTy()
589 ? Builder.CreatePtrToInt(Val, NewTy)
590 : Builder.CreateBitCast(Val, NewTy);
591
592 auto *NewRMWI = Builder.CreateAtomicRMW(AtomicRMWInst::Xchg, Addr, NewVal,
593 RMWI->getAlign(), RMWI->getOrdering(),
594 RMWI->getSyncScopeID());
595 NewRMWI->setVolatile(RMWI->isVolatile());
596 copyMetadataForAtomic(*NewRMWI, *RMWI);
597 LLVM_DEBUG(dbgs() << "Replaced " << *RMWI << " with " << *NewRMWI << "\n");
598
599 Value *NewRVal = RMWI->getType()->isPointerTy()
600 ? Builder.CreateIntToPtr(NewRMWI, RMWI->getType())
601 : Builder.CreateBitCast(NewRMWI, RMWI->getType());
602 RMWI->replaceAllUsesWith(NewRVal);
603 RMWI->eraseFromParent();
604 return NewRMWI;
605}
606
607bool AtomicExpandImpl::tryExpandAtomicLoad(LoadInst *LI) {
608 switch (TLI->shouldExpandAtomicLoadInIR(LI)) {
609 case TargetLoweringBase::AtomicExpansionKind::None:
610 return false;
611 case TargetLoweringBase::AtomicExpansionKind::LLSC:
612 expandAtomicOpToLLSC(
613 LI, LI->getType(), LI->getPointerOperand(), LI->getAlign(),
614 LI->getOrdering(),
615 [](IRBuilderBase &Builder, Value *Loaded) { return Loaded; });
616 return true;
617 case TargetLoweringBase::AtomicExpansionKind::LLOnly:
618 return expandAtomicLoadToLL(LI);
619 case TargetLoweringBase::AtomicExpansionKind::CmpXChg:
620 return expandAtomicLoadToCmpXchg(LI);
621 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
622 LI->setAtomic(AtomicOrdering::NotAtomic);
623 return true;
624 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
625 TLI->emitExpandAtomicLoad(LI);
626 return true;
627 default:
628 llvm_unreachable("Unhandled case in tryExpandAtomicLoad");
629 }
630}
631
632bool AtomicExpandImpl::tryExpandAtomicStore(StoreInst *SI) {
633 switch (TLI->shouldExpandAtomicStoreInIR(SI)) {
634 case TargetLoweringBase::AtomicExpansionKind::None:
635 return false;
636 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
637 TLI->emitExpandAtomicStore(SI);
638 return true;
639 case TargetLoweringBase::AtomicExpansionKind::Expand:
640 expandAtomicStoreToXChg(SI);
641 return true;
642 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
643 SI->setAtomic(AtomicOrdering::NotAtomic);
644 return true;
645 default:
646 llvm_unreachable("Unhandled case in tryExpandAtomicStore");
647 }
648}
649
650bool AtomicExpandImpl::expandAtomicLoadToLL(LoadInst *LI) {
651 ReplacementIRBuilder Builder(LI, *DL);
652
653 // On some architectures, load-linked instructions are atomic for larger
654 // sizes than normal loads. For example, the only 64-bit load guaranteed
655 // to be single-copy atomic by ARM is an ldrexd (A3.5.3).
656 Value *Val = TLI->emitLoadLinked(Builder, LI->getType(),
657 LI->getPointerOperand(), LI->getOrdering());
659
660 LI->replaceAllUsesWith(Val);
661 LI->eraseFromParent();
662
663 return true;
664}
665
666bool AtomicExpandImpl::expandAtomicLoadToCmpXchg(LoadInst *LI) {
667 ReplacementIRBuilder Builder(LI, *DL);
668 AtomicOrdering Order = LI->getOrdering();
669 if (Order == AtomicOrdering::Unordered)
670 Order = AtomicOrdering::Monotonic;
671
672 Value *Addr = LI->getPointerOperand();
673 Type *Ty = LI->getType();
674
675 // cmpxchg supports only integer and pointer operands. If the load type is
676 // FP or vector, run the cmpxchg on the same-sized integer and bitcast the
677 // result back; mirrors createCmpXchgInstFun.
678 bool NeedBitcast = Ty->isFloatingPointTy() || Ty->isVectorTy();
679 Type *CmpXchgTy = Ty;
680 if (NeedBitcast)
681 CmpXchgTy = Builder.getIntNTy(Ty->getPrimitiveSizeInBits());
682 Constant *DummyVal = Constant::getNullValue(CmpXchgTy);
683
684 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
685 Addr, DummyVal, DummyVal, LI->getAlign(), Order,
687 LI->getSyncScopeID());
688 Pair->setVolatile(LI->isVolatile());
689 Value *Loaded = Builder.CreateExtractValue(Pair, 0, "loaded");
690 if (NeedBitcast)
691 Loaded = Builder.CreateBitCast(Loaded, Ty);
692
693 LI->replaceAllUsesWith(Loaded);
694 LI->eraseFromParent();
695
696 return true;
697}
698
699/// Convert an atomic store of a non-integral type to an integer store of the
700/// equivalent bitwidth. We used to not support floating point or vector
701/// atomics in the IR at all. The backends learned to deal with the bitcast
702/// idiom because that was the only way of expressing the notion of a atomic
703/// float or vector store. The long term plan is to teach each backend to
704/// instruction select from the original atomic store, but as a migration
705/// mechanism, we convert back to the old format which the backends understand.
706/// Each backend will need individual work to recognize the new format.
707StoreInst *AtomicExpandImpl::convertAtomicStoreToIntegerType(StoreInst *SI) {
708 ReplacementIRBuilder Builder(SI, *DL);
709 auto *M = SI->getModule();
710 Type *NewTy = getCorrespondingIntegerType(SI->getValueOperand()->getType(),
711 M->getDataLayout());
712 Value *NewVal = SI->getValueOperand()->getType()->isPtrOrPtrVectorTy()
713 ? Builder.CreatePtrToInt(SI->getValueOperand(), NewTy)
714 : Builder.CreateBitCast(SI->getValueOperand(), NewTy);
715
716 Value *Addr = SI->getPointerOperand();
717
718 StoreInst *NewSI = Builder.CreateStore(NewVal, Addr, SI->getProperties());
719 LLVM_DEBUG(dbgs() << "Replaced " << *SI << " with " << *NewSI << "\n");
720 SI->eraseFromParent();
721 return NewSI;
722}
723
724void AtomicExpandImpl::expandAtomicStoreToXChg(StoreInst *SI) {
725 // This function is only called on atomic stores that are too large to be
726 // atomic if implemented as a native store. So we replace them by an
727 // atomic swap, that can be implemented for example as a ldrex/strex on ARM
728 // or lock cmpxchg8/16b on X86, as these are atomic for larger sizes.
729 // It is the responsibility of the target to only signal expansion via
730 // shouldExpandAtomicRMW in cases where this is required and possible.
731 ReplacementIRBuilder Builder(SI, *DL);
732 AtomicOrdering Ordering = SI->getOrdering();
733 assert(Ordering != AtomicOrdering::NotAtomic);
734 AtomicOrdering RMWOrdering = Ordering == AtomicOrdering::Unordered
735 ? AtomicOrdering::Monotonic
736 : Ordering;
737 AtomicRMWInst *AI = Builder.CreateAtomicRMW(
738 AtomicRMWInst::Xchg, SI->getPointerOperand(), SI->getValueOperand(),
739 SI->getAlign(), RMWOrdering, SI->getSyncScopeID());
740 AI->setVolatile(SI->isVolatile());
741 SI->eraseFromParent();
742
743 // Now we have an appropriate swap instruction, lower it as usual.
744 tryExpandAtomicRMW(AI);
745}
746
747static void createCmpXchgInstFun(IRBuilderBase &Builder, Value *Addr,
748 Value *Loaded, Value *NewVal, Align AddrAlign,
749 AtomicOrdering MemOpOrder, SyncScope::ID SSID,
750 bool IsVolatile, Value *&Success,
751 Value *&NewLoaded, Instruction *MetadataSrc) {
752 Type *OrigTy = NewVal->getType();
753
754 // This code can go away when cmpxchg supports FP and vector types.
755 assert(!OrigTy->isPointerTy());
756 bool NeedBitcast = OrigTy->isFloatingPointTy() || OrigTy->isVectorTy();
757 if (NeedBitcast) {
758 IntegerType *IntTy = Builder.getIntNTy(OrigTy->getPrimitiveSizeInBits());
759 NewVal = Builder.CreateBitCast(NewVal, IntTy);
760 Loaded = Builder.CreateBitCast(Loaded, IntTy);
761 }
762
763 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
764 Addr, Loaded, NewVal, AddrAlign, MemOpOrder,
766 Pair->setVolatile(IsVolatile);
767 if (MetadataSrc)
768 copyMetadataForAtomic(*Pair, *MetadataSrc);
769
770 Success = Builder.CreateExtractValue(Pair, 1, "success");
771 NewLoaded = Builder.CreateExtractValue(Pair, 0, "newloaded");
772
773 if (NeedBitcast)
774 NewLoaded = Builder.CreateBitCast(NewLoaded, OrigTy);
775}
776
777bool AtomicExpandImpl::tryExpandAtomicRMW(AtomicRMWInst *AI) {
778 LLVMContext &Ctx = AI->getModule()->getContext();
779 TargetLowering::AtomicExpansionKind Kind = TLI->shouldExpandAtomicRMWInIR(AI);
780 switch (Kind) {
781 case TargetLoweringBase::AtomicExpansionKind::None:
782 return false;
783 case TargetLoweringBase::AtomicExpansionKind::LLSC: {
784 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
785 unsigned ValueSize = getAtomicOpSize(AI);
786 if (ValueSize < MinCASSize) {
787 expandPartwordAtomicRMW(AI,
788 TargetLoweringBase::AtomicExpansionKind::LLSC);
789 } else {
790 auto PerformOp = [&](IRBuilderBase &Builder, Value *Loaded) {
791 return buildAtomicRMWValue(AI->getOperation(), Builder, Loaded,
792 AI->getValOperand());
793 };
794 expandAtomicOpToLLSC(AI, AI->getType(), AI->getPointerOperand(),
795 AI->getAlign(), AI->getOrdering(), PerformOp);
796 }
797 return true;
798 }
799 case TargetLoweringBase::AtomicExpansionKind::CmpXChg: {
800 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
801 unsigned ValueSize = getAtomicOpSize(AI);
802 if (ValueSize < MinCASSize) {
803 expandPartwordAtomicRMW(AI,
804 TargetLoweringBase::AtomicExpansionKind::CmpXChg);
805 } else {
807 Ctx.getSyncScopeNames(SSNs);
808 auto MemScope = SSNs[AI->getSyncScopeID()].empty()
809 ? "system"
810 : SSNs[AI->getSyncScopeID()];
811 OptimizationRemarkEmitter ORE(AI->getFunction());
812 ORE.emit([&]() {
813 return OptimizationRemark(DEBUG_TYPE, "Passed", AI)
814 << "A compare and swap loop was generated for an atomic "
815 << AI->getOperationName(AI->getOperation()) << " operation at "
816 << MemScope << " memory scope";
817 });
818 expandAtomicRMWToCmpXchg(AI, createCmpXchgInstFun);
819 }
820 return true;
821 }
822 case TargetLoweringBase::AtomicExpansionKind::MaskedIntrinsic: {
823 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
824 unsigned ValueSize = getAtomicOpSize(AI);
825 if (ValueSize < MinCASSize) {
827 // Widen And/Or/Xor and give the target another chance at expanding it.
830 tryExpandAtomicRMW(widenPartwordAtomicRMW(AI));
831 return true;
832 }
833 }
834 expandAtomicRMWToMaskedIntrinsic(AI);
835 return true;
836 }
837 case TargetLoweringBase::AtomicExpansionKind::BitTestIntrinsic: {
839 return true;
840 }
841 case TargetLoweringBase::AtomicExpansionKind::CmpArithIntrinsic: {
843 return true;
844 }
845 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
846 return lowerAtomicRMWInst(AI);
847 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
848 TLI->emitExpandAtomicRMW(AI);
849 return true;
850 default:
851 llvm_unreachable("Unhandled case in tryExpandAtomicRMW");
852 }
853}
854
855namespace {
856
857struct PartwordMaskValues {
858 // These three fields are guaranteed to be set by createMaskInstrs.
859 Type *WordType = nullptr;
860 Type *ValueType = nullptr;
861 Type *IntValueType = nullptr;
862 Value *AlignedAddr = nullptr;
863 Align AlignedAddrAlignment;
864 // The remaining fields can be null.
865 Value *ShiftAmt = nullptr;
866 Value *Mask = nullptr;
867 Value *Inv_Mask = nullptr;
868};
869
870[[maybe_unused]]
871raw_ostream &operator<<(raw_ostream &O, const PartwordMaskValues &PMV) {
872 auto PrintObj = [&O](auto *V) {
873 if (V)
874 O << *V;
875 else
876 O << "nullptr";
877 O << '\n';
878 };
879 O << "PartwordMaskValues {\n";
880 O << " WordType: ";
881 PrintObj(PMV.WordType);
882 O << " ValueType: ";
883 PrintObj(PMV.ValueType);
884 O << " AlignedAddr: ";
885 PrintObj(PMV.AlignedAddr);
886 O << " AlignedAddrAlignment: " << PMV.AlignedAddrAlignment.value() << '\n';
887 O << " ShiftAmt: ";
888 PrintObj(PMV.ShiftAmt);
889 O << " Mask: ";
890 PrintObj(PMV.Mask);
891 O << " Inv_Mask: ";
892 PrintObj(PMV.Inv_Mask);
893 O << "}\n";
894 return O;
895}
896
897} // end anonymous namespace
898
899/// This is a helper function which builds instructions to provide
900/// values necessary for partword atomic operations. It takes an
901/// incoming address, Addr, and ValueType, and constructs the address,
902/// shift-amounts and masks needed to work with a larger value of size
903/// WordSize.
904///
905/// AlignedAddr: Addr rounded down to a multiple of WordSize
906///
907/// ShiftAmt: Number of bits to right-shift a WordSize value loaded
908/// from AlignAddr for it to have the same value as if
909/// ValueType was loaded from Addr.
910///
911/// Mask: Value to mask with the value loaded from AlignAddr to
912/// include only the part that would've been loaded from Addr.
913///
914/// Inv_Mask: The inverse of Mask.
915static PartwordMaskValues createMaskInstrs(IRBuilderBase &Builder,
917 Value *Addr, Align AddrAlign,
918 unsigned MinWordSize) {
919 PartwordMaskValues PMV;
920
921 Module *M = I->getModule();
922 LLVMContext &Ctx = M->getContext();
923 const DataLayout &DL = M->getDataLayout();
924 unsigned ValueSize = DL.getTypeStoreSize(ValueType);
925
926 PMV.ValueType = PMV.IntValueType = ValueType;
927 if (PMV.ValueType->isFloatingPointTy() || PMV.ValueType->isVectorTy())
928 PMV.IntValueType =
929 Type::getIntNTy(Ctx, ValueType->getPrimitiveSizeInBits());
930
931 PMV.WordType = MinWordSize > ValueSize ? Type::getIntNTy(Ctx, MinWordSize * 8)
932 : ValueType;
933 if (PMV.ValueType == PMV.WordType) {
934 PMV.AlignedAddr = Addr;
935 PMV.AlignedAddrAlignment = AddrAlign;
936 PMV.ShiftAmt = ConstantInt::get(PMV.ValueType, 0);
937 PMV.Mask = ConstantInt::get(PMV.ValueType, ~0, /*isSigned*/ true);
938 return PMV;
939 }
940
941 PMV.AlignedAddrAlignment = Align(MinWordSize);
942
943 assert(ValueSize < MinWordSize);
944
945 PointerType *PtrTy = cast<PointerType>(Addr->getType());
946 IntegerType *IntTy = DL.getIndexType(Ctx, PtrTy->getAddressSpace());
947 Value *PtrLSB;
948
949 if (AddrAlign < MinWordSize) {
950 PMV.AlignedAddr = Builder.CreateIntrinsic(
951 Intrinsic::ptrmask, {PtrTy, IntTy},
952 {Addr, ConstantInt::getSigned(IntTy, ~(uint64_t)(MinWordSize - 1))},
953 nullptr, "AlignedAddr");
954
955 Value *AddrInt = Builder.CreatePtrToInt(Addr, IntTy);
956 PtrLSB = Builder.CreateAnd(AddrInt, MinWordSize - 1, "PtrLSB");
957 } else {
958 // If the alignment is high enough, the LSB are known 0.
959 PMV.AlignedAddr = Addr;
960 PtrLSB = ConstantInt::getNullValue(IntTy);
961 }
962
963 if (DL.isLittleEndian()) {
964 // turn bytes into bits
965 PMV.ShiftAmt = Builder.CreateShl(PtrLSB, 3);
966 } else {
967 // turn bytes into bits, and count from the other side.
968 PMV.ShiftAmt = Builder.CreateShl(
969 Builder.CreateXor(PtrLSB, MinWordSize - ValueSize), 3);
970 }
971
972 PMV.ShiftAmt = Builder.CreateTrunc(PMV.ShiftAmt, PMV.WordType, "ShiftAmt");
973 PMV.Mask = Builder.CreateShl(
974 ConstantInt::get(PMV.WordType, (1 << (ValueSize * 8)) - 1), PMV.ShiftAmt,
975 "Mask");
976
977 PMV.Inv_Mask = Builder.CreateNot(PMV.Mask, "Inv_Mask");
978
979 return PMV;
980}
981
982static Value *extractMaskedValue(IRBuilderBase &Builder, Value *WideWord,
983 const PartwordMaskValues &PMV) {
984 assert(WideWord->getType() == PMV.WordType && "Widened type mismatch");
985 if (PMV.WordType == PMV.ValueType)
986 return WideWord;
987
988 Value *Shift = Builder.CreateLShr(WideWord, PMV.ShiftAmt, "shifted");
989 Value *Trunc = Builder.CreateTrunc(Shift, PMV.IntValueType, "extracted");
990 return Builder.CreateBitCast(Trunc, PMV.ValueType);
991}
992
993static Value *insertMaskedValue(IRBuilderBase &Builder, Value *WideWord,
994 Value *Updated, const PartwordMaskValues &PMV) {
995 assert(WideWord->getType() == PMV.WordType && "Widened type mismatch");
996 assert(Updated->getType() == PMV.ValueType && "Value type mismatch");
997 if (PMV.WordType == PMV.ValueType)
998 return Updated;
999
1000 Updated = Builder.CreateBitCast(Updated, PMV.IntValueType);
1001
1002 Value *ZExt = Builder.CreateZExt(Updated, PMV.WordType, "extended");
1003 Value *Shift =
1004 Builder.CreateShl(ZExt, PMV.ShiftAmt, "shifted", /*HasNUW*/ true);
1005 Value *And = Builder.CreateAnd(WideWord, PMV.Inv_Mask, "unmasked");
1006 Value *Or = Builder.CreateOr(And, Shift, "inserted");
1007 return Or;
1008}
1009
1010/// Emit IR to implement a masked version of a given atomicrmw
1011/// operation. (That is, only the bits under the Mask should be
1012/// affected by the operation)
1014 IRBuilderBase &Builder, Value *Loaded,
1015 Value *Shifted_Inc, Value *Inc,
1016 const PartwordMaskValues &PMV) {
1017 // TODO: update to use
1018 // https://graphics.stanford.edu/~seander/bithacks.html#MaskedMerge in order
1019 // to merge bits from two values without requiring PMV.Inv_Mask.
1020 switch (Op) {
1021 case AtomicRMWInst::Xchg: {
1022 Value *Loaded_MaskOut = Builder.CreateAnd(Loaded, PMV.Inv_Mask);
1023 Value *FinalVal = Builder.CreateOr(Loaded_MaskOut, Shifted_Inc);
1024 return FinalVal;
1025 }
1026 case AtomicRMWInst::Or:
1027 case AtomicRMWInst::Xor:
1028 case AtomicRMWInst::And:
1029 llvm_unreachable("Or/Xor/And handled by widenPartwordAtomicRMW");
1030 case AtomicRMWInst::Add:
1031 case AtomicRMWInst::Sub:
1032 case AtomicRMWInst::Nand: {
1033 // The other arithmetic ops need to be masked into place.
1034 Value *NewVal = buildAtomicRMWValue(Op, Builder, Loaded, Shifted_Inc);
1035 Value *NewVal_Masked = Builder.CreateAnd(NewVal, PMV.Mask);
1036 Value *Loaded_MaskOut = Builder.CreateAnd(Loaded, PMV.Inv_Mask);
1037 Value *FinalVal = Builder.CreateOr(Loaded_MaskOut, NewVal_Masked);
1038 return FinalVal;
1039 }
1040 case AtomicRMWInst::Max:
1041 case AtomicRMWInst::Min:
1056 // Finally, other ops will operate on the full value, so truncate down to
1057 // the original size, and expand out again after doing the
1058 // operation. Bitcasts will be inserted for FP values.
1059 Value *Loaded_Extract = extractMaskedValue(Builder, Loaded, PMV);
1060 Value *NewVal = buildAtomicRMWValue(Op, Builder, Loaded_Extract, Inc);
1061 Value *FinalVal = insertMaskedValue(Builder, Loaded, NewVal, PMV);
1062 return FinalVal;
1063 }
1064 default:
1065 llvm_unreachable("Unknown atomic op");
1066 }
1067}
1068
1069/// Expand a sub-word atomicrmw operation into an appropriate
1070/// word-sized operation.
1071///
1072/// It will create an LL/SC or cmpxchg loop, as appropriate, the same
1073/// way as a typical atomicrmw expansion. The only difference here is
1074/// that the operation inside of the loop may operate upon only a
1075/// part of the value.
1076void AtomicExpandImpl::expandPartwordAtomicRMW(
1077 AtomicRMWInst *AI, TargetLoweringBase::AtomicExpansionKind ExpansionKind) {
1078 // Widen And/Or/Xor and give the target another chance at expanding it.
1082 tryExpandAtomicRMW(widenPartwordAtomicRMW(AI));
1083 return;
1084 }
1085 AtomicOrdering MemOpOrder = AI->getOrdering();
1086 SyncScope::ID SSID = AI->getSyncScopeID();
1087
1088 ReplacementIRBuilder Builder(AI, *DL);
1089
1090 PartwordMaskValues PMV =
1091 createMaskInstrs(Builder, AI, AI->getType(), AI->getPointerOperand(),
1092 AI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1093
1094 Value *ValOperand_Shifted = nullptr;
1097 Value *ValOp = Builder.CreateBitCast(AI->getValOperand(), PMV.IntValueType);
1098 ValOperand_Shifted =
1099 Builder.CreateShl(Builder.CreateZExt(ValOp, PMV.WordType), PMV.ShiftAmt,
1100 "ValOperand_Shifted");
1101 }
1102
1103 auto PerformPartwordOp = [&](IRBuilderBase &Builder, Value *Loaded) {
1104 return performMaskedAtomicOp(Op, Builder, Loaded, ValOperand_Shifted,
1105 AI->getValOperand(), PMV);
1106 };
1107
1108 Value *OldResult;
1109 if (ExpansionKind == TargetLoweringBase::AtomicExpansionKind::CmpXChg) {
1110 OldResult = insertRMWCmpXchgLoop(Builder, PMV.WordType, PMV.AlignedAddr,
1111 PMV.AlignedAddrAlignment, MemOpOrder, SSID,
1112 AI->isVolatile(), PerformPartwordOp,
1114 } else {
1115 assert(ExpansionKind == TargetLoweringBase::AtomicExpansionKind::LLSC);
1116 OldResult = insertRMWLLSCLoop(Builder, PMV.WordType, PMV.AlignedAddr,
1117 PMV.AlignedAddrAlignment, MemOpOrder,
1118 PerformPartwordOp);
1119 }
1120
1121 Value *FinalOldResult = extractMaskedValue(Builder, OldResult, PMV);
1122 AI->replaceAllUsesWith(FinalOldResult);
1123 AI->eraseFromParent();
1124}
1125
1126// Widen the bitwise atomicrmw (or/xor/and) to the minimum supported width.
1127AtomicRMWInst *AtomicExpandImpl::widenPartwordAtomicRMW(AtomicRMWInst *AI) {
1128 ReplacementIRBuilder Builder(AI, *DL);
1130
1132 Op == AtomicRMWInst::And) &&
1133 "Unable to widen operation");
1134
1135 PartwordMaskValues PMV =
1136 createMaskInstrs(Builder, AI, AI->getType(), AI->getPointerOperand(),
1137 AI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1138
1139 Value *ValOperand_Shifted =
1140 Builder.CreateShl(Builder.CreateZExt(AI->getValOperand(), PMV.WordType),
1141 PMV.ShiftAmt, "ValOperand_Shifted");
1142
1143 Value *NewOperand;
1144
1145 if (Op == AtomicRMWInst::And)
1146 NewOperand =
1147 Builder.CreateOr(ValOperand_Shifted, PMV.Inv_Mask, "AndOperand");
1148 else
1149 NewOperand = ValOperand_Shifted;
1150
1151 AtomicRMWInst *NewAI = Builder.CreateAtomicRMW(
1152 Op, PMV.AlignedAddr, NewOperand, PMV.AlignedAddrAlignment,
1153 AI->getOrdering(), AI->getSyncScopeID());
1154
1155 NewAI->setVolatile(AI->isVolatile());
1156 copyMetadataForAtomic(*NewAI, *AI);
1157
1158 Value *FinalOldResult = extractMaskedValue(Builder, NewAI, PMV);
1159 AI->replaceAllUsesWith(FinalOldResult);
1160 AI->eraseFromParent();
1161 return NewAI;
1162}
1163
1164bool AtomicExpandImpl::expandPartwordCmpXchg(AtomicCmpXchgInst *CI) {
1165 // The basic idea here is that we're expanding a cmpxchg of a
1166 // smaller memory size up to a word-sized cmpxchg. To do this, we
1167 // need to add a retry-loop for strong cmpxchg, so that
1168 // modifications to other parts of the word don't cause a spurious
1169 // failure.
1170
1171 // This generates code like the following:
1172 // [[Setup mask values PMV.*]]
1173 // %NewVal_Shifted = shl i32 %NewVal, %PMV.ShiftAmt
1174 // %Cmp_Shifted = shl i32 %Cmp, %PMV.ShiftAmt
1175 // %InitLoaded = load i32* %addr
1176 // %InitLoaded_MaskOut = and i32 %InitLoaded, %PMV.Inv_Mask
1177 // br partword.cmpxchg.loop
1178 // partword.cmpxchg.loop:
1179 // %Loaded_MaskOut = phi i32 [ %InitLoaded_MaskOut, %entry ],
1180 // [ %OldVal_MaskOut, %partword.cmpxchg.failure ]
1181 // %FullWord_NewVal = or i32 %Loaded_MaskOut, %NewVal_Shifted
1182 // %FullWord_Cmp = or i32 %Loaded_MaskOut, %Cmp_Shifted
1183 // %NewCI = cmpxchg i32* %PMV.AlignedAddr, i32 %FullWord_Cmp,
1184 // i32 %FullWord_NewVal success_ordering failure_ordering
1185 // %OldVal = extractvalue { i32, i1 } %NewCI, 0
1186 // %Success = extractvalue { i32, i1 } %NewCI, 1
1187 // br i1 %Success, label %partword.cmpxchg.end,
1188 // label %partword.cmpxchg.failure
1189 // partword.cmpxchg.failure:
1190 // %OldVal_MaskOut = and i32 %OldVal, %PMV.Inv_Mask
1191 // %ShouldContinue = icmp ne i32 %Loaded_MaskOut, %OldVal_MaskOut
1192 // br i1 %ShouldContinue, label %partword.cmpxchg.loop,
1193 // label %partword.cmpxchg.end
1194 // partword.cmpxchg.end:
1195 // %tmp1 = lshr i32 %OldVal, %PMV.ShiftAmt
1196 // %FinalOldVal = trunc i32 %tmp1 to i8
1197 // %tmp2 = insertvalue { i8, i1 } undef, i8 %FinalOldVal, 0
1198 // %Res = insertvalue { i8, i1 } %25, i1 %Success, 1
1199
1200 Value *Addr = CI->getPointerOperand();
1201 Value *Cmp = CI->getCompareOperand();
1202 Value *NewVal = CI->getNewValOperand();
1203
1204 BasicBlock *BB = CI->getParent();
1205 Function *F = BB->getParent();
1206 ReplacementIRBuilder Builder(CI, *DL);
1207 LLVMContext &Ctx = Builder.getContext();
1208
1209 BasicBlock *EndBB =
1210 BB->splitBasicBlock(CI->getIterator(), "partword.cmpxchg.end");
1211 auto FailureBB =
1212 BasicBlock::Create(Ctx, "partword.cmpxchg.failure", F, EndBB);
1213 auto LoopBB = BasicBlock::Create(Ctx, "partword.cmpxchg.loop", F, FailureBB);
1214
1215 // The split call above "helpfully" added a branch at the end of BB
1216 // (to the wrong place).
1217 std::prev(BB->end())->eraseFromParent();
1218 Builder.SetInsertPoint(BB);
1219
1220 PartwordMaskValues PMV =
1221 createMaskInstrs(Builder, CI, CI->getCompareOperand()->getType(), Addr,
1222 CI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1223
1224 // Shift the incoming values over, into the right location in the word.
1225 Value *NewVal_Shifted =
1226 Builder.CreateShl(Builder.CreateZExt(NewVal, PMV.WordType), PMV.ShiftAmt);
1227 Value *Cmp_Shifted =
1228 Builder.CreateShl(Builder.CreateZExt(Cmp, PMV.WordType), PMV.ShiftAmt);
1229
1230 // Load the entire current word, and mask into place the expected and new
1231 // values
1232 LoadInst *InitLoaded = Builder.CreateLoad(PMV.WordType, PMV.AlignedAddr);
1233 Value *InitLoaded_MaskOut = Builder.CreateAnd(InitLoaded, PMV.Inv_Mask);
1234 Builder.CreateBr(LoopBB);
1235
1236 // partword.cmpxchg.loop:
1237 Builder.SetInsertPoint(LoopBB);
1238 PHINode *Loaded_MaskOut = Builder.CreatePHI(PMV.WordType, 2);
1239 Loaded_MaskOut->addIncoming(InitLoaded_MaskOut, BB);
1240
1241 // The initial load must be atomic with the same synchronization scope
1242 // to avoid a data race with concurrent stores. If the instruction being
1243 // emulated is volatile, issue a volatile load.
1244 // addIncoming is done first so that any replaceAllUsesWith calls during
1245 // normalization correctly update the PHI incoming value.
1246 InitLoaded->setVolatile(CI->isVolatile());
1248 InitLoaded->setAtomic(AtomicOrdering::Monotonic, CI->getSyncScopeID());
1249 // The newly created load might need to be lowered further. Because it is
1250 // created in the same block as the atomicrmw, the AtomicExpand loop will
1251 // not process it again.
1252 processAtomicInstr(InitLoaded);
1253 }
1254
1255 // Mask/Or the expected and new values into place in the loaded word.
1256 Value *FullWord_NewVal = Builder.CreateOr(Loaded_MaskOut, NewVal_Shifted);
1257 Value *FullWord_Cmp = Builder.CreateOr(Loaded_MaskOut, Cmp_Shifted);
1258 AtomicCmpXchgInst *NewCI = Builder.CreateAtomicCmpXchg(
1259 PMV.AlignedAddr, FullWord_Cmp, FullWord_NewVal, PMV.AlignedAddrAlignment,
1261 NewCI->setVolatile(CI->isVolatile());
1262 // When we're building a strong cmpxchg, we need a loop, so you
1263 // might think we could use a weak cmpxchg inside. But, using strong
1264 // allows the below comparison for ShouldContinue, and we're
1265 // expecting the underlying cmpxchg to be a machine instruction,
1266 // which is strong anyways.
1267 NewCI->setWeak(CI->isWeak());
1268
1269 Value *OldVal = Builder.CreateExtractValue(NewCI, 0);
1270 Value *Success = Builder.CreateExtractValue(NewCI, 1);
1271
1272 if (CI->isWeak())
1273 Builder.CreateBr(EndBB);
1274 else
1275 Builder.CreateCondBr(Success, EndBB, FailureBB);
1276
1277 // partword.cmpxchg.failure:
1278 Builder.SetInsertPoint(FailureBB);
1279 // Upon failure, verify that the masked-out part of the loaded value
1280 // has been modified. If it didn't, abort the cmpxchg, since the
1281 // masked-in part must've.
1282 Value *OldVal_MaskOut = Builder.CreateAnd(OldVal, PMV.Inv_Mask);
1283 Value *ShouldContinue = Builder.CreateICmpNE(Loaded_MaskOut, OldVal_MaskOut);
1284 Builder.CreateCondBr(ShouldContinue, LoopBB, EndBB);
1285
1286 // Add the second value to the phi from above
1287 Loaded_MaskOut->addIncoming(OldVal_MaskOut, FailureBB);
1288
1289 // partword.cmpxchg.end:
1290 Builder.SetInsertPoint(CI);
1291
1292 Value *FinalOldVal = extractMaskedValue(Builder, OldVal, PMV);
1293 Value *Res = PoisonValue::get(CI->getType());
1294 Res = Builder.CreateInsertValue(Res, FinalOldVal, 0);
1295 Res = Builder.CreateInsertValue(Res, Success, 1);
1296
1297 CI->replaceAllUsesWith(Res);
1298 CI->eraseFromParent();
1299 return true;
1300}
1301
1302void AtomicExpandImpl::expandAtomicOpToLLSC(
1303 Instruction *I, Type *ResultType, Value *Addr, Align AddrAlign,
1304 AtomicOrdering MemOpOrder,
1305 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp) {
1306 ReplacementIRBuilder Builder(I, *DL);
1307 Value *Loaded = insertRMWLLSCLoop(Builder, ResultType, Addr, AddrAlign,
1308 MemOpOrder, PerformOp);
1309
1310 I->replaceAllUsesWith(Loaded);
1311 I->eraseFromParent();
1312}
1313
1314void AtomicExpandImpl::expandAtomicRMWToMaskedIntrinsic(AtomicRMWInst *AI) {
1315 ReplacementIRBuilder Builder(AI, *DL);
1316
1317 PartwordMaskValues PMV =
1318 createMaskInstrs(Builder, AI, AI->getType(), AI->getPointerOperand(),
1319 AI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1320
1321 // The value operand must be sign-extended for signed min/max so that the
1322 // target's signed comparison instructions can be used. Otherwise, just
1323 // zero-ext.
1324 Instruction::CastOps CastOp = Instruction::ZExt;
1325 AtomicRMWInst::BinOp RMWOp = AI->getOperation();
1326 if (RMWOp == AtomicRMWInst::Max || RMWOp == AtomicRMWInst::Min)
1327 CastOp = Instruction::SExt;
1328
1329 Value *ValOperand_Shifted = Builder.CreateShl(
1330 Builder.CreateCast(CastOp, AI->getValOperand(), PMV.WordType),
1331 PMV.ShiftAmt, "ValOperand_Shifted");
1332 Value *OldResult = TLI->emitMaskedAtomicRMWIntrinsic(
1333 Builder, AI, PMV.AlignedAddr, ValOperand_Shifted, PMV.Mask, PMV.ShiftAmt,
1334 AI->getOrdering());
1335 Value *FinalOldResult = extractMaskedValue(Builder, OldResult, PMV);
1336 AI->replaceAllUsesWith(FinalOldResult);
1337 AI->eraseFromParent();
1338}
1339
1340void AtomicExpandImpl::expandAtomicCmpXchgToMaskedIntrinsic(
1341 AtomicCmpXchgInst *CI) {
1342 ReplacementIRBuilder Builder(CI, *DL);
1343
1344 PartwordMaskValues PMV = createMaskInstrs(
1345 Builder, CI, CI->getCompareOperand()->getType(), CI->getPointerOperand(),
1346 CI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1347
1348 Value *CmpVal_Shifted = Builder.CreateShl(
1349 Builder.CreateZExt(CI->getCompareOperand(), PMV.WordType), PMV.ShiftAmt,
1350 "CmpVal_Shifted");
1351 Value *NewVal_Shifted = Builder.CreateShl(
1352 Builder.CreateZExt(CI->getNewValOperand(), PMV.WordType), PMV.ShiftAmt,
1353 "NewVal_Shifted");
1355 Builder, CI, PMV.AlignedAddr, CmpVal_Shifted, NewVal_Shifted, PMV.Mask,
1356 CI->getMergedOrdering());
1357 Value *FinalOldVal = extractMaskedValue(Builder, OldVal, PMV);
1358 Value *Res = PoisonValue::get(CI->getType());
1359 Res = Builder.CreateInsertValue(Res, FinalOldVal, 0);
1360 Value *Success = Builder.CreateICmpEQ(
1361 CmpVal_Shifted, Builder.CreateAnd(OldVal, PMV.Mask), "Success");
1362 Res = Builder.CreateInsertValue(Res, Success, 1);
1363
1364 CI->replaceAllUsesWith(Res);
1365 CI->eraseFromParent();
1366}
1367
1368Value *AtomicExpandImpl::insertRMWLLSCLoop(
1369 IRBuilderBase &Builder, Type *ResultTy, Value *Addr, Align AddrAlign,
1370 AtomicOrdering MemOpOrder,
1371 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp) {
1372 LLVMContext &Ctx = Builder.getContext();
1373 BasicBlock *BB = Builder.GetInsertBlock();
1374 Function *F = BB->getParent();
1375
1376 assert(AddrAlign >= F->getDataLayout().getTypeStoreSize(ResultTy) &&
1377 "Expected at least natural alignment at this point.");
1378
1379 // Given: atomicrmw some_op iN* %addr, iN %incr ordering
1380 //
1381 // The standard expansion we produce is:
1382 // [...]
1383 // atomicrmw.start:
1384 // %loaded = @load.linked(%addr)
1385 // %new = some_op iN %loaded, %incr
1386 // %stored = @store_conditional(%new, %addr)
1387 // %try_again = icmp i32 ne %stored, 0
1388 // br i1 %try_again, label %loop, label %atomicrmw.end
1389 // atomicrmw.end:
1390 // [...]
1391 BasicBlock *ExitBB =
1392 BB->splitBasicBlock(Builder.GetInsertPoint(), "atomicrmw.end");
1393 BasicBlock *LoopBB = BasicBlock::Create(Ctx, "atomicrmw.start", F, ExitBB);
1394
1395 // The split call above "helpfully" added a branch at the end of BB (to the
1396 // wrong place).
1397 std::prev(BB->end())->eraseFromParent();
1398 Builder.SetInsertPoint(BB);
1399 Builder.CreateBr(LoopBB);
1400
1401 // Start the main loop block now that we've taken care of the preliminaries.
1402 Builder.SetInsertPoint(LoopBB);
1403 Value *Loaded = TLI->emitLoadLinked(Builder, ResultTy, Addr, MemOpOrder);
1404
1405 Value *NewVal = PerformOp(Builder, Loaded);
1406
1407 Value *StoreSuccess =
1408 TLI->emitStoreConditional(Builder, NewVal, Addr, MemOpOrder);
1409 Value *TryAgain = Builder.CreateICmpNE(
1410 StoreSuccess, ConstantInt::get(IntegerType::get(Ctx, 32), 0), "tryagain");
1411
1412 Instruction *CondBr = Builder.CreateCondBr(TryAgain, LoopBB, ExitBB);
1413
1414 // Atomic RMW expands to a Load-linked / Store-Conditional loop, because it is
1415 // hard to predict precise branch weigths we mark the branch as "unknown"
1416 // (50/50) to prevent misleading optimizations.
1418
1419 Builder.SetInsertPoint(ExitBB, ExitBB->begin());
1420 return Loaded;
1421}
1422
1423/// Convert an atomic cmpxchg of a non-integral type to an integer cmpxchg of
1424/// the equivalent bitwidth. We used to not support pointer cmpxchg in the
1425/// IR. As a migration step, we convert back to what use to be the standard
1426/// way to represent a pointer cmpxchg so that we can update backends one by
1427/// one.
1428AtomicCmpXchgInst *
1429AtomicExpandImpl::convertCmpXchgToIntegerType(AtomicCmpXchgInst *CI) {
1430 auto *M = CI->getModule();
1431 Type *NewTy = getCorrespondingIntegerType(CI->getCompareOperand()->getType(),
1432 M->getDataLayout());
1433
1434 ReplacementIRBuilder Builder(CI, *DL);
1435
1436 Value *Addr = CI->getPointerOperand();
1437
1438 Value *NewCmp = Builder.CreatePtrToInt(CI->getCompareOperand(), NewTy);
1439 Value *NewNewVal = Builder.CreatePtrToInt(CI->getNewValOperand(), NewTy);
1440
1441 auto *NewCI = Builder.CreateAtomicCmpXchg(
1442 Addr, NewCmp, NewNewVal, CI->getAlign(), CI->getSuccessOrdering(),
1443 CI->getFailureOrdering(), CI->getSyncScopeID());
1444 NewCI->setVolatile(CI->isVolatile());
1445 NewCI->setWeak(CI->isWeak());
1446 LLVM_DEBUG(dbgs() << "Replaced " << *CI << " with " << *NewCI << "\n");
1447
1448 Value *OldVal = Builder.CreateExtractValue(NewCI, 0);
1449 Value *Succ = Builder.CreateExtractValue(NewCI, 1);
1450
1451 OldVal = Builder.CreateIntToPtr(OldVal, CI->getCompareOperand()->getType());
1452
1453 Value *Res = PoisonValue::get(CI->getType());
1454 Res = Builder.CreateInsertValue(Res, OldVal, 0);
1455 Res = Builder.CreateInsertValue(Res, Succ, 1);
1456
1457 CI->replaceAllUsesWith(Res);
1458 CI->eraseFromParent();
1459 return NewCI;
1460}
1461
1462bool AtomicExpandImpl::expandAtomicCmpXchg(AtomicCmpXchgInst *CI) {
1463 AtomicOrdering SuccessOrder = CI->getSuccessOrdering();
1464 AtomicOrdering FailureOrder = CI->getFailureOrdering();
1465 Value *Addr = CI->getPointerOperand();
1466 BasicBlock *BB = CI->getParent();
1467 Function *F = BB->getParent();
1468 LLVMContext &Ctx = F->getContext();
1469 // If shouldInsertFencesForAtomic() returns true, then the target does not
1470 // want to deal with memory orders, and emitLeading/TrailingFence should take
1471 // care of everything. Otherwise, emitLeading/TrailingFence are no-op and we
1472 // should preserve the ordering.
1473 bool ShouldInsertFencesForAtomic = TLI->shouldInsertFencesForAtomic(CI);
1474 AtomicOrdering MemOpOrder = ShouldInsertFencesForAtomic
1475 ? AtomicOrdering::Monotonic
1476 : CI->getMergedOrdering();
1477
1478 // In implementations which use a barrier to achieve release semantics, we can
1479 // delay emitting this barrier until we know a store is actually going to be
1480 // attempted. The cost of this delay is that we need 2 copies of the block
1481 // emitting the load-linked, affecting code size.
1482 //
1483 // Ideally, this logic would be unconditional except for the minsize check
1484 // since in other cases the extra blocks naturally collapse down to the
1485 // minimal loop. Unfortunately, this puts too much stress on later
1486 // optimisations so we avoid emitting the extra logic in those cases too.
1487 bool HasReleasedLoadBB = !CI->isWeak() && ShouldInsertFencesForAtomic &&
1488 SuccessOrder != AtomicOrdering::Monotonic &&
1489 SuccessOrder != AtomicOrdering::Acquire &&
1490 !F->hasMinSize();
1491
1492 // There's no overhead for sinking the release barrier in a weak cmpxchg, so
1493 // do it even on minsize.
1494 bool UseUnconditionalReleaseBarrier = F->hasMinSize() && !CI->isWeak();
1495
1496 // Given: cmpxchg some_op iN* %addr, iN %desired, iN %new success_ord fail_ord
1497 //
1498 // The full expansion we produce is:
1499 // [...]
1500 // %aligned.addr = ...
1501 // cmpxchg.start:
1502 // %unreleasedload = @load.linked(%aligned.addr)
1503 // %unreleasedload.extract = extract value from %unreleasedload
1504 // %should_store = icmp eq %unreleasedload.extract, %desired
1505 // br i1 %should_store, label %cmpxchg.releasingstore,
1506 // label %cmpxchg.nostore
1507 // cmpxchg.releasingstore:
1508 // fence?
1509 // br label cmpxchg.trystore
1510 // cmpxchg.trystore:
1511 // %loaded.trystore = phi [%unreleasedload, %cmpxchg.releasingstore],
1512 // [%releasedload, %cmpxchg.releasedload]
1513 // %updated.new = insert %new into %loaded.trystore
1514 // %stored = @store_conditional(%updated.new, %aligned.addr)
1515 // %success = icmp eq i32 %stored, 0
1516 // br i1 %success, label %cmpxchg.success,
1517 // label %cmpxchg.releasedload/%cmpxchg.failure
1518 // cmpxchg.releasedload:
1519 // %releasedload = @load.linked(%aligned.addr)
1520 // %releasedload.extract = extract value from %releasedload
1521 // %should_store = icmp eq %releasedload.extract, %desired
1522 // br i1 %should_store, label %cmpxchg.trystore,
1523 // label %cmpxchg.failure
1524 // cmpxchg.success:
1525 // fence?
1526 // br label %cmpxchg.end
1527 // cmpxchg.nostore:
1528 // %loaded.nostore = phi [%unreleasedload, %cmpxchg.start],
1529 // [%releasedload,
1530 // %cmpxchg.releasedload/%cmpxchg.trystore]
1531 // @load_linked_fail_balance()?
1532 // br label %cmpxchg.failure
1533 // cmpxchg.failure:
1534 // fence?
1535 // br label %cmpxchg.end
1536 // cmpxchg.end:
1537 // %loaded.exit = phi [%loaded.nostore, %cmpxchg.failure],
1538 // [%loaded.trystore, %cmpxchg.trystore]
1539 // %success = phi i1 [true, %cmpxchg.success], [false, %cmpxchg.failure]
1540 // %loaded = extract value from %loaded.exit
1541 // %restmp = insertvalue { iN, i1 } undef, iN %loaded, 0
1542 // %res = insertvalue { iN, i1 } %restmp, i1 %success, 1
1543 // [...]
1544 BasicBlock *ExitBB = BB->splitBasicBlock(CI->getIterator(), "cmpxchg.end");
1545 auto FailureBB = BasicBlock::Create(Ctx, "cmpxchg.failure", F, ExitBB);
1546 auto NoStoreBB = BasicBlock::Create(Ctx, "cmpxchg.nostore", F, FailureBB);
1547 auto SuccessBB = BasicBlock::Create(Ctx, "cmpxchg.success", F, NoStoreBB);
1548 auto ReleasedLoadBB =
1549 BasicBlock::Create(Ctx, "cmpxchg.releasedload", F, SuccessBB);
1550 auto TryStoreBB =
1551 BasicBlock::Create(Ctx, "cmpxchg.trystore", F, ReleasedLoadBB);
1552 auto ReleasingStoreBB =
1553 BasicBlock::Create(Ctx, "cmpxchg.fencedstore", F, TryStoreBB);
1554 auto StartBB = BasicBlock::Create(Ctx, "cmpxchg.start", F, ReleasingStoreBB);
1555
1556 ReplacementIRBuilder Builder(CI, *DL);
1557
1558 // The split call above "helpfully" added a branch at the end of BB (to the
1559 // wrong place), but we might want a fence too. It's easiest to just remove
1560 // the branch entirely.
1561 std::prev(BB->end())->eraseFromParent();
1562 Builder.SetInsertPoint(BB);
1563 if (ShouldInsertFencesForAtomic && UseUnconditionalReleaseBarrier)
1564 TLI->emitLeadingFence(Builder, CI, SuccessOrder);
1565
1566 PartwordMaskValues PMV =
1567 createMaskInstrs(Builder, CI, CI->getCompareOperand()->getType(), Addr,
1568 CI->getAlign(), TLI->getMinCmpXchgSizeInBits() / 8);
1569 Builder.CreateBr(StartBB);
1570
1571 // Start the main loop block now that we've taken care of the preliminaries.
1572 Builder.SetInsertPoint(StartBB);
1573 Value *UnreleasedLoad =
1574 TLI->emitLoadLinked(Builder, PMV.WordType, PMV.AlignedAddr, MemOpOrder);
1575 Value *UnreleasedLoadExtract =
1576 extractMaskedValue(Builder, UnreleasedLoad, PMV);
1577 Value *ShouldStore = Builder.CreateICmpEQ(
1578 UnreleasedLoadExtract, CI->getCompareOperand(), "should_store");
1579
1580 // If the cmpxchg doesn't actually need any ordering when it fails, we can
1581 // jump straight past that fence instruction (if it exists).
1582 Builder.CreateCondBr(ShouldStore, ReleasingStoreBB, NoStoreBB,
1583 MDBuilder(F->getContext()).createLikelyBranchWeights());
1584
1585 Builder.SetInsertPoint(ReleasingStoreBB);
1586 if (ShouldInsertFencesForAtomic && !UseUnconditionalReleaseBarrier)
1587 TLI->emitLeadingFence(Builder, CI, SuccessOrder);
1588 Builder.CreateBr(TryStoreBB);
1589
1590 Builder.SetInsertPoint(TryStoreBB);
1591 PHINode *LoadedTryStore =
1592 Builder.CreatePHI(PMV.WordType, 2, "loaded.trystore");
1593 LoadedTryStore->addIncoming(UnreleasedLoad, ReleasingStoreBB);
1594 Value *NewValueInsert =
1595 insertMaskedValue(Builder, LoadedTryStore, CI->getNewValOperand(), PMV);
1596 Value *StoreSuccess = TLI->emitStoreConditional(Builder, NewValueInsert,
1597 PMV.AlignedAddr, MemOpOrder);
1598 StoreSuccess = Builder.CreateICmpEQ(
1599 StoreSuccess, ConstantInt::get(Type::getInt32Ty(Ctx), 0), "success");
1600 BasicBlock *RetryBB = HasReleasedLoadBB ? ReleasedLoadBB : StartBB;
1601 Builder.CreateCondBr(StoreSuccess, SuccessBB,
1602 CI->isWeak() ? FailureBB : RetryBB,
1603 MDBuilder(F->getContext()).createLikelyBranchWeights());
1604
1605 Builder.SetInsertPoint(ReleasedLoadBB);
1606 Value *SecondLoad;
1607 if (HasReleasedLoadBB) {
1608 SecondLoad =
1609 TLI->emitLoadLinked(Builder, PMV.WordType, PMV.AlignedAddr, MemOpOrder);
1610 Value *SecondLoadExtract = extractMaskedValue(Builder, SecondLoad, PMV);
1611 ShouldStore = Builder.CreateICmpEQ(SecondLoadExtract,
1612 CI->getCompareOperand(), "should_store");
1613
1614 // If the cmpxchg doesn't actually need any ordering when it fails, we can
1615 // jump straight past that fence instruction (if it exists).
1616 Builder.CreateCondBr(
1617 ShouldStore, TryStoreBB, NoStoreBB,
1618 MDBuilder(F->getContext()).createLikelyBranchWeights());
1619 // Update PHI node in TryStoreBB.
1620 LoadedTryStore->addIncoming(SecondLoad, ReleasedLoadBB);
1621 } else
1622 Builder.CreateUnreachable();
1623
1624 // Make sure later instructions don't get reordered with a fence if
1625 // necessary.
1626 Builder.SetInsertPoint(SuccessBB);
1627 if (ShouldInsertFencesForAtomic ||
1629 TLI->emitTrailingFence(Builder, CI, SuccessOrder);
1630 Builder.CreateBr(ExitBB);
1631
1632 Builder.SetInsertPoint(NoStoreBB);
1633 PHINode *LoadedNoStore =
1634 Builder.CreatePHI(UnreleasedLoad->getType(), 2, "loaded.nostore");
1635 LoadedNoStore->addIncoming(UnreleasedLoad, StartBB);
1636 if (HasReleasedLoadBB)
1637 LoadedNoStore->addIncoming(SecondLoad, ReleasedLoadBB);
1638
1639 // In the failing case, where we don't execute the store-conditional, the
1640 // target might want to balance out the load-linked with a dedicated
1641 // instruction (e.g., on ARM, clearing the exclusive monitor).
1643 Builder.CreateBr(FailureBB);
1644
1645 Builder.SetInsertPoint(FailureBB);
1646 PHINode *LoadedFailure =
1647 Builder.CreatePHI(UnreleasedLoad->getType(), 2, "loaded.failure");
1648 LoadedFailure->addIncoming(LoadedNoStore, NoStoreBB);
1649 if (CI->isWeak())
1650 LoadedFailure->addIncoming(LoadedTryStore, TryStoreBB);
1651 if (ShouldInsertFencesForAtomic)
1652 TLI->emitTrailingFence(Builder, CI, FailureOrder);
1653 Builder.CreateBr(ExitBB);
1654
1655 // Finally, we have control-flow based knowledge of whether the cmpxchg
1656 // succeeded or not. We expose this to later passes by converting any
1657 // subsequent "icmp eq/ne %loaded, %oldval" into a use of an appropriate
1658 // PHI.
1659 Builder.SetInsertPoint(ExitBB, ExitBB->begin());
1660 PHINode *LoadedExit =
1661 Builder.CreatePHI(UnreleasedLoad->getType(), 2, "loaded.exit");
1662 LoadedExit->addIncoming(LoadedTryStore, SuccessBB);
1663 LoadedExit->addIncoming(LoadedFailure, FailureBB);
1664 PHINode *Success = Builder.CreatePHI(Type::getInt1Ty(Ctx), 2, "success");
1665 Success->addIncoming(ConstantInt::getTrue(Ctx), SuccessBB);
1666 Success->addIncoming(ConstantInt::getFalse(Ctx), FailureBB);
1667
1668 // This is the "exit value" from the cmpxchg expansion. It may be of
1669 // a type wider than the one in the cmpxchg instruction.
1670 Value *LoadedFull = LoadedExit;
1671
1672 Builder.SetInsertPoint(ExitBB, std::next(Success->getIterator()));
1673 Value *Loaded = extractMaskedValue(Builder, LoadedFull, PMV);
1674
1675 // Look for any users of the cmpxchg that are just comparing the loaded value
1676 // against the desired one, and replace them with the CFG-derived version.
1678 for (auto *User : CI->users()) {
1679 ExtractValueInst *EV = dyn_cast<ExtractValueInst>(User);
1680 if (!EV)
1681 continue;
1682
1683 assert(EV->getNumIndices() == 1 && EV->getIndices()[0] <= 1 &&
1684 "weird extraction from { iN, i1 }");
1685
1686 if (EV->getIndices()[0] == 0)
1687 EV->replaceAllUsesWith(Loaded);
1688 else
1690
1691 PrunedInsts.push_back(EV);
1692 }
1693
1694 // We can remove the instructions now we're no longer iterating through them.
1695 for (auto *EV : PrunedInsts)
1696 EV->eraseFromParent();
1697
1698 if (!CI->use_empty()) {
1699 // Some use of the full struct return that we don't understand has happened,
1700 // so we've got to reconstruct it properly.
1701 Value *Res;
1702 Res = Builder.CreateInsertValue(PoisonValue::get(CI->getType()), Loaded, 0);
1703 Res = Builder.CreateInsertValue(Res, Success, 1);
1704
1705 CI->replaceAllUsesWith(Res);
1706 }
1707
1708 CI->eraseFromParent();
1709 return true;
1710}
1711
1712bool AtomicExpandImpl::isIdempotentRMW(AtomicRMWInst *RMWI) {
1713 if (RMWI->isVolatile())
1714 return false;
1715 // TODO: Add floating point support.
1716 auto C = dyn_cast<ConstantInt>(RMWI->getValOperand());
1717 if (!C)
1718 return false;
1719
1720 switch (RMWI->getOperation()) {
1721 case AtomicRMWInst::Add:
1722 case AtomicRMWInst::Sub:
1723 case AtomicRMWInst::Or:
1724 case AtomicRMWInst::Xor:
1725 return C->isZero();
1726 case AtomicRMWInst::And:
1727 return C->isMinusOne();
1728 case AtomicRMWInst::Min:
1729 return C->isMaxValue(true);
1730 case AtomicRMWInst::Max:
1731 return C->isMinValue(true);
1733 return C->isMaxValue(false);
1735 return C->isMinValue(false);
1736 default:
1737 return false;
1738 }
1739}
1740
1741bool AtomicExpandImpl::simplifyIdempotentRMW(AtomicRMWInst *RMWI) {
1742 if (auto ResultingLoad = TLI->lowerIdempotentRMWIntoFencedLoad(RMWI)) {
1743 tryExpandAtomicLoad(ResultingLoad);
1744 return true;
1745 }
1746 return false;
1747}
1748
1749Value *AtomicExpandImpl::insertRMWCmpXchgLoop(
1750 IRBuilderBase &Builder, Type *ResultTy, Value *Addr, Align AddrAlign,
1751 AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile,
1752 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp,
1753 CreateCmpXchgInstFun CreateCmpXchg, Instruction *MetadataSrc) {
1754 LLVMContext &Ctx = Builder.getContext();
1755 BasicBlock *BB = Builder.GetInsertBlock();
1756 Function *F = BB->getParent();
1757
1758 // Given: atomicrmw some_op iN* %addr, iN %incr ordering
1759 //
1760 // The standard expansion we produce is:
1761 // [...]
1762 // %init_loaded = load atomic iN* %addr
1763 // br label %loop
1764 // loop:
1765 // %loaded = phi iN [ %init_loaded, %entry ], [ %new_loaded, %loop ]
1766 // %new = some_op iN %loaded, %incr
1767 // %pair = cmpxchg iN* %addr, iN %loaded, iN %new
1768 // %new_loaded = extractvalue { iN, i1 } %pair, 0
1769 // %success = extractvalue { iN, i1 } %pair, 1
1770 // br i1 %success, label %atomicrmw.end, label %loop
1771 // atomicrmw.end:
1772 // [...]
1773 BasicBlock *ExitBB =
1774 BB->splitBasicBlock(Builder.GetInsertPoint(), "atomicrmw.end");
1775 BasicBlock *LoopBB = BasicBlock::Create(Ctx, "atomicrmw.start", F, ExitBB);
1776
1777 // The split call above "helpfully" added a branch at the end of BB (to the
1778 // wrong place), but we want a load. It's easiest to just remove
1779 // the branch entirely.
1780 std::prev(BB->end())->eraseFromParent();
1781 Builder.SetInsertPoint(BB);
1782 LoadInst *InitLoaded = Builder.CreateAlignedLoad(ResultTy, Addr, AddrAlign);
1783 Builder.CreateBr(LoopBB);
1784
1785 // Start the main loop block now that we've taken care of the preliminaries.
1786 Builder.SetInsertPoint(LoopBB);
1787 PHINode *Loaded = Builder.CreatePHI(ResultTy, 2, "loaded");
1788 Loaded->addIncoming(InitLoaded, BB);
1789
1790 // The initial load must be atomic with the same synchronization scope
1791 // to avoid a data race with concurrent stores. If the instruction being
1792 // emulated is volatile, issue a volatile load.
1793 // addIncoming is done first so that any replaceAllUsesWith calls during
1794 // normalization correctly update the PHI incoming value.
1795 InitLoaded->setVolatile(IsVolatile);
1797 InitLoaded->setAtomic(AtomicOrdering::Monotonic, SSID);
1798 // The newly created load might need to be lowered further. Because it is
1799 // created in the same block as the atomicrmw, the AtomicExpand loop will
1800 // not process it again.
1801 processAtomicInstr(InitLoaded);
1802 }
1803
1804 Value *NewVal = PerformOp(Builder, Loaded);
1805
1806 Value *NewLoaded = nullptr;
1807 Value *Success = nullptr;
1808
1809 CreateCmpXchg(Builder, Addr, Loaded, NewVal, AddrAlign,
1810 MemOpOrder == AtomicOrdering::Unordered
1811 ? AtomicOrdering::Monotonic
1812 : MemOpOrder,
1813 SSID, IsVolatile, Success, NewLoaded, MetadataSrc);
1814 assert(Success && NewLoaded);
1815
1816 Loaded->addIncoming(NewLoaded, LoopBB);
1817
1818 Instruction *CondBr = Builder.CreateCondBr(Success, ExitBB, LoopBB);
1819
1820 // Atomic RMW expands to a cmpxchg loop, Since precise branch weights
1821 // cannot be easily determined here, we mark the branch as "unknown" (50/50)
1822 // to prevent misleading optimizations.
1824
1825 Builder.SetInsertPoint(ExitBB, ExitBB->begin());
1826 return NewLoaded;
1827}
1828
1829bool AtomicExpandImpl::tryExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) {
1830 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
1831 unsigned ValueSize = getAtomicOpSize(CI);
1832
1833 switch (TLI->shouldExpandAtomicCmpXchgInIR(CI)) {
1834 default:
1835 llvm_unreachable("Unhandled case in tryExpandAtomicCmpXchg");
1836 case TargetLoweringBase::AtomicExpansionKind::None:
1837 if (ValueSize < MinCASSize)
1838 return expandPartwordCmpXchg(CI);
1839 return false;
1840 case TargetLoweringBase::AtomicExpansionKind::LLSC: {
1841 return expandAtomicCmpXchg(CI);
1842 }
1843 case TargetLoweringBase::AtomicExpansionKind::MaskedIntrinsic:
1844 expandAtomicCmpXchgToMaskedIntrinsic(CI);
1845 return true;
1846 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
1847 return lowerAtomicCmpXchgInst(CI);
1848 case TargetLoweringBase::AtomicExpansionKind::CustomExpand: {
1849 TLI->emitExpandAtomicCmpXchg(CI);
1850 return true;
1851 }
1852 }
1853}
1854
1855bool AtomicExpandImpl::expandAtomicRMWToCmpXchg(
1856 AtomicRMWInst *AI, CreateCmpXchgInstFun CreateCmpXchg) {
1857 ReplacementIRBuilder Builder(AI, AI->getDataLayout());
1858 Builder.setIsFPConstrained(
1859 AI->getFunction()->hasFnAttribute(Attribute::StrictFP));
1860
1861 // FIXME: If FP exceptions are observable, we should force them off for the
1862 // loop for the FP atomics.
1863 Value *Loaded = AtomicExpandImpl::insertRMWCmpXchgLoop(
1864 Builder, AI->getType(), AI->getPointerOperand(), AI->getAlign(),
1865 AI->getOrdering(), AI->getSyncScopeID(), AI->isVolatile(),
1866 [&](IRBuilderBase &Builder, Value *Loaded) {
1867 return buildAtomicRMWValue(AI->getOperation(), Builder, Loaded,
1868 AI->getValOperand());
1869 },
1870 CreateCmpXchg, /*MetadataSrc=*/AI);
1871
1872 AI->replaceAllUsesWith(Loaded);
1873 AI->eraseFromParent();
1874 return true;
1875}
1876
1877// In order to use one of the sized library calls such as
1878// __atomic_fetch_add_4, the alignment must be sufficient, the size
1879// must be one of the potentially-specialized sizes, and the value
1880// type must actually exist in C on the target (otherwise, the
1881// function wouldn't actually be defined.)
1882static bool canUseSizedAtomicCall(unsigned Size, Align Alignment,
1883 const DataLayout &DL) {
1884 // TODO: "LargestSize" is an approximation for "largest type that
1885 // you can express in C". It seems to be the case that int128 is
1886 // supported on all 64-bit platforms, otherwise only up to 64-bit
1887 // integers are supported. If we get this wrong, then we'll try to
1888 // call a sized libcall that doesn't actually exist. There should
1889 // really be some more reliable way in LLVM of determining integer
1890 // sizes which are valid in the target's C ABI...
1891 unsigned LargestSize = DL.getLargestLegalIntTypeSizeInBits() >= 64 ? 16 : 8;
1892 return Alignment >= Size &&
1893 (Size == 1 || Size == 2 || Size == 4 || Size == 8 || Size == 16) &&
1894 Size <= LargestSize;
1895}
1896
1897void AtomicExpandImpl::expandAtomicLoadToLibcall(LoadInst *I) {
1898 static const RTLIB::Libcall Libcalls[6] = {
1899 RTLIB::ATOMIC_LOAD, RTLIB::ATOMIC_LOAD_1, RTLIB::ATOMIC_LOAD_2,
1900 RTLIB::ATOMIC_LOAD_4, RTLIB::ATOMIC_LOAD_8, RTLIB::ATOMIC_LOAD_16};
1901 unsigned Size = getAtomicOpSize(I);
1902
1903 bool Expanded = expandAtomicOpToLibcall(
1904 I, Size, I->getAlign(), I->getPointerOperand(), nullptr, nullptr,
1905 I->getOrdering(), AtomicOrdering::NotAtomic, Libcalls);
1906 if (!Expanded)
1907 handleUnsupportedAtomicSize(I, "atomic load");
1908}
1909
1910void AtomicExpandImpl::expandAtomicStoreToLibcall(StoreInst *I) {
1911 static const RTLIB::Libcall Libcalls[6] = {
1912 RTLIB::ATOMIC_STORE, RTLIB::ATOMIC_STORE_1, RTLIB::ATOMIC_STORE_2,
1913 RTLIB::ATOMIC_STORE_4, RTLIB::ATOMIC_STORE_8, RTLIB::ATOMIC_STORE_16};
1914 unsigned Size = getAtomicOpSize(I);
1915
1916 bool Expanded = expandAtomicOpToLibcall(
1917 I, Size, I->getAlign(), I->getPointerOperand(), I->getValueOperand(),
1918 nullptr, I->getOrdering(), AtomicOrdering::NotAtomic, Libcalls);
1919 if (!Expanded)
1920 handleUnsupportedAtomicSize(I, "atomic store");
1921}
1922
1923void AtomicExpandImpl::expandAtomicCASToLibcall(AtomicCmpXchgInst *I,
1924 const Twine &AtomicOpName,
1925 Instruction *DiagnosticInst) {
1926 static const RTLIB::Libcall Libcalls[6] = {
1927 RTLIB::ATOMIC_COMPARE_EXCHANGE, RTLIB::ATOMIC_COMPARE_EXCHANGE_1,
1928 RTLIB::ATOMIC_COMPARE_EXCHANGE_2, RTLIB::ATOMIC_COMPARE_EXCHANGE_4,
1929 RTLIB::ATOMIC_COMPARE_EXCHANGE_8, RTLIB::ATOMIC_COMPARE_EXCHANGE_16};
1930 unsigned Size = getAtomicOpSize(I);
1931
1932 bool Expanded = expandAtomicOpToLibcall(
1933 I, Size, I->getAlign(), I->getPointerOperand(), I->getNewValOperand(),
1934 I->getCompareOperand(), I->getSuccessOrdering(), I->getFailureOrdering(),
1935 Libcalls);
1936 if (!Expanded)
1937 handleUnsupportedAtomicSize(I, AtomicOpName, DiagnosticInst);
1938}
1939
1941 static const RTLIB::Libcall LibcallsXchg[6] = {
1942 RTLIB::ATOMIC_EXCHANGE, RTLIB::ATOMIC_EXCHANGE_1,
1943 RTLIB::ATOMIC_EXCHANGE_2, RTLIB::ATOMIC_EXCHANGE_4,
1944 RTLIB::ATOMIC_EXCHANGE_8, RTLIB::ATOMIC_EXCHANGE_16};
1945 static const RTLIB::Libcall LibcallsAdd[6] = {
1946 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_ADD_1,
1947 RTLIB::ATOMIC_FETCH_ADD_2, RTLIB::ATOMIC_FETCH_ADD_4,
1948 RTLIB::ATOMIC_FETCH_ADD_8, RTLIB::ATOMIC_FETCH_ADD_16};
1949 static const RTLIB::Libcall LibcallsSub[6] = {
1950 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_SUB_1,
1951 RTLIB::ATOMIC_FETCH_SUB_2, RTLIB::ATOMIC_FETCH_SUB_4,
1952 RTLIB::ATOMIC_FETCH_SUB_8, RTLIB::ATOMIC_FETCH_SUB_16};
1953 static const RTLIB::Libcall LibcallsAnd[6] = {
1954 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_AND_1,
1955 RTLIB::ATOMIC_FETCH_AND_2, RTLIB::ATOMIC_FETCH_AND_4,
1956 RTLIB::ATOMIC_FETCH_AND_8, RTLIB::ATOMIC_FETCH_AND_16};
1957 static const RTLIB::Libcall LibcallsOr[6] = {
1958 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_OR_1,
1959 RTLIB::ATOMIC_FETCH_OR_2, RTLIB::ATOMIC_FETCH_OR_4,
1960 RTLIB::ATOMIC_FETCH_OR_8, RTLIB::ATOMIC_FETCH_OR_16};
1961 static const RTLIB::Libcall LibcallsXor[6] = {
1962 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_XOR_1,
1963 RTLIB::ATOMIC_FETCH_XOR_2, RTLIB::ATOMIC_FETCH_XOR_4,
1964 RTLIB::ATOMIC_FETCH_XOR_8, RTLIB::ATOMIC_FETCH_XOR_16};
1965 static const RTLIB::Libcall LibcallsNand[6] = {
1966 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_NAND_1,
1967 RTLIB::ATOMIC_FETCH_NAND_2, RTLIB::ATOMIC_FETCH_NAND_4,
1968 RTLIB::ATOMIC_FETCH_NAND_8, RTLIB::ATOMIC_FETCH_NAND_16};
1969
1970 switch (Op) {
1972 llvm_unreachable("Should not have BAD_BINOP.");
1974 return ArrayRef(LibcallsXchg);
1975 case AtomicRMWInst::Add:
1976 return ArrayRef(LibcallsAdd);
1977 case AtomicRMWInst::Sub:
1978 return ArrayRef(LibcallsSub);
1979 case AtomicRMWInst::And:
1980 return ArrayRef(LibcallsAnd);
1981 case AtomicRMWInst::Or:
1982 return ArrayRef(LibcallsOr);
1983 case AtomicRMWInst::Xor:
1984 return ArrayRef(LibcallsXor);
1986 return ArrayRef(LibcallsNand);
1987 case AtomicRMWInst::Max:
1988 case AtomicRMWInst::Min:
2003 // No atomic libcalls are available for these.
2004 return {};
2005 }
2006 llvm_unreachable("Unexpected AtomicRMW operation.");
2007}
2008
2009void AtomicExpandImpl::expandAtomicRMWToLibcall(AtomicRMWInst *I) {
2010 ArrayRef<RTLIB::Libcall> Libcalls = GetRMWLibcall(I->getOperation());
2011
2012 unsigned Size = getAtomicOpSize(I);
2013
2014 bool Success = false;
2015 if (!Libcalls.empty())
2016 Success = expandAtomicOpToLibcall(
2017 I, Size, I->getAlign(), I->getPointerOperand(), I->getValOperand(),
2018 nullptr, I->getOrdering(), AtomicOrdering::NotAtomic, Libcalls);
2019
2020 // The expansion failed: either there were no libcalls at all for
2021 // the operation (min/max), or there were only size-specialized
2022 // libcalls (add/sub/etc) and we needed a generic. So, expand to a
2023 // CAS libcall, via a CAS loop, instead.
2024 if (!Success) {
2025 expandAtomicRMWToCmpXchg(
2026 I, [this, I](IRBuilderBase &Builder, Value *Addr, Value *Loaded,
2027 Value *NewVal, Align Alignment, AtomicOrdering MemOpOrder,
2028 SyncScope::ID SSID, bool IsVolatile, Value *&Success,
2029 Value *&NewLoaded, Instruction *MetadataSrc) {
2030 // Create the CAS instruction normally...
2031 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
2032 Addr, Loaded, NewVal, Alignment, MemOpOrder,
2034 Pair->setVolatile(IsVolatile);
2035 if (MetadataSrc)
2036 copyMetadataForAtomic(*Pair, *MetadataSrc);
2037
2038 Success = Builder.CreateExtractValue(Pair, 1, "success");
2039 NewLoaded = Builder.CreateExtractValue(Pair, 0, "newloaded");
2040
2041 // ...and then expand the CAS into a libcall.
2042 expandAtomicCASToLibcall(
2043 Pair,
2044 "atomicrmw " + AtomicRMWInst::getOperationName(I->getOperation()),
2045 MetadataSrc);
2046 });
2047 }
2048}
2049
2050// A helper routine for the above expandAtomic*ToLibcall functions.
2051//
2052// 'Libcalls' contains an array of enum values for the particular
2053// ATOMIC libcalls to be emitted. All of the other arguments besides
2054// 'I' are extracted from the Instruction subclass by the
2055// caller. Depending on the particular call, some will be null.
2056bool AtomicExpandImpl::expandAtomicOpToLibcall(
2057 Instruction *I, unsigned Size, Align Alignment, Value *PointerOperand,
2058 Value *ValueOperand, Value *CASExpected, AtomicOrdering Ordering,
2059 AtomicOrdering Ordering2, ArrayRef<RTLIB::Libcall> Libcalls) {
2060 assert(Libcalls.size() == 6);
2061
2062 LLVMContext &Ctx = I->getContext();
2063 Module *M = I->getModule();
2064 const DataLayout &DL = M->getDataLayout();
2065 IRBuilder<> Builder(I);
2066 IRBuilder<> AllocaBuilder(&I->getFunction()->getEntryBlock().front());
2067
2068 bool UseSizedLibcall = canUseSizedAtomicCall(Size, Alignment, DL);
2069 Type *SizedIntTy = Type::getIntNTy(Ctx, Size * 8);
2070
2071 if (M->getTargetTriple().isOSWindows() && M->getTargetTriple().isX86_64() &&
2072 Size == 16) {
2073 // x86_64 Windows passes i128 as an XMM vector; on return, it is in
2074 // XMM0, and as a parameter, it is passed indirectly. The generic lowering
2075 // rules handles this correctly if we pass it as a v2i64 rather than
2076 // i128. This is what Clang does in the frontend for such types as well
2077 // (see WinX86_64ABIInfo::classify in Clang).
2078 SizedIntTy = FixedVectorType::get(Type::getInt64Ty(Ctx), 2);
2079 }
2080
2081 const Align AllocaAlignment = DL.getPrefTypeAlign(SizedIntTy);
2082
2083 // TODO: the "order" argument type is "int", not int32. So
2084 // getInt32Ty may be wrong if the arch uses e.g. 16-bit ints.
2085 assert(Ordering != AtomicOrdering::NotAtomic && "expect atomic MO");
2086 Constant *OrderingVal =
2087 ConstantInt::get(Type::getInt32Ty(Ctx), (int)toCABI(Ordering));
2088 Constant *Ordering2Val = nullptr;
2089 if (CASExpected) {
2090 assert(Ordering2 != AtomicOrdering::NotAtomic && "expect atomic MO");
2091 Ordering2Val =
2092 ConstantInt::get(Type::getInt32Ty(Ctx), (int)toCABI(Ordering2));
2093 }
2094 bool HasResult = I->getType() != Type::getVoidTy(Ctx);
2095
2096 RTLIB::Libcall RTLibType;
2097 if (UseSizedLibcall) {
2098 switch (Size) {
2099 case 1:
2100 RTLibType = Libcalls[1];
2101 break;
2102 case 2:
2103 RTLibType = Libcalls[2];
2104 break;
2105 case 4:
2106 RTLibType = Libcalls[3];
2107 break;
2108 case 8:
2109 RTLibType = Libcalls[4];
2110 break;
2111 case 16:
2112 RTLibType = Libcalls[5];
2113 break;
2114 }
2115 } else if (Libcalls[0] != RTLIB::UNKNOWN_LIBCALL) {
2116 RTLibType = Libcalls[0];
2117 } else {
2118 // Can't use sized function, and there's no generic for this
2119 // operation, so give up.
2120 return false;
2121 }
2122
2123 RTLIB::LibcallImpl LibcallImpl = LibcallLowering->getLibcallImpl(RTLibType);
2124 if (LibcallImpl == RTLIB::Unsupported) {
2125 // This target does not implement the requested atomic libcall so give up.
2126 return false;
2127 }
2128
2129 // Build up the function call. There's two kinds. First, the sized
2130 // variants. These calls are going to be one of the following (with
2131 // N=1,2,4,8,16):
2132 // iN __atomic_load_N(iN *ptr, int ordering)
2133 // void __atomic_store_N(iN *ptr, iN val, int ordering)
2134 // iN __atomic_{exchange|fetch_*}_N(iN *ptr, iN val, int ordering)
2135 // bool __atomic_compare_exchange_N(iN *ptr, iN *expected, iN desired,
2136 // int success_order, int failure_order)
2137 //
2138 // Note that these functions can be used for non-integer atomic
2139 // operations, the values just need to be bitcast to integers on the
2140 // way in and out.
2141 //
2142 // And, then, the generic variants. They look like the following:
2143 // void __atomic_load(size_t size, void *ptr, void *ret, int ordering)
2144 // void __atomic_store(size_t size, void *ptr, void *val, int ordering)
2145 // void __atomic_exchange(size_t size, void *ptr, void *val, void *ret,
2146 // int ordering)
2147 // bool __atomic_compare_exchange(size_t size, void *ptr, void *expected,
2148 // void *desired, int success_order,
2149 // int failure_order)
2150 //
2151 // The different signatures are built up depending on the
2152 // 'UseSizedLibcall', 'CASExpected', 'ValueOperand', and 'HasResult'
2153 // variables.
2154
2155 AllocaInst *AllocaCASExpected = nullptr;
2156 AllocaInst *AllocaValue = nullptr;
2157 AllocaInst *AllocaResult = nullptr;
2158
2159 Type *ResultTy;
2161 AttributeList Attr;
2162
2163 // 'size' argument.
2164 if (!UseSizedLibcall) {
2165 // Note, getIntPtrType is assumed equivalent to size_t.
2166 Args.push_back(ConstantInt::get(DL.getIntPtrType(Ctx), Size));
2167 }
2168
2169 // 'ptr' argument.
2170 // note: This assumes all address spaces share a common libfunc
2171 // implementation and that addresses are convertable. For systems without
2172 // that property, we'd need to extend this mechanism to support AS-specific
2173 // families of atomic intrinsics.
2174 Value *PtrVal = PointerOperand;
2175 PtrVal = Builder.CreateAddrSpaceCast(PtrVal, PointerType::getUnqual(Ctx));
2176 Args.push_back(PtrVal);
2177
2178 // 'expected' argument, if present.
2179 if (CASExpected) {
2180 AllocaCASExpected = AllocaBuilder.CreateAlloca(CASExpected->getType());
2181 AllocaCASExpected->setAlignment(AllocaAlignment);
2182 Builder.CreateLifetimeStart(AllocaCASExpected);
2183 Builder.CreateAlignedStore(CASExpected, AllocaCASExpected, AllocaAlignment);
2184 Args.push_back(AllocaCASExpected);
2185 }
2186
2187 // 'val' argument ('desired' for cas), if present.
2188 if (ValueOperand) {
2189 if (UseSizedLibcall) {
2190 Value *IntValue =
2191 Builder.CreateBitPreservingCastChain(DL, ValueOperand, SizedIntTy);
2192 Args.push_back(IntValue);
2193 } else {
2194 AllocaValue = AllocaBuilder.CreateAlloca(ValueOperand->getType());
2195 AllocaValue->setAlignment(AllocaAlignment);
2196 Builder.CreateLifetimeStart(AllocaValue);
2197 Builder.CreateAlignedStore(ValueOperand, AllocaValue, AllocaAlignment);
2198 Args.push_back(AllocaValue);
2199 }
2200 }
2201
2202 // 'ret' argument.
2203 if (!CASExpected && HasResult && !UseSizedLibcall) {
2204 AllocaResult = AllocaBuilder.CreateAlloca(I->getType());
2205 AllocaResult->setAlignment(AllocaAlignment);
2206 Builder.CreateLifetimeStart(AllocaResult);
2207 Args.push_back(AllocaResult);
2208 }
2209
2210 // 'ordering' ('success_order' for cas) argument.
2211 Args.push_back(OrderingVal);
2212
2213 // 'failure_order' argument, if present.
2214 if (Ordering2Val)
2215 Args.push_back(Ordering2Val);
2216
2217 // Now, the return type.
2218 if (CASExpected) {
2219 ResultTy = Type::getInt1Ty(Ctx);
2220 Attr = Attr.addRetAttribute(Ctx, Attribute::ZExt);
2221 } else if (HasResult && UseSizedLibcall)
2222 ResultTy = SizedIntTy;
2223 else
2224 ResultTy = Type::getVoidTy(Ctx);
2225
2226 // Done with setting up arguments and return types, create the call:
2228 for (Value *Arg : Args)
2229 ArgTys.push_back(Arg->getType());
2230 FunctionType *FnType = FunctionType::get(ResultTy, ArgTys, false);
2231 FunctionCallee LibcallFn = M->getOrInsertFunction(
2233 Attr);
2234 CallInst *Call = Builder.CreateCall(LibcallFn, Args);
2235 Call->setAttributes(Attr);
2236 Value *Result = Call;
2237
2238 // And then, extract the results...
2239 if (ValueOperand && !UseSizedLibcall)
2240 Builder.CreateLifetimeEnd(AllocaValue);
2241
2242 if (CASExpected) {
2243 // The final result from the CAS is {load of 'expected' alloca, bool result
2244 // from call}
2245 Type *FinalResultTy = I->getType();
2246 Value *V = PoisonValue::get(FinalResultTy);
2247 Value *ExpectedOut = Builder.CreateAlignedLoad(
2248 CASExpected->getType(), AllocaCASExpected, AllocaAlignment);
2249 Builder.CreateLifetimeEnd(AllocaCASExpected);
2250 V = Builder.CreateInsertValue(V, ExpectedOut, 0);
2251 V = Builder.CreateInsertValue(V, Result, 1);
2253 } else if (HasResult) {
2254 Value *V;
2255 if (UseSizedLibcall) {
2256 // Add bitcasts from Result's scalar type to I's <n x ptr> vector type
2257 auto *PtrTy = dyn_cast<PointerType>(I->getType()->getScalarType());
2258 auto *VTy = dyn_cast<VectorType>(I->getType());
2259 if (VTy && PtrTy && !Result->getType()->isVectorTy()) {
2260 unsigned AS = PtrTy->getAddressSpace();
2261 Value *BC = Builder.CreateBitCast(
2262 Result, VTy->getWithNewType(DL.getIntPtrType(Ctx, AS)));
2263 V = Builder.CreateIntToPtr(BC, I->getType());
2264 } else
2265 V = Builder.CreateBitOrPointerCast(Result, I->getType());
2266 } else {
2267 V = Builder.CreateAlignedLoad(I->getType(), AllocaResult,
2268 AllocaAlignment);
2269 Builder.CreateLifetimeEnd(AllocaResult);
2270 }
2271 I->replaceAllUsesWith(V);
2272 }
2273 I->eraseFromParent();
2274 return true;
2275}
#define Success
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static Value * performMaskedAtomicOp(AtomicRMWInst::BinOp Op, IRBuilderBase &Builder, Value *Loaded, Value *Shifted_Inc, Value *Inc, const PartwordMaskValues &PMV)
Emit IR to implement a masked version of a given atomicrmw operation.
static PartwordMaskValues createMaskInstrs(IRBuilderBase &Builder, Instruction *I, Type *ValueType, Value *Addr, Align AddrAlign, unsigned MinWordSize)
This is a helper function which builds instructions to provide values necessary for partword atomic o...
static bool canUseSizedAtomicCall(unsigned Size, Align Alignment, const DataLayout &DL)
static void createCmpXchgInstFun(IRBuilderBase &Builder, Value *Addr, Value *Loaded, Value *NewVal, Align AddrAlign, AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile, Value *&Success, Value *&NewLoaded, Instruction *MetadataSrc)
static Value * extractMaskedValue(IRBuilderBase &Builder, Value *WideWord, const PartwordMaskValues &PMV)
Expand Atomic static false unsigned getAtomicOpSize(LoadInst *LI)
static void writeUnsupportedAtomicSizeReason(const TargetLowering *TLI, Inst *I, raw_ostream &OS)
static bool atomicSizeSupported(const TargetLowering *TLI, Inst *I)
static Value * insertMaskedValue(IRBuilderBase &Builder, Value *WideWord, Value *Updated, const PartwordMaskValues &PMV)
static void copyMetadataForAtomic(Instruction &Dest, const Instruction &Source)
Copy metadata that's safe to preserve when widening atomics.
static ArrayRef< RTLIB::Libcall > GetRMWLibcall(AtomicRMWInst::BinOp Op)
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static bool runOnFunction(Function &F, bool PostInlining)
#define DEBUG_TYPE
Module.h This file contains the declarations for the Module class.
static bool isIdempotentRMW(AtomicRMWInst &RMWI)
Return true if and only if the given instruction does not modify the memory location referenced.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Machine Check Debug Module
This file provides utility for Memory Model Relaxation Annotations (MMRAs).
#define T
FunctionAnalysisManager FAM
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
This file contains the declarations for profiling metadata utility functions.
const char * Msg
This file defines the SmallString class.
This file defines the SmallVector class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
This file describes how to lower LLVM code to machine code.
Target-Independent Code Generator Pass Configuration Options pass.
void setAlignment(Align Align)
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
An instruction that atomically checks whether a specified value is in a memory location,...
AtomicOrdering getMergedOrdering() const
Returns a single ordering which is at least as strong as both the success and failure orderings for t...
void setWeak(bool IsWeak)
bool isVolatile() const
Return true if this is a cmpxchg from a volatile memory location.
AtomicOrdering getFailureOrdering() const
Returns the failure ordering constraint of this cmpxchg instruction.
static AtomicOrdering getStrongestFailureOrdering(AtomicOrdering SuccessOrdering)
Returns the strongest permitted ordering on failure, given the desired ordering on success.
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
bool isWeak() const
Return true if this cmpxchg may spuriously fail.
void setVolatile(bool V)
Specify whether this is a volatile cmpxchg.
AtomicOrdering getSuccessOrdering() const
Returns the success ordering constraint of this cmpxchg instruction.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this cmpxchg instruction.
LLVM_ABI PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)
an instruction that atomically reads a memory location, combines it with another value,...
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
bool isVolatile() const
Return true if this is a RMW on a volatile memory location.
void setVolatile(bool V)
Specify whether this is a volatile RMW or not.
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Add
*p = old + v
@ FAdd
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ FMinimum
*p = minimum(old, v) minimum matches the behavior of llvm.minimum.
@ Min
*p = old <signed v ? old : v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FMaximum
*p = maximum(old, v) maximum matches the behavior of llvm.maximum.
@ FSub
*p = old - v
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMaximumNum
*p = maximumnum(old, v) maximumnum matches the behavior of llvm.maximumnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ FMinimumNum
*p = minimumnum(old, v) minimumnum matches the behavior of llvm.minimumnum.
@ Nand
*p = ~(old & v)
Value * getPointerOperand()
BinOp getOperation() const
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
static LLVM_ABI StringRef getOperationName(BinOp Op)
AtomicOrdering getOrdering() const
Returns the ordering constraint of this rmw instruction.
iterator end()
Definition BasicBlock.h:474
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:461
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
reverse_iterator rbegin()
Definition BasicBlock.h:477
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
Definition BasicBlock.h:206
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
reverse_iterator rend()
Definition BasicBlock.h:479
void setAttributes(AttributeList A)
Set the attributes for this call.
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
ArrayRef< unsigned > getIndices() const
unsigned getNumIndices() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:867
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
BasicBlockListType::iterator iterator
Definition Function.h:70
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:723
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
AtomicCmpXchgInst * CreateAtomicCmpXchg(Value *Ptr, Value *Cmp, Value *New, MaybeAlign Align, AtomicOrdering SuccessOrdering, AtomicOrdering FailureOrdering, SyncScope::ID SSID=SyncScope::System)
Definition IRBuilder.h:1968
Value * CreateInsertValue(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2716
LLVM_ABI CallInst * CreateLifetimeStart(Value *Ptr)
Create a lifetime.start intrinsic.
LLVM_ABI CallInst * CreateLifetimeEnd(Value *Ptr)
Create a lifetime.end intrinsic.
LoadInst * CreateAlignedLoad(Type *Ty, Value *Ptr, MaybeAlign Align, const char *Name)
Definition IRBuilder.h:1934
CondBrInst * CreateCondBr(Value *Cond, BasicBlock *True, BasicBlock *False, MDNode *BranchWeights=nullptr, MDNode *Unpredictable=nullptr)
Create a conditional 'br Cond, TrueDest, FalseDest' instruction.
Definition IRBuilder.h:1216
UnreachableInst * CreateUnreachable()
Definition IRBuilder.h:1358
Value * CreateExtractValue(Value *Agg, ArrayRef< unsigned > Idxs, const Twine &Name="")
Definition IRBuilder.h:2709
BasicBlock::iterator GetInsertPoint() const
Definition IRBuilder.h:176
Value * CreateIntToPtr(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2238
Value * CreateCast(Instruction::CastOps Op, Value *V, Type *DestTy, const Twine &Name="", MDNode *FPMathTag=nullptr, FMFSource FMFSource={})
Definition IRBuilder.h:2277
BasicBlock * GetInsertBlock() const
Definition IRBuilder.h:175
LLVM_ABI Value * CreateBitPreservingCastChain(const DataLayout &DL, Value *V, Type *NewTy)
Create a chain of casts to convert V to NewTy, preserving the bit pattern of V.
Value * CreateICmpNE(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2379
UncondBrInst * CreateBr(BasicBlock *Dest)
Create an unconditional 'br label X' instruction.
Definition IRBuilder.h:1210
Value * CreateBitOrPointerCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2325
PHINode * CreatePHI(Type *Ty, unsigned NumReservedValues, const Twine &Name="")
Definition IRBuilder.h:2540
Value * CreateICmpEQ(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2375
void setIsFPConstrained(bool IsCon)
Enable/Disable use of constrained floating point math.
Definition IRBuilder.h:306
Value * CreateBitCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2243
LoadInst * CreateLoad(Type *Ty, Value *Ptr, const char *Name)
Provided to resolve 'CreateLoad(Ty, Ptr, "...")' correctly, instead of converting the string to 'bool...
Definition IRBuilder.h:1906
Value * CreateShl(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1511
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2121
LLVMContext & getContext() const
Definition IRBuilder.h:177
Value * CreateAnd(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:1570
Value * CreatePtrToInt(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2233
CallInst * CreateCall(FunctionType *FTy, Value *Callee, ArrayRef< Value * > Args={}, const Twine &Name="", MDNode *FPMathTag=nullptr)
Definition IRBuilder.h:2554
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Definition IRBuilder.h:181
StoreInst * CreateAlignedStore(Value *Val, Value *Ptr, MaybeAlign Align, bool isVolatile=false)
Definition IRBuilder.h:1953
Value * CreateOr(Value *LHS, Value *RHS, const Twine &Name="", bool IsDisjoint=false)
Definition IRBuilder.h:1592
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2248
AtomicRMWInst * CreateAtomicRMW(AtomicRMWInst::BinOp Op, Value *Ptr, Value *Val, MaybeAlign Align, AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System, bool Elementwise=false)
Definition IRBuilder.h:1981
Provides an 'InsertHelper' that calls a user-provided callback after performing the default insertion...
Definition IRBuilder.h:75
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2893
InstSimplifyFolder - Use InstructionSimplify to fold operations to existing values.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void moveAfter(Instruction *MovePos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
LLVM_ABI void getSyncScopeNames(SmallVectorImpl< StringRef > &SSNs) const
getSyncScopeNames - Populates client supplied SmallVector with synchronization scope names registered...
Tracks which library functions to use for a particular subtarget.
RTLIB::LibcallImpl getLibcallImpl(RTLIB::Libcall Call) const
Return the lowering's selection of implementation call for Call.
An instruction for reading from memory.
Value * getPointerOperand()
bool isVolatile() const
Return true if this is a load from a volatile memory location.
void setAtomic(AtomicOrdering Ordering, SyncScope::ID SSID=SyncScope::System)
Sets the ordering constraint and the synchronization scope ID of this load instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this load instruction.
void setVolatile(bool V)
Specify whether this is a volatile load or not.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this load instruction.
LoadStoreInstProperties getProperties() const
Returns the properties of this load instruction.
Align getAlign() const
Return the alignment of the access that is being performed.
Metadata node.
Definition Metadata.h:1069
Record a mapping from subtarget to LibcallLoweringInfo.
const LibcallLoweringInfo & getLibcallLowering(const TargetSubtargetInfo &Subtarget) const
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
LLVMContext & getContext() const
Get the global data context.
Definition Module.h:327
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
virtual void getAnalysisUsage(AnalysisUsage &) const
getAnalysisUsage - This function should be overriden by passes that need analysis information to do t...
Definition Pass.cpp:112
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses none()
Convenience factory function for the empty preserved set.
Definition Analysis.h:115
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
virtual Value * emitStoreConditional(IRBuilderBase &Builder, Value *Val, Value *Addr, AtomicOrdering Ord) const
Perform a store-conditional operation to Addr.
EVT getMemValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
virtual void emitBitTestAtomicRMWIntrinsic(AtomicRMWInst *AI) const
Perform a bit test atomicrmw using a target-specific intrinsic.
virtual AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
virtual bool shouldInsertFencesForAtomic(const Instruction *I) const
Whether AtomicExpandPass should automatically insert fences and reduce ordering for this atomic.
virtual AtomicOrdering atomicOperationOrderAfterFenceSplit(const Instruction *I) const
virtual void emitExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) const
Perform a cmpxchg expansion using a target-specific method.
unsigned getMinCmpXchgSizeInBits() const
Returns the size of the smallest cmpxchg or ll/sc instruction the backend supports.
virtual Value * emitMaskedAtomicRMWIntrinsic(IRBuilderBase &Builder, AtomicRMWInst *AI, Value *AlignedAddr, Value *Incr, Value *Mask, Value *ShiftAmt, AtomicOrdering Ord) const
Perform a masked atomicrmw using a target-specific intrinsic.
virtual AtomicExpansionKind shouldExpandAtomicCmpXchgInIR(const AtomicCmpXchgInst *AI) const
Returns how the given atomic cmpxchg should be expanded by the IR-level AtomicExpand pass.
virtual Value * emitLoadLinked(IRBuilderBase &Builder, Type *ValueTy, Value *Addr, AtomicOrdering Ord) const
Perform a load-linked operation on Addr, returning a "Value *" with the corresponding pointee type.
virtual void emitExpandAtomicRMW(AtomicRMWInst *AI) const
Perform a atomicrmw expansion using a target-specific way.
virtual void emitAtomicCmpXchgNoStoreLLBalance(IRBuilderBase &Builder) const
virtual void emitExpandAtomicStore(StoreInst *SI) const
Perform a atomic store using a target-specific way.
virtual AtomicExpansionKind shouldCastAtomicRMWIInIR(AtomicRMWInst *RMWI) const
Returns how the given atomic atomicrmw should be cast by the IR-level AtomicExpand pass.
virtual bool shouldInsertTrailingSeqCstFenceForAtomicStore(const Instruction *I) const
Whether AtomicExpandPass should automatically insert a seq_cst trailing fence without reducing the or...
virtual AtomicExpansionKind shouldExpandAtomicLoadInIR(LoadInst *LI) const
Returns how the given (atomic) load should be expanded by the IR-level AtomicExpand pass.
virtual Value * emitMaskedAtomicCmpXchgIntrinsic(IRBuilderBase &Builder, AtomicCmpXchgInst *CI, Value *AlignedAddr, Value *CmpVal, Value *NewVal, Value *Mask, AtomicOrdering Ord) const
Perform a masked cmpxchg using a target-specific intrinsic.
virtual bool shouldIssueAtomicLoadForAtomicEmulationLoop(void) const
unsigned getMaxAtomicSizeInBitsSupported() const
Returns the maximum atomic operation size (in bits) supported by the backend.
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
virtual void emitExpandAtomicLoad(LoadInst *LI) const
Perform a atomic load using a target-specific way.
virtual AtomicExpansionKind shouldExpandAtomicStoreInIR(StoreInst *SI) const
Returns how the given (atomic) store should be expanded by the IR-level AtomicExpand pass into.
virtual void emitCmpArithAtomicRMWIntrinsic(AtomicRMWInst *AI) const
Perform a atomicrmw which the result is only used by comparison, using a target-specific intrinsic.
virtual AtomicExpansionKind shouldCastAtomicStoreInIR(StoreInst *SI) const
Returns how the given (atomic) store should be cast by the IR-level AtomicExpand pass into.
virtual Instruction * emitTrailingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const
virtual AtomicExpansionKind shouldCastAtomicLoadInIR(LoadInst *LI) const
Returns how the given (atomic) load should be cast by the IR-level AtomicExpand pass.
virtual Instruction * emitLeadingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const
Inserts in the IR a target-specific intrinsic specifying a fence.
virtual LoadInst * lowerIdempotentRMWIntoFencedLoad(AtomicRMWInst *RMWI) const
On some platforms, an AtomicRMW that never actually modifies the value (such as fetch_add of 0) can b...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
Primary interface to the complete machine description for the target machine.
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
Target-Independent Code Generator Pass Configuration Options.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:285
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:313
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
iterator_range< user_iterator > users()
Definition Value.h:426
bool use_empty() const
Definition Value.h:346
An efficient, type-erasing, non-owning reference to a callable.
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:81
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI bool canInstructionHaveMMRAs(const Instruction &I)
LLVM_ABI void setExplicitlyUnknownBranchWeightsIfProfiled(Instruction &I, StringRef PassName, const Function *F=nullptr)
Like setExplicitlyUnknownBranchWeights(...), but only sets unknown branch weights in the new instruct...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
OuterAnalysisManagerProxy< ModuleAnalysisManager, Function > ModuleAnalysisManagerFunctionProxy
Provide the ModuleAnalysisManager to Function proxy.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
bool isReleaseOrStronger(AtomicOrdering AO)
AtomicOrderingCABI toCABI(AtomicOrdering AO)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI Value * buildAtomicRMWValue(AtomicRMWInst::BinOp Op, IRBuilderBase &Builder, Value *Loaded, Value *Val)
Emit IR to implement the given atomicrmw operation on values in registers, returning the new value.
AtomicOrdering
Atomic ordering for LLVM's memory model.
IRBuilder(LLVMContext &, FolderTy, InserterTy, MDNode *, ArrayRef< OperandBundleDef >) -> IRBuilder< FolderTy, InserterTy >
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
ArrayRef(const T &OneElt) -> ArrayRef< T >
bool isAcquireOrStronger(AtomicOrdering AO)
constexpr unsigned BitWidth
LLVM_ABI bool lowerAtomicCmpXchgInst(AtomicCmpXchgInst *CXI)
Convert the given Cmpxchg into primitive load and compare.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool lowerAtomicRMWInst(AtomicRMWInst *RMWI)
Convert the given RMWI into primitive load and stores, assuming that doing so is legal.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
LLVM_ABI FunctionPass * createAtomicExpandLegacyPass()
AtomicExpandPass - At IR level this pass replace atomic instructions with __atomic_* library calls,...
LLVM_ABI char & AtomicExpandID
AtomicExpandID – Lowers atomic operations in terms of either cmpxchg load-linked/store-conditional lo...
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
Definition ValueTypes.h:435
Matching combinators.
static StringRef getLibcallImplName(RTLIB::LibcallImpl CallImpl)
Get the libcall routine name for the specified libcall implementation.