31#include "llvm/IR/IntrinsicsAMDGPU.h"
36#define DEBUG_TYPE "amdgpu-atomic-optimizer"
43struct ReplacementInfo {
67class AMDGPUAtomicOptimizerImpl
80 Value *
const Identity)
const;
82 Value *
const Identity)
const;
85 std::pair<Value *, Value *>
91 bool ValDivergent,
bool IsLDS)
const;
94 AMDGPUAtomicOptimizerImpl() =
delete;
99 :
F(
F), UA(UA),
DL(
F.getDataLayout()), DTU(DTU), ST(ST),
101 ScanImpl(ScanImpl) {}
111char AMDGPUAtomicOptimizer::ID = 0;
115bool AMDGPUAtomicOptimizer::runOnFunction(
Function &
F) {
116 if (skipFunction(
F)) {
121 getAnalysis<UniformityInfoWrapperPass>().getUniformityInfo();
124 getAnalysisIfAvailable<DominatorTreeWrapperPass>();
126 DomTreeUpdater::UpdateStrategy::Lazy);
132 return AMDGPUAtomicOptimizerImpl(
F, UA, DTU, ST, ScanImpl).run();
140 DomTreeUpdater::UpdateStrategy::Lazy);
143 bool IsChanged = AMDGPUAtomicOptimizerImpl(
F, UA, DTU, ST, ScanImpl).run();
154bool AMDGPUAtomicOptimizerImpl::run() {
162 if (ToReplace.empty())
165 for (
auto &[
I,
Op, ValIdx, ValDivergent, IsLDS] : ToReplace)
166 optimizeAtomic(*
I,
Op, ValIdx, ValDivergent, IsLDS);
172 switch (Ty->getTypeID()) {
177 unsigned Size = Ty->getIntegerBitWidth();
185void AMDGPUAtomicOptimizerImpl::visitAtomicRMWInst(AtomicRMWInst &
I) {
186 if (
I.getType()->isVectorTy() ||
I.isVolatile())
190 switch (
I.getPointerAddressSpace()) {
221 !(
I.getType()->isFloatTy() ||
I.getType()->isDoubleTy())) {
225 const unsigned PtrIdx = 0;
226 const unsigned ValIdx = 1;
241 if (ScanImpl == ScanOptions::DPP && !ST.hasDPP())
252 if (IsLDS && ValDivergent && ScanImpl == ScanOptions::Iterative &&
254 TargetLowering::AtomicExpansionKind::None)
260 ToReplace.push_back({&
I,
Op, ValIdx, ValDivergent, IsLDS});
263void AMDGPUAtomicOptimizerImpl::visitIntrinsicInst(IntrinsicInst &
I) {
264 if (
I.getType()->isVectorTy())
269 switch (
I.getIntrinsicID()) {
272 case Intrinsic::amdgcn_struct_buffer_atomic_add:
273 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_add:
274 case Intrinsic::amdgcn_raw_buffer_atomic_add:
275 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_add:
278 case Intrinsic::amdgcn_struct_buffer_atomic_sub:
279 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_sub:
280 case Intrinsic::amdgcn_raw_buffer_atomic_sub:
281 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub:
284 case Intrinsic::amdgcn_struct_buffer_atomic_and:
285 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_and:
286 case Intrinsic::amdgcn_raw_buffer_atomic_and:
287 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_and:
290 case Intrinsic::amdgcn_struct_buffer_atomic_or:
291 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_or:
292 case Intrinsic::amdgcn_raw_buffer_atomic_or:
293 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_or:
296 case Intrinsic::amdgcn_struct_buffer_atomic_xor:
297 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_xor:
298 case Intrinsic::amdgcn_raw_buffer_atomic_xor:
299 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_xor:
302 case Intrinsic::amdgcn_struct_buffer_atomic_smin:
303 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_smin:
304 case Intrinsic::amdgcn_raw_buffer_atomic_smin:
305 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_smin:
308 case Intrinsic::amdgcn_struct_buffer_atomic_umin:
309 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_umin:
310 case Intrinsic::amdgcn_raw_buffer_atomic_umin:
311 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_umin:
314 case Intrinsic::amdgcn_struct_buffer_atomic_smax:
315 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_smax:
316 case Intrinsic::amdgcn_raw_buffer_atomic_smax:
317 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_smax:
320 case Intrinsic::amdgcn_struct_buffer_atomic_umax:
321 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_umax:
322 case Intrinsic::amdgcn_raw_buffer_atomic_umax:
323 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_umax:
332 const unsigned ValIdx = 0;
341 if (ScanImpl == ScanOptions::DPP && !ST.hasDPP())
350 for (
unsigned Idx = 1;
Idx <
I.getNumOperands();
Idx++) {
359 ToReplace.push_back({&
I,
Op, ValIdx, ValDivergent,
false});
372 return B.CreateBinOp(Instruction::Add,
LHS,
RHS);
376 return B.CreateBinOp(Instruction::Sub,
LHS,
RHS);
380 return B.CreateBinOp(Instruction::And,
LHS,
RHS);
382 return B.CreateBinOp(Instruction::Or,
LHS,
RHS);
384 return B.CreateBinOp(Instruction::Xor,
LHS,
RHS);
399 return B.CreateMaxNum(
LHS,
RHS);
401 return B.CreateMinNum(
LHS,
RHS);
412 Value *
const Identity)
const {
413 Type *AtomicTy =
V->getType();
420 B.CreateIntrinsic(Intrinsic::amdgcn_update_dpp, AtomicTy,
421 {Identity, V, B.getInt32(DPP::ROW_XMASK0 | 1 << Idx),
422 B.getInt32(0xf), B.getInt32(0xf), B.getFalse()}));
426 assert(ST.hasPermlane16Insts());
427 Value *Permlanex16Call =
428 B.CreateIntrinsic(AtomicTy, Intrinsic::amdgcn_permlanex16,
430 B.getInt32(0),
B.getFalse(),
B.getFalse()});
438 Value *Permlane64Call =
439 B.CreateIntrinsic(AtomicTy, Intrinsic::amdgcn_permlane64,
V);
446 M, Intrinsic::amdgcn_readlane, AtomicTy);
447 Value *Lane0 =
B.CreateCall(ReadLane, {
V,
B.getInt32(0)});
448 Value *Lane32 =
B.CreateCall(ReadLane, {
V,
B.getInt32(32)});
456 Value *Identity)
const {
457 Type *AtomicTy =
V->getType();
460 M, Intrinsic::amdgcn_update_dpp, AtomicTy);
465 B.CreateCall(UpdateDPP,
466 {Identity, V, B.getInt32(DPP::ROW_SHR0 | 1 << Idx),
467 B.getInt32(0xf), B.getInt32(0xf), B.getFalse()}));
469 if (ST.hasDPPBroadcasts()) {
473 B.CreateCall(UpdateDPP,
474 {Identity, V, B.getInt32(DPP::BCAST15), B.getInt32(0xa),
475 B.getInt32(0xf), B.getFalse()}));
478 B.CreateCall(UpdateDPP,
479 {Identity, V, B.getInt32(DPP::BCAST31), B.getInt32(0xc),
480 B.getInt32(0xf), B.getFalse()}));
487 assert(ST.hasPermlane16Insts());
489 B.CreateIntrinsic(AtomicTy, Intrinsic::amdgcn_permlanex16,
491 B.getInt32(-1),
B.getFalse(),
B.getFalse()});
493 Value *UpdateDPPCall =
B.CreateCall(
495 B.getInt32(0xa),
B.getInt32(0xf),
B.getFalse()});
500 Value *
const Lane31 =
B.CreateIntrinsic(
501 AtomicTy, Intrinsic::amdgcn_readlane, {
V,
B.getInt32(31)});
503 Value *UpdateDPPCall =
B.CreateCall(
505 B.getInt32(0xc),
B.getInt32(0xf),
B.getFalse()});
516 Value *Identity)
const {
517 Type *AtomicTy =
V->getType();
520 M, Intrinsic::amdgcn_update_dpp, AtomicTy);
521 if (ST.hasDPPWavefrontShifts()) {
523 V =
B.CreateCall(UpdateDPP,
525 B.getInt32(0xf),
B.getFalse()});
528 M, Intrinsic::amdgcn_readlane, AtomicTy);
530 M, Intrinsic::amdgcn_writelane, AtomicTy);
535 V =
B.CreateCall(UpdateDPP,
537 B.getInt32(0xf),
B.getInt32(0xf),
B.getFalse()});
540 V =
B.CreateCall(WriteLane, {
B.CreateCall(ReadLane, {Old,
B.getInt32(15)}),
547 {
B.CreateCall(ReadLane, {Old,
B.getInt32(31)}),
B.getInt32(32),
V});
552 {
B.CreateCall(ReadLane, {Old,
B.getInt32(47)}),
B.getInt32(48),
V});
564std::pair<Value *, Value *> AMDGPUAtomicOptimizerImpl::buildScanIteratively(
566 Instruction &
I, BasicBlock *ComputeLoop, BasicBlock *ComputeEnd)
const {
567 auto *Ty =
I.getType();
569 auto *EntryBB =
I.getParent();
570 auto NeedResult = !
I.use_empty();
573 B.CreateIntrinsic(Intrinsic::amdgcn_ballot, WaveTy,
B.getTrue());
576 B.SetInsertPoint(ComputeLoop);
580 PHINode *OldValuePhi =
nullptr;
582 OldValuePhi =
B.CreatePHI(Ty, 2,
"OldValuePhi");
585 auto *ActiveBits =
B.CreatePHI(WaveTy, 2,
"ActiveBits");
586 ActiveBits->addIncoming(Ballot, EntryBB);
590 B.CreateIntrinsic(Intrinsic::cttz, WaveTy, {ActiveBits,
B.getTrue()});
592 auto *LaneIdxInt =
B.CreateTrunc(FF1,
B.getInt32Ty());
595 Value *LaneValue =
B.CreateIntrinsic(
V->getType(), Intrinsic::amdgcn_readlane,
600 Value *OldValue =
nullptr;
602 OldValue =
B.CreateIntrinsic(
V->getType(), Intrinsic::amdgcn_writelane,
603 {Accumulator, LaneIdxInt, OldValuePhi});
609 Accumulator->addIncoming(NewAccumulator, ComputeLoop);
613 auto *
Mask =
B.CreateShl(ConstantInt::get(WaveTy, 1), FF1);
615 auto *InverseMask =
B.CreateXor(Mask, ConstantInt::getAllOnesValue(WaveTy));
616 auto *NewActiveBits =
B.CreateAnd(ActiveBits, InverseMask);
617 ActiveBits->addIncoming(NewActiveBits, ComputeLoop);
620 auto *IsEnd =
B.CreateICmpEQ(NewActiveBits, ConstantInt::get(WaveTy, 0));
621 B.CreateCondBr(IsEnd, ComputeEnd, ComputeLoop);
623 B.SetInsertPoint(ComputeEnd);
625 return {OldValue, NewAccumulator};
631 const unsigned BitWidth = Ty->getPrimitiveSizeInBits();
666 "Atomic Op yet to be ported to use Wave Reduction intrinsics.");
669 return Intrinsic::amdgcn_wave_reduce_add;
672 return Intrinsic::amdgcn_wave_reduce_fadd;
674 return Intrinsic::amdgcn_wave_reduce_and;
676 return Intrinsic::amdgcn_wave_reduce_or;
678 return Intrinsic::amdgcn_wave_reduce_xor;
680 return Intrinsic::amdgcn_wave_reduce_umax;
682 return Intrinsic::amdgcn_wave_reduce_max;
684 return Intrinsic::amdgcn_wave_reduce_fmax;
686 return Intrinsic::amdgcn_wave_reduce_umin;
688 return Intrinsic::amdgcn_wave_reduce_min;
690 return Intrinsic::amdgcn_wave_reduce_fmin;
699void AMDGPUAtomicOptimizerImpl::optimizeAtomic(Instruction &
I,
710 if (IsLDS && ValDivergent && ScanImpl == ScanOptions::DPP) {
711 if (MDNode *MD =
I.getMetadata(
"amdgpu.expected.active.lanes")) {
713 constexpr unsigned ActiveLanesThreshold = 5;
714 if (CI->getValue().ule(ActiveLanesThreshold))
723 B.setIsFPConstrained(
I.getFunction()->hasFnAttribute(Attribute::StrictFP));
740 Value *
const Cond =
B.CreateIntrinsic(Intrinsic::amdgcn_ps_live, {});
748 B.SetInsertPoint(&
I);
751 Type *
const Ty =
I.getType();
752 Type *Int32Ty =
B.getInt32Ty();
754 [[maybe_unused]]
const unsigned TyBitWidth =
DL.getTypeSizeInBits(Ty);
758 Value *
V =
I.getOperand(ValIdx);
763 CallInst *
const Ballot =
B.CreateIntrinsicWithoutFolding(
764 Intrinsic::amdgcn_ballot, WaveTy,
B.getTrue());
773 B.CreateIntrinsic(Intrinsic::amdgcn_mbcnt_lo, {Ballot,
B.getInt32(0)});
775 Value *
const ExtractLo =
B.CreateTrunc(Ballot, Int32Ty);
776 Value *
const ExtractHi =
B.CreateTrunc(
B.CreateLShr(Ballot, 32), Int32Ty);
777 Mbcnt =
B.CreateIntrinsic(Intrinsic::amdgcn_mbcnt_lo,
778 {ExtractLo,
B.getInt32(0)});
779 Mbcnt =
B.CreateIntrinsic(Intrinsic::amdgcn_mbcnt_hi, {ExtractHi, Mbcnt});
783 LLVMContext &
C =
F->getContext();
784 const bool NeedResult = !
I.use_empty();
785 const bool UseWaveReductionIntrinsic = !ValDivergent || !NeedResult;
797 Value *ExclScan =
nullptr;
798 Value *NewV =
nullptr;
802 if (UseWaveReductionIntrinsic) {
804 unsigned Strategy = ScanImpl == ScanOptions::DPP ? 2 : 1;
806 NewV =
B.CreateIntrinsic(WaveRedIntrinsic, Ty, {
V,
B.getInt32(Strategy)});
810 assert(ValDivergent && NeedResult);
811 if (ScanImpl == ScanOptions::DPP) {
815 B.CreateIntrinsic(Intrinsic::amdgcn_set_inactive, Ty, {
V, Identity});
816 if (!NeedResult && ST.hasPermlane16Insts()) {
820 NewV = buildReduction(
B, ScanOp, NewV, Identity);
822 NewV = buildScan(
B, ScanOp, NewV, Identity);
824 ExclScan = buildShiftRight(
B, NewV, Identity);
829 NewV =
B.CreateIntrinsic(Ty, Intrinsic::amdgcn_readlane,
830 {NewV, LastLaneIdx});
833 NewV =
B.CreateIntrinsic(Intrinsic::amdgcn_strict_wwm, Ty, NewV);
834 }
else if (ScanImpl == ScanOptions::Iterative) {
838 std::tie(ExclScan, NewV) = buildScanIteratively(
B, ScanOp, Identity,
V,
I,
839 ComputeLoop, ComputeEnd);
848 Value *
const Cond =
B.CreateICmpEQ(Mbcnt,
B.getInt32(0));
870 if (NeedResult && ValDivergent && ScanImpl == ScanOptions::Iterative) {
876 B.SetInsertPoint(ComputeEnd);
878 B.Insert(Terminator);
882 B.SetInsertPoint(OriginalBB);
883 B.CreateBr(ComputeLoop);
887 {{DominatorTree::Insert, OriginalBB, ComputeLoop},
888 {DominatorTree::Insert, ComputeLoop, ComputeEnd}});
893 DomTreeUpdates.push_back({DominatorTree::Insert, ComputeEnd, Succ});
894 DomTreeUpdates.push_back({DominatorTree::Delete, OriginalBB, Succ});
899 Predecessor = ComputeEnd;
901 Predecessor = OriginalBB;
904 B.SetInsertPoint(SingleLaneTerminator);
914 B.SetInsertPoint(&
I);
918 PHINode *
const PHI =
B.CreatePHI(Ty, 2);
920 PHI->addIncoming(NewI, SingleLaneTerminator->
getParent());
927 ReadlaneVal =
B.CreateZExt(
PHI,
B.getInt32Ty());
929 Value *BroadcastI =
B.CreateIntrinsic(
930 ReadlaneVal->
getType(), Intrinsic::amdgcn_readfirstlane, ReadlaneVal);
932 BroadcastI =
B.CreateTrunc(BroadcastI, Ty);
938 Value *LaneOffset =
nullptr;
940 if (ScanImpl == ScanOptions::DPP) {
942 B.CreateIntrinsic(Intrinsic::amdgcn_strict_wwm, Ty, ExclScan);
943 }
else if (ScanImpl == ScanOptions::Iterative) {
944 LaneOffset = ExclScan;
949 Mbcnt = isAtomicFloatingPointTy ?
B.CreateUIToFP(Mbcnt, Ty)
950 :
B.CreateIntCast(Mbcnt, Ty,
false);
966 LaneOffset =
B.CreateSelect(
Cond, Identity,
V);
969 LaneOffset =
buildMul(
B,
V,
B.CreateAnd(Mbcnt, 1));
973 LaneOffset =
B.CreateFMul(
V, Mbcnt);
979 if (isAtomicFloatingPointTy) {
997 PHINode *
const PHI =
B.CreatePHI(Ty, 2);
999 PHI->addIncoming(Result,
I.getParent());
1000 I.replaceAllUsesWith(
PHI);
1003 I.replaceAllUsesWith(Result);
1008 I.eraseFromParent();
1012 "AMDGPU atomic optimizations",
false,
false)
1019 return new AMDGPUAtomicOptimizer(ScanStrategy);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static Constant * getIdentityValueForAtomicOp(Type *const Ty, AtomicRMWInst::BinOp Op)
static bool isLegalCrossLaneType(Type *Ty)
static Value * buildMul(IRBuilder<> &B, Value *LHS, Value *RHS)
static Value * buildNonAtomicBinOp(IRBuilder<> &B, AtomicRMWInst::BinOp Op, Value *LHS, Value *RHS)
static Intrinsic::ID getWaveReductionIntrinsic(AtomicRMWInst::BinOp Op)
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool runOnFunction(Function &F, bool PostInlining)
AMD GCN specific subclass of TargetSubtarget.
Machine Check Debug Module
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
const SmallVectorImpl< MachineOperand > & Cond
static void visit(BasicBlock &Start, std::function< bool(BasicBlock *)> op)
Target-Independent Code Generator Pass Configuration Options pass.
bool isSingleLaneExecution(const Function &Kernel) const
Return true if only a single workitem can be active in a wave.
unsigned getWavefrontSize() const
static APFloat getNaN(const fltSemantics &Sem, bool Negative=false, uint64_t payload=0)
Factory for NaN values.
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
static APInt getMinValue(unsigned numBits)
Gets minimum unsigned value of APInt for a specific bit width.
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
an instruction that atomically reads a memory location, combines it with another value,...
static bool isFPOperation(BinOp Op)
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Min
*p = old <signed v ? old : v
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
LLVM_ABI InstListType::const_iterator getFirstNonPHIIt() const
Returns an iterator to the first instruction in this block that is not a PHINode instruction.
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ ICMP_SLT
signed less than
@ ICMP_UGT
unsigned greater than
@ ICMP_SGT
signed greater than
@ ICMP_ULT
unsigned less than
This is the shared class of boolean and integer constants.
bool isOne() const
This is just a convenience method to make client code smaller for a common case.
This is an important base class in LLVM.
A parsed version of the target data layout string in and methods for querying it.
Analysis pass which computes a DominatorTree.
Legacy analysis pass which computes a DominatorTree.
DominatorTree & getDomTree()
FunctionPass class - This class is used to implement most global optimizations.
bool hasPermLane64() const
const SITargetLowering * getTargetLowering() const override
void applyUpdates(ArrayRef< UpdateT > Updates)
Submit updates to all available trees.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Base class for instruction visitors.
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
void addIncoming(Value *V, BasicBlock *BB)
Add an incoming value to the end of the PHI list.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserve()
Mark an analysis as preserved.
AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Primary interface to the complete machine description for the target machine.
const STC & getSubtarget(const Function &F) const
This method returns a pointer to the specified type of TargetSubtargetInfo.
Target-Independent Code Generator Pass Configuration Options.
TMC & getTM() const
Get the right type of TargetMachine for this target.
The instances of the Type class are immutable: once they are created, they are never changed.
@ FloatTyID
32-bit floating point type
@ IntegerTyID
Arbitrary bit width integers.
@ DoubleTyID
64-bit floating point type
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
void setOperand(unsigned i, Value *Val)
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
const ParentTy * getParent() const
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ LOCAL_ADDRESS
Address space for local memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ BasicBlock
Various leaf nodes.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
friend class Instruction
Iterator for Instructions in a `BasicBlock.
This is an optimization pass for GlobalISel generic memory operations.
GenericUniformityInfo< SSAContext > UniformityInfo
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
FunctionPass * createAMDGPUAtomicOptimizerPass(ScanOptions ScanStrategy)
IRBuilder(LLVMContext &, FolderTy, InserterTy) -> IRBuilder< FolderTy, InserterTy >
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
DWARFExpression::Operation Op
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
char & AMDGPUAtomicOptimizerID
LLVM_ABI Instruction * SplitBlockAndInsertIfThen(Value *Cond, BasicBlock::iterator SplitBefore, bool Unreachable, MDNode *BranchWeights=nullptr, DomTreeUpdater *DTU=nullptr, LoopInfo *LI=nullptr, BasicBlock *ThenBlock=nullptr)
Split the containing block at the specified instruction - everything before SplitBefore stays in the ...
AnalysisManager< Function > FunctionAnalysisManager
Convenience typedef for the Function analysis manager.
PreservedAnalyses run(Function &F, FunctionAnalysisManager &AM)