30#include "llvm/IR/IntrinsicsAMDGPU.h"
33#define GET_GICOMBINER_DEPS
34#include "AMDGPUGenPreLegalizeGICombiner.inc"
35#undef GET_GICOMBINER_DEPS
37#define DEBUG_TYPE "amdgpu-postlegalizer-combiner"
43#define GET_GICOMBINER_TYPES
44#include "AMDGPUGenPostLegalizeGICombiner.inc"
45#undef GET_GICOMBINER_TYPES
47class AMDGPUPostLegalizerCombinerImpl :
public Combiner {
49 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig;
56 AMDGPUPostLegalizerCombinerImpl(
59 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
63 static const char *
getName() {
return "AMDGPUPostLegalizerCombinerImpl"; }
68 struct FMinFMaxLegacyInfo {
76 FMinFMaxLegacyInfo &Info)
const;
78 const FMinFMaxLegacyInfo &Info)
const;
88 struct CvtF32UByteMatchInfo {
94 CvtF32UByteMatchInfo &MatchInfo)
const;
96 const CvtF32UByteMatchInfo &MatchInfo)
const;
102 bool matchCombineSignExtendInReg(
103 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchInfo)
const;
104 void applyCombineSignExtendInReg(
105 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchInfo)
const;
112 bool matchCombine_s_mul_u64(
MachineInstr &
MI,
unsigned &NewOpcode)
const;
115#define GET_GICOMBINER_CLASS_MEMBERS
116#define AMDGPUSubtarget GCNSubtarget
117#include "AMDGPUGenPostLegalizeGICombiner.inc"
118#undef GET_GICOMBINER_CLASS_MEMBERS
119#undef AMDGPUSubtarget
122#define GET_GICOMBINER_IMPL
123#define AMDGPUSubtarget GCNSubtarget
124#include "AMDGPUGenPostLegalizeGICombiner.inc"
125#undef AMDGPUSubtarget
126#undef GET_GICOMBINER_IMPL
128AMDGPUPostLegalizerCombinerImpl::AMDGPUPostLegalizerCombinerImpl(
131 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
133 :
Combiner(MF, CInfo, &VT, CSEInfo), RuleConfig(RuleConfig), STI(STI),
134 TII(*STI.getInstrInfo()),
135 Helper(Observer,
B,
false, &VT, MDT, LI, STI),
137#include
"AMDGPUGenPostLegalizeGICombiner.inc"
142bool AMDGPUPostLegalizerCombinerImpl::tryCombineAll(
MachineInstr &
MI)
const {
143 if (tryCombineAllImpl(
MI))
146 switch (
MI.getOpcode()) {
147 case TargetOpcode::G_SHL:
148 case TargetOpcode::G_LSHR:
149 case TargetOpcode::G_ASHR:
159bool AMDGPUPostLegalizerCombinerImpl::matchFMinFMaxLegacy(
160 MachineInstr &
MI, MachineInstr &FCmp, FMinFMaxLegacyInfo &Info)
const {
173 if ((
Info.LHS != True ||
Info.RHS != False) &&
174 (
Info.LHS != False ||
Info.RHS != True))
180 if (
Info.LHS != True)
195void AMDGPUPostLegalizerCombinerImpl::applySelectFCmpToFMinFMaxLegacy(
196 MachineInstr &
MI,
const FMinFMaxLegacyInfo &Info)
const {
198 : AMDGPU::G_AMDGPU_FMIN_LEGACY;
208 B.buildInstr(
Opc, {
MI.getOperand(0)}, {
X,
Y},
MI.getFlags());
210 MI.eraseFromParent();
213bool AMDGPUPostLegalizerCombinerImpl::matchUCharToFloat(
214 MachineInstr &
MI)
const {
221 LLT Ty = MRI.getType(DstReg);
224 unsigned SrcSize = MRI.getType(SrcReg).getSizeInBits();
225 assert(SrcSize == 16 || SrcSize == 32 || SrcSize == 64);
233void AMDGPUPostLegalizerCombinerImpl::applyUCharToFloat(
234 MachineInstr &
MI)
const {
239 LLT Ty = MRI.getType(DstReg);
240 LLT SrcTy = MRI.getType(SrcReg);
242 SrcReg =
B.buildAnyExtOrTrunc(
S32, SrcReg).getReg(0);
245 B.buildInstr(AMDGPU::G_AMDGPU_CVT_F32_UBYTE0, {DstReg}, {SrcReg},
248 auto Cvt0 =
B.buildInstr(AMDGPU::G_AMDGPU_CVT_F32_UBYTE0, {
S32}, {SrcReg},
250 B.buildFPTrunc(DstReg, Cvt0,
MI.getFlags());
253 MI.eraseFromParent();
256bool AMDGPUPostLegalizerCombinerImpl::matchFDivSqrtToRsqF16(
257 MachineInstr &
MI)
const {
259 return MRI.hasOneNonDBGUse(Sqrt);
262void AMDGPUPostLegalizerCombinerImpl::applyFDivSqrtToRsqF16(
266 LLT DstTy = MRI.getType(Dst);
267 uint32_t
Flags =
MI.getFlags();
268 Register RSQ =
B.buildIntrinsic(Intrinsic::amdgcn_rsq, {DstTy})
272 B.buildFMul(Dst, RSQ,
Y, Flags);
273 MI.eraseFromParent();
276bool AMDGPUPostLegalizerCombinerImpl::matchCvtF32UByteN(
277 MachineInstr &
MI, CvtF32UByteMatchInfo &MatchInfo)
const {
287 const unsigned Offset =
MI.getOpcode() - AMDGPU::G_AMDGPU_CVT_F32_UBYTE0;
289 unsigned ShiftOffset = 8 *
Offset;
291 ShiftOffset += ShiftAmt;
293 ShiftOffset -= ShiftAmt;
295 MatchInfo.CvtVal = Src0;
296 MatchInfo.ShiftOffset = ShiftOffset;
297 return ShiftOffset < 32 && ShiftOffset >= 8 && (ShiftOffset % 8) == 0;
304void AMDGPUPostLegalizerCombinerImpl::applyCvtF32UByteN(
305 MachineInstr &
MI,
const CvtF32UByteMatchInfo &MatchInfo)
const {
306 unsigned NewOpc = AMDGPU::G_AMDGPU_CVT_F32_UBYTE0 + MatchInfo.ShiftOffset / 8;
310 LLT SrcTy = MRI.getType(MatchInfo.CvtVal);
313 CvtSrc =
B.buildAnyExt(
S32, CvtSrc).getReg(0);
317 B.buildInstr(NewOpc, {
MI.getOperand(0)}, {CvtSrc},
MI.getFlags());
318 MI.eraseFromParent();
321bool AMDGPUPostLegalizerCombinerImpl::matchRemoveFcanonicalize(
322 MachineInstr &
MI)
const {
323 const SITargetLowering *TLI =
static_cast<const SITargetLowering *
>(
324 MF.getSubtarget().getTargetLowering());
334bool AMDGPUPostLegalizerCombinerImpl::matchCombineSignExtendInReg(
335 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchData)
const {
337 if (!MRI.hasOneNonDBGUse(LoadReg))
342 MachineInstr *LoadMI = MRI.getVRegDef(LoadReg);
343 int64_t Width =
MI.getOperand(2).getImm();
345 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
346 MatchData = {LoadMI, AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE};
348 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
349 MatchData = {LoadMI, AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT};
351 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_UBYTE:
352 MatchData = {LoadMI, AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SBYTE};
354 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_USHORT:
355 MatchData = {LoadMI, AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SSHORT};
363void AMDGPUPostLegalizerCombinerImpl::applyCombineSignExtendInReg(
364 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchData)
const {
365 auto [LoadMI, NewOpcode] = MatchData;
369 Register SignExtendInsnDst =
MI.getOperand(0).getReg();
372 MI.eraseFromParent();
375bool AMDGPUPostLegalizerCombinerImpl::matchCombine_s_mul_u64(
376 MachineInstr &
MI,
unsigned &NewOpcode)
const {
382 if (VT->getKnownBits(Src1).countMinLeadingZeros() >= 32 &&
383 VT->getKnownBits(Src0).countMinLeadingZeros() >= 32) {
384 NewOpcode = AMDGPU::G_AMDGPU_S_MUL_U64_U32;
388 if (VT->computeNumSignBits(Src1) >= 33 &&
389 VT->computeNumSignBits(Src0) >= 33) {
390 NewOpcode = AMDGPU::G_AMDGPU_S_MUL_I64_I32;
400runCombiner(
MachineFunction &MF, GISelValueTracking *VT, GISelCSEInfo *CSEInfo,
401 MachineDominatorTree *MDT,
402 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
406 const LegalizerInfo *LI =
ST.getLegalizerInfo();
408 CombinerInfo CInfo(
false,
410 F.hasOptSize(),
F.hasMinSize());
412 CInfo.MaxIterations = 1;
413 CInfo.ObserverLvl = CombinerInfo::ObserverLevel::SinglePass;
415 CInfo.EnableFullDCE =
false;
416 AMDGPUPostLegalizerCombinerImpl Impl(MF, CInfo, *VT, CSEInfo, RuleConfig, ST,
418 return Impl.combineMachineInstrs();
421class AMDGPUPostLegalizerCombinerLegacy :
public MachineFunctionPass {
425 AMDGPUPostLegalizerCombinerLegacy(
bool IsOptNone =
false);
427 StringRef getPassName()
const override {
428 return "AMDGPUPostLegalizerCombiner";
433 void getAnalysisUsage(AnalysisUsage &AU)
const override;
437 AMDGPUPostLegalizerCombinerImplRuleConfig RuleConfig;
441void AMDGPUPostLegalizerCombinerLegacy::getAnalysisUsage(
442 AnalysisUsage &AU)
const {
445 AU.
addRequired<GISelValueTrackingAnalysisLegacy>();
455AMDGPUPostLegalizerCombinerLegacy::AMDGPUPostLegalizerCombinerLegacy(
457 : MachineFunctionPass(
ID), IsOptNone(IsOptNone) {
458 if (!RuleConfig.parseCommandLineOption())
462bool AMDGPUPostLegalizerCombinerLegacy::runOnMachineFunction(
470 GISelValueTracking *VT =
471 &getAnalysis<GISelValueTrackingAnalysisLegacy>().get(MF);
472 GISelCSEAnalysisWrapper &
Wrapper =
473 getAnalysis<GISelCSEAnalysisWrapperPass>().getCSEWrapper();
474 GISelCSEInfo *CSEInfo =
476 MachineDominatorTree *MDT =
478 : &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
480 return runCombiner(MF, VT, CSEInfo, MDT, RuleConfig, EnableOpt);
483char AMDGPUPostLegalizerCombinerLegacy::ID = 0;
485 "Combine AMDGPU machine instrs after legalization",
false,
490 "Combine AMDGPU machine instrs after legalization",
false,
494 return new AMDGPUPostLegalizerCombinerLegacy(IsOptNone);
503 AMDGPUPostLegalizerCombinerImplRuleConfig RuleConfig;
504 if (!RuleConfig.parseCommandLineOption())
514 if (!runCombiner(MF, &VT, CSEInfo, MDT, RuleConfig,
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define GET_GICOMBINER_CONSTRUCTOR_INITS
amdgpu aa AMDGPU Address space based Alias Analysis Wrapper
This contains common combine transformations that may be used in a combine pass.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
Provides AMDGPU specific target descriptions.
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
Provides analysis for continuously CSEing during GISel passes.
This contains common combine transformations that may be used in a combine pass,or by the target else...
Option class for Targets to specify which operations are combined how and when.
This contains the base class for all Combiners generated by TableGen.
AMD GCN specific subclass of TargetSubtarget.
Provides analysis for querying information about KnownBits during GISel passes.
const HexagonInstrInfo * TII
Contains matchers for matching SSA Machine Instructions.
Promote Memory to Register
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static StringRef getName(Value *V)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Target-Independent Code Generator Pass Configuration Options pass.
bool canIgnoreLegacyMinMaxTies(const MachineInstr &MI, Register LHS, Register RHS) const
fmin_legacy/fmax_legacy select s1 on NaN, and on a +0.0/-0.0 tie (s1 for min, s0 for max).
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represents analyses that only rely on functions' control flow.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Predicate getUnorderedPredicate() const
GISelValueTracking * getValueTracking() const
LLVM_ABI bool tryCombineShiftToUnmerge(MachineInstr &MI, unsigned TargetShiftAmount) const
FunctionPass class - This class is used to implement most global optimizations.
The actual analysis pass wrapper.
To use KnownBitsInfo analysis in a pass, KnownBitsInfo &Info = getAnalysis<GISelValueTrackingInfoAnal...
bool maskedValueIsZero(Register Val, const APInt &Mask)
constexpr bool isScalar() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
Register getReg() const
getReg - Returns the register number.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Wrapper class representing virtual and physical registers.
bool isCanonicalized(SelectionDAG &DAG, SDValue Op, SDNodeFlags UserFlags={}, unsigned MaxDepth=5) const
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
operand_type_match m_Reg()
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
ConstantMatch< APInt > m_ICst(APInt &Cst)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_LSHR, false > m_GLShr(const LHS &L, const RHS &R)
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
This is an optimization pass for GlobalISel generic memory operations.
FunctionPass * createAMDGPUPostLegalizeCombinerLegacy(bool IsOptNone)
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI std::unique_ptr< CSEConfigBase > getStandardCSEConfigForOpt(CodeGenOptLevel Level)
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI void getSelectionDAGFallbackAnalysisUsage(AnalysisUsage &AU)
Modify analysis usage so it preserves passes required for the SelectionDAG fallback.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.