29#define GET_GICOMBINER_DEPS
30#include "AMDGPUGenPreLegalizeGICombiner.inc"
31#undef GET_GICOMBINER_DEPS
33#define DEBUG_TYPE "amdgpu-prelegalizer-combiner"
39#define GET_GICOMBINER_TYPES
40#include "AMDGPUGenPreLegalizeGICombiner.inc"
41#undef GET_GICOMBINER_TYPES
43class AMDGPUPreLegalizerCombinerImpl :
public Combiner {
45 const AMDGPUPreLegalizerCombinerImplRuleConfig &RuleConfig;
50 AMDGPUPreLegalizerCombinerImpl(
53 const AMDGPUPreLegalizerCombinerImplRuleConfig &RuleConfig,
57 static const char *
getName() {
return "AMDGPUPreLegalizerCombinerImpl"; }
62 struct ClampI64ToI16MatchInfo {
70 ClampI64ToI16MatchInfo &MatchInfo)
const;
73 const ClampI64ToI16MatchInfo &MatchInfo)
const;
76#define GET_GICOMBINER_CLASS_MEMBERS
77#define AMDGPUSubtarget GCNSubtarget
78#include "AMDGPUGenPreLegalizeGICombiner.inc"
79#undef GET_GICOMBINER_CLASS_MEMBERS
83#define GET_GICOMBINER_IMPL
84#define AMDGPUSubtarget GCNSubtarget
85#include "AMDGPUGenPreLegalizeGICombiner.inc"
87#undef GET_GICOMBINER_IMPL
89AMDGPUPreLegalizerCombinerImpl::AMDGPUPreLegalizerCombinerImpl(
92 const AMDGPUPreLegalizerCombinerImplRuleConfig &RuleConfig,
94 :
Combiner(MF, CInfo, &VT, CSEInfo), RuleConfig(RuleConfig), STI(STI),
95 Helper(Observer,
B,
true, &VT, MDT, LI, STI),
97#include
"AMDGPUGenPreLegalizeGICombiner.inc"
102bool AMDGPUPreLegalizerCombinerImpl::tryCombineAll(
MachineInstr &
MI)
const {
103 if (tryCombineAllImpl(
MI))
108bool AMDGPUPreLegalizerCombinerImpl::matchClampI64ToI16(
110 ClampI64ToI16MatchInfo &MatchInfo)
const {
111 assert(
MI.getOpcode() == TargetOpcode::G_TRUNC &&
"Invalid instruction!");
114 const LLT SrcType = MRI.
getType(
MI.getOperand(1).getReg());
118 const LLT DstType = MRI.
getType(
MI.getOperand(0).getReg());
126 auto IsApplicableForCombine = [&MatchInfo](
bool OuterIsMin) ->
bool {
127 const int64_t
Lo = OuterIsMin ? MatchInfo.Cmp2 : MatchInfo.Cmp1;
128 const int64_t
Hi = OuterIsMin ? MatchInfo.Cmp1 : MatchInfo.Cmp2;
131 const int64_t Min = std::numeric_limits<int16_t>::min();
132 const int64_t
Max = std::numeric_limits<int16_t>::max();
145 return IsApplicableForCombine(
true);
153 return IsApplicableForCombine(
false);
167void AMDGPUPreLegalizerCombinerImpl::applyClampI64ToI16(
168 MachineInstr &
MI,
const ClampI64ToI16MatchInfo &MatchInfo)
const {
174 auto Unmerge =
B.buildUnmerge(I32, Src);
176 assert(
MI.getOpcode() != AMDGPU::G_AMDGPU_CVT_PK_I16_I32);
180 B.buildInstr(AMDGPU::G_AMDGPU_CVT_PK_I16_I32, {
V2S16},
181 {Unmerge.getReg(0), Unmerge.getReg(1)},
MI.getFlags());
183 auto MinBoundary = std::min(MatchInfo.Cmp1, MatchInfo.Cmp2);
184 auto MaxBoundary = std::max(MatchInfo.Cmp1, MatchInfo.Cmp2);
185 auto MinBoundaryDst =
B.buildConstant(I32, MinBoundary);
186 auto MaxBoundaryDst =
B.buildConstant(I32, MaxBoundary);
190 auto Med3 =
B.buildInstr(
191 AMDGPU::G_AMDGPU_SMED3, {
I32},
192 {MinBoundaryDst.getReg(0),
Bitcast.getReg(0), MaxBoundaryDst.getReg(0)},
195 B.buildTrunc(
MI.getOperand(0).getReg(), Med3);
197 MI.eraseFromParent();
203class AMDGPUPreLegalizerCombiner :
public MachineFunctionPass {
207 AMDGPUPreLegalizerCombiner(
bool IsOptNone =
false);
209 StringRef getPassName()
const override {
210 return "AMDGPUPreLegalizerCombiner";
215 void getAnalysisUsage(AnalysisUsage &AU)
const override;
219 AMDGPUPreLegalizerCombinerImplRuleConfig RuleConfig;
223void AMDGPUPreLegalizerCombiner::getAnalysisUsage(AnalysisUsage &AU)
const {
227 AU.
addRequired<GISelValueTrackingAnalysisLegacy>();
238AMDGPUPreLegalizerCombiner::AMDGPUPreLegalizerCombiner(
bool IsOptNone)
239 : MachineFunctionPass(
ID), IsOptNone(IsOptNone) {
240 if (!RuleConfig.parseCommandLineOption())
244bool AMDGPUPreLegalizerCombiner::runOnMachineFunction(
MachineFunction &MF) {
247 auto *TPC = &getAnalysis<TargetPassConfig>();
252 &getAnalysis<GISelValueTrackingAnalysisLegacy>().get(MF);
256 getAnalysis<GISelCSEAnalysisWrapperPass>().getCSEWrapper();
257 auto *CSEInfo = &
Wrapper.get(TPC->getCSEConfig());
262 : &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
264 nullptr, EnableOpt,
F.hasOptSize(),
F.hasMinSize());
266 CInfo.MaxIterations = 1;
270 CInfo.EnableFullDCE =
true;
271 AMDGPUPreLegalizerCombinerImpl Impl(MF, CInfo, *VT, CSEInfo, RuleConfig, STI,
273 return Impl.combineMachineInstrs();
276char AMDGPUPreLegalizerCombiner::ID = 0;
278 "Combine AMDGPU machine instrs before legalization",
283 "Combine AMDGPU machine instrs before legalization",
false,
287 return new AMDGPUPreLegalizerCombiner(IsOptNone);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define GET_GICOMBINER_CONSTRUCTOR_INITS
amdgpu aa AMDGPU Address space based Alias Analysis Wrapper
This contains common combine transformations that may be used in a combine pass.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
Provides analysis for continuously CSEing during GISel passes.
This contains common combine transformations that may be used in a combine pass,or by the target else...
Option class for Targets to specify which operations are combined how and when.
This contains the base class for all Combiners generated by TableGen.
AMD GCN specific subclass of TargetSubtarget.
Provides analysis for querying information about KnownBits during GISel passes.
Contains matchers for matching SSA Machine Instructions.
Promote Memory to Register
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static StringRef getName(Value *V)
Target-Independent Code Generator Pass Configuration Options pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
FunctionPass class - This class is used to implement most global optimizations.
const LegalizerInfo * getLegalizerInfo() const override
Simple wrapper that does the following.
To use KnownBitsInfo analysis in a pass, KnownBitsInfo &Info = getAnalysis<GISelValueTrackingInfoAnal...
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
static LLT integer(unsigned SizeInBits)
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Representation of each machine instruction.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
Wrapper class representing virtual and physical registers.
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
Target-Independent Code Generator Pass Configuration Options.
@ Bitcast
Perform the operation on a different, but equivalently sized type.
operand_type_match m_Reg()
ConstantMatch< APInt > m_ICst(APInt &Cst)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SMIN, true > m_GSMin(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SMAX, true > m_GSMax(const LHS &L, const RHS &R)
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI void getSelectionDAGFallbackAnalysisUsage(AnalysisUsage &AU)
Modify analysis usage so it preserves passes required for the SelectionDAG fallback.
FunctionPass * createAMDGPUPreLegalizeCombiner(bool IsOptNone)
@ SinglePass
Enables Observer-based DCE and additional heuristics that retry combining defined and used instructio...