26#include "llvm/IR/IntrinsicsAMDGPU.h"
29#define GET_GICOMBINER_DEPS
30#include "AMDGPUGenPreLegalizeGICombiner.inc"
31#undef GET_GICOMBINER_DEPS
33#define DEBUG_TYPE "amdgpu-postlegalizer-combiner"
39#define GET_GICOMBINER_TYPES
40#include "AMDGPUGenPostLegalizeGICombiner.inc"
41#undef GET_GICOMBINER_TYPES
43class AMDGPUPostLegalizerCombinerImpl :
public Combiner {
45 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig;
52 AMDGPUPostLegalizerCombinerImpl(
55 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
59 static const char *
getName() {
return "AMDGPUPostLegalizerCombinerImpl"; }
64 struct FMinFMaxLegacyInfo {
72 FMinFMaxLegacyInfo &Info)
const;
74 const FMinFMaxLegacyInfo &Info)
const;
84 struct CvtF32UByteMatchInfo {
90 CvtF32UByteMatchInfo &MatchInfo)
const;
92 const CvtF32UByteMatchInfo &MatchInfo)
const;
98 bool matchCombineSignExtendInReg(
99 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchInfo)
const;
100 void applyCombineSignExtendInReg(
101 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchInfo)
const;
108 bool matchCombine_s_mul_u64(
MachineInstr &
MI,
unsigned &NewOpcode)
const;
111#define GET_GICOMBINER_CLASS_MEMBERS
112#define AMDGPUSubtarget GCNSubtarget
113#include "AMDGPUGenPostLegalizeGICombiner.inc"
114#undef GET_GICOMBINER_CLASS_MEMBERS
115#undef AMDGPUSubtarget
118#define GET_GICOMBINER_IMPL
119#define AMDGPUSubtarget GCNSubtarget
120#include "AMDGPUGenPostLegalizeGICombiner.inc"
121#undef AMDGPUSubtarget
122#undef GET_GICOMBINER_IMPL
124AMDGPUPostLegalizerCombinerImpl::AMDGPUPostLegalizerCombinerImpl(
127 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
129 :
Combiner(MF, CInfo, &VT, CSEInfo), RuleConfig(RuleConfig), STI(STI),
130 TII(*STI.getInstrInfo()),
131 Helper(Observer,
B,
false, &VT, MDT, LI, STI),
133#include
"AMDGPUGenPostLegalizeGICombiner.inc"
138bool AMDGPUPostLegalizerCombinerImpl::tryCombineAll(
MachineInstr &
MI)
const {
139 if (tryCombineAllImpl(
MI))
142 switch (
MI.getOpcode()) {
143 case TargetOpcode::G_SHL:
144 case TargetOpcode::G_LSHR:
145 case TargetOpcode::G_ASHR:
155bool AMDGPUPostLegalizerCombinerImpl::matchFMinFMaxLegacy(
156 MachineInstr &
MI, MachineInstr &FCmp, FMinFMaxLegacyInfo &Info)
const {
169 if ((
Info.LHS != True ||
Info.RHS != False) &&
170 (
Info.LHS != False ||
Info.RHS != True))
176 if (
Info.LHS != True)
183void AMDGPUPostLegalizerCombinerImpl::applySelectFCmpToFMinFMaxLegacy(
184 MachineInstr &
MI,
const FMinFMaxLegacyInfo &Info)
const {
186 : AMDGPU::G_AMDGPU_FMIN_LEGACY;
196 B.buildInstr(
Opc, {
MI.getOperand(0)}, {
X,
Y},
MI.getFlags());
198 MI.eraseFromParent();
201bool AMDGPUPostLegalizerCombinerImpl::matchUCharToFloat(
202 MachineInstr &
MI)
const {
209 LLT Ty = MRI.getType(DstReg);
212 unsigned SrcSize = MRI.getType(SrcReg).getSizeInBits();
213 assert(SrcSize == 16 || SrcSize == 32 || SrcSize == 64);
221void AMDGPUPostLegalizerCombinerImpl::applyUCharToFloat(
222 MachineInstr &
MI)
const {
227 LLT Ty = MRI.getType(DstReg);
228 LLT SrcTy = MRI.getType(SrcReg);
230 SrcReg =
B.buildAnyExtOrTrunc(
S32, SrcReg).getReg(0);
233 B.buildInstr(AMDGPU::G_AMDGPU_CVT_F32_UBYTE0, {DstReg}, {SrcReg},
236 auto Cvt0 =
B.buildInstr(AMDGPU::G_AMDGPU_CVT_F32_UBYTE0, {
S32}, {SrcReg},
238 B.buildFPTrunc(DstReg, Cvt0,
MI.getFlags());
241 MI.eraseFromParent();
244bool AMDGPUPostLegalizerCombinerImpl::matchFDivSqrtToRsqF16(
245 MachineInstr &
MI)
const {
247 return MRI.hasOneNonDBGUse(Sqrt);
250void AMDGPUPostLegalizerCombinerImpl::applyFDivSqrtToRsqF16(
254 LLT DstTy = MRI.getType(Dst);
255 uint32_t
Flags =
MI.getFlags();
256 Register RSQ =
B.buildIntrinsic(Intrinsic::amdgcn_rsq, {DstTy})
260 B.buildFMul(Dst, RSQ,
Y, Flags);
261 MI.eraseFromParent();
264bool AMDGPUPostLegalizerCombinerImpl::matchCvtF32UByteN(
265 MachineInstr &
MI, CvtF32UByteMatchInfo &MatchInfo)
const {
275 const unsigned Offset =
MI.getOpcode() - AMDGPU::G_AMDGPU_CVT_F32_UBYTE0;
277 unsigned ShiftOffset = 8 *
Offset;
279 ShiftOffset += ShiftAmt;
281 ShiftOffset -= ShiftAmt;
283 MatchInfo.CvtVal = Src0;
284 MatchInfo.ShiftOffset = ShiftOffset;
285 return ShiftOffset < 32 && ShiftOffset >= 8 && (ShiftOffset % 8) == 0;
292void AMDGPUPostLegalizerCombinerImpl::applyCvtF32UByteN(
293 MachineInstr &
MI,
const CvtF32UByteMatchInfo &MatchInfo)
const {
294 unsigned NewOpc = AMDGPU::G_AMDGPU_CVT_F32_UBYTE0 + MatchInfo.ShiftOffset / 8;
298 LLT SrcTy = MRI.getType(MatchInfo.CvtVal);
301 CvtSrc =
B.buildAnyExt(
S32, CvtSrc).getReg(0);
305 B.buildInstr(NewOpc, {
MI.getOperand(0)}, {CvtSrc},
MI.getFlags());
306 MI.eraseFromParent();
309bool AMDGPUPostLegalizerCombinerImpl::matchRemoveFcanonicalize(
310 MachineInstr &
MI)
const {
311 const SITargetLowering *TLI =
static_cast<const SITargetLowering *
>(
312 MF.getSubtarget().getTargetLowering());
322bool AMDGPUPostLegalizerCombinerImpl::matchCombineSignExtendInReg(
323 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchData)
const {
325 if (!MRI.hasOneNonDBGUse(LoadReg))
330 MachineInstr *LoadMI = MRI.getVRegDef(LoadReg);
331 int64_t Width =
MI.getOperand(2).getImm();
333 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
334 MatchData = {LoadMI, AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE};
336 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
337 MatchData = {LoadMI, AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT};
339 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_UBYTE:
340 MatchData = {LoadMI, AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SBYTE};
342 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_USHORT:
343 MatchData = {LoadMI, AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SSHORT};
351void AMDGPUPostLegalizerCombinerImpl::applyCombineSignExtendInReg(
352 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchData)
const {
353 auto [LoadMI, NewOpcode] = MatchData;
357 Register SignExtendInsnDst =
MI.getOperand(0).getReg();
360 MI.eraseFromParent();
363bool AMDGPUPostLegalizerCombinerImpl::matchCombine_s_mul_u64(
364 MachineInstr &
MI,
unsigned &NewOpcode)
const {
370 if (VT->getKnownBits(Src1).countMinLeadingZeros() >= 32 &&
371 VT->getKnownBits(Src0).countMinLeadingZeros() >= 32) {
372 NewOpcode = AMDGPU::G_AMDGPU_S_MUL_U64_U32;
376 if (VT->computeNumSignBits(Src1) >= 33 &&
377 VT->computeNumSignBits(Src0) >= 33) {
378 NewOpcode = AMDGPU::G_AMDGPU_S_MUL_I64_I32;
387class AMDGPUPostLegalizerCombiner :
public MachineFunctionPass {
391 AMDGPUPostLegalizerCombiner(
bool IsOptNone =
false);
393 StringRef getPassName()
const override {
394 return "AMDGPUPostLegalizerCombiner";
399 void getAnalysisUsage(AnalysisUsage &AU)
const override;
403 AMDGPUPostLegalizerCombinerImplRuleConfig RuleConfig;
407void AMDGPUPostLegalizerCombiner::getAnalysisUsage(AnalysisUsage &AU)
const {
410 AU.
addRequired<GISelValueTrackingAnalysisLegacy>();
418AMDGPUPostLegalizerCombiner::AMDGPUPostLegalizerCombiner(
bool IsOptNone)
419 : MachineFunctionPass(
ID), IsOptNone(IsOptNone) {
420 if (!RuleConfig.parseCommandLineOption())
424bool AMDGPUPostLegalizerCombiner::runOnMachineFunction(
MachineFunction &MF) {
436 &getAnalysis<GISelValueTrackingAnalysisLegacy>().get(MF);
439 : &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
442 LI, EnableOpt,
F.hasOptSize(),
F.hasMinSize());
444 CInfo.MaxIterations = 1;
447 CInfo.EnableFullDCE =
false;
448 AMDGPUPostLegalizerCombinerImpl Impl(MF, CInfo, *VT,
nullptr,
449 RuleConfig, ST, MDT, LI);
450 return Impl.combineMachineInstrs();
453char AMDGPUPostLegalizerCombiner::ID = 0;
455 "Combine AMDGPU machine instrs after legalization",
false,
459 "Combine AMDGPU machine instrs after legalization",
false,
463 return new AMDGPUPostLegalizerCombiner(IsOptNone);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define GET_GICOMBINER_CONSTRUCTOR_INITS
This contains common combine transformations that may be used in a combine pass.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This contains common combine transformations that may be used in a combine pass,or by the target else...
Option class for Targets to specify which operations are combined how and when.
This contains the base class for all Combiners generated by TableGen.
AMD GCN specific subclass of TargetSubtarget.
Provides analysis for querying information about KnownBits during GISel passes.
const HexagonInstrInfo * TII
Contains matchers for matching SSA Machine Instructions.
Promote Memory to Register
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static StringRef getName(Value *V)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Target-Independent Code Generator Pass Configuration Options pass.
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Predicate getUnorderedPredicate() const
GISelValueTracking * getValueTracking() const
LLVM_ABI bool tryCombineShiftToUnmerge(MachineInstr &MI, unsigned TargetShiftAmount) const
FunctionPass class - This class is used to implement most global optimizations.
To use KnownBitsInfo analysis in a pass, KnownBitsInfo &Info = getAnalysis<GISelValueTrackingInfoAnal...
bool maskedValueIsZero(Register Val, const APInt &Mask)
constexpr bool isScalar() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
Register getReg() const
getReg - Returns the register number.
Wrapper class representing virtual and physical registers.
bool isCanonicalized(SelectionDAG &DAG, SDValue Op, SDNodeFlags UserFlags={}, unsigned MaxDepth=5) const
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
operand_type_match m_Reg()
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
ConstantMatch< APInt > m_ICst(APInt &Cst)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_LSHR, false > m_GLShr(const LHS &L, const RHS &R)
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI void getSelectionDAGFallbackAnalysisUsage(AnalysisUsage &AU)
Modify analysis usage so it preserves passes required for the SelectionDAG fallback.
FunctionPass * createAMDGPUPostLegalizeCombiner(bool IsOptNone)
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ SinglePass
Enables Observer-based DCE and additional heuristics that retry combining defined and used instructio...