14#include "llvm/IR/IntrinsicsAMDGPU.h"
25 TII(*
STI.getInstrInfo()) {}
29 switch (
MI.getOpcode()) {
35 case AMDGPU::G_FMINNUM:
36 case AMDGPU::G_FMAXNUM:
37 case AMDGPU::G_FMINNUM_IEEE:
38 case AMDGPU::G_FMAXNUM_IEEE:
39 case AMDGPU::G_FMINIMUM:
40 case AMDGPU::G_FMAXIMUM:
43 case AMDGPU::G_INTRINSIC_TRUNC:
44 case AMDGPU::G_FPTRUNC:
46 case AMDGPU::G_FNEARBYINT:
47 case AMDGPU::G_INTRINSIC_ROUND:
48 case AMDGPU::G_INTRINSIC_ROUNDEVEN:
49 case AMDGPU::G_FCANONICALIZE:
50 case AMDGPU::G_AMDGPU_RCP_IFLAG:
51 case AMDGPU::G_AMDGPU_FMIN_LEGACY:
52 case AMDGPU::G_AMDGPU_FMAX_LEGACY:
54 case AMDGPU::G_INTRINSIC: {
56 switch (IntrinsicID) {
57 case Intrinsic::amdgcn_rcp:
58 case Intrinsic::amdgcn_rcp_legacy:
59 case Intrinsic::amdgcn_sin:
60 case Intrinsic::amdgcn_fmul_legacy:
61 case Intrinsic::amdgcn_fmed3:
62 case Intrinsic::amdgcn_fma_legacy:
86 if (!
MI.memoperands().empty())
89 switch (
MI.getOpcode()) {
91 case AMDGPU::G_SELECT:
94 case TargetOpcode::INLINEASM:
95 case TargetOpcode::INLINEASM_BR:
96 case AMDGPU::G_INTRINSIC_W_SIDE_EFFECTS:
97 case AMDGPU::G_INTRINSIC_CONVERGENT_W_SIDE_EFFECTS:
98 case AMDGPU::G_BITCAST:
99 case AMDGPU::G_ANYEXT:
100 case AMDGPU::G_BUILD_VECTOR:
101 case AMDGPU::G_BUILD_VECTOR_TRUNC:
104 case AMDGPU::G_INTRINSIC:
105 case AMDGPU::G_INTRINSIC_CONVERGENT: {
107 switch (IntrinsicID) {
108 case Intrinsic::amdgcn_interp_p1:
109 case Intrinsic::amdgcn_interp_p2:
110 case Intrinsic::amdgcn_interp_mov:
111 case Intrinsic::amdgcn_interp_p1_f16:
112 case Intrinsic::amdgcn_interp_p2_f16:
113 case Intrinsic::amdgcn_div_scale:
131 unsigned NumMayIncreaseSize = 0;
153 APInt(64, 0x3fc45f306dc9c882));
163 std::optional<FPValueAndVReg> FPValReg;
165 if (FPValReg->Value.isZero() && !FPValReg->Value.isNegative())
169 if (ST.hasInv2PiInlineImm() &&
isInv2Pi(FPValReg->Value))
177 case AMDGPU::G_FMAXNUM:
178 return AMDGPU::G_FMINNUM;
179 case AMDGPU::G_FMINNUM:
180 return AMDGPU::G_FMAXNUM;
181 case AMDGPU::G_FMAXNUM_IEEE:
182 return AMDGPU::G_FMINNUM_IEEE;
183 case AMDGPU::G_FMINNUM_IEEE:
184 return AMDGPU::G_FMAXNUM_IEEE;
185 case AMDGPU::G_FMAXIMUM:
186 return AMDGPU::G_FMINIMUM;
187 case AMDGPU::G_FMINIMUM:
188 return AMDGPU::G_FMAXIMUM;
189 case AMDGPU::G_AMDGPU_FMAX_LEGACY:
190 return AMDGPU::G_AMDGPU_FMIN_LEGACY;
191 case AMDGPU::G_AMDGPU_FMIN_LEGACY:
192 return AMDGPU::G_AMDGPU_FMAX_LEGACY;
201 MatchInfo =
MRI.getVRegDef(Src);
207 if (
MRI.hasOneNonDBGUse(Src)) {
218 case AMDGPU::G_FMINNUM:
219 case AMDGPU::G_FMAXNUM:
220 case AMDGPU::G_FMINNUM_IEEE:
221 case AMDGPU::G_FMAXNUM_IEEE:
222 case AMDGPU::G_FMINIMUM:
223 case AMDGPU::G_FMAXIMUM:
224 case AMDGPU::G_AMDGPU_FMIN_LEGACY:
225 case AMDGPU::G_AMDGPU_FMAX_LEGACY:
235 case AMDGPU::G_FPEXT:
236 case AMDGPU::G_INTRINSIC_TRUNC:
237 case AMDGPU::G_FPTRUNC:
238 case AMDGPU::G_FRINT:
239 case AMDGPU::G_FNEARBYINT:
240 case AMDGPU::G_INTRINSIC_ROUND:
241 case AMDGPU::G_INTRINSIC_ROUNDEVEN:
243 case AMDGPU::G_FCANONICALIZE:
244 case AMDGPU::G_AMDGPU_RCP_IFLAG:
246 case AMDGPU::G_INTRINSIC:
247 case AMDGPU::G_INTRINSIC_CONVERGENT: {
249 switch (IntrinsicID) {
250 case Intrinsic::amdgcn_rcp:
251 case Intrinsic::amdgcn_rcp_legacy:
252 case Intrinsic::amdgcn_sin:
253 case Intrinsic::amdgcn_fmul_legacy:
254 case Intrinsic::amdgcn_fmed3:
256 case Intrinsic::amdgcn_fma_legacy:
286 Reg =
Builder.buildFNeg(
MRI.getType(Reg), Reg).getReg(0);
299 YReg =
Builder.buildFNeg(
MRI.getType(YReg), YReg).getReg(0);
304 Builder.setInstrAndDebugLoc(*MatchInfo);
317 case AMDGPU::G_FMINNUM:
318 case AMDGPU::G_FMAXNUM:
319 case AMDGPU::G_FMINNUM_IEEE:
320 case AMDGPU::G_FMAXNUM_IEEE:
321 case AMDGPU::G_FMINIMUM:
322 case AMDGPU::G_FMAXIMUM:
323 case AMDGPU::G_AMDGPU_FMIN_LEGACY:
324 case AMDGPU::G_AMDGPU_FMAX_LEGACY: {
336 case AMDGPU::G_FPEXT:
337 case AMDGPU::G_INTRINSIC_TRUNC:
338 case AMDGPU::G_FRINT:
339 case AMDGPU::G_FNEARBYINT:
340 case AMDGPU::G_INTRINSIC_ROUND:
341 case AMDGPU::G_INTRINSIC_ROUNDEVEN:
343 case AMDGPU::G_FCANONICALIZE:
344 case AMDGPU::G_AMDGPU_RCP_IFLAG:
345 case AMDGPU::G_FPTRUNC:
348 case AMDGPU::G_INTRINSIC:
349 case AMDGPU::G_INTRINSIC_CONVERGENT: {
351 switch (IntrinsicID) {
352 case Intrinsic::amdgcn_rcp:
353 case Intrinsic::amdgcn_rcp_legacy:
354 case Intrinsic::amdgcn_sin:
357 case Intrinsic::amdgcn_fmul_legacy:
360 case Intrinsic::amdgcn_fmed3:
365 case Intrinsic::amdgcn_fma_legacy:
381 if (
MRI.hasOneNonDBGUse(MatchInfoDst)) {
397 Builder.setInstrAndDebugLoc(*NextInst);
398 Builder.buildFNeg(MatchInfoDst, NegatedMatchInfo,
MI.getFlags());
401 MI.eraseFromParent();
407 if (!
MRI.hasOneNonDBGUse(Round))
419 Builder.setInstrAndDebugLoc(Fabs);
437 bool LosesInfo =
true;
449 assert(
MI.getOpcode() == TargetOpcode::G_FPTRUNC);
470 LLT Ty =
MRI.getType(Src0);
471 auto A1 =
Builder.buildFMinNumIEEE(Ty, Src0, Src1);
472 auto B1 =
Builder.buildFMaxNumIEEE(Ty, Src0, Src1);
473 auto C1 =
Builder.buildFMaxNumIEEE(Ty, A1, Src2);
474 Builder.buildFMinNumIEEE(
MI.getOperand(0), B1, C1);
475 MI.eraseFromParent();
481 assert(
MI.getOpcode() == TargetOpcode::G_FMUL);
486 LLT DestTy =
MRI.getType(Dst);
499 const auto SelectTrueVal =
503 const auto SelectFalseVal =
508 if (SelectTrueVal->isNegative() != SelectFalseVal->isNegative())
513 if (ScalarDestTy ==
LLT::float32() &&
TII.isInlineConstant(*SelectTrueVal) &&
514 TII.isInlineConstant(*SelectFalseVal))
517 int SelectTrueLog2Val = SelectTrueVal->getExactLog2Abs();
518 if (SelectTrueLog2Val == INT_MIN)
520 int SelectFalseLog2Val = SelectFalseVal->getExactLog2Abs();
521 if (SelectFalseLog2Val == INT_MIN)
526 auto NewSel =
Builder.buildSelect(
527 IntDestTy, SelectCondReg,
528 Builder.buildConstant(IntDestTy, SelectTrueLog2Val),
529 Builder.buildConstant(IntDestTy, SelectFalseLog2Val));
532 if (SelectTrueVal->isNegative()) {
534 Builder.buildFNeg(DestTy, XReg,
MRI.getVRegDef(XReg)->getFlags());
535 Builder.buildFLdexp(Dst, NegX, NewSel,
MI.getFlags());
537 Builder.buildFLdexp(Dst, XReg, NewSel,
MI.getFlags());
549 const uint64_t Val = Res->Value.getZExtValue();
550 unsigned MaskIdx = 0;
551 unsigned MaskLen = 0;
556 return MaskLen >= 32 && ((MaskIdx == 0) || (MaskIdx == 64 - MaskLen));
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static LLVM_READONLY bool hasSourceMods(const MachineInstr &MI)
static bool isInv2Pi(const APFloat &APF)
static bool isFPExtFromF16OrConst(const MachineRegisterInfo &MRI, Register Reg)
static bool mayIgnoreSignedZero(MachineInstr &MI)
static bool isConstantCostlierToNegate(MachineInstr &MI, Register Reg, MachineRegisterInfo &MRI)
static bool allUsesHaveSourceMods(MachineInstr &MI, MachineRegisterInfo &MRI, unsigned CostThreshold=4)
static LLVM_READONLY bool opMustUseVOP3Encoding(const MachineInstr &MI, const MachineRegisterInfo &MRI)
returns true if the operation will definitely need to use a 64-bit encoding, and thus will use a VOP3...
static unsigned inverseMinMax(unsigned Opc)
static LLVM_READNONE bool fnegFoldsIntoMI(const MachineInstr &MI)
This contains common combine transformations that may be used in a combine pass.
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
AMD GCN specific subclass of TargetSubtarget.
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
Interface for Targets to specify which operations they can successfully select and how the others sho...
Contains matchers for matching SSA Machine Instructions.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
bool matchFoldFAbsFptrunc(MachineInstr &Fabs, MachineInstr &Fptrunc) const
AMDGPUCombinerHelper(GISelChangeObserver &Observer, MachineIRBuilder &B, bool IsPreLegalize, GISelValueTracking *VT, MachineDominatorTree *MDT, const LegalizerInfo *LI, const GCNSubtarget &STI)
bool matchConstantIs32BitMask(Register Reg) const
bool matchCombineFmulWithSelectToFldexp(MachineInstr &MI, MachineInstr &Sel, std::function< void(MachineIRBuilder &)> &MatchInfo) const
LLVM_ABI CombinerHelper(GISelChangeObserver &Observer, MachineIRBuilder &B, bool IsPreLegalize, GISelValueTracking *VT=nullptr, MachineDominatorTree *MDT=nullptr, const LegalizerInfo *LI=nullptr)
bool matchExpandPromotedF16FMed3(MachineInstr &MI, Register Src0, Register Src1, Register Src2) const
void applyFoldableFneg(MachineInstr &MI, MachineInstr *&MatchInfo) const
bool matchFoldableFneg(MachineInstr &MI, MachineInstr *&MatchInfo) const
void applyFoldFAbsFptrunc(MachineInstr &Fabs, MachineInstr &Fptrunc) const
void applyExpandPromotedF16FMed3(MachineInstr &MI, Register Src0, Register Src1, Register Src2) const
static const fltSemantics & IEEEsingle()
static const fltSemantics & IEEEdouble()
static constexpr roundingMode rmNearestTiesToEven
static const fltSemantics & IEEEhalf()
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
bool bitwiseIsEqual(const APFloat &RHS) const
Class for arbitrary precision integers.
LLVM_ABI void replaceRegWith(MachineRegisterInfo &MRI, Register FromReg, Register ToReg) const
MachineRegisterInfo::replaceRegWith() and inform the observer of the changes.
LLVM_ABI void replaceRegOpWith(MachineRegisterInfo &MRI, MachineOperand &FromRegOp, Register ToReg) const
Replace a single register operand with a new register and inform the observer of the changes.
LLVM_ABI void replaceOpcodeWith(MachineInstr &FromMI, unsigned ToOpcode) const
Replace the opcode in instruction with a new opcode and inform the observer of the changes.
MachineRegisterInfo & MRI
LLVM_ABI bool isLegalOrBeforeLegalizer(const LegalityQuery &Query) const
MachineDominatorTree * MDT
GISelChangeObserver & Observer
MachineIRBuilder & Builder
ConstantFP - Floating Point Values [float, double].
const APFloat & getValueAPF() const
Abstract class that contains various methods for clients to notify about changes.
static constexpr LLT float64()
Get a 64-bit IEEE double value.
constexpr unsigned getScalarSizeInBits() const
constexpr LLT changeElementType(LLT NewEltTy) const
If this type is a vector, return a vector with the same number of elements but the new element type.
LLT getScalarType() const
static constexpr LLT float16()
Get a 16-bit IEEE half value.
static LLT integer(unsigned SizeInBits)
static constexpr LLT float32()
Get a 32-bit IEEE float value.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
Helper class to build MachineInstr.
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineOperand & getOperand(unsigned i) const
uint32_t getFlags() const
Return the MI flags bitvector.
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
MachineOperand class - Representation of each machine instruction operand.
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
Wrapper class representing virtual and physical registers.
The instances of the Type class are immutable: once they are created, they are never changed.
A Use represents the edge between a Value definition and its users.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
operand_type_match m_Reg()
UnaryOp_match< SrcTy, TargetOpcode::G_FPEXT > m_GFPExt(const SrcTy &Src)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
UnaryOp_match< SrcTy, TargetOpcode::G_FNEG > m_GFNeg(const SrcTy &Src)
GFCstAndRegMatch m_GFCst(std::optional< FPValueAndVReg > &FPValReg)
GFCstOrSplatGFCstMatch m_GFCstOrSplat(std::optional< FPValueAndVReg > &FPValReg)
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI std::optional< APFloat > isConstantOrConstantSplatVectorFP(Register Def, const MachineRegisterInfo &MRI)
Determines if Def defines a float constant integer or a splat vector of float constant integers.
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
static cl::opt< unsigned > CostThreshold("dfa-cost-threshold", cl::desc("Maximum cost accepted for the transformation"), cl::Hidden, cl::init(50))