24#define DEBUG_TYPE "si-fold-operands"
45 unsigned DefSubReg = AMDGPU::NoSubRegister;
50 FoldableDef() =
delete;
52 unsigned DefSubReg = AMDGPU::NoSubRegister)
53 : DefRC(DefRC), DefSubReg(DefSubReg), Kind(FoldOp.
getType()) {
56 ImmToFold = FoldOp.
getImm();
57 }
else if (FoldOp.
isFI()) {
58 FrameIndexToFold = FoldOp.
getIndex();
68 unsigned DefSubReg = AMDGPU::NoSubRegister)
69 : ImmToFold(FoldImm), DefRC(DefRC), DefSubReg(DefSubReg),
74 FoldableDef Copy(*
this);
75 Copy.DefSubReg =
TRI.composeSubRegIndices(DefSubReg, SubReg);
83 return OpToFold->getReg();
86 unsigned getSubReg()
const {
88 return OpToFold->getSubReg();
99 return FrameIndexToFold;
107 std::optional<int64_t> getEffectiveImmVal()
const {
115 unsigned OpIdx)
const {
118 std::optional<int64_t> ImmToFold = getEffectiveImmVal();
125 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
128 if (DefSubReg != AMDGPU::NoSubRegister)
131 return TII.isOperandLegal(
MI, OpIdx, &TmpOp);
136 if (DefSubReg != AMDGPU::NoSubRegister)
138 return TII.isOperandLegal(
MI, OpIdx, OpToFold);
145struct FoldCandidate {
153 bool Commuted =
false,
int ShrinkOp = -1)
154 :
UseMI(
MI), Def(Def), ShrinkOpcode(ShrinkOp), UseOpNo(OpNo),
155 Commuted(Commuted) {}
157 bool isFI()
const {
return Def.isFI(); }
161 return Def.FrameIndexToFold;
164 bool isImm()
const {
return Def.isImm(); }
166 bool isReg()
const {
return Def.isReg(); }
170 bool isGlobal()
const {
return Def.isGlobal(); }
172 bool needsShrink()
const {
return ShrinkOpcode != -1; }
175class SIFoldOperandsImpl {
186 const FoldableDef &OpToFold)
const;
189 unsigned convertToVALUOp(
unsigned Opc,
bool UseVOP3 =
false)
const {
191 case AMDGPU::S_ADD_I32: {
192 if (ST->hasAddNoCarryInsts())
193 return UseVOP3 ? AMDGPU::V_ADD_U32_e64 : AMDGPU::V_ADD_U32_e32;
194 return UseVOP3 ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_ADD_CO_U32_e32;
196 case AMDGPU::S_OR_B32:
197 return UseVOP3 ? AMDGPU::V_OR_B32_e64 : AMDGPU::V_OR_B32_e32;
198 case AMDGPU::S_AND_B32:
199 return UseVOP3 ? AMDGPU::V_AND_B32_e64 : AMDGPU::V_AND_B32_e32;
200 case AMDGPU::S_MUL_I32:
201 return AMDGPU::V_MUL_LO_U32_e64;
203 return AMDGPU::INSTRUCTION_LIST_END;
207 bool foldCopyToVGPROfScalarAddOfFrameIndex(
Register DstReg,
Register SrcReg,
213 int64_t ImmVal)
const;
217 int64_t ImmVal)
const;
221 const FoldableDef &OpToFold)
const;
224 bool isTemporallyDivergentUse(
const FoldableDef &OpToFold,
232 getRegSeqInit(
SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
235 std::pair<int64_t, const TargetRegisterClass *>
249 struct ANDMaskResult {
255 std::optional<ANDMaskResult> getANDMaskRegOperand(
MachineInstr &AndMI)
const;
260 bool foldInstOperand(
MachineInstr &
MI,
const FoldableDef &OpToFold)
const;
262 bool foldCopyToAGPRRegSequence(
MachineInstr *CopyMI)
const;
269 std::pair<const MachineOperand *, int> isOMod(
const MachineInstr &
MI)
const;
279 SIFoldOperandsImpl() =
default;
294 &getAnalysis<MachineLoopInfoWrapperPass>().getLI();
295 return SIFoldOperandsImpl().run(MF, MLI);
298 StringRef getPassName()
const override {
return "SI Fold Operands"; }
320char SIFoldOperandsLegacy::ID = 0;
329 TRI.getSubRegisterClass(RC, MO.getSubReg()))
337 case AMDGPU::V_MAC_F32_e64:
338 return AMDGPU::V_MAD_F32_e64;
339 case AMDGPU::V_MAC_F16_e64:
340 return AMDGPU::V_MAD_F16_e64;
341 case AMDGPU::V_FMAC_F32_e64:
342 return AMDGPU::V_FMA_F32_e64;
343 case AMDGPU::V_FMAC_F16_e64:
344 return AMDGPU::V_FMA_F16_gfx9_e64;
345 case AMDGPU::V_FMAC_F16_t16_e64:
346 return AMDGPU::V_FMA_F16_gfx9_t16_e64;
347 case AMDGPU::V_FMAC_F16_fake16_e64:
348 return AMDGPU::V_FMA_F16_gfx9_fake16_e64;
349 case AMDGPU::V_FMAC_LEGACY_F32_e64:
350 return AMDGPU::V_FMA_LEGACY_F32_e64;
351 case AMDGPU::V_FMAC_F64_e64:
352 return AMDGPU::V_FMA_F64_e64;
354 return AMDGPU::INSTRUCTION_LIST_END;
360 const FoldableDef &OpToFold)
const {
361 if (!OpToFold.isFI())
364 const unsigned Opc =
UseMI.getOpcode();
366 case AMDGPU::S_ADD_I32:
367 case AMDGPU::S_ADD_U32:
368 case AMDGPU::V_ADD_U32_e32:
369 case AMDGPU::V_ADD_CO_U32_e32:
373 return UseMI.getOperand(OpNo == 1 ? 2 : 1).isImm() &&
375 case AMDGPU::V_ADD_U32_e64:
376 case AMDGPU::V_ADD_CO_U32_e64:
377 return UseMI.getOperand(OpNo == 2 ? 3 : 2).isImm() &&
384 return OpNo == AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
388 int SIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::saddr);
392 int VIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
393 return OpNo == VIdx && SIdx == -1;
399bool SIFoldOperandsImpl::foldCopyToVGPROfScalarAddOfFrameIndex(
404 if (
TRI->isVGPR(*MRI, DstReg) &&
TRI->isSGPRReg(*MRI, SrcReg) &&
407 if (!Def ||
Def->getNumOperands() != 4)
410 MachineOperand *Src0 = &
Def->getOperand(1);
411 MachineOperand *Src1 = &
Def->getOperand(2);
422 const bool UseVOP3 = !Src0->
isImm() ||
TII->isInlineConstant(*Src0);
423 unsigned NewOp = convertToVALUOp(
Def->getOpcode(), UseVOP3);
424 if (NewOp == AMDGPU::INSTRUCTION_LIST_END ||
425 !
Def->getOperand(3).isDead())
428 MachineBasicBlock *
MBB =
Def->getParent();
430 if (NewOp != AMDGPU::V_ADD_CO_U32_e32) {
431 MachineInstrBuilder
Add =
434 if (
Add->getDesc().getNumDefs() == 2) {
436 Add.addDef(CarryOutReg, RegState::Dead);
440 Add.add(*Src0).add(*Src1).setMIFlags(
Def->getFlags());
444 Def->eraseFromParent();
445 MI.eraseFromParent();
449 assert(NewOp == AMDGPU::V_ADD_CO_U32_e32);
460 Def->eraseFromParent();
461 MI.eraseFromParent();
470 return new SIFoldOperandsLegacy();
473bool SIFoldOperandsImpl::canUseImmWithOpSel(
const MachineInstr *
MI,
475 int64_t ImmVal)
const {
482 int OpNo =
MI->getOperandNo(&Old);
484 unsigned Opcode =
MI->getOpcode();
485 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
507bool SIFoldOperandsImpl::tryFoldImmWithOpSel(MachineInstr *
MI,
unsigned UseOpNo,
508 int64_t ImmVal)
const {
509 MachineOperand &Old =
MI->getOperand(UseOpNo);
510 unsigned Opcode =
MI->getOpcode();
511 int OpNo =
MI->getOperandNo(&Old);
512 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
524 AMDGPU::OpName ModName = AMDGPU::OpName::NUM_OPERAND_NAMES;
525 unsigned SrcIdx = ~0;
526 if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0)) {
527 ModName = AMDGPU::OpName::src0_modifiers;
529 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1)) {
530 ModName = AMDGPU::OpName::src1_modifiers;
532 }
else if (OpNo == AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src2)) {
533 ModName = AMDGPU::OpName::src2_modifiers;
536 assert(ModName != AMDGPU::OpName::NUM_OPERAND_NAMES);
537 int ModIdx = AMDGPU::getNamedOperandIdx(Opcode, ModName);
538 MachineOperand &
Mod =
MI->getOperand(ModIdx);
539 unsigned ModVal =
Mod.getImm();
545 uint32_t
Imm = (
static_cast<uint32_t
>(ImmHi) << 16) | ImmLo;
550 auto tryFoldToInline = [&](uint32_t
Imm) ->
bool {
559 uint16_t
Lo =
static_cast<uint16_t
>(
Imm);
560 uint16_t
Hi =
static_cast<uint16_t
>(
Imm >> 16);
566 if (ST->hasBF16InlineConstFromUpperFP32() &&
570 Mod.setImm(NewModVal);
575 if (
static_cast<int16_t
>(
Lo) < 0) {
576 int32_t SExt =
static_cast<int16_t
>(
Lo);
578 Mod.setImm(NewModVal);
593 uint32_t Swapped = (
static_cast<uint32_t
>(
Lo) << 16) |
Hi;
604 if (tryFoldToInline(
Imm))
613 bool IsUAdd = Opcode == AMDGPU::V_PK_ADD_U16;
614 bool IsUSub = Opcode == AMDGPU::V_PK_SUB_U16;
615 if (SrcIdx == 1 && (IsUAdd || IsUSub)) {
617 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::clamp);
618 bool Clamp =
MI->getOperand(ClampIdx).getImm() != 0;
621 uint16_t NegLo = -
static_cast<uint16_t
>(
Imm);
622 uint16_t NegHi = -
static_cast<uint16_t
>(
Imm >> 16);
623 uint32_t NegImm = (
static_cast<uint32_t
>(NegHi) << 16) | NegLo;
625 if (tryFoldToInline(NegImm)) {
627 IsUAdd ? AMDGPU::V_PK_SUB_U16 : AMDGPU::V_PK_ADD_U16;
628 MI->setDesc(
TII->get(NegOpcode));
637bool SIFoldOperandsImpl::updateOperand(FoldCandidate &Fold)
const {
638 MachineInstr *
MI = Fold.UseMI;
639 MachineOperand &Old =
MI->getOperand(Fold.UseOpNo);
642 std::optional<int64_t> ImmVal;
644 ImmVal = Fold.Def.getEffectiveImmVal();
646 if (ImmVal && canUseImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal)) {
647 if (tryFoldImmWithOpSel(Fold.UseMI, Fold.UseOpNo, *ImmVal))
653 int OpNo =
MI->getOperandNo(&Old);
654 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
660 if ((Fold.isImm() || Fold.isFI() || Fold.isGlobal()) && Fold.needsShrink()) {
668 int Op32 = Fold.ShrinkOpcode;
669 MachineOperand &Dst0 =
MI->getOperand(0);
670 MachineOperand &Dst1 =
MI->getOperand(1);
678 MachineInstr *Inst32 =
TII->buildShrunkInst(*
MI, Op32);
680 if (HaveNonDbgCarryUse) {
683 .
addReg(AMDGPU::VCC, RegState::Kill);
693 for (
unsigned I =
MI->getNumOperands() - 1;
I > 0; --
I)
694 MI->removeOperand(
I);
695 MI->setDesc(
TII->get(AMDGPU::IMPLICIT_DEF));
698 TII->commuteInstruction(*Inst32,
false);
702 assert(!Fold.needsShrink() &&
"not handled");
707 if (NewMFMAOpc == -1)
709 MI->setDesc(
TII->get(NewMFMAOpc));
710 MI->untieRegOperand(0);
711 const MCInstrDesc &MCID =
MI->getDesc();
712 for (
unsigned I = 0;
I <
MI->getNumDefs(); ++
I)
714 MI->getOperand(
I).setIsEarlyClobber(
true);
719 int OpNo =
MI->getOperandNo(&Old);
720 if (!
TII->isOperandLegal(*
MI, OpNo, &New))
723 if (ST->hasBF16InlineConstFromUpperFP32() &&
725 AMDGPU::getNamedOperandIdx(
MI->getOpcode(), AMDGPU::OpName::src0)) {
726 unsigned Opcode =
MI->getOpcode();
727 uint8_t OpType =
TII->get(Opcode).operands()[OpNo].OperandType;
730 TII->isInlineConstant(*ImmVal, OpType)) {
733 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0_modifiers);
736 MachineOperand &ModOp =
MI->getOperand(Mod0);
747 if (Fold.isGlobal()) {
748 Old.
ChangeToGA(Fold.Def.OpToFold->getGlobal(),
749 Fold.Def.OpToFold->getOffset(),
750 Fold.Def.OpToFold->getTargetFlags());
759 MachineOperand *
New = Fold.Def.OpToFold;
763 TII->getRegClass(
MI->getDesc(), Fold.UseOpNo)) {
765 TRI->getRegClassForReg(*MRI,
New->getReg());
768 if (
New->getSubReg()) {
770 TRI->getMatchingSuperRegClass(NewRC, OpRC,
New->getSubReg());
776 if (
New->getReg().isVirtual() &&
779 <<
TRI->getRegClassName(ConstrainRC) <<
'\n');
786 if (Old.
getSubReg() == AMDGPU::lo16 &&
TRI->isSGPRReg(*MRI,
New->getReg()))
788 if (
New->getReg().isPhysical()) {
796 if (
MI->isBundledWithPred()) {
798 for (MachineOperand &MO : Header.operands()) {
799 if (MO.getReg() == OldReg) {
800 MO.setReg(
New->getReg());
801 MO.setSubReg(
New->getSubReg());
810 FoldCandidate &&Entry) {
812 for (FoldCandidate &Fold : FoldList)
813 if (Fold.UseMI == Entry.UseMI && Fold.UseOpNo == Entry.UseOpNo)
815 LLVM_DEBUG(
dbgs() <<
"Append " << (Entry.Commuted ?
"commuted" :
"normal")
816 <<
" operand " << Entry.UseOpNo <<
"\n " << *Entry.UseMI);
822 const FoldableDef &FoldOp,
823 bool Commuted =
false,
int ShrinkOp = -1) {
825 FoldCandidate(
MI, OpNo, FoldOp, Commuted, ShrinkOp));
833 if (!ST->hasPKF32InstsReplicatingLower32BitsOfScalarInput())
843 const FoldableDef &OpToFold) {
844 assert(OpToFold.isImm() &&
"Expected immediate operand");
845 uint64_t ImmVal = OpToFold.getEffectiveImmVal().value();
851bool SIFoldOperandsImpl::tryAddToFoldList(
852 SmallVectorImpl<FoldCandidate> &FoldList, MachineInstr *
MI,
unsigned OpNo,
853 const FoldableDef &OpToFold)
const {
854 const unsigned Opc =
MI->getOpcode();
856 auto tryToFoldAsFMAAKorMK = [&]() {
857 if (!OpToFold.isImm())
860 const bool TryAK = OpNo == 3;
861 const unsigned NewOpc = TryAK ? AMDGPU::S_FMAAK_F32 : AMDGPU::S_FMAMK_F32;
862 MI->setDesc(
TII->get(NewOpc));
865 bool FoldAsFMAAKorMK =
866 tryAddToFoldList(FoldList,
MI, TryAK ? 3 : 2, OpToFold);
867 if (FoldAsFMAAKorMK) {
869 MI->untieRegOperand(3);
872 MachineOperand &Op1 =
MI->getOperand(1);
873 MachineOperand &Op2 =
MI->getOperand(2);
890 bool IsLegal = OpToFold.isOperandLegal(*
TII, *
MI, OpNo);
891 if (!IsLegal && OpToFold.isImm()) {
892 if (std::optional<int64_t> ImmVal = OpToFold.getEffectiveImmVal())
893 IsLegal = canUseImmWithOpSel(
MI, OpNo, *ImmVal);
899 if (NewOpc != AMDGPU::INSTRUCTION_LIST_END) {
902 MI->setDesc(
TII->get(NewOpc));
907 bool FoldAsMAD = tryAddToFoldList(FoldList,
MI, OpNo, OpToFold);
909 MI->untieRegOperand(OpNo);
913 MI->removeOperand(
MI->getNumExplicitOperands() - 1);
919 if (
Opc == AMDGPU::S_FMAC_F32 && OpNo == 3) {
920 if (tryToFoldAsFMAAKorMK())
925 if (OpToFold.isImm()) {
927 if (
Opc == AMDGPU::S_SETREG_B32)
928 ImmOpc = AMDGPU::S_SETREG_IMM32_B32;
929 else if (
Opc == AMDGPU::S_SETREG_B32_mode)
930 ImmOpc = AMDGPU::S_SETREG_IMM32_B32_mode;
932 MI->setDesc(
TII->get(ImmOpc));
941 bool CanCommute =
TII->findCommutedOpIndices(*
MI, OpNo, CommuteOpNo);
945 MachineOperand &
Op =
MI->getOperand(OpNo);
946 MachineOperand &CommutedOp =
MI->getOperand(CommuteOpNo);
952 if (!
Op.isReg() || !CommutedOp.
isReg())
957 if (
Op.isReg() && CommutedOp.
isReg() &&
958 (
Op.getReg() == CommutedOp.
getReg() &&
962 if (!
TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo))
966 if (!OpToFold.isOperandLegal(*
TII, *
MI, CommuteOpNo)) {
967 if ((
Opc != AMDGPU::V_ADD_CO_U32_e64 &&
Opc != AMDGPU::V_SUB_CO_U32_e64 &&
968 Opc != AMDGPU::V_SUBREV_CO_U32_e64) ||
969 (!OpToFold.isImm() && !OpToFold.isFI() && !OpToFold.isGlobal())) {
970 TII->commuteInstruction(*
MI,
false, OpNo, CommuteOpNo);
976 MachineOperand &OtherOp =
MI->getOperand(OpNo);
977 if (!OtherOp.
isReg() ||
984 unsigned MaybeCommutedOpc =
MI->getOpcode();
998 if (
Opc == AMDGPU::S_FMAC_F32 &&
999 (OpNo != 1 || !
MI->getOperand(1).isIdenticalTo(
MI->getOperand(2)))) {
1000 if (tryToFoldAsFMAAKorMK())
1006 if (OpToFold.isImm() &&
1015bool SIFoldOperandsImpl::isUseSafeToFold(
const MachineInstr &
MI,
1016 const MachineOperand &UseMO)
const {
1018 return !
TII->isSDWA(
MI);
1025 if (
MI.modifiesRegister(
TRI.getExec(), &
TRI))
1033bool SIFoldOperandsImpl::isTemporallyDivergentUse(
1034 const FoldableDef &OpToFold,
const MachineInstr &
UseMI)
const {
1035 if (!OpToFold.isReg())
1037 const MachineInstr *
DefMI = OpToFold.DefMI;
1040 !
TRI->isSGPRReg(*MRI, OpToFold.getReg()))
1052 SubDef &&
TII.isFoldableCopy(*SubDef);
1054 unsigned SrcIdx =
TII.getFoldableCopySrcIdx(*SubDef);
1063 if (
SrcOp.getSubReg())
1071 MachineInstr &RegSeq,
1072 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs)
const {
1088 else if (!
TRI->getCommonSubClass(RC, OpRC))
1093 Defs.emplace_back(&SrcOp, SubRegIdx);
1098 if (DefSrc && (DefSrc->
isReg() || DefSrc->
isImm())) {
1099 Defs.emplace_back(DefSrc, SubRegIdx);
1103 Defs.emplace_back(&SrcOp, SubRegIdx);
1113 SmallVectorImpl<std::pair<MachineOperand *, unsigned>> &Defs,
1116 if (!Def || !
Def->isRegSequence())
1119 return getRegSeqInit(*Def, Defs);
1122std::pair<int64_t, const TargetRegisterClass *>
1123SIFoldOperandsImpl::isRegSeqSplat(MachineInstr &RegSeq)
const {
1129 bool TryToMatchSplat64 =
false;
1131 std::optional<int64_t>
Imm;
1132 for (
unsigned I = 0,
E = Defs.
size();
I !=
E; ++
I) {
1133 const MachineOperand *
Op = Defs[
I].first;
1137 if (!Def ||
Def->isImplicitDef())
1143 int64_t SubImm =
Op->getImm();
1149 if (
Imm != SubImm) {
1150 if (
I == 1 && (
E & 1) == 0) {
1153 TryToMatchSplat64 =
true;
1161 if (!TryToMatchSplat64) {
1163 return {*
Imm, SrcRC};
1170 for (
unsigned I = 0,
E = Defs.
size();
I !=
E;
I += 2) {
1171 const MachineOperand *Op0 = Defs[
I].first;
1172 const MachineOperand *Op1 = Defs[
I + 1].first;
1177 unsigned SubReg0 = Defs[
I].second;
1178 unsigned SubReg1 = Defs[
I + 1].second;
1182 if (
TRI->getChannelFromSubReg(SubReg0) + 1 !=
1183 TRI->getChannelFromSubReg(SubReg1))
1186 if (
TRI->getSubRegIdxSize(SubReg0) != 32)
1191 SplatVal64 = MergedVal;
1192 else if (SplatVal64 != MergedVal)
1199 return {SplatVal64, RC64};
1202bool SIFoldOperandsImpl::tryFoldRegSeqSplat(
1203 MachineInstr *
UseMI,
unsigned UseOpIdx, int64_t SplatVal,
1206 if (UseOpIdx >=
Desc.getNumOperands())
1213 int16_t RCID =
TII->getOpRegClassID(
Desc.operands()[UseOpIdx]);
1222 if (SplatVal != 0 && SplatVal != -1) {
1226 uint8_t OpTy =
Desc.operands()[UseOpIdx].OperandType;
1233 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0);
1240 OpRC =
TRI->getSubRegisterClass(OpRC, AMDGPU::sub0_sub1);
1246 if (!
TRI->getCommonSubClass(OpRC, SplatRC))
1251 if (!
TII->isOperandLegal(*
UseMI, UseOpIdx, &TmpOp))
1257bool SIFoldOperandsImpl::tryToFoldACImm(
1258 const FoldableDef &OpToFold, MachineInstr *
UseMI,
unsigned UseOpIdx,
1259 SmallVectorImpl<FoldCandidate> &FoldList)
const {
1261 if (UseOpIdx >=
Desc.getNumOperands())
1268 if (OpToFold.isImm() && OpToFold.isOperandLegal(*
TII, *
UseMI, UseOpIdx)) {
1279bool SIFoldOperandsImpl::foldOperand(
1280 FoldableDef OpToFold, MachineInstr *
UseMI,
int UseOpIdx,
1281 SmallVectorImpl<FoldCandidate> &FoldList,
1282 SmallVectorImpl<MachineInstr *> &CopiesToReplace)
const {
1286 if (!isUseSafeToFold(*
UseMI, *UseOp))
1289 if (isTemporallyDivergentUse(OpToFold, *
UseMI))
1293 if (UseOp->
isReg() && OpToFold.isReg()) {
1297 if (UseOp->
getSubReg() != AMDGPU::NoSubRegister &&
1299 !
TRI->isSGPRReg(*MRI, OpToFold.getReg())))
1312 std::tie(SplatVal, SplatRC) = isRegSeqSplat(*
UseMI);
1317 for (
unsigned I = 0;
I != UsesToProcess.size(); ++
I) {
1318 MachineOperand *RSUse = UsesToProcess[
I];
1319 MachineInstr *RSUseMI = RSUse->
getParent();
1329 if (tryFoldRegSeqSplat(RSUseMI, OpNo, SplatVal, SplatRC)) {
1330 FoldableDef SplatDef(SplatVal, SplatRC);
1338 if (RSUse->
getSubReg() != RegSeqDstSubReg)
1344 FoldList, CopiesToReplace);
1350 if (tryToFoldACImm(OpToFold,
UseMI, UseOpIdx, FoldList))
1353 if (frameIndexMayFold(*
UseMI, UseOpIdx, OpToFold)) {
1358 if (
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::srsrc)->getReg() !=
1364 MachineOperand &SOff =
1365 *
TII->getNamedOperand(*
UseMI, AMDGPU::OpName::soffset);
1376 TII->getNamedOperand(*
UseMI, AMDGPU::OpName::cpol)->getImm();
1391 bool FoldingImmLike =
1392 OpToFold.isImm() || OpToFold.isFI() || OpToFold.isGlobal();
1411 for (
unsigned MovOp :
1412 {AMDGPU::S_MOV_B32, AMDGPU::V_MOV_B32_e32, AMDGPU::S_MOV_B64,
1413 AMDGPU::V_MOV_B64_PSEUDO, AMDGPU::V_MOV_B16_t16_e64,
1414 AMDGPU::V_ACCVGPR_WRITE_B32_e64, AMDGPU::AV_MOV_B32_IMM_PSEUDO,
1415 AMDGPU::AV_MOV_B64_IMM_PSEUDO}) {
1416 const MCInstrDesc &MovDesc =
TII->get(MovOp);
1426 const int SrcIdx = MovOp == AMDGPU::V_MOV_B16_t16_e64 ? 2 : 1;
1428 int16_t RegClassID =
TII->getOpRegClassID(MovDesc.
operands()[SrcIdx]);
1429 if (RegClassID != -1) {
1433 MovSrcRC =
TRI->getMatchingSuperRegClass(SrcRC, MovSrcRC, UseSubReg);
1437 if (MovOp == AMDGPU::AV_MOV_B32_IMM_PSEUDO &&
1438 (!OpToFold.isImm() ||
1439 !
TII->isImmOperandLegal(MovDesc, SrcIdx,
1440 *OpToFold.getEffectiveImmVal())))
1453 if (!OpToFold.isImm() ||
1454 !
TII->isImmOperandLegal(MovDesc, 1, *OpToFold.getEffectiveImmVal()))
1460 while (ImpOpI != ImpOpE) {
1467 if (MovOp == AMDGPU::V_MOV_B16_t16_e64) {
1469 MachineOperand NewSrcOp(SrcOp);
1491 LLVM_DEBUG(
dbgs() <<
"Folding " << *OpToFold.OpToFold <<
"\n into "
1496 unsigned SubRegIdx = OpToFold.getSubReg();
1510 static_assert(AMDGPU::sub1_hi16 == 12,
"Subregister layout has changed");
1515 if (SubRegIdx > AMDGPU::sub1) {
1516 LaneBitmask
M =
TRI->getSubRegIndexLaneMask(SubRegIdx);
1517 M |=
M.getLane(
M.getHighestLane() - 1);
1518 SmallVector<unsigned, 4> Indexes;
1519 TRI->getCoveringSubRegIndexes(
TRI->getRegClassForReg(*MRI,
UseReg), M,
1521 assert(Indexes.
size() == 1 &&
"Expected one 32-bit subreg to cover");
1522 SubRegIdx = Indexes[0];
1524 }
else if (
TII->getOpSize(*
UseMI, 1) == 4)
1527 SubRegIdx = AMDGPU::sub0;
1532 OpToFold.OpToFold->setIsKill(
false);
1537 if (foldCopyToAGPRRegSequence(
UseMI))
1542 if (UseOpc == AMDGPU::V_READFIRSTLANE_B32 ||
1543 (UseOpc == AMDGPU::V_READLANE_B32 &&
1545 AMDGPU::getNamedOperandIdx(UseOpc, AMDGPU::OpName::src0))) {
1550 if (FoldingImmLike) {
1553 *OpToFold.DefMI, *
UseMI))
1559 if (OpToFold.isImm()) {
1561 *OpToFold.getEffectiveImmVal());
1562 }
else if (OpToFold.isFI())
1565 assert(OpToFold.isGlobal());
1567 OpToFold.OpToFold->getOffset(),
1568 OpToFold.OpToFold->getTargetFlags());
1574 if (OpToFold.isReg() &&
TRI->isSGPRReg(*MRI, OpToFold.getReg())) {
1577 *OpToFold.DefMI, *
UseMI))
1599 UseDesc.
operands()[UseOpIdx].RegClass == -1)
1607 Changed |= tryAddToFoldList(FoldList,
UseMI, UseOpIdx, OpToFold);
1614 case AMDGPU::S_ADD_I32:
1615 case AMDGPU::S_ADD_U32:
1618 case AMDGPU::S_SUB_I32:
1619 case AMDGPU::S_SUB_U32:
1622 case AMDGPU::V_AND_B32_e64:
1623 case AMDGPU::V_AND_B32_e32:
1624 case AMDGPU::S_AND_B32:
1627 case AMDGPU::V_OR_B32_e64:
1628 case AMDGPU::V_OR_B32_e32:
1629 case AMDGPU::S_OR_B32:
1632 case AMDGPU::V_XOR_B32_e64:
1633 case AMDGPU::V_XOR_B32_e32:
1634 case AMDGPU::S_XOR_B32:
1637 case AMDGPU::S_XNOR_B32:
1640 case AMDGPU::S_NAND_B32:
1643 case AMDGPU::S_NOR_B32:
1646 case AMDGPU::S_ANDN2_B32:
1649 case AMDGPU::S_ORN2_B32:
1652 case AMDGPU::V_LSHL_B32_e64:
1653 case AMDGPU::V_LSHL_B32_e32:
1654 case AMDGPU::S_LSHL_B32:
1656 Result =
LHS << (
RHS & 31);
1658 case AMDGPU::V_LSHLREV_B32_e64:
1659 case AMDGPU::V_LSHLREV_B32_e32:
1660 Result =
RHS << (
LHS & 31);
1662 case AMDGPU::V_LSHR_B32_e64:
1663 case AMDGPU::V_LSHR_B32_e32:
1664 case AMDGPU::S_LSHR_B32:
1665 Result =
LHS >> (
RHS & 31);
1667 case AMDGPU::V_LSHRREV_B32_e64:
1668 case AMDGPU::V_LSHRREV_B32_e32:
1669 Result =
RHS >> (
LHS & 31);
1671 case AMDGPU::V_ASHR_I32_e64:
1672 case AMDGPU::V_ASHR_I32_e32:
1673 case AMDGPU::S_ASHR_I32:
1674 Result =
static_cast<int32_t
>(
LHS) >> (
RHS & 31);
1676 case AMDGPU::V_ASHRREV_I32_e64:
1677 case AMDGPU::V_ASHRREV_I32_e32:
1678 Result =
static_cast<int32_t
>(
RHS) >> (
LHS & 31);
1686 return IsScalar ? AMDGPU::S_MOV_B32 : AMDGPU::V_MOV_B32_e32;
1692bool SIFoldOperandsImpl::tryConstantFoldOp(MachineInstr *
MI)
const {
1693 if (!
MI->allImplicitDefsAreDead())
1696 unsigned Opc =
MI->getOpcode();
1698 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
1702 MachineOperand *Src0 = &
MI->getOperand(Src0Idx);
1703 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*MRI, *Src0);
1705 if ((
Opc == AMDGPU::V_NOT_B32_e64 ||
Opc == AMDGPU::V_NOT_B32_e32 ||
1706 Opc == AMDGPU::S_NOT_B32) &&
1708 MI->getOperand(1).ChangeToImmediate(~*Src0Imm);
1709 TII->mutateAndCleanupImplicit(
1714 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
1718 MachineOperand *Src1 = &
MI->getOperand(Src1Idx);
1719 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
1721 if (!Src0Imm && !Src1Imm)
1727 if (Src0Imm && Src1Imm) {
1732 bool IsSGPR =
TRI->isSGPRReg(*MRI,
MI->getOperand(0).getReg());
1736 MI->getOperand(Src0Idx).ChangeToImmediate(NewImm);
1737 MI->removeOperand(Src1Idx);
1744 if (
Opc == AMDGPU::S_SUB_I32 ||
Opc == AMDGPU::S_SUB_U32) {
1745 if (Src1Imm &&
static_cast<int32_t
>(*Src1Imm) == 0) {
1747 MI->removeOperand(Src1Idx);
1748 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1754 if (!
MI->isCommutable())
1757 if (Src0Imm && !Src1Imm) {
1763 int32_t Src1Val =
static_cast<int32_t
>(*Src1Imm);
1764 if (
Opc == AMDGPU::S_ADD_I32 ||
Opc == AMDGPU::S_ADD_U32) {
1767 MI->removeOperand(Src1Idx);
1768 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1774 if (
Opc == AMDGPU::V_OR_B32_e64 ||
1775 Opc == AMDGPU::V_OR_B32_e32 ||
1776 Opc == AMDGPU::S_OR_B32) {
1779 MI->removeOperand(Src1Idx);
1780 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1781 }
else if (Src1Val == -1) {
1783 MI->removeOperand(Src0Idx);
1784 TII->mutateAndCleanupImplicit(
1792 if (
Opc == AMDGPU::V_AND_B32_e64 ||
Opc == AMDGPU::V_AND_B32_e32 ||
1793 Opc == AMDGPU::S_AND_B32) {
1796 MI->removeOperand(Src0Idx);
1797 TII->mutateAndCleanupImplicit(
1799 }
else if (Src1Val == -1) {
1801 MI->removeOperand(Src1Idx);
1802 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1809 if (
Opc == AMDGPU::V_XOR_B32_e64 ||
Opc == AMDGPU::V_XOR_B32_e32 ||
1810 Opc == AMDGPU::S_XOR_B32) {
1813 MI->removeOperand(Src1Idx);
1814 TII->mutateAndCleanupImplicit(*
MI,
TII->get(AMDGPU::COPY));
1823bool SIFoldOperandsImpl::tryFoldCndMask(MachineInstr &
MI)
const {
1824 unsigned Opc =
MI.getOpcode();
1825 if (
Opc != AMDGPU::V_CNDMASK_B32_e32 &&
Opc != AMDGPU::V_CNDMASK_B32_e64 &&
1826 Opc != AMDGPU::V_CNDMASK_B64_PSEUDO)
1829 MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
1830 MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
1832 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
1836 std::optional<int64_t> Src0Imm =
TII->getImmOrMaterializedImm(*MRI, *Src0);
1837 if (!Src0Imm || *Src0Imm != *Src1Imm)
1842 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1_modifiers);
1844 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
1845 if ((Src1ModIdx != -1 &&
MI.getOperand(Src1ModIdx).getImm() != 0) ||
1846 (Src0ModIdx != -1 &&
MI.getOperand(Src0ModIdx).getImm() != 0))
1852 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
1854 MI.removeOperand(Src2Idx);
1855 MI.removeOperand(AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1));
1856 if (Src1ModIdx != -1)
1857 MI.removeOperand(Src1ModIdx);
1858 if (Src0ModIdx != -1)
1859 MI.removeOperand(Src0ModIdx);
1860 TII->mutateAndCleanupImplicit(
MI, NewDesc);
1867std::optional<SIFoldOperandsImpl::ANDMaskResult>
1868SIFoldOperandsImpl::getANDMaskRegOperand(MachineInstr &AndMI)
const {
1870 if (
Opc != AMDGPU::V_AND_B32_e64 &&
Opc != AMDGPU::V_AND_B32_e32 &&
1871 Opc != AMDGPU::S_AND_B32)
1872 return std::nullopt;
1874 std::optional<int64_t> MaskImm =
1879 MaskImm =
TII->getImmOrMaterializedImm(*MRI, AndMI.
getOperand(2));
1883 return std::nullopt;
1896bool SIFoldOperandsImpl::tryFoldRedundantAND(MachineInstr &ChildMI)
const {
1901 std::optional<ANDMaskResult> ChildResult = getANDMaskRegOperand(ChildMI);
1905 if (!ChildResult->Reg.isVirtual())
1908 MachineInstr *ParentMI = MRI->
getVRegDef(ChildResult->Reg);
1912 int64_t ParentMask = 0;
1913 std::optional<ANDMaskResult> ParentResult = getANDMaskRegOperand(*ParentMI);
1916 ParentMask = ParentResult->Mask;
1919 ParentMask = 0xffff;
1925 if ((ParentMask & ChildResult->Mask) != ParentMask)
1947bool SIFoldOperandsImpl::foldInstOperand(MachineInstr &
MI,
1948 const FoldableDef &OpToFold)
const {
1952 SmallVector<MachineInstr *, 4> CopiesToReplace;
1954 MachineOperand &Dst =
MI.getOperand(0);
1959 for (
auto *U : UsesToProcess) {
1960 MachineInstr *
UseMI =
U->getParent();
1962 FoldableDef SubOpToFold = OpToFold.getWithSubReg(*
TRI,
U->getSubReg());
1967 if (CopiesToReplace.
empty() && FoldList.
empty())
1971 for (MachineInstr *Copy : CopiesToReplace)
1972 Copy->addImplicitDefUseOperands(*MF);
1974 SetVector<MachineInstr *> ConstantFoldCandidates;
1975 for (FoldCandidate &Fold : FoldList) {
1976 assert(!Fold.isReg() || Fold.Def.OpToFold);
1977 if (Fold.isReg() && Fold.getReg().isVirtual()) {
1979 const MachineInstr *
DefMI = Fold.Def.DefMI;
1987 assert(Fold.Def.OpToFold && Fold.isReg());
1994 <<
static_cast<int>(Fold.UseOpNo) <<
" of "
1998 ConstantFoldCandidates.
insert(Fold.UseMI);
2000 }
else if (Fold.Commuted) {
2002 TII->commuteInstruction(*Fold.UseMI,
false);
2006 for (MachineInstr *
MI : ConstantFoldCandidates) {
2007 if (tryConstantFoldOp(
MI)) {
2017bool SIFoldOperandsImpl::foldCopyToAGPRRegSequence(MachineInstr *CopyMI)
const {
2024 if (!
TRI->isAGPRClass(DefRC))
2036 DenseMap<TargetInstrInfo::RegSubRegPair, Register> VGPRCopies;
2045 unsigned NumFoldable = 0;
2047 for (
unsigned I = 1;
I != NumRegSeqOperands;
I += 2) {
2064 DefRC, &AMDGPU::AGPR_32RegClass, SubRegIdx);
2084 TRI->getMatchingSuperRegClass(DefRC, InputRC, SubRegIdx);
2095 if (NumFoldable == 0)
2098 CopyMI->
setDesc(
TII->get(AMDGPU::REG_SEQUENCE));
2102 for (
auto [Def, DestSubIdx] : NewDefs) {
2103 if (!
Def->isReg()) {
2107 BuildMI(
MBB, CopyMI,
DL,
TII->get(AMDGPU::V_ACCVGPR_WRITE_B32_e64), Tmp)
2112 Def->setIsKill(
false);
2114 Register &VGPRCopy = VGPRCopies[Src];
2117 TRI->getSubRegisterClass(UseRC, DestSubIdx);
2142 B.addImm(DestSubIdx);
2149bool SIFoldOperandsImpl::tryFoldFoldableCopy(
2150 MachineInstr &
MI, MachineOperand *&CurrentKnownM0Val)
const {
2154 if (DstReg == AMDGPU::M0) {
2155 MachineOperand &NewM0Val =
MI.getOperand(1);
2156 if (CurrentKnownM0Val && CurrentKnownM0Val->
isIdenticalTo(NewM0Val)) {
2157 MI.eraseFromParent();
2168 MachineOperand *OpToFoldPtr;
2169 if (
MI.getOpcode() == AMDGPU::V_MOV_B16_t16_e64) {
2171 if (
TII->hasAnyModifiersSet(
MI))
2173 OpToFoldPtr = &
MI.getOperand(2);
2175 OpToFoldPtr = &
MI.getOperand(1);
2176 MachineOperand &OpToFold = *OpToFoldPtr;
2180 if (!FoldingImm && !OpToFold.
isReg())
2185 !
TRI->isConstantPhysReg(OpToFold.
getReg()))
2214 if (
MI.getOpcode() == AMDGPU::COPY && OpToFold.
isReg() &&
2216 if (DstRC == &AMDGPU::SReg_32RegClass &&
2225 if (OpToFold.
isReg() &&
MI.isCopy() && !
MI.getOperand(1).getSubReg()) {
2226 if (foldCopyToAGPRRegSequence(&
MI))
2230 FoldableDef
Def(OpToFold, DstRC);
2231 bool Changed = foldInstOperand(
MI, Def);
2238 auto *InstToErase = &
MI;
2240 auto &SrcOp = InstToErase->getOperand(1);
2242 InstToErase->eraseFromParent();
2244 InstToErase =
nullptr;
2248 if (!InstToErase || !
TII->isFoldableCopy(*InstToErase))
2252 if (InstToErase && InstToErase->isRegSequence() &&
2254 InstToErase->eraseFromParent();
2264 return OpToFold.
isReg() &&
2265 foldCopyToVGPROfScalarAddOfFrameIndex(DstReg, OpToFold.
getReg(),
MI);
2270const MachineOperand *
2271SIFoldOperandsImpl::isClamp(
const MachineInstr &
MI)
const {
2272 unsigned Op =
MI.getOpcode();
2274 case AMDGPU::V_MAX_F32_e64:
2275 case AMDGPU::V_MAX_F16_e64:
2276 case AMDGPU::V_MAX_F16_t16_e64:
2277 case AMDGPU::V_MAX_F16_fake16_e64:
2278 case AMDGPU::V_MAX_F64_e64:
2279 case AMDGPU::V_MAX_NUM_F64_e64:
2280 case AMDGPU::V_PK_MAX_F16:
2281 case AMDGPU::V_MAX_BF16_PSEUDO_e64:
2282 case AMDGPU::V_PK_MAX_NUM_BF16: {
2283 if (
MI.mayRaiseFPException())
2286 if (!
TII->getNamedOperand(
MI, AMDGPU::OpName::clamp)->getImm())
2290 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2291 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2295 Src0->
getSubReg() != AMDGPU::NoSubRegister)
2299 if (
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2303 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers)->getImm();
2305 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers)->getImm();
2309 unsigned UnsetMods =
2310 (
Op == AMDGPU::V_PK_MAX_F16 ||
Op == AMDGPU::V_PK_MAX_NUM_BF16)
2313 if (Src0Mods != UnsetMods && Src1Mods != UnsetMods)
2323bool SIFoldOperandsImpl::tryFoldClamp(MachineInstr &
MI) {
2324 const MachineOperand *ClampSrc = isClamp(
MI);
2340 if (
Def->mayRaiseFPException())
2343 MachineOperand *DefClamp =
TII->getNamedOperand(*Def, AMDGPU::OpName::clamp);
2347 LLVM_DEBUG(
dbgs() <<
"Folding clamp " << *DefClamp <<
" into " << *Def);
2353 Register MIDstReg =
MI.getOperand(0).getReg();
2354 if (
TRI->isSGPRReg(*MRI, DefReg)) {
2363 MI.eraseFromParent();
2368 if (
TII->convertToThreeAddress(*Def,
nullptr,
nullptr))
2369 Def->eraseFromParent();
2376 case AMDGPU::V_MUL_F64_e64:
2377 case AMDGPU::V_MUL_F64_pseudo_e64: {
2379 case 0x3fe0000000000000:
2381 case 0x4000000000000000:
2383 case 0x4010000000000000:
2389 case AMDGPU::V_MUL_F32_e64: {
2390 switch (
static_cast<uint32_t>(Val)) {
2401 case AMDGPU::V_MUL_F16_e64:
2402 case AMDGPU::V_MUL_F16_t16_e64:
2403 case AMDGPU::V_MUL_F16_fake16_e64: {
2404 switch (
static_cast<uint16_t>(Val)) {
2415 case AMDGPU::V_PK_MUL_BF16: {
2416 switch (
static_cast<uint16_t>(Val)) {
2435std::pair<const MachineOperand *, int>
2436SIFoldOperandsImpl::isOMod(
const MachineInstr &
MI)
const {
2437 unsigned Op =
MI.getOpcode();
2439 case AMDGPU::V_MUL_F64_e64:
2440 case AMDGPU::V_MUL_F64_pseudo_e64:
2441 case AMDGPU::V_MUL_F32_e64:
2442 case AMDGPU::V_MUL_F16_t16_e64:
2443 case AMDGPU::V_MUL_F16_fake16_e64:
2444 case AMDGPU::V_MUL_F16_e64: {
2446 if ((
Op == AMDGPU::V_MUL_F32_e64 &&
2448 ((
Op == AMDGPU::V_MUL_F64_e64 ||
Op == AMDGPU::V_MUL_F64_pseudo_e64 ||
2449 Op == AMDGPU::V_MUL_F16_e64 ||
Op == AMDGPU::V_MUL_F16_t16_e64 ||
2450 Op == AMDGPU::V_MUL_F16_fake16_e64) &&
2453 MI.mayRaiseFPException())
2456 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2457 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2460 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
2466 TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) ||
2467 TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) ||
2468 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2469 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2472 return {Src0, OMod};
2474 case AMDGPU::V_ADD_F64_e64:
2475 case AMDGPU::V_ADD_F64_pseudo_e64:
2476 case AMDGPU::V_ADD_F32_e64:
2477 case AMDGPU::V_ADD_F16_e64:
2478 case AMDGPU::V_ADD_F16_t16_e64:
2479 case AMDGPU::V_ADD_F16_fake16_e64: {
2481 if ((
Op == AMDGPU::V_ADD_F32_e64 &&
2483 ((
Op == AMDGPU::V_ADD_F64_e64 ||
Op == AMDGPU::V_ADD_F64_pseudo_e64 ||
2484 Op == AMDGPU::V_ADD_F16_e64 ||
Op == AMDGPU::V_ADD_F16_t16_e64 ||
2485 Op == AMDGPU::V_ADD_F16_fake16_e64) &&
2490 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2491 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2495 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src0_modifiers) &&
2496 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::src1_modifiers) &&
2497 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp) &&
2498 !
TII->hasModifiersSet(
MI, AMDGPU::OpName::omod))
2503 case AMDGPU::V_PK_MUL_BF16: {
2508 MI.mayRaiseFPException())
2511 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2512 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2515 std::optional<int64_t> Src1Imm =
TII->getImmOrMaterializedImm(*MRI, *Src1);
2519 int OMod =
getOModValue(AMDGPU::V_PK_MUL_BF16, *Src1Imm);
2524 const MachineOperand *Src0Mods =
2525 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
2526 const MachineOperand *Src1Mods =
2527 TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
2530 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2531 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2534 return {Src0, OMod};
2536 case AMDGPU::V_PK_ADD_BF16: {
2542 const MachineOperand *Src0 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src0);
2543 const MachineOperand *Src1 =
TII->getNamedOperand(
MI, AMDGPU::OpName::src1);
2550 const MachineOperand *Src0Mods =
2551 TII->getNamedOperand(
MI, AMDGPU::OpName::src0_modifiers);
2552 const MachineOperand *Src1Mods =
2553 TII->getNamedOperand(
MI, AMDGPU::OpName::src1_modifiers);
2556 TII->hasModifiersSet(
MI, AMDGPU::OpName::omod) ||
2557 TII->hasModifiersSet(
MI, AMDGPU::OpName::clamp))
2568bool SIFoldOperandsImpl::tryFoldOMod(MachineInstr &
MI) {
2569 const MachineOperand *RegOp;
2571 std::tie(RegOp, OMod) = isOMod(
MI);
2573 RegOp->
getSubReg() != AMDGPU::NoSubRegister ||
2578 Register OModSrcReg =
Def->getOperand(0).getReg();
2582 if (
Def->isRegSequence() &&
Def->getNumOperands() == 5 &&
2583 Def->getOperand(2).getImm() == AMDGPU::lo16) {
2585 bool CanLookThrough =
true;
2586 MachineInstr *Hi16Def = MRI->
getVRegDef(
Def->getOperand(3).getReg());
2588 CanLookThrough =
false;
2590 if (CanLookThrough) {
2601 MachineOperand *DefOMod =
TII->getNamedOperand(*Def, AMDGPU::OpName::omod);
2605 if (
Def->mayRaiseFPException())
2610 if (
TII->hasModifiersSet(*Def, AMDGPU::OpName::clamp))
2620 MI.eraseFromParent();
2625 if (
TII->convertToThreeAddress(*Def,
nullptr,
nullptr))
2626 Def->eraseFromParent();
2633bool SIFoldOperandsImpl::tryFoldSGPRSplatRegSequence(MachineInstr &
MI) {
2636 if (!ST->hasPackedFP64SingleSGPROps() && !ST->hasPackedU64SingleSGPROps())
2643 if (!
TRI->isSGPRClass(RegClass) ||
TRI->getRegSizeInBits(*RegClass) != 128)
2647 if (!getRegSeqInit(Defs,
Reg))
2651 if (Defs.
size() <= 1)
2654 const auto &[FirstOp,
_] = Defs.
front();
2655 if (!FirstOp->isReg())
2658 Register FirstReg = FirstOp->getReg();
2659 unsigned FirstSubReg = FirstOp->getSubReg();
2662 if (!
TRI->isSGPRClass(FirstRegClass))
2667 const auto &[
Op,
_] =
Def;
2668 return Op->isReg() &&
Op->getReg() == FirstReg &&
2669 Op->getSubReg() == FirstSubReg;
2681 MachineInstrBuilder
RS =
BuildMI(*
MI.getParent(),
MI,
MI.getDebugLoc(),
2682 TII->get(AMDGPU::REG_SEQUENCE), NewDst);
2685 FirstOp->setIsKill(
false);
2687 RS.addImm(Defs[0].second);
2692 for (
unsigned i = 1; i < Defs.
size(); ++i) {
2693 RS.addReg(UndefReg, RegState::Undef);
2694 RS.addImm(Defs[i].second);
2703 MI.eraseFromParent();
2709bool SIFoldOperandsImpl::tryFoldRegSequence(MachineInstr &
MI) {
2713 if (tryFoldSGPRSplatRegSequence(
MI))
2716 auto Reg =
MI.getOperand(0).getReg();
2718 if (!ST->hasGFX90AInsts() || !
TRI->isVGPR(*MRI,
Reg) ||
2723 if (!getRegSeqInit(Defs,
Reg))
2726 for (
auto &[
Op, SubIdx] : Defs) {
2729 if (
TRI->isAGPR(*MRI,
Op->getReg()))
2732 const MachineInstr *SubDef = MRI->
getVRegDef(
Op->getReg());
2740 MachineInstr *
UseMI =
Op->getParent();
2749 if (
Op->getSubReg())
2755 if (!OpRC || !
TRI->isVectorSuperClass(OpRC))
2761 TII->get(AMDGPU::REG_SEQUENCE), Dst);
2763 for (
auto &[Def, SubIdx] : Defs) {
2764 Def->setIsKill(
false);
2765 if (
TRI->isAGPR(*MRI,
Def->getReg())) {
2776 if (!
TII->isOperandLegal(*
UseMI, OpIdx,
Op)) {
2778 RS->eraseFromParent();
2787 MI.eraseFromParent();
2795 Register &OutReg,
unsigned &OutSubReg) {
2805 if (
TRI.isAGPR(MRI, CopySrcReg)) {
2806 OutReg = CopySrcReg;
2815 if (!CopySrcDef || !CopySrcDef->
isCopy())
2822 OtherCopySrc.
getSubReg() != AMDGPU::NoSubRegister ||
2823 !
TRI.isAGPR(MRI, OtherCopySrcReg))
2826 OutReg = OtherCopySrcReg;
2860bool SIFoldOperandsImpl::tryFoldPhiAGPR(MachineInstr &
PHI) {
2864 if (!
TRI->isVGPR(*MRI, PhiOut))
2870 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2871 MachineOperand &MO =
PHI.getOperand(K);
2873 if (!Copy || !
Copy->isCopy())
2877 unsigned AGPRRegMask = AMDGPU::NoSubRegister;
2882 if (
const auto *SubRC =
TRI->getSubRegisterClass(CopyInRC, AGPRRegMask))
2893 bool IsAGPR32 = (ARC == &AMDGPU::AGPR_32RegClass);
2897 for (
unsigned K = 1;
K <
PHI.getNumExplicitOperands();
K += 2) {
2898 MachineOperand &MO =
PHI.getOperand(K);
2902 MachineBasicBlock *InsertMBB =
nullptr;
2905 unsigned CopyOpc = AMDGPU::COPY;
2910 if (
Def->isCopy()) {
2912 unsigned AGPRSubReg = AMDGPU::NoSubRegister;
2925 MachineOperand &CopyIn =
Def->getOperand(1);
2928 CopyOpc = AMDGPU::V_ACCVGPR_WRITE_B32_e64;
2931 InsertMBB =
Def->getParent();
2939 MachineInstr *
MI =
BuildMI(*InsertMBB, InsertPt,
PHI.getDebugLoc(),
2940 TII->get(CopyOpc), NewReg)
2950 PHI.getOperand(0).setReg(NewReg);
2956 TII->get(AMDGPU::COPY), PhiOut)
2964bool SIFoldOperandsImpl::tryFoldLoad(MachineInstr &
MI) {
2966 if (!ST->hasGFX90AInsts() ||
MI.getNumExplicitDefs() != 1)
2969 MachineOperand &
Def =
MI.getOperand(0);
2986 while (!
Users.empty()) {
2987 const MachineInstr *
I =
Users.pop_back_val();
2988 if (!
I->isCopy() && !
I->isRegSequence())
2990 Register DstReg =
I->getOperand(0).getReg();
2994 if (
TRI->isAGPR(*MRI, DstReg))
2998 Users.push_back(&U);
3003 if (!
TII->isOperandLegal(
MI, 0, &Def)) {
3008 while (!MoveRegs.
empty()) {
3050bool SIFoldOperandsImpl::tryOptimizeAGPRPhis(MachineBasicBlock &
MBB) {
3053 if (ST->hasGFX90AInsts())
3057 DenseMap<std::pair<Register, unsigned>, std::vector<MachineOperand *>>
3060 for (
auto &
MI :
MBB) {
3064 if (!
TRI->isAGPR(*MRI,
MI.getOperand(0).getReg()))
3067 for (
unsigned K = 1;
K <
MI.getNumOperands();
K += 2) {
3068 MachineOperand &PhiMO =
MI.getOperand(K);
3078 for (
const auto &[Entry, MOs] : RegToMO) {
3079 if (MOs.size() == 1)
3084 MachineBasicBlock *DefMBB =
Def->getParent();
3091 MachineInstr *VGPRCopy =
3093 TII->get(AMDGPU::V_ACCVGPR_READ_B32_e64), TempVGPR)
3099 TII->get(AMDGPU::COPY), TempAGPR)
3103 for (MachineOperand *MO : MOs) {
3115bool SIFoldOperandsImpl::run(
MachineFunction &MF,
const MachineLoopInfo *MLI) {
3121 MFI = MF.
getInfo<SIMachineFunctionInfo>();
3132 MachineOperand *CurrentKnownM0Val =
nullptr;
3140 if (tryConstantFoldOp(&
MI)) {
3145 if (tryFoldRedundantAND(
MI)) {
3150 if (
MI.isRegSequence() && tryFoldRegSequence(
MI)) {
3155 if (
MI.isPHI() && tryFoldPhiAGPR(
MI)) {
3160 if (
MI.mayLoad() && tryFoldLoad(
MI)) {
3165 if (
TII->isFoldableCopy(
MI)) {
3166 Changed |= tryFoldFoldableCopy(
MI, CurrentKnownM0Val);
3171 if (CurrentKnownM0Val &&
MI.modifiesRegister(AMDGPU::M0,
TRI))
3172 CurrentKnownM0Val =
nullptr;
3192 bool Changed = SIFoldOperandsImpl().run(MF, MLI);
MachineInstrBuilder & UseMI
MachineInstrBuilder MachineInstrBuilder & DefMI
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static bool updateOperand(Instruction *Inst, unsigned Idx, Instruction *Mat)
Updates the operand at Idx in instruction Inst with the result of instruction Mat.
This file builds on the ADT/GraphTraits.h file to build generic depth first graph iterator.
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
const HexagonInstrInfo * TII
iv Induction Variable Users
Register const TargetRegisterInfo * TRI
Promote Memory to Register
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static bool loopModifiesExec(const MachineLoop &L, const SIRegisterInfo &TRI)
static unsigned macToMad(unsigned Opc)
static bool isAGPRCopy(const SIRegisterInfo &TRI, const MachineRegisterInfo &MRI, const MachineInstr &Copy, Register &OutReg, unsigned &OutSubReg)
Checks whether Copy is a AGPR -> VGPR copy.
static void appendFoldCandidate(SmallVectorImpl< FoldCandidate > &FoldList, FoldCandidate &&Entry)
static const TargetRegisterClass * getRegOpRC(const MachineRegisterInfo &MRI, const TargetRegisterInfo &TRI, const MachineOperand &MO)
static bool evalBinaryInstruction(unsigned Opcode, int32_t &Result, uint32_t LHS, uint32_t RHS)
static int getOModValue(unsigned Opc, int64_t Val)
static unsigned getMovOpc(bool IsScalar)
static MachineOperand * lookUpCopyChain(const SIInstrInfo &TII, const MachineRegisterInfo &MRI, Register SrcReg)
static bool checkImmOpForPKF32InstrReplicatesLower32BitsOfScalarOperand(const FoldableDef &OpToFold)
static bool isPKF32InstrReplicatesLower32BitsOfScalarOperand(const GCNSubtarget *ST, MachineInstr *MI, unsigned OpNo)
Interface definition for SIInstrInfo.
Interface definition for SIRegisterInfo.
static int Lookup(ArrayRef< TableEntry > Table, unsigned Opcode)
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Represents analyses that only rely on functions' control flow.
FunctionPass class - This class is used to implement most global optimizations.
const SIInstrInfo * getInstrInfo() const override
bool hasDOTOpSelHazard() const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
const HexagonRegisterInfo & getRegisterInfo() const
bool contains(const LoopT *L) const
Return true if the specified loop is contained within this loop.
LoopT * getLoopFor(const BlockT *BB) const
Return the inner most loop that BB lives in.
ArrayRef< MCOperandInfo > operands() const
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
bool isVariadic() const
Return true if this instruction can have a variable number of operands.
This holds information about one operand of a machine instruction, indicating the register class for ...
uint8_t OperandType
Information about the type of the operand.
bool hasSuperClassEq(const MCRegisterClass *RC) const
Returns true if RC is a super-class of or equal to this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
bool hasSubClassEq(const MCRegisterClass *RC) const
Returns true if RC is a sub-class of or equal to this class.
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI iterator SkipPHIsLabelsAndDebug(iterator I, Register Reg=Register(), bool SkipPseudoOp=true)
Return the first instruction in MBB after I that is not a PHI, label or debug.
LLVM_ABI LivenessQueryResult computeRegisterLiveness(const TargetRegisterInfo *TRI, MCRegister Reg, const_iterator Before, unsigned Neighborhood=10) const
Return whether (physical) register Reg has been defined and not killed as of just before Before.
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
LivenessQueryResult
Possible outcome of a register liveness query to computeRegisterLiveness()
@ LQR_Dead
Register is known to be fully dead.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
Properties which a MachineFunction may have at a given point in time.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool isImplicitDef() const
const MachineBasicBlock * getParent() const
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
LLVM_ABI bool allImplicitDefsAreDead() const
Return true if all the implicit defs of this instruction are dead.
unsigned getNumOperands() const
Retuns the total number of operands.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
unsigned getOperandNo(const_mop_iterator I) const
Returns the number of the operand iterator I points to.
LLVM_ABI unsigned getNumExplicitOperands() const
Returns the number of non-implicit operands.
mop_range implicit_operands()
const MCInstrDesc & getDesc() const
Returns the target instruction descriptor of this MachineInstr.
void clearFlag(MIFlag Flag)
clearFlag - Clear a MI flag.
bool isRegSequence() const
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
const DebugLoc & getDebugLoc() const
Returns the debug location id of this MachineInstr.
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
Analysis pass that exposes the MachineLoopInfo for a machine function.
MachineOperand class - Representation of each machine instruction operand.
void setSubReg(unsigned subReg)
unsigned getSubReg() const
LLVM_ABI unsigned getOperandNo() const
Returns the index of this operand in the instruction that it belongs to.
LLVM_ABI void substVirtReg(Register Reg, unsigned SubIdx, const TargetRegisterInfo &)
substVirtReg - Substitute the current register with the virtual subregister Reg:SubReg.
LLVM_ABI void ChangeToFrameIndex(int Idx, unsigned TargetFlags=0)
Replace this operand with a frame index.
void setImm(int64_t immVal)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
LLVM_ABI void ChangeToImmediate(int64_t ImmVal, unsigned TargetFlags=0)
ChangeToImmediate - Replace this operand with a new immediate operand of the specified value.
LLVM_ABI void ChangeToGA(const GlobalValue *GV, int64_t Offset, unsigned TargetFlags=0)
ChangeToGA - Replace this operand with a new global address operand.
void setIsKill(bool Val=true)
LLVM_ABI void ChangeToRegister(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isDebug=false)
ChangeToRegister - Replace this operand with a new register operand of the specified value.
MachineInstr * getParent()
getParent - Return the instruction that this operand belongs to.
LLVM_ABI void substPhysReg(MCRegister Reg, const TargetRegisterInfo &)
substPhysReg - Substitute the current register with the physical register Reg, taking any existing Su...
static MachineOperand CreateImm(int64_t Val)
bool isGlobal() const
isGlobal - Tests if this is a MO_GlobalAddress operand.
MachineOperandType getType() const
getType - Returns the MachineOperandType for this operand.
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
@ MO_Immediate
Immediate operand.
@ MO_GlobalAddress
Address of a global value.
@ MO_FrameIndex
Abstract Stack Frame Index.
@ MO_Register
Register operand.
static MachineOperand CreateFI(int Idx)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
use_nodbg_iterator use_nodbg_begin(Register RegNo) const
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
iterator_range< use_nodbg_iterator > use_nodbg_operands(Register Reg) const
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool hasOneNonDBGUser(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug instruction using the specified regis...
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
void setRegAllocationHint(Register VReg, unsigned Type, Register PrefReg)
setRegAllocationHint - Specify a register allocation hint for the specified virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI const TargetRegisterClass * constrainRegClass(Register Reg, const TargetRegisterClass *RC, unsigned MinNumRegs=0)
constrainRegClass - Constrain the register class of the specified virtual register to be a common sub...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Wrapper class representing virtual and physical registers.
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
static bool hasSameClamp(const MachineInstr &A, const MachineInstr &B)
static std::optional< int64_t > extractSubregFromImm(int64_t ImmVal, unsigned SubRegIndex)
Return the extracted immediate value in a subregister use from a constant materialized in a super reg...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
SIModeRegisterDefaults getMode() const
bool insert(const value_type &X)
Insert a new element into the SetVector.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
Represent a constant reference to a string, i.e.
static const unsigned CommuteAnyOperandIndex
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
LLVM_READONLY int32_t getMFMAEarlyClobberOp(uint32_t Opcode)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
LLVM_READONLY int32_t getVOPe32(uint32_t Opcode)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_REG_INLINE_AC_FP64
LLVM_READONLY int32_t getFlatScratchInstSSfromSV(uint32_t Opcode)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
constexpr bool isVOP3(const T &...O)
constexpr bool isMAI(const T &...O)
constexpr bool isSWMMAC(const T &...O)
constexpr bool isVOP3P(const T &...O)
constexpr bool isWMMA(const T &...O)
constexpr bool isDOT(const T &...O)
constexpr bool isPacked(const T &...O)
NodeAddr< DefNode * > Def
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
TargetInstrInfo::RegSubRegPair getRegSubRegPair(const MachineOperand &O)
Create RegSubRegPair from a register MachineOperand.
MachineBasicBlock::instr_iterator getBundleStart(MachineBasicBlock::instr_iterator I)
Returns an iterator to the first instruction in the bundle containing I.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
bool execMayBeModifiedBeforeUse(const MachineRegisterInfo &MRI, Register VReg, const MachineInstr &DefMI, const MachineInstr &UseMI)
Return false if EXEC is not changed between the def of VReg at DefMI and the use at UseMI.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
FunctionPass * createSIFoldOperandsLegacyPass()
char & SIFoldOperandsLegacyID
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
@ Sub
Subtraction of integers.
DWARFExpression::Operation Op
iterator_range< pointer_iterator< WrappedIteratorT > > make_pointer_range(RangeT &&Range)
iterator_range< df_iterator< T > > depth_first(const T &G)
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
MCRegisterClass TargetRegisterClass
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
bool IEEE
Floating point opcodes that support exception flag gathering quiet and propagate signaling NaN inputs...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.