LLVM 24.0.0git
AMDGPUBaseInfo.cpp
Go to the documentation of this file.
1//===- AMDGPUBaseInfo.cpp - AMDGPU Base encoding information --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "AMDGPUBaseInfo.h"
10#include "AMDGPU.h"
11#include "AMDGPUAsmUtils.h"
12#include "AMDKernelCodeT.h"
17#include "llvm/IR/Attributes.h"
18#include "llvm/IR/Constants.h"
19#include "llvm/IR/Function.h"
20#include "llvm/IR/GlobalValue.h"
21#include "llvm/IR/IntrinsicsAMDGPU.h"
22#include "llvm/IR/IntrinsicsR600.h"
23#include "llvm/IR/LLVMContext.h"
24#include "llvm/IR/Metadata.h"
25#include "llvm/MC/MCInstrInfo.h"
30#include <optional>
31
32#define GET_INSTRINFO_NAMED_OPS
33#define GET_INSTRMAP_INFO
34#include "AMDGPUGenInstrInfo.inc"
35
37 "amdhsa-code-object-version", llvm::cl::Hidden,
39 llvm::cl::desc("Set default AMDHSA Code Object Version (module flag "
40 "or asm directive still take priority if present)"));
41
42namespace {
43
44/// \returns Bit mask for given bit \p Shift and bit \p Width.
45unsigned getBitMask(unsigned Shift, unsigned Width) {
46 return ((1 << Width) - 1) << Shift;
47}
48
49/// Packs \p Src into \p Dst for given bit \p Shift and bit \p Width.
50///
51/// \returns Packed \p Dst.
52unsigned packBits(unsigned Src, unsigned Dst, unsigned Shift, unsigned Width) {
53 unsigned Mask = getBitMask(Shift, Width);
54 return ((Src << Shift) & Mask) | (Dst & ~Mask);
55}
56
57/// Unpacks bits from \p Src for given bit \p Shift and bit \p Width.
58///
59/// \returns Unpacked bits.
60unsigned unpackBits(unsigned Src, unsigned Shift, unsigned Width) {
61 return (Src & getBitMask(Shift, Width)) >> Shift;
62}
63
64/// \returns Vmcnt bit shift (lower bits).
65unsigned getVmcntBitShiftLo(unsigned VersionMajor) {
66 return VersionMajor >= 11 ? 10 : 0;
67}
68
69/// \returns Vmcnt bit width (lower bits).
70unsigned getVmcntBitWidthLo(unsigned VersionMajor) {
71 return VersionMajor >= 11 ? 6 : 4;
72}
73
74/// \returns Expcnt bit shift.
75unsigned getExpcntBitShift(unsigned VersionMajor) {
76 return VersionMajor >= 11 ? 0 : 4;
77}
78
79/// \returns Expcnt bit width.
80unsigned getExpcntBitWidth(unsigned VersionMajor) { return 3; }
81
82/// \returns Lgkmcnt bit shift.
83unsigned getLgkmcntBitShift(unsigned VersionMajor) {
84 return VersionMajor >= 11 ? 4 : 8;
85}
86
87/// \returns Lgkmcnt bit width.
88unsigned getLgkmcntBitWidth(unsigned VersionMajor) {
89 return VersionMajor >= 10 ? 6 : 4;
90}
91
92/// \returns Vmcnt bit shift (higher bits).
93unsigned getVmcntBitShiftHi(unsigned VersionMajor) { return 14; }
94
95/// \returns Vmcnt bit width (higher bits).
96unsigned getVmcntBitWidthHi(unsigned VersionMajor) {
97 return (VersionMajor == 9 || VersionMajor == 10) ? 2 : 0;
98}
99
100/// \returns Loadcnt bit width
101unsigned getLoadcntBitWidth(unsigned VersionMajor) {
102 return VersionMajor >= 12 ? 6 : 0;
103}
104
105/// \returns Samplecnt bit width.
106unsigned getSamplecntBitWidth(unsigned VersionMajor) {
107 return VersionMajor >= 12 ? 6 : 0;
108}
109
110/// \returns Bvhcnt bit width.
111unsigned getBvhcntBitWidth(unsigned VersionMajor) {
112 return VersionMajor >= 12 ? 3 : 0;
113}
114
115/// \returns Dscnt bit width.
116unsigned getDscntBitWidth(unsigned VersionMajor) {
117 return VersionMajor >= 12 ? 6 : 0;
118}
119
120/// \returns Dscnt bit shift in combined S_WAIT instructions.
121unsigned getDscntBitShift(unsigned VersionMajor) { return 0; }
122
123/// \returns Storecnt or Vscnt bit width, depending on VersionMajor.
124unsigned getStorecntBitWidth(unsigned VersionMajor) {
125 return VersionMajor >= 10 ? 6 : 0;
126}
127
128/// \returns Kmcnt bit width.
129unsigned getKmcntBitWidth(unsigned VersionMajor) {
130 return VersionMajor >= 12 ? 5 : 0;
131}
132
133/// \returns Xcnt bit width.
134unsigned getXcntBitWidth(unsigned VersionMajor, unsigned VersionMinor) {
135 return VersionMajor == 12 && VersionMinor == 5 ? 6 : 0;
136}
137
138/// \returns Asynccnt bit width.
139unsigned getAsynccntBitWidth(unsigned VersionMajor, unsigned VersionMinor) {
140 return VersionMajor == 12 && VersionMinor == 5 ? 6 : 0;
141}
142
143/// \returns shift for Loadcnt/Storecnt in combined S_WAIT instructions.
144unsigned getLoadcntStorecntBitShift(unsigned VersionMajor) {
145 return VersionMajor >= 12 ? 8 : 0;
146}
147
148/// \returns VaSdst bit width
149inline unsigned getVaSdstBitWidth() { return 3; }
150
151/// \returns VaSdst bit shift
152inline unsigned getVaSdstBitShift() { return 9; }
153
154/// \returns VmVsrc bit width
155inline unsigned getVmVsrcBitWidth() { return 3; }
156
157/// \returns VmVsrc bit shift
158inline unsigned getVmVsrcBitShift() { return 2; }
159
160/// \returns VaVdst bit width
161inline unsigned getVaVdstBitWidth() { return 4; }
162
163/// \returns VaVdst bit shift
164inline unsigned getVaVdstBitShift() { return 12; }
165
166/// \returns VaVcc bit width
167inline unsigned getVaVccBitWidth() { return 1; }
168
169/// \returns VaVcc bit shift
170inline unsigned getVaVccBitShift() { return 1; }
171
172/// \returns SaSdst bit width
173inline unsigned getSaSdstBitWidth() { return 1; }
174
175/// \returns SaSdst bit shift
176inline unsigned getSaSdstBitShift() { return 0; }
177
178/// \returns VaSsrc width
179inline unsigned getVaSsrcBitWidth() { return 1; }
180
181/// \returns VaSsrc bit shift
182inline unsigned getVaSsrcBitShift() { return 8; }
183
184/// \returns HoldCnt bit shift
185inline unsigned getHoldCntWidth(unsigned VersionMajor, unsigned VersionMinor) {
186 static constexpr const unsigned MinMajor = 10;
187 static constexpr const unsigned MinMinor = 3;
188 return std::tie(VersionMajor, VersionMinor) >= std::tie(MinMajor, MinMinor)
189 ? 1
190 : 0;
191}
192
193/// \returns HoldCnt bit shift
194inline unsigned getHoldCntBitShift() { return 7; }
195
196} // end anonymous namespace
197
198namespace llvm {
199
200namespace AMDGPU {
201
202/// \returns true if the target supports signed immediate offset for SMRD
203/// instructions.
205 return isGFX9Plus(ST);
206}
207
208/// \returns True if \p STI is AMDHSA.
209bool isHsaAbi(const MCSubtargetInfo &STI) {
210 return STI.getTargetTriple().getOS() == Triple::AMDHSA;
211}
212
215 M.getModuleFlag("amdhsa_code_object_version"))) {
216 return (unsigned)Ver->getZExtValue() / 100;
217 }
218
220}
221
225
226unsigned getAMDHSACodeObjectVersion(unsigned ABIVersion) {
227 switch (ABIVersion) {
229 return 4;
231 return 5;
233 return 6;
234 default:
236 }
237}
238
239uint8_t getELFABIVersion(const Triple &T, unsigned CodeObjectVersion) {
240 if (T.getOS() != Triple::AMDHSA)
241 return 0;
242
243 switch (CodeObjectVersion) {
244 case 4:
246 case 5:
248 case 6:
250 default:
251 report_fatal_error("Unsupported AMDHSA Code Object Version " +
252 Twine(CodeObjectVersion));
253 }
254}
255
256unsigned getMultigridSyncArgImplicitArgPosition(unsigned CodeObjectVersion) {
257 switch (CodeObjectVersion) {
258 case AMDHSA_COV4:
259 return 48;
260 case AMDHSA_COV5:
261 case AMDHSA_COV6:
262 default:
264 }
265}
266
267// FIXME: All such magic numbers about the ABI should be in a
268// central TD file.
269unsigned getHostcallImplicitArgPosition(unsigned CodeObjectVersion) {
270 switch (CodeObjectVersion) {
271 case AMDHSA_COV4:
272 return 24;
273 case AMDHSA_COV5:
274 case AMDHSA_COV6:
275 default:
277 }
278}
279
280unsigned getDefaultQueueImplicitArgPosition(unsigned CodeObjectVersion) {
281 switch (CodeObjectVersion) {
282 case AMDHSA_COV4:
283 return 32;
284 case AMDHSA_COV5:
285 case AMDHSA_COV6:
286 default:
288 }
289}
290
291unsigned getCompletionActionImplicitArgPosition(unsigned CodeObjectVersion) {
292 switch (CodeObjectVersion) {
293 case AMDHSA_COV4:
294 return 40;
295 case AMDHSA_COV5:
296 case AMDHSA_COV6:
297 default:
299 }
300}
301
302#define GET_MIMGBaseOpcodesTable_IMPL
303#define GET_MIMGDimInfoTable_IMPL
304#define GET_MIMGInfoTable_IMPL
305#define GET_MIMGLZMappingTable_IMPL
306#define GET_MIMGMIPMappingTable_IMPL
307#define GET_MIMGBiasMappingTable_IMPL
308#define GET_MIMGOffsetMappingTable_IMPL
309#define GET_MIMGG16MappingTable_IMPL
310#define GET_MAIInstInfoTable_IMPL
311#define GET_WMMAInstInfoTable_IMPL
312#include "AMDGPUGenSearchableTables.inc"
313
314int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding,
315 unsigned VDataDwords, unsigned VAddrDwords) {
316 const MIMGInfo *Info =
317 getMIMGOpcodeHelper(BaseOpcode, MIMGEncoding, VDataDwords, VAddrDwords);
318 return Info ? Info->Opcode : -1;
319}
320
322 const MIMGInfo *Info = getMIMGInfo(Opc);
323 return Info ? getMIMGBaseOpcodeInfo(Info->BaseOpcode) : nullptr;
324}
325
326int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels) {
327 const MIMGInfo *OrigInfo = getMIMGInfo(Opc);
328 const MIMGInfo *NewInfo =
329 getMIMGOpcodeHelper(OrigInfo->BaseOpcode, OrigInfo->MIMGEncoding,
330 NewChannels, OrigInfo->VAddrDwords);
331 return NewInfo ? NewInfo->Opcode : -1;
332}
333
334unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode,
335 const MIMGDimInfo *Dim, bool IsA16,
336 bool IsG16Supported) {
337 unsigned AddrWords = BaseOpcode->NumExtraArgs;
338 unsigned AddrComponents = (BaseOpcode->Coordinates ? Dim->NumCoords : 0) +
339 (BaseOpcode->LodOrClampOrMip ? 1 : 0);
340 if (IsA16)
341 AddrWords += divideCeil(AddrComponents, 2);
342 else
343 AddrWords += AddrComponents;
344
345 // Note: For subtargets that support A16 but not G16, enabling A16 also
346 // enables 16 bit gradients.
347 // For subtargets that support A16 (operand) and G16 (done with a different
348 // instruction encoding), they are independent.
349
350 if (BaseOpcode->Gradients) {
351 if ((IsA16 && !IsG16Supported) || BaseOpcode->G16)
352 // There are two gradients per coordinate, we pack them separately.
353 // For the 3d case,
354 // we get (dy/du, dx/du) (-, dz/du) (dy/dv, dx/dv) (-, dz/dv)
355 AddrWords += alignTo<2>(Dim->NumGradients / 2);
356 else
357 AddrWords += Dim->NumGradients;
358 }
359 return AddrWords;
360}
361
372
381
386
391
395
399
403
408
416
421
424 bool IsX;
425 bool IsY;
426};
427
428#define GET_FP4FP8DstByteSelTable_DECL
429#define GET_FP4FP8DstByteSelTable_IMPL
430
435
441
442#define GET_DPMACCInstructionTable_DECL
443#define GET_DPMACCInstructionTable_IMPL
444#define GET_MTBUFInfoTable_DECL
445#define GET_MTBUFInfoTable_IMPL
446#define GET_MUBUFInfoTable_DECL
447#define GET_MUBUFInfoTable_IMPL
448#define GET_SMInfoTable_DECL
449#define GET_SMInfoTable_IMPL
450#define GET_VOP1InfoTable_DECL
451#define GET_VOP1InfoTable_IMPL
452#define GET_VOP2InfoTable_DECL
453#define GET_VOP2InfoTable_IMPL
454#define GET_VOP3InfoTable_DECL
455#define GET_VOP3InfoTable_IMPL
456#define GET_VOPC64DPPTable_DECL
457#define GET_VOPC64DPPTable_IMPL
458#define GET_VOPC64DPP8Table_DECL
459#define GET_VOPC64DPP8Table_IMPL
460#define GET_VOPCAsmOnlyInfoTable_DECL
461#define GET_VOPCAsmOnlyInfoTable_IMPL
462#define GET_VOP3CAsmOnlyInfoTable_DECL
463#define GET_VOP3CAsmOnlyInfoTable_IMPL
464#define GET_VOPDComponentTable_DECL
465#define GET_VOPDComponentTable_IMPL
466#define GET_VOPDPairs_DECL
467#define GET_VOPDPairs_IMPL
468#define GET_VOPDXYTable_DECL
469#define GET_VOPDXYTable_IMPL
470#define GET_VOPTrue16Table_DECL
471#define GET_VOPTrue16Table_IMPL
472#define GET_True16D16Table_IMPL
473#define GET_WMMAOpcode2AddrMappingTable_DECL
474#define GET_WMMAOpcode2AddrMappingTable_IMPL
475#define GET_WMMAOpcode3AddrMappingTable_DECL
476#define GET_WMMAOpcode3AddrMappingTable_IMPL
477#define GET_getMFMA_F8F6F4_WithSize_DECL
478#define GET_getMFMA_F8F6F4_WithSize_IMPL
479#define GET_isMFMA_F8F6F4Table_IMPL
480#define GET_isCvtScaleF32_F32F16ToF8F4Table_IMPL
481
482#include "AMDGPUGenSearchableTables.inc"
483
484int getMTBUFBaseOpcode(unsigned Opc) {
485 const MTBUFInfo *Info = getMTBUFInfoFromOpcode(Opc);
486 return Info ? Info->BaseOpcode : -1;
487}
488
489int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements) {
490 const MTBUFInfo *Info =
491 getMTBUFInfoFromBaseOpcodeAndElements(BaseOpc, Elements);
492 return Info ? Info->Opcode : -1;
493}
494
495int getMTBUFElements(unsigned Opc) {
496 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
497 return Info ? Info->elements : 0;
498}
499
500bool getMTBUFHasVAddr(unsigned Opc) {
501 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
502 return Info && Info->has_vaddr;
503}
504
505bool getMTBUFHasSrsrc(unsigned Opc) {
506 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
507 return Info && Info->has_srsrc;
508}
509
510bool getMTBUFHasSoffset(unsigned Opc) {
511 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
512 return Info && Info->has_soffset;
513}
514
515int getMUBUFBaseOpcode(unsigned Opc) {
516 const MUBUFInfo *Info = getMUBUFInfoFromOpcode(Opc);
517 return Info ? Info->BaseOpcode : -1;
518}
519
520int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements) {
521 const MUBUFInfo *Info =
522 getMUBUFInfoFromBaseOpcodeAndElements(BaseOpc, Elements);
523 return Info ? Info->Opcode : -1;
524}
525
526int getMUBUFElements(unsigned Opc) {
527 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
528 return Info ? Info->elements : 0;
529}
530
531bool getMUBUFHasVAddr(unsigned Opc) {
532 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
533 return Info && Info->has_vaddr;
534}
535
536bool getMUBUFHasSrsrc(unsigned Opc) {
537 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
538 return Info && Info->has_srsrc;
539}
540
541bool getMUBUFHasSoffset(unsigned Opc) {
542 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
543 return Info && Info->has_soffset;
544}
545
546bool getMUBUFIsBufferInv(unsigned Opc) {
547 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
548 return Info && Info->IsBufferInv;
549}
550
551bool getMUBUFTfe(unsigned Opc) {
552 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
553 return Info && Info->tfe;
554}
555
556bool getSMEMIsBuffer(unsigned Opc) {
557 const SMInfo *Info = getSMEMOpcodeHelper(Opc);
558 return Info && Info->IsBuffer;
559}
560
561bool getVOP1IsSingle(unsigned Opc) {
562 const VOPInfo *Info = getVOP1OpcodeHelper(Opc);
563 return !Info || Info->IsSingle;
564}
565
566bool getVOP2IsSingle(unsigned Opc) {
567 const VOPInfo *Info = getVOP2OpcodeHelper(Opc);
568 return !Info || Info->IsSingle;
569}
570
571bool getVOP3IsSingle(unsigned Opc) {
572 const VOPInfo *Info = getVOP3OpcodeHelper(Opc);
573 return !Info || Info->IsSingle;
574}
575
576bool isVOPC64DPP(unsigned Opc) {
577 return isVOPC64DPPOpcodeHelper(Opc) || isVOPC64DPP8OpcodeHelper(Opc);
578}
579
580bool isVOPCAsmOnly(unsigned Opc) { return isVOPCAsmOnlyOpcodeHelper(Opc); }
581
582bool getMAIIsDGEMM(unsigned Opc) {
583 const MAIInstInfo *Info = getMAIInstInfoHelper(Opc);
584 return Info && Info->is_dgemm;
585}
586
587bool getMAIIsGFX940XDL(unsigned Opc) {
588 const MAIInstInfo *Info = getMAIInstInfoHelper(Opc);
589 return Info && Info->is_gfx940_xdl;
590}
591
592bool getWMMAIsXDL(unsigned Opc) {
593 const WMMAInstInfo *Info = getWMMAInstInfoHelper(Opc);
594 return Info ? Info->is_wmma_xdl : false;
595}
596
597bool getHasMatrixScale(unsigned Opc) {
598 const WMMAInstInfo *Info = getWMMAInstInfoHelper(Opc);
599 return Info && Info->HasMatrixScale;
600}
601
603 switch (EncodingVal) {
606 return 6;
608 return 4;
611 default:
612 return 8;
613 }
614
615 llvm_unreachable("covered switch over mfma scale formats");
616}
617
619 unsigned BLGP,
620 unsigned F8F8Opcode) {
621 uint8_t SrcANumRegs = mfmaScaleF8F6F4FormatToNumRegs(CBSZ);
622 uint8_t SrcBNumRegs = mfmaScaleF8F6F4FormatToNumRegs(BLGP);
623 return getMFMA_F8F6F4_InstWithNumRegs(SrcANumRegs, SrcBNumRegs, F8F8Opcode);
624}
625
627 switch (Fmt) {
630 return 16;
633 return 12;
635 return 8;
636 }
637
638 llvm_unreachable("covered switch over wmma scale formats");
639}
640
642 unsigned FmtB,
643 unsigned F8F8Opcode) {
644 uint8_t SrcANumRegs = wmmaScaleF8F6F4FormatToNumRegs(FmtA);
645 uint8_t SrcBNumRegs = wmmaScaleF8F6F4FormatToNumRegs(FmtB);
646 return getMFMA_F8F6F4_InstWithNumRegs(SrcANumRegs, SrcBNumRegs, F8F8Opcode);
647}
648
649bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale,
650 unsigned BFmt, unsigned BScale) {
651 auto isValid = [](unsigned Fmt, unsigned Scale) -> bool {
652 switch (Fmt) {
657 if (Scale != WMMA::MATRIX_SCALE_FMT_E8)
658 return false;
659 break;
661 if (Scale != WMMA::MATRIX_SCALE_FMT_E8 &&
664 return false;
665 break;
666 }
667 return true;
668 };
669
670 if (!isValid(AFmt, AScale) || !isValid(BFmt, BScale))
671 return false;
672
673 if (AFmt == WMMA::MATRIX_FMT_FP4 && BFmt == WMMA::MATRIX_FMT_FP4 &&
674 AScale != BScale)
675 return false;
676
677 return true;
678}
679
681 if (ST.hasFeature(AMDGPU::FeatureGFX13Insts))
683 if (ST.hasFeature(AMDGPU::FeatureGFX1250Insts))
685 if (ST.hasFeature(AMDGPU::FeatureGFX12Insts))
687 if (ST.hasFeature(AMDGPU::FeatureGFX11_7Insts))
689 if (ST.hasFeature(AMDGPU::FeatureGFX11Insts))
691 llvm_unreachable("Subtarget generation does not support VOPD!");
692}
693
694CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3) {
695 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(Opc) : 0;
696 Opc = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : Opc;
697 // Normalize through VOPDComponentTable so that e32 and e64 variants
698 // of the same logical opcode all share a single entry.
699 const VOPDComponentInfo *Info = getVOPDComponentHelper(Opc);
700 if (!Info)
701 return {false, false};
702 unsigned Key =
703 (Info->VOPDOp << 5) | (EncodingFamily << 1) | (VOPD3 ? 1u : 0u);
704 const VOPDXYInfo *XYInfo = getVOPDXYInfo(Key);
705 if (!XYInfo)
706 return {false, false};
707 return {XYInfo->IsX, XYInfo->IsY};
708}
709
710unsigned getVOPDOpcode(unsigned Opc, bool VOPD3) {
711 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(Opc) : 0;
712 Opc = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : Opc;
713 const VOPDComponentInfo *Info = getVOPDComponentHelper(Opc);
714 return Info ? Info->VOPDOp : ~0u;
715}
716
717bool isVOPD(unsigned Opc) {
718 return AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src0X);
719}
720
721bool isMAC(unsigned Opc) {
722 return Opc == AMDGPU::V_MAC_F32_e64_gfx6_gfx7 ||
723 Opc == AMDGPU::V_MAC_F32_e64_gfx10 ||
724 Opc == AMDGPU::V_MAC_F32_e64_vi ||
725 Opc == AMDGPU::V_MAC_LEGACY_F32_e64_gfx6_gfx7 ||
726 Opc == AMDGPU::V_MAC_LEGACY_F32_e64_gfx10 ||
727 Opc == AMDGPU::V_MAC_F16_e64_vi ||
728 Opc == AMDGPU::V_FMAC_F64_e64_gfx90a ||
729 Opc == AMDGPU::V_FMAC_F64_e64_gfx12 ||
730 Opc == AMDGPU::V_FMAC_F64_e64_gfx13 ||
731 Opc == AMDGPU::V_FMAC_F32_e64_gfx10 ||
732 Opc == AMDGPU::V_FMAC_F32_e64_gfx11 ||
733 Opc == AMDGPU::V_FMAC_F32_e64_gfx12 ||
734 Opc == AMDGPU::V_FMAC_F32_e64_gfx13 ||
735 Opc == AMDGPU::V_FMAC_F32_e64_vi ||
736 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64_gfx10 ||
737 Opc == AMDGPU::V_FMAC_DX9_ZERO_F32_e64_gfx11 ||
738 Opc == AMDGPU::V_FMAC_F16_e64_gfx10 ||
739 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx11 ||
740 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx11 ||
741 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx12 ||
742 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx12 ||
743 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx13 ||
744 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx13 ||
745 Opc == AMDGPU::V_DOT2C_F32_F16_e64_vi ||
746 Opc == AMDGPU::V_DOT2C_F32_BF16_e64_vi ||
747 Opc == AMDGPU::V_DOT2C_I32_I16_e64_vi ||
748 Opc == AMDGPU::V_DOT4C_I32_I8_e64_vi ||
749 Opc == AMDGPU::V_DOT8C_I32_I4_e64_vi;
750}
751
752bool isPermlane16(unsigned Opc) {
753 return Opc == AMDGPU::V_PERMLANE16_B32_gfx10 ||
754 Opc == AMDGPU::V_PERMLANEX16_B32_gfx10 ||
755 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx11 ||
756 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx11 ||
757 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx12 ||
758 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx13 ||
759 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx12 ||
760 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx13 ||
761 Opc == AMDGPU::V_PERMLANE16_VAR_B32_e64_gfx12 ||
762 Opc == AMDGPU::V_PERMLANE16_VAR_B32_e64_gfx13 ||
763 Opc == AMDGPU::V_PERMLANEX16_VAR_B32_e64_gfx12 ||
764 Opc == AMDGPU::V_PERMLANEX16_VAR_B32_e64_gfx13;
765}
766
768 return Opc == AMDGPU::V_CVT_F32_BF8_e64_gfx12 ||
769 Opc == AMDGPU::V_CVT_F32_FP8_e64_gfx12 ||
770 Opc == AMDGPU::V_CVT_F32_BF8_e64_dpp_gfx12 ||
771 Opc == AMDGPU::V_CVT_F32_FP8_e64_dpp_gfx12 ||
772 Opc == AMDGPU::V_CVT_F32_BF8_e64_dpp8_gfx12 ||
773 Opc == AMDGPU::V_CVT_F32_FP8_e64_dpp8_gfx12 ||
774 Opc == AMDGPU::V_CVT_PK_F32_BF8_fake16_e64_gfx12 ||
775 Opc == AMDGPU::V_CVT_PK_F32_FP8_fake16_e64_gfx12 ||
776 Opc == AMDGPU::V_CVT_PK_F32_BF8_t16_e64_gfx12 ||
777 Opc == AMDGPU::V_CVT_PK_F32_FP8_t16_e64_gfx12;
778}
779
780bool isGenericAtomic(unsigned Opc) {
781 return Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SWAP ||
782 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_ADD ||
783 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB ||
784 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMIN ||
785 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMIN ||
786 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMAX ||
787 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMAX ||
788 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_AND ||
789 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_OR ||
790 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_XOR ||
791 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_INC ||
792 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_DEC ||
793 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FADD ||
794 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMIN ||
795 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMAX ||
796 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_CMPSWAP ||
797 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB_CLAMP_U32 ||
798 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_COND_SUB_U32 ||
799 Opc == AMDGPU::G_AMDGPU_ATOMIC_CMPXCHG;
800}
801
802bool isAsyncStore(unsigned Opc) {
803 return Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B8_gfx1250 ||
804 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B32_gfx1250 ||
805 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B64_gfx1250 ||
806 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B128_gfx1250 ||
807 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B8_SADDR_gfx1250 ||
808 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B32_SADDR_gfx1250 ||
809 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B64_SADDR_gfx1250 ||
810 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B128_SADDR_gfx1250;
811}
812
813bool isTensorStore(unsigned Opc) {
814 return Opc == TENSOR_STORE_FROM_LDS_d2_gfx1250 ||
815 Opc == TENSOR_STORE_FROM_LDS_d4_gfx1250;
816}
817
818unsigned getTemporalHintType(const MCInstrDesc TID) {
819 if (SIInstrFlags::isAtomic(TID))
821 unsigned Opc = TID.getOpcode();
822 // Async and Tensor store should have the temporal hint type of TH_TYPE_STORE
823 if (TID.mayStore() &&
824 (isAsyncStore(Opc) || isTensorStore(Opc) || !TID.mayLoad()))
825 return CPol::TH_TYPE_STORE;
826
827 // This will default to returning TH_TYPE_LOAD when neither MayStore nor
828 // MayLoad flag is present which is the case with instructions like
829 // image_get_resinfo.
830 return CPol::TH_TYPE_LOAD;
831}
832
833bool isTrue16Inst(unsigned Opc) {
834 const VOPTrue16Info *Info = getTrue16OpcodeHelper(Opc);
835 return Info && Info->IsTrue16;
836}
837
839 const FP4FP8DstByteSelInfo *Info = getFP4FP8DstByteSelHelper(Opc);
840 if (!Info)
841 return FPType::None;
842 if (Info->HasFP8DstByteSel)
843 return FPType::FP8;
844 if (Info->HasFP4DstByteSel)
845 return FPType::FP4;
846
847 return FPType::None;
848}
849
850bool isDPMACCInstruction(unsigned Opc) {
851 const DPMACCInstructionInfo *Info = getDPMACCInstructionHelper(Opc);
852 return Info && Info->IsDPMACCInstruction;
853}
854
855unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc) {
856 const WMMAOpcodeMappingInfo *Info = getWMMAMappingInfoFrom2AddrOpcode(Opc);
857 return Info ? Info->Opcode3Addr : ~0u;
858}
859
860unsigned mapWMMA3AddrTo2AddrOpcode(unsigned Opc) {
861 const WMMAOpcodeMappingInfo *Info = getWMMAMappingInfoFrom3AddrOpcode(Opc);
862 return Info ? Info->Opcode2Addr : ~0u;
863}
864
865// Wrapper for Tablegen'd function. enum Subtarget is not defined in any
866// header files, so we need to wrap it in a function that takes unsigned
867// instead.
868int32_t getMCOpcode(uint32_t Opcode, unsigned Gen) {
869 return getMCOpcodeGen(Opcode, static_cast<Subtarget>(Gen));
870}
871
872unsigned getBitOp2(unsigned Opc) {
873 switch (Opc) {
874 default:
875 return 0;
876 case AMDGPU::V_AND_B32_e32:
877 return 0x40;
878 case AMDGPU::V_OR_B32_e32:
879 return 0x54;
880 case AMDGPU::V_XOR_B32_e32:
881 return 0x14;
882 case AMDGPU::V_XNOR_B32_e32:
883 return 0x41;
884 }
885}
886
887int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily,
888 bool VOPD3) {
889 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(OpY) : 0;
890 OpY = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : OpY;
891 const VOPDInfo *Info =
892 getVOPDInfoFromComponentOpcodes(OpX, OpY, EncodingFamily, VOPD3);
893 return Info ? Info->Opcode : -1;
894}
895
896std::pair<unsigned, unsigned> getVOPDComponents(unsigned VOPDOpcode) {
897 const VOPDInfo *Info = getVOPDOpcodeHelper(VOPDOpcode);
898 assert(Info);
899 const auto *OpX = getVOPDBaseFromComponent(Info->OpX);
900 const auto *OpY = getVOPDBaseFromComponent(Info->OpY);
901 assert(OpX && OpY);
902 return {OpX->BaseVOP, OpY->BaseVOP};
903}
904
905namespace VOPD {
906
907ComponentProps::ComponentProps(const MCInstrDesc &OpDesc, bool VOP3Layout) {
909
912 auto TiedIdx = OpDesc.getOperandConstraint(Component::SRC2, MCOI::TIED_TO);
913 assert(TiedIdx == -1 || TiedIdx == Component::DST);
914 HasSrc2Acc = TiedIdx != -1;
915 Opcode = OpDesc.getOpcode();
916
917 IsVOP3 = VOP3Layout || SIInstrFlags::isVOP3(OpDesc);
918 SrcOperandsNum = AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src2) ? 3
919 : AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::imm) ? 3
920 : AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src1) ? 2
921 : 1;
922 assert(SrcOperandsNum <= Component::MAX_SRC_NUM);
923
924 if (Opcode == AMDGPU::V_CNDMASK_B32_e32 ||
925 Opcode == AMDGPU::V_CNDMASK_B32_e64) {
926 // CNDMASK is an awkward exception, it has FP modifiers, but not FP
927 // operands.
928 NumVOPD3Mods = 2;
929 if (IsVOP3)
930 SrcOperandsNum = 3;
931 } else if (Opcode == AMDGPU::V_DOT2_F32_F16 ||
932 Opcode == AMDGPU::V_DOT2_F32_BF16) {
933 // VOP3P opcodes that have VOPD but don't have VOP2 version. Using VOPD3
934 // path in getIndexOfSrcInMCOperands to get correct src operand indexes,
935 // but generating VOPD, not VOPD3.
936 NumVOPD3Mods = SrcOperandsNum;
937 } else if (isSISrcFPOperand(OpDesc,
938 getNamedOperandIdx(Opcode, OpName::src0))) {
939 // All FP VOPD instructions have Neg modifiers for all operands except
940 // for tied src2.
941 NumVOPD3Mods = SrcOperandsNum;
942 if (HasSrc2Acc)
943 --NumVOPD3Mods;
944 }
945
946 if (SIInstrFlags::isVOP3(OpDesc))
947 return;
948
949 auto OperandsNum = OpDesc.getNumOperands();
950 unsigned CompOprIdx;
951 for (CompOprIdx = Component::SRC1; CompOprIdx < OperandsNum; ++CompOprIdx) {
952 if (OpDesc.operands()[CompOprIdx].OperandType == AMDGPU::OPERAND_KIMM32) {
953 MandatoryLiteralIdx = CompOprIdx;
954 break;
955 }
956 }
957}
958
960 return getNamedOperandIdx(Opcode, OpName::bitop3);
961}
962
963unsigned ComponentInfo::getIndexInParsedOperands(unsigned CompOprIdx) const {
964 assert(CompOprIdx < Component::MAX_OPR_NUM);
965
966 if (CompOprIdx == Component::DST)
968
969 auto CompSrcIdx = CompOprIdx - Component::DST_NUM;
970 if (CompSrcIdx < getCompParsedSrcOperandsNum())
971 return getIndexOfSrcInParsedOperands(CompSrcIdx);
972
973 // The specified operand does not exist.
974 return 0;
975}
976
978 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
979 const MCRegisterInfo &MRI, bool SkipSrc, bool AllowSameVGPR,
980 bool VOPD3) const {
981
982 auto OpXRegs = getRegIndices(ComponentIndex::X, GetRegIdx,
983 CompInfo[ComponentIndex::X].isVOP3());
984 auto OpYRegs = getRegIndices(ComponentIndex::Y, GetRegIdx,
985 CompInfo[ComponentIndex::Y].isVOP3());
986
987 const auto banksOverlap = [&MRI](MCRegister X, MCRegister Y,
988 unsigned BanksMask) -> bool {
989 MCRegister BaseX = MRI.getSubReg(X, AMDGPU::sub0);
990 MCRegister BaseY = MRI.getSubReg(Y, AMDGPU::sub0);
991 if (!BaseX)
992 BaseX = X;
993 if (!BaseY)
994 BaseY = Y;
995 if ((BaseX.id() & BanksMask) == (BaseY.id() & BanksMask))
996 return true;
997 if (BaseX != X /* This is 64-bit register */ &&
998 ((BaseX.id() + 1) & BanksMask) == (BaseY.id() & BanksMask))
999 return true;
1000 if (BaseY != Y &&
1001 (BaseX.id() & BanksMask) == ((BaseY.id() + 1) & BanksMask))
1002 return true;
1003
1004 // If both are 64-bit bank conflict will be detected yet while checking
1005 // the first subreg.
1006 return false;
1007 };
1008
1009 unsigned CompOprIdx;
1010 for (CompOprIdx = 0; CompOprIdx < Component::MAX_OPR_NUM; ++CompOprIdx) {
1011 unsigned BanksMasks = VOPD3 ? VOPD3_VGPR_BANK_MASKS[CompOprIdx]
1012 : VOPD_VGPR_BANK_MASKS[CompOprIdx];
1013 if (!OpXRegs[CompOprIdx] || !OpYRegs[CompOprIdx])
1014 continue;
1015
1016 if (getVGPREncodingMSBs(OpXRegs[CompOprIdx], MRI) !=
1017 getVGPREncodingMSBs(OpYRegs[CompOprIdx], MRI))
1018 return CompOprIdx;
1019
1020 if (SkipSrc && CompOprIdx >= Component::DST_NUM)
1021 continue;
1022
1023 if (CompOprIdx < Component::DST_NUM) {
1024 // Even if we do not check vdst parity, vdst operands still shall not
1025 // overlap.
1026 if (MRI.regsOverlap(OpXRegs[CompOprIdx], OpYRegs[CompOprIdx]))
1027 return CompOprIdx;
1028 if (VOPD3) // No need to check dst parity.
1029 continue;
1030 }
1031
1032 if (banksOverlap(OpXRegs[CompOprIdx], OpYRegs[CompOprIdx], BanksMasks) &&
1033 (!AllowSameVGPR || CompOprIdx < Component::DST_NUM ||
1034 OpXRegs[CompOprIdx] != OpYRegs[CompOprIdx]))
1035 return CompOprIdx;
1036 }
1037
1038 return {};
1039}
1040
1041// Return an array of VGPR registers [DST,SRC0,SRC1,SRC2] used
1042// by the specified component. If an operand is unused
1043// or is not a VGPR, the corresponding value is 0.
1044//
1045// GetRegIdx(Component, MCOperandIdx) must return a VGPR register index
1046// for the specified component and MC operand. The callback must return 0
1047// if the operand is not a register or not a VGPR.
1049InstInfo::getRegIndices(unsigned CompIdx,
1050 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
1051 bool VOPD3) const {
1052 assert(CompIdx < COMPONENTS_NUM);
1053
1054 const auto &Comp = CompInfo[CompIdx];
1056
1057 RegIndices[DST] = GetRegIdx(CompIdx, Comp.getIndexOfDstInMCOperands());
1058
1059 for (unsigned CompOprIdx : {SRC0, SRC1, SRC2}) {
1060 unsigned CompSrcIdx = CompOprIdx - DST_NUM;
1061 RegIndices[CompOprIdx] =
1062 Comp.hasRegSrcOperand(CompSrcIdx)
1063 ? GetRegIdx(CompIdx,
1064 Comp.getIndexOfSrcInMCOperands(CompSrcIdx, VOPD3))
1065 : MCRegister();
1066 }
1067 return RegIndices;
1068}
1069
1070} // namespace VOPD
1071
1073 return VOPD::InstInfo(OpX, OpY);
1074}
1075
1077 const MCInstrInfo *InstrInfo) {
1078 auto [OpX, OpY] = getVOPDComponents(VOPDOpcode);
1079 const auto &OpXDesc = InstrInfo->get(OpX);
1080 const auto &OpYDesc = InstrInfo->get(OpY);
1081 bool VOPD3 = SIInstrFlags::isVOPD3(*InstrInfo, VOPDOpcode);
1083 VOPD::ComponentInfo OpYInfo(OpYDesc, OpXInfo, VOPD3);
1084 return VOPD::InstInfo(OpXInfo, OpYInfo);
1085}
1086
1088 StringRef FeatureString) {
1090 STI.getFeatureBits().test(FeatureXNACKOnOffModes)
1091 ? TargetIDSetting::Any
1092 : TargetIDSetting::Unsupported,
1093 STI.getFeatureBits().test(FeatureSupportsSRAMECC)
1094 ? TargetIDSetting::Any
1095 : TargetIDSetting::Unsupported);
1096
1097 // Check if xnack or sramecc is explicitly enabled or disabled. In the
1098 // absence of the target features we assume we must generate code that can run
1099 // in any environment.
1100 SubtargetFeatures Features(FeatureString);
1101 std::optional<bool> XnackRequested;
1102 std::optional<bool> SramEccRequested;
1103
1104 for (const std::string &Feature : Features.getFeatures()) {
1105 if (Feature == "+xnack")
1106 XnackRequested = true;
1107 else if (Feature == "-xnack")
1108 XnackRequested = false;
1109 else if (Feature == "+sramecc")
1110 SramEccRequested = true;
1111 else if (Feature == "-sramecc")
1112 SramEccRequested = false;
1113 }
1114
1115 // Only allow changing xnack setting if the target supports on/off modes.
1116 // Targets without on/off mode support keep their initial setting
1117 // (Unsupported).
1118
1119 bool XnackSupported = STI.getFeatureBits().test(FeatureXNACKOnOffModes);
1120 bool SramEccSupported = TargetID.isSramEccSupported();
1121
1122 if (XnackRequested) {
1123 if (XnackSupported) {
1124 TargetID.setXnackSetting(*XnackRequested ? TargetIDSetting::On
1125 : TargetIDSetting::Off);
1126 } else {
1127 // If a specific xnack setting was requested and this GPU does not support
1128 // xnack emit a warning. Setting will remain set to "Unsupported".
1129 if (*XnackRequested) {
1130 errs() << "warning: xnack 'On' was requested for a processor that does "
1131 "not support it!\n";
1132 } else {
1133 errs() << "warning: xnack 'Off' was requested for a processor that "
1134 "does not support it!\n";
1135 }
1136 }
1137 }
1138
1139 if (SramEccRequested) {
1140 if (SramEccSupported) {
1141 TargetID.setSramEccSetting(*SramEccRequested ? TargetIDSetting::On
1142 : TargetIDSetting::Off);
1143 } else {
1144 // If a specific sramecc setting was requested and this GPU does not
1145 // support sramecc emit a warning. Setting will remain set to
1146 // "Unsupported".
1147 if (*SramEccRequested) {
1148 errs() << "warning: sramecc 'On' was requested for a processor that "
1149 "does not support it!\n";
1150 } else {
1151 errs() << "warning: sramecc 'Off' was requested for a processor that "
1152 "does not support it!\n";
1153 }
1154 }
1155 }
1156
1157 return TargetID;
1158}
1159
1160namespace IsaInfo {
1161
1163 if (STI.getFeatureBits().test(FeatureInstCacheLineSize128))
1164 return 128;
1165 if (STI.getFeatureBits().test(FeatureInstCacheLineSize64))
1166 return 64;
1167 return 64;
1168}
1169
1170unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
1171 if (STI.getFeatureBits().test(FeatureWavefrontSize16))
1172 return 16;
1173 if (STI.getFeatureBits().test(FeatureWavefrontSize32))
1174 return 32;
1175
1176 return 64;
1177}
1178
1179// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
1180// not depend on how many SIMDs a work-group runs on.
1182 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
1183 return 32768;
1184 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
1185 return 65536;
1186 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
1187 return 163840;
1188 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
1189 return 196608;
1190 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
1191 return 327680;
1192 return 32768;
1193}
1194
1195// Total physical size of LDS on the block, in bytes. On targets with
1196// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
1197// addressable size (gfx10/11/12, 128k physical and 64k addressable). On other
1198// targets it is equal to the addressable size.
1199static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
1200 unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
1201 if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
1202 return 2 * Addressable;
1203 return Addressable;
1204}
1205
1206// Sizes in use, by generation (addressable / physical block):
1207// gfx6 : 32 KiB
1208// gfx7 / gfx8 / gfx9: 64 KiB
1209// gfx9.5 (gfx950) : 160 KiB
1210// gfx10 / 11 / 12 : 64 KiB addressable, 128 KiB physical block
1211// gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
1212// gfx13 : 192 KiB on four SIMDs, 96 KiB on two
1213// Total available in the current mode. The physical size is halved when a
1214// work-group runs on two SIMDs.
1216 unsigned Size = getPhysicalLocalMemorySize(STI);
1217 if (!isFullSIMDMode(STI))
1218 Size /= 2;
1219 return Size;
1220}
1221
1222// What one work-group can allocate in the current mode. This is the HW
1223// addressable cap, but never more than the total available in the current mode.
1225 return std::min(getMaxHWAddressableLocalMemorySize(STI),
1226 getLocalMemorySize(STI));
1227}
1228
1230 unsigned FlatWorkGroupSize) {
1231 assert(FlatWorkGroupSize != 0);
1232 if (!STI.getTargetTriple().isAMDGCN())
1233 return 8;
1234 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1235 unsigned MaxWaves =
1237 unsigned N = getWavesPerWorkGroup(STI, FlatWorkGroupSize);
1238 if (N == 1) {
1239 // Single-wave workgroups don't consume barrier resources.
1240 return MaxWaves;
1241 }
1242
1243 unsigned MaxBarriers = 16;
1244 if (isGFX10Plus(STI) && !STI.getFeatureBits().test(FeatureCuMode))
1245 MaxBarriers = 32;
1246
1247 return std::min(MaxWaves / N, MaxBarriers);
1248}
1249
1251 unsigned FlatWorkGroupSize) {
1252 return divideCeil(getWavesPerWorkGroup(STI, FlatWorkGroupSize),
1254}
1255
1256unsigned getMinFlatWorkGroupSize(const MCSubtargetInfo &STI) { return 1; }
1257
1259 unsigned FlatWorkGroupSize) {
1260 return divideCeil(FlatWorkGroupSize, getWavefrontSize(STI));
1261}
1262
1263unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI) { return 8; }
1264
1265// Per-wave SGPRs reserved for the trap handler when enabled.
1266static unsigned getSGPRTrapHandlerReserve(const MCSubtargetInfo &STI) {
1267 return STI.getFeatureBits().test(FeatureTrapHandler) ? TRAP_NUM_SGPRS : 0;
1268}
1269
1270// Per-wave SGPR budget (before the addressable clamp): take off the trap
1271// reserve, round down to \p Granule. Shared by getMinNumSGPRs() and
1272// getMaxNumSGPRs(); getOccupancyWithNumSGPRs() is the closed-form algebraic
1273// inverse of this same budget (it does not call this helper), so the two encode
1274// one model.
1275static unsigned getSGPRBudgetPerWave(unsigned TotalNumSGPRs,
1276 unsigned WavesPerEU, unsigned TrapReserve,
1277 unsigned Granule) {
1278 assert(WavesPerEU != 0 && Granule != 0);
1279 unsigned Budget = TotalNumSGPRs / WavesPerEU;
1280 Budget -= std::min(Budget, TrapReserve);
1281 return alignDown(Budget, Granule);
1282}
1283
1284unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU) {
1285 assert(WavesPerEU != 0);
1286
1288 if (Version.Major >= 10)
1289 return 0;
1290
1291 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1292 if (WavesPerEU >= getMaxWavesPerEU(Kind))
1293 return 0;
1294
1295 unsigned MinNumSGPRs =
1296 getSGPRBudgetPerWave(getTotalNumSGPRs(Kind), WavesPerEU + 1,
1298 getSGPRAllocGranule(Kind)) +
1299 1;
1300 return std::min(MinNumSGPRs, getAddressableNumSGPRs(Kind));
1301}
1302
1303unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1304 bool Addressable) {
1305 assert(WavesPerEU != 0);
1306
1307 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1308 unsigned AddressableNumSGPRs = getAddressableNumSGPRs(Kind);
1310 if (Version.Major >= 10)
1311 return Addressable ? AddressableNumSGPRs : 108;
1312 if (Version.Major >= 8 && !Addressable)
1313 AddressableNumSGPRs = 112;
1314 unsigned MaxNumSGPRs = getSGPRBudgetPerWave(
1315 getTotalNumSGPRs(Kind), WavesPerEU, getSGPRTrapHandlerReserve(STI),
1316 getSGPRAllocGranule(Kind));
1317 return std::min(MaxNumSGPRs, AddressableNumSGPRs);
1318}
1319
1321 // From GFX10 on the SGPR file is large enough that SGPRs never limit
1322 // occupancy. Kept as one capability so callers don't each test the version.
1323 return getIsaVersion(STI.getCPU()).Major < 10;
1324}
1325
1326unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
1327 bool FlatScrUsed, bool XNACKUsed) {
1328 unsigned ExtraSGPRs = 0;
1329 if (VCCUsed)
1330 ExtraSGPRs = 2;
1331
1333 if (Version.Major >= 10)
1334 return ExtraSGPRs;
1335
1336 if (Version.Major < 8) {
1337 if (FlatScrUsed)
1338 ExtraSGPRs = 4;
1339 } else {
1340 if (XNACKUsed)
1341 ExtraSGPRs = 4;
1342
1343 if (FlatScrUsed ||
1344 STI.getFeatureBits().test(AMDGPU::FeatureArchitectedFlatScratch))
1345 ExtraSGPRs = 6;
1346 }
1347
1348 return ExtraSGPRs;
1349}
1350
1351unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
1352 bool FlatScrUsed) {
1353 return getNumExtraSGPRs(STI, VCCUsed, FlatScrUsed,
1354 STI.getFeatureBits().test(AMDGPU::FeatureXNACK));
1355}
1356
1357static unsigned getGranulatedNumRegisterBlocks(unsigned NumRegs,
1358 unsigned Granule) {
1359 return divideCeil(std::max(1u, NumRegs), Granule);
1360}
1361
1362unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
1363 // SGPRBlocks is actual number of SGPR blocks minus 1.
1365 1;
1366}
1367
1369 unsigned DynamicVGPRBlockSize,
1370 std::optional<bool> EnableWavefrontSize32) {
1371 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1372 return 8;
1373
1374 if (DynamicVGPRBlockSize != 0)
1375 return DynamicVGPRBlockSize;
1376
1377 bool IsWave32 = EnableWavefrontSize32
1378 ? *EnableWavefrontSize32
1379 : STI.getFeatureBits().test(FeatureWavefrontSize32);
1380
1381 if (STI.getFeatureBits().test(Feature1536VGPRs))
1382 return IsWave32 ? 24 : 12;
1383
1384 if (hasGFX10_3Insts(STI))
1385 return IsWave32 ? 16 : 8;
1386
1387 return IsWave32 ? 8 : 4;
1388}
1389
1391 std::optional<bool> EnableWavefrontSize32) {
1392 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1393 return 8;
1394
1395 bool IsWave32 = EnableWavefrontSize32
1396 ? *EnableWavefrontSize32
1397 : STI.getFeatureBits().test(FeatureWavefrontSize32);
1398
1399 if (STI.getFeatureBits().test(Feature1024AddressableVGPRs))
1400 return IsWave32 ? 16 : 8;
1401
1402 return IsWave32 ? 8 : 4;
1403}
1404
1405unsigned getArchVGPRAllocGranule() { return 4; }
1406
1407unsigned getTotalNumVGPRs(const MCSubtargetInfo &STI) {
1408 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1409 return 512;
1410 if (!isGFX10Plus(STI))
1411 return 256;
1412 bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
1413 if (STI.getFeatureBits().test(Feature1536VGPRs))
1414 return IsWave32 ? 1536 : 768;
1415 return IsWave32 ? 1024 : 512;
1416}
1417
1419 const auto &Features = STI.getFeatureBits();
1420 if (Features.test(Feature1024AddressableVGPRs))
1421 return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
1422 return 256;
1423}
1424
1426 unsigned DynamicVGPRBlockSize) {
1427 const auto &Features = STI.getFeatureBits();
1428 if (Features.test(FeatureGFX90AInsts))
1429 return 512;
1430
1431 if (DynamicVGPRBlockSize != 0) {
1432 // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
1433 return MaxDynamicVGPRBlocks *
1434 getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
1435 }
1436 return getAddressableNumArchVGPRs(STI);
1437}
1438
1440 unsigned NumVGPRs,
1441 unsigned DynamicVGPRBlockSize) {
1443 NumVGPRs, getVGPRAllocGranule(STI, DynamicVGPRBlockSize),
1445}
1446
1447unsigned getNumWavesPerEUWithNumVGPRs(unsigned NumVGPRs, unsigned Granule,
1448 unsigned MaxWaves,
1449 unsigned TotalNumVGPRs) {
1450 if (NumVGPRs < Granule)
1451 return MaxWaves;
1452 unsigned RoundedRegs = alignTo(NumVGPRs, Granule);
1453 return std::min(std::max(TotalNumVGPRs / RoundedRegs, 1u), MaxWaves);
1454}
1455
1456unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves,
1457 unsigned TotalNumSGPRs, unsigned Granule,
1458 unsigned TrapReserve) {
1459 // Closed-form inverse of getMaxNumSGPRs(): the budget condition
1460 // SGPRs <= alignDown(TotalNumSGPRs / W - TrapReserve, Granule)
1461 // solves to W <= TotalNumSGPRs / (alignTo(SGPRs, Granule) + TrapReserve).
1462 unsigned PerWave = alignTo(SGPRs, Granule) + TrapReserve;
1463 return PerWave ? std::clamp(TotalNumSGPRs / PerWave, 1u, MaxWaves) : MaxWaves;
1464}
1465
1466unsigned getOccupancyWithNumSGPRs(const MCSubtargetInfo &STI, unsigned SGPRs) {
1467 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1468 unsigned MaxWaves = getMaxWavesPerEU(Kind);
1469
1470 if (!isSGPROccupancyLimited(STI))
1471 return MaxWaves;
1472
1473 return getOccupancyWithNumSGPRs(SGPRs, MaxWaves, getTotalNumSGPRs(Kind),
1474 getSGPRAllocGranule(Kind),
1476}
1477
1478unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1479 unsigned DynamicVGPRBlockSize) {
1480 assert(WavesPerEU != 0);
1481
1482 // In dynamic VGPR mode, (static) occupancy does not depend on VGPR usage,
1483 // so getMaxNumVGPRs does not depend on WavesPerEU, and thus we need to return
1484 // zero because there is no nonzero VGPR usage N where going below N
1485 // achieves higher (static) occupancy.
1486 bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
1487 if (DynamicVGPREnabled)
1488 return 0;
1489
1490 unsigned MaxWavesPerEU = getMaxWavesPerEU(parseArchAMDGCN(STI.getCPU()));
1491 if (WavesPerEU >= MaxWavesPerEU)
1492 return 0;
1493
1494 unsigned TotNumVGPRs = getTotalNumVGPRs(STI);
1495 unsigned AddrsableNumVGPRs =
1496 getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
1497 unsigned Granule = getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
1498 unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
1499
1500 if (MaxNumVGPRs == alignDown(TotNumVGPRs / MaxWavesPerEU, Granule))
1501 return 0;
1502
1503 unsigned MinWavesPerEU = getNumWavesPerEUWithNumVGPRs(STI, AddrsableNumVGPRs,
1504 DynamicVGPRBlockSize);
1505 if (WavesPerEU < MinWavesPerEU)
1506 return getMinNumVGPRs(STI, MinWavesPerEU, DynamicVGPRBlockSize);
1507
1508 unsigned MaxNumVGPRsNext = alignDown(TotNumVGPRs / (WavesPerEU + 1), Granule);
1509 unsigned MinNumVGPRs = 1 + std::min(MaxNumVGPRs - Granule, MaxNumVGPRsNext);
1510 return std::min(MinNumVGPRs, AddrsableNumVGPRs);
1511}
1512
1513unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1514 unsigned DynamicVGPRBlockSize) {
1515 assert(WavesPerEU != 0);
1516
1517 // In dynamic VGPR mode, WavesPerEU does not imply a VGPR limit.
1518 bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
1519 unsigned MaxNumVGPRs =
1520 DynamicVGPREnabled
1521 ? getTotalNumVGPRs(STI)
1522 : alignDown(getTotalNumVGPRs(STI) / WavesPerEU,
1523 getVGPRAllocGranule(STI, DynamicVGPRBlockSize));
1524 unsigned AddressableNumVGPRs =
1525 getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
1526 return std::min(MaxNumVGPRs, AddressableNumVGPRs);
1527}
1528
1529unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs,
1530 std::optional<bool> EnableWavefrontSize32) {
1532 NumVGPRs, getVGPREncodingGranule(STI, EnableWavefrontSize32)) -
1533 1;
1534}
1535
1537 unsigned NumVGPRs,
1538 unsigned DynamicVGPRBlockSize,
1539 std::optional<bool> EnableWavefrontSize32) {
1541 NumVGPRs,
1542 getVGPRAllocGranule(STI, DynamicVGPRBlockSize, EnableWavefrontSize32));
1543}
1544} // end namespace IsaInfo
1545
1547 const MCSubtargetInfo &STI) {
1549 KernelCode.amd_kernel_code_version_major = 1;
1550 KernelCode.amd_kernel_code_version_minor = 2;
1551 KernelCode.amd_machine_kind = 1; // AMD_MACHINE_KIND_AMDGPU
1552 KernelCode.amd_machine_version_major = Version.Major;
1553 KernelCode.amd_machine_version_minor = Version.Minor;
1554 KernelCode.amd_machine_version_stepping = Version.Stepping;
1556 if (STI.getFeatureBits().test(FeatureWavefrontSize32)) {
1557 KernelCode.wavefront_size = 5;
1559 } else {
1560 KernelCode.wavefront_size = 6;
1561 }
1562
1563 // If the code object does not support indirect functions, then the value must
1564 // be 0xffffffff.
1565 KernelCode.call_convention = -1;
1566
1567 // These alignment values are specified in powers of two, so alignment =
1568 // 2^n. The minimum alignment is 2^4 = 16.
1569 KernelCode.kernarg_segment_alignment = 4;
1570 KernelCode.group_segment_alignment = 4;
1571 KernelCode.private_segment_alignment = 4;
1572
1573 if (Version.Major >= 10) {
1574 KernelCode.compute_pgm_resource_registers |=
1575 S_00B848_WGP_MODE(STI.getFeatureBits().test(FeatureCuMode) ? 0 : 1) |
1577 }
1578}
1579
1582}
1583
1586}
1587
1589 unsigned AS = GV->getAddressSpace();
1590 return AS == AMDGPUAS::CONSTANT_ADDRESS ||
1592}
1593
1595 return TT.getArch() == Triple::r600;
1596}
1597
1598static bool isValidRegPrefix(char C) {
1599 return C == 'v' || C == 's' || C == 'a';
1600}
1601
1602std::tuple<char, unsigned, unsigned> parseAsmPhysRegName(StringRef RegName) {
1603 char Kind = RegName.front();
1604 if (!isValidRegPrefix(Kind))
1605 return {};
1606
1607 RegName = RegName.drop_front();
1608 if (RegName.consume_front("[")) {
1609 unsigned Idx, End;
1610 bool Failed = RegName.consumeInteger(10, Idx);
1611 Failed |= !RegName.consume_front(":");
1612 Failed |= RegName.consumeInteger(10, End);
1613 Failed |= !RegName.consume_back("]");
1614 if (!Failed) {
1615 unsigned NumRegs = End - Idx + 1;
1616 if (NumRegs > 1)
1617 return {Kind, Idx, NumRegs};
1618 }
1619 } else {
1620 unsigned Idx;
1621 bool Failed = RegName.getAsInteger(10, Idx);
1622 if (!Failed)
1623 return {Kind, Idx, 1};
1624 }
1625
1626 return {};
1627}
1628
1629std::tuple<char, unsigned, unsigned>
1631 StringRef RegName = Constraint;
1632 if (!RegName.consume_front("{") || !RegName.consume_back("}"))
1633 return {};
1635}
1636
1637std::pair<unsigned, unsigned>
1639 std::pair<unsigned, unsigned> Default,
1640 bool OnlyFirstRequired) {
1641 if (auto Attr = getIntegerPairAttribute(F, Name, OnlyFirstRequired))
1642 return {Attr->first, Attr->second.value_or(Default.second)};
1643 return Default;
1644}
1645
1646std::optional<std::pair<unsigned, std::optional<unsigned>>>
1648 bool OnlyFirstRequired) {
1649 Attribute A = F.getFnAttribute(Name);
1650 if (!A.isStringAttribute())
1651 return std::nullopt;
1652
1653 LLVMContext &Ctx = F.getContext();
1654 std::pair<unsigned, std::optional<unsigned>> Ints;
1655 std::pair<StringRef, StringRef> Strs = A.getValueAsString().split(',');
1656 if (Strs.first.trim().getAsInteger(0, Ints.first)) {
1657 Ctx.emitError("can't parse first integer attribute " + Name);
1658 return std::nullopt;
1659 }
1660 unsigned Second = 0;
1661 if (Strs.second.trim().getAsInteger(0, Second)) {
1662 if (!OnlyFirstRequired || !Strs.second.trim().empty()) {
1663 Ctx.emitError("can't parse second integer attribute " + Name);
1664 return std::nullopt;
1665 }
1666 } else {
1667 Ints.second = Second;
1668 }
1669
1670 return Ints;
1671}
1672
1674 unsigned Size,
1675 unsigned DefaultVal) {
1676 std::optional<SmallVector<unsigned>> R =
1678 return R.has_value() ? *R : SmallVector<unsigned>(Size, DefaultVal);
1679}
1680
1681std::optional<SmallVector<unsigned>>
1683 assert(Size > 2);
1684 LLVMContext &Ctx = F.getContext();
1685
1686 Attribute A = F.getFnAttribute(Name);
1687 if (!A.isValid())
1688 return std::nullopt;
1689 if (!A.isStringAttribute()) {
1690 Ctx.emitError(Name + " is not a string attribute");
1691 return std::nullopt;
1692 }
1693
1695
1696 StringRef S = A.getValueAsString();
1697 unsigned i = 0;
1698 for (; !S.empty() && i < Size; i++) {
1699 std::pair<StringRef, StringRef> Strs = S.split(',');
1700 unsigned IntVal;
1701 if (Strs.first.trim().getAsInteger(0, IntVal)) {
1702 Ctx.emitError("can't parse integer attribute " + Strs.first + " in " +
1703 Name);
1704 return std::nullopt;
1705 }
1706 Vals[i] = IntVal;
1707 S = Strs.second;
1708 }
1709
1710 if (!S.empty() || i < Size) {
1711 Ctx.emitError("attribute " + Name +
1712 " has incorrect number of integers; expected " +
1714 return std::nullopt;
1715 }
1716 return Vals;
1717}
1718
1720 return getIntegerVecAttribute(F, "amdgpu-max-num-workgroups", 3,
1721 std::numeric_limits<uint32_t>::max());
1722}
1723
1724bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val) {
1725 assert((MD.getNumOperands() % 2 == 0) && "invalid number of operands!");
1726 for (unsigned I = 0, E = MD.getNumOperands() / 2; I != E; ++I) {
1727 auto Low =
1728 mdconst::extract<ConstantInt>(MD.getOperand(2 * I + 0))->getValue();
1729 auto High =
1730 mdconst::extract<ConstantInt>(MD.getOperand(2 * I + 1))->getValue();
1731 // There are two types of [A; B) ranges:
1732 // A < B, e.g. [4; 5) which is a range that only includes 4.
1733 // A > B, e.g. [5; 4) which is a range that wraps around and includes
1734 // everything except 4.
1735 if (Low.ult(High)) {
1736 if (Low.ule(Val) && High.ugt(Val))
1737 return true;
1738 } else {
1739 if (Low.uge(Val) && High.ult(Val))
1740 return true;
1741 }
1742 }
1743
1744 return false;
1745}
1746
1748 return (1 << (getVmcntBitWidthLo(Version.Major) +
1749 getVmcntBitWidthHi(Version.Major))) -
1750 1;
1751}
1752
1754 return (1 << getLoadcntBitWidth(Version.Major)) - 1;
1755}
1756
1758 return (1 << getSamplecntBitWidth(Version.Major)) - 1;
1759}
1760
1762 return (1 << getBvhcntBitWidth(Version.Major)) - 1;
1763}
1764
1766 return (1 << getExpcntBitWidth(Version.Major)) - 1;
1767}
1768
1770 return (1 << getLgkmcntBitWidth(Version.Major)) - 1;
1771}
1772
1774 return (1 << getDscntBitWidth(Version.Major)) - 1;
1775}
1776
1778 return (1 << getKmcntBitWidth(Version.Major)) - 1;
1779}
1780
1782 return (1 << getXcntBitWidth(Version.Major, Version.Minor)) - 1;
1783}
1784
1786 return (1 << getAsynccntBitWidth(Version.Major, Version.Minor)) - 1;
1787}
1788
1790 return (1 << getStorecntBitWidth(Version.Major)) - 1;
1791}
1792
1794 unsigned VmcntLo = getBitMask(getVmcntBitShiftLo(Version.Major),
1795 getVmcntBitWidthLo(Version.Major));
1796 unsigned Expcnt = getBitMask(getExpcntBitShift(Version.Major),
1797 getExpcntBitWidth(Version.Major));
1798 unsigned Lgkmcnt = getBitMask(getLgkmcntBitShift(Version.Major),
1799 getLgkmcntBitWidth(Version.Major));
1800 unsigned VmcntHi = getBitMask(getVmcntBitShiftHi(Version.Major),
1801 getVmcntBitWidthHi(Version.Major));
1802 return VmcntLo | Expcnt | Lgkmcnt | VmcntHi;
1803}
1804
1805unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt) {
1806 unsigned VmcntLo = unpackBits(Waitcnt, getVmcntBitShiftLo(Version.Major),
1807 getVmcntBitWidthLo(Version.Major));
1808 unsigned VmcntHi = unpackBits(Waitcnt, getVmcntBitShiftHi(Version.Major),
1809 getVmcntBitWidthHi(Version.Major));
1810 return VmcntLo | VmcntHi << getVmcntBitWidthLo(Version.Major);
1811}
1812
1813unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt) {
1814 return unpackBits(Waitcnt, getExpcntBitShift(Version.Major),
1815 getExpcntBitWidth(Version.Major));
1816}
1817
1818unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt) {
1819 return unpackBits(Waitcnt, getLgkmcntBitShift(Version.Major),
1820 getLgkmcntBitWidth(Version.Major));
1821}
1822
1823unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt) {
1824 return unpackBits(Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1825 getLoadcntBitWidth(Version.Major));
1826}
1827
1828unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt) {
1829 return unpackBits(Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1830 getStorecntBitWidth(Version.Major));
1831}
1832
1833unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt) {
1834 return unpackBits(Waitcnt, getDscntBitShift(Version.Major),
1835 getDscntBitWidth(Version.Major));
1836}
1837
1838void decodeWaitcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned &Vmcnt,
1839 unsigned &Expcnt, unsigned &Lgkmcnt) {
1840 Vmcnt = decodeVmcnt(Version, Waitcnt);
1841 Expcnt = decodeExpcnt(Version, Waitcnt);
1842 Lgkmcnt = decodeLgkmcnt(Version, Waitcnt);
1843}
1844
1845unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt,
1846 unsigned Vmcnt) {
1847 Waitcnt = packBits(Vmcnt, Waitcnt, getVmcntBitShiftLo(Version.Major),
1848 getVmcntBitWidthLo(Version.Major));
1849 return packBits(Vmcnt >> getVmcntBitWidthLo(Version.Major), Waitcnt,
1850 getVmcntBitShiftHi(Version.Major),
1851 getVmcntBitWidthHi(Version.Major));
1852}
1853
1854unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt,
1855 unsigned Expcnt) {
1856 return packBits(Expcnt, Waitcnt, getExpcntBitShift(Version.Major),
1857 getExpcntBitWidth(Version.Major));
1858}
1859
1860unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt,
1861 unsigned Lgkmcnt) {
1862 return packBits(Lgkmcnt, Waitcnt, getLgkmcntBitShift(Version.Major),
1863 getLgkmcntBitWidth(Version.Major));
1864}
1865
1866unsigned encodeWaitcnt(const IsaVersion &Version, unsigned Vmcnt,
1867 unsigned Expcnt, unsigned Lgkmcnt) {
1868 unsigned Waitcnt = getWaitcntBitMask(Version);
1870 Waitcnt = encodeExpcnt(Version, Waitcnt, Expcnt);
1871 Waitcnt = encodeLgkmcnt(Version, Waitcnt, Lgkmcnt);
1872 return Waitcnt;
1873}
1874
1876 bool IsStore) {
1877 unsigned Dscnt = getBitMask(getDscntBitShift(Version.Major),
1878 getDscntBitWidth(Version.Major));
1879 if (IsStore) {
1880 unsigned Storecnt = getBitMask(getLoadcntStorecntBitShift(Version.Major),
1881 getStorecntBitWidth(Version.Major));
1882 return Dscnt | Storecnt;
1883 }
1884 unsigned Loadcnt = getBitMask(getLoadcntStorecntBitShift(Version.Major),
1885 getLoadcntBitWidth(Version.Major));
1886 return Dscnt | Loadcnt;
1887}
1888
1889static unsigned encodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt,
1890 unsigned Loadcnt) {
1891 return packBits(Loadcnt, Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1892 getLoadcntBitWidth(Version.Major));
1893}
1894
1895static unsigned encodeStorecnt(const IsaVersion &Version, unsigned Waitcnt,
1896 unsigned Storecnt) {
1897 return packBits(Storecnt, Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1898 getStorecntBitWidth(Version.Major));
1899}
1900
1901static unsigned encodeDscnt(const IsaVersion &Version, unsigned Waitcnt,
1902 unsigned Dscnt) {
1903 return packBits(Dscnt, Waitcnt, getDscntBitShift(Version.Major),
1904 getDscntBitWidth(Version.Major));
1905}
1906
1907unsigned encodeLoadcntDscnt(const IsaVersion &Version, unsigned Loadcnt,
1908 unsigned Dscnt) {
1909 unsigned Waitcnt = getCombinedCountBitMask(Version, false);
1910 Waitcnt = encodeLoadcnt(Version, Waitcnt, Loadcnt);
1912 return Waitcnt;
1913}
1914
1915unsigned encodeStorecntDscnt(const IsaVersion &Version, unsigned Storecnt,
1916 unsigned Dscnt) {
1917 unsigned Waitcnt = getCombinedCountBitMask(Version, true);
1918 Waitcnt = encodeStorecnt(Version, Waitcnt, Storecnt);
1920 return Waitcnt;
1921}
1922
1923//===----------------------------------------------------------------------===//
1924// Custom Operand Values
1925//===----------------------------------------------------------------------===//
1926
1928 int Size,
1929 const MCSubtargetInfo &STI) {
1930 unsigned Enc = 0;
1931 for (int Idx = 0; Idx < Size; ++Idx) {
1932 const auto &Op = Opr[Idx];
1933 if (Op.isSupported(STI))
1934 Enc |= Op.encode(Op.Default);
1935 }
1936 return Enc;
1937}
1938
1940 int Size, unsigned Code,
1941 bool &HasNonDefaultVal,
1942 const MCSubtargetInfo &STI) {
1943 unsigned UsedOprMask = 0;
1944 HasNonDefaultVal = false;
1945 for (int Idx = 0; Idx < Size; ++Idx) {
1946 const auto &Op = Opr[Idx];
1947 if (!Op.isSupported(STI))
1948 continue;
1949 UsedOprMask |= Op.getMask();
1950 unsigned Val = Op.decode(Code);
1951 if (!Op.isValid(Val))
1952 return false;
1953 HasNonDefaultVal |= (Val != Op.Default);
1954 }
1955 return (Code & ~UsedOprMask) == 0;
1956}
1957
1958static bool decodeCustomOperand(const CustomOperandVal *Opr, int Size,
1959 unsigned Code, int &Idx, StringRef &Name,
1960 unsigned &Val, bool &IsDefault,
1961 const MCSubtargetInfo &STI) {
1962 while (Idx < Size) {
1963 const auto &Op = Opr[Idx++];
1964 if (Op.isSupported(STI)) {
1965 Name = Op.Name;
1966 Val = Op.decode(Code);
1967 IsDefault = (Val == Op.Default);
1968 return true;
1969 }
1970 }
1971
1972 return false;
1973}
1974
1976 int64_t InputVal) {
1977 if (InputVal < 0 || InputVal > Op.Max)
1978 return OPR_VAL_INVALID;
1979 return Op.encode(InputVal);
1980}
1981
1982static int encodeCustomOperand(const CustomOperandVal *Opr, int Size,
1983 const StringRef Name, int64_t InputVal,
1984 unsigned &UsedOprMask,
1985 const MCSubtargetInfo &STI) {
1986 int InvalidId = OPR_ID_UNKNOWN;
1987 for (int Idx = 0; Idx < Size; ++Idx) {
1988 const auto &Op = Opr[Idx];
1989 if (Op.Name == Name) {
1990 if (!Op.isSupported(STI)) {
1991 InvalidId = OPR_ID_UNSUPPORTED;
1992 continue;
1993 }
1994 auto OprMask = Op.getMask();
1995 if (OprMask & UsedOprMask)
1996 return OPR_ID_DUPLICATE;
1997 UsedOprMask |= OprMask;
1998 return encodeCustomOperandVal(Op, InputVal);
1999 }
2000 }
2001 return InvalidId;
2002}
2003
2004//===----------------------------------------------------------------------===//
2005// DepCtr
2006//===----------------------------------------------------------------------===//
2007
2008namespace DepCtr {
2009
2011 static int Default = -1;
2012 if (Default == -1)
2014 return Default;
2015}
2016
2017bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal,
2018 const MCSubtargetInfo &STI) {
2020 HasNonDefaultVal, STI);
2021}
2022
2023bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val,
2024 bool &IsDefault, const MCSubtargetInfo &STI) {
2025 return decodeCustomOperand(DepCtrInfo, DEP_CTR_SIZE, Code, Id, Name, Val,
2026 IsDefault, STI);
2027}
2028
2029int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask,
2030 const MCSubtargetInfo &STI) {
2031 return encodeCustomOperand(DepCtrInfo, DEP_CTR_SIZE, Name, Val, UsedOprMask,
2032 STI);
2033}
2034
2035unsigned getVaVdstBitMask() { return (1 << getVaVdstBitWidth()) - 1; }
2036
2037unsigned getVaSdstBitMask() { return (1 << getVaSdstBitWidth()) - 1; }
2038
2039unsigned getVaSsrcBitMask() { return (1 << getVaSsrcBitWidth()) - 1; }
2040
2042 return (1 << getHoldCntWidth(Version.Major, Version.Minor)) - 1;
2043}
2044
2045unsigned getVmVsrcBitMask() { return (1 << getVmVsrcBitWidth()) - 1; }
2046
2047unsigned getVaVccBitMask() { return (1 << getVaVccBitWidth()) - 1; }
2048
2049unsigned getSaSdstBitMask() { return (1 << getSaSdstBitWidth()) - 1; }
2050
2051unsigned decodeFieldVmVsrc(unsigned Encoded) {
2052 return unpackBits(Encoded, getVmVsrcBitShift(), getVmVsrcBitWidth());
2053}
2054
2055unsigned decodeFieldVaVdst(unsigned Encoded) {
2056 return unpackBits(Encoded, getVaVdstBitShift(), getVaVdstBitWidth());
2057}
2058
2059unsigned decodeFieldSaSdst(unsigned Encoded) {
2060 return unpackBits(Encoded, getSaSdstBitShift(), getSaSdstBitWidth());
2061}
2062
2063unsigned decodeFieldVaSdst(unsigned Encoded) {
2064 return unpackBits(Encoded, getVaSdstBitShift(), getVaSdstBitWidth());
2065}
2066
2067unsigned decodeFieldVaVcc(unsigned Encoded) {
2068 return unpackBits(Encoded, getVaVccBitShift(), getVaVccBitWidth());
2069}
2070
2071unsigned decodeFieldVaSsrc(unsigned Encoded) {
2072 return unpackBits(Encoded, getVaSsrcBitShift(), getVaSsrcBitWidth());
2073}
2074
2075unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version) {
2076 return unpackBits(Encoded, getHoldCntBitShift(),
2077 getHoldCntWidth(Version.Major, Version.Minor));
2078}
2079
2080unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc) {
2081 return packBits(VmVsrc, Encoded, getVmVsrcBitShift(), getVmVsrcBitWidth());
2082}
2083
2084unsigned encodeFieldVmVsrc(unsigned VmVsrc, const MCSubtargetInfo &STI) {
2085 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2086 return encodeFieldVmVsrc(Encoded, VmVsrc);
2087}
2088
2089unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst) {
2090 return packBits(VaVdst, Encoded, getVaVdstBitShift(), getVaVdstBitWidth());
2091}
2092
2093unsigned encodeFieldVaVdst(unsigned VaVdst, const MCSubtargetInfo &STI) {
2094 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2095 return encodeFieldVaVdst(Encoded, VaVdst);
2096}
2097
2098unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst) {
2099 return packBits(SaSdst, Encoded, getSaSdstBitShift(), getSaSdstBitWidth());
2100}
2101
2102unsigned encodeFieldSaSdst(unsigned SaSdst, const MCSubtargetInfo &STI) {
2103 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2104 return encodeFieldSaSdst(Encoded, SaSdst);
2105}
2106
2107unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst) {
2108 return packBits(VaSdst, Encoded, getVaSdstBitShift(), getVaSdstBitWidth());
2109}
2110
2111unsigned encodeFieldVaSdst(unsigned VaSdst, const MCSubtargetInfo &STI) {
2112 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2113 return encodeFieldVaSdst(Encoded, VaSdst);
2114}
2115
2116unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc) {
2117 return packBits(VaVcc, Encoded, getVaVccBitShift(), getVaVccBitWidth());
2118}
2119
2120unsigned encodeFieldVaVcc(unsigned VaVcc, const MCSubtargetInfo &STI) {
2121 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2122 return encodeFieldVaVcc(Encoded, VaVcc);
2123}
2124
2125unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc) {
2126 return packBits(VaSsrc, Encoded, getVaSsrcBitShift(), getVaSsrcBitWidth());
2127}
2128
2129unsigned encodeFieldVaSsrc(unsigned VaSsrc, const MCSubtargetInfo &STI) {
2130 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2131 return encodeFieldVaSsrc(Encoded, VaSsrc);
2132}
2133
2134unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt,
2135 const IsaVersion &Version) {
2136 return packBits(HoldCnt, Encoded, getHoldCntBitShift(),
2137 getHoldCntWidth(Version.Major, Version.Minor));
2138}
2139
2140unsigned encodeFieldHoldCnt(unsigned HoldCnt, const MCSubtargetInfo &STI) {
2141 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2142 return encodeFieldHoldCnt(Encoded, HoldCnt, getIsaVersion(STI.getCPU()));
2143}
2144
2145} // namespace DepCtr
2146
2147//===----------------------------------------------------------------------===//
2148// exp tgt
2149//===----------------------------------------------------------------------===//
2150
2151namespace Exp {
2152
2153struct ExpTgt {
2155 unsigned Tgt;
2156 unsigned MaxIndex;
2157};
2158
2159// clang-format off
2160static constexpr ExpTgt ExpTgtInfo[] = {
2161 {{"null"}, ET_NULL, ET_NULL_MAX_IDX},
2162 {{"mrtz"}, ET_MRTZ, ET_MRTZ_MAX_IDX},
2163 {{"prim"}, ET_PRIM, ET_PRIM_MAX_IDX},
2164 {{"mrt"}, ET_MRT0, ET_MRT_MAX_IDX},
2165 {{"pos"}, ET_POS0, ET_POS_MAX_IDX},
2166 {{"dual_src_blend"},ET_DUAL_SRC_BLEND0, ET_DUAL_SRC_BLEND_MAX_IDX},
2167 {{"param"}, ET_PARAM0, ET_PARAM_MAX_IDX},
2168};
2169// clang-format on
2170
2171bool getTgtName(unsigned Id, StringRef &Name, int &Index) {
2172 for (const ExpTgt &Val : ExpTgtInfo) {
2173 if (Val.Tgt <= Id && Id <= Val.Tgt + Val.MaxIndex) {
2174 Index = (Val.MaxIndex == 0) ? -1 : (Id - Val.Tgt);
2175 Name = Val.Name;
2176 return true;
2177 }
2178 }
2179 return false;
2180}
2181
2182unsigned getTgtId(const StringRef Name) {
2183
2184 for (const ExpTgt &Val : ExpTgtInfo) {
2185 if (Val.MaxIndex == 0 && Name == Val.Name)
2186 return Val.Tgt;
2187
2188 if (Val.MaxIndex > 0 && Name.starts_with(Val.Name)) {
2189 StringRef Suffix = Name.drop_front(Val.Name.size());
2190
2191 unsigned Id;
2192 if (Suffix.getAsInteger(10, Id) || Id > Val.MaxIndex)
2193 return ET_INVALID;
2194
2195 // Disable leading zeroes
2196 if (Suffix.size() > 1 && Suffix[0] == '0')
2197 return ET_INVALID;
2198
2199 return Val.Tgt + Id;
2200 }
2201 }
2202 return ET_INVALID;
2203}
2204
2205bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI) {
2206 switch (Id) {
2207 case ET_NULL:
2208 return !isGFX11Plus(STI);
2209 case ET_POS4:
2210 case ET_PRIM:
2211 return isGFX10Plus(STI);
2212 case ET_DUAL_SRC_BLEND0:
2213 case ET_DUAL_SRC_BLEND1:
2214 return isGFX11Plus(STI);
2215 default:
2216 if (Id >= ET_PARAM0 && Id <= ET_PARAM31)
2217 return !isGFX11Plus(STI) || isGFX13Plus(STI);
2218 return true;
2219 }
2220}
2221
2222} // namespace Exp
2223
2224//===----------------------------------------------------------------------===//
2225// MTBUF Format
2226//===----------------------------------------------------------------------===//
2227
2228namespace MTBUFFormat {
2229
2230int64_t getDfmt(const StringRef Name) {
2231 for (int Id = DFMT_MIN; Id <= DFMT_MAX; ++Id) {
2232 if (Name == DfmtSymbolic[Id])
2233 return Id;
2234 }
2235 return DFMT_UNDEF;
2236}
2237
2239 assert(Id <= DFMT_MAX);
2240 return DfmtSymbolic[Id];
2241}
2242
2244 if (isSI(STI) || isCI(STI))
2245 return NfmtSymbolicSICI;
2246 if (isVI(STI) || isGFX9(STI))
2247 return NfmtSymbolicVI;
2248 return NfmtSymbolicGFX10;
2249}
2250
2251int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI) {
2252 const auto *lookupTable = getNfmtLookupTable(STI);
2253 for (int Id = NFMT_MIN; Id <= NFMT_MAX; ++Id) {
2254 if (Name == lookupTable[Id])
2255 return Id;
2256 }
2257 return NFMT_UNDEF;
2258}
2259
2260StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI) {
2261 assert(Id <= NFMT_MAX);
2262 return getNfmtLookupTable(STI)[Id];
2263}
2264
2265bool isValidDfmtNfmt(unsigned Id, const MCSubtargetInfo &STI) {
2266 unsigned Dfmt;
2267 unsigned Nfmt;
2268 decodeDfmtNfmt(Id, Dfmt, Nfmt);
2269 return isValidNfmt(Nfmt, STI);
2270}
2271
2272bool isValidNfmt(unsigned Id, const MCSubtargetInfo &STI) {
2273 return !getNfmtName(Id, STI).empty();
2274}
2275
2276int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt) {
2277 return (Dfmt << DFMT_SHIFT) | (Nfmt << NFMT_SHIFT);
2278}
2279
2280void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt) {
2281 Dfmt = (Format >> DFMT_SHIFT) & DFMT_MASK;
2282 Nfmt = (Format >> NFMT_SHIFT) & NFMT_MASK;
2283}
2284
2285int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI) {
2286 if (isGFX11Plus(STI)) {
2287 for (int Id = UfmtGFX11::UFMT_FIRST; Id <= UfmtGFX11::UFMT_LAST; ++Id) {
2288 if (Name == UfmtSymbolicGFX11[Id])
2289 return Id;
2290 }
2291 } else {
2292 for (int Id = UfmtGFX10::UFMT_FIRST; Id <= UfmtGFX10::UFMT_LAST; ++Id) {
2293 if (Name == UfmtSymbolicGFX10[Id])
2294 return Id;
2295 }
2296 }
2297 return UFMT_UNDEF;
2298}
2299
2301 if (isValidUnifiedFormat(Id, STI))
2302 return isGFX10(STI) ? UfmtSymbolicGFX10[Id] : UfmtSymbolicGFX11[Id];
2303 return "";
2304}
2305
2306bool isValidUnifiedFormat(unsigned Id, const MCSubtargetInfo &STI) {
2307 return isGFX10(STI) ? Id <= UfmtGFX10::UFMT_LAST : Id <= UfmtGFX11::UFMT_LAST;
2308}
2309
2310int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt,
2311 const MCSubtargetInfo &STI) {
2312 int64_t Fmt = encodeDfmtNfmt(Dfmt, Nfmt);
2313 if (isGFX11Plus(STI)) {
2314 for (int Id = UfmtGFX11::UFMT_FIRST; Id <= UfmtGFX11::UFMT_LAST; ++Id) {
2315 if (Fmt == DfmtNfmt2UFmtGFX11[Id])
2316 return Id;
2317 }
2318 } else {
2319 for (int Id = UfmtGFX10::UFMT_FIRST; Id <= UfmtGFX10::UFMT_LAST; ++Id) {
2320 if (Fmt == DfmtNfmt2UFmtGFX10[Id])
2321 return Id;
2322 }
2323 }
2324 return UFMT_UNDEF;
2325}
2326
2327bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI) {
2328 return isGFX10Plus(STI) ? (Val <= UFMT_MAX) : (Val <= DFMT_NFMT_MAX);
2329}
2330
2332 if (isGFX10Plus(STI))
2333 return UFMT_DEFAULT;
2334 return DFMT_NFMT_DEFAULT;
2335}
2336
2337} // namespace MTBUFFormat
2338
2339//===----------------------------------------------------------------------===//
2340// SendMsg
2341//===----------------------------------------------------------------------===//
2342
2343namespace SendMsg {
2344
2348
2349bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI) {
2350 return (MsgId & ~(getMsgIdMask(STI))) == 0;
2351}
2352
2353bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI,
2354 bool Strict) {
2355 assert(isValidMsgId(MsgId, STI));
2356
2357 if (!Strict)
2358 return 0 <= OpId && isUInt<OP_WIDTH_>(OpId);
2359
2360 if (msgRequiresOp(MsgId, STI)) {
2361 if (MsgId == ID_GS_PreGFX11 && OpId == OP_GS_NOP)
2362 return false;
2363
2364 return !getMsgOpName(MsgId, OpId, STI).empty();
2365 }
2366
2367 return OpId == OP_NONE_;
2368}
2369
2370bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId,
2371 const MCSubtargetInfo &STI, bool Strict) {
2372 assert(isValidMsgOp(MsgId, OpId, STI, Strict));
2373
2374 if (!Strict)
2376
2377 if (!isGFX11Plus(STI)) {
2378 switch (MsgId) {
2379 case ID_GS_PreGFX11:
2382 return (OpId == OP_GS_NOP)
2385 }
2386 }
2387 return StreamId == STREAM_ID_NONE_;
2388}
2389
2390bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI) {
2391 return MsgId == ID_SYSMSG ||
2392 (!isGFX11Plus(STI) &&
2393 (MsgId == ID_GS_PreGFX11 || MsgId == ID_GS_DONE_PreGFX11));
2394}
2395
2396bool msgSupportsStream(int64_t MsgId, int64_t OpId,
2397 const MCSubtargetInfo &STI) {
2398 return !isGFX11Plus(STI) &&
2399 (MsgId == ID_GS_PreGFX11 || MsgId == ID_GS_DONE_PreGFX11) &&
2400 OpId != OP_GS_NOP;
2401}
2402
2403void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId,
2404 uint16_t &StreamId, const MCSubtargetInfo &STI) {
2405 MsgId = Val & getMsgIdMask(STI);
2406 if (isGFX11Plus(STI)) {
2407 OpId = 0;
2408 StreamId = 0;
2409 } else {
2410 OpId = (Val & OP_MASK_) >> OP_SHIFT_;
2412 }
2413}
2414
2416 return MsgId | (OpId << OP_SHIFT_) | (StreamId << STREAM_ID_SHIFT_);
2417}
2418
2419bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI) {
2420 // Explicitly list message types that are known to not use m0.
2421 // This is safer than excluding only GS_ALLOC_REQ, in case new message
2422 // types are added in the future that do use m0.
2423 if (isGFX11Plus(STI)) {
2424 switch (MsgId) {
2426 return true;
2427 default:
2428 break;
2429 }
2430 }
2431 switch (MsgId) {
2432 case ID_SAVEWAVE:
2433 case ID_STALL_WAVE_GEN:
2434 case ID_HALT_WAVES:
2435 case ID_ORDERED_PS_DONE:
2437 case ID_GET_DOORBELL:
2438 case ID_GET_DDID:
2439 case ID_SYSMSG:
2440 return true;
2441 default:
2442 return false;
2443 }
2444}
2445
2446} // namespace SendMsg
2447
2448//===----------------------------------------------------------------------===//
2449//
2450//===----------------------------------------------------------------------===//
2451
2453 return F.getFnAttributeAsParsedInteger("InitialPSInputAddr", 0);
2454}
2455
2457 // As a safe default always respond as if PS has color exports.
2458 return F.getFnAttributeAsParsedInteger(
2459 "amdgpu-color-export",
2460 F.getCallingConv() == CallingConv::AMDGPU_PS ? 1 : 0) != 0;
2461}
2462
2464 return F.getFnAttributeAsParsedInteger("amdgpu-depth-export", 0) != 0;
2465}
2466
2468 unsigned BlockSize =
2469 F.getFnAttributeAsParsedInteger("amdgpu-dynamic-vgpr-block-size", 0);
2470
2471 if (BlockSize == 16 || BlockSize == 32)
2472 return BlockSize;
2473
2474 return 0;
2475}
2476
2477bool hasXNACK(const MCSubtargetInfo &STI) {
2478 return STI.hasFeature(AMDGPU::FeatureXNACK);
2479}
2480
2482 return STI.hasFeature(AMDGPU::FeatureMIMG_R128) &&
2483 !STI.hasFeature(AMDGPU::FeatureR128A16);
2484}
2485
2486bool hasA16(const MCSubtargetInfo &STI) {
2487 return STI.hasFeature(AMDGPU::FeatureA16);
2488}
2489
2490bool hasG16(const MCSubtargetInfo &STI) {
2491 return STI.hasFeature(AMDGPU::FeatureG16);
2492}
2493
2495 return !STI.hasFeature(AMDGPU::FeatureUnpackedD16VMem) && !isCI(STI) &&
2496 !isSI(STI);
2497}
2498
2499bool hasGDS(const MCSubtargetInfo &STI) {
2500 return STI.hasFeature(AMDGPU::FeatureGDS);
2501}
2502
2503unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler) {
2504 auto Version = getIsaVersion(STI.getCPU());
2505 if (Version.Major == 10)
2506 return Version.Minor >= 3 ? 13 : 5;
2507 if (Version.Major == 11)
2508 return 5;
2509 if (Version.Major >= 12)
2510 return HasSampler ? 4 : 5;
2511 return 0;
2512}
2513
2515 if (isGFX1250Plus(STI))
2516 return 32;
2517 return 16;
2518}
2519
2520bool isSI(const MCSubtargetInfo &STI) {
2521 return STI.hasFeature(AMDGPU::FeatureSouthernIslands);
2522}
2523
2524bool isCI(const MCSubtargetInfo &STI) {
2525 return STI.hasFeature(AMDGPU::FeatureSeaIslands);
2526}
2527
2528bool isVI(const MCSubtargetInfo &STI) {
2529 return STI.hasFeature(AMDGPU::FeatureVolcanicIslands);
2530}
2531
2532bool isGFX9(const MCSubtargetInfo &STI) {
2533 return STI.hasFeature(AMDGPU::FeatureGFX9);
2534}
2535
2537 return isGFX9(STI) || isGFX10(STI);
2538}
2539
2541 return isGFX9(STI) || isGFX10(STI) || isGFX11(STI);
2542}
2543
2545 return isVI(STI) || isGFX9(STI) || isGFX10(STI);
2546}
2547
2548bool isGFX8Plus(const MCSubtargetInfo &STI) {
2549 return isVI(STI) || isGFX9Plus(STI);
2550}
2551
2552bool isGFX9Plus(const MCSubtargetInfo &STI) {
2553 return isGFX9(STI) || isGFX10Plus(STI);
2554}
2555
2556bool isNotGFX9Plus(const MCSubtargetInfo &STI) { return !isGFX9Plus(STI); }
2557
2559 return STI.hasFeature(AMDGPU::FeaturePopsExitingWaveID);
2560}
2561
2563 return STI.hasFeature(AMDGPU::FeatureApertureRegs) &&
2564 !STI.hasFeature(AMDGPU::FeatureGloballyAddressableScratch);
2565}
2566
2567bool isGFX10(const MCSubtargetInfo &STI) {
2568 return STI.hasFeature(AMDGPU::FeatureGFX10);
2569}
2570
2572 return isGFX10(STI) || isGFX11(STI);
2573}
2574
2576 return isGFX10(STI) || isGFX11Plus(STI);
2577}
2578
2579bool isGFX11(const MCSubtargetInfo &STI) {
2580 return STI.hasFeature(AMDGPU::FeatureGFX11);
2581}
2582
2584 return isGFX11(STI) || isGFX12Plus(STI);
2585}
2586
2587bool isGFX12(const MCSubtargetInfo &STI) {
2588 return STI.getFeatureBits()[AMDGPU::FeatureGFX12];
2589}
2590
2592 return isGFX12(STI) || isGFX13Plus(STI);
2593}
2594
2595bool isNotGFX12Plus(const MCSubtargetInfo &STI) { return !isGFX12Plus(STI); }
2596
2597bool isGFX1250(const MCSubtargetInfo &STI) {
2598 return STI.getFeatureBits()[AMDGPU::FeatureGFX1250Insts] && !isGFX13(STI);
2599}
2600
2602 return isGFX1250(STI) || !STI.getFeatureBits().test(FeatureCuMode);
2603}
2604
2606 return STI.getFeatureBits()[AMDGPU::FeatureGFX1250Insts];
2607}
2608
2609bool isGFX13(const MCSubtargetInfo &STI) {
2610 return STI.getFeatureBits()[AMDGPU::FeatureGFX13];
2611}
2612
2613bool isGFX13Plus(const MCSubtargetInfo &STI) { return isGFX13(STI); }
2614
2616 if (isGFX1250(STI))
2617 return false;
2618 return isGFX10Plus(STI);
2619}
2620
2621bool isNotGFX11Plus(const MCSubtargetInfo &STI) { return !isGFX11Plus(STI); }
2622
2624 return isSI(STI) || isCI(STI) || isVI(STI) || isGFX9(STI);
2625}
2626
2628 return isGFX10(STI) && !AMDGPU::isGFX10_BEncoding(STI);
2629}
2630
2632 return STI.hasFeature(AMDGPU::FeatureGCN3Encoding);
2633}
2634
2636 return STI.hasFeature(AMDGPU::FeatureGFX10_BEncoding);
2637}
2638
2640 return STI.hasFeature(AMDGPU::FeatureGFX10_3Insts);
2641}
2642
2644 return isGFX10_BEncoding(STI) && !isGFX12Plus(STI);
2645}
2646
2647bool isGFX90A(const MCSubtargetInfo &STI) {
2648 return STI.hasFeature(AMDGPU::FeatureGFX90AInsts);
2649}
2650
2651bool isGFX940(const MCSubtargetInfo &STI) {
2652 return STI.hasFeature(AMDGPU::FeatureGFX940Insts);
2653}
2654
2656 return STI.hasFeature(AMDGPU::FeatureArchitectedFlatScratch);
2657}
2658
2660 return STI.hasFeature(AMDGPU::FeatureMAIInsts);
2661}
2662
2663bool hasVOPD(const MCSubtargetInfo &STI) {
2664 return STI.hasFeature(AMDGPU::FeatureVOPDInsts);
2665}
2666
2668 return STI.hasFeature(AMDGPU::FeatureDPPSrc1SGPR);
2669}
2670
2672 return STI.hasFeature(AMDGPU::FeatureKernargPreload);
2673}
2674
2675int32_t getTotalNumVGPRs(bool has90AInsts, int32_t ArgNumAGPR,
2676 int32_t ArgNumVGPR) {
2677 if (has90AInsts && ArgNumAGPR)
2678 return alignTo(ArgNumVGPR, 4) + ArgNumAGPR;
2679 return std::max(ArgNumVGPR, ArgNumAGPR);
2680}
2681
2683 const MCRegisterClass &SGPRClass =
2684 TRI->getRegClass(AMDGPU::SReg_32RegClassID);
2685 const MCRegister FirstSubReg = TRI->getSubReg(Reg, AMDGPU::sub0);
2686 return SGPRClass.contains(FirstSubReg != 0 ? FirstSubReg : Reg) ||
2687 Reg == AMDGPU::SCC;
2688}
2689
2693
2694#define MAP_REG2REG \
2695 using namespace AMDGPU; \
2696 switch (Reg.id()) { \
2697 default: \
2698 return Reg; \
2699 CASE_CI_VI(FLAT_SCR) \
2700 CASE_CI_VI(FLAT_SCR_LO) \
2701 CASE_CI_VI(FLAT_SCR_HI) \
2702 CASE_VI_GFX9PLUS(TTMP0) \
2703 CASE_VI_GFX9PLUS(TTMP1) \
2704 CASE_VI_GFX9PLUS(TTMP2) \
2705 CASE_VI_GFX9PLUS(TTMP3) \
2706 CASE_VI_GFX9PLUS(TTMP4) \
2707 CASE_VI_GFX9PLUS(TTMP5) \
2708 CASE_VI_GFX9PLUS(TTMP6) \
2709 CASE_VI_GFX9PLUS(TTMP7) \
2710 CASE_VI_GFX9PLUS(TTMP8) \
2711 CASE_VI_GFX9PLUS(TTMP9) \
2712 CASE_VI_GFX9PLUS(TTMP10) \
2713 CASE_VI_GFX9PLUS(TTMP11) \
2714 CASE_VI_GFX9PLUS(TTMP12) \
2715 CASE_VI_GFX9PLUS(TTMP13) \
2716 CASE_VI_GFX9PLUS(TTMP14) \
2717 CASE_VI_GFX9PLUS(TTMP15) \
2718 CASE_VI_GFX9PLUS(TTMP0_TTMP1) \
2719 CASE_VI_GFX9PLUS(TTMP2_TTMP3) \
2720 CASE_VI_GFX9PLUS(TTMP4_TTMP5) \
2721 CASE_VI_GFX9PLUS(TTMP6_TTMP7) \
2722 CASE_VI_GFX9PLUS(TTMP8_TTMP9) \
2723 CASE_VI_GFX9PLUS(TTMP10_TTMP11) \
2724 CASE_VI_GFX9PLUS(TTMP12_TTMP13) \
2725 CASE_VI_GFX9PLUS(TTMP14_TTMP15) \
2726 CASE_VI_GFX9PLUS(TTMP0_TTMP1_TTMP2_TTMP3) \
2727 CASE_VI_GFX9PLUS(TTMP4_TTMP5_TTMP6_TTMP7) \
2728 CASE_VI_GFX9PLUS(TTMP8_TTMP9_TTMP10_TTMP11) \
2729 CASE_VI_GFX9PLUS(TTMP12_TTMP13_TTMP14_TTMP15) \
2730 CASE_VI_GFX9PLUS(TTMP0_TTMP1_TTMP2_TTMP3_TTMP4_TTMP5_TTMP6_TTMP7) \
2731 CASE_VI_GFX9PLUS(TTMP4_TTMP5_TTMP6_TTMP7_TTMP8_TTMP9_TTMP10_TTMP11) \
2732 CASE_VI_GFX9PLUS(TTMP8_TTMP9_TTMP10_TTMP11_TTMP12_TTMP13_TTMP14_TTMP15) \
2733 CASE_VI_GFX9PLUS( \
2734 TTMP0_TTMP1_TTMP2_TTMP3_TTMP4_TTMP5_TTMP6_TTMP7_TTMP8_TTMP9_TTMP10_TTMP11_TTMP12_TTMP13_TTMP14_TTMP15) \
2735 CASE_GFXPRE11_GFX11PLUS(M0) \
2736 CASE_GFXPRE11_GFX11PLUS(SGPR_NULL) \
2737 CASE_GFXPRE11_GFX11PLUS_TO(SGPR_NULL64, SGPR_NULL) \
2738 }
2739
2740#define CASE_CI_VI(node) \
2741 assert(!isSI(STI)); \
2742 case node: \
2743 return isCI(STI) ? node##_ci : node##_vi;
2744
2745#define CASE_VI_GFX9PLUS(node) \
2746 case node: \
2747 return isGFX9Plus(STI) ? node##_gfx9plus : node##_vi;
2748
2749#define CASE_GFXPRE11_GFX11PLUS(node) \
2750 case node: \
2751 return isGFX11Plus(STI) ? node##_gfx11plus : node##_gfxpre11;
2752
2753#define CASE_GFXPRE11_GFX11PLUS_TO(node, result) \
2754 case node: \
2755 return isGFX11Plus(STI) ? result##_gfx11plus : result##_gfxpre11;
2756
2758 if (STI.getTargetTriple().getArch() == Triple::r600)
2759 return Reg;
2761}
2762
2763#undef CASE_CI_VI
2764#undef CASE_VI_GFX9PLUS
2765#undef CASE_GFXPRE11_GFX11PLUS
2766#undef CASE_GFXPRE11_GFX11PLUS_TO
2767
2768#define CASE_CI_VI(node) \
2769 case node##_ci: \
2770 case node##_vi: \
2771 return node;
2772#define CASE_VI_GFX9PLUS(node) \
2773 case node##_vi: \
2774 case node##_gfx9plus: \
2775 return node;
2776#define CASE_GFXPRE11_GFX11PLUS(node) \
2777 case node##_gfx11plus: \
2778 case node##_gfxpre11: \
2779 return node;
2780#define CASE_GFXPRE11_GFX11PLUS_TO(node, result)
2781
2783
2785 switch (Reg.id()) {
2786 case AMDGPU::SRC_SHARED_BASE_LO:
2787 case AMDGPU::SRC_SHARED_BASE:
2788 case AMDGPU::SRC_SHARED_LIMIT_LO:
2789 case AMDGPU::SRC_SHARED_LIMIT:
2790 case AMDGPU::SRC_PRIVATE_BASE_LO:
2791 case AMDGPU::SRC_PRIVATE_BASE:
2792 case AMDGPU::SRC_PRIVATE_LIMIT_LO:
2793 case AMDGPU::SRC_PRIVATE_LIMIT:
2794 case AMDGPU::SRC_FLAT_SCRATCH_BASE_LO:
2795 case AMDGPU::SRC_FLAT_SCRATCH_BASE_HI:
2796 case AMDGPU::SRC_POPS_EXITING_WAVE_ID:
2797 return true;
2798 case AMDGPU::SRC_VCCZ:
2799 case AMDGPU::SRC_EXECZ:
2800 case AMDGPU::SRC_SCC:
2801 return true;
2802 case AMDGPU::SGPR_NULL:
2803 return true;
2804 default:
2805 return false;
2806 }
2807}
2808
2809#undef CASE_CI_VI
2810#undef CASE_VI_GFX9PLUS
2811#undef CASE_GFXPRE11_GFX11PLUS
2812#undef CASE_GFXPRE11_GFX11PLUS_TO
2813#undef MAP_REG2REG
2814
2815bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2816 assert(OpNo < Desc.NumOperands);
2817 unsigned OpType = Desc.operands()[OpNo].OperandType;
2818 return OpType >= AMDGPU::OPERAND_KIMM_FIRST &&
2819 OpType <= AMDGPU::OPERAND_KIMM_LAST;
2820}
2821
2822bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2823 assert(OpNo < Desc.NumOperands);
2824 unsigned OpType = Desc.operands()[OpNo].OperandType;
2825 switch (OpType) {
2840 return true;
2841 default:
2842 return false;
2843 }
2844}
2845
2846bool isSISrcInlinableOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2847 assert(OpNo < Desc.NumOperands);
2848 unsigned OpType = Desc.operands()[OpNo].OperandType;
2849 return (OpType >= AMDGPU::OPERAND_REG_INLINE_C_FIRST &&
2853}
2854
2855// Avoid using MCRegisterClass::getSize, since that function will go away
2856// (move from MC* level to Target* level). Return size in bits.
2857unsigned getRegBitWidth(unsigned RCID) {
2858 switch (RCID) {
2859 case AMDGPU::VGPR_16RegClassID:
2860 case AMDGPU::VGPR_16_Lo128RegClassID:
2861 case AMDGPU::SGPR_LO16RegClassID:
2862 case AMDGPU::AGPR_LO16RegClassID:
2863 return 16;
2864 case AMDGPU::SGPR_32RegClassID:
2865 case AMDGPU::VGPR_32RegClassID:
2866 case AMDGPU::VGPR_32_Lo256RegClassID:
2867 case AMDGPU::VRegOrLds_32RegClassID:
2868 case AMDGPU::AGPR_32RegClassID:
2869 case AMDGPU::VS_32RegClassID:
2870 case AMDGPU::AV_32RegClassID:
2871 case AMDGPU::SReg_32RegClassID:
2872 case AMDGPU::SReg_32_XM0RegClassID:
2873 case AMDGPU::SRegOrLds_32RegClassID:
2874 return 32;
2875 case AMDGPU::SGPR_64RegClassID:
2876 case AMDGPU::VS_64RegClassID:
2877 case AMDGPU::SReg_64RegClassID:
2878 case AMDGPU::VReg_64RegClassID:
2879 case AMDGPU::AReg_64RegClassID:
2880 case AMDGPU::SReg_64_XEXECRegClassID:
2881 case AMDGPU::VReg_64_Align2RegClassID:
2882 case AMDGPU::AReg_64_Align2RegClassID:
2883 case AMDGPU::AV_64RegClassID:
2884 case AMDGPU::AV_64_Align2RegClassID:
2885 case AMDGPU::VReg_64_Lo256_Align2RegClassID:
2886 case AMDGPU::VS_64_Lo256RegClassID:
2887 return 64;
2888 case AMDGPU::SGPR_96RegClassID:
2889 case AMDGPU::SReg_96RegClassID:
2890 case AMDGPU::VReg_96RegClassID:
2891 case AMDGPU::AReg_96RegClassID:
2892 case AMDGPU::VReg_96_Align2RegClassID:
2893 case AMDGPU::AReg_96_Align2RegClassID:
2894 case AMDGPU::AV_96RegClassID:
2895 case AMDGPU::AV_96_Align2RegClassID:
2896 case AMDGPU::VReg_96_Lo256_Align2RegClassID:
2897 return 96;
2898 case AMDGPU::SGPR_128RegClassID:
2899 case AMDGPU::SReg_128RegClassID:
2900 case AMDGPU::VReg_128RegClassID:
2901 case AMDGPU::AReg_128RegClassID:
2902 case AMDGPU::VReg_128_Align2RegClassID:
2903 case AMDGPU::AReg_128_Align2RegClassID:
2904 case AMDGPU::AV_128RegClassID:
2905 case AMDGPU::AV_128_Align2RegClassID:
2906 case AMDGPU::SReg_128_XNULLRegClassID:
2907 case AMDGPU::VReg_128_Lo256_Align2RegClassID:
2908 return 128;
2909 case AMDGPU::SGPR_160RegClassID:
2910 case AMDGPU::SReg_160RegClassID:
2911 case AMDGPU::VReg_160RegClassID:
2912 case AMDGPU::AReg_160RegClassID:
2913 case AMDGPU::VReg_160_Align2RegClassID:
2914 case AMDGPU::AReg_160_Align2RegClassID:
2915 case AMDGPU::AV_160RegClassID:
2916 case AMDGPU::AV_160_Align2RegClassID:
2917 case AMDGPU::VReg_160_Lo256_Align2RegClassID:
2918 return 160;
2919 case AMDGPU::SGPR_192RegClassID:
2920 case AMDGPU::SReg_192RegClassID:
2921 case AMDGPU::VReg_192RegClassID:
2922 case AMDGPU::AReg_192RegClassID:
2923 case AMDGPU::VReg_192_Align2RegClassID:
2924 case AMDGPU::AReg_192_Align2RegClassID:
2925 case AMDGPU::AV_192RegClassID:
2926 case AMDGPU::AV_192_Align2RegClassID:
2927 case AMDGPU::VReg_192_Lo256_Align2RegClassID:
2928 return 192;
2929 case AMDGPU::SGPR_224RegClassID:
2930 case AMDGPU::SReg_224RegClassID:
2931 case AMDGPU::VReg_224RegClassID:
2932 case AMDGPU::AReg_224RegClassID:
2933 case AMDGPU::VReg_224_Align2RegClassID:
2934 case AMDGPU::AReg_224_Align2RegClassID:
2935 case AMDGPU::AV_224RegClassID:
2936 case AMDGPU::AV_224_Align2RegClassID:
2937 case AMDGPU::VReg_224_Lo256_Align2RegClassID:
2938 return 224;
2939 case AMDGPU::SGPR_256RegClassID:
2940 case AMDGPU::SReg_256RegClassID:
2941 case AMDGPU::VReg_256RegClassID:
2942 case AMDGPU::AReg_256RegClassID:
2943 case AMDGPU::VReg_256_Align2RegClassID:
2944 case AMDGPU::AReg_256_Align2RegClassID:
2945 case AMDGPU::AV_256RegClassID:
2946 case AMDGPU::AV_256_Align2RegClassID:
2947 case AMDGPU::SReg_256_XNULLRegClassID:
2948 case AMDGPU::VReg_256_Lo256_Align2RegClassID:
2949 return 256;
2950 case AMDGPU::SGPR_288RegClassID:
2951 case AMDGPU::SReg_288RegClassID:
2952 case AMDGPU::VReg_288RegClassID:
2953 case AMDGPU::AReg_288RegClassID:
2954 case AMDGPU::VReg_288_Align2RegClassID:
2955 case AMDGPU::AReg_288_Align2RegClassID:
2956 case AMDGPU::AV_288RegClassID:
2957 case AMDGPU::AV_288_Align2RegClassID:
2958 case AMDGPU::VReg_288_Lo256_Align2RegClassID:
2959 return 288;
2960 case AMDGPU::SGPR_320RegClassID:
2961 case AMDGPU::SReg_320RegClassID:
2962 case AMDGPU::VReg_320RegClassID:
2963 case AMDGPU::AReg_320RegClassID:
2964 case AMDGPU::VReg_320_Align2RegClassID:
2965 case AMDGPU::AReg_320_Align2RegClassID:
2966 case AMDGPU::AV_320RegClassID:
2967 case AMDGPU::AV_320_Align2RegClassID:
2968 case AMDGPU::VReg_320_Lo256_Align2RegClassID:
2969 return 320;
2970 case AMDGPU::SGPR_352RegClassID:
2971 case AMDGPU::SReg_352RegClassID:
2972 case AMDGPU::VReg_352RegClassID:
2973 case AMDGPU::AReg_352RegClassID:
2974 case AMDGPU::VReg_352_Align2RegClassID:
2975 case AMDGPU::AReg_352_Align2RegClassID:
2976 case AMDGPU::AV_352RegClassID:
2977 case AMDGPU::AV_352_Align2RegClassID:
2978 case AMDGPU::VReg_352_Lo256_Align2RegClassID:
2979 return 352;
2980 case AMDGPU::SGPR_384RegClassID:
2981 case AMDGPU::SReg_384RegClassID:
2982 case AMDGPU::VReg_384RegClassID:
2983 case AMDGPU::AReg_384RegClassID:
2984 case AMDGPU::VReg_384_Align2RegClassID:
2985 case AMDGPU::AReg_384_Align2RegClassID:
2986 case AMDGPU::AV_384RegClassID:
2987 case AMDGPU::AV_384_Align2RegClassID:
2988 case AMDGPU::VReg_384_Lo256_Align2RegClassID:
2989 return 384;
2990 case AMDGPU::SGPR_512RegClassID:
2991 case AMDGPU::SReg_512RegClassID:
2992 case AMDGPU::VReg_512RegClassID:
2993 case AMDGPU::AReg_512RegClassID:
2994 case AMDGPU::VReg_512_Align2RegClassID:
2995 case AMDGPU::AReg_512_Align2RegClassID:
2996 case AMDGPU::AV_512RegClassID:
2997 case AMDGPU::AV_512_Align2RegClassID:
2998 case AMDGPU::VReg_512_Lo256_Align2RegClassID:
2999 return 512;
3000 case AMDGPU::SGPR_1024RegClassID:
3001 case AMDGPU::SReg_1024RegClassID:
3002 case AMDGPU::VReg_1024RegClassID:
3003 case AMDGPU::AReg_1024RegClassID:
3004 case AMDGPU::VReg_1024_Align2RegClassID:
3005 case AMDGPU::AReg_1024_Align2RegClassID:
3006 case AMDGPU::AV_1024RegClassID:
3007 case AMDGPU::AV_1024_Align2RegClassID:
3008 case AMDGPU::VReg_1024_Lo256_Align2RegClassID:
3009 return 1024;
3010 default:
3011 llvm_unreachable("Unexpected register class");
3012 }
3013}
3014
3015unsigned getRegBitWidth(const MCRegisterClass &RC) {
3016 return getRegBitWidth(RC.getID());
3017}
3018
3019bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi) {
3021 return true;
3022
3023 uint64_t Val = static_cast<uint64_t>(Literal);
3024 return (Val == llvm::bit_cast<uint64_t>(0.0)) ||
3025 (Val == llvm::bit_cast<uint64_t>(1.0)) ||
3026 (Val == llvm::bit_cast<uint64_t>(-1.0)) ||
3027 (Val == llvm::bit_cast<uint64_t>(0.5)) ||
3028 (Val == llvm::bit_cast<uint64_t>(-0.5)) ||
3029 (Val == llvm::bit_cast<uint64_t>(2.0)) ||
3030 (Val == llvm::bit_cast<uint64_t>(-2.0)) ||
3031 (Val == llvm::bit_cast<uint64_t>(4.0)) ||
3032 (Val == llvm::bit_cast<uint64_t>(-4.0)) ||
3033 (Val == 0x3fc45f306dc9c882 && HasInv2Pi);
3034}
3035
3036bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi) {
3038 return true;
3039
3040 // The actual type of the operand does not seem to matter as long
3041 // as the bits match one of the inline immediate values. For example:
3042 //
3043 // -nan has the hexadecimal encoding of 0xfffffffe which is -2 in decimal,
3044 // so it is a legal inline immediate.
3045 //
3046 // 1065353216 has the hexadecimal encoding 0x3f800000 which is 1.0f in
3047 // floating-point, so it is a legal inline immediate.
3048
3049 uint32_t Val = static_cast<uint32_t>(Literal);
3050 return (Val == llvm::bit_cast<uint32_t>(0.0f)) ||
3051 (Val == llvm::bit_cast<uint32_t>(1.0f)) ||
3052 (Val == llvm::bit_cast<uint32_t>(-1.0f)) ||
3053 (Val == llvm::bit_cast<uint32_t>(0.5f)) ||
3054 (Val == llvm::bit_cast<uint32_t>(-0.5f)) ||
3055 (Val == llvm::bit_cast<uint32_t>(2.0f)) ||
3056 (Val == llvm::bit_cast<uint32_t>(-2.0f)) ||
3057 (Val == llvm::bit_cast<uint32_t>(4.0f)) ||
3058 (Val == llvm::bit_cast<uint32_t>(-4.0f)) ||
3059 (Val == 0x3e22f983 && HasInv2Pi);
3060}
3061
3062bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi) {
3063 if (!HasInv2Pi)
3064 return false;
3066 return true;
3067 uint16_t Val = static_cast<uint16_t>(Literal);
3068 return Val == 0x3F00 || // 0.5
3069 Val == 0xBF00 || // -0.5
3070 Val == 0x3F80 || // 1.0
3071 Val == 0xBF80 || // -1.0
3072 Val == 0x4000 || // 2.0
3073 Val == 0xC000 || // -2.0
3074 Val == 0x4080 || // 4.0
3075 Val == 0xC080 || // -4.0
3076 Val == 0x3E22; // 1.0 / (2.0 * pi)
3077}
3078
3079bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi) {
3080 return isInlinableLiteral32(Literal, HasInv2Pi);
3081}
3082
3083bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi) {
3084 if (!HasInv2Pi)
3085 return false;
3087 return true;
3088 uint16_t Val = static_cast<uint16_t>(Literal);
3089 return Val == 0x3C00 || // 1.0
3090 Val == 0xBC00 || // -1.0
3091 Val == 0x3800 || // 0.5
3092 Val == 0xB800 || // -0.5
3093 Val == 0x4000 || // 2.0
3094 Val == 0xC000 || // -2.0
3095 Val == 0x4400 || // 4.0
3096 Val == 0xC400 || // -4.0
3097 Val == 0x3118; // 1/2pi
3098}
3099
3100std::optional<unsigned> getInlineEncodingV216(bool IsFloat, uint32_t Literal) {
3101 // Unfortunately, the Instruction Set Architecture Reference Guide is
3102 // misleading about how the inline operands work for (packed) 16-bit
3103 // instructions. In a nutshell, the actual HW behavior is:
3104 //
3105 // - integer encodings (-16 .. 64) are always produced as sign-extended
3106 // 32-bit values
3107 // - float encodings are produced as:
3108 // - for F16 instructions: corresponding half-precision float values in
3109 // the LSBs, 0 in the MSBs
3110 // - for UI16 instructions: corresponding single-precision float value
3111 int32_t Signed = static_cast<int32_t>(Literal);
3112 if (Signed >= 0 && Signed <= 64)
3113 return 128 + Signed;
3114
3115 if (Signed >= -16 && Signed <= -1)
3116 return 192 + std::abs(Signed);
3117
3118 if (IsFloat) {
3119 // clang-format off
3120 switch (Literal) {
3121 case 0x3800: return 240; // 0.5
3122 case 0xB800: return 241; // -0.5
3123 case 0x3C00: return 242; // 1.0
3124 case 0xBC00: return 243; // -1.0
3125 case 0x4000: return 244; // 2.0
3126 case 0xC000: return 245; // -2.0
3127 case 0x4400: return 246; // 4.0
3128 case 0xC400: return 247; // -4.0
3129 case 0x3118: return 248; // 1.0 / (2.0 * pi)
3130 default: break;
3131 }
3132 // clang-format on
3133 } else {
3134 // clang-format off
3135 switch (Literal) {
3136 case 0x3F000000: return 240; // 0.5
3137 case 0xBF000000: return 241; // -0.5
3138 case 0x3F800000: return 242; // 1.0
3139 case 0xBF800000: return 243; // -1.0
3140 case 0x40000000: return 244; // 2.0
3141 case 0xC0000000: return 245; // -2.0
3142 case 0x40800000: return 246; // 4.0
3143 case 0xC0800000: return 247; // -4.0
3144 case 0x3E22F983: return 248; // 1.0 / (2.0 * pi)
3145 default: break;
3146 }
3147 // clang-format on
3148 }
3149
3150 return {};
3151}
3152
3153// Encoding of the literal as an inline constant for a V_PK_*_IU16 instruction
3154// or nullopt.
3155std::optional<unsigned> getInlineEncodingV2I16(uint32_t Literal) {
3156 return getInlineEncodingV216(false, Literal);
3157}
3158
3159// Encoding of the literal as an inline constant for a V_PK_*_BF16 instruction
3160// or nullopt.
3161std::optional<unsigned> getInlineEncodingV2BF16(uint32_t Literal) {
3162 int32_t Signed = static_cast<int32_t>(Literal);
3163 if (Signed >= 0 && Signed <= 64)
3164 return 128 + Signed;
3165
3166 if (Signed >= -16 && Signed <= -1)
3167 return 192 + std::abs(Signed);
3168
3169 // clang-format off
3170 switch (Literal) {
3171 case 0x3F00: return 240; // 0.5
3172 case 0xBF00: return 241; // -0.5
3173 case 0x3F80: return 242; // 1.0
3174 case 0xBF80: return 243; // -1.0
3175 case 0x4000: return 244; // 2.0
3176 case 0xC000: return 245; // -2.0
3177 case 0x4080: return 246; // 4.0
3178 case 0xC080: return 247; // -4.0
3179 case 0x3E22: return 248; // 1.0 / (2.0 * pi)
3180 default: break;
3181 }
3182 // clang-format on
3183
3184 return std::nullopt;
3185}
3186
3187// Encoding of the literal as an inline constant for a V_PK_*_F16 instruction
3188// or nullopt.
3189std::optional<unsigned> getInlineEncodingV2F16(uint32_t Literal) {
3190 return getInlineEncodingV216(true, Literal);
3191}
3192
3193// Encoding of the literal as an inline constant for V_PK_FMAC_F16 instruction
3194// or nullopt. This accounts for different inline constant behavior:
3195// - Pre-GFX11: fp16 inline constants have the value in low 16 bits, 0 in high
3196// - GFX11+: fp16 inline constants are duplicated into both halves
3198 bool IsGFX11Plus) {
3199 // Pre-GFX11 behavior: f16 in low bits, 0 in high bits
3200 if (!IsGFX11Plus)
3201 return getInlineEncodingV216(/*IsFloat=*/true, Literal);
3202
3203 // GFX11+ behavior: f16 duplicated in both halves
3204 // First, check for sign-extended integer inline constants (-16 to 64)
3205 // These work the same across all generations
3206 int32_t Signed = static_cast<int32_t>(Literal);
3207 if (Signed >= 0 && Signed <= 64)
3208 return 128 + Signed;
3209
3210 if (Signed >= -16 && Signed <= -1)
3211 return 192 + std::abs(Signed);
3212
3213 // For float inline constants on GFX11+, both halves must be equal
3214 uint16_t Lo = static_cast<uint16_t>(Literal);
3215 uint16_t Hi = static_cast<uint16_t>(Literal >> 16);
3216 if (Lo != Hi)
3217 return std::nullopt;
3218 return getInlineEncodingV216(/*IsFloat=*/true, Lo);
3219}
3220
3221// Whether the given literal can be inlined for a V_PK_* instruction.
3223 switch (OpType) {
3226 return getInlineEncodingV216(false, Literal).has_value();
3229 return getInlineEncodingV216(true, Literal).has_value();
3231 llvm_unreachable("OPERAND_REG_IMM_V2FP16_SPLAT is not supported");
3236 return false;
3237 default:
3238 llvm_unreachable("bad packed operand type");
3239 }
3240}
3241
3242// Whether the given literal can be inlined for a V_PK_*_IU16 instruction.
3246
3247// Whether the given literal can be inlined for a V_PK_*_BF16 instruction.
3251
3252// Whether the given literal can be inlined for a V_PK_*_F16 instruction.
3256
3257// Whether the given literal can be inlined for V_PK_FMAC_F16 instruction.
3259 return getPKFMACF16InlineEncoding(Literal, IsGFX11Plus).has_value();
3260}
3261
3262bool isValid32BitLiteral(uint64_t Val, bool IsFP64) {
3263 if (IsFP64)
3264 return !Lo_32(Val);
3265
3266 return isUInt<32>(Val) || isInt<32>(Val);
3267}
3268
3269int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit) {
3270 switch (Type) {
3271 default:
3272 break;
3277 return Imm & 0xffff;
3291 return Lo_32(Imm);
3294 return IsLit ? Imm : Hi_32(Imm);
3295 }
3296 return Imm;
3297}
3298
3300 const Function *F = A->getParent();
3301
3302 // Arguments to compute shaders are never a source of divergence.
3303 CallingConv::ID CC = F->getCallingConv();
3304 switch (CC) {
3307 return true;
3318 // For non-compute shaders, SGPR inputs are marked with either inreg or
3319 // byval. Everything else is in VGPRs.
3320 return A->hasAttribute(Attribute::InReg) ||
3321 A->hasAttribute(Attribute::ByVal);
3322 default:
3323 // TODO: treat i1 as divergent?
3324 return A->hasAttribute(Attribute::InReg);
3325 }
3326}
3327
3328bool isArgPassedInSGPR(const CallBase *CB, unsigned ArgNo) {
3329 // Arguments to compute shaders are never a source of divergence.
3331 switch (CC) {
3334 return true;
3345 // For non-compute shaders, SGPR inputs are marked with either inreg or
3346 // byval. Everything else is in VGPRs.
3347 return CB->paramHasAttr(ArgNo, Attribute::InReg) ||
3348 CB->paramHasAttr(ArgNo, Attribute::ByVal);
3349 default:
3350 return CB->paramHasAttr(ArgNo, Attribute::InReg);
3351 }
3352}
3353
3354static bool hasSMEMByteOffset(const MCSubtargetInfo &ST) {
3355 return isGCN3Encoding(ST) || isGFX10Plus(ST);
3356}
3357
3359 int64_t EncodedOffset) {
3360 if (isGFX12Plus(ST))
3361 return isUInt<23>(EncodedOffset);
3362
3363 return hasSMEMByteOffset(ST) ? isUInt<20>(EncodedOffset)
3364 : isUInt<8>(EncodedOffset);
3365}
3366
3368 int64_t EncodedOffset, bool IsBuffer) {
3369 if (isGFX12Plus(ST)) {
3370 if (IsBuffer && EncodedOffset < 0)
3371 return false;
3372 return isInt<24>(EncodedOffset);
3373 }
3374
3375 return !IsBuffer && hasSMRDSignedImmOffset(ST) && isInt<21>(EncodedOffset);
3376}
3377
3378static bool isDwordAligned(uint64_t ByteOffset) {
3379 return (ByteOffset & 3) == 0;
3380}
3381
3383 uint64_t ByteOffset) {
3384 if (hasSMEMByteOffset(ST))
3385 return ByteOffset;
3386
3387 assert(isDwordAligned(ByteOffset));
3388 return ByteOffset >> 2;
3389}
3390
3391std::optional<int64_t> getSMRDEncodedOffset(const MCSubtargetInfo &ST,
3392 int64_t ByteOffset, bool IsBuffer,
3393 bool HasSOffset) {
3394 // For unbuffered smem loads, it is illegal for the Immediate Offset to be
3395 // negative if the resulting (Offset + (M0 or SOffset or zero) is negative.
3396 // Handle case where SOffset is not present.
3397 if (!IsBuffer && !HasSOffset && ByteOffset < 0 && hasSMRDSignedImmOffset(ST))
3398 return std::nullopt;
3399
3400 if (isGFX12Plus(ST)) // 24 bit signed offsets
3401 return isInt<24>(ByteOffset) ? std::optional<int64_t>(ByteOffset)
3402 : std::nullopt;
3403
3404 // The signed version is always a byte offset.
3405 if (!IsBuffer && hasSMRDSignedImmOffset(ST)) {
3407 return isInt<20>(ByteOffset) ? std::optional<int64_t>(ByteOffset)
3408 : std::nullopt;
3409 }
3410
3411 if (!isDwordAligned(ByteOffset) && !hasSMEMByteOffset(ST))
3412 return std::nullopt;
3413
3414 int64_t EncodedOffset = convertSMRDOffsetUnits(ST, ByteOffset);
3415 return isLegalSMRDEncodedUnsignedOffset(ST, EncodedOffset)
3416 ? std::optional<int64_t>(EncodedOffset)
3417 : std::nullopt;
3418}
3419
3420std::optional<int64_t> getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST,
3421 int64_t ByteOffset) {
3422 if (!isCI(ST) || !isDwordAligned(ByteOffset))
3423 return std::nullopt;
3424
3425 int64_t EncodedOffset = convertSMRDOffsetUnits(ST, ByteOffset);
3426 return isUInt<32>(EncodedOffset) ? std::optional<int64_t>(EncodedOffset)
3427 : std::nullopt;
3428}
3429
3431 if (ST.getFeatureBits().test(FeatureFlatOffsetBits12))
3432 return 12;
3433 if (ST.getFeatureBits().test(FeatureFlatOffsetBits24))
3434 return 24;
3435 return 13;
3436}
3437
3438namespace {
3439
3440struct SourceOfDivergence {
3441 unsigned Intr;
3442};
3443const SourceOfDivergence *lookupSourceOfDivergence(unsigned Intr);
3444
3445struct AlwaysUniform {
3446 unsigned Intr;
3447};
3448const AlwaysUniform *lookupAlwaysUniform(unsigned Intr);
3449
3450#define GET_SourcesOfDivergence_IMPL
3451#define GET_UniformIntrinsics_IMPL
3452#define GET_Gfx9BufferFormat_IMPL
3453#define GET_Gfx10BufferFormat_IMPL
3454#define GET_Gfx11PlusBufferFormat_IMPL
3455
3456#include "AMDGPUGenSearchableTables.inc"
3457
3458} // end anonymous namespace
3459
3460bool isIntrinsicSourceOfDivergence(unsigned IntrID) {
3461 return lookupSourceOfDivergence(IntrID);
3462}
3463
3464bool isIntrinsicAlwaysUniform(unsigned IntrID) {
3465 return lookupAlwaysUniform(IntrID);
3466}
3467
3469 uint8_t NumComponents,
3470 uint8_t NumFormat,
3471 const MCSubtargetInfo &STI) {
3472 return isGFX11Plus(STI) ? getGfx11PlusBufferFormatInfo(
3473 BitsPerComp, NumComponents, NumFormat)
3474 : isGFX10(STI)
3475 ? getGfx10BufferFormatInfo(BitsPerComp, NumComponents, NumFormat)
3476 : getGfx9BufferFormatInfo(BitsPerComp, NumComponents, NumFormat);
3477}
3478
3480 const MCSubtargetInfo &STI) {
3481 return isGFX11Plus(STI) ? getGfx11PlusBufferFormatInfo(Format)
3482 : isGFX10(STI) ? getGfx10BufferFormatInfo(Format)
3483 : getGfx9BufferFormatInfo(Format);
3484}
3485
3487 const MCRegisterInfo &MRI) {
3488 const unsigned VGPRClasses[] = {
3489 AMDGPU::VGPR_16RegClassID, AMDGPU::VGPR_32RegClassID,
3490 AMDGPU::VReg_64RegClassID, AMDGPU::VReg_96RegClassID,
3491 AMDGPU::VReg_128RegClassID, AMDGPU::VReg_160RegClassID,
3492 AMDGPU::VReg_192RegClassID, AMDGPU::VReg_224RegClassID,
3493 AMDGPU::VReg_256RegClassID, AMDGPU::VReg_288RegClassID,
3494 AMDGPU::VReg_320RegClassID, AMDGPU::VReg_352RegClassID,
3495 AMDGPU::VReg_384RegClassID, AMDGPU::VReg_512RegClassID,
3496 AMDGPU::VReg_1024RegClassID};
3497
3498 for (unsigned RCID : VGPRClasses) {
3499 const MCRegisterClass &RC = MRI.getRegClass(RCID);
3500 if (RC.contains(Reg))
3501 return &RC;
3502 }
3503
3504 return nullptr;
3505}
3506
3508 unsigned Enc = MRI.getEncodingValue(Reg);
3509 unsigned Idx = Enc & AMDGPU::HWEncoding::REG_IDX_MASK;
3510 return Idx >> 8;
3511}
3512
3514 const MCRegisterInfo &MRI) {
3515 unsigned Enc = MRI.getEncodingValue(Reg);
3516 unsigned Idx = Enc & AMDGPU::HWEncoding::REG_IDX_MASK;
3517 if (Idx >= 0x100)
3518 return MCRegister();
3519
3520 const MCRegisterClass *RC = getVGPRPhysRegClass(Reg, MRI);
3521 if (!RC)
3522 return MCRegister();
3523
3524 Idx |= MSBs << 8;
3525 if (RC->getID() == AMDGPU::VGPR_16RegClassID) {
3526 // This class has 2048 registers with interleaved lo16 and hi16.
3527 Idx *= 2;
3529 ++Idx;
3530 }
3531
3532 return RC->getRegister(Idx);
3533}
3534
3535static std::optional<unsigned>
3536convertSetRegImmToVgprMSBs(unsigned Imm, unsigned Simm16,
3537 bool HasSetregVGPRMSBFixup) {
3538 constexpr unsigned VGPRMSBShift =
3540
3541 auto [HwRegId, Offset, Size] = Hwreg::HwregEncoding::decode(Simm16);
3542 if (HwRegId != Hwreg::ID_MODE ||
3543 (!HasSetregVGPRMSBFixup && (Offset + Size) < VGPRMSBShift))
3544 return {};
3545 // If there is SetregVGPRMSBFixup then Offset is ignored.
3546 if (!HasSetregVGPRMSBFixup)
3547 Imm <<= Offset;
3548 Imm = (Imm & Hwreg::VGPR_MSB_MASK) >> VGPRMSBShift;
3549 if (!HasSetregVGPRMSBFixup)
3551 return llvm::rotr<uint8_t>(static_cast<uint8_t>(Imm), /*R=*/2);
3552}
3553
3554std::optional<unsigned> convertSetRegImmToVgprMSBs(const MachineInstr &MI,
3555 bool HasSetregVGPRMSBFixup) {
3556 assert(MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32);
3557 return convertSetRegImmToVgprMSBs(MI.getOperand(0).getImm(),
3558 MI.getOperand(1).getImm(),
3559 HasSetregVGPRMSBFixup);
3560}
3561
3562std::optional<unsigned> convertSetRegImmToVgprMSBs(const MCInst &MI,
3563 bool HasSetregVGPRMSBFixup) {
3564 assert(MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32_gfx12);
3565 return convertSetRegImmToVgprMSBs(MI.getOperand(0).getImm(),
3566 MI.getOperand(1).getImm(),
3567 HasSetregVGPRMSBFixup);
3568}
3569
3570std::pair<const AMDGPU::OpName *, const AMDGPU::OpName *>
3572 static const AMDGPU::OpName VOPOps[4] = {
3573 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2,
3574 AMDGPU::OpName::vdst};
3575 static const AMDGPU::OpName VDSOps[4] = {
3576 AMDGPU::OpName::addr, AMDGPU::OpName::data0, AMDGPU::OpName::data1,
3577 AMDGPU::OpName::vdst};
3578 static const AMDGPU::OpName FLATOps[4] = {
3579 AMDGPU::OpName::vaddr, AMDGPU::OpName::vdata,
3580 AMDGPU::OpName::NUM_OPERAND_NAMES, AMDGPU::OpName::vdst};
3581 static const AMDGPU::OpName BUFOps[4] = {
3582 AMDGPU::OpName::vaddr, AMDGPU::OpName::NUM_OPERAND_NAMES,
3583 AMDGPU::OpName::NUM_OPERAND_NAMES, AMDGPU::OpName::vdata};
3584 static const AMDGPU::OpName VIMGOps[4] = {
3585 AMDGPU::OpName::vaddr0, AMDGPU::OpName::vaddr1, AMDGPU::OpName::vaddr2,
3586 AMDGPU::OpName::vdata};
3587
3588 // For VOPD instructions MSB of a corresponding Y component operand VGPR
3589 // address is supposed to match X operand, otherwise VOPD shall not be
3590 // combined.
3591 static const AMDGPU::OpName VOPDOpsX[4] = {
3592 AMDGPU::OpName::src0X, AMDGPU::OpName::vsrc1X, AMDGPU::OpName::vsrc2X,
3593 AMDGPU::OpName::vdstX};
3594 static const AMDGPU::OpName VOPDOpsY[4] = {
3595 AMDGPU::OpName::src0Y, AMDGPU::OpName::vsrc1Y, AMDGPU::OpName::vsrc2Y,
3596 AMDGPU::OpName::vdstY};
3597
3598 // VOP2 MADMK instructions use src0, imm, src1 scheme.
3599 static const AMDGPU::OpName VOP2MADMKOps[4] = {
3600 AMDGPU::OpName::src0, AMDGPU::OpName::NUM_OPERAND_NAMES,
3601 AMDGPU::OpName::src1, AMDGPU::OpName::vdst};
3602 static const AMDGPU::OpName VOPDFMAMKOpsX[4] = {
3603 AMDGPU::OpName::src0X, AMDGPU::OpName::NUM_OPERAND_NAMES,
3604 AMDGPU::OpName::vsrc1X, AMDGPU::OpName::vdstX};
3605 static const AMDGPU::OpName VOPDFMAMKOpsY[4] = {
3606 AMDGPU::OpName::src0Y, AMDGPU::OpName::NUM_OPERAND_NAMES,
3607 AMDGPU::OpName::vsrc1Y, AMDGPU::OpName::vdstY};
3608
3612 switch (Desc.getOpcode()) {
3613 // LD_SCALE operands ignore MSB.
3614 case AMDGPU::V_WMMA_LD_SCALE_PAIRED_B32:
3615 case AMDGPU::V_WMMA_LD_SCALE_PAIRED_B32_gfx1250:
3616 case AMDGPU::V_WMMA_LD_SCALE16_PAIRED_B64:
3617 case AMDGPU::V_WMMA_LD_SCALE16_PAIRED_B64_gfx1250:
3618 return {};
3619 case AMDGPU::V_FMAMK_F16:
3620 case AMDGPU::V_FMAMK_F16_t16:
3621 case AMDGPU::V_FMAMK_F16_t16_gfx12:
3622 case AMDGPU::V_FMAMK_F16_fake16:
3623 case AMDGPU::V_FMAMK_F16_fake16_gfx12:
3624 case AMDGPU::V_FMAMK_F32:
3625 case AMDGPU::V_FMAMK_F32_gfx12:
3626 case AMDGPU::V_FMAMK_F64:
3627 case AMDGPU::V_FMAMK_F64_gfx1250:
3628 return {VOP2MADMKOps, nullptr};
3629 default:
3630 break;
3631 }
3632 return {VOPOps, nullptr};
3633 }
3634
3636 return {VDSOps, nullptr};
3637
3639 return {FLATOps, nullptr};
3640
3642 return {BUFOps, nullptr};
3643
3645 return {VIMGOps, nullptr};
3646
3647 if (AMDGPU::isVOPD(Desc.getOpcode())) {
3648 auto [OpX, OpY] = getVOPDComponents(Desc.getOpcode());
3649 return {(OpX == AMDGPU::V_FMAMK_F32) ? VOPDFMAMKOpsX : VOPDOpsX,
3650 (OpY == AMDGPU::V_FMAMK_F32) ? VOPDFMAMKOpsY : VOPDOpsY};
3651 }
3652
3654
3656 llvm_unreachable("Sample and export VGPR lowering is not implemented and"
3657 " these instructions are not expected on gfx1250");
3658
3659 return {};
3660}
3661
3662bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode) {
3663 const MCInstrDesc &Desc = MII.get(Opcode);
3665 return Desc.mayLoad() && !Desc.mayStore() && !getSMEMIsBuffer(Opcode);
3667 return false;
3668
3669 // Only SV and SVS modes are supported.
3670 if (SIInstrFlags::isFlatScratch(MII, Opcode))
3671 return hasNamedOperand(Opcode, OpName::vaddr);
3672
3673 // Only GVS mode is supported.
3674 return hasNamedOperand(Opcode, OpName::vaddr) &&
3675 hasNamedOperand(Opcode, OpName::saddr);
3676
3677 return false;
3678}
3679
3680bool hasAny64BitVGPROperands(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
3681 const MCSubtargetInfo &ST) {
3682 for (auto OpName : {OpName::vdst, OpName::src0, OpName::src1, OpName::src2}) {
3683 int Idx = getNamedOperandIdx(OpDesc.getOpcode(), OpName);
3684 if (Idx == -1)
3685 continue;
3686
3687 const MCOperandInfo &OpInfo = OpDesc.operands()[Idx];
3688 int16_t RegClass = MII.getOpRegClassID(
3689 OpInfo, ST.getHwMode(MCSubtargetInfo::HwMode_RegInfo));
3690 if (RegClass == AMDGPU::VReg_64RegClassID ||
3691 RegClass == AMDGPU::VReg_64_Align2RegClassID)
3692 return true;
3693 }
3694
3695 return false;
3696}
3697
3698bool isDPALU_DPP32BitOpc(unsigned Opc) {
3699 switch (Opc) {
3700 case AMDGPU::V_MUL_LO_U32_e64:
3701 case AMDGPU::V_MUL_LO_U32_e64_dpp:
3702 case AMDGPU::V_MUL_LO_U32_e64_dpp_gfx1250:
3703 case AMDGPU::V_MUL_HI_U32_e64:
3704 case AMDGPU::V_MUL_HI_U32_e64_dpp:
3705 case AMDGPU::V_MUL_HI_U32_e64_dpp_gfx1250:
3706 case AMDGPU::V_MUL_HI_I32_e64:
3707 case AMDGPU::V_MUL_HI_I32_e64_dpp:
3708 case AMDGPU::V_MUL_HI_I32_e64_dpp_gfx1250:
3709 case AMDGPU::V_MAD_U32_e64:
3710 case AMDGPU::V_MAD_U32_e64_dpp:
3711 case AMDGPU::V_MAD_U32_e64_dpp_gfx1250:
3712 return true;
3713 default:
3714 return false;
3715 }
3716}
3717
3718bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
3719 const MCSubtargetInfo &ST) {
3720 if (!ST.hasFeature(AMDGPU::FeatureDPALU_DPP))
3721 return false;
3722
3723 if (isDPALU_DPP32BitOpc(OpDesc.getOpcode()))
3724 return ST.hasFeature(AMDGPU::FeatureGFX1250Insts);
3725
3726 return hasAny64BitVGPROperands(OpDesc, MII, ST);
3727}
3728
3730 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
3731 return 64;
3732 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
3733 return 128;
3734 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
3735 return 256;
3736 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
3737 return 320;
3738 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
3739 return 512;
3740 return 64; // In sync with getAddressableLocalMemorySize
3741}
3742
3744 switch (Opc) {
3745 case AMDGPU::V_PK_ADD_F32_gfx1250:
3746 case AMDGPU::V_PK_ADD_F32_gfx1250_gfx12:
3747 case AMDGPU::V_PK_MUL_F32_gfx1250:
3748 case AMDGPU::V_PK_MUL_F32_gfx1250_gfx12:
3749 case AMDGPU::V_PK_FMA_F32_gfx1250:
3750 case AMDGPU::V_PK_FMA_F32_gfx1250_gfx12:
3751 return true;
3752 default:
3753 return false;
3754 }
3755}
3756
3758 switch (Opc) {
3759 case AMDGPU::V_PK_ADD_F64:
3760 case AMDGPU::V_PK_ADD_F64_gfx1250:
3761 case AMDGPU::V_PK_MUL_F64:
3762 case AMDGPU::V_PK_MUL_F64_gfx1250:
3763 case AMDGPU::V_PK_FMA_F64:
3764 case AMDGPU::V_PK_FMA_F64_gfx1250:
3765 case AMDGPU::V_PK_MAX_NUM_F64:
3766 case AMDGPU::V_PK_MAX_NUM_F64_gfx1250:
3767 case AMDGPU::V_PK_MIN_NUM_F64:
3768 case AMDGPU::V_PK_MIN_NUM_F64_gfx1250:
3769 case AMDGPU::V_PK_ADD_NC_U64:
3770 case AMDGPU::V_PK_ADD_NC_U64_gfx1250:
3771 case AMDGPU::V_PK_SUB_NC_U64:
3772 case AMDGPU::V_PK_SUB_NC_U64_gfx1250:
3773 case AMDGPU::V_PK_LSHL_ADD_U64:
3774 case AMDGPU::V_PK_LSHL_ADD_U64_gfx1250:
3775 return true;
3776 default:
3777 return false;
3778 }
3779}
3780
3784
3785const std::array<unsigned, 3> &ClusterDimsAttr::getDims() const {
3786 assert(isFixedDims() && "expect kind to be FixedDims");
3787 return Dims;
3788}
3789
3790std::string ClusterDimsAttr::to_string() const {
3791 SmallString<10> Buffer;
3792 raw_svector_ostream OS(Buffer);
3793
3794 switch (getKind()) {
3795 case Kind::Unknown:
3796 return "";
3797 case Kind::NoCluster: {
3798 OS << EncoNoCluster << ',' << EncoNoCluster << ',' << EncoNoCluster;
3799 return Buffer.c_str();
3800 }
3801 case Kind::VariableDims: {
3802 OS << EncoVariableDims << ',' << EncoVariableDims << ','
3803 << EncoVariableDims;
3804 return Buffer.c_str();
3805 }
3806 case Kind::FixedDims: {
3807 OS << Dims[0] << ',' << Dims[1] << ',' << Dims[2];
3808 return Buffer.c_str();
3809 }
3810 }
3811 llvm_unreachable("Unknown ClusterDimsAttr kind");
3812}
3813
3815 std::optional<SmallVector<unsigned>> Attr =
3816 getIntegerVecAttribute(F, "amdgpu-cluster-dims", /*Size=*/3);
3818
3819 if (!Attr.has_value())
3820 AttrKind = Kind::Unknown;
3821 else if (all_of(*Attr, equal_to(EncoNoCluster)))
3822 AttrKind = Kind::NoCluster;
3823 else if (all_of(*Attr, equal_to(EncoVariableDims)))
3824 AttrKind = Kind::VariableDims;
3825
3826 ClusterDimsAttr A(AttrKind);
3827 if (AttrKind == Kind::FixedDims)
3828 A.Dims = {(*Attr)[0], (*Attr)[1], (*Attr)[2]};
3829
3830 return A;
3831}
3832
3833std::optional<APFloat> evaluateRcp(const APFloat &Val) {
3834 const fltSemantics &Sem = Val.getSemantics();
3835
3836 // v_rcp_f16/bf16 are correctly rounded.
3837 if (&Sem == &APFloat::IEEEhalf() || &Sem == &APFloat::BFloat())
3838 return APFloat::getOne(Sem) / Val;
3839
3840 // v_rcp_f32/f64 always flush a denormal input to zero (preserving sign)
3841 // before reciprocating.
3842 APFloat Arg = Val;
3843 if (Arg.isDenormal())
3844 Arg = APFloat::getZero(Sem, Arg.isNegative());
3845
3846 APFloat Result = APFloat::getOne(Sem) / Arg;
3847
3848 // v_rcp_f32/f64 always flush a denormal result to zero (preserving sign).
3849 if (Result.isDenormal())
3850 Result = APFloat::getZero(Sem, Result.isNegative());
3851
3852 // v_rcp_f32/f64 only approximate the reciprocal, except for these special
3853 // cases where the result is exact.
3854 if (!Result.isZero() && !Result.isInfinity() && !Result.isNaN() &&
3855 !Result.isOne() && !Result.isMinusOne())
3856 return std::nullopt;
3857
3858 return Result;
3859}
3860
3861} // namespace AMDGPU
3862
3864 switch (S) {
3865 case (AMDGPU::TargetIDSetting::Unsupported):
3866 OS << "Unsupported";
3867 break;
3868 case (AMDGPU::TargetIDSetting::Any):
3869 OS << "Any";
3870 break;
3871 case (AMDGPU::TargetIDSetting::Off):
3872 OS << "Off";
3873 break;
3874 case (AMDGPU::TargetIDSetting::On):
3875 OS << "On";
3876 break;
3877 }
3878 return OS;
3879}
3880
3881} // namespace llvm
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static llvm::cl::opt< unsigned > DefaultAMDHSACodeObjectVersion("amdhsa-code-object-version", llvm::cl::Hidden, llvm::cl::init(llvm::AMDGPU::AMDHSA_COV6), llvm::cl::desc("Set default AMDHSA Code Object Version (module flag " "or asm directive still take priority if present)"))
#define MAP_REG2REG
unsigned uint64_t
Provides AMDGPU specific target descriptions.
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file contains the simple types necessary to represent the attributes associated with functions a...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
IRTranslator LLVM IR MI
#define RegName(no)
#define F(x, y, z)
Definition MD5.cpp:54
Register Reg
Register const TargetRegisterInfo * TRI
This file contains the declarations for metadata subclasses.
#define T
uint64_t High
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
#define S_00B848_MEM_ORDERED(x)
Definition SIDefines.h:1482
#define S_00B848_WGP_MODE(x)
Definition SIDefines.h:1479
#define S_00B848_FWD_PROGRESS(x)
Definition SIDefines.h:1485
This file contains some functions that are useful when dealing with strings.
static const int BlockSize
Definition TarWriter.cpp:33
static ClusterDimsAttr get(const Function &F)
const std::array< unsigned, 3 > & getDims() const
void setSramEccSetting(TargetIDSetting NewSramEccSetting)
Sets sramecc setting to NewSramEccSetting.
void setXnackSetting(TargetIDSetting NewXnackSetting)
Sets xnack setting to NewXnackSetting.
unsigned getIndexInParsedOperands(unsigned CompOprIdx) const
unsigned getIndexOfSrcInParsedOperands(unsigned CompSrcIdx) const
std::optional< unsigned > getInvalidCompOperandIndex(std::function< MCRegister(unsigned, unsigned)> GetRegIdx, const MCRegisterInfo &MRI, bool SkipSrc=false, bool AllowSameVGPR=false, bool VOPD3=false) const
std::array< MCRegister, Component::MAX_OPR_NUM > RegIndices
Represents the counter values to wait for in an s_waitcnt instruction.
static const fltSemantics & BFloat()
Definition APFloat.h:303
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
bool isNegative() const
Definition APFloat.h:1575
bool isDenormal() const
Definition APFloat.h:1576
const fltSemantics & getSemantics() const
Definition APFloat.h:1583
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1184
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1175
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
constexpr bool test(unsigned I) const
unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
bool mayStore() const
Return true if this instruction could possibly modify memory.
bool mayLoad() const
Return true if this instruction could possibly read memory.
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
unsigned getOpcode() const
Return the opcode number for this descriptor.
Interface to description of machine instruction set.
Definition MCInstrInfo.h:27
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Definition MCInstrInfo.h:79
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
MCRegisterClass - Base class of TargetRegisterClass.
unsigned getID() const
getID() - Return the register class ID number.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
uint16_t getEncodingValue(MCRegister Reg) const
Returns the encoding for Reg.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr unsigned id() const
Definition MCRegister.h:82
Generic base class for all target subtargets.
bool hasFeature(unsigned Feature) const
const Triple & getTargetTriple() const
const FeatureBitset & getFeatureBits() const
StringRef getCPU() const
Metadata node.
Definition Metadata.h:1069
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1426
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1432
Representation of each machine instruction.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
const char * c_str()
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
Manages the enabling and disabling of subtarget specific features.
const std::vector< std::string > & getFeatures() const
Returns the vector of individual subtarget features.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
OSType getOS() const
Get the parsed operating system type of this triple.
Definition Triple.h:522
ArchType getArch() const
Get the parsed architecture type of this triple.
Definition Triple.h:513
bool isAMDGCN() const
Tests whether the target is AMDGCN.
Definition Triple.h:991
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an SmallVector or SmallString.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
unsigned decodeFieldVaVcc(unsigned Encoded)
unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc)
unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version)
bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val, bool &IsDefault, const MCSubtargetInfo &STI)
unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt, const IsaVersion &Version)
unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc)
unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst)
unsigned decodeFieldSaSdst(unsigned Encoded)
unsigned getHoldCntBitMask(const IsaVersion &Version)
unsigned decodeFieldVaSdst(unsigned Encoded)
unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc)
unsigned decodeFieldVaSsrc(unsigned Encoded)
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
const CustomOperandVal DepCtrInfo[]
bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal, const MCSubtargetInfo &STI)
unsigned decodeFieldVaVdst(unsigned Encoded)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
unsigned decodeFieldVmVsrc(unsigned Encoded)
unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
static constexpr ExpTgt ExpTgtInfo[]
bool getTgtName(unsigned Id, StringRef &Name, int &Index)
unsigned getTgtId(const StringRef Name)
constexpr uint32_t VersionMinor
HSA metadata minor version.
constexpr uint32_t VersionMajor
HSA metadata major version.
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize)
static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI)
bool isSGPROccupancyLimited(const MCSubtargetInfo &STI)
unsigned getArchVGPRAllocGranule()
For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage, returns the allocation granule...
static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI)
static unsigned getSGPRTrapHandlerReserve(const MCSubtargetInfo &STI)
unsigned getMinFlatWorkGroupSize(const MCSubtargetInfo &STI)
unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, std::optional< bool > EnableWavefrontSize32)
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU)
unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, bool Addressable)
unsigned getWavefrontSize(const MCSubtargetInfo &STI)
unsigned getWavesPerEUForWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getInstCacheLineSize(const MCSubtargetInfo &STI)
static constexpr unsigned MaxDynamicVGPRBlocks
Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
static unsigned getSGPRBudgetPerWave(unsigned TotalNumSGPRs, unsigned WavesPerEU, unsigned TrapReserve, unsigned Granule)
unsigned getTotalNumVGPRs(const MCSubtargetInfo &STI)
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getWavesPerWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves, unsigned TotalNumSGPRs, unsigned Granule, unsigned TrapReserve)
unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs)
unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed, bool FlatScrUsed, bool XNACKUsed)
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
static unsigned getGranulatedNumRegisterBlocks(unsigned NumRegs, unsigned Granule)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
StringLiteral const UfmtSymbolicGFX11[]
bool isValidUnifiedFormat(unsigned Id, const MCSubtargetInfo &STI)
unsigned getDefaultFormatEncoding(const MCSubtargetInfo &STI)
StringRef getUnifiedFormatName(unsigned Id, const MCSubtargetInfo &STI)
unsigned const DfmtNfmt2UFmtGFX10[]
StringLiteral const DfmtSymbolic[]
static StringLiteral const * getNfmtLookupTable(const MCSubtargetInfo &STI)
bool isValidNfmt(unsigned Id, const MCSubtargetInfo &STI)
StringLiteral const NfmtSymbolicGFX10[]
bool isValidDfmtNfmt(unsigned Id, const MCSubtargetInfo &STI)
int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt, const MCSubtargetInfo &STI)
StringRef getDfmtName(unsigned Id)
int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt)
int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI)
bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI)
StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI)
unsigned const DfmtNfmt2UFmtGFX11[]
StringLiteral const NfmtSymbolicVI[]
StringLiteral const NfmtSymbolicSICI[]
int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI)
int64_t getDfmt(const StringRef Name)
StringLiteral const UfmtSymbolicGFX10[]
void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt)
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId, uint16_t &StreamId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI)
Returns true if the message does not use the m0 operand.
StringRef getMsgOpName(int64_t MsgId, uint64_t Encoding, const MCSubtargetInfo &STI)
Map from an encoding to the symbolic name for a sendmsg operation.
static uint64_t getMsgIdMask(const MCSubtargetInfo &STI)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
constexpr unsigned VOPD_VGPR_BANK_MASKS[]
constexpr unsigned COMPONENTS_NUM
constexpr unsigned VOPD3_VGPR_BANK_MASKS[]
bool isGCN3Encoding(const MCSubtargetInfo &STI)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isGFX10_BEncoding(const MCSubtargetInfo &STI)
bool isInlineValue(MCRegister Reg)
bool isGFX10_GFX11(const MCSubtargetInfo &STI)
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
uint64_t convertSMRDOffsetUnits(const MCSubtargetInfo &ST, uint64_t ByteOffset)
Convert ByteOffset to dwords if the subtarget uses dword SMRD immediate offsets.
static unsigned encodeStorecnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Storecnt)
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
static bool hasSMEMByteOffset(const MCSubtargetInfo &ST)
LLVM_ABI unsigned getMaxWavesPerEU(GPUKind AK)
bool isVOPCAsmOnly(unsigned Opc)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords)
bool getMTBUFHasSrsrc(unsigned Opc)
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool getWMMAIsXDL(unsigned Opc)
static std::optional< unsigned > convertSetRegImmToVgprMSBs(unsigned Imm, unsigned Simm16, bool HasSetregVGPRMSBFixup)
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
static bool isSymbolicCustomOperandEncoding(const CustomOperandVal *Opr, int Size, unsigned Code, bool &HasNonDefaultVal, const MCSubtargetInfo &STI)
bool isGFX10Before1030(const MCSubtargetInfo &STI)
bool isSISrcInlinableOperand(const MCInstrDesc &Desc, unsigned OpNo)
Does this operand support only inlinable literals?
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
const int OPR_ID_UNSUPPORTED
void initDefaultAMDKernelCodeT(AMDGPUMCKernelCodeT &KernelCode, const MCSubtargetInfo &STI)
bool shouldEmitConstantsToTextSection(const Triple &TT)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
int getMTBUFElements(unsigned Opc)
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
static int encodeCustomOperandVal(const CustomOperandVal &Op, int64_t InputVal)
unsigned getTemporalHintType(const MCInstrDesc TID)
int32_t getTotalNumVGPRs(bool has90AInsts, int32_t ArgNumAGPR, int32_t ArgNumVGPR)
bool isGFX10(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV216(bool IsFloat, uint32_t Literal)
FPType getFPDstSelType(unsigned Opc)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
const MCRegisterClass * getVGPRPhysRegClass(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned encodeLoadcntDscnt(const IsaVersion &Version, const Waitcnt &Decoded)
bool getHasMatrixScale(unsigned Opc)
bool hasPackedD16(const MCSubtargetInfo &STI)
unsigned getStorecntBitMask(const IsaVersion &Version)
bool isFullSIMDMode(const MCSubtargetInfo &STI)
unsigned getLdsDwGranularity(const MCSubtargetInfo &ST)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
bool isGFX10_3_GFX11(const MCSubtargetInfo &STI)
bool isGFX13(const MCSubtargetInfo &STI)
unsigned getAsynccntBitMask(const IsaVersion &Version)
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val)
Checks if Val is inside MD, a !range-like metadata.
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI, StringRef FeatureString)
Construct TargetID from MCSubtargetInfo.
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
unsigned getVOPDOpcode(unsigned Opc, bool VOPD3)
bool isGroupSegment(const GlobalValue *GV)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
bool getMTBUFHasSoffset(unsigned Opc)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool hasXNACK(const MCSubtargetInfo &STI)
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
static unsigned getCombinedCountBitMask(const IsaVersion &Version, bool IsStore)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
bool isVOPC64DPP(unsigned Opc)
int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool getMAIIsGFX940XDL(unsigned Opc)
bool isSI(const MCSubtargetInfo &STI)
unsigned getDefaultAMDHSACodeObjectVersion()
LLVM_ABI unsigned getTotalNumSGPRs(GPUKind AK)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
bool isReadOnlySegment(const GlobalValue *GV)
Waitcnt decodeWaitcnt(const IsaVersion &Version, unsigned Encoded)
bool isArgPassedInSGPR(const Argument *A)
bool isIntrinsicAlwaysUniform(unsigned IntrID)
int getMUBUFBaseOpcode(unsigned Opc)
unsigned encodeWaitcnt(const IsaVersion &Version, const Waitcnt &Decoded)
unsigned getAMDHSACodeObjectVersion(const Module &M)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool getVOP3IsSingle(unsigned Opc)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
bool isGFX9(const MCSubtargetInfo &STI)
bool isDPALU_DPP32BitOpc(unsigned Opc)
bool getVOP1IsSingle(unsigned Opc)
static bool isDwordAligned(uint64_t ByteOffset)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
bool getHasColorExport(const Function &F)
GPUKind
GPU kinds supported by the AMDGPU target.
int getMTBUFBaseOpcode(unsigned Opc)
bool isGFX90A(const MCSubtargetInfo &STI)
unsigned getSamplecntBitMask(const IsaVersion &Version)
unsigned getDefaultQueueImplicitArgPosition(unsigned CodeObjectVersion)
std::tuple< char, unsigned, unsigned > parseAsmPhysRegName(StringRef RegName)
Returns a valid charcode or 0 in the first entry if this is a valid physical register name.
bool getHasDepthExport(const Function &F)
bool isGFX8_GFX9_GFX10(const MCSubtargetInfo &STI)
bool getMUBUFHasVAddr(unsigned Opc)
bool isTrue16Inst(unsigned Opc)
LLVM_ABI unsigned getSGPRAllocGranule(GPUKind AK)
unsigned getVGPREncodingMSBs(MCRegister Reg, const MCRegisterInfo &MRI)
std::pair< unsigned, unsigned > getVOPDComponents(unsigned VOPDOpcode)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned getInitialPSInputAddr(const Function &F)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool isAsyncStore(unsigned Opc)
unsigned getDynamicVGPRBlockSize(const Function &F)
unsigned getKmcntBitMask(const IsaVersion &Version)
MCRegister getVGPRWithMSBs(MCRegister Reg, unsigned MSBs, const MCRegisterInfo &MRI)
If Reg is a low VGPR return a corresponding high VGPR with MSBs set.
unsigned getVmcntBitMask(const IsaVersion &Version)
bool isNotGFX10Plus(const MCSubtargetInfo &STI)
bool hasMAIInsts(const MCSubtargetInfo &STI)
unsigned getBitOp2(unsigned Opc)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
unsigned getXcntBitMask(const IsaVersion &Version)
bool isGenericAtomic(unsigned Opc)
const MFMA_F8F6F4_Info * getWMMA_F8F6F4_WithFormatArgs(unsigned FmtA, unsigned FmtB, unsigned F8F8Opcode)
bool isGFX8Plus(const MCSubtargetInfo &STI)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getLgkmcntBitMask(const IsaVersion &Version)
bool getMUBUFTfe(unsigned Opc)
unsigned getBvhcntBitMask(const IsaVersion &Version)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasGFX10_3Insts(const MCSubtargetInfo &STI)
unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt)
std::pair< const AMDGPU::OpName *, const AMDGPU::OpName * > getVGPRLoweringOperandTables(const MCInstrDesc &Desc)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool isGFX13Plus(const MCSubtargetInfo &STI)
unsigned getExpcntBitMask(const IsaVersion &Version)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
bool getMUBUFHasSoffset(unsigned Opc)
bool isNotGFX11Plus(const MCSubtargetInfo &STI)
bool isGFX11Plus(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV2F16(uint32_t Literal)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
std::tuple< char, unsigned, unsigned > parseAsmConstraintPhysReg(StringRef Constraint)
Returns a valid charcode or 0 in the first entry if this is a valid physical register constraint.
unsigned getHostcallImplicitArgPosition(unsigned CodeObjectVersion)
static unsigned getDefaultCustomOperandEncoding(const CustomOperandVal *Opr, int Size, const MCSubtargetInfo &STI)
static unsigned encodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Loadcnt)
bool isGFX10Plus(const MCSubtargetInfo &STI)
static bool decodeCustomOperand(const CustomOperandVal *Opr, int Size, unsigned Code, int &Idx, StringRef &Name, unsigned &Val, bool &IsDefault, const MCSubtargetInfo &STI)
static bool isValidRegPrefix(char C)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
AMDGPU::TargetID TargetID
bool isGlobalSegment(const GlobalValue *GV)
SmallVector< unsigned > getMaxNumWorkGroups(const Function &F)
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:446
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_C_LAST
Definition SIDefines.h:487
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:439
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:455
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:452
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:457
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:441
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:438
@ OPERAND_REG_INLINE_AC_FIRST
Definition SIDefines.h:489
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:440
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:443
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:458
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:469
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:470
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:444
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FIRST
Definition SIDefines.h:486
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:454
@ OPERAND_REG_INLINE_AC_LAST
Definition SIDefines.h:490
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:450
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:456
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:471
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:453
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:461
std::optional< unsigned > getPKFMACF16InlineEncoding(uint32_t Literal, bool IsGFX11Plus)
bool isNotGFX9Plus(const MCSubtargetInfo &STI)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isCvt_F32_Fp8_Bf8_e64(unsigned Opc)
std::optional< unsigned > getInlineEncodingV2I16(uint32_t Literal)
unsigned encodeStorecntDscnt(const IsaVersion &Version, const Waitcnt &Decoded)
bool isGFX1250(const MCSubtargetInfo &STI)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool isTensorStore(unsigned Opc)
bool getMUBUFIsBufferInv(unsigned Opc)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
std::optional< unsigned > getInlineEncodingV2BF16(uint32_t Literal)
static int encodeCustomOperand(const CustomOperandVal *Opr, int Size, const StringRef Name, int64_t InputVal, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
bool isMAC(unsigned Opc)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
bool getVOP2IsSingle(unsigned Opc)
bool getMAIIsDGEMM(unsigned Opc)
Returns true if MAI operation is a double precision GEMM.
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const int OPR_ID_UNKNOWN
unsigned getCompletionActionImplicitArgPosition(unsigned CodeObjectVersion)
SmallVector< unsigned > getIntegerVecAttribute(const Function &F, StringRef Name, unsigned Size, unsigned DefaultVal)
unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt)
bool isGFX1250Plus(const MCSubtargetInfo &STI)
int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels)
bool isNotGFX12Plus(const MCSubtargetInfo &STI)
bool getMTBUFHasVAddr(unsigned Opc)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
uint8_t getELFABIVersion(const Triple &T, unsigned CodeObjectVersion)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
unsigned getLoadcntBitMask(const IsaVersion &Version)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily, bool VOPD3)
static unsigned encodeDscnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Dscnt)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
const MFMA_F8F6F4_Info * getMFMA_F8F6F4_WithFormatArgs(unsigned CBSZ, unsigned BLGP, unsigned F8F8Opcode)
unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getMultigridSyncArgImplicitArgPosition(unsigned CodeObjectVersion)
bool isGFX9_GFX10_GFX11(const MCSubtargetInfo &STI)
bool isGFX9_GFX10(const MCSubtargetInfo &STI)
int getMUBUFElements(unsigned Opc)
const GcnBufferFormatInfo * getGcnBufferFormatInfo(uint8_t BitsPerComp, uint8_t NumComponents, uint8_t NumFormat, const MCSubtargetInfo &STI)
unsigned mapWMMA3AddrTo2AddrOpcode(unsigned Opc)
bool isPermlane16(unsigned Opc)
bool getMUBUFHasSrsrc(unsigned Opc)
unsigned getDscntBitMask(const IsaVersion &Version)
bool hasAny64BitVGPROperands(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ ELFABIVERSION_AMDGPU_HSA_V4
Definition ELF.h:384
@ ELFABIVERSION_AMDGPU_HSA_V5
Definition ELF.h:385
@ ELFABIVERSION_AMDGPU_HSA_V6
Definition ELF.h:386
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:236
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:239
constexpr bool isVOP1(const T &...O)
Definition SIDefines.h:230
constexpr bool isVOP2(const T &...O)
Definition SIDefines.h:233
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:286
constexpr bool isBuffer(const T &...O)
Definition SIDefines.h:267
constexpr bool isVIMAGE(const T &...O)
Definition SIDefines.h:277
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:271
constexpr bool isVOP3Like(const T &...O)
Definition SIDefines.h:245
constexpr bool isFlatScratch(const T &...O)
Definition SIDefines.h:361
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:274
constexpr bool isVOPD3(const T &...O)
Definition SIDefines.h:385
constexpr bool isEXP(const T &...O)
Definition SIDefines.h:283
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:280
constexpr bool isDS(const T &...O)
Definition SIDefines.h:289
constexpr bool isAtomic(const T &...O)
Definition SIDefines.h:396
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:255
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract_or_null(Y &&MD)
Extract a Value from Metadata, allowing null.
Definition Metadata.h:683
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:668
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:578
constexpr T rotr(T V, int R)
Definition bit.h:399
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
std::string utostr(uint64_t X, bool isNeg=false)
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
Definition STLExtras.h:2173
Op::Description Desc
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
To bit_cast(const From &from) noexcept
Definition bit.h:90
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
constexpr int countr_zero_constexpr(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:190
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
@ AlwaysUniform
The result value is always uniform.
Definition Uniformity.h:23
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
#define N
AMD Kernel Code Object (amd_kernel_code_t).
static std::tuple< typename Fields::ValueType... > decode(uint64_t Encoded)
Instruction set architecture version.