LLVM 24.0.0git
AMDGPUBaseInfo.cpp
Go to the documentation of this file.
1//===- AMDGPUBaseInfo.cpp - AMDGPU Base encoding information --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "AMDGPUBaseInfo.h"
10#include "AMDGPU.h"
11#include "AMDGPUAsmUtils.h"
12#include "AMDKernelCodeT.h"
17#include "llvm/IR/Attributes.h"
18#include "llvm/IR/Constants.h"
19#include "llvm/IR/Function.h"
20#include "llvm/IR/GlobalValue.h"
21#include "llvm/IR/IntrinsicsAMDGPU.h"
22#include "llvm/IR/IntrinsicsR600.h"
23#include "llvm/IR/LLVMContext.h"
24#include "llvm/IR/Metadata.h"
25#include "llvm/MC/MCInstrInfo.h"
30#include <optional>
31
32#define GET_INSTRINFO_NAMED_OPS
33#define GET_INSTRMAP_INFO
34#include "AMDGPUGenInstrInfo.inc"
35
37 "amdhsa-code-object-version", llvm::cl::Hidden,
39 llvm::cl::desc("Set default AMDHSA Code Object Version (module flag "
40 "or asm directive still take priority if present)"));
41
42namespace {
43
44/// \returns Bit mask for given bit \p Shift and bit \p Width.
45unsigned getBitMask(unsigned Shift, unsigned Width) {
46 return ((1 << Width) - 1) << Shift;
47}
48
49/// Packs \p Src into \p Dst for given bit \p Shift and bit \p Width.
50///
51/// \returns Packed \p Dst.
52unsigned packBits(unsigned Src, unsigned Dst, unsigned Shift, unsigned Width) {
53 unsigned Mask = getBitMask(Shift, Width);
54 return ((Src << Shift) & Mask) | (Dst & ~Mask);
55}
56
57/// Unpacks bits from \p Src for given bit \p Shift and bit \p Width.
58///
59/// \returns Unpacked bits.
60unsigned unpackBits(unsigned Src, unsigned Shift, unsigned Width) {
61 return (Src & getBitMask(Shift, Width)) >> Shift;
62}
63
64/// \returns Vmcnt bit shift (lower bits).
65unsigned getVmcntBitShiftLo(unsigned VersionMajor) {
66 return VersionMajor >= 11 ? 10 : 0;
67}
68
69/// \returns Vmcnt bit width (lower bits).
70unsigned getVmcntBitWidthLo(unsigned VersionMajor) {
71 return VersionMajor >= 11 ? 6 : 4;
72}
73
74/// \returns Expcnt bit shift.
75unsigned getExpcntBitShift(unsigned VersionMajor) {
76 return VersionMajor >= 11 ? 0 : 4;
77}
78
79/// \returns Expcnt bit width.
80unsigned getExpcntBitWidth(unsigned VersionMajor) { return 3; }
81
82/// \returns Lgkmcnt bit shift.
83unsigned getLgkmcntBitShift(unsigned VersionMajor) {
84 return VersionMajor >= 11 ? 4 : 8;
85}
86
87/// \returns Lgkmcnt bit width.
88unsigned getLgkmcntBitWidth(unsigned VersionMajor) {
89 return VersionMajor >= 10 ? 6 : 4;
90}
91
92/// \returns Vmcnt bit shift (higher bits).
93unsigned getVmcntBitShiftHi(unsigned VersionMajor) { return 14; }
94
95/// \returns Vmcnt bit width (higher bits).
96unsigned getVmcntBitWidthHi(unsigned VersionMajor) {
97 return (VersionMajor == 9 || VersionMajor == 10) ? 2 : 0;
98}
99
100/// \returns Loadcnt bit width
101unsigned getLoadcntBitWidth(unsigned VersionMajor) {
102 return VersionMajor >= 12 ? 6 : 0;
103}
104
105/// \returns Samplecnt bit width.
106unsigned getSamplecntBitWidth(unsigned VersionMajor) {
107 return VersionMajor >= 12 ? 6 : 0;
108}
109
110/// \returns Bvhcnt bit width.
111unsigned getBvhcntBitWidth(unsigned VersionMajor) {
112 return VersionMajor >= 12 ? 3 : 0;
113}
114
115/// \returns Dscnt bit width.
116unsigned getDscntBitWidth(unsigned VersionMajor) {
117 return VersionMajor >= 12 ? 6 : 0;
118}
119
120/// \returns Dscnt bit shift in combined S_WAIT instructions.
121unsigned getDscntBitShift(unsigned VersionMajor) { return 0; }
122
123/// \returns Storecnt or Vscnt bit width, depending on VersionMajor.
124unsigned getStorecntBitWidth(unsigned VersionMajor) {
125 return VersionMajor >= 10 ? 6 : 0;
126}
127
128/// \returns Kmcnt bit width.
129unsigned getKmcntBitWidth(unsigned VersionMajor) {
130 return VersionMajor >= 12 ? 5 : 0;
131}
132
133/// \returns Xcnt bit width.
134unsigned getXcntBitWidth(unsigned VersionMajor, unsigned VersionMinor) {
135 return VersionMajor == 12 && VersionMinor == 5 ? 6 : 0;
136}
137
138/// \returns Asynccnt bit width.
139unsigned getAsynccntBitWidth(unsigned VersionMajor, unsigned VersionMinor) {
140 return VersionMajor == 12 && VersionMinor == 5 ? 6 : 0;
141}
142
143/// \returns shift for Loadcnt/Storecnt in combined S_WAIT instructions.
144unsigned getLoadcntStorecntBitShift(unsigned VersionMajor) {
145 return VersionMajor >= 12 ? 8 : 0;
146}
147
148/// \returns VaSdst bit width
149inline unsigned getVaSdstBitWidth() { return 3; }
150
151/// \returns VaSdst bit shift
152inline unsigned getVaSdstBitShift() { return 9; }
153
154/// \returns VmVsrc bit width
155inline unsigned getVmVsrcBitWidth() { return 3; }
156
157/// \returns VmVsrc bit shift
158inline unsigned getVmVsrcBitShift() { return 2; }
159
160/// \returns VaVdst bit width
161inline unsigned getVaVdstBitWidth() { return 4; }
162
163/// \returns VaVdst bit shift
164inline unsigned getVaVdstBitShift() { return 12; }
165
166/// \returns VaVcc bit width
167inline unsigned getVaVccBitWidth() { return 1; }
168
169/// \returns VaVcc bit shift
170inline unsigned getVaVccBitShift() { return 1; }
171
172/// \returns SaSdst bit width
173inline unsigned getSaSdstBitWidth() { return 1; }
174
175/// \returns SaSdst bit shift
176inline unsigned getSaSdstBitShift() { return 0; }
177
178/// \returns VaSsrc width
179inline unsigned getVaSsrcBitWidth() { return 1; }
180
181/// \returns VaSsrc bit shift
182inline unsigned getVaSsrcBitShift() { return 8; }
183
184/// \returns HoldCnt bit shift
185inline unsigned getHoldCntWidth(unsigned VersionMajor, unsigned VersionMinor) {
186 static constexpr const unsigned MinMajor = 10;
187 static constexpr const unsigned MinMinor = 3;
188 return std::tie(VersionMajor, VersionMinor) >= std::tie(MinMajor, MinMinor)
189 ? 1
190 : 0;
191}
192
193/// \returns HoldCnt bit shift
194inline unsigned getHoldCntBitShift() { return 7; }
195
196} // end anonymous namespace
197
198namespace llvm {
199
200namespace AMDGPU {
201
202/// \returns true if the target supports signed immediate offset for SMRD
203/// instructions.
205 return isGFX9Plus(ST);
206}
207
208/// \returns True if \p STI is AMDHSA.
209bool isHsaAbi(const MCSubtargetInfo &STI) {
210 return STI.getTargetTriple().getOS() == Triple::AMDHSA;
211}
212
215 M.getModuleFlag("amdhsa_code_object_version"))) {
216 return (unsigned)Ver->getZExtValue() / 100;
217 }
218
220}
221
225
226unsigned getAMDHSACodeObjectVersion(unsigned ABIVersion) {
227 switch (ABIVersion) {
229 return 4;
231 return 5;
233 return 6;
234 default:
236 }
237}
238
239uint8_t getELFABIVersion(const Triple &T, unsigned CodeObjectVersion) {
240 if (T.getOS() != Triple::AMDHSA)
241 return 0;
242
243 switch (CodeObjectVersion) {
244 case 4:
246 case 5:
248 case 6:
250 default:
251 report_fatal_error("Unsupported AMDHSA Code Object Version " +
252 Twine(CodeObjectVersion));
253 }
254}
255
256unsigned getMultigridSyncArgImplicitArgPosition(unsigned CodeObjectVersion) {
257 switch (CodeObjectVersion) {
258 case AMDHSA_COV4:
259 return 48;
260 case AMDHSA_COV5:
261 case AMDHSA_COV6:
262 default:
264 }
265}
266
267// FIXME: All such magic numbers about the ABI should be in a
268// central TD file.
269unsigned getHostcallImplicitArgPosition(unsigned CodeObjectVersion) {
270 switch (CodeObjectVersion) {
271 case AMDHSA_COV4:
272 return 24;
273 case AMDHSA_COV5:
274 case AMDHSA_COV6:
275 default:
277 }
278}
279
280unsigned getDefaultQueueImplicitArgPosition(unsigned CodeObjectVersion) {
281 switch (CodeObjectVersion) {
282 case AMDHSA_COV4:
283 return 32;
284 case AMDHSA_COV5:
285 case AMDHSA_COV6:
286 default:
288 }
289}
290
291unsigned getCompletionActionImplicitArgPosition(unsigned CodeObjectVersion) {
292 switch (CodeObjectVersion) {
293 case AMDHSA_COV4:
294 return 40;
295 case AMDHSA_COV5:
296 case AMDHSA_COV6:
297 default:
299 }
300}
301
302#define GET_MIMGBaseOpcodesTable_IMPL
303#define GET_MIMGDimInfoTable_IMPL
304#define GET_MIMGInfoTable_IMPL
305#define GET_MIMGLZMappingTable_IMPL
306#define GET_MIMGMIPMappingTable_IMPL
307#define GET_MIMGBiasMappingTable_IMPL
308#define GET_MIMGOffsetMappingTable_IMPL
309#define GET_MIMGG16MappingTable_IMPL
310#define GET_MAIInstInfoTable_IMPL
311#define GET_WMMAInstInfoTable_IMPL
312#include "AMDGPUGenSearchableTables.inc"
313
314int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding,
315 unsigned VDataDwords, unsigned VAddrDwords) {
316 const MIMGInfo *Info =
317 getMIMGOpcodeHelper(BaseOpcode, MIMGEncoding, VDataDwords, VAddrDwords);
318 return Info ? Info->Opcode : -1;
319}
320
322 const MIMGInfo *Info = getMIMGInfo(Opc);
323 return Info ? getMIMGBaseOpcodeInfo(Info->BaseOpcode) : nullptr;
324}
325
326int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels) {
327 const MIMGInfo *OrigInfo = getMIMGInfo(Opc);
328 const MIMGInfo *NewInfo =
329 getMIMGOpcodeHelper(OrigInfo->BaseOpcode, OrigInfo->MIMGEncoding,
330 NewChannels, OrigInfo->VAddrDwords);
331 return NewInfo ? NewInfo->Opcode : -1;
332}
333
334unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode,
335 const MIMGDimInfo *Dim, bool IsA16,
336 bool IsG16Supported) {
337 unsigned AddrWords = BaseOpcode->NumExtraArgs;
338 unsigned AddrComponents = (BaseOpcode->Coordinates ? Dim->NumCoords : 0) +
339 (BaseOpcode->LodOrClampOrMip ? 1 : 0);
340 if (IsA16)
341 AddrWords += divideCeil(AddrComponents, 2);
342 else
343 AddrWords += AddrComponents;
344
345 // Note: For subtargets that support A16 but not G16, enabling A16 also
346 // enables 16 bit gradients.
347 // For subtargets that support A16 (operand) and G16 (done with a different
348 // instruction encoding), they are independent.
349
350 if (BaseOpcode->Gradients) {
351 if ((IsA16 && !IsG16Supported) || BaseOpcode->G16)
352 // There are two gradients per coordinate, we pack them separately.
353 // For the 3d case,
354 // we get (dy/du, dx/du) (-, dz/du) (dy/dv, dx/dv) (-, dz/dv)
355 AddrWords += alignTo<2>(Dim->NumGradients / 2);
356 else
357 AddrWords += Dim->NumGradients;
358 }
359 return AddrWords;
360}
361
372
381
386
391
395
399
403
408
416
421
424 bool IsX;
425 bool IsY;
426};
427
428#define GET_FP4FP8DstByteSelTable_DECL
429#define GET_FP4FP8DstByteSelTable_IMPL
430
435
441
442#define GET_DPMACCInstructionTable_DECL
443#define GET_DPMACCInstructionTable_IMPL
444#define GET_MTBUFInfoTable_DECL
445#define GET_MTBUFInfoTable_IMPL
446#define GET_MUBUFInfoTable_DECL
447#define GET_MUBUFInfoTable_IMPL
448#define GET_SMInfoTable_DECL
449#define GET_SMInfoTable_IMPL
450#define GET_VOP1InfoTable_DECL
451#define GET_VOP1InfoTable_IMPL
452#define GET_VOP2InfoTable_DECL
453#define GET_VOP2InfoTable_IMPL
454#define GET_VOP3InfoTable_DECL
455#define GET_VOP3InfoTable_IMPL
456#define GET_VOPC64DPPTable_DECL
457#define GET_VOPC64DPPTable_IMPL
458#define GET_VOPC64DPP8Table_DECL
459#define GET_VOPC64DPP8Table_IMPL
460#define GET_VOPCAsmOnlyInfoTable_DECL
461#define GET_VOPCAsmOnlyInfoTable_IMPL
462#define GET_VOP3CAsmOnlyInfoTable_DECL
463#define GET_VOP3CAsmOnlyInfoTable_IMPL
464#define GET_VOPDComponentTable_DECL
465#define GET_VOPDComponentTable_IMPL
466#define GET_VOPDPairs_DECL
467#define GET_VOPDPairs_IMPL
468#define GET_VOPDXYTable_DECL
469#define GET_VOPDXYTable_IMPL
470#define GET_VOPTrue16Table_DECL
471#define GET_VOPTrue16Table_IMPL
472#define GET_True16D16Table_IMPL
473#define GET_WMMAOpcode2AddrMappingTable_DECL
474#define GET_WMMAOpcode2AddrMappingTable_IMPL
475#define GET_WMMAOpcode3AddrMappingTable_DECL
476#define GET_WMMAOpcode3AddrMappingTable_IMPL
477#define GET_getMFMA_F8F6F4_WithSize_DECL
478#define GET_getMFMA_F8F6F4_WithSize_IMPL
479#define GET_isMFMA_F8F6F4Table_IMPL
480#define GET_isCvtScaleF32_F32F16ToF8F4Table_IMPL
481
482#include "AMDGPUGenSearchableTables.inc"
483
484int getMTBUFBaseOpcode(unsigned Opc) {
485 const MTBUFInfo *Info = getMTBUFInfoFromOpcode(Opc);
486 return Info ? Info->BaseOpcode : -1;
487}
488
489int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements) {
490 const MTBUFInfo *Info =
491 getMTBUFInfoFromBaseOpcodeAndElements(BaseOpc, Elements);
492 return Info ? Info->Opcode : -1;
493}
494
495int getMTBUFElements(unsigned Opc) {
496 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
497 return Info ? Info->elements : 0;
498}
499
500bool getMTBUFHasVAddr(unsigned Opc) {
501 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
502 return Info && Info->has_vaddr;
503}
504
505bool getMTBUFHasSrsrc(unsigned Opc) {
506 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
507 return Info && Info->has_srsrc;
508}
509
510bool getMTBUFHasSoffset(unsigned Opc) {
511 const MTBUFInfo *Info = getMTBUFOpcodeHelper(Opc);
512 return Info && Info->has_soffset;
513}
514
515int getMUBUFBaseOpcode(unsigned Opc) {
516 const MUBUFInfo *Info = getMUBUFInfoFromOpcode(Opc);
517 return Info ? Info->BaseOpcode : -1;
518}
519
520int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements) {
521 const MUBUFInfo *Info =
522 getMUBUFInfoFromBaseOpcodeAndElements(BaseOpc, Elements);
523 return Info ? Info->Opcode : -1;
524}
525
526int getMUBUFElements(unsigned Opc) {
527 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
528 return Info ? Info->elements : 0;
529}
530
531bool getMUBUFHasVAddr(unsigned Opc) {
532 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
533 return Info && Info->has_vaddr;
534}
535
536bool getMUBUFHasSrsrc(unsigned Opc) {
537 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
538 return Info && Info->has_srsrc;
539}
540
541bool getMUBUFHasSoffset(unsigned Opc) {
542 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
543 return Info && Info->has_soffset;
544}
545
546bool getMUBUFIsBufferInv(unsigned Opc) {
547 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
548 return Info && Info->IsBufferInv;
549}
550
551bool getMUBUFTfe(unsigned Opc) {
552 const MUBUFInfo *Info = getMUBUFOpcodeHelper(Opc);
553 return Info && Info->tfe;
554}
555
556bool getSMEMIsBuffer(unsigned Opc) {
557 const SMInfo *Info = getSMEMOpcodeHelper(Opc);
558 return Info && Info->IsBuffer;
559}
560
561bool getVOP1IsSingle(unsigned Opc) {
562 const VOPInfo *Info = getVOP1OpcodeHelper(Opc);
563 return !Info || Info->IsSingle;
564}
565
566bool getVOP2IsSingle(unsigned Opc) {
567 const VOPInfo *Info = getVOP2OpcodeHelper(Opc);
568 return !Info || Info->IsSingle;
569}
570
571bool getVOP3IsSingle(unsigned Opc) {
572 const VOPInfo *Info = getVOP3OpcodeHelper(Opc);
573 return !Info || Info->IsSingle;
574}
575
576bool isVOPC64DPP(unsigned Opc) {
577 return isVOPC64DPPOpcodeHelper(Opc) || isVOPC64DPP8OpcodeHelper(Opc);
578}
579
580bool isVOPCAsmOnly(unsigned Opc) { return isVOPCAsmOnlyOpcodeHelper(Opc); }
581
582bool getMAIIsDGEMM(unsigned Opc) {
583 const MAIInstInfo *Info = getMAIInstInfoHelper(Opc);
584 return Info && Info->is_dgemm;
585}
586
587bool getMAIIsGFX940XDL(unsigned Opc) {
588 const MAIInstInfo *Info = getMAIInstInfoHelper(Opc);
589 return Info && Info->is_gfx940_xdl;
590}
591
592bool getWMMAIsXDL(unsigned Opc) {
593 const WMMAInstInfo *Info = getWMMAInstInfoHelper(Opc);
594 return Info ? Info->is_wmma_xdl : false;
595}
596
597bool getHasMatrixScale(unsigned Opc) {
598 const WMMAInstInfo *Info = getWMMAInstInfoHelper(Opc);
599 return Info && Info->HasMatrixScale;
600}
601
603 switch (EncodingVal) {
606 return 6;
608 return 4;
611 default:
612 return 8;
613 }
614
615 llvm_unreachable("covered switch over mfma scale formats");
616}
617
619 unsigned BLGP,
620 unsigned F8F8Opcode) {
621 uint8_t SrcANumRegs = mfmaScaleF8F6F4FormatToNumRegs(CBSZ);
622 uint8_t SrcBNumRegs = mfmaScaleF8F6F4FormatToNumRegs(BLGP);
623 return getMFMA_F8F6F4_InstWithNumRegs(SrcANumRegs, SrcBNumRegs, F8F8Opcode);
624}
625
627 switch (Fmt) {
630 return 16;
633 return 12;
635 return 8;
636 }
637
638 llvm_unreachable("covered switch over wmma scale formats");
639}
640
642 unsigned FmtB,
643 unsigned F8F8Opcode) {
644 uint8_t SrcANumRegs = wmmaScaleF8F6F4FormatToNumRegs(FmtA);
645 uint8_t SrcBNumRegs = wmmaScaleF8F6F4FormatToNumRegs(FmtB);
646 return getMFMA_F8F6F4_InstWithNumRegs(SrcANumRegs, SrcBNumRegs, F8F8Opcode);
647}
648
649bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale,
650 unsigned BFmt, unsigned BScale) {
651 auto isValid = [](unsigned Fmt, unsigned Scale) -> bool {
652 switch (Fmt) {
657 if (Scale != WMMA::MATRIX_SCALE_FMT_E8)
658 return false;
659 break;
661 if (Scale != WMMA::MATRIX_SCALE_FMT_E8 &&
664 return false;
665 break;
666 }
667 return true;
668 };
669
670 if (!isValid(AFmt, AScale) || !isValid(BFmt, BScale))
671 return false;
672
673 if (AFmt == WMMA::MATRIX_FMT_FP4 && BFmt == WMMA::MATRIX_FMT_FP4 &&
674 AScale != BScale)
675 return false;
676
677 return true;
678}
679
681 if (ST.hasFeature(AMDGPU::FeatureGFX13Insts))
683 if (ST.hasFeature(AMDGPU::FeatureGFX1250Insts))
685 if (ST.hasFeature(AMDGPU::FeatureGFX12Insts))
687 if (ST.hasFeature(AMDGPU::FeatureGFX11_7Insts))
689 if (ST.hasFeature(AMDGPU::FeatureGFX11Insts))
691 llvm_unreachable("Subtarget generation does not support VOPD!");
692}
693
694CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3) {
695 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(Opc) : 0;
696 Opc = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : Opc;
697 // Normalize through VOPDComponentTable so that e32 and e64 variants
698 // of the same logical opcode all share a single entry.
699 const VOPDComponentInfo *Info = getVOPDComponentHelper(Opc);
700 if (!Info)
701 return {false, false};
702 unsigned Key =
703 (Info->VOPDOp << 5) | (EncodingFamily << 1) | (VOPD3 ? 1u : 0u);
704 const VOPDXYInfo *XYInfo = getVOPDXYInfo(Key);
705 if (!XYInfo)
706 return {false, false};
707 return {XYInfo->IsX, XYInfo->IsY};
708}
709
710unsigned getVOPDOpcode(unsigned Opc, bool VOPD3) {
711 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(Opc) : 0;
712 Opc = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : Opc;
713 const VOPDComponentInfo *Info = getVOPDComponentHelper(Opc);
714 return Info ? Info->VOPDOp : ~0u;
715}
716
717bool isVOPD(unsigned Opc) {
718 return AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src0X);
719}
720
721bool isMAC(unsigned Opc) {
722 return Opc == AMDGPU::V_MAC_F32_e64_gfx6_gfx7 ||
723 Opc == AMDGPU::V_MAC_F32_e64_gfx10 ||
724 Opc == AMDGPU::V_MAC_F32_e64_vi ||
725 Opc == AMDGPU::V_MAC_LEGACY_F32_e64_gfx6_gfx7 ||
726 Opc == AMDGPU::V_MAC_LEGACY_F32_e64_gfx10 ||
727 Opc == AMDGPU::V_MAC_F16_e64_vi ||
728 Opc == AMDGPU::V_FMAC_F64_e64_gfx90a ||
729 Opc == AMDGPU::V_FMAC_F64_e64_gfx12 ||
730 Opc == AMDGPU::V_FMAC_F64_e64_gfx13 ||
731 Opc == AMDGPU::V_FMAC_F32_e64_gfx10 ||
732 Opc == AMDGPU::V_FMAC_F32_e64_gfx11 ||
733 Opc == AMDGPU::V_FMAC_F32_e64_gfx12 ||
734 Opc == AMDGPU::V_FMAC_F32_e64_gfx13 ||
735 Opc == AMDGPU::V_FMAC_F32_e64_vi ||
736 Opc == AMDGPU::V_FMAC_LEGACY_F32_e64_gfx10 ||
737 Opc == AMDGPU::V_FMAC_DX9_ZERO_F32_e64_gfx11 ||
738 Opc == AMDGPU::V_FMAC_F16_e64_gfx10 ||
739 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx11 ||
740 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx11 ||
741 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx12 ||
742 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx12 ||
743 Opc == AMDGPU::V_FMAC_F16_t16_e64_gfx13 ||
744 Opc == AMDGPU::V_FMAC_F16_fake16_e64_gfx13 ||
745 Opc == AMDGPU::V_DOT2C_F32_F16_e64_vi ||
746 Opc == AMDGPU::V_DOT2C_F32_BF16_e64_vi ||
747 Opc == AMDGPU::V_DOT2C_I32_I16_e64_vi ||
748 Opc == AMDGPU::V_DOT4C_I32_I8_e64_vi ||
749 Opc == AMDGPU::V_DOT8C_I32_I4_e64_vi;
750}
751
752bool isPermlane16(unsigned Opc) {
753 return Opc == AMDGPU::V_PERMLANE16_B32_gfx10 ||
754 Opc == AMDGPU::V_PERMLANEX16_B32_gfx10 ||
755 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx11 ||
756 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx11 ||
757 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx12 ||
758 Opc == AMDGPU::V_PERMLANE16_B32_e64_gfx13 ||
759 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx12 ||
760 Opc == AMDGPU::V_PERMLANEX16_B32_e64_gfx13 ||
761 Opc == AMDGPU::V_PERMLANE16_VAR_B32_e64_gfx12 ||
762 Opc == AMDGPU::V_PERMLANE16_VAR_B32_e64_gfx13 ||
763 Opc == AMDGPU::V_PERMLANEX16_VAR_B32_e64_gfx12 ||
764 Opc == AMDGPU::V_PERMLANEX16_VAR_B32_e64_gfx13;
765}
766
768 return Opc == AMDGPU::V_CVT_F32_BF8_e64_gfx12 ||
769 Opc == AMDGPU::V_CVT_F32_FP8_e64_gfx12 ||
770 Opc == AMDGPU::V_CVT_F32_BF8_e64_dpp_gfx12 ||
771 Opc == AMDGPU::V_CVT_F32_FP8_e64_dpp_gfx12 ||
772 Opc == AMDGPU::V_CVT_F32_BF8_e64_dpp8_gfx12 ||
773 Opc == AMDGPU::V_CVT_F32_FP8_e64_dpp8_gfx12 ||
774 Opc == AMDGPU::V_CVT_PK_F32_BF8_fake16_e64_gfx12 ||
775 Opc == AMDGPU::V_CVT_PK_F32_FP8_fake16_e64_gfx12 ||
776 Opc == AMDGPU::V_CVT_PK_F32_BF8_t16_e64_gfx12 ||
777 Opc == AMDGPU::V_CVT_PK_F32_FP8_t16_e64_gfx12;
778}
779
780bool isGenericAtomic(unsigned Opc) {
781 return Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SWAP ||
782 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_ADD ||
783 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB ||
784 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMIN ||
785 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMIN ||
786 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SMAX ||
787 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_UMAX ||
788 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_AND ||
789 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_OR ||
790 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_XOR ||
791 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_INC ||
792 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_DEC ||
793 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FADD ||
794 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMIN ||
795 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_FMAX ||
796 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_CMPSWAP ||
797 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_SUB_CLAMP_U32 ||
798 Opc == AMDGPU::G_AMDGPU_BUFFER_ATOMIC_COND_SUB_U32 ||
799 Opc == AMDGPU::G_AMDGPU_ATOMIC_CMPXCHG;
800}
801
802bool isAsyncStore(unsigned Opc) {
803 return Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B8_gfx1250 ||
804 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B32_gfx1250 ||
805 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B64_gfx1250 ||
806 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B128_gfx1250 ||
807 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B8_SADDR_gfx1250 ||
808 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B32_SADDR_gfx1250 ||
809 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B64_SADDR_gfx1250 ||
810 Opc == GLOBAL_STORE_ASYNC_FROM_LDS_B128_SADDR_gfx1250;
811}
812
813bool isTensorStore(unsigned Opc) {
814 return Opc == TENSOR_STORE_FROM_LDS_d2_gfx1250 ||
815 Opc == TENSOR_STORE_FROM_LDS_d4_gfx1250;
816}
817
818unsigned getTemporalHintType(const MCInstrDesc TID) {
819 if (SIInstrFlags::isAtomic(TID))
821 unsigned Opc = TID.getOpcode();
822 // Async and Tensor store should have the temporal hint type of TH_TYPE_STORE
823 if (TID.mayStore() &&
824 (isAsyncStore(Opc) || isTensorStore(Opc) || !TID.mayLoad()))
825 return CPol::TH_TYPE_STORE;
826
827 // This will default to returning TH_TYPE_LOAD when neither MayStore nor
828 // MayLoad flag is present which is the case with instructions like
829 // image_get_resinfo.
830 return CPol::TH_TYPE_LOAD;
831}
832
833bool isTrue16Inst(unsigned Opc) {
834 const VOPTrue16Info *Info = getTrue16OpcodeHelper(Opc);
835 return Info && Info->IsTrue16;
836}
837
839 const FP4FP8DstByteSelInfo *Info = getFP4FP8DstByteSelHelper(Opc);
840 if (!Info)
841 return FPType::None;
842 if (Info->HasFP8DstByteSel)
843 return FPType::FP8;
844 if (Info->HasFP4DstByteSel)
845 return FPType::FP4;
846
847 return FPType::None;
848}
849
850bool isDPMACCInstruction(unsigned Opc) {
851 const DPMACCInstructionInfo *Info = getDPMACCInstructionHelper(Opc);
852 return Info && Info->IsDPMACCInstruction;
853}
854
855unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc) {
856 const WMMAOpcodeMappingInfo *Info = getWMMAMappingInfoFrom2AddrOpcode(Opc);
857 return Info ? Info->Opcode3Addr : ~0u;
858}
859
860unsigned mapWMMA3AddrTo2AddrOpcode(unsigned Opc) {
861 const WMMAOpcodeMappingInfo *Info = getWMMAMappingInfoFrom3AddrOpcode(Opc);
862 return Info ? Info->Opcode2Addr : ~0u;
863}
864
865// Wrapper for Tablegen'd function. enum Subtarget is not defined in any
866// header files, so we need to wrap it in a function that takes unsigned
867// instead.
868int32_t getMCOpcode(uint32_t Opcode, unsigned Gen) {
869 return getMCOpcodeGen(Opcode, static_cast<Subtarget>(Gen));
870}
871
872unsigned getBitOp2(unsigned Opc) {
873 switch (Opc) {
874 default:
875 return 0;
876 case AMDGPU::V_AND_B32_e32:
877 return 0x40;
878 case AMDGPU::V_OR_B32_e32:
879 return 0x54;
880 case AMDGPU::V_XOR_B32_e32:
881 return 0x14;
882 case AMDGPU::V_XNOR_B32_e32:
883 return 0x41;
884 }
885}
886
887int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily,
888 bool VOPD3) {
889 bool IsConvertibleToBitOp = VOPD3 ? getBitOp2(OpY) : 0;
890 OpY = IsConvertibleToBitOp ? (unsigned)AMDGPU::V_BITOP3_B32_e64 : OpY;
891 const VOPDInfo *Info =
892 getVOPDInfoFromComponentOpcodes(OpX, OpY, EncodingFamily, VOPD3);
893 return Info ? Info->Opcode : -1;
894}
895
896std::pair<unsigned, unsigned> getVOPDComponents(unsigned VOPDOpcode) {
897 const VOPDInfo *Info = getVOPDOpcodeHelper(VOPDOpcode);
898 assert(Info);
899 const auto *OpX = getVOPDBaseFromComponent(Info->OpX);
900 const auto *OpY = getVOPDBaseFromComponent(Info->OpY);
901 assert(OpX && OpY);
902 return {OpX->BaseVOP, OpY->BaseVOP};
903}
904
905namespace VOPD {
906
907ComponentProps::ComponentProps(const MCInstrDesc &OpDesc, bool VOP3Layout) {
909
912 auto TiedIdx = OpDesc.getOperandConstraint(Component::SRC2, MCOI::TIED_TO);
913 assert(TiedIdx == -1 || TiedIdx == Component::DST);
914 HasSrc2Acc = TiedIdx != -1;
915 Opcode = OpDesc.getOpcode();
916
917 IsVOP3 = VOP3Layout || SIInstrFlags::isVOP3(OpDesc);
918 SrcOperandsNum = AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src2) ? 3
919 : AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::imm) ? 3
920 : AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::src1) ? 2
921 : 1;
922 assert(SrcOperandsNum <= Component::MAX_SRC_NUM);
923
924 if (Opcode == AMDGPU::V_CNDMASK_B32_e32 ||
925 Opcode == AMDGPU::V_CNDMASK_B32_e64) {
926 // CNDMASK is an awkward exception, it has FP modifiers, but not FP
927 // operands.
928 NumVOPD3Mods = 2;
929 if (IsVOP3)
930 SrcOperandsNum = 3;
931 } else if (Opcode == AMDGPU::V_DOT2_F32_F16 ||
932 Opcode == AMDGPU::V_DOT2_F32_BF16) {
933 // VOP3P opcodes that have VOPD but don't have VOP2 version. Using VOPD3
934 // path in getIndexOfSrcInMCOperands to get correct src operand indexes,
935 // but generating VOPD, not VOPD3.
936 NumVOPD3Mods = SrcOperandsNum;
937 } else if (isSISrcFPOperand(OpDesc,
938 getNamedOperandIdx(Opcode, OpName::src0))) {
939 // All FP VOPD instructions have Neg modifiers for all operands except
940 // for tied src2.
941 NumVOPD3Mods = SrcOperandsNum;
942 if (HasSrc2Acc)
943 --NumVOPD3Mods;
944 }
945
946 if (SIInstrFlags::isVOP3(OpDesc))
947 return;
948
949 auto OperandsNum = OpDesc.getNumOperands();
950 unsigned CompOprIdx;
951 for (CompOprIdx = Component::SRC1; CompOprIdx < OperandsNum; ++CompOprIdx) {
952 if (OpDesc.operands()[CompOprIdx].OperandType == AMDGPU::OPERAND_KIMM32) {
953 MandatoryLiteralIdx = CompOprIdx;
954 break;
955 }
956 }
957}
958
960 return getNamedOperandIdx(Opcode, OpName::bitop3);
961}
962
963unsigned ComponentInfo::getIndexInParsedOperands(unsigned CompOprIdx) const {
964 assert(CompOprIdx < Component::MAX_OPR_NUM);
965
966 if (CompOprIdx == Component::DST)
968
969 auto CompSrcIdx = CompOprIdx - Component::DST_NUM;
970 if (CompSrcIdx < getCompParsedSrcOperandsNum())
971 return getIndexOfSrcInParsedOperands(CompSrcIdx);
972
973 // The specified operand does not exist.
974 return 0;
975}
976
978 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
979 const MCRegisterInfo &MRI, bool SkipSrc, bool AllowSameVGPR,
980 bool VOPD3) const {
981
982 auto OpXRegs = getRegIndices(ComponentIndex::X, GetRegIdx,
983 CompInfo[ComponentIndex::X].isVOP3());
984 auto OpYRegs = getRegIndices(ComponentIndex::Y, GetRegIdx,
985 CompInfo[ComponentIndex::Y].isVOP3());
986
987 const auto banksOverlap = [&MRI](MCRegister X, MCRegister Y,
988 unsigned BanksMask) -> bool {
989 MCRegister BaseX = MRI.getSubReg(X, AMDGPU::sub0);
990 MCRegister BaseY = MRI.getSubReg(Y, AMDGPU::sub0);
991 if (!BaseX)
992 BaseX = X;
993 if (!BaseY)
994 BaseY = Y;
995 if ((BaseX.id() & BanksMask) == (BaseY.id() & BanksMask))
996 return true;
997 if (BaseX != X /* This is 64-bit register */ &&
998 ((BaseX.id() + 1) & BanksMask) == (BaseY.id() & BanksMask))
999 return true;
1000 if (BaseY != Y &&
1001 (BaseX.id() & BanksMask) == ((BaseY.id() + 1) & BanksMask))
1002 return true;
1003
1004 // If both are 64-bit bank conflict will be detected yet while checking
1005 // the first subreg.
1006 return false;
1007 };
1008
1009 unsigned CompOprIdx;
1010 for (CompOprIdx = 0; CompOprIdx < Component::MAX_OPR_NUM; ++CompOprIdx) {
1011 unsigned BanksMasks = VOPD3 ? VOPD3_VGPR_BANK_MASKS[CompOprIdx]
1012 : VOPD_VGPR_BANK_MASKS[CompOprIdx];
1013 if (!OpXRegs[CompOprIdx] || !OpYRegs[CompOprIdx])
1014 continue;
1015
1016 if (getVGPREncodingMSBs(OpXRegs[CompOprIdx], MRI) !=
1017 getVGPREncodingMSBs(OpYRegs[CompOprIdx], MRI))
1018 return CompOprIdx;
1019
1020 if (SkipSrc && CompOprIdx >= Component::DST_NUM)
1021 continue;
1022
1023 if (CompOprIdx < Component::DST_NUM) {
1024 // Even if we do not check vdst parity, vdst operands still shall not
1025 // overlap.
1026 if (MRI.regsOverlap(OpXRegs[CompOprIdx], OpYRegs[CompOprIdx]))
1027 return CompOprIdx;
1028 if (VOPD3) // No need to check dst parity.
1029 continue;
1030 }
1031
1032 if (banksOverlap(OpXRegs[CompOprIdx], OpYRegs[CompOprIdx], BanksMasks) &&
1033 (!AllowSameVGPR || CompOprIdx < Component::DST_NUM ||
1034 OpXRegs[CompOprIdx] != OpYRegs[CompOprIdx]))
1035 return CompOprIdx;
1036 }
1037
1038 return {};
1039}
1040
1041// Return an array of VGPR registers [DST,SRC0,SRC1,SRC2] used
1042// by the specified component. If an operand is unused
1043// or is not a VGPR, the corresponding value is 0.
1044//
1045// GetRegIdx(Component, MCOperandIdx) must return a VGPR register index
1046// for the specified component and MC operand. The callback must return 0
1047// if the operand is not a register or not a VGPR.
1049InstInfo::getRegIndices(unsigned CompIdx,
1050 std::function<MCRegister(unsigned, unsigned)> GetRegIdx,
1051 bool VOPD3) const {
1052 assert(CompIdx < COMPONENTS_NUM);
1053
1054 const auto &Comp = CompInfo[CompIdx];
1056
1057 RegIndices[DST] = GetRegIdx(CompIdx, Comp.getIndexOfDstInMCOperands());
1058
1059 for (unsigned CompOprIdx : {SRC0, SRC1, SRC2}) {
1060 unsigned CompSrcIdx = CompOprIdx - DST_NUM;
1061 RegIndices[CompOprIdx] =
1062 Comp.hasRegSrcOperand(CompSrcIdx)
1063 ? GetRegIdx(CompIdx,
1064 Comp.getIndexOfSrcInMCOperands(CompSrcIdx, VOPD3))
1065 : MCRegister();
1066 }
1067 return RegIndices;
1068}
1069
1070} // namespace VOPD
1071
1073 return VOPD::InstInfo(OpX, OpY);
1074}
1075
1077 const MCInstrInfo *InstrInfo) {
1078 auto [OpX, OpY] = getVOPDComponents(VOPDOpcode);
1079 const auto &OpXDesc = InstrInfo->get(OpX);
1080 const auto &OpYDesc = InstrInfo->get(OpY);
1081 bool VOPD3 = SIInstrFlags::isVOPD3(*InstrInfo, VOPDOpcode);
1083 VOPD::ComponentInfo OpYInfo(OpYDesc, OpXInfo, VOPD3);
1084 return VOPD::InstInfo(OpXInfo, OpYInfo);
1085}
1086
1088 StringRef FeatureString) {
1090 STI.getFeatureBits().test(FeatureXNACKOnOffModes)
1091 ? TargetIDSetting::Any
1092 : TargetIDSetting::Unsupported,
1093 STI.getFeatureBits().test(FeatureSupportsSRAMECC)
1094 ? TargetIDSetting::Any
1095 : TargetIDSetting::Unsupported);
1096
1097 // Check if xnack or sramecc is explicitly enabled or disabled. In the
1098 // absence of the target features we assume we must generate code that can run
1099 // in any environment.
1100 SubtargetFeatures Features(FeatureString);
1101 std::optional<bool> XnackRequested;
1102 std::optional<bool> SramEccRequested;
1103
1104 for (const std::string &Feature : Features.getFeatures()) {
1105 if (Feature == "+xnack")
1106 XnackRequested = true;
1107 else if (Feature == "-xnack")
1108 XnackRequested = false;
1109 else if (Feature == "+sramecc")
1110 SramEccRequested = true;
1111 else if (Feature == "-sramecc")
1112 SramEccRequested = false;
1113 }
1114
1115 // Only allow changing xnack setting if the target supports on/off modes.
1116 // Targets without on/off mode support keep their initial setting
1117 // (Unsupported).
1118
1119 bool XnackSupported = STI.getFeatureBits().test(FeatureXNACKOnOffModes);
1120 bool SramEccSupported = TargetID.isSramEccSupported();
1121
1122 if (XnackRequested) {
1123 if (XnackSupported) {
1124 TargetID.setXnackSetting(*XnackRequested ? TargetIDSetting::On
1125 : TargetIDSetting::Off);
1126 } else {
1127 // If a specific xnack setting was requested and this GPU does not support
1128 // xnack emit a warning. Setting will remain set to "Unsupported".
1129 if (*XnackRequested) {
1130 errs() << "warning: xnack 'On' was requested for a processor that does "
1131 "not support it!\n";
1132 } else {
1133 errs() << "warning: xnack 'Off' was requested for a processor that "
1134 "does not support it!\n";
1135 }
1136 }
1137 }
1138
1139 if (SramEccRequested) {
1140 if (SramEccSupported) {
1141 TargetID.setSramEccSetting(*SramEccRequested ? TargetIDSetting::On
1142 : TargetIDSetting::Off);
1143 } else {
1144 // If a specific sramecc setting was requested and this GPU does not
1145 // support sramecc emit a warning. Setting will remain set to
1146 // "Unsupported".
1147 if (*SramEccRequested) {
1148 errs() << "warning: sramecc 'On' was requested for a processor that "
1149 "does not support it!\n";
1150 } else {
1151 errs() << "warning: sramecc 'Off' was requested for a processor that "
1152 "does not support it!\n";
1153 }
1154 }
1155 }
1156
1157 return TargetID;
1158}
1159
1160namespace IsaInfo {
1161
1163 if (STI.getFeatureBits().test(FeatureInstCacheLineSize128))
1164 return 128;
1165 if (STI.getFeatureBits().test(FeatureInstCacheLineSize64))
1166 return 64;
1167 return 64;
1168}
1169
1170unsigned getWavefrontSize(const MCSubtargetInfo &STI) {
1171 if (STI.getFeatureBits().test(FeatureWavefrontSize16))
1172 return 16;
1173 if (STI.getFeatureBits().test(FeatureWavefrontSize32))
1174 return 32;
1175
1176 return 64;
1177}
1178
1179// Maximum LDS a single work-group can address. This is a fixed HW cap. It does
1180// not depend on how many SIMDs a work-group runs on.
1182 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
1183 return 32768;
1184 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
1185 return 65536;
1186 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
1187 return 163840;
1188 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
1189 return 196608;
1190 if (STI.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
1191 return 327680;
1192 return 32768;
1193}
1194
1195// Total physical size of LDS on the block, in bytes. On targets with
1196// FeatureHalfAddressablePhysicalLocalMemory the physical block is twice the
1197// addressable size (gfx10/11/12, 128k physical and 64k addressable). On other
1198// targets it is equal to the addressable size.
1199static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI) {
1200 unsigned Addressable = getMaxHWAddressableLocalMemorySize(STI);
1201 if (STI.getFeatureBits().test(FeatureHalfAddressablePhysicalLocalMemory))
1202 return 2 * Addressable;
1203 return Addressable;
1204}
1205
1206// Sizes in use, by generation (addressable / physical block):
1207// gfx6 : 32 KiB
1208// gfx7 / gfx8 / gfx9: 64 KiB
1209// gfx9.5 (gfx950) : 160 KiB
1210// gfx10 / 11 / 12 : 64 KiB addressable, 128 KiB physical block
1211// gfx12.5 (gfx1250) : 320 KiB (always runs on four SIMDs)
1212// gfx13 : 192 KiB on four SIMDs, 96 KiB on two
1213// Total available in the current mode. The physical size is halved when a
1214// work-group runs on two SIMDs.
1216 unsigned Size = getPhysicalLocalMemorySize(STI);
1217 if (!isFullSIMDMode(STI))
1218 Size /= 2;
1219 return Size;
1220}
1221
1222// What one work-group can allocate in the current mode. This is the HW
1223// addressable cap, but never more than the total available in the current mode.
1225 return std::min(getMaxHWAddressableLocalMemorySize(STI),
1226 getLocalMemorySize(STI));
1227}
1228
1230 unsigned FlatWorkGroupSize) {
1231 assert(FlatWorkGroupSize != 0);
1232 if (!STI.getTargetTriple().isAMDGCN())
1233 return 8;
1234 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1235 unsigned MaxWaves =
1237 unsigned N = getWavesPerWorkGroup(STI, FlatWorkGroupSize);
1238 if (N == 1) {
1239 // Single-wave workgroups don't consume barrier resources.
1240 return MaxWaves;
1241 }
1242
1243 unsigned MaxBarriers = 16;
1244 if (isGFX10Plus(STI) && !STI.getFeatureBits().test(FeatureCuMode))
1245 MaxBarriers = 32;
1246
1247 return std::min(MaxWaves / N, MaxBarriers);
1248}
1249
1251 unsigned FlatWorkGroupSize) {
1252 return divideCeil(getWavesPerWorkGroup(STI, FlatWorkGroupSize),
1254}
1255
1257 unsigned FlatWorkGroupSize) {
1258 return divideCeil(FlatWorkGroupSize, getWavefrontSize(STI));
1259}
1260
1261unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI) { return 8; }
1262
1263// Per-wave SGPRs reserved for the trap handler when enabled.
1264static unsigned getSGPRTrapHandlerReserve(const MCSubtargetInfo &STI) {
1265 return STI.getFeatureBits().test(FeatureTrapHandler) ? TRAP_NUM_SGPRS : 0;
1266}
1267
1268// Per-wave SGPR budget (before the addressable clamp): take off the trap
1269// reserve, round down to \p Granule. Shared by getMinNumSGPRs() and
1270// getMaxNumSGPRs(); getOccupancyWithNumSGPRs() is the closed-form algebraic
1271// inverse of this same budget (it does not call this helper), so the two encode
1272// one model.
1273static unsigned getSGPRBudgetPerWave(unsigned TotalNumSGPRs,
1274 unsigned WavesPerEU, unsigned TrapReserve,
1275 unsigned Granule) {
1276 assert(WavesPerEU != 0 && Granule != 0);
1277 unsigned Budget = TotalNumSGPRs / WavesPerEU;
1278 Budget -= std::min(Budget, TrapReserve);
1279 return alignDown(Budget, Granule);
1280}
1281
1282unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU) {
1283 assert(WavesPerEU != 0);
1284
1286 if (Version.Major >= 10)
1287 return 0;
1288
1289 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1290 if (WavesPerEU >= getMaxWavesPerEU(Kind))
1291 return 0;
1292
1293 unsigned MinNumSGPRs =
1294 getSGPRBudgetPerWave(getTotalNumSGPRs(Kind), WavesPerEU + 1,
1296 getSGPRAllocGranule(Kind)) +
1297 1;
1298 return std::min(MinNumSGPRs, getAddressableNumSGPRs(Kind));
1299}
1300
1301unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1302 bool Addressable) {
1303 assert(WavesPerEU != 0);
1304
1305 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1306 unsigned AddressableNumSGPRs = getAddressableNumSGPRs(Kind);
1308 if (Version.Major >= 10)
1309 return Addressable ? AddressableNumSGPRs : 108;
1310 if (Version.Major >= 8 && !Addressable)
1311 AddressableNumSGPRs = 112;
1312 unsigned MaxNumSGPRs = getSGPRBudgetPerWave(
1313 getTotalNumSGPRs(Kind), WavesPerEU, getSGPRTrapHandlerReserve(STI),
1314 getSGPRAllocGranule(Kind));
1315 return std::min(MaxNumSGPRs, AddressableNumSGPRs);
1316}
1317
1319 // From GFX10 on the SGPR file is large enough that SGPRs never limit
1320 // occupancy. Kept as one capability so callers don't each test the version.
1321 return getIsaVersion(STI.getCPU()).Major < 10;
1322}
1323
1324unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
1325 bool FlatScrUsed, bool XNACKUsed) {
1326 unsigned ExtraSGPRs = 0;
1327 if (VCCUsed)
1328 ExtraSGPRs = 2;
1329
1331 if (Version.Major >= 10)
1332 return ExtraSGPRs;
1333
1334 if (Version.Major < 8) {
1335 if (FlatScrUsed)
1336 ExtraSGPRs = 4;
1337 } else {
1338 if (XNACKUsed)
1339 ExtraSGPRs = 4;
1340
1341 if (FlatScrUsed ||
1342 STI.getFeatureBits().test(AMDGPU::FeatureArchitectedFlatScratch))
1343 ExtraSGPRs = 6;
1344 }
1345
1346 return ExtraSGPRs;
1347}
1348
1349unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed,
1350 bool FlatScrUsed) {
1351 return getNumExtraSGPRs(STI, VCCUsed, FlatScrUsed,
1352 STI.getFeatureBits().test(AMDGPU::FeatureXNACK));
1353}
1354
1355static unsigned getGranulatedNumRegisterBlocks(unsigned NumRegs,
1356 unsigned Granule) {
1357 return divideCeil(std::max(1u, NumRegs), Granule);
1358}
1359
1360unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs) {
1361 // SGPRBlocks is actual number of SGPR blocks minus 1.
1363 1;
1364}
1365
1367 unsigned DynamicVGPRBlockSize,
1368 std::optional<bool> EnableWavefrontSize32) {
1369 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1370 return 8;
1371
1372 if (DynamicVGPRBlockSize != 0)
1373 return DynamicVGPRBlockSize;
1374
1375 bool IsWave32 = EnableWavefrontSize32
1376 ? *EnableWavefrontSize32
1377 : STI.getFeatureBits().test(FeatureWavefrontSize32);
1378
1379 if (STI.getFeatureBits().test(Feature1536VGPRs))
1380 return IsWave32 ? 24 : 12;
1381
1382 if (hasGFX10_3Insts(STI))
1383 return IsWave32 ? 16 : 8;
1384
1385 return IsWave32 ? 8 : 4;
1386}
1387
1389 std::optional<bool> EnableWavefrontSize32) {
1390 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1391 return 8;
1392
1393 bool IsWave32 = EnableWavefrontSize32
1394 ? *EnableWavefrontSize32
1395 : STI.getFeatureBits().test(FeatureWavefrontSize32);
1396
1397 if (STI.getFeatureBits().test(Feature1024AddressableVGPRs))
1398 return IsWave32 ? 16 : 8;
1399
1400 return IsWave32 ? 8 : 4;
1401}
1402
1403unsigned getArchVGPRAllocGranule() { return 4; }
1404
1405unsigned getTotalNumVGPRs(const MCSubtargetInfo &STI) {
1406 if (STI.getFeatureBits().test(FeatureGFX90AInsts))
1407 return 512;
1408 if (!isGFX10Plus(STI))
1409 return 256;
1410 bool IsWave32 = STI.getFeatureBits().test(FeatureWavefrontSize32);
1411 if (STI.getFeatureBits().test(Feature1536VGPRs))
1412 return IsWave32 ? 1536 : 768;
1413 return IsWave32 ? 1024 : 512;
1414}
1415
1417 const auto &Features = STI.getFeatureBits();
1418 if (Features.test(Feature1024AddressableVGPRs))
1419 return Features.test(FeatureWavefrontSize32) ? 1024 : 512;
1420 return 256;
1421}
1422
1424 unsigned DynamicVGPRBlockSize) {
1425 const auto &Features = STI.getFeatureBits();
1426 if (Features.test(FeatureGFX90AInsts))
1427 return 512;
1428
1429 if (DynamicVGPRBlockSize != 0) {
1430 // On GFX12 we can allocate at most MaxDynamicVGPRBlocks blocks of VGPRs.
1431 return MaxDynamicVGPRBlocks *
1432 getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
1433 }
1434 return getAddressableNumArchVGPRs(STI);
1435}
1436
1438 unsigned NumVGPRs,
1439 unsigned DynamicVGPRBlockSize) {
1441 NumVGPRs, getVGPRAllocGranule(STI, DynamicVGPRBlockSize),
1443}
1444
1445unsigned getNumWavesPerEUWithNumVGPRs(unsigned NumVGPRs, unsigned Granule,
1446 unsigned MaxWaves,
1447 unsigned TotalNumVGPRs) {
1448 if (NumVGPRs < Granule)
1449 return MaxWaves;
1450 unsigned RoundedRegs = alignTo(NumVGPRs, Granule);
1451 return std::min(std::max(TotalNumVGPRs / RoundedRegs, 1u), MaxWaves);
1452}
1453
1454unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves,
1455 unsigned TotalNumSGPRs, unsigned Granule,
1456 unsigned TrapReserve) {
1457 // Closed-form inverse of getMaxNumSGPRs(): the budget condition
1458 // SGPRs <= alignDown(TotalNumSGPRs / W - TrapReserve, Granule)
1459 // solves to W <= TotalNumSGPRs / (alignTo(SGPRs, Granule) + TrapReserve).
1460 unsigned PerWave = alignTo(SGPRs, Granule) + TrapReserve;
1461 return PerWave ? std::clamp(TotalNumSGPRs / PerWave, 1u, MaxWaves) : MaxWaves;
1462}
1463
1464unsigned getOccupancyWithNumSGPRs(const MCSubtargetInfo &STI, unsigned SGPRs) {
1465 GPUKind Kind = parseArchAMDGCN(STI.getCPU());
1466 unsigned MaxWaves = getMaxWavesPerEU(Kind);
1467
1468 if (!isSGPROccupancyLimited(STI))
1469 return MaxWaves;
1470
1471 return getOccupancyWithNumSGPRs(SGPRs, MaxWaves, getTotalNumSGPRs(Kind),
1472 getSGPRAllocGranule(Kind),
1474}
1475
1476unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1477 unsigned DynamicVGPRBlockSize) {
1478 assert(WavesPerEU != 0);
1479
1480 // In dynamic VGPR mode, (static) occupancy does not depend on VGPR usage,
1481 // so getMaxNumVGPRs does not depend on WavesPerEU, and thus we need to return
1482 // zero because there is no nonzero VGPR usage N where going below N
1483 // achieves higher (static) occupancy.
1484 bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
1485 if (DynamicVGPREnabled)
1486 return 0;
1487
1488 unsigned MaxWavesPerEU = getMaxWavesPerEU(parseArchAMDGCN(STI.getCPU()));
1489 if (WavesPerEU >= MaxWavesPerEU)
1490 return 0;
1491
1492 unsigned TotNumVGPRs = getTotalNumVGPRs(STI);
1493 unsigned AddrsableNumVGPRs =
1494 getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
1495 unsigned Granule = getVGPRAllocGranule(STI, DynamicVGPRBlockSize);
1496 unsigned MaxNumVGPRs = alignDown(TotNumVGPRs / WavesPerEU, Granule);
1497
1498 if (MaxNumVGPRs == alignDown(TotNumVGPRs / MaxWavesPerEU, Granule))
1499 return 0;
1500
1501 unsigned MinWavesPerEU = getNumWavesPerEUWithNumVGPRs(STI, AddrsableNumVGPRs,
1502 DynamicVGPRBlockSize);
1503 if (WavesPerEU < MinWavesPerEU)
1504 return getMinNumVGPRs(STI, MinWavesPerEU, DynamicVGPRBlockSize);
1505
1506 unsigned MaxNumVGPRsNext = alignDown(TotNumVGPRs / (WavesPerEU + 1), Granule);
1507 unsigned MinNumVGPRs = 1 + std::min(MaxNumVGPRs - Granule, MaxNumVGPRsNext);
1508 return std::min(MinNumVGPRs, AddrsableNumVGPRs);
1509}
1510
1511unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU,
1512 unsigned DynamicVGPRBlockSize) {
1513 assert(WavesPerEU != 0);
1514
1515 // In dynamic VGPR mode, WavesPerEU does not imply a VGPR limit.
1516 bool DynamicVGPREnabled = (DynamicVGPRBlockSize != 0);
1517 unsigned MaxNumVGPRs =
1518 DynamicVGPREnabled
1519 ? getTotalNumVGPRs(STI)
1520 : alignDown(getTotalNumVGPRs(STI) / WavesPerEU,
1521 getVGPRAllocGranule(STI, DynamicVGPRBlockSize));
1522 unsigned AddressableNumVGPRs =
1523 getAddressableNumVGPRs(STI, DynamicVGPRBlockSize);
1524 return std::min(MaxNumVGPRs, AddressableNumVGPRs);
1525}
1526
1527unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs,
1528 std::optional<bool> EnableWavefrontSize32) {
1530 NumVGPRs, getVGPREncodingGranule(STI, EnableWavefrontSize32)) -
1531 1;
1532}
1533
1535 unsigned NumVGPRs,
1536 unsigned DynamicVGPRBlockSize,
1537 std::optional<bool> EnableWavefrontSize32) {
1539 NumVGPRs,
1540 getVGPRAllocGranule(STI, DynamicVGPRBlockSize, EnableWavefrontSize32));
1541}
1542} // end namespace IsaInfo
1543
1545 const MCSubtargetInfo &STI) {
1547 KernelCode.amd_kernel_code_version_major = 1;
1548 KernelCode.amd_kernel_code_version_minor = 2;
1549 KernelCode.amd_machine_kind = 1; // AMD_MACHINE_KIND_AMDGPU
1550 KernelCode.amd_machine_version_major = Version.Major;
1551 KernelCode.amd_machine_version_minor = Version.Minor;
1552 KernelCode.amd_machine_version_stepping = Version.Stepping;
1554 if (STI.getFeatureBits().test(FeatureWavefrontSize32)) {
1555 KernelCode.wavefront_size = 5;
1557 } else {
1558 KernelCode.wavefront_size = 6;
1559 }
1560
1561 // If the code object does not support indirect functions, then the value must
1562 // be 0xffffffff.
1563 KernelCode.call_convention = -1;
1564
1565 // These alignment values are specified in powers of two, so alignment =
1566 // 2^n. The minimum alignment is 2^4 = 16.
1567 KernelCode.kernarg_segment_alignment = 4;
1568 KernelCode.group_segment_alignment = 4;
1569 KernelCode.private_segment_alignment = 4;
1570
1571 if (Version.Major >= 10) {
1572 KernelCode.compute_pgm_resource_registers |=
1573 S_00B848_WGP_MODE(STI.getFeatureBits().test(FeatureCuMode) ? 0 : 1) |
1575 }
1576}
1577
1580}
1581
1584}
1585
1587 unsigned AS = GV->getAddressSpace();
1588 return AS == AMDGPUAS::CONSTANT_ADDRESS ||
1590}
1591
1593 return TT.getArch() == Triple::r600;
1594}
1595
1596static bool isValidRegPrefix(char C) {
1597 return C == 'v' || C == 's' || C == 'a';
1598}
1599
1600std::tuple<char, unsigned, unsigned> parseAsmPhysRegName(StringRef RegName) {
1601 char Kind = RegName.front();
1602 if (!isValidRegPrefix(Kind))
1603 return {};
1604
1605 RegName = RegName.drop_front();
1606 if (RegName.consume_front("[")) {
1607 unsigned Idx, End;
1608 bool Failed = RegName.consumeInteger(10, Idx);
1609 Failed |= !RegName.consume_front(":");
1610 Failed |= RegName.consumeInteger(10, End);
1611 Failed |= !RegName.consume_back("]");
1612 if (!Failed) {
1613 unsigned NumRegs = End - Idx + 1;
1614 if (NumRegs > 1)
1615 return {Kind, Idx, NumRegs};
1616 }
1617 } else {
1618 unsigned Idx;
1619 bool Failed = RegName.getAsInteger(10, Idx);
1620 if (!Failed)
1621 return {Kind, Idx, 1};
1622 }
1623
1624 return {};
1625}
1626
1627std::tuple<char, unsigned, unsigned>
1629 StringRef RegName = Constraint;
1630 if (!RegName.consume_front("{") || !RegName.consume_back("}"))
1631 return {};
1633}
1634
1635std::pair<unsigned, unsigned>
1637 std::pair<unsigned, unsigned> Default,
1638 bool OnlyFirstRequired) {
1639 if (auto Attr = getIntegerPairAttribute(F, Name, OnlyFirstRequired))
1640 return {Attr->first, Attr->second.value_or(Default.second)};
1641 return Default;
1642}
1643
1644std::optional<std::pair<unsigned, std::optional<unsigned>>>
1646 bool OnlyFirstRequired) {
1647 Attribute A = F.getFnAttribute(Name);
1648 if (!A.isStringAttribute())
1649 return std::nullopt;
1650
1651 LLVMContext &Ctx = F.getContext();
1652 std::pair<unsigned, std::optional<unsigned>> Ints;
1653 std::pair<StringRef, StringRef> Strs = A.getValueAsString().split(',');
1654 if (Strs.first.trim().getAsInteger(0, Ints.first)) {
1655 Ctx.emitError("can't parse first integer attribute " + Name);
1656 return std::nullopt;
1657 }
1658 unsigned Second = 0;
1659 if (Strs.second.trim().getAsInteger(0, Second)) {
1660 if (!OnlyFirstRequired || !Strs.second.trim().empty()) {
1661 Ctx.emitError("can't parse second integer attribute " + Name);
1662 return std::nullopt;
1663 }
1664 } else {
1665 Ints.second = Second;
1666 }
1667
1668 return Ints;
1669}
1670
1672 unsigned Size,
1673 unsigned DefaultVal) {
1674 std::optional<SmallVector<unsigned>> R =
1676 return R.has_value() ? *R : SmallVector<unsigned>(Size, DefaultVal);
1677}
1678
1679std::optional<SmallVector<unsigned>>
1681 assert(Size > 2);
1682 LLVMContext &Ctx = F.getContext();
1683
1684 Attribute A = F.getFnAttribute(Name);
1685 if (!A.isValid())
1686 return std::nullopt;
1687 if (!A.isStringAttribute()) {
1688 Ctx.emitError(Name + " is not a string attribute");
1689 return std::nullopt;
1690 }
1691
1693
1694 StringRef S = A.getValueAsString();
1695 unsigned i = 0;
1696 for (; !S.empty() && i < Size; i++) {
1697 std::pair<StringRef, StringRef> Strs = S.split(',');
1698 unsigned IntVal;
1699 if (Strs.first.trim().getAsInteger(0, IntVal)) {
1700 Ctx.emitError("can't parse integer attribute " + Strs.first + " in " +
1701 Name);
1702 return std::nullopt;
1703 }
1704 Vals[i] = IntVal;
1705 S = Strs.second;
1706 }
1707
1708 if (!S.empty() || i < Size) {
1709 Ctx.emitError("attribute " + Name +
1710 " has incorrect number of integers; expected " +
1712 return std::nullopt;
1713 }
1714 return Vals;
1715}
1716
1718 return getIntegerVecAttribute(F, "amdgpu-max-num-workgroups", 3,
1719 std::numeric_limits<uint32_t>::max());
1720}
1721
1722bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val) {
1723 assert((MD.getNumOperands() % 2 == 0) && "invalid number of operands!");
1724 for (unsigned I = 0, E = MD.getNumOperands() / 2; I != E; ++I) {
1725 auto Low =
1726 mdconst::extract<ConstantInt>(MD.getOperand(2 * I + 0))->getValue();
1727 auto High =
1728 mdconst::extract<ConstantInt>(MD.getOperand(2 * I + 1))->getValue();
1729 // There are two types of [A; B) ranges:
1730 // A < B, e.g. [4; 5) which is a range that only includes 4.
1731 // A > B, e.g. [5; 4) which is a range that wraps around and includes
1732 // everything except 4.
1733 if (Low.ult(High)) {
1734 if (Low.ule(Val) && High.ugt(Val))
1735 return true;
1736 } else {
1737 if (Low.uge(Val) && High.ult(Val))
1738 return true;
1739 }
1740 }
1741
1742 return false;
1743}
1744
1746 return (1 << (getVmcntBitWidthLo(Version.Major) +
1747 getVmcntBitWidthHi(Version.Major))) -
1748 1;
1749}
1750
1752 return (1 << getLoadcntBitWidth(Version.Major)) - 1;
1753}
1754
1756 return (1 << getSamplecntBitWidth(Version.Major)) - 1;
1757}
1758
1760 return (1 << getBvhcntBitWidth(Version.Major)) - 1;
1761}
1762
1764 return (1 << getExpcntBitWidth(Version.Major)) - 1;
1765}
1766
1768 return (1 << getLgkmcntBitWidth(Version.Major)) - 1;
1769}
1770
1772 return (1 << getDscntBitWidth(Version.Major)) - 1;
1773}
1774
1776 return (1 << getKmcntBitWidth(Version.Major)) - 1;
1777}
1778
1780 return (1 << getXcntBitWidth(Version.Major, Version.Minor)) - 1;
1781}
1782
1784 return (1 << getAsynccntBitWidth(Version.Major, Version.Minor)) - 1;
1785}
1786
1788 return (1 << getStorecntBitWidth(Version.Major)) - 1;
1789}
1790
1792 unsigned VmcntLo = getBitMask(getVmcntBitShiftLo(Version.Major),
1793 getVmcntBitWidthLo(Version.Major));
1794 unsigned Expcnt = getBitMask(getExpcntBitShift(Version.Major),
1795 getExpcntBitWidth(Version.Major));
1796 unsigned Lgkmcnt = getBitMask(getLgkmcntBitShift(Version.Major),
1797 getLgkmcntBitWidth(Version.Major));
1798 unsigned VmcntHi = getBitMask(getVmcntBitShiftHi(Version.Major),
1799 getVmcntBitWidthHi(Version.Major));
1800 return VmcntLo | Expcnt | Lgkmcnt | VmcntHi;
1801}
1802
1803unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt) {
1804 unsigned VmcntLo = unpackBits(Waitcnt, getVmcntBitShiftLo(Version.Major),
1805 getVmcntBitWidthLo(Version.Major));
1806 unsigned VmcntHi = unpackBits(Waitcnt, getVmcntBitShiftHi(Version.Major),
1807 getVmcntBitWidthHi(Version.Major));
1808 return VmcntLo | VmcntHi << getVmcntBitWidthLo(Version.Major);
1809}
1810
1811unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt) {
1812 return unpackBits(Waitcnt, getExpcntBitShift(Version.Major),
1813 getExpcntBitWidth(Version.Major));
1814}
1815
1816unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt) {
1817 return unpackBits(Waitcnt, getLgkmcntBitShift(Version.Major),
1818 getLgkmcntBitWidth(Version.Major));
1819}
1820
1821unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt) {
1822 return unpackBits(Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1823 getLoadcntBitWidth(Version.Major));
1824}
1825
1826unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt) {
1827 return unpackBits(Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1828 getStorecntBitWidth(Version.Major));
1829}
1830
1831unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt) {
1832 return unpackBits(Waitcnt, getDscntBitShift(Version.Major),
1833 getDscntBitWidth(Version.Major));
1834}
1835
1836void decodeWaitcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned &Vmcnt,
1837 unsigned &Expcnt, unsigned &Lgkmcnt) {
1838 Vmcnt = decodeVmcnt(Version, Waitcnt);
1839 Expcnt = decodeExpcnt(Version, Waitcnt);
1840 Lgkmcnt = decodeLgkmcnt(Version, Waitcnt);
1841}
1842
1843unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt,
1844 unsigned Vmcnt) {
1845 Waitcnt = packBits(Vmcnt, Waitcnt, getVmcntBitShiftLo(Version.Major),
1846 getVmcntBitWidthLo(Version.Major));
1847 return packBits(Vmcnt >> getVmcntBitWidthLo(Version.Major), Waitcnt,
1848 getVmcntBitShiftHi(Version.Major),
1849 getVmcntBitWidthHi(Version.Major));
1850}
1851
1852unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt,
1853 unsigned Expcnt) {
1854 return packBits(Expcnt, Waitcnt, getExpcntBitShift(Version.Major),
1855 getExpcntBitWidth(Version.Major));
1856}
1857
1858unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt,
1859 unsigned Lgkmcnt) {
1860 return packBits(Lgkmcnt, Waitcnt, getLgkmcntBitShift(Version.Major),
1861 getLgkmcntBitWidth(Version.Major));
1862}
1863
1864unsigned encodeWaitcnt(const IsaVersion &Version, unsigned Vmcnt,
1865 unsigned Expcnt, unsigned Lgkmcnt) {
1866 unsigned Waitcnt = getWaitcntBitMask(Version);
1868 Waitcnt = encodeExpcnt(Version, Waitcnt, Expcnt);
1869 Waitcnt = encodeLgkmcnt(Version, Waitcnt, Lgkmcnt);
1870 return Waitcnt;
1871}
1872
1874 bool IsStore) {
1875 unsigned Dscnt = getBitMask(getDscntBitShift(Version.Major),
1876 getDscntBitWidth(Version.Major));
1877 if (IsStore) {
1878 unsigned Storecnt = getBitMask(getLoadcntStorecntBitShift(Version.Major),
1879 getStorecntBitWidth(Version.Major));
1880 return Dscnt | Storecnt;
1881 }
1882 unsigned Loadcnt = getBitMask(getLoadcntStorecntBitShift(Version.Major),
1883 getLoadcntBitWidth(Version.Major));
1884 return Dscnt | Loadcnt;
1885}
1886
1887static unsigned encodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt,
1888 unsigned Loadcnt) {
1889 return packBits(Loadcnt, Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1890 getLoadcntBitWidth(Version.Major));
1891}
1892
1893static unsigned encodeStorecnt(const IsaVersion &Version, unsigned Waitcnt,
1894 unsigned Storecnt) {
1895 return packBits(Storecnt, Waitcnt, getLoadcntStorecntBitShift(Version.Major),
1896 getStorecntBitWidth(Version.Major));
1897}
1898
1899static unsigned encodeDscnt(const IsaVersion &Version, unsigned Waitcnt,
1900 unsigned Dscnt) {
1901 return packBits(Dscnt, Waitcnt, getDscntBitShift(Version.Major),
1902 getDscntBitWidth(Version.Major));
1903}
1904
1905unsigned encodeLoadcntDscnt(const IsaVersion &Version, unsigned Loadcnt,
1906 unsigned Dscnt) {
1907 unsigned Waitcnt = getCombinedCountBitMask(Version, false);
1908 Waitcnt = encodeLoadcnt(Version, Waitcnt, Loadcnt);
1910 return Waitcnt;
1911}
1912
1913unsigned encodeStorecntDscnt(const IsaVersion &Version, unsigned Storecnt,
1914 unsigned Dscnt) {
1915 unsigned Waitcnt = getCombinedCountBitMask(Version, true);
1916 Waitcnt = encodeStorecnt(Version, Waitcnt, Storecnt);
1918 return Waitcnt;
1919}
1920
1921//===----------------------------------------------------------------------===//
1922// Custom Operand Values
1923//===----------------------------------------------------------------------===//
1924
1926 int Size,
1927 const MCSubtargetInfo &STI) {
1928 unsigned Enc = 0;
1929 for (int Idx = 0; Idx < Size; ++Idx) {
1930 const auto &Op = Opr[Idx];
1931 if (Op.isSupported(STI))
1932 Enc |= Op.encode(Op.Default);
1933 }
1934 return Enc;
1935}
1936
1938 int Size, unsigned Code,
1939 bool &HasNonDefaultVal,
1940 const MCSubtargetInfo &STI) {
1941 unsigned UsedOprMask = 0;
1942 HasNonDefaultVal = false;
1943 for (int Idx = 0; Idx < Size; ++Idx) {
1944 const auto &Op = Opr[Idx];
1945 if (!Op.isSupported(STI))
1946 continue;
1947 UsedOprMask |= Op.getMask();
1948 unsigned Val = Op.decode(Code);
1949 if (!Op.isValid(Val))
1950 return false;
1951 HasNonDefaultVal |= (Val != Op.Default);
1952 }
1953 return (Code & ~UsedOprMask) == 0;
1954}
1955
1956static bool decodeCustomOperand(const CustomOperandVal *Opr, int Size,
1957 unsigned Code, int &Idx, StringRef &Name,
1958 unsigned &Val, bool &IsDefault,
1959 const MCSubtargetInfo &STI) {
1960 while (Idx < Size) {
1961 const auto &Op = Opr[Idx++];
1962 if (Op.isSupported(STI)) {
1963 Name = Op.Name;
1964 Val = Op.decode(Code);
1965 IsDefault = (Val == Op.Default);
1966 return true;
1967 }
1968 }
1969
1970 return false;
1971}
1972
1974 int64_t InputVal) {
1975 if (InputVal < 0 || InputVal > Op.Max)
1976 return OPR_VAL_INVALID;
1977 return Op.encode(InputVal);
1978}
1979
1980static int encodeCustomOperand(const CustomOperandVal *Opr, int Size,
1981 const StringRef Name, int64_t InputVal,
1982 unsigned &UsedOprMask,
1983 const MCSubtargetInfo &STI) {
1984 int InvalidId = OPR_ID_UNKNOWN;
1985 for (int Idx = 0; Idx < Size; ++Idx) {
1986 const auto &Op = Opr[Idx];
1987 if (Op.Name == Name) {
1988 if (!Op.isSupported(STI)) {
1989 InvalidId = OPR_ID_UNSUPPORTED;
1990 continue;
1991 }
1992 auto OprMask = Op.getMask();
1993 if (OprMask & UsedOprMask)
1994 return OPR_ID_DUPLICATE;
1995 UsedOprMask |= OprMask;
1996 return encodeCustomOperandVal(Op, InputVal);
1997 }
1998 }
1999 return InvalidId;
2000}
2001
2002//===----------------------------------------------------------------------===//
2003// DepCtr
2004//===----------------------------------------------------------------------===//
2005
2006namespace DepCtr {
2007
2009 static int Default = -1;
2010 if (Default == -1)
2012 return Default;
2013}
2014
2015bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal,
2016 const MCSubtargetInfo &STI) {
2018 HasNonDefaultVal, STI);
2019}
2020
2021bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val,
2022 bool &IsDefault, const MCSubtargetInfo &STI) {
2023 return decodeCustomOperand(DepCtrInfo, DEP_CTR_SIZE, Code, Id, Name, Val,
2024 IsDefault, STI);
2025}
2026
2027int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask,
2028 const MCSubtargetInfo &STI) {
2029 return encodeCustomOperand(DepCtrInfo, DEP_CTR_SIZE, Name, Val, UsedOprMask,
2030 STI);
2031}
2032
2033unsigned getVaVdstBitMask() { return (1 << getVaVdstBitWidth()) - 1; }
2034
2035unsigned getVaSdstBitMask() { return (1 << getVaSdstBitWidth()) - 1; }
2036
2037unsigned getVaSsrcBitMask() { return (1 << getVaSsrcBitWidth()) - 1; }
2038
2040 return (1 << getHoldCntWidth(Version.Major, Version.Minor)) - 1;
2041}
2042
2043unsigned getVmVsrcBitMask() { return (1 << getVmVsrcBitWidth()) - 1; }
2044
2045unsigned getVaVccBitMask() { return (1 << getVaVccBitWidth()) - 1; }
2046
2047unsigned getSaSdstBitMask() { return (1 << getSaSdstBitWidth()) - 1; }
2048
2049unsigned decodeFieldVmVsrc(unsigned Encoded) {
2050 return unpackBits(Encoded, getVmVsrcBitShift(), getVmVsrcBitWidth());
2051}
2052
2053unsigned decodeFieldVaVdst(unsigned Encoded) {
2054 return unpackBits(Encoded, getVaVdstBitShift(), getVaVdstBitWidth());
2055}
2056
2057unsigned decodeFieldSaSdst(unsigned Encoded) {
2058 return unpackBits(Encoded, getSaSdstBitShift(), getSaSdstBitWidth());
2059}
2060
2061unsigned decodeFieldVaSdst(unsigned Encoded) {
2062 return unpackBits(Encoded, getVaSdstBitShift(), getVaSdstBitWidth());
2063}
2064
2065unsigned decodeFieldVaVcc(unsigned Encoded) {
2066 return unpackBits(Encoded, getVaVccBitShift(), getVaVccBitWidth());
2067}
2068
2069unsigned decodeFieldVaSsrc(unsigned Encoded) {
2070 return unpackBits(Encoded, getVaSsrcBitShift(), getVaSsrcBitWidth());
2071}
2072
2073unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version) {
2074 return unpackBits(Encoded, getHoldCntBitShift(),
2075 getHoldCntWidth(Version.Major, Version.Minor));
2076}
2077
2078unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc) {
2079 return packBits(VmVsrc, Encoded, getVmVsrcBitShift(), getVmVsrcBitWidth());
2080}
2081
2082unsigned encodeFieldVmVsrc(unsigned VmVsrc, const MCSubtargetInfo &STI) {
2083 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2084 return encodeFieldVmVsrc(Encoded, VmVsrc);
2085}
2086
2087unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst) {
2088 return packBits(VaVdst, Encoded, getVaVdstBitShift(), getVaVdstBitWidth());
2089}
2090
2091unsigned encodeFieldVaVdst(unsigned VaVdst, const MCSubtargetInfo &STI) {
2092 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2093 return encodeFieldVaVdst(Encoded, VaVdst);
2094}
2095
2096unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst) {
2097 return packBits(SaSdst, Encoded, getSaSdstBitShift(), getSaSdstBitWidth());
2098}
2099
2100unsigned encodeFieldSaSdst(unsigned SaSdst, const MCSubtargetInfo &STI) {
2101 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2102 return encodeFieldSaSdst(Encoded, SaSdst);
2103}
2104
2105unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst) {
2106 return packBits(VaSdst, Encoded, getVaSdstBitShift(), getVaSdstBitWidth());
2107}
2108
2109unsigned encodeFieldVaSdst(unsigned VaSdst, const MCSubtargetInfo &STI) {
2110 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2111 return encodeFieldVaSdst(Encoded, VaSdst);
2112}
2113
2114unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc) {
2115 return packBits(VaVcc, Encoded, getVaVccBitShift(), getVaVccBitWidth());
2116}
2117
2118unsigned encodeFieldVaVcc(unsigned VaVcc, const MCSubtargetInfo &STI) {
2119 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2120 return encodeFieldVaVcc(Encoded, VaVcc);
2121}
2122
2123unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc) {
2124 return packBits(VaSsrc, Encoded, getVaSsrcBitShift(), getVaSsrcBitWidth());
2125}
2126
2127unsigned encodeFieldVaSsrc(unsigned VaSsrc, const MCSubtargetInfo &STI) {
2128 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2129 return encodeFieldVaSsrc(Encoded, VaSsrc);
2130}
2131
2132unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt,
2133 const IsaVersion &Version) {
2134 return packBits(HoldCnt, Encoded, getHoldCntBitShift(),
2135 getHoldCntWidth(Version.Major, Version.Minor));
2136}
2137
2138unsigned encodeFieldHoldCnt(unsigned HoldCnt, const MCSubtargetInfo &STI) {
2139 unsigned Encoded = getDefaultDepCtrEncoding(STI);
2140 return encodeFieldHoldCnt(Encoded, HoldCnt, getIsaVersion(STI.getCPU()));
2141}
2142
2143} // namespace DepCtr
2144
2145//===----------------------------------------------------------------------===//
2146// exp tgt
2147//===----------------------------------------------------------------------===//
2148
2149namespace Exp {
2150
2151struct ExpTgt {
2153 unsigned Tgt;
2154 unsigned MaxIndex;
2155};
2156
2157// clang-format off
2158static constexpr ExpTgt ExpTgtInfo[] = {
2159 {{"null"}, ET_NULL, ET_NULL_MAX_IDX},
2160 {{"mrtz"}, ET_MRTZ, ET_MRTZ_MAX_IDX},
2161 {{"prim"}, ET_PRIM, ET_PRIM_MAX_IDX},
2162 {{"mrt"}, ET_MRT0, ET_MRT_MAX_IDX},
2163 {{"pos"}, ET_POS0, ET_POS_MAX_IDX},
2164 {{"dual_src_blend"},ET_DUAL_SRC_BLEND0, ET_DUAL_SRC_BLEND_MAX_IDX},
2165 {{"param"}, ET_PARAM0, ET_PARAM_MAX_IDX},
2166};
2167// clang-format on
2168
2169bool getTgtName(unsigned Id, StringRef &Name, int &Index) {
2170 for (const ExpTgt &Val : ExpTgtInfo) {
2171 if (Val.Tgt <= Id && Id <= Val.Tgt + Val.MaxIndex) {
2172 Index = (Val.MaxIndex == 0) ? -1 : (Id - Val.Tgt);
2173 Name = Val.Name;
2174 return true;
2175 }
2176 }
2177 return false;
2178}
2179
2180unsigned getTgtId(const StringRef Name) {
2181
2182 for (const ExpTgt &Val : ExpTgtInfo) {
2183 if (Val.MaxIndex == 0 && Name == Val.Name)
2184 return Val.Tgt;
2185
2186 if (Val.MaxIndex > 0 && Name.starts_with(Val.Name)) {
2187 StringRef Suffix = Name.drop_front(Val.Name.size());
2188
2189 unsigned Id;
2190 if (Suffix.getAsInteger(10, Id) || Id > Val.MaxIndex)
2191 return ET_INVALID;
2192
2193 // Disable leading zeroes
2194 if (Suffix.size() > 1 && Suffix[0] == '0')
2195 return ET_INVALID;
2196
2197 return Val.Tgt + Id;
2198 }
2199 }
2200 return ET_INVALID;
2201}
2202
2203bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI) {
2204 switch (Id) {
2205 case ET_NULL:
2206 return !isGFX11Plus(STI);
2207 case ET_POS4:
2208 case ET_PRIM:
2209 return isGFX10Plus(STI);
2210 case ET_DUAL_SRC_BLEND0:
2211 case ET_DUAL_SRC_BLEND1:
2212 return isGFX11Plus(STI);
2213 default:
2214 if (Id >= ET_PARAM0 && Id <= ET_PARAM31)
2215 return !isGFX11Plus(STI) || isGFX13Plus(STI);
2216 return true;
2217 }
2218}
2219
2220} // namespace Exp
2221
2222//===----------------------------------------------------------------------===//
2223// MTBUF Format
2224//===----------------------------------------------------------------------===//
2225
2226namespace MTBUFFormat {
2227
2228int64_t getDfmt(const StringRef Name) {
2229 for (int Id = DFMT_MIN; Id <= DFMT_MAX; ++Id) {
2230 if (Name == DfmtSymbolic[Id])
2231 return Id;
2232 }
2233 return DFMT_UNDEF;
2234}
2235
2237 assert(Id <= DFMT_MAX);
2238 return DfmtSymbolic[Id];
2239}
2240
2242 if (isSI(STI) || isCI(STI))
2243 return NfmtSymbolicSICI;
2244 if (isVI(STI) || isGFX9(STI))
2245 return NfmtSymbolicVI;
2246 return NfmtSymbolicGFX10;
2247}
2248
2249int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI) {
2250 const auto *lookupTable = getNfmtLookupTable(STI);
2251 for (int Id = NFMT_MIN; Id <= NFMT_MAX; ++Id) {
2252 if (Name == lookupTable[Id])
2253 return Id;
2254 }
2255 return NFMT_UNDEF;
2256}
2257
2258StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI) {
2259 assert(Id <= NFMT_MAX);
2260 return getNfmtLookupTable(STI)[Id];
2261}
2262
2263bool isValidDfmtNfmt(unsigned Id, const MCSubtargetInfo &STI) {
2264 unsigned Dfmt;
2265 unsigned Nfmt;
2266 decodeDfmtNfmt(Id, Dfmt, Nfmt);
2267 return isValidNfmt(Nfmt, STI);
2268}
2269
2270bool isValidNfmt(unsigned Id, const MCSubtargetInfo &STI) {
2271 return !getNfmtName(Id, STI).empty();
2272}
2273
2274int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt) {
2275 return (Dfmt << DFMT_SHIFT) | (Nfmt << NFMT_SHIFT);
2276}
2277
2278void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt) {
2279 Dfmt = (Format >> DFMT_SHIFT) & DFMT_MASK;
2280 Nfmt = (Format >> NFMT_SHIFT) & NFMT_MASK;
2281}
2282
2283int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI) {
2284 if (isGFX11Plus(STI)) {
2285 for (int Id = UfmtGFX11::UFMT_FIRST; Id <= UfmtGFX11::UFMT_LAST; ++Id) {
2286 if (Name == UfmtSymbolicGFX11[Id])
2287 return Id;
2288 }
2289 } else {
2290 for (int Id = UfmtGFX10::UFMT_FIRST; Id <= UfmtGFX10::UFMT_LAST; ++Id) {
2291 if (Name == UfmtSymbolicGFX10[Id])
2292 return Id;
2293 }
2294 }
2295 return UFMT_UNDEF;
2296}
2297
2299 if (isValidUnifiedFormat(Id, STI))
2300 return isGFX10(STI) ? UfmtSymbolicGFX10[Id] : UfmtSymbolicGFX11[Id];
2301 return "";
2302}
2303
2304bool isValidUnifiedFormat(unsigned Id, const MCSubtargetInfo &STI) {
2305 return isGFX10(STI) ? Id <= UfmtGFX10::UFMT_LAST : Id <= UfmtGFX11::UFMT_LAST;
2306}
2307
2308int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt,
2309 const MCSubtargetInfo &STI) {
2310 int64_t Fmt = encodeDfmtNfmt(Dfmt, Nfmt);
2311 if (isGFX11Plus(STI)) {
2312 for (int Id = UfmtGFX11::UFMT_FIRST; Id <= UfmtGFX11::UFMT_LAST; ++Id) {
2313 if (Fmt == DfmtNfmt2UFmtGFX11[Id])
2314 return Id;
2315 }
2316 } else {
2317 for (int Id = UfmtGFX10::UFMT_FIRST; Id <= UfmtGFX10::UFMT_LAST; ++Id) {
2318 if (Fmt == DfmtNfmt2UFmtGFX10[Id])
2319 return Id;
2320 }
2321 }
2322 return UFMT_UNDEF;
2323}
2324
2325bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI) {
2326 return isGFX10Plus(STI) ? (Val <= UFMT_MAX) : (Val <= DFMT_NFMT_MAX);
2327}
2328
2330 if (isGFX10Plus(STI))
2331 return UFMT_DEFAULT;
2332 return DFMT_NFMT_DEFAULT;
2333}
2334
2335} // namespace MTBUFFormat
2336
2337//===----------------------------------------------------------------------===//
2338// SendMsg
2339//===----------------------------------------------------------------------===//
2340
2341namespace SendMsg {
2342
2346
2347bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI) {
2348 return (MsgId & ~(getMsgIdMask(STI))) == 0;
2349}
2350
2351bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI,
2352 bool Strict) {
2353 assert(isValidMsgId(MsgId, STI));
2354
2355 if (!Strict)
2356 return 0 <= OpId && isUInt<OP_WIDTH_>(OpId);
2357
2358 if (msgRequiresOp(MsgId, STI)) {
2359 if (MsgId == ID_GS_PreGFX11 && OpId == OP_GS_NOP)
2360 return false;
2361
2362 return !getMsgOpName(MsgId, OpId, STI).empty();
2363 }
2364
2365 return OpId == OP_NONE_;
2366}
2367
2368bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId,
2369 const MCSubtargetInfo &STI, bool Strict) {
2370 assert(isValidMsgOp(MsgId, OpId, STI, Strict));
2371
2372 if (!Strict)
2374
2375 if (!isGFX11Plus(STI)) {
2376 switch (MsgId) {
2377 case ID_GS_PreGFX11:
2380 return (OpId == OP_GS_NOP)
2383 }
2384 }
2385 return StreamId == STREAM_ID_NONE_;
2386}
2387
2388bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI) {
2389 return MsgId == ID_SYSMSG ||
2390 (!isGFX11Plus(STI) &&
2391 (MsgId == ID_GS_PreGFX11 || MsgId == ID_GS_DONE_PreGFX11));
2392}
2393
2394bool msgSupportsStream(int64_t MsgId, int64_t OpId,
2395 const MCSubtargetInfo &STI) {
2396 return !isGFX11Plus(STI) &&
2397 (MsgId == ID_GS_PreGFX11 || MsgId == ID_GS_DONE_PreGFX11) &&
2398 OpId != OP_GS_NOP;
2399}
2400
2401void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId,
2402 uint16_t &StreamId, const MCSubtargetInfo &STI) {
2403 MsgId = Val & getMsgIdMask(STI);
2404 if (isGFX11Plus(STI)) {
2405 OpId = 0;
2406 StreamId = 0;
2407 } else {
2408 OpId = (Val & OP_MASK_) >> OP_SHIFT_;
2410 }
2411}
2412
2414 return MsgId | (OpId << OP_SHIFT_) | (StreamId << STREAM_ID_SHIFT_);
2415}
2416
2417bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI) {
2418 // Explicitly list message types that are known to not use m0.
2419 // This is safer than excluding only GS_ALLOC_REQ, in case new message
2420 // types are added in the future that do use m0.
2421 if (isGFX11Plus(STI)) {
2422 switch (MsgId) {
2424 return true;
2425 default:
2426 break;
2427 }
2428 }
2429 switch (MsgId) {
2430 case ID_SAVEWAVE:
2431 case ID_STALL_WAVE_GEN:
2432 case ID_HALT_WAVES:
2433 case ID_ORDERED_PS_DONE:
2435 case ID_GET_DOORBELL:
2436 case ID_GET_DDID:
2437 case ID_SYSMSG:
2438 return true;
2439 default:
2440 return false;
2441 }
2442}
2443
2444} // namespace SendMsg
2445
2446//===----------------------------------------------------------------------===//
2447//
2448//===----------------------------------------------------------------------===//
2449
2451 return F.getFnAttributeAsParsedInteger("InitialPSInputAddr", 0);
2452}
2453
2455 // As a safe default always respond as if PS has color exports.
2456 return F.getFnAttributeAsParsedInteger(
2457 "amdgpu-color-export",
2458 F.getCallingConv() == CallingConv::AMDGPU_PS ? 1 : 0) != 0;
2459}
2460
2462 return F.getFnAttributeAsParsedInteger("amdgpu-depth-export", 0) != 0;
2463}
2464
2466 unsigned BlockSize =
2467 F.getFnAttributeAsParsedInteger("amdgpu-dynamic-vgpr-block-size", 0);
2468
2469 if (BlockSize == 16 || BlockSize == 32)
2470 return BlockSize;
2471
2472 return 0;
2473}
2474
2475bool hasXNACK(const MCSubtargetInfo &STI) {
2476 return STI.hasFeature(AMDGPU::FeatureXNACK);
2477}
2478
2480 return STI.hasFeature(AMDGPU::FeatureMIMG_R128) &&
2481 !STI.hasFeature(AMDGPU::FeatureR128A16);
2482}
2483
2484bool hasA16(const MCSubtargetInfo &STI) {
2485 return STI.hasFeature(AMDGPU::FeatureA16);
2486}
2487
2488bool hasG16(const MCSubtargetInfo &STI) {
2489 return STI.hasFeature(AMDGPU::FeatureG16);
2490}
2491
2493 return !STI.hasFeature(AMDGPU::FeatureUnpackedD16VMem) && !isCI(STI) &&
2494 !isSI(STI);
2495}
2496
2497bool hasGDS(const MCSubtargetInfo &STI) {
2498 return STI.hasFeature(AMDGPU::FeatureGDS);
2499}
2500
2501unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler) {
2502 auto Version = getIsaVersion(STI.getCPU());
2503 if (Version.Major == 10)
2504 return Version.Minor >= 3 ? 13 : 5;
2505 if (Version.Major == 11)
2506 return 5;
2507 if (Version.Major >= 12)
2508 return HasSampler ? 4 : 5;
2509 return 0;
2510}
2511
2513 if (isGFX1250Plus(STI))
2514 return 32;
2515 return 16;
2516}
2517
2518bool isSI(const MCSubtargetInfo &STI) {
2519 return STI.hasFeature(AMDGPU::FeatureSouthernIslands);
2520}
2521
2522bool isCI(const MCSubtargetInfo &STI) {
2523 return STI.hasFeature(AMDGPU::FeatureSeaIslands);
2524}
2525
2526bool isVI(const MCSubtargetInfo &STI) {
2527 return STI.hasFeature(AMDGPU::FeatureVolcanicIslands);
2528}
2529
2530bool isGFX9(const MCSubtargetInfo &STI) {
2531 return STI.hasFeature(AMDGPU::FeatureGFX9);
2532}
2533
2535 return isGFX9(STI) || isGFX10(STI);
2536}
2537
2539 return isGFX9(STI) || isGFX10(STI) || isGFX11(STI);
2540}
2541
2543 return isVI(STI) || isGFX9(STI) || isGFX10(STI);
2544}
2545
2546bool isGFX8Plus(const MCSubtargetInfo &STI) {
2547 return isVI(STI) || isGFX9Plus(STI);
2548}
2549
2550bool isGFX9Plus(const MCSubtargetInfo &STI) {
2551 return isGFX9(STI) || isGFX10Plus(STI);
2552}
2553
2554bool isNotGFX9Plus(const MCSubtargetInfo &STI) { return !isGFX9Plus(STI); }
2555
2557 return STI.hasFeature(AMDGPU::FeaturePopsExitingWaveID);
2558}
2559
2561 return STI.hasFeature(AMDGPU::FeatureApertureRegs) &&
2562 !STI.hasFeature(AMDGPU::FeatureGloballyAddressableScratch);
2563}
2564
2565bool isGFX10(const MCSubtargetInfo &STI) {
2566 return STI.hasFeature(AMDGPU::FeatureGFX10);
2567}
2568
2570 return isGFX10(STI) || isGFX11(STI);
2571}
2572
2574 return isGFX10(STI) || isGFX11Plus(STI);
2575}
2576
2577bool isGFX11(const MCSubtargetInfo &STI) {
2578 return STI.hasFeature(AMDGPU::FeatureGFX11);
2579}
2580
2582 return isGFX11(STI) || isGFX12Plus(STI);
2583}
2584
2585bool isGFX12(const MCSubtargetInfo &STI) {
2586 return STI.getFeatureBits()[AMDGPU::FeatureGFX12];
2587}
2588
2590 return isGFX12(STI) || isGFX13Plus(STI);
2591}
2592
2593bool isNotGFX12Plus(const MCSubtargetInfo &STI) { return !isGFX12Plus(STI); }
2594
2595bool isGFX1250(const MCSubtargetInfo &STI) {
2596 return STI.getFeatureBits()[AMDGPU::FeatureGFX1250Insts] && !isGFX13(STI);
2597}
2598
2600 return isGFX1250(STI) || !STI.getFeatureBits().test(FeatureCuMode);
2601}
2602
2604 return STI.getFeatureBits()[AMDGPU::FeatureGFX1250Insts];
2605}
2606
2607bool isGFX13(const MCSubtargetInfo &STI) {
2608 return STI.getFeatureBits()[AMDGPU::FeatureGFX13];
2609}
2610
2611bool isGFX13Plus(const MCSubtargetInfo &STI) { return isGFX13(STI); }
2612
2614 if (isGFX1250(STI))
2615 return false;
2616 return isGFX10Plus(STI);
2617}
2618
2619bool isNotGFX11Plus(const MCSubtargetInfo &STI) { return !isGFX11Plus(STI); }
2620
2622 return isSI(STI) || isCI(STI) || isVI(STI) || isGFX9(STI);
2623}
2624
2626 return isGFX10(STI) && !AMDGPU::isGFX10_BEncoding(STI);
2627}
2628
2630 return STI.hasFeature(AMDGPU::FeatureGCN3Encoding);
2631}
2632
2634 return STI.hasFeature(AMDGPU::FeatureGFX10_BEncoding);
2635}
2636
2638 return STI.hasFeature(AMDGPU::FeatureGFX10_3Insts);
2639}
2640
2642 return isGFX10_BEncoding(STI) && !isGFX12Plus(STI);
2643}
2644
2645bool isGFX90A(const MCSubtargetInfo &STI) {
2646 return STI.hasFeature(AMDGPU::FeatureGFX90AInsts);
2647}
2648
2649bool isGFX940(const MCSubtargetInfo &STI) {
2650 return STI.hasFeature(AMDGPU::FeatureGFX940Insts);
2651}
2652
2654 return STI.hasFeature(AMDGPU::FeatureArchitectedFlatScratch);
2655}
2656
2658 return STI.hasFeature(AMDGPU::FeatureMAIInsts);
2659}
2660
2661bool hasVOPD(const MCSubtargetInfo &STI) {
2662 return STI.hasFeature(AMDGPU::FeatureVOPDInsts);
2663}
2664
2666 return STI.hasFeature(AMDGPU::FeatureDPPSrc1SGPR);
2667}
2668
2670 return STI.hasFeature(AMDGPU::FeatureKernargPreload);
2671}
2672
2673int32_t getTotalNumVGPRs(bool has90AInsts, int32_t ArgNumAGPR,
2674 int32_t ArgNumVGPR) {
2675 if (has90AInsts && ArgNumAGPR)
2676 return alignTo(ArgNumVGPR, 4) + ArgNumAGPR;
2677 return std::max(ArgNumVGPR, ArgNumAGPR);
2678}
2679
2681 const MCRegisterClass &SGPRClass =
2682 TRI->getRegClass(AMDGPU::SReg_32RegClassID);
2683 const MCRegister FirstSubReg = TRI->getSubReg(Reg, AMDGPU::sub0);
2684 return SGPRClass.contains(FirstSubReg != 0 ? FirstSubReg : Reg) ||
2685 Reg == AMDGPU::SCC;
2686}
2687
2691
2692#define MAP_REG2REG \
2693 using namespace AMDGPU; \
2694 switch (Reg.id()) { \
2695 default: \
2696 return Reg; \
2697 CASE_CI_VI(FLAT_SCR) \
2698 CASE_CI_VI(FLAT_SCR_LO) \
2699 CASE_CI_VI(FLAT_SCR_HI) \
2700 CASE_VI_GFX9PLUS(TTMP0) \
2701 CASE_VI_GFX9PLUS(TTMP1) \
2702 CASE_VI_GFX9PLUS(TTMP2) \
2703 CASE_VI_GFX9PLUS(TTMP3) \
2704 CASE_VI_GFX9PLUS(TTMP4) \
2705 CASE_VI_GFX9PLUS(TTMP5) \
2706 CASE_VI_GFX9PLUS(TTMP6) \
2707 CASE_VI_GFX9PLUS(TTMP7) \
2708 CASE_VI_GFX9PLUS(TTMP8) \
2709 CASE_VI_GFX9PLUS(TTMP9) \
2710 CASE_VI_GFX9PLUS(TTMP10) \
2711 CASE_VI_GFX9PLUS(TTMP11) \
2712 CASE_VI_GFX9PLUS(TTMP12) \
2713 CASE_VI_GFX9PLUS(TTMP13) \
2714 CASE_VI_GFX9PLUS(TTMP14) \
2715 CASE_VI_GFX9PLUS(TTMP15) \
2716 CASE_VI_GFX9PLUS(TTMP0_TTMP1) \
2717 CASE_VI_GFX9PLUS(TTMP2_TTMP3) \
2718 CASE_VI_GFX9PLUS(TTMP4_TTMP5) \
2719 CASE_VI_GFX9PLUS(TTMP6_TTMP7) \
2720 CASE_VI_GFX9PLUS(TTMP8_TTMP9) \
2721 CASE_VI_GFX9PLUS(TTMP10_TTMP11) \
2722 CASE_VI_GFX9PLUS(TTMP12_TTMP13) \
2723 CASE_VI_GFX9PLUS(TTMP14_TTMP15) \
2724 CASE_VI_GFX9PLUS(TTMP0_TTMP1_TTMP2_TTMP3) \
2725 CASE_VI_GFX9PLUS(TTMP4_TTMP5_TTMP6_TTMP7) \
2726 CASE_VI_GFX9PLUS(TTMP8_TTMP9_TTMP10_TTMP11) \
2727 CASE_VI_GFX9PLUS(TTMP12_TTMP13_TTMP14_TTMP15) \
2728 CASE_VI_GFX9PLUS(TTMP0_TTMP1_TTMP2_TTMP3_TTMP4_TTMP5_TTMP6_TTMP7) \
2729 CASE_VI_GFX9PLUS(TTMP4_TTMP5_TTMP6_TTMP7_TTMP8_TTMP9_TTMP10_TTMP11) \
2730 CASE_VI_GFX9PLUS(TTMP8_TTMP9_TTMP10_TTMP11_TTMP12_TTMP13_TTMP14_TTMP15) \
2731 CASE_VI_GFX9PLUS( \
2732 TTMP0_TTMP1_TTMP2_TTMP3_TTMP4_TTMP5_TTMP6_TTMP7_TTMP8_TTMP9_TTMP10_TTMP11_TTMP12_TTMP13_TTMP14_TTMP15) \
2733 CASE_GFXPRE11_GFX11PLUS(M0) \
2734 CASE_GFXPRE11_GFX11PLUS(SGPR_NULL) \
2735 CASE_GFXPRE11_GFX11PLUS_TO(SGPR_NULL64, SGPR_NULL) \
2736 }
2737
2738#define CASE_CI_VI(node) \
2739 assert(!isSI(STI)); \
2740 case node: \
2741 return isCI(STI) ? node##_ci : node##_vi;
2742
2743#define CASE_VI_GFX9PLUS(node) \
2744 case node: \
2745 return isGFX9Plus(STI) ? node##_gfx9plus : node##_vi;
2746
2747#define CASE_GFXPRE11_GFX11PLUS(node) \
2748 case node: \
2749 return isGFX11Plus(STI) ? node##_gfx11plus : node##_gfxpre11;
2750
2751#define CASE_GFXPRE11_GFX11PLUS_TO(node, result) \
2752 case node: \
2753 return isGFX11Plus(STI) ? result##_gfx11plus : result##_gfxpre11;
2754
2756 if (STI.getTargetTriple().getArch() == Triple::r600)
2757 return Reg;
2759}
2760
2761#undef CASE_CI_VI
2762#undef CASE_VI_GFX9PLUS
2763#undef CASE_GFXPRE11_GFX11PLUS
2764#undef CASE_GFXPRE11_GFX11PLUS_TO
2765
2766#define CASE_CI_VI(node) \
2767 case node##_ci: \
2768 case node##_vi: \
2769 return node;
2770#define CASE_VI_GFX9PLUS(node) \
2771 case node##_vi: \
2772 case node##_gfx9plus: \
2773 return node;
2774#define CASE_GFXPRE11_GFX11PLUS(node) \
2775 case node##_gfx11plus: \
2776 case node##_gfxpre11: \
2777 return node;
2778#define CASE_GFXPRE11_GFX11PLUS_TO(node, result)
2779
2781
2783 switch (Reg.id()) {
2784 case AMDGPU::SRC_SHARED_BASE_LO:
2785 case AMDGPU::SRC_SHARED_BASE:
2786 case AMDGPU::SRC_SHARED_LIMIT_LO:
2787 case AMDGPU::SRC_SHARED_LIMIT:
2788 case AMDGPU::SRC_PRIVATE_BASE_LO:
2789 case AMDGPU::SRC_PRIVATE_BASE:
2790 case AMDGPU::SRC_PRIVATE_LIMIT_LO:
2791 case AMDGPU::SRC_PRIVATE_LIMIT:
2792 case AMDGPU::SRC_FLAT_SCRATCH_BASE_LO:
2793 case AMDGPU::SRC_FLAT_SCRATCH_BASE_HI:
2794 case AMDGPU::SRC_POPS_EXITING_WAVE_ID:
2795 return true;
2796 case AMDGPU::SRC_VCCZ:
2797 case AMDGPU::SRC_EXECZ:
2798 case AMDGPU::SRC_SCC:
2799 return true;
2800 case AMDGPU::SGPR_NULL:
2801 return true;
2802 default:
2803 return false;
2804 }
2805}
2806
2807#undef CASE_CI_VI
2808#undef CASE_VI_GFX9PLUS
2809#undef CASE_GFXPRE11_GFX11PLUS
2810#undef CASE_GFXPRE11_GFX11PLUS_TO
2811#undef MAP_REG2REG
2812
2813bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2814 assert(OpNo < Desc.NumOperands);
2815 unsigned OpType = Desc.operands()[OpNo].OperandType;
2816 return OpType >= AMDGPU::OPERAND_KIMM_FIRST &&
2817 OpType <= AMDGPU::OPERAND_KIMM_LAST;
2818}
2819
2820bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2821 assert(OpNo < Desc.NumOperands);
2822 unsigned OpType = Desc.operands()[OpNo].OperandType;
2823 switch (OpType) {
2838 return true;
2839 default:
2840 return false;
2841 }
2842}
2843
2844bool isSISrcInlinableOperand(const MCInstrDesc &Desc, unsigned OpNo) {
2845 assert(OpNo < Desc.NumOperands);
2846 unsigned OpType = Desc.operands()[OpNo].OperandType;
2847 return (OpType >= AMDGPU::OPERAND_REG_INLINE_C_FIRST &&
2851}
2852
2853// Avoid using MCRegisterClass::getSize, since that function will go away
2854// (move from MC* level to Target* level). Return size in bits.
2855unsigned getRegBitWidth(unsigned RCID) {
2856 switch (RCID) {
2857 case AMDGPU::VGPR_16RegClassID:
2858 case AMDGPU::VGPR_16_Lo128RegClassID:
2859 case AMDGPU::SGPR_LO16RegClassID:
2860 case AMDGPU::AGPR_LO16RegClassID:
2861 return 16;
2862 case AMDGPU::SGPR_32RegClassID:
2863 case AMDGPU::VGPR_32RegClassID:
2864 case AMDGPU::VGPR_32_Lo256RegClassID:
2865 case AMDGPU::VRegOrLds_32RegClassID:
2866 case AMDGPU::AGPR_32RegClassID:
2867 case AMDGPU::VS_32RegClassID:
2868 case AMDGPU::AV_32RegClassID:
2869 case AMDGPU::SReg_32RegClassID:
2870 case AMDGPU::SReg_32_XM0RegClassID:
2871 case AMDGPU::SRegOrLds_32RegClassID:
2872 return 32;
2873 case AMDGPU::SGPR_64RegClassID:
2874 case AMDGPU::VS_64RegClassID:
2875 case AMDGPU::SReg_64RegClassID:
2876 case AMDGPU::VReg_64RegClassID:
2877 case AMDGPU::AReg_64RegClassID:
2878 case AMDGPU::SReg_64_XEXECRegClassID:
2879 case AMDGPU::VReg_64_Align2RegClassID:
2880 case AMDGPU::AReg_64_Align2RegClassID:
2881 case AMDGPU::AV_64RegClassID:
2882 case AMDGPU::AV_64_Align2RegClassID:
2883 case AMDGPU::VReg_64_Lo256_Align2RegClassID:
2884 case AMDGPU::VS_64_Lo256RegClassID:
2885 return 64;
2886 case AMDGPU::SGPR_96RegClassID:
2887 case AMDGPU::SReg_96RegClassID:
2888 case AMDGPU::VReg_96RegClassID:
2889 case AMDGPU::AReg_96RegClassID:
2890 case AMDGPU::VReg_96_Align2RegClassID:
2891 case AMDGPU::AReg_96_Align2RegClassID:
2892 case AMDGPU::AV_96RegClassID:
2893 case AMDGPU::AV_96_Align2RegClassID:
2894 case AMDGPU::VReg_96_Lo256_Align2RegClassID:
2895 return 96;
2896 case AMDGPU::SGPR_128RegClassID:
2897 case AMDGPU::SReg_128RegClassID:
2898 case AMDGPU::VReg_128RegClassID:
2899 case AMDGPU::AReg_128RegClassID:
2900 case AMDGPU::VReg_128_Align2RegClassID:
2901 case AMDGPU::AReg_128_Align2RegClassID:
2902 case AMDGPU::AV_128RegClassID:
2903 case AMDGPU::AV_128_Align2RegClassID:
2904 case AMDGPU::SReg_128_XNULLRegClassID:
2905 case AMDGPU::VReg_128_Lo256_Align2RegClassID:
2906 return 128;
2907 case AMDGPU::SGPR_160RegClassID:
2908 case AMDGPU::SReg_160RegClassID:
2909 case AMDGPU::VReg_160RegClassID:
2910 case AMDGPU::AReg_160RegClassID:
2911 case AMDGPU::VReg_160_Align2RegClassID:
2912 case AMDGPU::AReg_160_Align2RegClassID:
2913 case AMDGPU::AV_160RegClassID:
2914 case AMDGPU::AV_160_Align2RegClassID:
2915 case AMDGPU::VReg_160_Lo256_Align2RegClassID:
2916 return 160;
2917 case AMDGPU::SGPR_192RegClassID:
2918 case AMDGPU::SReg_192RegClassID:
2919 case AMDGPU::VReg_192RegClassID:
2920 case AMDGPU::AReg_192RegClassID:
2921 case AMDGPU::VReg_192_Align2RegClassID:
2922 case AMDGPU::AReg_192_Align2RegClassID:
2923 case AMDGPU::AV_192RegClassID:
2924 case AMDGPU::AV_192_Align2RegClassID:
2925 case AMDGPU::VReg_192_Lo256_Align2RegClassID:
2926 return 192;
2927 case AMDGPU::SGPR_224RegClassID:
2928 case AMDGPU::SReg_224RegClassID:
2929 case AMDGPU::VReg_224RegClassID:
2930 case AMDGPU::AReg_224RegClassID:
2931 case AMDGPU::VReg_224_Align2RegClassID:
2932 case AMDGPU::AReg_224_Align2RegClassID:
2933 case AMDGPU::AV_224RegClassID:
2934 case AMDGPU::AV_224_Align2RegClassID:
2935 case AMDGPU::VReg_224_Lo256_Align2RegClassID:
2936 return 224;
2937 case AMDGPU::SGPR_256RegClassID:
2938 case AMDGPU::SReg_256RegClassID:
2939 case AMDGPU::VReg_256RegClassID:
2940 case AMDGPU::AReg_256RegClassID:
2941 case AMDGPU::VReg_256_Align2RegClassID:
2942 case AMDGPU::AReg_256_Align2RegClassID:
2943 case AMDGPU::AV_256RegClassID:
2944 case AMDGPU::AV_256_Align2RegClassID:
2945 case AMDGPU::SReg_256_XNULLRegClassID:
2946 case AMDGPU::VReg_256_Lo256_Align2RegClassID:
2947 return 256;
2948 case AMDGPU::SGPR_288RegClassID:
2949 case AMDGPU::SReg_288RegClassID:
2950 case AMDGPU::VReg_288RegClassID:
2951 case AMDGPU::AReg_288RegClassID:
2952 case AMDGPU::VReg_288_Align2RegClassID:
2953 case AMDGPU::AReg_288_Align2RegClassID:
2954 case AMDGPU::AV_288RegClassID:
2955 case AMDGPU::AV_288_Align2RegClassID:
2956 case AMDGPU::VReg_288_Lo256_Align2RegClassID:
2957 return 288;
2958 case AMDGPU::SGPR_320RegClassID:
2959 case AMDGPU::SReg_320RegClassID:
2960 case AMDGPU::VReg_320RegClassID:
2961 case AMDGPU::AReg_320RegClassID:
2962 case AMDGPU::VReg_320_Align2RegClassID:
2963 case AMDGPU::AReg_320_Align2RegClassID:
2964 case AMDGPU::AV_320RegClassID:
2965 case AMDGPU::AV_320_Align2RegClassID:
2966 case AMDGPU::VReg_320_Lo256_Align2RegClassID:
2967 return 320;
2968 case AMDGPU::SGPR_352RegClassID:
2969 case AMDGPU::SReg_352RegClassID:
2970 case AMDGPU::VReg_352RegClassID:
2971 case AMDGPU::AReg_352RegClassID:
2972 case AMDGPU::VReg_352_Align2RegClassID:
2973 case AMDGPU::AReg_352_Align2RegClassID:
2974 case AMDGPU::AV_352RegClassID:
2975 case AMDGPU::AV_352_Align2RegClassID:
2976 case AMDGPU::VReg_352_Lo256_Align2RegClassID:
2977 return 352;
2978 case AMDGPU::SGPR_384RegClassID:
2979 case AMDGPU::SReg_384RegClassID:
2980 case AMDGPU::VReg_384RegClassID:
2981 case AMDGPU::AReg_384RegClassID:
2982 case AMDGPU::VReg_384_Align2RegClassID:
2983 case AMDGPU::AReg_384_Align2RegClassID:
2984 case AMDGPU::AV_384RegClassID:
2985 case AMDGPU::AV_384_Align2RegClassID:
2986 case AMDGPU::VReg_384_Lo256_Align2RegClassID:
2987 return 384;
2988 case AMDGPU::SGPR_512RegClassID:
2989 case AMDGPU::SReg_512RegClassID:
2990 case AMDGPU::VReg_512RegClassID:
2991 case AMDGPU::AReg_512RegClassID:
2992 case AMDGPU::VReg_512_Align2RegClassID:
2993 case AMDGPU::AReg_512_Align2RegClassID:
2994 case AMDGPU::AV_512RegClassID:
2995 case AMDGPU::AV_512_Align2RegClassID:
2996 case AMDGPU::VReg_512_Lo256_Align2RegClassID:
2997 return 512;
2998 case AMDGPU::SGPR_1024RegClassID:
2999 case AMDGPU::SReg_1024RegClassID:
3000 case AMDGPU::VReg_1024RegClassID:
3001 case AMDGPU::AReg_1024RegClassID:
3002 case AMDGPU::VReg_1024_Align2RegClassID:
3003 case AMDGPU::AReg_1024_Align2RegClassID:
3004 case AMDGPU::AV_1024RegClassID:
3005 case AMDGPU::AV_1024_Align2RegClassID:
3006 case AMDGPU::VReg_1024_Lo256_Align2RegClassID:
3007 return 1024;
3008 default:
3009 llvm_unreachable("Unexpected register class");
3010 }
3011}
3012
3013unsigned getRegBitWidth(const MCRegisterClass &RC) {
3014 return getRegBitWidth(RC.getID());
3015}
3016
3017bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi) {
3019 return true;
3020
3021 uint64_t Val = static_cast<uint64_t>(Literal);
3022 return (Val == llvm::bit_cast<uint64_t>(0.0)) ||
3023 (Val == llvm::bit_cast<uint64_t>(1.0)) ||
3024 (Val == llvm::bit_cast<uint64_t>(-1.0)) ||
3025 (Val == llvm::bit_cast<uint64_t>(0.5)) ||
3026 (Val == llvm::bit_cast<uint64_t>(-0.5)) ||
3027 (Val == llvm::bit_cast<uint64_t>(2.0)) ||
3028 (Val == llvm::bit_cast<uint64_t>(-2.0)) ||
3029 (Val == llvm::bit_cast<uint64_t>(4.0)) ||
3030 (Val == llvm::bit_cast<uint64_t>(-4.0)) ||
3031 (Val == 0x3fc45f306dc9c882 && HasInv2Pi);
3032}
3033
3034bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi) {
3036 return true;
3037
3038 // The actual type of the operand does not seem to matter as long
3039 // as the bits match one of the inline immediate values. For example:
3040 //
3041 // -nan has the hexadecimal encoding of 0xfffffffe which is -2 in decimal,
3042 // so it is a legal inline immediate.
3043 //
3044 // 1065353216 has the hexadecimal encoding 0x3f800000 which is 1.0f in
3045 // floating-point, so it is a legal inline immediate.
3046
3047 uint32_t Val = static_cast<uint32_t>(Literal);
3048 return (Val == llvm::bit_cast<uint32_t>(0.0f)) ||
3049 (Val == llvm::bit_cast<uint32_t>(1.0f)) ||
3050 (Val == llvm::bit_cast<uint32_t>(-1.0f)) ||
3051 (Val == llvm::bit_cast<uint32_t>(0.5f)) ||
3052 (Val == llvm::bit_cast<uint32_t>(-0.5f)) ||
3053 (Val == llvm::bit_cast<uint32_t>(2.0f)) ||
3054 (Val == llvm::bit_cast<uint32_t>(-2.0f)) ||
3055 (Val == llvm::bit_cast<uint32_t>(4.0f)) ||
3056 (Val == llvm::bit_cast<uint32_t>(-4.0f)) ||
3057 (Val == 0x3e22f983 && HasInv2Pi);
3058}
3059
3060bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi) {
3061 if (!HasInv2Pi)
3062 return false;
3064 return true;
3065 uint16_t Val = static_cast<uint16_t>(Literal);
3066 return Val == 0x3F00 || // 0.5
3067 Val == 0xBF00 || // -0.5
3068 Val == 0x3F80 || // 1.0
3069 Val == 0xBF80 || // -1.0
3070 Val == 0x4000 || // 2.0
3071 Val == 0xC000 || // -2.0
3072 Val == 0x4080 || // 4.0
3073 Val == 0xC080 || // -4.0
3074 Val == 0x3E22; // 1.0 / (2.0 * pi)
3075}
3076
3077bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi) {
3078 return isInlinableLiteral32(Literal, HasInv2Pi);
3079}
3080
3081bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi) {
3082 if (!HasInv2Pi)
3083 return false;
3085 return true;
3086 uint16_t Val = static_cast<uint16_t>(Literal);
3087 return Val == 0x3C00 || // 1.0
3088 Val == 0xBC00 || // -1.0
3089 Val == 0x3800 || // 0.5
3090 Val == 0xB800 || // -0.5
3091 Val == 0x4000 || // 2.0
3092 Val == 0xC000 || // -2.0
3093 Val == 0x4400 || // 4.0
3094 Val == 0xC400 || // -4.0
3095 Val == 0x3118; // 1/2pi
3096}
3097
3098std::optional<unsigned> getInlineEncodingV216(bool IsFloat, uint32_t Literal) {
3099 // Unfortunately, the Instruction Set Architecture Reference Guide is
3100 // misleading about how the inline operands work for (packed) 16-bit
3101 // instructions. In a nutshell, the actual HW behavior is:
3102 //
3103 // - integer encodings (-16 .. 64) are always produced as sign-extended
3104 // 32-bit values
3105 // - float encodings are produced as:
3106 // - for F16 instructions: corresponding half-precision float values in
3107 // the LSBs, 0 in the MSBs
3108 // - for UI16 instructions: corresponding single-precision float value
3109 int32_t Signed = static_cast<int32_t>(Literal);
3110 if (Signed >= 0 && Signed <= 64)
3111 return 128 + Signed;
3112
3113 if (Signed >= -16 && Signed <= -1)
3114 return 192 + std::abs(Signed);
3115
3116 if (IsFloat) {
3117 // clang-format off
3118 switch (Literal) {
3119 case 0x3800: return 240; // 0.5
3120 case 0xB800: return 241; // -0.5
3121 case 0x3C00: return 242; // 1.0
3122 case 0xBC00: return 243; // -1.0
3123 case 0x4000: return 244; // 2.0
3124 case 0xC000: return 245; // -2.0
3125 case 0x4400: return 246; // 4.0
3126 case 0xC400: return 247; // -4.0
3127 case 0x3118: return 248; // 1.0 / (2.0 * pi)
3128 default: break;
3129 }
3130 // clang-format on
3131 } else {
3132 // clang-format off
3133 switch (Literal) {
3134 case 0x3F000000: return 240; // 0.5
3135 case 0xBF000000: return 241; // -0.5
3136 case 0x3F800000: return 242; // 1.0
3137 case 0xBF800000: return 243; // -1.0
3138 case 0x40000000: return 244; // 2.0
3139 case 0xC0000000: return 245; // -2.0
3140 case 0x40800000: return 246; // 4.0
3141 case 0xC0800000: return 247; // -4.0
3142 case 0x3E22F983: return 248; // 1.0 / (2.0 * pi)
3143 default: break;
3144 }
3145 // clang-format on
3146 }
3147
3148 return {};
3149}
3150
3151// Encoding of the literal as an inline constant for a V_PK_*_IU16 instruction
3152// or nullopt.
3153std::optional<unsigned> getInlineEncodingV2I16(uint32_t Literal) {
3154 return getInlineEncodingV216(false, Literal);
3155}
3156
3157// Encoding of the literal as an inline constant for a V_PK_*_BF16 instruction
3158// or nullopt.
3159std::optional<unsigned> getInlineEncodingV2BF16(uint32_t Literal) {
3160 int32_t Signed = static_cast<int32_t>(Literal);
3161 if (Signed >= 0 && Signed <= 64)
3162 return 128 + Signed;
3163
3164 if (Signed >= -16 && Signed <= -1)
3165 return 192 + std::abs(Signed);
3166
3167 // clang-format off
3168 switch (Literal) {
3169 case 0x3F00: return 240; // 0.5
3170 case 0xBF00: return 241; // -0.5
3171 case 0x3F80: return 242; // 1.0
3172 case 0xBF80: return 243; // -1.0
3173 case 0x4000: return 244; // 2.0
3174 case 0xC000: return 245; // -2.0
3175 case 0x4080: return 246; // 4.0
3176 case 0xC080: return 247; // -4.0
3177 case 0x3E22: return 248; // 1.0 / (2.0 * pi)
3178 default: break;
3179 }
3180 // clang-format on
3181
3182 return std::nullopt;
3183}
3184
3185// Encoding of the literal as an inline constant for a V_PK_*_F16 instruction
3186// or nullopt.
3187std::optional<unsigned> getInlineEncodingV2F16(uint32_t Literal) {
3188 return getInlineEncodingV216(true, Literal);
3189}
3190
3191// Encoding of the literal as an inline constant for V_PK_FMAC_F16 instruction
3192// or nullopt. This accounts for different inline constant behavior:
3193// - Pre-GFX11: fp16 inline constants have the value in low 16 bits, 0 in high
3194// - GFX11+: fp16 inline constants are duplicated into both halves
3196 bool IsGFX11Plus) {
3197 // Pre-GFX11 behavior: f16 in low bits, 0 in high bits
3198 if (!IsGFX11Plus)
3199 return getInlineEncodingV216(/*IsFloat=*/true, Literal);
3200
3201 // GFX11+ behavior: f16 duplicated in both halves
3202 // First, check for sign-extended integer inline constants (-16 to 64)
3203 // These work the same across all generations
3204 int32_t Signed = static_cast<int32_t>(Literal);
3205 if (Signed >= 0 && Signed <= 64)
3206 return 128 + Signed;
3207
3208 if (Signed >= -16 && Signed <= -1)
3209 return 192 + std::abs(Signed);
3210
3211 // For float inline constants on GFX11+, both halves must be equal
3212 uint16_t Lo = static_cast<uint16_t>(Literal);
3213 uint16_t Hi = static_cast<uint16_t>(Literal >> 16);
3214 if (Lo != Hi)
3215 return std::nullopt;
3216 return getInlineEncodingV216(/*IsFloat=*/true, Lo);
3217}
3218
3219// Whether the given literal can be inlined for a V_PK_* instruction.
3221 switch (OpType) {
3224 return getInlineEncodingV216(false, Literal).has_value();
3227 return getInlineEncodingV216(true, Literal).has_value();
3229 llvm_unreachable("OPERAND_REG_IMM_V2FP16_SPLAT is not supported");
3234 return false;
3235 default:
3236 llvm_unreachable("bad packed operand type");
3237 }
3238}
3239
3240// Whether the given literal can be inlined for a V_PK_*_IU16 instruction.
3244
3245// Whether the given literal can be inlined for a V_PK_*_BF16 instruction.
3249
3250// Whether the given literal can be inlined for a V_PK_*_F16 instruction.
3254
3255// Whether the given literal can be inlined for V_PK_FMAC_F16 instruction.
3257 return getPKFMACF16InlineEncoding(Literal, IsGFX11Plus).has_value();
3258}
3259
3260bool isValid32BitLiteral(uint64_t Val, bool IsFP64) {
3261 if (IsFP64)
3262 return !Lo_32(Val);
3263
3264 return isUInt<32>(Val) || isInt<32>(Val);
3265}
3266
3267int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit) {
3268 switch (Type) {
3269 default:
3270 break;
3275 return Imm & 0xffff;
3289 return Lo_32(Imm);
3292 return IsLit ? Imm : Hi_32(Imm);
3293 }
3294 return Imm;
3295}
3296
3298 const Function *F = A->getParent();
3299
3300 // Arguments to compute shaders are never a source of divergence.
3301 CallingConv::ID CC = F->getCallingConv();
3302 switch (CC) {
3305 return true;
3316 // For non-compute shaders, SGPR inputs are marked with either inreg or
3317 // byval. Everything else is in VGPRs.
3318 return A->hasAttribute(Attribute::InReg) ||
3319 A->hasAttribute(Attribute::ByVal);
3320 default:
3321 // TODO: treat i1 as divergent?
3322 return A->hasAttribute(Attribute::InReg);
3323 }
3324}
3325
3326bool isArgPassedInSGPR(const CallBase *CB, unsigned ArgNo) {
3327 // Arguments to compute shaders are never a source of divergence.
3329 switch (CC) {
3332 return true;
3343 // For non-compute shaders, SGPR inputs are marked with either inreg or
3344 // byval. Everything else is in VGPRs.
3345 return CB->paramHasAttr(ArgNo, Attribute::InReg) ||
3346 CB->paramHasAttr(ArgNo, Attribute::ByVal);
3347 default:
3348 return CB->paramHasAttr(ArgNo, Attribute::InReg);
3349 }
3350}
3351
3352static bool hasSMEMByteOffset(const MCSubtargetInfo &ST) {
3353 return isGCN3Encoding(ST) || isGFX10Plus(ST);
3354}
3355
3357 int64_t EncodedOffset) {
3358 if (isGFX12Plus(ST))
3359 return isUInt<23>(EncodedOffset);
3360
3361 return hasSMEMByteOffset(ST) ? isUInt<20>(EncodedOffset)
3362 : isUInt<8>(EncodedOffset);
3363}
3364
3366 int64_t EncodedOffset, bool IsBuffer) {
3367 if (isGFX12Plus(ST)) {
3368 if (IsBuffer && EncodedOffset < 0)
3369 return false;
3370 return isInt<24>(EncodedOffset);
3371 }
3372
3373 return !IsBuffer && hasSMRDSignedImmOffset(ST) && isInt<21>(EncodedOffset);
3374}
3375
3376static bool isDwordAligned(uint64_t ByteOffset) {
3377 return (ByteOffset & 3) == 0;
3378}
3379
3381 uint64_t ByteOffset) {
3382 if (hasSMEMByteOffset(ST))
3383 return ByteOffset;
3384
3385 assert(isDwordAligned(ByteOffset));
3386 return ByteOffset >> 2;
3387}
3388
3389std::optional<int64_t> getSMRDEncodedOffset(const MCSubtargetInfo &ST,
3390 int64_t ByteOffset, bool IsBuffer,
3391 bool HasSOffset) {
3392 // For unbuffered smem loads, it is illegal for the Immediate Offset to be
3393 // negative if the resulting (Offset + (M0 or SOffset or zero) is negative.
3394 // Handle case where SOffset is not present.
3395 if (!IsBuffer && !HasSOffset && ByteOffset < 0 && hasSMRDSignedImmOffset(ST))
3396 return std::nullopt;
3397
3398 if (isGFX12Plus(ST)) // 24 bit signed offsets
3399 return isInt<24>(ByteOffset) ? std::optional<int64_t>(ByteOffset)
3400 : std::nullopt;
3401
3402 // The signed version is always a byte offset.
3403 if (!IsBuffer && hasSMRDSignedImmOffset(ST)) {
3405 return isInt<20>(ByteOffset) ? std::optional<int64_t>(ByteOffset)
3406 : std::nullopt;
3407 }
3408
3409 if (!isDwordAligned(ByteOffset) && !hasSMEMByteOffset(ST))
3410 return std::nullopt;
3411
3412 int64_t EncodedOffset = convertSMRDOffsetUnits(ST, ByteOffset);
3413 return isLegalSMRDEncodedUnsignedOffset(ST, EncodedOffset)
3414 ? std::optional<int64_t>(EncodedOffset)
3415 : std::nullopt;
3416}
3417
3418std::optional<int64_t> getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST,
3419 int64_t ByteOffset) {
3420 if (!isCI(ST) || !isDwordAligned(ByteOffset))
3421 return std::nullopt;
3422
3423 int64_t EncodedOffset = convertSMRDOffsetUnits(ST, ByteOffset);
3424 return isUInt<32>(EncodedOffset) ? std::optional<int64_t>(EncodedOffset)
3425 : std::nullopt;
3426}
3427
3429 if (ST.getFeatureBits().test(FeatureFlatOffsetBits12))
3430 return 12;
3431 if (ST.getFeatureBits().test(FeatureFlatOffsetBits24))
3432 return 24;
3433 return 13;
3434}
3435
3436namespace {
3437
3438struct SourceOfDivergence {
3439 unsigned Intr;
3440};
3441const SourceOfDivergence *lookupSourceOfDivergence(unsigned Intr);
3442
3443struct AlwaysUniform {
3444 unsigned Intr;
3445};
3446const AlwaysUniform *lookupAlwaysUniform(unsigned Intr);
3447
3448#define GET_SourcesOfDivergence_IMPL
3449#define GET_UniformIntrinsics_IMPL
3450#define GET_Gfx9BufferFormat_IMPL
3451#define GET_Gfx10BufferFormat_IMPL
3452#define GET_Gfx11PlusBufferFormat_IMPL
3453
3454#include "AMDGPUGenSearchableTables.inc"
3455
3456} // end anonymous namespace
3457
3458bool isIntrinsicSourceOfDivergence(unsigned IntrID) {
3459 return lookupSourceOfDivergence(IntrID);
3460}
3461
3462bool isIntrinsicAlwaysUniform(unsigned IntrID) {
3463 return lookupAlwaysUniform(IntrID);
3464}
3465
3467 uint8_t NumComponents,
3468 uint8_t NumFormat,
3469 const MCSubtargetInfo &STI) {
3470 return isGFX11Plus(STI) ? getGfx11PlusBufferFormatInfo(
3471 BitsPerComp, NumComponents, NumFormat)
3472 : isGFX10(STI)
3473 ? getGfx10BufferFormatInfo(BitsPerComp, NumComponents, NumFormat)
3474 : getGfx9BufferFormatInfo(BitsPerComp, NumComponents, NumFormat);
3475}
3476
3478 const MCSubtargetInfo &STI) {
3479 return isGFX11Plus(STI) ? getGfx11PlusBufferFormatInfo(Format)
3480 : isGFX10(STI) ? getGfx10BufferFormatInfo(Format)
3481 : getGfx9BufferFormatInfo(Format);
3482}
3483
3485 const MCRegisterInfo &MRI) {
3486 const unsigned VGPRClasses[] = {
3487 AMDGPU::VGPR_16RegClassID, AMDGPU::VGPR_32RegClassID,
3488 AMDGPU::VReg_64RegClassID, AMDGPU::VReg_96RegClassID,
3489 AMDGPU::VReg_128RegClassID, AMDGPU::VReg_160RegClassID,
3490 AMDGPU::VReg_192RegClassID, AMDGPU::VReg_224RegClassID,
3491 AMDGPU::VReg_256RegClassID, AMDGPU::VReg_288RegClassID,
3492 AMDGPU::VReg_320RegClassID, AMDGPU::VReg_352RegClassID,
3493 AMDGPU::VReg_384RegClassID, AMDGPU::VReg_512RegClassID,
3494 AMDGPU::VReg_1024RegClassID};
3495
3496 for (unsigned RCID : VGPRClasses) {
3497 const MCRegisterClass &RC = MRI.getRegClass(RCID);
3498 if (RC.contains(Reg))
3499 return &RC;
3500 }
3501
3502 return nullptr;
3503}
3504
3506 unsigned Enc = MRI.getEncodingValue(Reg);
3507 unsigned Idx = Enc & AMDGPU::HWEncoding::REG_IDX_MASK;
3508 return Idx >> 8;
3509}
3510
3512 const MCRegisterInfo &MRI) {
3513 unsigned Enc = MRI.getEncodingValue(Reg);
3514 unsigned Idx = Enc & AMDGPU::HWEncoding::REG_IDX_MASK;
3515 if (Idx >= 0x100)
3516 return MCRegister();
3517
3518 const MCRegisterClass *RC = getVGPRPhysRegClass(Reg, MRI);
3519 if (!RC)
3520 return MCRegister();
3521
3522 Idx |= MSBs << 8;
3523 if (RC->getID() == AMDGPU::VGPR_16RegClassID) {
3524 // This class has 2048 registers with interleaved lo16 and hi16.
3525 Idx *= 2;
3527 ++Idx;
3528 }
3529
3530 return RC->getRegister(Idx);
3531}
3532
3533static std::optional<unsigned>
3534convertSetRegImmToVgprMSBs(unsigned Imm, unsigned Simm16,
3535 bool HasSetregVGPRMSBFixup) {
3536 constexpr unsigned VGPRMSBShift =
3538
3539 auto [HwRegId, Offset, Size] = Hwreg::HwregEncoding::decode(Simm16);
3540 if (HwRegId != Hwreg::ID_MODE ||
3541 (!HasSetregVGPRMSBFixup && (Offset + Size) < VGPRMSBShift))
3542 return {};
3543 // If there is SetregVGPRMSBFixup then Offset is ignored.
3544 if (!HasSetregVGPRMSBFixup)
3545 Imm <<= Offset;
3546 Imm = (Imm & Hwreg::VGPR_MSB_MASK) >> VGPRMSBShift;
3547 if (!HasSetregVGPRMSBFixup)
3549 return llvm::rotr<uint8_t>(static_cast<uint8_t>(Imm), /*R=*/2);
3550}
3551
3552std::optional<unsigned> convertSetRegImmToVgprMSBs(const MachineInstr &MI,
3553 bool HasSetregVGPRMSBFixup) {
3554 assert(MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32);
3555 return convertSetRegImmToVgprMSBs(MI.getOperand(0).getImm(),
3556 MI.getOperand(1).getImm(),
3557 HasSetregVGPRMSBFixup);
3558}
3559
3560std::optional<unsigned> convertSetRegImmToVgprMSBs(const MCInst &MI,
3561 bool HasSetregVGPRMSBFixup) {
3562 assert(MI.getOpcode() == AMDGPU::S_SETREG_IMM32_B32_gfx12);
3563 return convertSetRegImmToVgprMSBs(MI.getOperand(0).getImm(),
3564 MI.getOperand(1).getImm(),
3565 HasSetregVGPRMSBFixup);
3566}
3567
3568std::pair<const AMDGPU::OpName *, const AMDGPU::OpName *>
3570 static const AMDGPU::OpName VOPOps[4] = {
3571 AMDGPU::OpName::src0, AMDGPU::OpName::src1, AMDGPU::OpName::src2,
3572 AMDGPU::OpName::vdst};
3573 static const AMDGPU::OpName VDSOps[4] = {
3574 AMDGPU::OpName::addr, AMDGPU::OpName::data0, AMDGPU::OpName::data1,
3575 AMDGPU::OpName::vdst};
3576 static const AMDGPU::OpName FLATOps[4] = {
3577 AMDGPU::OpName::vaddr, AMDGPU::OpName::vdata,
3578 AMDGPU::OpName::NUM_OPERAND_NAMES, AMDGPU::OpName::vdst};
3579 static const AMDGPU::OpName BUFOps[4] = {
3580 AMDGPU::OpName::vaddr, AMDGPU::OpName::NUM_OPERAND_NAMES,
3581 AMDGPU::OpName::NUM_OPERAND_NAMES, AMDGPU::OpName::vdata};
3582 static const AMDGPU::OpName VIMGOps[4] = {
3583 AMDGPU::OpName::vaddr0, AMDGPU::OpName::vaddr1, AMDGPU::OpName::vaddr2,
3584 AMDGPU::OpName::vdata};
3585
3586 // For VOPD instructions MSB of a corresponding Y component operand VGPR
3587 // address is supposed to match X operand, otherwise VOPD shall not be
3588 // combined.
3589 static const AMDGPU::OpName VOPDOpsX[4] = {
3590 AMDGPU::OpName::src0X, AMDGPU::OpName::vsrc1X, AMDGPU::OpName::vsrc2X,
3591 AMDGPU::OpName::vdstX};
3592 static const AMDGPU::OpName VOPDOpsY[4] = {
3593 AMDGPU::OpName::src0Y, AMDGPU::OpName::vsrc1Y, AMDGPU::OpName::vsrc2Y,
3594 AMDGPU::OpName::vdstY};
3595
3596 // VOP2 MADMK instructions use src0, imm, src1 scheme.
3597 static const AMDGPU::OpName VOP2MADMKOps[4] = {
3598 AMDGPU::OpName::src0, AMDGPU::OpName::NUM_OPERAND_NAMES,
3599 AMDGPU::OpName::src1, AMDGPU::OpName::vdst};
3600 static const AMDGPU::OpName VOPDFMAMKOpsX[4] = {
3601 AMDGPU::OpName::src0X, AMDGPU::OpName::NUM_OPERAND_NAMES,
3602 AMDGPU::OpName::vsrc1X, AMDGPU::OpName::vdstX};
3603 static const AMDGPU::OpName VOPDFMAMKOpsY[4] = {
3604 AMDGPU::OpName::src0Y, AMDGPU::OpName::NUM_OPERAND_NAMES,
3605 AMDGPU::OpName::vsrc1Y, AMDGPU::OpName::vdstY};
3606
3610 switch (Desc.getOpcode()) {
3611 // LD_SCALE operands ignore MSB.
3612 case AMDGPU::V_WMMA_LD_SCALE_PAIRED_B32:
3613 case AMDGPU::V_WMMA_LD_SCALE_PAIRED_B32_gfx1250:
3614 case AMDGPU::V_WMMA_LD_SCALE16_PAIRED_B64:
3615 case AMDGPU::V_WMMA_LD_SCALE16_PAIRED_B64_gfx1250:
3616 return {};
3617 case AMDGPU::V_FMAMK_F16:
3618 case AMDGPU::V_FMAMK_F16_t16:
3619 case AMDGPU::V_FMAMK_F16_t16_gfx12:
3620 case AMDGPU::V_FMAMK_F16_fake16:
3621 case AMDGPU::V_FMAMK_F16_fake16_gfx12:
3622 case AMDGPU::V_FMAMK_F32:
3623 case AMDGPU::V_FMAMK_F32_gfx12:
3624 case AMDGPU::V_FMAMK_F64:
3625 case AMDGPU::V_FMAMK_F64_gfx1250:
3626 return {VOP2MADMKOps, nullptr};
3627 default:
3628 break;
3629 }
3630 return {VOPOps, nullptr};
3631 }
3632
3634 return {VDSOps, nullptr};
3635
3637 return {FLATOps, nullptr};
3638
3640 return {BUFOps, nullptr};
3641
3643 return {VIMGOps, nullptr};
3644
3645 if (AMDGPU::isVOPD(Desc.getOpcode())) {
3646 auto [OpX, OpY] = getVOPDComponents(Desc.getOpcode());
3647 return {(OpX == AMDGPU::V_FMAMK_F32) ? VOPDFMAMKOpsX : VOPDOpsX,
3648 (OpY == AMDGPU::V_FMAMK_F32) ? VOPDFMAMKOpsY : VOPDOpsY};
3649 }
3650
3652
3654 llvm_unreachable("Sample and export VGPR lowering is not implemented and"
3655 " these instructions are not expected on gfx1250");
3656
3657 return {};
3658}
3659
3660bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode) {
3661 const MCInstrDesc &Desc = MII.get(Opcode);
3663 return Desc.mayLoad() && !Desc.mayStore() && !getSMEMIsBuffer(Opcode);
3665 return false;
3666
3667 // Only SV and SVS modes are supported.
3668 if (SIInstrFlags::isFlatScratch(MII, Opcode))
3669 return hasNamedOperand(Opcode, OpName::vaddr);
3670
3671 // Only GVS mode is supported.
3672 return hasNamedOperand(Opcode, OpName::vaddr) &&
3673 hasNamedOperand(Opcode, OpName::saddr);
3674
3675 return false;
3676}
3677
3678bool hasAny64BitVGPROperands(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
3679 const MCSubtargetInfo &ST) {
3680 for (auto OpName : {OpName::vdst, OpName::src0, OpName::src1, OpName::src2}) {
3681 int Idx = getNamedOperandIdx(OpDesc.getOpcode(), OpName);
3682 if (Idx == -1)
3683 continue;
3684
3685 const MCOperandInfo &OpInfo = OpDesc.operands()[Idx];
3686 int16_t RegClass = MII.getOpRegClassID(
3687 OpInfo, ST.getHwMode(MCSubtargetInfo::HwMode_RegInfo));
3688 if (RegClass == AMDGPU::VReg_64RegClassID ||
3689 RegClass == AMDGPU::VReg_64_Align2RegClassID)
3690 return true;
3691 }
3692
3693 return false;
3694}
3695
3696bool isDPALU_DPP32BitOpc(unsigned Opc) {
3697 switch (Opc) {
3698 case AMDGPU::V_MUL_LO_U32_e64:
3699 case AMDGPU::V_MUL_LO_U32_e64_dpp:
3700 case AMDGPU::V_MUL_LO_U32_e64_dpp_gfx1250:
3701 case AMDGPU::V_MUL_HI_U32_e64:
3702 case AMDGPU::V_MUL_HI_U32_e64_dpp:
3703 case AMDGPU::V_MUL_HI_U32_e64_dpp_gfx1250:
3704 case AMDGPU::V_MUL_HI_I32_e64:
3705 case AMDGPU::V_MUL_HI_I32_e64_dpp:
3706 case AMDGPU::V_MUL_HI_I32_e64_dpp_gfx1250:
3707 case AMDGPU::V_MAD_U32_e64:
3708 case AMDGPU::V_MAD_U32_e64_dpp:
3709 case AMDGPU::V_MAD_U32_e64_dpp_gfx1250:
3710 return true;
3711 default:
3712 return false;
3713 }
3714}
3715
3716bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII,
3717 const MCSubtargetInfo &ST) {
3718 if (!ST.hasFeature(AMDGPU::FeatureDPALU_DPP))
3719 return false;
3720
3721 if (isDPALU_DPP32BitOpc(OpDesc.getOpcode()))
3722 return ST.hasFeature(AMDGPU::FeatureGFX1250Insts);
3723
3724 return hasAny64BitVGPROperands(OpDesc, MII, ST);
3725}
3726
3728 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize32768))
3729 return 64;
3730 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize65536))
3731 return 128;
3732 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize196608))
3733 return 256;
3734 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize163840))
3735 return 320;
3736 if (ST.getFeatureBits().test(FeatureAddressableLocalMemorySize327680))
3737 return 512;
3738 return 64; // In sync with getAddressableLocalMemorySize
3739}
3740
3742 switch (Opc) {
3743 case AMDGPU::V_PK_ADD_F32_gfx1250:
3744 case AMDGPU::V_PK_ADD_F32_gfx1250_gfx12:
3745 case AMDGPU::V_PK_MUL_F32_gfx1250:
3746 case AMDGPU::V_PK_MUL_F32_gfx1250_gfx12:
3747 case AMDGPU::V_PK_FMA_F32_gfx1250:
3748 case AMDGPU::V_PK_FMA_F32_gfx1250_gfx12:
3749 return true;
3750 default:
3751 return false;
3752 }
3753}
3754
3756 switch (Opc) {
3757 case AMDGPU::V_PK_ADD_F64:
3758 case AMDGPU::V_PK_ADD_F64_gfx1250:
3759 case AMDGPU::V_PK_MUL_F64:
3760 case AMDGPU::V_PK_MUL_F64_gfx1250:
3761 case AMDGPU::V_PK_FMA_F64:
3762 case AMDGPU::V_PK_FMA_F64_gfx1250:
3763 case AMDGPU::V_PK_MAX_NUM_F64:
3764 case AMDGPU::V_PK_MAX_NUM_F64_gfx1250:
3765 case AMDGPU::V_PK_MIN_NUM_F64:
3766 case AMDGPU::V_PK_MIN_NUM_F64_gfx1250:
3767 case AMDGPU::V_PK_ADD_NC_U64:
3768 case AMDGPU::V_PK_ADD_NC_U64_gfx1250:
3769 case AMDGPU::V_PK_SUB_NC_U64:
3770 case AMDGPU::V_PK_SUB_NC_U64_gfx1250:
3771 case AMDGPU::V_PK_LSHL_ADD_U64:
3772 case AMDGPU::V_PK_LSHL_ADD_U64_gfx1250:
3773 return true;
3774 default:
3775 return false;
3776 }
3777}
3778
3782
3783const std::array<unsigned, 3> &ClusterDimsAttr::getDims() const {
3784 assert(isFixedDims() && "expect kind to be FixedDims");
3785 return Dims;
3786}
3787
3788std::string ClusterDimsAttr::to_string() const {
3789 SmallString<10> Buffer;
3790 raw_svector_ostream OS(Buffer);
3791
3792 switch (getKind()) {
3793 case Kind::Unknown:
3794 return "";
3795 case Kind::NoCluster: {
3796 OS << EncoNoCluster << ',' << EncoNoCluster << ',' << EncoNoCluster;
3797 return Buffer.c_str();
3798 }
3799 case Kind::VariableDims: {
3800 OS << EncoVariableDims << ',' << EncoVariableDims << ','
3801 << EncoVariableDims;
3802 return Buffer.c_str();
3803 }
3804 case Kind::FixedDims: {
3805 OS << Dims[0] << ',' << Dims[1] << ',' << Dims[2];
3806 return Buffer.c_str();
3807 }
3808 }
3809 llvm_unreachable("Unknown ClusterDimsAttr kind");
3810}
3811
3813 std::optional<SmallVector<unsigned>> Attr =
3814 getIntegerVecAttribute(F, "amdgpu-cluster-dims", /*Size=*/3);
3816
3817 if (!Attr.has_value())
3818 AttrKind = Kind::Unknown;
3819 else if (all_of(*Attr, equal_to(EncoNoCluster)))
3820 AttrKind = Kind::NoCluster;
3821 else if (all_of(*Attr, equal_to(EncoVariableDims)))
3822 AttrKind = Kind::VariableDims;
3823
3824 ClusterDimsAttr A(AttrKind);
3825 if (AttrKind == Kind::FixedDims)
3826 A.Dims = {(*Attr)[0], (*Attr)[1], (*Attr)[2]};
3827
3828 return A;
3829}
3830
3831std::optional<APFloat> evaluateRcp(const APFloat &Val) {
3832 const fltSemantics &Sem = Val.getSemantics();
3833
3834 // v_rcp_f16/bf16 are correctly rounded.
3835 if (&Sem == &APFloat::IEEEhalf() || &Sem == &APFloat::BFloat())
3836 return APFloat::getOne(Sem) / Val;
3837
3838 // v_rcp_f32/f64 always flush a denormal input to zero (preserving sign)
3839 // before reciprocating.
3840 APFloat Arg = Val;
3841 if (Arg.isDenormal())
3842 Arg = APFloat::getZero(Sem, Arg.isNegative());
3843
3844 APFloat Result = APFloat::getOne(Sem) / Arg;
3845
3846 // v_rcp_f32/f64 always flush a denormal result to zero (preserving sign).
3847 if (Result.isDenormal())
3848 Result = APFloat::getZero(Sem, Result.isNegative());
3849
3850 // v_rcp_f32/f64 only approximate the reciprocal, except for these special
3851 // cases where the result is exact.
3852 if (!Result.isZero() && !Result.isInfinity() && !Result.isNaN() &&
3853 !Result.isOne() && !Result.isMinusOne())
3854 return std::nullopt;
3855
3856 return Result;
3857}
3858
3859} // namespace AMDGPU
3860
3862 switch (S) {
3863 case (AMDGPU::TargetIDSetting::Unsupported):
3864 OS << "Unsupported";
3865 break;
3866 case (AMDGPU::TargetIDSetting::Any):
3867 OS << "Any";
3868 break;
3869 case (AMDGPU::TargetIDSetting::Off):
3870 OS << "Off";
3871 break;
3872 case (AMDGPU::TargetIDSetting::On):
3873 OS << "On";
3874 break;
3875 }
3876 return OS;
3877}
3878
3879} // namespace llvm
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static llvm::cl::opt< unsigned > DefaultAMDHSACodeObjectVersion("amdhsa-code-object-version", llvm::cl::Hidden, llvm::cl::init(llvm::AMDGPU::AMDHSA_COV6), llvm::cl::desc("Set default AMDHSA Code Object Version (module flag " "or asm directive still take priority if present)"))
#define MAP_REG2REG
unsigned uint64_t
Provides AMDGPU specific target descriptions.
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file contains the simple types necessary to represent the attributes associated with functions a...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
IRTranslator LLVM IR MI
#define RegName(no)
#define F(x, y, z)
Definition MD5.cpp:54
Register Reg
Register const TargetRegisterInfo * TRI
This file contains the declarations for metadata subclasses.
#define T
uint64_t High
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
#define S_00B848_MEM_ORDERED(x)
Definition SIDefines.h:1482
#define S_00B848_WGP_MODE(x)
Definition SIDefines.h:1479
#define S_00B848_FWD_PROGRESS(x)
Definition SIDefines.h:1485
This file contains some functions that are useful when dealing with strings.
static const int BlockSize
Definition TarWriter.cpp:33
static ClusterDimsAttr get(const Function &F)
const std::array< unsigned, 3 > & getDims() const
void setSramEccSetting(TargetIDSetting NewSramEccSetting)
Sets sramecc setting to NewSramEccSetting.
void setXnackSetting(TargetIDSetting NewXnackSetting)
Sets xnack setting to NewXnackSetting.
unsigned getIndexInParsedOperands(unsigned CompOprIdx) const
unsigned getIndexOfSrcInParsedOperands(unsigned CompSrcIdx) const
std::optional< unsigned > getInvalidCompOperandIndex(std::function< MCRegister(unsigned, unsigned)> GetRegIdx, const MCRegisterInfo &MRI, bool SkipSrc=false, bool AllowSameVGPR=false, bool VOPD3=false) const
std::array< MCRegister, Component::MAX_OPR_NUM > RegIndices
Represents the counter values to wait for in an s_waitcnt instruction.
static const fltSemantics & BFloat()
Definition APFloat.h:303
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
bool isNegative() const
Definition APFloat.h:1583
bool isDenormal() const
Definition APFloat.h:1584
const fltSemantics & getSemantics() const
Definition APFloat.h:1591
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1183
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
constexpr bool test(unsigned I) const
unsigned getAddressSpace() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
Describe properties that are true of each instruction in the target description file.
unsigned getNumOperands() const
Return the number of declared MachineOperands for this MachineInstruction.
ArrayRef< MCOperandInfo > operands() const
bool mayStore() const
Return true if this instruction could possibly modify memory.
bool mayLoad() const
Return true if this instruction could possibly read memory.
unsigned getNumDefs() const
Return the number of MachineOperands that are register definitions.
int getOperandConstraint(unsigned OpNum, MCOI::OperandConstraint Constraint) const
Returns the value of the specified operand constraint if it is present.
unsigned getOpcode() const
Return the opcode number for this descriptor.
Interface to description of machine instruction set.
Definition MCInstrInfo.h:27
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Definition MCInstrInfo.h:79
This holds information about one operand of a machine instruction, indicating the register class for ...
Definition MCInstrDesc.h:88
MCRegisterClass - Base class of TargetRegisterClass.
unsigned getID() const
getID() - Return the register class ID number.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
uint16_t getEncodingValue(MCRegister Reg) const
Returns the encoding for Reg.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr unsigned id() const
Definition MCRegister.h:82
Generic base class for all target subtargets.
bool hasFeature(unsigned Feature) const
const Triple & getTargetTriple() const
const FeatureBitset & getFeatureBits() const
StringRef getCPU() const
Metadata node.
Definition Metadata.h:1069
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1426
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1432
Representation of each machine instruction.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
const char * c_str()
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
Manages the enabling and disabling of subtarget specific features.
const std::vector< std::string > & getFeatures() const
Returns the vector of individual subtarget features.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
OSType getOS() const
Get the parsed operating system type of this triple.
Definition Triple.h:522
ArchType getArch() const
Get the parsed architecture type of this triple.
Definition Triple.h:513
bool isAMDGCN() const
Tests whether the target is AMDGCN.
Definition Triple.h:991
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an SmallVector or SmallString.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
unsigned decodeFieldVaVcc(unsigned Encoded)
unsigned encodeFieldVaVcc(unsigned Encoded, unsigned VaVcc)
unsigned decodeFieldHoldCnt(unsigned Encoded, const IsaVersion &Version)
bool decodeDepCtr(unsigned Code, int &Id, StringRef &Name, unsigned &Val, bool &IsDefault, const MCSubtargetInfo &STI)
unsigned encodeFieldHoldCnt(unsigned Encoded, unsigned HoldCnt, const IsaVersion &Version)
unsigned encodeFieldVaSsrc(unsigned Encoded, unsigned VaSsrc)
unsigned encodeFieldVaVdst(unsigned Encoded, unsigned VaVdst)
unsigned decodeFieldSaSdst(unsigned Encoded)
unsigned getHoldCntBitMask(const IsaVersion &Version)
unsigned decodeFieldVaSdst(unsigned Encoded)
unsigned encodeFieldVmVsrc(unsigned Encoded, unsigned VmVsrc)
unsigned decodeFieldVaSsrc(unsigned Encoded)
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
unsigned encodeFieldSaSdst(unsigned Encoded, unsigned SaSdst)
const CustomOperandVal DepCtrInfo[]
bool isSymbolicDepCtrEncoding(unsigned Code, bool &HasNonDefaultVal, const MCSubtargetInfo &STI)
unsigned decodeFieldVaVdst(unsigned Encoded)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
unsigned decodeFieldVmVsrc(unsigned Encoded)
unsigned encodeFieldVaSdst(unsigned Encoded, unsigned VaSdst)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
static constexpr ExpTgt ExpTgtInfo[]
bool getTgtName(unsigned Id, StringRef &Name, int &Index)
unsigned getTgtId(const StringRef Name)
constexpr uint32_t VersionMinor
HSA metadata minor version.
constexpr uint32_t VersionMajor
HSA metadata major version.
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize)
static unsigned getMaxHWAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getAddressableNumArchVGPRs(const MCSubtargetInfo &STI)
bool isSGPROccupancyLimited(const MCSubtargetInfo &STI)
unsigned getArchVGPRAllocGranule()
For subtargets with a unified VGPR file and mixed ArchVGPR/AGPR usage, returns the allocation granule...
static unsigned getPhysicalLocalMemorySize(const MCSubtargetInfo &STI)
static unsigned getSGPRTrapHandlerReserve(const MCSubtargetInfo &STI)
unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getEncodedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, std::optional< bool > EnableWavefrontSize32)
unsigned getMaxWorkGroupsPerCU(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getMinNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU)
unsigned getMaxNumSGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, bool Addressable)
unsigned getWavefrontSize(const MCSubtargetInfo &STI)
unsigned getWavesPerEUForWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getInstCacheLineSize(const MCSubtargetInfo &STI)
static constexpr unsigned MaxDynamicVGPRBlocks
Maximum number of VGPR blocks that can be allocated in dynamic VGPR mode.
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
static unsigned getSGPRBudgetPerWave(unsigned TotalNumSGPRs, unsigned WavesPerEU, unsigned TrapReserve, unsigned Granule)
unsigned getTotalNumVGPRs(const MCSubtargetInfo &STI)
unsigned getMinNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
unsigned getAddressableNumVGPRs(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize)
unsigned getWavesPerWorkGroup(const MCSubtargetInfo &STI, unsigned FlatWorkGroupSize)
unsigned getAllocatedNumVGPRBlocks(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves, unsigned TotalNumSGPRs, unsigned Granule, unsigned TrapReserve)
unsigned getNumSGPRBlocks(const MCSubtargetInfo &STI, unsigned NumSGPRs)
unsigned getNumExtraSGPRs(const MCSubtargetInfo &STI, bool VCCUsed, bool FlatScrUsed, bool XNACKUsed)
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getMaxNumVGPRs(const MCSubtargetInfo &STI, unsigned WavesPerEU, unsigned DynamicVGPRBlockSize)
static unsigned getGranulatedNumRegisterBlocks(unsigned NumRegs, unsigned Granule)
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
StringLiteral const UfmtSymbolicGFX11[]
bool isValidUnifiedFormat(unsigned Id, const MCSubtargetInfo &STI)
unsigned getDefaultFormatEncoding(const MCSubtargetInfo &STI)
StringRef getUnifiedFormatName(unsigned Id, const MCSubtargetInfo &STI)
unsigned const DfmtNfmt2UFmtGFX10[]
StringLiteral const DfmtSymbolic[]
static StringLiteral const * getNfmtLookupTable(const MCSubtargetInfo &STI)
bool isValidNfmt(unsigned Id, const MCSubtargetInfo &STI)
StringLiteral const NfmtSymbolicGFX10[]
bool isValidDfmtNfmt(unsigned Id, const MCSubtargetInfo &STI)
int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt, const MCSubtargetInfo &STI)
StringRef getDfmtName(unsigned Id)
int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt)
int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI)
bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI)
StringRef getNfmtName(unsigned Id, const MCSubtargetInfo &STI)
unsigned const DfmtNfmt2UFmtGFX11[]
StringLiteral const NfmtSymbolicVI[]
StringLiteral const NfmtSymbolicSICI[]
int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI)
int64_t getDfmt(const StringRef Name)
StringLiteral const UfmtSymbolicGFX10[]
void decodeDfmtNfmt(unsigned Format, unsigned &Dfmt, unsigned &Nfmt)
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
void decodeMsg(unsigned Val, uint16_t &MsgId, uint16_t &OpId, uint16_t &StreamId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgDoesNotUseM0(int64_t MsgId, const MCSubtargetInfo &STI)
Returns true if the message does not use the m0 operand.
StringRef getMsgOpName(int64_t MsgId, uint64_t Encoding, const MCSubtargetInfo &STI)
Map from an encoding to the symbolic name for a sendmsg operation.
static uint64_t getMsgIdMask(const MCSubtargetInfo &STI)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
constexpr unsigned VOPD_VGPR_BANK_MASKS[]
constexpr unsigned COMPONENTS_NUM
constexpr unsigned VOPD3_VGPR_BANK_MASKS[]
bool isGCN3Encoding(const MCSubtargetInfo &STI)
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isGFX10_BEncoding(const MCSubtargetInfo &STI)
bool isInlineValue(MCRegister Reg)
bool isGFX10_GFX11(const MCSubtargetInfo &STI)
bool isInlinableLiteralV216(uint32_t Literal, uint8_t OpType)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
uint64_t convertSMRDOffsetUnits(const MCSubtargetInfo &ST, uint64_t ByteOffset)
Convert ByteOffset to dwords if the subtarget uses dword SMRD immediate offsets.
static unsigned encodeStorecnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Storecnt)
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
static bool hasSMEMByteOffset(const MCSubtargetInfo &ST)
LLVM_ABI unsigned getMaxWavesPerEU(GPUKind AK)
bool isVOPCAsmOnly(unsigned Opc)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords)
bool getMTBUFHasSrsrc(unsigned Opc)
std::optional< int64_t > getSMRDEncodedLiteralOffset32(const MCSubtargetInfo &ST, int64_t ByteOffset)
bool getWMMAIsXDL(unsigned Opc)
static std::optional< unsigned > convertSetRegImmToVgprMSBs(unsigned Imm, unsigned Simm16, bool HasSetregVGPRMSBFixup)
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
static bool isSymbolicCustomOperandEncoding(const CustomOperandVal *Opr, int Size, unsigned Code, bool &HasNonDefaultVal, const MCSubtargetInfo &STI)
bool isGFX10Before1030(const MCSubtargetInfo &STI)
bool isSISrcInlinableOperand(const MCInstrDesc &Desc, unsigned OpNo)
Does this operand support only inlinable literals?
unsigned mapWMMA2AddrTo3AddrOpcode(unsigned Opc)
const int OPR_ID_UNSUPPORTED
void initDefaultAMDKernelCodeT(AMDGPUMCKernelCodeT &KernelCode, const MCSubtargetInfo &STI)
bool shouldEmitConstantsToTextSection(const Triple &TT)
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isDPMACCInstruction(unsigned Opc)
int getMTBUFElements(unsigned Opc)
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
static int encodeCustomOperandVal(const CustomOperandVal &Op, int64_t InputVal)
unsigned getTemporalHintType(const MCInstrDesc TID)
int32_t getTotalNumVGPRs(bool has90AInsts, int32_t ArgNumAGPR, int32_t ArgNumVGPR)
bool isGFX10(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2BF16(uint32_t Literal)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV216(bool IsFloat, uint32_t Literal)
FPType getFPDstSelType(unsigned Opc)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
const MCRegisterClass * getVGPRPhysRegClass(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned encodeLoadcntDscnt(const IsaVersion &Version, const Waitcnt &Decoded)
bool getHasMatrixScale(unsigned Opc)
bool hasPackedD16(const MCSubtargetInfo &STI)
unsigned getStorecntBitMask(const IsaVersion &Version)
bool isFullSIMDMode(const MCSubtargetInfo &STI)
unsigned getLdsDwGranularity(const MCSubtargetInfo &ST)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
bool isGFX10_3_GFX11(const MCSubtargetInfo &STI)
bool isGFX13(const MCSubtargetInfo &STI)
unsigned getAsynccntBitMask(const IsaVersion &Version)
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val)
Checks if Val is inside MD, a !range-like metadata.
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
TargetID createAMDGPUTargetID(const MCSubtargetInfo &STI, StringRef FeatureString)
Construct TargetID from MCSubtargetInfo.
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
unsigned getVOPDOpcode(unsigned Opc, bool VOPD3)
bool isGroupSegment(const GlobalValue *GV)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
bool getMTBUFHasSoffset(unsigned Opc)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool hasXNACK(const MCSubtargetInfo &STI)
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
static unsigned getCombinedCountBitMask(const IsaVersion &Version, bool IsStore)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
bool isVOPC64DPP(unsigned Opc)
int getMUBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool getMAIIsGFX940XDL(unsigned Opc)
bool isSI(const MCSubtargetInfo &STI)
unsigned getDefaultAMDHSACodeObjectVersion()
LLVM_ABI unsigned getTotalNumSGPRs(GPUKind AK)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
bool isReadOnlySegment(const GlobalValue *GV)
Waitcnt decodeWaitcnt(const IsaVersion &Version, unsigned Encoded)
bool isArgPassedInSGPR(const Argument *A)
bool isIntrinsicAlwaysUniform(unsigned IntrID)
int getMUBUFBaseOpcode(unsigned Opc)
unsigned encodeWaitcnt(const IsaVersion &Version, const Waitcnt &Decoded)
unsigned getAMDHSACodeObjectVersion(const Module &M)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
LLVM_READONLY bool hasNamedOperand(uint64_t Opcode, OpName NamedIdx)
bool getVOP3IsSingle(unsigned Opc)
bool isPackedSingleSGPR64BitInst(unsigned Opc)
The opcode is a packed 64-bit instruction which only reads low 64 bits of a scalar operand and propag...
bool isGFX9(const MCSubtargetInfo &STI)
bool isDPALU_DPP32BitOpc(unsigned Opc)
bool getVOP1IsSingle(unsigned Opc)
static bool isDwordAligned(uint64_t ByteOffset)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
bool getHasColorExport(const Function &F)
GPUKind
GPU kinds supported by the AMDGPU target.
int getMTBUFBaseOpcode(unsigned Opc)
bool isGFX90A(const MCSubtargetInfo &STI)
unsigned getSamplecntBitMask(const IsaVersion &Version)
unsigned getDefaultQueueImplicitArgPosition(unsigned CodeObjectVersion)
std::tuple< char, unsigned, unsigned > parseAsmPhysRegName(StringRef RegName)
Returns a valid charcode or 0 in the first entry if this is a valid physical register name.
bool getHasDepthExport(const Function &F)
bool isGFX8_GFX9_GFX10(const MCSubtargetInfo &STI)
bool getMUBUFHasVAddr(unsigned Opc)
bool isTrue16Inst(unsigned Opc)
LLVM_ABI unsigned getSGPRAllocGranule(GPUKind AK)
unsigned getVGPREncodingMSBs(MCRegister Reg, const MCRegisterInfo &MRI)
std::pair< unsigned, unsigned > getVOPDComponents(unsigned VOPDOpcode)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned getInitialPSInputAddr(const Function &F)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool isAsyncStore(unsigned Opc)
unsigned getDynamicVGPRBlockSize(const Function &F)
unsigned getKmcntBitMask(const IsaVersion &Version)
MCRegister getVGPRWithMSBs(MCRegister Reg, unsigned MSBs, const MCRegisterInfo &MRI)
If Reg is a low VGPR return a corresponding high VGPR with MSBs set.
unsigned getVmcntBitMask(const IsaVersion &Version)
bool isNotGFX10Plus(const MCSubtargetInfo &STI)
bool hasMAIInsts(const MCSubtargetInfo &STI)
unsigned getBitOp2(unsigned Opc)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
unsigned getXcntBitMask(const IsaVersion &Version)
bool isGenericAtomic(unsigned Opc)
const MFMA_F8F6F4_Info * getWMMA_F8F6F4_WithFormatArgs(unsigned FmtA, unsigned FmtB, unsigned F8F8Opcode)
bool isGFX8Plus(const MCSubtargetInfo &STI)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
unsigned getLgkmcntBitMask(const IsaVersion &Version)
bool getMUBUFTfe(unsigned Opc)
unsigned getBvhcntBitMask(const IsaVersion &Version)
bool hasSMRDSignedImmOffset(const MCSubtargetInfo &ST)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasGFX10_3Insts(const MCSubtargetInfo &STI)
unsigned decodeDscnt(const IsaVersion &Version, unsigned Waitcnt)
std::pair< const AMDGPU::OpName *, const AMDGPU::OpName * > getVGPRLoweringOperandTables(const MCInstrDesc &Desc)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
int getMTBUFOpcode(unsigned BaseOpc, unsigned Elements)
bool isGFX13Plus(const MCSubtargetInfo &STI)
unsigned getExpcntBitMask(const IsaVersion &Version)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
int32_t getMCOpcode(uint32_t Opcode, unsigned Gen)
bool getMUBUFHasSoffset(unsigned Opc)
bool isNotGFX11Plus(const MCSubtargetInfo &STI)
bool isGFX11Plus(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV2F16(uint32_t Literal)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
std::tuple< char, unsigned, unsigned > parseAsmConstraintPhysReg(StringRef Constraint)
Returns a valid charcode or 0 in the first entry if this is a valid physical register constraint.
unsigned getHostcallImplicitArgPosition(unsigned CodeObjectVersion)
static unsigned getDefaultCustomOperandEncoding(const CustomOperandVal *Opr, int Size, const MCSubtargetInfo &STI)
static unsigned encodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Loadcnt)
bool isGFX10Plus(const MCSubtargetInfo &STI)
static bool decodeCustomOperand(const CustomOperandVal *Opr, int Size, unsigned Code, int &Idx, StringRef &Name, unsigned &Val, bool &IsDefault, const MCSubtargetInfo &STI)
static bool isValidRegPrefix(char C)
std::optional< int64_t > getSMRDEncodedOffset(const MCSubtargetInfo &ST, int64_t ByteOffset, bool IsBuffer, bool HasSOffset)
AMDGPU::TargetID TargetID
bool isGlobalSegment(const GlobalValue *GV)
SmallVector< unsigned > getMaxNumWorkGroups(const Function &F)
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:446
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_C_LAST
Definition SIDefines.h:487
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:439
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:455
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:452
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:457
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:441
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:438
@ OPERAND_REG_INLINE_AC_FIRST
Definition SIDefines.h:489
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:440
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:443
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:458
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:469
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:470
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:444
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FIRST
Definition SIDefines.h:486
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:454
@ OPERAND_REG_INLINE_AC_LAST
Definition SIDefines.h:490
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:450
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:456
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:471
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:453
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:461
std::optional< unsigned > getPKFMACF16InlineEncoding(uint32_t Literal, bool IsGFX11Plus)
bool isNotGFX9Plus(const MCSubtargetInfo &STI)
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isCvt_F32_Fp8_Bf8_e64(unsigned Opc)
std::optional< unsigned > getInlineEncodingV2I16(uint32_t Literal)
unsigned encodeStorecntDscnt(const IsaVersion &Version, const Waitcnt &Decoded)
bool isGFX1250(const MCSubtargetInfo &STI)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool isSingleSGPRReadInst(unsigned Opc)
Packed instructions that read a single SGPR for SGPR operands, except for 64-bit elements which read ...
bool isTensorStore(unsigned Opc)
bool getMUBUFIsBufferInv(unsigned Opc)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
std::optional< unsigned > getInlineEncodingV2BF16(uint32_t Literal)
static int encodeCustomOperand(const CustomOperandVal *Opr, int Size, const StringRef Name, int64_t InputVal, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
bool isMAC(unsigned Opc)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
bool getVOP2IsSingle(unsigned Opc)
bool getMAIIsDGEMM(unsigned Opc)
Returns true if MAI operation is a double precision GEMM.
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const int OPR_ID_UNKNOWN
unsigned getCompletionActionImplicitArgPosition(unsigned CodeObjectVersion)
SmallVector< unsigned > getIntegerVecAttribute(const Function &F, StringRef Name, unsigned Size, unsigned DefaultVal)
unsigned decodeStorecnt(const IsaVersion &Version, unsigned Waitcnt)
bool isGFX1250Plus(const MCSubtargetInfo &STI)
int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels)
bool isNotGFX12Plus(const MCSubtargetInfo &STI)
bool getMTBUFHasVAddr(unsigned Opc)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
uint8_t getELFABIVersion(const Triple &T, unsigned CodeObjectVersion)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
unsigned getLoadcntBitMask(const IsaVersion &Version)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
int getVOPDFull(unsigned OpX, unsigned OpY, unsigned EncodingFamily, bool VOPD3)
static unsigned encodeDscnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Dscnt)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
const MFMA_F8F6F4_Info * getMFMA_F8F6F4_WithFormatArgs(unsigned CBSZ, unsigned BLGP, unsigned F8F8Opcode)
unsigned decodeLoadcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getMultigridSyncArgImplicitArgPosition(unsigned CodeObjectVersion)
bool isGFX9_GFX10_GFX11(const MCSubtargetInfo &STI)
bool isGFX9_GFX10(const MCSubtargetInfo &STI)
int getMUBUFElements(unsigned Opc)
const GcnBufferFormatInfo * getGcnBufferFormatInfo(uint8_t BitsPerComp, uint8_t NumComponents, uint8_t NumFormat, const MCSubtargetInfo &STI)
unsigned mapWMMA3AddrTo2AddrOpcode(unsigned Opc)
bool isPermlane16(unsigned Opc)
bool getMUBUFHasSrsrc(unsigned Opc)
unsigned getDscntBitMask(const IsaVersion &Version)
bool hasAny64BitVGPROperands(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ ELFABIVERSION_AMDGPU_HSA_V4
Definition ELF.h:384
@ ELFABIVERSION_AMDGPU_HSA_V5
Definition ELF.h:385
@ ELFABIVERSION_AMDGPU_HSA_V6
Definition ELF.h:386
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:236
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:239
constexpr bool isVOP1(const T &...O)
Definition SIDefines.h:230
constexpr bool isVOP2(const T &...O)
Definition SIDefines.h:233
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:286
constexpr bool isBuffer(const T &...O)
Definition SIDefines.h:267
constexpr bool isVIMAGE(const T &...O)
Definition SIDefines.h:277
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:271
constexpr bool isVOP3Like(const T &...O)
Definition SIDefines.h:245
constexpr bool isFlatScratch(const T &...O)
Definition SIDefines.h:361
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:274
constexpr bool isVOPD3(const T &...O)
Definition SIDefines.h:385
constexpr bool isEXP(const T &...O)
Definition SIDefines.h:283
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:280
constexpr bool isDS(const T &...O)
Definition SIDefines.h:289
constexpr bool isAtomic(const T &...O)
Definition SIDefines.h:396
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:255
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract_or_null(Y &&MD)
Extract a Value from Metadata, allowing null.
Definition Metadata.h:683
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:668
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
constexpr T rotr(T V, int R)
Definition bit.h:399
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
std::string utostr(uint64_t X, bool isNeg=false)
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
Definition STLExtras.h:2173
Op::Description Desc
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
To bit_cast(const From &from) noexcept
Definition bit.h:90
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
constexpr int countr_zero_constexpr(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:190
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
@ AlwaysUniform
The result value is always uniform.
Definition Uniformity.h:23
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
#define N
AMD Kernel Code Object (amd_kernel_code_t).
static std::tuple< typename Fields::ValueType... > decode(uint64_t Encoded)
Instruction set architecture version.