LLVM 24.0.0git
GCNSubtarget.cpp
Go to the documentation of this file.
1//===-- GCNSubtarget.cpp - GCN Subtarget Information ----------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Implements the GCN specific subclass of TargetSubtarget.
11//
12//===----------------------------------------------------------------------===//
13
14#include "GCNSubtarget.h"
15#include "AMDGPUCallLowering.h"
17#include "AMDGPULegalizerInfo.h"
20#include "AMDGPUTargetMachine.h"
29#include "llvm/IR/MDBuilder.h"
31#include <algorithm>
32
33using namespace llvm;
34
35#define DEBUG_TYPE "gcn-subtarget"
36
37#define GET_SUBTARGETINFO_TARGET_DESC
38#define GET_SUBTARGETINFO_CTOR
39#define AMDGPUSubtarget GCNSubtarget
40#include "AMDGPUGenSubtargetInfo.inc"
41#undef AMDGPUSubtarget
42
44 "amdgpu-vgpr-index-mode",
45 cl::desc("Use GPR indexing mode instead of movrel for vector indexing"),
46 cl::init(false));
47
48static cl::opt<bool> UseAA("amdgpu-use-aa-in-codegen",
49 cl::desc("Enable the use of AA during codegen."),
50 cl::init(true));
51
53 NSAThreshold("amdgpu-nsa-threshold",
54 cl::desc("Number of addresses from which to enable MIMG NSA."),
56
58
60 // Legacy triples without a subarch default to the first target that supports
61 // flat addressing for HSA, otherwise the first amdgcn target.
62 if (TT.getSubArch() == Triple::NoSubArch)
63 return TT.getOS() == Triple::AMDHSA ? AMDGPUSubtarget::SEA_ISLANDS
65
66 switch (AMDGPU::getMajorSubArch(TT.getSubArch())) {
91 default:
92 reportFatalUsageError("invalid subarch for amdgpu");
93 }
94}
95
97 StringRef GPU,
98 StringRef FS) {
99 // Determine default and user-specified characteristics
100 //
101 // We want to be able to turn these off, but making this a subtarget feature
102 // for SI has the unhelpful behavior that it unsets everything else if you
103 // disable it.
104 //
105 // Similarly we want enable-prt-strict-null to be on by default and not to
106 // unset everything else if it is disabled
107
108 SmallString<256> FullFS("+load-store-opt,+enable-ds128,");
109
110 // Turn on features that HSA ABI requires. Also turn on FlatForGlobal by
111 // default
112 if (isAmdHsaOS())
113 FullFS += "+flat-for-global,+unaligned-access-mode,+trap-handler,";
114
115 FullFS += "+enable-prt-strict-null,"; // This is overridden by a disable in FS
116
117 // Disable mutually exclusive bits.
118 if (FS.contains_insensitive("+wavefrontsize")) {
119 if (!FS.contains_insensitive("wavefrontsize16"))
120 FullFS += "-wavefrontsize16,";
121 if (!FS.contains_insensitive("wavefrontsize32"))
122 FullFS += "-wavefrontsize32,";
123 if (!FS.contains_insensitive("wavefrontsize64"))
124 FullFS += "-wavefrontsize64,";
125 }
126
127 FullFS += FS;
128
129 ParseSubtargetFeatures(GPU, /*TuneCPU*/ GPU, FullFS);
130
131 // Implement the "generic" processors, which acts as the default when no
132 // generation features are enabled (e.g for -mcpu=''). HSA OS defaults to
133 // the first amdgcn target that supports flat addressing. Other OSes defaults
134 // to the first amdgcn target.
137 // Assume wave64 for the unknown target, if not explicitly set.
138 if (getWavefrontSizeLog2() == 0)
140 } else if (!hasFeature(AMDGPU::FeatureWavefrontSize32) &&
141 !hasFeature(AMDGPU::FeatureWavefrontSize64)) {
142 // If there is no default wave size it must be a generation before gfx10,
143 // these have FeatureWavefrontSize64 in their definition already. For gfx10+
144 // set wave32 as a default.
145 ToggleFeature(AMDGPU::FeatureWavefrontSize32);
147 }
148
149 // We don't support FP64 for EG/NI atm.
151
152 // Targets must either support 64-bit offsets for MUBUF instructions, and/or
153 // support flat operations, otherwise they cannot access a 64-bit global
154 // address space
155 assert(hasAddr64() || hasFlat());
156 // Unless +-flat-for-global is specified, turn on FlatForGlobal for targets
157 // that do not support ADDR64 variants of MUBUF instructions. Such targets
158 // cannot use a 64 bit offset with a MUBUF instruction to access the global
159 // address space
160 if (!hasAddr64() && !FS.contains("flat-for-global") && !UseFlatForGlobal) {
161 ToggleFeature(AMDGPU::FeatureUseFlatForGlobal);
162 UseFlatForGlobal = true;
163 }
164 // Unless +-flat-for-global is specified, use MUBUF instructions for global
165 // address space access if flat operations are not available.
166 if (!hasFlat() && !FS.contains("flat-for-global") && UseFlatForGlobal) {
167 ToggleFeature(AMDGPU::FeatureUseFlatForGlobal);
168 UseFlatForGlobal = false;
169 }
170
171 // Set defaults if needed.
172 if (MaxPrivateElementSize == 0)
174
175 if (LDSBankCount == 0)
176 LDSBankCount = 32;
177
178 if (MaxWavesPerEU == 0)
179 MaxWavesPerEU = 10;
180
181 if (FlatOffsetBitWidth == 0)
183
187 // LDS Allocation Granularity calculated in bytes from dwords
189 AMDGPU::getLdsDwGranularity(*this) * sizeof(uint32_t);
190
193
194 // InstCacheLineSize is set from TableGen subtarget features
195 // (FeatureInstCacheLineSize64 / FeatureInstCacheLineSize128).
196 // Fall back to 64 if no feature was specified (e.g. generic targets).
197 if (InstCacheLineSize == 0)
199
201 "InstCacheLineSize must be a power of 2");
202
203 return *this;
204}
205
207 LLVMContext &Ctx = F.getContext();
208 if (hasFeature(AMDGPU::FeatureWavefrontSize32) &&
209 hasFeature(AMDGPU::FeatureWavefrontSize64)) {
210 Ctx.diagnose(DiagnosticInfoUnsupported(
211 F, "must specify exactly one of wavefrontsize32 and wavefrontsize64"));
212 }
213}
214
215// TODO: Validate subarch for subtarget
216
218 const GCNTargetMachine &TM, bool BufferOOBRelaxed,
222 : // clang-format off
223 AMDGPUGenSubtargetInfo(TT, GPU, /*TuneCPU*/ GPU, FS),
224 AMDGPUSubtarget(TT),
225 TargetID(AMDGPU::createAMDGPUTargetID(*this, "")),
226 InstrItins(getInstrItineraryForCPU(GPU)),
229 InstrInfo(initializeSubtargetDependencies(TT, GPU, FS)),
230 TLInfo(TM, *this),
231 // Frame index expansion sometimes assumes the low bit of SP is 0
232 FrameLowering(TargetFrameLowering::StackGrowsUp, getStackAlignment(), 0,
233 /*TransAl=*/Align(4)) {
234
235 // clang-format on
236
237 // Apply the module flag's xnack setting if the target supports on/off modes.
238 // Targets without on/off mode support have xnack always on and ignore module
239 // flags.
240 if (hasXNACKOnOffModes())
241 TargetID.setXnackSetting(XnackSetting);
242
243 // Apply the module flag's sramecc setting if the target supports it.
244 if (supportsSRAMECC())
245 TargetID.setSramEccSetting(SramEccSetting);
246
247 LLVM_DEBUG(dbgs() << "xnack setting for subtarget: "
248 << TargetID.getXnackSetting() << '\n');
249 LLVM_DEBUG(dbgs() << "sramecc setting for subtarget: "
250 << TargetID.getSramEccSetting() << '\n');
251
254
255 TSInfo = std::make_unique<AMDGPUSelectionDAGInfo>();
256
257 CallLoweringInfo = std::make_unique<AMDGPUCallLowering>(*getTargetLowering());
258 InlineAsmLoweringInfo =
259 std::make_unique<InlineAsmLowering>(getTargetLowering());
260 Legalizer = std::make_unique<AMDGPULegalizerInfo>(*this, TM);
261 RegBankInfo = std::make_unique<AMDGPURegisterBankInfo>(*this);
262 InstSelector =
263 std::make_unique<AMDGPUInstructionSelector>(*this, *RegBankInfo);
264}
265
267 return TSInfo.get();
268}
269
270unsigned GCNSubtarget::getConstantBusLimit(unsigned Opcode) const {
271 if (getGeneration() < GFX10)
272 return 1;
273
274 switch (Opcode) {
275 case AMDGPU::V_LSHLREV_B64_e64:
276 case AMDGPU::V_LSHLREV_B64_gfx10:
277 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
278 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
279 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
280 case AMDGPU::V_LSHL_B64_e64:
281 case AMDGPU::V_LSHRREV_B64_e64:
282 case AMDGPU::V_LSHRREV_B64_gfx10:
283 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
284 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
285 case AMDGPU::V_LSHR_B64_e64:
286 case AMDGPU::V_ASHRREV_I64_e64:
287 case AMDGPU::V_ASHRREV_I64_gfx10:
288 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
289 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
290 case AMDGPU::V_ASHR_I64_e64:
291 return 1;
292 }
293
294 return 2;
295}
296
297/// This list was mostly derived from experimentation.
298bool GCNSubtarget::zeroesHigh16BitsOfDest(unsigned Opcode) const {
299 switch (Opcode) {
300 case AMDGPU::V_CVT_F16_F32_e32:
301 case AMDGPU::V_CVT_F16_F32_e64:
302 case AMDGPU::V_CVT_F16_U16_e32:
303 case AMDGPU::V_CVT_F16_U16_e64:
304 case AMDGPU::V_CVT_F16_I16_e32:
305 case AMDGPU::V_CVT_F16_I16_e64:
306 case AMDGPU::V_RCP_F16_e64:
307 case AMDGPU::V_RCP_F16_e32:
308 case AMDGPU::V_RSQ_F16_e64:
309 case AMDGPU::V_RSQ_F16_e32:
310 case AMDGPU::V_SQRT_F16_e64:
311 case AMDGPU::V_SQRT_F16_e32:
312 case AMDGPU::V_LOG_F16_e64:
313 case AMDGPU::V_LOG_F16_e32:
314 case AMDGPU::V_EXP_F16_e64:
315 case AMDGPU::V_EXP_F16_e32:
316 case AMDGPU::V_SIN_F16_e64:
317 case AMDGPU::V_SIN_F16_e32:
318 case AMDGPU::V_COS_F16_e64:
319 case AMDGPU::V_COS_F16_e32:
320 case AMDGPU::V_FLOOR_F16_e64:
321 case AMDGPU::V_FLOOR_F16_e32:
322 case AMDGPU::V_CEIL_F16_e64:
323 case AMDGPU::V_CEIL_F16_e32:
324 case AMDGPU::V_TRUNC_F16_e64:
325 case AMDGPU::V_TRUNC_F16_e32:
326 case AMDGPU::V_RNDNE_F16_e64:
327 case AMDGPU::V_RNDNE_F16_e32:
328 case AMDGPU::V_FRACT_F16_e64:
329 case AMDGPU::V_FRACT_F16_e32:
330 case AMDGPU::V_FREXP_MANT_F16_e64:
331 case AMDGPU::V_FREXP_MANT_F16_e32:
332 case AMDGPU::V_FREXP_EXP_I16_F16_e64:
333 case AMDGPU::V_FREXP_EXP_I16_F16_e32:
334 case AMDGPU::V_LDEXP_F16_e64:
335 case AMDGPU::V_LDEXP_F16_e32:
336 case AMDGPU::V_LSHLREV_B16_e64:
337 case AMDGPU::V_LSHLREV_B16_e32:
338 case AMDGPU::V_LSHRREV_B16_e64:
339 case AMDGPU::V_LSHRREV_B16_e32:
340 case AMDGPU::V_ASHRREV_I16_e64:
341 case AMDGPU::V_ASHRREV_I16_e32:
342 case AMDGPU::V_ADD_U16_e64:
343 case AMDGPU::V_ADD_U16_e32:
344 case AMDGPU::V_SUB_U16_e64:
345 case AMDGPU::V_SUB_U16_e32:
346 case AMDGPU::V_SUBREV_U16_e64:
347 case AMDGPU::V_SUBREV_U16_e32:
348 case AMDGPU::V_MUL_LO_U16_e64:
349 case AMDGPU::V_MUL_LO_U16_e32:
350 case AMDGPU::V_ADD_F16_e64:
351 case AMDGPU::V_ADD_F16_e32:
352 case AMDGPU::V_SUB_F16_e64:
353 case AMDGPU::V_SUB_F16_e32:
354 case AMDGPU::V_SUBREV_F16_e64:
355 case AMDGPU::V_SUBREV_F16_e32:
356 case AMDGPU::V_MUL_F16_e64:
357 case AMDGPU::V_MUL_F16_e32:
358 case AMDGPU::V_MAX_F16_e64:
359 case AMDGPU::V_MAX_F16_e32:
360 case AMDGPU::V_MIN_F16_e64:
361 case AMDGPU::V_MIN_F16_e32:
362 case AMDGPU::V_MAX_U16_e64:
363 case AMDGPU::V_MAX_U16_e32:
364 case AMDGPU::V_MIN_U16_e64:
365 case AMDGPU::V_MIN_U16_e32:
366 case AMDGPU::V_MAX_I16_e64:
367 case AMDGPU::V_MAX_I16_e32:
368 case AMDGPU::V_MIN_I16_e64:
369 case AMDGPU::V_MIN_I16_e32:
370 case AMDGPU::V_MAD_F16_e64:
371 case AMDGPU::V_MAD_U16_e64:
372 case AMDGPU::V_MAD_I16_e64:
373 case AMDGPU::V_FMA_F16_e64:
374 case AMDGPU::V_DIV_FIXUP_F16_e64:
375 // On gfx10, all 16-bit instructions preserve the high bits.
377 case AMDGPU::V_MADAK_F16:
378 case AMDGPU::V_MADMK_F16:
379 case AMDGPU::V_MAC_F16_e64:
380 case AMDGPU::V_MAC_F16_e32:
381 case AMDGPU::V_FMAMK_F16:
382 case AMDGPU::V_FMAAK_F16:
383 case AMDGPU::V_FMAC_F16_e64:
384 case AMDGPU::V_FMAC_F16_e32:
385 // In gfx9, the preferred handling of the unused high 16-bits changed. Most
386 // instructions maintain the legacy behavior of 0ing. Some instructions
387 // changed to preserving the high bits.
389 case AMDGPU::V_MAD_MIXLO_F16:
390 case AMDGPU::V_MAD_MIXHI_F16:
391 default:
392 return false;
393 }
394}
395
397 const SchedRegion &Region) const {
398 // Track register pressure so the scheduler can try to decrease
399 // pressure once register usage is above the threshold defined by
400 // SIRegisterInfo::getRegPressureSetLimit()
401 Policy.ShouldTrackPressure = true;
402
403 const Function &F = Region.RegionBegin->getMF()->getFunction();
404 if (AMDGPU::getSchedStrategy(F) == "coexec") {
405 Policy.OnlyTopDown = true;
406 Policy.OnlyBottomUp = false;
407 return;
408 }
409
410 // Enabling both top down and bottom up scheduling seems to give us less
411 // register spills than just using one of these approaches on its own.
412 Policy.OnlyTopDown = false;
413 Policy.OnlyBottomUp = false;
414
415 // Enabling ShouldTrackLaneMasks crashes the SI Machine Scheduler.
416 if (!enableSIScheduler())
417 Policy.ShouldTrackLaneMasks = true;
418}
419
421 const SchedRegion &Region) const {
422 const Function &F = Region.RegionBegin->getMF()->getFunction();
423 Attribute PostRADirectionAttr = F.getFnAttribute("amdgpu-post-ra-direction");
424 if (!PostRADirectionAttr.isValid())
425 return;
426
427 StringRef PostRADirectionStr = PostRADirectionAttr.getValueAsString();
428 if (PostRADirectionStr == "topdown") {
429 Policy.OnlyTopDown = true;
430 Policy.OnlyBottomUp = false;
431 } else if (PostRADirectionStr == "bottomup") {
432 Policy.OnlyTopDown = false;
433 Policy.OnlyBottomUp = true;
434 } else if (PostRADirectionStr == "bidirectional") {
435 Policy.OnlyTopDown = false;
436 Policy.OnlyBottomUp = false;
437 } else {
439 F, F.getSubprogram(), "invalid value for postRA direction attribute");
440 F.getContext().diagnose(Diag);
441 }
442
443 LLVM_DEBUG({
444 const char *DirStr = "default";
445 if (Policy.OnlyTopDown && !Policy.OnlyBottomUp)
446 DirStr = "topdown";
447 else if (!Policy.OnlyTopDown && Policy.OnlyBottomUp)
448 DirStr = "bottomup";
449 else if (!Policy.OnlyTopDown && !Policy.OnlyBottomUp)
450 DirStr = "bidirectional";
451
452 dbgs() << "Post-MI-sched direction (" << F.getName() << "): " << DirStr
453 << '\n';
454 });
455}
456
461
463 if (isWave32()) {
464 // Fix implicit $vcc operands after MIParser has verified that they match
465 // the instruction definitions.
466 for (auto &MBB : MF) {
467 for (auto &MI : MBB)
468 InstrInfo.fixImplicitOperands(MI);
469 }
470 }
471}
472
474 return InstrInfo.pseudoToMCOpcode(AMDGPU::V_MAD_F16_e64) != -1;
475}
476
478 return hasVGPRIndexMode() && (!hasMovrel() || EnableVGPRIndexMode);
479}
480
481bool GCNSubtarget::useAA() const { return UseAA; }
482
483unsigned GCNSubtarget::getOccupancyWithNumSGPRs(unsigned SGPRs) const {
485}
486
487unsigned
489 unsigned DynamicVGPRBlockSize) const {
491 DynamicVGPRBlockSize);
492}
493
494unsigned
495GCNSubtarget::getBaseReservedNumSGPRs(const bool HasFlatScratch) const {
497 return 2; // VCC. FLAT_SCRATCH and XNACK are no longer in SGPRs.
498
499 if (HasFlatScratch || HasArchitectedFlatScratch) {
501 return 6; // FLAT_SCRATCH, XNACK, VCC (in that order).
503 return 4; // FLAT_SCRATCH, VCC (in that order).
504 }
505
506 if (isXNACKEnabled())
507 return 4; // XNACK, VCC (in that order).
508 return 2; // VCC.
509}
510
515
517 // In principle we do not need to reserve SGPR pair used for flat_scratch if
518 // we know flat instructions do not access the stack anywhere in the
519 // program. For now assume it's needed if we have flat instructions.
520 const bool KernelUsesFlatScratch = hasFlatAddressSpace();
521 return getBaseReservedNumSGPRs(KernelUsesFlatScratch);
522}
523
524std::pair<unsigned, unsigned>
526 unsigned NumSGPRs, unsigned NumVGPRs) const {
527 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
528 auto [MinOcc, MaxOcc] = getOccupancyWithWorkGroupSizes(LDSSize, F);
529 unsigned SGPROcc = getOccupancyWithNumSGPRs(NumSGPRs);
530 unsigned VGPROcc = getOccupancyWithNumVGPRs(NumVGPRs, DynamicVGPRBlockSize);
531
532 // Maximum occupancy may be further limited by high SGPR/VGPR usage.
533 MaxOcc = std::min({MaxOcc, SGPROcc, VGPROcc});
534 return {std::min(MinOcc, MaxOcc), MaxOcc};
535}
536
538 const Function &F, std::pair<unsigned, unsigned> WavesPerEU,
539 unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const {
540 // Compute maximum number of SGPRs function can use using default/requested
541 // minimum number of waves per execution unit.
542 unsigned MaxNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, false);
543 unsigned MaxAddressableNumSGPRs = getMaxNumSGPRs(WavesPerEU.first, true);
544
545 // Check if maximum number of SGPRs was explicitly requested using
546 // "amdgpu-num-sgpr" attribute.
547 unsigned Requested =
548 F.getFnAttributeAsParsedInteger("amdgpu-num-sgpr", MaxNumSGPRs);
549
550 if (Requested != MaxNumSGPRs) {
551 // Make sure requested value does not violate subtarget's specifications.
552 if (Requested && (Requested <= ReservedNumSGPRs))
553 Requested = 0;
554
555 // If more SGPRs are required to support the input user/system SGPRs,
556 // increase to accommodate them.
557 //
558 // FIXME: This really ends up using the requested number of SGPRs + number
559 // of reserved special registers in total. Theoretically you could re-use
560 // the last input registers for these special registers, but this would
561 // require a lot of complexity to deal with the weird aliasing.
562 unsigned InputNumSGPRs = PreloadedSGPRs;
563 if (Requested && Requested < InputNumSGPRs)
564 Requested = InputNumSGPRs;
565
566 // Make sure requested value is compatible with values implied by
567 // default/requested minimum/maximum number of waves per execution unit.
568 if (Requested && Requested > getMaxNumSGPRs(WavesPerEU.first, false))
569 Requested = 0;
570 if (WavesPerEU.second && Requested &&
571 Requested < getMinNumSGPRs(WavesPerEU.second))
572 Requested = 0;
573
574 if (Requested)
575 MaxNumSGPRs = Requested;
576 }
577
578 if (hasSGPRInitBug())
580
581 return std::min(MaxNumSGPRs - ReservedNumSGPRs, MaxAddressableNumSGPRs);
582}
583
585 const Function &F = MF.getFunction();
589}
590
592 using USI = GCNUserSGPRUsageInfo;
593 // Max number of user SGPRs
594 const unsigned MaxUserSGPRs =
595 USI::getNumUserSGPRForField(USI::PrivateSegmentBufferID) +
596 USI::getNumUserSGPRForField(USI::DispatchPtrID) +
597 USI::getNumUserSGPRForField(USI::QueuePtrID) +
598 USI::getNumUserSGPRForField(USI::KernargSegmentPtrID) +
599 USI::getNumUserSGPRForField(USI::DispatchIdID) +
600 USI::getNumUserSGPRForField(USI::FlatScratchInitID) +
601 USI::getNumUserSGPRForField(USI::ImplicitBufferPtrID);
602
603 // Max number of system SGPRs
604 const unsigned MaxSystemSGPRs = 1 + // WorkGroupIDX
605 1 + // WorkGroupIDY
606 1 + // WorkGroupIDZ
607 1 + // WorkGroupInfo
608 1; // private segment wave byte offset
609
610 // Max number of synthetic SGPRs
611 const unsigned SyntheticSGPRs = 1; // LDSKernelId
612
613 return MaxUserSGPRs + MaxSystemSGPRs + SyntheticSGPRs;
614}
615
620
622 const Function &F, std::pair<unsigned, unsigned> NumVGPRBounds) const {
623 const auto [Min, Max] = NumVGPRBounds;
624
625 // Check if maximum number of VGPRs was explicitly requested using
626 // "amdgpu-num-vgpr" attribute.
627
628 unsigned Requested = F.getFnAttributeAsParsedInteger("amdgpu-num-vgpr", Max);
629 if (Requested != Max && hasGFX90AInsts())
630 Requested *= 2;
631
632 // Make sure requested value is inside the range of possible VGPR usage.
633 return std::clamp(Requested, Min, Max);
634}
635
637 unsigned DynamicVGPRBlockSize = AMDGPU::getDynamicVGPRBlockSize(F);
638 std::pair<unsigned, unsigned> Waves = getWavesPerEU(F);
639 return getBaseMaxNumVGPRs(
640 F, {getMinNumVGPRs(Waves.second, DynamicVGPRBlockSize),
641 getMaxNumVGPRs(Waves.first, DynamicVGPRBlockSize)});
642}
643
645 return getMaxNumVGPRs(MF.getFunction());
646}
647
648std::pair<unsigned, unsigned>
650 const unsigned MaxVectorRegs = getMaxNumVGPRs(F);
651
652 unsigned MaxNumVGPRs = MaxVectorRegs;
653 unsigned MaxNumAGPRs = 0;
654 unsigned NumArchVGPRs = getAddressableNumArchVGPRs();
655
656 // On GFX90A, the number of VGPRs and AGPRs need not be equal. Theoretically,
657 // a wave may have up to 512 total vector registers combining together both
658 // VGPRs and AGPRs. Hence, in an entry function without calls and without
659 // AGPRs used within it, it is possible to use the whole vector register
660 // budget for VGPRs.
661 //
662 // TODO: it shall be possible to estimate maximum AGPR/VGPR pressure and split
663 // register file accordingly.
664 if (hasGFX90AInsts()) {
665 unsigned MinNumAGPRs = 0;
666 const unsigned TotalNumAGPRs = AMDGPU::AGPR_32RegClass.getNumRegs();
667
668 const std::pair<unsigned, unsigned> DefaultNumAGPR = {~0u, ~0u};
669
670 // TODO: The lower bound should probably force the number of required
671 // registers up, overriding amdgpu-waves-per-eu.
672 std::tie(MinNumAGPRs, MaxNumAGPRs) =
673 AMDGPU::getIntegerPairAttribute(F, "amdgpu-agpr-alloc", DefaultNumAGPR,
674 /*OnlyFirstRequired=*/true);
675
676 if (MinNumAGPRs == DefaultNumAGPR.first) {
677 // Default to splitting half the registers if AGPRs are required.
678 MinNumAGPRs = MaxNumAGPRs = MaxVectorRegs / 2;
679 } else {
680 // Align to accum_offset's allocation granularity.
681 MinNumAGPRs = alignTo(MinNumAGPRs, 4);
682
683 MinNumAGPRs = std::min(MinNumAGPRs, TotalNumAGPRs);
684 }
685
686 // Clamp values to be inbounds of our limits, and ensure min <= max.
687
688 MaxNumAGPRs = std::min(std::max(MinNumAGPRs, MaxNumAGPRs), MaxVectorRegs);
689 MinNumAGPRs = std::min({MinNumAGPRs, TotalNumAGPRs, MaxNumAGPRs});
690
691 MaxNumVGPRs = std::min(MaxVectorRegs - MinNumAGPRs, NumArchVGPRs);
692 MaxNumAGPRs = std::min(MaxVectorRegs - MaxNumVGPRs, MaxNumAGPRs);
693
694 assert(MaxNumVGPRs + MaxNumAGPRs <= MaxVectorRegs &&
695 MaxNumAGPRs <= TotalNumAGPRs && MaxNumVGPRs <= NumArchVGPRs &&
696 "invalid register counts");
697 } else if (hasMAIInsts()) {
698 // On gfx908 the number of AGPRs always equals the number of VGPRs.
699 MaxNumAGPRs = MaxNumVGPRs = MaxVectorRegs;
700 }
701
702 return std::pair(MaxNumVGPRs, MaxNumAGPRs);
703}
704
705// Check to which source operand UseOpIdx points to and return a pointer to the
706// operand of the corresponding source modifier.
707// Return nullptr if UseOpIdx either doesn't point to src0/1/2 or if there is no
708// operand for the corresponding source modifier.
709static const MachineOperand *
711 const SIInstrInfo &InstrInfo) {
712 AMDGPU::OpName UseName =
713 AMDGPU::getOperandIdxName(UseI.getOpcode(), UseOpIdx);
714 switch (UseName) {
715 case AMDGPU::OpName::src0:
716 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src0_modifiers);
717 case AMDGPU::OpName::src1:
718 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src1_modifiers);
719 case AMDGPU::OpName::src2:
720 return InstrInfo.getNamedOperand(UseI, AMDGPU::OpName::src2_modifiers);
721 default:
722 return nullptr;
723 }
724}
725
726// Get the subreg idx of the subreg that is used by the given instruction
727// operand, considering the given op_sel modifier.
728// Return 0 if the whole register is used or as a conservative fallback.
730 const SIInstrInfo &InstrInfo,
731 const MachineInstr &I,
732 const MachineOperand &Op) {
733 if (!InstrInfo.isVOP3P(I) || InstrInfo.isWMMA(I) || InstrInfo.isSWMMAC(I))
734 return AMDGPU::NoSubRegister;
735
736 const MachineOperand *OpMod =
737 getVOP3PSourceModifierFromOpIdx(I, Op.getOperandNo(), InstrInfo);
738 if (!OpMod)
739 return AMDGPU::NoSubRegister;
740
741 // Note: the FMA_MIX* and MAD_MIX* instructions have different semantics for
742 // the op_sel and op_sel_hi source modifiers:
743 // - op_sel: selects low/high operand bits as input to the operation;
744 // has only meaning for 16-bit source operands
745 // - op_sel_hi: specifies the size of the source operands (16 or 32 bits);
746 // a value of 0 indicates 32 bit, 1 indicates 16 bit
747 // For the other VOP3P instructions, the semantics are:
748 // - op_sel: selects low/high operand bits as input to the operation which
749 // results in the lower-half of the destination
750 // - op_sel_hi: selects the low/high operand bits as input to the operation
751 // which results in the higher-half of the destination
752 int64_t OpSel = OpMod->getImm() & SISrcMods::OP_SEL_0;
753 int64_t OpSelHi = OpMod->getImm() & SISrcMods::OP_SEL_1;
754
755 // Check if all parts of the register are being used (= op_sel and op_sel_hi
756 // differ for VOP3P or op_sel_hi=0 for VOP3PMix). In that case we can return
757 // early.
758 if ((!InstrInfo.isVOP3PMix(I) && (!OpSel || !OpSelHi) &&
759 (OpSel || OpSelHi)) ||
760 (InstrInfo.isVOP3PMix(I) && !OpSelHi))
761 return AMDGPU::NoSubRegister;
762
763 const MachineRegisterInfo &MRI = I.getParent()->getParent()->getRegInfo();
764 const TargetRegisterClass *RC = TRI.getRegClassForOperandReg(MRI, Op);
765
766 if (unsigned SubRegIdx = OpSel ? AMDGPU::sub1 : AMDGPU::sub0;
767 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
768 return SubRegIdx;
769 if (unsigned SubRegIdx = OpSel ? AMDGPU::hi16 : AMDGPU::lo16;
770 TRI.getSubClassWithSubReg(RC, SubRegIdx) == RC)
771 return SubRegIdx;
772
773 return AMDGPU::NoSubRegister;
774}
775
776Register GCNSubtarget::getRealSchedDependency(const MachineInstr &DefI,
777 int DefOpIdx,
778 const MachineInstr &UseI,
779 int UseOpIdx) const {
780 const SIRegisterInfo *TRI = getRegisterInfo();
781 const MachineOperand &DefOp = DefI.getOperand(DefOpIdx);
782 const MachineOperand &UseOp = UseI.getOperand(UseOpIdx);
783 Register DefReg = DefOp.getReg();
784 Register UseReg = UseOp.getReg();
785
786 // If the registers aren't restricted to a sub-register, there is no point in
787 // further analysis. This check makes only sense for virtual registers because
788 // physical registers may form a tuple and thus be part of a superregister
789 // although they are not a subregister themselves (vgpr0 is a "subreg" of
790 // vgpr0_vgpr1 without being a subreg in itself).
791 unsigned DefSubRegIdx = DefOp.getSubReg();
792 if (DefReg.isVirtual() && DefSubRegIdx == AMDGPU::NoSubRegister)
793 return DefReg;
794 unsigned UseSubRegIdx = getEffectiveSubRegIdx(*TRI, InstrInfo, UseI, UseOp);
795 if (UseReg.isVirtual() && UseSubRegIdx == AMDGPU::NoSubRegister)
796 return DefReg;
797
798 if (!TRI->checkSubRegInterference(DefReg, DefSubRegIdx, UseReg, UseSubRegIdx))
799 return Register(); // No real dependency
800
801 // UseReg might be smaller or larger than DefReg, depending on the subreg and
802 // on whether DefReg is a subreg, too. -> Find the smaller one. This does not
803 // apply to virtual registers because we cannot construct a subreg for them.
804 if (DefReg.isVirtual())
805 return DefReg;
806 MCRegister DefMCReg =
807 DefSubRegIdx ? TRI->getSubReg(DefReg, DefSubRegIdx) : DefReg.asMCReg();
808 MCRegister UseMCReg =
809 UseSubRegIdx ? TRI->getSubReg(UseReg, UseSubRegIdx) : UseReg.asMCReg();
810 return TRI->isSubRegisterEq(DefMCReg, UseMCReg) ? UseMCReg : DefMCReg;
811}
812
814 SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep,
815 const TargetSchedModel *SchedModel) const {
816 if (Dep.getKind() != SDep::Kind::Data || !Dep.getReg() || !Def->isInstr() ||
817 !Use->isInstr())
818 return;
819
820 MachineInstr *DefI = Def->getInstr();
821 MachineInstr *UseI = Use->getInstr();
822
823 // Check for false latency on $tensorcnt / $asynccnt dependencies
824 if (Dep.getReg() == AMDGPU::TENSORcnt || Dep.getReg() == AMDGPU::ASYNCcnt) {
825 unsigned UseOp = UseI->getOpcode();
826 // Do not adjust latency for load->s_wait
827 bool IsBarrierCase =
828 InstrInfo.isLDSDMA(*DefI) &&
829 (UseOp == AMDGPU::S_WAIT_TENSORCNT || UseOp == AMDGPU::S_WAIT_ASYNCCNT);
830 if (!IsBarrierCase) {
831 Dep.setLatency(1);
832 return;
833 }
834 }
835
836 if (Register Reg = getRealSchedDependency(*DefI, DefOpIdx, *UseI, UseOpIdx)) {
837 Dep.setReg(Reg);
838 } else {
839 Dep = SDep(Def, SDep::Artificial);
840 return; // This is not a data dependency anymore.
841 }
842
843 if (DefI->isBundle()) {
845 auto Reg = Dep.getReg();
848 unsigned Lat = 0;
849 for (++I; I != E && I->isBundledWithPred(); ++I) {
850 if (I->isMetaInstruction())
851 continue;
852 if (I->modifiesRegister(Reg, TRI))
853 Lat = InstrInfo.getInstrLatency(getInstrItineraryData(), *I);
854 else if (Lat)
855 --Lat;
856 }
857 Dep.setLatency(Lat);
858 } else if (UseI->isBundle()) {
860 auto Reg = Dep.getReg();
863 unsigned Lat = InstrInfo.getInstrLatency(getInstrItineraryData(), *DefI);
864 for (++I; I != E && I->isBundledWithPred() && Lat; ++I) {
865 if (I->isMetaInstruction())
866 continue;
867 if (I->readsRegister(Reg, TRI))
868 break;
869 --Lat;
870 }
871 Dep.setLatency(Lat);
872 } else if (Dep.getLatency() == 0 && Dep.getReg() == AMDGPU::VCC_LO) {
873 // Work around the fact that SIInstrInfo::fixImplicitOperands modifies
874 // implicit operands which come from the MCInstrDesc, which can fool
875 // ScheduleDAGInstrs::addPhysRegDataDeps into treating them as implicit
876 // pseudo operands.
877 Dep.setLatency(InstrInfo.getSchedModel().computeOperandLatency(
878 DefI, DefOpIdx, UseI, UseOpIdx));
879 }
880}
881
884 return 0; // Not MIMG encoding.
885
886 if (NSAThreshold.getNumOccurrences() > 0)
887 return std::max(NSAThreshold.getValue(), 2u);
888
890 "amdgpu-nsa-threshold", -1);
891 if (Value > 0)
892 return std::max(Value, 2);
893
894 return NSAThreshold;
895}
896
898 const GCNSubtarget &ST)
899 : ST(ST) {
900 const CallingConv::ID CC = F.getCallingConv();
901 const bool IsKernel =
903
904 if (IsKernel && (!F.arg_empty() || ST.getImplicitArgNumBytes(F) != 0))
905 KernargSegmentPtr = true;
906
907 bool IsAmdHsaOrMesa = ST.isAmdHsaOrMesa(F);
908 if (IsAmdHsaOrMesa && !ST.hasFlatScratchEnabled())
909 PrivateSegmentBuffer = true;
910 else if (ST.isMesaGfxShader(F))
911 ImplicitBufferPtr = true;
912
913 if (!AMDGPU::isGraphics(CC)) {
914 if (!F.hasFnAttribute("amdgpu-no-dispatch-ptr"))
915 DispatchPtr = true;
916
917 // FIXME: Can this always be disabled with < COv5?
918 if (!F.hasFnAttribute("amdgpu-no-queue-ptr"))
919 QueuePtr = true;
920
921 if (!F.hasFnAttribute("amdgpu-no-dispatch-id"))
922 DispatchID = true;
923 }
924
925 if (ST.hasFlatAddressSpace() && AMDGPU::isEntryFunctionCC(CC) &&
926 (IsAmdHsaOrMesa || ST.hasFlatScratchEnabled()) &&
927 // FlatScratchInit cannot be true for graphics CC if
928 // hasFlatScratchEnabled() is false.
929 (ST.hasFlatScratchEnabled() ||
930 (!AMDGPU::isGraphics(CC) &&
931 !F.hasFnAttribute("amdgpu-no-flat-scratch-init"))) &&
932 !ST.hasArchitectedFlatScratch()) {
933 FlatScratchInit = true;
934 }
935
937 NumUsedUserSGPRs += getNumUserSGPRForField(ImplicitBufferPtrID);
938
941
942 if (hasDispatchPtr())
943 NumUsedUserSGPRs += getNumUserSGPRForField(DispatchPtrID);
944
945 if (hasQueuePtr())
946 NumUsedUserSGPRs += getNumUserSGPRForField(QueuePtrID);
947
949 NumUsedUserSGPRs += getNumUserSGPRForField(KernargSegmentPtrID);
950
951 if (hasDispatchID())
952 NumUsedUserSGPRs += getNumUserSGPRForField(DispatchIdID);
953
954 if (hasFlatScratchInit())
955 NumUsedUserSGPRs += getNumUserSGPRForField(FlatScratchInitID);
956
958 NumUsedUserSGPRs += getNumUserSGPRForField(PrivateSegmentSizeID);
959}
960
962 assert(NumKernargPreloadSGPRs + NumSGPRs <= AMDGPU::getMaxNumUserSGPRs(ST));
963 NumKernargPreloadSGPRs += NumSGPRs;
964 NumUsedUserSGPRs += NumSGPRs;
965}
966
968 return AMDGPU::getMaxNumUserSGPRs(ST) - NumUsedUserSGPRs;
969}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static cl::opt< bool > UseAA("aarch64-use-aa", cl::init(true), cl::desc("Enable the use of AA during codegen."))
This file describes how to lower LLVM calls to machine code calls.
This file declares the targeting of the InstructionSelector class for AMDGPU.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
This file declares the targeting of the RegisterBankInfo class for AMDGPU.
static cl::opt< bool > SramEccSetting("amdgpu-sramecc", cl::desc("Force amdgpu.sramecc for testing"), cl::ReallyHidden)
static cl::opt< bool > XnackSetting("amdgpu-xnack", cl::desc("Force amdgpu.xnack value for testing"), cl::ReallyHidden)
The AMDGPU TargetMachine interface definition for hw codegen targets.
MachineBasicBlock & MBB
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static AMDGPUSubtarget::Generation computeDefaultGeneration(const Triple &TT)
static cl::opt< unsigned > NSAThreshold("amdgpu-nsa-threshold", cl::desc("Number of addresses from which to enable MIMG NSA."), cl::init(2), cl::Hidden)
static cl::opt< bool > EnableVGPRIndexMode("amdgpu-vgpr-index-mode", cl::desc("Use GPR indexing mode instead of movrel for vector indexing"), cl::init(false))
static cl::opt< bool > UseAA("amdgpu-use-aa-in-codegen", cl::desc("Enable the use of AA during codegen."), cl::init(true))
static const MachineOperand * getVOP3PSourceModifierFromOpIdx(const MachineInstr &UseI, int UseOpIdx, const SIInstrInfo &InstrInfo)
static unsigned getEffectiveSubRegIdx(const SIRegisterInfo &TRI, const SIInstrInfo &InstrInfo, const MachineInstr &I, const MachineOperand &Op)
AMD GCN specific subclass of TargetSubtarget.
static Register UseReg(const MachineOperand &MO)
IRTranslator LLVM IR MI
This file describes how to lower LLVM inline asm to machine code INLINEASM.
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
if(PassOpts->AAPipeline)
This file defines the SmallString class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
std::pair< unsigned, unsigned > getWavesPerEU(const Function &F) const
std::pair< unsigned, unsigned > getOccupancyWithWorkGroupSizes(uint32_t LDSBytes, const Function &F) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getWavefrontSizeLog2() const
AMDGPUSubtarget(const Triple &TT)
unsigned AddressableLocalMemorySize
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:105
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
bool isValid() const
Return true if the attribute is any kind of attribute.
Definition Attributes.h:261
Diagnostic information for optimization failures.
Diagnostic information for unsupported feature in backend.
uint64_t getFnAttributeAsParsedInteger(StringRef Kind, uint64_t Default=0) const
For a string attribute Kind, parse attribute as an integer.
Definition Function.cpp:774
bool hasFlat() const
InstrItineraryData InstrItins
bool useVGPRIndexMode() const
void mirFileLoaded(MachineFunction &MF) const override
unsigned MaxPrivateElementSize
unsigned getAddressableNumArchVGPRs() const
unsigned getMinNumSGPRs(unsigned WavesPerEU) const
void ParseSubtargetFeatures(StringRef CPU, StringRef TuneCPU, StringRef FS)
void overridePipelinerPolicy(MachinePipelinerPolicy &Policy) const override
unsigned getConstantBusLimit(unsigned Opcode) const
const InstrItineraryData * getInstrItineraryData() const override
void adjustSchedDependency(SUnit *Def, int DefOpIdx, SUnit *Use, int UseOpIdx, SDep &Dep, const TargetSchedModel *SchedModel) const override
void overridePostRASchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
Align getStackAlignment() const
const bool BufferOOBRelaxed
bool hasMadF16() const
unsigned getMinNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
const SIRegisterInfo * getRegisterInfo() const override
unsigned getBaseMaxNumVGPRs(const Function &F, std::pair< unsigned, unsigned > NumVGPRBounds) const
bool zeroesHigh16BitsOfDest(unsigned Opcode) const
Returns if the result of this instruction with a 16-bit result returned in a 32-bit register implicit...
unsigned getBaseMaxNumSGPRs(const Function &F, std::pair< unsigned, unsigned > WavesPerEU, unsigned PreloadedSGPRs, unsigned ReservedNumSGPRs) const
unsigned getMaxNumPreloadedSGPRs() const
GCNSubtarget & initializeSubtargetDependencies(const Triple &TT, StringRef GPU, StringRef FS)
void overrideSchedPolicy(MachineSchedPolicy &Policy, const SchedRegion &Region) const override
std::pair< unsigned, unsigned > computeOccupancy(const Function &F, unsigned LDSSize=0, unsigned NumSGPRs=0, unsigned NumVGPRs=0) const
Subtarget's minimum/maximum occupancy, in number of waves per EU, that can be achieved when the only ...
unsigned getMaxNumVGPRs(unsigned WavesPerEU, unsigned DynamicVGPRBlockSize) const
AMDGPU::TargetID TargetID
const SITargetLowering * getTargetLowering() const override
unsigned getNSAThreshold(const MachineFunction &MF) const
GCNSubtarget(const Triple &TT, StringRef GPU, StringRef FS, const GCNTargetMachine &TM, bool BufferOOBRelaxed=false, bool TBufferOOBRelaxed=false, AMDGPU::TargetIDSetting XnackSetting=AMDGPU::TargetIDSetting::Any, AMDGPU::TargetIDSetting SramEccSetting=AMDGPU::TargetIDSetting::Any)
unsigned getReservedNumSGPRs(const MachineFunction &MF) const
const bool TBufferOOBRelaxed
bool useAA() const override
bool isWave32() const
unsigned getOccupancyWithNumVGPRs(unsigned VGPRs, unsigned DynamicVGPRBlockSize) const
Return the maximum number of waves per SIMD for kernels using VGPRs VGPRs.
unsigned InstCacheLineSize
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs) const
Return the maximum number of waves per SIMD for kernels using SGPRs SGPRs.
Generation getGeneration() const
unsigned getMaxNumSGPRs(unsigned WavesPerEU, bool Addressable) const
std::pair< unsigned, unsigned > getMaxNumVectorRegs(const Function &F) const
Return a pair of maximum numbers of VGPRs and AGPRs that meet the number of waves per execution unit ...
bool isXNACKEnabled() const
unsigned getBaseReservedNumSGPRs(const bool HasFlatScratch) const
bool hasAddr64() const
void checkSubtargetFeatures(const Function &F) const
Diagnose inconsistent subtarget features before attempting to codegen function F.
~GCNSubtarget() override
const SelectionDAGTargetInfo * getSelectionDAGInfo() const override
static unsigned getNumUserSGPRForField(UserSGPRID ID)
void allocKernargPreloadSGPRs(unsigned NumSGPRs)
bool hasPrivateSegmentBuffer() const
GCNUserSGPRUsageInfo(const Function &F, const GCNSubtarget &ST)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Instructions::const_iterator const_instr_iterator
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
const MachineBasicBlock * getParent() const
bool isBundle() const
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
unsigned getSubReg() const
int64_t getImm() const
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
Scheduling dependency.
Definition ScheduleDAG.h:52
Kind getKind() const
Returns an enum value representing the kind of the dependence.
@ Data
Regular data dependence (aka true-dependence).
Definition ScheduleDAG.h:56
void setLatency(unsigned Lat)
Sets the latency for this edge.
@ Artificial
Arbitrary strong DAG edge (no real dependence).
Definition ScheduleDAG.h:75
unsigned getLatency() const
Returns the latency value for this edge, which roughly means the minimum number of cycles that must e...
Register getReg() const
Returns the register associated with this edge.
void setReg(Register Reg)
Assigns the associated register for this edge.
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
std::pair< unsigned, unsigned > getWavesPerEU() const
GCNUserSGPRUsageInfo & getUserSGPRInfo()
Scheduling unit. This is a node in the scheduling DAG.
Targets can subclass this to parameterize the SelectionDAG lowering and instruction selection process...
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Information about stack frame layout on the target.
Provide an instruction scheduling machine model to CodeGen passes.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
@ AMDGPUSubArch1250S
Definition Triple.h:271
@ AMDGPUSubArch9
Definition Triple.h:218
@ AMDGPUSubArch9_4
Definition Triple.h:231
@ AMDGPUSubArch6
Definition Triple.h:196
@ AMDGPUSubArch10_3
Definition Triple.h:241
@ AMDGPUSubArch90A
Definition Triple.h:229
@ AMDGPUSubArch810
Definition Triple.h:216
@ AMDGPUSubArch11
Definition Triple.h:250
@ AMDGPUSubArch7
Definition Triple.h:201
@ AMDGPUSubArch12_5
Definition Triple.h:270
@ AMDGPUSubArch10_1
Definition Triple.h:235
@ AMDGPUSubArch11_7
Definition Triple.h:261
@ AMDGPUSubArch8
Definition Triple.h:209
@ AMDGPUSubArch13
Definition Triple.h:275
@ AMDGPUSubArch12
Definition Triple.h:266
@ AMDGPUSubArch908
Definition Triple.h:228
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
self_iterator getIterator()
Definition ilist_node.h:123
unsigned getNumWavesPerEUWithNumVGPRs(const MCSubtargetInfo &STI, unsigned NumVGPRs, unsigned DynamicVGPRBlockSize)
unsigned getAddressableLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getOccupancyWithNumSGPRs(unsigned SGPRs, unsigned MaxWaves, unsigned TotalNumSGPRs, unsigned Granule, unsigned TrapReserve)
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
StringRef getSchedStrategy(const Function &F)
constexpr unsigned getNumWorkGroupSIMDs(bool FullSIMDMode)
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
bool isFullSIMDMode(const MCSubtargetInfo &STI)
unsigned getLdsDwGranularity(const MCSubtargetInfo &ST)
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
unsigned getDynamicVGPRBlockSize(const Function &F)
LLVM_ABI Triple::SubArchType getMajorSubArch(Triple::SubArchType SubArch)
std::pair< unsigned, unsigned > getIntegerPairAttribute(const Function &F, StringRef Name, std::pair< unsigned, unsigned > Default, bool OnlyFirstRequired)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ SPIR_KERNEL
Used for SPIR kernel functions.
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
DWARFExpression::Operation Op
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Software pipelining policy for a loop, which a target can customize by implementing TargetSubtargetIn...
bool ShouldLimitRegPressure
Limit the register pressure of the scheduled loop, retrying at a higher II when a schedule needs too ...
Define a generic scheduling policy for targets that don't provide their own MachineSchedStrategy.
bool ShouldTrackLaneMasks
Track LaneMasks to allow reordering of independent subregister writes of the same vreg.
A region of an MBB for scheduling.