LLVM 24.0.0git
SIFrameLowering.cpp
Go to the documentation of this file.
1//===----------------------- SIFrameLowering.cpp --------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//==-----------------------------------------------------------------------===//
8
9#include "SIFrameLowering.h"
10#include "AMDGPU.h"
11#include "AMDGPULaneMaskUtils.h"
12#include "GCNSubtarget.h"
15#include "SISpillUtils.h"
21#include "llvm/Support/LEB128.h"
23
24using namespace llvm;
25
26#define DEBUG_TYPE "frame-info"
27
29 "amdgpu-spill-vgpr-to-agpr",
30 cl::desc("Enable spilling VGPRs to AGPRs"),
32 cl::init(true));
33
34static constexpr unsigned SGPRBitSize = 32;
35static constexpr unsigned SGPRByteSize = SGPRBitSize / 8;
36static constexpr unsigned VGPRLaneBitSize = 32;
37
38// Find a register matching \p RC from \p LiveUnits which is unused and
39// available throughout the function. On failure, returns AMDGPU::NoRegister.
40// TODO: Rewrite the loop here to iterate over MCRegUnits instead of
41// MCRegisters. This should reduce the number of iterations and avoid redundant
42// checking.
44 const LiveRegUnits &LiveUnits,
45 const TargetRegisterClass &RC) {
46 for (MCRegister Reg : RC) {
47 if (!MRI.isPhysRegUsed(Reg) && LiveUnits.available(Reg) &&
48 !MRI.isReserved(Reg))
49 return Reg;
50 }
51 return MCRegister();
52}
53
54static void encodeDwarfRegisterLocation(int DwarfReg, raw_ostream &OS) {
55 assert(DwarfReg >= 0);
56 if (DwarfReg < 32) {
57 OS << uint8_t(dwarf::DW_OP_reg0 + DwarfReg);
58 } else {
59 OS << uint8_t(dwarf::DW_OP_regx);
60 encodeULEB128(DwarfReg, OS);
61 }
62}
63
65 int64_t DwarfStackPtrReg) {
66 assert(ST.hasFlatScratchEnabled());
67
68 // When flat scratch is enabled, the stack pointer is an address in the
69 // private_lane DWARF address space (i.e. swizzled), but in order to
70 // accurately and efficiently describe things like masked spills of vector
71 // registers we want to define the CFA to be an address in the private_wave
72 // DWARF address space (i.e. unswizzled). To achieve this we scale the stack
73 // pointer by the wavefront size, implemented as (SP << wave_size_log2).
74 const unsigned WavefrontSizeLog2 = ST.getWavefrontSizeLog2();
75 assert(WavefrontSizeLog2 < 32);
76
79 encodeDwarfRegisterLocation(DwarfStackPtrReg, OSBlock);
80 OSBlock << uint8_t(dwarf::DW_OP_deref_size) << uint8_t(SGPRByteSize)
81 << uint8_t(dwarf::DW_OP_lit0 + WavefrontSizeLog2)
82 << uint8_t(dwarf::DW_OP_shl)
83 << uint8_t(dwarf::DW_OP_lit0 +
84 dwarf::DW_ASPACE_LLVM_AMDGPU_private_wave)
85 << uint8_t(dwarf::DW_OP_LLVM_user)
86 << uint8_t(dwarf::DW_OP_LLVM_form_aspace_address);
87
88 SmallString<20> CFIInst;
89 raw_svector_ostream OSCFIInst(CFIInst);
90 OSCFIInst << uint8_t(dwarf::DW_CFA_def_cfa_expression);
91 encodeULEB128(Block.size(), OSCFIInst);
92 OSCFIInst << Block;
93
94 return MCCFIInstruction::createEscape(nullptr, OSCFIInst.str());
95}
96
97void SIFrameLowering::emitDefCFA(MachineBasicBlock &MBB,
99 DebugLoc const &DL, MCRegister StackPtrReg,
100 bool AspaceAlreadyDefined,
101 MachineInstr::MIFlag Flags) const {
103 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
104 const SIRegisterInfo *TRI = ST.getRegisterInfo();
105
106 int64_t DwarfStackPtrReg = TRI->getDwarfRegNum(StackPtrReg, false);
107 MCCFIInstruction CFIInst =
108 ST.hasFlatScratchEnabled()
109 ? createScaledCFAInPrivateWave(ST, DwarfStackPtrReg)
110 : (AspaceAlreadyDefined
111 ? MCCFIInstruction::createLLVMDefAspaceCfa(
112 nullptr, DwarfStackPtrReg, 0,
113 dwarf::DW_ASPACE_LLVM_AMDGPU_private_wave, SMLoc())
114 : MCCFIInstruction::createDefCfaRegister(nullptr,
115 DwarfStackPtrReg));
116 buildCFI(MBB, MBBI, DL, CFIInst, Flags);
117}
118
119// Find a scratch register that we can use in the prologue. We avoid using
120// callee-save registers since they may appear to be free when this is called
121// from canUseAsPrologue (during shrink wrapping), but then no longer be free
122// when this is called from emitPrologue.
124 MachineRegisterInfo &MRI, LiveRegUnits &LiveUnits,
125 const TargetRegisterClass &RC, bool Unused = false) {
126 // Mark callee saved registers as used so we will not choose them.
127 const MCPhysReg *CSRegs = MRI.getCalleeSavedRegs();
128 for (unsigned i = 0; CSRegs[i]; ++i)
129 LiveUnits.addReg(CSRegs[i]);
130
131 // We are looking for a register that can be used throughout the entire
132 // function, so any use is unacceptable.
133 if (Unused)
134 return findUnusedRegister(MRI, LiveUnits, RC);
135
136 for (MCRegister Reg : RC) {
137 if (LiveUnits.available(Reg) && !MRI.isReserved(Reg))
138 return Reg;
139 }
140
141 return MCRegister();
142}
143
144/// Query target location for spilling SGPRs
145/// \p IncludeScratchCopy : Also look for free scratch SGPRs
147 MachineFunction &MF, LiveRegUnits &LiveUnits, Register SGPR,
148 const TargetRegisterClass &RC = AMDGPU::SReg_32_XM0_XEXECRegClass,
149 bool IncludeScratchCopy = true) {
151 MachineFrameInfo &FrameInfo = MF.getFrameInfo();
152
153 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
154 const SIRegisterInfo *TRI = ST.getRegisterInfo();
155 unsigned Size = TRI->getSpillSize(RC);
156 Align Alignment = TRI->getSpillAlign(RC);
157
158 // We need to save and restore the given SGPR.
159
160 Register ScratchSGPR;
161 // 1: Try to save the given register into an unused scratch SGPR. The
162 // LiveUnits should have all the callee saved registers marked as used. For
163 // certain cases we skip copy to scratch SGPR.
164 if (IncludeScratchCopy)
165 ScratchSGPR = findUnusedRegister(MF.getRegInfo(), LiveUnits, RC);
166
167 if (!ScratchSGPR) {
168 int FI = FrameInfo.CreateStackObject(Size, Alignment, true, nullptr,
170
171 if (TRI->spillSGPRToVGPR() &&
172 MFI->allocateSGPRSpillToVGPRLane(MF, FI, /*SpillToPhysVGPRLane=*/true,
173 /*IsPrologEpilog=*/true)) {
174 // 2: There's no free lane to spill, and no free register to save the
175 // SGPR, so we're forced to take another VGPR to use for the spill.
179
180 LLVM_DEBUG(auto Spill = MFI->getSGPRSpillToPhysicalVGPRLanes(FI).front();
181 dbgs() << printReg(SGPR, TRI) << " requires fallback spill to "
182 << printReg(Spill.VGPR, TRI) << ':' << Spill.Lane
183 << '\n';);
184 } else {
185 // Remove dead <FI> index
187 // 3: If all else fails, spill the register to memory.
188 FI = FrameInfo.CreateSpillStackObject(Size, Alignment);
190 SGPR,
192 LLVM_DEBUG(dbgs() << "Reserved FI " << FI << " for spilling "
193 << printReg(SGPR, TRI) << '\n');
194 }
195 } else {
199 LiveUnits.addReg(ScratchSGPR);
200 LLVM_DEBUG(dbgs() << "Saving " << printReg(SGPR, TRI) << " with copy to "
201 << printReg(ScratchSGPR, TRI) << '\n');
202 }
203}
204
205// We need to specially emit stack operations here because a different frame
206// register is used than in the rest of the function, as getFrameRegister would
207// use.
208static void buildPrologSpill(const GCNSubtarget &ST, const SIRegisterInfo &TRI,
209 const SIMachineFunctionInfo &FuncInfo,
210 LiveRegUnits &LiveUnits, MachineFunction &MF,
213 Register SpillReg, int FI, Register FrameReg,
214 int64_t DwordOff = 0) {
215 unsigned Opc = ST.hasFlatScratchEnabled() ? AMDGPU::SCRATCH_STORE_DWORD_SADDR
216 : AMDGPU::BUFFER_STORE_DWORD_OFFSET;
217
218 MachineFrameInfo &FrameInfo = MF.getFrameInfo();
221 PtrInfo, MachineMemOperand::MOStore, FrameInfo.getObjectSize(FI),
222 FrameInfo.getObjectAlign(FI));
223 LiveUnits.addReg(SpillReg);
224 bool IsKill = !MBB.isLiveIn(SpillReg);
225 TRI.buildSpillLoadStore(MBB, I, DL, Opc, FI, SpillReg, IsKill, FrameReg,
226 DwordOff, MMO, nullptr, &LiveUnits);
227 if (IsKill)
228 LiveUnits.removeReg(SpillReg);
229}
230
231static void buildEpilogRestore(const GCNSubtarget &ST,
232 const SIRegisterInfo &TRI,
233 const SIMachineFunctionInfo &FuncInfo,
234 LiveRegUnits &LiveUnits, MachineFunction &MF,
237 const DebugLoc &DL, Register SpillReg, int FI,
238 Register FrameReg, int64_t DwordOff = 0) {
239 unsigned Opc = ST.hasFlatScratchEnabled() ? AMDGPU::SCRATCH_LOAD_DWORD_SADDR
240 : AMDGPU::BUFFER_LOAD_DWORD_OFFSET;
241
242 MachineFrameInfo &FrameInfo = MF.getFrameInfo();
245 PtrInfo, MachineMemOperand::MOLoad, FrameInfo.getObjectSize(FI),
246 FrameInfo.getObjectAlign(FI));
247 TRI.buildSpillLoadStore(MBB, I, DL, Opc, FI, SpillReg, false, FrameReg,
248 DwordOff, MMO, nullptr, &LiveUnits);
249}
250
252 const DebugLoc &DL, const SIInstrInfo *TII,
253 Register TargetReg) {
254 MachineFunction *MF = MBB.getParent();
256 const SIRegisterInfo *TRI = &TII->getRegisterInfo();
257 const MCInstrDesc &SMovB32 = TII->get(AMDGPU::S_MOV_B32);
258 Register TargetLo = TRI->getSubReg(TargetReg, AMDGPU::sub0);
259 Register TargetHi = TRI->getSubReg(TargetReg, AMDGPU::sub1);
260
261 if (MFI->getGITPtrHigh() != 0xffffffff) {
262 BuildMI(MBB, I, DL, SMovB32, TargetHi)
263 .addImm(MFI->getGITPtrHigh())
264 .addReg(TargetReg, RegState::ImplicitDefine);
265 } else {
266 const MCInstrDesc &GetPC64 = TII->get(AMDGPU::S_GETPC_B64_pseudo);
267 BuildMI(MBB, I, DL, GetPC64, TargetReg);
268 }
269 Register GitPtrLo = MFI->getGITPtrLoReg(*MF);
270 MF->getRegInfo().addLiveIn(GitPtrLo);
271 MBB.addLiveIn(GitPtrLo);
272 BuildMI(MBB, I, DL, SMovB32, TargetLo)
273 .addReg(GitPtrLo);
274}
275
276static void initLiveUnits(LiveRegUnits &LiveUnits, const SIRegisterInfo &TRI,
277 const SIMachineFunctionInfo *FuncInfo,
279 MachineBasicBlock::iterator MBBI, bool IsProlog) {
280 if (LiveUnits.empty()) {
281 LiveUnits.init(TRI);
282 if (IsProlog) {
283 LiveUnits.addLiveIns(MBB);
284 } else {
285 // In epilog.
286 LiveUnits.addLiveOuts(MBB);
287 LiveUnits.stepBackward(*MBBI);
288 }
289 }
290}
291
292namespace llvm {
293
294// SpillBuilder to save/restore special SGPR spills like the one needed for FP,
295// BP, etc. These spills are delayed until the current function's frame is
296// finalized. For a given register, the builder uses the
297// PrologEpilogSGPRSaveRestoreInfo to decide the spill method.
301 MachineFunction &MF;
302 const GCNSubtarget &ST;
303 MachineFrameInfo &MFI;
304 SIMachineFunctionInfo *FuncInfo;
305 const SIInstrInfo *TII;
306 const SIRegisterInfo &TRI;
307 const MCRegisterInfo *MCRI;
308 const SIFrameLowering *TFI;
309 Register SuperReg;
311 LiveRegUnits &LiveUnits;
312 const DebugLoc &DL;
313 Register FrameReg;
314 ArrayRef<int16_t> SplitParts;
315 unsigned NumSubRegs;
316 unsigned EltSize = 4;
317 bool IsFramePtrPrologSpill;
318 bool NeedsFrameMoves;
319
320 static bool isExec(Register Reg) {
321 return Reg == AMDGPU::EXEC_LO || Reg == AMDGPU::EXEC;
322 }
323
324 /// If this builder requires SuperReg-based CFI, which is emitted after all
325 /// SubRegs are actually spilled, return the Register which should be used
326 /// as input to getDwarfRegNum. Otherwise, CFI should be generated per-SubReg.
327 ///
328 /// Note: Most spills handled by this builder generate CFI after each
329 /// SubReg spill, as each SubReg maps directly to a CFI register via
330 /// getDwarfRegNum(SubReg, false). All other cases currently currently
331 /// correspond to the SuperReg directly.
332 MCRegister getCFISuperReg() const {
333 if (IsFramePtrPrologSpill)
334 return FuncInfo->getFrameOffsetReg();
335 // FIXME: CFI for EXEC needs a fix by accurately computing the spill
336 // offset for both the low and high components.
337 if (isExec(SuperReg))
338 return AMDGPU::EXEC;
339 return {};
340 }
341
342 void saveToMemory(const int FI) const {
343 MachineRegisterInfo &MRI = MF.getRegInfo();
344 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
345 assert(!MFI.isDeadObjectIndex(FI));
346
347 initLiveUnits(LiveUnits, TRI, FuncInfo, MF, MBB, MI, /*IsProlog*/ true);
348
350 MRI, LiveUnits, AMDGPU::VGPR_32RegClass);
351 if (!TmpVGPR)
352 report_fatal_error("failed to find free scratch register");
353
354 auto BuildCFI = [&](Register Reg) {
355 TFI->buildCFI(MBB, MI, DL,
357 nullptr, MCRI->getDwarfRegNum(Reg, false),
358 MFI.getObjectOffset(FI) * ST.getWavefrontSize()));
359 };
360 MCRegister CFISuperReg = getCFISuperReg();
361 for (unsigned I = 0, DwordOff = 0; I < NumSubRegs; ++I) {
362 Register SubReg = NumSubRegs == 1
363 ? SuperReg
364 : Register(TRI.getSubReg(SuperReg, SplitParts[I]));
365 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), TmpVGPR)
366 .addReg(SubReg);
367
368 buildPrologSpill(ST, TRI, *FuncInfo, LiveUnits, MF, MBB, MI, DL, TmpVGPR,
369 FI, FrameReg, DwordOff);
370 if (NeedsFrameMoves && !CFISuperReg)
371 BuildCFI(SubReg);
372 DwordOff += 4;
373 }
374 if (NeedsFrameMoves && CFISuperReg)
375 BuildCFI(CFISuperReg);
376 }
377
378 void saveToVGPRLane(const int FI) const {
379 assert(!MFI.isDeadObjectIndex(FI));
380
381 assert(MFI.getStackID(FI) == TargetStackID::SGPRSpill);
383 FuncInfo->getSGPRSpillToPhysicalVGPRLanes(FI);
384 assert(Spill.size() == NumSubRegs);
385
386 MCRegister CFISuperReg = getCFISuperReg();
387 for (unsigned I = 0; I < NumSubRegs; ++I) {
388 Register SubReg = NumSubRegs == 1
389 ? SuperReg
390 : Register(TRI.getSubReg(SuperReg, SplitParts[I]));
391 BuildMI(MBB, MI, DL, TII->get(AMDGPU::SI_SPILL_S32_TO_VGPR),
392 Spill[I].VGPR)
393 .addReg(SubReg)
394 .addImm(Spill[I].Lane)
395 .addReg(Spill[I].VGPR, RegState::Undef);
396 if (NeedsFrameMoves && !CFISuperReg)
397 TFI->buildCFIForSGPRToVGPRSpill(MBB, MI, DL, SubReg, Spill[I].VGPR,
398 Spill[I].Lane);
399 }
400 if (NeedsFrameMoves && CFISuperReg)
401 TFI->buildCFIForSGPRToVGPRSpill(MBB, MI, DL, CFISuperReg, Spill);
402 }
403
404 void copyToScratchSGPR(Register DstReg) const {
405 BuildMI(MBB, MI, DL, TII->get(AMDGPU::COPY), DstReg)
406 .addReg(SuperReg)
408 if (NeedsFrameMoves) {
409 const TargetRegisterClass *RC = TRI.getPhysRegBaseClass(DstReg);
410 ArrayRef<int16_t> DstSplitParts = TRI.getRegSplitParts(RC, EltSize);
411 assert(NumSubRegs == (DstSplitParts.empty() ? 1 : DstSplitParts.size()));
412 MCRegister CFISuperReg = getCFISuperReg();
413 if (!CFISuperReg)
414 CFISuperReg = SuperReg;
415 int64_t DwarfCFISuperReg = MCRI->getDwarfRegNum(CFISuperReg, false);
416 int64_t DwarfDstSuperReg = MCRI->getDwarfRegNum(DstReg, false);
417 if (DwarfCFISuperReg >= 0 && DwarfDstSuperReg >= 0) {
418 TFI->buildCFI(MBB, MI, DL,
420 nullptr, DwarfCFISuperReg, DwarfDstSuperReg));
421 } else if (isExec(CFISuperReg)) {
422 assert(NumSubRegs == 2 && "EXEC larger than 64-bit");
423 TFI->buildCFIForRegToSGPRPairSpill(MBB, MI, DL, CFISuperReg, DstReg);
424 } else {
425 for (unsigned I = 0; I < NumSubRegs; ++I) {
426 MCRegister SrcSubReg = TRI.getSubReg(SuperReg, SplitParts[I]);
427 MCRegister DstSubReg = TRI.getSubReg(DstReg, DstSplitParts[I]);
428 TFI->buildCFI(MBB, MI, DL,
430 nullptr, MCRI->getDwarfRegNum(SrcSubReg, false),
431 MCRI->getDwarfRegNum(DstSubReg, false)));
432 }
433 }
434 }
435 }
436
437 void restoreFromMemory(const int FI) {
438 MachineRegisterInfo &MRI = MF.getRegInfo();
439 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
440
441 initLiveUnits(LiveUnits, TRI, FuncInfo, MF, MBB, MI, /*IsProlog*/ false);
443 MRI, LiveUnits, AMDGPU::VGPR_32RegClass);
444 if (!TmpVGPR)
445 report_fatal_error("failed to find free scratch register");
446
447 for (unsigned I = 0, DwordOff = 0; I < NumSubRegs; ++I) {
448 MCRegister SubReg = NumSubRegs == 1
449 ? SuperReg.asMCReg()
450 : TRI.getSubReg(SuperReg, SplitParts[I]);
451
452 buildEpilogRestore(ST, TRI, *FuncInfo, LiveUnits, MF, MBB, MI, DL,
453 TmpVGPR, FI, FrameReg, DwordOff);
454 assert(SubReg.isPhysical());
455
456 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), SubReg)
457 .addReg(TmpVGPR, RegState::Kill);
458 DwordOff += 4;
459 }
460 }
461
462 void restoreFromVGPRLane(const int FI) {
463 assert(MFI.getStackID(FI) == TargetStackID::SGPRSpill);
465 FuncInfo->getSGPRSpillToPhysicalVGPRLanes(FI);
466 assert(Spill.size() == NumSubRegs);
467
468 for (unsigned I = 0; I < NumSubRegs; ++I) {
469 MCRegister SubReg = NumSubRegs == 1
470 ? SuperReg.asMCReg()
471 : TRI.getSubReg(SuperReg, SplitParts[I]);
472 BuildMI(MBB, MI, DL, TII->get(AMDGPU::SI_RESTORE_S32_FROM_VGPR), SubReg)
473 .addReg(Spill[I].VGPR)
474 .addImm(Spill[I].Lane);
475 }
476 }
477
478 void copyFromScratchSGPR(Register SrcReg) const {
479 BuildMI(MBB, MI, DL, TII->get(AMDGPU::COPY), SuperReg)
480 .addReg(SrcReg)
482 }
483
484public:
489 const DebugLoc &DL, const SIInstrInfo *TII,
490 const SIRegisterInfo &TRI,
491 LiveRegUnits &LiveUnits, Register FrameReg,
492 bool IsFramePtrPrologSpill = false)
493 : MI(MI), MBB(MBB), MF(*MBB.getParent()),
494 ST(MF.getSubtarget<GCNSubtarget>()), MFI(MF.getFrameInfo()),
495 FuncInfo(MF.getInfo<SIMachineFunctionInfo>()), TII(TII), TRI(TRI),
496 MCRI(MF.getContext().getRegisterInfo()), TFI(ST.getFrameLowering()),
497 SuperReg(Reg), SI(SI), LiveUnits(LiveUnits), DL(DL), FrameReg(FrameReg),
498 IsFramePtrPrologSpill(IsFramePtrPrologSpill),
499 NeedsFrameMoves(MF.needsFrameMoves()) {
500 const TargetRegisterClass *RC = TRI.getPhysRegBaseClass(SuperReg);
501 SplitParts = TRI.getRegSplitParts(RC, EltSize);
502 NumSubRegs = SplitParts.empty() ? 1 : SplitParts.size();
503
504 assert(SuperReg != AMDGPU::M0 && "m0 should never spill");
505 }
506
507 void save() {
508 switch (SI.getKind()) {
510 return saveToMemory(SI.getIndex());
512 return saveToVGPRLane(SI.getIndex());
514 return copyToScratchSGPR(SI.getReg());
515 }
516 }
517
518 void restore() {
519 switch (SI.getKind()) {
521 return restoreFromMemory(SI.getIndex());
523 return restoreFromVGPRLane(SI.getIndex());
525 return copyFromScratchSGPR(SI.getReg());
526 }
527 }
528};
529
530} // namespace llvm
531
532// Emit flat scratch setup code, assuming `MFI->hasFlatScratchInit()`
533void SIFrameLowering::emitEntryFunctionFlatScratchInit(
535 const DebugLoc &DL, Register ScratchWaveOffsetReg) const {
536 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
537 const SIInstrInfo *TII = ST.getInstrInfo();
538 const SIRegisterInfo *TRI = &TII->getRegisterInfo();
539 const SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
540
541 // We don't need this if we only have spills since there is no user facing
542 // scratch.
543
544 // TODO: If we know we don't have flat instructions earlier, we can omit
545 // this from the input registers.
546 //
547 // TODO: We only need to know if we access scratch space through a flat
548 // pointer. Because we only detect if flat instructions are used at all,
549 // this will be used more often than necessary on VI.
550
551 Register FlatScrInitLo;
552 Register FlatScrInitHi;
553
554 if (ST.isAmdPalOS()) {
555 // Extract the scratch offset from the descriptor in the GIT
556 LiveRegUnits LiveUnits;
557 LiveUnits.init(*TRI);
558 LiveUnits.addLiveIns(MBB);
559
560 // Find unused reg to load flat scratch init into
561 MachineRegisterInfo &MRI = MF.getRegInfo();
562 Register FlatScrInit = AMDGPU::NoRegister;
563 ArrayRef<MCPhysReg> AllSGPR64s = TRI->getAllSGPR64(MF);
564 unsigned NumPreloaded = (MFI->getNumPreloadedSGPRs() + 1) / 2;
565 AllSGPR64s = AllSGPR64s.slice(
566 std::min(static_cast<unsigned>(AllSGPR64s.size()), NumPreloaded));
567 Register GITPtrLoReg = MFI->getGITPtrLoReg(MF);
568 for (MCPhysReg Reg : AllSGPR64s) {
569 if (LiveUnits.available(Reg) && !MRI.isReserved(Reg) &&
570 MRI.isAllocatable(Reg) && !TRI->isSubRegisterEq(Reg, GITPtrLoReg)) {
571 FlatScrInit = Reg;
572 break;
573 }
574 }
575 assert(FlatScrInit && "Failed to find free register for scratch init");
576
577 FlatScrInitLo = TRI->getSubReg(FlatScrInit, AMDGPU::sub0);
578 FlatScrInitHi = TRI->getSubReg(FlatScrInit, AMDGPU::sub1);
579
580 buildGitPtr(MBB, I, DL, TII, FlatScrInit);
581
582 // We now have the GIT ptr - now get the scratch descriptor from the entry
583 // at offset 0 (or offset 16 for a compute shader).
584 MachinePointerInfo PtrInfo(AMDGPUAS::CONSTANT_ADDRESS);
585 const MCInstrDesc &LoadDwordX2 = TII->get(AMDGPU::S_LOAD_DWORDX2_IMM);
586 auto *MMO = MF.getMachineMemOperand(
587 PtrInfo,
590 8, Align(4));
591 unsigned Offset =
593 const GCNSubtarget &Subtarget = MF.getSubtarget<GCNSubtarget>();
594 unsigned EncodedOffset = AMDGPU::convertSMRDOffsetUnits(Subtarget, Offset);
595 BuildMI(MBB, I, DL, LoadDwordX2, FlatScrInit)
596 .addReg(FlatScrInit)
597 .addImm(EncodedOffset) // offset
598 .addImm(0) // cpol
599 .addMemOperand(MMO);
600
601 // Mask the offset in [47:0] of the descriptor
602 const MCInstrDesc &SAndB32 = TII->get(AMDGPU::S_AND_B32);
603 auto And = BuildMI(MBB, I, DL, SAndB32, FlatScrInitHi)
604 .addReg(FlatScrInitHi)
605 .addImm(0xffff);
606 And->getOperand(3).setIsDead(); // Mark SCC as dead.
607 } else {
608 Register FlatScratchInitReg =
610 assert(FlatScratchInitReg);
611
612 MachineRegisterInfo &MRI = MF.getRegInfo();
613 MRI.addLiveIn(FlatScratchInitReg);
614 MBB.addLiveIn(FlatScratchInitReg);
615
616 FlatScrInitLo = TRI->getSubReg(FlatScratchInitReg, AMDGPU::sub0);
617 FlatScrInitHi = TRI->getSubReg(FlatScratchInitReg, AMDGPU::sub1);
618 }
619
620 // Do a 64-bit pointer add.
621 if (ST.flatScratchIsPointer()) {
622 if (ST.getGeneration() >= AMDGPUSubtarget::GFX10) {
623 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADD_U32), FlatScrInitLo)
624 .addReg(FlatScrInitLo)
625 .addReg(ScratchWaveOffsetReg);
626 auto Addc = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADDC_U32),
627 FlatScrInitHi)
628 .addReg(FlatScrInitHi)
629 .addImm(0);
630 Addc->getOperand(3).setIsDead(); // Mark SCC as dead.
631
632 using namespace AMDGPU::Hwreg;
633 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_SETREG_B32))
634 .addReg(FlatScrInitLo)
635 .addImm(int16_t(HwregEncoding::encode(ID_FLAT_SCR_LO, 0, 32)));
636 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_SETREG_B32))
637 .addReg(FlatScrInitHi)
638 .addImm(int16_t(HwregEncoding::encode(ID_FLAT_SCR_HI, 0, 32)));
639 return;
640 }
641
642 // For GFX9.
643 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADD_U32), AMDGPU::FLAT_SCR_LO)
644 .addReg(FlatScrInitLo)
645 .addReg(ScratchWaveOffsetReg);
646 auto Addc = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADDC_U32),
647 AMDGPU::FLAT_SCR_HI)
648 .addReg(FlatScrInitHi)
649 .addImm(0);
650 Addc->getOperand(3).setIsDead(); // Mark SCC as dead.
651
652 return;
653 }
654
655 assert(ST.getGeneration() < AMDGPUSubtarget::GFX9);
656
657 // Copy the size in bytes.
658 BuildMI(MBB, I, DL, TII->get(AMDGPU::COPY), AMDGPU::FLAT_SCR_LO)
659 .addReg(FlatScrInitHi, RegState::Kill);
660
661 // Add wave offset in bytes to private base offset.
662 // See comment in AMDKernelCodeT.h for enable_sgpr_flat_scratch_init.
663 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADD_I32), FlatScrInitLo)
664 .addReg(FlatScrInitLo)
665 .addReg(ScratchWaveOffsetReg);
666
667 // Convert offset to 256-byte units.
668 auto LShr = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_LSHR_B32),
669 AMDGPU::FLAT_SCR_HI)
670 .addReg(FlatScrInitLo, RegState::Kill)
671 .addImm(8);
672 LShr->getOperand(3).setIsDead(); // Mark SCC as dead.
673}
674
675// Note SGPRSpill stack IDs should only be used for SGPR spilling to VGPRs, not
676// memory. They should have been removed by now.
678 for (int I = MFI.getObjectIndexBegin(), E = MFI.getObjectIndexEnd();
679 I != E; ++I) {
680 if (!MFI.isDeadObjectIndex(I))
681 return false;
682 }
683
684 return true;
685}
686
687// Shift down registers reserved for the scratch RSRC.
688Register SIFrameLowering::getEntryFunctionReservedScratchRsrcReg(
689 MachineFunction &MF) const {
690
691 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
692 const SIInstrInfo *TII = ST.getInstrInfo();
693 const SIRegisterInfo *TRI = &TII->getRegisterInfo();
694 MachineRegisterInfo &MRI = MF.getRegInfo();
695 SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
696
697 assert(MFI->isEntryFunction());
698
699 Register ScratchRsrcReg = MFI->getScratchRSrcReg();
700
701 if (!ScratchRsrcReg || (!MRI.isPhysRegUsed(ScratchRsrcReg) &&
703 return Register();
704
705 if (ST.hasSGPRInitBug() ||
706 ScratchRsrcReg != TRI->reservedPrivateSegmentBufferReg(MF))
707 return ScratchRsrcReg;
708
709 // We reserved the last registers for this. Shift it down to the end of those
710 // which were actually used.
711 //
712 // FIXME: It might be safer to use a pseudoregister before replacement.
713
714 // FIXME: We should be able to eliminate unused input registers. We only
715 // cannot do this for the resources required for scratch access. For now we
716 // skip over user SGPRs and may leave unused holes.
717
718 unsigned NumPreloaded = (MFI->getNumPreloadedSGPRs() + 3) / 4;
719 ArrayRef<MCPhysReg> AllSGPR128s = TRI->getAllSGPR128(MF);
720 AllSGPR128s = AllSGPR128s.slice(std::min(static_cast<unsigned>(AllSGPR128s.size()), NumPreloaded));
721
722 // Skip the last N reserved elements because they should have already been
723 // reserved for VCC etc.
724 Register GITPtrLoReg = MFI->getGITPtrLoReg(MF);
725 for (MCPhysReg Reg : AllSGPR128s) {
726 // Pick the first unallocated one. Make sure we don't clobber the other
727 // reserved input we needed. Also for PAL, make sure we don't clobber
728 // the GIT pointer passed in SGPR0 or SGPR8.
729 if (!MRI.isPhysRegUsed(Reg) && MRI.isAllocatable(Reg) &&
730 (!GITPtrLoReg || !TRI->isSubRegisterEq(Reg, GITPtrLoReg))) {
731 MRI.replaceRegWith(ScratchRsrcReg, Reg);
733 MRI.reserveReg(Reg, TRI);
734 return Reg;
735 }
736 }
737
738 return ScratchRsrcReg;
739}
740
741static unsigned getScratchScaleFactor(const GCNSubtarget &ST) {
742 return ST.hasFlatScratchEnabled() ? 1 : ST.getWavefrontSize();
743}
744
746 MachineBasicBlock &MBB) const {
747 assert(&MF.front() == &MBB && "Shrink-wrapping not yet supported");
748
749 // FIXME: If we only have SGPR spills, we won't actually be using scratch
750 // memory since these spill to VGPRs. We should be cleaning up these unused
751 // SGPR spill frame indices somewhere.
752
753 // FIXME: We still have implicit uses on SGPR spill instructions in case they
754 // need to spill to vector memory. It's likely that will not happen, but at
755 // this point it appears we need the setup. This part of the prolog should be
756 // emitted after frame indices are eliminated.
757
758 // FIXME: Remove all of the isPhysRegUsed checks
759
761 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
762 const SIInstrInfo *TII = ST.getInstrInfo();
763 const SIRegisterInfo *TRI = &TII->getRegisterInfo();
765 const Function &F = MF.getFunction();
766 MachineFrameInfo &FrameInfo = MF.getFrameInfo();
767
768 assert(MFI->isEntryFunction());
769
770 // Debug location must be unknown since the first debug location is used to
771 // determine the end of the prologue.
772 DebugLoc DL;
774
775 if (MF.needsFrameMoves()) {
776 // On entry the SP/FP are not set up, so we need to define the CFA in terms
777 // of a literal location expression.
778 static const char CFAEncodedInstUserOpsArr[] = {
779 dwarf::DW_CFA_def_cfa_expression,
780 4, // length
781 static_cast<char>(dwarf::DW_OP_lit0),
782 static_cast<char>(dwarf::DW_OP_lit0 +
783 dwarf::DW_ASPACE_LLVM_AMDGPU_private_wave),
784 static_cast<char>(dwarf::DW_OP_LLVM_user),
785 static_cast<char>(dwarf::DW_OP_LLVM_form_aspace_address)};
786 static StringRef CFAEncodedInstUserOps =
787 StringRef(CFAEncodedInstUserOpsArr, sizeof(CFAEncodedInstUserOpsArr));
788 buildCFI(MBB, I, DL,
789 MCCFIInstruction::createEscape(nullptr, CFAEncodedInstUserOps,
790 SMLoc(),
791 "CFA is 0 in private_wave aspace"));
792 // Unwinding halts when the return address (PC) is undefined.
793 buildCFI(MBB, I, DL,
795 nullptr, TRI->getDwarfRegNum(AMDGPU::PC_REG, false)));
796 }
797
798 Register PreloadedScratchWaveOffsetReg = MFI->getPreloadedReg(
800
801 // We need to do the replacement of the private segment buffer register even
802 // if there are no stack objects. There could be stores to undef or a
803 // constant without an associated object.
804 //
805 // This will return `Register()` in cases where there are no actual
806 // uses of the SRSRC.
807 Register ScratchRsrcReg;
808 if (!ST.hasFlatScratchEnabled())
809 ScratchRsrcReg = getEntryFunctionReservedScratchRsrcReg(MF);
810
811 // Make the selected register live throughout the function.
812 if (ScratchRsrcReg) {
813 for (MachineBasicBlock &OtherBB : MF) {
814 if (&OtherBB != &MBB) {
815 OtherBB.addLiveIn(ScratchRsrcReg);
816 }
817 }
818 }
819
820 // Now that we have fixed the reserved SRSRC we need to locate the
821 // (potentially) preloaded SRSRC.
822 Register PreloadedScratchRsrcReg;
823 if (ST.isAmdHsaOrMesa(F)) {
824 PreloadedScratchRsrcReg =
826 if (ScratchRsrcReg && PreloadedScratchRsrcReg) {
827 // We added live-ins during argument lowering, but since they were not
828 // used they were deleted. We're adding the uses now, so add them back.
829 MRI.addLiveIn(PreloadedScratchRsrcReg);
830 MBB.addLiveIn(PreloadedScratchRsrcReg);
831 }
832 }
833
834 // We found the SRSRC first because it needs four registers and has an
835 // alignment requirement. If the SRSRC that we found is clobbering with
836 // the scratch wave offset, which may be in a fixed SGPR or a free SGPR
837 // chosen by SITargetLowering::allocateSystemSGPRs, COPY the scratch
838 // wave offset to a free SGPR.
839 Register ScratchWaveOffsetReg;
840 if (PreloadedScratchWaveOffsetReg &&
841 TRI->isSubRegisterEq(ScratchRsrcReg, PreloadedScratchWaveOffsetReg)) {
842 ArrayRef<MCPhysReg> AllSGPRs = TRI->getAllSGPR32(MF);
843 unsigned NumPreloaded = MFI->getNumPreloadedSGPRs();
844 AllSGPRs = AllSGPRs.slice(
845 std::min(static_cast<unsigned>(AllSGPRs.size()), NumPreloaded));
846 Register GITPtrLoReg = MFI->getGITPtrLoReg(MF);
847 for (MCPhysReg Reg : AllSGPRs) {
848 if (!MRI.isPhysRegUsed(Reg) && MRI.isAllocatable(Reg) &&
849 !TRI->isSubRegisterEq(ScratchRsrcReg, Reg) && GITPtrLoReg != Reg) {
850 ScratchWaveOffsetReg = Reg;
851 BuildMI(MBB, I, DL, TII->get(AMDGPU::COPY), ScratchWaveOffsetReg)
852 .addReg(PreloadedScratchWaveOffsetReg, RegState::Kill);
853 break;
854 }
855 }
856
857 // FIXME: We can spill incoming arguments and restore at the end of the
858 // prolog.
859 if (!ScratchWaveOffsetReg)
861 "could not find temporary scratch offset register in prolog");
862 } else {
863 ScratchWaveOffsetReg = PreloadedScratchWaveOffsetReg;
864 }
865 assert(ScratchWaveOffsetReg || !PreloadedScratchWaveOffsetReg);
866
867 unsigned Offset = FrameInfo.getStackSize() * getScratchScaleFactor(ST);
868 if (!mayReserveScratchForCWSR(MF)) {
869 if (hasFP(MF)) {
871 assert(FPReg != AMDGPU::FP_REG);
872 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_MOV_B32), FPReg).addImm(0);
873 }
874
877 assert(SPReg != AMDGPU::SP_REG);
878 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_MOV_B32), SPReg).addImm(Offset);
879 }
880 } else {
881 // We need to check if we're on a compute queue - if we are, then the CWSR
882 // trap handler may need to store some VGPRs on the stack. The first VGPR
883 // block is saved separately, so we only need to allocate space for any
884 // additional VGPR blocks used. For now, we will make sure there's enough
885 // room for the theoretical maximum number of VGPRs that can be allocated.
886 // FIXME: Figure out if the shader uses fewer VGPRs in practice.
887 assert(hasFP(MF));
889 assert(FPReg != AMDGPU::FP_REG);
890 unsigned VGPRSize = llvm::alignTo(
891 (ST.getAddressableNumVGPRs(MFI->getDynamicVGPRBlockSize()) -
893 MFI->getDynamicVGPRBlockSize())) *
894 4,
895 FrameInfo.getMaxAlign());
897
898 BuildMI(MBB, I, DL, TII->get(AMDGPU::GET_STACK_BASE), FPReg);
901 assert(SPReg != AMDGPU::SP_REG);
902
903 // If at least one of the constants can be inlined, then we can use
904 // s_cselect. Otherwise, use a mov and cmovk.
905 if (AMDGPU::isInlinableLiteral32(Offset, ST.hasInv2PiInlineImm()) ||
907 ST.hasInv2PiInlineImm())) {
908 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_CSELECT_B32), SPReg)
909 .addImm(Offset + VGPRSize)
910 .addImm(Offset);
911 } else {
912 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_MOV_B32), SPReg).addImm(Offset);
913 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_CMOVK_I32), SPReg)
914 .addImm(Offset + VGPRSize);
915 }
916 }
917 }
918
919 bool NeedsFlatScratchInit =
921 (MRI.isPhysRegUsed(AMDGPU::FLAT_SCR) || FrameInfo.hasCalls() ||
922 (!allStackObjectsAreDead(FrameInfo) && ST.hasFlatScratchEnabled()));
923
924 if ((NeedsFlatScratchInit || ScratchRsrcReg) &&
925 PreloadedScratchWaveOffsetReg && !ST.hasArchitectedFlatScratch()) {
926 MRI.addLiveIn(PreloadedScratchWaveOffsetReg);
927 MBB.addLiveIn(PreloadedScratchWaveOffsetReg);
928 }
929
930 if (NeedsFlatScratchInit) {
931 emitEntryFunctionFlatScratchInit(MF, MBB, I, DL, ScratchWaveOffsetReg);
932 }
933
934 if (ScratchRsrcReg) {
935 emitEntryFunctionScratchRsrcRegSetup(MF, MBB, I, DL,
936 PreloadedScratchRsrcReg,
937 ScratchRsrcReg, ScratchWaveOffsetReg);
938 }
939}
940
941// Emit scratch RSRC setup code, assuming `ScratchRsrcReg != AMDGPU::NoReg`
942void SIFrameLowering::emitEntryFunctionScratchRsrcRegSetup(
944 const DebugLoc &DL, Register PreloadedScratchRsrcReg,
945 Register ScratchRsrcReg, Register ScratchWaveOffsetReg) const {
946
947 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
948 const SIInstrInfo *TII = ST.getInstrInfo();
949 const SIRegisterInfo *TRI = &TII->getRegisterInfo();
951 const Function &Fn = MF.getFunction();
952
953 if (ST.isAmdPalOS()) {
954 // The pointer to the GIT is formed from the offset passed in and either
955 // the amdgpu-git-ptr-high function attribute or the top part of the PC
956 Register Rsrc01 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub0_sub1);
957 Register Rsrc03 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub3);
958
959 buildGitPtr(MBB, I, DL, TII, Rsrc01);
960
961 // We now have the GIT ptr - now get the scratch descriptor from the entry
962 // at offset 0 (or offset 16 for a compute shader).
964 const MCInstrDesc &LoadDwordX4 = TII->get(AMDGPU::S_LOAD_DWORDX4_IMM);
965 auto *MMO = MF.getMachineMemOperand(
966 PtrInfo,
969 16, Align(4));
970 unsigned Offset = Fn.getCallingConv() == CallingConv::AMDGPU_CS ? 16 : 0;
971 const GCNSubtarget &Subtarget = MF.getSubtarget<GCNSubtarget>();
972 unsigned EncodedOffset = AMDGPU::convertSMRDOffsetUnits(Subtarget, Offset);
973 BuildMI(MBB, I, DL, LoadDwordX4, ScratchRsrcReg)
974 .addReg(Rsrc01)
975 .addImm(EncodedOffset) // offset
976 .addImm(0) // cpol
977 .addReg(ScratchRsrcReg, RegState::ImplicitDefine)
978 .addMemOperand(MMO);
979
980 // The driver will always set the SRD for wave 64 (bits 118:117 of
981 // descriptor / bits 22:21 of third sub-reg will be 0b11)
982 // If the shader is actually wave32 we have to modify the const_index_stride
983 // field of the descriptor 3rd sub-reg (bits 22:21) to 0b10 (stride=32). The
984 // reason the driver does this is that there can be cases where it presents
985 // 2 shaders with different wave size (e.g. VsFs).
986 // TODO: convert to using SCRATCH instructions or multiple SRD buffers
987 if (ST.isWave32()) {
988 const MCInstrDesc &SBitsetB32 = TII->get(AMDGPU::S_BITSET0_B32);
989 BuildMI(MBB, I, DL, SBitsetB32, Rsrc03)
990 .addImm(21)
991 .addReg(Rsrc03);
992 }
993 } else if (ST.isMesaGfxShader(Fn) || !PreloadedScratchRsrcReg) {
994 assert(!ST.isAmdHsaOrMesa(Fn));
995 const MCInstrDesc &SMovB32 = TII->get(AMDGPU::S_MOV_B32);
996
997 Register Rsrc2 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub2);
998 Register Rsrc3 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub3);
999
1000 // Use relocations to get the pointer, and setup the other bits manually.
1001 uint64_t Rsrc23 = TII->getScratchRsrcWords23();
1002
1004 Register Rsrc01 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub0_sub1);
1005
1007 const MCInstrDesc &Mov64 = TII->get(AMDGPU::S_MOV_B64);
1008
1009 BuildMI(MBB, I, DL, Mov64, Rsrc01)
1011 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1012 } else {
1013 const MCInstrDesc &LoadDwordX2 = TII->get(AMDGPU::S_LOAD_DWORDX2_IMM);
1014
1015 MachinePointerInfo PtrInfo(AMDGPUAS::CONSTANT_ADDRESS);
1016 auto *MMO = MF.getMachineMemOperand(
1017 PtrInfo,
1020 8, Align(4));
1021 BuildMI(MBB, I, DL, LoadDwordX2, Rsrc01)
1023 .addImm(0) // offset
1024 .addImm(0) // cpol
1025 .addMemOperand(MMO)
1026 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1027
1030 }
1031 } else {
1032 Register Rsrc0 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub0);
1033 Register Rsrc1 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub1);
1034
1035 BuildMI(MBB, I, DL, SMovB32, Rsrc0)
1036 .addExternalSymbol("SCRATCH_RSRC_DWORD0")
1037 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1038
1039 BuildMI(MBB, I, DL, SMovB32, Rsrc1)
1040 .addExternalSymbol("SCRATCH_RSRC_DWORD1")
1041 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1042 }
1043
1044 BuildMI(MBB, I, DL, SMovB32, Rsrc2)
1045 .addImm(Lo_32(Rsrc23))
1046 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1047
1048 BuildMI(MBB, I, DL, SMovB32, Rsrc3)
1049 .addImm(Hi_32(Rsrc23))
1050 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1051 } else if (ST.isAmdHsaOrMesa(Fn)) {
1052 assert(PreloadedScratchRsrcReg);
1053
1054 if (ScratchRsrcReg != PreloadedScratchRsrcReg) {
1055 BuildMI(MBB, I, DL, TII->get(AMDGPU::COPY), ScratchRsrcReg)
1056 .addReg(PreloadedScratchRsrcReg, RegState::Kill);
1057 }
1058 }
1059
1060 // Add the scratch wave offset into the scratch RSRC.
1061 //
1062 // We only want to update the first 48 bits, which is the base address
1063 // pointer, without touching the adjacent 16 bits of flags. We know this add
1064 // cannot carry-out from bit 47, otherwise the scratch allocation would be
1065 // impossible to fit in the 48-bit global address space.
1066 //
1067 // TODO: Evaluate if it is better to just construct an SRD using the flat
1068 // scratch init and some constants rather than update the one we are passed.
1069 Register ScratchRsrcSub0 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub0);
1070 Register ScratchRsrcSub1 = TRI->getSubReg(ScratchRsrcReg, AMDGPU::sub1);
1071
1072 // We cannot Kill ScratchWaveOffsetReg here because we allow it to be used in
1073 // the kernel body via inreg arguments.
1074 BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADD_U32), ScratchRsrcSub0)
1075 .addReg(ScratchRsrcSub0)
1076 .addReg(ScratchWaveOffsetReg)
1077 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1078 auto Addc = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADDC_U32), ScratchRsrcSub1)
1079 .addReg(ScratchRsrcSub1)
1080 .addImm(0)
1081 .addReg(ScratchRsrcReg, RegState::ImplicitDefine);
1082 Addc->getOperand(3).setIsDead(); // Mark SCC as dead.
1083}
1084
1086 switch (ID) {
1090 return true;
1095 return false;
1096 }
1097 llvm_unreachable("Invalid TargetStackID::Value");
1098}
1099
1100void SIFrameLowering::emitPrologueEntryCFI(MachineBasicBlock &MBB,
1102 const DebugLoc &DL) const {
1103 const MachineFunction &MF = *MBB.getParent();
1104 const MachineRegisterInfo &MRI = MF.getRegInfo();
1105 const MCRegisterInfo *MCRI = MF.getContext().getRegisterInfo();
1106 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1107 const SIRegisterInfo &TRI = ST.getInstrInfo()->getRegisterInfo();
1108 MCRegister StackPtrReg =
1109 MF.getInfo<SIMachineFunctionInfo>()->getStackPtrOffsetReg();
1110
1111 emitDefCFA(MBB, MBBI, DL, StackPtrReg, /*AspaceAlreadyDefined=*/true,
1113
1114 buildCFIForRegToSGPRPairSpill(MBB, MBBI, DL, AMDGPU::PC_REG,
1115 TRI.getReturnAddressReg(MF));
1116
1117 BitVector IsCalleeSaved(TRI.getNumRegs());
1118 const MCPhysReg *CSRegs = MRI.getCalleeSavedRegs();
1119 for (unsigned I = 0; CSRegs[I]; ++I) {
1120 IsCalleeSaved.set(CSRegs[I]);
1121 }
1122 auto ProcessReg = [&](MCPhysReg Reg) {
1123 // VCC is not preserved across calls.
1124 if (Reg == AMDGPU::VCC || Reg == AMDGPU::VCC_LO || Reg == AMDGPU::VCC_HI)
1125 return;
1126 if (IsCalleeSaved.test(Reg) || !MRI.isPhysRegModified(Reg))
1127 return;
1128 unsigned DwarfReg = MCRI->getDwarfRegNum(Reg, false);
1129 buildCFI(MBB, MBBI, DL,
1130 MCCFIInstruction::createUndefined(nullptr, DwarfReg));
1131 };
1132
1133 // Emit CFI rules for caller saved Arch VGPRs which are clobbered
1134 unsigned NumArchVGPRs = ST.has1024AddressableVGPRs() ? 1024 : 256;
1135 for_each(AMDGPU::VGPR_32RegClass.getRegisters().take_front(NumArchVGPRs),
1136 ProcessReg);
1137
1138 // Emit CFI rules for caller saved Accum VGPRs which are clobbered
1139 if (ST.hasMAIInsts()) {
1140 for_each(AMDGPU::AGPR_32RegClass.getRegisters(), ProcessReg);
1141 }
1142
1143 // Emit CFI rules for caller saved SGPRs which are clobbered
1144 for_each(AMDGPU::SGPR_32RegClass.getRegisters(), ProcessReg);
1145}
1146
1147// Activate only the inactive lanes when \p EnableInactiveLanes is true.
1148// Otherwise, activate all lanes. It returns the saved exec.
1150 MachineFunction &MF,
1153 const DebugLoc &DL, bool IsProlog,
1154 bool EnableInactiveLanes) {
1155 Register ScratchExecCopy;
1156 MachineRegisterInfo &MRI = MF.getRegInfo();
1157 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1158 const SIInstrInfo *TII = ST.getInstrInfo();
1159 const SIRegisterInfo &TRI = TII->getRegisterInfo();
1161
1162 initLiveUnits(LiveUnits, TRI, FuncInfo, MF, MBB, MBBI, IsProlog);
1163
1164 if (FuncInfo->isWholeWaveFunction()) {
1165 // Whole wave functions already have a copy of the original EXEC mask that
1166 // we can use.
1167 assert(IsProlog && "Epilog should look at return, not setup");
1168 ScratchExecCopy =
1169 TII->getWholeWaveFunctionSetup(MF)->getOperand(0).getReg();
1170 assert(ScratchExecCopy && "Couldn't find copy of EXEC");
1171 } else {
1172 ScratchExecCopy = findScratchNonCalleeSaveRegister(
1173 MRI, LiveUnits, *TRI.getWaveMaskRegClass());
1174 }
1175
1176 if (!ScratchExecCopy)
1177 report_fatal_error("failed to find free scratch register");
1178
1179 LiveUnits.addReg(ScratchExecCopy);
1180
1181 const unsigned SaveExecOpc =
1182 ST.isWave32() ? (EnableInactiveLanes ? AMDGPU::S_XOR_SAVEEXEC_B32
1183 : AMDGPU::S_OR_SAVEEXEC_B32)
1184 : (EnableInactiveLanes ? AMDGPU::S_XOR_SAVEEXEC_B64
1185 : AMDGPU::S_OR_SAVEEXEC_B64);
1186 auto SaveExec =
1187 BuildMI(MBB, MBBI, DL, TII->get(SaveExecOpc), ScratchExecCopy).addImm(-1);
1188 SaveExec->getOperand(3).setIsDead(); // Mark SCC as dead.
1189
1190 return ScratchExecCopy;
1191}
1192
1196 LiveRegUnits &LiveUnits, Register FrameReg, Register FramePtrRegScratchCopy,
1197 const bool NeedsFrameMoves) const {
1199 MachineFrameInfo &MFI = MF.getFrameInfo();
1200 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1201 const SIInstrInfo *TII = ST.getInstrInfo();
1202 const SIRegisterInfo &TRI = TII->getRegisterInfo();
1203 const MCRegisterInfo *MCRI = MF.getContext().getRegisterInfo();
1204 MachineRegisterInfo &MRI = MF.getRegInfo();
1206
1207 // Spill Whole-Wave Mode VGPRs. Save only the inactive lanes of the scratch
1208 // registers. However, save all lanes of callee-saved VGPRs. Due to this, we
1209 // might end up flipping the EXEC bits twice.
1210 Register ScratchExecCopy;
1211 SmallVector<std::pair<Register, int>, 2> WWMCalleeSavedRegs, WWMScratchRegs;
1212 FuncInfo->splitWWMSpillRegisters(MF, WWMCalleeSavedRegs, WWMScratchRegs);
1213 if (!WWMScratchRegs.empty())
1214 ScratchExecCopy =
1215 buildScratchExecCopy(LiveUnits, MF, MBB, MBBI, DL,
1216 /*IsProlog*/ true, /*EnableInactiveLanes*/ true);
1217
1218 auto StoreWWMRegisters =
1220 for (const auto &Reg : WWMRegs) {
1221 Register VGPR = Reg.first;
1222 int FI = Reg.second;
1223 buildPrologSpill(ST, TRI, *FuncInfo, LiveUnits, MF, MBB, MBBI, DL,
1224 VGPR, FI, FrameReg);
1225 if (NeedsFrameMoves) {
1226 // We spill the entire VGPR, so we can get away with just cfi_offset
1227 buildCFI(MBB, MBBI, DL,
1229 nullptr, MCRI->getDwarfRegNum(VGPR, false),
1230 MFI.getObjectOffset(FI) * ST.getWavefrontSize()));
1231 }
1232 }
1233 };
1234
1235 for (const Register Reg : make_first_range(WWMScratchRegs)) {
1236 if (!MRI.isReserved(Reg)) {
1237 MRI.addLiveIn(Reg);
1238 MBB.addLiveIn(Reg);
1239 }
1240 }
1241 StoreWWMRegisters(WWMScratchRegs);
1242
1243 auto EnableAllLanes = [&]() {
1244 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg).addImm(-1);
1245 };
1246
1247 if (!WWMCalleeSavedRegs.empty()) {
1248 if (ScratchExecCopy) {
1249 EnableAllLanes();
1250 } else {
1251 ScratchExecCopy = buildScratchExecCopy(LiveUnits, MF, MBB, MBBI, DL,
1252 /*IsProlog*/ true,
1253 /*EnableInactiveLanes*/ false);
1254 }
1255 }
1256
1257 StoreWWMRegisters(WWMCalleeSavedRegs);
1258 if (FuncInfo->isWholeWaveFunction()) {
1259 // If we have already saved some WWM CSR registers, then the EXEC is already
1260 // -1 and we don't need to do anything else. Otherwise, save the original
1261 // EXEC into the setup register and set EXEC to -1 here.
1262 if (!ScratchExecCopy)
1263 buildScratchExecCopy(LiveUnits, MF, MBB, MBBI, DL, /*IsProlog*/ true,
1264 /*EnableInactiveLanes*/ false);
1265 else if (WWMCalleeSavedRegs.empty())
1266 EnableAllLanes();
1267 } else if (ScratchExecCopy) {
1268 // FIXME: Split block and make terminator.
1269 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg)
1270 .addReg(ScratchExecCopy, RegState::Kill);
1271 LiveUnits.addReg(ScratchExecCopy);
1272 }
1273
1274 Register FramePtrReg = FuncInfo->getFrameOffsetReg();
1275
1276 for (const auto &Spill : FuncInfo->getPrologEpilogSGPRSpills()) {
1277 // Special handle FP spill:
1278 // Skip if FP is saved to a scratch SGPR, the save has already been emitted.
1279 // Otherwise, FP has been moved to a temporary register and spill it
1280 // instead.
1281 bool IsFramePtrPrologSpill = Spill.first == FramePtrReg;
1282 Register Reg = IsFramePtrPrologSpill ? FramePtrRegScratchCopy : Spill.first;
1283 if (!Reg)
1284 continue;
1285
1286 PrologEpilogSGPRSpillBuilder SB(Reg, Spill.second, MBB, MBBI, DL, TII, TRI,
1287 LiveUnits, FrameReg, IsFramePtrPrologSpill);
1288 SB.save();
1289 }
1290
1291 // If a copy to scratch SGPR has been chosen for any of the SGPR spills, make
1292 // such scratch registers live throughout the function.
1293 SmallVector<Register, 1> ScratchSGPRs;
1294 FuncInfo->getAllScratchSGPRCopyDstRegs(ScratchSGPRs);
1295 if (!ScratchSGPRs.empty()) {
1296 for (MachineBasicBlock &MBB : MF) {
1297 for (MCPhysReg Reg : ScratchSGPRs)
1298 MBB.addLiveIn(Reg);
1299
1300 MBB.sortUniqueLiveIns();
1301 }
1302 if (!LiveUnits.empty()) {
1303 for (MCPhysReg Reg : ScratchSGPRs)
1304 LiveUnits.addReg(Reg);
1305 }
1306 }
1307
1308 // Remove the spill entry created for EXEC. It is needed only for CFISaves in
1309 // the prologue.
1310 if (TRI.isCFISavedRegsSpillEnabled())
1311 FuncInfo->removePrologEpilogSGPRSpillEntry(TRI.getExec());
1312}
1313
1317 LiveRegUnits &LiveUnits, Register FrameReg,
1318 Register FramePtrRegScratchCopy) const {
1319 const SIMachineFunctionInfo *FuncInfo = MF.getInfo<SIMachineFunctionInfo>();
1320 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1321 const SIInstrInfo *TII = ST.getInstrInfo();
1322 const SIRegisterInfo &TRI = TII->getRegisterInfo();
1324 Register FramePtrReg = FuncInfo->getFrameOffsetReg();
1325
1326 for (const auto &Spill : FuncInfo->getPrologEpilogSGPRSpills()) {
1327 // Special handle FP restore:
1328 // Skip if FP needs to be restored from the scratch SGPR. Otherwise, restore
1329 // the FP value to a temporary register. The frame pointer should be
1330 // overwritten only at the end when all other spills are restored from
1331 // current frame.
1332 Register Reg =
1333 Spill.first == FramePtrReg ? FramePtrRegScratchCopy : Spill.first;
1334 if (!Reg)
1335 continue;
1336
1337 PrologEpilogSGPRSpillBuilder SB(Reg, Spill.second, MBB, MBBI, DL, TII, TRI,
1338 LiveUnits, FrameReg);
1339 SB.restore();
1340 }
1341
1342 // Restore Whole-Wave Mode VGPRs. Restore only the inactive lanes of the
1343 // scratch registers. However, restore all lanes of callee-saved VGPRs. Due to
1344 // this, we might end up flipping the EXEC bits twice.
1345 Register ScratchExecCopy;
1346 SmallVector<std::pair<Register, int>, 2> WWMCalleeSavedRegs, WWMScratchRegs;
1347 FuncInfo->splitWWMSpillRegisters(MF, WWMCalleeSavedRegs, WWMScratchRegs);
1348 auto RestoreWWMRegisters =
1350 for (const auto &Reg : WWMRegs) {
1351 Register VGPR = Reg.first;
1352 int FI = Reg.second;
1353 buildEpilogRestore(ST, TRI, *FuncInfo, LiveUnits, MF, MBB, MBBI, DL,
1354 VGPR, FI, FrameReg);
1355 }
1356 };
1357
1358 if (FuncInfo->isWholeWaveFunction()) {
1359 // For whole wave functions, the EXEC is already -1 at this point.
1360 // Therefore, we can restore the CSR WWM registers right away.
1361 RestoreWWMRegisters(WWMCalleeSavedRegs);
1362
1363 // The original EXEC is the first operand of the return instruction.
1364 MachineInstr &Return = MBB.instr_back();
1365 unsigned Opcode = Return.getOpcode();
1366 switch (Opcode) {
1367 case AMDGPU::SI_WHOLE_WAVE_FUNC_RETURN:
1368 Opcode = AMDGPU::SI_RETURN;
1369 break;
1370 case AMDGPU::SI_TCRETURN_GFX_WholeWave:
1371 Opcode = AMDGPU::SI_TCRETURN_GFX;
1372 break;
1373 default:
1374 llvm_unreachable("Unexpected return inst");
1375 }
1376 Register OrigExec = Return.getOperand(0).getReg();
1377
1378 if (!WWMScratchRegs.empty()) {
1379 BuildMI(MBB, MBBI, DL, TII->get(LMC.XorOpc), LMC.ExecReg)
1380 .addReg(OrigExec)
1381 .addImm(-1);
1382 RestoreWWMRegisters(WWMScratchRegs);
1383 }
1384
1385 // Restore original EXEC.
1386 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg).addReg(OrigExec);
1387
1388 // Drop the first operand and update the opcode.
1389 Return.removeOperand(0);
1390 Return.setDesc(TII->get(Opcode));
1391
1392 return;
1393 }
1394
1395 if (!WWMScratchRegs.empty()) {
1396 ScratchExecCopy =
1397 buildScratchExecCopy(LiveUnits, MF, MBB, MBBI, DL,
1398 /*IsProlog=*/false, /*EnableInactiveLanes=*/true);
1399 }
1400 RestoreWWMRegisters(WWMScratchRegs);
1401 if (!WWMCalleeSavedRegs.empty()) {
1402 if (ScratchExecCopy) {
1403 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg).addImm(-1);
1404 } else {
1405 ScratchExecCopy = buildScratchExecCopy(LiveUnits, MF, MBB, MBBI, DL,
1406 /*IsProlog*/ false,
1407 /*EnableInactiveLanes*/ false);
1408 }
1409 }
1410
1411 RestoreWWMRegisters(WWMCalleeSavedRegs);
1412 if (ScratchExecCopy) {
1413 // FIXME: Split block and make terminator.
1414 BuildMI(MBB, MBBI, DL, TII->get(LMC.MovOpc), LMC.ExecReg)
1415 .addReg(ScratchExecCopy, RegState::Kill);
1416 }
1417}
1418
1420 MachineBasicBlock &MBB) const {
1422 if (FuncInfo->isEntryFunction()) {
1424 return;
1425 }
1426
1427 MachineFrameInfo &MFI = MF.getFrameInfo();
1428 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1429 const SIInstrInfo *TII = ST.getInstrInfo();
1430 const SIRegisterInfo &TRI = TII->getRegisterInfo();
1431 MachineRegisterInfo &MRI = MF.getRegInfo();
1432
1433 Register StackPtrReg = FuncInfo->getStackPtrOffsetReg();
1434 Register FramePtrReg = FuncInfo->getFrameOffsetReg();
1435 Register BasePtrReg =
1436 TRI.hasBasePointer(MF) ? TRI.getBaseRegister() : Register();
1437 LiveRegUnits LiveUnits;
1438
1440 // DebugLoc must be unknown since the first instruction with DebugLoc is used
1441 // to determine the end of the prologue.
1442 DebugLoc DL;
1443
1444 bool HasFP = false;
1445 bool HasBP = false;
1446 uint32_t NumBytes = MFI.getStackSize();
1447 uint32_t RoundedSize = NumBytes;
1448
1449 // Functions that never return don't need to save and restore the FP or BP.
1450 const Function &F = MF.getFunction();
1451 bool SavesStackRegs =
1452 !F.hasFnAttribute(Attribute::NoReturn) && !FuncInfo->isChainFunction();
1453
1454 const bool NeedsFrameMoves = MF.needsFrameMoves();
1455
1456 if (NeedsFrameMoves)
1457 emitPrologueEntryCFI(MBB, MBBI, DL);
1458
1459 if (TRI.hasStackRealignment(MF))
1460 HasFP = true;
1461
1462 Register FramePtrRegScratchCopy;
1463 if (!HasFP && !hasFP(MF)) {
1464 // Emit the CSR spill stores with SP base register.
1465 emitCSRSpillStores(MF, MBB, MBBI, DL, LiveUnits, StackPtrReg,
1466 FramePtrRegScratchCopy, NeedsFrameMoves);
1467 } else if (SavesStackRegs) {
1468 // CSR spill stores will use FP as base register.
1469 Register SGPRForFPSaveRestoreCopy =
1470 FuncInfo->getScratchSGPRCopyDstReg(FramePtrReg);
1471
1472 initLiveUnits(LiveUnits, TRI, FuncInfo, MF, MBB, MBBI, /*IsProlog*/ true);
1473 if (SGPRForFPSaveRestoreCopy) {
1474 // Copy FP to the scratch register now and emit the CFI entry. It avoids
1475 // the extra FP copy needed in the other two cases when FP is spilled to
1476 // memory or to a VGPR lane.
1478 FramePtrReg,
1479 FuncInfo->getPrologEpilogSGPRSaveRestoreInfo(FramePtrReg), MBB, MBBI,
1480 DL, TII, TRI, LiveUnits, FramePtrReg,
1481 /*IsFramePtrPrologSpill*/ true);
1482 SB.save();
1483 LiveUnits.addReg(SGPRForFPSaveRestoreCopy);
1484 } else {
1485 // Copy FP into a new scratch register so that its previous value can be
1486 // spilled after setting up the new frame.
1487 FramePtrRegScratchCopy = findScratchNonCalleeSaveRegister(
1488 MRI, LiveUnits, AMDGPU::SReg_32_XM0_XEXECRegClass);
1489 if (!FramePtrRegScratchCopy)
1490 report_fatal_error("failed to find free scratch register");
1491
1492 LiveUnits.addReg(FramePtrRegScratchCopy);
1493 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::COPY), FramePtrRegScratchCopy)
1494 .addReg(FramePtrReg);
1495 }
1496 }
1497
1498 if (HasFP) {
1499 const unsigned Alignment = MFI.getMaxAlign().value();
1500
1501 RoundedSize += Alignment;
1502 if (LiveUnits.empty()) {
1503 LiveUnits.init(TRI);
1504 LiveUnits.addLiveIns(MBB);
1505 }
1506
1507 // s_add_i32 s33, s32, NumBytes
1508 // s_and_b32 s33, s33, 0b111...0000
1509 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::S_ADD_I32), FramePtrReg)
1510 .addReg(StackPtrReg)
1511 .addImm((Alignment - 1) * getScratchScaleFactor(ST))
1513 auto And = BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::S_AND_B32), FramePtrReg)
1514 .addReg(FramePtrReg, RegState::Kill)
1515 .addImm(-Alignment * getScratchScaleFactor(ST))
1517 And->getOperand(3).setIsDead(); // Mark SCC as dead.
1518 FuncInfo->setIsStackRealigned(true);
1519 } else if ((HasFP = hasFP(MF))) {
1520 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::COPY), FramePtrReg)
1521 .addReg(StackPtrReg)
1523 }
1524
1525 // If FP is used, emit the CSR spills with FP base register.
1526 if (HasFP) {
1527 emitCSRSpillStores(MF, MBB, MBBI, DL, LiveUnits, FramePtrReg,
1528 FramePtrRegScratchCopy, NeedsFrameMoves);
1529 if (FramePtrRegScratchCopy)
1530 LiveUnits.removeReg(FramePtrRegScratchCopy);
1531 }
1532
1533 // If we need a base pointer, set it up here. It's whatever the value of
1534 // the stack pointer is at this point. Any variable size objects will be
1535 // allocated after this, so we can still use the base pointer to reference
1536 // the incoming arguments.
1537 if ((HasBP = TRI.hasBasePointer(MF))) {
1538 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::COPY), BasePtrReg)
1539 .addReg(StackPtrReg)
1541 }
1542
1543 if (HasFP) {
1544 if (NeedsFrameMoves)
1545 emitDefCFA(MBB, MBBI, DL, FramePtrReg, /*AspaceAlreadyDefined=*/false,
1547 }
1548
1549 if (HasFP && RoundedSize != 0) {
1550 auto Add = BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::S_ADD_I32), StackPtrReg)
1551 .addReg(StackPtrReg)
1552 .addImm(RoundedSize * getScratchScaleFactor(ST))
1554 Add->getOperand(3).setIsDead(); // Mark SCC as dead.
1555 }
1556
1557 bool FPSaved = FuncInfo->hasPrologEpilogSGPRSpillEntry(FramePtrReg);
1558 (void)FPSaved;
1559 assert((!HasFP || FPSaved || !SavesStackRegs) &&
1560 "Needed to save FP but didn't save it anywhere");
1561
1562 // If we allow spilling to AGPRs we may have saved FP but then spill
1563 // everything into AGPRs instead of the stack.
1564 assert((HasFP || !FPSaved || !SavesStackRegs || EnableSpillVGPRToAGPR) &&
1565 "Saved FP but didn't need it");
1566
1567 bool BPSaved = FuncInfo->hasPrologEpilogSGPRSpillEntry(BasePtrReg);
1568 (void)BPSaved;
1569 assert((!HasBP || BPSaved || !SavesStackRegs) &&
1570 "Needed to save BP but didn't save it anywhere");
1571
1572 assert((HasBP || !BPSaved) && "Saved BP but didn't need it");
1573
1574 if (FuncInfo->isWholeWaveFunction()) {
1575 // SI_WHOLE_WAVE_FUNC_SETUP has outlived its purpose.
1576 TII->getWholeWaveFunctionSetup(MF)->eraseFromParent();
1577 }
1578}
1579
1581 MachineBasicBlock &MBB) const {
1582 const SIMachineFunctionInfo *FuncInfo = MF.getInfo<SIMachineFunctionInfo>();
1583 if (FuncInfo->isEntryFunction())
1584 return;
1585
1586 const MachineFrameInfo &MFI = MF.getFrameInfo();
1587 if (FuncInfo->isChainFunction() && !MFI.hasTailCall())
1588 return;
1589
1590 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1591 const SIInstrInfo *TII = ST.getInstrInfo();
1592 const SIRegisterInfo &TRI = TII->getRegisterInfo();
1593 MachineRegisterInfo &MRI = MF.getRegInfo();
1594 LiveRegUnits LiveUnits;
1595 // Get the insert location for the epilogue. If there were no terminators in
1596 // the block, get the last instruction.
1598 DebugLoc DL;
1599 if (!MBB.empty()) {
1600 MBBI = MBB.getLastNonDebugInstr();
1601 if (MBBI != MBB.end())
1602 DL = MBBI->getDebugLoc();
1603
1604 MBBI = MBB.getFirstTerminator();
1605 }
1606
1607 uint32_t NumBytes = MFI.getStackSize();
1608 uint32_t RoundedSize = FuncInfo->isStackRealigned()
1609 ? NumBytes + MFI.getMaxAlign().value()
1610 : NumBytes;
1611 const Register StackPtrReg = FuncInfo->getStackPtrOffsetReg();
1612 Register FramePtrReg = FuncInfo->getFrameOffsetReg();
1613 bool FPSaved = FuncInfo->hasPrologEpilogSGPRSpillEntry(FramePtrReg);
1614
1615 if (RoundedSize != 0) {
1616 if (TRI.hasBasePointer(MF)) {
1617 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::COPY), StackPtrReg)
1618 .addReg(TRI.getBaseRegister())
1620 } else if (hasFP(MF)) {
1621 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::COPY), StackPtrReg)
1622 .addReg(FramePtrReg)
1624 }
1625 }
1626
1627 Register FramePtrRegScratchCopy;
1628 Register SGPRForFPSaveRestoreCopy =
1629 FuncInfo->getScratchSGPRCopyDstReg(FramePtrReg);
1630 if (FPSaved) {
1631 // CSR spill restores should use FP as base register. If
1632 // SGPRForFPSaveRestoreCopy is not true, restore the previous value of FP
1633 // into a new scratch register and copy to FP later when other registers are
1634 // restored from the current stack frame.
1635 initLiveUnits(LiveUnits, TRI, FuncInfo, MF, MBB, MBBI, /*IsProlog*/ false);
1636 if (SGPRForFPSaveRestoreCopy) {
1637 LiveUnits.addReg(SGPRForFPSaveRestoreCopy);
1638 } else {
1639 FramePtrRegScratchCopy = findScratchNonCalleeSaveRegister(
1640 MRI, LiveUnits, AMDGPU::SReg_32_XM0_XEXECRegClass);
1641 if (!FramePtrRegScratchCopy)
1642 report_fatal_error("failed to find free scratch register");
1643
1644 LiveUnits.addReg(FramePtrRegScratchCopy);
1645 }
1646
1647 emitCSRSpillRestores(MF, MBB, MBBI, DL, LiveUnits, FramePtrReg,
1648 FramePtrRegScratchCopy);
1649 }
1650
1651 if (hasFP(MF) && MF.needsFrameMoves()) {
1652 emitDefCFA(MBB, MBBI, DL, StackPtrReg, /*AspaceAlreadyDefined=*/false,
1654 }
1655
1656 if (FPSaved) {
1657 // Insert the copy to restore FP.
1658 Register SrcReg = SGPRForFPSaveRestoreCopy ? SGPRForFPSaveRestoreCopy
1659 : FramePtrRegScratchCopy;
1661 BuildMI(MBB, MBBI, DL, TII->get(AMDGPU::COPY), FramePtrReg)
1662 .addReg(SrcReg);
1663 if (SGPRForFPSaveRestoreCopy)
1665 } else {
1666 // Insert the CSR spill restores with SP as the base register.
1667 emitCSRSpillRestores(MF, MBB, MBBI, DL, LiveUnits, StackPtrReg,
1668 FramePtrRegScratchCopy);
1669 }
1670}
1671
1672#ifndef NDEBUG
1674 const MachineFrameInfo &MFI = MF.getFrameInfo();
1675 const SIMachineFunctionInfo *FuncInfo = MF.getInfo<SIMachineFunctionInfo>();
1676 for (int I = MFI.getObjectIndexBegin(), E = MFI.getObjectIndexEnd();
1677 I != E; ++I) {
1678 if (!MFI.isDeadObjectIndex(I) &&
1681 return false;
1682 }
1683 }
1684
1685 return true;
1686}
1687#endif
1688
1690 int FI,
1691 Register &FrameReg) const {
1692 const SIRegisterInfo *RI = MF.getSubtarget<GCNSubtarget>().getRegisterInfo();
1693
1694 FrameReg = RI->getFrameRegister(MF);
1696}
1697
1699 MachineFunction &MF,
1700 RegScavenger *RS) const {
1701 MachineFrameInfo &MFI = MF.getFrameInfo();
1702
1703 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1704 const SIInstrInfo *TII = ST.getInstrInfo();
1705 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1706 MachineRegisterInfo &MRI = MF.getRegInfo();
1708
1709 const bool SpillVGPRToAGPR = ST.hasMAIInsts() && FuncInfo->hasSpilledVGPRs()
1711
1712 if (SpillVGPRToAGPR) {
1713 // To track the spill frame indices handled in this pass.
1714 BitVector SpillFIs(MFI.getObjectIndexEnd(), false);
1715 BitVector NonVGPRSpillFIs(MFI.getObjectIndexEnd(), false);
1716
1717 bool SeenDbgInstr = false;
1718
1719 for (MachineBasicBlock &MBB : MF) {
1721 int FrameIndex;
1722 if (MI.isDebugInstr())
1723 SeenDbgInstr = true;
1724
1725 if (TII->isVGPRSpill(MI)) {
1726 // Try to eliminate stack used by VGPR spills before frame
1727 // finalization.
1728 unsigned FIOp = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
1729 AMDGPU::OpName::vaddr);
1730 int FI = MI.getOperand(FIOp).getIndex();
1731 Register VReg =
1732 TII->getNamedOperand(MI, AMDGPU::OpName::vdata)->getReg();
1733 if (FuncInfo->allocateVGPRSpillToAGPR(MF, FI,
1734 TRI->isAGPR(MRI, VReg))) {
1735 assert(RS != nullptr);
1736 RS->enterBasicBlockEnd(MBB);
1737 RS->backward(std::next(MI.getIterator()));
1738 TRI->eliminateFrameIndex(MI, 0, FIOp, RS);
1739 SpillFIs.set(FI);
1740 continue;
1741 }
1742 } else if (TII->isStoreToStackSlot(MI, FrameIndex) ||
1743 TII->isLoadFromStackSlot(MI, FrameIndex))
1744 if (!MFI.isFixedObjectIndex(FrameIndex))
1745 NonVGPRSpillFIs.set(FrameIndex);
1746 }
1747 }
1748
1749 // Stack slot coloring may assign different objects to the same stack slot.
1750 // If not, then the VGPR to AGPR spill slot is dead.
1751 for (unsigned FI : SpillFIs.set_bits())
1752 if (!NonVGPRSpillFIs.test(FI))
1753 FuncInfo->setVGPRToAGPRSpillDead(FI);
1754
1755 for (MachineBasicBlock &MBB : MF) {
1756 for (MCPhysReg Reg : FuncInfo->getVGPRSpillAGPRs())
1757 MBB.addLiveIn(Reg);
1758
1759 for (MCPhysReg Reg : FuncInfo->getAGPRSpillVGPRs())
1760 MBB.addLiveIn(Reg);
1761
1762 MBB.sortUniqueLiveIns();
1763
1764 if (!SpillFIs.empty() && SeenDbgInstr)
1765 clearDebugInfoForSpillFIs(MFI, MBB, SpillFIs);
1766 }
1767 }
1768
1769 // At this point we've already allocated all spilled SGPRs to VGPRs if we
1770 // can. Any remaining SGPR spills will go to memory, so move them back to the
1771 // default stack.
1772 bool HaveSGPRToVMemSpill =
1773 FuncInfo->removeDeadFrameIndices(MFI, /*ResetSGPRSpillStackIDs*/ true);
1775 "SGPR spill should have been removed in SILowerSGPRSpills");
1776
1777 // FIXME: The other checks should be redundant with allStackObjectsAreDead,
1778 // but currently hasNonSpillStackObjects is set only from source
1779 // allocas. Stack temps produced from legalization are not counted currently.
1780 if (!allStackObjectsAreDead(MFI)) {
1781 assert(RS && "RegScavenger required if spilling");
1782
1783 // Add an emergency spill slot
1784 RS->addScavengingFrameIndex(FuncInfo->getScavengeFI(MFI, *TRI));
1785
1786 if (HaveSGPRToVMemSpill && FuncInfo->hasNoWWMPoolSGPRSpillFallback()) {
1787 // The no-WWM-pool fallback can reach SGPR-to-memory lowering while an
1788 // ordinary frame-index scavenge is live. It may then need one slot for
1789 // its temporary VGPR and another for recursive address materialization.
1790 RS->addScavengingFrameIndex(MFI.CreateSpillStackObject(4, Align(4)));
1791 RS->addScavengingFrameIndex(MFI.CreateSpillStackObject(4, Align(4)));
1792 } else if (HaveSGPRToVMemSpill &&
1794 // Existing large-frame SGPR-to-memory spills need one additional VGPR
1795 // emergency frame index.
1796 RS->addScavengingFrameIndex(MFI.CreateSpillStackObject(4, Align(4)));
1797 }
1798 }
1799}
1800
1802 MachineFunction &MF, RegScavenger *RS) const {
1803 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1804 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1805 MachineRegisterInfo &MRI = MF.getRegInfo();
1807
1808 if (ST.hasMAIInsts() && !ST.hasGFX90AInsts()) {
1809 // On gfx908, we had initially reserved highest available VGPR for AGPR
1810 // copy. Now since we are done with RA, check if there exist an unused VGPR
1811 // which is lower than the eariler reserved VGPR before RA. If one exist,
1812 // use it for AGPR copy instead of one reserved before RA.
1813 Register VGPRForAGPRCopy = FuncInfo->getVGPRForAGPRCopy();
1814 Register UnusedLowVGPR =
1815 TRI->findUnusedRegister(MRI, &AMDGPU::VGPR_32RegClass, MF);
1816 if (UnusedLowVGPR && (TRI->getHWRegIndex(UnusedLowVGPR) <
1817 TRI->getHWRegIndex(VGPRForAGPRCopy))) {
1818 // Reserve this newly identified VGPR (for AGPR copy)
1819 // reserved registers should already be frozen at this point
1820 // so we can avoid calling MRI.freezeReservedRegs and just use
1821 // MRI.reserveReg
1822 FuncInfo->setVGPRForAGPRCopy(UnusedLowVGPR);
1823 MRI.reserveReg(UnusedLowVGPR, TRI);
1824 }
1825 }
1826 // We initally reserved the highest available SGPR pair for long branches
1827 // now, after RA, we shift down to a lower unused one if one exists
1828 Register LongBranchReservedReg = FuncInfo->getLongBranchReservedReg();
1829 Register UnusedLowSGPR =
1830 TRI->findUnusedRegister(MRI, &AMDGPU::SGPR_64RegClass, MF);
1831 // If LongBranchReservedReg is null then we didn't find a long branch
1832 // and never reserved a register to begin with so there is nothing to
1833 // shift down. Then if UnusedLowSGPR is null, there isn't available lower
1834 // register to use so just keep the original one we set.
1835 if (LongBranchReservedReg && UnusedLowSGPR) {
1836 FuncInfo->setLongBranchReservedReg(UnusedLowSGPR);
1837 MRI.reserveReg(UnusedLowSGPR, TRI);
1838 }
1839}
1840
1841// The special SGPR spills like the one needed for FP, BP or any reserved
1842// registers delayed until frame lowering.
1844 MachineFunction &MF, BitVector &SavedVGPRs,
1845 bool NeedExecCopyReservedReg) const {
1846 MachineFrameInfo &FrameInfo = MF.getFrameInfo();
1847 MachineRegisterInfo &MRI = MF.getRegInfo();
1849 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1850 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1851 LiveRegUnits LiveUnits;
1852 LiveUnits.init(*TRI);
1853 // Initially mark callee saved registers as used so we will not choose them
1854 // while looking for scratch SGPRs.
1855 const MCPhysReg *CSRegs = MF.getRegInfo().getCalleeSavedRegs();
1856 for (unsigned I = 0; CSRegs[I]; ++I)
1857 LiveUnits.addReg(CSRegs[I]);
1858
1859 const TargetRegisterClass &RC = *TRI->getWaveMaskRegClass();
1860
1861 Register ReservedRegForExecCopy = MFI->getSGPRForEXECCopy();
1862 if (NeedExecCopyReservedReg ||
1863 (ReservedRegForExecCopy &&
1864 MRI.isPhysRegUsed(ReservedRegForExecCopy, /*SkipRegMaskTest=*/true))) {
1865 MRI.reserveReg(ReservedRegForExecCopy, TRI);
1866 Register UnusedScratchReg = findUnusedRegister(MRI, LiveUnits, RC);
1867 if (UnusedScratchReg) {
1868 // If found any unused scratch SGPR, reserve the register itself for Exec
1869 // copy and there is no need for any spills in that case.
1870 MFI->setSGPRForEXECCopy(UnusedScratchReg);
1871 MRI.replaceRegWith(ReservedRegForExecCopy, UnusedScratchReg);
1872 LiveUnits.addReg(UnusedScratchReg);
1873 } else {
1874 // Needs spill.
1875 assert(!MFI->hasPrologEpilogSGPRSpillEntry(ReservedRegForExecCopy) &&
1876 "Re-reserving spill slot for EXEC copy register");
1877 getVGPRSpillLaneOrTempRegister(MF, LiveUnits, ReservedRegForExecCopy, RC,
1878 /*IncludeScratchCopy=*/false);
1879 }
1880 } else if (ReservedRegForExecCopy) {
1881 // Reset it at this point. There are no whole-wave copies and spills
1882 // encountered.
1883 MFI->setSGPRForEXECCopy(AMDGPU::NoRegister);
1884 }
1885
1886 if (TRI->isCFISavedRegsSpillEnabled()) {
1887 Register Exec = TRI->getExec();
1889 "Re-reserving spill slot for EXEC");
1890 getVGPRSpillLaneOrTempRegister(MF, LiveUnits, Exec, RC);
1891 }
1892
1893 // Functions that don't return to the caller don't need to preserve
1894 // the FP and BP.
1895 const Function &F = MF.getFunction();
1896 if (F.hasFnAttribute(Attribute::NoReturn) ||
1897 AMDGPU::isChainCC(F.getCallingConv()))
1898 return;
1899
1900 // hasFP only knows about stack objects that already exist. We're now
1901 // determining the stack slots that will be created, so we have to predict
1902 // them. Stack objects force FP usage with calls.
1903 //
1904 // Note a new VGPR CSR may be introduced if one is used for the spill, but we
1905 // don't want to report it here.
1906 //
1907 // FIXME: Is this really hasReservedCallFrame?
1908 const bool WillHaveFP =
1909 FrameInfo.hasCalls() &&
1910 (SavedVGPRs.any() || !allStackObjectsAreDead(FrameInfo));
1911
1912 if (WillHaveFP || hasFP(MF)) {
1913 Register FramePtrReg = MFI->getFrameOffsetReg();
1914 assert(!MFI->hasPrologEpilogSGPRSpillEntry(FramePtrReg) &&
1915 "Re-reserving spill slot for FP");
1916 getVGPRSpillLaneOrTempRegister(MF, LiveUnits, FramePtrReg);
1917 }
1918
1919 if (TRI->hasBasePointer(MF)) {
1920 Register BasePtrReg = TRI->getBaseRegister();
1921 assert(!MFI->hasPrologEpilogSGPRSpillEntry(BasePtrReg) &&
1922 "Re-reserving spill slot for BP");
1923 getVGPRSpillLaneOrTempRegister(MF, LiveUnits, BasePtrReg);
1924 }
1925}
1926
1927// Only report VGPRs to generic code.
1929 BitVector &SavedVGPRs,
1930 RegScavenger *RS) const {
1932
1933 // If this is a function with the amdgpu_cs_chain[_preserve] calling
1934 // convention and it doesn't contain any calls to llvm.amdgcn.cs.chain, then
1935 // we don't need to save and restore anything.
1936 if (MFI->isChainFunction() && !MF.getFrameInfo().hasTailCall())
1937 return;
1938
1940
1941 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
1942 const SIRegisterInfo *TRI = ST.getRegisterInfo();
1943 const SIInstrInfo *TII = ST.getInstrInfo();
1944 bool NeedExecCopyReservedReg = false;
1945
1946 MachineInstr *ReturnMI = nullptr;
1947 for (MachineBasicBlock &MBB : MF) {
1948 for (MachineInstr &MI : MBB) {
1949 // TODO: Walking through all MBBs here would be a bad heuristic. Better
1950 // handle them elsewhere.
1951 if (TII->isWWMRegSpillOpcode(MI.getOpcode()))
1952 NeedExecCopyReservedReg = true;
1953 else if (MI.getOpcode() == AMDGPU::SI_RETURN ||
1954 MI.getOpcode() == AMDGPU::SI_RETURN_TO_EPILOG ||
1955 MI.getOpcode() == AMDGPU::SI_WHOLE_WAVE_FUNC_RETURN ||
1956 (MFI->isChainFunction() &&
1957 TII->isChainCallOpcode(MI.getOpcode()))) {
1958 // We expect all return to be the same size.
1959 assert(!ReturnMI ||
1960 (count_if(MI.operands(), [](auto Op) { return Op.isReg(); }) ==
1961 count_if(ReturnMI->operands(), [](auto Op) { return Op.isReg(); })));
1962 ReturnMI = &MI;
1963 }
1964 }
1965 }
1966
1967 SmallVector<Register> SortedWWMVGPRs;
1968 for (Register Reg : MFI->getWWMReservedRegs()) {
1969 // The shift-back is needed only for the VGPRs used for SGPR spills and they
1970 // are of 32-bit size. SIPreAllocateWWMRegs pass can add tuples into WWM
1971 // reserved registers.
1972 const TargetRegisterClass *RC = TRI->getPhysRegBaseClass(Reg);
1973 if (TRI->getRegSizeInBits(*RC) != 32)
1974 continue;
1975 SortedWWMVGPRs.push_back(Reg);
1976 }
1977
1978 sort(SortedWWMVGPRs, std::greater<Register>());
1979 MFI->shiftWwmVGPRsToLowestRange(MF, SortedWWMVGPRs, SavedVGPRs);
1980
1981 if (MFI->isEntryFunction())
1982 return;
1983
1984 if (MFI->isWholeWaveFunction()) {
1985 // In practice, all the VGPRs are WWM registers, and we will need to save at
1986 // least their inactive lanes. Add them to WWMReservedRegs.
1987 assert(!NeedExecCopyReservedReg &&
1988 "Whole wave functions can use the reg mapped for their i1 argument");
1989
1990 unsigned NumArchVGPRs = ST.getAddressableNumArchVGPRs();
1991 for (MCRegister Reg :
1992 AMDGPU::VGPR_32RegClass.getRegisters().take_front(NumArchVGPRs))
1993 if (MF.getRegInfo().isPhysRegModified(Reg)) {
1994 MFI->reserveWWMRegister(Reg);
1995 MF.begin()->addLiveIn(Reg);
1996 }
1997 MF.begin()->sortUniqueLiveIns();
1998 }
1999
2000 // Remove any VGPRs used in the return value because these do not need to be saved.
2001 // This prevents CSR restore from clobbering return VGPRs.
2002 if (ReturnMI) {
2003 for (auto &Op : ReturnMI->operands()) {
2004 if (Op.isReg())
2005 SavedVGPRs.reset(Op.getReg());
2006 }
2007 }
2008
2009 // Create the stack objects for WWM registers now.
2010 for (Register Reg : MFI->getWWMReservedRegs()) {
2011 const TargetRegisterClass *RC = TRI->getPhysRegBaseClass(Reg);
2012 MFI->allocateWWMSpill(MF, Reg, TRI->getSpillSize(*RC),
2013 TRI->getSpillAlign(*RC));
2014 }
2015
2016 // Ignore the SGPRs the default implementation found.
2017 SavedVGPRs.clearBitsNotInMask(TRI->getAllVectorRegMask());
2018
2019 // Do not save AGPRs prior to GFX90A because there was no easy way to do so.
2020 // In gfx908 there was do AGPR loads and stores and thus spilling also
2021 // require a temporary VGPR.
2022 if (!ST.hasGFX90AInsts())
2023 SavedVGPRs.clearBitsInMask(TRI->getAllAGPRRegMask());
2024
2025 determinePrologEpilogSGPRSaves(MF, SavedVGPRs, NeedExecCopyReservedReg);
2026
2027 // The Whole-Wave VGPRs need to be specially inserted in the prolog, so don't
2028 // allow the default insertion to handle them.
2029 for (auto &Reg : MFI->getWWMSpills())
2030 SavedVGPRs.reset(Reg.first);
2031}
2032
2034 BitVector &SavedRegs,
2035 RegScavenger *RS) const {
2038 if (MFI->isEntryFunction())
2039 return;
2040
2041 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2042 const SIRegisterInfo *TRI = ST.getRegisterInfo();
2043
2044 // The SP is specifically managed and we don't want extra spills of it.
2045 SavedRegs.reset(MFI->getStackPtrOffsetReg());
2046
2047 const BitVector AllSavedRegs = SavedRegs;
2048 SavedRegs.clearBitsInMask(TRI->getAllVectorRegMask());
2049
2050 // We have to anticipate introducing CSR VGPR spills or spill of caller
2051 // save VGPR reserved for SGPR spills as we now always create stack entry
2052 // for it, if we don't have any stack objects already, since we require a FP
2053 // if there is a call and stack. We will allocate a VGPR for SGPR spills if
2054 // there are any SGPR spills. Whether they are CSR spills or otherwise.
2055 MachineFrameInfo &FrameInfo = MF.getFrameInfo();
2056 const bool WillHaveFP =
2057 FrameInfo.hasCalls() && (AllSavedRegs.any() || MFI->hasSpilledSGPRs());
2058
2059 // FP will be specially managed like SP.
2060 if (WillHaveFP || hasFP(MF))
2061 SavedRegs.reset(MFI->getFrameOffsetReg());
2062
2063 // Return address use with return instruction is hidden through the SI_RETURN
2064 // pseudo. Given that and since the IPRA computes actual register usage and
2065 // does not use CSR list, the clobbering of return address by function calls
2066 // (D117243) or otherwise (D120922) is ignored/not seen by the IPRA's register
2067 // usage collection. This will ensure save/restore of return address happens
2068 // in those scenarios.
2069 const MachineRegisterInfo &MRI = MF.getRegInfo();
2070 Register RetAddrReg = TRI->getReturnAddressReg(MF);
2071 if (!MFI->isEntryFunction() &&
2072 (FrameInfo.hasCalls() || MRI.isPhysRegModified(RetAddrReg))) {
2073 SavedRegs.set(TRI->getSubReg(RetAddrReg, AMDGPU::sub0));
2074 SavedRegs.set(TRI->getSubReg(RetAddrReg, AMDGPU::sub1));
2075 }
2076}
2077
2079 const GCNSubtarget &ST,
2080 std::vector<CalleeSavedInfo> &CSI) {
2082 MachineFrameInfo &MFI = MF.getFrameInfo();
2083 const SIRegisterInfo *TRI = ST.getRegisterInfo();
2084
2085 assert(
2086 llvm::is_sorted(CSI,
2087 [](const CalleeSavedInfo &A, const CalleeSavedInfo &B) {
2088 return A.getReg() < B.getReg();
2089 }) &&
2090 "Callee saved registers not sorted");
2091
2092 auto CanUseBlockOps = [&](const CalleeSavedInfo &CSI) {
2093 return !CSI.isSpilledToReg() &&
2094 TRI->getPhysRegBaseClass(CSI.getReg()) == &AMDGPU::VGPR_32RegClass &&
2095 !FuncInfo->isWWMReservedRegister(CSI.getReg());
2096 };
2097
2098 auto CSEnd = CSI.end();
2099 for (auto CSIt = CSI.begin(); CSIt != CSEnd; ++CSIt) {
2100 Register Reg = CSIt->getReg();
2101 if (!CanUseBlockOps(*CSIt))
2102 continue;
2103
2104 // Find all the regs that will fit in a 32-bit mask starting at the current
2105 // reg and build said mask. It should have 1 for every register that's
2106 // included, with the current register as the least significant bit.
2107 uint32_t Mask = 1;
2108 CSEnd = std::remove_if(
2109 CSIt + 1, CSEnd, [&](const CalleeSavedInfo &CSI) -> bool {
2110 if (CanUseBlockOps(CSI) && CSI.getReg() < Reg + 32) {
2111 Mask |= 1 << (CSI.getReg() - Reg);
2112 return true;
2113 } else {
2114 return false;
2115 }
2116 });
2117
2118 const TargetRegisterClass *BlockRegClass = TRI->getRegClassForBlockOp(MF);
2119 Register RegBlock =
2120 TRI->getMatchingSuperReg(Reg, AMDGPU::sub0, BlockRegClass);
2121 if (!RegBlock) {
2122 // We couldn't find a super register for the block. This can happen if
2123 // the register we started with is too high (e.g. v232 if the maximum is
2124 // v255). We therefore try to get the last register block and figure out
2125 // the mask from there.
2126 Register LastBlockStart =
2127 AMDGPU::VGPR0 + alignDown(Reg - AMDGPU::VGPR0, 32);
2128 RegBlock =
2129 TRI->getMatchingSuperReg(LastBlockStart, AMDGPU::sub0, BlockRegClass);
2130 assert(RegBlock && TRI->isSubRegister(RegBlock, Reg) &&
2131 "Couldn't find super register");
2132 int RegDelta = Reg - LastBlockStart;
2133 assert(RegDelta > 0 && llvm::countl_zero(Mask) >= RegDelta &&
2134 "Bad shift amount");
2135 Mask <<= RegDelta;
2136 }
2137
2138 FuncInfo->setMaskForVGPRBlockOps(RegBlock, Mask);
2139
2140 // The stack objects can be a bit smaller than the register block if we know
2141 // some of the high bits of Mask are 0. This may happen often with calling
2142 // conventions where the caller and callee-saved VGPRs are interleaved at
2143 // a small boundary (e.g. 8 or 16).
2144 int UnusedBits = llvm::countl_zero(Mask);
2145 unsigned BlockSize = TRI->getSpillSize(*BlockRegClass) - UnusedBits * 4;
2146 int FrameIdx =
2147 MFI.CreateStackObject(BlockSize, TRI->getSpillAlign(*BlockRegClass),
2148 /*isSpillSlot=*/true);
2149 MFI.setIsCalleeSavedObjectIndex(FrameIdx, true);
2150
2151 CSIt->setFrameIdx(FrameIdx);
2152 CSIt->setReg(RegBlock);
2153 }
2154 CSI.erase(CSEnd, CSI.end());
2155}
2156
2159 std::vector<CalleeSavedInfo> &CSI) const {
2160 if (CSI.empty())
2161 return true; // Early exit if no callee saved registers are modified!
2162
2163 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2164 bool UseVGPRBlocks = ST.useVGPRBlockOpsForCSR();
2165
2166 if (UseVGPRBlocks)
2167 assignSlotsUsingVGPRBlocks(MF, ST, CSI);
2168
2169 return assignCalleeSavedSpillSlotsImpl(MF, TRI, CSI) || UseVGPRBlocks;
2170}
2171
2174 std::vector<CalleeSavedInfo> &CSI) const {
2175 if (CSI.empty())
2176 return true; // Early exit if no callee saved registers are modified!
2177
2178 const SIMachineFunctionInfo *FuncInfo = MF.getInfo<SIMachineFunctionInfo>();
2179 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2180 const SIRegisterInfo *RI = ST.getRegisterInfo();
2181 Register FramePtrReg = FuncInfo->getFrameOffsetReg();
2182 Register BasePtrReg = RI->getBaseRegister();
2183 Register SGPRForFPSaveRestoreCopy =
2184 FuncInfo->getScratchSGPRCopyDstReg(FramePtrReg);
2185 Register SGPRForBPSaveRestoreCopy =
2186 FuncInfo->getScratchSGPRCopyDstReg(BasePtrReg);
2187 if (!SGPRForFPSaveRestoreCopy && !SGPRForBPSaveRestoreCopy)
2188 return false;
2189
2190 unsigned NumModifiedRegs = 0;
2191
2192 if (SGPRForFPSaveRestoreCopy)
2193 NumModifiedRegs++;
2194 if (SGPRForBPSaveRestoreCopy)
2195 NumModifiedRegs++;
2196
2197 for (auto &CS : CSI) {
2198 if (CS.getReg() == FramePtrReg.asMCReg() && SGPRForFPSaveRestoreCopy) {
2199 CS.setDstReg(SGPRForFPSaveRestoreCopy);
2200 if (--NumModifiedRegs)
2201 break;
2202 } else if (CS.getReg() == BasePtrReg.asMCReg() &&
2203 SGPRForBPSaveRestoreCopy) {
2204 CS.setDstReg(SGPRForBPSaveRestoreCopy);
2205 if (--NumModifiedRegs)
2206 break;
2207 }
2208 }
2209
2210 return false;
2211}
2212
2214 const MachineFunction &MF) const {
2215
2216 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2217 const MachineFrameInfo &MFI = MF.getFrameInfo();
2218 const SIInstrInfo *TII = ST.getInstrInfo();
2219 uint64_t EstStackSize = MFI.estimateStackSize(MF);
2220 uint64_t MaxOffset = EstStackSize - 1;
2221
2222 // We need the emergency stack slots to be allocated in range of the
2223 // MUBUF/flat scratch immediate offset from the base register, so assign these
2224 // first at the incoming SP position.
2225 //
2226 // TODO: We could try sorting the objects to find a hole in the first bytes
2227 // rather than allocating as close to possible. This could save a lot of space
2228 // on frames with alignment requirements.
2229 if (ST.hasFlatScratchEnabled()) {
2230 if (TII->isLegalFLATOffset(MaxOffset, AMDGPUAS::PRIVATE_ADDRESS,
2232 return false;
2233 } else {
2234 if (TII->isLegalMUBUFImmOffset(MaxOffset))
2235 return false;
2236 }
2237
2238 return true;
2239}
2240
2241/// Return the set of all root registers of regunits live-in to @p MBB.
2242///
2243/// Intended to avoid using the expensive @c MCRegAliasIterator when deciding
2244/// if a register to be spilled is already live-in (see @c isAnyRootLiveIn).
2246 const SIRegisterInfo &TRI) {
2247 SparseBitVector<> LiveInRoots;
2248 for (const auto &LI : MBB.liveins()) {
2249 for (MCRegUnitMaskIterator MI(LI.PhysReg, &TRI); MI.isValid(); ++MI) {
2250 auto [Unit, UnitLaneMask] = *MI;
2251 if ((LI.LaneMask & UnitLaneMask).none())
2252 continue;
2253 for (MCRegUnitRootIterator RI(Unit, &TRI); RI.isValid(); ++RI)
2254 LiveInRoots.set(*RI);
2255 }
2256 }
2257 return LiveInRoots;
2258}
2259
2260/// Returns true iff any root of @p Reg is in @p LiveInRoots
2261/// (see @c buildLiveInRoots).
2262static bool isAnyRootLiveIn(const SparseBitVector<> &LiveInRoots,
2263 const SIRegisterInfo &TRI, MCRegister Reg) {
2264 for (MCRegUnitIterator UI(Reg, &TRI); UI.isValid(); ++UI) {
2265 for (MCRegUnitRootIterator RI(*UI, &TRI); RI.isValid(); ++RI) {
2266 if (LiveInRoots.test(*RI))
2267 return true;
2268 }
2269 }
2270 return false;
2271}
2272
2273void SIFrameLowering::spillCalleeSavedRegisterWithoutBlockOps(
2275 const CalleeSavedInfo &CS, const SIInstrInfo *TII,
2276 const SIRegisterInfo &TRI,
2277 const std::optional<SparseBitVector<>> &LiveInRoots) const {
2278 MCRegister Reg = CS.getReg();
2279
2280 // We assume a sortUniqueLiveIns later
2281 MBB.addLiveIn(Reg);
2282
2283 if (CS.isSpilledToReg()) {
2284 BuildMI(MBB, MI, DebugLoc(), TII->get(TargetOpcode::COPY), CS.getDstReg())
2285 .addReg(Reg, getKillRegState(true));
2286 } else {
2287 const TargetRegisterClass *RC = TRI.getMinimalPhysRegClass(Reg);
2288 bool IsKill = true;
2289 // If this value was already livein, we probably have a direct use of
2290 // the incoming register value, so don't kill at the spill point. This
2291 // happens since we pass some special inputs (workgroup IDs) in the
2292 // callee saved range.
2293 if (LiveInRoots)
2294 IsKill = !isAnyRootLiveIn(*LiveInRoots, TRI, Reg);
2295 TII->storeRegToStackSlotCFI(MBB, MI, Reg, IsKill, CS.getFrameIdx(), RC);
2296 }
2297}
2298
2301 ArrayRef<CalleeSavedInfo> CSI, const TargetRegisterInfo *OrigTRI) const {
2302 auto &TRI = *static_cast<const SIRegisterInfo *>(OrigTRI);
2303 MachineFunction *MF = MBB.getParent();
2304 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
2305 const SIInstrInfo *TII = ST.getInstrInfo();
2306
2307 std::optional<SparseBitVector<>> LiveInRoots;
2308 if (MBB.getParent()->getRegInfo().tracksLiveness())
2309 LiveInRoots = buildLiveInRoots(MBB, TRI);
2310
2311 if (!ST.useVGPRBlockOpsForCSR()) {
2312 for (const CalleeSavedInfo &CS : CSI)
2313 spillCalleeSavedRegisterWithoutBlockOps(MBB, MI, CS, TII, TRI,
2314 LiveInRoots);
2315 if (LiveInRoots)
2316 MBB.sortUniqueLiveIns();
2317 return true;
2318 }
2319
2320 MachineFrameInfo &FrameInfo = MF->getFrameInfo();
2322
2323 const TargetRegisterClass *BlockRegClass = TRI.getRegClassForBlockOp(*MF);
2324 for (const CalleeSavedInfo &CS : CSI) {
2325 Register Reg = CS.getReg();
2326 if (!BlockRegClass->contains(Reg) ||
2327 !FuncInfo->hasMaskForVGPRBlockOps(Reg)) {
2328 spillCalleeSavedRegisterWithoutBlockOps(MBB, MI, CS, TII, TRI,
2329 LiveInRoots);
2330 continue;
2331 }
2332
2333 // Build a scratch block store.
2334 uint32_t Mask = FuncInfo->getMaskForVGPRBlockOps(Reg);
2335 int FrameIndex = CS.getFrameIdx();
2336 MachinePointerInfo PtrInfo =
2337 MachinePointerInfo::getFixedStack(*MF, FrameIndex);
2338 MachineMemOperand *MMO =
2340 FrameInfo.getObjectSize(FrameIndex),
2341 FrameInfo.getObjectAlign(FrameIndex));
2342
2343 BuildMI(MBB, MI, MI->getDebugLoc(),
2344 TII->get(AMDGPU::SI_BLOCK_SPILL_V1024_CFI_SAVE))
2345 .addReg(Reg, getKillRegState(false))
2346 .addFrameIndex(FrameIndex)
2347 .addReg(FuncInfo->getStackPtrOffsetReg())
2348 .addImm(0)
2349 .addImm(Mask)
2350 .addMemOperand(MMO);
2351
2352 FuncInfo->setHasSpilledVGPRs();
2353
2354 // Add the register to the liveins. This is necessary because if any of the
2355 // VGPRs in the register block is reserved (e.g. if it's a WWM register),
2356 // then the whole block will be marked as reserved and `updateLiveness` will
2357 // skip it.
2358 if (LiveInRoots)
2359 MBB.addLiveIn(Reg);
2360 }
2361 if (LiveInRoots)
2362 MBB.sortUniqueLiveIns();
2363
2364 return true;
2365}
2366
2370 const TargetRegisterInfo *OrigTRI) const {
2371 auto &TRI = *static_cast<const SIRegisterInfo *>(OrigTRI);
2372 MachineFunction *MF = MBB.getParent();
2373 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
2374 if (!ST.useVGPRBlockOpsForCSR())
2375 return false;
2376
2378 MachineFrameInfo &MFI = MF->getFrameInfo();
2379 const SIInstrInfo *TII = ST.getInstrInfo();
2380 const TargetRegisterClass *BlockRegClass = TRI.getRegClassForBlockOp(*MF);
2381 for (const CalleeSavedInfo &CS : reverse(CSI)) {
2382 Register Reg = CS.getReg();
2383 if (!BlockRegClass->contains(Reg) ||
2384 !FuncInfo->hasMaskForVGPRBlockOps(Reg)) {
2386 continue;
2387 }
2388
2389 // Build a scratch block load.
2390 uint32_t Mask = FuncInfo->getMaskForVGPRBlockOps(Reg);
2391 int FrameIndex = CS.getFrameIdx();
2392 MachinePointerInfo PtrInfo =
2393 MachinePointerInfo::getFixedStack(*MF, FrameIndex);
2395 PtrInfo, MachineMemOperand::MOLoad, MFI.getObjectSize(FrameIndex),
2396 MFI.getObjectAlign(FrameIndex));
2397
2398 auto MIB = BuildMI(MBB, MI, MI->getDebugLoc(),
2399 TII->get(AMDGPU::SI_BLOCK_SPILL_V1024_RESTORE), Reg)
2400 .addFrameIndex(FrameIndex)
2401 .addReg(FuncInfo->getStackPtrOffsetReg())
2402 .addImm(0)
2403 .addImm(Mask)
2404 .addMemOperand(MMO);
2405 TRI.addImplicitUsesForBlockCSRLoad(MIB, Reg);
2406
2407 // Add the register to the liveins. This is necessary because if any of the
2408 // VGPRs in the register block is reserved (e.g. if it's a WWM register),
2409 // then the whole block will be marked as reserved and `updateLiveness` will
2410 // skip it.
2411 MBB.addLiveIn(Reg);
2412 }
2413
2414 MBB.sortUniqueLiveIns();
2415 return true;
2416}
2417
2419 MachineFunction &MF,
2422 int64_t Amount = I->getOperand(0).getImm();
2423 if (Amount == 0)
2424 return MBB.erase(I);
2425
2426 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2427 const SIInstrInfo *TII = ST.getInstrInfo();
2428 const DebugLoc &DL = I->getDebugLoc();
2429 unsigned Opc = I->getOpcode();
2430 bool IsDestroy = Opc == TII->getCallFrameDestroyOpcode();
2431 uint64_t CalleePopAmount = IsDestroy ? I->getOperand(1).getImm() : 0;
2432
2433 if (!hasReservedCallFrame(MF)) {
2434 Amount = alignTo(Amount, getStackAlign());
2435 assert(isUInt<32>(Amount) && "exceeded stack address space size");
2438
2439 Amount *= getScratchScaleFactor(ST);
2440 if (IsDestroy)
2441 Amount = -Amount;
2442 auto Add = BuildMI(MBB, I, DL, TII->get(AMDGPU::S_ADD_I32), SPReg)
2443 .addReg(SPReg)
2444 .addImm(Amount);
2445 Add->getOperand(3).setIsDead(); // Mark SCC as dead.
2446 } else if (CalleePopAmount != 0) {
2447 llvm_unreachable("is this used?");
2448 }
2449
2450 return MBB.erase(I);
2451}
2452
2453/// Returns true if the frame will require a reference to the stack pointer.
2454///
2455/// This is the set of conditions common to setting up the stack pointer in a
2456/// kernel, and for using a frame pointer in a callable function.
2457///
2458/// FIXME: Should also check hasOpaqueSPAdjustment and if any inline asm
2459/// references SP.
2461 return MFI.hasVarSizedObjects() || MFI.hasStackMap() || MFI.hasPatchPoint();
2462}
2463
2464// The FP for kernels is always known 0, so we never really need to setup an
2465// explicit register for it. However, DisableFramePointerElim will force us to
2466// use a register for it.
2468 const MachineFrameInfo &MFI = MF.getFrameInfo();
2469
2470 // For entry functions we can use an immediate offset in most cases,
2471 // so the presence of calls doesn't imply we need a distinct frame pointer.
2472 if (MFI.hasCalls() &&
2474 // All offsets are unsigned, so need to be addressed in the same direction
2475 // as stack growth.
2476
2477 // FIXME: This function is pretty broken, since it can be called before the
2478 // frame layout is determined or CSR spills are inserted.
2479 return MFI.getStackSize() != 0;
2480 }
2481
2482 return frameTriviallyRequiresSP(MFI) || MFI.isFrameAddressTaken() ||
2483 MF.getSubtarget<GCNSubtarget>().getRegisterInfo()->hasStackRealignment(
2484 MF) ||
2487}
2488
2490 const MachineFunction &MF) const {
2491 return MF.getInfo<SIMachineFunctionInfo>()->isDynamicVGPREnabled() &&
2494}
2495
2496// This is essentially a reduced version of hasFP for entry functions. Since the
2497// stack pointer is known 0 on entry to kernels, we never really need an FP
2498// register. We may need to initialize the stack pointer depending on the frame
2499// properties, which logically overlaps many of the cases where an ordinary
2500// function would require an FP.
2502 const MachineFunction &MF) const {
2503 // Callable functions always require a stack pointer reference.
2505 "only expected to call this for entry points functions");
2506
2507 const MachineFrameInfo &MFI = MF.getFrameInfo();
2508
2509 // Entry points ordinarily don't need to initialize SP. We have to set it up
2510 // for callees if there are any. Also note tail calls are only possible via
2511 // the `llvm.amdgcn.cs.chain` intrinsic.
2512 if (MFI.hasCalls() || MFI.hasTailCall())
2513 return true;
2514
2515 // We still need to initialize the SP if we're doing anything weird that
2516 // references the SP, like variable sized stack objects.
2517 return frameTriviallyRequiresSP(MFI);
2518}
2519
2522 const DebugLoc &DL,
2523 const MCCFIInstruction &CFIInst,
2524 MachineInstr::MIFlag Flag) const {
2525 MachineFunction &MF = *MBB.getParent();
2526 const SIInstrInfo *TII = MF.getSubtarget<GCNSubtarget>().getInstrInfo();
2527 return BuildMI(MBB, MBBI, DL, TII->get(TargetOpcode::CFI_INSTRUCTION))
2528 .addCFIIndex(MF.addFrameInst(CFIInst))
2529 .setMIFlag(Flag);
2530}
2531
2534 const DebugLoc &DL, const MCRegister Reg, const MCRegister RegCopy) const {
2535 MachineFunction &MF = *MBB.getParent();
2536 const MCRegisterInfo &MCRI = *MF.getContext().getRegisterInfo();
2537 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2538
2539 unsigned MaskReg = MCRI.getDwarfRegNum(
2540 ST.isWave32() ? AMDGPU::EXEC_LO : AMDGPU::EXEC, false);
2542 nullptr, MCRI.getDwarfRegNum(Reg, false),
2543 MCRI.getDwarfRegNum(RegCopy, false), VGPRLaneBitSize, MaskReg,
2544 ST.getWavefrontSize());
2545 return buildCFI(MBB, MBBI, DL, std::move(CFIInst));
2546}
2547
2550 const DebugLoc &DL, const MCRegister SGPR, const MCRegister VGPR,
2551 const int Lane) const {
2552 const MachineFunction &MF = *MBB.getParent();
2553 const MCRegisterInfo &MCRI = *MF.getContext().getRegisterInfo();
2554
2555 int DwarfSGPR = MCRI.getDwarfRegNum(SGPR, false);
2556 int DwarfVGPR = MCRI.getDwarfRegNum(VGPR, false);
2557 assert(DwarfSGPR != -1 && DwarfVGPR != -1);
2558 assert(Lane != -1 && "Expected a lane to be present");
2559
2560 // Build a CFI instruction that represents a SGPR spilled to a single lane of
2561 // a VGPR.
2563 unsigned(Lane), VGPRLaneBitSize};
2564 auto CFIInst =
2565 MCCFIInstruction::createLLVMVectorRegisters(nullptr, DwarfSGPR, {VR});
2566 return buildCFI(MBB, MBBI, DL, std::move(CFIInst));
2567}
2568
2571 const DebugLoc &DL, MCRegister SGPR,
2572 ArrayRef<SIRegisterInfo::SpilledReg> VGPRSpills) const {
2573 if (VGPRSpills.size() == 1u)
2574 return buildCFIForSGPRToVGPRSpill(MBB, MBBI, DL, SGPR, VGPRSpills[0].VGPR,
2575 VGPRSpills[0].Lane);
2576 const MachineFunction &MF = *MBB.getParent();
2577 const MCRegisterInfo &MCRI = *MF.getContext().getRegisterInfo();
2578
2579 int DwarfSGPR = MCRI.getDwarfRegNum(SGPR, false);
2580 assert(DwarfSGPR != -1);
2581
2582 // Build a CFI instruction that represents a SGPR spilled to multiple lanes of
2583 // multiple VGPRs.
2584
2586 for (SIRegisterInfo::SpilledReg Spill : VGPRSpills) {
2587 int DwarfVGPR = MCRI.getDwarfRegNum(Spill.VGPR, false);
2588 assert(DwarfVGPR != -1);
2589 assert(Spill.hasLane() && "Expected a lane to be present");
2590 VGPRs.push_back(
2591 {unsigned(DwarfVGPR), unsigned(Spill.Lane), VGPRLaneBitSize});
2592 }
2593
2594 auto CFIInst = MCCFIInstruction::createLLVMVectorRegisters(nullptr, DwarfSGPR,
2595 std::move(VGPRs));
2596 return buildCFI(MBB, MBBI, DL, std::move(CFIInst));
2597}
2598
2601 const DebugLoc &DL, MCRegister SGPR, int64_t Offset) const {
2602 MachineFunction &MF = *MBB.getParent();
2603 const MCRegisterInfo &MCRI = *MF.getContext().getRegisterInfo();
2604 return buildCFI(MBB, MBBI, DL,
2606 nullptr, MCRI.getDwarfRegNum(SGPR, false), Offset));
2607}
2608
2611 const DebugLoc &DL, MCRegister VGPR, int64_t Offset) const {
2612 const MachineFunction &MF = *MBB.getParent();
2613 const MCRegisterInfo &MCRI = *MF.getContext().getRegisterInfo();
2614 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2615
2616 int DwarfVGPR = MCRI.getDwarfRegNum(VGPR, false);
2617 assert(DwarfVGPR != -1);
2618
2619 unsigned MaskReg = MCRI.getDwarfRegNum(
2620 ST.isWave32() ? AMDGPU::EXEC_LO : AMDGPU::EXEC, false);
2622 nullptr, DwarfVGPR, VGPRLaneBitSize, MaskReg, ST.getWavefrontSize(),
2623 Offset);
2624 return buildCFI(MBB, MBBI, DL, std::move(CFIInst));
2625}
2626
2629 const DebugLoc &DL, const MCRegister Reg, const MCRegister SGPRPair) const {
2630 const MachineFunction &MF = *MBB.getParent();
2631 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
2632 const SIRegisterInfo &TRI = *ST.getRegisterInfo();
2633
2634 MCRegister SGPR0 = TRI.getSubReg(SGPRPair, AMDGPU::sub0);
2635 MCRegister SGPR1 = TRI.getSubReg(SGPRPair, AMDGPU::sub1);
2636
2637 int DwarfReg = TRI.getDwarfRegNum(Reg, false);
2638 int DwarfSGPR0 = TRI.getDwarfRegNum(SGPR0, false);
2639 int DwarfSGPR1 = TRI.getDwarfRegNum(SGPR1, false);
2640 assert(DwarfReg != -1 && DwarfSGPR0 != -1 && DwarfSGPR1 != -1);
2641
2643 nullptr, DwarfReg, DwarfSGPR0, SGPRBitSize, DwarfSGPR1, SGPRBitSize);
2644 return buildCFI(MBB, MBBI, DL, std::move(CFIInst));
2645}
2646
2649 const DebugLoc &DL, MCRegister Reg) const {
2650 const MachineFunction &MF = *MBB.getParent();
2651 const MCRegisterInfo &MCRI = *MF.getContext().getRegisterInfo();
2652 int DwarfReg = MCRI.getDwarfRegNum(Reg, /*isEH=*/false);
2653 auto CFIInst = MCCFIInstruction::createSameValue(nullptr, DwarfReg);
2654 return buildCFI(MBB, MBBI, DL, std::move(CFIInst));
2655}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
Provides AMDGPU specific target descriptions.
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static const Function * getParent(const Value *V)
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains constants used for implementing Dwarf debug support.
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
A set of register units.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static constexpr MCPhysReg FPReg
static constexpr MCPhysReg SPReg
This file declares the machine register scavenger class.
static void buildEpilogRestore(const GCNSubtarget &ST, const SIRegisterInfo &TRI, const SIMachineFunctionInfo &FuncInfo, LiveRegUnits &LiveUnits, MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register SpillReg, int FI, Register FrameReg, int64_t DwordOff=0)
static cl::opt< bool > EnableSpillVGPRToAGPR("amdgpu-spill-vgpr-to-agpr", cl::desc("Enable spilling VGPRs to AGPRs"), cl::ReallyHidden, cl::init(true))
static void getVGPRSpillLaneOrTempRegister(MachineFunction &MF, LiveRegUnits &LiveUnits, Register SGPR, const TargetRegisterClass &RC=AMDGPU::SReg_32_XM0_XEXECRegClass, bool IncludeScratchCopy=true)
Query target location for spilling SGPRs IncludeScratchCopy : Also look for free scratch SGPRs.
static void buildGitPtr(MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, const SIInstrInfo *TII, Register TargetReg)
static bool allStackObjectsAreDead(const MachineFrameInfo &MFI)
static MCCFIInstruction createScaledCFAInPrivateWave(const GCNSubtarget &ST, int64_t DwarfStackPtrReg)
static void buildPrologSpill(const GCNSubtarget &ST, const SIRegisterInfo &TRI, const SIMachineFunctionInfo &FuncInfo, LiveRegUnits &LiveUnits, MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator I, const DebugLoc &DL, Register SpillReg, int FI, Register FrameReg, int64_t DwordOff=0)
static Register buildScratchExecCopy(LiveRegUnits &LiveUnits, MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, bool IsProlog, bool EnableInactiveLanes)
static void encodeDwarfRegisterLocation(int DwarfReg, raw_ostream &OS)
static constexpr unsigned SGPRBitSize
static bool frameTriviallyRequiresSP(const MachineFrameInfo &MFI)
Returns true if the frame will require a reference to the stack pointer.
static SparseBitVector buildLiveInRoots(const MachineBasicBlock &MBB, const SIRegisterInfo &TRI)
Return the set of all root registers of regunits live-in to MBB.
static void initLiveUnits(LiveRegUnits &LiveUnits, const SIRegisterInfo &TRI, const SIMachineFunctionInfo *FuncInfo, MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, bool IsProlog)
static constexpr unsigned VGPRLaneBitSize
static bool allSGPRSpillsAreDead(const MachineFunction &MF)
static MCRegister findScratchNonCalleeSaveRegister(MachineRegisterInfo &MRI, LiveRegUnits &LiveUnits, const TargetRegisterClass &RC, bool Unused=false)
static MCRegister findUnusedRegister(MachineRegisterInfo &MRI, const LiveRegUnits &LiveUnits, const TargetRegisterClass &RC)
static constexpr unsigned SGPRByteSize
static void assignSlotsUsingVGPRBlocks(MachineFunction &MF, const GCNSubtarget &ST, std::vector< CalleeSavedInfo > &CSI)
static bool isAnyRootLiveIn(const SparseBitVector<> &LiveInRoots, const SIRegisterInfo &TRI, MCRegister Reg)
Returns true iff any root of Reg is in LiveInRoots (see buildLiveInRoots).
static unsigned getScratchScaleFactor(const GCNSubtarget &ST)
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
#define LLVM_DEBUG(...)
Definition Debug.h:119
static const int BlockSize
Definition TarWriter.cpp:33
static const LaneMaskConstants & get(const GCNSubtarget &ST)
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
ArrayRef< T > slice(size_t N, size_t M) const
slice(n, m) - Chop off the first N elements of the array, and keep M elements in the array.
Definition ArrayRef.h:185
bool test(unsigned Idx) const
Returns true if bit Idx is set.
Definition BitVector.h:482
BitVector & reset()
Reset all bits in the bitvector.
Definition BitVector.h:409
void clearBitsNotInMask(const uint32_t *Mask, unsigned MaskWords=~0u)
Clear a bit in this vector for every '0' bit in Mask.
Definition BitVector.h:760
BitVector & set()
Set all bits in the bitvector.
Definition BitVector.h:366
bool any() const
Returns true if any bit is set.
Definition BitVector.h:189
void clearBitsInMask(const uint32_t *Mask, unsigned MaskWords=~0u)
Clear any bits in this vector that are set in Mask.
Definition BitVector.h:748
iterator_range< const_set_bits_iterator > set_bits() const
Definition BitVector.h:159
bool empty() const
Returns whether there are no bits in this bitvector.
Definition BitVector.h:175
The CalleeSavedInfo class tracks the information need to locate where a callee saved register is in t...
MCRegister getReg() const
MCRegister getDstReg() const
A debug info location.
Definition DebugLoc.h:126
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:272
const HexagonRegisterInfo & getRegisterInfo() const
A set of register units used to track register liveness.
bool available(MCRegister Reg) const
Returns true if no part of physical register Reg is live.
void init(const TargetRegisterInfo &TRI)
Initialize and clear the set.
void addReg(MCRegister Reg)
Adds register units covered by physical register Reg.
LLVM_ABI void stepBackward(const MachineInstr &MI)
Updates liveness when stepping backwards over the instruction MI.
LLVM_ABI void addLiveOuts(const MachineBasicBlock &MBB)
Adds registers living out of block MBB.
void removeReg(MCRegister Reg)
Removes all register units covered by physical register Reg.
bool empty() const
Returns true if the set is empty.
LLVM_ABI void addLiveIns(const MachineBasicBlock &MBB)
Adds registers living into block MBB.
static MCCFIInstruction createLLVMVectorOffset(MCSymbol *L, unsigned Register, unsigned RegisterSizeInBits, unsigned MaskRegister, unsigned MaskRegisterSizeInBits, int64_t Offset, SMLoc Loc={})
.cfi_llvm_vector_offset Previous value of Register is saved at Offset from CFA.
Definition MCDwarf.h:797
static MCCFIInstruction createUndefined(MCSymbol *L, unsigned Register, SMLoc Loc={})
.cfi_undefined From now on the previous value of Register can't be restored anymore.
Definition MCDwarf.h:732
static MCCFIInstruction createLLVMVectorRegisters(MCSymbol *L, unsigned Register, ArrayRef< VectorRegisterWithLane > VectorRegisters, SMLoc Loc={})
.cfi_llvm_vector_registers Previous value of Register is saved in lanes of vector registers.
Definition MCDwarf.h:787
static MCCFIInstruction createLLVMVectorRegisterMask(MCSymbol *L, unsigned Register, unsigned SpillRegister, unsigned SpillRegisterLaneSizeInBits, unsigned MaskRegister, unsigned MaskRegisterSizeInBits, SMLoc Loc={})
.cfi_llvm_vector_register_mask Previous value of Register is saved in SpillRegister,...
Definition MCDwarf.h:808
static MCCFIInstruction createRegister(MCSymbol *L, unsigned Register1, unsigned Register2, SMLoc Loc={})
.cfi_register Previous value of Register1 is saved in register Register2.
Definition MCDwarf.h:685
static MCCFIInstruction createOffset(MCSymbol *L, unsigned Register, int64_t Offset, SMLoc Loc={})
.cfi_offset Previous value of Register is saved at offset Offset from CFA.
Definition MCDwarf.h:670
static MCCFIInstruction createLLVMRegisterPair(MCSymbol *L, unsigned Register, unsigned R1, unsigned R1SizeInBits, unsigned R2, unsigned R2SizeInBits, SMLoc Loc={})
.cfi_llvm_register_pair Previous value of Register is saved in R1:R2.
Definition MCDwarf.h:777
static MCCFIInstruction createEscape(MCSymbol *L, StringRef Vals, SMLoc Loc={}, StringRef Comment="")
.cfi_escape Allows the user to add arbitrary bytes to the unwind info.
Definition MCDwarf.h:756
static MCCFIInstruction createSameValue(MCSymbol *L, unsigned Register, SMLoc Loc={})
.cfi_same_value Current value of Register is the same as in the previous frame.
Definition MCDwarf.h:739
const MCRegisterInfo * getRegisterInfo() const
Definition MCContext.h:411
Describe properties that are true of each instruction in the target description file.
bool isValid() const
Returns true if this iterator is not yet at the end.
MCRegUnitMaskIterator enumerates a list of register units and their associated lane masks for Reg.
MCRegUnitRootIterator enumerates the root registers of a register unit.
bool isValid() const
Check if the iterator is at the end of the list.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
virtual int64_t getDwarfRegNum(MCRegister Reg, bool isEH) const
Map a target register to an equivalent dwarf register number.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition MCRegister.h:72
void addLiveIn(MCRegister PhysReg, LaneBitmask LaneMask=LaneBitmask::getAll())
Adds the specified register as a live in.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
MachineInstrBundleIterator< MachineInstr > iterator
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool hasVarSizedObjects() const
This method may be called any time after instruction selection is complete to determine if the stack ...
uint64_t getStackSize() const
Return the number of bytes that must be allocated to hold all of the fixed size frame objects.
bool hasCalls() const
Return true if the current function has any function calls.
bool isFrameAddressTaken() const
This method may be called any time after instruction selection is complete to determine if there is a...
Align getMaxAlign() const
Return alignment of this function's frame.
bool hasPatchPoint() const
This method may be called any time after instruction selection is complete to determine if there is a...
bool hasTailCall() const
Returns true if the function contains a tail call.
bool hasStackMap() const
This method may be called any time after instruction selection is complete to determine if there is a...
LLVM_ABI int CreateSpillStackObject(uint64_t Size, Align Alignment, TargetStackID::Value StackID=TargetStackID::Default)
Create a new statically sized stack object that represents a spill slot, returning a nonnegative iden...
void RemoveStackObject(int ObjectIdx)
Remove or mark dead a statically sized stack object.
int getObjectIndexEnd() const
Return one past the maximum frame object index.
uint8_t getStackID(int ObjectIdx) const
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
int getObjectIndexBegin() const
Return the minimum frame object index.
bool isDeadObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a dead object.
unsigned addFrameInst(const MCCFIInstruction &Inst)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
bool needsFrameMoves() const
True if this function needs frame moves for debug or exceptions.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineBasicBlock & front() const
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & addExternalSymbol(const char *FnName, unsigned TargetFlags=0) const
const MachineInstrBuilder & addCFIIndex(unsigned CFIIndex) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & setMIFlag(MachineInstr::MIFlag Flag) const
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
mop_range operands()
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
void setIsDead(bool Val=true)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
bool isReserved(MCRegister PhysReg) const
isReserved - Returns true when PhysReg is a reserved register.
bool isAllocatable(MCRegister PhysReg) const
isAllocatable - Returns true when PhysReg belongs to an allocatable register class and it hasn't been...
LLVM_ABI const MCPhysReg * getCalleeSavedRegs() const
Returns list of callee saved registers.
void reserveReg(MCRegister PhysReg, const TargetRegisterInfo *TRI)
reserveReg – Mark a register as reserved so checks like isAllocatable will not suggest using it.
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
LLVM_ABI bool isPhysRegModified(MCRegister PhysReg, bool SkipNoReturnDef=false) const
Return true if the specified register is modified in this function.
LLVM_ABI bool isPhysRegUsed(MCRegister PhysReg, bool SkipRegMaskTest=false) const
Return true if the specified register is modified or read in this function.
Represent a mutable reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:294
PrologEpilogSGPRSpillBuilder(Register Reg, const PrologEpilogSGPRSaveRestoreInfo SI, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const DebugLoc &DL, const SIInstrInfo *TII, const SIRegisterInfo &TRI, LiveRegUnits &LiveUnits, Register FrameReg, bool IsFramePtrPrologSpill=false)
Wrapper class representing virtual and physical registers.
Definition Register.h:20
MCRegister asMCReg() const
Utility to check-convert this value to a MCRegister.
Definition Register.h:107
void determinePrologEpilogSGPRSaves(MachineFunction &MF, BitVector &SavedRegs, bool NeedExecCopyReservedReg) const
MachineInstr * buildCFIForSGPRToVMEMSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister SGPR, int64_t Offset) const
Create a CFI index describing a spill of a SGPR to VMEM and build a MachineInstr around it.
void emitCSRSpillRestores(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, LiveRegUnits &LiveUnits, Register FrameReg, Register FramePtrRegScratchCopy) const
StackOffset getFrameIndexReference(const MachineFunction &MF, int FI, Register &FrameReg) const override
getFrameIndexReference - This method should return the base register and offset used to reference a f...
void processFunctionBeforeFrameFinalized(MachineFunction &MF, RegScavenger *RS=nullptr) const override
processFunctionBeforeFrameFinalized - This method is called immediately before the specified function...
bool mayReserveScratchForCWSR(const MachineFunction &MF) const
bool allocateScavengingFrameIndexesNearIncomingSP(const MachineFunction &MF) const override
Control the placement of special register scavenging spill slots when allocating a stack frame.
bool requiresStackPointerReference(const MachineFunction &MF) const
void emitEntryFunctionPrologue(MachineFunction &MF, MachineBasicBlock &MBB) const
void determineCalleeSaves(MachineFunction &MF, BitVector &SavedRegs, RegScavenger *RS=nullptr) const override
This method determines which of the registers reported by TargetRegisterInfo::getCalleeSavedRegs() sh...
bool hasFPImpl(const MachineFunction &MF) const override
bool assignCalleeSavedSpillSlotsImpl(MachineFunction &MF, const TargetRegisterInfo *TRI, std::vector< CalleeSavedInfo > &CSI) const
MachineInstr * buildCFIForVRegToVRegSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCRegister Reg, const MCRegister RegCopy) const
Create a CFI index describing a spill of the VGPR/AGPR Reg to another VGPR/AGPR RegCopy and build a M...
bool spillCalleeSavedRegisters(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, ArrayRef< CalleeSavedInfo > CSI, const TargetRegisterInfo *TRI) const override
spillCalleeSavedRegisters - Issues instruction(s) to spill all callee saved registers and returns tru...
MachineInstr * buildCFIForRegToSGPRPairSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister Reg, MCRegister SGPRPair) const
MachineInstr * buildCFIForVGPRToVMEMSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister VGPR, int64_t Offset) const
Create a CFI index describing a spill of a VGPR to VMEM and build a MachineInstr around it.
MachineInstr * buildCFIForSGPRToVGPRSpill(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCRegister SGPR, const MCRegister VGPR, const int Lane) const
Create a CFI index describing a spill of an SGPR to a single lane of a VGPR and build a MachineInstr ...
bool assignCalleeSavedSpillSlots(MachineFunction &MF, const TargetRegisterInfo *TRI, std::vector< CalleeSavedInfo > &CSI) const override
assignCalleeSavedSpillSlots - Allows target to override spill slot assignment logic.
void determineCalleeSavesSGPR(MachineFunction &MF, BitVector &SavedRegs, RegScavenger *RS=nullptr) const
void emitEpilogue(MachineFunction &MF, MachineBasicBlock &MBB) const override
MachineInstr * buildCFIForSameValue(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, MCRegister Reg) const
MachineInstr * buildCFI(MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, const MCCFIInstruction &CFIInst, MachineInstr::MIFlag flag=MachineInstr::FrameSetup) const
Create a CFI index for CFIInst and build a MachineInstr around it.
void emitCSRSpillStores(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MBBI, const DebugLoc &DL, LiveRegUnits &LiveUnits, Register FrameReg, Register FramePtrRegScratchCopy, const bool NeedsFrameMoves) const
void processFunctionBeforeFrameIndicesReplaced(MachineFunction &MF, RegScavenger *RS=nullptr) const override
processFunctionBeforeFrameIndicesReplaced - This method is called immediately before MO_FrameIndex op...
bool isSupportedStackID(TargetStackID::Value ID) const override
void emitPrologue(MachineFunction &MF, MachineBasicBlock &MBB) const override
emitProlog/emitEpilog - These methods insert prolog and epilog code into the function.
MachineBasicBlock::iterator eliminateCallFramePseudoInstr(MachineFunction &MF, MachineBasicBlock &MBB, MachineBasicBlock::iterator MI) const override
This method is called during prolog/epilog code insertion to eliminate call frame setup and destroy p...
bool restoreCalleeSavedRegisters(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, MutableArrayRef< CalleeSavedInfo > CSI, const TargetRegisterInfo *TRI) const override
restoreCalleeSavedRegisters - Issues instruction(s) to restore all callee saved registers and returns...
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
ArrayRef< PrologEpilogSGPRSpill > getPrologEpilogSGPRSpills() const
const WWMSpillsMap & getWWMSpills() const
void getAllScratchSGPRCopyDstRegs(SmallVectorImpl< Register > &Regs) const
ArrayRef< MCPhysReg > getAGPRSpillVGPRs() const
void removePrologEpilogSGPRSpillEntry(Register Reg)
void shiftWwmVGPRsToLowestRange(MachineFunction &MF, SmallVectorImpl< Register > &WWMVGPRs, BitVector &SavedVGPRs)
void setMaskForVGPRBlockOps(Register RegisterBlock, uint32_t Mask)
GCNUserSGPRUsageInfo & getUserSGPRInfo()
void allocateWWMSpill(MachineFunction &MF, Register VGPR, uint64_t Size=4, Align Alignment=Align(4))
void setVGPRToAGPRSpillDead(int FrameIndex)
Register getScratchRSrcReg() const
Returns the physical register reserved for use as the resource descriptor for scratch accesses.
ArrayRef< MCPhysReg > getVGPRSpillAGPRs() const
int getScavengeFI(MachineFrameInfo &MFI, const SIRegisterInfo &TRI)
uint32_t getMaskForVGPRBlockOps(Register RegisterBlock) const
bool hasMaskForVGPRBlockOps(Register RegisterBlock) const
bool hasPrologEpilogSGPRSpillEntry(Register Reg) const
Register getGITPtrLoReg(const MachineFunction &MF) const
void setVGPRForAGPRCopy(Register NewVGPRForAGPRCopy)
bool allocateVGPRSpillToAGPR(MachineFunction &MF, int FI, bool isAGPRtoVGPR)
Reserve AGPRs or VGPRs to support spilling for FrameIndex FI.
void splitWWMSpillRegisters(MachineFunction &MF, SmallVectorImpl< std::pair< Register, int > > &CalleeSavedRegs, SmallVectorImpl< std::pair< Register, int > > &ScratchRegs) const
bool isWWMReservedRegister(Register Reg) const
ArrayRef< SIRegisterInfo::SpilledReg > getSGPRSpillToPhysicalVGPRLanes(int FrameIndex) const
bool allocateSGPRSpillToVGPRLane(MachineFunction &MF, int FI, bool SpillToPhysVGPRLane=false, bool IsPrologEpilog=false)
void setLongBranchReservedReg(Register Reg)
void setHasSpilledVGPRs(bool Spill=true)
bool removeDeadFrameIndices(MachineFrameInfo &MFI, bool ResetSGPRSpillStackIDs)
If ResetSGPRSpillStackIDs is true, reset the stack ID from sgpr-spill to the default stack.
void setScratchReservedForDynamicVGPRs(unsigned SizeInBytes)
MCRegister getPreloadedReg(AMDGPUFunctionArgInfo::PreloadedValue Value) const
bool checkIndexInPrologEpilogSGPRSpills(int FI) const
const ReservedRegSet & getWWMReservedRegs() const
const PrologEpilogSGPRSaveRestoreInfo & getPrologEpilogSGPRSaveRestoreInfo(Register Reg) const
void setIsStackRealigned(bool Realigned=true)
void addToPrologEpilogSGPRSpills(Register Reg, PrologEpilogSGPRSaveRestoreInfo SI)
Register getScratchSGPRCopyDstReg(Register Reg) const
Register getFrameRegister(const MachineFunction &MF) const override
Represents a location in source code.
Definition SMLoc.h:22
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
void set(unsigned Idx)
bool test(unsigned Idx) const
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
int64_t getFixed() const
Returns the fixed component of the stack.
Definition TypeSize.h:46
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool hasFP(const MachineFunction &MF) const
hasFP - Return true if the specified function should have a dedicated frame pointer register.
virtual bool hasReservedCallFrame(const MachineFunction &MF) const
hasReservedCallFrame - Under normal circumstances, when a frame pointer is not required,...
virtual void determineCalleeSaves(MachineFunction &MF, BitVector &SavedRegs, RegScavenger *RS=nullptr) const
This method determines which of the registers reported by TargetRegisterInfo::getCalleeSavedRegs() sh...
void restoreCalleeSavedRegister(MachineBasicBlock &MBB, MachineBasicBlock::iterator MI, const CalleeSavedInfo &CS, const TargetInstrInfo *TII, const TargetRegisterInfo *TRI) const
Align getStackAlign() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
TargetOptions Options
LLVM_ABI bool DisableFramePointerElim(const MachineFunction &MF) const
DisableFramePointerElim - This returns true if frame pointer elimination optimization should be disab...
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an SmallVector or SmallString.
StringRef str() const
Return a StringRef for the vector contents.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ PRIVATE_ADDRESS
Address space for private memory.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
unsigned getVGPRAllocGranule(const MCSubtargetInfo &STI, unsigned DynamicVGPRBlockSize, std::optional< bool > EnableWavefrontSize32)
uint64_t convertSMRDOffsetUnits(const MCSubtargetInfo &ST, uint64_t ByteOffset)
Convert ByteOffset to dwords if the subtarget uses dword SMRD immediate offsets.
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isCompute(CallingConv::ID CC)
LLVM_READNONE constexpr bool isChainCC(CallingConv::ID CC)
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
initializer< Ty > init(const Ty &Val)
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:578
UnaryFunction for_each(R &&Range, UnaryFunction F)
Provide wrappers to std::for_each which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1732
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
constexpr RegState getKillRegState(bool B)
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:633
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
void clearDebugInfoForSpillFIs(MachineFrameInfo &MFI, MachineBasicBlock &MBB, const BitVector &SpillFIs)
Replace frame index operands with null registers in debug value instructions for the specified spill ...
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
auto reverse(ContainerTy &&C)
Definition STLExtras.h:407
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1636
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
auto make_first_range(ContainerTy &&c)
Given a container of pairs, return a range over the first elements.
Definition STLExtras.h:1399
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1970
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
@ And
Bitwise or logical AND of integers.
@ Add
Sum of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
Definition STLExtras.h:2019
unsigned encodeULEB128(uint64_t Value, raw_ostream &OS, unsigned PadTo=0)
Utility function to encode a ULEB128 value to an output stream.
Definition LEB128.h:79
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.