LLVM 24.0.0git
AMDGPUHWEvents.cpp
Go to the documentation of this file.
1//===- AMDGPUHWEvents.cpp ---------------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "AMDGPUHWEvents.h"
10#include "GCNSubtarget.h"
11#include "SIInstrInfo.h"
13#include "llvm/Support/Debug.h"
15
16namespace llvm {
17namespace AMDGPU {
18
19#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
20LLVM_DUMP_METHOD void HWEvents::dump() const { dbgs() << *this << "\n"; }
21#endif
22
24 const SIInstrInfo &TII) {
25 if (TII.isVALU(Inst, /*AllowLDSDMA=*/false)) {
26 // Core/Side-, DP-, XDL- and TRANS-MACC VALU instructions complete
27 // out-of-order with respect to each other, so each of these classes
28 // has its own event.
29
30 if (TII.isXDL(Inst))
31 return HWEvents::VGPR_XDL_READ | HWEvents::VGPR_XDL_WRITE;
32
33 if (TII.isTRANS(Inst))
34 return HWEvents::VGPR_TRANS_READ | HWEvents::VGPR_TRANS_WRITE;
35
37 return HWEvents::VGPR_DPMACC_READ | HWEvents::VGPR_DPMACC_WRITE;
38
39 return HWEvents::VGPR_CSMACC_READ | HWEvents::VGPR_CSMACC_WRITE;
40 }
41
42 // FLAT and LDS instructions may read their VGPR sources out-of-order
43 // with respect to each other and all other VMEM instructions, so
44 // each of these also has a separate event.
45
46 if (TII.isFLAT(Inst))
47 return HWEvents::VGPR_FLAT_READ;
48
49 if (TII.isDS(Inst))
50 return HWEvents::VGPR_LDS_READ;
51
52 if (TII.isVMEM(Inst) || TII.isVIMAGE(Inst) || TII.isVSAMPLE(Inst))
53 return HWEvents::VGPR_VMEM_READ;
54
55 // Otherwise, no hazard.
56 return HWEvents::NONE;
57}
58
60 const SIInstrInfo &TII) {
61 switch (Inst.getOpcode()) {
62 // FIXME: GLOBAL_INV needs to be tracked with xcnt too.
63 case AMDGPU::GLOBAL_INV:
64 case AMDGPU::BUFFER_INV:
65 return HWEvents::VMEM_INV_ACCESS; // tracked using loadcnt/vmcnt, but
66 // doesn't write VGPRs
67 case AMDGPU::GLOBAL_WB:
68 case AMDGPU::GLOBAL_WBINV:
69 return HWEvents::VMEM_WRITE_ACCESS; // tracked using storecnt
70 default:
71 break;
72 }
73
75 // LDS DMA loads are also stores, but on the LDS side. On the VMEM side
76 // these should use VM_CNT.
78 return HWEvents::VMEM_READ_ACCESS;
79
80 if (Inst.mayStore() &&
81 (!Inst.mayLoad() || SIInstrInfo::isAtomicNoRet(Inst))) {
82 if (TII.mayAccessScratch(Inst))
83 return HWEvents::SCRATCH_WRITE_ACCESS;
84 return HWEvents::VMEM_WRITE_ACCESS;
85 }
86
87 if (SIInstrInfo::isFLAT(Inst))
88 return HWEvents::VMEM_READ_ACCESS;
89
90 if (SIInstrInfo::isImage(Inst)) {
92 const AMDGPU::MIMGBaseOpcodeInfo *BaseInfo =
93 AMDGPU::getMIMGBaseOpcodeInfo(Info->BaseOpcode);
94
95 if (BaseInfo->BVH)
96 return HWEvents::VMEM_BVH_READ_ACCESS;
97
98 // We have to make an additional check for isVSAMPLE here since some
99 // instructions don't have a sampler, but are still classified as sampler
100 // instructions for the purposes of e.g. waitcnt.
101 if (BaseInfo->Sampler || BaseInfo->MSAA || SIInstrInfo::isVSAMPLE(Inst))
102 return HWEvents::VMEM_SAMPLER_READ_ACCESS;
103 }
104
105 return HWEvents::VMEM_READ_ACCESS;
106}
107
109 const GCNSubtarget &ST, const SIInstrInfo &TII,
110 bool TgSplit) {
111 if (TII.isDS(Inst) && TII.usesLGKM_CNT(Inst)) {
112 if (TII.isAlwaysGDS(Inst.getOpcode()) ||
113 TII.hasModifiersSet(Inst, AMDGPU::OpName::gds))
114 return HWEvents::GDS_ACCESS | HWEvents::GDS_GPR_LOCK;
115
116 return HWEvents::LDS_ACCESS;
117 }
118
119 if (TII.isFLAT(Inst)) {
121 return getSimplifiedVMEMEventsFor(Inst, TII);
122
123 assert(Inst.mayLoadOrStore());
124 HWEvents E = HWEvents::NONE;
125 if (TII.mayAccessVMEMThroughFlat(Inst)) {
126 if (ST.hasWaitXcnt())
127 E |= HWEvents::VMEM_GROUP;
129 }
130
131 if (TII.mayAccessLDSThroughFlat(Inst, TgSplit))
132 E |= HWEvents::LDS_ACCESS;
133
135 E |= HWEvents::ASYNC_ACCESS;
136
137 return E;
138 }
139
141 return HWEvents::TENSOR_ACCESS;
142
143 if (SIInstrInfo::isVMEM(Inst) &&
145 Inst.getOpcode() == AMDGPU::BUFFER_INV ||
146 Inst.getOpcode() == AMDGPU::BUFFER_WBL2)) {
147 // BUFFER_INV increments VM_CNT. BUFFER_WBL2 also needs tracking because an
148 // S_WAITCNT vmcnt(0) must follow it to ensure the writeback has completed.
150 if (ST.hasWaitXcnt())
151 E |= HWEvents::VMEM_GROUP;
152 if (ST.vmemWriteNeedsExpWaitcnt() &&
153 (Inst.mayStore() || SIInstrInfo::isAtomicRet(Inst)))
154 E |= HWEvents::VMW_GPR_LOCK;
155
156 return E;
157 }
158
159 if (TII.isSMRD(Inst)) {
160 if (ST.hasWaitXcnt())
161 return HWEvents::SMEM_GROUP | HWEvents::SMEM_ACCESS;
162 return HWEvents::SMEM_ACCESS;
163 }
164
165 if (SIInstrInfo::isLDSDIR(Inst)) {
166 return HWEvents::EXP_LDS_ACCESS;
167 }
168
169 if (SIInstrInfo::isEXP(Inst)) {
170 unsigned Imm = TII.getNamedOperand(Inst, AMDGPU::OpName::tgt)->getImm();
172 return HWEvents::EXP_PARAM_ACCESS;
174 return HWEvents::EXP_POS_ACCESS;
175 return HWEvents::EXP_GPR_LOCK;
176 }
177
179 return HWEvents::SCC_WRITE;
180 }
181
182 switch (Inst.getOpcode()) {
183 case AMDGPU::S_SENDMSG:
184 case AMDGPU::S_SENDMSG_RTN_B32:
185 case AMDGPU::S_SENDMSG_RTN_B64:
186 case AMDGPU::S_SENDMSGHALT:
187 return HWEvents::SQ_MESSAGE;
188 case AMDGPU::S_MEMTIME:
189 case AMDGPU::S_MEMREALTIME:
190 case AMDGPU::S_GET_BARRIER_STATE_M0:
191 case AMDGPU::S_GET_BARRIER_STATE_IMM:
192 return HWEvents::SMEM_ACCESS;
193 }
194
195 return HWEvents::NONE;
196}
197
199 bool IsExpertMode, bool TgSplit) {
200 const SIInstrInfo &TII = *ST.getInstrInfo();
201
202 if (IsExpertMode)
203 return getEventsForImpl(Inst, ST, TII, TgSplit) |
205 return getEventsForImpl(Inst, ST, TII, TgSplit);
206}
207} // namespace AMDGPU
208
210 ListSeparator LS(" | ");
211#define AMDGPU_HW_EVENT(E, V) \
212 if (Events & AMDGPU::HWEvents::E) \
213 OS << LS << #E << " ";
214#include "AMDGPUHWEvents.def"
215 return OS;
216}
217
218} // namespace llvm
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define LLVM_DUMP_METHOD
Mark debug helper function definitions like dump() that should not be stripped from debug builds.
Definition Compiler.h:678
AMD GCN specific subclass of TargetSubtarget.
const HexagonInstrInfo * TII
Interface definition for SIInstrInfo.
This file contains some functions that are useful when dealing with strings.
Bit mask of hardware events.
A helper class to return the specified delimiter string after the first invocation of operator String...
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
bool mayLoadOrStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read or modify memory.
bool mayLoad(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly read memory.
bool mayStore(QueryType Type=AnyInBundle) const
Return true if this instruction could possibly modify memory.
static bool isVMEM(const MachineInstr &MI)
static bool isEXP(const MachineInstr &MI)
static bool mayWriteLDSThroughDMA(const MachineInstr &MI)
static bool usesTENSOR_CNT(const MachineInstr &MI)
static bool isLDSDIR(const MachineInstr &MI)
static bool isVSAMPLE(const MachineInstr &MI)
static bool isAtomicRet(const MachineInstr &MI)
static bool isImage(const MachineInstr &MI)
static bool isGFX12CacheInvOrWBInst(unsigned Opc)
static bool isSBarrierSCCWrite(unsigned Opcode)
static bool usesASYNC_CNT(const MachineInstr &MI)
static bool isFLAT(const MachineInstr &MI)
static bool isAtomicNoRet(const MachineInstr &MI)
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isDPMACCInstruction(unsigned Opc)
static HWEvents getExpertSchedulingEventType(const MachineInstr &Inst, const SIInstrInfo &TII)
HWEvents getSimplifiedVMEMEventsFor(const MachineInstr &Inst, const SIInstrInfo &TII)
HWEvents getEventsFor(const MachineInstr &Inst, const GCNSubtarget &ST, bool IsExpertMode, bool TgSplit)
bool getMUBUFIsBufferInv(unsigned Opc)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
static HWEvents getEventsForImpl(const MachineInstr &Inst, const GCNSubtarget &ST, const SIInstrInfo &TII, bool TgSplit)
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)