26#include "llvm/IR/IntrinsicsAMDGPU.h"
29#define GET_GICOMBINER_DEPS
30#include "AMDGPUGenPreLegalizeGICombiner.inc"
31#undef GET_GICOMBINER_DEPS
33#define DEBUG_TYPE "amdgpu-postlegalizer-combiner"
39#define GET_GICOMBINER_TYPES
40#include "AMDGPUGenPostLegalizeGICombiner.inc"
41#undef GET_GICOMBINER_TYPES
43class AMDGPUPostLegalizerCombinerImpl :
public Combiner {
45 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig;
52 AMDGPUPostLegalizerCombinerImpl(
55 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
59 static const char *
getName() {
return "AMDGPUPostLegalizerCombinerImpl"; }
64 struct FMinFMaxLegacyInfo {
72 FMinFMaxLegacyInfo &Info)
const;
74 const FMinFMaxLegacyInfo &Info)
const;
84 struct CvtF32UByteMatchInfo {
90 CvtF32UByteMatchInfo &MatchInfo)
const;
92 const CvtF32UByteMatchInfo &MatchInfo)
const;
98 bool matchCombineSignExtendInReg(
99 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchInfo)
const;
100 void applyCombineSignExtendInReg(
101 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchInfo)
const;
108 bool matchCombine_s_mul_u64(
MachineInstr &
MI,
unsigned &NewOpcode)
const;
111#define GET_GICOMBINER_CLASS_MEMBERS
112#define AMDGPUSubtarget GCNSubtarget
113#include "AMDGPUGenPostLegalizeGICombiner.inc"
114#undef GET_GICOMBINER_CLASS_MEMBERS
115#undef AMDGPUSubtarget
118#define GET_GICOMBINER_IMPL
119#define AMDGPUSubtarget GCNSubtarget
120#include "AMDGPUGenPostLegalizeGICombiner.inc"
121#undef AMDGPUSubtarget
122#undef GET_GICOMBINER_IMPL
124AMDGPUPostLegalizerCombinerImpl::AMDGPUPostLegalizerCombinerImpl(
127 const AMDGPUPostLegalizerCombinerImplRuleConfig &RuleConfig,
129 :
Combiner(MF, CInfo, &VT, CSEInfo), RuleConfig(RuleConfig), STI(STI),
130 TII(*STI.getInstrInfo()),
131 Helper(Observer,
B,
false, &VT, MDT, LI, STI),
133#include
"AMDGPUGenPostLegalizeGICombiner.inc"
138bool AMDGPUPostLegalizerCombinerImpl::tryCombineAll(
MachineInstr &
MI)
const {
139 if (tryCombineAllImpl(
MI))
142 switch (
MI.getOpcode()) {
143 case TargetOpcode::G_SHL:
144 case TargetOpcode::G_LSHR:
145 case TargetOpcode::G_ASHR:
155bool AMDGPUPostLegalizerCombinerImpl::matchFMinFMaxLegacy(
156 MachineInstr &
MI, MachineInstr &FCmp, FMinFMaxLegacyInfo &Info)
const {
169 if ((
Info.LHS != True ||
Info.RHS != False) &&
170 (
Info.LHS != False ||
Info.RHS != True))
176 if (
Info.LHS != True)
191void AMDGPUPostLegalizerCombinerImpl::applySelectFCmpToFMinFMaxLegacy(
192 MachineInstr &
MI,
const FMinFMaxLegacyInfo &Info)
const {
194 : AMDGPU::G_AMDGPU_FMIN_LEGACY;
204 B.buildInstr(
Opc, {
MI.getOperand(0)}, {
X,
Y},
MI.getFlags());
206 MI.eraseFromParent();
209bool AMDGPUPostLegalizerCombinerImpl::matchUCharToFloat(
210 MachineInstr &
MI)
const {
217 LLT Ty = MRI.getType(DstReg);
220 unsigned SrcSize = MRI.getType(SrcReg).getSizeInBits();
221 assert(SrcSize == 16 || SrcSize == 32 || SrcSize == 64);
229void AMDGPUPostLegalizerCombinerImpl::applyUCharToFloat(
230 MachineInstr &
MI)
const {
235 LLT Ty = MRI.getType(DstReg);
236 LLT SrcTy = MRI.getType(SrcReg);
238 SrcReg =
B.buildAnyExtOrTrunc(
S32, SrcReg).getReg(0);
241 B.buildInstr(AMDGPU::G_AMDGPU_CVT_F32_UBYTE0, {DstReg}, {SrcReg},
244 auto Cvt0 =
B.buildInstr(AMDGPU::G_AMDGPU_CVT_F32_UBYTE0, {
S32}, {SrcReg},
246 B.buildFPTrunc(DstReg, Cvt0,
MI.getFlags());
249 MI.eraseFromParent();
252bool AMDGPUPostLegalizerCombinerImpl::matchFDivSqrtToRsqF16(
253 MachineInstr &
MI)
const {
255 return MRI.hasOneNonDBGUse(Sqrt);
258void AMDGPUPostLegalizerCombinerImpl::applyFDivSqrtToRsqF16(
262 LLT DstTy = MRI.getType(Dst);
263 uint32_t
Flags =
MI.getFlags();
264 Register RSQ =
B.buildIntrinsic(Intrinsic::amdgcn_rsq, {DstTy})
268 B.buildFMul(Dst, RSQ,
Y, Flags);
269 MI.eraseFromParent();
272bool AMDGPUPostLegalizerCombinerImpl::matchCvtF32UByteN(
273 MachineInstr &
MI, CvtF32UByteMatchInfo &MatchInfo)
const {
283 const unsigned Offset =
MI.getOpcode() - AMDGPU::G_AMDGPU_CVT_F32_UBYTE0;
285 unsigned ShiftOffset = 8 *
Offset;
287 ShiftOffset += ShiftAmt;
289 ShiftOffset -= ShiftAmt;
291 MatchInfo.CvtVal = Src0;
292 MatchInfo.ShiftOffset = ShiftOffset;
293 return ShiftOffset < 32 && ShiftOffset >= 8 && (ShiftOffset % 8) == 0;
300void AMDGPUPostLegalizerCombinerImpl::applyCvtF32UByteN(
301 MachineInstr &
MI,
const CvtF32UByteMatchInfo &MatchInfo)
const {
302 unsigned NewOpc = AMDGPU::G_AMDGPU_CVT_F32_UBYTE0 + MatchInfo.ShiftOffset / 8;
306 LLT SrcTy = MRI.getType(MatchInfo.CvtVal);
309 CvtSrc =
B.buildAnyExt(
S32, CvtSrc).getReg(0);
313 B.buildInstr(NewOpc, {
MI.getOperand(0)}, {CvtSrc},
MI.getFlags());
314 MI.eraseFromParent();
317bool AMDGPUPostLegalizerCombinerImpl::matchRemoveFcanonicalize(
318 MachineInstr &
MI)
const {
319 const SITargetLowering *TLI =
static_cast<const SITargetLowering *
>(
320 MF.getSubtarget().getTargetLowering());
330bool AMDGPUPostLegalizerCombinerImpl::matchCombineSignExtendInReg(
331 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchData)
const {
333 if (!MRI.hasOneNonDBGUse(LoadReg))
338 MachineInstr *LoadMI = MRI.getVRegDef(LoadReg);
339 int64_t Width =
MI.getOperand(2).getImm();
341 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
342 MatchData = {LoadMI, AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE};
344 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
345 MatchData = {LoadMI, AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT};
347 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_UBYTE:
348 MatchData = {LoadMI, AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SBYTE};
350 case AMDGPU::G_AMDGPU_S_BUFFER_LOAD_USHORT:
351 MatchData = {LoadMI, AMDGPU::G_AMDGPU_S_BUFFER_LOAD_SSHORT};
359void AMDGPUPostLegalizerCombinerImpl::applyCombineSignExtendInReg(
360 MachineInstr &
MI, std::pair<MachineInstr *, unsigned> &MatchData)
const {
361 auto [LoadMI, NewOpcode] = MatchData;
365 Register SignExtendInsnDst =
MI.getOperand(0).getReg();
368 MI.eraseFromParent();
371bool AMDGPUPostLegalizerCombinerImpl::matchCombine_s_mul_u64(
372 MachineInstr &
MI,
unsigned &NewOpcode)
const {
378 if (VT->getKnownBits(Src1).countMinLeadingZeros() >= 32 &&
379 VT->getKnownBits(Src0).countMinLeadingZeros() >= 32) {
380 NewOpcode = AMDGPU::G_AMDGPU_S_MUL_U64_U32;
384 if (VT->computeNumSignBits(Src1) >= 33 &&
385 VT->computeNumSignBits(Src0) >= 33) {
386 NewOpcode = AMDGPU::G_AMDGPU_S_MUL_I64_I32;
395class AMDGPUPostLegalizerCombiner :
public MachineFunctionPass {
399 AMDGPUPostLegalizerCombiner(
bool IsOptNone =
false);
401 StringRef getPassName()
const override {
402 return "AMDGPUPostLegalizerCombiner";
407 void getAnalysisUsage(AnalysisUsage &AU)
const override;
411 AMDGPUPostLegalizerCombinerImplRuleConfig RuleConfig;
415void AMDGPUPostLegalizerCombiner::getAnalysisUsage(AnalysisUsage &AU)
const {
418 AU.
addRequired<GISelValueTrackingAnalysisLegacy>();
426AMDGPUPostLegalizerCombiner::AMDGPUPostLegalizerCombiner(
bool IsOptNone)
427 : MachineFunctionPass(
ID), IsOptNone(IsOptNone) {
428 if (!RuleConfig.parseCommandLineOption())
432bool AMDGPUPostLegalizerCombiner::runOnMachineFunction(
MachineFunction &MF) {
444 &getAnalysis<GISelValueTrackingAnalysisLegacy>().get(MF);
447 : &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
450 LI, EnableOpt,
F.hasOptSize(),
F.hasMinSize());
452 CInfo.MaxIterations = 1;
455 CInfo.EnableFullDCE =
false;
456 AMDGPUPostLegalizerCombinerImpl Impl(MF, CInfo, *VT,
nullptr,
457 RuleConfig, ST, MDT, LI);
458 return Impl.combineMachineInstrs();
461char AMDGPUPostLegalizerCombiner::ID = 0;
463 "Combine AMDGPU machine instrs after legalization",
false,
467 "Combine AMDGPU machine instrs after legalization",
false,
471 return new AMDGPUPostLegalizerCombiner(IsOptNone);
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define GET_GICOMBINER_CONSTRUCTOR_INITS
This contains common combine transformations that may be used in a combine pass.
This file declares the targeting of the Machinelegalizer class for AMDGPU.
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This contains common combine transformations that may be used in a combine pass,or by the target else...
Option class for Targets to specify which operations are combined how and when.
This contains the base class for all Combiners generated by TableGen.
AMD GCN specific subclass of TargetSubtarget.
Provides analysis for querying information about KnownBits during GISel passes.
const HexagonInstrInfo * TII
Contains matchers for matching SSA Machine Instructions.
Promote Memory to Register
#define INITIALIZE_PASS_DEPENDENCY(depName)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
static StringRef getName(Value *V)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Target-Independent Code Generator Pass Configuration Options pass.
bool canIgnoreLegacyMinMaxTies(const MachineInstr &MI, Register LHS, Register RHS) const
fmin_legacy/fmax_legacy select s1 on NaN, and on a +0.0/-0.0 tie (s1 for min, s0 for max).
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
@ FCMP_ULT
1 1 0 0 True if unordered or less than
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Predicate getUnorderedPredicate() const
GISelValueTracking * getValueTracking() const
LLVM_ABI bool tryCombineShiftToUnmerge(MachineInstr &MI, unsigned TargetShiftAmount) const
FunctionPass class - This class is used to implement most global optimizations.
To use KnownBitsInfo analysis in a pass, KnownBitsInfo &Info = getAnalysis<GISelValueTrackingInfoAnal...
bool maskedValueIsZero(Register Val, const APInt &Mask)
constexpr bool isScalar() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI void setDesc(const MCInstrDesc &TID)
Replace the instruction descriptor (thus opcode) of the current instruction with a new one.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
Register getReg() const
getReg - Returns the register number.
Wrapper class representing virtual and physical registers.
bool isCanonicalized(SelectionDAG &DAG, SDValue Op, SDNodeFlags UserFlags={}, unsigned MaxDepth=5) const
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
operand_type_match m_Reg()
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
ConstantMatch< APInt > m_ICst(APInt &Cst)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_SHL, false > m_GShl(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_LSHR, false > m_GLShr(const LHS &L, const RHS &R)
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
LLVM_ABI void getSelectionDAGFallbackAnalysisUsage(AnalysisUsage &AU)
Modify analysis usage so it preserves passes required for the SelectionDAG fallback.
FunctionPass * createAMDGPUPostLegalizeCombiner(bool IsOptNone)
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
@ SinglePass
Enables Observer-based DCE and additional heuristics that retry combining defined and used instructio...