51 cl::desc(
"Use partial reduction intrinsics for "
52 "all supported unordered reductions."));
61 "should not try to widen irregular types");
76 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
85 if (!VPBB->getParent())
88 auto EndIter = Term ? Term->getIterator() : VPBB->end();
93 VPValue *VPV = Ingredient.getVPSingleValue();
114 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
116 nullptr , IsConsecutive,
117 *VPI, Ingredient.getDebugLoc());
119 bool IsConsecutive = IsConsecutiveAccess(
120 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
122 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
123 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
126 Ingredient.operands(), *VPI,
127 Ingredient.getDebugLoc(),
GEP);
139 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
144 if (VectorID == Intrinsic::assume ||
145 VectorID == Intrinsic::lifetime_end ||
146 VectorID == Intrinsic::lifetime_start ||
147 VectorID == Intrinsic::sideeffect ||
148 VectorID == Intrinsic::pseudoprobe) {
153 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
154 VectorID != Intrinsic::pseudoprobe;
158 Ingredient.getDebugLoc());
161 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
162 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
166 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
170 *VPI, Ingredient.getDebugLoc());
174 "inductions must be created earlier");
183 "Only recpies with zero or one defined values expected");
184 Ingredient.eraseFromParent();
195 const Loop *L =
nullptr;
200 if (
A->getOpcode() != Instruction::Store ||
201 B->getOpcode() != Instruction::Store)
214 const APInt *Distance;
220 Type *TyA =
A->getOperand(0)->getScalarType();
221 uint64_t SizeA =
DL.getTypeStoreSize(TyA);
222 Type *TyB =
B->getOperand(0)->getScalarType();
223 uint64_t SizeB =
DL.getTypeStoreSize(TyB);
228 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
230 auto VFs =
B->getParent()->getPlan()->vectorFactors();
241 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
242 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
251 return ExcludeRecipes.contains(
Store) ||
252 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
265 std::optional<SinkStoreInfo> SinkInfo = {}) {
266 bool CheckReads = SinkInfo.has_value();
270 if (SinkInfo && SinkInfo->shouldSkip(R))
274 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
299template <
unsigned Opcode>
304 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
305 "Only Load and Store opcodes supported");
306 constexpr bool IsLoad = (Opcode == Instruction::Load);
309 RecipesByAddressAndType;
314 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
318 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
322 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
327 for (
auto &Group :
Groups) {
342 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
349 if (Candidate->getParent() == SinkTo ||
350 all_of(Candidate->operands(),
351 [](
VPValue *
Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
363 WorkList.
insert({SinkTo, Candidate});
375 for (
auto &Recipe : *VPBB)
377 InsertIfValidSinkCandidate(VPBB,
Op);
381 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
384 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
389 auto UsersOutsideSinkTo =
391 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
393 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
394 return !U->usesFirstLaneOnly(SinkCandidate);
397 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
399 if (NeedsDuplicating) {
403 if (
auto *SinkCandidateRepR =
408 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
409 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
413 Clone = SinkCandidate->
clone();
423 InsertIfValidSinkCandidate(SinkTo,
Op);
432 if (EntryBB->getNumSuccessors() != 2)
437 if (!Succ0 || !Succ1)
440 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
442 if (Succ0->getSingleSuccessor() == Succ1)
444 if (Succ1->getSingleSuccessor() == Succ0)
461 if (!Region1->isReplicator())
463 auto *MiddleBasicBlock =
465 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
470 if (!Region2 || !Region2->isReplicator())
473 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
474 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
475 if (!Mask1 || Mask1 != Mask2)
478 assert(Mask1 && Mask2 &&
"both region must have conditions");
484 if (TransformedRegions.
contains(Region1))
491 if (!Then1 || !Then2)
499 std::optional<BlockFrequency> Freq1 =
502 if (Freq1 && Freq2) {
526 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
532 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
533 Phi1ToMove.eraseFromParent();
536 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
550 TransformedRegions.
insert(Region1);
553 return !TransformedRegions.
empty();
561 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
562 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
563 auto *BlockInMask = PredRecipe->
getMask();
578 BOMRecipe->setExecutionFrequency(RecipeWithoutMask->getExecutionFrequency(),
580 RecipeWithoutMask->clearExecutionFrequency();
589 Region->setParent(ParentRegion);
595 RecipeWithoutMask->getDebugLoc());
596 Exiting->appendRecipe(PHIRecipe);
609 if (RepR->isPredicated())
628 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
640 if (!VPBB->getParent())
644 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
653 R.moveBefore(*PredVPBB, PredVPBB->
end());
655 auto *ParentRegion = VPBB->getParent();
656 if (ParentRegion && ParentRegion->getExiting() == VPBB)
657 ParentRegion->setExiting(PredVPBB);
661 return !WorkList.
empty();
668 bool ShouldSimplify =
true;
669 while (ShouldSimplify) {
685 if (!
IV ||
IV->getTruncInst())
700 for (
auto *U : FindMyCast->
users()) {
702 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
703 FoundUserCast = UserCast;
710 FindMyCast = FoundUserCast;
712 if (FindMyCast !=
IV)
736 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
738 PhiR->eraseFromParent();
794 Def->user_empty() || !Def->getUnderlyingValue() ||
795 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
808 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
810 Def->getUnderlyingInstr());
811 Clone->insertAfter(Def);
812 Def->replaceAllUsesWith(Clone);
813 Def->eraseFromParent();
828 PtrIV->replaceAllUsesWith(PtrAdd);
835 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
836 return U->usesScalars(WideIV);
845 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
848 Plan, ID.getKind(), ID.getInductionOpcode(),
850 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
851 WideIV->getDebugLoc(), Builder, WrapFlags);
854 if (!HasOnlyVectorVFs) {
856 "plans containing a scalar VF cannot also include scalable VFs");
857 WideIV->replaceAllUsesWith(Steps);
860 WideIV->replaceUsesWithIf(Steps,
861 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
863 return U.usesFirstLaneOnly(WideIV);
864 return U.usesScalars(WideIV);
880 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
885 if (!Def || Def->getNumOperands() != 2)
893 auto IsWideIVInc = [&]() {
894 auto &ID = WideIV->getInductionDescriptor();
897 VPValue *IVStep = WideIV->getStepValue();
898 switch (ID.getInductionOpcode()) {
899 case Instruction::Add:
901 case Instruction::FAdd:
903 case Instruction::FSub:
906 case Instruction::Sub: {
926 return IsWideIVInc() ? WideIV :
nullptr;
950 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
952 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
953 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
958 if (Incoming != WideIV) {
960 EndValue =
B.createAdd(EndValue, One,
DL);
965 VPValue *Start = WideIV->getStartValue();
966 VPValue *Step = WideIV->getStepValue();
967 EndValue =
B.createDerivedIV(
969 Start, EndValue, Step);
983 if (WideIntOrFp && WideIntOrFp->getTruncInst())
993 Start, VectorTC, Step);
1025 assert(EndValue &&
"Must have computed the end value up front");
1030 if (Incoming != WideIV)
1042 auto *Zero = Plan.
getZero(StepTy);
1043 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1048 return B.createNaryOp(
1049 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1051 : Instruction::FAdd,
1052 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1069 const SCEV *Start, *Step;
1087 VPValue *ExitCount = Builder.createOverflowingOp(
1090 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1099 VPBuilder VectorPHBuilder(VectorPH, VectorPH->getFirstNonPhi());
1109 EndValues[WideIV] = EndValue;
1119 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1120 R.eraseFromParent();
1129 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1131 if (PredVPBB == MiddleVPBB) {
1133 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1136 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1139 Plan, ExitIRI->getOperand(Idx), PSE);
1142 ExitIRI->setOperand(Idx, Escape);
1159 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1163 ExpR->replaceAllUsesWith(V->second);
1167 ExpR->eraseFromParent();
1174 bool CanCreateNewRecipe) {
1199 return Plan.
getZero(Def->getScalarType());
1214 if (CanCreateNewRecipe &&
1219 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1220 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1221 return Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z));
1226 return Def->getOperand(1);
1231 return Builder.createLogicalAnd(
X,
Y);
1241 if (CanCreateNewRecipe &&
1245 return Builder.createLogicalOr(Z,
Y);
1249 if (CanCreateNewRecipe &&
1251 return Builder.createNot(
C);
1255 Def->setOperand(0,
C);
1256 Def->setOperand(1,
Y);
1257 Def->setOperand(2,
X);
1262 if (CanCreateNewRecipe &&
1266 Y->getScalarType()->isIntegerTy(1))
1267 return Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z));
1271 if (CanCreateNewRecipe &&
1277 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
X,
Y,
1278 Def->getDebugLoc());
1305 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1309 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1310 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1311 RepR->getDebugLoc());
1312 Unmasked->insertBefore(RepR);
1326 bool CanCreateNewRecipe =
1333 Def->getScalarType() ==
A->getScalarType())
1337 Type *TruncTy = Def->getScalarType();
1338 Type *ATy =
A->getScalarType();
1339 if (TruncTy == ATy) {
1348 : Instruction::ZExt;
1351 if (
auto *UnderlyingExt = Z->getUnderlyingValue()) {
1353 Ext->setUnderlyingValue(UnderlyingExt);
1357 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
A, TruncTy);
1375 return Plan.
getZero(Def->getScalarType());
1381 return Builder.createSub(Plan.
getZero(
A->getScalarType()),
A,
1382 Def->getDebugLoc(),
"", NW);
1385 if (CanCreateNewRecipe &&
1393 return Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW);
1400 Def->getDebugLoc());
1407 MulR->hasNoSignedWrap() &&
1409 return Builder.createNaryOp(
1412 Def->getDebugLoc());
1417 return Builder.createNaryOp(
1433 return match(U, m_Not(m_Specific(Cmp))) ||
1434 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1436 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1443 R->setOperand(1,
Y);
1444 R->setOperand(2,
X);
1448 R->replaceAllUsesWith(Cmp);
1453 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1454 Cmp->setDebugLoc(Def->getDebugLoc());
1467 if (
Op->getNumUsers() > 1 ||
1471 }
else if (!UnpairedCmp) {
1472 UnpairedCmp =
Op->getDefiningRecipe();
1476 UnpairedCmp =
nullptr;
1483 if (NewOps.
size() < Def->getNumOperands()) {
1492 if (CanCreateNewRecipe &&
1503 A->getScalarType() == Def->getScalarType())
1508 Type *WideStepTy = Def->getScalarType();
1509 if (
X->getScalarType() != WideStepTy)
1510 X = Builder.createWidenCast(Instruction::Trunc,
X, WideStepTy);
1519 Def->getScalarType()->isIntegerTy(1)) {
1520 Def->setOperand(1, Plan.
getTrue());
1521 Def->setOperand(0,
Y);
1528 return Def->getOperand(0);
1534 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1550 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1556 return BuildVector->getOperand(Idx);
1565 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1566 return U.usesFirstLaneOnly(Def);
1576 "broadcast operand must be single-scalar");
1577 Def->setOperand(0, Z);
1582 Def->replaceUsesWithIf(
1583 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1588 if (Def->getNumOperands() == 1) {
1589 return Def->getOperand(0);
1593 return Phi->getOperand(0);
1599 if (Def->getNumOperands() == 1 &&
1624 return Builder.createNaryOp(Instruction::ExtractElement, {
A, LaneToExtract},
1625 Def->getDebugLoc());
1639 if (IVInc->getNumUsers() == 2) {
1644 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1645 Def->replaceAllUsesWith(IVInc);
1647 Inc->replaceAllUsesWith(Phi);
1648 Phi->setOperand(0,
Y);
1658 return VPR->getOperand(0);
1664 return Steps->getOperand(0);
1670 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1672 return PhiR && PhiR->isInLoop();
1692 Def->replaceAllUsesWith(New);
1693 Def->eraseFromParent();
1696 Def->eraseFromParent();
1715 R.getVPSingleValue()->replaceAllUsesWith(
X);
1731 while (!Worklist.
empty()) {
1740 R->replaceAllUsesWith(
1741 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1745static std::optional<Instruction::BinaryOps>
1748 case Intrinsic::masked_udiv:
1749 return Instruction::UDiv;
1750 case Intrinsic::masked_sdiv:
1751 return Instruction::SDiv;
1752 case Intrinsic::masked_urem:
1753 return Instruction::URem;
1754 case Intrinsic::masked_srem:
1755 return Instruction::SRem;
1772 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1776 if (RepR && RepR->getOpcode() == Instruction::Store &&
1779 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1780 true ,
nullptr , *RepR ,
1781 *RepR , RepR->getDebugLoc());
1782 Clone->insertBefore(RepOrWidenR);
1784 VPValue *ExtractOp = Clone->getOperand(0);
1790 Clone->setOperand(0, ExtractOp);
1791 RepR->eraseFromParent();
1803 VPValue *SafeDivisor = Builder.createSelect(
1804 IntrR->getOperand(2), IntrR->getOperand(1),
1806 VPValue *Clone = Builder.createNaryOp(
1807 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1810 IntrR->eraseFromParent();
1819 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1828 return !U->usesScalars(
Op);
1832 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1835 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1836 IntroducesBCastOf(Op)))
1840 bool LiveInNeedsBroadcast =
1841 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1842 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1843 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1850 RepOrWidenR->getUnderlyingInstr());
1851 Clone->insertBefore(RepOrWidenR);
1852 RepOrWidenR->replaceAllUsesWith(Clone);
1854 RepOrWidenR->eraseFromParent();
1890 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1891 UniqueValues.
insert(Blend->getIncomingValue(0));
1892 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1894 UniqueValues.
insert(Blend->getIncomingValue(
I));
1896 if (UniqueValues.
size() == 1) {
1897 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1898 Blend->eraseFromParent();
1902 if (Blend->isNormalized())
1908 unsigned StartIndex = 0;
1909 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1921 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1923 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1924 if (
I == StartIndex)
1926 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1927 OperandsWithMask.
push_back(Blend->getMask(
I));
1932 OperandsWithMask, *Blend, Blend->getDebugLoc());
1933 NewBlend->insertBefore(&R);
1935 VPValue *DeadMask = Blend->getMask(StartIndex);
1937 Blend->eraseFromParent();
1942 if (NewBlend->getNumOperands() == 3 &&
1944 VPValue *Inc0 = NewBlend->getOperand(0);
1945 VPValue *Inc1 = NewBlend->getOperand(1);
1946 VPValue *OldMask = NewBlend->getOperand(2);
1947 NewBlend->setOperand(0, Inc1);
1948 NewBlend->setOperand(1, Inc0);
1949 NewBlend->setOperand(2, NewMask);
1976 APInt MaxVal = AlignedTC - 1;
1979 unsigned NewBitWidth =
1985 bool MadeChange =
false;
2010 "canonical IV is not expected to have a truncation");
2015 NewWideIV->insertBefore(WideIV);
2022 Cmp->replaceAllUsesWith(
2023 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
2037 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
2039 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
2053 const SCEV *VectorTripCount =
2058 "Trip count SCEV must be computable");
2073 bool MadeChange =
false;
2081 for (
VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2090 Builder.setInsertPoint(Extract);
2093 Start = Builder.createAdd(
2098 Extract->eraseFromParent();
2113 auto *Term = &ExitingVPBB->
back();
2119 bool MatchedCanIVInc =
2125 if (MatchedCanIVInc ||
2133 const SCEV *VectorTripCount =
2139 "Trip count SCEV must be computable");
2158 Term->setOperand(1, Plan.
getTrue());
2163 {}, Term->getDebugLoc());
2165 Term->eraseFromParent();
2173 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2174 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2193 RecurKind RK = PhiR->getRecurrenceKind();
2200 RecWithFlags->dropPoisonGeneratingFlags();
2206struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2215 return GEP->getSourceElementType();
2218 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2219 [](
auto *
I) {
return I->getSourceElementType(); })
2220 .
Default([](
auto *) {
return nullptr; });
2224 static bool canHandle(
const VPSingleDefRecipe *Def) {
2233 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2234 C->second == Instruction::ExtractValue)))
2240 if (
Def->mayWriteToMemory())
2242 return !
Def->mayReadFromMemory() ||
2247 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2250 getGEPSourceElementType(Def),
Def->getScalarType(),
2253 if (RFlags->hasPredicate())
2256 return hash_combine(Result, SIVSteps->getInductionOpcode());
2265 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2266 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2269 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2271 !
equal(
L->operands(),
R->operands()))
2275 "must have valid opcode info for both recipes");
2277 if (LFlags->hasPredicate() &&
2278 LFlags->getPredicate() !=
2282 if (LSIV->getInductionOpcode() !=
2297 const VPRegionBlock *RegionL =
L->getRegion();
2298 const VPRegionBlock *RegionR =
R->getRegion();
2301 L->getParent() !=
R->getParent())
2303 return L->getScalarType() ==
R->getScalarType();
2322 if (R.mayWriteToMemory())
2325 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2328 auto [It, Inserted] =
2329 (IsLoad ? LoadCSEMap : CSEMap).try_emplace(Def, Def);
2334 if (!VPDT.
dominates(V->getParent(), VPBB))
2339 if (EarlierLoad->getAlign() <
Load->getAlign()) {
2346 EarlierLoad->intersect(*
Load);
2351 Def->replaceAllUsesWith(V);
2362 bool Sinking =
false) {
2391 "Expected vector prehader's successor to be the vector loop region");
2399 return !Op->isDefinedOutsideLoopRegions();
2402 R.moveBefore(*Preheader, Preheader->
end());
2422 assert(!RepR->isPredicated() &&
2423 "Expected prior transformation of predicated replicates to "
2424 "replicate regions");
2429 if (!RepR->isSingleScalar())
2433 if (RepR->getOpcode() == Instruction::Store &&
2434 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2439 assert((!R.mayWriteToMemory() ||
2440 (RepR && RepR->getOpcode() == Instruction::Store &&
2441 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2442 "The only recipes that may write to memory are expected to be "
2443 "stores with invariant pointer-operand");
2453 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2454 auto *UserR = cast<VPRecipeBase>(U);
2455 VPBasicBlock *Parent = UserR->getParent();
2457 if (SinkBB && SinkBB != Parent)
2462 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2463 Parent->getSinglePredecessor() != LoopRegion;
2473 "Defining block must dominate sink block");
2498 VPValue *ResultVPV = R.getVPSingleValue();
2500 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2501 if (!NewResSizeInBits)
2514 (void)OldResSizeInBits;
2522 VPW->dropPoisonGeneratingFlags();
2524 assert((OldResSizeInBits != NewResSizeInBits ||
2526 "Only ICmps should not need extending the result.");
2532 if (OldResSizeInBits != NewResSizeInBits) {
2534 Instruction::ZExt, ResultVPV, OldResTy);
2536 Ext->setOperand(0, ResultVPV);
2546 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2547 if (OpSizeInBits == NewResSizeInBits)
2549 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2550 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2556 Builder.setInsertPoint(&R);
2557 ProcessedIter->second =
2558 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2560 Op = ProcessedIter->second;
2564 NWR->insertBefore(&R);
2568 VPValue *Replacement = NWR->getVPSingleValue();
2569 if (OldResSizeInBits != NewResSizeInBits)
2575 R.eraseFromParent();
2581 std::optional<VPDominatorTree> VPDT;
2589 bool SimplifiedPhi =
false;
2599 assert(VPBB->getNumSuccessors() == 2 &&
2600 "Two successors expected for BranchOnCond");
2601 unsigned RemovedIdx;
2612 "There must be a single edge between VPBB and its successor");
2617 SimplifiedPhi =
true;
2621 if (!PhiR || PhiR->getNumIncoming() != 1)
2623 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2624 PhiR->eraseFromParent();
2629 VPBB->back().eraseFromParent();
2641 if (Reachable.contains(
B))
2652 for (
VPValue *Def : R.definedValues())
2653 Def->replaceAllUsesWith(&Tmp);
2654 R.eraseFromParent();
2658 return SimplifiedPhi;
2684 auto GetSimplifiedLiveInViaSCEV = [&](
VPValue *VPV) ->
VPValue * {
2693 if (
VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2694 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2705 "expected to run before loop regions are created");
2707 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U,
2714 return VPDT.
dominates(Header, R->getParent());
2718 Value *StrideV = Stride->getValue();
2719 const APInt *StrideConst;
2726 CanUseVersionedStride);
2740 CanUseVersionedStride);
2742 RewriteMap[StrideV] = StrideExpr;
2749 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2752 if (NewSCEV != ScevExpr) {
2754 ExpSCEV->replaceAllUsesWith(NewExp);
2765 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2770 while (!Worklist.
empty()) {
2773 if (!Visited.
insert(CurRec).second)
2795 RecWithFlags->isDisjoint()) {
2798 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2799 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2800 RecWithFlags->replaceAllUsesWith(New);
2801 RecWithFlags->eraseFromParent();
2804 RecWithFlags->dropPoisonGeneratingFlags();
2809 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2810 "found instruction with poison generating flags not covered by "
2811 "VPRecipeWithIRFlags");
2816 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2838 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2839 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2840 match(WidenRec->getMask(), m_UnlessHdrMask))
2841 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2843 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2844 if (AddrDef && InterleaveRec->getMask() &&
2845 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2846 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2856 const bool &EpilogueAllowed) {
2857 if (InterleaveGroups.empty())
2868 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2875 for (
const auto *IG : InterleaveGroups) {
2878 for (
auto *Member : IG->members())
2880 StartMember = Member;
2888 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2894 StoredValues.
push_back(StoreR->getStoredValue());
2901 bool NeedsMaskForGaps =
2902 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2903 (!StoredValues.
empty() && !IG->isFull());
2906 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2910 "Dead member in non-load group?");
2915 InsertPos->getAsRecipe()))
2916 InsertPos = MemberR;
2917 IRInsertPos = &InsertPos->getIngredient();
2927 VPValue *Addr = Start->getAddr();
2929 if (IG->getIndex(StartMember) != 0 ||
2937 assert(IG->getIndex(IRInsertPos) != 0 &&
2938 "index of insert position shouldn't be zero");
2942 IG->getIndex(IRInsertPos),
2946 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2952 if (IG->isReverse()) {
2955 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2956 ReversePtr->insertBefore(InsertPosR);
2960 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2962 VPIG->insertBefore(InsertPosR);
2965 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2968 if (!Member->getType()->isVoidTy()) {
2986static std::optional<VPValue *>
3039 VPValue *UncountableCondition =
nullptr;
3043 return std::nullopt;
3046 Worklist.
push_back(UncountableCondition);
3047 while (!Worklist.
empty()) {
3051 if (V->isDefinedOutsideLoopRegions())
3057 if (V->getNumUsers() > 1)
3058 return std::nullopt;
3070 return std::nullopt;
3074 return std::nullopt;
3082 return std::nullopt;
3087 if (Recipes.
empty() ||
3089 return std::nullopt;
3091 return UncountableCondition;
3147 for (
auto &Exit : Exits) {
3148 if (Exit.EarlyExitingVPBB == LatchVPBB)
3152 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3153 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3164 std::optional<VPValue *>
Cond =
3180 assert(
Load &&
"Couldn't find exactly one load");
3183 "Uncountable exit condition load is conditional.");
3197 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3221 while (InsertIt != HeaderVPBB->
end() &&
3223 erase(ConditionRecipes, &*InsertIt);
3226 for (
auto *Recipe :
reverse(ConditionRecipes))
3227 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3231 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3233 Type *IVScalarTy =
IV->getScalarType();
3239 "uncountable.exit.mask");
3244 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3246 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3256 "Expected BranchOnCond terminator for MiddleVPBB");
3267 auto Phis = ScalarPH->
phis();
3277 "Continuing from different IV");
3299 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3301 for (
auto [EarlyExitingVPBB, ExitBlock] :
3305 VPValue *CondOfEarlyExitingVPBB;
3306 [[maybe_unused]]
bool Matched =
3307 match(EarlyExitingVPBB->getTerminator(),
3309 assert(Matched &&
"Terminator must be BranchOnCond");
3313 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3314 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3316 TrueSucc == ExitBlock
3317 ? CondOfEarlyExitingVPBB
3318 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3324 "exit condition must dominate the latch");
3332 assert(!Exits.
empty() &&
"must have at least one early exit");
3339 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3342 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3348 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3349 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3351 Exits[
I].EarlyExitingVPBB) &&
3352 "RPO sort must place dominating exits before dominated ones");
3358 VPValue *Combined = Exits[0].CondToExit;
3371 "Unexpected terminator");
3372 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3373 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3374 LatchExitingBranch->eraseFromParent();
3377 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3378 LatchVPBB->clearSuccessors();
3383 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3384 MiddleVPBB->clearPredecessors();
3385 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3387 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3392 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3396 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3404 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3407 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3439 for (
auto [Exit, VectorEarlyExitVPBB] :
3440 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3441 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3453 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3454 VPValue *NewIncoming = IncomingVal;
3456 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3461 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3462 ExitIRI->addIncoming(NewIncoming);
3465 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3499 bool IsLastDispatch = (
I + 2 == Exits.
size());
3501 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3507 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3510 CurrentBB = FalseBB;
3525 VPValue *VecOp = Red->getVecOp();
3528 if (Red->isPartialReduction())
3532 auto IsExtendedRedValidAndClampRange =
3545 "getExtendedReductionCost only supports integer types");
3546 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3547 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3548 Red->getFastMathFlagsOrNone(),
CostKind);
3549 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3557 IsExtendedRedValidAndClampRange(
3578 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3579 Opcode != Instruction::FAdd)
3583 if (Red->isPartialReduction())
3589 auto IsMulAccValidAndClampRange =
3601 (Ext0->getOpcode() != Ext1->getOpcode() ||
3602 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3606 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3608 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3615 ExtCost += Ext0->computeCost(VF, Ctx);
3617 ExtCost += Ext1->computeCost(VF, Ctx);
3619 ExtCost += OuterExt->computeCost(VF, Ctx);
3621 return MulAccCost.
isValid() &&
3622 MulAccCost < ExtCost + MulCost + RedCost;
3627 VPValue *VecOp = Red->getVecOp();
3665 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3667 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3668 Mul->setOperand(1, ExtB);
3678 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3683 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3690 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3707 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3716 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3717 Ext0->getOpcode() == Ext1->getOpcode() &&
3718 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3720 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3721 *Ext0, *Ext0, Ext0->getDebugLoc());
3722 NewExt0->insertBefore(Ext0);
3727 Ext->getScalarType(),
nullptr, *Ext1,
3728 *Ext1, Ext1->getDebugLoc());
3731 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3732 NewMul->insertBefore(
Mul);
3733 Ext->replaceAllUsesWith(NewMul);
3734 Ext->eraseFromParent();
3735 Mul->eraseFromParent();
3749 if (Red->isPartialReduction())
3753 auto IP = std::next(Red->getIterator());
3754 auto *VPBB = Red->getParent();
3764 Red->replaceAllUsesWith(AbstractR);
3787 return CommonMetadata;
3790template <
unsigned Opcode>
3795 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3796 "Only Load and Store opcodes supported");
3797 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3804 for (
auto Recipes :
Groups) {
3805 if (Recipes.size() < 2)
3810 "Expected all recipes in group to have the same load-store type");
3817 VPValue *MaskI = RecipeI->getMask();
3823 bool HasComplementaryMask =
false;
3828 VPValue *MaskJ = RecipeJ->getMask();
3837 if (HasComplementaryMask) {
3838 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3848template <
typename InstType>
3866 for (
auto &Group :
Groups) {
3886 return R->isSingleScalar() == IsSingleScalar;
3888 "all members in group must agree on IsSingleScalar");
3893 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3894 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3896 UnpredicatedLoad->insertBefore(EarliestLoad);
3900 Load->replaceAllUsesWith(UnpredicatedLoad);
3901 Load->eraseFromParent();
3910 if (!StoreLoc || !StoreLoc->AATags.Scope)
3917 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3929 for (
auto &Group :
Groups) {
3942 VPValue *SelectedValue = Group[0]->getOperand(0);
3945 bool IsSingleScalar = Group[0]->isSingleScalar();
3946 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3947 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3948 "all members in group must agree on IsSingleScalar");
3949 VPValue *Mask = Group[
I]->getMask();
3951 SelectedValue = Builder.createSelect(
3954 Value->getScalarType()));
3962 StoreWithMinAlign->getUnderlyingInstr(),
3963 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3964 nullptr, *LastStore, CommonMetadata);
3965 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
3969 Store->eraseFromParent();
3984 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
3989 if (Member0Op == OpV)
3999 return !IsScalable && !W->getMask() && W->isConsecutive() &&
4002 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
4017 if (R->getScalarType() != WideMember0->getScalarType())
4019 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
4023 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
4026 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
4031 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
4032 const auto &[OpIdx, OpV] =
P;
4033 return !
canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
4044static std::optional<ElementCount>
4048 if (!InterleaveR || InterleaveR->
getMask())
4049 return std::nullopt;
4051 Type *GroupElementTy =
nullptr;
4055 return Op->getScalarType() == GroupElementTy;
4057 return std::nullopt;
4061 return Op->getScalarType() == GroupElementTy;
4063 return std::nullopt;
4067 if (IG->getFactor() != IG->getNumMembers())
4068 return std::nullopt;
4074 assert(
Size.isScalable() == VF.isScalable() &&
4075 "if Size is scalable, VF must be scalable and vice versa");
4076 return Size.getKnownMinValue();
4080 unsigned MinVal = VF.getKnownMinValue();
4082 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
4085 return std::nullopt;
4093 return RepR && RepR->isSingleScalar();
4107 if (V->isDefinedOutsideLoopRegions()) {
4110 return M->isDefinedOutsideLoopRegions() &&
4111 M->getScalarType() == V->getScalarType();
4113 "expected distinct loop-invariant values of matching scalar type");
4128 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4130 for (
VPValue *Member : Members)
4131 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4132 WideMember0->setOperand(
4141 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4143 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4144 *LoadGroup, LoadGroup->getDebugLoc());
4150 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4151 "must be a single scalar load");
4152 NarrowedOps.
insert(RepR);
4157 VPValue *PtrOp = WideLoad->getAddr();
4159 PtrOp = VecPtr->getOperand(0);
4164 nullptr, {}, *WideLoad);
4165 N->insertBefore(WideLoad);
4170std::unique_ptr<VPlan>
4190 "unexpected branch-on-count");
4193 std::optional<ElementCount> VFToOptimize;
4207 if (R.mayWriteToMemory() && !InterleaveR)
4213 return any_of(V->users(), [&](VPUser *U) {
4214 auto *UR = cast<VPRecipeBase>(U);
4215 return UR->getParent()->getParent() != VectorLoop;
4232 std::optional<ElementCount> NarrowedVF =
4234 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4236 VFToOptimize = NarrowedVF;
4239 if (InterleaveR->getStoredValues().empty())
4244 auto *Member0 = InterleaveR->getStoredValues()[0];
4254 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4257 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4258 return IR && IR->getInterleaveGroup()->isFull() &&
4259 IR->getVPValue(Op.index()) == Op.value();
4268 VFToOptimize->isScalable()))
4273 if (StoreGroups.empty())
4277 bool RequiresScalarEpilogue =
4288 std::unique_ptr<VPlan> NewPlan;
4290 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4291 Plan.
setVF(*VFToOptimize);
4292 NewPlan->removeVF(*VFToOptimize);
4299 for (
auto *StoreGroup : StoreGroups) {
4301 NarrowedOps, Preheader);
4307 StoreGroup->getDebugLoc());
4314 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4320 if (VFToOptimize->isScalable()) {
4323 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4331 materializeVectorTripCount(Plan, VectorPH,
false,
4332 RequiresScalarEpilogue, Step);
4337 removeDeadRecipes(Plan);
4340 "All VPVectorPointerRecipes should have been removed");
4360 "Cannot handle loops with uncountable early exits");
4367 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4374 if (
any_of(RecurSplice->users(),
4375 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4456 {},
"vector.recur.extract.for.phi");
4459 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4473 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4474 VPValue *InvariantCandidate = BinOp->getOperand(1);
4476 std::swap(WidenIVCandidate, InvariantCandidate);
4490 auto *ClonedOp = BinOp->
clone();
4491 if (ClonedOp->getOperand(0) == WidenIV) {
4492 ClonedOp->setOperand(0, ScalarIV);
4494 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4495 ClonedOp->setOperand(1, ScalarIV);
4509 return std::nullopt;
4514 return std::nullopt;
4526 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4527 bool UseMax) -> std::optional<APSInt> {
4529 for (
bool Signed : {
true,
false}) {
4538 return std::nullopt;
4546 PhiR->getRecurrenceKind()))
4555 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4569 !
match(FindLastSelect,
4578 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4583 "IVOfExpressionToSink not being an AddRec must imply "
4584 "FindLastExpression not being an AddRec.");
4593 bool UseMax = *StepDirection;
4594 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4595 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4602 if (IVOfExpressionToSink) {
4603 const SCEV *FindLastExpressionSCEV =
4605 if (std::optional<bool> NewUseMax =
4607 if (
auto NewSentinel =
4608 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4611 SentinelVal = *NewSentinel;
4612 UseSigned = NewSentinel->isSigned();
4613 UseMax = *NewUseMax;
4614 IVSCEV = FindLastExpressionSCEV;
4615 IVOfExpressionToSink =
nullptr;
4625 if (AR->hasNoSignedWrap())
4627 else if (AR->hasNoUnsignedWrap())
4637 VPValue *NewFindLastSelect = BackedgeVal;
4639 if (!SentinelVal || IVOfExpressionToSink) {
4642 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4643 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4644 if (
match(FindLastSelect,
4646 SelectCond = LoopBuilder.
createNot(SelectCond);
4653 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4656 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4665 VPIRFlags Flags(MinMaxKind,
false,
false,
4671 NewFindLastSelect, Flags, ExitDL);
4674 VPValue *VectorRegionExitingVal = ReducedIV;
4675 if (IVOfExpressionToSink)
4676 VectorRegionExitingVal =
4678 ReducedIV, IVOfExpressionToSink);
4681 VPValue *StartVPV = PhiR->getStartValue();
4688 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4698 AnyOfPhi->insertAfter(PhiR);
4705 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4718 PhiR->hasUsesOutsideReductionChain());
4719 NewPhiR->insertBefore(PhiR);
4720 PhiR->replaceAllUsesWith(NewPhiR);
4721 PhiR->eraseFromParent();
4728struct ReductionExtend {
4729 Type *SrcType =
nullptr;
4730 ExtendKind Kind = ExtendKind::PR_None;
4736struct ExtendedReductionOperand {
4740 ReductionExtend ExtendA, ExtendB;
4748struct VPPartialReductionChain {
4751 VPWidenRecipe *ReductionBinOp =
nullptr;
4753 ExtendedReductionOperand ExtendedOp;
4760 unsigned AccumulatorOpIdx;
4761 unsigned ScaleFactor;
4764 VPBlendRecipe *Blend =
nullptr;
4769static std::optional<unsigned>
4773 "Expected a non-normalized blend with two incoming values");
4779 return std::nullopt;
4780 return FirstIncomingHasOneUse ? 0 : 1;
4792 if (!
Op->hasOneUse() ||
4798 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4799 Op->getOperand(1), NarrowTy);
4801 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4810 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4812 assert(Ext->getOpcode() ==
4814 "Expected both the LHS and RHS extends to be the same");
4815 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4818 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4819 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4820 auto *
Max = Builder.insert(
4822 {FreezeX, FreezeY}, SrcTy));
4823 auto *Min = Builder.insert(
4825 {FreezeX, FreezeY}, SrcTy));
4826 auto *AbsDiff = Builder.insert(
4829 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4830 Op->getScalarType());
4842 if (!
Mul->hasOneUse() ||
4843 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4844 MulLHS->getOpcode() != MulRHS->getOpcode())
4847 auto *NewLHS = Builder.createWidenCast(
4848 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4849 auto *NewRHS = MulLHS == MulRHS
4851 : Builder.createWidenCast(MulRHS->getOpcode(),
4852 MulRHS->getOperand(0),
4853 Ext->getScalarType());
4854 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4855 Builder.insert(NewMul);
4856 Op->replaceAllUsesWith(NewMul);
4857 Op->eraseFromParent();
4858 Mul->eraseFromParent();
4867 VPValue *VecOp = Red->getVecOp();
4921static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4929 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4932 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4948 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4950 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4955 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
4967 Builder.insert(NegRecipe);
4968 ExtendedOp = NegRecipe;
4983 std::optional<unsigned> BlendReductionIdx =
4984 getBlendReductionUpdateValueIdx(Chain.Blend);
4985 assert(BlendReductionIdx &&
4987 "Expected blend to contain the reduction update");
5004 assert((!ExitValue || IsLastInChain) &&
5005 "if we found ExitValue, it must match RdxPhi's backedge value");
5016 PartialRed->insertBefore(WidenRecipe);
5026 E->insertBefore(WidenRecipe);
5027 PartialRed->replaceAllUsesWith(
E);
5040 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
5041 StartInst->setOperand(2, NewScaleFactor);
5049 VPValue *OldStartValue = StartInst->getOperand(0);
5050 StartInst->setOperand(0, StartInst->getOperand(1));
5054 assert(RdxResult &&
"Could not find reduction result");
5057 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
5058 : Instruction::BinaryOps::Sub;
5064 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
5070 const VPPartialReductionChain &Link,
5073 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5074 std::optional<unsigned> BinOpc = std::nullopt;
5076 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5077 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
5079 std::optional<llvm::FastMathFlags>
Flags;
5083 auto GetLinkOpcode = [&Link]() ->
unsigned {
5086 return Instruction::Add;
5088 return Instruction::FAdd;
5090 return Link.ReductionBinOp->
getOpcode();
5095 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5096 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5117static std::optional<ExtendedReductionOperand>
5120 "Op should be operand of UpdateR");
5128 if (
Op->hasOneUse() &&
5137 Type *RHSInputType =
Y->getScalarType();
5138 if (LHSInputType != RHSInputType ||
5139 LHSExt->getOpcode() != RHSExt->getOpcode())
5140 return std::nullopt;
5143 return ExtendedReductionOperand{
5145 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5149 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5152 VPValue *CastSource = CastRecipe->getOperand(0);
5153 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5163 return ExtendedReductionOperand{
5170 if (!
Op->hasOneUse())
5171 return std::nullopt;
5176 return std::nullopt;
5186 return std::nullopt;
5190 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5193 const APInt *RHSConst =
nullptr;
5199 return std::nullopt;
5203 if (Cast && OuterExtKind &&
5204 getPartialReductionExtendKind(Cast) != OuterExtKind)
5205 return std::nullopt;
5207 Type *RHSInputType = LHSInputType;
5208 ExtendKind RHSExtendKind = LHSExtendKind;
5211 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5214 return ExtendedReductionOperand{
5215 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5222static std::optional<SmallVector<VPPartialReductionChain>>
5229 return std::nullopt;
5239 VPValue *CurrentValue = ExitValue;
5240 while (CurrentValue != RedPhiR) {
5242 std::optional<unsigned> BlendReductionIdx;
5246 return std::nullopt;
5248 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5249 if (!BlendReductionIdx)
5250 return std::nullopt;
5257 return std::nullopt;
5264 std::optional<ExtendedReductionOperand> ExtendedOp =
5265 matchExtendedReductionOperand(UpdateR,
Op);
5267 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5269 return std::nullopt;
5277 return std::nullopt;
5279 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5282 return std::nullopt;
5284 VPPartialReductionChain Link(
5285 {UpdateR, *ExtendedOp, RK,
5290 CurrentValue = PrevValue;
5295 std::reverse(Chain.
begin(), Chain.
end());
5315 if (
auto Chains = getScaledReductions(RedPhiR))
5316 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
5327 for (
auto *Rdx : UnorderedReductions) {
5343 ? std::make_optional(Rdx->getFastMathFlagsOrNone())
5347 Backedge->getOpcode(), ScalarTy,
nullptr,
5349 std::nullopt, CostCtx.
CostKind, FMF);
5350 return PRCost <= CurrentCost;
5356 Rdx->getRecurrenceKind(), Rdx->getFastMathFlagsOrNone(),
5357 Backedge->getUnderlyingInstr(), Rdx, OtherOp,
nullptr,
5360 Partial->insertBefore(Backedge);
5361 Backedge->replaceAllUsesWith(Partial);
5362 Backedge->eraseFromParent();
5365 if (ChainsByPhi.
empty())
5373 for (
const auto &[
_, Chains] : ChainsByPhi)
5374 for (
const VPPartialReductionChain &Chain : Chains) {
5375 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5377 PartialReductionBlends.
insert(Chain.Blend);
5378 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5384 auto ExtendUsersValid = [&](
VPValue *Ext) {
5386 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5390 auto IsProfitablePartialReductionChainForVF =
5397 for (
const VPPartialReductionChain &Link : Chain) {
5398 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5399 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5403 PartialCost += LinkCost;
5404 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5406 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5407 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5410 RegularCost += Extend->computeCost(VF, CostCtx);
5412 return PartialCost.
isValid() && PartialCost < RegularCost;
5420 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5421 for (
const VPPartialReductionChain &Chain : Chains) {
5422 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5426 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5428 return PhiR == RedPhiR;
5432 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5434 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5440 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5449 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5450 return RepR && RepR->getOpcode() == Instruction::Store;
5461 return IsProfitablePartialReductionChainForVF(Chains, VF);
5467 for (
auto &[Phi, Chains] : ChainsByPhi)
5468 for (
const VPPartialReductionChain &Chain : Chains)
5469 transformToPartialReduction(Chain, Plan, Phi);
5484 if (VPI && VPI->getUnderlyingValue() &&
5495 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5498 if (!ProcessVPInst(VPI))
5507 assert(New->getParent() &&
"New recipe must have been inserted");
5508 if (VPI->
getOpcode() == Instruction::Load)
5517 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5524 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5526 VPI, FinalRedStoresBuilder))
5535 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5548 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5552 return Scalarize(VPI);
5559 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5561 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5571 const SCEV *PtrSCEV =
5573 bool IsSingleScalarLoad =
5579 I, Ptr, IsSingleScalarLoad,
5588 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5590 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5594 std::optional<int64_t> Stride =
5596 if (Stride != 1 && Stride != -1)
5627 return ReplaceWith(VPI,
Load);
5636 auto *StoreR = Builder.createWidenStore(
5639 return ReplaceWith(VPI, StoreR);
5646 return ReplaceWith(VPI, Recipe);
5648 return Scalarize(VPI);
5671 if (VPI->mayHaveSideEffects())
5675 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5680 if (VPI->getOpcode() == Instruction::Add &&
5689 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
5690 *VPI, VPI->getDebugLoc(),
I);
5691 Recipe->insertBefore(VPI);
5692 VPI->replaceAllUsesWith(Recipe);
5693 VPI->eraseFromParent();
5703 switch (Param.ParamKind) {
5704 case VFParamKind::Vector:
5705 case VFParamKind::GlobalPredicate:
5707 case VFParamKind::OMP_Uniform:
5708 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5709 SE->isLoopInvariant(
5710 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5712 case VFParamKind::OMP_Linear:
5713 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5714 m_scev_AffineAddRec(
5715 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5716 m_SpecificLoop(L)));
5733 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5734 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5737 if (It == Mappings.end())
5744struct CallWideningDecision {
5745 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5746 CallWideningDecision(KindTy Kind,
Function *Variant =
nullptr)
5769 return CallWideningDecision::KindTy::Scalarize;
5779 return CallWideningDecision::KindTy::Scalarize;
5783 false, VF, CostCtx);
5798 return CallWideningDecision::KindTy::Intrinsic;
5802 if (VecFunc && ScalarCost >= VecCallCost)
5803 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5805 return CallWideningDecision::KindTy::Scalarize;
5815 if (!VPI || !VPI->getUnderlyingValue() ||
5816 VPI->getOpcode() != Instruction::Call)
5821 VPI->op_begin() + CI->arg_size());
5823 CallWideningDecision Decision =
5832 switch (Decision.Kind) {
5833 case CallWideningDecision::KindTy::Intrinsic: {
5837 *VPI, VPI->getDebugLoc());
5840 case CallWideningDecision::KindTy::VectorVariant: {
5844 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
5845 Ops.push_back(Mask);
5847 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5849 *VPI, VPI->getDebugLoc());
5852 case CallWideningDecision::KindTy::Scalarize:
5858 VPI->replaceAllUsesWith(Replacement);
5859 VPI->eraseFromParent();
5881 if (!MemR || MemR->isConsecutive())
5884 VPValue *Ptr = MemR->getAddr();
5896 VPValue *StoredValue =
nullptr;
5900 StoredValue = StoreR->getStoredValue();
5902 IntrinID = Intrinsic::experimental_vp_strided_store;
5906 IntrinID = Intrinsic::experimental_vp_strided_load;
5909 Align Alignment = MemR->getAlign();
5912 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
5917 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
5918 return StridedLoadStoreCost < CurrentCost;
5929 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
5934 I32VF = Builder.createScalarZExtOrTrunc(
5948 "Stride type from SCEV must match the index type");
5949 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5952 auto *
Offset = Builder.createOverflowingOp(
5953 Instruction::Mul, {CanIV, StrideInBytes},
5954 {AddRecPtr->hasNoUnsignedWrap(),
false});
5958 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5961 VPValue *NewPtr = Builder.createVectorPointer(
5965 VPValue *Mask = MemR->getMask();
5970 Ops.push_back(StoredValue);
5971 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
5973 auto *StridedR = Builder.createWidenMemIntrinsic(
5976 *MemR, R.getDebugLoc());
5979 R.eraseFromParent();
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
VPBlockBase * getSinglePredecessor() const
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPRegionBlock * getRegion()
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
bool isOrdered() const
Returns true, if the phi is part of an ordered reduction.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
bool isInLoop() const
Returns true if the phi is part of an in-loop reduction.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
VPValue * getStartValue() const
Returns the start value of the induction.
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS*X will result in a value whose quantity matches our ...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS*X will result in a value whose quantity matches our own.
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
int_pred_ty< is_zero_int, 1 > m_False()
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
int_pred_ty< is_one, 1 > m_True()
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
initializer< Ty > init(const Ty &Val)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
ReductionStyle getReductionStyle(bool InLoop, bool Ordered, unsigned ScaleFactor)
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...