53 "should not try to widen irregular types");
68 auto IsConsecutiveAccess = [&](
VPValue *Addr,
Type *AccessTy) {
77 if (!VPBB->getParent())
80 auto EndIter = Term ? Term->getIterator() : VPBB->end();
85 VPValue *VPV = Ingredient.getVPSingleValue();
106 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
108 nullptr , IsConsecutive,
109 *VPI, Ingredient.getDebugLoc());
111 bool IsConsecutive = IsConsecutiveAccess(
112 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
114 *
Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
115 nullptr , IsConsecutive, *VPI, Ingredient.getDebugLoc());
118 Ingredient.operands(), *VPI,
119 Ingredient.getDebugLoc(),
GEP);
131 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
136 if (VectorID == Intrinsic::assume ||
137 VectorID == Intrinsic::lifetime_end ||
138 VectorID == Intrinsic::lifetime_start ||
139 VectorID == Intrinsic::sideeffect ||
140 VectorID == Intrinsic::pseudoprobe) {
145 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
146 VectorID != Intrinsic::pseudoprobe;
150 Ingredient.getDebugLoc());
153 *CI, VectorID,
drop_end(Ingredient.operands()), CI->getType(),
154 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
158 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
162 *VPI, Ingredient.getDebugLoc());
166 "inductions must be created earlier");
175 "Only recpies with zero or one defined values expected");
176 Ingredient.eraseFromParent();
187 const Loop *L =
nullptr;
192 if (
A->getOpcode() != Instruction::Store ||
193 B->getOpcode() != Instruction::Store)
206 const APInt *Distance;
212 Type *TyA =
A->getOperand(0)->getScalarType();
213 uint64_t SizeA =
DL.getTypeStoreSize(TyA);
214 Type *TyB =
B->getOperand(0)->getScalarType();
215 uint64_t SizeB =
DL.getTypeStoreSize(TyB);
220 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
222 auto VFs =
B->getParent()->getPlan()->vectorFactors();
226 return Distance->
abs().
uge(
234 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
235 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
244 return ExcludeRecipes.contains(
Store) ||
245 (
Store && isNoAliasViaDistance(
Store, &GroupLeader));
258 std::optional<SinkStoreInfo> SinkInfo = {}) {
259 bool CheckReads = SinkInfo.has_value();
263 if (SinkInfo && SinkInfo->shouldSkip(R))
267 if (!
R.mayWriteToMemory() && !(CheckReads &&
R.mayReadFromMemory()))
292template <
unsigned Opcode>
297 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
298 "Only Load and Store opcodes supported");
299 constexpr bool IsLoad = (Opcode == Instruction::Load);
302 RecipesByAddressAndType;
307 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
311 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
315 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
320 for (
auto &Group :
Groups) {
335 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
342 if (Candidate->getParent() == SinkTo ||
343 all_of(Candidate->operands(),
344 [](
VPValue *
Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
356 WorkList.
insert({SinkTo, Candidate});
368 for (
auto &Recipe : *VPBB)
370 InsertIfValidSinkCandidate(VPBB,
Op);
374 for (
unsigned I = 0;
I != WorkList.
size(); ++
I) {
377 std::tie(SinkTo, SinkCandidate) = WorkList[
I];
382 auto UsersOutsideSinkTo =
384 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
386 if (
any_of(UsersOutsideSinkTo, [SinkCandidate](
VPUser *U) {
387 return !U->usesFirstLaneOnly(SinkCandidate);
390 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
392 if (NeedsDuplicating) {
396 if (
auto *SinkCandidateRepR =
401 SinkCandidateRepR->getOpcode(), SinkCandidate->
operands(),
402 nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
406 Clone = SinkCandidate->
clone();
416 InsertIfValidSinkCandidate(SinkTo,
Op);
425 if (EntryBB->getNumSuccessors() != 2)
430 if (!Succ0 || !Succ1)
433 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
435 if (Succ0->getSingleSuccessor() == Succ1)
437 if (Succ1->getSingleSuccessor() == Succ0)
454 if (!Region1->isReplicator())
456 auto *MiddleBasicBlock =
458 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
463 if (!Region2 || !Region2->isReplicator())
466 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
467 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
468 if (!Mask1 || Mask1 != Mask2)
471 assert(Mask1 && Mask2 &&
"both region must have conditions");
477 if (TransformedRegions.
contains(Region1))
484 if (!Then1 || !Then2)
504 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
510 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
511 Phi1ToMove.eraseFromParent();
514 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
528 TransformedRegions.
insert(Region1);
531 return !TransformedRegions.
empty();
539 std::string RegionName = (
Twine(
"pred.") + Instr->getOpcodeName()).str();
540 assert(Instr->getParent() &&
"Predicated instruction not in any basic block");
541 auto *BlockInMask = PredRecipe->
getMask();
562 Region->setParent(ParentRegion);
568 RecipeWithoutMask->getDebugLoc());
569 Exiting->appendRecipe(PHIRecipe);
582 if (RepR->isPredicated())
601 if (ParentRegion && ParentRegion->
getExiting() == CurrentBlock)
613 if (!VPBB->getParent())
617 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
626 R.moveBefore(*PredVPBB, PredVPBB->
end());
628 auto *ParentRegion = VPBB->getParent();
629 if (ParentRegion && ParentRegion->getExiting() == VPBB)
630 ParentRegion->setExiting(PredVPBB);
634 return !WorkList.
empty();
641 bool ShouldSimplify =
true;
642 while (ShouldSimplify) {
658 if (!
IV ||
IV->getTruncInst())
673 for (
auto *U : FindMyCast->
users()) {
675 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
676 FoundUserCast = UserCast;
683 FindMyCast = FoundUserCast;
685 if (FindMyCast !=
IV)
707 VPUser *PhiUser = PhiR->getSingleUser();
713 PhiR->replaceAllUsesWith(Start);
714 PhiR->eraseFromParent();
752 Def->user_empty() || !Def->getUnderlyingValue() ||
753 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
766 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
768 Def->getUnderlyingInstr());
769 Clone->insertAfter(Def);
770 Def->replaceAllUsesWith(Clone);
771 Def->eraseFromParent();
786 PtrIV->replaceAllUsesWith(PtrAdd);
793 if (HasOnlyVectorVFs &&
none_of(WideIV->users(), [WideIV](
VPUser *U) {
794 return U->usesScalars(WideIV);
803 WrapFlags = {
static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
806 Plan, ID.getKind(), ID.getInductionOpcode(),
808 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
809 WideIV->getDebugLoc(), Builder, WrapFlags);
812 if (!HasOnlyVectorVFs) {
814 "plans containing a scalar VF cannot also include scalable VFs");
815 WideIV->replaceAllUsesWith(Steps);
818 WideIV->replaceUsesWithIf(Steps,
819 [WideIV, HasScalableVF](
VPUser &U,
unsigned) {
821 return U.usesFirstLaneOnly(WideIV);
822 return U.usesScalars(WideIV);
838 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
843 if (!Def || Def->getNumOperands() != 2)
851 auto IsWideIVInc = [&]() {
852 auto &ID = WideIV->getInductionDescriptor();
855 VPValue *IVStep = WideIV->getStepValue();
856 switch (ID.getInductionOpcode()) {
857 case Instruction::Add:
859 case Instruction::FAdd:
861 case Instruction::FSub:
864 case Instruction::Sub: {
884 return IsWideIVInc() ? WideIV :
nullptr;
908 VPValue *FirstActiveLane =
B.createFirstActiveLane(Mask,
DL);
910 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType,
DL);
911 VPValue *EndValue =
B.createAdd(CanonicalIV, FirstActiveLane,
DL);
916 if (Incoming != WideIV) {
918 EndValue =
B.createAdd(EndValue, One,
DL);
923 VPIRValue *Start = WideIV->getStartValue();
924 VPValue *Step = WideIV->getStepValue();
925 EndValue =
B.createDerivedIV(
927 Start, EndValue, Step);
941 if (WideIntOrFp && WideIntOrFp->getTruncInst())
951 Start, VectorTC, Step);
983 assert(EndValue &&
"Must have computed the end value up front");
988 if (Incoming != WideIV)
1000 auto *Zero = Plan.
getZero(StepTy);
1001 return B.createPtrAdd(EndValue,
B.createSub(Zero, Step),
1006 return B.createNaryOp(
1007 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1009 : Instruction::FAdd,
1010 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1025 const SCEV *Start, *Step;
1043 VPValue *ExitCount = Builder.createOverflowingOp(
1046 return Builder.createDerivedIV(Kind,
nullptr, StartVPV, ExitCount,
1055 VPBuilder VectorPHBuilder(VectorPH, VectorPH->begin());
1065 EndValues[WideIV] = EndValue;
1075 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1076 R.eraseFromParent();
1085 for (
auto [Idx, PredVPBB] :
enumerate(ExitVPBB->getPredecessors())) {
1087 if (PredVPBB == MiddleVPBB) {
1089 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1092 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1095 Plan, ExitIRI->getOperand(Idx), PSE);
1098 ExitIRI->setOperand(Idx, Escape);
1115 const auto &[V, Inserted] = SCEV2VPV.
try_emplace(ExpR->getSCEV(), ExpR);
1119 ExpR->replaceAllUsesWith(V->second);
1123 ExpR->eraseFromParent();
1130 bool CanCreateNewRecipe) {
1131 VPlan *Plan = Def->getParent()->getPlan();
1157 return Plan->
getZero(Def->getScalarType());
1172 if (CanCreateNewRecipe &&
1177 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1178 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1179 return Builder.createLogicalAnd(
X, Builder.createOr(
Y, Z));
1184 return Def->getOperand(1);
1189 return Builder.createLogicalAnd(
X,
Y);
1200 if (CanCreateNewRecipe &&
1202 return Builder.createNot(
C);
1206 Def->setOperand(0,
C);
1207 Def->setOperand(1,
Y);
1208 Def->setOperand(2,
X);
1213 if (CanCreateNewRecipe &&
1217 Y->getScalarType()->isIntegerTy(1))
1218 return Builder.createOr(
Y, Builder.createLogicalAnd(
X, Z));
1222 if (CanCreateNewRecipe &&
1228 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1),
X,
Y,
1229 Def->getDebugLoc());
1238 VPlan *Plan = Def->getParent()->getPlan();
1258 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1262 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1263 RepR->isSingleScalar(),
nullptr, *RepR, *RepR,
1264 RepR->getDebugLoc());
1265 Unmasked->insertBefore(RepR);
1279 bool CanCreateNewRecipe =
1284 Type *TruncTy = Def->getScalarType();
1285 Type *ATy =
A->getScalarType();
1286 if (TruncTy == ATy) {
1295 : Instruction::ZExt;
1298 if (
auto *UnderlyingExt = Z->getUnderlyingValue()) {
1300 Ext->setUnderlyingValue(UnderlyingExt);
1304 auto *Trunc = Builder.createWidenCast(Instruction::Trunc,
A, TruncTy);
1321 return Plan->
getZero(Def->getScalarType());
1327 return Builder.createSub(Plan->
getZero(
A->getScalarType()),
A,
1328 Def->getDebugLoc(),
"", NW);
1331 if (CanCreateNewRecipe &&
1339 return Builder.createSub(
X,
Y, Def->getDebugLoc(),
"", NW);
1346 Def->getDebugLoc());
1353 MulR->hasNoSignedWrap() &&
1355 return Builder.createNaryOp(
1358 Def->getDebugLoc());
1363 return Builder.createNaryOp(
1379 return match(U, m_Not(m_Specific(Cmp))) ||
1380 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1382 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1389 R->setOperand(1,
Y);
1390 R->setOperand(2,
X);
1394 R->replaceAllUsesWith(Cmp);
1399 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1400 Cmp->setDebugLoc(Def->getDebugLoc());
1413 if (
Op->getNumUsers() > 1 ||
1417 }
else if (!UnpairedCmp) {
1418 UnpairedCmp =
Op->getDefiningRecipe();
1422 UnpairedCmp =
nullptr;
1429 if (NewOps.
size() < Def->getNumOperands()) {
1438 if (CanCreateNewRecipe &&
1449 A->getScalarType() == Def->getScalarType())
1454 Type *WideStepTy = Def->getScalarType();
1455 if (
X->getScalarType() != WideStepTy)
1456 X = Builder.createWidenCast(Instruction::Trunc,
X, WideStepTy);
1465 Def->getScalarType()->isIntegerTy(1)) {
1466 Def->setOperand(1, Plan->
getTrue());
1467 Def->setOperand(0,
Y);
1474 return Def->getOperand(0);
1480 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1496 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1502 return BuildVector->getOperand(Idx);
1511 Def->replaceUsesWithIf(Def->getOperand(0), [Def](
VPUser &U,
unsigned) {
1512 return U.usesFirstLaneOnly(Def);
1522 "broadcast operand must be single-scalar");
1523 Def->setOperand(0, Z);
1528 Def->replaceUsesWithIf(
1529 X, [Def](
const VPUser &U,
unsigned) {
return U.usesScalars(Def); });
1534 if (Def->getNumOperands() == 1) {
1535 return Def->getOperand(0);
1539 return Phi->getOperand(0);
1545 if (Def->getNumOperands() == 1 &&
1570 return Builder.createNaryOp(Instruction::ExtractElement, {
A, LaneToExtract},
1571 Def->getDebugLoc());
1585 if (IVInc->getNumUsers() == 2) {
1590 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1591 Def->replaceAllUsesWith(IVInc);
1593 Inc->replaceAllUsesWith(Phi);
1594 Phi->setOperand(0,
Y);
1604 return VPR->getOperand(0);
1610 return Steps->getOperand(0);
1616 Def->replaceUsesWithIf(StartV, [](
const VPUser &U,
unsigned Idx) {
1618 return PhiR && PhiR->isInLoop();
1638 Def->replaceAllUsesWith(New);
1639 Def->eraseFromParent();
1642 Def->eraseFromParent();
1661 R.getVPSingleValue()->replaceAllUsesWith(
X);
1677 while (!Worklist.
empty()) {
1686 R->replaceAllUsesWith(
1687 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(
X,
Y)));
1691static std::optional<Instruction::BinaryOps>
1694 case Intrinsic::masked_udiv:
1695 return Instruction::UDiv;
1696 case Intrinsic::masked_sdiv:
1697 return Instruction::SDiv;
1698 case Intrinsic::masked_urem:
1699 return Instruction::URem;
1700 case Intrinsic::masked_srem:
1701 return Instruction::SRem;
1718 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1722 if (RepR && RepR->getOpcode() == Instruction::Store &&
1725 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1726 true ,
nullptr , *RepR ,
1727 *RepR , RepR->getDebugLoc());
1728 Clone->insertBefore(RepOrWidenR);
1730 VPValue *ExtractOp = Clone->getOperand(0);
1736 Clone->setOperand(0, ExtractOp);
1737 RepR->eraseFromParent();
1749 VPValue *SafeDivisor = Builder.createSelect(
1750 IntrR->getOperand(2), IntrR->getOperand(1),
1752 VPValue *Clone = Builder.createNaryOp(
1753 *
Opc, {IntrR->getOperand(0), SafeDivisor},
1756 IntrR->eraseFromParent();
1765 auto IntroducesBCastOf = [](
const VPValue *
Op) {
1774 return !U->usesScalars(
Op);
1778 if (
any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1781 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1782 IntroducesBCastOf(Op)))
1786 bool LiveInNeedsBroadcast =
1787 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1788 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1789 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1796 RepOrWidenR->getUnderlyingInstr());
1797 Clone->insertBefore(RepOrWidenR);
1798 RepOrWidenR->replaceAllUsesWith(Clone);
1800 RepOrWidenR->eraseFromParent();
1836 if (Blend->isNormalized() || !
match(Blend->getMask(0),
m_False()))
1837 UniqueValues.
insert(Blend->getIncomingValue(0));
1838 for (
unsigned I = 1;
I != Blend->getNumIncomingValues(); ++
I)
1840 UniqueValues.
insert(Blend->getIncomingValue(
I));
1842 if (UniqueValues.
size() == 1) {
1843 Blend->replaceAllUsesWith(*UniqueValues.
begin());
1844 Blend->eraseFromParent();
1848 if (Blend->isNormalized())
1854 unsigned StartIndex = 0;
1855 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1867 OperandsWithMask.
push_back(Blend->getIncomingValue(StartIndex));
1869 for (
unsigned I = 0;
I != Blend->getNumIncomingValues(); ++
I) {
1870 if (
I == StartIndex)
1872 OperandsWithMask.
push_back(Blend->getIncomingValue(
I));
1873 OperandsWithMask.
push_back(Blend->getMask(
I));
1878 OperandsWithMask, *Blend, Blend->getDebugLoc());
1879 NewBlend->insertBefore(&R);
1881 VPValue *DeadMask = Blend->getMask(StartIndex);
1883 Blend->eraseFromParent();
1888 if (NewBlend->getNumOperands() == 3 &&
1890 VPValue *Inc0 = NewBlend->getOperand(0);
1891 VPValue *Inc1 = NewBlend->getOperand(1);
1892 VPValue *OldMask = NewBlend->getOperand(2);
1893 NewBlend->setOperand(0, Inc1);
1894 NewBlend->setOperand(1, Inc0);
1895 NewBlend->setOperand(2, NewMask);
1922 APInt MaxVal = AlignedTC - 1;
1925 unsigned NewBitWidth =
1931 bool MadeChange =
false;
1956 "canonical IV is not expected to have a truncation");
1961 NewWideIV->insertBefore(WideIV);
1968 Cmp->replaceAllUsesWith(
1969 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
1983 return any_of(
Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
1985 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
1999 const SCEV *VectorTripCount =
2004 "Trip count SCEV must be computable");
2019 bool MadeChange =
false;
2027 for (
VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2036 Builder.setInsertPoint(Extract);
2039 Start = Builder.createAdd(
2044 Extract->eraseFromParent();
2059 auto *Term = &ExitingVPBB->
back();
2074 const SCEV *VectorTripCount =
2080 "Trip count SCEV must be computable");
2099 Term->setOperand(1, Plan.
getTrue());
2104 {}, Term->getDebugLoc());
2106 Term->eraseFromParent();
2114 assert(Plan.
hasVF(BestVF) &&
"BestVF is not available in Plan");
2115 assert(Plan.
hasUF(BestUF) &&
"BestUF is not available in Plan");
2134 RecurKind RK = PhiR->getRecurrenceKind();
2141 RecWithFlags->dropPoisonGeneratingFlags();
2147struct VPCSEDenseMapInfo :
public DenseMapInfo<VPSingleDefRecipe *> {
2156 return GEP->getSourceElementType();
2159 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2160 [](
auto *
I) {
return I->getSourceElementType(); })
2161 .
Default([](
auto *) {
return nullptr; });
2165 static bool canHandle(
const VPSingleDefRecipe *Def) {
2174 if (!
C || (!
C->first && (
C->second == Instruction::InsertValue ||
2175 C->second == Instruction::ExtractValue)))
2179 return !
Def->mayReadOrWriteMemory();
2183 static unsigned getHashValue(
const VPSingleDefRecipe *Def) {
2186 getGEPSourceElementType(Def),
Def->getScalarType(),
2189 if (RFlags->hasPredicate())
2192 return hash_combine(Result, SIVSteps->getInductionOpcode());
2197 static bool isEqual(
const VPSingleDefRecipe *L,
const VPSingleDefRecipe *R) {
2198 if (
L->getVPRecipeID() !=
R->getVPRecipeID() ||
2201 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2203 !
equal(
L->operands(),
R->operands()))
2207 "must have valid opcode info for both recipes");
2209 if (LFlags->hasPredicate() &&
2210 LFlags->getPredicate() !=
2214 if (LSIV->getInductionOpcode() !=
2224 const VPRegionBlock *RegionL =
L->getRegion();
2225 const VPRegionBlock *RegionR =
R->getRegion();
2228 L->getParent() !=
R->getParent())
2230 return L->getScalarType() ==
R->getScalarType();
2246 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2250 if (!VPDT.
dominates(V->getParent(), VPBB))
2255 Def->replaceAllUsesWith(V);
2268 bool Sinking =
false) {
2297 "Expected vector prehader's successor to be the vector loop region");
2305 return !Op->isDefinedOutsideLoopRegions();
2308 R.moveBefore(*Preheader, Preheader->
end());
2328 assert(!RepR->isPredicated() &&
2329 "Expected prior transformation of predicated replicates to "
2330 "replicate regions");
2335 if (!RepR->isSingleScalar())
2339 if (RepR->getOpcode() == Instruction::Store &&
2340 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2345 assert((!R.mayWriteToMemory() ||
2346 (RepR && RepR->getOpcode() == Instruction::Store &&
2347 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2348 "The only recipes that may write to memory are expected to be "
2349 "stores with invariant pointer-operand");
2359 if (
any_of(Def->users(), [&SinkBB, &LoopRegion](
VPUser *U) {
2360 auto *UserR = cast<VPRecipeBase>(U);
2361 VPBasicBlock *Parent = UserR->getParent();
2363 if (SinkBB && SinkBB != Parent)
2368 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2369 Parent->getSinglePredecessor() != LoopRegion;
2379 "Defining block must dominate sink block");
2404 VPValue *ResultVPV = R.getVPSingleValue();
2406 unsigned NewResSizeInBits = MinBWs.
lookup(UI);
2407 if (!NewResSizeInBits)
2420 (void)OldResSizeInBits;
2428 VPW->dropPoisonGeneratingFlags();
2430 assert((OldResSizeInBits != NewResSizeInBits ||
2432 "Only ICmps should not need extending the result.");
2438 if (OldResSizeInBits != NewResSizeInBits) {
2440 Instruction::ZExt, ResultVPV, OldResTy);
2442 Ext->setOperand(0, ResultVPV);
2452 unsigned OpSizeInBits =
Op->getScalarType()->getScalarSizeInBits();
2453 if (OpSizeInBits == NewResSizeInBits)
2455 assert(OpSizeInBits > NewResSizeInBits &&
"nothing to truncate");
2456 auto [ProcessedIter, Inserted] = ProcessedTruncs.
try_emplace(
Op);
2462 Builder.setInsertPoint(&R);
2463 ProcessedIter->second =
2464 Builder.createWidenCast(Instruction::Trunc,
Op, NewResTy);
2466 Op = ProcessedIter->second;
2470 NWR->insertBefore(&R);
2474 VPValue *Replacement = NWR->getVPSingleValue();
2475 if (OldResSizeInBits != NewResSizeInBits)
2481 R.eraseFromParent();
2487 std::optional<VPDominatorTree> VPDT;
2495 bool SimplifiedPhi =
false;
2505 assert(VPBB->getNumSuccessors() == 2 &&
2506 "Two successors expected for BranchOnCond");
2507 unsigned RemovedIdx;
2518 "There must be a single edge between VPBB and its successor");
2523 SimplifiedPhi =
true;
2527 if (!PhiR || PhiR->getNumIncoming() != 1)
2529 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2530 PhiR->eraseFromParent();
2535 VPBB->back().eraseFromParent();
2547 if (Reachable.contains(
B))
2558 for (
VPValue *Def : R.definedValues())
2559 Def->replaceAllUsesWith(&Tmp);
2560 R.eraseFromParent();
2564 return SimplifiedPhi;
2590 auto GetSimplifiedLiveInViaSCEV = [&](
VPValue *VPV) ->
VPValue * {
2599 if (
VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2600 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2612 "expected to run before loop regions are created");
2614 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](
VPUser &U,
2621 return VPDT.
dominates(Header, R->getParent());
2624 for (
const SCEV *Stride : StridesMap.
values()) {
2626 const APInt *StrideConst;
2633 CanUseVersionedStride);
2647 CanUseVersionedStride);
2649 RewriteMap[StrideV] = StrideExpr;
2656 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2659 if (NewSCEV != ScevExpr) {
2661 ExpSCEV->replaceAllUsesWith(NewExp);
2672 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](
VPRecipeBase *Root) {
2677 while (!Worklist.
empty()) {
2680 if (!Visited.
insert(CurRec).second)
2702 RecWithFlags->isDisjoint()) {
2705 Builder.createAdd(
A,
B, RecWithFlags->getDebugLoc());
2706 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2707 RecWithFlags->replaceAllUsesWith(New);
2708 RecWithFlags->eraseFromParent();
2711 RecWithFlags->dropPoisonGeneratingFlags();
2716 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2717 "found instruction with poison generating flags not covered by "
2718 "VPRecipeWithIRFlags");
2723 if (
VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2745 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2746 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2747 match(WidenRec->getMask(), m_UnlessHdrMask))
2748 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2750 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2751 if (AddrDef && InterleaveRec->getMask() &&
2752 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2753 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2763 const bool &EpilogueAllowed) {
2764 if (InterleaveGroups.empty())
2775 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2782 for (
const auto *IG : InterleaveGroups) {
2785 for (
auto *Member : IG->members())
2787 StartMember = Member;
2795 for (
unsigned I = 0;
I < IG->getFactor(); ++
I) {
2801 StoredValues.
push_back(StoreR->getStoredValue());
2808 bool NeedsMaskForGaps =
2809 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2810 (!StoredValues.
empty() && !IG->isFull());
2813 auto *InsertPos = IRMemberToRecipe.
lookup(IRInsertPos);
2817 "Dead member in non-load group?");
2822 InsertPos->getAsRecipe()))
2823 InsertPos = MemberR;
2824 IRInsertPos = &InsertPos->getIngredient();
2834 VPValue *Addr = Start->getAddr();
2836 if (IG->getIndex(StartMember) != 0 ||
2844 assert(IG->getIndex(IRInsertPos) != 0 &&
2845 "index of insert position shouldn't be zero");
2849 IG->getIndex(IRInsertPos),
2853 Addr =
B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2859 if (IG->isReverse()) {
2862 -(int64_t)IG->getFactor(), NW, InsertPosR->
getDebugLoc());
2863 ReversePtr->insertBefore(InsertPosR);
2867 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2869 VPIG->insertBefore(InsertPosR);
2872 for (
unsigned i = 0; i < IG->getFactor(); ++i)
2875 if (!Member->getType()->isVoidTy()) {
2893static std::optional<VPValue *>
2946 VPValue *UncountableCondition =
nullptr;
2950 return std::nullopt;
2953 Worklist.
push_back(UncountableCondition);
2954 while (!Worklist.
empty()) {
2958 if (V->isDefinedOutsideLoopRegions())
2964 if (V->getNumUsers() > 1)
2965 return std::nullopt;
2977 return std::nullopt;
2981 return std::nullopt;
2989 return std::nullopt;
2994 if (Recipes.
empty() ||
2996 return std::nullopt;
2998 return UncountableCondition;
3054 for (
auto &Exit : Exits) {
3055 if (Exit.EarlyExitingVPBB == LatchVPBB)
3059 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3060 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3071 std::optional<VPValue *>
Cond =
3087 assert(
Load &&
"Couldn't find exactly one load");
3090 "Uncountable exit condition load is conditional.");
3104 DL.getTypeStoreSize(
Load->getScalarType()).getFixedValue());
3128 while (InsertIt != HeaderVPBB->
end() &&
3130 erase(ConditionRecipes, &*InsertIt);
3133 for (
auto *Recipe :
reverse(ConditionRecipes))
3134 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3138 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3140 Type *IVScalarTy =
IV->getScalarType();
3146 "uncountable.exit.mask");
3151 if (R.mayReadOrWriteMemory() && &R !=
Load) {
3153 if (!VPDT.
dominates(R.getParent(), LatchVPBB))
3163 "Expected BranchOnCond terminator for MiddleVPBB");
3174 auto Phis = ScalarPH->
phis();
3184 "Continuing from different IV");
3206 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3208 for (
auto [EarlyExitingVPBB, ExitBlock] :
3212 VPValue *CondOfEarlyExitingVPBB;
3213 [[maybe_unused]]
bool Matched =
3214 match(EarlyExitingVPBB->getTerminator(),
3216 assert(Matched &&
"Terminator must be BranchOnCond");
3220 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3221 auto *CondToEarlyExit = EarlyExitingBuilder.
createNaryOp(
3223 TrueSucc == ExitBlock
3224 ? CondOfEarlyExitingVPBB
3225 : EarlyExitingBuilder.
createNot(CondOfEarlyExitingVPBB));
3231 "exit condition must dominate the latch");
3239 assert(!Exits.
empty() &&
"must have at least one early exit");
3246 for (
const auto &[Num, VPB] :
enumerate(RPOT))
3249 return RPOIdx[
A.EarlyExitingVPBB] < RPOIdx[
B.EarlyExitingVPBB];
3255 for (
unsigned I = 0;
I + 1 < Exits.
size(); ++
I)
3256 for (
unsigned J =
I + 1; J < Exits.
size(); ++J)
3258 Exits[
I].EarlyExitingVPBB) &&
3259 "RPO sort must place dominating exits before dominated ones");
3265 VPValue *Combined = Exits[0].CondToExit;
3278 "Unexpected terminator");
3279 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3280 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3281 LatchExitingBranch->eraseFromParent();
3284 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3285 LatchVPBB->clearSuccessors();
3290 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3291 MiddleVPBB->clearPredecessors();
3292 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3294 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3299 for (
unsigned Idx = 0; Idx != Exits.
size(); ++Idx) {
3303 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3311 Exits.
size() == 1 ? VectorEarlyExitVPBBs[0]
3314 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3346 for (
auto [Exit, VectorEarlyExitVPBB] :
3347 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3348 auto &[EarlyExitingVPBB, EarlyExitVPBB,
_] = Exit;
3360 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3361 VPValue *NewIncoming = IncomingVal;
3363 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3368 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3369 ExitIRI->addIncoming(NewIncoming);
3372 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3406 bool IsLastDispatch = (
I + 2 == Exits.
size());
3408 IsLastDispatch ? VectorEarlyExitVPBBs.
back()
3414 VectorEarlyExitVPBBs[
I]->setPredecessors({CurrentBB});
3417 CurrentBB = FalseBB;
3432 VPValue *VecOp = Red->getVecOp();
3434 assert(!Red->isPartialReduction() &&
3435 "This path does not support partial reductions");
3438 auto IsExtendedRedValidAndClampRange =
3451 "getExtendedReductionCost only supports integer types");
3452 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3453 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3454 Red->getFastMathFlagsOrNone(),
CostKind);
3455 return ExtRedCost.
isValid() && ExtRedCost < ExtCost + RedCost;
3463 IsExtendedRedValidAndClampRange(
3484 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3485 Opcode != Instruction::FAdd)
3488 assert(!Red->isPartialReduction() &&
3489 "This path does not support partial reductions");
3493 auto IsMulAccValidAndClampRange =
3505 (Ext0->getOpcode() != Ext1->getOpcode() ||
3506 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3510 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3512 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3519 ExtCost += Ext0->computeCost(VF, Ctx);
3521 ExtCost += Ext1->computeCost(VF, Ctx);
3523 ExtCost += OuterExt->computeCost(VF, Ctx);
3525 return MulAccCost.
isValid() &&
3526 MulAccCost < ExtCost + MulCost + RedCost;
3531 VPValue *VecOp = Red->getVecOp();
3569 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3571 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3572 Mul->setOperand(1, ExtB);
3582 ExtendAndReplaceConstantOp(RecipeA, RecipeB,
B,
Mul);
3587 IsMulAccValidAndClampRange(
Mul, RecipeA, RecipeB,
nullptr)) {
3594 if (!
Sub && IsMulAccValidAndClampRange(
Mul,
nullptr,
nullptr,
nullptr))
3611 ExtendAndReplaceConstantOp(Ext0, Ext1,
B,
Mul);
3620 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3621 Ext0->getOpcode() == Ext1->getOpcode() &&
3622 IsMulAccValidAndClampRange(
Mul, Ext0, Ext1, Ext) &&
Mul->hasOneUse()) {
3624 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(),
nullptr,
3625 *Ext0, *Ext0, Ext0->getDebugLoc());
3626 NewExt0->insertBefore(Ext0);
3631 Ext->getScalarType(),
nullptr, *Ext1,
3632 *Ext1, Ext1->getDebugLoc());
3635 auto *NewMul =
Mul->cloneWithOperands({NewExt0, NewExt1});
3636 NewMul->insertBefore(
Mul);
3637 Ext->replaceAllUsesWith(NewMul);
3638 Ext->eraseFromParent();
3639 Mul->eraseFromParent();
3653 assert(!Red->isPartialReduction() &&
3654 "This path does not support partial reductions");
3657 auto IP = std::next(Red->getIterator());
3658 auto *VPBB = Red->getParent();
3668 Red->replaceAllUsesWith(AbstractR);
3688 return CommonMetadata;
3691template <
unsigned Opcode>
3696 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3697 "Only Load and Store opcodes supported");
3698 [[maybe_unused]]
constexpr bool IsLoad = (Opcode == Instruction::Load);
3705 for (
auto Recipes :
Groups) {
3706 if (Recipes.size() < 2)
3711 "Expected all recipes in group to have the same load-store type");
3718 VPValue *MaskI = RecipeI->getMask();
3724 bool HasComplementaryMask =
false;
3729 VPValue *MaskJ = RecipeJ->getMask();
3738 if (HasComplementaryMask) {
3739 assert(Group.
size() >= 2 &&
"must have at least 2 entries");
3749template <
typename InstType>
3767 for (
auto &Group :
Groups) {
3787 return R->isSingleScalar() == IsSingleScalar;
3789 "all members in group must agree on IsSingleScalar");
3794 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3795 IsSingleScalar,
nullptr, *EarliestLoad, CommonMetadata);
3797 UnpredicatedLoad->insertBefore(EarliestLoad);
3801 Load->replaceAllUsesWith(UnpredicatedLoad);
3802 Load->eraseFromParent();
3811 if (!StoreLoc || !StoreLoc->AATags.Scope)
3818 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3830 for (
auto &Group :
Groups) {
3843 VPValue *SelectedValue = Group[0]->getOperand(0);
3846 bool IsSingleScalar = Group[0]->isSingleScalar();
3847 for (
unsigned I = 1;
I < Group.size(); ++
I) {
3848 assert(IsSingleScalar == Group[
I]->isSingleScalar() &&
3849 "all members in group must agree on IsSingleScalar");
3850 VPValue *Mask = Group[
I]->getMask();
3852 SelectedValue = Builder.createSelect(
3855 Value->getScalarType()));
3863 StoreWithMinAlign->getUnderlyingInstr(),
3864 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3865 nullptr, *LastStore, CommonMetadata);
3866 UnpredicatedStore->insertBefore(*InsertBB, LastStore->
getIterator());
3870 Store->eraseFromParent();
3885 VPValue *OpV,
unsigned Idx,
bool IsScalable) {
3890 if (Member0Op == OpV)
3900 return !IsScalable && !W->getMask() && W->isConsecutive() &&
3903 return IR->getInterleaveGroup()->isFull() &&
IR->getVPValue(Idx) == OpV;
3918 if (R->getScalarType() != WideMember0->getScalarType())
3920 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
3924 for (
unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
3927 OpsI.
push_back(
Op->getDefiningRecipe()->getOperand(Idx));
3932 if (
any_of(
enumerate(OpsI), [WideMember0, Idx, IsScalable](
const auto &
P) {
3933 const auto &[OpIdx, OpV] =
P;
3934 return !
canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
3945static std::optional<ElementCount>
3949 if (!InterleaveR || InterleaveR->
getMask())
3950 return std::nullopt;
3952 Type *GroupElementTy =
nullptr;
3956 return Op->getScalarType() == GroupElementTy;
3958 return std::nullopt;
3962 return Op->getScalarType() == GroupElementTy;
3964 return std::nullopt;
3968 if (IG->getFactor() != IG->getNumMembers())
3969 return std::nullopt;
3975 assert(
Size.isScalable() == VF.isScalable() &&
3976 "if Size is scalable, VF must be scalable and vice versa");
3977 return Size.getKnownMinValue();
3981 unsigned MinVal = VF.getKnownMinValue();
3983 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
3986 return std::nullopt;
3994 return RepR && RepR->isSingleScalar();
4008 if (V->isDefinedOutsideLoopRegions()) {
4011 return M->isDefinedOutsideLoopRegions() &&
4012 M->getScalarType() == V->getScalarType();
4014 "expected distinct loop-invariant values of matching scalar type");
4029 for (
unsigned Idx = 0,
E = WideMember0->getNumOperands(); Idx !=
E; ++Idx) {
4031 for (
VPValue *Member : Members)
4032 OpsI.
push_back(Member->getDefiningRecipe()->getOperand(Idx));
4033 WideMember0->setOperand(
4042 auto *LI =
cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4044 *LI, LoadGroup->getAddr(), LoadGroup->getMask(),
true,
4045 *LoadGroup, LoadGroup->getDebugLoc());
4051 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4052 "must be a single scalar load");
4053 NarrowedOps.
insert(RepR);
4058 VPValue *PtrOp = WideLoad->getAddr();
4060 PtrOp = VecPtr->getOperand(0);
4065 nullptr, {}, *WideLoad);
4066 N->insertBefore(WideLoad);
4071std::unique_ptr<VPlan>
4091 "unexpected branch-on-count");
4094 std::optional<ElementCount> VFToOptimize;
4108 if (R.mayWriteToMemory() && !InterleaveR)
4114 return any_of(V->users(), [&](VPUser *U) {
4115 auto *UR = cast<VPRecipeBase>(U);
4116 return UR->getParent()->getParent() != VectorLoop;
4133 std::optional<ElementCount> NarrowedVF =
4135 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4137 VFToOptimize = NarrowedVF;
4140 if (InterleaveR->getStoredValues().empty())
4145 auto *Member0 = InterleaveR->getStoredValues()[0];
4155 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4158 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4159 return IR && IR->getInterleaveGroup()->isFull() &&
4160 IR->getVPValue(Op.index()) == Op.value();
4169 VFToOptimize->isScalable()))
4174 if (StoreGroups.empty())
4178 bool RequiresScalarEpilogue =
4189 std::unique_ptr<VPlan> NewPlan;
4191 NewPlan = std::unique_ptr<VPlan>(Plan.
duplicate());
4192 Plan.
setVF(*VFToOptimize);
4193 NewPlan->removeVF(*VFToOptimize);
4200 for (
auto *StoreGroup : StoreGroups) {
4202 NarrowedOps, Preheader);
4208 StoreGroup->getDebugLoc());
4215 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4221 if (VFToOptimize->isScalable()) {
4224 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale,
UF},
4232 materializeVectorTripCount(Plan, VectorPH,
false,
4233 RequiresScalarEpilogue, Step);
4238 removeDeadRecipes(Plan);
4241 "All VPVectorPointerRecipes should have been removed");
4261 "Cannot handle loops with uncountable early exits");
4268 assert(RecurSplice &&
"expected FirstOrderRecurrenceSplice");
4275 if (
any_of(RecurSplice->users(),
4276 [](
VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4357 {},
"vector.recur.extract.for.phi");
4360 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4374 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4375 VPValue *InvariantCandidate = BinOp->getOperand(1);
4377 std::swap(WidenIVCandidate, InvariantCandidate);
4391 auto *ClonedOp = BinOp->
clone();
4392 if (ClonedOp->getOperand(0) == WidenIV) {
4393 ClonedOp->setOperand(0, ScalarIV);
4395 assert(ClonedOp->getOperand(1) == WidenIV &&
"one operand must be WideIV");
4396 ClonedOp->setOperand(1, ScalarIV);
4410 return std::nullopt;
4415 return std::nullopt;
4427 auto CheckSentinel = [&SE](
const SCEV *IVSCEV,
4428 bool UseMax) -> std::optional<APSInt> {
4430 for (
bool Signed : {
true,
false}) {
4439 return std::nullopt;
4447 PhiR->getRecurrenceKind()))
4456 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4470 !
match(FindLastSelect,
4479 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4484 "IVOfExpressionToSink not being an AddRec must imply "
4485 "FindLastExpression not being an AddRec.");
4494 bool UseMax = *StepDirection;
4495 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4496 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4503 if (IVOfExpressionToSink) {
4504 const SCEV *FindLastExpressionSCEV =
4506 if (std::optional<bool> NewUseMax =
4508 if (
auto NewSentinel =
4509 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4512 SentinelVal = *NewSentinel;
4513 UseSigned = NewSentinel->isSigned();
4514 UseMax = *NewUseMax;
4515 IVSCEV = FindLastExpressionSCEV;
4516 IVOfExpressionToSink =
nullptr;
4526 if (AR->hasNoSignedWrap())
4528 else if (AR->hasNoUnsignedWrap())
4538 VPValue *NewFindLastSelect = BackedgeVal;
4540 if (!SentinelVal || IVOfExpressionToSink) {
4543 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4544 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4545 if (
match(FindLastSelect,
4547 SelectCond = LoopBuilder.
createNot(SelectCond);
4554 if (SelectCond !=
Cond || IVOfExpressionToSink) {
4557 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4566 VPIRFlags Flags(MinMaxKind,
false,
false,
4572 NewFindLastSelect, Flags, ExitDL);
4575 VPValue *VectorRegionExitingVal = ReducedIV;
4576 if (IVOfExpressionToSink)
4577 VectorRegionExitingVal =
4579 ReducedIV, IVOfExpressionToSink);
4582 VPValue *StartVPV = PhiR->getStartValue();
4589 NewRdxResult = MiddleBuilder.
createSelect(Cmp, VectorRegionExitingVal,
4599 AnyOfPhi->insertAfter(PhiR);
4606 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4619 PhiR->hasUsesOutsideReductionChain());
4620 NewPhiR->insertBefore(PhiR);
4621 PhiR->replaceAllUsesWith(NewPhiR);
4622 PhiR->eraseFromParent();
4629struct ReductionExtend {
4630 Type *SrcType =
nullptr;
4631 ExtendKind Kind = ExtendKind::PR_None;
4637struct ExtendedReductionOperand {
4641 ReductionExtend ExtendA, ExtendB;
4649struct VPPartialReductionChain {
4652 VPWidenRecipe *ReductionBinOp =
nullptr;
4654 ExtendedReductionOperand ExtendedOp;
4661 unsigned AccumulatorOpIdx;
4662 unsigned ScaleFactor;
4665 VPBlendRecipe *Blend =
nullptr;
4670static std::optional<unsigned>
4674 "Expected a non-normalized blend with two incoming values");
4680 return std::nullopt;
4681 return FirstIncomingHasOneUse ? 0 : 1;
4693 if (!
Op->hasOneUse() ||
4699 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4700 Op->getOperand(1), NarrowTy);
4702 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4711 auto *
Sub =
Op->getOperand(0)->getDefiningRecipe();
4713 assert(Ext->getOpcode() ==
4715 "Expected both the LHS and RHS extends to be the same");
4716 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4719 auto *FreezeX = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
X}));
4720 auto *FreezeY = Builder.insert(
new VPWidenRecipe(Instruction::Freeze, {
Y}));
4721 auto *
Max = Builder.insert(
4723 {FreezeX, FreezeY}, SrcTy));
4724 auto *Min = Builder.insert(
4726 {FreezeX, FreezeY}, SrcTy));
4727 auto *AbsDiff = Builder.insert(
4730 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4731 Op->getScalarType());
4743 if (!
Mul->hasOneUse() ||
4744 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4745 MulLHS->getOpcode() != MulRHS->getOpcode())
4748 auto *NewLHS = Builder.createWidenCast(
4749 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4750 auto *NewRHS = MulLHS == MulRHS
4752 : Builder.createWidenCast(MulRHS->getOpcode(),
4753 MulRHS->getOperand(0),
4754 Ext->getScalarType());
4755 auto *NewMul =
Mul->cloneWithOperands({NewLHS, NewRHS});
4756 Builder.insert(NewMul);
4757 Op->replaceAllUsesWith(NewMul);
4758 Op->eraseFromParent();
4759 Mul->eraseFromParent();
4768 VPValue *VecOp = Red->getVecOp();
4822static void transformToPartialReduction(
const VPPartialReductionChain &Chain,
4830 WidenRecipe->
getOperand(1 - Chain.AccumulatorOpIdx));
4833 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4849 if ((WidenRecipe->
getOpcode() == Instruction::Sub &&
4851 (WidenRecipe->
getOpcode() == Instruction::FSub &&
4856 if (WidenRecipe->
getOpcode() == Instruction::FSub) {
4868 Builder.insert(NegRecipe);
4869 ExtendedOp = NegRecipe;
4884 std::optional<unsigned> BlendReductionIdx =
4885 getBlendReductionUpdateValueIdx(Chain.Blend);
4886 assert(BlendReductionIdx &&
4888 "Expected blend to contain the reduction update");
4905 assert((!ExitValue || IsLastInChain) &&
4906 "if we found ExitValue, it must match RdxPhi's backedge value");
4917 PartialRed->insertBefore(WidenRecipe);
4927 E->insertBefore(WidenRecipe);
4928 PartialRed->replaceAllUsesWith(
E);
4941 auto *NewScaleFactor = Plan.
getConstantInt(32, Chain.ScaleFactor);
4942 StartInst->setOperand(2, NewScaleFactor);
4950 VPValue *OldStartValue = StartInst->getOperand(0);
4951 StartInst->setOperand(0, StartInst->getOperand(1));
4955 assert(RdxResult &&
"Could not find reduction result");
4958 unsigned SubOpc = Chain.RK ==
RecurKind::FSub ? Instruction::BinaryOps::FSub
4959 : Instruction::BinaryOps::Sub;
4965 [&NewResult](
VPUser &U,
unsigned Idx) {
return &
U != NewResult; });
4971 const VPPartialReductionChain &Link,
4974 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
4975 std::optional<unsigned> BinOpc = std::nullopt;
4977 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
4978 BinOpc = ExtendedOp.ExtendsUser->
getOpcode();
4980 std::optional<llvm::FastMathFlags>
Flags;
4984 auto GetLinkOpcode = [&Link]() ->
unsigned {
4987 return Instruction::Add;
4989 return Instruction::FAdd;
4991 return Link.ReductionBinOp->
getOpcode();
4996 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
4997 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5018static std::optional<ExtendedReductionOperand>
5021 "Op should be operand of UpdateR");
5029 if (
Op->hasOneUse() &&
5038 Type *RHSInputType =
Y->getScalarType();
5039 if (LHSInputType != RHSInputType ||
5040 LHSExt->getOpcode() != RHSExt->getOpcode())
5041 return std::nullopt;
5044 return ExtendedReductionOperand{
5046 {LHSInputType, getPartialReductionExtendKind(LHSExt)},
5050 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5053 VPValue *CastSource = CastRecipe->getOperand(0);
5054 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5064 return ExtendedReductionOperand{
5071 if (!
Op->hasOneUse())
5072 return std::nullopt;
5077 return std::nullopt;
5087 return std::nullopt;
5091 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5094 const APInt *RHSConst =
nullptr;
5100 return std::nullopt;
5104 if (Cast && OuterExtKind &&
5105 getPartialReductionExtendKind(Cast) != OuterExtKind)
5106 return std::nullopt;
5108 Type *RHSInputType = LHSInputType;
5109 ExtendKind RHSExtendKind = LHSExtendKind;
5112 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5115 return ExtendedReductionOperand{
5116 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5123static std::optional<SmallVector<VPPartialReductionChain>>
5130 return std::nullopt;
5140 VPValue *CurrentValue = ExitValue;
5141 while (CurrentValue != RedPhiR) {
5143 std::optional<unsigned> BlendReductionIdx;
5147 return std::nullopt;
5149 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5150 if (!BlendReductionIdx)
5151 return std::nullopt;
5158 return std::nullopt;
5165 std::optional<ExtendedReductionOperand> ExtendedOp =
5166 matchExtendedReductionOperand(UpdateR,
Op);
5168 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5170 return std::nullopt;
5178 return std::nullopt;
5180 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5183 return std::nullopt;
5185 VPPartialReductionChain Link(
5186 {UpdateR, *ExtendedOp, RK,
5191 CurrentValue = PrevValue;
5196 std::reverse(Chain.
begin(), Chain.
end());
5215 if (
auto Chains = getScaledReductions(RedPhiR))
5216 ChainsByPhi.
try_emplace(RedPhiR, std::move(*Chains));
5219 if (ChainsByPhi.
empty())
5227 for (
const auto &[
_, Chains] : ChainsByPhi)
5228 for (
const VPPartialReductionChain &Chain : Chains) {
5229 PartialReductionOps.
insert(Chain.ExtendedOp.ExtendsUser);
5231 PartialReductionBlends.
insert(Chain.Blend);
5232 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5238 auto ExtendUsersValid = [&](
VPValue *Ext) {
5240 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5244 auto IsProfitablePartialReductionChainForVF =
5251 for (
const VPPartialReductionChain &Link : Chain) {
5252 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5253 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5257 PartialCost += LinkCost;
5258 RegularCost += Link.ReductionBinOp->
computeCost(VF, CostCtx);
5260 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5261 RegularCost += ExtendedOp.ExtendsUser->
computeCost(VF, CostCtx);
5264 RegularCost += Extend->computeCost(VF, CostCtx);
5266 return PartialCost.
isValid() && PartialCost < RegularCost;
5274 for (
auto &[RedPhiR, Chains] : ChainsByPhi) {
5275 for (
const VPPartialReductionChain &Chain : Chains) {
5276 if (!
all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5280 auto UseIsValid = [&, RedPhiR = RedPhiR](
VPUser *U) {
5282 return PhiR == RedPhiR;
5286 return Blend == Chain.Blend || PartialReductionBlends.
contains(Blend);
5288 return Chain.ScaleFactor == ScaledReductionMap.
lookup_or(R, 0) ||
5294 if (!
all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5303 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5304 return RepR && RepR->getOpcode() == Instruction::Store;
5315 return IsProfitablePartialReductionChainForVF(Chains, VF);
5321 for (
auto &[Phi, Chains] : ChainsByPhi)
5322 for (
const VPPartialReductionChain &Chain : Chains)
5323 transformToPartialReduction(Chain, Plan, Phi);
5338 if (VPI && VPI->getUnderlyingValue() &&
5349 auto ProcessSubset = [&](
VPlan &,
auto ProcessVPInst) {
5352 if (!ProcessVPInst(VPI))
5361 assert(New->getParent() &&
"New recipe must have been inserted");
5362 if (VPI->
getOpcode() == Instruction::Load)
5371 return ReplaceWith(VPI,
VPBuilder(VPI).insert(
5378 "lowerMemoryIdioms", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5380 VPI, FinalRedStoresBuilder))
5389 return ReplaceWith(VPI,
VPBuilder(VPI).insert(Histogram));
5402 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5406 return Scalarize(VPI);
5413 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5415 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5425 const SCEV *PtrSCEV =
5427 bool IsSingleScalarLoad =
5433 I, Ptr, IsSingleScalarLoad,
5442 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](
VPInstruction *VPI) {
5444 bool IsLoad = VPI->
getOpcode() == Instruction::Load;
5448 std::optional<int64_t> Stride =
5450 if (Stride != 1 && Stride != -1)
5481 return ReplaceWith(VPI,
Load);
5490 auto *StoreR = Builder.createWidenStore(
5493 return ReplaceWith(VPI, StoreR);
5500 return ReplaceWith(VPI, Recipe);
5502 return Scalarize(VPI);
5525 if (VPI->mayHaveSideEffects())
5529 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5534 if (VPI->getOpcode() == Instruction::Add &&
5543 VPI->getOpcode(), VPI->operandsWithoutMask(),
nullptr, *VPI,
5544 *VPI, VPI->getDebugLoc(),
I);
5545 Recipe->insertBefore(VPI);
5546 VPI->replaceAllUsesWith(Recipe);
5547 VPI->eraseFromParent();
5557 switch (Param.ParamKind) {
5558 case VFParamKind::Vector:
5559 case VFParamKind::GlobalPredicate:
5561 case VFParamKind::OMP_Uniform:
5562 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5563 SE->isLoopInvariant(
5564 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5566 case VFParamKind::OMP_Linear:
5567 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5568 m_scev_AffineAddRec(
5569 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5570 m_SpecificLoop(L)));
5587 const auto *It =
find_if(Mappings, [&](
const VFInfo &Info) {
5588 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5591 if (It == Mappings.end())
5598struct CallWideningDecision {
5599 enum class KindTy { Scalarize,
Intrinsic, VectorVariant };
5600 CallWideningDecision(KindTy Kind,
Function *Variant =
nullptr)
5623 return CallWideningDecision::KindTy::Scalarize;
5633 return CallWideningDecision::KindTy::Scalarize;
5637 false, VF, CostCtx);
5652 return CallWideningDecision::KindTy::Intrinsic;
5656 if (VecFunc && ScalarCost >= VecCallCost)
5657 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5659 return CallWideningDecision::KindTy::Scalarize;
5669 if (!VPI || !VPI->getUnderlyingValue() ||
5670 VPI->getOpcode() != Instruction::Call)
5675 VPI->op_begin() + CI->arg_size());
5677 CallWideningDecision Decision =
5686 switch (Decision.Kind) {
5687 case CallWideningDecision::KindTy::Intrinsic: {
5691 *VPI, VPI->getDebugLoc());
5694 case CallWideningDecision::KindTy::VectorVariant: {
5698 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.
getTrue();
5699 Ops.push_back(Mask);
5701 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5703 *VPI, VPI->getDebugLoc());
5706 case CallWideningDecision::KindTy::Scalarize:
5712 VPI->replaceAllUsesWith(Replacement);
5713 VPI->eraseFromParent();
5735 if (!MemR || MemR->isConsecutive())
5738 VPValue *Ptr = MemR->getAddr();
5750 VPValue *StoredValue =
nullptr;
5754 StoredValue = StoreR->getStoredValue();
5756 IntrinID = Intrinsic::experimental_vp_strided_store;
5760 IntrinID = Intrinsic::experimental_vp_strided_load;
5763 Align Alignment = MemR->getAlign();
5766 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
5771 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
5772 return StridedLoadStoreCost < CurrentCost;
5783 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
5788 I32VF = Builder.createScalarZExtOrTrunc(
5802 "Stride type from SCEV must match the index type");
5803 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5806 auto *
Offset = Builder.createOverflowingOp(
5807 Instruction::Mul, {CanIV, StrideInBytes},
5808 {AddRecPtr->hasNoUnsignedWrap(),
false});
5812 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV,
Offset, NWFlags);
5815 VPValue *NewPtr = Builder.createVectorPointer(
5819 VPValue *Mask = MemR->getMask();
5824 Ops.push_back(StoredValue);
5825 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
5827 auto *StridedR = Builder.createWidenMemIntrinsic(
5830 *MemR, R.getDebugLoc());
5833 R.eraseFromParent();
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
iv Induction Variable Users
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Legalize the Machine IR a function s Machine IR
This file provides utility analysis objects describing memory locations.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
static const uint32_t IV[8]
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
unsigned getActiveBits() const
Compute the number of active bits in the value.
APInt abs() const
Get the absolute value.
unsigned getBitWidth() const
Return the number of bits in the APInt.
int32_t exactLogBase2() const
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
An arbitrary precision integer that knows its signedness.
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
const T & back() const
Get the last element.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
const Function * getParent() const
Return the enclosing method, or null if none.
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
@ ICMP_ULE
unsigned less or equal
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
static DebugLoc getUnknown()
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
static constexpr ElementCount getScalable(ScalarTy MinVal)
constexpr bool isScalar() const
Exactly one element.
Convenience struct for specifying and reasoning about fast-math flags.
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Represents a single loop in the control flow graph.
This class implements a map that also provides access to all stored values in a deterministic order.
ValueT lookup(const KeyT &Key) const
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
unsigned getOpcode() const
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
size_type size() const
Determine the number of elements in the SetVector.
bool insert(const value_type &X)
Insert a new element into the SetVector.
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
The instances of the Type class are immutable: once they are created, they are never changed.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
bool isPointerTy() const
True if this is an instance of PointerType.
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
iterator begin()
Recipe iterator methods.
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
const VPRecipeBase & front() const
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
const VPRecipeBase & back() const
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
VPRegionBlock * getParent()
const VPBasicBlock * getExitingBasicBlock() const
size_t getNumSuccessors() const
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
const VPBlocksTy & getPredecessors() const
VPBlockBase * getSinglePredecessor() const
const VPBasicBlock * getEntryBasicBlock() const
VPBlockBase * getSingleSuccessor() const
const VPBlocksTy & getSuccessors() const
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
A special type of VPBasicBlock that wraps an existing IR basic block.
Class to record and manage LLVM IR flags.
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
This is a concrete Recipe that models a single VPlan-level instruction.
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
@ ExtractPenultimateElement
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
@ BuildVector
Creates a fixed-width vector containing all operands.
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
unsigned getOpcode() const
VPValue * getMask() const
Returns the mask for the VPInstruction.
const InterleaveGroup< Instruction > * getInterleaveGroup() const
VPValue * getMask() const
Return the mask used by this recipe.
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
VPRegionBlock * getRegion()
VPBasicBlock * getParent()
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
A recipe for handling reduction phis.
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
A recipe to represent inloop, ordered or partial reduction operations.
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
const VPBlockBase * getEntry() const
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
const VPBlockBase * getExiting() const
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
bool isPredicated() const
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Lightweight SCEV-to-VPlan expander.
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
void setOperand(unsigned I, VPValue *New)
unsigned getNumOperands() const
VPValue * getOperand(unsigned N) const
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
void replaceAllUsesWith(VPValue *New)
unsigned getNumUsers() const
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
A recipe for widening Call instructions using library calls.
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Instruction::CastOps getOpcode() const
A recipe for handling GEP instructions.
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
VPIRValue * getStartValue() const
Returns the start value of the induction.
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
VPValue * getStepValue()
Returns the step value of the induction.
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
A recipe for widening vector intrinsics.
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
unsigned getOpcode() const
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
bool hasVF(ElementCount VF) const
const DataLayout & getDataLayout() const
LLVMContext & getContext() const
VPBasicBlock * getEntry()
bool hasScalableVF() const
VPValue * getTripCount() const
The trip count of the original loop.
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
bool hasUF(unsigned UF) const
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
VPSymbolicValue & getVectorTripCount()
The vector trip count.
VPValue * getBackedgeTakenCount() const
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
void setVF(ElementCount VF)
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
bool hasScalarVFOnly() const
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
LLVM Value Representation.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
constexpr ScalarTy getFixedValue() const
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
constexpr LeafTy multiplyCoefficientBy(ScalarTy RHS) const
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_isa< To... > m_Isa()
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_False()
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_True()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
NodeAddr< DefNode * > Def
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPIRValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
void stable_sort(R &&Range)
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
auto cast_or_null(const Y &Val)
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
auto dyn_cast_or_null(const Y &Val)
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
auto reverse(ContainerTy &&C)
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
void sort(IteratorTy Start, IteratorTy End)
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Type * getType() const
Returns the type of the underlying IR value.
A recipe for widening load operations, using the address to load from and an optional mask.
A recipe for widening store operations, using the stored value, the address to store to and an option...