LLVM 24.0.0git
VPlanTransforms.cpp
Go to the documentation of this file.
1//===-- VPlanTransforms.cpp - Utility VPlan to VPlan transforms -----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file implements a set of utility VPlan to VPlan transformations.
11///
12//===----------------------------------------------------------------------===//
13
14#include "VPlanTransforms.h"
15#include "VPRecipeBuilder.h"
16#include "VPlan.h"
17#include "VPlanAnalysis.h"
18#include "VPlanCFG.h"
19#include "VPlanDominatorTree.h"
20#include "VPlanHelpers.h"
21#include "VPlanPatternMatch.h"
22#include "VPlanUtils.h"
23#include "llvm/ADT/APInt.h"
25#include "llvm/ADT/STLExtras.h"
26#include "llvm/ADT/SetVector.h"
28#include "llvm/ADT/TypeSwitch.h"
30#include "llvm/Analysis/Loads.h"
36#include "llvm/IR/Intrinsics.h"
37#include "llvm/IR/Metadata.h"
41
42using namespace llvm;
43using namespace VPlanPatternMatch;
44using namespace SCEVPatternMatch;
45
46/// If the pointer operand \p Addr of a memory access is an affine AddRec
47/// w.r.t. \p L with a constant stride, return the stride in units of
48/// \p AccessTy. Otherwise return std::nullopt.
49static std::optional<int64_t> getConstantStride(VPValue *Addr, Type *AccessTy,
51 const Loop *L) {
52 assert(!hasIrregularType(AccessTy, L->getHeader()->getDataLayout()) &&
53 "should not try to widen irregular types");
54 const SCEV *AddrSCEV = vputils::getSCEVExprForVPValue(Addr, PSE, L);
55 auto *AddRec = dyn_cast<SCEVAddRecExpr>(AddrSCEV);
56 if (!AddRec)
57 return {};
58
59 return getStrideFromAddRec(AddRec, L, AccessTy, /*Ptr=*/nullptr, PSE);
60}
61
64 Loop *OuterLoop) {
65
66 // Returns true if the access of \p AccessTy at \p Addr can be widened to a
67 // consecutive vector access.
68 auto IsConsecutiveAccess = [&](VPValue *Addr, Type *AccessTy) {
69 return !hasIrregularType(AccessTy, Plan.getDataLayout()) &&
70 getConstantStride(Addr, AccessTy, PSE, OuterLoop) == 1;
71 };
72
74 Plan.getVectorLoopRegion());
76 // Skip blocks outside region
77 if (!VPBB->getParent())
78 break;
79 VPRecipeBase *Term = VPBB->getTerminator();
80 auto EndIter = Term ? Term->getIterator() : VPBB->end();
81 // Introduce each ingredient into VPlan.
82 for (VPRecipeBase &Ingredient :
83 make_early_inc_range(make_range(VPBB->begin(), EndIter))) {
84
85 VPValue *VPV = Ingredient.getVPSingleValue();
86 if (!VPV->getUnderlyingValue())
87 continue;
88
90
91 // Atomic accesses and fences have ordering/atomicity semantics that
92 // cannot be preserved by lane-wise widening.
94 return false;
95
96 VPRecipeBase *NewRecipe = nullptr;
97 if (auto *PhiR = dyn_cast<VPPhi>(&Ingredient)) {
98 auto *Phi = cast<PHINode>(PhiR->getUnderlyingValue());
99 NewRecipe = new VPWidenPHIRecipe(PhiR->operands(), PhiR->getDebugLoc(),
100 Phi->getName());
101 } else if (auto *VPI = dyn_cast<VPInstruction>(&Ingredient)) {
102 assert(!isa<PHINode>(Inst) && "phis should be handled above");
103 // Create VPWidenMemoryRecipe for loads and stores.
104 if (LoadInst *Load = dyn_cast<LoadInst>(Inst)) {
105 bool IsConsecutive =
106 IsConsecutiveAccess(VPI->getOperand(0), VPI->getScalarType());
107 NewRecipe = new VPWidenLoadRecipe(*Load, Ingredient.getOperand(0),
108 nullptr /*Mask*/, IsConsecutive,
109 *VPI, Ingredient.getDebugLoc());
110 } else if (StoreInst *Store = dyn_cast<StoreInst>(Inst)) {
111 bool IsConsecutive = IsConsecutiveAccess(
112 VPI->getOperand(1), VPI->getOperand(0)->getScalarType());
113 NewRecipe = new VPWidenStoreRecipe(
114 *Store, Ingredient.getOperand(1), Ingredient.getOperand(0),
115 nullptr /*Mask*/, IsConsecutive, *VPI, Ingredient.getDebugLoc());
117 NewRecipe = new VPWidenGEPRecipe(GEP->getSourceElementType(),
118 Ingredient.operands(), *VPI,
119 Ingredient.getDebugLoc(), GEP);
120 } else if (CallInst *CI = dyn_cast<CallInst>(Inst)) {
121 Intrinsic::ID VectorID = getVectorIntrinsicIDForCall(CI, &TLI);
122 if (VectorID == Intrinsic::not_intrinsic)
123 return false;
124
125 // The noalias.scope.decl intrinsic declares a noalias scope that
126 // is valid for a single iteration. Emitting it as a single-scalar
127 // replicate would incorrectly extend the scope across multiple
128 // original iterations packed into one vector iteration.
129 // FIXME: If we want to vectorize this loop, then we have to drop
130 // all the associated !alias.scope and !noalias.
131 if (VectorID == Intrinsic::experimental_noalias_scope_decl)
132 return false;
133
134 // These intrinsics are recognized by getVectorIntrinsicIDForCall
135 // but are not widenable. Emit them as replicate instead of widening.
136 if (VectorID == Intrinsic::assume ||
137 VectorID == Intrinsic::lifetime_end ||
138 VectorID == Intrinsic::lifetime_start ||
139 VectorID == Intrinsic::sideeffect ||
140 VectorID == Intrinsic::pseudoprobe) {
141 // If the operand of llvm.assume holds before vectorization, it will
142 // also hold per lane.
143 // llvm.pseudoprobe requires to be duplicated per lane for accurate
144 // sample count.
145 const bool IsSingleScalar = VectorID != Intrinsic::assume &&
146 VectorID != Intrinsic::pseudoprobe;
147 NewRecipe = new VPReplicateRecipe(CI, Ingredient.operands(),
148 /*IsSingleScalar=*/IsSingleScalar,
149 /*Mask=*/nullptr, *VPI, *VPI,
150 Ingredient.getDebugLoc());
151 } else {
152 NewRecipe = new VPWidenIntrinsicRecipe(
153 *CI, VectorID, drop_end(Ingredient.operands()), CI->getType(),
154 VPIRFlags(*CI), *VPI, CI->getDebugLoc());
155 }
156 } else if (auto *CI = dyn_cast<CastInst>(Inst)) {
157 NewRecipe = new VPWidenCastRecipe(
158 CI->getOpcode(), Ingredient.getOperand(0), CI->getType(), CI,
159 VPIRFlags(*CI), VPIRMetadata(*CI));
160 } else {
161 NewRecipe = new VPWidenRecipe(*Inst, Ingredient.operands(), *VPI,
162 *VPI, Ingredient.getDebugLoc());
163 }
164 } else {
166 "inductions must be created earlier");
167 continue;
168 }
169
170 NewRecipe->insertBefore(&Ingredient);
171 if (NewRecipe->getNumDefinedValues() == 1)
172 VPV->replaceAllUsesWith(NewRecipe->getVPSingleValue());
173 else
174 assert(NewRecipe->getNumDefinedValues() == 0 &&
175 "Only recpies with zero or one defined values expected");
176 Ingredient.eraseFromParent();
177 }
178 }
179 return true;
180}
181
182/// Helper for extra no-alias checks via known-safe recipe and SCEV.
185 VPReplicateRecipe &GroupLeader;
186 PredicatedScalarEvolution *PSE = nullptr;
187 const Loop *L = nullptr;
188
189 // Return true if \p A and \p B are known to not alias for all VFs in the
190 // plan, checked via the distance between the accesses
191 bool isNoAliasViaDistance(VPReplicateRecipe *A, VPReplicateRecipe *B) const {
192 if (A->getOpcode() != Instruction::Store ||
193 B->getOpcode() != Instruction::Store)
194 return false;
195
196 if (!PSE || !L)
197 return A == B;
198
199 VPValue *AddrA = A->getOperand(1);
200 const SCEV *SCEVA = vputils::getSCEVExprForVPValue(AddrA, *PSE, L);
201 VPValue *AddrB = B->getOperand(1);
202 const SCEV *SCEVB = vputils::getSCEVExprForVPValue(AddrB, *PSE, L);
204 return false;
205
206 const APInt *Distance;
207 ScalarEvolution &SE = *PSE->getSE();
208 if (!match(SE.getMinusSCEV(SCEVA, SCEVB), m_scev_APInt(Distance)))
209 return false;
210
211 const DataLayout &DL = SE.getDataLayout();
212 Type *TyA = A->getOperand(0)->getScalarType();
213 uint64_t SizeA = DL.getTypeStoreSize(TyA);
214 Type *TyB = B->getOperand(0)->getScalarType();
215 uint64_t SizeB = DL.getTypeStoreSize(TyB);
216
217 // Use the maximum store size to ensure no overlap from either direction.
218 // Currently only handles fixed sizes, as it is only used for
219 // replicating VPReplicateRecipes.
220 uint64_t MaxStoreSize = std::max(SizeA, SizeB);
221
222 auto VFs = B->getParent()->getPlan()->vectorFactors();
224 if (MaxVF.isScalable())
225 return false;
226 return Distance->abs().uge(
227 MaxVF.multiplyCoefficientBy(MaxStoreSize).getFixedValue());
228 }
229
230public:
233 const Loop &L)
234 : ExcludeRecipes(ExcludeRecipes.begin(), ExcludeRecipes.end()),
235 GroupLeader(GroupLeader), PSE(&PSE), L(&L) {}
236
237 SinkStoreInfo(VPReplicateRecipe &GroupLeader) : GroupLeader(GroupLeader) {}
238
239 /// Return true if \p R should be skipped during alias checking, either
240 /// because it's in the exclude set or because no-alias can be proven via
241 /// SCEV.
242 bool shouldSkip(VPRecipeBase &R) const {
244 return ExcludeRecipes.contains(Store) ||
245 (Store && isNoAliasViaDistance(Store, &GroupLeader));
246 }
247};
248
249/// Check if a memory operation doesn't alias with memory operations using
250/// scoped noalias metadata, in blocks in the single-successor chain between \p
251/// FirstBB and \p LastBB. If \p SinkInfo is std::nullopt, only recipes that may
252/// write to memory are checked (for load hoisting). Otherwise recipes that both
253/// read and write memory are checked, and SCEV is used to prove no-alias
254/// between the group leader and other replicate recipes (for store sinking).
255static bool
257 VPBasicBlock *FirstBB, VPBasicBlock *LastBB,
258 std::optional<SinkStoreInfo> SinkInfo = {}) {
259 bool CheckReads = SinkInfo.has_value();
260 for (VPBasicBlock *VPBB :
262 for (VPRecipeBase &R : *VPBB) {
263 if (SinkInfo && SinkInfo->shouldSkip(R))
264 continue;
265
266 // Skip recipes that don't need checking.
267 if (!R.mayWriteToMemory() && !(CheckReads && R.mayReadFromMemory()))
268 continue;
269
271 if (!Loc)
272 // Conservatively assume aliasing for memory operations without
273 // location.
274 return false;
275
277 return false;
278 }
279 }
280 return true;
281}
282
283/// Get the value type of the replicate load or store. \p IsLoad indicates
284/// whether it is a load.
286 return (IsLoad ? R : R->getOperand(0))->getScalarType();
287}
288
289/// Collect either replicated Loads or Stores grouped by their address SCEV and
290/// their load-store type, in a deep-traversal of the vector loop region in \p
291/// Plan.
292template <unsigned Opcode>
295 VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L,
296 function_ref<bool(VPReplicateRecipe *)> FilterFn) {
297 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
298 "Only Load and Store opcodes supported");
299 constexpr bool IsLoad = (Opcode == Instruction::Load);
302 RecipesByAddressAndType;
305 for (VPRecipeBase &R : *VPBB) {
306 auto *RepR = dyn_cast<VPReplicateRecipe>(&R);
307 if (!RepR || RepR->getOpcode() != Opcode || !FilterFn(RepR))
308 continue;
309
310 // For loads, operand 0 is address; for stores, operand 1 is address.
311 VPValue *Addr = RepR->getOperand(IsLoad ? 0 : 1);
312 const Type *LoadStoreTy = getLoadStoreValueType(RepR, IsLoad);
313 const SCEV *AddrSCEV = vputils::getSCEVExprForVPValue(Addr, PSE, L);
314 if (!isa<SCEVCouldNotCompute>(AddrSCEV))
315 RecipesByAddressAndType[{AddrSCEV, LoadStoreTy}].push_back(RepR);
316 }
317 }
318 auto Groups = to_vector(RecipesByAddressAndType.values());
319 VPDominatorTree VPDT(Plan);
320 for (auto &Group : Groups) {
321 // Sort mem ops by dominance order, with earliest (most dominating) first.
323 return VPDT.properlyDominates(A, B);
324 });
325 }
326 return Groups;
327}
328
329static bool sinkScalarOperands(VPlan &Plan) {
330 auto Iter = vp_depth_first_deep(Plan.getEntry());
331 bool ScalarVFOnly = Plan.hasScalarVFOnly();
332 bool Changed = false;
333
335 auto InsertIfValidSinkCandidate = [ScalarVFOnly, &WorkList](
336 VPBasicBlock *SinkTo, VPValue *Op) {
337 auto *Candidate = dyn_cast<VPSingleDefRecipe>(Op);
339 VPInstruction>(Candidate))
340 return;
341
342 if (Candidate->getParent() == SinkTo ||
343 all_of(Candidate->operands(),
344 [](VPValue *Op) { return Op->isDefinedOutsideLoopRegions(); }) ||
345 vputils::cannotHoistOrSinkRecipe(*Candidate, /*Sinking=*/true))
346 return;
347
348 if (!ScalarVFOnly && !vputils::doesGeneratePerAllLanes(Candidate))
349 return;
350
351 // Only single-scalar VPInstructions can be sunk.
352 if (auto *VPI = dyn_cast<VPInstruction>(Candidate))
353 if (!vputils::isSingleScalar(VPI))
354 return;
355
356 WorkList.insert({SinkTo, Candidate});
357 };
358
359 // First, collect the operands of all recipes in replicate blocks as seeds for
360 // sinking.
362 VPBasicBlock *EntryVPBB = VPR->getEntryBasicBlock();
363 if (!VPR->isReplicator() || EntryVPBB->getSuccessors().size() != 2)
364 continue;
365 VPBasicBlock *VPBB = cast<VPBasicBlock>(EntryVPBB->getSuccessors().front());
366 if (VPBB->getSingleSuccessor() != VPR->getExitingBasicBlock())
367 continue;
368 for (auto &Recipe : *VPBB)
369 for (VPValue *Op : Recipe.operands())
370 InsertIfValidSinkCandidate(VPBB, Op);
371 }
372
373 // Try to sink each replicate or scalar IV steps recipe in the worklist.
374 for (unsigned I = 0; I != WorkList.size(); ++I) {
375 VPBasicBlock *SinkTo;
376 VPSingleDefRecipe *SinkCandidate;
377 std::tie(SinkTo, SinkCandidate) = WorkList[I];
378
379 // All recipe users of SinkCandidate must be in the same block SinkTo or all
380 // users outside of SinkTo must only use the first lane of SinkCandidate. In
381 // the latter case, we need to duplicate SinkCandidate.
382 auto UsersOutsideSinkTo =
383 make_filter_range(SinkCandidate->users(), [SinkTo](VPUser *U) {
384 return cast<VPRecipeBase>(U)->getParent() != SinkTo;
385 });
386 if (any_of(UsersOutsideSinkTo, [SinkCandidate](VPUser *U) {
387 return !U->usesFirstLaneOnly(SinkCandidate);
388 }))
389 continue;
390 bool NeedsDuplicating = !UsersOutsideSinkTo.empty();
391
392 if (NeedsDuplicating) {
393 if (ScalarVFOnly)
394 continue;
395 VPSingleDefRecipe *Clone;
396 if (auto *SinkCandidateRepR =
397 dyn_cast<VPReplicateRecipe>(SinkCandidate)) {
398 // TODO: Handle converting to uniform recipes as separate transform,
399 // then cloning should be sufficient here.
401 SinkCandidateRepR->getOpcode(), SinkCandidate->operands(),
402 /*Mask=*/nullptr, *SinkCandidateRepR, *SinkCandidateRepR,
403 SinkCandidate->getDebugLoc(), SinkCandidate->getUnderlyingInstr());
404 // TODO: add ".cloned" suffix to name of Clone's VPValue.
405 } else {
406 Clone = SinkCandidate->clone();
407 }
408
409 Clone->insertBefore(SinkCandidate);
410 SinkCandidate->replaceUsesWithIf(Clone, [SinkTo](VPUser &U, unsigned) {
411 return cast<VPRecipeBase>(&U)->getParent() != SinkTo;
412 });
413 }
414 SinkCandidate->moveBefore(*SinkTo, SinkTo->getFirstNonPhi());
415 for (VPValue *Op : SinkCandidate->operands())
416 InsertIfValidSinkCandidate(SinkTo, Op);
417 Changed = true;
418 }
419 return Changed;
420}
421
422/// If \p R is a triangle region, return the 'then' block of the triangle.
424 auto *EntryBB = cast<VPBasicBlock>(R->getEntry());
425 if (EntryBB->getNumSuccessors() != 2)
426 return nullptr;
427
428 auto *Succ0 = dyn_cast<VPBasicBlock>(EntryBB->getSuccessors()[0]);
429 auto *Succ1 = dyn_cast<VPBasicBlock>(EntryBB->getSuccessors()[1]);
430 if (!Succ0 || !Succ1)
431 return nullptr;
432
433 if (Succ0->getNumSuccessors() + Succ1->getNumSuccessors() != 1)
434 return nullptr;
435 if (Succ0->getSingleSuccessor() == Succ1)
436 return Succ0;
437 if (Succ1->getSingleSuccessor() == Succ0)
438 return Succ1;
439 return nullptr;
440}
441
442// Merge replicate regions in their successor region, if a replicate region
443// is connected to a successor replicate region with the same predicate by a
444// single, empty VPBasicBlock.
446 SmallPtrSet<VPRegionBlock *, 4> TransformedRegions;
447
448 // Collect replicate regions followed by an empty block, followed by another
449 // replicate region with matching masks to process front. This is to avoid
450 // iterator invalidation issues while merging regions.
453 vp_depth_first_deep(Plan.getEntry()))) {
454 if (!Region1->isReplicator())
455 continue;
456 auto *MiddleBasicBlock =
457 dyn_cast_or_null<VPBasicBlock>(Region1->getSingleSuccessor());
458 if (!MiddleBasicBlock || !MiddleBasicBlock->empty())
459 continue;
460
461 auto *Region2 =
462 dyn_cast_or_null<VPRegionBlock>(MiddleBasicBlock->getSingleSuccessor());
463 if (!Region2 || !Region2->isReplicator())
464 continue;
465
466 VPValue *Mask1 = Region1->getEntryBranchOnMask()->getOperand(0);
467 VPValue *Mask2 = Region2->getEntryBranchOnMask()->getOperand(0);
468 if (!Mask1 || Mask1 != Mask2)
469 continue;
470
471 assert(Mask1 && Mask2 && "both region must have conditions");
472 WorkList.push_back(Region1);
473 }
474
475 // Move recipes from Region1 to its successor region, if both are triangles.
476 for (VPRegionBlock *Region1 : WorkList) {
477 if (TransformedRegions.contains(Region1))
478 continue;
479 auto *MiddleBasicBlock = cast<VPBasicBlock>(Region1->getSingleSuccessor());
480 auto *Region2 = cast<VPRegionBlock>(MiddleBasicBlock->getSingleSuccessor());
481
482 VPBasicBlock *Then1 = getPredicatedThenBlock(Region1);
483 VPBasicBlock *Then2 = getPredicatedThenBlock(Region2);
484 if (!Then1 || !Then2)
485 continue;
486
487 // Note: No fusion-preventing memory dependencies are expected in either
488 // region. Such dependencies should be rejected during earlier dependence
489 // checks, which guarantee accesses can be re-ordered for vectorization.
490 //
491 // Move recipes to the successor region.
492 for (VPRecipeBase &ToMove : make_early_inc_range(reverse(*Then1)))
493 ToMove.moveBefore(*Then2, Then2->getFirstNonPhi());
494
495 auto *Merge1 = cast<VPBasicBlock>(Then1->getSingleSuccessor());
496 auto *Merge2 = cast<VPBasicBlock>(Then2->getSingleSuccessor());
497
498 // Move VPPredInstPHIRecipes from the merge block to the successor region's
499 // merge block. Update all users inside the successor region to use the
500 // original values.
501 for (VPRecipeBase &Phi1ToMove : make_early_inc_range(reverse(*Merge1))) {
502 VPValue *PredInst1 =
503 cast<VPPredInstPHIRecipe>(&Phi1ToMove)->getOperand(0);
504 VPValue *Phi1ToMoveV = Phi1ToMove.getVPSingleValue();
505 Phi1ToMoveV->replaceUsesWithIf(PredInst1, [Then2](VPUser &U, unsigned) {
506 return cast<VPRecipeBase>(&U)->getParent() == Then2;
507 });
508
509 // Remove phi recipes that are unused after merging the regions.
510 if (Phi1ToMove.getVPSingleValue()->user_empty()) {
511 Phi1ToMove.eraseFromParent();
512 continue;
513 }
514 Phi1ToMove.moveBefore(*Merge2, Merge2->begin());
515 }
516
517 // Remove the dead recipes in Region1's entry block.
518 for (VPRecipeBase &R :
519 make_early_inc_range(reverse(*Region1->getEntryBasicBlock())))
520 R.eraseFromParent();
521
522 // Finally, remove the first region.
523 for (VPBlockBase *Pred : make_early_inc_range(Region1->getPredecessors())) {
524 VPBlockUtils::disconnectBlocks(Pred, Region1);
525 VPBlockUtils::connectBlocks(Pred, MiddleBasicBlock);
526 }
527 VPBlockUtils::disconnectBlocks(Region1, MiddleBasicBlock);
528 TransformedRegions.insert(Region1);
529 }
530
531 return !TransformedRegions.empty();
532}
533
535 VPRegionBlock *ParentRegion,
536 VPlan &Plan) {
537 Instruction *Instr = PredRecipe->getUnderlyingInstr();
538 // Build the triangular if-then region.
539 std::string RegionName = (Twine("pred.") + Instr->getOpcodeName()).str();
540 assert(Instr->getParent() && "Predicated instruction not in any basic block");
541 auto *BlockInMask = PredRecipe->getMask();
542 auto *MaskDef = BlockInMask->getDefiningRecipe();
543 auto *BOMRecipe = new VPBranchOnMaskRecipe(
544 BlockInMask, MaskDef ? MaskDef->getDebugLoc() : DebugLoc::getUnknown());
545 auto *Entry =
546 Plan.createVPBasicBlock(Twine(RegionName) + ".entry", BOMRecipe);
547
548 // Replace predicated replicate recipe with a replicate recipe without a
549 // mask but in the replicate region.
550 auto *RecipeWithoutMask = new VPReplicateRecipe(
551 PredRecipe->getUnderlyingInstr(), PredRecipe->operandsWithoutMask(),
552 PredRecipe->isSingleScalar(), nullptr /*Mask*/, *PredRecipe, *PredRecipe,
553 PredRecipe->getDebugLoc());
554 auto *Pred =
555 Plan.createVPBasicBlock(Twine(RegionName) + ".if", RecipeWithoutMask);
556 auto *Exiting = Plan.createVPBasicBlock(Twine(RegionName) + ".continue");
558 Plan.createReplicateRegion(Entry, Exiting, RegionName);
559
560 // Note: first set Entry as region entry and then connect successors starting
561 // from it in order, to propagate the "parent" of each VPBasicBlock.
562 Region->setParent(ParentRegion);
563 VPBlockUtils::insertTwoBlocksAfter(Pred, Exiting, Entry);
564 VPBlockUtils::connectBlocks(Pred, Exiting);
565
566 if (!PredRecipe->user_empty()) {
567 auto *PHIRecipe = new VPPredInstPHIRecipe(RecipeWithoutMask,
568 RecipeWithoutMask->getDebugLoc());
569 Exiting->appendRecipe(PHIRecipe);
570 PredRecipe->replaceAllUsesWith(PHIRecipe);
571 }
572 PredRecipe->eraseFromParent();
573 return Region;
574}
575
576static void addReplicateRegions(VPlan &Plan) {
579 vp_depth_first_deep(Plan.getEntry()))) {
580 for (VPRecipeBase &R : *VPBB)
581 if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
582 if (RepR->isPredicated())
583 WorkList.push_back(RepR);
584 }
585 }
586
587 unsigned BBNum = 0;
588 for (VPReplicateRecipe *RepR : WorkList) {
589 VPBasicBlock *CurrentBlock = RepR->getParent();
590 VPBasicBlock *SplitBlock = CurrentBlock->splitAt(RepR->getIterator());
591
592 BasicBlock *OrigBB = RepR->getUnderlyingInstr()->getParent();
593 SplitBlock->setName(
594 OrigBB->hasName() ? OrigBB->getName() + "." + Twine(BBNum++) : "");
595 // Record predicated instructions for above packing optimizations.
597 createReplicateRegion(RepR, CurrentBlock->getParent(), Plan);
599
600 VPRegionBlock *ParentRegion = Region->getParent();
601 if (ParentRegion && ParentRegion->getExiting() == CurrentBlock)
602 ParentRegion->setExiting(SplitBlock);
603 }
604}
605
609 vp_depth_first_deep(Plan.getEntry()))) {
610 // Don't fold the blocks in the skeleton of the Plan into their single
611 // predecessors for now.
612 // TODO: Remove restriction once more of the skeleton is modeled in VPlan.
613 if (!VPBB->getParent())
614 continue;
615 auto *PredVPBB =
616 dyn_cast_or_null<VPBasicBlock>(VPBB->getSinglePredecessor());
617 if (!PredVPBB || PredVPBB->getNumSuccessors() != 1 ||
618 isa<VPIRBasicBlock>(PredVPBB))
619 continue;
620 WorkList.push_back(VPBB);
621 }
622
623 for (VPBasicBlock *VPBB : WorkList) {
624 VPBasicBlock *PredVPBB = cast<VPBasicBlock>(VPBB->getSinglePredecessor());
625 for (VPRecipeBase &R : make_early_inc_range(*VPBB))
626 R.moveBefore(*PredVPBB, PredVPBB->end());
627 VPBlockUtils::disconnectBlocks(PredVPBB, VPBB);
628 auto *ParentRegion = VPBB->getParent();
629 if (ParentRegion && ParentRegion->getExiting() == VPBB)
630 ParentRegion->setExiting(PredVPBB);
631 VPBlockUtils::transferSuccessors(VPBB, PredVPBB);
632 // VPBB is now dead and will be cleaned up when the plan gets destroyed.
633 }
634 return !WorkList.empty();
635}
636
638 // Convert masked VPReplicateRecipes to if-then region blocks.
640
641 bool ShouldSimplify = true;
642 while (ShouldSimplify) {
643 ShouldSimplify = sinkScalarOperands(Plan);
644 ShouldSimplify |= mergeReplicateRegionsIntoSuccessors(Plan);
645 ShouldSimplify |= mergeBlocksIntoPredecessors(Plan);
646 }
647}
648
649/// Remove redundant casts of inductions.
650///
651/// Such redundant casts are casts of induction variables that can be ignored,
652/// because we already proved that the casted phi is equal to the uncasted phi
653/// in the vectorized loop. There is no need to vectorize the cast - the same
654/// value can be used for both the phi and casts in the vector loop.
656 for (auto &Phi : Plan.getVectorLoopRegion()->getEntryBasicBlock()->phis()) {
658 if (!IV || IV->getTruncInst())
659 continue;
660
661 // A sequence of IR Casts has potentially been recorded for IV, which
662 // *must be bypassed* when the IV is vectorized, because the vectorized IV
663 // will produce the desired casted value. This sequence forms a def-use
664 // chain and is provided in reverse order, ending with the cast that uses
665 // the IV phi. Search for the recipe of the last cast in the chain and
666 // replace it with the original IV. Note that only the final cast is
667 // expected to have users outside the cast-chain and the dead casts left
668 // over will be cleaned up later.
669 ArrayRef<Instruction *> Casts = IV->getInductionDescriptor().getCastInsts();
670 VPValue *FindMyCast = IV;
671 for (Instruction *IRCast : reverse(Casts)) {
672 VPSingleDefRecipe *FoundUserCast = nullptr;
673 for (auto *U : FindMyCast->users()) {
674 auto *UserCast = dyn_cast<VPSingleDefRecipe>(U);
675 if (UserCast && UserCast->getUnderlyingValue() == IRCast) {
676 FoundUserCast = UserCast;
677 break;
678 }
679 }
680 // A cast recipe in the chain may have been removed by earlier DCE.
681 if (!FoundUserCast)
682 break;
683 FindMyCast = FoundUserCast;
684 }
685 if (FindMyCast != IV)
686 FindMyCast->replaceAllUsesWith(IV);
687 }
688}
689
692 Plan.getEntry());
694 // The recipes in the block are processed in reverse order, to catch chains
695 // of dead recipes.
696 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB))) {
697 if (vputils::isDeadRecipe(R)) {
698 R.eraseFromParent();
699 continue;
700 }
701
702 // Check if R is a dead VPPhi <-> update cycle and remove it.
703 VPValue *Start, *Incoming;
704 if (!match(&R, m_VPPhi(m_VPValue(Start), m_VPValue(Incoming))))
705 continue;
706 auto *PhiR = cast<VPPhi>(&R);
707 VPUser *PhiUser = PhiR->getSingleUser();
708 if (!PhiUser)
709 continue;
710 if (PhiUser != Incoming->getDefiningRecipe() ||
711 Incoming->getNumUsers() != 1)
712 continue;
713 PhiR->replaceAllUsesWith(Start);
714 PhiR->eraseFromParent();
715 Incoming->getDefiningRecipe()->eraseFromParent();
716 }
717 }
718}
719
720/// Legalize VPWidenPointerInductionRecipe, by replacing it with a PtrAdd
721/// (IndStart, ScalarIVSteps (0, Step)) if only its scalar values are used, as
722/// VPWidenPointerInductionRecipe will generate vectors only. If some users
723/// require vectors while other require scalars, the scalar uses need to extract
724/// the scalars from the generated vectors (Note that this is different to how
725/// int/fp inductions are handled). Legalize extract-from-ends using uniform
726/// VPReplicateRecipe of wide inductions to use regular VPReplicateRecipe, so
727/// the correct end value is available. Also optimize
728/// VPWidenIntOrFpInductionRecipe, if any of its users needs scalar values, by
729/// providing them scalar steps built on the canonical scalar IV and update the
730/// original IV's users. This is an optional optimization to reduce the needs of
731/// vector extracts.
734 bool HasOnlyVectorVFs = !Plan.hasScalarVFOnly();
735
737 for (VPRecipeBase &Phi : HeaderVPBB->phis())
738 if (auto *PhiR = dyn_cast<VPWidenInductionRecipe>(&Phi))
739 WideIVs.push_back(PhiR);
740
741 // Try to narrow wide and replicating recipes to uniform recipes, based on
742 // VPlan analysis.
743 // TODO: Apply to all recipes in the future, to replace legacy uniformity
744 // analysis.
745 for (VPWidenInductionRecipe *PhiR : WideIVs) {
747 for (VPUser *U : reverse(Users)) {
748 auto *Def = dyn_cast<VPRecipeWithIRFlags>(U);
749 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
750 // Skip recipes that shouldn't be narrowed.
751 if (!Def || !isa<VPReplicateRecipe, VPWidenRecipe>(Def) ||
752 Def->user_empty() || !Def->getUnderlyingValue() ||
753 (RepR && (RepR->isSingleScalar() || RepR->isPredicated())))
754 continue;
755
756 // Skip recipes that may have other lanes than their first used.
758 continue;
759
760 // TODO: Support scalarizing ExtractValue.
761 if (match(Def,
763 continue;
764
766 Def->getUnderlyingInstr()->getOpcode(), Def->operands(),
767 /*Mask=*/nullptr, *Def, {}, DebugLoc::getUnknown(),
768 Def->getUnderlyingInstr());
769 Clone->insertAfter(Def);
770 Def->replaceAllUsesWith(Clone);
771 Def->eraseFromParent();
772 }
773 }
774
775 VPBuilder Builder(HeaderVPBB, HeaderVPBB->getFirstNonPhi());
776 for (VPWidenInductionRecipe *PhiR : WideIVs) {
777 // Replace wide pointer inductions which have only their scalars used by
778 // PtrAdd(IndStart, ScalarIVSteps (0, Step)).
779 if (auto *PtrIV = dyn_cast<VPWidenPointerInductionRecipe>(PhiR)) {
780 if (!Plan.hasScalarVFOnly() &&
781 !PtrIV->onlyScalarsGenerated(Plan.hasScalableVF()))
782 continue;
783
784 VPValue *PtrAdd =
785 vputils::scalarizeVPWidenPointerInduction(PtrIV, Plan, Builder);
786 PtrIV->replaceAllUsesWith(PtrAdd);
787 continue;
788 }
789
790 // Replace widened induction with scalar steps for users that only use
791 // scalars.
792 auto *WideIV = cast<VPWidenIntOrFpInductionRecipe>(PhiR);
793 if (HasOnlyVectorVFs && none_of(WideIV->users(), [WideIV](VPUser *U) {
794 return U->usesScalars(WideIV);
795 }))
796 continue;
797
798 const InductionDescriptor &ID = WideIV->getInductionDescriptor();
799 VPIRFlags::WrapFlagsTy WrapFlags;
800 // We can preserve nuw when the step is non-negative.
801 const APInt *Step;
802 if (match(WideIV->getStepValue(), m_APInt(Step)) && Step->isNonNegative())
803 WrapFlags = {static_cast<bool>(WideIV->getNoWrapFlagsOrNone().HasNUW),
804 false};
806 Plan, ID.getKind(), ID.getInductionOpcode(),
807 dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
808 WideIV->getTruncInst(), WideIV->getStartValue(), WideIV->getStepValue(),
809 WideIV->getDebugLoc(), Builder, WrapFlags);
810
811 // Update scalar users of IV to use Step instead.
812 if (!HasOnlyVectorVFs) {
813 assert(!Plan.hasScalableVF() &&
814 "plans containing a scalar VF cannot also include scalable VFs");
815 WideIV->replaceAllUsesWith(Steps);
816 } else {
817 bool HasScalableVF = Plan.hasScalableVF();
818 WideIV->replaceUsesWithIf(Steps,
819 [WideIV, HasScalableVF](VPUser &U, unsigned) {
820 if (HasScalableVF)
821 return U.usesFirstLaneOnly(WideIV);
822 return U.usesScalars(WideIV);
823 });
824 }
825 }
826}
827
828/// Check if \p VPV is an untruncated wide induction, either before or after the
829/// increment. If so return the header IV (before the increment), otherwise
830/// return null.
833 auto *WideIV = dyn_cast<VPWidenInductionRecipe>(VPV);
834 if (WideIV) {
835 // VPV itself is a wide induction, separately compute the end value for exit
836 // users if it is not a truncated IV.
837 auto *IntOrFpIV = dyn_cast<VPWidenIntOrFpInductionRecipe>(WideIV);
838 return (IntOrFpIV && IntOrFpIV->getTruncInst()) ? nullptr : WideIV;
839 }
840
841 // Check if VPV is an optimizable induction increment.
842 VPRecipeBase *Def = VPV->getDefiningRecipe();
843 if (!Def || Def->getNumOperands() != 2)
844 return nullptr;
845 WideIV = dyn_cast<VPWidenInductionRecipe>(Def->getOperand(0));
846 if (!WideIV)
847 WideIV = dyn_cast<VPWidenInductionRecipe>(Def->getOperand(1));
848 if (!WideIV)
849 return nullptr;
850
851 auto IsWideIVInc = [&]() {
852 auto &ID = WideIV->getInductionDescriptor();
853
854 // Check if VPV increments the induction by the induction step.
855 VPValue *IVStep = WideIV->getStepValue();
856 switch (ID.getInductionOpcode()) {
857 case Instruction::Add:
858 return match(VPV, m_c_Add(m_Specific(WideIV), m_Specific(IVStep)));
859 case Instruction::FAdd:
860 return match(VPV, m_c_FAdd(m_Specific(WideIV), m_Specific(IVStep)));
861 case Instruction::FSub:
862 return match(VPV, m_Binary<Instruction::FSub>(m_Specific(WideIV),
863 m_Specific(IVStep)));
864 case Instruction::Sub: {
865 // IVStep will be the negated step of the subtraction. Check if Step == -1
866 // * IVStep.
867 VPValue *Step;
868 if (!match(VPV, m_Sub(m_VPValue(), m_VPValue(Step))))
869 return false;
870 const SCEV *IVStepSCEV = vputils::getSCEVExprForVPValue(IVStep, PSE);
871 const SCEV *StepSCEV = vputils::getSCEVExprForVPValue(Step, PSE);
872 ScalarEvolution &SE = *PSE.getSE();
873 return !isa<SCEVCouldNotCompute>(IVStepSCEV) &&
874 !isa<SCEVCouldNotCompute>(StepSCEV) &&
875 IVStepSCEV == SE.getNegativeSCEV(StepSCEV);
876 }
877 default:
878 return ID.getKind() == InductionDescriptor::IK_PtrInduction &&
879 match(VPV, m_GetElementPtr(m_Specific(WideIV),
880 m_Specific(WideIV->getStepValue())));
881 }
882 llvm_unreachable("should have been covered by switch above");
883 };
884 return IsWideIVInc() ? WideIV : nullptr;
885}
886
887/// Attempts to optimize the induction variable exit values for users in the
888/// early exit block.
891 VPValue *Incoming, *Mask;
893 m_VPValue(Incoming))))
894 return nullptr;
895
896 auto *WideIV = getOptimizableIVOf(Incoming, PSE);
897 if (!WideIV)
898 return nullptr;
899
900 // Calculate the final index.
901 VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
902 auto *CanonicalIV = LoopRegion->getCanonicalIV();
903 Type *CanonicalIVType = LoopRegion->getCanonicalIVType();
904 auto *ExtractR = cast<VPInstruction>(Op);
905 VPBuilder B(ExtractR);
906
907 DebugLoc DL = ExtractR->getDebugLoc();
908 VPValue *FirstActiveLane = B.createFirstActiveLane(Mask, DL);
909 FirstActiveLane =
910 B.createScalarZExtOrTrunc(FirstActiveLane, CanonicalIVType, DL);
911 VPValue *EndValue = B.createAdd(CanonicalIV, FirstActiveLane, DL);
912
913 // `getOptimizableIVOf()` always returns the pre-incremented IV, so if it
914 // changed it means the exit is using the incremented value, so we need to
915 // add the step.
916 if (Incoming != WideIV) {
917 VPValue *One = Plan.getConstantInt(CanonicalIVType, 1);
918 EndValue = B.createAdd(EndValue, One, DL);
919 }
920
921 if (!match(WideIV, m_CanonicalWidenIV())) {
922 const InductionDescriptor &ID = WideIV->getInductionDescriptor();
923 VPIRValue *Start = WideIV->getStartValue();
924 VPValue *Step = WideIV->getStepValue();
925 EndValue = B.createDerivedIV(
926 ID.getKind(), dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
927 Start, EndValue, Step);
928 }
929
930 return EndValue;
931}
932
933/// Compute the end value for \p WideIV, unless it is truncated. Creates a
934/// VPDerivedIVRecipe for non-canonical inductions.
936 VPBuilder &VectorPHBuilder,
937 VPValue *VectorTC) {
938 auto *WideIntOrFp = dyn_cast<VPWidenIntOrFpInductionRecipe>(WideIV);
939 // Truncated wide inductions resume from the last lane of their vector value
940 // in the last vector iteration which is handled elsewhere.
941 if (WideIntOrFp && WideIntOrFp->getTruncInst())
942 return nullptr;
943
944 VPIRValue *Start = WideIV->getStartValue();
945 VPValue *Step = WideIV->getStepValue();
946 const InductionDescriptor &ID = WideIV->getInductionDescriptor();
947 VPValue *EndValue = VectorTC;
948 if (!match(WideIV, m_CanonicalWidenIV())) {
949 EndValue = VectorPHBuilder.createDerivedIV(
950 ID.getKind(), dyn_cast_or_null<FPMathOperator>(ID.getInductionBinOp()),
951 Start, VectorTC, Step);
952 }
953
954 // EndValue is derived from the vector trip count (which has the same type as
955 // the widest induction) and thus may be wider than the induction here.
956 Type *ScalarTypeOfWideIV = WideIV->getScalarType();
957 if (ScalarTypeOfWideIV != EndValue->getScalarType()) {
958 EndValue = VectorPHBuilder.createScalarCast(Instruction::Trunc, EndValue,
959 ScalarTypeOfWideIV,
960 WideIV->getDebugLoc());
961 }
962
963 return EndValue;
964}
965
966/// Attempts to optimize the induction variable exit values for users in the
967/// exit block coming from the latch in the original scalar loop.
968static VPValue *
972 VPValue *Incoming;
975 m_VPValue(Incoming)))))
976 return nullptr;
977
978 VPWidenInductionRecipe *WideIV = getOptimizableIVOf(Incoming, PSE);
979 if (!WideIV)
980 return nullptr;
981
982 VPValue *EndValue = EndValues.lookup(WideIV);
983 assert(EndValue && "Must have computed the end value up front");
984
985 // `getOptimizableIVOf()` always returns the pre-incremented IV, so if it
986 // changed it means the exit is using the incremented value, so we don't
987 // need to subtract the step.
988 if (Incoming != WideIV)
989 return EndValue;
990
991 // Otherwise, subtract the step from the EndValue.
992 auto *ExtractR = cast<VPInstruction>(Op);
993 VPBuilder B(ExtractR);
994 VPValue *Step = WideIV->getStepValue();
995 Type *ScalarTy = WideIV->getScalarType();
996 if (ScalarTy->isIntegerTy())
997 return B.createSub(EndValue, Step, DebugLoc::getUnknown(), "ind.escape");
998 if (ScalarTy->isPointerTy()) {
999 Type *StepTy = Step->getScalarType();
1000 auto *Zero = Plan.getZero(StepTy);
1001 return B.createPtrAdd(EndValue, B.createSub(Zero, Step),
1002 DebugLoc::getUnknown(), "ind.escape");
1003 }
1004 if (ScalarTy->isFloatingPointTy()) {
1005 const auto &ID = WideIV->getInductionDescriptor();
1006 return B.createNaryOp(
1007 ID.getInductionBinOp()->getOpcode() == Instruction::FAdd
1008 ? Instruction::FSub
1009 : Instruction::FAdd,
1010 {EndValue, Step}, {ID.getInductionBinOp()->getFastMathFlags()});
1011 }
1012 llvm_unreachable("all possible induction types must be handled");
1013 return nullptr;
1014}
1015
1018 VPValue *ResumeTC,
1019 const Loop *L) {
1020 VPValue *Incoming;
1022 return nullptr;
1023
1024 const SCEV *IncomingSCEV = vputils::getSCEVExprForVPValue(Incoming, PSE, L);
1025 const SCEV *Start, *Step;
1026 if (!match(IncomingSCEV, m_scev_AffineAddRec(m_SCEV(Start), m_SCEV(Step),
1027 m_SpecificLoop(L))))
1028 return nullptr;
1029
1030 auto *ExtractR = cast<VPInstruction>(Op);
1031 DebugLoc DL = ExtractR->getDebugLoc();
1032 VPBuilder Builder(ExtractR);
1033 VPSCEVExpander Expander(Builder, *PSE.getSE(), DL);
1034 VPValue *StartVPV = Expander.expand(Start);
1035 VPValue *StepVPV = Expander.expand(Step);
1036
1037 Type *StartTy = StartVPV->getScalarType();
1038 assert(StartTy->isIntOrPtrTy() && "The type must be SCEVable");
1042 Type *TCTy = ResumeTC->getScalarType();
1043 VPValue *ExitCount = Builder.createOverflowingOp(
1044 Instruction::Sub, {ResumeTC, Plan.getConstantInt(TCTy, 1)},
1045 {/*HasNUW=*/true, /*HasNSW=*/false}, DebugLoc::getUnknown());
1046 return Builder.createDerivedIV(Kind, /*FPBinOp=*/nullptr, StartVPV, ExitCount,
1047 StepVPV);
1048}
1049
1051 VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L) {
1052 // Compute end values for all inductions.
1053 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
1054 auto *VectorPH = cast<VPBasicBlock>(VectorRegion->getSinglePredecessor());
1055 VPBuilder VectorPHBuilder(VectorPH, VectorPH->begin());
1057 VPValue *ResumeTC =
1058 Plan.hasTailFolded() ? Plan.getTripCount() : &Plan.getVectorTripCount();
1059 for (auto &Phi : VectorRegion->getEntryBasicBlock()->phis()) {
1060 auto *WideIV = dyn_cast<VPWidenInductionRecipe>(&Phi);
1061 if (!WideIV)
1062 continue;
1063 if (VPValue *EndValue =
1064 tryToComputeEndValueForInduction(WideIV, VectorPHBuilder, ResumeTC))
1065 EndValues[WideIV] = EndValue;
1066 }
1067
1068 VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
1069 for (VPRecipeBase &R : make_early_inc_range(*MiddleVPBB)) {
1070 VPValue *Op;
1071 if (!match(&R, m_ExitingIVValue(m_VPValue(Op))))
1072 continue;
1073 auto *WideIV = cast<VPWidenInductionRecipe>(Op);
1074 if (VPValue *EndValue = EndValues.lookup(WideIV)) {
1075 R.getVPSingleValue()->replaceAllUsesWith(EndValue);
1076 R.eraseFromParent();
1077 }
1078 }
1079
1080 // Then, optimize exit block users.
1081 for (VPIRBasicBlock *ExitVPBB : Plan.getExitBlocks()) {
1082 for (VPRecipeBase &R : ExitVPBB->phis()) {
1083 auto *ExitIRI = cast<VPIRPhi>(&R);
1084
1085 for (auto [Idx, PredVPBB] : enumerate(ExitVPBB->getPredecessors())) {
1086 VPValue *Escape = nullptr;
1087 if (PredVPBB == MiddleVPBB) {
1089 Plan, ExitIRI->getOperand(Idx), EndValues, PSE);
1090 if (!Escape)
1092 Plan, ExitIRI->getOperand(Idx), PSE, ResumeTC, L);
1093 } else {
1095 Plan, ExitIRI->getOperand(Idx), PSE);
1096 }
1097 if (Escape)
1098 ExitIRI->setOperand(Idx, Escape);
1099 }
1100 }
1101 }
1102}
1103
1104/// Remove redundant ExpandSCEVRecipes in \p Plan's entry block by replacing
1105/// them with already existing recipes expanding the same SCEV expression.
1108
1109 for (VPRecipeBase &R :
1111 auto *ExpR = dyn_cast<VPExpandSCEVRecipe>(&R);
1112 if (!ExpR)
1113 continue;
1114
1115 const auto &[V, Inserted] = SCEV2VPV.try_emplace(ExpR->getSCEV(), ExpR);
1116 if (Inserted)
1117 continue;
1118
1119 ExpR->replaceAllUsesWith(V->second);
1120 if (ExpR == Plan.getTripCount())
1121 Plan.resetTripCount(V->second);
1122
1123 ExpR->eraseFromParent();
1124 }
1125}
1126
1127/// Try to simplify logical and bitwise recipes in \p Def.
1129 VPBuilder &Builder,
1130 bool CanCreateNewRecipe) {
1131 VPlan *Plan = Def->getParent()->getPlan();
1132
1133 // Simplify (X && Y) | (X && !Y) -> X.
1134 // TODO: Split up into simpler, modular combines: (X && Y) | (X && Z) into X
1135 // && (Y | Z) and (X | !X) into true. This requires queuing newly created
1136 // recipes to be visited during simplification.
1137 VPValue *X, *Y, *Z;
1138 if (match(Def,
1141 return X;
1142
1143 // x | AllOnes -> AllOnes
1144 if (match(Def, m_c_BinaryOr(m_VPValue(X), m_AllOnes())))
1145 return Plan->getAllOnesValue(Def->getScalarType());
1146
1147 // x | 0 -> x
1148 if (match(Def, m_c_BinaryOr(m_VPValue(X), m_ZeroInt())))
1149 return X;
1150
1151 // x | !x -> AllOnes
1153 return Plan->getAllOnesValue(Def->getScalarType());
1154
1155 // x & 0 -> 0
1156 if (match(Def, m_c_BinaryAnd(m_VPValue(X), m_ZeroInt())))
1157 return Plan->getZero(Def->getScalarType());
1158
1159 // x & AllOnes -> x
1160 if (match(Def, m_c_BinaryAnd(m_VPValue(X), m_AllOnes())))
1161 return X;
1162
1163 // x && false -> false
1164 if (match(Def, m_c_LogicalAnd(m_VPValue(X), m_False())))
1165 return Plan->getFalse();
1166
1167 // x && true -> x
1168 if (match(Def, m_c_LogicalAnd(m_VPValue(X), m_True())))
1169 return X;
1170
1171 // (x && y) | (x && z) -> x && (y | z)
1172 if (CanCreateNewRecipe &&
1175 // Simplify only if one of the operands has one use to avoid creating an
1176 // extra recipe.
1177 (!Def->getOperand(0)->hasMoreThanOneUniqueUser() ||
1178 !Def->getOperand(1)->hasMoreThanOneUniqueUser()))
1179 return Builder.createLogicalAnd(X, Builder.createOr(Y, Z));
1180
1181 // x && (x && y) -> x && y
1182 if (match(Def, m_LogicalAnd(m_VPValue(X),
1184 return Def->getOperand(1);
1185
1186 // x && (y && x) -> x && y
1187 if (match(Def, m_LogicalAnd(m_VPValue(X),
1189 return Builder.createLogicalAnd(X, Y);
1190
1191 // x && !x -> 0
1193 return Plan->getFalse();
1194
1195 if (match(Def, m_Select(m_VPValue(), m_VPValue(X), m_Deferred(X))))
1196 return X;
1197
1198 // select c, false, true -> not c
1199 VPValue *C;
1200 if (CanCreateNewRecipe &&
1201 match(Def, m_Select(m_VPValue(C), m_False(), m_True())))
1202 return Builder.createNot(C);
1203
1204 // select !c, x, y -> select c, y, x
1205 if (match(Def, m_Select(m_Not(m_VPValue(C)), m_VPValue(X), m_VPValue(Y)))) {
1206 Def->setOperand(0, C);
1207 Def->setOperand(1, Y);
1208 Def->setOperand(2, X);
1209 return Def;
1210 }
1211
1212 // select x, (i1 y | z), y -> y | (x && z)
1213 if (CanCreateNewRecipe &&
1214 match(Def, m_Select(m_VPValue(X),
1216 m_Deferred(Y))) &&
1217 Y->getScalarType()->isIntegerTy(1))
1218 return Builder.createOr(Y, Builder.createLogicalAnd(X, Z));
1219
1220 // select %M0, (select %M1, %X, %Y), %Y -> select (%M0 && %M1), %X, %Y
1221 VPValue *Mask0, *Mask1;
1222 if (CanCreateNewRecipe &&
1223 match(Def,
1224 m_SelectLike(m_VPValue(Mask0),
1226 m_VPValue(Y))),
1227 m_Deferred(Y))))
1228 return Builder.createSelect(Builder.createLogicalAnd(Mask0, Mask1), X, Y,
1229 Def->getDebugLoc());
1230
1231 return nullptr;
1232}
1233
1234/// Try to simplify VPSingleDefRecipe \p Def. Returns a new recipe if it should
1235/// be replaced, or the existing recipe if it was modified. Returns nullptr if
1236/// nothing was simplified.
1238 VPlan *Plan = Def->getParent()->getPlan();
1239
1240 // Simplification of live-in IR values for SingleDef recipes using
1241 // InstSimplifyFolder.
1242 const DataLayout &DL = Plan->getDataLayout();
1243 if (VPValue *V = vputils::tryToFoldLiveIns(*Def, Def->operands(), DL))
1244 return V;
1245
1246 // Fold PredPHI LiveIn -> LiveIn.
1247 if (auto *PredPHI = dyn_cast<VPPredInstPHIRecipe>(Def)) {
1248 VPValue *Op = PredPHI->getOperand(0);
1249 if (isa<VPIRValue>(Op))
1250 return Op;
1251 }
1252
1253 // Drop the mask of a predicated store masked by the header mask (which is
1254 // guaranteed to be true at least for the first lane) and both the stored
1255 // value and the address are uniform across VF and UF. The header mask is
1256 // still the abstract region value here.
1257 if (auto *RepR = dyn_cast<VPReplicateRecipe>(Def);
1258 RepR && RepR->isPredicated() && RepR->getOpcode() == Instruction::Store &&
1259 all_of(RepR->operandsWithoutMask(), vputils::isUniformAcrossVFsAndUFs) &&
1260 match(RepR->getMask(), m_HeaderMask())) {
1261 auto *Unmasked = new VPReplicateRecipe(
1262 RepR->getUnderlyingInstr(), RepR->operandsWithoutMask(),
1263 RepR->isSingleScalar(), /*Mask=*/nullptr, *RepR, *RepR,
1264 RepR->getDebugLoc());
1265 Unmasked->insertBefore(RepR);
1266 return Unmasked;
1267 }
1268
1269 VPBuilder Builder(Def);
1270
1271 // Avoid replacing VPInstructions with underlying values with new
1272 // VPInstructions, as we would fail to create widen/replicate recpes from the
1273 // new VPInstructions without an underlying value, and miss out on some
1274 // transformations that only apply to widened/replicated recipes later, by
1275 // doing so.
1276 // TODO: We should also not replace non-VPInstructions like VPWidenRecipe with
1277 // VPInstructions without underlying values, as those will get skipped during
1278 // cost computation.
1279 bool CanCreateNewRecipe =
1280 !isa<VPInstruction>(Def) || !Def->getUnderlyingValue();
1281
1282 VPValue *A, *Z;
1283
1284 // A bitcast to the same type is a no-op.
1285 if (match(Def, m_BitCast(m_VPValue(A))) &&
1286 Def->getScalarType() == A->getScalarType())
1287 return A;
1288
1289 if (match(Def, m_Trunc(m_VPValue(Z, m_ZExtOrSExt(m_VPValue(A)))))) {
1290 Type *TruncTy = Def->getScalarType();
1291 Type *ATy = A->getScalarType();
1292 if (TruncTy == ATy) {
1293 return A;
1294 } else {
1295 // Don't replace a non-widened cast recipe with a widened cast.
1296 if (!isa<VPWidenCastRecipe>(Def))
1297 return nullptr;
1298 if (ATy->getScalarSizeInBits() < TruncTy->getScalarSizeInBits()) {
1299
1300 unsigned ExtOpcode = match(Z, m_SExt(m_VPValue())) ? Instruction::SExt
1301 : Instruction::ZExt;
1302 auto *Ext = Builder.createWidenCast(Instruction::CastOps(ExtOpcode), A,
1303 TruncTy);
1304 if (auto *UnderlyingExt = Z->getUnderlyingValue()) {
1305 // UnderlyingExt has distinct return type, used to retain legacy cost.
1306 Ext->setUnderlyingValue(UnderlyingExt);
1307 }
1308 return Ext;
1309 } else if (ATy->getScalarSizeInBits() > TruncTy->getScalarSizeInBits()) {
1310 auto *Trunc = Builder.createWidenCast(Instruction::Trunc, A, TruncTy);
1311 return Trunc;
1312 }
1313 }
1314 }
1315
1316 if (VPValue *V = simplifyLogicalRecipe(Def, Builder, CanCreateNewRecipe))
1317 return V;
1318
1319 VPValue *X, *Y;
1320 if (match(Def, m_c_Add(m_VPValue(A), m_ZeroInt())))
1321 return A;
1322
1323 if (match(Def, m_c_Mul(m_VPValue(A), m_One())))
1324 return A;
1325
1326 if (match(Def, m_c_Mul(m_VPValue(A), m_ZeroInt())))
1327 return Plan->getZero(Def->getScalarType());
1328
1329 if (CanCreateNewRecipe && match(Def, m_c_Mul(m_VPValue(A), m_AllOnes()))) {
1330 // Preserve nsw from the Mul on the new Sub.
1332 false, cast<VPRecipeWithIRFlags>(Def)->hasNoSignedWrap()};
1333 return Builder.createSub(Plan->getZero(A->getScalarType()), A,
1334 Def->getDebugLoc(), "", NW);
1335 }
1336
1337 if (CanCreateNewRecipe &&
1338 match(Def, m_c_Add(m_VPValue(X),
1339 m_VPValue(Z, m_Sub(m_ZeroInt(), m_VPValue(Y)))))) {
1340 // Preserve nsw from the Add and the Sub, if it's present on both, on the
1341 // new Sub.
1343 false, cast<VPRecipeWithIRFlags>(Def)->hasNoSignedWrap() &&
1344 cast<VPRecipeWithIRFlags>(Z)->hasNoSignedWrap()};
1345 return Builder.createSub(X, Y, Def->getDebugLoc(), "", NW);
1346 }
1347
1348 const APInt *APC;
1349 if (CanCreateNewRecipe && match(Def, m_URem(m_VPValue(X), m_APInt(APC))) &&
1350 APC->isPowerOf2())
1351 return Builder.createAnd(X, Plan->getConstantInt(*APC - 1),
1352 Def->getDebugLoc());
1353
1354 if (CanCreateNewRecipe && match(Def, m_c_Mul(m_VPValue(A), m_APInt(APC))) &&
1355 APC->isPowerOf2()) {
1356 auto *MulR = cast<VPRecipeWithIRFlags>(Def);
1357 unsigned ShiftAmt = APC->exactLogBase2();
1358 VPIRFlags::WrapFlagsTy NW(MulR->hasNoUnsignedWrap(),
1359 MulR->hasNoSignedWrap() &&
1360 ShiftAmt != APC->getBitWidth() - 1);
1361 return Builder.createNaryOp(
1362 Instruction::Shl,
1363 {A, Plan->getConstantInt(APC->getBitWidth(), ShiftAmt)}, NW,
1364 Def->getDebugLoc());
1365 }
1366
1367 if (CanCreateNewRecipe && match(Def, m_UDiv(m_VPValue(A), m_APInt(APC))) &&
1368 APC->isPowerOf2())
1369 return Builder.createNaryOp(
1370 Instruction::LShr,
1371 {A, Plan->getConstantInt(APC->getBitWidth(), APC->exactLogBase2())},
1372 *cast<VPRecipeWithIRFlags>(Def), Def->getDebugLoc());
1373
1374 if (match(Def, m_Not(m_VPValue(A)))) {
1375 if (match(A, m_Not(m_VPValue(A))))
1376 return A;
1377
1378 // Try to fold Not into compares by adjusting the predicate in-place.
1379 CmpPredicate Pred;
1380 if (match(A, m_Cmp(Pred, m_VPValue(), m_VPValue()))) {
1381 auto *Cmp = cast<VPRecipeWithIRFlags>(A);
1382 // Only fold if every user is a Not of the cmp, or a select using the cmp
1383 // solely as its condition.
1384 if (all_of(Cmp->users(), [Cmp](VPUser *U) {
1385 return match(U, m_Not(m_Specific(Cmp))) ||
1386 (match(U, m_Select(m_Specific(Cmp), m_VPValue(),
1387 m_VPValue())) &&
1388 U->getOperand(1) != Cmp && U->getOperand(2) != Cmp);
1389 })) {
1390 Cmp->setPredicate(CmpInst::getInversePredicate(Pred));
1391 for (VPUser *U : to_vector(Cmp->users())) {
1392 auto *R = cast<VPSingleDefRecipe>(U);
1393 if (match(R, m_Select(m_Specific(Cmp), m_VPValue(X), m_VPValue(Y)))) {
1394 // select (cmp pred), x, y -> select (cmp inv_pred), y, x
1395 R->setOperand(1, Y);
1396 R->setOperand(2, X);
1397 } else {
1398 // not (cmp pred) -> cmp inv_pred
1399 assert(match(R, m_Not(m_Specific(Cmp))) && "Unexpected user");
1400 R->replaceAllUsesWith(Cmp);
1401 }
1402 }
1403 // If Cmp doesn't have a debug location, use the one from the negation,
1404 // to preserve the location.
1405 if (!Cmp->getDebugLoc() && Def->getDebugLoc())
1406 Cmp->setDebugLoc(Def->getDebugLoc());
1407 return Def;
1408 }
1409 }
1410 }
1411
1412 // Fold any-of (fcmp uno %A, %A), (fcmp uno %B, %B), ... ->
1413 // any-of (fcmp uno %A, %B), ...
1414 if (match(Def, m_AnyOf())) {
1416 VPRecipeBase *UnpairedCmp = nullptr;
1417 for (VPValue *Op : Def->operands()) {
1418 VPValue *X;
1419 if (Op->getNumUsers() > 1 ||
1421 m_Deferred(X)))) {
1422 NewOps.push_back(Op);
1423 } else if (!UnpairedCmp) {
1424 UnpairedCmp = Op->getDefiningRecipe();
1425 } else {
1426 NewOps.push_back(Builder.createFCmp(CmpInst::FCMP_UNO,
1427 UnpairedCmp->getOperand(0), X));
1428 UnpairedCmp = nullptr;
1429 }
1430 }
1431
1432 if (UnpairedCmp)
1433 NewOps.push_back(UnpairedCmp->getVPSingleValue());
1434
1435 if (NewOps.size() < Def->getNumOperands()) {
1436 VPValue *NewAnyOf = Builder.createNaryOp(VPInstruction::AnyOf, NewOps);
1437 return NewAnyOf;
1438 }
1439 }
1440
1441 // Fold (fcmp uno %X, %X) or (fcmp uno %Y, %Y) -> fcmp uno %X, %Y
1442 // This is useful for fmax/fmin without fast-math flags, where we need to
1443 // check if any operand is NaN.
1444 if (CanCreateNewRecipe &&
1445 match(Def,
1446 m_BinaryOr(
1449 return Builder.createFCmp(CmpInst::FCMP_UNO, X, Y);
1450
1451 // Remove redundant DerviedIVs, that is 0 + A * 1 -> A and 0 + 0 * x -> 0.
1452 if ((match(Def, m_DerivedIV(m_ZeroInt(), m_VPValue(A), m_One())) ||
1454 m_VPValue()))) &&
1455 A->getScalarType() == Def->getScalarType())
1456 return A;
1457
1459 m_One()))) {
1460 Type *WideStepTy = Def->getScalarType();
1461 if (X->getScalarType() != WideStepTy)
1462 X = Builder.createWidenCast(Instruction::Trunc, X, WideStepTy);
1463 return X;
1464 }
1465
1466 // For i1 vp.merges produced by AnyOf reductions:
1467 // vp.merge true, (or x, y), x, evl -> vp.merge y, true, x, evl
1469 m_VPValue(X), m_VPValue())) &&
1471 Def->getScalarType()->isIntegerTy(1)) {
1472 Def->setOperand(1, Plan->getTrue());
1473 Def->setOperand(0, Y);
1474 return Def;
1475 }
1476
1477 // Simplify MaskedCond with no block mask to its single operand.
1479 !cast<VPInstruction>(Def)->isMasked())
1480 return Def->getOperand(0);
1481
1482 // Look through ExtractLastLane.
1483 if (match(Def, m_ExtractLastLane(m_VPValue(A)))) {
1484 if (match(A, m_BuildVector())) {
1485 auto *BuildVector = cast<VPInstruction>(A);
1486 return BuildVector->getOperand(BuildVector->getNumOperands() - 1);
1487 }
1488
1489 if (match(A, m_Broadcast(m_VPValue(X))))
1490 return X;
1491
1493 return A;
1494
1495 if (Plan->hasScalarVFOnly())
1496 return A;
1497 }
1498
1499 // Look through ExtractPenultimateElement (BuildVector ....).
1501 auto *BuildVector = cast<VPInstruction>(Def->getOperand(0));
1502 return BuildVector->getOperand(BuildVector->getNumOperands() - 2);
1503 }
1504
1505 uint64_t Idx;
1507 auto *BuildVector = cast<VPInstruction>(Def->getOperand(0));
1508 return BuildVector->getOperand(Idx);
1509 }
1510
1511 if (match(Def, m_BuildVector()) && all_equal(Def->operands()))
1512 return Builder.createNaryOp(VPInstruction::Broadcast, Def->getOperand(0));
1513
1514 // Replace uses of a BuildVector by users that only use its first lane with
1515 // its first operand directly.
1516 if (match(Def, m_BuildVector())) {
1517 Def->replaceUsesWithIf(Def->getOperand(0), [Def](VPUser &U, unsigned) {
1518 return U.usesFirstLaneOnly(Def);
1519 });
1520 return Def;
1521 }
1522
1523 // Look through broadcast of single-scalar when used as select conditions; in
1524 // that case the scalar condition can be used directly.
1525 if (match(Def,
1528 "broadcast operand must be single-scalar");
1529 Def->setOperand(0, Z);
1530 return Def;
1531 }
1532
1533 if (match(Def, m_Broadcast(m_VPValue(X)))) {
1534 Def->replaceUsesWithIf(
1535 X, [Def](const VPUser &U, unsigned) { return U.usesScalars(Def); });
1536 return Def;
1537 }
1538
1540 if (Def->getNumOperands() == 1) {
1541 return Def->getOperand(0);
1542 }
1543 if (auto *Phi = dyn_cast<VPFirstOrderRecurrencePHIRecipe>(Def)) {
1544 if (all_equal(Phi->incoming_values()))
1545 return Phi->getOperand(0);
1546 }
1547 return nullptr;
1548 }
1549
1550 VPIRValue *IRV;
1551 if (Def->getNumOperands() == 1 &&
1553 return IRV;
1554
1555 // Some simplifications can only be applied after unrolling. Perform them
1556 // below.
1557 if (!Plan->isUnrolled())
1558 return nullptr;
1559
1560 // After unrolling, extract-lane may be used to extract values from multiple
1561 // scalar sources. Only simplify when extracting from a single scalar source.
1562 VPValue *LaneToExtract;
1563 if (match(Def, m_ExtractLane(m_VPValue(LaneToExtract), m_VPValue(A)))) {
1564 // Simplify extract-lane(%lane_num, %scalar_val) -> %scalar_val.
1566 return A;
1567
1568 // Replace extract-lane(0, canonical-WIDEN-INDUCTION) with the region's
1569 // scalar canonical IV.
1571 if (match(LaneToExtract, m_ZeroInt()) &&
1572 match(A, m_CanonicalWidenIV(WidenIV)))
1573 return WidenIV->getRegion()->getCanonicalIV();
1574
1575 // Simplify extract-lane with single source to extract-element.
1576 return Builder.createNaryOp(Instruction::ExtractElement, {A, LaneToExtract},
1577 Def->getDebugLoc());
1578 }
1579
1580 // Look for cycles where Def is of the form:
1581 // X = phi(0, IVInc) ; used only by IVInc, or by IVInc and Inc = X + Y
1582 // IVInc = X + Step ; used by X and Def
1583 // Def = IVInc + Y
1584 // Fold the increment Y into the phi's start value, replace Def with IVInc,
1585 // and if Inc exists, replace it with X.
1586 VPValue *IVInc;
1587 if (match(Def, m_Add(m_VPValue(IVInc, m_Add(m_VPValue(X), m_VPValue())),
1588 m_VPValue(Y))) &&
1589 isa<VPIRValue>(Y) && match(X, m_VPPhi(m_ZeroInt(), m_Specific(IVInc)))) {
1590 auto *Phi = cast<VPPhi>(X);
1591 if (IVInc->getNumUsers() == 2) {
1592 // If Phi has a second user (besides IVInc's defining recipe), it must
1593 // be Inc = Phi + Y for the fold to apply.
1595 findUserOf(Phi, m_Add(m_Specific(Phi), m_Specific(Y))));
1596 if (Phi->getNumUsers() == 1 || (Phi->getNumUsers() == 2 && Inc)) {
1597 Def->replaceAllUsesWith(IVInc);
1598 if (Inc)
1599 Inc->replaceAllUsesWith(Phi);
1600 Phi->setOperand(0, Y);
1601 return Def;
1602 }
1603 }
1604 }
1605
1606 // Simplify unrolled VectorPointer without offset, or with zero offset, to
1607 // just the pointer operand.
1608 if (auto *VPR = dyn_cast<VPVectorPointerRecipe>(Def))
1609 if (!VPR->getVFxPart() || match(VPR->getVFxPart(), m_ZeroInt()))
1610 return VPR->getOperand(0);
1611
1612 // VPScalarIVSteps after unrolling can be replaced by their start value, if
1613 // the start index is zero and only the first lane 0 is demanded.
1614 if (auto *Steps = dyn_cast<VPScalarIVStepsRecipe>(Def))
1615 if (!Steps->getStartIndex() && vputils::onlyFirstLaneUsed(Steps))
1616 return Steps->getOperand(0);
1617
1618 // Simplify redundant ReductionStartVector recipes after unrolling.
1619 VPValue *StartV;
1621 m_VPValue(StartV), m_VPValue(), m_VPValue()))) {
1622 Def->replaceUsesWithIf(StartV, [](const VPUser &U, unsigned Idx) {
1623 auto *PhiR = dyn_cast<VPReductionPHIRecipe>(&U);
1624 return PhiR && PhiR->isInLoop();
1625 });
1626 return Def;
1627 }
1628
1629 if (Plan->getConcreteUF() == 1 && match(Def, m_ExtractLastPart(m_VPValue(A))))
1630 return A;
1631
1632 return nullptr;
1633}
1634
1637 Plan.getEntry());
1639 for (VPRecipeBase &R : make_early_inc_range(*VPBB))
1640 if (auto *Def = dyn_cast<VPSingleDefRecipe>(&R))
1641 if (VPValue *New = simplifyRecipe(Def)) {
1642 if (New != Def) {
1643 // Replace the recipe with a new one.
1644 Def->replaceAllUsesWith(New);
1645 Def->eraseFromParent();
1646 } else if (vputils::isDeadRecipe(R)) {
1647 // Recipe was modified - it may be dead now.
1648 Def->eraseFromParent();
1649 }
1650 }
1651 }
1652}
1653
1655 // Pull out reverses from any elementwise op.
1656 // binop(reverse(x), reverse(y)) -> reverse(binop(x,y))
1658 Plan, [](VPValue *&X) { return m_Reverse(m_VPValue(X)); },
1659 [](auto *X) { return new VPInstruction(VPInstruction::Reverse, X); });
1660
1661 // reverse(reverse(x)) -> x
1662 VPValue *X;
1665 for (VPRecipeBase &R : make_early_inc_range(*VPBB))
1666 if (match(&R, m_Reverse(m_Reverse(m_VPValue(X)))))
1667 R.getVPSingleValue()->replaceAllUsesWith(X);
1668}
1669
1670/// Reassociate (headermask && x) && y -> headermask && (x && y) to allow the
1671/// header mask to be simplified further when tail folding, e.g. in
1672/// optimizeEVLMasks.
1673static void reassociateHeaderMask(VPlan &Plan) {
1674 VPValue *HeaderMask = Plan.getVectorLoopRegion()->getHeaderMask();
1675 if (!HeaderMask)
1676 return;
1677
1678 SmallVector<VPUser *> Worklist;
1679 for (VPUser *U : HeaderMask->users())
1680 if (match(U, m_LogicalAnd(m_Specific(HeaderMask), m_VPValue())))
1682
1683 while (!Worklist.empty()) {
1684 auto *R = dyn_cast<VPSingleDefRecipe>(Worklist.pop_back_val());
1685 VPValue *X, *Y;
1686 if (!R || !match(R, m_LogicalAnd(
1687 m_LogicalAnd(m_Specific(HeaderMask), m_VPValue(X)),
1688 m_VPValue(Y))))
1689 continue;
1690 append_range(Worklist, R->users());
1691 VPBuilder Builder(R);
1692 R->replaceAllUsesWith(
1693 Builder.createLogicalAnd(HeaderMask, Builder.createLogicalAnd(X, Y)));
1694 }
1695}
1696
1697static std::optional<Instruction::BinaryOps>
1699 switch (ID) {
1700 case Intrinsic::masked_udiv:
1701 return Instruction::UDiv;
1702 case Intrinsic::masked_sdiv:
1703 return Instruction::SDiv;
1704 case Intrinsic::masked_urem:
1705 return Instruction::URem;
1706 case Intrinsic::masked_srem:
1707 return Instruction::SRem;
1708 default:
1709 return {};
1710 }
1711}
1712
1714 if (Plan.hasScalarVFOnly())
1715 return;
1716
1718 vp_depth_first_deep(Plan.getEntry()))) {
1719 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB))) {
1722 continue;
1723 auto *RepR = dyn_cast<VPReplicateRecipe>(&R);
1724 if (RepR && (RepR->isSingleScalar() || RepR->isPredicated()))
1725 continue;
1726
1727 auto *RepOrWidenR = cast<VPRecipeWithIRFlags>(&R);
1728 if (RepR && RepR->getOpcode() == Instruction::Store &&
1729 vputils::isSingleScalar(RepR->getOperand(1))) {
1730 auto *Clone = new VPReplicateRecipe(
1731 RepOrWidenR->getUnderlyingInstr(), RepOrWidenR->operands(),
1732 true /*IsSingleScalar*/, nullptr /*Mask*/, *RepR /*Flags*/,
1733 *RepR /*Metadata*/, RepR->getDebugLoc());
1734 Clone->insertBefore(RepOrWidenR);
1735 VPBuilder Builder(Clone);
1736 VPValue *ExtractOp = Clone->getOperand(0);
1737 if (vputils::isUniformAcrossVFsAndUFs(RepR->getOperand(1)))
1738 ExtractOp =
1739 Builder.createNaryOp(VPInstruction::ExtractLastPart, ExtractOp);
1740 ExtractOp =
1741 Builder.createNaryOp(VPInstruction::ExtractLastLane, ExtractOp);
1742 Clone->setOperand(0, ExtractOp);
1743 RepR->eraseFromParent();
1744 continue;
1745 }
1746
1747 // Narrow llvm.masked.{u,s}{div,rem} intrinsics with a safe divisor.
1748 if (auto *IntrR = dyn_cast<VPWidenIntrinsicRecipe>(RepOrWidenR)) {
1749 if (!vputils::onlyFirstLaneUsed(IntrR))
1750 continue;
1751 auto Opc = getUnmaskedDivRemOpcode(IntrR->getVectorIntrinsicID());
1752 if (!Opc)
1753 continue;
1754 VPBuilder Builder(IntrR);
1755 VPValue *SafeDivisor = Builder.createSelect(
1756 IntrR->getOperand(2), IntrR->getOperand(1),
1757 Plan.getConstantInt(IntrR->getScalarType(), 1));
1758 VPValue *Clone = Builder.createNaryOp(
1759 *Opc, {IntrR->getOperand(0), SafeDivisor},
1760 VPIRFlags::getDefaultFlags(*Opc), IntrR->getDebugLoc());
1761 IntrR->replaceAllUsesWith(Clone);
1762 IntrR->eraseFromParent();
1763 continue;
1764 }
1765
1766 // Skip recipes that aren't single scalars.
1767 if (!vputils::isSingleScalar(RepOrWidenR))
1768 continue;
1769
1770 // Predicate to check if a user of Op introduces extra broadcasts.
1771 auto IntroducesBCastOf = [](const VPValue *Op) {
1772 return [Op](const VPUser *U) {
1773 if (auto *VPI = dyn_cast<VPInstruction>(U)) {
1777 VPI->getOpcode()))
1778 return false;
1779 }
1780 return !U->usesScalars(Op);
1781 };
1782 };
1783
1784 if (any_of(RepOrWidenR->users(), IntroducesBCastOf(RepOrWidenR)) &&
1785 none_of(RepOrWidenR->operands(), [&](VPValue *Op) {
1786 if (any_of(
1787 make_filter_range(Op->users(), not_equal_to(RepOrWidenR)),
1788 IntroducesBCastOf(Op)))
1789 return false;
1790 // Non-constant live-ins require broadcasts, while constants do not
1791 // need explicit broadcasts.
1792 bool LiveInNeedsBroadcast =
1793 isa<VPIRValue>(Op) && !isa<VPConstant>(Op);
1794 auto *OpR = dyn_cast<VPReplicateRecipe>(Op);
1795 return LiveInNeedsBroadcast || (OpR && OpR->isSingleScalar());
1796 }))
1797 continue;
1798
1799 auto *Clone = VPBuilder::createSingleScalarOp(
1800 vputils::getOpcode(RepOrWidenR), RepOrWidenR->operands(),
1801 /*Mask=*/nullptr, *RepOrWidenR, {}, DebugLoc::getUnknown(),
1802 RepOrWidenR->getUnderlyingInstr());
1803 Clone->insertBefore(RepOrWidenR);
1804 RepOrWidenR->replaceAllUsesWith(Clone);
1805 if (vputils::isDeadRecipe(*RepOrWidenR))
1806 RepOrWidenR->eraseFromParent();
1807 }
1808 }
1809}
1810
1811/// Try to see if all of \p Blend's masks share a common value logically and'ed
1812/// and remove it from the masks.
1814 if (Blend->isNormalized())
1815 return;
1816 VPValue *CommonEdgeMask;
1817 if (!match(Blend->getMask(0),
1818 m_LogicalAnd(m_VPValue(CommonEdgeMask), m_VPValue())))
1819 return;
1820 for (unsigned I = 0; I < Blend->getNumIncomingValues(); I++)
1821 if (!match(Blend->getMask(I),
1822 m_LogicalAnd(m_Specific(CommonEdgeMask), m_VPValue())))
1823 return;
1824 for (unsigned I = 0; I < Blend->getNumIncomingValues(); I++)
1825 Blend->setMask(I, Blend->getMask(I)->getDefiningRecipe()->getOperand(1));
1826}
1827
1828/// Normalize and simplify VPBlendRecipes. Should be run after simplifyRecipes
1829/// to make sure the masks are simplified.
1830static void simplifyBlends(VPlan &Plan) {
1833 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
1834 auto *Blend = dyn_cast<VPBlendRecipe>(&R);
1835 if (!Blend)
1836 continue;
1837
1838 removeCommonBlendMask(Blend);
1839
1840 // Try to remove redundant blend recipes.
1841 SmallPtrSet<VPValue *, 4> UniqueValues;
1842 if (Blend->isNormalized() || !match(Blend->getMask(0), m_False()))
1843 UniqueValues.insert(Blend->getIncomingValue(0));
1844 for (unsigned I = 1; I != Blend->getNumIncomingValues(); ++I)
1845 if (!match(Blend->getMask(I), m_False()))
1846 UniqueValues.insert(Blend->getIncomingValue(I));
1847
1848 if (UniqueValues.size() == 1) {
1849 Blend->replaceAllUsesWith(*UniqueValues.begin());
1850 Blend->eraseFromParent();
1851 continue;
1852 }
1853
1854 if (Blend->isNormalized())
1855 continue;
1856
1857 // Normalize the blend so its first incoming value is used as the initial
1858 // value with the others blended into it.
1859
1860 unsigned StartIndex = 0;
1861 for (unsigned I = 0; I != Blend->getNumIncomingValues(); ++I) {
1862 // If a value's mask is used only by the blend then is can be deadcoded.
1863 // TODO: Find the most expensive mask that can be deadcoded, or a mask
1864 // that's used by multiple blends where it can be removed from them all.
1865 VPValue *Mask = Blend->getMask(I);
1866 if (Mask->hasOneUse() && !match(Mask, m_False())) {
1867 StartIndex = I;
1868 break;
1869 }
1870 }
1871
1872 SmallVector<VPValue *, 4> OperandsWithMask;
1873 OperandsWithMask.push_back(Blend->getIncomingValue(StartIndex));
1874
1875 for (unsigned I = 0; I != Blend->getNumIncomingValues(); ++I) {
1876 if (I == StartIndex)
1877 continue;
1878 OperandsWithMask.push_back(Blend->getIncomingValue(I));
1879 OperandsWithMask.push_back(Blend->getMask(I));
1880 }
1881
1882 auto *NewBlend =
1883 new VPBlendRecipe(cast_or_null<PHINode>(Blend->getUnderlyingValue()),
1884 OperandsWithMask, *Blend, Blend->getDebugLoc());
1885 NewBlend->insertBefore(&R);
1886
1887 VPValue *DeadMask = Blend->getMask(StartIndex);
1888 Blend->replaceAllUsesWith(NewBlend);
1889 Blend->eraseFromParent();
1891
1892 /// Simplify BLEND %a, %b, Not(%mask) -> BLEND %b, %a, %mask.
1893 VPValue *NewMask;
1894 if (NewBlend->getNumOperands() == 3 &&
1895 match(NewBlend->getMask(1), m_Not(m_VPValue(NewMask)))) {
1896 VPValue *Inc0 = NewBlend->getOperand(0);
1897 VPValue *Inc1 = NewBlend->getOperand(1);
1898 VPValue *OldMask = NewBlend->getOperand(2);
1899 NewBlend->setOperand(0, Inc1);
1900 NewBlend->setOperand(1, Inc0);
1901 NewBlend->setOperand(2, NewMask);
1902 if (OldMask->user_empty())
1903 cast<VPInstruction>(OldMask)->eraseFromParent();
1904 }
1905 }
1906 }
1907}
1908
1909/// Optimize the width of vector induction variables in \p Plan based on a known
1910/// constant Trip Count, \p BestVF and \p BestUF.
1912 ElementCount BestVF,
1913 unsigned BestUF) {
1914 // Only proceed if we have not completely removed the vector region.
1915 if (!Plan.getVectorLoopRegion())
1916 return false;
1917
1918 const APInt *TC;
1919 if (!BestVF.isFixed() || !match(Plan.getTripCount(), m_APInt(TC)))
1920 return false;
1921
1922 // Calculate the minimum power-of-2 bit width that can fit the known TC, VF
1923 // and UF. Returns at least 8.
1924 auto ComputeBitWidth = [](APInt TC, uint64_t Align) {
1925 APInt AlignedTC =
1928 APInt MaxVal = AlignedTC - 1;
1929 return std::max<unsigned>(PowerOf2Ceil(MaxVal.getActiveBits()), 8);
1930 };
1931 unsigned NewBitWidth =
1932 ComputeBitWidth(*TC, BestVF.getKnownMinValue() * BestUF);
1933
1934 LLVMContext &Ctx = Plan.getContext();
1935 auto *NewIVTy = IntegerType::get(Ctx, NewBitWidth);
1936
1937 bool MadeChange = false;
1938
1939 VPBasicBlock *HeaderVPBB = Plan.getVectorLoopRegion()->getEntryBasicBlock();
1940 for (VPRecipeBase &Phi : HeaderVPBB->phis()) {
1941 // Currently only handle canonical IVs as it is trivial to replace the start
1942 // and stop values, and we currently only perform the optimization when the
1943 // IV has a single use.
1945 if (!match(&Phi, m_CanonicalWidenIV(WideIV)))
1946 continue;
1947 if (WideIV->hasMoreThanOneUniqueUser() ||
1948 NewIVTy == WideIV->getScalarType())
1949 continue;
1950
1951 // Currently only handle cases where the single user is a header-mask
1952 // comparison with the backedge-taken-count.
1953 VPUser *SingleUser = WideIV->getSingleUser();
1954 if (!SingleUser ||
1955 !match(SingleUser,
1956 m_ICmp(m_Specific(WideIV),
1958 continue;
1959
1960 // Update IV operands and comparison bound to use new narrower type.
1961 assert(!WideIV->getTruncInst() &&
1962 "canonical IV is not expected to have a truncation");
1963 auto *NewWideIV = new VPWidenIntOrFpInductionRecipe(
1964 WideIV->getPHINode(), Plan.getZero(NewIVTy),
1965 Plan.getConstantInt(NewIVTy, 1), WideIV->getVFValue(),
1966 WideIV->getInductionDescriptor(), *WideIV, WideIV->getDebugLoc());
1967 NewWideIV->insertBefore(WideIV);
1968
1969 auto *NewBTC = new VPWidenCastRecipe(
1970 Instruction::Trunc, Plan.getOrCreateBackedgeTakenCount(), NewIVTy,
1971 nullptr, VPIRFlags::getDefaultFlags(Instruction::Trunc));
1972 Plan.getVectorPreheader()->appendRecipe(NewBTC);
1973 auto *Cmp = cast<VPInstruction>(WideIV->getSingleUser());
1974 Cmp->replaceAllUsesWith(
1975 VPBuilder(Cmp).createICmp(Cmp->getPredicate(), NewWideIV, NewBTC));
1976
1977 MadeChange = true;
1978 }
1979
1980 return MadeChange;
1981}
1982
1983/// Return true if \p Cond is known to be true for given \p BestVF and \p
1984/// BestUF.
1986 ElementCount BestVF, unsigned BestUF,
1989 return any_of(Cond->getDefiningRecipe()->operands(), [&Plan, BestVF, BestUF,
1990 &PSE](VPValue *C) {
1991 return isConditionTrueViaVFAndUF(C, Plan, BestVF, BestUF, PSE);
1992 });
1993
1994 auto *CanIV = Plan.getVectorLoopRegion()->getCanonicalIV();
1997 m_c_Add(m_Specific(CanIV), m_Specific(&Plan.getVFxUF())),
1998 m_Specific(&Plan.getVectorTripCount()))))
1999 return false;
2000
2001 // The compare checks CanIV + VFxUF == vector trip count. The vector trip
2002 // count is not conveniently available as SCEV so far, so we compare directly
2003 // against the original trip count. This is stricter than necessary, as we
2004 // will only return true if the trip count == vector trip count.
2005 const SCEV *VectorTripCount =
2007 if (isa<SCEVCouldNotCompute>(VectorTripCount))
2008 VectorTripCount = vputils::getSCEVExprForVPValue(Plan.getTripCount(), PSE);
2009 assert(!isa<SCEVCouldNotCompute>(VectorTripCount) &&
2010 "Trip count SCEV must be computable");
2011 ScalarEvolution &SE = *PSE.getSE();
2012 ElementCount NumElements = BestVF.multiplyCoefficientBy(BestUF);
2013 const SCEV *C = SE.getElementCount(VectorTripCount->getType(), NumElements);
2014 return SE.isKnownPredicate(CmpInst::ICMP_EQ, VectorTripCount, C);
2015}
2016
2017// Replaces ExtractVectorForPart instructions with ICMP when the VF is scalar
2018// and the source is a WideActiveLaneMask. The unused mask is removed later
2019// when removing dead recipes.
2021 ElementCount BestVF) {
2022 if (!BestVF.isScalar())
2023 return false;
2024
2025 bool MadeChange = false;
2026 VPBuilder Builder;
2027 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
2028 VPBasicBlock *PreheaderVPBB = Plan.getVectorPreheader();
2029 VPBasicBlock *ExitingVPBB = VectorRegion->getExitingBasicBlock();
2030
2031 VPValue *Start, *TC;
2032 uint64_t Idx;
2033 for (VPBasicBlock *VPBB : {PreheaderVPBB, ExitingVPBB}) {
2034 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
2037 m_VPValue()),
2038 m_ConstantInt(Idx))))
2039 continue;
2040
2041 auto *Extract = cast<VPInstruction>(&R);
2042 Builder.setInsertPoint(Extract);
2043
2044 if (Idx > 0)
2045 Start = Builder.createAdd(
2046 Start, Plan.getConstantInt(Start->getScalarType(), Idx));
2047
2048 VPValue *ICmp = Builder.createICmp(CmpInst::ICMP_ULT, Start, TC);
2049 Extract->replaceAllUsesWith(ICmp);
2050 Extract->eraseFromParent();
2051 MadeChange = true;
2052 }
2053 }
2054
2055 return MadeChange;
2056}
2057
2058/// Try to simplify the branch condition of \p Plan. This may restrict the
2059/// resulting plan to \p BestVF and \p BestUF.
2061 unsigned BestUF,
2063 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
2064 VPBasicBlock *ExitingVPBB = VectorRegion->getExitingBasicBlock();
2065 auto *Term = &ExitingVPBB->back();
2066 VPValue *Cond;
2067 auto m_CanIVInc = m_Add(m_VPValue(), m_Specific(&Plan.getVFxUF()));
2068 // Check if the branch condition compares the canonical IV increment (for main
2069 // loop), or the canonical IV increment plus an offset (for epilog loop).
2070 if (match(Term, m_BranchOnCount(
2071 m_CombineOr(m_CanIVInc, m_c_Add(m_CanIVInc, m_LiveIn())),
2072 m_VPValue())) ||
2073 match(Term,
2076 m_ZeroInt()))))) {
2077 // Try to simplify the branch condition if VectorTC <= VF * UF when the
2078 // latch terminator is BranchOnCount or
2079 // BranchOnCond(Not(ExtractVectorForPart(WideActiveLaneMask), 0))
2080 const SCEV *VectorTripCount =
2082 if (isa<SCEVCouldNotCompute>(VectorTripCount))
2083 VectorTripCount =
2085 assert(!isa<SCEVCouldNotCompute>(VectorTripCount) &&
2086 "Trip count SCEV must be computable");
2087 ScalarEvolution &SE = *PSE.getSE();
2088 ElementCount NumElements = BestVF.multiplyCoefficientBy(BestUF);
2089 const SCEV *C = SE.getElementCount(VectorTripCount->getType(), NumElements);
2090 if (!SE.isKnownPredicate(CmpInst::ICMP_ULE, VectorTripCount, C))
2091 return false;
2092 } else if (match(Term, m_BranchOnCond(m_VPValue(Cond))) ||
2094 // For BranchOnCond, check if we can prove the condition to be true using VF
2095 // and UF.
2096 if (!isConditionTrueViaVFAndUF(Cond, Plan, BestVF, BestUF, PSE))
2097 return false;
2098 } else {
2099 return false;
2100 }
2101
2102 // The vector loop region only executes once. Convert terminator of the
2103 // exiting block to exit in the first iteration.
2104 if (match(Term, m_BranchOnTwoConds())) {
2105 Term->setOperand(1, Plan.getTrue());
2106 return true;
2107 }
2108
2109 auto *BOC = new VPInstruction(VPInstruction::BranchOnCond, Plan.getTrue(), {},
2110 {}, Term->getDebugLoc());
2111 ExitingVPBB->appendRecipe(BOC);
2112 Term->eraseFromParent();
2113
2114 return true;
2115}
2116
2118 unsigned BestUF,
2120 assert(Plan.hasVF(BestVF) && "BestVF is not available in Plan");
2121 assert(Plan.hasUF(BestUF) && "BestUF is not available in Plan");
2122
2123 bool MadeChange =
2124 simplifyBranchConditionForVFAndUF(Plan, BestVF, BestUF, PSE);
2125 MadeChange |= replaceMaskWithCompareForScalarPlan(Plan, BestVF);
2126 MadeChange |= optimizeVectorInductionWidthForTCAndVFUF(Plan, BestVF, BestUF);
2127
2128 if (MadeChange) {
2129 Plan.setVF(BestVF);
2130 assert(Plan.getConcreteUF() == BestUF && "BestUF must match the Plan's UF");
2131 }
2132}
2133
2135 for (VPRecipeBase &R :
2137 auto *PhiR = dyn_cast<VPReductionPHIRecipe>(&R);
2138 if (!PhiR)
2139 continue;
2140 RecurKind RK = PhiR->getRecurrenceKind();
2141 if (RK != RecurKind::Add && RK != RecurKind::Mul && RK != RecurKind::Sub &&
2143 continue;
2144
2146 if (auto *RecWithFlags = dyn_cast<VPRecipeWithIRFlags>(U)) {
2147 RecWithFlags->dropPoisonGeneratingFlags();
2148 }
2149 }
2150}
2151
2152namespace {
2153struct VPCSEDenseMapInfo : public DenseMapInfo<VPSingleDefRecipe *> {
2154 /// If recipe \p R will lower to a GEP with a non-i8 source element type,
2155 /// return that source element type.
2156 static Type *getGEPSourceElementType(const VPSingleDefRecipe *R) {
2157 // All VPInstructions that lower to GEPs must have the i8 source element
2158 // type (as they are PtrAdds), so we omit it.
2160 .Case([](const VPReplicateRecipe *I) -> Type * {
2161 if (auto *GEP = dyn_cast<GetElementPtrInst>(I->getUnderlyingValue()))
2162 return GEP->getSourceElementType();
2163 return nullptr;
2164 })
2165 .Case<VPVectorPointerRecipe, VPWidenGEPRecipe>(
2166 [](auto *I) { return I->getSourceElementType(); })
2167 .Default([](auto *) { return nullptr; });
2168 }
2169
2170 /// Returns true if recipe \p Def can be safely handed for CSE.
2171 static bool canHandle(const VPSingleDefRecipe *Def) {
2172 // We can extend the list of handled recipes in the future,
2173 // provided we account for the data embedded in them while checking for
2174 // equality or hashing.
2176
2177 // The issue with (Insert|Extract)Value is that the index of the
2178 // insert/extract is not a proper operand in LLVM IR, and hence also not in
2179 // VPlan.
2180 if (!C || (!C->first && (C->second == Instruction::InsertValue ||
2181 C->second == Instruction::ExtractValue)))
2182 return false;
2183
2184 // During CSE, we can only handle non-memory recipes, as memory can alias.
2185 return !Def->mayReadOrWriteMemory();
2186 }
2187
2188 /// Hash the underlying data of \p Def.
2189 static unsigned getHashValue(const VPSingleDefRecipe *Def) {
2190 hash_code Result = hash_combine(
2191 Def->getVPRecipeID(), vputils::getOpcodeOrIntrinsicID(Def),
2192 getGEPSourceElementType(Def), Def->getScalarType(),
2194 if (auto *RFlags = dyn_cast<VPRecipeWithIRFlags>(Def))
2195 if (RFlags->hasPredicate())
2196 return hash_combine(Result, RFlags->getPredicate());
2197 if (auto *SIVSteps = dyn_cast<VPScalarIVStepsRecipe>(Def))
2198 return hash_combine(Result, SIVSteps->getInductionOpcode());
2199 return Result;
2200 }
2201
2202 /// Check equality of underlying data of \p L and \p R.
2203 static bool isEqual(const VPSingleDefRecipe *L, const VPSingleDefRecipe *R) {
2204 if (L->getVPRecipeID() != R->getVPRecipeID() ||
2207 getGEPSourceElementType(L) != getGEPSourceElementType(R) ||
2209 !equal(L->operands(), R->operands()))
2210 return false;
2213 "must have valid opcode info for both recipes");
2214 if (auto *LFlags = dyn_cast<VPRecipeWithIRFlags>(L))
2215 if (LFlags->hasPredicate() &&
2216 LFlags->getPredicate() !=
2217 cast<VPRecipeWithIRFlags>(R)->getPredicate())
2218 return false;
2219 if (auto *LSIV = dyn_cast<VPScalarIVStepsRecipe>(L))
2220 if (LSIV->getInductionOpcode() !=
2221 cast<VPScalarIVStepsRecipe>(R)->getInductionOpcode())
2222 return false;
2223 // Phi recipes can only be equal if they are in the same VPBB, as they
2224 // implicitly depend on their predecessors.
2225 if (isa<VPWidenPHIRecipe>(L) && L->getParent() != R->getParent())
2226 return false;
2227 // Recipes in replicate regions implicitly depend on predicate. If either
2228 // recipe is in a replicate region, only consider them equal if both have
2229 // the same parent.
2230 const VPRegionBlock *RegionL = L->getRegion();
2231 const VPRegionBlock *RegionR = R->getRegion();
2232 if (((RegionL && RegionL->isReplicator()) ||
2233 (RegionR && RegionR->isReplicator())) &&
2234 L->getParent() != R->getParent())
2235 return false;
2236 return L->getScalarType() == R->getScalarType();
2237 }
2238};
2239} // end anonymous namespace
2240
2241/// Perform a common-subexpression-elimination of VPSingleDefRecipes on the \p
2242/// Plan.
2244 VPDominatorTree VPDT(Plan);
2246
2248 Plan.getEntry());
2250 for (VPRecipeBase &R : *VPBB) {
2251 auto *Def = dyn_cast<VPSingleDefRecipe>(&R);
2252 if (!Def || !VPCSEDenseMapInfo::canHandle(Def))
2253 continue;
2254 if (VPSingleDefRecipe *V = CSEMap.lookup(Def)) {
2255 // V must dominate Def for a valid replacement.
2256 if (!VPDT.dominates(V->getParent(), VPBB))
2257 continue;
2258 // Only keep flags present on both V and Def.
2259 if (auto *RFlags = dyn_cast<VPRecipeWithIRFlags>(V))
2260 RFlags->intersectFlags(*cast<VPRecipeWithIRFlags>(Def));
2261 Def->replaceAllUsesWith(V);
2262 continue;
2263 }
2264 CSEMap[Def] = Def;
2265 }
2266 }
2267}
2268
2269/// Return true if we do not know how to (mechanically) hoist or sink a
2270/// non-memory or memory recipe \p R out of a loop region. When sinking, passing
2271/// \p Sinking = true ensures that assumes aren't sunk.
2273 VPBasicBlock *LastBB,
2274 bool Sinking = false) {
2275 if (!isa<VPReplicateRecipe>(R) || !R.mayReadOrWriteMemory() ||
2277 return vputils::cannotHoistOrSinkRecipe(R, Sinking);
2278
2279 // Check that the memory operation doesn't alias between FirstBB and LastBB.
2280 auto MemLoc = vputils::getMemoryLocation(R);
2281
2282 // TODO: Could make use of SinkStoreInfo::isNoAliasViaDistance by collecting
2283 // stores upfront, and constructing a full SinkStoreInfo.
2284 auto SinkInfo =
2285 Sinking ? std::make_optional(SinkStoreInfo(cast<VPReplicateRecipe>(R)))
2286 : std::nullopt;
2287
2288 return !MemLoc ||
2289 !canHoistOrSinkWithNoAliasCheck(*MemLoc, FirstBB, LastBB, SinkInfo);
2290}
2291
2292/// Move loop-invariant recipes out of the vector loop region in \p Plan.
2293static void licm(VPlan &Plan) {
2294 VPBasicBlock *Preheader = Plan.getVectorPreheader();
2295
2296 // Hoist any loop invariant recipes from the vector loop region to the
2297 // preheader. Preform a shallow traversal of the vector loop region, to
2298 // exclude recipes in replicate regions. Since the top-level blocks in the
2299 // vector loop region are guaranteed to execute if the vector pre-header is,
2300 // we don't need to check speculation safety.
2301 VPRegionBlock *LoopRegion = Plan.getVectorLoopRegion();
2302 assert(Preheader->getSingleSuccessor() == LoopRegion &&
2303 "Expected vector prehader's successor to be the vector loop region");
2305 vp_depth_first_shallow(LoopRegion->getEntry()))) {
2306 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
2307 if (cannotHoistOrSinkRecipe(R, LoopRegion->getEntryBasicBlock(),
2308 LoopRegion->getExitingBasicBlock()))
2309 continue;
2310 if (any_of(R.operands(), [](VPValue *Op) {
2311 return !Op->isDefinedOutsideLoopRegions();
2312 }))
2313 continue;
2314 R.moveBefore(*Preheader, Preheader->end());
2315 }
2316 }
2317
2318#ifndef NDEBUG
2319 VPDominatorTree VPDT(Plan);
2320#endif
2321 // Sink recipes with no users inside the vector loop region if all users are
2322 // in the same exit block of the region.
2323 // TODO: Extend to sink recipes from inner loops.
2325 LoopRegion->getEntry());
2327 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB))) {
2328 if (cannotHoistOrSinkRecipe(R, LoopRegion->getEntryBasicBlock(),
2329 LoopRegion->getExitingBasicBlock(),
2330 /*Sinking=*/true))
2331 continue;
2332
2333 if (auto *RepR = dyn_cast<VPReplicateRecipe>(&R)) {
2334 assert(!RepR->isPredicated() &&
2335 "Expected prior transformation of predicated replicates to "
2336 "replicate regions");
2337 // narrowToSingleScalarRecipes should have already maximally narrowed
2338 // replicates to single-scalar replicates.
2339 // TODO: When unrolling, replicateByVF doesn't handle sunk
2340 // non-single-scalar replicates correctly.
2341 if (!RepR->isSingleScalar())
2342 continue;
2343
2344 // The pointer operand of stores must be loop-invariant.
2345 if (RepR->getOpcode() == Instruction::Store &&
2346 !RepR->getOperand(1)->isDefinedOutsideLoopRegions())
2347 continue;
2348 }
2349
2350 [[maybe_unused]] auto *RepR = dyn_cast<VPReplicateRecipe>(&R);
2351 assert((!R.mayWriteToMemory() ||
2352 (RepR && RepR->getOpcode() == Instruction::Store &&
2353 RepR->getOperand(1)->isDefinedOutsideLoopRegions())) &&
2354 "The only recipes that may write to memory are expected to be "
2355 "stores with invariant pointer-operand");
2356
2357 // TODO: Use R.definedValues() instead of casting to VPSingleDefRecipe to
2358 // support recipes with multiple defined values (e.g., interleaved loads).
2359 auto *Def = cast<VPSingleDefRecipe>(&R);
2360
2361 // Cannot sink the recipe if the user is defined in a loop region or a
2362 // non-successor of the vector loop region. Cannot sink if user is a phi
2363 // either.
2364 VPBasicBlock *SinkBB = nullptr;
2365 if (any_of(Def->users(), [&SinkBB, &LoopRegion](VPUser *U) {
2366 auto *UserR = cast<VPRecipeBase>(U);
2367 VPBasicBlock *Parent = UserR->getParent();
2368 // TODO: Support sinking when users are in multiple blocks.
2369 if (SinkBB && SinkBB != Parent)
2370 return true;
2371 SinkBB = Parent;
2372 // TODO: If the user is a PHI node, we should check the block of
2373 // incoming value. Support PHI node users if needed.
2374 return UserR->isPhi() || Parent->getEnclosingLoopRegion() ||
2375 Parent->getSinglePredecessor() != LoopRegion;
2376 }))
2377 continue;
2378
2379 if (!SinkBB)
2380 SinkBB = cast<VPBasicBlock>(LoopRegion->getSingleSuccessor());
2381
2382 // TODO: This will need to be a check instead of a assert after
2383 // conditional branches in vectorized loops are supported.
2384 assert(VPDT.properlyDominates(VPBB, SinkBB) &&
2385 "Defining block must dominate sink block");
2386 // TODO: Clone the recipe if users are on multiple exit paths, instead of
2387 // just moving.
2388 Def->moveBefore(*SinkBB, SinkBB->getFirstNonPhi());
2389 }
2390 }
2391}
2392
2394 VPlan &Plan, const MapVector<Instruction *, uint64_t> &MinBWs) {
2395 if (Plan.hasScalarVFOnly())
2396 return;
2397 // Keep track of created truncates, so they can be re-used. Note that we
2398 // cannot use RAUW after creating a new truncate, as this would could make
2399 // other uses have different types for their operands, making them invalidly
2400 // typed.
2402 VPBasicBlock *PH = Plan.getVectorPreheader();
2405 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
2408 continue;
2409
2410 VPValue *ResultVPV = R.getVPSingleValue();
2411 auto *UI = cast_or_null<Instruction>(ResultVPV->getUnderlyingValue());
2412 unsigned NewResSizeInBits = MinBWs.lookup(UI);
2413 if (!NewResSizeInBits)
2414 continue;
2415
2416 // If the value wasn't vectorized, we must maintain the original scalar
2417 // type. Skip those here, after incrementing NumProcessedRecipes. Also
2418 // skip casts which do not need to be handled explicitly here, as
2419 // redundant casts will be removed during recipe simplification.
2421 continue;
2422
2423 Type *OldResTy = ResultVPV->getScalarType();
2424 unsigned OldResSizeInBits = OldResTy->getScalarSizeInBits();
2425 assert(OldResTy->isIntegerTy() && "only integer types supported");
2426 (void)OldResSizeInBits;
2427
2428 auto *NewResTy = IntegerType::get(Plan.getContext(), NewResSizeInBits);
2429
2430 // Any wrapping introduced by shrinking this operation shouldn't be
2431 // considered undefined behavior. So, we can't unconditionally copy
2432 // arithmetic wrapping flags to VPW.
2433 if (auto *VPW = dyn_cast<VPRecipeWithIRFlags>(&R))
2434 VPW->dropPoisonGeneratingFlags();
2435
2436 assert((OldResSizeInBits != NewResSizeInBits ||
2437 match(&R, m_ICmp(m_VPValue(), m_VPValue()))) &&
2438 "Only ICmps should not need extending the result.");
2439 assert(!isa<VPWidenStoreRecipe>(&R) && "stores cannot be narrowed");
2440
2441 // For loads/intrinsics we don't recreate the recipe; just wrap the
2442 // original wide result in a ZExt to OldResTy.
2444 if (OldResSizeInBits != NewResSizeInBits) {
2446 Instruction::ZExt, ResultVPV, OldResTy);
2447 ResultVPV->replaceAllUsesWith(Ext);
2448 Ext->setOperand(0, ResultVPV);
2449 }
2450 continue;
2451 }
2452
2453 // Shrink operands by introducing truncates as needed.
2454 unsigned StartIdx =
2455 match(&R, m_Select(m_VPValue(), m_VPValue(), m_VPValue())) ? 1 : 0;
2456 SmallVector<VPValue *> NewOperands(R.operands());
2457 for (VPValue *&Op : drop_begin(NewOperands, StartIdx)) {
2458 unsigned OpSizeInBits = Op->getScalarType()->getScalarSizeInBits();
2459 if (OpSizeInBits == NewResSizeInBits)
2460 continue;
2461 assert(OpSizeInBits > NewResSizeInBits && "nothing to truncate");
2462 auto [ProcessedIter, Inserted] = ProcessedTruncs.try_emplace(Op);
2463 if (Inserted) {
2464 VPBuilder Builder;
2465 if (isa<VPIRValue>(Op))
2466 Builder.setInsertPoint(PH);
2467 else
2468 Builder.setInsertPoint(&R);
2469 ProcessedIter->second =
2470 Builder.createWidenCast(Instruction::Trunc, Op, NewResTy);
2471 }
2472 Op = ProcessedIter->second;
2473 }
2474
2475 auto *NWR = cast<VPWidenRecipe>(&R)->cloneWithOperands(NewOperands);
2476 NWR->insertBefore(&R);
2477
2478 // Wrap NWR in a ZExt to preserve the original wide type for downstream
2479 // users (unless this is an ICmp, which produces i1 regardless).
2480 VPValue *Replacement = NWR->getVPSingleValue();
2481 if (OldResSizeInBits != NewResSizeInBits)
2482 Replacement =
2484 .createWidenCast(Instruction::ZExt, Replacement, OldResTy)
2485 ->getVPSingleValue();
2486 ResultVPV->replaceAllUsesWith(Replacement);
2487 R.eraseFromParent();
2488 }
2489 }
2490}
2491
2492bool VPlanTransforms::removeBranchOnConst(VPlan &Plan, bool OnlyLatches) {
2493 std::optional<VPDominatorTree> VPDT;
2494 if (OnlyLatches)
2495 VPDT.emplace(Plan);
2496
2497 // Collect all blocks before modifying the CFG so we can identify unreachable
2498 // ones after constant branch removal.
2500
2501 bool SimplifiedPhi = false;
2502 for (VPBasicBlock *VPBB : VPBlockUtils::blocksOnly<VPBasicBlock>(AllBlocks)) {
2503 VPValue *Cond;
2504 // Skip blocks that are not terminated by BranchOnCond.
2505 if (VPBB->empty() || !match(&VPBB->back(), m_BranchOnCond(m_VPValue(Cond))))
2506 continue;
2507
2508 if (OnlyLatches && !VPBlockUtils::isLatch(VPBB, *VPDT))
2509 continue;
2510
2511 assert(VPBB->getNumSuccessors() == 2 &&
2512 "Two successors expected for BranchOnCond");
2513 unsigned RemovedIdx;
2514 if (match(Cond, m_True()))
2515 RemovedIdx = 1;
2516 else if (match(Cond, m_False()))
2517 RemovedIdx = 0;
2518 else
2519 continue;
2520
2521 VPBasicBlock *RemovedSucc =
2522 cast<VPBasicBlock>(VPBB->getSuccessors()[RemovedIdx]);
2523 assert(count(RemovedSucc->getPredecessors(), VPBB) == 1 &&
2524 "There must be a single edge between VPBB and its successor");
2525 // Values coming from VPBB into phi recipes of RemovedSucc are removed from
2526 // these recipes and single-entry header phis are removed.
2527 for (VPRecipeBase &R : make_early_inc_range(RemovedSucc->phis())) {
2528 cast<VPPhiAccessors>(&R)->removeIncomingValueFor(VPBB);
2529 SimplifiedPhi = true;
2530 // Remove now invalid header phis that are left single-entry after
2531 // removing their backedges.
2532 auto *PhiR = dyn_cast<VPHeaderPHIRecipe>(&R);
2533 if (!PhiR || PhiR->getNumIncoming() != 1)
2534 continue;
2535 PhiR->replaceAllUsesWith(PhiR->getOperand(0));
2536 PhiR->eraseFromParent();
2537 }
2538
2539 // Disconnect blocks and remove the terminator.
2540 VPBlockUtils::disconnectBlocks(VPBB, RemovedSucc);
2541 VPBB->back().eraseFromParent();
2542 }
2543
2544 // Compute which blocks are still reachable from the entry after constant
2545 // branch removal.
2548
2549 // Detach all unreachable blocks from their successors, removing their recipes
2550 // and incoming values from phi recipes.
2551 VPSymbolicValue Tmp(nullptr);
2552 for (VPBlockBase *B : AllBlocks) {
2553 if (Reachable.contains(B))
2554 continue;
2555 for (VPBlockBase *Succ : to_vector(B->successors())) {
2556 if (auto *SuccBB = dyn_cast<VPBasicBlock>(Succ))
2557 for (VPRecipeBase &R : SuccBB->phis())
2558 cast<VPPhiAccessors>(&R)->removeIncomingValueFor(B);
2560 }
2561 for (VPBasicBlock *DeadBB :
2563 for (VPRecipeBase &R : make_early_inc_range(*DeadBB)) {
2564 for (VPValue *Def : R.definedValues())
2565 Def->replaceAllUsesWith(&Tmp);
2566 R.eraseFromParent();
2567 }
2568 }
2569 }
2570 return SimplifiedPhi;
2571}
2572
2593
2596 auto GetSimplifiedLiveInViaSCEV = [&](VPValue *VPV) -> VPValue * {
2597 const SCEV *Expr = vputils::getSCEVExprForVPValue(VPV, PSE);
2598 const APInt *C;
2599 if (match(Expr, m_scev_APInt(C)))
2600 return Plan.getConstantInt(*C);
2601 return nullptr;
2602 };
2603
2604 for (VPValue *LiveIn : to_vector(Plan.getLiveIns())) {
2605 if (VPValue *SimplifiedLiveIn = GetSimplifiedLiveInViaSCEV(LiveIn))
2606 LiveIn->replaceAllUsesWith(SimplifiedLiveIn);
2607 }
2608}
2609
2611 VPlan &Plan, PredicatedScalarEvolution &PSE,
2612 const SymbolicStrideMap &StridesMap, const VPDominatorTree &VPDT) {
2613 // Replace VPValues for known constant strides guaranteed by predicated scalar
2614 // evolution that are guaranteed to be guarded by the runtime checks; that is,
2615 // blocks dominated by the vector header.
2616 assert(!Plan.getVectorLoopRegion() &&
2617 "expected to run before loop regions are created");
2618 const auto &[Header, _] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
2619 auto CanUseVersionedStride = [&VPDT, Header = Header, &Plan](VPUser &U,
2620 unsigned Idx) {
2621 auto *R = cast<VPRecipeBase>(&U);
2622 // Skip phis if the loop if loop is not yet guarded.
2623 if (isa<VPPhiAccessors>(R) &&
2624 Header == Plan.getEntry()->getSingleSuccessor())
2625 return false;
2626 return VPDT.dominates(Header, R->getParent());
2627 };
2628 ValueToSCEVMapTy RewriteMap;
2629 for (const SCEVUnknown *Stride : StridesMap.values()) {
2630 Value *StrideV = Stride->getValue();
2631 const APInt *StrideConst;
2632 const SCEV *StrideExpr = PSE.getSCEV(StrideV);
2633 if (!match(StrideExpr, m_scev_APInt(StrideConst)))
2634 // Only handle constant strides for now.
2635 continue;
2636 if (VPValue *StrideVPV = Plan.getLiveIn(StrideV))
2637 StrideVPV->replaceUsesWithIf(Plan.getConstantInt(*StrideConst),
2638 CanUseVersionedStride);
2639
2640 // The versioned value may not be used in the loop directly but through an
2641 // integral cast (sext/zext/trunc). Add new live-ins in those cases.
2642 for (Value *U : StrideV->users()) {
2644 continue;
2645 VPValue *StrideVPV = Plan.getLiveIn(U);
2646 if (!StrideVPV)
2647 continue;
2648 unsigned BW = U->getType()->getScalarSizeInBits();
2649 APInt C = isa<SExtInst>(U) ? StrideConst->sext(BW)
2650 : StrideConst->zextOrTrunc(BW);
2651 StrideVPV->replaceUsesWithIf(Plan.getConstantInt(C),
2652 CanUseVersionedStride);
2653 }
2654 RewriteMap[StrideV] = StrideExpr;
2655 }
2656
2657 for (VPRecipeBase &R : *Plan.getEntry()) {
2658 auto *ExpSCEV = dyn_cast<VPExpandSCEVRecipe>(&R);
2659 if (!ExpSCEV)
2660 continue;
2661 const SCEV *ScevExpr = ExpSCEV->getSCEV();
2662 auto *NewSCEV =
2663 SCEVParameterRewriter::rewrite(ScevExpr, *PSE.getSE(), RewriteMap);
2664 if (NewSCEV != ScevExpr) {
2665 VPValue *NewExp = vputils::getOrCreateVPValueForSCEVExpr(Plan, NewSCEV);
2666 ExpSCEV->replaceAllUsesWith(NewExp);
2667 if (Plan.getTripCount() == ExpSCEV)
2668 Plan.resetTripCount(NewExp);
2669 }
2670 }
2671}
2672
2674 // Collect recipes in the backward slice of `Root` that may generate a poison
2675 // value that is used after vectorization.
2677 auto CollectPoisonGeneratingInstrsInBackwardSlice([&](VPRecipeBase *Root) {
2679 Worklist.push_back(Root);
2680
2681 // Traverse the backward slice of Root through its use-def chain.
2682 while (!Worklist.empty()) {
2683 VPRecipeBase *CurRec = Worklist.pop_back_val();
2684
2685 if (!Visited.insert(CurRec).second)
2686 continue;
2687
2688 // Prune search if we find another recipe generating a widen memory
2689 // instruction. Widen memory instructions involved in address computation
2690 // will lead to gather/scatter instructions, which don't need to be
2691 // handled.
2693 VPHeaderPHIRecipe>(CurRec))
2694 continue;
2695
2696 // This recipe contributes to the address computation of a widen
2697 // load/store. If the underlying instruction has poison-generating flags,
2698 // drop them directly.
2699 if (auto *RecWithFlags = dyn_cast<VPRecipeWithIRFlags>(CurRec)) {
2700 VPValue *A, *B;
2701 // Dropping disjoint from an OR may yield incorrect results, as some
2702 // analysis may have converted it to an Add implicitly (e.g. SCEV used
2703 // for dependence analysis). Instead, replace it with an equivalent Add.
2704 // This is possible as all users of the disjoint OR only access lanes
2705 // where the operands are disjoint or poison otherwise.
2706 if (match(RecWithFlags, m_BinaryOr(m_VPValue(A), m_VPValue(B))) &&
2707 RecWithFlags->isDisjoint()) {
2708 VPBuilder Builder(RecWithFlags);
2709 VPInstruction *New =
2710 Builder.createAdd(A, B, RecWithFlags->getDebugLoc());
2711 New->setUnderlyingValue(RecWithFlags->getUnderlyingValue());
2712 RecWithFlags->replaceAllUsesWith(New);
2713 RecWithFlags->eraseFromParent();
2714 CurRec = New;
2715 } else
2716 RecWithFlags->dropPoisonGeneratingFlags();
2717 } else {
2720 (void)Instr;
2721 assert((!Instr || !Instr->hasPoisonGeneratingFlags()) &&
2722 "found instruction with poison generating flags not covered by "
2723 "VPRecipeWithIRFlags");
2724 }
2725
2726 // Add new definitions to the worklist.
2727 for (VPValue *Operand : CurRec->operands())
2728 if (VPRecipeBase *OpDef = Operand->getDefiningRecipe())
2729 Worklist.push_back(OpDef);
2730 }
2731 });
2732
2733 // We want to exclude the tail folding case, as we don't need to drop flags
2734 // for operations computing the first lane in this case: the first lane of the
2735 // header mask must always be true. For reverse memory accesses, the mask is
2736 // wrapped in a Reverse, which is just a permutation of the header mask, so
2737 // peel it off before checking. The header mask is still the abstract region
2738 // value at this point (materialization happens later).
2739 auto m_UnlessHdrMask = m_Unless( // NOLINT
2741
2742 // Traverse all the recipes in the VPlan and collect the poison-generating
2743 // recipes in the backward slice starting at the address of a VPWidenRecipe or
2744 // VPInterleaveRecipe.
2745 auto Iter =
2748 for (VPRecipeBase &Recipe : *VPBB) {
2749 if (auto *WidenRec = dyn_cast<VPWidenMemoryRecipe>(&Recipe)) {
2750 VPRecipeBase *AddrDef = WidenRec->getAddr()->getDefiningRecipe();
2751 if (AddrDef && WidenRec->isConsecutive() && WidenRec->getMask() &&
2752 match(WidenRec->getMask(), m_UnlessHdrMask))
2753 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2754 } else if (auto *InterleaveRec = dyn_cast<VPInterleaveRecipe>(&Recipe)) {
2755 VPRecipeBase *AddrDef = InterleaveRec->getAddr()->getDefiningRecipe();
2756 if (AddrDef && InterleaveRec->getMask() &&
2757 match(InterleaveRec->getMask(), m_UnlessHdrMask))
2758 CollectPoisonGeneratingInstrsInBackwardSlice(AddrDef);
2759 }
2760 }
2761 }
2762}
2763
2765 VPlan &Plan,
2767 &InterleaveGroups,
2768 const bool &EpilogueAllowed) {
2769 if (InterleaveGroups.empty())
2770 return;
2771
2773 for (VPBasicBlock *VPBB :
2776 for (VPRecipeBase &R : make_filter_range(*VPBB, [](VPRecipeBase &R) {
2777 return isa<VPWidenMemoryRecipe>(&R);
2778 })) {
2779 auto *MemR = cast<VPWidenMemoryRecipe>(&R);
2780 IRMemberToRecipe[&MemR->getIngredient()] = MemR;
2781 }
2782
2783 // Interleave memory: for each Interleave Group we marked earlier as relevant
2784 // for this VPlan, replace the Recipes widening its memory instructions with a
2785 // single VPInterleaveRecipe at its insertion point.
2786 VPDominatorTree VPDT(Plan);
2787 for (const auto *IG : InterleaveGroups) {
2788 VPWidenMemoryRecipe *Start = nullptr;
2789 Instruction *StartMember = nullptr;
2790 for (auto *Member : IG->members())
2791 if (VPWidenMemoryRecipe *R = IRMemberToRecipe.lookup(Member)) {
2792 StartMember = Member;
2793 Start = R;
2794 break;
2795 }
2796 if (!StartMember) // All member recipes are dead, so the group is dead.
2797 continue;
2798 VPIRMetadata InterleaveMD(*Start);
2799 SmallVector<VPValue *, 4> StoredValues;
2800 for (unsigned I = 0; I < IG->getFactor(); ++I) {
2801 Instruction *MemberI = IG->getMember(I);
2802 if (!MemberI)
2803 continue;
2804 if (VPWidenMemoryRecipe *MemoryR = IRMemberToRecipe.lookup(MemberI)) {
2805 if (auto *StoreR = dyn_cast<VPWidenStoreRecipe>(MemoryR->getAsRecipe()))
2806 StoredValues.push_back(StoreR->getStoredValue());
2807 InterleaveMD.intersect(*MemoryR);
2808 } else {
2809 InterleaveMD.intersect(VPIRMetadata(*MemberI));
2810 }
2811 }
2812
2813 bool NeedsMaskForGaps =
2814 (IG->requiresScalarEpilogue() && !EpilogueAllowed) ||
2815 (!StoredValues.empty() && !IG->isFull());
2816
2817 Instruction *IRInsertPos = IG->getInsertPos();
2818 auto *InsertPos = IRMemberToRecipe.lookup(IRInsertPos);
2819 if (!InsertPos) {
2820 // InsertPos member is dead: find a new member that is alive.
2821 assert(isa<VPWidenLoadRecipe>(Start->getAsRecipe()) &&
2822 "Dead member in non-load group?");
2823 InsertPos = Start;
2824 for (Instruction *Member : IG->members())
2825 if (VPWidenMemoryRecipe *MemberR = IRMemberToRecipe.lookup(Member))
2826 if (VPDT.properlyDominates(MemberR->getAsRecipe(),
2827 InsertPos->getAsRecipe()))
2828 InsertPos = MemberR;
2829 IRInsertPos = &InsertPos->getIngredient();
2830 }
2831 VPRecipeBase *InsertPosR = InsertPos->getAsRecipe();
2832
2834 if (auto *Gep = dyn_cast<GetElementPtrInst>(
2835 getLoadStorePointerOperand(IRInsertPos)->stripPointerCasts()))
2836 NW = Gep->getNoWrapFlags().withoutNoUnsignedWrap();
2837
2838 // Get or create the start address for the interleave group.
2839 VPValue *Addr = Start->getAddr();
2840 VPRecipeBase *AddrDef = Addr->getDefiningRecipe();
2841 if (IG->getIndex(StartMember) != 0 ||
2842 (AddrDef && !VPDT.properlyDominates(AddrDef, InsertPosR))) {
2843 // Either member zero's recipe is dead, or we cannot re-use the address of
2844 // member zero because it does not dominate the insert position. Instead,
2845 // use the address of the insert position and create a PtrAdd adjusting it
2846 // to the address of member zero.
2847 // TODO: Hoist Addr's defining recipe (and any operands as needed) to
2848 // InsertPos or sink loads above zero members to join it.
2849 assert(IG->getIndex(IRInsertPos) != 0 &&
2850 "index of insert position shouldn't be zero");
2851 auto &DL = IRInsertPos->getDataLayout();
2852 APInt Offset(32,
2853 DL.getTypeAllocSize(getLoadStoreType(IRInsertPos)) *
2854 IG->getIndex(IRInsertPos),
2855 /*IsSigned=*/true);
2856 VPValue *OffsetVPV = Plan.getConstantInt(-Offset);
2857 VPBuilder B(InsertPosR);
2858 Addr = B.createNoWrapPtrAdd(InsertPos->getAddr(), OffsetVPV, NW);
2859 }
2860 // If the group is reverse, adjust the index to refer to the last vector
2861 // lane instead of the first. We adjust the index from the first vector
2862 // lane, rather than directly getting the pointer for lane VF - 1, because
2863 // the pointer operand of the interleaved access is supposed to be uniform.
2864 if (IG->isReverse()) {
2865 auto *ReversePtr = new VPVectorEndPointerRecipe(
2866 Addr, &Plan.getVF(), getLoadStoreType(IRInsertPos),
2867 -(int64_t)IG->getFactor(), NW, InsertPosR->getDebugLoc());
2868 ReversePtr->insertBefore(InsertPosR);
2869 Addr = ReversePtr;
2870 }
2871 auto *VPIG = new VPInterleaveRecipe(
2872 IG, Addr, StoredValues, InsertPos->getMask(), NeedsMaskForGaps,
2873 InterleaveMD, InsertPosR->getDebugLoc());
2874 VPIG->insertBefore(InsertPosR);
2875
2876 unsigned J = 0;
2877 for (unsigned i = 0; i < IG->getFactor(); ++i)
2878 if (Instruction *Member = IG->getMember(i)) {
2879 VPWidenMemoryRecipe *MemberR = IRMemberToRecipe.lookup(Member);
2880 if (!Member->getType()->isVoidTy()) {
2881 if (MemberR) {
2882 VPValue *OriginalV = MemberR->getAsRecipe()->getVPSingleValue();
2883 OriginalV->replaceAllUsesWith(VPIG->getVPValue(J));
2884 }
2885 J++;
2886 }
2887 if (MemberR)
2888 MemberR->getAsRecipe()->eraseFromParent();
2889 }
2890 }
2891}
2892
2893/// Returns the VPValue representing the uncountable exit comparison used by
2894/// AnyOf if the recipes it depends on can be traced back to live-ins and
2895/// the addresses (in GEP/PtrAdd form) of any (non-masked) load used in
2896/// generating the values for the comparison. The recipes are stored in
2897/// \p Recipes.
2898static std::optional<VPValue *>
2900 VPBasicBlock *LatchVPBB) {
2901 // Given a plain CFG VPlan loop with countable latch exiting block
2902 // \p LatchVPBB, we're looking to match the recipes contributing to the
2903 // uncountable exit condition comparison (here, vp<%4>) back to either
2904 // live-ins or the address nodes for the load used as part of the uncountable
2905 // exit comparison so that we can either move them within the loop, or copy
2906 // them to the preheader depending on the chosen method for dealing with
2907 // stores in uncountable exit loops.
2908 //
2909 // Currently, the address of the load is restricted to a GEP with 2 operands
2910 // and a live-in base address. This constraint may be relaxed later.
2911 //
2912 // VPlan ' for UF>=1' {
2913 // Live-in vp<%0> = VF * UF
2914 // Live-in vp<%1> = vector-trip-count
2915 // Live-in ir<20> = original trip-count
2916 //
2917 // ir-bb<entry>:
2918 // Successor(s): scalar.ph, vector.ph
2919 //
2920 // vector.ph:
2921 // Successor(s): for.body
2922 //
2923 // for.body:
2924 // EMIT vp<%2> = phi ir<0>, vp<%index.next>
2925 // EMIT-SCALAR ir<%iv> = phi [ ir<0>, vector.ph ], [ ir<%iv.next>, for.inc ]
2926 // EMIT ir<%uncountable.addr> = getelementptr inbounds nuw ir<%pred>,ir<%iv>
2927 // EMIT ir<%uncountable.val> = load ir<%uncountable.addr>
2928 // EMIT ir<%uncountable.cond> = icmp sgt ir<%uncountable.val>, ir<500>
2929 // EMIT vp<%3> = masked-cond ir<%uncountable.cond>
2930 // Successor(s): for.inc
2931 //
2932 // for.inc:
2933 // EMIT ir<%iv.next> = add nuw nsw ir<%iv>, ir<1>
2934 // EMIT ir<%countable.cond> = icmp eq ir<%iv.next>, ir<20>
2935 // EMIT vp<%index.next> = add nuw vp<%2>, vp<%0>
2936 // EMIT vp<%4> = any-of ir<%3>
2937 // EMIT vp<%5> = icmp eq vp<%index.next>, vp<%1>
2938 // EMIT branch-on-two-conds vp<%4>, vp<%5>
2939 // Successor(s): middle.block, middle.block, for.body
2940 //
2941 // middle.block:
2942 // Successor(s): ir-bb<exit>, scalar.ph
2943 //
2944 // ir-bb<exit>:
2945 // No successors
2946 //
2947 // scalar.ph:
2948 // }
2949
2950 // Find the uncountable loop exit condition.
2951 VPValue *UncountableCondition = nullptr;
2952 if (!match(LatchVPBB->getTerminator(),
2953 m_BranchOnTwoConds(m_AnyOf(m_VPValue(UncountableCondition)),
2954 m_VPValue())))
2955 return std::nullopt;
2956
2958 Worklist.push_back(UncountableCondition);
2959 while (!Worklist.empty()) {
2960 VPValue *V = Worklist.pop_back_val();
2961
2962 // Any value defined outside the loop does not need to be copied.
2963 if (V->isDefinedOutsideLoopRegions())
2964 continue;
2965
2966 // FIXME: Remove the single user restriction; it's here because we're
2967 // starting with the simplest set of loops we can, and multiple
2968 // users means needing to add PHI nodes in the transform.
2969 if (V->getNumUsers() > 1)
2970 return std::nullopt;
2971
2972 VPValue *Op1, *Op2;
2973 // Walk back through recipes until we find at least one load from memory.
2974 if (match(V, m_ICmp(m_VPValue(Op1), m_VPValue(Op2)))) {
2975 Worklist.push_back(Op1);
2976 Worklist.push_back(Op2);
2977 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
2978 } else if (match(V, m_VPInstruction<Instruction::Load>(m_VPValue(Op1)))) {
2979 VPRecipeBase *GepR = Op1->getDefiningRecipe();
2980 // Only matching base + single offset term for now.
2981 if (GepR->getNumOperands() != 2)
2982 return std::nullopt;
2983 // Matching a GEP with a loop-invariant base ptr.
2985 m_LiveIn(), m_VPValue())))
2986 return std::nullopt;
2987 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
2988 Recipes.push_back(cast<VPInstruction>(GepR));
2990 m_VPValue(Op1)))) {
2991 Worklist.push_back(Op1);
2992 Recipes.push_back(cast<VPInstruction>(V->getDefiningRecipe()));
2993 } else
2994 return std::nullopt;
2995 }
2996
2997 // If we couldn't match anything, don't return the condition. It may be
2998 // defined outside the loop.
2999 if (Recipes.empty() ||
3001 return std::nullopt;
3002
3003 return UncountableCondition;
3004}
3005
3011
3012/// Update \p Plan to mask memory operations in the loop based on whether the
3013/// early exit is taken or not.
3014///
3015/// We're currently expecting to find a loop with properties similar to the
3016/// following:
3017///
3018/// for.body:
3019/// ir<%indvars.iv> = WIDEN-INDUCTION nuw nsw ir<0>, ir<1>, vp<%0>
3020/// EMIT ir<%arrayidx> = getelementptr inbounds nuw ir<@c>, ir<%indvars.iv>
3021/// EMIT-SCALAR ir<%0> = load ir<%arrayidx>
3022/// EMIT ir<%cmp1> = icmp sgt ir<%0>, ir<5>
3023/// EMIT vp<%1> = masked-cond ir<%cmp1>
3024/// Successor(s): if.end
3025///
3026/// if.end:
3027/// EMIT ir<%arrayidx3> = getelementptr inbounds nuw ir<@src>, ir<%indvars.iv>
3028/// EMIT-SCALAR ir<%2> = load ir<%arrayidx3>
3029/// EMIT ir<%add> = add nsw ir<%2>, ir<42>
3030/// EMIT ir<%arrayidx5> = getelementptr inbounds nuw ir<@dst>, ir<%indvars.iv>
3031/// EMIT store ir<%add>, ir<%arrayidx5>
3032/// EMIT ir<%indvars.iv.next> = add nuw nsw ir<%indvars.iv>, ir<1>
3033/// EMIT vp<%3> = any-of ir<%1>
3034/// EMIT ir<%exitcond.not> = icmp eq ir<%indvars.iv.next>, ir<10000>
3035/// EMIT branch-on-two-conds vp<%3>, ir<%exitcond.not>
3036/// Successor(s): middle.block, middle.block, for.body
3037///
3038/// We currently expect LoopVectorizationLegality to ensure that:
3039/// * There must also be a counted exit. We will need to support speculative
3040/// or first-faulting loads before we can remove this restriction.
3041/// * Any stores within the loop must not alias with the load used for the
3042/// uncountable exit. We can relax this a bit with runtime aliasing checks.
3043/// * Other memory operations in the loop can take place before or after the
3044/// uncountable exit, but must also be unconditional. We need to support
3045/// combining the conditions in VPlanPredicator.
3046/// * The loop must have a single unconditional load contributing to the
3047/// uncountable exit comparison, and the other term must be loop-invariant.
3048/// Improving upon this requires work in getRecipesForUncountableExit to
3049/// handle more complex recipe graphs.
3052 VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB,
3053 Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT,
3054 AssumptionCache *AC) {
3055
3056 // Disconnect early exiting blocks from successors, remove branches. We
3057 // currently don't support multiple uses for recipes involved in creating
3058 // the uncountable exit condition.
3059 for (auto &Exit : Exits) {
3060 if (Exit.EarlyExitingVPBB == LatchVPBB)
3061 continue;
3062
3063 for (VPRecipeBase &R : Exit.EarlyExitVPBB->phis())
3064 cast<VPIRPhi>(&R)->removeIncomingValueFor(Exit.EarlyExitingVPBB);
3065 Exit.EarlyExitingVPBB->getTerminator()->eraseFromParent();
3066 VPBlockUtils::disconnectBlocks(Exit.EarlyExitingVPBB, Exit.EarlyExitVPBB);
3067 }
3068
3069 VPDominatorTree VPDT(Plan);
3070
3071 // We can abandon a VPlan entirely if we return false here, so we shouldn't
3072 // crash if some earlier assumptions on scalar IR don't hold for the vplan
3073 // version of the loop.
3074 SmallVector<VPInstruction *, 8> ConditionRecipes;
3075
3076 std::optional<VPValue *> Cond =
3077 getRecipesForUncountableExit(ConditionRecipes, LatchVPBB);
3078 if (!Cond)
3079 return false;
3080
3081 // Find load contributing to condition.
3082 // At the moment LoopVectorizationLegality only supports a single
3083 // early-exit expression with a compare and a single load that must
3084 // be unconditional.
3085 // TODO: Support more than one load.
3086 auto *Load =
3087 find_singleton<VPInstruction>(ConditionRecipes, [](auto *I, bool _) {
3089 ? I
3090 : nullptr;
3091 });
3092 assert(Load && "Couldn't find exactly one load");
3093 // TODO: Support conditional loads for uncountable exits.
3094 assert(VPDT.dominates(Load->getParent(), LatchVPBB) &&
3095 "Uncountable exit condition load is conditional.");
3096 VPInstruction *Ptr = cast<VPInstruction>(Load->getOperand(0));
3097
3098 // Ensure that we are guaranteed to be able to dereference the memory used
3099 // for determining the uncountable exit for the maximum possible number of
3100 // scalar iterations of the loop.
3101 //
3102 // TODO: Support first-faulting loads in cases where we don't know whether
3103 // all possible addresses are dereferenceable.
3104 {
3106 const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, TheLoop);
3107 const DataLayout &DL = Plan.getDataLayout();
3108 APInt EltSize(DL.getIndexTypeSizeInBits(Ptr->getScalarType()),
3109 DL.getTypeStoreSize(Load->getScalarType()).getFixedValue());
3111 PtrSCEV, cast<LoadInst>(Load->getUnderlyingInstr())->getAlign(),
3112 PSE.getSE()->getConstant(EltSize), TheLoop, *PSE.getSE(), DT, AC,
3113 &Predicates))
3114 return false;
3115 }
3116
3117 // Check for a single GEP for the condition load to see if we can link it to
3118 // a widen IV recipe with a step of 1; we're only interested in contiguous
3119 // accesses for the condition load right now.
3120 auto *IV = cast<VPWidenInductionRecipe>(&HeaderVPBB->front());
3121 if (!match(IV->getStartValue(), m_SpecificInt(0)) ||
3122 !match(IV->getStepValue(), m_SpecificInt(1)))
3123 return false;
3125 m_Specific(IV))))
3126 return false;
3127
3128 // We want to guarantee that the uncountable exit condition (and the mask
3129 // we will generate from it) are available for all operations in the loop
3130 // that need to be masked. If the condition recipes are not already the first
3131 // recipes in the header after the last phi, move them there.
3132 auto InsertIt = HeaderVPBB->getFirstNonPhi();
3133 while (InsertIt != HeaderVPBB->end() &&
3134 is_contained(ConditionRecipes, &*InsertIt)) {
3135 erase(ConditionRecipes, &*InsertIt);
3136 InsertIt++;
3137 }
3138 for (auto *Recipe : reverse(ConditionRecipes))
3139 Recipe->moveBefore(*HeaderVPBB, InsertIt);
3140
3141 // Create a mask to represent all lanes that fully execute in the vector loop,
3142 // stopping short of any early exit.
3143 VPBuilder MaskBuilder(HeaderVPBB, InsertIt);
3144 VPValue *FirstActive = MaskBuilder.createFirstActiveLane(*Cond);
3145 Type *IVScalarTy = IV->getScalarType();
3146 VPValue *Zero = Plan.getZero(IVScalarTy);
3147 FirstActive =
3148 MaskBuilder.createScalarZExtOrTrunc(FirstActive, IVScalarTy, DebugLoc());
3150 {Zero, FirstActive}, DebugLoc(),
3151 "uncountable.exit.mask");
3152
3153 // Convert all other memory operations to use the mask.
3154 for (VPBasicBlock *VPBB : vp_rpo_plain_cfg_loop_body(HeaderVPBB))
3155 for (VPRecipeBase &R : *VPBB)
3156 if (R.mayReadOrWriteMemory() && &R != Load) {
3157 // TODO: Handle conditional memory operations in the loop.
3158 if (!VPDT.dominates(R.getParent(), LatchVPBB))
3159 return false;
3160 cast<VPInstruction>(&R)->addMask(Mask);
3161 }
3162
3163 // Update middle block branch to compare (IV + however many lanes were active)
3164 // against the full trip count, since we may be exiting the vector loop early.
3165 // If we didn't take an early exit, we should get the equivalent of VF from
3166 // the FirstActiveLane.
3167 assert(match(MiddleVPBB->getTerminator(), m_BranchOnCond()) &&
3168 "Expected BranchOnCond terminator for MiddleVPBB");
3169 VPBuilder MiddleBuilder(MiddleVPBB->getTerminator());
3170 VPValue *ScalarIV = MiddleBuilder.createNaryOp(VPInstruction::ExtractLane,
3171 {Zero, IV}, DebugLoc());
3172 VPValue *ExitIV = MiddleBuilder.createAdd(ScalarIV, FirstActive);
3173 VPValue *FullTC =
3174 MiddleBuilder.createICmp(CmpInst::ICMP_EQ, ExitIV, Plan.getTripCount());
3175 MiddleVPBB->getTerminator()->setOperand(0, FullTC);
3176
3177 // Update resume phi in scalar.ph.
3178 VPBasicBlock *ScalarPH = Plan.getScalarPreheader();
3179 auto Phis = ScalarPH->phis();
3180 // TODO: Handle more than one Phi; re-derive from IV.
3181 // TODO: Handle reductions.
3182 if (range_size(Phis) != 1)
3183 return false;
3184 VPPhi *ContinueIV = cast<VPPhi>(Phis.begin());
3185 // Make sure we're referring to the same IV.
3186 assert(
3187 match(ContinueIV->getOperand(0),
3189 "Continuing from different IV");
3190 ContinueIV->setOperand(0, ExitIV);
3191 return true;
3192}
3193
3195 VPlan &Plan, Loop *TheLoop, PredicatedScalarEvolution &PSE,
3197#ifndef NDEBUG
3198 VPDominatorTree VPDT(Plan);
3199#endif
3200
3201 auto *MiddleVPBB = VPBlockUtils::getPlainCFGMiddleBlock(Plan);
3202 auto [HeaderVPBB, LatchVPBB] = VPBlockUtils::getPlainCFGHeaderAndLatch(Plan);
3203
3204 // Dereferenceability is checked separately for uncountable exit loops with
3205 // stores, as only the loads contributing to the exit condition need to
3206 // be checked.
3207 if (Style == UncountableExitStyle::ReadOnly &&
3208 !areAllLoadsDereferenceable(HeaderVPBB, TheLoop, PSE, DT, AC))
3209 return false;
3210
3211 VPBuilder LatchBuilder(LatchVPBB->getTerminator());
3213 for (auto [EarlyExitingVPBB, ExitBlock] :
3214 vputils::getEarlyExits(Plan, MiddleVPBB)) {
3215 // Collect condition for this early exit.
3216 VPBlockBase *TrueSucc = EarlyExitingVPBB->getSuccessors()[0];
3217 VPValue *CondOfEarlyExitingVPBB;
3218 [[maybe_unused]] bool Matched =
3219 match(EarlyExitingVPBB->getTerminator(),
3220 m_BranchOnCond(m_VPValue(CondOfEarlyExitingVPBB)));
3221 assert(Matched && "Terminator must be BranchOnCond");
3222
3223 // Insert the MaskedCond in the EarlyExitingVPBB so the predicator adds
3224 // the correct block mask.
3225 VPBuilder EarlyExitingBuilder(EarlyExitingVPBB->getTerminator());
3226 auto *CondToEarlyExit = EarlyExitingBuilder.createNaryOp(
3228 TrueSucc == ExitBlock
3229 ? CondOfEarlyExitingVPBB
3230 : EarlyExitingBuilder.createNot(CondOfEarlyExitingVPBB));
3231 assert((isa<VPIRValue>(CondOfEarlyExitingVPBB) ||
3232 !VPDT.properlyDominates(EarlyExitingVPBB, LatchVPBB) ||
3233 VPDT.properlyDominates(
3234 CondOfEarlyExitingVPBB->getDefiningRecipe()->getParent(),
3235 LatchVPBB)) &&
3236 "exit condition must dominate the latch");
3237 Exits.push_back({
3238 EarlyExitingVPBB,
3239 ExitBlock,
3240 CondToEarlyExit,
3241 });
3242 }
3243
3244 assert(!Exits.empty() && "must have at least one early exit");
3245 // Sort exits by RPO order to get correct program order. RPO gives a
3246 // topological ordering of the CFG, ensuring upstream exits are checked
3247 // before downstream exits in the dispatch chain.
3249 HeaderVPBB);
3251 for (const auto &[Num, VPB] : enumerate(RPOT))
3252 RPOIdx[VPB] = Num;
3253 llvm::sort(Exits, [&RPOIdx](const EarlyExitInfo &A, const EarlyExitInfo &B) {
3254 return RPOIdx[A.EarlyExitingVPBB] < RPOIdx[B.EarlyExitingVPBB];
3255 });
3256#ifndef NDEBUG
3257 // After RPO sorting, verify that for any pair where one exit dominates
3258 // another, the dominating exit comes first. This is guaranteed by RPO
3259 // (topological order) and is required for the dispatch chain correctness.
3260 for (unsigned I = 0; I + 1 < Exits.size(); ++I)
3261 for (unsigned J = I + 1; J < Exits.size(); ++J)
3262 assert(!VPDT.properlyDominates(Exits[J].EarlyExitingVPBB,
3263 Exits[I].EarlyExitingVPBB) &&
3264 "RPO sort must place dominating exits before dominated ones");
3265#endif
3266
3267 // Build the AnyOf condition for the latch terminator using logical OR
3268 // to avoid poison propagation from later exit conditions when an earlier
3269 // exit is taken.
3270 VPValue *Combined = Exits[0].CondToExit;
3271 for (const EarlyExitInfo &Info : drop_begin(Exits))
3272 Combined = LatchBuilder.createLogicalOr(Combined, Info.CondToExit);
3273
3274 VPValue *IsAnyExitTaken =
3275 LatchBuilder.createNaryOp(VPInstruction::AnyOf, {Combined});
3276
3277 // Create a comparison for the latch exit condition and replace the
3278 // BranchOnCond with a BranchOnTwoConds. The original BranchOnCond's condition
3279 // is used as the latch-exit condition; canonical IV recipes have not been
3280 // introduced yet, so there is no BranchOnCount to derive the condition from.
3281 auto *LatchExitingBranch = cast<VPInstruction>(LatchVPBB->getTerminator());
3282 assert(LatchExitingBranch->getOpcode() == VPInstruction::BranchOnCond &&
3283 "Unexpected terminator");
3284 VPValue *IsLatchExitTaken = LatchExitingBranch->getOperand(0);
3285 DebugLoc LatchDL = LatchExitingBranch->getDebugLoc();
3286 LatchExitingBranch->eraseFromParent();
3287 LatchBuilder.setInsertPoint(LatchVPBB);
3289 {IsAnyExitTaken, IsLatchExitTaken}, LatchDL);
3290 LatchVPBB->clearSuccessors();
3291
3293 // If handling the exiting lane in the scalar loop, combine the exit
3294 // conditions into a single BranchOnCond.
3295 LatchVPBB->setSuccessors({MiddleVPBB, MiddleVPBB, HeaderVPBB});
3296 MiddleVPBB->clearPredecessors();
3297 MiddleVPBB->setPredecessors({LatchVPBB, LatchVPBB});
3299 Plan, Exits, HeaderVPBB, LatchVPBB, MiddleVPBB, TheLoop, PSE, DT, AC);
3300 }
3301
3302 // Create the vector.early.exit blocks.
3303 SmallVector<VPBasicBlock *> VectorEarlyExitVPBBs(Exits.size());
3304 for (unsigned Idx = 0; Idx != Exits.size(); ++Idx) {
3305 Twine BlockSuffix = Exits.size() == 1 ? "" : Twine(".") + Twine(Idx);
3306 VPBasicBlock *VectorEarlyExitVPBB =
3307 Plan.createVPBasicBlock("vector.early.exit" + BlockSuffix);
3308 VectorEarlyExitVPBBs[Idx] = VectorEarlyExitVPBB;
3309 }
3310
3311 // Create the dispatch block (or reuse the single exit block if only one
3312 // exit). The dispatch block computes the first active lane of the combined
3313 // condition and, for multiple exits, chains through conditions to determine
3314 // which exit to take.
3315 VPBasicBlock *DispatchVPBB =
3316 Exits.size() == 1 ? VectorEarlyExitVPBBs[0]
3317 : Plan.createVPBasicBlock("vector.early.exit.check");
3318 DispatchVPBB->setPredecessors({LatchVPBB});
3319 LatchVPBB->setSuccessors({DispatchVPBB, MiddleVPBB, HeaderVPBB});
3320 VPBuilder DispatchBuilder(DispatchVPBB, DispatchVPBB->begin());
3321 VPValue *FirstActiveLane = DispatchBuilder.createFirstActiveLane(
3322 {Combined}, DebugLoc::getUnknown(), "first.active.lane");
3323
3324 // For each early exit, disconnect the original exiting block
3325 // (early.exiting.I) from the exit block (ir-bb<exit.I>) and route through a
3326 // new vector.early.exit block. Update ir-bb<exit.I>'s phis to extract their
3327 // values at the first active lane:
3328 //
3329 // Input:
3330 // early.exiting.I:
3331 // ...
3332 // EMIT branch-on-cond vp<%cond.I>
3333 // Successor(s): in.loop.succ, ir-bb<exit.I>
3334 //
3335 // ir-bb<exit.I>:
3336 // IR %phi = phi [ vp<%incoming.I>, early.exiting.I ], ...
3337 //
3338 // Output:
3339 // early.exiting.I:
3340 // ...
3341 // Successor(s): in.loop.succ
3342 //
3343 // vector.early.exit.I:
3344 // EMIT vp<%exit.val> = extract-lane vp<%first.lane>, vp<%incoming.I>
3345 // Successor(s): ir-bb<exit.I>
3346 //
3347 // ir-bb<exit.I>:
3348 // IR %phi = phi ... (extra operand: vp<%exit.val> from
3349 // vector.early.exit.I)
3350 //
3351 for (auto [Exit, VectorEarlyExitVPBB] :
3352 zip_equal(Exits, VectorEarlyExitVPBBs)) {
3353 auto &[EarlyExitingVPBB, EarlyExitVPBB, _] = Exit;
3354 // Adjust the phi nodes in EarlyExitVPBB.
3355 // 1. remove incoming values from EarlyExitingVPBB,
3356 // 2. extract the incoming value at FirstActiveLane
3357 // 3. add back the extracts as last operands for the phis
3358 // Then adjust the CFG, removing the edge between EarlyExitingVPBB and
3359 // EarlyExitVPBB and adding a new edge between VectorEarlyExitVPBB and
3360 // EarlyExitVPBB. The extracts at FirstActiveLane are now the incoming
3361 // values from VectorEarlyExitVPBB.
3362 for (VPRecipeBase &R : EarlyExitVPBB->phis()) {
3363 auto *ExitIRI = cast<VPIRPhi>(&R);
3364 VPValue *IncomingVal =
3365 ExitIRI->getIncomingValueForBlock(EarlyExitingVPBB);
3366 VPValue *NewIncoming = IncomingVal;
3367 if (!isa<VPIRValue>(IncomingVal)) {
3368 VPBuilder EarlyExitBuilder(VectorEarlyExitVPBB);
3369 NewIncoming = EarlyExitBuilder.createNaryOp(
3370 VPInstruction::ExtractLane, {FirstActiveLane, IncomingVal},
3371 DebugLoc::getUnknown(), "early.exit.value");
3372 }
3373 ExitIRI->removeIncomingValueFor(EarlyExitingVPBB);
3374 ExitIRI->addIncoming(NewIncoming);
3375 }
3376
3377 EarlyExitingVPBB->getTerminator()->eraseFromParent();
3378 VPBlockUtils::disconnectBlocks(EarlyExitingVPBB, EarlyExitVPBB);
3379 VPBlockUtils::connectBlocks(VectorEarlyExitVPBB, EarlyExitVPBB);
3380 }
3381
3382 // Chain through exits: for each exit, check if its condition is true at
3383 // the first active lane. If so, take that exit; otherwise, try the next.
3384 // The last exit needs no check since it must be taken if all others fail.
3385 //
3386 // For 3 exits (cond.0, cond.1, cond.2), this creates:
3387 //
3388 // latch:
3389 // ...
3390 // EMIT vp<%combined> = logical-or vp<%cond.0>, vp<%cond.1>, vp<%cond.2>
3391 // ...
3392 //
3393 // vector.early.exit.check:
3394 // EMIT vp<%first.lane> = first-active-lane vp<%combined>
3395 // EMIT vp<%at.cond.0> = extract-lane vp<%first.lane>, vp<%cond.0>
3396 // EMIT branch-on-cond vp<%at.cond.0>
3397 // Successor(s): vector.early.exit.0, vector.early.exit.check.0
3398 //
3399 // vector.early.exit.check.0:
3400 // EMIT vp<%at.cond.1> = extract-lane vp<%first.lane>, vp<%cond.1>
3401 // EMIT branch-on-cond vp<%at.cond.1>
3402 // Successor(s): vector.early.exit.1, vector.early.exit.2
3403 VPBasicBlock *CurrentBB = DispatchVPBB;
3404 for (auto [I, Exit] : enumerate(ArrayRef(Exits).drop_back())) {
3405 VPValue *LaneVal = DispatchBuilder.createNaryOp(
3406 VPInstruction::ExtractLane, {FirstActiveLane, Exit.CondToExit},
3407 DebugLoc::getUnknown(), "exit.cond.at.lane");
3408
3409 // For the last dispatch, branch directly to the last exit on false;
3410 // otherwise, create a new check block.
3411 bool IsLastDispatch = (I + 2 == Exits.size());
3412 VPBasicBlock *FalseBB =
3413 IsLastDispatch ? VectorEarlyExitVPBBs.back()
3414 : Plan.createVPBasicBlock(
3415 Twine("vector.early.exit.check.") + Twine(I));
3416
3417 DispatchBuilder.createNaryOp(VPInstruction::BranchOnCond, {LaneVal});
3418 CurrentBB->setSuccessors({VectorEarlyExitVPBBs[I], FalseBB});
3419 VectorEarlyExitVPBBs[I]->setPredecessors({CurrentBB});
3420 FalseBB->setPredecessors({CurrentBB});
3421
3422 CurrentBB = FalseBB;
3423 DispatchBuilder.setInsertPoint(CurrentBB);
3424 }
3425
3426 return true;
3427}
3428
3429/// This function tries convert extended in-loop reductions to
3430/// VPExpressionRecipe and clamp the \p Range if it is beneficial and
3431/// valid. The created recipe must be decomposed to its constituent
3432/// recipes before execution.
3433static VPExpressionRecipe *
3435 VFRange &Range) {
3436 Type *RedTy = Red->getScalarType();
3437 VPValue *VecOp = Red->getVecOp();
3438
3439 assert(!Red->isPartialReduction() &&
3440 "This path does not support partial reductions");
3441
3442 // Clamp the range if using extended-reduction is profitable.
3443 auto IsExtendedRedValidAndClampRange =
3444 [&](unsigned Opcode, Instruction::CastOps ExtOpc, Type *SrcTy) -> bool {
3446 [&](ElementCount VF) {
3447 auto *SrcVecTy = cast<VectorType>(toVectorTy(SrcTy, VF));
3449
3451 InstructionCost ExtCost =
3452 cast<VPWidenCastRecipe>(VecOp)->computeCost(VF, Ctx);
3453 InstructionCost RedCost = Red->computeCost(VF, Ctx);
3454
3455 assert(!RedTy->isFloatingPointTy() &&
3456 "getExtendedReductionCost only supports integer types");
3457 ExtRedCost = Ctx.TTI.getExtendedReductionCost(
3458 Opcode, ExtOpc == Instruction::CastOps::ZExt, RedTy, SrcVecTy,
3459 Red->getFastMathFlagsOrNone(), CostKind);
3460 return ExtRedCost.isValid() && ExtRedCost < ExtCost + RedCost;
3461 },
3462 Range);
3463 };
3464
3465 VPValue *A;
3466 // Match reduce(ext)).
3468 IsExtendedRedValidAndClampRange(
3469 RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind()),
3470 cast<VPWidenCastRecipe>(VecOp)->getOpcode(), A->getScalarType()))
3471 return new VPExpressionRecipe(cast<VPWidenCastRecipe>(VecOp), Red);
3472
3473 return nullptr;
3474}
3475
3476/// This function tries convert extended in-loop reductions to
3477/// VPExpressionRecipe and clamp the \p Range if it is beneficial
3478/// and valid. The created VPExpressionRecipe must be decomposed to its
3479/// constituent recipes before execution. Patterns of the
3480/// VPExpressionRecipe:
3481/// reduce.add(mul(...)),
3482/// reduce.add(mul(ext(A), ext(B))),
3483/// reduce.add(ext(mul(ext(A), ext(B)))).
3484/// reduce.fadd(fmul(ext(A), ext(B)))
3485static VPExpressionRecipe *
3487 VPCostContext &Ctx, VFRange &Range) {
3488 unsigned Opcode = RecurrenceDescriptor::getOpcode(Red->getRecurrenceKind());
3489 if (Opcode != Instruction::Add && Opcode != Instruction::Sub &&
3490 Opcode != Instruction::FAdd)
3491 return nullptr;
3492
3493 assert(!Red->isPartialReduction() &&
3494 "This path does not support partial reductions");
3495 Type *RedTy = Red->getScalarType();
3496
3497 // Clamp the range if using multiply-accumulate-reduction is profitable.
3498 auto IsMulAccValidAndClampRange =
3500 VPWidenCastRecipe *OuterExt) -> bool {
3502 [&](ElementCount VF) {
3504 Type *SrcTy = Ext0 ? Ext0->getOperand(0)->getScalarType() : RedTy;
3505 InstructionCost MulAccCost;
3506
3507 // getMulAccReductionCost for in-loop reductions does not support
3508 // mixed or floating-point extends.
3509 if (Ext0 && Ext1 &&
3510 (Ext0->getOpcode() != Ext1->getOpcode() ||
3511 Ext0->getOpcode() == Instruction::CastOps::FPExt))
3512 return false;
3513
3514 bool IsZExt =
3515 !Ext0 || Ext0->getOpcode() == Instruction::CastOps::ZExt;
3516 auto *SrcVecTy = cast<VectorType>(toVectorTy(SrcTy, VF));
3517 MulAccCost = Ctx.TTI.getMulAccReductionCost(IsZExt, Opcode, RedTy,
3518 SrcVecTy, CostKind);
3519
3520 InstructionCost MulCost = Mul->computeCost(VF, Ctx);
3521 InstructionCost RedCost = Red->computeCost(VF, Ctx);
3522 InstructionCost ExtCost = 0;
3523 if (Ext0)
3524 ExtCost += Ext0->computeCost(VF, Ctx);
3525 if (Ext1)
3526 ExtCost += Ext1->computeCost(VF, Ctx);
3527 if (OuterExt)
3528 ExtCost += OuterExt->computeCost(VF, Ctx);
3529
3530 return MulAccCost.isValid() &&
3531 MulAccCost < ExtCost + MulCost + RedCost;
3532 },
3533 Range);
3534 };
3535
3536 VPValue *VecOp = Red->getVecOp();
3537 VPRecipeBase *Sub = nullptr;
3538 VPValue *A, *B;
3539 VPValue *Tmp = nullptr;
3540
3541 if (RedTy->isFloatingPointTy())
3542 return nullptr;
3543
3544 // Sub reductions could have a sub between the add reduction and vec op.
3545 if (match(VecOp, m_Sub(m_ZeroInt(), m_VPValue(Tmp)))) {
3546 Sub = VecOp->getDefiningRecipe();
3547 VecOp = Tmp;
3548 }
3549
3550 // If ValB is a constant and can be safely extended, truncate it to the same
3551 // type as ExtA's operand, then extend it to the same type as ExtA. This
3552 // creates two uniform extends that can more easily be matched by the rest of
3553 // the bundling code. The ExtB reference, ValB and operand 1 of Mul are all
3554 // replaced with the new extend of the constant.
3555 auto ExtendAndReplaceConstantOp = [](VPWidenCastRecipe *ExtA,
3556 VPWidenCastRecipe *&ExtB, VPValue *&ValB,
3557 VPWidenRecipe *Mul) {
3558 if (!ExtA || ExtB || !isa<VPIRValue>(ValB))
3559 return;
3560 Type *NarrowTy = ExtA->getOperand(0)->getScalarType();
3561 Instruction::CastOps ExtOpc = ExtA->getOpcode();
3562 const APInt *Const;
3563 if (!match(ValB, m_APInt(Const)) ||
3565 Const, NarrowTy, TTI::getPartialReductionExtendKind(ExtOpc)))
3566 return;
3567 // The truncate ensures that the type of each extended operand is the
3568 // same, and it's been proven that the constant can be extended from
3569 // NarrowTy safely. Necessary since ExtA's extended operand would be
3570 // e.g. an i8, while the const will likely be an i32. This will be
3571 // elided by later optimisations.
3572 VPBuilder Builder(Mul);
3573 auto *Trunc =
3574 Builder.createWidenCast(Instruction::CastOps::Trunc, ValB, NarrowTy);
3575 Type *WideTy = ExtA->getScalarType();
3576 ValB = ExtB = Builder.createWidenCast(ExtOpc, Trunc, WideTy);
3577 Mul->setOperand(1, ExtB);
3578 };
3579
3580 // Try to match reduce.add(mul(...)).
3581 if (match(VecOp, m_Mul(m_VPValue(A), m_VPValue(B)))) {
3582 auto *RecipeA = dyn_cast<VPWidenCastRecipe>(A);
3583 auto *RecipeB = dyn_cast<VPWidenCastRecipe>(B);
3584 auto *Mul = cast<VPWidenRecipe>(VecOp);
3585
3586 // Convert reduce.add(mul(ext, const)) to reduce.add(mul(ext, ext(const)))
3587 ExtendAndReplaceConstantOp(RecipeA, RecipeB, B, Mul);
3588
3589 // Match reduce.add/sub(mul(ext, ext)).
3590 if (RecipeA && RecipeB && match(RecipeA, m_ZExtOrSExt(m_VPValue())) &&
3591 match(RecipeB, m_ZExtOrSExt(m_VPValue())) &&
3592 IsMulAccValidAndClampRange(Mul, RecipeA, RecipeB, nullptr)) {
3593 if (Sub)
3594 return new VPExpressionRecipe(RecipeA, RecipeB, Mul,
3595 cast<VPWidenRecipe>(Sub), Red);
3596 return new VPExpressionRecipe(RecipeA, RecipeB, Mul, Red);
3597 }
3598 // TODO: Add an expression type for this variant with a negated mul
3599 if (!Sub && IsMulAccValidAndClampRange(Mul, nullptr, nullptr, nullptr))
3600 return new VPExpressionRecipe(Mul, Red);
3601 }
3602 // TODO: Add an expression type for negated versions of other expression
3603 // variants.
3604 if (Sub)
3605 return nullptr;
3606
3607 // Match reduce.add(ext(mul(A, B))).
3608 if (match(VecOp, m_ZExtOrSExt(m_Mul(m_VPValue(A), m_VPValue(B))))) {
3609 auto *Ext = cast<VPWidenCastRecipe>(VecOp);
3610 auto *Mul = cast<VPWidenRecipe>(Ext->getOperand(0));
3611 auto *Ext0 = dyn_cast<VPWidenCastRecipe>(A);
3612 auto *Ext1 = dyn_cast<VPWidenCastRecipe>(B);
3613
3614 // reduce.add(ext(mul(ext, const)))
3615 // -> reduce.add(ext(mul(ext, ext(const))))
3616 ExtendAndReplaceConstantOp(Ext0, Ext1, B, Mul);
3617
3618 // reduce.add(ext(mul(ext(A), ext(B))))
3619 // -> reduce.add(mul(wider_ext(A), wider_ext(B)))
3620 // The inner extends must either have the same opcode as the outer extend or
3621 // be the same, in which case the multiply can never result in a negative
3622 // value and the outer extend can be folded away by doing wider
3623 // extends for the operands of the mul.
3624 if (Ext0 && Ext1 &&
3625 (Ext->getOpcode() == Ext0->getOpcode() || Ext0 == Ext1) &&
3626 Ext0->getOpcode() == Ext1->getOpcode() &&
3627 IsMulAccValidAndClampRange(Mul, Ext0, Ext1, Ext) && Mul->hasOneUse()) {
3628 auto *NewExt0 = new VPWidenCastRecipe(
3629 Ext0->getOpcode(), Ext0->getOperand(0), Ext->getScalarType(), nullptr,
3630 *Ext0, *Ext0, Ext0->getDebugLoc());
3631 NewExt0->insertBefore(Ext0);
3632
3633 VPWidenCastRecipe *NewExt1 = NewExt0;
3634 if (Ext0 != Ext1) {
3635 NewExt1 = new VPWidenCastRecipe(Ext1->getOpcode(), Ext1->getOperand(0),
3636 Ext->getScalarType(), nullptr, *Ext1,
3637 *Ext1, Ext1->getDebugLoc());
3638 NewExt1->insertBefore(Ext1);
3639 }
3640 auto *NewMul = Mul->cloneWithOperands({NewExt0, NewExt1});
3641 NewMul->insertBefore(Mul);
3642 Ext->replaceAllUsesWith(NewMul);
3643 Ext->eraseFromParent();
3644 Mul->eraseFromParent();
3645 return new VPExpressionRecipe(NewExt0, NewExt1, NewMul, Red);
3646 }
3647 }
3648 return nullptr;
3649}
3650
3651/// This function tries to create abstract recipes from the reduction recipe for
3652/// following optimizations and cost estimation.
3654 VPCostContext &Ctx,
3655 VFRange &Range) {
3656 // Creation of VPExpressions for partial reductions is entirely handled in
3657 // transformToPartialReduction.
3658 assert(!Red->isPartialReduction() &&
3659 "This path does not support partial reductions");
3660
3661 VPExpressionRecipe *AbstractR = nullptr;
3662 auto IP = std::next(Red->getIterator());
3663 auto *VPBB = Red->getParent();
3664 if (auto *MulAcc = tryToMatchAndCreateMulAccumulateReduction(Red, Ctx, Range))
3665 AbstractR = MulAcc;
3666 else if (auto *ExtRed = tryToMatchAndCreateExtendedReduction(Red, Ctx, Range))
3667 AbstractR = ExtRed;
3668 // Cannot create abstract inloop reduction recipes.
3669 if (!AbstractR)
3670 return;
3671
3672 AbstractR->insertBefore(*VPBB, IP);
3673 Red->replaceAllUsesWith(AbstractR);
3674}
3675
3686
3687// Collect common metadata from a group of replicate recipes by intersecting
3688// metadata from all recipes in the group.
3690 VPIRMetadata CommonMetadata = *Recipes.front();
3691 for (VPReplicateRecipe *Recipe : drop_begin(Recipes))
3692 CommonMetadata.intersect(*Recipe);
3693 return CommonMetadata;
3694}
3695
3696template <unsigned Opcode>
3700 const Loop *L) {
3701 static_assert(Opcode == Instruction::Load || Opcode == Instruction::Store,
3702 "Only Load and Store opcodes supported");
3703 [[maybe_unused]] constexpr bool IsLoad = (Opcode == Instruction::Load);
3704
3705 // For each address, collect operations with the same or complementary masks.
3708 Plan, PSE, L,
3709 [](VPReplicateRecipe *RepR) { return RepR->isPredicated(); });
3710 for (auto Recipes : Groups) {
3711 if (Recipes.size() < 2)
3712 continue;
3713
3715 map_range(Recipes, bind_back<getLoadStoreValueType>(IsLoad))) &&
3716 "Expected all recipes in group to have the same load-store type");
3717
3718 // Collect groups with the same or complementary masks.
3719 for (VPReplicateRecipe *&RecipeI : Recipes) {
3720 if (!RecipeI)
3721 continue;
3722
3723 VPValue *MaskI = RecipeI->getMask();
3725 Group.push_back(RecipeI);
3726 RecipeI = nullptr;
3727
3728 // Find all operations with the same or complementary masks.
3729 bool HasComplementaryMask = false;
3730 for (VPReplicateRecipe *&RecipeJ : Recipes) {
3731 if (!RecipeJ)
3732 continue;
3733
3734 VPValue *MaskJ = RecipeJ->getMask();
3735 // Check if any operation in the group has a complementary mask with
3736 // another, that is M1 == NOT(M2) or M2 == NOT(M1).
3737 HasComplementaryMask |= match(MaskI, m_Not(m_Specific(MaskJ))) ||
3738 match(MaskJ, m_Not(m_Specific(MaskI)));
3739 Group.push_back(RecipeJ);
3740 RecipeJ = nullptr;
3741 }
3742
3743 if (HasComplementaryMask) {
3744 assert(Group.size() >= 2 && "must have at least 2 entries");
3745 AllGroups.push_back(std::move(Group));
3746 }
3747 }
3748 }
3749
3750 return AllGroups;
3751}
3752
3753// Find the recipe with minimum alignment in the group.
3754template <typename InstType>
3755static VPReplicateRecipe *
3757 return *min_element(Group, [](VPReplicateRecipe *A, VPReplicateRecipe *B) {
3758 return cast<InstType>(A->getUnderlyingInstr())->getAlign() <
3759 cast<InstType>(B->getUnderlyingInstr())->getAlign();
3760 });
3761}
3762
3765 const Loop *L) {
3766 auto Groups =
3768 if (Groups.empty())
3769 return;
3770
3771 // Process each group of loads.
3772 for (auto &Group : Groups) {
3773 // Try to use the earliest (most dominating) load to replace all others.
3774 VPReplicateRecipe *EarliestLoad = Group[0];
3775 VPBasicBlock *FirstBB = EarliestLoad->getParent();
3776 VPBasicBlock *LastBB = Group.back()->getParent();
3777
3778 // Check that the load doesn't alias with stores between first and last.
3779 auto LoadLoc = vputils::getMemoryLocation(*EarliestLoad);
3780 if (!LoadLoc || !canHoistOrSinkWithNoAliasCheck(*LoadLoc, FirstBB, LastBB))
3781 continue;
3782
3783 // Collect common metadata from all loads in the group.
3784 VPIRMetadata CommonMetadata = getCommonMetadata(Group);
3785
3786 // Find the load with minimum alignment to use.
3787 auto *LoadWithMinAlign = findRecipeWithMinAlign<LoadInst>(Group);
3788
3789 bool IsSingleScalar = EarliestLoad->isSingleScalar();
3790 assert(all_of(Group,
3791 [IsSingleScalar](VPReplicateRecipe *R) {
3792 return R->isSingleScalar() == IsSingleScalar;
3793 }) &&
3794 "all members in group must agree on IsSingleScalar");
3795
3796 // Create an unpredicated version of the earliest load with common
3797 // metadata.
3798 auto *UnpredicatedLoad = new VPReplicateRecipe(
3799 LoadWithMinAlign->getUnderlyingInstr(), {EarliestLoad->getOperand(0)},
3800 IsSingleScalar, /*Mask=*/nullptr, *EarliestLoad, CommonMetadata);
3801
3802 UnpredicatedLoad->insertBefore(EarliestLoad);
3803
3804 // Replace all loads in the group with the unpredicated load.
3805 for (VPReplicateRecipe *Load : Group) {
3806 Load->replaceAllUsesWith(UnpredicatedLoad);
3807 Load->eraseFromParent();
3808 }
3809 }
3810}
3811
3812static bool
3814 PredicatedScalarEvolution &PSE, const Loop &L) {
3815 auto StoreLoc = vputils::getMemoryLocation(*StoresToSink.front());
3816 if (!StoreLoc || !StoreLoc->AATags.Scope)
3817 return false;
3818
3819 // When sinking a group of stores, all members of the group alias each other.
3820 // Skip them during the alias checks.
3821 VPBasicBlock *FirstBB = StoresToSink.front()->getParent();
3822 VPBasicBlock *LastBB = StoresToSink.back()->getParent();
3823 SinkStoreInfo SinkInfo(StoresToSink, *StoresToSink[0], PSE, L);
3824 return canHoistOrSinkWithNoAliasCheck(*StoreLoc, FirstBB, LastBB, SinkInfo);
3825}
3826
3829 const Loop *L) {
3830 auto Groups =
3832 if (Groups.empty())
3833 return;
3834
3835 for (auto &Group : Groups) {
3836 if (!canSinkStoreWithNoAliasCheck(Group, PSE, *L))
3837 continue;
3838
3839 // Use the last (most dominated) store's location for the unconditional
3840 // store.
3841 VPReplicateRecipe *LastStore = Group.back();
3842 VPBasicBlock *InsertBB = LastStore->getParent();
3843
3844 // Collect common alias metadata from all stores in the group.
3845 VPIRMetadata CommonMetadata = getCommonMetadata(Group);
3846
3847 // Build select chain for stored values.
3848 VPValue *SelectedValue = Group[0]->getOperand(0);
3849 VPBuilder Builder(InsertBB, LastStore->getIterator());
3850
3851 bool IsSingleScalar = Group[0]->isSingleScalar();
3852 for (unsigned I = 1; I < Group.size(); ++I) {
3853 assert(IsSingleScalar == Group[I]->isSingleScalar() &&
3854 "all members in group must agree on IsSingleScalar");
3855 VPValue *Mask = Group[I]->getMask();
3856 VPValue *Value = Group[I]->getOperand(0);
3857 SelectedValue = Builder.createSelect(
3858 Mask, Value, SelectedValue, Group[I]->getDebugLoc(), "",
3859 VPIRFlags::getDefaultFlags(Instruction::Select,
3860 Value->getScalarType()));
3861 }
3862
3863 // Find the store with minimum alignment to use.
3864 auto *StoreWithMinAlign = findRecipeWithMinAlign<StoreInst>(Group);
3865
3866 // Create unconditional store with selected value and common metadata.
3867 auto *UnpredicatedStore = new VPReplicateRecipe(
3868 StoreWithMinAlign->getUnderlyingInstr(),
3869 {SelectedValue, LastStore->getOperand(1)}, IsSingleScalar,
3870 /*Mask=*/nullptr, *LastStore, CommonMetadata);
3871 UnpredicatedStore->insertBefore(*InsertBB, LastStore->getIterator());
3872
3873 // Remove all predicated stores from the group.
3874 for (VPReplicateRecipe *Store : Group)
3875 Store->eraseFromParent();
3876 }
3877}
3878
3879/// Returns true if \p V is VPWidenLoadRecipe or VPInterleaveRecipe that can be
3880/// converted to a narrower recipe. \p V is used by a wide recipe that feeds a
3881/// store interleave group at index \p Idx, \p WideMember0 is the recipe feeding
3882/// the same interleave group at index 0. A VPWidenLoadRecipe can be narrowed to
3883/// an index-independent load if it feeds all wide ops at all indices (\p OpV
3884/// must be the operand at index \p OpIdx for both the recipe at lane 0, \p
3885/// WideMember0). A VPInterleaveRecipe can be narrowed to a wide load, if \p V
3886/// is defined at \p Idx of a load interleave group.
3887/// A live-in or recipe defined outside the loop region can be converted, if it
3888/// is the same across all lanes, or we can create a BuildVector for it.
3889static bool canNarrowLoad(VPSingleDefRecipe *WideMember0, unsigned OpIdx,
3890 VPValue *OpV, unsigned Idx, bool IsScalable) {
3891 VPValue *Member0Op = WideMember0->getOperand(OpIdx);
3892 if (Member0Op->isDefinedOutsideLoopRegions()) {
3893 // Operand matches Member0, broadcast across all fields for both live-ins
3894 // and recipes.
3895 if (Member0Op == OpV)
3896 return true;
3897 // Otherwise distinct per-field VPValues are assembled into a BuildVector.
3898 return !IsScalable && OpV->isDefinedOutsideLoopRegions() &&
3899 OpV->getScalarType() == Member0Op->getScalarType();
3900 }
3901 VPRecipeBase *Member0OpR = Member0Op->getDefiningRecipe();
3902 if (auto *W = dyn_cast<VPWidenLoadRecipe>(Member0OpR))
3903 // For scalable VFs, the narrowed plan processes vscale iterations at once,
3904 // so a shared wide load cannot be narrowed to a uniform scalar; bail out.
3905 return !IsScalable && !W->getMask() && W->isConsecutive() &&
3906 Member0Op == OpV;
3907 if (auto *IR = dyn_cast<VPInterleaveRecipe>(Member0OpR))
3908 return IR->getInterleaveGroup()->isFull() && IR->getVPValue(Idx) == OpV;
3909 return false;
3910}
3911
3912static bool canNarrowOps(ArrayRef<VPValue *> Ops, bool IsScalable) {
3914 auto *WideMember0 = dyn_cast<VPRecipeWithIRFlags>(Ops[0]);
3915 if (!WideMember0)
3916 return false;
3917 for (VPValue *V : Ops) {
3919 return false;
3920 auto *R = cast<VPRecipeWithIRFlags>(V);
3921 if (vputils::getOpcode(R) != vputils::getOpcode(WideMember0))
3922 return false;
3923 if (R->getScalarType() != WideMember0->getScalarType())
3924 return false;
3925 if (R->hasPredicate() && R->getPredicate() != WideMember0->getPredicate())
3926 return false;
3927 }
3928
3929 for (unsigned Idx = 0; Idx != WideMember0->getNumOperands(); ++Idx) {
3931 for (VPValue *Op : Ops)
3932 OpsI.push_back(Op->getDefiningRecipe()->getOperand(Idx));
3933
3934 if (canNarrowOps(OpsI, IsScalable))
3935 continue;
3936
3937 if (any_of(enumerate(OpsI), [WideMember0, Idx, IsScalable](const auto &P) {
3938 const auto &[OpIdx, OpV] = P;
3939 return !canNarrowLoad(WideMember0, Idx, OpV, OpIdx, IsScalable);
3940 }))
3941 return false;
3942 }
3943
3944 return true;
3945}
3946
3947/// Returns VF from \p VFs if \p IR is a full interleave group with factor and
3948/// number of members both equal to VF. The interleave group must also access
3949/// the full vector width.
3950static std::optional<ElementCount>
3953 const TargetTransformInfo &TTI) {
3954 if (!InterleaveR || InterleaveR->getMask())
3955 return std::nullopt;
3956
3957 Type *GroupElementTy = nullptr;
3958 if (InterleaveR->getStoredValues().empty()) {
3959 GroupElementTy = InterleaveR->getVPValue(0)->getScalarType();
3960 if (!all_of(InterleaveR->definedValues(), [GroupElementTy](VPValue *Op) {
3961 return Op->getScalarType() == GroupElementTy;
3962 }))
3963 return std::nullopt;
3964 } else {
3965 GroupElementTy = InterleaveR->getStoredValues()[0]->getScalarType();
3966 if (!all_of(InterleaveR->getStoredValues(), [GroupElementTy](VPValue *Op) {
3967 return Op->getScalarType() == GroupElementTy;
3968 }))
3969 return std::nullopt;
3970 }
3971
3972 auto IG = InterleaveR->getInterleaveGroup();
3973 if (IG->getFactor() != IG->getNumMembers())
3974 return std::nullopt;
3975
3976 auto GetVectorBitWidthForVF = [&TTI](ElementCount VF) {
3977 TypeSize Size = TTI.getRegisterBitWidth(
3980 assert(Size.isScalable() == VF.isScalable() &&
3981 "if Size is scalable, VF must be scalable and vice versa");
3982 return Size.getKnownMinValue();
3983 };
3984
3985 for (ElementCount VF : VFs) {
3986 unsigned MinVal = VF.getKnownMinValue();
3987 unsigned GroupSize = GroupElementTy->getScalarSizeInBits() * MinVal;
3988 if (IG->getFactor() == MinVal && GroupSize == GetVectorBitWidthForVF(VF))
3989 return {VF};
3990 }
3991 return std::nullopt;
3992}
3993
3994/// Returns true if \p VPValue is a narrow VPValue.
3995static bool isAlreadyNarrow(VPValue *VPV) {
3996 if (isa<VPIRValue>(VPV))
3997 return true;
3998 auto *RepR = dyn_cast<VPReplicateRecipe>(VPV);
3999 return RepR && RepR->isSingleScalar();
4000}
4001
4002// Convert the wide recipes defining the VPValues in \p Members feeding an
4003// interleave group to a single narrow variant. The first member is reused as
4004// the narrowed recipe. BuildVectors for live-in operands are inserted into \p
4005// Preheader.
4007 SmallPtrSetImpl<VPValue *> &NarrowedOps,
4008 VPBasicBlock *Preheader) {
4009 VPValue *V = Members.front();
4010 if (NarrowedOps.contains(V))
4011 return V;
4012
4013 if (V->isDefinedOutsideLoopRegions()) {
4014 assert(all_of(Members,
4015 [V](VPValue *M) {
4016 return M->isDefinedOutsideLoopRegions() &&
4017 M->getScalarType() == V->getScalarType();
4018 }) &&
4019 "expected distinct loop-invariant values of matching scalar type");
4020 auto *BV = new VPInstruction(VPInstruction::BuildVector, Members);
4021 Preheader->appendRecipe(BV);
4022 NarrowedOps.insert(BV);
4023 return BV;
4024 }
4025
4026 if (isAlreadyNarrow(V))
4027 return V;
4028
4029 VPRecipeBase *R = V->getDefiningRecipe();
4031 auto *WideMember0 = cast<VPRecipeWithIRFlags>(R);
4032 for (VPValue *Member : Members.drop_front())
4033 WideMember0->intersectFlags(*cast<VPRecipeWithIRFlags>(Member));
4034 for (unsigned Idx = 0, E = WideMember0->getNumOperands(); Idx != E; ++Idx) {
4036 for (VPValue *Member : Members)
4037 OpsI.push_back(Member->getDefiningRecipe()->getOperand(Idx));
4038 WideMember0->setOperand(
4039 Idx, narrowInterleaveGroupOp(OpsI, NarrowedOps, Preheader));
4040 }
4041 return V;
4042 }
4043
4044 if (auto *LoadGroup = dyn_cast<VPInterleaveRecipe>(R)) {
4045 // Narrow interleave group to wide load, as transformed VPlan will only
4046 // process one original iteration.
4047 auto *LI = cast<LoadInst>(LoadGroup->getInterleaveGroup()->getInsertPos());
4048 auto *L = VPBuilder(LoadGroup).createWidenLoad(
4049 *LI, LoadGroup->getAddr(), LoadGroup->getMask(), /*Consecutive=*/true,
4050 *LoadGroup, LoadGroup->getDebugLoc());
4051 NarrowedOps.insert(L);
4052 return L;
4053 }
4054
4055 if (auto *RepR = dyn_cast<VPReplicateRecipe>(R)) {
4056 assert(RepR->isSingleScalar() && RepR->getOpcode() == Instruction::Load &&
4057 "must be a single scalar load");
4058 NarrowedOps.insert(RepR);
4059 return RepR;
4060 }
4061
4062 auto *WideLoad = cast<VPWidenLoadRecipe>(R);
4063 VPValue *PtrOp = WideLoad->getAddr();
4064 if (auto *VecPtr = dyn_cast<VPVectorPointerRecipe>(PtrOp))
4065 PtrOp = VecPtr->getOperand(0);
4066 // Narrow wide load to uniform scalar load, as transformed VPlan will only
4067 // process one original iteration.
4068 auto *N = new VPReplicateRecipe(&WideLoad->getIngredient(), {PtrOp},
4069 /*IsUniform*/ true,
4070 /*Mask*/ nullptr, {}, *WideLoad);
4071 N->insertBefore(WideLoad);
4072 NarrowedOps.insert(N);
4073 return N;
4074}
4075
4076std::unique_ptr<VPlan>
4078 const TargetTransformInfo &TTI) {
4079 VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
4080
4081 if (!VectorLoop)
4082 return nullptr;
4083
4084 // Only handle single-block loops for now.
4085 if (VectorLoop->getEntryBasicBlock() != VectorLoop->getExitingBasicBlock())
4086 return nullptr;
4087
4088 // Skip plans when we may not be able to properly narrow.
4089 VPBasicBlock *Exiting = VectorLoop->getExitingBasicBlock();
4090 if (!match(&Exiting->back(), m_BranchOnCount()))
4091 return nullptr;
4092
4093 assert(match(&Exiting->back(),
4095 m_Specific(&Plan.getVectorTripCount()))) &&
4096 "unexpected branch-on-count");
4097
4099 std::optional<ElementCount> VFToOptimize;
4100 for (auto &R : *VectorLoop->getEntryBasicBlock()) {
4103 continue;
4104
4105 // Bail out on recipes not supported at the moment:
4106 // * phi recipes other than the canonical induction
4107 // * recipes writing to memory except interleave groups
4108 // Only support plans with a canonical induction phi.
4109 if (R.isPhi())
4110 return nullptr;
4111
4112 auto *InterleaveR = dyn_cast<VPInterleaveRecipe>(&R);
4113 if (R.mayWriteToMemory() && !InterleaveR)
4114 return nullptr;
4115
4116 // Bail out if any recipe defines a vector value used outside the
4117 // vector loop region.
4118 if (any_of(R.definedValues(), [&](VPValue *V) {
4119 return any_of(V->users(), [&](VPUser *U) {
4120 auto *UR = cast<VPRecipeBase>(U);
4121 return UR->getParent()->getParent() != VectorLoop;
4122 });
4123 }))
4124 return nullptr;
4125
4126 // All other ops are allowed, but we reject uses that cannot be converted
4127 // when checking all allowed consumers (store interleave groups) below.
4128 if (!InterleaveR)
4129 continue;
4130
4131 // Try to find a single VF, where all interleave groups are consecutive and
4132 // saturate the full vector width. If we already have a candidate VF, check
4133 // if it is applicable for the current InterleaveR, otherwise look for a
4134 // suitable VF across the Plan's VFs.
4136 VFToOptimize ? SmallVector<ElementCount>({*VFToOptimize})
4137 : to_vector(Plan.vectorFactors());
4138 std::optional<ElementCount> NarrowedVF =
4139 isConsecutiveInterleaveGroup(InterleaveR, VFs, TTI);
4140 if (!NarrowedVF || (VFToOptimize && NarrowedVF != VFToOptimize))
4141 return nullptr;
4142 VFToOptimize = NarrowedVF;
4143
4144 // Skip read interleave groups.
4145 if (InterleaveR->getStoredValues().empty())
4146 continue;
4147
4148 // Narrow interleave groups, if all operands are already matching narrow
4149 // ops.
4150 auto *Member0 = InterleaveR->getStoredValues()[0];
4151 if (isAlreadyNarrow(Member0) &&
4152 all_of(InterleaveR->getStoredValues(), equal_to(Member0))) {
4153 StoreGroups.push_back(InterleaveR);
4154 continue;
4155 }
4156
4157 // For now, we only support full interleave groups storing load interleave
4158 // groups.
4159 if (all_of(enumerate(InterleaveR->getStoredValues()), [](auto Op) {
4160 VPRecipeBase *DefR = Op.value()->getDefiningRecipe();
4161 if (!DefR)
4162 return false;
4163 auto *IR = dyn_cast<VPInterleaveRecipe>(DefR);
4164 return IR && IR->getInterleaveGroup()->isFull() &&
4165 IR->getVPValue(Op.index()) == Op.value();
4166 })) {
4167 StoreGroups.push_back(InterleaveR);
4168 continue;
4169 }
4170
4171 // Check if all values feeding InterleaveR are matching wide recipes, which
4172 // operands that can be narrowed.
4173 if (!canNarrowOps(InterleaveR->getStoredValues(),
4174 VFToOptimize->isScalable()))
4175 return nullptr;
4176 StoreGroups.push_back(InterleaveR);
4177 }
4178
4179 if (StoreGroups.empty())
4180 return nullptr;
4181
4182 VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
4183 bool RequiresScalarEpilogue =
4184 MiddleVPBB->getNumSuccessors() == 1 &&
4185 MiddleVPBB->getSingleSuccessor() == Plan.getScalarPreheader();
4186 // Bail out for tail-folding (middle block with a single successor to exit).
4187 if (MiddleVPBB->getNumSuccessors() != 2 && !RequiresScalarEpilogue)
4188 return nullptr;
4189
4190 // All interleave groups in Plan can be narrowed for VFToOptimize. Split the
4191 // original Plan into 2: a) a new clone which contains all VFs of Plan, except
4192 // VFToOptimize, and b) the original Plan with VFToOptimize as single VF.
4193 // TODO: Handle cases where only some interleave groups can be narrowed.
4194 std::unique_ptr<VPlan> NewPlan;
4195 if (size(Plan.vectorFactors()) != 1) {
4196 NewPlan = std::unique_ptr<VPlan>(Plan.duplicate());
4197 Plan.setVF(*VFToOptimize);
4198 NewPlan->removeVF(*VFToOptimize);
4199 }
4200
4201 // Convert InterleaveGroup \p R to a single VPWidenLoadRecipe.
4202 SmallPtrSet<VPValue *, 4> NarrowedOps;
4203 VPBasicBlock *Preheader = Plan.getVectorPreheader();
4204 // Narrow operation tree rooted at store groups.
4205 for (auto *StoreGroup : StoreGroups) {
4206 VPValue *Res = narrowInterleaveGroupOp(StoreGroup->getStoredValues(),
4207 NarrowedOps, Preheader);
4208 auto *SI =
4209 cast<StoreInst>(StoreGroup->getInterleaveGroup()->getInsertPos());
4210 VPBuilder(StoreGroup)
4211 .createWidenStore(*SI, StoreGroup->getAddr(), Res, nullptr,
4212 /*Consecutive=*/true, *StoreGroup,
4213 StoreGroup->getDebugLoc());
4214 StoreGroup->eraseFromParent();
4215 }
4216
4217 // Adjust induction to reflect that the transformed plan only processes one
4218 // original iteration.
4220 Type *CanIVTy = VectorLoop->getCanonicalIVType();
4221 VPBasicBlock *VectorPH = Plan.getVectorPreheader();
4222 VPBuilder PHBuilder(VectorPH, VectorPH->begin());
4223
4224 VPValue *UF = &Plan.getUF();
4225 VPValue *Step;
4226 if (VFToOptimize->isScalable()) {
4227 VPValue *VScale =
4228 PHBuilder.createElementCount(CanIVTy, ElementCount::getScalable(1));
4229 Step = PHBuilder.createOverflowingOp(Instruction::Mul, {VScale, UF},
4230 {true, false});
4231 Plan.getVF().replaceAllUsesWith(VScale);
4232 } else {
4233 Step = UF;
4234 Plan.getVF().replaceAllUsesWith(Plan.getConstantInt(CanIVTy, 1));
4235 }
4236 // Materialize vector trip count with the narrowed step.
4237 materializeVectorTripCount(Plan, VectorPH, /*TailByMasking=*/false,
4238 RequiresScalarEpilogue, Step);
4239
4240 CanIVInc->setOperand(1, Step);
4241 Plan.getVFxUF().replaceAllUsesWith(Step);
4242
4243 removeDeadRecipes(Plan);
4244 assert(none_of(*VectorLoop->getEntryBasicBlock(),
4246 "All VPVectorPointerRecipes should have been removed");
4247 return NewPlan;
4248}
4249
4251 VFRange &Range) {
4252 VPRegionBlock *VectorRegion = Plan.getVectorLoopRegion();
4253 auto *MiddleVPBB = Plan.getMiddleBlock();
4254 VPBuilder MiddleBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
4255
4256 auto IsScalableOne = [](ElementCount VF) -> bool {
4257 return VF == ElementCount::getScalable(1);
4258 };
4259
4260 for (auto &HeaderPhi : VectorRegion->getEntryBasicBlock()->phis()) {
4261 auto *FOR = dyn_cast<VPFirstOrderRecurrencePHIRecipe>(&HeaderPhi);
4262 if (!FOR)
4263 continue;
4264
4265 assert(VectorRegion->getSingleSuccessor() == Plan.getMiddleBlock() &&
4266 "Cannot handle loops with uncountable early exits");
4267
4268 // Find the existing splice for this FOR, created in
4269 // createHeaderPhiRecipes. All uses of FOR have already been replaced with
4270 // RecurSplice there; only RecurSplice itself still references FOR.
4271 auto *RecurSplice =
4273 assert(RecurSplice && "expected FirstOrderRecurrenceSplice");
4274
4275 // For VF vscale x 1, if vscale = 1, we are unable to extract the
4276 // penultimate value of the recurrence. Instead we rely on the existing
4277 // extract of the last element from the result of
4278 // VPInstruction::FirstOrderRecurrenceSplice.
4279 // TODO: Consider vscale_range info and UF.
4280 if (any_of(RecurSplice->users(),
4281 [](VPUser *U) { return !cast<VPRecipeBase>(U)->getRegion(); }) &&
4283 Range))
4284 return;
4285
4286 // This is the second phase of vectorizing first-order recurrences, creating
4287 // extracts for users outside the loop. An overview of the transformation is
4288 // described below. Suppose we have the following loop with some use after
4289 // the loop of the last a[i-1],
4290 //
4291 // for (int i = 0; i < n; ++i) {
4292 // t = a[i - 1];
4293 // b[i] = a[i] - t;
4294 // }
4295 // use t;
4296 //
4297 // There is a first-order recurrence on "a". For this loop, the shorthand
4298 // scalar IR looks like:
4299 //
4300 // scalar.ph:
4301 // s.init = a[-1]
4302 // br scalar.body
4303 //
4304 // scalar.body:
4305 // i = phi [0, scalar.ph], [i+1, scalar.body]
4306 // s1 = phi [s.init, scalar.ph], [s2, scalar.body]
4307 // s2 = a[i]
4308 // b[i] = s2 - s1
4309 // br cond, scalar.body, exit.block
4310 //
4311 // exit.block:
4312 // use = lcssa.phi [s1, scalar.body]
4313 //
4314 // In this example, s1 is a recurrence because it's value depends on the
4315 // previous iteration. In the first phase of vectorization, we created a
4316 // VPFirstOrderRecurrencePHIRecipe v1 for s1. Now we create the extracts
4317 // for users in the scalar preheader and exit block.
4318 //
4319 // vector.ph:
4320 // v_init = vector(..., ..., ..., a[-1])
4321 // br vector.body
4322 //
4323 // vector.body
4324 // i = phi [0, vector.ph], [i+4, vector.body]
4325 // v1 = phi [v_init, vector.ph], [v2, vector.body]
4326 // v2 = a[i, i+1, i+2, i+3]
4327 // v1' = splice(v1(3), v2(0, 1, 2))
4328 // b[i, i+1, i+2, i+3] = v2 - v1'
4329 // br cond, vector.body, middle.block
4330 //
4331 // middle.block:
4332 // vector.recur.extract.for.phi = v2(2)
4333 // vector.recur.extract = v2(3)
4334 // br cond, scalar.ph, exit.block
4335 //
4336 // scalar.ph:
4337 // scalar.recur.init = phi [vector.recur.extract, middle.block],
4338 // [s.init, otherwise]
4339 // br scalar.body
4340 //
4341 // scalar.body:
4342 // i = phi [0, scalar.ph], [i+1, scalar.body]
4343 // s1 = phi [scalar.recur.init, scalar.ph], [s2, scalar.body]
4344 // s2 = a[i]
4345 // b[i] = s2 - s1
4346 // br cond, scalar.body, exit.block
4347 //
4348 // exit.block:
4349 // lo = lcssa.phi [s1, scalar.body],
4350 // [vector.recur.extract.for.phi, middle.block]
4351 //
4352 // Update extracts of the splice in the middle block: they extract the
4353 // penultimate element of the recurrence.
4355 make_range(MiddleVPBB->getFirstNonPhi(), MiddleVPBB->end()))) {
4356 if (!match(&R, m_ExtractLastLaneOfLastPart(m_Specific(RecurSplice))))
4357 continue;
4358
4359 auto *ExtractR = cast<VPInstruction>(&R);
4360 VPValue *PenultimateElement = MiddleBuilder.createNaryOp(
4361 VPInstruction::ExtractPenultimateElement, RecurSplice->getOperand(1),
4362 {}, "vector.recur.extract.for.phi");
4363 for (VPUser *ExitU : to_vector(ExtractR->users())) {
4364 if (auto *ExitPhi = dyn_cast<VPIRPhi>(ExitU))
4365 ExitPhi->replaceUsesOfWith(ExtractR, PenultimateElement);
4366 }
4367 }
4368 }
4369}
4370
4371/// Check if \p V is a binary expression of a widened IV and a loop-invariant
4372/// value. Returns the widened IV if found, nullptr otherwise.
4374 auto *BinOp = dyn_cast<VPWidenRecipe>(V);
4375 if (!BinOp || !Instruction::isBinaryOp(BinOp->getOpcode()) ||
4376 Instruction::isIntDivRem(BinOp->getOpcode()))
4377 return nullptr;
4378
4379 VPValue *WidenIVCandidate = BinOp->getOperand(0);
4380 VPValue *InvariantCandidate = BinOp->getOperand(1);
4381 if (!isa<VPWidenIntOrFpInductionRecipe>(WidenIVCandidate))
4382 std::swap(WidenIVCandidate, InvariantCandidate);
4383
4384 if (!InvariantCandidate->isDefinedOutsideLoopRegions())
4385 return nullptr;
4386
4387 return dyn_cast<VPWidenIntOrFpInductionRecipe>(WidenIVCandidate);
4388}
4389
4390/// Create a scalar version of \p BinOp, with its \p WidenIV operand replaced
4391/// by \p ScalarIV, and place it after \p ScalarIV's defining recipe.
4395 BinOp->getNumOperands() == 2 && "BinOp must have 2 operands");
4396 auto *ClonedOp = BinOp->clone();
4397 if (ClonedOp->getOperand(0) == WidenIV) {
4398 ClonedOp->setOperand(0, ScalarIV);
4399 } else {
4400 assert(ClonedOp->getOperand(1) == WidenIV && "one operand must be WideIV");
4401 ClonedOp->setOperand(1, ScalarIV);
4402 }
4403 ClonedOp->insertAfter(ScalarIV->getDefiningRecipe());
4404 return ClonedOp;
4405}
4406
4407/// If \p S is an affine AddRec, returns true if its step is known to be
4408/// positive and false if it is known to be negative. Returns std::nullopt if
4409/// \p S is not an affine AddRec, or if the sign of its step cannot be
4410/// determined.
4411static std::optional<bool> getStepDirection(const SCEV *S,
4412 ScalarEvolution &SE) {
4413 const SCEV *Step;
4414 if (!match(S, m_scev_AffineAddRec(m_SCEV(), m_SCEV(Step))))
4415 return std::nullopt;
4416 if (SE.isKnownPositive(Step))
4417 return true;
4418 if (SE.isKnownNegative(Step))
4419 return false;
4420 return std::nullopt;
4421}
4422
4425 Loop &L) {
4426 ScalarEvolution &SE = *PSE.getSE();
4427 VPRegionBlock *VectorLoopRegion = Plan.getVectorLoopRegion();
4428
4429 // Helper lambda to check if the IV range excludes the sentinel value. Try
4430 // signed first, then unsigned. Return an excluded sentinel if found,
4431 // otherwise return std::nullopt.
4432 auto CheckSentinel = [&SE](const SCEV *IVSCEV,
4433 bool UseMax) -> std::optional<APSInt> {
4434 unsigned BW = IVSCEV->getType()->getScalarSizeInBits();
4435 for (bool Signed : {true, false}) {
4436 APSInt Sentinel = UseMax ? APSInt::getMinValue(BW, /*Unsigned=*/!Signed)
4437 : APSInt::getMaxValue(BW, /*Unsigned=*/!Signed);
4438
4439 ConstantRange IVRange =
4440 Signed ? SE.getSignedRange(IVSCEV) : SE.getUnsignedRange(IVSCEV);
4441 if (!IVRange.contains(Sentinel))
4442 return Sentinel;
4443 }
4444 return std::nullopt;
4445 };
4446
4447 VPValue *HeaderMask = VectorLoopRegion->getHeaderMask();
4448 for (VPRecipeBase &Phi :
4449 make_early_inc_range(VectorLoopRegion->getEntryBasicBlock()->phis())) {
4450 auto *PhiR = dyn_cast<VPReductionPHIRecipe>(&Phi);
4452 PhiR->getRecurrenceKind()))
4453 continue;
4454
4455 Type *PhiTy = PhiR->getScalarType();
4456 if (PhiTy->isPointerTy() || PhiTy->isFloatingPointTy())
4457 continue;
4458
4459 // If there's a header mask, the backedge select will not be the find-last
4460 // select.
4461 VPValue *BackedgeVal = PhiR->getBackedgeValue();
4462 auto *FindLastSelect = cast<VPSingleDefRecipe>(BackedgeVal);
4463 if (HeaderMask &&
4464 !match(BackedgeVal,
4465 m_Select(m_Specific(HeaderMask),
4466 m_VPSingleDefRecipe(FindLastSelect), m_Specific(PhiR))))
4467 continue;
4468
4469 // Get the find-last expression from the find-last select of the reduction
4470 // phi. The find-last select should be a select between the phi and the
4471 // find-last expression.
4472 VPValue *Cond, *FindLastExpression;
4473 if (!match(FindLastSelect, m_SelectLike(m_VPValue(Cond), m_Specific(PhiR),
4474 m_VPValue(FindLastExpression))) &&
4475 !match(FindLastSelect,
4476 m_SelectLike(m_VPValue(Cond), m_VPValue(FindLastExpression),
4477 m_Specific(PhiR))))
4478 continue;
4479
4480 // Check if FindLastExpression is a simple expression of a widened IV. If
4481 // so, we can track the underlying IV instead and sink the expression.
4482 auto *IVOfExpressionToSink = getExpressionIV(FindLastExpression);
4483 const SCEV *IVSCEV = vputils::getSCEVExprForVPValue(
4484 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression, PSE,
4485 &L);
4486 if (!match(IVSCEV, m_scev_AffineAddRec(m_SCEV(), m_SCEV()))) {
4487 assert(!match(vputils::getSCEVExprForVPValue(FindLastExpression, PSE, &L),
4489 "IVOfExpressionToSink not being an AddRec must imply "
4490 "FindLastExpression not being an AddRec.");
4491 continue;
4492 }
4493
4494 // Determine direction from the step of IVSCEV, if possible.
4495 std::optional<bool> StepDirection = getStepDirection(IVSCEV, SE);
4496 if (!StepDirection)
4497 continue;
4498
4499 bool UseMax = *StepDirection;
4500 std::optional<APSInt> SentinelVal = CheckSentinel(IVSCEV, UseMax);
4501 bool UseSigned = SentinelVal && SentinelVal->isSigned();
4502
4503 // Sinking an expression will disable epilogue vectorization. Only use it,
4504 // if FindLastExpression cannot be vectorized via a sentinel. Sinking may
4505 // also prevent vectorizing using a sentinel (e.g., if the expression is a
4506 // multiply or divide by large constant, respectively), which also makes
4507 // sinking undesirable.
4508 if (IVOfExpressionToSink) {
4509 const SCEV *FindLastExpressionSCEV =
4510 vputils::getSCEVExprForVPValue(FindLastExpression, PSE, &L);
4511 if (std::optional<bool> NewUseMax =
4512 getStepDirection(FindLastExpressionSCEV, SE)) {
4513 if (auto NewSentinel =
4514 CheckSentinel(FindLastExpressionSCEV, *NewUseMax)) {
4515 // The original expression already has a sentinel, so prefer not
4516 // sinking to keep epilogue vectorization possible.
4517 SentinelVal = *NewSentinel;
4518 UseSigned = NewSentinel->isSigned();
4519 UseMax = *NewUseMax;
4520 IVSCEV = FindLastExpressionSCEV;
4521 IVOfExpressionToSink = nullptr;
4522 }
4523 }
4524 }
4525
4526 // If no sentinel was found, fall back to a boolean AnyOf reduction to track
4527 // if the condition was ever true. Requires the IV to not wrap, otherwise we
4528 // cannot use min/max.
4529 if (!SentinelVal) {
4530 auto *AR = cast<SCEVAddRecExpr>(IVSCEV);
4531 if (AR->hasNoSignedWrap())
4532 UseSigned = true;
4533 else if (AR->hasNoUnsignedWrap())
4534 UseSigned = false;
4535 else
4536 continue;
4537 }
4538
4540 BackedgeVal,
4542
4543 VPValue *NewFindLastSelect = BackedgeVal;
4544 VPValue *SelectCond = Cond;
4545 if (!SentinelVal || IVOfExpressionToSink) {
4546 // When we need to create a new select, normalize the condition so that
4547 // PhiR is the last operand and include the header mask if needed.
4548 DebugLoc DL = FindLastSelect->getDefiningRecipe()->getDebugLoc();
4549 VPBuilder LoopBuilder(FindLastSelect->getDefiningRecipe());
4550 if (match(FindLastSelect,
4552 SelectCond = LoopBuilder.createNot(SelectCond);
4553
4554 // When tail folding, mask the condition with the header mask to prevent
4555 // propagating poison from inactive lanes in the last vector iteration.
4556 if (HeaderMask)
4557 SelectCond = LoopBuilder.createLogicalAnd(HeaderMask, SelectCond);
4558
4559 if (SelectCond != Cond || IVOfExpressionToSink) {
4560 NewFindLastSelect = LoopBuilder.createSelect(
4561 SelectCond,
4562 IVOfExpressionToSink ? IVOfExpressionToSink : FindLastExpression,
4563 PhiR, DL);
4564 }
4565 }
4566
4567 // Create the reduction result in the middle block using sentinel directly.
4568 RecurKind MinMaxKind =
4569 UseMax ? (UseSigned ? RecurKind::SMax : RecurKind::UMax)
4570 : (UseSigned ? RecurKind::SMin : RecurKind::UMin);
4571 VPIRFlags Flags(MinMaxKind, /*IsOrdered=*/false, /*IsInLoop=*/false,
4572 FastMathFlags());
4573 DebugLoc ExitDL = RdxResult->getDebugLoc();
4574 VPBuilder MiddleBuilder(RdxResult);
4575 VPValue *ReducedIV =
4577 NewFindLastSelect, Flags, ExitDL);
4578
4579 // If IVOfExpressionToSink is an expression to sink, sink it now.
4580 VPValue *VectorRegionExitingVal = ReducedIV;
4581 if (IVOfExpressionToSink)
4582 VectorRegionExitingVal =
4583 cloneBinOpForScalarIV(cast<VPWidenRecipe>(FindLastExpression),
4584 ReducedIV, IVOfExpressionToSink);
4585
4586 VPValue *NewRdxResult;
4587 VPValue *StartVPV = PhiR->getStartValue();
4588 if (SentinelVal) {
4589 // Sentinel-based approach: reduce IVs with min/max, compare against
4590 // sentinel to detect if condition was ever true, select accordingly.
4591 VPValue *Sentinel = Plan.getConstantInt(*SentinelVal);
4592 auto *Cmp = MiddleBuilder.createICmp(CmpInst::ICMP_NE, ReducedIV,
4593 Sentinel, ExitDL);
4594 NewRdxResult = MiddleBuilder.createSelect(Cmp, VectorRegionExitingVal,
4595 StartVPV, ExitDL);
4596 StartVPV = Sentinel;
4597 } else {
4598 // Introduce a boolean AnyOf reduction to track if the condition was ever
4599 // true in the loop. Use it to select the initial start value, if it was
4600 // never true.
4601 auto *AnyOfPhi = new VPReductionPHIRecipe(
4602 /*Phi=*/nullptr, RecurKind::Or, *Plan.getFalse(), *Plan.getFalse(),
4603 RdxUnordered{1}, {}, /*HasUsesOutsideReductionChain=*/false);
4604 AnyOfPhi->insertAfter(PhiR);
4605
4606 VPBuilder LoopBuilder(BackedgeVal->getDefiningRecipe());
4607 VPValue *OrVal = LoopBuilder.createOr(AnyOfPhi, SelectCond);
4608 AnyOfPhi->setOperand(1, OrVal);
4609
4610 NewRdxResult = MiddleBuilder.createAnyOfReduction(
4611 OrVal, VectorRegionExitingVal, StartVPV, ExitDL);
4612
4613 // Initialize the IV reduction phi with the neutral element, not the
4614 // original start value, to ensure correct min/max reduction results.
4615 StartVPV = Plan.getOrAddLiveIn(
4616 getRecurrenceIdentity(MinMaxKind, IVSCEV->getType(), {}));
4617 }
4618 RdxResult->replaceAllUsesWith(NewRdxResult);
4619 RdxResult->eraseFromParent();
4620
4621 auto *NewPhiR = new VPReductionPHIRecipe(
4622 cast<PHINode>(PhiR->getUnderlyingInstr()), RecurKind::FindIV, *StartVPV,
4623 *NewFindLastSelect, RdxUnordered{1}, {},
4624 PhiR->hasUsesOutsideReductionChain());
4625 NewPhiR->insertBefore(PhiR);
4626 PhiR->replaceAllUsesWith(NewPhiR);
4627 PhiR->eraseFromParent();
4628 }
4629}
4630
4631namespace {
4632
4633using ExtendKind = TTI::PartialReductionExtendKind;
4634struct ReductionExtend {
4635 Type *SrcType = nullptr;
4636 ExtendKind Kind = ExtendKind::PR_None;
4637};
4638
4639/// Describes the extends used to compute the extended reduction operand.
4640/// ExtendB is optional. If ExtendB is present, ExtendsUser is a binary
4641/// operation.
4642struct ExtendedReductionOperand {
4643 /// The recipe that consumes the extends.
4644 VPWidenRecipe *ExtendsUser = nullptr;
4645 /// Extend descriptions (inputs to getPartialReductionCost).
4646 ReductionExtend ExtendA, ExtendB;
4647};
4648
4649/// A chain of recipes that form a partial reduction. Matches either
4650/// reduction_bin_op (extended op, accumulator), or
4651/// reduction_bin_op (accumulator, extended op).
4652/// The possible forms of the "extended op" are listed in
4653/// matchExtendedReductionOperand.
4654struct VPPartialReductionChain {
4655 /// The top-level binary operation that forms the reduction to a scalar
4656 /// after the loop body.
4657 VPWidenRecipe *ReductionBinOp = nullptr;
4658 /// The user of the extends that is then reduced.
4659 ExtendedReductionOperand ExtendedOp;
4660 /// The recurrence kind for the entire partial reduction chain.
4661 /// This allows distinguishing between Sub and AddWithSub recurrences,
4662 /// when the ReductionBinOp is a Instruction::Sub.
4663 RecurKind RK;
4664 /// The index of the accumulator operand of ReductionBinOp. The extended op
4665 /// is `1 - AccumulatorOpIdx`.
4666 unsigned AccumulatorOpIdx;
4667 unsigned ScaleFactor;
4668 /// Optional blend to represent predication for the block that updates the
4669 /// reduction.
4670 VPBlendRecipe *Blend = nullptr;
4671};
4672
4673// Return the incoming index of the single-use value in the blend, which is
4674// expected to be the predicated reduction update.
4675static std::optional<unsigned>
4676getBlendReductionUpdateValueIdx(VPBlendRecipe *Blend) {
4677 assert(Blend && !Blend->isNormalized() &&
4678 Blend->getNumIncomingValues() == 2 &&
4679 "Expected a non-normalized blend with two incoming values");
4680 bool FirstIncomingHasOneUse = Blend->getIncomingValue(0)->hasOneUse();
4681
4682 // Only the update value should have one use (the blend). The previous
4683 // value should always have at least two uses, the blend and the reduction.
4684 if (FirstIncomingHasOneUse == Blend->getIncomingValue(1)->hasOneUse())
4685 return std::nullopt;
4686 return FirstIncomingHasOneUse ? 0 : 1;
4687}
4688
4689static VPSingleDefRecipe *
4690optimizeExtendsForPartialReduction(VPSingleDefRecipe *Op) {
4691 // reduce.add(mul(ext(A), C))
4692 // -> reduce.add(mul(ext(A), ext(trunc(C))))
4693 const APInt *Const;
4694 if (match(Op, m_Mul(m_ZExtOrSExt(m_VPValue()), m_APInt(Const)))) {
4695 auto *ExtA = cast<VPWidenCastRecipe>(Op->getOperand(0));
4696 Instruction::CastOps ExtOpc = ExtA->getOpcode();
4697 Type *NarrowTy = ExtA->getOperand(0)->getScalarType();
4698 if (!Op->hasOneUse() ||
4700 Const, NarrowTy, TTI::getPartialReductionExtendKind(ExtOpc)))
4701 return Op;
4702
4703 VPBuilder Builder(Op);
4704 auto *Trunc = Builder.createWidenCast(Instruction::CastOps::Trunc,
4705 Op->getOperand(1), NarrowTy);
4706 Type *WideTy = ExtA->getScalarType();
4707 Op->setOperand(1, Builder.createWidenCast(ExtOpc, Trunc, WideTy));
4708 return Op;
4709 }
4710
4711 // reduce.add(abs(sub(ext(A), ext(B))))
4712 // -> reduce.add(ext(absolute-difference(A, B)))
4713 VPValue *X, *Y;
4716 auto *Sub = Op->getOperand(0)->getDefiningRecipe();
4717 auto *Ext = cast<VPWidenCastRecipe>(Sub->getOperand(0));
4718 assert(Ext->getOpcode() ==
4719 cast<VPWidenCastRecipe>(Sub->getOperand(1))->getOpcode() &&
4720 "Expected both the LHS and RHS extends to be the same");
4721 bool IsSigned = Ext->getOpcode() == Instruction::SExt;
4722 VPBuilder Builder(Op);
4723 Type *SrcTy = X->getScalarType();
4724 auto *FreezeX = Builder.insert(new VPWidenRecipe(Instruction::Freeze, {X}));
4725 auto *FreezeY = Builder.insert(new VPWidenRecipe(Instruction::Freeze, {Y}));
4726 auto *Max = Builder.insert(
4727 new VPWidenIntrinsicRecipe(IsSigned ? Intrinsic::smax : Intrinsic::umax,
4728 {FreezeX, FreezeY}, SrcTy));
4729 auto *Min = Builder.insert(
4730 new VPWidenIntrinsicRecipe(IsSigned ? Intrinsic::smin : Intrinsic::umin,
4731 {FreezeX, FreezeY}, SrcTy));
4732 auto *AbsDiff = Builder.insert(
4733 new VPWidenRecipe(Instruction::Sub, {Max, Min},
4734 VPIRFlags::getDefaultFlags(Instruction::Sub)));
4735 return Builder.createWidenCast(Instruction::CastOps::ZExt, AbsDiff,
4736 Op->getScalarType());
4737 }
4738
4739 // reduce.add(ext(mul(ext(A), ext(B))))
4740 // -> reduce.add(mul(wider_ext(A), wider_ext(B)))
4741 // TODO: Support this optimization for float types.
4743 m_ZExtOrSExt(m_VPValue()))))) {
4744 auto *Ext = cast<VPWidenCastRecipe>(Op);
4745 auto *Mul = cast<VPWidenRecipe>(Ext->getOperand(0));
4746 auto *MulLHS = cast<VPWidenCastRecipe>(Mul->getOperand(0));
4747 auto *MulRHS = cast<VPWidenCastRecipe>(Mul->getOperand(1));
4748 if (!Mul->hasOneUse() ||
4749 (Ext->getOpcode() != MulLHS->getOpcode() && MulLHS != MulRHS) ||
4750 MulLHS->getOpcode() != MulRHS->getOpcode())
4751 return Op;
4752 VPBuilder Builder(Mul);
4753 auto *NewLHS = Builder.createWidenCast(
4754 MulLHS->getOpcode(), MulLHS->getOperand(0), Ext->getScalarType());
4755 auto *NewRHS = MulLHS == MulRHS
4756 ? NewLHS
4757 : Builder.createWidenCast(MulRHS->getOpcode(),
4758 MulRHS->getOperand(0),
4759 Ext->getScalarType());
4760 auto *NewMul = Mul->cloneWithOperands({NewLHS, NewRHS});
4761 Builder.insert(NewMul);
4762 Op->replaceAllUsesWith(NewMul);
4763 Op->eraseFromParent();
4764 Mul->eraseFromParent();
4765 return NewMul;
4766 }
4767
4768 return Op;
4769}
4770
4771static VPExpressionRecipe *
4772createPartialReductionExpression(VPReductionRecipe *Red) {
4773 VPValue *VecOp = Red->getVecOp();
4774
4775 // reduce.[f]add(ext(op))
4776 // -> VPExpressionRecipe(op, red)
4777 if (match(VecOp, m_WidenAnyExtend(m_VPValue())))
4778 return new VPExpressionRecipe(cast<VPWidenCastRecipe>(VecOp), Red);
4779
4780 // reduce.[f]add(neg(ext(op)))
4781 // -> VPExpressionRecipe(op, sub/neg, red)
4782 if (match(VecOp, m_AnyNeg(m_WidenAnyExtend(m_VPValue())))) {
4783 auto *Neg = cast<VPWidenRecipe>(VecOp);
4784 auto *Ext =
4785 cast<VPWidenCastRecipe>(Neg->getOperand(Neg->getNumOperands() - 1));
4786 return new VPExpressionRecipe(Ext, Neg, Red);
4787 }
4788
4789 // reduce.[f]add([f]mul(ext(a), ext(b)))
4790 // -> VPExpressionRecipe(a, b, mul, red)
4791 if (match(VecOp, m_FMul(m_FPExt(m_VPValue()), m_FPExt(m_VPValue()))) ||
4792 match(VecOp,
4794 auto *Mul = cast<VPWidenRecipe>(VecOp);
4795 auto *ExtA = cast<VPWidenCastRecipe>(Mul->getOperand(0));
4796 auto *ExtB = cast<VPWidenCastRecipe>(Mul->getOperand(1));
4797 return new VPExpressionRecipe(ExtA, ExtB, Mul, Red);
4798 }
4799
4800 // reduce.fadd(fneg(fmul(fpext(a), fpext(b))))
4801 // -> VPExpressionRecipe(a, b, fmul, fsub, red)
4802 if (match(VecOp,
4804 auto *FNeg = cast<VPWidenRecipe>(VecOp);
4805 auto *FMul = cast<VPWidenRecipe>(FNeg->getOperand(0));
4806 auto *ExtA = cast<VPWidenCastRecipe>(FMul->getOperand(0));
4807 auto *ExtB = cast<VPWidenCastRecipe>(FMul->getOperand(1));
4808 return new VPExpressionRecipe(ExtA, ExtB, FMul, FNeg, Red);
4809 }
4810
4811 // reduce.add(neg(mul(ext(a), ext(b))))
4812 // -> VPExpressionRecipe(a, b, mul, sub, red)
4814 m_ZExtOrSExt(m_VPValue()))))) {
4815 auto *Sub = cast<VPWidenRecipe>(VecOp);
4816 auto *Mul = cast<VPWidenRecipe>(Sub->getOperand(1));
4817 auto *ExtA = cast<VPWidenCastRecipe>(Mul->getOperand(0));
4818 auto *ExtB = cast<VPWidenCastRecipe>(Mul->getOperand(1));
4819 return new VPExpressionRecipe(ExtA, ExtB, Mul, Sub, Red);
4820 }
4821
4822 llvm_unreachable("Unsupported expression");
4823}
4824
4825// Helper to transform a partial reduction chain into a partial reduction
4826// recipe. Assumes profitability has been checked.
4827static void transformToPartialReduction(const VPPartialReductionChain &Chain,
4828 VPlan &Plan,
4829 VPReductionPHIRecipe *RdxPhi) {
4830 VPWidenRecipe *WidenRecipe = Chain.ReductionBinOp;
4831 assert(WidenRecipe->getNumOperands() == 2 && "Expected binary operation");
4832
4833 VPValue *Accumulator = WidenRecipe->getOperand(Chain.AccumulatorOpIdx);
4834 auto *ExtendedOp = cast<VPSingleDefRecipe>(
4835 WidenRecipe->getOperand(1 - Chain.AccumulatorOpIdx));
4836
4837 // FIXME: Do these transforms before invoking the cost-model.
4838 ExtendedOp = optimizeExtendsForPartialReduction(ExtendedOp);
4839
4840 // Sub-reductions can be implemented in two ways:
4841 // (1) negate the operand in the vector loop (the default way).
4842 // (2) subtract the reduced value from the init value in the middle block.
4843 // Both ways keep the reduction itself as an 'add' reduction.
4844 //
4845 // The ISD nodes for partial reductions don't support folding the
4846 // sub/negation into its operands because the following is not a valid
4847 // transformation:
4848 // sub(0, mul(ext(a), ext(b)))
4849 // -> mul(ext(a), ext(sub(0, b)))
4850 //
4851 // It's therefore better to choose option (2) such that the partial
4852 // reduction is always positive (starting at '0') and to do a final
4853 // subtract in the middle block.
4854 if ((WidenRecipe->getOpcode() == Instruction::Sub &&
4855 Chain.RK != RecurKind::Sub) ||
4856 (WidenRecipe->getOpcode() == Instruction::FSub &&
4857 Chain.RK != RecurKind::FSub)) {
4858 VPBuilder Builder(WidenRecipe);
4859 Type *ElemTy = ExtendedOp->getScalarType();
4860 VPWidenRecipe *NegRecipe;
4861 if (WidenRecipe->getOpcode() == Instruction::FSub) {
4862 NegRecipe =
4863 new VPWidenRecipe(Instruction::FNeg, {ExtendedOp},
4864 VPIRFlags::getDefaultFlags(Instruction::FNeg),
4866 } else {
4867 auto *Zero = Plan.getZero(ElemTy);
4868 NegRecipe =
4869 new VPWidenRecipe(Instruction::Sub, {Zero, ExtendedOp},
4870 VPIRFlags::getDefaultFlags(Instruction::Sub),
4872 }
4873 Builder.insert(NegRecipe);
4874 ExtendedOp = NegRecipe;
4875 }
4876
4877 // Check if WidenRecipe is the final result of the reduction. If so, look
4878 // through the Select recipe introduced by tail-folding, otherwise look
4879 // through any Blend recipe introduced by predication for the block.
4880 VPValue *ExitSearch =
4881 Chain.Blend ? cast<VPValue>(Chain.Blend) : cast<VPValue>(WidenRecipe);
4882
4883 VPValue *Cond = nullptr;
4885 findUserOf(ExitSearch, m_Select(m_VPValue(Cond), m_Specific(ExitSearch),
4886 m_Specific(RdxPhi))));
4887
4888 if (Chain.Blend) {
4889 std::optional<unsigned> BlendReductionIdx =
4890 getBlendReductionUpdateValueIdx(Chain.Blend);
4891 assert(BlendReductionIdx &&
4892 Chain.Blend->getIncomingValue(*BlendReductionIdx) == WidenRecipe &&
4893 "Expected blend to contain the reduction update");
4894 VPValue *BlendCond = Chain.Blend->getMask(*BlendReductionIdx);
4895 Cond = ExitValue ? VPBuilder(WidenRecipe)
4896 .createLogicalAnd(Cond, BlendCond,
4897 WidenRecipe->getDebugLoc())
4898 : BlendCond;
4899 }
4900
4901 // When folding the tail, the inactive lanes of the reduction update are
4902 // computed from values that do not correspond to any scalar iteration
4903 // and must not be accumulated.
4904 if (!Cond)
4906
4907 bool IsLastInChain = RdxPhi->getBackedgeValue() == WidenRecipe ||
4908 RdxPhi->getBackedgeValue() == ExitValue ||
4909 RdxPhi->getBackedgeValue() == Chain.Blend;
4910 assert((!ExitValue || IsLastInChain) &&
4911 "if we found ExitValue, it must match RdxPhi's backedge value");
4912
4913 Type *PhiType = RdxPhi->getScalarType();
4914 RecurKind RdxKind =
4916 auto *PartialRed = new VPReductionRecipe(
4917 RdxKind,
4918 RdxKind == RecurKind::FAdd ? WidenRecipe->getFastMathFlagsOrNone()
4919 : FastMathFlags(),
4920 WidenRecipe->getUnderlyingInstr(), Accumulator, ExtendedOp, Cond,
4921 RdxUnordered{/*VFScaleFactor=*/Chain.ScaleFactor});
4922 PartialRed->insertBefore(WidenRecipe);
4923
4924 if (ExitValue)
4925 ExitValue->replaceAllUsesWith(PartialRed);
4926 if (Chain.Blend)
4927 Chain.Blend->replaceAllUsesWith(PartialRed);
4928 WidenRecipe->replaceAllUsesWith(PartialRed);
4929
4930 // For cost-model purposes, fold this into a VPExpression.
4931 VPExpressionRecipe *E = createPartialReductionExpression(PartialRed);
4932 E->insertBefore(WidenRecipe);
4933 PartialRed->replaceAllUsesWith(E);
4934
4935 // We only need to update the PHI node once, which is when we find the
4936 // last reduction in the chain.
4937 if (!IsLastInChain)
4938 return;
4939
4940 // Scale the PHI and ReductionStartVector by the VFScaleFactor
4941 assert(RdxPhi->getVFScaleFactor() == 1 && "scale factor must not be set");
4942 RdxPhi->setVFScaleFactor(Chain.ScaleFactor);
4943
4944 auto *StartInst = cast<VPInstruction>(RdxPhi->getStartValue());
4945 assert(StartInst->getOpcode() == VPInstruction::ReductionStartVector);
4946 auto *NewScaleFactor = Plan.getConstantInt(32, Chain.ScaleFactor);
4947 StartInst->setOperand(2, NewScaleFactor);
4948
4949 // If this is the last value in a sub-reduction chain, then update the PHI
4950 // node to start at `0` and update the reduction-result to subtract from
4951 // the PHI's start value.
4952 if (Chain.RK != RecurKind::Sub && Chain.RK != RecurKind::FSub)
4953 return;
4954
4955 VPValue *OldStartValue = StartInst->getOperand(0);
4956 StartInst->setOperand(0, StartInst->getOperand(1));
4957
4958 // Replace reduction_result by 'sub (startval, reductionresult)'.
4960 assert(RdxResult && "Could not find reduction result");
4961
4962 VPBuilder Builder = VPBuilder::getToInsertAfter(RdxResult);
4963 unsigned SubOpc = Chain.RK == RecurKind::FSub ? Instruction::BinaryOps::FSub
4964 : Instruction::BinaryOps::Sub;
4965 VPInstruction *NewResult = Builder.createNaryOp(
4966 SubOpc, {OldStartValue, RdxResult}, VPIRFlags::getDefaultFlags(SubOpc),
4967 RdxPhi->getDebugLoc());
4968 RdxResult->replaceUsesWithIf(
4969 NewResult,
4970 [&NewResult](VPUser &U, unsigned Idx) { return &U != NewResult; });
4971}
4972
4973/// Returns the cost of a link in a partial-reduction chain for a given VF.
4974static InstructionCost
4975getPartialReductionLinkCost(VPCostContext &CostCtx,
4976 const VPPartialReductionChain &Link,
4977 ElementCount VF) {
4978 Type *RdxType = Link.ReductionBinOp->getScalarType();
4979 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
4980 std::optional<unsigned> BinOpc = std::nullopt;
4981 // If ExtendB is not none, then the "ExtendsUser" is the binary operation.
4982 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
4983 BinOpc = ExtendedOp.ExtendsUser->getOpcode();
4984
4985 std::optional<llvm::FastMathFlags> Flags;
4986 if (RdxType->isFloatingPointTy())
4987 Flags = Link.ReductionBinOp->getFastMathFlagsOrNone();
4988
4989 auto GetLinkOpcode = [&Link]() -> unsigned {
4990 switch (Link.RK) {
4991 case RecurKind::Sub:
4992 return Instruction::Add;
4993 case RecurKind::FSub:
4994 return Instruction::FAdd;
4995 default:
4996 return Link.ReductionBinOp->getOpcode();
4997 }
4998 };
4999
5000 return CostCtx.TTI.getPartialReductionCost(
5001 GetLinkOpcode(), ExtendedOp.ExtendA.SrcType, ExtendedOp.ExtendB.SrcType,
5002 RdxType, VF, ExtendedOp.ExtendA.Kind, ExtendedOp.ExtendB.Kind, BinOpc,
5003 CostCtx.CostKind, Flags);
5004}
5005
5006static ExtendKind getPartialReductionExtendKind(VPWidenCastRecipe *Cast) {
5008}
5009
5010/// Checks if \p Op (which is an operand of \p UpdateR) is an extended reduction
5011/// operand. This is an operand where the source of the value (e.g. a load) has
5012/// been extended (sext, zext, or fpext) before it is used in the reduction.
5013///
5014/// Possible forms matched by this function:
5015/// - UpdateR(PrevValue, ext(...))
5016/// - UpdateR(PrevValue, mul(ext(...), ext(...)))
5017/// - UpdateR(PrevValue, mul(ext(...), Constant))
5018/// - UpdateR(PrevValue, ext(mul(ext(...), ext(...))))
5019/// - UpdateR(PrevValue, ext(mul(ext(...), Constant)))
5020/// - UpdateR(PrevValue, abs(sub(ext(...), ext(...)))
5021///
5022/// Note: The second operand of UpdateR corresponds to \p Op in the examples.
5023static std::optional<ExtendedReductionOperand>
5024matchExtendedReductionOperand(VPWidenRecipe *UpdateR, VPValue *Op) {
5025 assert(is_contained(UpdateR->operands(), Op) &&
5026 "Op should be operand of UpdateR");
5027
5028 // Try matching an absolute difference operand of the form
5029 // `abs(sub(ext(A), ext(B)))`. This will be later transformed into
5030 // `ext(absolute-difference(A, B))`. This allows us to perform the absolute
5031 // difference on a wider type and get the extend for "free" from the partial
5032 // reduction.
5033 VPValue *X, *Y;
5034 if (Op->hasOneUse() &&
5038 auto *Abs = cast<VPWidenIntrinsicRecipe>(Op);
5039 auto *Sub = cast<VPWidenRecipe>(Abs->getOperand(0));
5040 auto *LHSExt = cast<VPWidenCastRecipe>(Sub->getOperand(0));
5041 auto *RHSExt = cast<VPWidenCastRecipe>(Sub->getOperand(1));
5042 Type *LHSInputType = X->getScalarType();
5043 Type *RHSInputType = Y->getScalarType();
5044 if (LHSInputType != RHSInputType ||
5045 LHSExt->getOpcode() != RHSExt->getOpcode())
5046 return std::nullopt;
5047 // Note: This is essentially the same as matching ext(...) as we will
5048 // rewrite this operand to ext(absolute-difference(A, B)).
5049 return ExtendedReductionOperand{
5050 Sub,
5051 /*ExtendA=*/{LHSInputType, getPartialReductionExtendKind(LHSExt)},
5052 /*ExtendB=*/{}};
5053 }
5054
5055 std::optional<TTI::PartialReductionExtendKind> OuterExtKind;
5057 auto *CastRecipe = cast<VPWidenCastRecipe>(Op);
5058 VPValue *CastSource = CastRecipe->getOperand(0);
5059 OuterExtKind = getPartialReductionExtendKind(CastRecipe);
5060 if (match(CastSource, m_Mul(m_VPValue(), m_VPValue())) ||
5061 match(CastSource, m_FMul(m_VPValue(), m_VPValue()))) {
5062 // Match: ext(mul(...))
5063 // Record the outer extend kind and set `Op` to the mul. We can then match
5064 // this as a binary operation. Note: We can optimize out the outer extend
5065 // by widening the inner extends to match it. See
5066 // optimizeExtendsForPartialReduction.
5067 Op = CastSource;
5068 } else {
5069 return ExtendedReductionOperand{
5070 UpdateR,
5071 /*ExtendA=*/{CastSource->getScalarType(), *OuterExtKind},
5072 /*ExtendB=*/{}};
5073 }
5074 }
5075
5076 if (!Op->hasOneUse())
5077 return std::nullopt;
5078
5080 if (!MulOp ||
5081 !is_contained({Instruction::Mul, Instruction::FMul}, MulOp->getOpcode()))
5082 return std::nullopt;
5083
5084 // The rest of the matching assumes `Op` is a (possibly extended) mul
5085 // operation.
5086
5087 VPValue *LHS = MulOp->getOperand(0);
5088 VPValue *RHS = MulOp->getOperand(1);
5089
5090 // The LHS of the operation must always be an extend.
5092 return std::nullopt;
5093
5094 auto *LHSCast = cast<VPWidenCastRecipe>(LHS);
5095 Type *LHSInputType = LHSCast->getOperand(0)->getScalarType();
5096 ExtendKind LHSExtendKind = getPartialReductionExtendKind(LHSCast);
5097
5098 // The RHS of the operation can be an extend or a constant integer.
5099 const APInt *RHSConst = nullptr;
5100 VPWidenCastRecipe *RHSCast = nullptr;
5102 RHSCast = cast<VPWidenCastRecipe>(RHS);
5103 else if (!match(RHS, m_APInt(RHSConst)) ||
5104 !canConstantBeExtended(RHSConst, LHSInputType, LHSExtendKind))
5105 return std::nullopt;
5106
5107 // The outer extend kind must match the inner extends for folding.
5108 for (VPWidenCastRecipe *Cast : {LHSCast, RHSCast})
5109 if (Cast && OuterExtKind &&
5110 getPartialReductionExtendKind(Cast) != OuterExtKind)
5111 return std::nullopt;
5112
5113 Type *RHSInputType = LHSInputType;
5114 ExtendKind RHSExtendKind = LHSExtendKind;
5115 if (RHSCast) {
5116 RHSInputType = RHSCast->getOperand(0)->getScalarType();
5117 RHSExtendKind = getPartialReductionExtendKind(RHSCast);
5118 }
5119
5120 return ExtendedReductionOperand{
5121 MulOp, {LHSInputType, LHSExtendKind}, {RHSInputType, RHSExtendKind}};
5122}
5123
5124/// Examines each operation in the reduction chain corresponding to \p RedPhiR,
5125/// and determines if the target can use a cheaper operation with a wider
5126/// per-iteration input VF and narrower PHI VF. If successful, returns the chain
5127/// of operations in the reduction.
5128static std::optional<SmallVector<VPPartialReductionChain>>
5129getScaledReductions(VPReductionPHIRecipe *RedPhiR) {
5130 // Get the backedge value from the reduction PHI and find the
5131 // ComputeReductionResult that uses it (directly or through a select for
5132 // predicated reductions).
5133 auto *RdxResult = vputils::findComputeReductionResult(RedPhiR);
5134 if (!RdxResult)
5135 return std::nullopt;
5136 VPValue *ExitValue = RdxResult->getOperand(0);
5137 match(ExitValue, m_Select(m_VPValue(), m_VPValue(ExitValue), m_VPValue()));
5138
5140 RecurKind RK = RedPhiR->getRecurrenceKind();
5141 Type *PhiType = RedPhiR->getScalarType();
5142 TypeSize PHISize = PhiType->getPrimitiveSizeInBits();
5143
5144 // Work backwards from the ExitValue examining each reduction operation.
5145 VPValue *CurrentValue = ExitValue;
5146 while (CurrentValue != RedPhiR) {
5147 VPBlendRecipe *Blend = dyn_cast<VPBlendRecipe>(CurrentValue);
5148 std::optional<unsigned> BlendReductionIdx;
5149 if (Blend) {
5150 assert(!Blend->isNormalized() && "Expect Blend not to be normalized.");
5151 if (Blend->getNumIncomingValues() != 2)
5152 return std::nullopt;
5153
5154 BlendReductionIdx = getBlendReductionUpdateValueIdx(Blend);
5155 if (!BlendReductionIdx)
5156 return std::nullopt;
5157
5158 CurrentValue = Blend->getIncomingValue(*BlendReductionIdx);
5159 }
5160
5161 auto *UpdateR = dyn_cast<VPWidenRecipe>(CurrentValue);
5162 if (!UpdateR || !Instruction::isBinaryOp(UpdateR->getOpcode()))
5163 return std::nullopt;
5164
5165 VPValue *Op = UpdateR->getOperand(1);
5166 VPValue *PrevValue = UpdateR->getOperand(0);
5167
5168 // Find the extended operand. The other operand (PrevValue) is the next link
5169 // in the reduction chain.
5170 std::optional<ExtendedReductionOperand> ExtendedOp =
5171 matchExtendedReductionOperand(UpdateR, Op);
5172 if (!ExtendedOp) {
5173 ExtendedOp = matchExtendedReductionOperand(UpdateR, PrevValue);
5174 if (!ExtendedOp)
5175 return std::nullopt;
5176 std::swap(Op, PrevValue);
5177 }
5178
5179 // Look for VPBlend(reduce(PrevValue, Op), PrevValue), where
5180 // reduce is equal to CurrentValue. This can be lowered as
5181 // a conditional reduction by hoisting the select to the inputs.
5182 if (Blend && Blend->getIncomingValue(1 - *BlendReductionIdx) != PrevValue)
5183 return std::nullopt;
5184
5185 Type *ExtSrcType = ExtendedOp->ExtendA.SrcType;
5186 TypeSize ExtSrcSize = ExtSrcType->getPrimitiveSizeInBits();
5187 if (!PHISize.hasKnownScalarFactor(ExtSrcSize))
5188 return std::nullopt;
5189
5190 VPPartialReductionChain Link(
5191 {UpdateR, *ExtendedOp, RK,
5192 PrevValue == UpdateR->getOperand(0) ? 0U : 1U,
5193 static_cast<unsigned>(PHISize.getKnownScalarFactor(ExtSrcSize)),
5194 Blend});
5195 Chain.push_back(Link);
5196 CurrentValue = PrevValue;
5197 }
5198
5199 // The chain links were collected by traversing backwards from the exit value.
5200 // Reverse the chains so they are in program order.
5201 std::reverse(Chain.begin(), Chain.end());
5202 return Chain;
5203}
5204} // namespace
5205
5207 VPCostContext &CostCtx,
5208 VFRange &Range) {
5209 // Find all possible valid partial reductions, grouping chains by their PHI.
5210 // This grouping allows invalidating the whole chain, if any link is not a
5211 // valid partial reduction.
5213 ChainsByPhi;
5214 VPBasicBlock *HeaderVPBB = Plan.getVectorLoopRegion()->getEntryBasicBlock();
5215 for (VPRecipeBase &R : HeaderVPBB->phis()) {
5216 auto *RedPhiR = dyn_cast<VPReductionPHIRecipe>(&R);
5217 if (!RedPhiR)
5218 continue;
5219
5220 if (auto Chains = getScaledReductions(RedPhiR))
5221 ChainsByPhi.try_emplace(RedPhiR, std::move(*Chains));
5222 }
5223
5224 if (ChainsByPhi.empty())
5225 return;
5226
5227 // Build set of partial reduction operations and blends for user validation
5228 // and a map of reduction bin ops to their scale factors for scale validation.
5229 SmallPtrSet<VPRecipeBase *, 4> PartialReductionOps;
5230 SmallPtrSet<VPBlendRecipe *, 4> PartialReductionBlends;
5231 DenseMap<VPSingleDefRecipe *, unsigned> ScaledReductionMap;
5232 for (const auto &[_, Chains] : ChainsByPhi)
5233 for (const VPPartialReductionChain &Chain : Chains) {
5234 PartialReductionOps.insert(Chain.ExtendedOp.ExtendsUser);
5235 if (Chain.Blend)
5236 PartialReductionBlends.insert(Chain.Blend);
5237 ScaledReductionMap[Chain.ReductionBinOp] = Chain.ScaleFactor;
5238 }
5239
5240 // A partial reduction is invalid if any of its extends are used by
5241 // something that isn't another partial reduction. This is because the
5242 // extends are intended to be lowered along with the reduction itself.
5243 auto ExtendUsersValid = [&](VPValue *Ext) {
5244 return !isa<VPWidenCastRecipe>(Ext) || all_of(Ext->users(), [&](VPUser *U) {
5245 return PartialReductionOps.contains(cast<VPRecipeBase>(U));
5246 });
5247 };
5248
5249 auto IsProfitablePartialReductionChainForVF =
5250 [&](ArrayRef<VPPartialReductionChain> Chain, ElementCount VF) -> bool {
5251 InstructionCost PartialCost = 0, RegularCost = 0;
5252
5253 // The chain is a profitable partial reduction chain if the cost of handling
5254 // the entire chain is cheaper when using partial reductions than when
5255 // handling the entire chain using regular reductions.
5256 for (const VPPartialReductionChain &Link : Chain) {
5257 const ExtendedReductionOperand &ExtendedOp = Link.ExtendedOp;
5258 InstructionCost LinkCost = getPartialReductionLinkCost(CostCtx, Link, VF);
5259 if (!LinkCost.isValid())
5260 return false;
5261
5262 PartialCost += LinkCost;
5263 RegularCost += Link.ReductionBinOp->computeCost(VF, CostCtx);
5264 // If ExtendB is not none, then the "ExtendsUser" is the binary operation.
5265 if (ExtendedOp.ExtendB.Kind != ExtendKind::PR_None)
5266 RegularCost += ExtendedOp.ExtendsUser->computeCost(VF, CostCtx);
5267 for (VPValue *Op : ExtendedOp.ExtendsUser->operands())
5268 if (auto *Extend = dyn_cast<VPWidenCastRecipe>(Op))
5269 RegularCost += Extend->computeCost(VF, CostCtx);
5270 }
5271 return PartialCost.isValid() && PartialCost < RegularCost;
5272 };
5273
5274 // Validate chains: check that extends are only used by partial reductions,
5275 // and that reduction bin ops are only used by other partial reductions with
5276 // matching scale factors, are outside the loop region or the select
5277 // introduced by tail-folding. Otherwise we would create users of scaled
5278 // reductions where the types of the other operands don't match.
5279 for (auto &[RedPhiR, Chains] : ChainsByPhi) {
5280 for (const VPPartialReductionChain &Chain : Chains) {
5281 if (!all_of(Chain.ExtendedOp.ExtendsUser->operands(), ExtendUsersValid)) {
5282 Chains.clear();
5283 break;
5284 }
5285 auto UseIsValid = [&, RedPhiR = RedPhiR](VPUser *U) {
5286 if (auto *PhiR = dyn_cast<VPReductionPHIRecipe>(U))
5287 return PhiR == RedPhiR;
5288 auto *R = cast<VPSingleDefRecipe>(U);
5289
5290 if (auto *Blend = dyn_cast<VPBlendRecipe>(R))
5291 return Blend == Chain.Blend || PartialReductionBlends.contains(Blend);
5292
5293 return Chain.ScaleFactor == ScaledReductionMap.lookup_or(R, 0) ||
5295 m_Specific(Chain.ReductionBinOp))) ||
5296 match(R, m_Select(m_VPValue(), m_Specific(Chain.ReductionBinOp),
5297 m_Specific(RedPhiR)));
5298 };
5299 if (!all_of(Chain.ReductionBinOp->users(), UseIsValid)) {
5300 Chains.clear();
5301 break;
5302 }
5303
5304 // Check if the compute-reduction-result is used by a sunk store.
5305 // TODO: Also form partial reductions in those cases.
5306 if (auto *RdxResult = vputils::findComputeReductionResult(RedPhiR)) {
5307 if (any_of(RdxResult->users(), [](VPUser *U) {
5308 auto *RepR = dyn_cast<VPReplicateRecipe>(U);
5309 return RepR && RepR->getOpcode() == Instruction::Store;
5310 })) {
5311 Chains.clear();
5312 break;
5313 }
5314 }
5315 }
5316
5317 // Clear the chain if it is not profitable.
5319 [&, &Chains = Chains](ElementCount VF) {
5320 return IsProfitablePartialReductionChainForVF(Chains, VF);
5321 },
5322 Range))
5323 Chains.clear();
5324 }
5325
5326 for (auto &[Phi, Chains] : ChainsByPhi)
5327 for (const VPPartialReductionChain &Chain : Chains)
5328 transformToPartialReduction(Chain, Plan, Phi);
5329}
5330
5332 VPRecipeBuilder &RecipeBuilder,
5333 VPCostContext &CostCtx) {
5334 // Collect all loads/stores first. We will start with ones having simpler
5335 // decisions followed by more complex ones that are potentially
5336 // guided/dependent on the simpler ones.
5338 for (VPBasicBlock *VPBB :
5341 for (VPRecipeBase &R : *VPBB) {
5342 auto *VPI = dyn_cast<VPInstruction>(&R);
5343 if (VPI && VPI->getUnderlyingValue() &&
5344 is_contained({Instruction::Load, Instruction::Store},
5345 VPI->getOpcode()))
5346 MemOps.push_back(VPI);
5347 }
5348 }
5349
5350 // Few helpers to process different kinds of memory operations.
5351
5352 // To be used as argument to `VPlanTransforms::runPass` which explicitly
5353 // specified pass name, hence `VPlan &` parameter.
5354 auto ProcessSubset = [&](VPlan &, auto ProcessVPInst) {
5355 SmallVector<VPInstruction *> RemainingMemOps;
5356 for (VPInstruction *VPI : MemOps) {
5357 if (!ProcessVPInst(VPI))
5358 RemainingMemOps.push_back(VPI);
5359 }
5360
5361 MemOps.clear();
5362 std::swap(MemOps, RemainingMemOps);
5363 };
5364
5365 auto ReplaceWith = [&](VPInstruction *VPI, VPRecipeBase *New) {
5366 assert(New->getParent() && "New recipe must have been inserted");
5367 if (VPI->getOpcode() == Instruction::Load)
5368 VPI->replaceAllUsesWith(New->getVPSingleValue());
5369 VPI->eraseFromParent();
5370
5371 // VPI has been processed.
5372 return true;
5373 };
5374
5375 auto Scalarize = [&](VPInstruction *VPI) {
5376 return ReplaceWith(VPI, VPBuilder(VPI).insert(
5377 RecipeBuilder.handleReplication(VPI, Range)));
5378 };
5379
5380 VPBasicBlock *MiddleVPBB = Plan.getMiddleBlock();
5381 VPBuilder FinalRedStoresBuilder(MiddleVPBB, MiddleVPBB->getFirstNonPhi());
5383 "lowerMemoryIdioms", ProcessSubset, Plan, [&](VPInstruction *VPI) {
5384 if (RecipeBuilder.replaceWithFinalIfReductionStore(
5385 VPI, FinalRedStoresBuilder))
5386 return true;
5387
5388 // Filter out scalar VPlan for the remaining idioms.
5390 [](ElementCount VF) { return VF.isScalar(); }, Range))
5391 return false;
5392
5393 if (VPHistogramRecipe *Histogram = RecipeBuilder.widenIfHistogram(VPI))
5394 return ReplaceWith(VPI, VPBuilder(VPI).insert(Histogram));
5395
5396 return false;
5397 });
5398
5399 // Filter out scalar VPlan for the remaining memory operations.
5401 [](ElementCount VF) { return VF.isScalar(); }, Range))
5402 return;
5403
5404 // If the instruction's allocated size doesn't equal it's type size, it
5405 // requires padding and will be scalarized.
5407 "scalarizeMemOpsWithIrregularTypes", ProcessSubset, Plan,
5408 [&](VPInstruction *VPI) {
5410 if (hasIrregularType(getLoadStoreType(I), I->getDataLayout()))
5411 return Scalarize(VPI);
5412
5413 return false;
5414 });
5415
5416 if (!RecipeBuilder.prefersVectorizedAddressing()) {
5418 "makeVPlanMemOpDecision", ProcessSubset, Plan, [&](VPInstruction *VPI) {
5420 bool IsLoad = VPI->getOpcode() == Instruction::Load;
5421 if (RecipeBuilder.isPredicatedInst(I) || !IsLoad ||
5423 return false;
5424
5425 // Scalarize loads used as addresses, matching the legacy CM. The load
5426 // is single-scalar if the pointer is loop-invariant, otherwise it is
5427 // replicated per-lane. No mask is needed as the load is not
5428 // predicated.
5429 VPValue *Ptr = VPI->getOperand(0);
5430 const SCEV *PtrSCEV =
5431 vputils::getSCEVExprForVPValue(Ptr, CostCtx.PSE, CostCtx.L);
5432 bool IsSingleScalarLoad =
5433 !isa<SCEVCouldNotCompute>(PtrSCEV) &&
5434 CostCtx.PSE.getSE()->isLoopInvariant(PtrSCEV, CostCtx.L);
5435
5436 ReplaceWith(VPI,
5437 VPBuilder(VPI).insert(new VPReplicateRecipe(
5438 I, Ptr, /*IsSingleScalar=*/IsSingleScalarLoad,
5439 /*Mask=*/nullptr, *VPI, *VPI, VPI->getDebugLoc())));
5440 return true;
5441 });
5442 }
5443
5444 // Widen unit-stride consecutive accesses, matching the legacy CM. Both
5445 // forward (stride +1) and reverse (stride -1) accesses are handled.
5447 "widenConsecutiveMemOps", ProcessSubset, Plan, [&](VPInstruction *VPI) {
5449 bool IsLoad = VPI->getOpcode() == Instruction::Load;
5450 VPValue *Ptr = VPI->getOperand(!IsLoad);
5451 Type *ScalarTy =
5452 IsLoad ? VPI->getScalarType() : VPI->getOperand(0)->getScalarType();
5453 std::optional<int64_t> Stride =
5454 getConstantStride(Ptr, ScalarTy, CostCtx.PSE, CostCtx.L);
5455 if (Stride != 1 && Stride != -1)
5456 return false;
5457 bool Reverse = Stride == -1;
5458
5459 // A predicated access can only be widened (rather than scalarized) if
5460 // the target supports a masked load/store for it.
5461 // TODO: Determine if a load/store needs predication directly in VPlan.
5462 bool IsPredicated = RecipeBuilder.isPredicatedInst(I);
5463 if (IsPredicated && !CostCtx.Config.isLegalMaskedLoadOrStore(
5464 IsLoad, ScalarTy, getLoadStoreAlignment(I),
5466 return false;
5467
5468 VPBuilder Builder(VPI);
5469 VPSingleDefRecipe *VectorPtr = Builder.createConsecutiveVectorPointer(
5470 Ptr, ScalarTy, Reverse, VPI->getDebugLoc());
5471
5472 VPValue *Mask = IsPredicated ? VPI->getMask() : nullptr;
5473 // Reverse the mask so it matches the reversed access order.
5474 if (Reverse && Mask)
5475 Mask = Builder.createNaryOp(VPInstruction::Reverse, Mask,
5476 VPI->getDebugLoc());
5477
5478 if (IsLoad) {
5479 VPSingleDefRecipe *Load = Builder.createWidenLoad(
5480 *cast<LoadInst>(I), VectorPtr, Mask,
5481 /*Consecutive=*/true, *VPI, VPI->getDebugLoc());
5482 // Reverse the loaded values back into program order.
5483 if (Reverse)
5484 Load = Builder.createNaryOp(VPInstruction::Reverse, Load,
5485 VPI->getDebugLoc());
5486 return ReplaceWith(VPI, Load);
5487 }
5488
5489 VPValue *StoredVal = VPI->getOperand(0);
5490 if (Reverse)
5491 // Reverse the stored values so they are written in descending order.
5492 StoredVal = Builder.createNaryOp(VPInstruction::Reverse, StoredVal,
5493 VPI->getDebugLoc());
5494
5495 auto *StoreR = Builder.createWidenStore(
5496 *cast<StoreInst>(I), VectorPtr, StoredVal, Mask,
5497 /*Consecutive=*/true, *VPI, VPI->getDebugLoc());
5498 return ReplaceWith(VPI, StoreR);
5499 });
5500
5501 VPlanTransforms::runPass("delegateMemOpWideningToLegacyCM", ProcessSubset,
5502 Plan, [&](VPInstruction *VPI) {
5503 if (VPRecipeBase *Recipe =
5504 RecipeBuilder.tryToWidenMemory(VPI, Range))
5505 return ReplaceWith(VPI, Recipe);
5506
5507 return Scalarize(VPI);
5508 });
5509}
5510
5513 [&](ElementCount VF) { return VF.isScalar(); }, Range))
5514 return;
5515
5517 Plan.getEntry());
5519 for (VPRecipeBase &R : make_early_inc_range(reverse(*VPBB))) {
5520 auto *VPI = dyn_cast<VPInstruction>(&R);
5521 if (!VPI)
5522 continue;
5523
5524 auto *I = cast_or_null<Instruction>(VPI->getUnderlyingValue());
5525 // Wouldn't be able to create a `VPReplicateRecipe` anyway.
5526 if (!I)
5527 continue;
5528
5529 // If executing other lanes produces side-effects we can't avoid them.
5530 if (VPI->mayHaveSideEffects())
5531 continue;
5532
5533 // We want to drop the mask operand, verify we can safely do that.
5534 if (VPI->isMasked() && !VPI->isSafeToSpeculativelyExecute())
5535 continue;
5536
5537 // Avoid rewriting IV increment as that interferes with
5538 // `removeRedundantCanonicalIVs`.
5539 if (VPI->getOpcode() == Instruction::Add &&
5541 continue;
5542
5543 // Other lanes are needed - can't drop them.
5545 continue;
5546
5547 auto *Recipe = VPBuilder::createSingleScalarOp(
5548 VPI->getOpcode(), VPI->operandsWithoutMask(), /*Mask=*/nullptr, *VPI,
5549 *VPI, VPI->getDebugLoc(), I);
5550 Recipe->insertBefore(VPI);
5551 VPI->replaceAllUsesWith(Recipe);
5552 VPI->eraseFromParent();
5553 }
5554 }
5555}
5556
5557/// Returns true if \p Info's parameter kinds are compatible with \p Args.
5558static bool areVFParamsOk(const VFInfo &Info, ArrayRef<VPValue *> Args,
5559 PredicatedScalarEvolution &PSE, const Loop *L) {
5560 ScalarEvolution *SE = PSE.getSE();
5561 return all_of(Info.Shape.Parameters, [&](VFParameter Param) {
5562 switch (Param.ParamKind) {
5563 case VFParamKind::Vector:
5564 case VFParamKind::GlobalPredicate:
5565 return true;
5566 case VFParamKind::OMP_Uniform:
5567 return SE->isSCEVable(Args[Param.ParamPos]->getScalarType()) &&
5568 SE->isLoopInvariant(
5569 vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5570 L);
5571 case VFParamKind::OMP_Linear:
5572 return match(vputils::getSCEVExprForVPValue(Args[Param.ParamPos], PSE, L),
5573 m_scev_AffineAddRec(
5574 m_SCEV(), m_scev_SpecificSInt(Param.LinearStepOrPos),
5575 m_SpecificLoop(L)));
5576 default:
5577 return false;
5578 }
5579 });
5580}
5581
5582/// Find a vector variant of \p CI for \p VF, respecting \p MaskRequired.
5583/// Returns the variant function, or nullptr. Masked variants are assumed to
5584/// take the mask as a trailing parameter.
5586 ElementCount VF, bool MaskRequired,
5588 const Loop *L) {
5589 if (CI->isNoBuiltin())
5590 return nullptr;
5591 auto Mappings = VFDatabase::getMappings(*CI);
5592 const auto *It = find_if(Mappings, [&](const VFInfo &Info) {
5593 return Info.Shape.VF == VF && (!MaskRequired || Info.isMasked()) &&
5594 areVFParamsOk(Info, Args, PSE, L);
5595 });
5596 if (It == Mappings.end())
5597 return nullptr;
5598 return CI->getModule()->getFunction(It->VectorName);
5599}
5600
5601namespace {
5602/// The outcome of choosing how to widen a call at a given VF.
5603struct CallWideningDecision {
5604 enum class KindTy { Scalarize, Intrinsic, VectorVariant };
5605 CallWideningDecision(KindTy Kind, Function *Variant = nullptr)
5606 : Kind(Kind), Variant(Variant) {}
5607 KindTy Kind;
5608
5609 /// Set when Kind == VectorVariant.
5611
5612 bool operator==(const CallWideningDecision &Other) const {
5613 return Kind == Other.Kind && Variant == Other.Variant;
5614 }
5615};
5616} // namespace
5617
5618/// Pick the cheapest widening for the call \p VPI at \p VF among scalarization,
5619/// vector intrinsic, and vector library variant.
5620static CallWideningDecision decideCallWidening(VPInstruction &VPI,
5622 ElementCount VF,
5623 VPCostContext &CostCtx) {
5624 auto *CI = cast<CallInst>(VPI.getUnderlyingInstr());
5625
5626 // Scalar VFs and calls forced or known to scalarize always replicate.
5627 if (VF.isScalar() || CostCtx.willBeScalarized(CI, VF))
5628 return CallWideningDecision::KindTy::Scalarize;
5629
5630 auto *CalledFn = cast<Function>(
5632 Type *ResultTy = VPI.getScalarType();
5634 bool MaskRequired = CostCtx.isMaskRequired(CI);
5635
5636 // Pseudo intrinsics (assume, lifetime, ...) are always scalarized.
5638 return CallWideningDecision::KindTy::Scalarize;
5639
5640 InstructionCost ScalarCost =
5641 VPReplicateRecipe::computeCallCost(CalledFn, ResultTy, Ops,
5642 /*IsSingleScalar=*/false, VF, CostCtx);
5643
5644 Function *VecFunc =
5645 findVectorVariant(CI, Ops, VF, MaskRequired, CostCtx.PSE, CostCtx.L);
5647 if (VecFunc)
5648 VecCallCost = VPWidenCallRecipe::computeCallCost(VecFunc, CostCtx);
5649
5650 // Prefer the intrinsic if it is at least as cheap as scalarizing and any
5651 // available vector variant.
5652 if (ID) {
5654 VPWidenIntrinsicRecipe::computeCallCost(ID, Ops, VPI, VF, CostCtx);
5655 if (IntrinsicCost.isValid() && ScalarCost >= IntrinsicCost &&
5656 (!VecFunc || VecCallCost >= IntrinsicCost))
5657 return CallWideningDecision::KindTy::Intrinsic;
5658 }
5659
5660 // Otherwise, use a vector library variant when it beats scalarizing.
5661 if (VecFunc && ScalarCost >= VecCallCost)
5662 return {CallWideningDecision::KindTy::VectorVariant, VecFunc};
5663
5664 return CallWideningDecision::KindTy::Scalarize;
5665}
5666
5668 VPRecipeBuilder &RecipeBuilder,
5669 VPCostContext &CostCtx) {
5672 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
5673 auto *VPI = dyn_cast<VPInstruction>(&R);
5674 if (!VPI || !VPI->getUnderlyingValue() ||
5675 VPI->getOpcode() != Instruction::Call)
5676 continue;
5677
5678 auto *CI = cast<CallInst>(VPI->getUnderlyingInstr());
5679 SmallVector<VPValue *, 4> Ops(VPI->op_begin(),
5680 VPI->op_begin() + CI->arg_size());
5681
5682 CallWideningDecision Decision =
5683 decideCallWidening(*VPI, Ops, Range.Start, CostCtx);
5685 [&](ElementCount VF) {
5686 return Decision == decideCallWidening(*VPI, Ops, VF, CostCtx);
5687 },
5688 Range);
5689
5690 VPSingleDefRecipe *Replacement = nullptr;
5691 switch (Decision.Kind) {
5692 case CallWideningDecision::KindTy::Intrinsic: {
5694 Type *ResultTy = VPI->getScalarType();
5695 Replacement = new VPWidenIntrinsicRecipe(*CI, ID, Ops, ResultTy, *VPI,
5696 *VPI, VPI->getDebugLoc());
5697 break;
5698 }
5699 case CallWideningDecision::KindTy::VectorVariant: {
5700 // Masked variants take the mask as a trailing parameter, so they have
5701 // one more parameter than the original call's arguments.
5702 if (Decision.Variant->arg_size() > Ops.size()) {
5703 VPValue *Mask = VPI->isMasked() ? VPI->getMask() : Plan.getTrue();
5704 Ops.push_back(Mask);
5705 }
5706 Ops.push_back(VPI->getOperand(VPI->getNumOperandsWithoutMask() - 1));
5707 Replacement = new VPWidenCallRecipe(CI, Decision.Variant, Ops, *VPI,
5708 *VPI, VPI->getDebugLoc());
5709 break;
5710 }
5711 case CallWideningDecision::KindTy::Scalarize:
5712 Replacement = RecipeBuilder.handleReplication(VPI, Range);
5713 break;
5714 }
5715
5716 Replacement->insertBefore(VPI);
5717 VPI->replaceAllUsesWith(Replacement);
5718 VPI->eraseFromParent();
5719 }
5720 }
5721}
5722
5725 Loop &L, VPCostContext &Ctx,
5726 VFRange &Range) {
5727 if (Plan.hasScalarVFOnly())
5728 return;
5729
5730 VPRegionBlock *VectorLoop = Plan.getVectorLoopRegion();
5731 VPValue *I32VF = nullptr;
5733 vp_depth_first_shallow(VectorLoop->getEntry()))) {
5734 for (VPRecipeBase &R : make_early_inc_range(*VPBB)) {
5735 auto *MemR = dyn_cast<VPWidenMemoryRecipe>(&R);
5736 // TODO: Transform reverse access into strided access with -1 stride.
5737 // TODO: Transform gather/scatter with uniform address into strided access
5738 // with 0 stride.
5739 // TODO: Transform interleave access into multiple strided accesses.
5740 if (!MemR || MemR->isConsecutive())
5741 continue;
5742
5743 VPValue *Ptr = MemR->getAddr();
5744 // Check if this is a strided access by analyzing the address SCEV for an
5745 // affine addRec.
5746 const SCEV *PtrSCEV = vputils::getSCEVExprForVPValue(Ptr, PSE, &L);
5747 const SCEV *Start;
5748 const SCEVConstant *Step;
5749 // TODO: Support non-constant loop invariant stride.
5750 if (!match(PtrSCEV,
5752 m_SpecificLoop(&L))))
5753 continue;
5754
5755 VPValue *StoredValue = nullptr;
5756 Type *DataTy;
5757 Intrinsic::ID IntrinID;
5758 if (auto *StoreR = dyn_cast<VPWidenStoreRecipe>(&R)) {
5759 StoredValue = StoreR->getStoredValue();
5760 DataTy = StoredValue->getScalarType();
5761 IntrinID = Intrinsic::experimental_vp_strided_store;
5762 } else {
5763 auto *LoadR = cast<VPWidenLoadRecipe>(&R);
5764 DataTy = LoadR->getScalarType();
5765 IntrinID = Intrinsic::experimental_vp_strided_load;
5766 }
5767
5768 Align Alignment = MemR->getAlign();
5769 auto IsProfitable = [&](ElementCount VF) {
5770 Type *VectorTy = toVectorTy(DataTy, VF);
5771 if (!Ctx.TTI.isLegalStridedLoadStore(VectorTy, Alignment))
5772 return false;
5773 const InstructionCost CurrentCost = MemR->computeCost(VF, Ctx);
5774 const InstructionCost StridedLoadStoreCost =
5776 IntrinID, VectorTy, MemR->isMasked(), Alignment, Ctx);
5777 return StridedLoadStoreCost < CurrentCost;
5778 };
5779
5781 Range))
5782 continue;
5783
5784 // Invalidate the legacy widening decision so the cost of replaced load is
5785 // not counted during precomputeCosts.
5786 // TODO: Remove once the legacy exit cost computation is retired.
5787 for (ElementCount VF : Range)
5788 Ctx.invalidateWideningDecision(&MemR->getIngredient(), VF);
5789
5790 // Get VF as i32 for the vector length operand.
5791 if (!I32VF) {
5792 VPBuilder Builder(Plan.getVectorPreheader());
5793 I32VF = Builder.createScalarZExtOrTrunc(
5794 &Plan.getVF(), Type::getInt32Ty(Plan.getContext()),
5796 }
5797
5798 VPBuilder Builder(&R);
5799 // Create the base pointer of strided access.
5800 // TODO: reuse VPDerivedIVRecipe for base pointer computation when it
5801 // supports a general VPValue as the start value.
5802 VPValue *StartVPV =
5803 VPSCEVExpander(Builder, *PSE.getSE(), R.getDebugLoc()).expand(Start);
5804 VPValue *StrideInBytes = Plan.getOrAddLiveIn(Step->getValue());
5805 Type *IndexTy = Plan.getDataLayout().getIndexType(Ptr->getScalarType());
5806 assert(IndexTy == StrideInBytes->getScalarType() &&
5807 "Stride type from SCEV must match the index type");
5808 VPValue *CanIV = Builder.createScalarZExtOrTrunc(
5809 VectorLoop->getCanonicalIV(), IndexTy, DebugLoc::getUnknown());
5810 auto *AddRecPtr = cast<SCEVAddRecExpr>(PtrSCEV);
5811 auto *Offset = Builder.createOverflowingOp(
5812 Instruction::Mul, {CanIV, StrideInBytes},
5813 {AddRecPtr->hasNoUnsignedWrap(), /*HasNSW=*/false});
5814 GEPNoWrapFlags NWFlags = AddRecPtr->hasNoUnsignedWrap()
5817 VPValue *BasePtr = Builder.createNoWrapPtrAdd(StartVPV, Offset, NWFlags);
5818
5819 // Create a new vector pointer for strided access.
5820 VPValue *NewPtr = Builder.createVectorPointer(
5821 BasePtr, Type::getInt8Ty(Plan.getContext()), StrideInBytes, NWFlags,
5822 R.getDebugLoc());
5823
5824 VPValue *Mask = MemR->getMask();
5825 if (!Mask)
5826 Mask = Plan.getTrue();
5828 if (StoredValue)
5829 Ops.push_back(StoredValue);
5830 Ops.append({NewPtr, StrideInBytes, Mask, I32VF});
5831
5832 auto *StridedR = Builder.createWidenMemIntrinsic(
5833 IntrinID, Ops,
5834 StoredValue ? Type::getVoidTy(Plan.getContext()) : DataTy, Alignment,
5835 *MemR, R.getDebugLoc());
5836 if (!StoredValue)
5837 cast<VPWidenLoadRecipe>(&R)->replaceAllUsesWith(StridedR);
5838 R.eraseFromParent();
5839 }
5840 }
5841}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static bool isEqual(const Function &Caller, const Function &Callee)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static cl::opt< OutputCostKind > CostKind("cost-kind", cl::desc("Target cost kind"), cl::init(OutputCostKind::RecipThroughput), cl::values(clEnumValN(OutputCostKind::RecipThroughput, "throughput", "Reciprocal throughput"), clEnumValN(OutputCostKind::Latency, "latency", "Instruction latency"), clEnumValN(OutputCostKind::CodeSize, "code-size", "Code size"), clEnumValN(OutputCostKind::SizeAndLatency, "size-latency", "Code size and latency"), clEnumValN(OutputCostKind::All, "all", "Print all cost kinds")))
static cl::opt< IntrinsicCostStrategy > IntrinsicCost("intrinsic-cost-strategy", cl::desc("Costing strategy for intrinsic instructions"), cl::init(IntrinsicCostStrategy::InstructionCost), cl::values(clEnumValN(IntrinsicCostStrategy::InstructionCost, "instruction-cost", "Use TargetTransformInfo::getInstructionCost"), clEnumValN(IntrinsicCostStrategy::IntrinsicCost, "intrinsic-cost", "Use TargetTransformInfo::getIntrinsicInstrCost"), clEnumValN(IntrinsicCostStrategy::TypeBasedIntrinsicCost, "type-based-intrinsic-cost", "Calculate the intrinsic cost based only on argument types")))
@ Default
Hexagon Common GEP
#define _
iv Induction Variable Users
Definition IVUsers.cpp:48
iv users
Definition IVUsers.cpp:48
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
licm
Definition LICM.cpp:391
Legalize the Machine IR a function s Machine IR
Definition Legalizer.cpp:85
#define I(x, y, z)
Definition MD5.cpp:57
This file provides utility analysis objects describing memory locations.
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
#define P(N)
This file builds on the ADT/GraphTraits.h file to build a generic graph post order iterator.
const SmallVectorImpl< MachineOperand > & Cond
Func MI getDebugLoc()))
This file contains some templates that are useful if you are working with the STL at all.
This is the interface for a metadata-based scoped no-alias analysis.
This file implements a set that has insertion order iteration characteristics.
This file defines the SmallPtrSet class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
This file implements the TypeSwitch template, which mimics a switch() statement whose cases are type ...
This file implements dominator tree analysis for a single level of a VPlan's H-CFG.
This file contains the declarations of different VPlan-related auxiliary helpers.
static SmallVector< SmallVector< VPReplicateRecipe *, 4 > > collectComplementaryPredicatedMemOps(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
static void removeCommonBlendMask(VPBlendRecipe *Blend)
Try to see if all of Blend's masks share a common value logically and'ed and remove it from the masks...
static void tryToCreateAbstractReductionRecipe(VPReductionRecipe *Red, VPCostContext &Ctx, VFRange &Range)
This function tries to create abstract recipes from the reduction recipe for following optimizations ...
static VPReplicateRecipe * findRecipeWithMinAlign(ArrayRef< VPReplicateRecipe * > Group)
static bool handleUncountableExitsWithSideEffects(VPlan &Plan, SmallVectorImpl< EarlyExitInfo > &Exits, VPBasicBlock *HeaderVPBB, VPBasicBlock *LatchVPBB, VPBasicBlock *MiddleVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC)
Update Plan to mask memory operations in the loop based on whether the early exit is taken or not.
static CallWideningDecision decideCallWidening(VPInstruction &VPI, ArrayRef< VPValue * > Ops, ElementCount VF, VPCostContext &CostCtx)
Pick the cheapest widening for the call VPI at VF among scalarization, vector intrinsic,...
static bool areVFParamsOk(const VFInfo &Info, ArrayRef< VPValue * > Args, PredicatedScalarEvolution &PSE, const Loop *L)
Returns true if Info's parameter kinds are compatible with Args.
static std::optional< VPValue * > getRecipesForUncountableExit(SmallVectorImpl< VPInstruction * > &Recipes, VPBasicBlock *LatchVPBB)
Returns the VPValue representing the uncountable exit comparison used by AnyOf if the recipes it depe...
static bool sinkScalarOperands(VPlan &Plan)
static std::optional< int64_t > getConstantStride(VPValue *Addr, Type *AccessTy, PredicatedScalarEvolution &PSE, const Loop *L)
If the pointer operand Addr of a memory access is an affine AddRec w.r.t.
static bool simplifyBranchConditionForVFAndUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Try to simplify the branch condition of Plan.
static VPValue * cloneBinOpForScalarIV(VPWidenRecipe *BinOp, VPValue *ScalarIV, VPWidenIntOrFpInductionRecipe *WidenIV)
Create a scalar version of BinOp, with its WidenIV operand replaced by ScalarIV, and place it after S...
static VPWidenIntOrFpInductionRecipe * getExpressionIV(VPValue *V)
Check if V is a binary expression of a widened IV and a loop-invariant value.
static void removeRedundantInductionCasts(VPlan &Plan)
Remove redundant casts of inductions.
static bool isConditionTrueViaVFAndUF(VPValue *Cond, VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Return true if Cond is known to be true for given BestVF and BestUF.
static VPExpressionRecipe * tryToMatchAndCreateExtendedReduction(VPReductionRecipe *Red, VPCostContext &Ctx, VFRange &Range)
This function tries convert extended in-loop reductions to VPExpressionRecipe and clamp the Range if ...
static std::optional< ElementCount > isConsecutiveInterleaveGroup(VPInterleaveRecipe *InterleaveR, ArrayRef< ElementCount > VFs, const TargetTransformInfo &TTI)
Returns VF from VFs if IR is a full interleave group with factor and number of members both equal to ...
static Type * getLoadStoreValueType(VPReplicateRecipe *R, bool IsLoad)
Get the value type of the replicate load or store.
static VPIRMetadata getCommonMetadata(ArrayRef< VPReplicateRecipe * > Recipes)
static VPValue * simplifyLogicalRecipe(VPSingleDefRecipe *Def, VPBuilder &Builder, bool CanCreateNewRecipe)
Try to simplify logical and bitwise recipes in Def.
static bool mergeReplicateRegionsIntoSuccessors(VPlan &Plan)
static Function * findVectorVariant(CallInst *CI, ArrayRef< VPValue * > Args, ElementCount VF, bool MaskRequired, PredicatedScalarEvolution &PSE, const Loop *L)
Find a vector variant of CI for VF, respecting MaskRequired.
static VPWidenInductionRecipe * getOptimizableIVOf(VPValue *VPV, PredicatedScalarEvolution &PSE)
Check if VPV is an untruncated wide induction, either before or after the increment.
static bool canNarrowLoad(VPSingleDefRecipe *WideMember0, unsigned OpIdx, VPValue *OpV, unsigned Idx, bool IsScalable)
Returns true if V is VPWidenLoadRecipe or VPInterleaveRecipe that can be converted to a narrower reci...
static void legalizeAndOptimizeInductions(VPlan &Plan)
Legalize VPWidenPointerInductionRecipe, by replacing it with a PtrAdd (IndStart, ScalarIVSteps (0,...
static void addReplicateRegions(VPlan &Plan)
static VPValue * optimizeLatchExitIVUserViaSCEV(VPlan &Plan, VPValue *Op, PredicatedScalarEvolution &PSE, VPValue *ResumeTC, const Loop *L)
static SmallVector< SmallVector< VPReplicateRecipe *, 4 > > collectGroupedReplicateMemOps(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L, function_ref< bool(VPReplicateRecipe *)> FilterFn)
Collect either replicated Loads or Stores grouped by their address SCEV and their load-store type,...
static VPValue * tryToComputeEndValueForInduction(VPWidenInductionRecipe *WideIV, VPBuilder &VectorPHBuilder, VPValue *VectorTC)
Compute the end value for WideIV, unless it is truncated.
static bool replaceMaskWithCompareForScalarPlan(VPlan &Plan, ElementCount BestVF)
static void removeRedundantExpandSCEVRecipes(VPlan &Plan)
Remove redundant ExpandSCEVRecipes in Plan's entry block by replacing them with already existing reci...
static VPValue * optimizeEarlyExitInductionUser(VPlan &Plan, VPValue *Op, PredicatedScalarEvolution &PSE)
Attempts to optimize the induction variable exit values for users in the early exit block.
static VPValue * narrowInterleaveGroupOp(ArrayRef< VPValue * > Members, SmallPtrSetImpl< VPValue * > &NarrowedOps, VPBasicBlock *Preheader)
static VPValue * simplifyRecipe(VPSingleDefRecipe *Def)
Try to simplify VPSingleDefRecipe Def.
static VPValue * optimizeLatchExitInductionUser(VPlan &Plan, VPValue *Op, DenseMap< VPValue *, VPValue * > &EndValues, PredicatedScalarEvolution &PSE)
Attempts to optimize the induction variable exit values for users in the exit block coming from the l...
static void reassociateHeaderMask(VPlan &Plan)
Reassociate (headermask && x) && y -> headermask && (x && y) to allow the header mask to be simplifie...
static VPBasicBlock * getPredicatedThenBlock(VPRegionBlock *R)
If R is a triangle region, return the 'then' block of the triangle.
static bool canHoistOrSinkWithNoAliasCheck(const MemoryLocation &MemLoc, VPBasicBlock *FirstBB, VPBasicBlock *LastBB, std::optional< SinkStoreInfo > SinkInfo={})
Check if a memory operation doesn't alias with memory operations using scoped noalias metadata,...
static VPRegionBlock * createReplicateRegion(VPReplicateRecipe *PredRecipe, VPRegionBlock *ParentRegion, VPlan &Plan)
static void simplifyBlends(VPlan &Plan)
Normalize and simplify VPBlendRecipes.
static bool cannotHoistOrSinkRecipe(VPRecipeBase &R, VPBasicBlock *FirstBB, VPBasicBlock *LastBB, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink a non-memory or memory recipe R out...
static std::optional< Instruction::BinaryOps > getUnmaskedDivRemOpcode(Intrinsic::ID ID)
static bool isAlreadyNarrow(VPValue *VPV)
Returns true if VPValue is a narrow VPValue.
static bool canNarrowOps(ArrayRef< VPValue * > Ops, bool IsScalable)
static bool optimizeVectorInductionWidthForTCAndVFUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF)
Optimize the width of vector induction variables in Plan based on a known constant Trip Count,...
static VPExpressionRecipe * tryToMatchAndCreateMulAccumulateReduction(VPReductionRecipe *Red, VPCostContext &Ctx, VFRange &Range)
This function tries convert extended in-loop reductions to VPExpressionRecipe and clamp the Range if ...
static bool canSinkStoreWithNoAliasCheck(ArrayRef< VPReplicateRecipe * > StoresToSink, PredicatedScalarEvolution &PSE, const Loop &L)
static std::optional< bool > getStepDirection(const SCEV *S, ScalarEvolution &SE)
If S is an affine AddRec, returns true if its step is known to be positive and false if it is known t...
static void narrowToSingleScalarRecipes(VPlan &Plan)
This file provides utility VPlan to VPlan transformations.
#define RUN_VPLAN_PASS(PASS,...)
This file contains the declarations of the Vectorization Plan base classes:
static const X86InstrFMA3Group Groups[]
Value * RHS
Value * LHS
BinaryOperator * Mul
static const uint32_t IV[8]
Definition blake3_impl.h:83
Helper for extra no-alias checks via known-safe recipe and SCEV.
SinkStoreInfo(ArrayRef< VPReplicateRecipe * > ExcludeRecipes, VPReplicateRecipe &GroupLeader, PredicatedScalarEvolution &PSE, const Loop &L)
SinkStoreInfo(VPReplicateRecipe &GroupLeader)
bool shouldSkip(VPRecipeBase &R) const
Return true if R should be skipped during alias checking, either because it's in the exclude set or b...
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1077
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1533
APInt abs() const
Get the absolute value.
Definition APInt.h:1816
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1509
int32_t exactLogBase2() const
Definition APInt.h:1804
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:331
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
Definition APInt.cpp:1029
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
Definition APInt.h:437
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1226
An arbitrary precision integer that knows its signedness.
Definition APSInt.h:24
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
@ NoAlias
The two locations do not alias at all.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
const T & back() const
Get the last element.
Definition ArrayRef.h:150
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
const T & front() const
Get the first element.
Definition ArrayRef.h:144
A cache of @llvm.assume calls within a function.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
bool isNoBuiltin() const
Return true if the call should not be treated as a call to a builtin.
This class represents a function call, abstracting a target machine's calling convention.
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
Definition InstrTypes.h:852
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
This class represents a range of values.
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
LLVM_ABI IntegerType * getIndexType(LLVMContext &C, unsigned AddressSpace) const
Returns the type of a GEP index in AddressSpace.
A debug info location.
Definition DebugLoc.h:126
static DebugLoc getUnknown()
Definition DebugLoc.h:153
ValueT lookup(const_arg_type_t< KeyT > Val) const
Return the entry for the specified key, or a default constructed value if no such entry exists.
Definition DenseMap.h:250
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:299
ValueT lookup_or(const_arg_type_t< KeyT > Val, U &&Default) const
Definition DenseMap.h:260
bool dominates(const DomTreeNodeBase< NodeT > *A, const DomTreeNodeBase< NodeT > *B) const
dominates - Returns true iff A dominates B.
Concrete subclass of DominatorTreeBase that is used to compute a normal dominator tree.
Definition Dominators.h:122
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:312
constexpr bool isScalar() const
Exactly one element.
Definition TypeSize.h:320
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
size_t arg_size() const
Definition Function.h:885
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedWrap() const
GEPNoWrapFlags withoutNoUnsignedWrap() const
static GEPNoWrapFlags none()
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
A struct for saving information about induction variables.
InductionKind
This enum represents the kinds of inductions that we support.
@ IK_PtrInduction
Pointer induction var. Step = C.
@ IK_IntInduction
Integer induction variable. Step = C.
static InstructionCost getInvalid(CostType Val=0)
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
bool isBinaryOp() const
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
bool isIntDivRem() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
The group of interleaved loads/stores sharing the same stride and close to each other.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
static bool getDecisionAndClampRange(const std::function< bool(ElementCount)> &Predicate, VFRange &Range)
Test a Predicate on a Range of VF's.
Definition VPlan.cpp:1681
Represents a single loop in the control flow graph.
Definition LoopInfo.h:40
This class implements a map that also provides access to all stored values in a deterministic order.
Definition MapVector.h:38
ValueT lookup(const KeyT &Key) const
Definition MapVector.h:110
std::pair< iterator, bool > try_emplace(const KeyT &Key, Ts &&...Args)
Definition MapVector.h:118
bool empty() const
Definition MapVector.h:79
Representation for a specific memory location.
Function * getFunction(StringRef Name) const
Look up the specified function in the module symbol table.
Definition Module.cpp:235
Post-order traversal of a graph.
An interface layer with SCEV used to manage how we see SCEV expressions for values in the context of ...
ScalarEvolution * getSE() const
Returns the ScalarEvolution analysis used.
LLVM_ABI const SCEV * getSCEV(Value *V)
Returns the SCEV expression of V, in the context of the current SCEV predicate.
static LLVM_ABI unsigned getOpcode(RecurKind Kind)
Returns the opcode corresponding to the RecurrenceKind.
static bool isFindLastRecurrenceKind(RecurKind Kind)
Returns true if the recurrence kind is of the form select(cmp(),x,y) where one of (x,...
RegionT * getParent() const
Get the parent of the Region.
Definition RegionInfo.h:362
This class represents a constant integer value.
ConstantInt * getValue() const
static const SCEV * rewrite(const SCEV *Scev, ScalarEvolution &SE, ValueToSCEVMapTy &Map)
This means that we are dealing with an entirely unknown SCEV value, and only represent it as its LLVM...
This class represents an analyzed expression in the program.
Type * getType() const
Return the LLVM type of this SCEV expression.
The main scalar evolution driver.
const DataLayout & getDataLayout() const
Return the DataLayout associated with the module this SCEV instance is operating on.
LLVM_ABI const SCEV * getNegativeSCEV(const SCEV *V, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
Return the SCEV object corresponding to -V.
LLVM_ABI bool isKnownNegative(const SCEV *S)
Test if the given expression is known to be negative.
LLVM_ABI const SCEV * getConstant(ConstantInt *V)
LLVM_ABI const SCEV * getMinusSCEV(SCEVUse LHS, SCEVUse RHS, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap, unsigned Depth=0)
Return LHS-RHS.
ConstantRange getSignedRange(const SCEV *S)
Determine the signed range for a particular SCEV.
LLVM_ABI bool isLoopInvariant(const SCEV *S, const Loop *L)
Return true if the value of the given SCEV is unchanging in the specified loop.
LLVM_ABI bool isKnownPositive(const SCEV *S)
Test if the given expression is known to be positive.
LLVM_ABI const SCEV * getElementCount(Type *Ty, ElementCount EC, SCEV::NoWrapFlags Flags=SCEV::FlagAnyWrap)
ConstantRange getUnsignedRange(const SCEV *S)
Determine the unsigned range for a particular SCEV.
LLVM_ABI bool isKnownPredicate(CmpPredicate Pred, SCEVUse LHS, SCEVUse RHS)
Test if the given expression is known to satisfy the condition described by Pred, LHS,...
static LLVM_ABI AliasResult alias(const MemoryLocation &LocA, const MemoryLocation &LocB)
A vector that has set insertion semantics.
Definition SetVector.h:57
size_type size() const
Determine the number of elements in the SetVector.
Definition SetVector.h:103
bool insert(const value_type &X)
Insert a new element into the SetVector.
Definition SetVector.h:157
size_type size() const
A templated base class for SmallPtrSet which provides the typesafe interface that is common across al...
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
iterator begin() const
bool contains(ConstPtrType Ptr) const
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Provides information about what library functions are available for the current target.
This pass provides access to the codegen interfaces that are needed for IR-level transformations.
static LLVM_ABI PartialReductionExtendKind getPartialReductionExtendKind(Instruction *I)
Get the kind of extension that an instruction represents.
TargetCostKind
The kind of cost model.
@ TCK_RecipThroughput
Reciprocal throughput.
LLVM_ABI InstructionCost getPartialReductionCost(unsigned Opcode, Type *InputTypeA, Type *InputTypeB, Type *AccumType, ElementCount VF, PartialReductionExtendKind OpAExtend, PartialReductionExtendKind OpBExtend, std::optional< unsigned > BinOp, TTI::TargetCostKind CostKind, std::optional< FastMathFlags > FMF) const
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
This class implements a switch-like dispatch statement for a value of 'T' using dyn_cast functionalit...
Definition TypeSwitch.h:89
TypeSwitch< T, ResultT > & Case(CallableT &&caseFn)
Add a case on the given type.
Definition TypeSwitch.h:98
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:282
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:307
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntOrPtrTy() const
Return true if this is an integer type or a pointer type.
Definition Type.h:270
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
op_range operands()
Definition User.h:267
static SmallVector< VFInfo, 8 > getMappings(const CallInst &CI)
Retrieve all the VFInfo instances associated to the CallInst CI.
Definition VectorUtils.h:76
bool isLegalMaskedLoadOrStore(bool IsLoad, Type *ScalarTy, Align Alignment, unsigned AddressSpace) const
Returns true if the target machine supports a masked load (if IsLoad) or masked store of scalar type ...
VPBasicBlock serves as the leaf of the Hierarchical Control-Flow Graph.
Definition VPlan.h:4400
void appendRecipe(VPRecipeBase *Recipe)
Augment the existing recipes of a VPBasicBlock with an additional Recipe as the last recipe.
Definition VPlan.h:4475
iterator end()
Definition VPlan.h:4437
iterator begin()
Recipe iterator methods.
Definition VPlan.h:4435
iterator_range< iterator > phis()
Returns an iterator range over the PHI-like recipes in the block.
Definition VPlan.h:4488
iterator getFirstNonPhi()
Return the position of the first non-phi node recipe in the block.
Definition VPlan.cpp:266
VPBasicBlock * splitAt(iterator SplitAt)
Split current block at SplitAt by inserting a new block between the current block and its successors ...
Definition VPlan.cpp:584
const VPRecipeBase & front() const
Definition VPlan.h:4447
VPRecipeBase * getTerminator()
If the block has multiple successors, return the branch recipe terminating the block.
Definition VPlan.cpp:663
const VPRecipeBase & back() const
Definition VPlan.h:4449
A recipe for vectorizing a phi-node as a sequence of mask-based select instructions.
Definition VPlan.h:2963
VPValue * getIncomingValue(unsigned Idx) const
Return incoming value number Idx.
Definition VPlan.h:3010
VPValue * getMask(unsigned Idx) const
Return mask number Idx.
Definition VPlan.h:3015
unsigned getNumIncomingValues() const
Return the number of incoming values, taking into account when normalized the first incoming value wi...
Definition VPlan.h:3005
void setMask(unsigned Idx, VPValue *V)
Set mask number Idx to V.
Definition VPlan.h:3021
bool isNormalized() const
A normalized blend is one that has an odd number of operands, whereby the first operand does not have...
Definition VPlan.h:3001
VPBlockBase is the building block of the Hierarchical Control-Flow Graph.
Definition VPlan.h:93
void setSuccessors(ArrayRef< VPBlockBase * > NewSuccs)
Set each VPBasicBlock in NewSuccss as successor of this VPBlockBase.
Definition VPlan.h:314
VPRegionBlock * getParent()
Definition VPlan.h:191
const VPBasicBlock * getExitingBasicBlock() const
Definition VPlan.cpp:236
size_t getNumSuccessors() const
Definition VPlan.h:242
void setPredecessors(ArrayRef< VPBlockBase * > NewPreds)
Set each VPBasicBlock in NewPreds as predecessor of this VPBlockBase.
Definition VPlan.h:305
const VPBlocksTy & getPredecessors() const
Definition VPlan.h:227
VPBlockBase * getSinglePredecessor() const
Definition VPlan.h:238
const VPBasicBlock * getEntryBasicBlock() const
Definition VPlan.cpp:216
VPBlockBase * getSingleSuccessor() const
Definition VPlan.h:232
const VPBlocksTy & getSuccessors() const
Definition VPlan.h:216
static auto blocksAs(T &&Range)
Return an iterator range over Range with each block cast to BlockTy.
Definition VPlanUtils.h:405
static void insertOnEdge(VPBlockBase *From, VPBlockBase *To, VPBlockBase *BlockPtr)
Inserts BlockPtr on the edge between From and To.
Definition VPlanUtils.h:424
static bool isLatch(const VPBlockBase *VPB, const VPDominatorTree &VPDT)
Returns true if VPB is a loop latch, using isHeader().
static VPBasicBlock * getPlainCFGMiddleBlock(const VPlan &Plan)
Returns the middle block of Plan in plain CFG form (before regions are formed).
static void insertTwoBlocksAfter(VPBlockBase *IfTrue, VPBlockBase *IfFalse, VPBlockBase *BlockPtr)
Insert disconnected VPBlockBases IfTrue and IfFalse after BlockPtr.
Definition VPlanUtils.h:315
static void connectBlocks(VPBlockBase *From, VPBlockBase *To, unsigned PredIdx=-1u, unsigned SuccIdx=-1u)
Connect VPBlockBases From and To bi-directionally.
Definition VPlanUtils.h:333
static void disconnectBlocks(VPBlockBase *From, VPBlockBase *To)
Disconnect VPBlockBases From and To bi-directionally.
Definition VPlanUtils.h:351
static auto blocksOnly(T &&Range)
Return an iterator range over Range which only includes BlockTy blocks.
Definition VPlanUtils.h:387
static std::pair< VPBasicBlock *, VPBasicBlock * > getPlainCFGHeaderAndLatch(const VPlan &Plan)
Returns the header and latch of the outermost loop of Plan in plain CFG form (before regions are form...
static void transferSuccessors(VPBlockBase *Old, VPBlockBase *New)
Transfer successors from Old to New. New must have no successors.
Definition VPlanUtils.h:371
static SmallVector< VPBasicBlock * > blocksInSingleSuccessorChainBetween(VPBasicBlock *FirstBB, VPBasicBlock *LastBB)
Returns the blocks between FirstBB and LastBB, where FirstBB to LastBB forms a single-sucessor chain.
A recipe for generating conditional branches on the bits of a mask.
Definition VPlan.h:3513
VPlan-based builder utility analogous to IRBuilder.
VPInstruction * createFirstActiveLane(ArrayRef< VPValue * > Masks, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenStoreRecipe * createWidenStore(StoreInst &Store, VPValue *Addr, VPValue *StoredVal, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Store, storing StoredVal to Addr with Mask (may be null).
VPInstruction * createAdd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", VPRecipeWithIRFlags::WrapFlagsTy WrapFlags={false, false})
VPInstruction * createOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createLogicalOr(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPWidenLoadRecipe * createWidenLoad(LoadInst &Load, VPValue *Addr, VPValue *Mask, bool Consecutive, const VPIRMetadata &Metadata, DebugLoc DL)
Create a recipe widening Load, loading from Addr with Mask (may be null).
VPInstruction * createNot(VPValue *Operand, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createAnyOfReduction(VPValue *ChainOp, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown())
Create an AnyOf reduction pattern: or-reduce ChainOp, freeze the result, then select between TrueVal ...
Definition VPlan.cpp:1668
void setInsertPoint(const VPInsertPoint &IP)
Set the current insert point.
VPInstruction * createLogicalAnd(VPValue *LHS, VPValue *RHS, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
VPInstruction * createScalarCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy, DebugLoc DL, std::optional< VPIRFlags > Flags=std::nullopt, const VPIRMetadata &Metadata={})
VPValue * createScalarZExtOrTrunc(VPValue *Op, Type *ResultTy, DebugLoc DL)
static VPBuilder getToInsertAfter(VPRecipeBase *R)
Create a VPBuilder to insert after R.
VPDerivedIVRecipe * createDerivedIV(InductionDescriptor::InductionKind Kind, FPMathOperator *FPBinOp, VPValue *Start, VPValue *Current, VPValue *Step, const VPIRFlags::WrapFlagsTy &Flags={})
Convert Current to Start + Current * Step.
VPWidenCastRecipe * createWidenCast(Instruction::CastOps Opcode, VPValue *Op, Type *ResultTy)
VPInstruction * createICmp(CmpInst::Predicate Pred, VPValue *A, VPValue *B, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="")
Create a new ICmp VPInstruction with predicate Pred and operands A and B.
VPInstruction * createSelect(VPValue *Cond, VPValue *TrueVal, VPValue *FalseVal, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", std::optional< VPIRFlags > Flags=std::nullopt)
Create a select of TrueVal and FalseVal based on Cond, using the default flags for the result type,...
VPInstruction * createNaryOp(unsigned Opcode, ArrayRef< VPValue * > Operands, Instruction *Inst=nullptr, const VPIRFlags &Flags={}, const VPIRMetadata &MD={}, DebugLoc DL=DebugLoc::getUnknown(), const Twine &Name="", Type *ResultTy=nullptr)
Create an N-ary operation with Opcode, Operands and set Inst as its underlying Instruction.
static VPSingleDefRecipe * createSingleScalarOp(unsigned Opcode, ArrayRef< VPValue * > Operands, VPValue *Mask, const VPIRFlags &Flags, const VPIRMetadata &Metadata, DebugLoc DL, Instruction *UV)
Create a single-scalar recipe with Opcode and Operands without inserting it.
unsigned getNumDefinedValues() const
Returns the number of values defined by the VPDef.
Definition VPlanValue.h:578
VPValue * getVPSingleValue()
Returns the only VPValue defined by the VPDef.
Definition VPlanValue.h:551
VPValue * getVPValue(unsigned I)
Returns the VPValue with index I defined by the VPDef.
Definition VPlanValue.h:563
ArrayRef< VPRecipeValue * > definedValues()
Returns an ArrayRef of the values defined by the VPDef.
Definition VPlanValue.h:573
Template specialization of the standard LLVM dominator tree utility for VPBlockBases.
bool properlyDominates(const VPRecipeBase *A, const VPRecipeBase *B) const
A recipe to combine multiple recipes into a single 'expression' recipe, which should be considered a ...
Definition VPlan.h:3558
A pure virtual base class for all recipes modeling header phis, including phis for first order recurr...
Definition VPlan.h:2451
virtual VPValue * getBackedgeValue()
Returns the incoming value from the loop backedge.
Definition VPlan.h:2498
VPValue * getStartValue()
Returns the start value of the phi, if one is set.
Definition VPlan.h:2487
A recipe representing a sequence of load -> update -> store as part of a histogram operation.
Definition VPlan.h:2178
A special type of VPBasicBlock that wraps an existing IR basic block.
Definition VPlan.h:4553
Class to record and manage LLVM IR flags.
Definition VPlan.h:703
static VPIRFlags getDefaultFlags(unsigned Opcode, Type *ResultTy=nullptr)
Returns default flags for Opcode and scalar ResultTy for opcodes that support it, asserts otherwise.
LLVM_ABI_FOR_TEST FastMathFlags getFastMathFlagsOrNone() const
Helper to manage IR metadata for recipes.
Definition VPlan.h:1180
void intersect(const VPIRMetadata &MD)
Intersect this VPIRMetadata object with MD, keeping only metadata nodes that are common to both.
This is a concrete Recipe that models a single VPlan-level instruction.
Definition VPlan.h:1235
unsigned getNumOperandsWithoutMask() const
Returns the number of operands, excluding the mask if the VPInstruction is masked.
Definition VPlan.h:1485
@ ExtractLane
Extracts a single lane (first operand) from a set of vector operands.
Definition VPlan.h:1336
@ ReductionStartVector
Start vector for reductions with 3 operands: the original start value, the identity value for the red...
Definition VPlan.h:1332
@ BuildVector
Creates a fixed-width vector containing all operands.
Definition VPlan.h:1281
@ ComputeReductionResult
Reduce the operands to the final reduction result using the operation specified via the operation's V...
Definition VPlan.h:1289
unsigned getOpcode() const
Definition VPlan.h:1429
VPValue * getMask() const
Returns the mask for the VPInstruction.
Definition VPlan.h:1501
const InterleaveGroup< Instruction > * getInterleaveGroup() const
Definition VPlan.h:3116
VPValue * getMask() const
Return the mask used by this recipe.
Definition VPlan.h:3108
ArrayRef< VPValue * > getStoredValues() const
Return the VPValues stored by this interleave group.
Definition VPlan.h:3137
VPInterleaveRecipe is a recipe for transforming an interleave group of load or stores into one wide l...
Definition VPlan.h:3147
VPPredInstPHIRecipe is a recipe for generating the phi nodes needed when control converges back from ...
Definition VPlan.h:3719
VPRecipeBase is a base class modeling a sequence of one or more output IR instructions.
Definition VPlan.h:410
VPRegionBlock * getRegion()
Definition VPlan.h:4799
VPBasicBlock * getParent()
Definition VPlan.h:482
DebugLoc getDebugLoc() const
Returns the debug location of the recipe.
Definition VPlan.h:560
void moveBefore(VPBasicBlock &BB, iplist< VPRecipeBase >::iterator I)
Unlink this recipe and insert into BB before I.
void insertBefore(VPRecipeBase *InsertPos)
Insert an unlinked recipe into a basic block immediately before the specified recipe.
void insertAfter(VPRecipeBase *InsertPos)
Insert an unlinked Recipe into a basic block immediately after the specified Recipe.
iplist< VPRecipeBase >::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
Helper class to create VPRecipies from IR instructions.
VPHistogramRecipe * widenIfHistogram(VPInstruction *VPI)
If VPI represents a histogram operation (as determined by LoopVectorizationLegality) make that safe f...
bool prefersVectorizedAddressing() const
Returns true if the target prefers vectorized addressing.
VPRecipeBase * tryToWidenMemory(VPInstruction *VPI, VFRange &Range)
Check if the load or store instruction VPI should widened for Range.Start and potentially masked.
bool replaceWithFinalIfReductionStore(VPInstruction *VPI, VPBuilder &FinalRedStoresBuilder)
If VPI is a store of a reduction into an invariant address, delete it.
VPSingleDefRecipe * handleReplication(VPInstruction *VPI, VFRange &Range)
Build a replicating or single-scalar recipe for VPI.
bool isPredicatedInst(Instruction *I) const
Returns true if I needs to be predicated (i.e.
Type * getScalarType() const
Returns the scalar type of this VPRecipeValue.
Definition VPlanValue.h:354
A recipe for handling reduction phis.
Definition VPlan.h:2870
void setVFScaleFactor(unsigned ScaleFactor)
Set the VFScaleFactor for this reduction phi.
Definition VPlan.h:2921
unsigned getVFScaleFactor() const
Get the factor that the VF of this recipe's output should be scaled by, or 1 if it isn't scaled.
Definition VPlan.h:2914
RecurKind getRecurrenceKind() const
Returns the recurrence kind of the reduction.
Definition VPlan.h:2927
A recipe to represent inloop, ordered or partial reduction operations.
Definition VPlan.h:3240
VPRegionBlock represents a collection of VPBasicBlocks and VPRegionBlocks which form a Single-Entry-S...
Definition VPlan.h:4625
const VPBlockBase * getEntry() const
Definition VPlan.h:4669
bool isReplicator() const
An indicator whether this region is to generate multiple replicated instances of output IR correspond...
Definition VPlan.h:4701
void setExiting(VPBlockBase *ExitingBlock)
Set ExitingBlock as the exiting VPBlockBase of this VPRegionBlock.
Definition VPlan.h:4686
Type * getCanonicalIVType() const
Return the type of the canonical IV for loop regions.
Definition VPlan.h:4753
VPRegionValue * getCanonicalIV()
Return the canonical induction variable of the region, null for replicating regions.
Definition VPlan.h:4745
const VPBlockBase * getExiting() const
Definition VPlan.h:4681
VPRegionValue * getHeaderMask() const
Return the header mask of the region, or null if not set.
Definition VPlan.h:4758
VPReplicateRecipe replicates a given instruction producing multiple scalar copies of the original sca...
Definition VPlan.h:3405
bool isSingleScalar() const
Returns true if the recipe produces a single scalar value.
Definition VPlan.h:3464
static InstructionCost computeCallCost(Function *CalledFn, Type *ResultTy, ArrayRef< const VPValue * > ArgOps, bool IsSingleScalar, ElementCount VF, VPCostContext &Ctx)
Return the cost of scalarizing a call to CalledFn with argument operands ArgOps for a given VF.
operand_range operandsWithoutMask()
Return the recipe's operands, excluding the mask of a predicated recipe.
Definition VPlan.h:3492
bool isPredicated() const
Definition VPlan.h:3469
VPValue * getMask()
Return the mask of a predicated VPReplicateRecipe.
Definition VPlan.h:3486
Lightweight SCEV-to-VPlan expander.
Definition VPlanUtils.h:250
VPValue * expand(const SCEV *S)
Expand S into recipes and live-ins using the builder.
A recipe for handling phi nodes of integer and floating-point inductions, producing their scalar valu...
Definition VPlan.h:4255
VPSingleDefRecipe is a base class for recipes that model a sequence of one or more output IR that def...
Definition VPlan.h:618
Instruction * getUnderlyingInstr()
Returns the underlying instruction.
Definition VPlan.h:688
VPSingleDefRecipe * clone() override=0
Clone the current recipe.
A symbolic live-in VPValue, used for values like vector trip count, VF, and VFxUF.
Definition VPlanValue.h:217
This class augments VPValue with operands which provide the inverse def-use edges from VPValue's user...
Definition VPlanValue.h:401
operand_range operands()
Definition VPlanValue.h:474
void setOperand(unsigned I, VPValue *New)
Definition VPlanValue.h:447
unsigned getNumOperands() const
Definition VPlanValue.h:441
VPValue * getOperand(unsigned N) const
Definition VPlanValue.h:442
This is the base class of the VPlan Def/Use graph, used for modeling the data flow into,...
Definition VPlanValue.h:50
Type * getScalarType() const
Returns the scalar type of this VPValue, dispatching based on the concrete subclass.
Definition VPlan.cpp:149
Value * getLiveInIRValue() const
Return the underlying IR value for a VPIRValue.
Definition VPlan.cpp:143
bool isDefinedOutsideLoopRegions() const
Returns true if the VPValue is defined outside any loop.
Definition VPlan.cpp:1492
VPRecipeBase * getDefiningRecipe()
Returns the recipe defining this VPValue or nullptr if it is not defined by a recipe,...
Definition VPlan.cpp:130
bool hasMoreThanOneUniqueUser() const
Returns true if the value has more than one unique user.
Definition VPlanValue.h:164
Value * getUnderlyingValue() const
Return the underlying Value attached to this VPValue.
Definition VPlanValue.h:75
bool user_empty() const
Definition VPlanValue.h:161
bool hasOneUse() const
Definition VPlanValue.h:175
VPUser * getSingleUser()
Return the single user of this value, or nullptr if there is not exactly one user.
Definition VPlanValue.h:179
void replaceAllUsesWith(VPValue *New)
Definition VPlan.cpp:1495
unsigned getNumUsers() const
Definition VPlanValue.h:115
void replaceUsesWithIf(VPValue *New, llvm::function_ref< bool(VPUser &U, unsigned Idx)> ShouldReplace)
Go through the uses list for this VPValue and make each use point to New if the callback ShouldReplac...
Definition VPlan.cpp:1501
user_range users()
Definition VPlanValue.h:157
A recipe to compute a pointer to the last element of each part of a widened memory access for widened...
Definition VPlan.h:2281
A recipe for widening Call instructions using library calls.
Definition VPlan.h:2112
static InstructionCost computeCallCost(Function *Variant, VPCostContext &Ctx)
Return the cost of widening a call using the vector function Variant.
VPWidenCastRecipe is a recipe to create vector cast instructions.
Definition VPlan.h:1894
Instruction::CastOps getOpcode() const
Definition VPlan.h:1930
A recipe for handling GEP instructions.
Definition VPlan.h:2221
Base class for widened induction (VPWidenIntOrFpInductionRecipe and VPWidenPointerInductionRecipe),...
Definition VPlan.h:2525
VPIRValue * getStartValue() const
Returns the start value of the induction.
Definition VPlan.h:2573
PHINode * getPHINode() const
Returns the underlying PHINode if one exists, or null otherwise.
Definition VPlan.h:2591
VPValue * getStepValue()
Returns the step value of the induction.
Definition VPlan.h:2576
const InductionDescriptor & getInductionDescriptor() const
Returns the induction descriptor for the recipe.
Definition VPlan.h:2596
A recipe for handling phi nodes of integer and floating-point inductions, producing their vector valu...
Definition VPlan.h:2625
TruncInst * getTruncInst()
Returns the first defined value as TruncInst, if it is one or nullptr otherwise.
Definition VPlan.h:2684
A recipe for widening vector intrinsics.
Definition VPlan.h:1941
static InstructionCost computeCallCost(Intrinsic::ID ID, ArrayRef< const VPValue * > Operands, const VPRecipeWithIRFlags &R, ElementCount VF, VPCostContext &Ctx)
Compute the cost of a vector intrinsic with ID and Operands.
static InstructionCost computeMemIntrinsicCost(Intrinsic::ID IID, Type *Ty, bool IsMasked, Align Alignment, VPCostContext &Ctx)
Helper function for computing the cost of vector memory intrinsic.
A common mixin class for widening memory operations.
Definition VPlan.h:3755
virtual VPRecipeBase * getAsRecipe()=0
Return a VPRecipeBase* to the current object.
A recipe for widened phis.
Definition VPlan.h:2757
VPWidenRecipe is a recipe for producing a widened instruction using the opcode and operands of the re...
Definition VPlan.h:1828
InstructionCost computeCost(ElementCount VF, VPCostContext &Ctx) const override
Return the cost of this VPWidenRecipe.
VPWidenRecipe * clone() override
Clone the current recipe.
Definition VPlan.h:1854
unsigned getOpcode() const
Definition VPlan.h:1873
VPlan models a candidate for vectorization, encoding various decisions take to produce efficient outp...
Definition VPlan.h:4812
VPIRValue * getLiveIn(Value *V) const
Return the live-in VPIRValue for V, if there is one or nullptr otherwise.
Definition VPlan.h:5151
bool hasVF(ElementCount VF) const
Definition VPlan.h:5044
const DataLayout & getDataLayout() const
Definition VPlan.h:5026
LLVMContext & getContext() const
Definition VPlan.h:5022
VPBasicBlock * getEntry()
Definition VPlan.h:4908
bool hasScalableVF() const
Definition VPlan.h:5045
VPValue * getTripCount() const
The trip count of the original loop.
Definition VPlan.h:4980
VPValue * getOrCreateBackedgeTakenCount()
The backedge taken count of the original loop.
Definition VPlan.h:5001
iterator_range< SmallSetVector< ElementCount, 2 >::iterator > vectorFactors() const
Returns an iterator range over all VFs of the plan.
Definition VPlan.h:5051
VPIRValue * getFalse()
Return a VPIRValue wrapping i1 false.
Definition VPlan.h:5117
VPSymbolicValue & getVFxUF()
Returns VF * UF of the vector loop region.
Definition VPlan.h:5020
VPIRValue * getAllOnesValue(Type *Ty)
Return a VPIRValue wrapping the AllOnes value of type Ty.
Definition VPlan.h:5123
VPRegionBlock * createReplicateRegion(VPBlockBase *Entry, VPBlockBase *Exiting, const std::string &Name="")
Create a new replicate region with Entry, Exiting and Name.
Definition VPlan.h:5202
auto getLiveIns() const
Return the list of live-in VPValues available in the VPlan.
Definition VPlan.h:5154
bool hasUF(unsigned UF) const
Definition VPlan.h:5069
ArrayRef< VPIRBasicBlock * > getExitBlocks() const
Return an ArrayRef containing VPIRBasicBlocks wrapping the exit blocks of the original scalar loop.
Definition VPlan.h:4974
VPSymbolicValue & getVectorTripCount()
The vector trip count.
Definition VPlan.h:5010
VPValue * getBackedgeTakenCount() const
Definition VPlan.h:5007
VPIRValue * getOrAddLiveIn(Value *V)
Gets the live-in VPIRValue for V or adds a new live-in (if none exists yet) for V.
Definition VPlan.h:5094
VPIRValue * getZero(Type *Ty)
Return a VPIRValue wrapping the null value of type Ty.
Definition VPlan.h:5120
void setVF(ElementCount VF)
Definition VPlan.h:5032
bool isUnrolled() const
Returns true if the VPlan already has been unrolled, i.e.
Definition VPlan.h:5085
LLVM_ABI_FOR_TEST VPRegionBlock * getVectorLoopRegion()
Returns the VPRegionBlock of the vector loop.
Definition VPlan.cpp:1080
unsigned getConcreteUF() const
Returns the concrete UF of the plan, after unrolling.
Definition VPlan.h:5072
void resetTripCount(VPValue *NewTripCount)
Resets the trip count for the VPlan.
Definition VPlan.h:4994
VPBasicBlock * getMiddleBlock()
Returns the 'middle' block of the plan, that is the block that selects whether to execute the scalar ...
Definition VPlan.h:4950
VPBasicBlock * createVPBasicBlock(const Twine &Name, VPRecipeBase *Recipe=nullptr)
Create a new VPBasicBlock with Name and containing Recipe if present.
Definition VPlan.h:5177
VPIRValue * getTrue()
Return a VPIRValue wrapping i1 true.
Definition VPlan.h:5114
VPBasicBlock * getVectorPreheader() const
Returns the preheader of the vector loop region, if one exists, or null otherwise.
Definition VPlan.h:4913
VPSymbolicValue & getUF()
Returns the UF of the vector loop region.
Definition VPlan.h:5017
bool hasScalarVFOnly() const
Definition VPlan.h:5062
VPBasicBlock * getScalarPreheader() const
Return the VPBasicBlock for the preheader of the scalar loop.
Definition VPlan.h:4964
bool hasTailFolded() const
Returns true if the vector loop region is tail-folded.
Definition VPlan.h:4929
VPSymbolicValue & getVF()
Returns the VF of the vector loop region.
Definition VPlan.h:5013
LLVM_ABI_FOR_TEST VPlan * duplicate()
Clone the current VPlan, update all VPValues of the new VPlan and cloned recipes to refer to the clon...
Definition VPlan.cpp:1240
VPIRValue * getConstantInt(Type *Ty, uint64_t Val, bool IsSigned=false)
Return a VPIRValue wrapping a ConstantInt with the given type and value.
Definition VPlan.h:5128
LLVM Value Representation.
Definition Value.h:75
iterator_range< user_iterator > users()
Definition Value.h:426
bool hasName() const
Definition Value.h:261
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
constexpr bool hasKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns true if there exists a value X where RHS.multiplyCoefficientBy(X) will result in a value whos...
Definition TypeSize.h:269
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr ScalarTy getKnownScalarFactor(const FixedOrScalableQuantity &RHS) const
Returns a value X where RHS.multiplyCoefficientBy(X) will result in a value whose quantity matches ou...
Definition TypeSize.h:277
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr LeafTy multiplyCoefficientBy(ScalarTy RHS) const
Definition TypeSize.h:256
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
An efficient, type-erasing, non-owning reference to a callable.
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt RoundingUDiv(const APInt &A, const APInt &B, APInt::Rounding RM)
Return A unsign-divided by B, rounded by the given rounding mode.
Definition APInt.cpp:2799
std::variant< std::monostate, Loc::Single, Loc::Multi, Loc::MMI, Loc::EntryValue > Variant
Alias for the std::variant specialization base class of DbgVariable.
Definition DwarfDebug.h:190
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_unless< Pattern > m_Unless(const Pattern &P)
Match if the inner matcher does NOT match.
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
auto m_Cmp()
Matches any compare instruction and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::URem > m_URem(const LHS &L, const RHS &R)
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
LogicalOp_match< LHS, RHS, Instruction::And > m_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R either in the form of L & R or L ?
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
BinaryOp_match< LHS, RHS, Instruction::FMul > m_FMul(const LHS &L, const RHS &R)
bool match(Val *V, const Pattern &P)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
SpecificCmpClass_match< LHS, RHS, CmpInst > m_SpecificCmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::UDiv > m_UDiv(const LHS &L, const RHS &R)
SelectLike_match< CondTy, LTy, RTy > m_SelectLike(const CondTy &C, const LTy &TrueC, const RTy &FalseC)
Matches a value that behaves like a boolean-controlled select, i.e.
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinaryOp_match< LHS, RHS, Instruction::FAdd, true > m_c_FAdd(const LHS &L, const RHS &R)
Matches FAdd with LHS and RHS in either order.
LogicalOp_match< LHS, RHS, Instruction::And, true > m_c_LogicalAnd(const LHS &L, const RHS &R)
Matches L && R with LHS and RHS in either order.
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
bind_cst_ty m_scev_APInt(const APInt *&C)
Match an SCEV constant and bind it to an APInt.
specificloop_ty m_SpecificLoop(const Loop *L)
bool match(const SCEV *S, const Pattern &P)
SCEVAffineAddRec_match< Op0_t, Op1_t, match_isa< const Loop > > m_scev_AffineAddRec(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastLane, VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > > m_ExtractLastLaneOfLastPart(const Op0_t &Op0)
AllRecipe_commutative_match< Instruction::And, Op0_t, Op1_t > m_c_BinaryAnd(const Op0_t &Op0, const Op1_t &Op1)
Match a binary AND operation.
AllRecipe_match< Instruction::Or, Op0_t, Op1_t > m_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
Match a binary OR operation.
VPInstruction_match< VPInstruction::AnyOf > m_AnyOf()
AllRecipe_commutative_match< Instruction::Or, Op0_t, Op1_t > m_c_BinaryOr(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ComputeReductionResult, Op0_t > m_ComputeReductionResult(const Op0_t &Op0)
auto m_WidenAnyExtend(const Op0_t &Op0)
match_bind< VPIRValue > m_VPIRValue(VPIRValue *&V)
Match a VPIRValue.
VPInstruction_match< VPInstruction::WideActiveLaneMask, Op0_t, Op1_t, Op2_t > m_WideActiveLaneMask(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
auto m_VPPhi(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::BranchOnTwoConds > m_BranchOnTwoConds()
AllRecipe_match< Opcode, Op0_t, Op1_t > m_Binary(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::LastActiveLane, Op0_t > m_LastActiveLane(const Op0_t &Op0)
auto m_WidenIntrinsic(const T &...Ops)
canonical_widen_iv_match m_CanonicalWidenIV()
VPInstruction_match< VPInstruction::ExitingIVValue, Op0_t > m_ExitingIVValue(const Op0_t &Op0)
VPInstruction_match< Instruction::ExtractElement, Op0_t, Op1_t > m_ExtractElement(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_False()
VPInstruction_match< VPInstruction::ExtractLastLane, Op0_t > m_ExtractLastLane(const Op0_t &Op0)
match_bind< VPSingleDefRecipe > m_VPSingleDefRecipe(VPSingleDefRecipe *&V)
Match a VPSingleDefRecipe, capturing if we match.
VPInstruction_match< VPInstruction::BranchOnCount > m_BranchOnCount()
auto m_GetElementPtr(const Op0_t &Op0, const Op1_t &Op1)
specific_intval< 1 > m_True()
auto m_VPValue()
Match an arbitrary VPValue and ignore it.
VPInstruction_match< VPInstruction::ExtractVectorForPart, Op0_t, Op1_t > m_ExtractVectorForPart(const Op0_t &Op0, const Op1_t &Op1)
VPInstruction_match< VPInstruction::ExtractLastPart, Op0_t > m_ExtractLastPart(const Op0_t &Op0)
VPRecipeBase * findUserOf(VPValue *V, const MatchT &P)
If V is used by a recipe matching pattern P, return it.
VPInstruction_match< VPInstruction::Broadcast, Op0_t > m_Broadcast(const Op0_t &Op0)
header_mask_match m_HeaderMask()
VPInstruction_match< VPInstruction::BuildVector > m_BuildVector()
BuildVector is matches only its opcode, w/o matching its operands as the number of operands is not fi...
VPInstruction_match< VPInstruction::ExtractPenultimateElement, Op0_t > m_ExtractPenultimateElement(const Op0_t &Op0)
match_bind< VPInstruction > m_VPInstruction(VPInstruction *&V)
Match a VPInstruction, capturing if we match.
VPInstruction_match< VPInstruction::FirstActiveLane, Op0_t > m_FirstActiveLane(const Op0_t &Op0)
auto m_DerivedIV(const Op0_t &Op0, const Op1_t &Op1, const Op2_t &Op2)
VPInstruction_match< VPInstruction::BranchOnCond > m_BranchOnCond()
VPInstruction_match< VPInstruction::ExtractLane, Op0_t, Op1_t > m_ExtractLane(const Op0_t &Op0, const Op1_t &Op1)
auto m_AnyNeg(const Op0_t &Op0)
VPInstruction_match< VPInstruction::Reverse, Op0_t > m_Reverse(const Op0_t &Op0)
NodeAddr< DefNode * > Def
Definition RDFGraph.h:384
bool isSingleScalar(const VPValue *VPV)
Returns true if VPV is a single scalar, either because it produces the same value for all lanes or on...
VPValue * getOrCreateVPValueForSCEVExpr(VPlan &Plan, const SCEV *Expr)
Get or create a VPValue that corresponds to the expansion of Expr.
bool cannotHoistOrSinkRecipe(const VPRecipeBase &R, bool Sinking=false)
Return true if we do not know how to (mechanically) hoist or sink R.
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
VPInstruction * findComputeReductionResult(VPReductionPHIRecipe *PhiR)
Find the ComputeReductionResult recipe for PhiR, looking through selects inserted for predicated redu...
VPInstruction * findCanonicalIVIncrement(VPlan &Plan)
Find the canonical IV increment of Plan's vector loop region.
std::optional< MemoryLocation > getMemoryLocation(const VPRecipeBase &R)
Return a MemoryLocation for R with noalias metadata populated from R, if the recipe is supported and ...
bool onlyFirstLaneUsed(const VPValue *Def)
Returns true if only the first lane of Def is used.
VPIRValue * tryToFoldLiveIns(VPSingleDefRecipe &R, ArrayRef< VPValue * > Operands, const DataLayout &DL)
Try to fold R using InstSimplifyFolder.
SmallVector< std::pair< VPBasicBlock *, VPIRBasicBlock * > > getEarlyExits(const VPlan &Plan, const VPBlockBase *MiddleVPBB)
Returns the (early exiting block, exit block) pairs of Plan, i.e.
void recursivelyDeleteDeadRecipes(VPValue *V)
Recursively delete V and any of its operands that become dead.
bool doesGeneratePerAllLanes(const VPRecipeBase *R)
Returns true if R produces scalar values for all VF lanes.
bool isDeadRecipe(VPRecipeBase &R)
Returns true if R is dead, i.e.
VPRecipeBase * findRecipe(VPValue *Start, PredT Pred)
Search Start's users for a recipe satisfying Pred, looking through recipes with definitions.
Definition VPlanUtils.h:149
bool isUniformAcrossVFsAndUFs(const VPValue *V)
Checks if V is uniform across all VF lanes and UF parts.
bool isUsedByLoadStoreAddress(const VPValue *V)
Returns true if V is used as part of the address of another load or store.
std::optional< std::pair< bool, unsigned > > getOpcodeOrIntrinsicID(const VPValue *V)
Get the instruction opcode or intrinsic ID for the recipe defining V.
VPValue * scalarizeVPWidenPointerInduction(VPWidenPointerInductionRecipe *PtrIV, VPlan &Plan, VPBuilder &Builder)
Scalarize a VPWidenPointerInductionRecipe by replacing it with a PtrAdd (IndStart,...
const SCEV * getSCEVExprForVPValue(const VPValue *V, PredicatedScalarEvolution &PSE, const Loop *L=nullptr)
Return the SCEV expression for V.
void pullOutPermutations(VPlan &Plan, Match_t Perm, Builder Build)
Removes the permutation pattern Perm from any elementwise operations in the plan, by constructing a n...
Definition VPlanUtils.h:236
SmallVector< VPUser * > collectUsersRecursively(VPValue *V)
Collect all users of V, looking through recipes that define other values.
VPScalarIVStepsRecipe * createScalarIVSteps(VPlan &Plan, InductionDescriptor::InductionKind Kind, Instruction::BinaryOps InductionOpcode, FPMathOperator *FPBinOp, Instruction *TruncI, VPIRValue *StartV, VPValue *Step, DebugLoc DL, VPBuilder &Builder, const VPIRFlags::WrapFlagsTy &Flags={})
Create a scalar-iv-steps recipe over Plan's canonical IV for an induction of Kind with InductionOpcod...
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:315
SmallVector< VPBasicBlock * > vp_rpo_plain_cfg_loop_body(VPBasicBlock *Header)
Returns the VPBasicBlocks forming the loop body of a plain (pre-region) VPlan in reverse post-order s...
Definition VPlanCFG.h:262
@ Offset
Definition DWP.cpp:578
void stable_sort(R &&Range)
Definition STLExtras.h:2116
auto min_element(R &&Range)
Provide wrappers to std::min_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2078
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
unsigned getLoadStoreAddressSpace(const Value *I)
A helper function that returns the address space of the pointer operand of load or store instruction.
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1669
LLVM_ABI Intrinsic::ID getVectorIntrinsicIDForCall(const CallInst *CI, const TargetLibraryInfo *TLI)
Returns intrinsic ID for call.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:840
DenseMap< const Value *, const SCEV * > ValueToSCEVMapTy
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
const Value * getLoadStorePointerOperand(const Value *V)
A helper function that returns the pointer operand of a load or store instruction.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr from_range_t from_range
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2208
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:633
auto cast_or_null(const Y &Val)
Definition Casting.h:714
Align getLoadStoreAlignment(const Value *I)
A helper function that returns the alignment of load or store instruction.
iterator_range< df_iterator< VPBlockShallowTraversalWrapper< VPBlockBase * > > > vp_depth_first_shallow(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order.
Definition VPlanCFG.h:250
constexpr auto bind_back(FnT &&Fn, BindArgsT &&...BindArgs)
C++23 bind_back.
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
iterator_range< df_iterator< VPBlockDeepTraversalWrapper< VPBlockBase * > > > vp_depth_first_deep(VPBlockBase *G)
Returns an iterator range to traverse the graph starting at G in depth-first order while traversing t...
Definition VPlanCFG.h:285
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
Definition STLExtras.h:2173
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
auto map_range(ContainerTy &&C, FuncTy F)
Return a range that applies F to the elements of C.
Definition STLExtras.h:365
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
void erase(Container &C, ValueType V)
Wrapper function to remove a value from a container:
Definition STLExtras.h:2200
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
auto reverse(ContainerTy &&C)
Definition STLExtras.h:407
constexpr size_t range_size(R &&Range)
Returns the size of the Range, i.e., the number of elements.
Definition STLExtras.h:1694
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1636
DenseMap< Value *, const SCEVUnknown * > SymbolicStrideMap
Maps a pointer to its symbolic (non-constant) stride.
bool hasIrregularType(Type *Ty, const DataLayout &DL)
A helper function that returns true if the given type is irregular.
UncountableExitStyle
Different methods of handling early exits.
Definition VPlan.h:79
@ ReadOnly
No side effects to worry about, so we can process any uncountable exits in the loop and branch either...
Definition VPlan.h:83
@ MaskedHandleExitInScalarLoop
All memory operations other than the load(s) required to determine whether an uncountable exit occurr...
Definition VPlan.h:88
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1753
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
iterator_range< filter_iterator< detail::IterOfRange< RangeT >, PredicateT > > make_filter_range(RangeT &&Range, PredicateT Pred)
Convenience function that takes a range of elements and a predicate, and return a new filter_iterator...
Definition STLExtras.h:551
bool canConstantBeExtended(const APInt *C, Type *NarrowType, TTI::PartialReductionExtendKind ExtKind)
Check if a constant CI can be safely treated as having been extended from a narrower type with the gi...
Definition VPlan.cpp:1884
T * find_singleton(R &&Range, Predicate P, bool AllowRepeats=false)
Return the single value in Range that satisfies P(<member of Range> *, AllowRepeats)->T * returning n...
Definition STLExtras.h:1837
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:322
@ Other
Any other memory.
Definition ModRef.h:68
TargetTransformInfo TTI
RecurKind
These are the kinds of recurrences that we support.
@ UMin
Unsigned integer min implemented in terms of select(cmp()).
@ FindIV
FindIV reduction with select(icmp(),x,y) where one of (x,y) is a loop induction variable (increasing ...
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ FSub
Subtraction of floats.
@ FMul
Product of floats.
@ SMax
Signed integer max implemented in terms of select(cmp()).
@ SMin
Signed integer min implemented in terms of select(cmp()).
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
@ AddChainWithSubs
A chain of adds and subs.
@ FAdd
Sum of floats.
@ UMax
Unsigned integer max implemented in terms of select(cmp()).
LLVM_ABI Value * getRecurrenceIdentity(RecurKind K, Type *Tp, FastMathFlags FMF)
Given information about an recurrence kind, return the identity for the @llvm.vector....
LLVM_ABI BasicBlock * SplitBlock(BasicBlock *Old, BasicBlock::iterator SplitPt, DominatorTree *DT, LoopInfo *LI=nullptr, MemorySSAUpdater *MSSAU=nullptr, const Twine &BBName="")
Split the specified block at the specified instruction.
auto count(R &&Range, const E &Element)
Wrapper function around std::count to count the number of times an element Element occurs in the give...
Definition STLExtras.h:2012
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2088
ArrayRef(const T &OneElt) -> ArrayRef< T >
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1772
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
Type * getLoadStoreType(const Value *I)
A helper function that returns the type of a load or store instruction.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Definition STLExtras.h:2166
hash_code hash_combine(const Ts &...args)
Combine values into a single hash_code.
Definition Hashing.h:305
LLVM_ABI std::optional< int64_t > getStrideFromAddRec(const SCEVAddRecExpr *AR, const Loop *Lp, Type *AccessTy, Value *Ptr, PredicatedScalarEvolution &PSE)
If AR is an affine AddRec for Lp with a constant step, return the step in units of AccessTy's allocat...
bool equal(L &&LRange, R &&RRange)
Wrapper function around std::equal to detect if pair-wise elements between two ranges are the same.
Definition STLExtras.h:2146
Type * toVectorTy(Type *Scalar, ElementCount EC)
A helper function for converting Scalar types to vector types.
LLVM_ABI bool isDereferenceableAndAlignedInLoop(LoadInst *LI, Loop *L, ScalarEvolution &SE, DominatorTree &DT, AssumptionCache *AC=nullptr, SmallVectorImpl< const SCEVPredicate * > *Predicates=nullptr)
Return true if we can prove that the given load (which is assumed to be within the specified loop) wo...
Definition Loads.cpp:304
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
hash_code hash_combine_range(InputIteratorT first, InputIteratorT last)
Compute a hash_code for a sequence of values.
Definition Hashing.h:285
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
VPBasicBlock * EarlyExitingVPBB
VPIRBasicBlock * EarlyExitVPBB
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
An information struct used to provide DenseMap with the various necessary components for a given valu...
This reduction is unordered with the partial result scaled down by some factor.
Definition VPlan.h:2852
Holds the VFShape for a specific scalar to vector function mapping.
Encapsulates information needed to describe a parameter.
A range of powers-of-2 vectorization factors with fixed start and adjustable end.
Struct to hold various analysis needed for cost computations.
const VFSelectionContext & Config
static bool isFreeScalarIntrinsic(Intrinsic::ID ID)
Returns true if ID is a pseudo intrinsic that is dropped via scalarization rather than widened.
Definition VPlan.cpp:1990
bool isMaskRequired(Instruction *I) const
Forwards to LoopVectorizationCostModel::isMaskRequired.
PredicatedScalarEvolution & PSE
bool willBeScalarized(Instruction *I, ElementCount VF) const
Returns true if I is known to be scalarized at VF.
TargetTransformInfo::TargetCostKind CostKind
const TargetLibraryInfo & TLI
const TargetTransformInfo & TTI
A VPValue representing a live-in from the input IR or a constant.
Definition VPlanValue.h:279
Type * getType() const
Returns the type of the underlying IR value.
Definition VPlan.cpp:147
A recipe for widening load operations, using the address to load from and an optional mask.
Definition VPlan.h:3819
A recipe for widening store operations, using the stored value, the address to store to and an option...
Definition VPlan.h:3918
static void simplifyLiveInsWithSCEV(VPlan &Plan, PredicatedScalarEvolution &PSE)
Check Plan's live-ins and replace them with constants, if they can be simplified via SCEV.
static decltype(auto) runPass(StringRef PassName, PassTy &&Pass, VPlan &Plan, ArgsTy &&...Args)
Helper to run a VPlan pass Pass on VPlan, forwarding extra arguments to the pass.
static void createInterleaveGroups(VPlan &Plan, const SmallPtrSetImpl< const InterleaveGroup< Instruction > * > &InterleaveGroups, const bool &EpilogueAllowed)
static LLVM_ABI_FOR_TEST bool tryToConvertVPInstructionsToVPRecipes(VPlan &Plan, const TargetLibraryInfo &TLI, PredicatedScalarEvolution &PSE, Loop *OuterLoop)
Replaces the VPInstructions in Plan with corresponding widen recipes.
static void createAndOptimizeReplicateRegions(VPlan &Plan)
Wrap predicated VPReplicateRecipes with a mask operand in an if-then region block and remove the mask...
static std::unique_ptr< VPlan > narrowInterleaveGroups(VPlan &Plan, const TargetTransformInfo &TTI)
Try to find a single VF among Plan's VFs for which all interleave groups (with known minimum VF eleme...
static void makeMemOpWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert load/store VPInstructions in Plan into widened or replicate recipes.
static LLVM_ABI_FOR_TEST bool handleUncountableEarlyExits(VPlan &Plan, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC, UncountableExitStyle Style)
Update Plan to account for uncountable early exits by introducing appropriate branching logic in the ...
static void hoistPredicatedLoads(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Hoist predicated loads from the same address to the loop entry block, if they are guaranteed to execu...
static bool mergeBlocksIntoPredecessors(VPlan &Plan)
Remove redundant VPBasicBlocks by merging them into their single predecessor if the latter has a sing...
static void optimizeFindIVReductions(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L)
Optimize FindLast reductions selecting IVs (or expressions of IVs) by converting them to FindIV reduc...
static void convertToAbstractRecipes(VPlan &Plan, VPCostContext &Ctx, VFRange &Range)
This function converts initial recipes to the abstract recipes and clamps Range based on cost model f...
static void makeScalarizationDecisions(VPlan &Plan, VFRange &Range)
Make VPlan-based scalarization decision prior to delegating to the ones made by the legacy CM.
static bool areAllLoadsDereferenceable(VPBasicBlock *HeaderVPBB, Loop *TheLoop, PredicatedScalarEvolution &PSE, DominatorTree &DT, AssumptionCache *AC)
Check if all loads in the loop are dereferenceable.
static void optimizeInductionLiveOutUsers(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
If there's a single exit block, optimize its phi recipes that use exiting IV values by feeding them p...
static void simplifyReverses(VPlan &Plan)
Cancel out redundant reverses in Plan, e.g. reverse(reverse(x)) -> x.
static void makeCallWideningDecisions(VPlan &Plan, VFRange &Range, VPRecipeBuilder &RecipeBuilder, VPCostContext &CostCtx)
Convert call VPInstructions in Plan into widened call, vector intrinsic or replicate recipes based on...
static void adjustFirstOrderRecurrenceMiddleUsers(VPlan &Plan, VFRange &Range)
Adjust first-order recurrence users in the middle block: create penultimate element extracts for LCSS...
static void removeDeadRecipes(VPlan &Plan)
Remove dead recipes from Plan.
static void simplifyRecipes(VPlan &Plan)
Perform instcombine-like simplifications on recipes in Plan.
static void sinkPredicatedStores(VPlan &Plan, PredicatedScalarEvolution &PSE, const Loop *L)
Sink predicated stores to the same address with complementary predicates (P and NOT P) to an uncondit...
static bool removeBranchOnConst(VPlan &Plan, bool OnlyLatches=false)
Remove BranchOnCond recipes with true or false conditions together with removing dead edges to their ...
static void convertToStridedAccesses(VPlan &Plan, PredicatedScalarEvolution &PSE, Loop &L, VPCostContext &Ctx, VFRange &Range)
Transform widen memory recipes into strided access recipes when legal and profitable.
static void clearReductionWrapFlags(VPlan &Plan)
Clear NSW/NUW flags from reduction instructions if necessary.
static void createPartialReductions(VPlan &Plan, VPCostContext &CostCtx, VFRange &Range)
Detect and create partial reduction recipes for scaled reductions in Plan.
static void cse(VPlan &Plan)
Perform common-subexpression-elimination on Plan.
static void replaceSymbolicStrides(VPlan &Plan, PredicatedScalarEvolution &PSE, const SymbolicStrideMap &StridesMap, const VPDominatorTree &VPDT)
Replace symbolic strides from StridesMap in Plan with constants when possible.
static LLVM_ABI_FOR_TEST void optimize(VPlan &Plan)
Apply VPlan-to-VPlan optimizations to Plan, including induction recipe optimizations,...
static void truncateToMinimalBitwidths(VPlan &Plan, const MapVector< Instruction *, uint64_t > &MinBWs)
Insert truncates and extends for any truncated recipe.
static void dropPoisonGeneratingRecipes(VPlan &Plan)
Drop poison flags from recipes that may generate a poison value that is used after vectorization,...
static void optimizeForVFAndUF(VPlan &Plan, ElementCount BestVF, unsigned BestUF, PredicatedScalarEvolution &PSE)
Optimize Plan based on BestVF and BestUF.