LLVM 24.0.0git
InstCombineCalls.cpp
Go to the documentation of this file.
1//===- InstCombineCalls.cpp -----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the visitCall, visitInvoke, and visitCallBr functions.
10//
11//===----------------------------------------------------------------------===//
12
13#include "InstCombineInternal.h"
14#include "llvm/ADT/APFloat.h"
15#include "llvm/ADT/APInt.h"
16#include "llvm/ADT/APSInt.h"
17#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/Bitset.h"
22#include "llvm/ADT/Statistic.h"
28#include "llvm/Analysis/Loads.h"
33#include "llvm/IR/Attributes.h"
34#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
39#include "llvm/IR/DebugInfo.h"
41#include "llvm/IR/Function.h"
43#include "llvm/IR/InlineAsm.h"
44#include "llvm/IR/InstrTypes.h"
45#include "llvm/IR/Instruction.h"
48#include "llvm/IR/Intrinsics.h"
49#include "llvm/IR/IntrinsicsAArch64.h"
50#include "llvm/IR/IntrinsicsAMDGPU.h"
51#include "llvm/IR/IntrinsicsARM.h"
52#include "llvm/IR/IntrinsicsHexagon.h"
53#include "llvm/IR/LLVMContext.h"
54#include "llvm/IR/Metadata.h"
57#include "llvm/IR/Statepoint.h"
58#include "llvm/IR/Type.h"
59#include "llvm/IR/User.h"
60#include "llvm/IR/Value.h"
61#include "llvm/IR/ValueHandle.h"
66#include "llvm/Support/Debug.h"
77#include <algorithm>
78#include <cassert>
79#include <cstdint>
80#include <optional>
81#include <utility>
82#include <vector>
83
84#define DEBUG_TYPE "instcombine"
86
87using namespace llvm;
88using namespace PatternMatch;
89
90STATISTIC(NumSimplified, "Number of library calls simplified");
91
93 "instcombine-guard-widening-window",
94 cl::init(3),
95 cl::desc("How wide an instruction window to bypass looking for "
96 "another guard"));
97
98/// Return the specified type promoted as it would be to pass though a va_arg
99/// area.
101 if (IntegerType* ITy = dyn_cast<IntegerType>(Ty)) {
102 if (ITy->getBitWidth() < 32)
103 return Type::getInt32Ty(Ty->getContext());
104 }
105 return Ty;
106}
107
108/// Recognize a memcpy/memmove from a trivially otherwise unused alloca.
109/// TODO: This should probably be integrated with visitAllocSites, but that
110/// requires a deeper change to allow either unread or unwritten objects.
112 auto *Src = MI->getRawSource();
113 while (isa<GetElementPtrInst>(Src)) {
114 if (!Src->hasOneUse())
115 return false;
116 Src = cast<Instruction>(Src)->getOperand(0);
117 }
118 return isa<AllocaInst>(Src) && Src->hasOneUse();
119}
120
122 Align DstAlign = getKnownAlignment(MI->getRawDest(), DL, MI, &AC, &DT);
123 MaybeAlign CopyDstAlign = MI->getDestAlign();
124 if (!CopyDstAlign || *CopyDstAlign < DstAlign) {
125 MI->setDestAlignment(DstAlign);
126 return MI;
127 }
128
129 Align SrcAlign = getKnownAlignment(MI->getRawSource(), DL, MI, &AC, &DT);
130 MaybeAlign CopySrcAlign = MI->getSourceAlign();
131 if (!CopySrcAlign || *CopySrcAlign < SrcAlign) {
132 MI->setSourceAlignment(SrcAlign);
133 return MI;
134 }
135
136 // If we have a store to a location which is known constant, we can conclude
137 // that the store must be storing the constant value (else the memory
138 // wouldn't be constant), and this must be a noop.
139 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
140 // Set the size of the copy to 0, it will be deleted on the next iteration.
141 MI->setLength((uint64_t)0);
142 return MI;
143 }
144
145 // If the source is provably undef, the memcpy/memmove doesn't do anything
146 // (unless the transfer is volatile).
147 if (hasUndefSource(MI) && !MI->isVolatile()) {
148 // Set the size of the copy to 0, it will be deleted on the next iteration.
149 MI->setLength((uint64_t)0);
150 return MI;
151 }
152
153 // If MemCpyInst length is 1/2/4/8 bytes then replace memcpy with
154 // load/store.
155 ConstantInt *MemOpLength = dyn_cast<ConstantInt>(MI->getLength());
156 if (!MemOpLength) return nullptr;
157
158 // Source and destination pointer types are always "i8*" for intrinsic. See
159 // if the size is something we can handle with a single primitive load/store.
160 // A single load+store correctly handles overlapping memory in the memmove
161 // case.
162 uint64_t Size = MemOpLength->getLimitedValue();
163 assert(Size && "0-sized memory transferring should be removed already.");
164
165 if (Size > 8 || (Size&(Size-1)))
166 return nullptr; // If not 1/2/4/8 bytes, exit.
167
168 // If it is an atomic and alignment is less than the size then we will
169 // introduce the unaligned memory access which will be later transformed
170 // into libcall in CodeGen. This is not evident performance gain so disable
171 // it now.
172 if (MI->isAtomic())
173 if (*CopyDstAlign < Size || *CopySrcAlign < Size)
174 return nullptr;
175
176 // Use an integer load+store unless we can find something better.
177 IntegerType* IntType = IntegerType::get(MI->getContext(), Size<<3);
178
179 // If the memcpy has metadata describing the members, see if we can get the
180 // TBAA, scope and noalias tags describing our copy.
181 AAMDNodes AACopyMD = MI->getAAMetadata().adjustForAccess(Size);
182
183 Value *Src = MI->getArgOperand(1);
184 Value *Dest = MI->getArgOperand(0);
185 LoadInst *L = Builder.CreateLoad(IntType, Src);
186 // Alignment from the mem intrinsic will be better, so use it.
187 L->setAlignment(*CopySrcAlign);
188 L->setAAMetadata(AACopyMD);
189 MDNode *LoopMemParallelMD =
190 MI->getMetadata(LLVMContext::MD_mem_parallel_loop_access);
191 if (LoopMemParallelMD)
192 L->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
193 MDNode *AccessGroupMD = MI->getMetadata(LLVMContext::MD_access_group);
194 if (AccessGroupMD)
195 L->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
196
197 StoreInst *S = Builder.CreateStore(L, Dest);
198 // Alignment from the mem intrinsic will be better, so use it.
199 S->setAlignment(*CopyDstAlign);
200 S->setAAMetadata(AACopyMD);
201 if (LoopMemParallelMD)
202 S->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
203 if (AccessGroupMD)
204 S->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
205 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
206
207 if (auto *MT = dyn_cast<MemTransferInst>(MI)) {
208 // non-atomics can be volatile
209 L->setVolatile(MT->isVolatile());
210 S->setVolatile(MT->isVolatile());
211 }
212 if (MI->isAtomic()) {
213 // atomics have to be unordered
214 L->setOrdering(AtomicOrdering::Unordered);
216 }
217
218 // Set the size of the copy to 0, it will be deleted on the next iteration.
219 MI->setLength((uint64_t)0);
220 return MI;
221}
222
224 const Align KnownAlignment =
225 getKnownAlignment(MI->getDest(), DL, MI, &AC, &DT);
226 MaybeAlign MemSetAlign = MI->getDestAlign();
227 if (!MemSetAlign || *MemSetAlign < KnownAlignment) {
228 MI->setDestAlignment(KnownAlignment);
229 return MI;
230 }
231
232 // If we have a store to a location which is known constant, we can conclude
233 // that the store must be storing the constant value (else the memory
234 // wouldn't be constant), and this must be a noop.
235 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
236 // Set the size of the copy to 0, it will be deleted on the next iteration.
237 MI->setLength((uint64_t)0);
238 return MI;
239 }
240
241 // Remove memset with an undef value.
242 // FIXME: This is technically incorrect because it might overwrite a poison
243 // value. Change to PoisonValue once #52930 is resolved.
244 if (isa<UndefValue>(MI->getValue())) {
245 // Set the size of the copy to 0, it will be deleted on the next iteration.
246 MI->setLength((uint64_t)0);
247 return MI;
248 }
249
250 // Extract the length and validate the fill type.
251 ConstantInt *LenC = dyn_cast<ConstantInt>(MI->getLength());
252 Value *Fill = MI->getValue();
253 if (!LenC || !Fill->getType()->isIntegerTy(8))
254 return nullptr;
255 const uint64_t Len = LenC->getLimitedValue();
256 assert(Len && "0-sized memory setting should be removed already.");
257 const Align Alignment = MI->getDestAlign().valueOrOne();
258
259 // If it is an atomic and alignment is less than the size then we will
260 // introduce the unaligned memory access which will be later transformed
261 // into libcall in CodeGen. This is not evident performance gain so disable
262 // it now.
263 if (MI->isAtomic() && Alignment < Len)
264 return nullptr;
265
266 // memset(s,c,n) -> store s, c (for n=1,2,4,8)
267 if (Len <= 8 && isPowerOf2_32((uint32_t)Len)) {
268 Value *Dest = MI->getDest();
269
270 // Extract the fill value and store. A one-byte memset does not need
271 // replication so a nonconstant i8 fill can be stored directly.
272 Value *FillVal;
273 if (auto *FillC = dyn_cast<ConstantInt>(Fill))
274 FillVal = ConstantInt::get(MI->getContext(),
275 APInt::getSplat(Len * 8, FillC->getValue()));
276 else if (Len == 1)
277 FillVal = Fill;
278 else
279 return nullptr;
280
281 StoreInst *S = Builder.CreateStore(FillVal, Dest, MI->isVolatile());
282 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
283 for (DbgVariableRecord *DbgAssign : at::getDVRAssignmentMarkers(S)) {
284 if (llvm::is_contained(DbgAssign->location_ops(), Fill))
285 DbgAssign->replaceVariableLocationOp(Fill, FillVal);
286 }
287
288 S->setAlignment(Alignment);
289 if (MI->isAtomic())
291
292 // Set the size of the copy to 0, it will be deleted on the next iteration.
293 MI->setLength((uint64_t)0);
294 return MI;
295 }
296
297 return nullptr;
298}
299
300// TODO, Obvious Missing Transforms:
301// * Narrow width by halfs excluding zero/undef lanes
302Value *InstCombinerImpl::simplifyMaskedLoad(IntrinsicInst &II) {
303 Value *LoadPtr = II.getArgOperand(0);
304 const Align Alignment = II.getParamAlign(0).valueOrOne();
305 Value *Mask = II.getArgOperand(1);
306
307 // If the mask is all ones or poison, this is a plain vector load of the 1st
308 // argument.
309 if (match(Mask, m_AllOnesOrPoison())) {
310 LoadInst *L = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
311 "unmaskedload");
312 L->copyMetadata(II);
313 return L;
314 }
315
316 // If we can unconditionally load from this address, replace with a
317 // load/select idiom.
318 if (isDereferenceablePointer(LoadPtr, II.getType(),
320 LoadInst *LI = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
321 "unmaskedload");
322 LI->copyMetadata(II);
323 return Builder.CreateSelect(II.getArgOperand(1), LI, II.getArgOperand(2));
324 }
325
326 return nullptr;
327}
328
329// TODO, Obvious Missing Transforms:
330// * Single constant active lane -> store
331// * Narrow width by halfs excluding zero/undef lanes
332Instruction *InstCombinerImpl::simplifyMaskedStore(IntrinsicInst &II) {
333 Value *StorePtr = II.getArgOperand(1);
334 Align Alignment = II.getParamAlign(1).valueOrOne();
335 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
336 if (!ConstMask)
337 return nullptr;
338
339 // If the mask is all zeros or poison, this instruction does nothing.
340 if (match(ConstMask, m_ZeroOrPoison()))
342
343 // If the mask is all ones or poison, this is a plain vector store of the 1st
344 // argument.
345 if (match(ConstMask, m_AllOnesOrPoison())) {
346 StoreInst *S =
347 new StoreInst(II.getArgOperand(0), StorePtr, false, Alignment);
348 S->copyMetadata(II);
349 return S;
350 }
351
352 if (isa<ScalableVectorType>(ConstMask->getType()))
353 return nullptr;
354
355 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
356 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
357 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
358 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
359 PoisonElts))
360 return replaceOperand(II, 0, V);
361
362 return nullptr;
363}
364
365// TODO, Obvious Missing Transforms:
366// * Single constant active lane load -> load
367// * Dereferenceable address & few lanes -> scalarize speculative load/selects
368// * Adjacent vector addresses -> masked.load
369// * Narrow width by halfs excluding zero/undef lanes
370// * Vector incrementing address -> vector masked load
371Instruction *InstCombinerImpl::simplifyMaskedGather(IntrinsicInst &II) {
372 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(1));
373 if (!ConstMask)
374 return nullptr;
375
376 // Vector splat address w/known mask -> scalar load
377 // Fold the gather to load the source vector first lane
378 // because it is reloading the same value each time
379 if (ConstMask->isAllOnesValue())
380 if (auto *SplatPtr = getSplatValue(II.getArgOperand(0))) {
381 auto *VecTy = cast<VectorType>(II.getType());
382 const Align Alignment = II.getParamAlign(0).valueOrOne();
383 LoadInst *L = Builder.CreateAlignedLoad(VecTy->getElementType(), SplatPtr,
384 Alignment, "load.scalar");
385 Value *Shuf =
386 Builder.CreateVectorSplat(VecTy->getElementCount(), L, "broadcast");
388 }
389
390 return nullptr;
391}
392
393// TODO, Obvious Missing Transforms:
394// * Single constant active lane -> store
395// * Adjacent vector addresses -> masked.store
396// * Narrow store width by halfs excluding zero/undef lanes
397// * Vector incrementing address -> vector masked store
398Instruction *InstCombinerImpl::simplifyMaskedScatter(IntrinsicInst &II) {
399 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
400 if (!ConstMask)
401 return nullptr;
402
403 // If the mask is all zeros or poison, a scatter does nothing.
404 if (match(ConstMask, m_ZeroOrPoison()))
406
407 // Vector splat address -> scalar store
408 if (auto *SplatPtr = getSplatValue(II.getArgOperand(1))) {
409 // scatter(splat(value), splat(ptr), non-zero-mask) -> store value, ptr
410 if (auto *SplatValue = getSplatValue(II.getArgOperand(0))) {
411 if (maskContainsAllOneOrUndef(ConstMask)) {
412 Align Alignment = II.getParamAlign(1).valueOrOne();
413 StoreInst *S = new StoreInst(SplatValue, SplatPtr, /*IsVolatile=*/false,
414 Alignment);
415 S->copyMetadata(II);
416 return S;
417 }
418 }
419 // scatter(vector, splat(ptr), splat(true)) -> store extract(vector,
420 // lastlane), ptr
421 if (ConstMask->isAllOnesValue()) {
422 Align Alignment = II.getParamAlign(1).valueOrOne();
423 VectorType *WideLoadTy = cast<VectorType>(II.getArgOperand(1)->getType());
424 ElementCount VF = WideLoadTy->getElementCount();
425 Value *RunTimeVF = Builder.CreateElementCount(Builder.getInt32Ty(), VF);
426 Value *LastLane = Builder.CreateSub(RunTimeVF, Builder.getInt32(1));
427 Value *Extract =
428 Builder.CreateExtractElement(II.getArgOperand(0), LastLane);
429 StoreInst *S =
430 new StoreInst(Extract, SplatPtr, /*IsVolatile=*/false, Alignment);
431 S->copyMetadata(II);
432 return S;
433 }
434 }
435 if (isa<ScalableVectorType>(ConstMask->getType()))
436 return nullptr;
437
438 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
439 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
440 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
441 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
442 PoisonElts))
443 return replaceOperand(II, 0, V);
444 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(1), DemandedElts,
445 PoisonElts))
446 return replaceOperand(II, 1, V);
447
448 return nullptr;
449}
450
451/// This function transforms launder.invariant.group and strip.invariant.group
452/// like:
453/// launder(launder(%x)) -> launder(%x) (the result is not the argument)
454/// launder(strip(%x)) -> launder(%x)
455/// strip(strip(%x)) -> strip(%x) (the result is not the argument)
456/// strip(launder(%x)) -> strip(%x)
457/// This is legal because it preserves the most recent information about
458/// the presence or absence of invariant.group.
460 InstCombinerImpl &IC) {
461 auto *Arg = II.getArgOperand(0);
462 auto *StrippedArg = Arg->stripPointerCasts();
463 auto *StrippedInvariantGroupsArg = StrippedArg;
464 while (auto *Intr = dyn_cast<IntrinsicInst>(StrippedInvariantGroupsArg)) {
465 if (Intr->getIntrinsicID() != Intrinsic::launder_invariant_group &&
466 Intr->getIntrinsicID() != Intrinsic::strip_invariant_group)
467 break;
468 StrippedInvariantGroupsArg = Intr->getArgOperand(0)->stripPointerCasts();
469 }
470 if (StrippedArg == StrippedInvariantGroupsArg)
471 return nullptr; // No launders/strips to remove.
472
473 Value *Result = nullptr;
474
475 if (II.getIntrinsicID() == Intrinsic::launder_invariant_group)
476 Result = IC.Builder.CreateLaunderInvariantGroup(StrippedInvariantGroupsArg);
477 else if (II.getIntrinsicID() == Intrinsic::strip_invariant_group)
478 Result = IC.Builder.CreateStripInvariantGroup(StrippedInvariantGroupsArg);
479 else
481 "simplifyInvariantGroupIntrinsic only handles launder and strip");
482 if (Result->getType()->getPointerAddressSpace() !=
483 II.getType()->getPointerAddressSpace())
484 Result = IC.Builder.CreateAddrSpaceCast(Result, II.getType());
485
486 return cast<Instruction>(Result);
487}
488
490 assert((II.getIntrinsicID() == Intrinsic::cttz ||
491 II.getIntrinsicID() == Intrinsic::ctlz) &&
492 "Expected cttz or ctlz intrinsic");
493 bool IsTZ = II.getIntrinsicID() == Intrinsic::cttz;
494 Value *Op0 = II.getArgOperand(0);
495 Value *Op1 = II.getArgOperand(1);
496 Value *X;
497 // ctlz(bitreverse(x)) -> cttz(x)
498 // cttz(bitreverse(x)) -> ctlz(x)
499 if (match(Op0, m_BitReverse(m_Value(X)))) {
500 Intrinsic::ID ID = IsTZ ? Intrinsic::ctlz : Intrinsic::cttz;
501 Function *F =
502 Intrinsic::getOrInsertDeclaration(II.getModule(), ID, II.getType());
503 return CallInst::Create(F, {X, II.getArgOperand(1)});
504 }
505
506 if (II.getType()->isIntOrIntVectorTy(1)) {
507 // ctlz/cttz i1 Op0 --> not Op0
508 if (match(Op1, m_Zero()))
509 return BinaryOperator::CreateNot(Op0);
510 // If zero is poison, then the input can be assumed to be "true", so the
511 // instruction simplifies to "false".
512 assert(match(Op1, m_One()) && "Expected ctlz/cttz operand to be 0 or 1");
513 return IC.replaceInstUsesWith(II, ConstantInt::getNullValue(II.getType()));
514 }
515
516 // If ctlz/cttz is only used as a shift amount, set is_zero_poison to true.
517 if (II.hasOneUse() && match(Op1, m_Zero()) &&
518 match(II.user_back(), m_Shift(m_Value(), m_Specific(&II))))
519 return CallInst::Create(II.getCalledFunction(),
520 {Op0, IC.Builder.getTrue()});
521
522 Constant *C;
523
524 if (IsTZ) {
525 // cttz(-x) -> cttz(x)
526 if (match(Op0, m_Neg(m_Value(X))))
527 return CallInst::Create(II.getCalledFunction(), {X, Op1});
528
529 // cttz(-x & x) -> cttz(x)
530 if (match(Op0, m_c_And(m_Neg(m_Value(X)), m_Deferred(X))))
531 return CallInst::Create(II.getCalledFunction(), {X, Op1});
532
533 // cttz(mul(X, OddC)) -> cttz(X)
534 if (match(Op0, m_Mul(m_Value(X),
535 m_CheckedInt([](const APInt &C) { return C[0]; }))))
536 return CallInst::Create(II.getCalledFunction(), {X, Op1});
537
538 // cttz(sext(x)) -> cttz(zext(x))
539 if (match(Op0, m_OneUse(m_SExt(m_Value(X))))) {
540 auto *Zext = IC.Builder.CreateZExt(X, II.getType());
541 auto *CttzZext =
542 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, Zext, Op1);
543 return IC.replaceInstUsesWith(II, CttzZext);
544 }
545
546 // Zext doesn't change the number of trailing zeros, so narrow:
547 // cttz(zext(x)) -> zext(cttz(x)) if the 'ZeroIsPoison' parameter is 'true'.
548 if (match(Op0, m_OneUse(m_ZExt(m_Value(X)))) && match(Op1, m_One())) {
549 auto *Cttz = IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, X,
550 IC.Builder.getTrue());
551 auto *ZextCttz = IC.Builder.CreateZExt(Cttz, II.getType());
552 return IC.replaceInstUsesWith(II, ZextCttz);
553 }
554
555 // cttz(abs(x)) -> cttz(x)
556 // cttz(nabs(x)) -> cttz(x)
557 Value *Y;
559 if (SPF == SPF_ABS || SPF == SPF_NABS)
560 return CallInst::Create(II.getCalledFunction(), {X, Op1});
561
563 return CallInst::Create(II.getCalledFunction(), {X, Op1});
564
565 // cttz(shl(%const, %val), 1) --> add(cttz(%const, 1), %val)
566 if (match(Op0, m_Shl(m_ImmConstant(C), m_Value(X))) &&
567 match(Op1, m_One())) {
568 Value *ConstCttz =
569 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
570 return BinaryOperator::CreateAdd(ConstCttz, X);
571 }
572
573 // cttz(lshr exact (%const, %val), 1) --> sub(cttz(%const, 1), %val)
574 if (match(Op0, m_Exact(m_LShr(m_ImmConstant(C), m_Value(X)))) &&
575 match(Op1, m_One())) {
576 Value *ConstCttz =
577 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
578 return BinaryOperator::CreateSub(ConstCttz, X);
579 }
580
581 // cttz(add(lshr(UINT_MAX, %val), 1)) --> sub(width, %val)
582 if (match(Op0, m_Add(m_LShr(m_AllOnes(), m_Value(X)), m_One()))) {
583 Value *Width =
584 ConstantInt::get(II.getType(), II.getType()->getScalarSizeInBits());
585 return BinaryOperator::CreateSub(Width, X);
586 }
587 } else {
588 // ctlz(lshr(%const, %val), 1) --> add(ctlz(%const, 1), %val)
589 if (match(Op0, m_LShr(m_ImmConstant(C), m_Value(X))) &&
590 match(Op1, m_One())) {
591 Value *ConstCtlz =
592 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
593 return BinaryOperator::CreateAdd(ConstCtlz, X);
594 }
595
596 // ctlz(shl nuw (%const, %val), 1) --> sub(ctlz(%const, 1), %val)
597 if (match(Op0, m_NUWShl(m_ImmConstant(C), m_Value(X))) &&
598 match(Op1, m_One())) {
599 Value *ConstCtlz =
600 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
601 return BinaryOperator::CreateSub(ConstCtlz, X);
602 }
603
604 // ctlz(~x & (x - 1)) -> bitwidth - cttz(x, false)
605 if (Op0->hasOneUse() &&
606 match(Op0,
608 Type *Ty = II.getType();
609 unsigned BitWidth = Ty->getScalarSizeInBits();
610 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
611 {X, IC.Builder.getFalse()});
612 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
613 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
614 }
615 }
616
617 // cttz(Pow2) -> Log2(Pow2)
618 // ctlz(Pow2) -> BitWidth - 1 - Log2(Pow2)
619 if (auto *R = IC.tryGetLog2(Op0, match(Op1, m_One()))) {
620 if (IsTZ)
621 return IC.replaceInstUsesWith(II, R);
622 BinaryOperator *BO = BinaryOperator::CreateSub(
623 ConstantInt::get(R->getType(), R->getType()->getScalarSizeInBits() - 1),
624 R);
625 BO->setHasNoSignedWrap();
627 return BO;
628 }
629
631
632 // Create a mask for bits above (ctlz) or below (cttz) the first known one.
633 unsigned PossibleZeros = IsTZ ? Known.countMaxTrailingZeros()
634 : Known.countMaxLeadingZeros();
635 unsigned DefiniteZeros = IsTZ ? Known.countMinTrailingZeros()
636 : Known.countMinLeadingZeros();
637
638 // If all bits above (ctlz) or below (cttz) the first known one are known
639 // zero, this value is constant.
640 // FIXME: This should be in InstSimplify because we're replacing an
641 // instruction with a constant.
642 if (PossibleZeros == DefiniteZeros) {
643 auto *C = ConstantInt::get(Op0->getType(), DefiniteZeros);
644 return IC.replaceInstUsesWith(II, C);
645 }
646
647 // If the input to cttz/ctlz is known to be non-zero,
648 // then change the 'ZeroIsPoison' parameter to 'true'
649 // because we know the zero behavior can't affect the result.
650 if (!Known.One.isZero() ||
652 if (!match(II.getArgOperand(1), m_One()))
653 return CallInst::Create(II.getCalledFunction(),
654 {Op0, IC.Builder.getTrue()});
655 }
656
657 // Add range attribute since known bits can't completely reflect what we know.
658 unsigned BitWidth = Op0->getType()->getScalarSizeInBits();
659 if (BitWidth != 1 && !II.hasRetAttr(Attribute::Range) &&
660 !II.getMetadata(LLVMContext::MD_range)) {
661 ConstantRange Range(APInt(BitWidth, DefiniteZeros),
662 APInt(BitWidth, PossibleZeros + 1));
663 II.addRangeRetAttr(Range);
664 return &II;
665 }
666
667 return nullptr;
668}
669
671 assert(II.getIntrinsicID() == Intrinsic::ctpop &&
672 "Expected ctpop intrinsic");
673 Type *Ty = II.getType();
674 unsigned BitWidth = Ty->getScalarSizeInBits();
675 Value *Op0 = II.getArgOperand(0);
676 Value *X, *Y;
677
678 // ctpop(bitreverse(x)) -> ctpop(x)
679 // ctpop(bswap(x)) -> ctpop(x)
680 if (match(Op0, m_BitReverse(m_Value(X))) || match(Op0, m_BSwap(m_Value(X))))
681 return CallInst::Create(II.getCalledFunction(), X);
682
683 // ctpop(rot(x)) -> ctpop(x)
684 if ((match(Op0, m_FShl(m_Value(X), m_Value(Y), m_Value())) ||
685 match(Op0, m_FShr(m_Value(X), m_Value(Y), m_Value()))) &&
686 X == Y)
687 return CallInst::Create(II.getCalledFunction(), X);
688
689 // ctpop(x | -x) -> bitwidth - cttz(x, false)
690 if (Op0->hasOneUse() &&
691 match(Op0, m_c_Or(m_Value(X), m_Neg(m_Deferred(X))))) {
692 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
693 {X, IC.Builder.getFalse()});
694 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
695 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
696 }
697
698 // ctpop(~x & (x - 1)) -> cttz(x, false)
699 if (match(Op0,
701 Function *F =
702 Intrinsic::getOrInsertDeclaration(II.getModule(), Intrinsic::cttz, Ty);
703 return CallInst::Create(F, {X, IC.Builder.getFalse()});
704 }
705
706 // Zext doesn't change the number of set bits, so narrow:
707 // ctpop (zext X) --> zext (ctpop X)
708 if (match(Op0, m_OneUse(m_ZExt(m_Value(X))))) {
709 Value *NarrowPop = IC.Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, X);
710 return CastInst::Create(Instruction::ZExt, NarrowPop, Ty);
711 }
712
714 IC.computeKnownBits(Op0, Known, &II);
715
716 // If all bits are zero except for exactly one fixed bit, then the result
717 // must be 0 or 1, and we can get that answer by shifting to LSB:
718 // ctpop (X & 32) --> (X & 32) >> 5
719 // TODO: Investigate removing this as its likely unnecessary given the below
720 // `isKnownToBeAPowerOfTwo` check.
721 if ((~Known.Zero).isPowerOf2())
722 return BinaryOperator::CreateLShr(
723 Op0, ConstantInt::get(Ty, (~Known.Zero).exactLogBase2()));
724
725 // More generally we can also handle non-constant power of 2 patterns such as
726 // shl/shr(Pow2, X), (X & -X), etc... by transforming:
727 // ctpop(Pow2OrZero) --> icmp ne X, 0
728 if (IC.isKnownToBeAPowerOfTwo(Op0, /* OrZero */ true))
729 return CastInst::Create(Instruction::ZExt,
732 Ty);
733
734 // Add range attribute since known bits can't completely reflect what we know.
735 if (BitWidth != 1) {
736 ConstantRange OldRange =
737 II.getRange().value_or(ConstantRange::getFull(BitWidth));
738
739 unsigned Lower = Known.countMinPopulation();
740 unsigned Upper = Known.countMaxPopulation() + 1;
741
742 if (Lower == 0 && OldRange.contains(APInt::getZero(BitWidth)) &&
744 Lower = 1;
745
747 Range = Range.intersectWith(OldRange, ConstantRange::Unsigned);
748
749 if (Range != OldRange) {
750 II.addRangeRetAttr(Range);
751 return &II;
752 }
753 }
754
755 return nullptr;
756}
757
758/// Convert `tbl`/`tbx` intrinsics to shufflevector if the mask is constant, and
759/// at most two source operands are actually referenced.
761 bool IsExtension) {
762 // Bail out if the mask is not a constant.
763 auto *C = dyn_cast<Constant>(II.getArgOperand(II.arg_size() - 1));
764 if (!C)
765 return nullptr;
766
767 auto *RetTy = cast<FixedVectorType>(II.getType());
768 unsigned NumIndexes = RetTy->getNumElements();
769
770 // Only perform this transformation for <8 x i8> and <16 x i8> vector types.
771 if (!RetTy->getElementType()->isIntegerTy(8) ||
772 (NumIndexes != 8 && NumIndexes != 16))
773 return nullptr;
774
775 // For tbx instructions, the first argument is the "fallback" vector, which
776 // has the same length as the mask and return type.
777 unsigned int StartIndex = (unsigned)IsExtension;
778 auto *SourceTy =
779 cast<FixedVectorType>(II.getArgOperand(StartIndex)->getType());
780 // Note that the element count of each source vector does *not* need to be the
781 // same as the element count of the return type and mask! All source vectors
782 // must have the same element count as each other, though.
783 unsigned NumElementsPerSource = SourceTy->getNumElements();
784
785 // There are no tbl/tbx intrinsics for which the destination size exceeds the
786 // source size. However, our definitions of the intrinsics, at least in
787 // IntrinsicsAArch64.td, allow for arbitrary destination vector sizes, so it
788 // *could* technically happen.
789 if (NumIndexes > NumElementsPerSource)
790 return nullptr;
791
792 // The tbl/tbx intrinsics take several source operands followed by a mask
793 // operand.
794 unsigned int NumSourceOperands = II.arg_size() - 1 - (unsigned)IsExtension;
795
796 // Map input operands to shuffle indices. This also helpfully deduplicates the
797 // input arguments, in case the same value is passed as an argument multiple
798 // times.
799 SmallDenseMap<Value *, unsigned, 2> ValueToShuffleSlot;
800 Value *ShuffleOperands[2] = {PoisonValue::get(SourceTy),
801 PoisonValue::get(SourceTy)};
802
803 int Indexes[16];
804 for (unsigned I = 0; I < NumIndexes; ++I) {
805 Constant *COp = C->getAggregateElement(I);
806
807 if (!COp || (!isa<UndefValue>(COp) && !isa<ConstantInt>(COp)))
808 return nullptr;
809
810 if (isa<UndefValue>(COp)) {
811 Indexes[I] = -1;
812 continue;
813 }
814
815 uint64_t Index = cast<ConstantInt>(COp)->getZExtValue();
816 // The index of the input argument that this index references (0 = first
817 // source argument, etc).
818 unsigned SourceOperandIndex = Index / NumElementsPerSource;
819 // The index of the element at that source operand.
820 unsigned SourceOperandElementIndex = Index % NumElementsPerSource;
821
822 Value *SourceOperand;
823 if (SourceOperandIndex >= NumSourceOperands) {
824 // This index is out of bounds. Map it to index into either the fallback
825 // vector (tbx) or vector of zeroes (tbl).
826 SourceOperandIndex = NumSourceOperands;
827 if (IsExtension) {
828 // For out-of-bounds indices in tbx, choose the `I`th element of the
829 // fallback.
830 SourceOperand = II.getArgOperand(0);
831 SourceOperandElementIndex = I;
832 } else {
833 // Otherwise, choose some element from the dummy vector of zeroes (we'll
834 // always choose the first).
835 SourceOperand = Constant::getNullValue(SourceTy);
836 SourceOperandElementIndex = 0;
837 }
838 } else {
839 SourceOperand = II.getArgOperand(SourceOperandIndex + StartIndex);
840 }
841
842 // The source operand may be the fallback vector, which may not have the
843 // same number of elements as the source vector. In that case, we *could*
844 // choose to extend its length with another shufflevector, but it's simpler
845 // to just bail instead.
846 if (cast<FixedVectorType>(SourceOperand->getType())->getNumElements() !=
847 NumElementsPerSource)
848 return nullptr;
849
850 // We now know the source operand referenced by this index. Make it a
851 // shufflevector operand, if it isn't already.
852 unsigned NumSlots = ValueToShuffleSlot.size();
853 // This shuffle references more than two sources, and hence cannot be
854 // represented as a shufflevector.
855 if (NumSlots == 2 && !ValueToShuffleSlot.contains(SourceOperand))
856 return nullptr;
857
858 auto [It, Inserted] =
859 ValueToShuffleSlot.try_emplace(SourceOperand, NumSlots);
860 if (Inserted)
861 ShuffleOperands[It->getSecond()] = SourceOperand;
862
863 unsigned RemappedIndex =
864 (It->getSecond() * NumElementsPerSource) + SourceOperandElementIndex;
865 Indexes[I] = RemappedIndex;
866 }
867
869 ShuffleOperands[0], ShuffleOperands[1], ArrayRef(Indexes, NumIndexes));
870 return IC.replaceInstUsesWith(II, Shuf);
871}
872
873// Returns true iff the 2 intrinsics have the same operands, limiting the
874// comparison to the first NumOperands.
875static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E,
876 unsigned NumOperands) {
877 assert(I.arg_size() >= NumOperands && "Not enough operands");
878 assert(E.arg_size() >= NumOperands && "Not enough operands");
879 for (unsigned i = 0; i < NumOperands; i++)
880 if (I.getArgOperand(i) != E.getArgOperand(i))
881 return false;
882 return true;
883}
884
885// Remove trivially empty start/end intrinsic ranges, i.e. a start
886// immediately followed by an end (ignoring debuginfo or other
887// start/end intrinsics in between). As this handles only the most trivial
888// cases, tracking the nesting level is not needed:
889//
890// call @llvm.foo.start(i1 0)
891// call @llvm.foo.start(i1 0) ; This one won't be skipped: it will be removed
892// call @llvm.foo.end(i1 0)
893// call @llvm.foo.end(i1 0) ; &I
894static bool
896 std::function<bool(const IntrinsicInst &)> IsStart) {
897 // We start from the end intrinsic and scan backwards, so that InstCombine
898 // has already processed (and potentially removed) all the instructions
899 // before the end intrinsic.
900 BasicBlock::reverse_iterator BI(EndI), BE(EndI.getParent()->rend());
901 for (; BI != BE; ++BI) {
902 if (auto *I = dyn_cast<IntrinsicInst>(&*BI)) {
903 if (I->isDebugOrPseudoInst() ||
904 I->getIntrinsicID() == EndI.getIntrinsicID())
905 continue;
906 if (IsStart(*I)) {
907 if (haveSameOperands(EndI, *I, EndI.arg_size())) {
909 IC.eraseInstFromFunction(EndI);
910 return true;
911 }
912 // Skip start intrinsics that don't pair with this end intrinsic.
913 continue;
914 }
915 }
916 break;
917 }
918
919 return false;
920}
921
923 removeTriviallyEmptyRange(I, *this, [&I](const IntrinsicInst &II) {
924 // Bail out on the case where the source va_list of a va_copy is destroyed
925 // immediately by a follow-up va_end.
926 return II.getIntrinsicID() == Intrinsic::vastart ||
927 (II.getIntrinsicID() == Intrinsic::vacopy &&
928 I.getArgOperand(0) != II.getArgOperand(1));
929 });
930 return nullptr;
931}
932
934 assert(Call.arg_size() > 1 && "Need at least 2 args to swap");
935 Value *Arg0 = Call.getArgOperand(0), *Arg1 = Call.getArgOperand(1);
936 if (isa<Constant>(Arg0) && !isa<Constant>(Arg1)) {
937 Call.setArgOperand(0, Arg1);
938 Call.setArgOperand(1, Arg0);
939 AttributeList CallAttr = Call.getAttributes();
940 AttributeSet LHSAttr = CallAttr.getParamAttrs(0);
941 AttributeSet RHSAttr = CallAttr.getParamAttrs(1);
942 LLVMContext &Ctx = Call.getContext();
943 Call.setAttributes(CallAttr
944 .setAttributesAtIndex(
945 Ctx, AttributeList::FirstArgIndex + 0, RHSAttr)
946 .setAttributesAtIndex(
947 Ctx, AttributeList::FirstArgIndex + 1, LHSAttr));
948 return &Call;
949 }
950 return nullptr;
951}
952
953/// Creates a result tuple for an overflow intrinsic \p II with a given
954/// \p Result and a constant \p Overflow value.
956 Constant *Overflow) {
957 Constant *V[] = {PoisonValue::get(Result->getType()), Overflow};
958 StructType *ST = cast<StructType>(II->getType());
959 Constant *Struct = ConstantStruct::get(ST, V);
960 return InsertValueInst::Create(Struct, Result, 0);
961}
962
964InstCombinerImpl::foldIntrinsicWithOverflowCommon(IntrinsicInst *II) {
965 WithOverflowInst *WO = cast<WithOverflowInst>(II);
966 Value *OperationResult = nullptr;
967 Constant *OverflowResult = nullptr;
968 if (OptimizeOverflowCheck(WO->getBinaryOp(), WO->isSigned(), WO->getLHS(),
969 WO->getRHS(), *WO, OperationResult, OverflowResult))
970 return createOverflowTuple(WO, OperationResult, OverflowResult);
971
972 // See whether we can optimize the overflow check with assumption information.
973 for (User *U : WO->users()) {
974 if (!match(U, m_ExtractValue<1>(m_Value())))
975 continue;
976
977 for (auto &AssumeVH : AC.assumptionsFor(U)) {
978 if (!AssumeVH)
979 continue;
980 CallInst *I = cast<CallInst>(AssumeVH);
981 if (!match(I->getArgOperand(0), m_Not(m_Specific(U))))
982 continue;
983 if (!isValidAssumeForContext(I, II, /*DT=*/nullptr,
984 /*AllowEphemerals=*/true))
985 continue;
986 Value *Result =
987 Builder.CreateBinOp(WO->getBinaryOp(), WO->getLHS(), WO->getRHS());
988 Result->takeName(WO);
989 if (auto *Inst = dyn_cast<Instruction>(Result)) {
990 if (WO->isSigned())
991 Inst->setHasNoSignedWrap();
992 else
993 Inst->setHasNoUnsignedWrap();
994 }
995 return createOverflowTuple(WO, Result,
996 ConstantInt::getFalse(U->getType()));
997 }
998 }
999
1000 return nullptr;
1001}
1002
1003static bool inputDenormalIsIEEE(const Function &F, const Type *Ty) {
1004 Ty = Ty->getScalarType();
1005 return F.getDenormalMode(Ty->getFltSemantics()).Input == DenormalMode::IEEE;
1006}
1007
1008static bool inputDenormalIsDAZ(const Function &F, const Type *Ty) {
1009 Ty = Ty->getScalarType();
1010 return F.getDenormalMode(Ty->getFltSemantics()).inputsAreZero();
1011}
1012
1013/// \returns the compare predicate type if the test performed by
1014/// llvm.is.fpclass(x, \p Mask) is equivalent to fcmp o__ x, 0.0 with the
1015/// floating-point environment assumed for \p F for type \p Ty
1017 const Function &F, Type *Ty) {
1018 switch (static_cast<unsigned>(Mask)) {
1019 case fcZero:
1020 if (inputDenormalIsIEEE(F, Ty))
1021 return FCmpInst::FCMP_OEQ;
1022 break;
1023 case fcZero | fcSubnormal:
1024 if (inputDenormalIsDAZ(F, Ty))
1025 return FCmpInst::FCMP_OEQ;
1026 break;
1027 case fcPositive | fcNegZero:
1028 if (inputDenormalIsIEEE(F, Ty))
1029 return FCmpInst::FCMP_OGE;
1030 break;
1032 if (inputDenormalIsDAZ(F, Ty))
1033 return FCmpInst::FCMP_OGE;
1034 break;
1036 if (inputDenormalIsIEEE(F, Ty))
1037 return FCmpInst::FCMP_OGT;
1038 break;
1039 case fcNegative | fcPosZero:
1040 if (inputDenormalIsIEEE(F, Ty))
1041 return FCmpInst::FCMP_OLE;
1042 break;
1044 if (inputDenormalIsDAZ(F, Ty))
1045 return FCmpInst::FCMP_OLE;
1046 break;
1048 if (inputDenormalIsIEEE(F, Ty))
1049 return FCmpInst::FCMP_OLT;
1050 break;
1051 case fcPosNormal | fcPosInf:
1052 if (inputDenormalIsDAZ(F, Ty))
1053 return FCmpInst::FCMP_OGT;
1054 break;
1055 case fcNegNormal | fcNegInf:
1056 if (inputDenormalIsDAZ(F, Ty))
1057 return FCmpInst::FCMP_OLT;
1058 break;
1059 case ~fcZero & ~fcNan:
1060 if (inputDenormalIsIEEE(F, Ty))
1061 return FCmpInst::FCMP_ONE;
1062 break;
1063 case ~(fcZero | fcSubnormal) & ~fcNan:
1064 if (inputDenormalIsDAZ(F, Ty))
1065 return FCmpInst::FCMP_ONE;
1066 break;
1067 default:
1068 break;
1069 }
1070
1072}
1073
1074Instruction *InstCombinerImpl::foldIntrinsicIsFPClass(IntrinsicInst &II) {
1075 Value *Src0 = II.getArgOperand(0);
1076 Value *Src1 = II.getArgOperand(1);
1077 const ConstantInt *CMask = cast<ConstantInt>(Src1);
1078 FPClassTest Mask = static_cast<FPClassTest>(CMask->getZExtValue());
1079 const bool IsUnordered = (Mask & fcNan) == fcNan;
1080 const bool IsOrdered = (Mask & fcNan) == fcNone;
1081 const FPClassTest OrderedMask = Mask & ~fcNan;
1082 const FPClassTest OrderedInvertedMask = ~OrderedMask & ~fcNan;
1083
1084 const bool IsStrict =
1085 II.getFunction()->getAttributes().hasFnAttr(Attribute::StrictFP);
1086
1087 Value *FNegSrc;
1088 // is.fpclass (fneg x), mask -> is.fpclass x, (fneg mask)
1089 if (match(Src0, m_FNeg(m_Value(FNegSrc))))
1090 return CallInst::Create(
1091 II.getCalledFunction(),
1092 {FNegSrc, ConstantInt::get(Src1->getType(), fneg(Mask))});
1093
1094 Value *FAbsSrc;
1095 if (match(Src0, m_FAbs(m_Value(FAbsSrc))))
1096 return CallInst::Create(
1097 II.getCalledFunction(),
1098 {FAbsSrc, ConstantInt::get(Src1->getType(), inverse_fabs(Mask))});
1099
1100 if ((OrderedMask == fcInf || OrderedInvertedMask == fcInf) &&
1101 (IsOrdered || IsUnordered) && !IsStrict) {
1102 // is.fpclass(x, fcInf) -> fcmp oeq fabs(x), +inf
1103 // is.fpclass(x, ~fcInf) -> fcmp one fabs(x), +inf
1104 // is.fpclass(x, fcInf|fcNan) -> fcmp ueq fabs(x), +inf
1105 // is.fpclass(x, ~(fcInf|fcNan)) -> fcmp une fabs(x), +inf
1107 FCmpInst::Predicate Pred =
1108 IsUnordered ? FCmpInst::FCMP_UEQ : FCmpInst::FCMP_OEQ;
1109 if (OrderedInvertedMask == fcInf)
1110 Pred = IsUnordered ? FCmpInst::FCMP_UNE : FCmpInst::FCMP_ONE;
1111
1112 Value *Fabs = Builder.CreateFAbs(Src0);
1113 Value *CmpInf = Builder.CreateFCmp(Pred, Fabs, Inf);
1114 CmpInf->takeName(&II);
1115 return replaceInstUsesWith(II, CmpInf);
1116 }
1117
1118 if ((OrderedMask == fcPosInf || OrderedMask == fcNegInf) &&
1119 (IsOrdered || IsUnordered) && !IsStrict) {
1120 // is.fpclass(x, fcPosInf) -> fcmp oeq x, +inf
1121 // is.fpclass(x, fcNegInf) -> fcmp oeq x, -inf
1122 // is.fpclass(x, fcPosInf|fcNan) -> fcmp ueq x, +inf
1123 // is.fpclass(x, fcNegInf|fcNan) -> fcmp ueq x, -inf
1124 Constant *Inf =
1125 ConstantFP::getInfinity(Src0->getType(), OrderedMask == fcNegInf);
1126 Value *EqInf = IsUnordered ? Builder.CreateFCmpUEQ(Src0, Inf)
1127 : Builder.CreateFCmpOEQ(Src0, Inf);
1128
1129 EqInf->takeName(&II);
1130 return replaceInstUsesWith(II, EqInf);
1131 }
1132
1133 if ((OrderedInvertedMask == fcPosInf || OrderedInvertedMask == fcNegInf) &&
1134 (IsOrdered || IsUnordered) && !IsStrict) {
1135 // is.fpclass(x, ~fcPosInf) -> fcmp one x, +inf
1136 // is.fpclass(x, ~fcNegInf) -> fcmp one x, -inf
1137 // is.fpclass(x, ~fcPosInf|fcNan) -> fcmp une x, +inf
1138 // is.fpclass(x, ~fcNegInf|fcNan) -> fcmp une x, -inf
1140 OrderedInvertedMask == fcNegInf);
1141 Value *NeInf = IsUnordered ? Builder.CreateFCmpUNE(Src0, Inf)
1142 : Builder.CreateFCmpONE(Src0, Inf);
1143 NeInf->takeName(&II);
1144 return replaceInstUsesWith(II, NeInf);
1145 }
1146
1147 if (Mask == fcNan && !IsStrict) {
1148 // Equivalent of isnan. Replace with standard fcmp if we don't care about FP
1149 // exceptions.
1150 Value *IsNan =
1151 Builder.CreateFCmpUNO(Src0, ConstantFP::getZero(Src0->getType()));
1152 IsNan->takeName(&II);
1153 return replaceInstUsesWith(II, IsNan);
1154 }
1155
1156 if (Mask == (~fcNan & fcAllFlags) && !IsStrict) {
1157 // Equivalent of !isnan. Replace with standard fcmp.
1158 Value *FCmp =
1159 Builder.CreateFCmpORD(Src0, ConstantFP::getZero(Src0->getType()));
1160 FCmp->takeName(&II);
1161 return replaceInstUsesWith(II, FCmp);
1162 }
1163
1165
1166 // Try to replace with an fcmp with 0
1167 //
1168 // is.fpclass(x, fcZero) -> fcmp oeq x, 0.0
1169 // is.fpclass(x, fcZero | fcNan) -> fcmp ueq x, 0.0
1170 // is.fpclass(x, ~fcZero & ~fcNan) -> fcmp one x, 0.0
1171 // is.fpclass(x, ~fcZero) -> fcmp une x, 0.0
1172 //
1173 // is.fpclass(x, fcPosSubnormal | fcPosNormal | fcPosInf) -> fcmp ogt x, 0.0
1174 // is.fpclass(x, fcPositive | fcNegZero) -> fcmp oge x, 0.0
1175 //
1176 // is.fpclass(x, fcNegSubnormal | fcNegNormal | fcNegInf) -> fcmp olt x, 0.0
1177 // is.fpclass(x, fcNegative | fcPosZero) -> fcmp ole x, 0.0
1178 //
1179 if (!IsStrict && (IsOrdered || IsUnordered) &&
1180 (PredType = fpclassTestIsFCmp0(OrderedMask, *II.getFunction(),
1181 Src0->getType())) !=
1184 // Equivalent of == 0.
1185 Value *FCmp = Builder.CreateFCmp(
1186 IsUnordered ? FCmpInst::getUnorderedPredicate(PredType) : PredType,
1187 Src0, Zero);
1188
1189 FCmp->takeName(&II);
1190 return replaceInstUsesWith(II, FCmp);
1191 }
1192
1193 KnownFPClass Known =
1194 computeKnownFPClass(Src0, Mask, SQ.getWithInstruction(&II));
1195
1196 // If none of the tests which can return false are possible, fold to true.
1197 // fp_class (nnan x), ~(qnan|snan) -> true
1198 // fp_class (ninf x), ~(ninf|pinf) -> true
1199 if (Known.isKnownAlways(Mask))
1200 return replaceInstUsesWith(II, ConstantInt::get(II.getType(), true));
1201
1202 // Clear test bits we know must be false from the source value.
1203 // fp_class (nnan x), qnan|snan|other -> fp_class (nnan x), other
1204 // fp_class (ninf x), ninf|pinf|other -> fp_class (ninf x), other
1205 if ((Mask & Known.KnownFPClasses) != Mask) {
1206 II.setArgOperand(
1207 1, ConstantInt::get(Src1->getType(), Mask & Known.KnownFPClasses));
1208 return &II;
1209 }
1210
1211 return nullptr;
1212}
1213
1214static std::optional<bool> getKnownSign(Value *Op, const SimplifyQuery &SQ) {
1216 if (Known.isNonNegative())
1217 return false;
1218 if (Known.isNegative())
1219 return true;
1220
1221 Value *X, *Y;
1222 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1224
1225 return std::nullopt;
1226}
1227
1228static std::optional<bool> getKnownSignOrZero(Value *Op,
1229 const SimplifyQuery &SQ) {
1230 if (std::optional<bool> Sign = getKnownSign(Op, SQ))
1231 return Sign;
1232
1233 Value *X, *Y;
1234 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1236
1237 return std::nullopt;
1238}
1239
1240/// Return true if two values \p Op0 and \p Op1 are known to have the same sign.
1241static bool signBitMustBeTheSame(Value *Op0, Value *Op1,
1242 const SimplifyQuery &SQ) {
1243 std::optional<bool> Known1 = getKnownSign(Op1, SQ);
1244 if (!Known1)
1245 return false;
1246 std::optional<bool> Known0 = getKnownSign(Op0, SQ);
1247 if (!Known0)
1248 return false;
1249 return *Known0 == *Known1;
1250}
1251
1252// Determines if ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b)) is safe.
1253//
1254// This is true if, when the add saturates, the resulting ldexp is guaranteed to
1255// produce 0 or inf.
1256static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy) {
1257 const fltSemantics &FltSem = FpTy->getScalarType()->getFltSemantics();
1258 if (!APFloat::semanticsHasInf(FltSem))
1259 return false;
1260
1261 // Cap ExpBits at 32 because scalbn takes an int. This is sufficient for any
1262 // reasonable fp type (for example, `double` only has 11 exponent bits).
1263 unsigned ExpBits = std::min(ExpTy->getScalarSizeInBits(), 32u);
1264 int SignedMax = static_cast<int>(maxIntN(ExpBits));
1265 int SignedMin = static_cast<int>(minIntN(ExpBits));
1266 APFloat ScaledUp = scalbn(APFloat::getSmallest(FltSem), SignedMax,
1268 APFloat ScaledDown = scalbn(APFloat::getLargest(FltSem), SignedMin,
1270 return ScaledUp.isInfinity() && ScaledDown.isZero();
1271}
1272
1273/// Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0. This
1274/// can trigger other combines.
1276 InstCombiner::BuilderTy &Builder) {
1277 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1278 assert((MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin ||
1279 MinMaxID == Intrinsic::umax || MinMaxID == Intrinsic::umin) &&
1280 "Expected a min or max intrinsic");
1281
1282 // TODO: Match vectors with undef elements, but undef may not propagate.
1283 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
1284 Value *X;
1285 const APInt *C0, *C1;
1286 if (!match(Op0, m_OneUse(m_Add(m_Value(X), m_APInt(C0)))) ||
1287 !match(Op1, m_APInt(C1)))
1288 return nullptr;
1289
1290 // Check for necessary no-wrap and overflow constraints.
1291 bool IsSigned = MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin;
1292 auto *Add = cast<BinaryOperator>(Op0);
1293 if ((IsSigned && !Add->hasNoSignedWrap()) ||
1294 (!IsSigned && !Add->hasNoUnsignedWrap()))
1295 return nullptr;
1296
1297 // If the constant difference overflows, then instsimplify should reduce the
1298 // min/max to the add or C1.
1299 bool Overflow;
1300 APInt CDiff =
1301 IsSigned ? C1->ssub_ov(*C0, Overflow) : C1->usub_ov(*C0, Overflow);
1302 assert(!Overflow && "Expected simplify of min/max");
1303
1304 // min/max (add X, C0), C1 --> add (min/max X, C1 - C0), C0
1305 // Note: the "mismatched" no-overflow setting does not propagate.
1306 Constant *NewMinMaxC = ConstantInt::get(II->getType(), CDiff);
1307 Value *NewMinMax = Builder.CreateBinaryIntrinsic(MinMaxID, X, NewMinMaxC);
1308 return IsSigned ? BinaryOperator::CreateNSWAdd(NewMinMax, Add->getOperand(1))
1309 : BinaryOperator::CreateNUWAdd(NewMinMax, Add->getOperand(1));
1310}
1311/// Match a sadd_sat or ssub_sat which is using min/max to clamp the value.
1312Instruction *InstCombinerImpl::matchSAddSubSat(IntrinsicInst &MinMax1) {
1313 Type *Ty = MinMax1.getType();
1314
1315 // We are looking for a tree of:
1316 // max(INT_MIN, min(INT_MAX, add(sext(A), sext(B))))
1317 // Where the min and max could be reversed
1318 Instruction *MinMax2;
1319 BinaryOperator *AddSub;
1320 const APInt *MinValue, *MaxValue;
1321 if (match(&MinMax1, m_SMin(m_Instruction(MinMax2), m_APInt(MaxValue)))) {
1322 if (!match(MinMax2, m_SMax(m_BinOp(AddSub), m_APInt(MinValue))))
1323 return nullptr;
1324 } else if (match(&MinMax1,
1325 m_SMax(m_Instruction(MinMax2), m_APInt(MinValue)))) {
1326 if (!match(MinMax2, m_SMin(m_BinOp(AddSub), m_APInt(MaxValue))))
1327 return nullptr;
1328 } else
1329 return nullptr;
1330
1331 // Check that the constants clamp a saturate, and that the new type would be
1332 // sensible to convert to.
1333 if (!(*MaxValue + 1).isPowerOf2() || -*MinValue != *MaxValue + 1)
1334 return nullptr;
1335 // In what bitwidth can this be treated as saturating arithmetics?
1336 unsigned NewBitWidth = (*MaxValue + 1).logBase2() + 1;
1337 // FIXME: This isn't quite right for vectors, but using the scalar type is a
1338 // good first approximation for what should be done there.
1339 if (!shouldChangeType(Ty->getScalarType()->getIntegerBitWidth(), NewBitWidth))
1340 return nullptr;
1341
1342 // Also make sure that the inner min/max and the add/sub have one use.
1343 if (!MinMax2->hasOneUse() || !AddSub->hasOneUse())
1344 return nullptr;
1345
1346 // Create the new type (which can be a vector type)
1347 Type *NewTy = Ty->getWithNewBitWidth(NewBitWidth);
1348
1349 Intrinsic::ID IntrinsicID;
1350 if (AddSub->getOpcode() == Instruction::Add)
1351 IntrinsicID = Intrinsic::sadd_sat;
1352 else if (AddSub->getOpcode() == Instruction::Sub)
1353 IntrinsicID = Intrinsic::ssub_sat;
1354 else
1355 return nullptr;
1356
1357 // The two operands of the add/sub must be nsw-truncatable to the NewTy. This
1358 // is usually achieved via a sext from a smaller type.
1359 if (ComputeMaxSignificantBits(AddSub->getOperand(0), AddSub) > NewBitWidth ||
1360 ComputeMaxSignificantBits(AddSub->getOperand(1), AddSub) > NewBitWidth)
1361 return nullptr;
1362
1363 // Finally create and return the sat intrinsic, truncated to the new type
1364 Value *AT = Builder.CreateTrunc(AddSub->getOperand(0), NewTy);
1365 Value *BT = Builder.CreateTrunc(AddSub->getOperand(1), NewTy);
1366 Value *Sat = Builder.CreateIntrinsic(IntrinsicID, NewTy, {AT, BT});
1367 return CastInst::Create(Instruction::SExt, Sat, Ty);
1368}
1369
1370
1371/// If we have a clamp pattern like max (min X, 42), 41 -- where the output
1372/// can only be one of two possible constant values -- turn that into a select
1373/// of constants.
1375 InstCombiner::BuilderTy &Builder) {
1376 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
1377 Value *X;
1378 const APInt *C0, *C1;
1379 if (!match(I1, m_APInt(C1)) || !I0->hasOneUse())
1380 return nullptr;
1381
1383 switch (II->getIntrinsicID()) {
1384 case Intrinsic::smax:
1385 if (match(I0, m_SMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1386 Pred = ICmpInst::ICMP_SGT;
1387 break;
1388 case Intrinsic::smin:
1389 if (match(I0, m_SMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1390 Pred = ICmpInst::ICMP_SLT;
1391 break;
1392 case Intrinsic::umax:
1393 if (match(I0, m_UMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1394 Pred = ICmpInst::ICMP_UGT;
1395 break;
1396 case Intrinsic::umin:
1397 if (match(I0, m_UMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1398 Pred = ICmpInst::ICMP_ULT;
1399 break;
1400 default:
1401 llvm_unreachable("Expected min/max intrinsic");
1402 }
1403 if (Pred == CmpInst::BAD_ICMP_PREDICATE)
1404 return nullptr;
1405
1406 // max (min X, 42), 41 --> X > 41 ? 42 : 41
1407 // min (max X, 42), 43 --> X < 43 ? 42 : 43
1408 Value *Cmp = Builder.CreateICmp(Pred, X, I1);
1409 return SelectInst::Create(Cmp, ConstantInt::get(II->getType(), *C0), I1);
1410}
1411
1412/// If this min/max has a constant operand and an operand that is a matching
1413/// min/max with a constant operand, constant-fold the 2 constant operands.
1415 IRBuilderBase &Builder,
1416 const SimplifyQuery &SQ) {
1417 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1418 auto *LHS = dyn_cast<MinMaxIntrinsic>(II->getArgOperand(0));
1419 if (!LHS)
1420 return nullptr;
1421
1422 Constant *C0, *C1;
1423 if (!match(LHS->getArgOperand(1), m_ImmConstant(C0)) ||
1424 !match(II->getArgOperand(1), m_ImmConstant(C1)))
1425 return nullptr;
1426
1427 // max (max X, C0), C1 --> max X, (max C0, C1)
1428 // min (min X, C0), C1 --> min X, (min C0, C1)
1429 // umax (smax X, nneg C0), nneg C1 --> smax X, (umax C0, C1)
1430 // smin (umin X, nneg C0), nneg C1 --> umin X, (smin C0, C1)
1431 Intrinsic::ID InnerMinMaxID = LHS->getIntrinsicID();
1432 if (InnerMinMaxID != MinMaxID &&
1433 !(((MinMaxID == Intrinsic::umax && InnerMinMaxID == Intrinsic::smax) ||
1434 (MinMaxID == Intrinsic::smin && InnerMinMaxID == Intrinsic::umin)) &&
1435 isKnownNonNegative(C0, SQ) && isKnownNonNegative(C1, SQ)))
1436 return nullptr;
1437
1439 Value *CondC = Builder.CreateICmp(Pred, C0, C1);
1440 Value *NewC = Builder.CreateSelect(CondC, C0, C1);
1441 return Builder.CreateIntrinsic(InnerMinMaxID, II->getType(),
1442 {LHS->getArgOperand(0), NewC});
1443}
1444
1445/// If this min/max has a matching min/max operand with a constant, try to push
1446/// the constant operand into this instruction. This can enable more folds.
1447static Instruction *
1449 InstCombiner::BuilderTy &Builder) {
1450 // Match and capture a min/max operand candidate.
1451 Value *X, *Y;
1452 Constant *C;
1453 Instruction *Inner;
1455 m_Instruction(Inner),
1457 m_Value(Y))))
1458 return nullptr;
1459
1460 // The inner op must match. Check for constants to avoid infinite loops.
1461 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1462 auto *InnerMM = dyn_cast<IntrinsicInst>(Inner);
1463 if (!InnerMM || InnerMM->getIntrinsicID() != MinMaxID ||
1465 return nullptr;
1466
1467 // max (max X, C), Y --> max (max X, Y), C
1469 MinMaxID, II->getType());
1470 Value *NewInner = Builder.CreateBinaryIntrinsic(MinMaxID, X, Y);
1471 NewInner->takeName(Inner);
1472 return CallInst::Create(MinMax, {NewInner, C});
1473}
1474
1475/// Reduce a sequence of min/max intrinsics with a common operand.
1477 // Match 3 of the same min/max ops. Example: umin(umin(), umin()).
1478 auto *LHS = dyn_cast<IntrinsicInst>(II->getArgOperand(0));
1479 auto *RHS = dyn_cast<IntrinsicInst>(II->getArgOperand(1));
1480 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1481 if (!LHS || !RHS || LHS->getIntrinsicID() != MinMaxID ||
1482 RHS->getIntrinsicID() != MinMaxID ||
1483 (!LHS->hasOneUse() && !RHS->hasOneUse()))
1484 return nullptr;
1485
1486 Value *A = LHS->getArgOperand(0);
1487 Value *B = LHS->getArgOperand(1);
1488 Value *C = RHS->getArgOperand(0);
1489 Value *D = RHS->getArgOperand(1);
1490
1491 // Look for a common operand.
1492 Value *MinMaxOp = nullptr;
1493 Value *ThirdOp = nullptr;
1494 if (LHS->hasOneUse()) {
1495 // If the LHS is only used in this chain and the RHS is used outside of it,
1496 // reuse the RHS min/max because that will eliminate the LHS.
1497 if (D == A || C == A) {
1498 // min(min(a, b), min(c, a)) --> min(min(c, a), b)
1499 // min(min(a, b), min(a, d)) --> min(min(a, d), b)
1500 MinMaxOp = RHS;
1501 ThirdOp = B;
1502 } else if (D == B || C == B) {
1503 // min(min(a, b), min(c, b)) --> min(min(c, b), a)
1504 // min(min(a, b), min(b, d)) --> min(min(b, d), a)
1505 MinMaxOp = RHS;
1506 ThirdOp = A;
1507 }
1508 } else {
1509 assert(RHS->hasOneUse() && "Expected one-use operand");
1510 // Reuse the LHS. This will eliminate the RHS.
1511 if (D == A || D == B) {
1512 // min(min(a, b), min(c, a)) --> min(min(a, b), c)
1513 // min(min(a, b), min(c, b)) --> min(min(a, b), c)
1514 MinMaxOp = LHS;
1515 ThirdOp = C;
1516 } else if (C == A || C == B) {
1517 // min(min(a, b), min(b, d)) --> min(min(a, b), d)
1518 // min(min(a, b), min(c, b)) --> min(min(a, b), d)
1519 MinMaxOp = LHS;
1520 ThirdOp = D;
1521 }
1522 }
1523
1524 if (!MinMaxOp || !ThirdOp)
1525 return nullptr;
1526
1527 Module *Mod = II->getModule();
1528 Function *MinMax =
1529 Intrinsic::getOrInsertDeclaration(Mod, MinMaxID, II->getType());
1530 return CallInst::Create(MinMax, { MinMaxOp, ThirdOp });
1531}
1532
1533/// If all arguments of the intrinsic are unary shuffles with the same mask,
1534/// try to shuffle after the intrinsic.
1537 if (!II->getType()->isVectorTy() ||
1538 !isTriviallyVectorizable(II->getIntrinsicID()) ||
1539 !II->getCalledFunction()->isSpeculatable())
1540 return nullptr;
1541
1542 Value *X;
1543 Constant *C;
1544 ArrayRef<int> Mask;
1545 auto *NonConstArg = find_if_not(II->args(), [&II](Use &Arg) {
1546 return isa<Constant>(Arg.get()) ||
1547 isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1548 Arg.getOperandNo(), nullptr);
1549 });
1550 if (!NonConstArg ||
1551 !match(NonConstArg, m_Shuffle(m_Value(X), m_Poison(), m_Mask(Mask))))
1552 return nullptr;
1553
1554 // At least 1 operand must be a shuffle with 1 use because we are creating 2
1555 // instructions.
1556 if (none_of(II->args(), match_fn(m_OneUse(m_Shuffle(m_Value(), m_Value())))))
1557 return nullptr;
1558
1559 // See if all arguments are shuffled with the same mask.
1561 Type *SrcTy = X->getType();
1562 for (Use &Arg : II->args()) {
1563 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1564 Arg.getOperandNo(), nullptr))
1565 NewArgs.push_back(Arg);
1566 else if (match(&Arg,
1567 m_Shuffle(m_Value(X), m_Poison(), m_SpecificMask(Mask))) &&
1568 X->getType() == SrcTy)
1569 NewArgs.push_back(X);
1570 else if (match(&Arg, m_ImmConstant(C))) {
1571 // If it's a constant, try find the constant that would be shuffled to C.
1572 if (Constant *ShuffledC =
1573 unshuffleConstant(Mask, C, cast<VectorType>(SrcTy)))
1574 NewArgs.push_back(ShuffledC);
1575 else
1576 return nullptr;
1577 } else
1578 return nullptr;
1579 }
1580
1581 // intrinsic (shuf X, M), (shuf Y, M), ... --> shuf (intrinsic X, Y, ...), M
1582 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1583 // Result type might be a different vector width.
1584 // TODO: Check that the result type isn't widened?
1585 VectorType *ResTy =
1586 VectorType::get(II->getType()->getScalarType(), cast<VectorType>(SrcTy));
1587 Value *NewIntrinsic =
1588 Builder.CreateIntrinsic(ResTy, II->getIntrinsicID(), NewArgs, FPI);
1589 return new ShuffleVectorInst(NewIntrinsic, Mask);
1590}
1591
1592/// If all arguments of the intrinsic are reverses, try to pull the reverse
1593/// after the intrinsic.
1595 if (!II->getType()->isVectorTy() ||
1596 !isTriviallyVectorizable(II->getIntrinsicID()))
1597 return nullptr;
1598
1599 // At least 1 operand must be a reverse with 1 use because we are creating 2
1600 // instructions.
1601 if (none_of(II->args(), [](Value *V) {
1602 return match(V, m_OneUse(m_VecReverse(m_Value())));
1603 }))
1604 return nullptr;
1605
1606 Value *X;
1607 Constant *C;
1608 SmallVector<Value *> NewArgs;
1609 for (Use &Arg : II->args()) {
1610 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1611 Arg.getOperandNo(), nullptr))
1612 NewArgs.push_back(Arg);
1613 else if (match(&Arg, m_VecReverse(m_Value(X))))
1614 NewArgs.push_back(X);
1615 else if (isSplatValue(Arg))
1616 NewArgs.push_back(Arg);
1617 else if (match(&Arg, m_ImmConstant(C)))
1618 NewArgs.push_back(Builder.CreateVectorReverse(C));
1619 else
1620 return nullptr;
1621 }
1622
1623 // intrinsic (reverse X), (reverse Y), ... --> reverse (intrinsic X, Y, ...)
1624 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1625 Value *NewIntrinsic = Builder.CreateIntrinsic(
1626 II->getType(), II->getIntrinsicID(), NewArgs, FPI);
1627 return Builder.CreateVectorReverse(NewIntrinsic);
1628}
1629
1630/// Fold the following cases and accepts bswap and bitreverse intrinsics:
1631/// bswap(logic_op(bswap(x), y)) --> logic_op(x, bswap(y))
1632/// bswap(logic_op(bswap(x), bswap(y))) --> logic_op(x, y) (ignores multiuse)
1633template <Intrinsic::ID IntrID>
1635 InstCombiner::BuilderTy &Builder) {
1636 static_assert(IntrID == Intrinsic::bswap || IntrID == Intrinsic::bitreverse,
1637 "This helper only supports BSWAP and BITREVERSE intrinsics");
1638
1639 Value *X, *Y;
1640 // Find bitwise logic op. Check that it is a BinaryOperator explicitly so we
1641 // don't match ConstantExpr that aren't meaningful for this transform.
1644 Value *OldReorderX, *OldReorderY;
1646
1647 // If both X and Y are bswap/bitreverse, the transform reduces the number
1648 // of instructions even if there's multiuse.
1649 // If only one operand is bswap/bitreverse, we need to ensure the operand
1650 // have only one use.
1651 if (match(X, m_Intrinsic<IntrID>(m_Value(OldReorderX))) &&
1652 match(Y, m_Intrinsic<IntrID>(m_Value(OldReorderY)))) {
1653 return BinaryOperator::Create(Op, OldReorderX, OldReorderY);
1654 }
1655
1656 if (match(X, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderX))))) {
1657 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, Y);
1658 return BinaryOperator::Create(Op, OldReorderX, NewReorder);
1659 }
1660
1661 if (match(Y, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderY))))) {
1662 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, X);
1663 return BinaryOperator::Create(Op, NewReorder, OldReorderY);
1664 }
1665 }
1666 return nullptr;
1667}
1668
1669/// Helper to match idempotent binary intrinsics, namely, intrinsics where
1670/// `f(f(x, y), y) == f(x, y)` holds.
1672 switch (IID) {
1673 case Intrinsic::smax:
1674 case Intrinsic::smin:
1675 case Intrinsic::umax:
1676 case Intrinsic::umin:
1677 case Intrinsic::maximum:
1678 case Intrinsic::minimum:
1679 case Intrinsic::maximumnum:
1680 case Intrinsic::minimumnum:
1681 case Intrinsic::maxnum:
1682 case Intrinsic::minnum:
1683 return true;
1684 default:
1685 return false;
1686 }
1687}
1688
1689/// Attempt to simplify value-accumulating recurrences of kind:
1690/// %umax.acc = phi i8 [ %umax, %backedge ], [ %a, %entry ]
1691/// %umax = call i8 @llvm.umax.i8(i8 %umax.acc, i8 %b)
1692/// And let the idempotent binary intrinsic be hoisted, when the operands are
1693/// known to be loop-invariant.
1695 IntrinsicInst *II) {
1696 PHINode *PN;
1697 Value *Init, *OtherOp;
1698
1699 // A binary intrinsic recurrence with loop-invariant operands is equivalent to
1700 // `call @llvm.binary.intrinsic(Init, OtherOp)`.
1701 auto IID = II->getIntrinsicID();
1702 if (!isIdempotentBinaryIntrinsic(IID) ||
1704 !IC.getDominatorTree().dominates(OtherOp, PN))
1705 return nullptr;
1706
1707 auto *InvariantBinaryInst =
1708 IC.Builder.CreateBinaryIntrinsic(IID, Init, OtherOp);
1709 if (isa<FPMathOperator>(InvariantBinaryInst))
1710 cast<Instruction>(InvariantBinaryInst)->copyFastMathFlags(II);
1711 return InvariantBinaryInst;
1712}
1713
1714static Value *simplifyReductionOperand(Value *Arg, bool CanReorderLanes) {
1715 if (!CanReorderLanes)
1716 return nullptr;
1717
1718 Value *V;
1719 if (match(Arg, m_VecReverse(m_Value(V))))
1720 return V;
1721
1722 ArrayRef<int> Mask;
1723 if (!isa<FixedVectorType>(Arg->getType()) ||
1724 !match(Arg, m_Shuffle(m_Value(V), m_Undef(), m_Mask(Mask))) ||
1725 !cast<ShuffleVectorInst>(Arg)->isSingleSource())
1726 return nullptr;
1727
1728 int Sz = Mask.size();
1729 SmallBitVector UsedIndices(Sz);
1730 for (int Idx : Mask) {
1731 if (Idx == PoisonMaskElem || UsedIndices.test(Idx))
1732 return nullptr;
1733 UsedIndices.set(Idx);
1734 }
1735
1736 // Can remove shuffle iff just shuffled elements, no repeats, undefs, or
1737 // other changes.
1738 return UsedIndices.all() ? V : nullptr;
1739}
1740
1741/// Fold an unsigned minimum of trailing or leading zero bits counts:
1742/// umin(cttz(CtOp1, ZeroUndef), ConstOp) --> cttz(CtOp1 | (1 << ConstOp))
1743/// umin(ctlz(CtOp1, ZeroUndef), ConstOp) --> ctlz(CtOp1 | (SignedMin
1744/// >> ConstOp))
1745/// umin(cttz(CtOp1), cttz(CtOp2)) --> cttz(CtOp1 | CtOp2)
1746/// umin(ctlz(CtOp1), ctlz(CtOp2)) --> ctlz(CtOp1 | CtOp2)
1747template <Intrinsic::ID IntrID>
1748static Value *
1750 const DataLayout &DL,
1751 InstCombiner::BuilderTy &Builder) {
1752 static_assert(IntrID == Intrinsic::cttz || IntrID == Intrinsic::ctlz,
1753 "This helper only supports cttz and ctlz intrinsics");
1754
1755 Value *CtOp1, *CtOp2;
1756 Value *ZeroUndef1, *ZeroUndef2;
1757 if (!match(I0, m_OneUse(
1758 m_Intrinsic<IntrID>(m_Value(CtOp1), m_Value(ZeroUndef1)))))
1759 return nullptr;
1760
1761 if (match(I1,
1762 m_OneUse(m_Intrinsic<IntrID>(m_Value(CtOp2), m_Value(ZeroUndef2)))))
1763 return Builder.CreateBinaryIntrinsic(
1764 IntrID, Builder.CreateOr(CtOp1, CtOp2),
1765 Builder.CreateOr(ZeroUndef1, ZeroUndef2));
1766
1767 unsigned BitWidth = I1->getType()->getScalarSizeInBits();
1768 auto LessBitWidth = [BitWidth](auto &C) { return C.ult(BitWidth); };
1769 if (!match(I1, m_CheckedInt(LessBitWidth)))
1770 // We have a constant >= BitWidth (which can be handled by CVP)
1771 // or a non-splat vector with elements < and >= BitWidth
1772 return nullptr;
1773
1774 Type *Ty = I1->getType();
1776 IntrID == Intrinsic::cttz ? Instruction::Shl : Instruction::LShr,
1777 IntrID == Intrinsic::cttz
1778 ? ConstantInt::get(Ty, 1)
1779 : ConstantInt::get(Ty, APInt::getSignedMinValue(BitWidth)),
1780 cast<Constant>(I1), DL);
1781 return Builder.CreateBinaryIntrinsic(
1782 IntrID, Builder.CreateOr(CtOp1, NewConst),
1783 ConstantInt::getTrue(ZeroUndef1->getType()));
1784}
1785
1786/// Return whether "X LOp (Y ROp Z)" is always equal to
1787/// "(X LOp Y) ROp (X LOp Z)".
1789 bool HasNSW, Intrinsic::ID ROp) {
1790 switch (ROp) {
1791 case Intrinsic::umax:
1792 case Intrinsic::umin:
1793 if (HasNUW && LOp == Instruction::Add)
1794 return true;
1795 if (HasNUW && LOp == Instruction::Shl)
1796 return true;
1797 return false;
1798 case Intrinsic::smax:
1799 case Intrinsic::smin:
1800 return HasNSW && LOp == Instruction::Add;
1801 default:
1802 return false;
1803 }
1804}
1805
1806/// Return whether "(X ROp Y) LOp Z" is always equal to
1807/// "(X LOp Z) ROp (Y LOp Z)".
1809 bool HasNSW, Intrinsic::ID ROp) {
1810 if (Instruction::isCommutative(LOp) || LOp == Instruction::Shl)
1811 return leftDistributesOverRight(LOp, HasNUW, HasNSW, ROp);
1812 switch (ROp) {
1813 case Intrinsic::umax:
1814 case Intrinsic::umin:
1815 return HasNUW && LOp == Instruction::Sub;
1816 case Intrinsic::smax:
1817 case Intrinsic::smin:
1818 return HasNSW && LOp == Instruction::Sub;
1819 default:
1820 return false;
1821 }
1822}
1823
1824// Attempts to factorise a common term
1825// in an instruction that has the form "(A op' B) op (C op' D)
1826// where op is an intrinsic and op' is a binop
1827static Value *
1829 InstCombiner::BuilderTy &Builder) {
1830 Value *LHS = II->getOperand(0), *RHS = II->getOperand(1);
1831 Intrinsic::ID TopLevelOpcode = II->getIntrinsicID();
1832
1835
1836 if (!Op0 || !Op1)
1837 return nullptr;
1838
1839 if (Op0->getOpcode() != Op1->getOpcode())
1840 return nullptr;
1841
1842 if (!Op0->hasOneUse() || !Op1->hasOneUse())
1843 return nullptr;
1844
1845 Instruction::BinaryOps InnerOpcode =
1846 static_cast<Instruction::BinaryOps>(Op0->getOpcode());
1847 bool HasNUW = Op0->hasNoUnsignedWrap() && Op1->hasNoUnsignedWrap();
1848 bool HasNSW = Op0->hasNoSignedWrap() && Op1->hasNoSignedWrap();
1849
1850 Value *A = Op0->getOperand(0);
1851 Value *B = Op0->getOperand(1);
1852 Value *C = Op1->getOperand(0);
1853 Value *D = Op1->getOperand(1);
1854
1855 // Attempts to swap variables such that A equals C or B equals D,
1856 // if the inner operation is commutative.
1857 if (Op0->isCommutative() && A != C && B != D) {
1858 if (A == D || B == C)
1859 std::swap(C, D);
1860 else
1861 return nullptr;
1862 }
1863
1864 if (A == C &&
1865 leftDistributesOverRight(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1866 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, B, D);
1867 return Builder.CreateNoWrapBinOp(InnerOpcode, A, NewIntrinsic, HasNUW,
1868 HasNSW);
1869 }
1870 if (B == D &&
1871 rightDistributesOverLeft(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1872 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, A, C);
1873 return Builder.CreateNoWrapBinOp(InnerOpcode, NewIntrinsic, B, HasNUW,
1874 HasNSW);
1875 }
1876 return nullptr;
1877}
1878
1880 Value *Arg0 = II->getArgOperand(0);
1881 auto *ShiftConst = dyn_cast<Constant>(II->getArgOperand(1));
1882 if (!ShiftConst)
1883 return nullptr;
1884
1885 int ElemBits = Arg0->getType()->getScalarSizeInBits();
1886 bool AllPositive = true;
1887 bool AllNegative = true;
1888
1889 auto Check = [&](Constant *C) -> bool {
1890 if (auto *CI = dyn_cast_or_null<ConstantInt>(C)) {
1891 const APInt &V = CI->getValue();
1892 if (V.isNonNegative()) {
1893 AllNegative = false;
1894 return AllPositive && V.ult(ElemBits);
1895 }
1896 AllPositive = false;
1897 return AllNegative && V.sgt(-ElemBits);
1898 }
1899 return false;
1900 };
1901
1902 if (auto *VTy = dyn_cast<FixedVectorType>(Arg0->getType())) {
1903 for (unsigned I = 0, E = VTy->getNumElements(); I < E; ++I) {
1904 if (!Check(ShiftConst->getAggregateElement(I)))
1905 return nullptr;
1906 }
1907
1908 } else if (!Check(ShiftConst))
1909 return nullptr;
1910
1911 IRBuilderBase &B = IC.Builder;
1912 if (AllPositive)
1913 return IC.replaceInstUsesWith(*II, B.CreateShl(Arg0, ShiftConst));
1914
1915 Value *NegAmt = B.CreateNeg(ShiftConst);
1916 Intrinsic::ID IID = II->getIntrinsicID();
1917 const bool IsSigned =
1918 IID == Intrinsic::arm_neon_vshifts || IID == Intrinsic::aarch64_neon_sshl;
1919 Value *Result =
1920 IsSigned ? B.CreateAShr(Arg0, NegAmt) : B.CreateLShr(Arg0, NegAmt);
1921 return IC.replaceInstUsesWith(*II, Result);
1922}
1923
1924// If II is llvm.sin(x) or llvm.cos(x), and there is a matching
1925// llvm.cos(x) or llvm.sin(x) using the same argument, combine them
1926// into a single llvm.sincos(x) call. Returns the result for II
1927// extracted from sincos, or nullptr if no match is found.
1929 InstCombinerImpl &IC) {
1930 Intrinsic::ID IID = II->getIntrinsicID();
1931 bool IsSin = IID == Intrinsic::sin;
1932 Intrinsic::ID MatchID = IsSin ? Intrinsic::cos : Intrinsic::sin;
1933
1934 Value *Arg = II->getArgOperand(0);
1935
1936 // Don't bother looking through uses of constants.
1937 if (isa<Constant>(Arg))
1938 return nullptr;
1939
1940 // Look for a matching cos/sin intrinsic with the same argument.
1941 IntrinsicInst *Match = nullptr;
1942 for (User *U : Arg->users()) {
1943 if (auto *Cand = dyn_cast<IntrinsicInst>(U)) {
1944 if (Cand != II && !Cand->use_empty() &&
1945 Cand->getIntrinsicID() == MatchID) {
1946 Match = Cand;
1947 break;
1948 }
1949 }
1950 }
1951
1952 if (!Match)
1953 return nullptr;
1954
1955 // Insert sincos right after the argument definition.
1957 if (auto *ArgInst = dyn_cast<Instruction>(Arg)) {
1958 std::optional<BasicBlock::iterator> InsertPt =
1959 ArgInst->getInsertionPointAfterDef();
1960 if (!InsertPt)
1961 return nullptr;
1962 B.SetInsertPoint(*InsertPt);
1963 } else {
1964 BasicBlock &EntryBB = II->getFunction()->getEntryBlock();
1965 B.SetInsertPoint(&EntryBB, EntryBB.begin());
1966 }
1967
1969 II->getModule(), Intrinsic::sincos, Arg->getType());
1970 CallInst *SinCos = B.CreateCall(SinCosFunc, Arg, "sincos");
1971 // Intersect fast-math flags from the two calls.
1972 SinCos->setFastMathFlags(II->getFastMathFlags() & Match->getFastMathFlags());
1973 // Propagate the most-generic fpmath metadata from the two original calls.
1975 II->getMetadata(LLVMContext::MD_fpmath),
1976 Match->getMetadata(LLVMContext::MD_fpmath)))
1977 SinCos->setMetadata(LLVMContext::MD_fpmath, MD);
1978 Value *Sin = B.CreateExtractValue(SinCos, 0, "sin");
1979 Value *Cos = B.CreateExtractValue(SinCos, 1, "cos");
1980
1981 // Replace the matching call and erase it.
1982 IC.replaceInstUsesWith(*Match, IsSin ? Cos : Sin);
1983 IC.eraseInstFromFunction(*Match);
1984 return IsSin ? Sin : Cos;
1985}
1986
1987/// Fold an scmp/ucmp intrinsic whose operands are extended from a narrower
1988/// type:
1989/// scmp (sext X), (sext Y) --> scmp X, Y
1990/// scmp (zext X), (zext Y) --> ucmp X, Y
1991/// ucmp (ext X), (ext Y) --> ucmp X, Y
1992/// Both operands must use the same extend opcode and source type. A constant
1993/// operand is narrowed instead, if truncating and re-extending it gives back
1994/// the same constant.
1996 InstCombiner::BuilderTy &Builder,
1997 const DataLayout &DL) {
1998 // scmp/ucmp are not commutative, so the extend may be on either side.
1999 unsigned ExtIdx = 0;
2000 Value *X;
2001 if (!match(II->getArgOperand(0), m_ZExtOrSExt(m_Value(X)))) {
2002 ExtIdx = 1;
2003 if (!match(II->getArgOperand(1), m_ZExtOrSExt(m_Value(X))))
2004 return nullptr;
2005 }
2006
2007 auto CastOpc = static_cast<Instruction::CastOps>(
2008 cast<Operator>(II->getArgOperand(ExtIdx))->getOpcode());
2009 Type *NarrowTy = X->getType();
2010
2011 // The other operand must be the same kind of extend from the same type, or a
2012 // constant that can be narrowed losslessly.
2013 Value *OtherOp = II->getArgOperand(1 - ExtIdx);
2014 Value *Y;
2015 Constant *WideC;
2016 if (match(OtherOp, m_ZExtOrSExt(m_Value(Y)))) {
2017 if (cast<Operator>(OtherOp)->getOpcode() != CastOpc ||
2018 Y->getType() != NarrowTy)
2019 return nullptr;
2020 } else if (match(OtherOp, m_ImmConstant(WideC))) {
2021 Y = getLosslessInvCast(WideC, NarrowTy, CastOpc, DL);
2022 if (!Y)
2023 return nullptr;
2024 } else {
2025 return nullptr;
2026 }
2027
2028 // Both extends preserve the unsigned order, so an unsigned compare of the
2029 // narrow operands is always equivalent. The signed order is only preserved by
2030 // sext; zero extended values are non-negative, so a signed compare of those
2031 // is an unsigned compare of the narrow operands.
2032 Intrinsic::ID NewIID =
2033 II->getIntrinsicID() == Intrinsic::scmp && CastOpc == Instruction::SExt
2034 ? Intrinsic::scmp
2035 : Intrinsic::ucmp;
2036 if (ExtIdx != 0)
2037 std::swap(X, Y);
2038 return Builder.CreateIntrinsic(II->getType(), NewIID, {X, Y});
2039}
2040
2041/// CallInst simplification. This mostly only handles folding of intrinsic
2042/// instructions. For normal calls, it allows visitCallBase to do the heavy
2043/// lifting.
2045 // Don't try to simplify calls without uses. It will not do anything useful,
2046 // but will result in the following folds being skipped.
2047 if (!CI.use_empty()) {
2048 SmallVector<Value *, 8> Args(CI.args());
2049 if (Value *V = simplifyCall(&CI, CI.getCalledOperand(), Args,
2050 SQ.getWithInstruction(&CI)))
2051 return replaceInstUsesWith(CI, V);
2052 }
2053
2054 if (Value *FreedOp = getFreedOperand(&CI, &TLI))
2055 return visitFree(CI, FreedOp);
2056
2057 // If the caller function (i.e. us, the function that contains this CallInst)
2058 // is nounwind, mark the call as nounwind, even if the callee isn't.
2059 if (CI.getFunction()->doesNotThrow() && !CI.doesNotThrow()) {
2060 CI.setDoesNotThrow();
2061 return &CI;
2062 }
2063
2065 if (!II)
2066 return visitCallBase(CI);
2067
2068 // Intrinsics cannot occur in an invoke or a callbr, so handle them here
2069 // instead of in visitCallBase.
2070 if (auto *MI = dyn_cast<AnyMemIntrinsic>(II)) {
2071 if (auto NumBytes = MI->getLengthInBytes()) {
2072 // memmove/cpy/set of zero bytes is a noop.
2073 if (NumBytes->isZero())
2074 return eraseInstFromFunction(CI);
2075
2076 // For atomic unordered mem intrinsics if len is not a positive or
2077 // not a multiple of element size then behavior is undefined.
2078 if (MI->isAtomic() &&
2079 (NumBytes->isNegative() ||
2080 (NumBytes->getZExtValue() % MI->getElementSizeInBytes() != 0))) {
2082 assert(MI->getType()->isVoidTy() &&
2083 "non void atomic unordered mem intrinsic");
2084 return eraseInstFromFunction(*MI);
2085 }
2086 }
2087
2088 // No other transformations apply to volatile transfers.
2089 if (MI->isVolatile())
2090 return nullptr;
2091
2093 // memmove(x,x,size) -> noop.
2094 if (MTI->getSource() == MTI->getDest())
2095 return eraseInstFromFunction(CI);
2096 }
2097
2098 auto IsPointerUndefined = [MI](Value *Ptr) {
2099 return isa<ConstantPointerNull>(Ptr) &&
2101 MI->getFunction(),
2102 cast<PointerType>(Ptr->getType())->getAddressSpace());
2103 };
2104 bool SrcIsUndefined = false;
2105 // If we can determine a pointer alignment that is bigger than currently
2106 // set, update the alignment.
2107 if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
2109 return I;
2110 SrcIsUndefined = IsPointerUndefined(MTI->getRawSource());
2111 } else if (auto *MSI = dyn_cast<AnyMemSetInst>(MI)) {
2112 if (Instruction *I = SimplifyAnyMemSet(MSI))
2113 return I;
2114 }
2115
2116 // If src/dest is null, this memory intrinsic must be a noop.
2117 if (SrcIsUndefined || IsPointerUndefined(MI->getRawDest())) {
2118 Builder.CreateAssumption(Builder.CreateIsNull(MI->getLength()));
2119 return eraseInstFromFunction(CI);
2120 }
2121
2122 // If we have a memmove and the source operation is a constant global,
2123 // then the source and dest pointers can't alias, so we can change this
2124 // into a call to memcpy.
2125 if (auto *MMI = dyn_cast<AnyMemMoveInst>(MI)) {
2126 if (GlobalVariable *GVSrc = dyn_cast<GlobalVariable>(MMI->getSource()))
2127 if (GVSrc->isConstant()) {
2128 Module *M = CI.getModule();
2129 Intrinsic::ID MemCpyID =
2130 MMI->isAtomic()
2131 ? Intrinsic::memcpy_element_unordered_atomic
2132 : Intrinsic::memcpy;
2133 Type *Tys[3] = { CI.getArgOperand(0)->getType(),
2134 CI.getArgOperand(1)->getType(),
2135 CI.getArgOperand(2)->getType() };
2137 Intrinsic::getOrInsertDeclaration(M, MemCpyID, Tys));
2138 return II;
2139 }
2140 }
2141 }
2142
2143 // For fixed width vector result intrinsics, use the generic demanded vector
2144 // support.
2145 if (auto *IIFVTy = dyn_cast<FixedVectorType>(II->getType())) {
2146 auto VWidth = IIFVTy->getNumElements();
2147 APInt PoisonElts(VWidth, 0);
2148 APInt AllOnesEltMask(APInt::getAllOnes(VWidth));
2149 if (Value *V = SimplifyDemandedVectorElts(II, AllOnesEltMask, PoisonElts)) {
2150 if (V != II)
2151 return replaceInstUsesWith(*II, V);
2152 return II;
2153 }
2154 }
2155
2156 if (II->isCommutative()) {
2157 if (auto Pair = matchSymmetricPair(II->getOperand(0), II->getOperand(1))) {
2158 replaceOperand(*II, 0, Pair->first);
2159 replaceOperand(*II, 1, Pair->second);
2160 II->dropPoisonGeneratingAnnotations();
2161 II->dropUBImplyingAttrsAndMetadata();
2162 return II;
2163 }
2164
2165 if (CallInst *NewCall = canonicalizeConstantArg0ToArg1(CI))
2166 return NewCall;
2167 }
2168
2169 // Unused constrained FP intrinsic calls may have declared side effect, which
2170 // prevents it from being removed. In some cases however the side effect is
2171 // actually absent. To detect this case, call SimplifyConstrainedFPCall. If it
2172 // returns a replacement, the call may be removed.
2173 if (CI.use_empty() && isa<ConstrainedFPIntrinsic>(CI)) {
2174 if (simplifyConstrainedFPCall(&CI, SQ.getWithInstruction(&CI)))
2175 return eraseInstFromFunction(CI);
2176 }
2177
2178 Intrinsic::ID IID = II->getIntrinsicID();
2179 switch (IID) {
2180 case Intrinsic::objectsize: {
2181 SmallVector<Instruction *> InsertedInstructions;
2182 if (Value *V = lowerObjectSizeCall(II, DL, &TLI, AA, /*MustSucceed=*/false,
2183 &InsertedInstructions)) {
2184 for (Instruction *Inserted : InsertedInstructions)
2185 Worklist.add(Inserted);
2186 return replaceInstUsesWith(CI, V);
2187 }
2188 return nullptr;
2189 }
2190 case Intrinsic::abs: {
2191 Value *IIOperand = II->getArgOperand(0);
2192 bool IntMinIsPoison = cast<Constant>(II->getArgOperand(1))->isOneValue();
2193
2194 // abs(-x) -> abs(x)
2195 Value *X;
2196 if (match(IIOperand, m_Neg(m_Value(X))))
2197 return CallInst::Create(
2198 II->getCalledFunction(),
2199 {X,
2200 Builder.getInt1(IntMinIsPoison ||
2201 cast<Instruction>(IIOperand)->hasNoSignedWrap())});
2202
2203 if (match(IIOperand, m_c_Select(m_Neg(m_Value(X)), m_Deferred(X))))
2204 return CallInst::Create(II->getCalledFunction(),
2205 {X, II->getArgOperand(1)});
2206
2207 Value *Y;
2208 // abs(a * abs(b)) -> abs(a * b)
2209 if (match(IIOperand,
2212 bool NSW =
2213 cast<Instruction>(IIOperand)->hasNoSignedWrap() && IntMinIsPoison;
2214 auto *XY = NSW ? Builder.CreateNSWMul(X, Y) : Builder.CreateMul(X, Y);
2215 return CallInst::Create(II->getCalledFunction(),
2216 {XY, II->getArgOperand(1)});
2217 }
2218
2219 if (std::optional<bool> Known =
2220 getKnownSignOrZero(IIOperand, SQ.getWithInstruction(II))) {
2221 // abs(x) -> x if x >= 0 (include abs(x-y) --> x - y where x >= y)
2222 // abs(x) -> x if x > 0 (include abs(x-y) --> x - y where x > y)
2223 if (!*Known)
2224 return replaceInstUsesWith(*II, IIOperand);
2225
2226 // abs(x) -> -x if x < 0
2227 // abs(x) -> -x if x < = 0 (include abs(x-y) --> y - x where x <= y)
2228 if (IntMinIsPoison)
2229 return BinaryOperator::CreateNSWNeg(IIOperand);
2230 return BinaryOperator::CreateNeg(IIOperand);
2231 }
2232
2233 // abs (sext X) --> zext (abs X*)
2234 // Clear the IsIntMin (nsw) bit on the abs to allow narrowing.
2235 if (match(IIOperand, m_OneUse(m_SExt(m_Value(X))))) {
2236 Value *NarrowAbs =
2237 Builder.CreateBinaryIntrinsic(Intrinsic::abs, X, Builder.getFalse());
2238 return CastInst::Create(Instruction::ZExt, NarrowAbs, II->getType());
2239 }
2240
2241 // Match a complicated way to check if a number is odd/even:
2242 // abs (srem X, 2) --> and X, 1
2243 const APInt *C;
2244 if (match(IIOperand, m_SRem(m_Value(X), m_APInt(C))) && *C == 2)
2245 return BinaryOperator::CreateAnd(X, ConstantInt::get(II->getType(), 1));
2246
2247 break;
2248 }
2249 case Intrinsic::umin: {
2250 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2251 // umin(x, 1) == zext(x != 0)
2252 if (match(I1, m_One())) {
2253 assert(II->getType()->getScalarSizeInBits() != 1 &&
2254 "Expected simplify of umin with max constant");
2255 Value *Zero = Constant::getNullValue(I0->getType());
2256 Value *Cmp = Builder.CreateICmpNE(I0, Zero);
2257 return CastInst::Create(Instruction::ZExt, Cmp, II->getType());
2258 }
2259 // umin(cttz(x), const) --> cttz(x | (1 << const))
2260 if (Value *FoldedCttz =
2262 I0, I1, DL, Builder))
2263 return replaceInstUsesWith(*II, FoldedCttz);
2264 // umin(ctlz(x), const) --> ctlz(x | (SignedMin >> const))
2265 if (Value *FoldedCtlz =
2267 I0, I1, DL, Builder))
2268 return replaceInstUsesWith(*II, FoldedCtlz);
2269 [[fallthrough]];
2270 }
2271 case Intrinsic::umax: {
2272 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2273 Value *X, *Y;
2274 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_ZExt(m_Value(Y))) &&
2275 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2276 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2277 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2278 }
2279 Constant *C;
2280 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2281 I0->hasOneUse()) {
2282 if (Constant *NarrowC = getLosslessUnsignedTrunc(C, X->getType(), DL)) {
2283 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2284 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2285 }
2286 }
2287 // If C is not 0:
2288 // umax(nuw_shl(x, C), x + 1) -> x == 0 ? 1 : nuw_shl(x, C)
2289 // If C is not 0 or 1:
2290 // umax(nuw_mul(x, C), x + 1) -> x == 0 ? 1 : nuw_mul(x, C)
2291 auto foldMaxMulShift = [&](Value *A, Value *B) -> Instruction * {
2292 const APInt *C;
2293 Value *X;
2294 if (!match(A, m_NUWShl(m_Value(X), m_APInt(C))) &&
2295 !(match(A, m_NUWMul(m_Value(X), m_APInt(C))) && !C->isOne()))
2296 return nullptr;
2297 if (C->isZero())
2298 return nullptr;
2299 if (!match(B, m_OneUse(m_Add(m_Specific(X), m_One()))))
2300 return nullptr;
2301
2302 Value *Cmp = Builder.CreateICmpEQ(X, ConstantInt::get(X->getType(), 0));
2303 Value *NewSelect = nullptr;
2304 NewSelect = Builder.CreateSelectWithUnknownProfile(
2305 Cmp, ConstantInt::get(X->getType(), 1), A, DEBUG_TYPE);
2306 return replaceInstUsesWith(*II, NewSelect);
2307 };
2308
2309 if (IID == Intrinsic::umax) {
2310 if (Instruction *I = foldMaxMulShift(I0, I1))
2311 return I;
2312 if (Instruction *I = foldMaxMulShift(I1, I0))
2313 return I;
2314 }
2315
2316 // If both operands of unsigned min/max are sign-extended, it is still ok
2317 // to narrow the operation.
2318 [[fallthrough]];
2319 }
2320 case Intrinsic::smax:
2321 case Intrinsic::smin: {
2322 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2323 Value *X, *Y;
2324 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_SExt(m_Value(Y))) &&
2325 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2326 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2327 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2328 }
2329
2330 Constant *C;
2331 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2332 I0->hasOneUse()) {
2333 if (Constant *NarrowC = getLosslessSignedTrunc(C, X->getType(), DL)) {
2334 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2335 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2336 }
2337 }
2338
2339 // smax(smin(X, MinC), MaxC) -> smin(smax(X, MaxC), MinC) if MinC s>= MaxC
2340 // umax(umin(X, MinC), MaxC) -> umin(umax(X, MaxC), MinC) if MinC u>= MaxC
2341 const APInt *MinC, *MaxC;
2342 auto CreateCanonicalClampForm = [&](bool IsSigned) {
2343 auto MaxIID = IsSigned ? Intrinsic::smax : Intrinsic::umax;
2344 auto MinIID = IsSigned ? Intrinsic::smin : Intrinsic::umin;
2345 Value *NewMax = Builder.CreateBinaryIntrinsic(
2346 MaxIID, X, ConstantInt::get(X->getType(), *MaxC));
2347 return replaceInstUsesWith(
2348 *II, Builder.CreateBinaryIntrinsic(
2349 MinIID, NewMax, ConstantInt::get(X->getType(), *MinC)));
2350 };
2351 if (IID == Intrinsic::smax &&
2353 m_APInt(MinC)))) &&
2354 match(I1, m_APInt(MaxC)) && MinC->sgt(*MaxC))
2355 return CreateCanonicalClampForm(true);
2356 if (IID == Intrinsic::umax &&
2358 m_APInt(MinC)))) &&
2359 match(I1, m_APInt(MaxC)) && MinC->ugt(*MaxC))
2360 return CreateCanonicalClampForm(false);
2361
2362 // umin(i1 X, i1 Y) -> and i1 X, Y
2363 // smax(i1 X, i1 Y) -> and i1 X, Y
2364 if ((IID == Intrinsic::umin || IID == Intrinsic::smax) &&
2365 II->getType()->isIntOrIntVectorTy(1)) {
2366 return BinaryOperator::CreateAnd(I0, I1);
2367 }
2368
2369 // umax(i1 X, i1 Y) -> or i1 X, Y
2370 // smin(i1 X, i1 Y) -> or i1 X, Y
2371 if ((IID == Intrinsic::umax || IID == Intrinsic::smin) &&
2372 II->getType()->isIntOrIntVectorTy(1)) {
2373 return BinaryOperator::CreateOr(I0, I1);
2374 }
2375
2376 // smin(smax(X, -1), 1) -> scmp(X, 0)
2377 // smax(smin(X, 1), -1) -> scmp(X, 0)
2378 // At this point, smax(smin(X, 1), -1) is changed to smin(smax(X, -1)
2379 // And i1's have been changed to and/ors
2380 // So we only need to check for smin
2381 if (IID == Intrinsic::smin) {
2382 if (match(I0, m_OneUse(m_SMax(m_Value(X), m_AllOnes()))) &&
2383 match(I1, m_One())) {
2384 Value *Zero = ConstantInt::get(X->getType(), 0);
2385 return replaceInstUsesWith(
2386 CI,
2387 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {X, Zero}));
2388 }
2389 }
2390
2391 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2392 // smax (neg nsw X), (neg nsw Y) --> neg nsw (smin X, Y)
2393 // smin (neg nsw X), (neg nsw Y) --> neg nsw (smax X, Y)
2394 // TODO: Canonicalize neg after min/max if I1 is constant.
2395 if (match(I0, m_NSWNeg(m_Value(X))) && match(I1, m_NSWNeg(m_Value(Y))) &&
2396 (I0->hasOneUse() || I1->hasOneUse())) {
2398 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, X, Y);
2399 return BinaryOperator::CreateNSWNeg(InvMaxMin);
2400 }
2401 }
2402
2403 // (umax X, (xor X, Pow2))
2404 // -> (or X, Pow2)
2405 // (umin X, (xor X, Pow2))
2406 // -> (and X, ~Pow2)
2407 // (smax X, (xor X, Pos_Pow2))
2408 // -> (or X, Pos_Pow2)
2409 // (smin X, (xor X, Pos_Pow2))
2410 // -> (and X, ~Pos_Pow2)
2411 // (smax X, (xor X, Neg_Pow2))
2412 // -> (and X, ~Neg_Pow2)
2413 // (smin X, (xor X, Neg_Pow2))
2414 // -> (or X, Neg_Pow2)
2415 if ((match(I0, m_c_Xor(m_Specific(I1), m_Value(X))) ||
2416 match(I1, m_c_Xor(m_Specific(I0), m_Value(X)))) &&
2417 isKnownToBeAPowerOfTwo(X, /* OrZero */ true)) {
2418 bool UseOr = IID == Intrinsic::smax || IID == Intrinsic::umax;
2419 bool UseAndN = IID == Intrinsic::smin || IID == Intrinsic::umin;
2420
2421 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2422 auto KnownSign = getKnownSign(X, SQ.getWithInstruction(II));
2423 if (KnownSign == std::nullopt) {
2424 UseOr = false;
2425 UseAndN = false;
2426 } else if (*KnownSign /* true is Signed. */) {
2427 UseOr ^= true;
2428 UseAndN ^= true;
2429 Type *Ty = I0->getType();
2430 // Negative power of 2 must be IntMin. It's possible to be able to
2431 // prove negative / power of 2 without actually having known bits, so
2432 // just get the value by hand.
2434 Ty, APInt::getSignedMinValue(Ty->getScalarSizeInBits()));
2435 }
2436 }
2437 if (UseOr)
2438 return BinaryOperator::CreateOr(I0, X);
2439 else if (UseAndN)
2440 return BinaryOperator::CreateAnd(I0, Builder.CreateNot(X));
2441 }
2442
2443 // If we can eliminate ~A and Y is free to invert:
2444 // max ~A, Y --> ~(min A, ~Y)
2445 //
2446 // Examples:
2447 // max ~A, ~Y --> ~(min A, Y)
2448 // max ~A, C --> ~(min A, ~C)
2449 // max ~A, (max ~Y, ~Z) --> ~min( A, (min Y, Z))
2450 auto moveNotAfterMinMax = [&](Value *X, Value *Y) -> Instruction * {
2451 Value *A;
2452 if (match(X, m_OneUse(m_Not(m_Value(A)))) &&
2453 !isFreeToInvert(A, A->hasOneUse())) {
2454 if (Value *NotY = getFreelyInverted(Y, Y->hasOneUse(), &Builder)) {
2456 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, A, NotY);
2457 return BinaryOperator::CreateNot(InvMaxMin);
2458 }
2459 }
2460 return nullptr;
2461 };
2462
2463 if (Instruction *I = moveNotAfterMinMax(I0, I1))
2464 return I;
2465 if (Instruction *I = moveNotAfterMinMax(I1, I0))
2466 return I;
2467
2469 return I;
2470
2471 // minmax (X & NegPow2C, Y & NegPow2C) --> minmax(X, Y) & NegPow2C
2472 const APInt *RHSC;
2473 if (match(I0, m_OneUse(m_And(m_Value(X), m_NegatedPower2(RHSC)))) &&
2474 match(I1, m_OneUse(m_And(m_Value(Y), m_SpecificInt(*RHSC)))))
2475 return BinaryOperator::CreateAnd(Builder.CreateBinaryIntrinsic(IID, X, Y),
2476 ConstantInt::get(II->getType(), *RHSC));
2477
2478 // smax(X, -X) --> abs(X)
2479 // smin(X, -X) --> -abs(X)
2480 // umax(X, -X) --> -abs(X)
2481 // umin(X, -X) --> abs(X)
2482 if (isKnownNegation(I0, I1)) {
2483 // We can choose either operand as the input to abs(), but if we can
2484 // eliminate the only use of a value, that's better for subsequent
2485 // transforms/analysis.
2486 if (I0->hasOneUse() && !I1->hasOneUse())
2487 std::swap(I0, I1);
2488
2489 // This is some variant of abs(). See if we can propagate 'nsw' to the abs
2490 // operation and potentially its negation.
2491 bool IntMinIsPoison = isKnownNegation(I0, I1, /* NeedNSW */ true);
2492 Value *Abs = Builder.CreateBinaryIntrinsic(
2493 Intrinsic::abs, I0,
2494 ConstantInt::getBool(II->getContext(), IntMinIsPoison));
2495
2496 // We don't have a "nabs" intrinsic, so negate if needed based on the
2497 // max/min operation.
2498 if (IID == Intrinsic::smin || IID == Intrinsic::umax)
2499 Abs = Builder.CreateNeg(Abs, "nabs", IntMinIsPoison);
2500 return replaceInstUsesWith(CI, Abs);
2501 }
2502
2504 return Sel;
2505
2506 if (Instruction *SAdd = matchSAddSubSat(*II))
2507 return SAdd;
2508
2509 if (Value *NewMinMax = reassociateMinMaxWithConstants(II, Builder, SQ))
2510 return replaceInstUsesWith(*II, NewMinMax);
2511
2513 return R;
2514
2515 if (Instruction *NewMinMax = factorizeMinMaxTree(II))
2516 return NewMinMax;
2517
2518 // Try to fold minmax with constant RHS based on range information
2519 if (match(I1, m_APIntAllowPoison(RHSC))) {
2520 ICmpInst::Predicate Pred =
2522 bool IsSigned = MinMaxIntrinsic::isSigned(IID);
2524 I0, IsSigned, SQ.getWithInstruction(II));
2525 if (!LHS_CR.isFullSet()) {
2526 if (LHS_CR.icmp(Pred, *RHSC))
2527 return replaceInstUsesWith(*II, I0);
2528 if (LHS_CR.icmp(ICmpInst::getSwappedPredicate(Pred), *RHSC))
2529 return replaceInstUsesWith(*II,
2530 ConstantInt::get(II->getType(), *RHSC));
2531 }
2532 }
2533
2535 return replaceInstUsesWith(*II, V);
2536
2537 break;
2538 }
2539 case Intrinsic::scmp:
2540 case Intrinsic::ucmp: {
2542 return replaceInstUsesWith(CI, V);
2543
2544 if (IID == Intrinsic::ucmp)
2545 break;
2546
2547 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2548
2549 // scmp(X, 0) -> sext_or_trunc(X) if X is known to be one of -1, 0, 1.
2550 if (match(I1, m_Zero())) {
2551 ConstantRange Range = computeConstantRange(I0, /*ForSigned=*/true,
2552 SQ.getWithInstruction(II));
2553 if (Range.getSignedMin().sge(-1) && Range.getSignedMax().sle(1))
2554 return replaceInstUsesWith(
2555 CI, Builder.CreateSExtOrTrunc(I0, II->getType()));
2556 }
2557 Value *LHS, *RHS;
2558 if (match(I0, m_NSWSub(m_Value(LHS), m_Value(RHS))) && match(I1, m_Zero()))
2559 return replaceInstUsesWith(
2560 CI,
2561 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {LHS, RHS}));
2562 break;
2563 }
2564 case Intrinsic::bitreverse: {
2565 Value *IIOperand = II->getArgOperand(0);
2566 // bitrev (zext i1 X to ?) --> X ? SignBitC : 0
2567 Value *X;
2568 if (match(IIOperand, m_ZExt(m_Value(X))) &&
2569 X->getType()->isIntOrIntVectorTy(1)) {
2570 Type *Ty = II->getType();
2571 APInt SignBit = APInt::getSignMask(Ty->getScalarSizeInBits());
2572 return SelectInst::Create(X, ConstantInt::get(Ty, SignBit),
2574 }
2575
2576 if (Instruction *crossLogicOpFold =
2578 return crossLogicOpFold;
2579
2580 break;
2581 }
2582 case Intrinsic::bswap: {
2583 Value *IIOperand = II->getArgOperand(0);
2584
2585 // Try to canonicalize bswap-of-logical-shift-by-8-bit-multiple as
2586 // inverse-shift-of-bswap:
2587 // bswap (shl X, Y) --> lshr (bswap X), Y
2588 // bswap (lshr X, Y) --> shl (bswap X), Y
2589 Value *X, *Y;
2590 if (match(IIOperand, m_OneUse(m_LogicalShift(m_Value(X), m_Value(Y))))) {
2591 unsigned BitWidth = IIOperand->getType()->getScalarSizeInBits();
2593 Value *NewSwap = Builder.CreateUnaryIntrinsic(Intrinsic::bswap, X);
2594 BinaryOperator::BinaryOps InverseShift =
2595 cast<BinaryOperator>(IIOperand)->getOpcode() == Instruction::Shl
2596 ? Instruction::LShr
2597 : Instruction::Shl;
2598 return BinaryOperator::Create(InverseShift, NewSwap, Y);
2599 }
2600 }
2601
2602 KnownBits Known = computeKnownBits(IIOperand, II);
2603 uint64_t LZ = alignDown(Known.countMinLeadingZeros(), 8);
2604 uint64_t TZ = alignDown(Known.countMinTrailingZeros(), 8);
2605 unsigned BW = Known.getBitWidth();
2606
2607 // bswap(x) -> shift(x) if x has exactly one "active byte"
2608 if (BW - LZ - TZ == 8) {
2609 assert(LZ != TZ && "active byte cannot be in the middle");
2610 if (LZ > TZ) // -> shl(x) if the "active byte" is in the low part of x
2611 return BinaryOperator::CreateNUWShl(
2612 IIOperand, ConstantInt::get(IIOperand->getType(), LZ - TZ));
2613 // -> lshr(x) if the "active byte" is in the high part of x
2614 return BinaryOperator::CreateExactLShr(
2615 IIOperand, ConstantInt::get(IIOperand->getType(), TZ - LZ));
2616 }
2617
2618 // bswap(trunc(bswap(x))) -> trunc(lshr(x, c))
2619 if (match(IIOperand, m_Trunc(m_BSwap(m_Value(X))))) {
2620 unsigned C = X->getType()->getScalarSizeInBits() - BW;
2621 Value *CV = ConstantInt::get(X->getType(), C);
2622 Value *V = Builder.CreateLShr(X, CV);
2623 return new TruncInst(V, IIOperand->getType());
2624 }
2625
2626 if (Instruction *crossLogicOpFold =
2628 return crossLogicOpFold;
2629 }
2630
2631 // Try to fold into bitreverse if bswap is the root of the expression tree.
2632 if (Instruction *BitOp = matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ false,
2633 /*MatchBitReversals*/ true))
2634 return BitOp;
2635 break;
2636 }
2637 case Intrinsic::masked_load:
2638 if (Value *SimplifiedMaskedOp = simplifyMaskedLoad(*II))
2639 return replaceInstUsesWith(CI, SimplifiedMaskedOp);
2640 break;
2641 case Intrinsic::masked_store:
2642 return simplifyMaskedStore(*II);
2643 case Intrinsic::masked_gather:
2644 return simplifyMaskedGather(*II);
2645 case Intrinsic::masked_scatter:
2646 return simplifyMaskedScatter(*II);
2647 case Intrinsic::launder_invariant_group:
2648 case Intrinsic::strip_invariant_group:
2649 if (auto *SkippedBarrier = simplifyInvariantGroupIntrinsic(*II, *this))
2650 return replaceInstUsesWith(*II, SkippedBarrier);
2651 break;
2652 case Intrinsic::powi: {
2653 if (ConstantInt *Power = dyn_cast<ConstantInt>(II->getArgOperand(1))) {
2654 // 0 and 1 are handled in instsimplify
2655 // powi(x, -1) -> 1/x
2656 if (Power->isMinusOne())
2657 return BinaryOperator::CreateFDivFMF(ConstantFP::get(CI.getType(), 1.0),
2658 II->getArgOperand(0), II);
2659 // powi(x, 2) -> x*x
2660 if (Power->equalsInt(2))
2661 return BinaryOperator::CreateFMulFMF(II->getArgOperand(0),
2662 II->getArgOperand(0), II);
2663
2664 if (!Power->getValue()[0]) {
2665 Value *X;
2666 // If power is even:
2667 // powi(-x, p) -> powi(x, p)
2668 // powi(fabs(x), p) -> powi(x, p)
2669 // powi(copysign(x, y), p) -> powi(x, p)
2670 if (match(II->getArgOperand(0), m_FNeg(m_Value(X))) ||
2671 match(II->getArgOperand(0), m_FAbs(m_Value(X))) ||
2672 match(II->getArgOperand(0),
2674 return CallInst::Create(II->getCalledFunction(), {X, Power});
2675 }
2676 }
2677 if (ConstantFP *Base = dyn_cast<ConstantFP>(II->getArgOperand(0))) {
2678 Value *Exp = II->getArgOperand(1);
2679 Type *Ty = Base->getType();
2680 // powi(2.0, p) -> ldexp(1.0, p)
2681 if (II->hasApproxFunc() && Base->isExactlyValue(2.0)) {
2682 ConstantFP *One = ConstantFP::get(Ty, 1.0);
2683 if (auto *VTy = dyn_cast<VectorType>(Ty))
2684 Exp = Builder.CreateVectorSplat(VTy->getElementCount(), Exp);
2685 Value *Ldexp = Builder.CreateLdexp(One, Exp, II);
2686 return replaceInstUsesWith(*II, Ldexp);
2687 }
2688 }
2689 break;
2690 }
2691
2692 case Intrinsic::cttz:
2693 case Intrinsic::ctlz:
2694 if (auto *I = foldCttzCtlz(*II, *this))
2695 return I;
2696 break;
2697
2698 case Intrinsic::ctpop:
2699 if (auto *I = foldCtpop(*II, *this))
2700 return I;
2701 break;
2702
2703 case Intrinsic::fshl:
2704 case Intrinsic::fshr: {
2705 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
2706 Type *Ty = II->getType();
2707 unsigned BitWidth = Ty->getScalarSizeInBits();
2708 Constant *ShAmtC;
2709 if (match(II->getArgOperand(2), m_ImmConstant(ShAmtC))) {
2710 // Canonicalize a shift amount constant operand to modulo the bit-width.
2711 Constant *WidthC = ConstantInt::get(Ty, BitWidth);
2712 Constant *ModuloC =
2713 ConstantFoldBinaryOpOperands(Instruction::URem, ShAmtC, WidthC, DL);
2714 if (!ModuloC)
2715 return nullptr;
2716 if (ModuloC != ShAmtC)
2717 return CallInst::Create(II->getCalledFunction(), {Op0, Op1, ModuloC});
2718
2720 ShAmtC, DL),
2721 m_One()) &&
2722 "Shift amount expected to be modulo bitwidth");
2723
2724 // Canonicalize funnel shift right by constant to funnel shift left. This
2725 // is not entirely arbitrary. For historical reasons, the backend may
2726 // recognize rotate left patterns but miss rotate right patterns.
2727 if (IID == Intrinsic::fshr) {
2728 // fshr X, Y, C --> fshl X, Y, (BitWidth - C) if C is not zero.
2729 if (!isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2730 return nullptr;
2731
2732 Constant *LeftShiftC = ConstantExpr::getSub(WidthC, ShAmtC);
2733 Module *Mod = II->getModule();
2734 Function *Fshl =
2735 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::fshl, Ty);
2736 return CallInst::Create(Fshl, { Op0, Op1, LeftShiftC });
2737 }
2738 assert(IID == Intrinsic::fshl &&
2739 "All funnel shifts by simple constants should go left");
2740
2741 // fshl(X, 0, C) --> shl X, C
2742 // fshl(X, undef, C) --> shl X, C
2743 if (match(Op1, m_ZeroInt()) || match(Op1, m_Undef()))
2744 return BinaryOperator::CreateShl(Op0, ShAmtC);
2745
2746 // fshl(0, X, C) --> lshr X, (BW-C)
2747 // fshl(undef, X, C) --> lshr X, (BW-C)
2748 // Similar to fshr -> fshl fold above, this is only valid if C is not zero
2749 if ((match(Op0, m_ZeroInt()) || match(Op0, m_Undef())) &&
2750 isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2751 return BinaryOperator::CreateLShr(Op1,
2752 ConstantExpr::getSub(WidthC, ShAmtC));
2753
2754 // fshl i16 X, X, 8 --> bswap i16 X (reduce to more-specific form)
2755 if (Op0 == Op1 && BitWidth == 16 && match(ShAmtC, m_SpecificInt(8))) {
2756 Module *Mod = II->getModule();
2757 Function *Bswap =
2758 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::bswap, Ty);
2759 return CallInst::Create(Bswap, { Op0 });
2760 }
2761 if (Instruction *BitOp =
2762 matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ true,
2763 /*MatchBitReversals*/ true))
2764 return BitOp;
2765
2766 // R = fshl(X, X, C2)
2767 // fshl(R, R, C1) --> fshl(X, X, (C1 + C2) % bitsize)
2768 Value *InnerOp;
2769 const APInt *ShAmtInnerC, *ShAmtOuterC;
2770 if (match(Op0, m_FShl(m_Value(InnerOp), m_Deferred(InnerOp),
2771 m_APInt(ShAmtInnerC))) &&
2772 match(ShAmtC, m_APInt(ShAmtOuterC)) && Op0 == Op1) {
2773 APInt Sum = *ShAmtOuterC + *ShAmtInnerC;
2774 APInt Modulo = Sum.urem(APInt(Sum.getBitWidth(), BitWidth));
2775 if (Modulo.isZero())
2776 return replaceInstUsesWith(*II, InnerOp);
2777 Constant *ModuloC = ConstantInt::get(Ty, Modulo);
2779 {InnerOp, InnerOp, ModuloC});
2780 }
2781 }
2782
2783 // fshl(X, X, Neg(Y)) --> fshr(X, X, Y)
2784 // fshr(X, X, Neg(Y)) --> fshl(X, X, Y)
2785 // if BitWidth is a power-of-2
2786 Value *Y;
2787 if (Op0 == Op1 && isPowerOf2_32(BitWidth) &&
2788 match(II->getArgOperand(2), m_Neg(m_Value(Y)))) {
2789 Module *Mod = II->getModule();
2791 Mod, IID == Intrinsic::fshl ? Intrinsic::fshr : Intrinsic::fshl, Ty);
2792 return CallInst::Create(OppositeShift, {Op0, Op1, Y});
2793 }
2794
2795 // fshl(X, 0, Y) --> shl(X, and(Y, BitWidth - 1)) if bitwidth is a
2796 // power-of-2
2797 if (IID == Intrinsic::fshl && isPowerOf2_32(BitWidth) &&
2798 match(Op1, m_ZeroInt())) {
2799 Value *Op2 = II->getArgOperand(2);
2800 Value *And = Builder.CreateAnd(Op2, ConstantInt::get(Ty, BitWidth - 1));
2801 return BinaryOperator::CreateShl(Op0, And);
2802 }
2803
2804 // Left or right might be masked.
2806 return &CI;
2807
2808 // The shift amount (operand 2) of a funnel shift is modulo the bitwidth,
2809 // so only the low bits of the shift amount are demanded if the bitwidth is
2810 // a power-of-2.
2811 if (!isPowerOf2_32(BitWidth))
2812 break;
2814 KnownBits Op2Known(BitWidth);
2815 if (SimplifyDemandedBits(II, 2, Op2Demanded, Op2Known))
2816 return &CI;
2817 break;
2818 }
2819 case Intrinsic::pdep: {
2820 const APInt *MaskC;
2821 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2822 unsigned MaskIdx, MaskLen;
2823 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2824 // any single contiguous sequence of 1s anywhere in the mask simply
2825 // describes a subset of the input bits shifted to the appropriate
2826 // position. Replace with the straight forward IR.
2827 Value *Input = II->getArgOperand(0);
2828 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2829 Value *Shifted = Builder.CreateShl(Input, ShiftAmt);
2830 Value *Masked = Builder.CreateAnd(Shifted, II->getArgOperand(1));
2831 return replaceInstUsesWith(*II, Masked);
2832 }
2833 }
2834 break;
2835 }
2836 case Intrinsic::pext: {
2837 const APInt *MaskC;
2838 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2839 unsigned MaskIdx, MaskLen;
2840 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2841 // any single contiguous sequence of 1s anywhere in the mask simply
2842 // describes a subset of the input bits shifted to the appropriate
2843 // position. Replace with the straight forward IR.
2844 Value *Input = II->getArgOperand(0);
2845 Value *Masked = Builder.CreateAnd(Input, II->getArgOperand(1));
2846 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2847 Value *Shifted = Builder.CreateLShr(Masked, ShiftAmt);
2848 return replaceInstUsesWith(*II, Shifted);
2849 }
2850 }
2851 break;
2852 }
2853 case Intrinsic::ptrmask: {
2854 unsigned BitWidth = DL.getPointerTypeSizeInBits(II->getType());
2857 return II;
2858
2859 Value *InnerPtr, *InnerMask;
2860 bool Changed = false;
2861 // Combine:
2862 // (ptrmask (ptrmask p, A), B)
2863 // -> (ptrmask p, (and A, B))
2864 if (match(II->getArgOperand(0),
2866 m_Value(InnerMask))))) {
2867 assert(II->getArgOperand(1)->getType() == InnerMask->getType() &&
2868 "Mask types must match");
2869 // TODO: If InnerMask == Op1, we could copy attributes from inner
2870 // callsite -> outer callsite.
2871 Value *NewMask = Builder.CreateAnd(II->getArgOperand(1), InnerMask);
2872 replaceOperand(CI, 0, InnerPtr);
2873 replaceOperand(CI, 1, NewMask);
2874 Changed = true;
2875 }
2876
2877 // See if we can deduce non-null.
2878 if (!CI.hasRetAttr(Attribute::NonNull) &&
2879 (Known.isNonZero() ||
2880 isKnownNonZero(II, getSimplifyQuery().getWithInstruction(II)))) {
2881 CI.addRetAttr(Attribute::NonNull);
2882 Changed = true;
2883 }
2884
2885 unsigned NewAlignmentLog =
2887 std::min(BitWidth - 1, Known.countMinTrailingZeros()));
2888 // Known bits will capture if we had alignment information associated with
2889 // the pointer argument.
2890 if (NewAlignmentLog > Log2(CI.getRetAlign().valueOrOne())) {
2892 CI.getContext(), Align(uint64_t(1) << NewAlignmentLog)));
2893 Changed = true;
2894 }
2895 if (Changed)
2896 return &CI;
2897 break;
2898 }
2899 case Intrinsic::uadd_with_overflow:
2900 case Intrinsic::sadd_with_overflow: {
2901 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2902 return I;
2903
2904 // Given 2 constant operands whose sum does not overflow:
2905 // uaddo (X +nuw C0), C1 -> uaddo X, C0 + C1
2906 // saddo (X +nsw C0), C1 -> saddo X, C0 + C1
2907 Value *X;
2908 const APInt *C0, *C1;
2909 Value *Arg0 = II->getArgOperand(0);
2910 Value *Arg1 = II->getArgOperand(1);
2911 bool IsSigned = IID == Intrinsic::sadd_with_overflow;
2912 bool HasNWAdd = IsSigned
2913 ? match(Arg0, m_NSWAddLike(m_Value(X), m_APInt(C0)))
2914 : match(Arg0, m_NUWAddLike(m_Value(X), m_APInt(C0)));
2915 if (HasNWAdd && match(Arg1, m_APInt(C1))) {
2916 bool Overflow;
2917 APInt NewC =
2918 IsSigned ? C1->sadd_ov(*C0, Overflow) : C1->uadd_ov(*C0, Overflow);
2919 if (!Overflow)
2920 return replaceInstUsesWith(
2921 *II, Builder.CreateBinaryIntrinsic(
2922 IID, X, ConstantInt::get(Arg1->getType(), NewC)));
2923 }
2924 break;
2925 }
2926
2927 case Intrinsic::umul_with_overflow:
2928 case Intrinsic::smul_with_overflow:
2929 case Intrinsic::usub_with_overflow:
2930 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2931 return I;
2932 break;
2933
2934 case Intrinsic::ssub_with_overflow: {
2935 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2936 return I;
2937
2938 Constant *C;
2939 Value *Arg0 = II->getArgOperand(0);
2940 Value *Arg1 = II->getArgOperand(1);
2941 // Given a constant C that is not the minimum signed value
2942 // for an integer of a given bit width:
2943 //
2944 // ssubo X, C -> saddo X, -C
2945 if (match(Arg1, m_Constant(C)) && C->isNotMinSignedValue()) {
2946 Value *NegVal = ConstantExpr::getNeg(C);
2947 // Build a saddo call that is equivalent to the discovered
2948 // ssubo call.
2949 return replaceInstUsesWith(
2950 *II, Builder.CreateBinaryIntrinsic(Intrinsic::sadd_with_overflow,
2951 Arg0, NegVal));
2952 }
2953
2954 break;
2955 }
2956
2957 case Intrinsic::uadd_sat:
2958 case Intrinsic::sadd_sat:
2959 case Intrinsic::usub_sat:
2960 case Intrinsic::ssub_sat: {
2962 Type *Ty = SI->getType();
2963 Value *Arg0 = SI->getLHS();
2964 Value *Arg1 = SI->getRHS();
2965
2966 // Make use of known overflow information.
2967 OverflowResult OR = computeOverflow(SI->getBinaryOp(), SI->isSigned(),
2968 Arg0, Arg1, SI);
2969 switch (OR) {
2971 break;
2973 if (SI->isSigned())
2974 return BinaryOperator::CreateNSW(SI->getBinaryOp(), Arg0, Arg1);
2975 else
2976 return BinaryOperator::CreateNUW(SI->getBinaryOp(), Arg0, Arg1);
2978 unsigned BitWidth = Ty->getScalarSizeInBits();
2979 APInt Min = APSInt::getMinValue(BitWidth, !SI->isSigned());
2980 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Min));
2981 }
2983 unsigned BitWidth = Ty->getScalarSizeInBits();
2984 APInt Max = APSInt::getMaxValue(BitWidth, !SI->isSigned());
2985 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Max));
2986 }
2987 }
2988
2989 // usub_sat((sub nuw C, A), C1) -> usub_sat(usub_sat(C, C1), A)
2990 // which after that:
2991 // usub_sat((sub nuw C, A), C1) -> usub_sat(C - C1, A) if C1 u< C
2992 // usub_sat((sub nuw C, A), C1) -> 0 otherwise
2993 Constant *C, *C1;
2994 Value *A;
2995 if (IID == Intrinsic::usub_sat &&
2996 match(Arg0, m_NUWSub(m_ImmConstant(C), m_Value(A))) &&
2997 match(Arg1, m_ImmConstant(C1))) {
2998 auto *NewC = Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, C, C1);
2999 auto *NewSub =
3000 Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, NewC, A);
3001 return replaceInstUsesWith(*SI, NewSub);
3002 }
3003
3004 // ssub.sat(X, C) -> sadd.sat(X, -C) if C != MIN
3005 if (IID == Intrinsic::ssub_sat && match(Arg1, m_Constant(C)) &&
3006 C->isNotMinSignedValue()) {
3007 Value *NegVal = ConstantExpr::getNeg(C);
3008 return replaceInstUsesWith(
3009 *II, Builder.CreateBinaryIntrinsic(
3010 Intrinsic::sadd_sat, Arg0, NegVal));
3011 }
3012
3013 // sat(sat(X + Val2) + Val) -> sat(X + (Val+Val2))
3014 // sat(sat(X - Val2) - Val) -> sat(X - (Val+Val2))
3015 // if Val and Val2 have the same sign
3016 if (auto *Other = dyn_cast<IntrinsicInst>(Arg0)) {
3017 Value *X;
3018 const APInt *Val, *Val2;
3019 APInt NewVal;
3020 bool IsUnsigned =
3021 IID == Intrinsic::uadd_sat || IID == Intrinsic::usub_sat;
3022 if (Other->getIntrinsicID() == IID &&
3023 match(Arg1, m_APInt(Val)) &&
3024 match(Other->getArgOperand(0), m_Value(X)) &&
3025 match(Other->getArgOperand(1), m_APInt(Val2))) {
3026 if (IsUnsigned)
3027 NewVal = Val->uadd_sat(*Val2);
3028 else if (Val->isNonNegative() == Val2->isNonNegative()) {
3029 bool Overflow;
3030 NewVal = Val->sadd_ov(*Val2, Overflow);
3031 if (Overflow) {
3032 // Both adds together may add more than SignedMaxValue
3033 // without saturating the final result.
3034 break;
3035 }
3036 } else {
3037 // Cannot fold saturated addition with different signs.
3038 break;
3039 }
3040
3041 return replaceInstUsesWith(
3042 *II, Builder.CreateBinaryIntrinsic(
3043 IID, X, ConstantInt::get(II->getType(), NewVal)));
3044 }
3045 }
3046 break;
3047 }
3048
3049 case Intrinsic::minnum:
3050 case Intrinsic::maxnum:
3051 case Intrinsic::minimumnum:
3052 case Intrinsic::maximumnum:
3053 case Intrinsic::minimum:
3054 case Intrinsic::maximum: {
3055 Value *Arg0 = II->getArgOperand(0);
3056 Value *Arg1 = II->getArgOperand(1);
3057 Value *X, *Y;
3058 if (match(Arg0, m_FNeg(m_Value(X))) && match(Arg1, m_FNeg(m_Value(Y))) &&
3059 (Arg0->hasOneUse() || Arg1->hasOneUse())) {
3060 // If both operands are negated, invert the call and negate the result:
3061 // min(-X, -Y) --> -(max(X, Y))
3062 // max(-X, -Y) --> -(min(X, Y))
3063 Intrinsic::ID NewIID;
3064 switch (IID) {
3065 case Intrinsic::maxnum:
3066 NewIID = Intrinsic::minnum;
3067 break;
3068 case Intrinsic::minnum:
3069 NewIID = Intrinsic::maxnum;
3070 break;
3071 case Intrinsic::maximumnum:
3072 NewIID = Intrinsic::minimumnum;
3073 break;
3074 case Intrinsic::minimumnum:
3075 NewIID = Intrinsic::maximumnum;
3076 break;
3077 case Intrinsic::maximum:
3078 NewIID = Intrinsic::minimum;
3079 break;
3080 case Intrinsic::minimum:
3081 NewIID = Intrinsic::maximum;
3082 break;
3083 default:
3084 llvm_unreachable("unexpected intrinsic ID");
3085 }
3086 Value *NewCall = Builder.CreateBinaryIntrinsic(NewIID, X, Y, II);
3087 Instruction *FNeg = UnaryOperator::CreateFNeg(NewCall);
3088 FNeg->copyIRFlags(II);
3089 return FNeg;
3090 }
3091
3092 // m(m(X, C2), C1) -> m(X, C)
3093 const APFloat *C1, *C2;
3094 if (auto *M = dyn_cast<IntrinsicInst>(Arg0)) {
3095 if (M->getIntrinsicID() == IID && match(Arg1, m_APFloat(C1)) &&
3096 ((match(M->getArgOperand(0), m_Value(X)) &&
3097 match(M->getArgOperand(1), m_APFloat(C2))) ||
3098 (match(M->getArgOperand(1), m_Value(X)) &&
3099 match(M->getArgOperand(0), m_APFloat(C2))))) {
3100 APFloat Res(0.0);
3101 switch (IID) {
3102 case Intrinsic::maxnum:
3103 Res = maxnum(*C1, *C2);
3104 break;
3105 case Intrinsic::minnum:
3106 Res = minnum(*C1, *C2);
3107 break;
3108 case Intrinsic::maximumnum:
3109 Res = maximumnum(*C1, *C2);
3110 break;
3111 case Intrinsic::minimumnum:
3112 Res = minimumnum(*C1, *C2);
3113 break;
3114 case Intrinsic::maximum:
3115 Res = maximum(*C1, *C2);
3116 break;
3117 case Intrinsic::minimum:
3118 Res = minimum(*C1, *C2);
3119 break;
3120 default:
3121 llvm_unreachable("unexpected intrinsic ID");
3122 }
3123 // TODO: Conservatively intersecting FMF. If Res == C2, the transform
3124 // was a simplification (so Arg0 and its original flags could
3125 // propagate?)
3126 Value *V = Builder.CreateBinaryIntrinsic(
3127 IID, X, ConstantFP::get(Arg0->getType(), Res),
3129 return replaceInstUsesWith(*II, V);
3130 }
3131 }
3132
3133 // m((fpext X), (fpext Y)) -> fpext (m(X, Y))
3134 if (match(Arg0, m_FPExt(m_Value(X))) && match(Arg1, m_FPExt(m_Value(Y))) &&
3135 (Arg0->hasOneUse() || Arg1->hasOneUse()) &&
3136 X->getType() == Y->getType()) {
3137 Value *NewCall =
3138 Builder.CreateBinaryIntrinsic(IID, X, Y, II, II->getName());
3139 return new FPExtInst(NewCall, II->getType());
3140 }
3141
3142 // m(fpext X, C) -> fpext m(X, TruncC) if C can be losslessly truncated.
3143 Constant *C;
3144 if (match(Arg0, m_OneUse(m_FPExt(m_Value(X)))) &&
3145 match(Arg1, m_ImmConstant(C))) {
3146 if (Constant *TruncC =
3147 getLosslessInvCast(C, X->getType(), Instruction::FPExt, DL)) {
3148 Value *NewCall =
3149 Builder.CreateBinaryIntrinsic(IID, X, TruncC, II, II->getName());
3150 return new FPExtInst(NewCall, II->getType());
3151 }
3152 }
3153
3154 // max X, -X --> fabs X
3155 // min X, -X --> -(fabs X)
3156 // TODO: Remove one-use limitation? That is obviously better for max,
3157 // hence why we don't check for one-use for that. However,
3158 // it would be an extra instruction for min (fnabs), but
3159 // that is still likely better for analysis and codegen.
3160 auto IsMinMaxOrXNegX = [IID, &X](Value *Op0, Value *Op1) {
3161 if (match(Op0, m_FNeg(m_Value(X))) && match(Op1, m_Specific(X)))
3162 return Op0->hasOneUse() ||
3163 (IID != Intrinsic::minimum && IID != Intrinsic::minnum &&
3164 IID != Intrinsic::minimumnum);
3165 return false;
3166 };
3167
3168 if (IsMinMaxOrXNegX(Arg0, Arg1) || IsMinMaxOrXNegX(Arg1, Arg0)) {
3169 Value *R = Builder.CreateFAbs(X, II);
3170 if (IID == Intrinsic::minimum || IID == Intrinsic::minnum ||
3171 IID == Intrinsic::minimumnum)
3172 R = Builder.CreateFNegFMF(R, II);
3173 return replaceInstUsesWith(*II, R);
3174 }
3175
3176 break;
3177 }
3178 case Intrinsic::matrix_multiply: {
3179 // Optimize negation in matrix multiplication.
3180
3181 // -A * -B -> A * B
3182 Value *A, *B;
3183 if (match(II->getArgOperand(0), m_FNeg(m_Value(A))) &&
3184 match(II->getArgOperand(1), m_FNeg(m_Value(B)))) {
3185 replaceOperand(*II, 0, A);
3186 replaceOperand(*II, 1, B);
3187 return II;
3188 }
3189
3190 Value *Op0 = II->getOperand(0);
3191 Value *Op1 = II->getOperand(1);
3192 Value *OpNotNeg, *NegatedOp;
3193 unsigned NegatedOpArg, OtherOpArg;
3194 if (match(Op0, m_FNeg(m_Value(OpNotNeg)))) {
3195 NegatedOp = Op0;
3196 NegatedOpArg = 0;
3197 OtherOpArg = 1;
3198 } else if (match(Op1, m_FNeg(m_Value(OpNotNeg)))) {
3199 NegatedOp = Op1;
3200 NegatedOpArg = 1;
3201 OtherOpArg = 0;
3202 } else
3203 // Multiplication doesn't have a negated operand.
3204 break;
3205
3206 // Only optimize if the negated operand has only one use.
3207 if (!NegatedOp->hasOneUse())
3208 break;
3209
3210 Value *OtherOp = II->getOperand(OtherOpArg);
3211 VectorType *RetTy = cast<VectorType>(II->getType());
3212 VectorType *NegatedOpTy = cast<VectorType>(NegatedOp->getType());
3213 VectorType *OtherOpTy = cast<VectorType>(OtherOp->getType());
3214 ElementCount NegatedCount = NegatedOpTy->getElementCount();
3215 ElementCount OtherCount = OtherOpTy->getElementCount();
3216 ElementCount RetCount = RetTy->getElementCount();
3217 // (-A) * B -> A * (-B), if it is cheaper to negate B and vice versa.
3218 if (ElementCount::isKnownGT(NegatedCount, OtherCount) &&
3219 ElementCount::isKnownLT(OtherCount, RetCount)) {
3220 Value *InverseOtherOp = Builder.CreateFNeg(OtherOp);
3221 replaceOperand(*II, NegatedOpArg, OpNotNeg);
3222 replaceOperand(*II, OtherOpArg, InverseOtherOp);
3223 return II;
3224 }
3225 // (-A) * B -> -(A * B), if it is cheaper to negate the result
3226 if (ElementCount::isKnownGT(NegatedCount, RetCount)) {
3227 SmallVector<Value *, 5> NewArgs(II->args());
3228 NewArgs[NegatedOpArg] = OpNotNeg;
3229 Value *NewMul = Builder.CreateIntrinsic(II->getType(), IID, NewArgs, II);
3230 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(NewMul, II));
3231 }
3232 break;
3233 }
3234 case Intrinsic::fmuladd: {
3235 // Try to simplify the underlying FMul.
3236 if (Value *V =
3237 simplifyFMulInst(II->getArgOperand(0), II->getArgOperand(1),
3238 II->getFastMathFlags(), SQ.getWithInstruction(II)))
3239 return BinaryOperator::CreateFAddFMF(V, II->getArgOperand(2),
3240 II->getFastMathFlags());
3241
3242 [[fallthrough]];
3243 }
3244 case Intrinsic::fma: {
3245 // fma fneg(x), fneg(y), z -> fma x, y, z
3246 Value *Src0 = II->getArgOperand(0);
3247 Value *Src1 = II->getArgOperand(1);
3248 Value *Src2 = II->getArgOperand(2);
3249 Value *X, *Y;
3250 if (match(Src0, m_FNeg(m_Value(X))) && match(Src1, m_FNeg(m_Value(Y))))
3251 return replaceInstUsesWith(
3252 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, Y, Src2}, II));
3253
3254 // fma fabs(x), fabs(x), z -> fma x, x, z
3255 if (match(Src0, m_FAbs(m_Value(X))) && match(Src1, m_FAbs(m_Specific(X))))
3256 return replaceInstUsesWith(
3257 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, X, Src2}, II));
3258
3259 // Try to simplify the underlying FMul. We can only apply simplifications
3260 // that do not require rounding.
3261 if (Value *V = simplifyFMAFMul(Src0, Src1, II->getFastMathFlags(),
3262 SQ.getWithInstruction(II)))
3263 return BinaryOperator::CreateFAddFMF(V, Src2, II->getFastMathFlags());
3264
3265 // fma x, y, 0 -> fmul x, y
3266 // This is always valid for -0.0, but requires nsz for +0.0 as
3267 // -0.0 + 0.0 = 0.0, which would not be the same as the fmul on its own.
3268 if (match(Src2, m_NegZeroFP()) ||
3269 (match(Src2, m_PosZeroFP()) && II->getFastMathFlags().noSignedZeros()))
3270 return BinaryOperator::CreateFMulFMF(Src0, Src1, II);
3271
3272 // fma x, -1.0, y -> fsub y, x
3273 if (match(Src1, m_SpecificFP(-1.0)))
3274 return BinaryOperator::CreateFSubFMF(Src2, Src0, II);
3275
3276 break;
3277 }
3278 case Intrinsic::copysign: {
3279 Value *Mag = II->getArgOperand(0), *Sign = II->getArgOperand(1);
3280 if (std::optional<bool> KnownSignBit = computeKnownFPSignBit(
3281 Sign, getSimplifyQuery().getWithInstruction(II))) {
3282 if (*KnownSignBit) {
3283 // If we know that the sign argument is negative, reduce to FNABS:
3284 // copysign Mag, -Sign --> fneg (fabs Mag)
3285 Value *Fabs = Builder.CreateFAbs(Mag, II);
3286 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(Fabs, II));
3287 }
3288
3289 // If we know that the sign argument is positive, reduce to FABS:
3290 // copysign Mag, +Sign --> fabs Mag
3291 Value *Fabs = Builder.CreateFAbs(Mag, II);
3292 return replaceInstUsesWith(*II, Fabs);
3293 }
3294
3295 // Propagate sign argument through nested calls:
3296 // copysign Mag, (copysign ?, X) --> copysign Mag, X
3297 Value *X;
3299 Value *CopySign =
3300 Builder.CreateCopySign(Mag, X, FMFSource::intersect(II, Sign));
3301 return replaceInstUsesWith(*II, CopySign);
3302 }
3303
3304 // Clear sign-bit of constant magnitude:
3305 // copysign -MagC, X --> copysign MagC, X
3306 // TODO: Support constant folding for fabs
3307 const APFloat *MagC;
3308 if (match(Mag, m_APFloat(MagC)) && MagC->isNegative()) {
3309 APFloat PosMagC = *MagC;
3310 PosMagC.clearSign();
3311 return replaceInstUsesWith(
3312 *II, Builder.CreateCopySign(ConstantFP::get(Mag->getType(), PosMagC),
3313 Sign, II));
3314 }
3315
3316 // Peek through changes of magnitude's sign-bit. This call rewrites those:
3317 // copysign (fabs X), Sign --> copysign X, Sign
3318 // copysign (fneg X), Sign --> copysign X, Sign
3319 if (match(Mag, m_FAbs(m_Value(X))) || match(Mag, m_FNeg(m_Value(X))))
3320 return replaceInstUsesWith(*II, Builder.CreateCopySign(X, Sign, II));
3321
3322 // copysign(floor(fabs(X)), X) --> copysign(trunc(X), X)
3323 // copysign ignores the sign bit of its magnitude argument (implicit fabs),
3324 // so replacing floor(fabs(X)) with trunc(X) is correct for all inputs
3325 // including NaN without requiring nnan. The m_FAbs match also ensures
3326 // the floor argument is non-negative, so floor == trunc.
3327 Value *FAbsArg;
3328 if (match(Mag, m_Intrinsic<Intrinsic::floor>(m_FAbs(m_Value(FAbsArg)))) &&
3329 FAbsArg == Sign) {
3330 Value *Trunc = Builder.CreateUnaryIntrinsic(Intrinsic::trunc, Sign, II);
3331 return replaceInstUsesWith(*II, Builder.CreateCopySign(Trunc, Sign, II));
3332 }
3333
3334 Type *SignEltTy = Sign->getType()->getScalarType();
3335
3336 Value *CastSrc;
3337 if (match(Sign,
3339 CastSrc->getType()->isIntOrIntVectorTy() &&
3343 APInt::getSignMask(Known.getBitWidth()), Known,
3344 SQ))
3345 return II;
3346 }
3347
3348 break;
3349 }
3350 case Intrinsic::fabs: {
3351 Value *Cond, *TVal, *FVal;
3352 Value *Arg = II->getArgOperand(0);
3353 Value *X;
3354 // fabs (-X) --> fabs (X)
3355 if (match(Arg, m_FNeg(m_Value(X)))) {
3356 Value *Fabs = Builder.CreateFAbs(X, II);
3357 return replaceInstUsesWith(CI, Fabs);
3358 }
3359
3360 if (match(Arg, m_Select(m_Value(Cond), m_Value(TVal), m_Value(FVal)))) {
3361 // fabs (select Cond, TrueC, FalseC) --> select Cond, AbsT, AbsF
3362 if (Arg->hasOneUse() ? (isa<Constant>(TVal) || isa<Constant>(FVal))
3363 : (isa<Constant>(TVal) && isa<Constant>(FVal))) {
3364 CallInst *AbsT = Builder.CreateCall(II->getCalledFunction(), {TVal});
3365 CallInst *AbsF = Builder.CreateCall(II->getCalledFunction(), {FVal});
3366 SelectInst *SI = SelectInst::Create(Cond, AbsT, AbsF);
3367 SI->setFastMathFlags(II->getFastMathFlags() |
3368 cast<SelectInst>(Arg)->getFastMathFlags());
3369 // Can't copy nsz to select, as even with the nsz flag the fabs result
3370 // always has the sign bit unset.
3371 SI->setHasNoSignedZeros(false);
3372 return SI;
3373 }
3374 // fabs (select Cond, -FVal, FVal) --> fabs FVal
3375 if (match(TVal, m_FNeg(m_Specific(FVal))))
3376 return replaceInstUsesWith(*II, Builder.CreateFAbs(FVal, II));
3377 // fabs (select Cond, TVal, -TVal) --> fabs TVal
3378 if (match(FVal, m_FNeg(m_Specific(TVal))))
3379 return replaceInstUsesWith(*II, Builder.CreateFAbs(TVal, II));
3380 }
3381
3382 Value *Magnitude, *Sign;
3383 if (match(II->getArgOperand(0),
3384 m_CopySign(m_Value(Magnitude), m_Value(Sign)))) {
3385 // fabs (copysign x, y) -> (fabs x)
3386 Value *AbsSign = Builder.CreateFAbs(Magnitude, II);
3387 return replaceInstUsesWith(*II, AbsSign);
3388 }
3389
3390 [[fallthrough]];
3391 }
3392 case Intrinsic::ceil:
3393 case Intrinsic::floor:
3394 case Intrinsic::round:
3395 case Intrinsic::roundeven:
3396 case Intrinsic::nearbyint:
3397 case Intrinsic::rint:
3398 case Intrinsic::trunc: {
3399 Value *ExtSrc;
3400 if (match(II->getArgOperand(0), m_OneUse(m_FPExt(m_Value(ExtSrc))))) {
3401 // Narrow the call: intrinsic (fpext x) -> fpext (intrinsic x)
3402 Value *NarrowII = Builder.CreateUnaryIntrinsic(IID, ExtSrc, II);
3403 return new FPExtInst(NarrowII, II->getType());
3404 }
3405 break;
3406 }
3407 case Intrinsic::cos:
3408 case Intrinsic::amdgcn_cos:
3409 case Intrinsic::cosh: {
3410 Value *X, *Sign;
3411 Value *Src = II->getArgOperand(0);
3412 if (match(Src, m_FNeg(m_Value(X))) || match(Src, m_FAbs(m_Value(X))) ||
3413 match(Src, m_CopySign(m_Value(X), m_Value(Sign)))) {
3414 // f(-x) --> f(x)
3415 // f(fabs(x)) --> f(x)
3416 // f(copysign(x, y)) --> f(x)
3417 // for f in {cos, cosh}
3418 return replaceInstUsesWith(*II, Builder.CreateUnaryIntrinsic(IID, X, II));
3419 }
3420 if (IID == Intrinsic::cos) {
3421 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3422 return replaceInstUsesWith(*II, Result);
3423 }
3424 break;
3425 }
3426 case Intrinsic::sin:
3427 case Intrinsic::amdgcn_sin:
3428 case Intrinsic::sinh:
3429 case Intrinsic::tan:
3430 case Intrinsic::tanh: {
3431 Value *X;
3432 if (match(II->getArgOperand(0), m_OneUse(m_FNeg(m_Value(X))))) {
3433 // f(-x) --> -f(x)
3434 // for f in {sin, sinh, tan, tanh}
3435 Value *NewFunc = Builder.CreateUnaryIntrinsic(IID, X, II);
3436 return UnaryOperator::CreateFNegFMF(NewFunc, II);
3437 }
3438 if (IID == Intrinsic::sin) {
3439 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3440 return replaceInstUsesWith(*II, Result);
3441 }
3442 break;
3443 }
3444 case Intrinsic::ldexp: {
3445 Value *Src = II->getArgOperand(0);
3446 Value *Exp = II->getArgOperand(1);
3447
3448 // ldexp(x, K) -> fmul x, 2^K
3449 uint64_t ConstExp;
3450 if (match(Exp, m_ConstantInt(ConstExp))) {
3451 const fltSemantics &FPTy =
3452 Src->getType()->getScalarType()->getFltSemantics();
3453
3454 APFloat Scaled = scalbn(APFloat::getOne(FPTy), static_cast<int>(ConstExp),
3456 if (!Scaled.isZero() && !Scaled.isInfinity()) {
3457 // Skip overflow and underflow cases.
3458 Constant *FPConst = ConstantFP::get(Src->getType(), Scaled);
3459 return BinaryOperator::CreateFMulFMF(Src, FPConst, II);
3460 }
3461 }
3462
3463 // ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b))
3464 //
3465 // A danger is if the first ldexp would overflow to infinity or underflow to
3466 // zero, but the combined exponent avoids it.
3467 //
3468 // We ignore this with reassoc, or if we know both exponents have the same
3469 // sign (since then we'd just double down on the over/underflow which would
3470 // occur anyway).
3471 //
3472 // ldexp can take arbitrary integer types, so we also need to ensure that
3473 // our exponent type is wide enough so that if sadd.sat(a, b) saturates,
3474 // then ldexp at the saturated exponent saturates to inf or zero as well.
3475 //
3476 // TODO: Could do better if we had range tracking for the input value
3477 // exponent. Also could broaden sign check to cover == 0 case.
3478 Value *InnerSrc;
3479 Value *InnerExp;
3481 m_Value(InnerSrc), m_Value(InnerExp)))) &&
3482 Exp->getType() == InnerExp->getType()) {
3483 FastMathFlags FMF = II->getFastMathFlags();
3484 FastMathFlags InnerFlags = cast<FPMathOperator>(Src)->getFastMathFlags();
3485
3486 if (ldexpSaturatingAddIsSafe(II->getType(), Exp->getType()) &&
3487 ((FMF.allowReassoc() && InnerFlags.allowReassoc()) ||
3488 signBitMustBeTheSame(Exp, InnerExp, SQ.getWithInstruction(II)))) {
3489 Value *NewExp =
3490 Builder.CreateBinaryIntrinsic(Intrinsic::sadd_sat, InnerExp, Exp);
3491 return replaceInstUsesWith(
3492 *II, Builder.CreateLdexp(InnerSrc, NewExp, FMF | InnerFlags));
3493 }
3494 }
3495
3496 // ldexp(x, zext(i1 y)) -> fmul x, (select y, 2.0, 1.0)
3497 // ldexp(x, sext(i1 y)) -> fmul x, (select y, 0.5, 1.0)
3498 Value *ExtSrc;
3499 if (match(Exp, m_ZExt(m_Value(ExtSrc))) &&
3500 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3501 Value *Select =
3502 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 2.0),
3503 ConstantFP::get(II->getType(), 1.0));
3505 }
3506 if (match(Exp, m_SExt(m_Value(ExtSrc))) &&
3507 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3508 Value *Select =
3509 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 0.5),
3510 ConstantFP::get(II->getType(), 1.0));
3512 }
3513
3514 // ldexp(x, c ? exp : 0) -> c ? ldexp(x, exp) : x
3515 // ldexp(x, c ? 0 : exp) -> c ? x : ldexp(x, exp)
3516 ///
3517 // TODO: If we cared, should insert a canonicalize for x
3518 Value *SelectCond, *SelectLHS, *SelectRHS;
3519 if (match(II->getArgOperand(1),
3520 m_OneUse(m_Select(m_Value(SelectCond), m_Value(SelectLHS),
3521 m_Value(SelectRHS))))) {
3522 Value *NewLdexp = nullptr;
3523 Value *Select = nullptr;
3524 if (match(SelectRHS, m_ZeroInt())) {
3525 NewLdexp = Builder.CreateLdexp(Src, SelectLHS, II);
3526 Select = Builder.CreateSelect(SelectCond, NewLdexp, Src);
3527 } else if (match(SelectLHS, m_ZeroInt())) {
3528 NewLdexp = Builder.CreateLdexp(Src, SelectRHS, II);
3529 Select = Builder.CreateSelect(SelectCond, Src, NewLdexp);
3530 }
3531
3532 if (NewLdexp) {
3533 Select->takeName(II);
3534 return replaceInstUsesWith(*II, Select);
3535 }
3536 }
3537
3538 break;
3539 }
3540 case Intrinsic::ptrauth_auth:
3541 case Intrinsic::ptrauth_resign: {
3542 // (sign|resign) + (auth|resign) can be folded by omitting the middle
3543 // sign+auth component if the key and discriminator match.
3544 bool NeedSign = II->getIntrinsicID() == Intrinsic::ptrauth_resign;
3545 Value *Ptr = II->getArgOperand(0);
3546 Value *Key = II->getArgOperand(1);
3547 Value *Disc = II->getArgOperand(2);
3548 Value *DS = nullptr;
3549 if (auto Bundle = II->getOperandBundle(LLVMContext::OB_deactivation_symbol))
3550 DS = Bundle->Inputs[0];
3551
3552 // AuthKey will be the key we need to end up authenticating against in
3553 // whatever we replace this sequence with.
3554 Value *AuthKey = nullptr, *AuthDisc = nullptr, *BasePtr;
3555 if (const auto *CI = dyn_cast<CallBase>(Ptr)) {
3556 Value *OtherDS = nullptr;
3557 if (auto Bundle =
3559 OtherDS = Bundle->Inputs[0];
3560 if (DS != OtherDS)
3561 break;
3562
3563 if (CI->getIntrinsicID() == Intrinsic::ptrauth_sign) {
3564 if (CI->getArgOperand(1) != Key || CI->getArgOperand(2) != Disc)
3565 break;
3566 } else if (CI->getIntrinsicID() == Intrinsic::ptrauth_resign) {
3567 // The resign intrinsic does not support deactivation symbols.
3568 assert(!DS);
3569 if (CI->getArgOperand(3) != Key || CI->getArgOperand(4) != Disc)
3570 break;
3571 AuthKey = CI->getArgOperand(1);
3572 AuthDisc = CI->getArgOperand(2);
3573 } else
3574 break;
3575 BasePtr = CI->getArgOperand(0);
3576 } else if (const auto *PtrToInt = dyn_cast<PtrToIntOperator>(Ptr)) {
3577 // ptrauth constants are equivalent to a call to @llvm.ptrauth.sign for
3578 // our purposes, so check for that too.
3579 const auto *CPA = dyn_cast<ConstantPtrAuth>(PtrToInt->getOperand(0));
3580 if (!CPA || DS || !CPA->isKnownCompatibleWith(Key, Disc, DL))
3581 break;
3582
3583 // resign(ptrauth(p,ks,ds),ks,ds,kr,dr) -> ptrauth(p,kr,dr)
3584 if (NeedSign && isa<ConstantInt>(II->getArgOperand(4))) {
3585 auto *SignKey = cast<ConstantInt>(II->getArgOperand(3));
3586 auto *SignDisc = cast<ConstantInt>(II->getArgOperand(4));
3587 auto *Null = ConstantPointerNull::get(Builder.getPtrTy());
3588 auto *NewCPA = ConstantPtrAuth::get(CPA->getPointer(), SignKey,
3589 SignDisc, /*AddrDisc=*/Null,
3590 /*DeactivationSymbol=*/Null);
3592 *II, ConstantExpr::getPointerCast(NewCPA, II->getType()));
3593 return eraseInstFromFunction(*II);
3594 }
3595
3596 // auth(ptrauth(p,k,d),k,d) -> p
3597 BasePtr = Builder.CreatePtrToInt(CPA->getPointer(), II->getType());
3598 } else
3599 break;
3600
3601 unsigned NewIntrin;
3602 if (AuthKey && NeedSign) {
3603 // resign(0,1) + resign(1,2) = resign(0, 2)
3604 NewIntrin = Intrinsic::ptrauth_resign;
3605 } else if (AuthKey) {
3606 // resign(0,1) + auth(1) = auth(0)
3607 NewIntrin = Intrinsic::ptrauth_auth;
3608 } else if (NeedSign) {
3609 // sign(0) + resign(0, 1) = sign(1)
3610 NewIntrin = Intrinsic::ptrauth_sign;
3611 } else {
3612 // sign(0) + auth(0) = nop
3613 replaceInstUsesWith(*II, BasePtr);
3614 return eraseInstFromFunction(*II);
3615 }
3616
3617 SmallVector<Value *, 4> CallArgs;
3618 CallArgs.push_back(BasePtr);
3619 if (AuthKey) {
3620 CallArgs.push_back(AuthKey);
3621 CallArgs.push_back(AuthDisc);
3622 }
3623
3624 if (NeedSign) {
3625 CallArgs.push_back(II->getArgOperand(3));
3626 CallArgs.push_back(II->getArgOperand(4));
3627 }
3628
3629 std::vector<OperandBundleDef> Bundles;
3630 if (DS)
3631 Bundles.push_back(OperandBundleDef("deactivation-symbol", DS));
3632
3633 Function *NewFn =
3634 Intrinsic::getOrInsertDeclaration(II->getModule(), NewIntrin);
3635 return CallInst::Create(NewFn, CallArgs, Bundles);
3636 }
3637 case Intrinsic::arm_neon_vtbl1:
3638 case Intrinsic::arm_neon_vtbl2:
3639 case Intrinsic::arm_neon_vtbl3:
3640 case Intrinsic::arm_neon_vtbl4:
3641 case Intrinsic::aarch64_neon_tbl1:
3642 case Intrinsic::aarch64_neon_tbl2:
3643 case Intrinsic::aarch64_neon_tbl3:
3644 case Intrinsic::aarch64_neon_tbl4:
3645 return simplifyNeonTbl(*II, *this, /*IsExtension=*/false);
3646 case Intrinsic::arm_neon_vtbx1:
3647 case Intrinsic::arm_neon_vtbx2:
3648 case Intrinsic::arm_neon_vtbx3:
3649 case Intrinsic::arm_neon_vtbx4:
3650 case Intrinsic::aarch64_neon_tbx1:
3651 case Intrinsic::aarch64_neon_tbx2:
3652 case Intrinsic::aarch64_neon_tbx3:
3653 case Intrinsic::aarch64_neon_tbx4:
3654 return simplifyNeonTbl(*II, *this, /*IsExtension=*/true);
3655
3656 case Intrinsic::arm_neon_vmulls:
3657 case Intrinsic::arm_neon_vmullu:
3658 case Intrinsic::aarch64_neon_smull:
3659 case Intrinsic::aarch64_neon_umull: {
3660 Value *Arg0 = II->getArgOperand(0);
3661 Value *Arg1 = II->getArgOperand(1);
3662
3663 // Handle mul by zero first:
3665 return replaceInstUsesWith(CI, ConstantAggregateZero::get(II->getType()));
3666 }
3667
3668 // Check for constant LHS & RHS - in this case we just simplify.
3669 bool Zext = (IID == Intrinsic::arm_neon_vmullu ||
3670 IID == Intrinsic::aarch64_neon_umull);
3671 VectorType *NewVT = cast<VectorType>(II->getType());
3672 if (Constant *CV0 = dyn_cast<Constant>(Arg0)) {
3673 if (Constant *CV1 = dyn_cast<Constant>(Arg1)) {
3674 Value *V0 = Builder.CreateIntCast(CV0, NewVT, /*isSigned=*/!Zext);
3675 Value *V1 = Builder.CreateIntCast(CV1, NewVT, /*isSigned=*/!Zext);
3676 return replaceInstUsesWith(CI, Builder.CreateMul(V0, V1));
3677 }
3678
3679 // Couldn't simplify - canonicalize constant to the RHS.
3680 std::swap(Arg0, Arg1);
3681 }
3682
3683 // Handle mul by one:
3684 if (Constant *CV1 = dyn_cast<Constant>(Arg1))
3685 if (ConstantInt *Splat =
3686 dyn_cast_or_null<ConstantInt>(CV1->getSplatValue()))
3687 if (Splat->isOne())
3688 return CastInst::CreateIntegerCast(Arg0, II->getType(),
3689 /*isSigned=*/!Zext);
3690
3691 break;
3692 }
3693 case Intrinsic::arm_neon_aesd:
3694 case Intrinsic::arm_neon_aese:
3695 case Intrinsic::aarch64_crypto_aesd:
3696 case Intrinsic::aarch64_crypto_aese:
3697 case Intrinsic::aarch64_sve_aesd:
3698 case Intrinsic::aarch64_sve_aese: {
3699 Value *DataArg = II->getArgOperand(0);
3700 Value *KeyArg = II->getArgOperand(1);
3701
3702 // Accept zero on either operand.
3703 if (!match(KeyArg, m_ZeroInt()))
3704 std::swap(KeyArg, DataArg);
3705
3706 // Try to use the builtin XOR in AESE and AESD to eliminate a prior XOR
3707 Value *Data, *Key;
3708 if (match(KeyArg, m_ZeroInt()) &&
3709 match(DataArg, m_Xor(m_Value(Data), m_Value(Key)))) {
3710 replaceOperand(*II, 0, Data);
3711 replaceOperand(*II, 1, Key);
3712 return II;
3713 }
3714 break;
3715 }
3716 case Intrinsic::arm_neon_vshifts:
3717 case Intrinsic::arm_neon_vshiftu:
3718 case Intrinsic::aarch64_neon_sshl:
3719 case Intrinsic::aarch64_neon_ushl:
3720 return foldNeonShift(II, *this);
3721 case Intrinsic::hexagon_V6_vandvrt:
3722 case Intrinsic::hexagon_V6_vandvrt_128B: {
3723 // Simplify Q -> V -> Q conversion.
3724 if (auto Op0 = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3725 Intrinsic::ID ID0 = Op0->getIntrinsicID();
3726 if (ID0 != Intrinsic::hexagon_V6_vandqrt &&
3727 ID0 != Intrinsic::hexagon_V6_vandqrt_128B)
3728 break;
3729 Value *Bytes = Op0->getArgOperand(1), *Mask = II->getArgOperand(1);
3730 uint64_t Bytes1 = computeKnownBits(Bytes, Op0).One.getZExtValue();
3731 uint64_t Mask1 = computeKnownBits(Mask, II).One.getZExtValue();
3732 // Check if every byte has common bits in Bytes and Mask.
3733 uint64_t C = Bytes1 & Mask1;
3734 if ((C & 0xFF) && (C & 0xFF00) && (C & 0xFF0000) && (C & 0xFF000000))
3735 return replaceInstUsesWith(*II, Op0->getArgOperand(0));
3736 }
3737 break;
3738 }
3739 case Intrinsic::stackrestore: {
3740 enum class ClassifyResult {
3741 None,
3742 Alloca,
3743 StackRestore,
3744 CallWithSideEffects,
3745 };
3746 auto Classify = [](const Instruction *I) {
3747 if (isa<AllocaInst>(I))
3748 return ClassifyResult::Alloca;
3749
3750 if (auto *CI = dyn_cast<CallInst>(I)) {
3751 if (auto *II = dyn_cast<IntrinsicInst>(CI)) {
3752 if (II->getIntrinsicID() == Intrinsic::stackrestore)
3753 return ClassifyResult::StackRestore;
3754
3755 if (II->mayHaveSideEffects())
3756 return ClassifyResult::CallWithSideEffects;
3757 } else {
3758 // Consider all non-intrinsic calls to be side effects
3759 return ClassifyResult::CallWithSideEffects;
3760 }
3761 }
3762
3763 return ClassifyResult::None;
3764 };
3765
3766 // If the stacksave and the stackrestore are in the same BB, and there is
3767 // no intervening call, alloca, or stackrestore of a different stacksave,
3768 // remove the restore. This can happen when variable allocas are DCE'd.
3769 if (IntrinsicInst *SS = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3770 if (SS->getIntrinsicID() == Intrinsic::stacksave &&
3771 SS->getParent() == II->getParent()) {
3772 BasicBlock::iterator BI(SS);
3773 bool CannotRemove = false;
3774 for (++BI; &*BI != II; ++BI) {
3775 switch (Classify(&*BI)) {
3776 case ClassifyResult::None:
3777 // So far so good, look at next instructions.
3778 break;
3779
3780 case ClassifyResult::StackRestore:
3781 // If we found an intervening stackrestore for a different
3782 // stacksave, we can't remove the stackrestore. Otherwise, continue.
3783 if (cast<IntrinsicInst>(*BI).getArgOperand(0) != SS)
3784 CannotRemove = true;
3785 break;
3786
3787 case ClassifyResult::Alloca:
3788 case ClassifyResult::CallWithSideEffects:
3789 // If we found an alloca, a non-intrinsic call, or an intrinsic
3790 // call with side effects, we can't remove the stackrestore.
3791 CannotRemove = true;
3792 break;
3793 }
3794 if (CannotRemove)
3795 break;
3796 }
3797
3798 if (!CannotRemove)
3799 return eraseInstFromFunction(CI);
3800 }
3801 }
3802
3803 // Scan down this block to see if there is another stack restore in the
3804 // same block without an intervening call/alloca.
3806 Instruction *TI = II->getParent()->getTerminator();
3807 bool CannotRemove = false;
3808 for (++BI; &*BI != TI; ++BI) {
3809 switch (Classify(&*BI)) {
3810 case ClassifyResult::None:
3811 // So far so good, look at next instructions.
3812 break;
3813
3814 case ClassifyResult::StackRestore:
3815 // If there is a stackrestore below this one, remove this one.
3816 return eraseInstFromFunction(CI);
3817
3818 case ClassifyResult::Alloca:
3819 case ClassifyResult::CallWithSideEffects:
3820 // If we found an alloca, a non-intrinsic call, or an intrinsic call
3821 // with side effects (such as llvm.stacksave and llvm.read_register),
3822 // we can't remove the stack restore.
3823 CannotRemove = true;
3824 break;
3825 }
3826 if (CannotRemove)
3827 break;
3828 }
3829
3830 // If the stack restore is in a return, resume, or unwind block and if there
3831 // are no allocas or calls between the restore and the return, nuke the
3832 // restore.
3833 if (!CannotRemove && (isa<ReturnInst>(TI) || isa<ResumeInst>(TI)))
3834 return eraseInstFromFunction(CI);
3835 break;
3836 }
3837 case Intrinsic::lifetime_end:
3838 // Asan needs to poison memory to detect invalid access which is possible
3839 // even for empty lifetime range.
3840 if (II->getFunction()->hasFnAttribute(Attribute::SanitizeAddress) ||
3841 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemory) ||
3842 II->getFunction()->hasFnAttribute(Attribute::SanitizeHWAddress) ||
3843 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag))
3844 break;
3845
3846 if (removeTriviallyEmptyRange(*II, *this, [](const IntrinsicInst &I) {
3847 return I.getIntrinsicID() == Intrinsic::lifetime_start;
3848 }))
3849 return nullptr;
3850 break;
3851 case Intrinsic::assume: {
3852 for (auto [Idx, OBU] : llvm::enumerate(II->operand_bundles())) {
3853 auto RemoveBundle = [&, Idx = Idx]() -> Instruction * {
3854 if (II->getNumOperandBundles() == 1)
3855 return eraseInstFromFunction(*II);
3857 };
3858
3859 switch (getBundleAttrFromOBU(OBU)) {
3860 case BundleAttr::None:
3861 llvm_unreachable("Unexpected Attribute");
3862 case BundleAttr::Align: {
3863 // Try to remove redundant alignment assumptions.
3864 auto [Ptr, _, OffsetPtr, Alignment, Offset] = getAssumeAlignInfo(OBU);
3865
3866 if (!Alignment)
3867 break;
3868
3869 // Remove align 1 and non-power-of-two bundles; they don't add any
3870 // useful information.
3871 if (*Alignment == 1 || !isPowerOf2_64(*Alignment))
3872 return RemoveBundle();
3873
3874 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3875 GEP &&
3876 GEP->getMaxPreservedAlignment(getDataLayout()) >= *Alignment) {
3877 Builder.CreateAlignmentAssumption(
3878 getDataLayout(), GEP->getPointerOperand(), *Alignment,
3879 OffsetPtr ? const_cast<Value *>(OffsetPtr->get()) : nullptr);
3880 return RemoveBundle();
3881 }
3882
3883 if (!Offset)
3884 break;
3885
3886 Value *BasePtr;
3887 const APInt *PtrOffset;
3888 if (match(Ptr.get(), m_PtrAdd(m_Value(BasePtr), m_APInt(PtrOffset)))) {
3889 auto PtrOffsetVal =
3890 PtrOffset->sextOrTrunc(DL.getIndexTypeSizeInBits(Ptr->getType()))
3891 .trySExtValue();
3892 if (!PtrOffsetVal)
3893 break;
3894 Builder.CreateAlignmentAssumption(
3895 DL, BasePtr, *Alignment,
3896 Builder.getInt64(*Offset - *PtrOffsetVal));
3897 return RemoveBundle();
3898 }
3899
3900 // Don't try to remove align assumptions for pointers derived from
3901 // arguments. We might lose information if the function gets inline and
3902 // the align argument attribute disappears.
3903 Value *UO = getUnderlyingObject(Ptr);
3904 if (!UO || isa<Argument>(UO))
3905 break;
3906
3907 // Compute known bits for the pointer and drop the assume if the
3908 // known alignment isn't increased by it.
3909 auto AlignMask = (*Alignment - 1);
3910 if (KnownBits KB = computeKnownBits(Ptr, II);
3911 (KB.Zero & AlignMask) == (~*Offset & AlignMask) &&
3912 (KB.One & AlignMask) == (*Offset & AlignMask))
3913 return RemoveBundle();
3914 break;
3915 }
3916
3917 case BundleAttr::Dereferenceable: {
3918 auto [Ptr, _, Count] = getAssumeDereferenceableInfo(OBU);
3919
3920 if (!Count)
3921 break;
3922
3923 if (*Count == 0 ||
3925 getSimplifyQuery().getWithInstruction(II)))
3926 return RemoveBundle();
3927
3928 break;
3929 }
3930
3931 case BundleAttr::Ignore:
3932 return RemoveBundle();
3933
3934 case BundleAttr::NonNull: {
3935 auto [Ptr] = llvm::getAssumeNonNullInfo(OBU);
3936
3937 // Drop assume if we can prove nonnull without it
3938 if (isKnownNonZero(Ptr, getSimplifyQuery().getWithInstruction(II)))
3939 return RemoveBundle();
3940
3941 // Fold the assume into metadata if it's valid at the load
3942 if (auto *LI = dyn_cast<LoadInst>(Ptr);
3943 LI &&
3944 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3945 MDNode *MD = MDNode::get(II->getContext(), {});
3946 LI->setMetadata(LLVMContext::MD_nonnull, MD);
3947 LI->setMetadata(LLVMContext::MD_noundef, MD);
3948 return RemoveBundle();
3949 }
3950
3951 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3952 GEP && GEP->isInBounds() &&
3953 !NullPointerIsDefined(II->getFunction(),
3954 Ptr->getType()->getPointerAddressSpace())) {
3955 Builder.CreateNonnullAssumption(GEP->stripInBoundsOffsets());
3956 return RemoveBundle();
3957 }
3958
3959 // TODO: apply nonnull return attributes to calls and invokes
3960 break;
3961 }
3962
3963 case BundleAttr::NoUndef: {
3964 auto [Val] = getAssumeNoUndefInfo(OBU);
3965
3967 return RemoveBundle();
3968
3969 if (auto *LI = dyn_cast<LoadInst>(Val);
3970 LI &&
3971 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3972 LI->setMetadata(LLVMContext::MD_noundef,
3973 MDNode::get(II->getContext(), {}));
3974 return RemoveBundle();
3975 }
3976
3977 } break;
3978
3979 case BundleAttr::SeparateStorage: {
3980 auto [Ptr1, Ptr2] = getAssumeSeparateStorageInfo(OBU);
3981 // Separate storage assumptions apply to the underlying allocations, not
3982 // any particular pointer within them. When evaluating the hints for AA
3983 // purposes we getUnderlyingObject them; by precomputing the answers
3984 // here we can avoid having to do so repeatedly there.
3985 auto MaybeSimplifyHint = [&](const Use &U) {
3986 Value *Hint = U.get();
3987 // Not having a limit is safe because InstCombine removes unreachable
3988 // code.
3989 Value *UnderlyingObject = getUnderlyingObject(Hint, /*MaxLookup*/ 0);
3990 if (Hint != UnderlyingObject)
3991 replaceUse(const_cast<Use &>(U), UnderlyingObject);
3992 };
3993 MaybeSimplifyHint(Ptr1);
3994 MaybeSimplifyHint(Ptr2);
3995 } break;
3996
3997 // TODO: Drop these assumes when they are redundant
3998 case BundleAttr::DereferenceableOrNull:
3999 break;
4000
4001 // This cannot be simplified
4002 case BundleAttr::Cold:
4003 break;
4004 }
4005 }
4006
4007 // If the assume has operand bundles, the folds below will never work, so
4008 // don't bother trying.
4009 if (II->hasOperandBundles())
4010 break;
4011
4012 Value *IIOperand = II->getArgOperand(0);
4013
4014 // Canonicalize assume(a && b) -> assume(a); assume(b);
4015 // Note: New assumption intrinsics created here are registered by
4016 // the InstCombineIRInserter object.
4017 Value *A, *B;
4018 if (match(IIOperand, m_LogicalAnd(m_Value(A), m_Value(B)))) {
4019 Builder.CreateAssumption(A);
4020 Builder.CreateAssumption(B);
4021 return eraseInstFromFunction(*II);
4022 }
4023 // assume(!(a || b)) -> assume(!a); assume(!b);
4024 if (match(IIOperand, m_Not(m_LogicalOr(m_Value(A), m_Value(B))))) {
4025 Builder.CreateAssumption(Builder.CreateNot(A));
4026 Builder.CreateAssumption(Builder.CreateNot(B));
4027 return eraseInstFromFunction(*II);
4028 }
4029
4030 // Convert nonnull assume like:
4031 // %A = icmp ne i32* %PTR, null
4032 // call void @llvm.assume(i1 %A)
4033 // into
4034 // call void @llvm.assume(i1 true) [ "nonnull"(i32* %PTR) ]
4035 if (match(IIOperand,
4037 A->getType()->isPointerTy()) {
4038 Builder.CreateNonnullAssumption(A);
4039 return eraseInstFromFunction(*II);
4040 }
4041
4042 // Convert alignment assume like:
4043 // %B = ptrtoint ptr %A to i64
4044 // %C = and i64 %B, Constant
4045 // %D = icmp eq i64 %C, 0
4046 // call void @llvm.assume(i1 %D)
4047 // into
4048 // call void @llvm.assume(i1 true) [ "align"(ptr [[A]], i64 Constant + 1)]
4049 uint64_t AlignMask = 1;
4050 if ((match(IIOperand, m_Not(m_Trunc(m_Value(A)))) ||
4051 match(IIOperand,
4053 m_And(m_Value(A), m_ConstantInt(AlignMask)),
4054 m_Zero())))) {
4055 if (isPowerOf2_64(AlignMask + 1) &&
4057 Builder.CreateAlignmentAssumption(getDataLayout(), A, AlignMask + 1);
4058 return eraseInstFromFunction(*II);
4059 }
4060 }
4061
4062 // Remove assumes on true/false
4063 if (auto *CI = dyn_cast<ConstantInt>(IIOperand);
4064 CI || isa<UndefValue, PoisonValue>(IIOperand)) {
4065 if (!CI || CI->isZero())
4067 return eraseInstFromFunction(*II);
4068 }
4069
4070 // Update the cache of affected values for this assumption (we might be
4071 // here because we just simplified the condition).
4072 AC.updateAffectedValues(cast<AssumeInst>(II));
4073 break;
4074 }
4075 case Intrinsic::experimental_guard: {
4076 // Is this guard followed by another guard? We scan forward over a small
4077 // fixed window of instructions to handle common cases with conditions
4078 // computed between guards.
4079 Instruction *NextInst = II->getNextNode();
4080 for (unsigned i = 0; i < GuardWideningWindow; i++) {
4081 // Note: Using context-free form to avoid compile time blow up
4082 if (!isSafeToSpeculativelyExecute(NextInst))
4083 break;
4084 NextInst = NextInst->getNextNode();
4085 }
4086 Value *NextCond = nullptr;
4087 if (match(NextInst,
4089 Value *CurrCond = II->getArgOperand(0);
4090
4091 // Remove a guard that it is immediately preceded by an identical guard.
4092 // Otherwise canonicalize guard(a); guard(b) -> guard(a & b).
4093 if (CurrCond != NextCond) {
4094 Instruction *MoveI = II->getNextNode();
4095 while (MoveI != NextInst) {
4096 auto *Temp = MoveI;
4097 MoveI = MoveI->getNextNode();
4098 Temp->moveBefore(II->getIterator());
4099 }
4100 replaceOperand(*II, 0, Builder.CreateAnd(CurrCond, NextCond));
4101 }
4102 eraseInstFromFunction(*NextInst);
4103 return II;
4104 }
4105 break;
4106 }
4107 case Intrinsic::vector_insert: {
4108 Value *Vec = II->getArgOperand(0);
4109 Value *SubVec = II->getArgOperand(1);
4110 Value *Idx = II->getArgOperand(2);
4111 auto *DstTy = dyn_cast<FixedVectorType>(II->getType());
4112 auto *VecTy = dyn_cast<FixedVectorType>(Vec->getType());
4113 auto *SubVecTy = dyn_cast<FixedVectorType>(SubVec->getType());
4114
4115 // Only canonicalize if the destination vector, Vec, and SubVec are all
4116 // fixed vectors.
4117 if (DstTy && VecTy && SubVecTy) {
4118 unsigned DstNumElts = DstTy->getNumElements();
4119 unsigned VecNumElts = VecTy->getNumElements();
4120 unsigned SubVecNumElts = SubVecTy->getNumElements();
4121 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4122
4123 // An insert that entirely overwrites Vec with SubVec is a nop.
4124 if (VecNumElts == SubVecNumElts)
4125 return replaceInstUsesWith(CI, SubVec);
4126
4127 // Widen SubVec into a vector of the same width as Vec, since
4128 // shufflevector requires the two input vectors to be the same width.
4129 // Elements beyond the bounds of SubVec within the widened vector are
4130 // undefined.
4131 SmallVector<int, 8> WidenMask;
4132 unsigned i;
4133 for (i = 0; i != SubVecNumElts; ++i)
4134 WidenMask.push_back(i);
4135 for (; i != VecNumElts; ++i)
4136 WidenMask.push_back(PoisonMaskElem);
4137
4138 Value *WidenShuffle = Builder.CreateShuffleVector(SubVec, WidenMask);
4139
4141 for (unsigned i = 0; i != IdxN; ++i)
4142 Mask.push_back(i);
4143 for (unsigned i = DstNumElts; i != DstNumElts + SubVecNumElts; ++i)
4144 Mask.push_back(i);
4145 for (unsigned i = IdxN + SubVecNumElts; i != DstNumElts; ++i)
4146 Mask.push_back(i);
4147
4148 Value *Shuffle = Builder.CreateShuffleVector(Vec, WidenShuffle, Mask);
4149 return replaceInstUsesWith(CI, Shuffle);
4150 }
4151 break;
4152 }
4153 case Intrinsic::vector_extract: {
4154 Value *Vec = II->getArgOperand(0);
4155 Value *Idx = II->getArgOperand(1);
4156
4157 Type *ReturnType = II->getType();
4158 // (extract_vector (insert_vector InsertTuple, InsertValue, InsertIdx),
4159 // ExtractIdx)
4160 unsigned ExtractIdx = cast<ConstantInt>(Idx)->getZExtValue();
4161 Value *InsertTuple, *InsertIdx, *InsertValue;
4163 m_Value(InsertValue),
4164 m_Value(InsertIdx))) &&
4165 InsertValue->getType() == ReturnType) {
4166 unsigned Index = cast<ConstantInt>(InsertIdx)->getZExtValue();
4167 // Case where we get the same index right after setting it.
4168 // extract.vector(insert.vector(InsertTuple, InsertValue, Idx), Idx) -->
4169 // InsertValue
4170 if (ExtractIdx == Index)
4171 return replaceInstUsesWith(CI, InsertValue);
4172 // If we are getting a different index than what was set in the
4173 // insert.vector intrinsic. We can just set the input tuple to the one up
4174 // in the chain. extract.vector(insert.vector(InsertTuple, InsertValue,
4175 // InsertIndex), ExtractIndex)
4176 // --> extract.vector(InsertTuple, ExtractIndex)
4177 else
4178 return replaceOperand(CI, 0, InsertTuple);
4179 }
4180
4181 ConstantInt *ALMUpperBound;
4183 m_Value(), m_ConstantInt(ALMUpperBound)))) {
4184 const auto &Attrs = II->getFunction()->getAttributes().getFnAttrs();
4185 unsigned VScaleMin = Attrs.getVScaleRangeMin();
4186 unsigned ScaleFactor =
4187 cast<VectorType>(ReturnType)->isScalableTy() ? VScaleMin : 1;
4188 if (ExtractIdx * ScaleFactor >= ALMUpperBound->getZExtValue())
4189 return replaceInstUsesWith(CI,
4190 ConstantVector::getNullValue(ReturnType));
4191 }
4192
4193 auto *DstTy = dyn_cast<VectorType>(ReturnType);
4194 auto *VecTy = dyn_cast<VectorType>(Vec->getType());
4195
4196 if (DstTy && VecTy) {
4197 auto DstEltCnt = DstTy->getElementCount();
4198 auto VecEltCnt = VecTy->getElementCount();
4199 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4200
4201 // Extracting the entirety of Vec is a nop.
4202 if (DstEltCnt == VecTy->getElementCount()) {
4203 replaceInstUsesWith(CI, Vec);
4204 return eraseInstFromFunction(CI);
4205 }
4206
4207 // Only canonicalize to shufflevector if the destination vector and
4208 // Vec are fixed vectors.
4209 if (VecEltCnt.isScalable() || DstEltCnt.isScalable())
4210 break;
4211
4213 for (unsigned i = 0; i != DstEltCnt.getKnownMinValue(); ++i)
4214 Mask.push_back(IdxN + i);
4215
4216 Value *Shuffle = Builder.CreateShuffleVector(Vec, Mask);
4217 return replaceInstUsesWith(CI, Shuffle);
4218 }
4219 break;
4220 }
4221 case Intrinsic::experimental_vp_reverse: {
4222 Value *X;
4223 Value *Vec = II->getArgOperand(0);
4224 Value *Mask = II->getArgOperand(1);
4225 if (!match(Mask, m_AllOnes()))
4226 break;
4227 Value *EVL = II->getArgOperand(2);
4228 // TODO: Canonicalize experimental.vp.reverse after unop/binops?
4229 // rev(unop rev(X)) --> unop X
4230 if (match(Vec,
4232 m_Value(X), m_AllOnes(), m_Specific(EVL)))))) {
4233 auto *OldUnOp = cast<UnaryOperator>(Vec);
4235 OldUnOp->getOpcode(), X, OldUnOp, OldUnOp->getName(),
4236 II->getIterator());
4237 return replaceInstUsesWith(CI, NewUnOp);
4238 }
4239 break;
4240 }
4241 case Intrinsic::vector_reduce_or:
4242 case Intrinsic::vector_reduce_and: {
4243 // Canonicalize logical or/and reductions:
4244 // Or reduction for i1 is represented as:
4245 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4246 // %res = cmp ne iReduxWidth %val, 0
4247 // And reduction for i1 is represented as:
4248 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4249 // %res = cmp eq iReduxWidth %val, 11111
4250 Value *Arg = II->getArgOperand(0);
4251 Value *Vect;
4252
4253 if (Value *NewOp =
4254 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4255 replaceUse(II->getOperandUse(0), NewOp);
4256 return II;
4257 }
4258
4259 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4260 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4261 if (FTy->getElementType() == Builder.getInt1Ty()) {
4262 Value *Res = Builder.CreateBitCast(
4263 Vect, Builder.getIntNTy(FTy->getNumElements()));
4264 if (IID == Intrinsic::vector_reduce_and) {
4265 Res = Builder.CreateICmpEQ(
4267 } else {
4268 assert(IID == Intrinsic::vector_reduce_or &&
4269 "Expected or reduction.");
4270 Res = Builder.CreateIsNotNull(Res);
4271 }
4272 if (Arg != Vect)
4273 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4274 II->getType());
4275 return replaceInstUsesWith(CI, Res);
4276 }
4277 }
4278 [[fallthrough]];
4279 }
4280 case Intrinsic::vector_reduce_add: {
4281 if (IID == Intrinsic::vector_reduce_add) {
4282 // Convert vector_reduce_add(ZExt(<n x i1>)) to
4283 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4284 // Convert vector_reduce_add(SExt(<n x i1>)) to
4285 // -ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4286 // Convert vector_reduce_add(<n x i1>) to
4287 // Trunc(ctpop(bitcast <n x i1> to in)).
4288 Value *Arg = II->getArgOperand(0);
4289 Value *Vect;
4290
4291 if (Value *NewOp =
4292 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4293 replaceUse(II->getOperandUse(0), NewOp);
4294 return II;
4295 }
4296
4297 // vector.reduce.add.vNiM(splat(%x)) -> mul(%x, N)
4298 if (Value *Splat = getSplatValue(Arg)) {
4299 ElementCount VecToReduceCount =
4300 cast<VectorType>(Arg->getType())->getElementCount();
4301 if (VecToReduceCount.isFixed()) {
4302 unsigned VectorSize = VecToReduceCount.getFixedValue();
4303 return BinaryOperator::CreateMul(
4304 Splat,
4305 ConstantInt::get(Splat->getType(), VectorSize, /*IsSigned=*/false,
4306 /*ImplicitTrunc=*/true));
4307 }
4308 }
4309
4310 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4311 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4312 if (FTy->getElementType() == Builder.getInt1Ty()) {
4313 Value *V = Builder.CreateBitCast(
4314 Vect, Builder.getIntNTy(FTy->getNumElements()));
4315 Value *Res = Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
4316 Res = Builder.CreateZExtOrTrunc(Res, II->getType());
4317 if (Arg != Vect &&
4318 cast<Instruction>(Arg)->getOpcode() == Instruction::SExt)
4319 Res = Builder.CreateNeg(Res);
4320 return replaceInstUsesWith(CI, Res);
4321 }
4322 }
4323 }
4324 [[fallthrough]];
4325 }
4326 case Intrinsic::vector_reduce_xor: {
4327 if (IID == Intrinsic::vector_reduce_xor) {
4328 // Exclusive disjunction reduction over the vector with
4329 // (potentially-extended) i1 element type is actually a
4330 // (potentially-extended) arithmetic `add` reduction over the original
4331 // non-extended value:
4332 // vector_reduce_xor(?ext(<n x i1>))
4333 // -->
4334 // ?ext(vector_reduce_add(<n x i1>))
4335 Value *Arg = II->getArgOperand(0);
4336 Value *Vect;
4337
4338 if (Value *NewOp =
4339 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4340 replaceUse(II->getOperandUse(0), NewOp);
4341 return II;
4342 }
4343
4344 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4345 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4346 if (VTy->getElementType() == Builder.getInt1Ty()) {
4347 Value *Res = Builder.CreateAddReduce(Vect);
4348 if (Arg != Vect)
4349 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4350 II->getType());
4351 return replaceInstUsesWith(CI, Res);
4352 }
4353 }
4354 }
4355 [[fallthrough]];
4356 }
4357 case Intrinsic::vector_reduce_mul: {
4358 if (IID == Intrinsic::vector_reduce_mul) {
4359 Value *Arg = II->getArgOperand(0);
4360
4361 if (Value *NewOp =
4362 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4363 replaceUse(II->getOperandUse(0), NewOp);
4364 return II;
4365 }
4366
4367 // vector_reduce_mul(zext(<n x i1>)), or
4368 // vector_reduce_mul(sext(<n x i1>)) (if n is even) -->
4369 // zext(vector_reduce_and(<n x i1>)).
4370 // (The sext case doesn't work if n is odd because multiplying an odd
4371 // number of -1's produces -1, not 1.)
4372 Value *Vect;
4373 bool IsZext = match(Arg, m_ZExt(m_Value(Vect))) &&
4374 Vect->getType()->isIntOrIntVectorTy(1);
4375 bool IsSext =
4376 match(Arg, m_SExt(m_Value(Vect))) &&
4377 Vect->getType()->isIntOrIntVectorTy(1) &&
4378 cast<VectorType>(Vect->getType())->getElementCount().isKnownEven();
4379 if (IsZext || IsSext) {
4380 Value *Res = Builder.CreateAndReduce(Vect);
4381 return CastInst::Create(Instruction::ZExt, Res, II->getType());
4382 }
4383
4384 // vector_reduce_mul(<n x i1>) --> vector_reduce_and(<n x i1>)
4385 if (Arg->getType()->isIntOrIntVectorTy(1))
4386 return replaceInstUsesWith(CI, Builder.CreateAndReduce(Arg));
4387 }
4388 [[fallthrough]];
4389 }
4390 case Intrinsic::vector_reduce_umin:
4391 case Intrinsic::vector_reduce_umax: {
4392 if (IID == Intrinsic::vector_reduce_umin ||
4393 IID == Intrinsic::vector_reduce_umax) {
4394 // UMin/UMax reduction over the vector with (potentially-extended)
4395 // i1 element type is actually a (potentially-extended)
4396 // logical `and`/`or` reduction over the original non-extended value:
4397 // vector_reduce_u{min,max}(?ext(<n x i1>))
4398 // -->
4399 // ?ext(vector_reduce_{and,or}(<n x i1>))
4400 Value *Arg = II->getArgOperand(0);
4401 Value *Vect;
4402
4403 if (Value *NewOp =
4404 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4405 replaceUse(II->getOperandUse(0), NewOp);
4406 return II;
4407 }
4408
4409 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4410 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4411 if (VTy->getElementType() == Builder.getInt1Ty()) {
4412 Value *Res = IID == Intrinsic::vector_reduce_umin
4413 ? Builder.CreateAndReduce(Vect)
4414 : Builder.CreateOrReduce(Vect);
4415 if (Arg != Vect)
4416 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4417 II->getType());
4418 return replaceInstUsesWith(CI, Res);
4419 }
4420 }
4421 }
4422 [[fallthrough]];
4423 }
4424 case Intrinsic::vector_reduce_smin:
4425 case Intrinsic::vector_reduce_smax: {
4426 if (IID == Intrinsic::vector_reduce_smin ||
4427 IID == Intrinsic::vector_reduce_smax) {
4428 // SMin/SMax reduction over the vector with (potentially-extended)
4429 // i1 element type is actually a (potentially-extended)
4430 // logical `and`/`or` reduction over the original non-extended value:
4431 // vector_reduce_s{min,max}(<n x i1>)
4432 // -->
4433 // vector_reduce_{or,and}(<n x i1>)
4434 // and
4435 // vector_reduce_s{min,max}(sext(<n x i1>))
4436 // -->
4437 // sext(vector_reduce_{or,and}(<n x i1>))
4438 // and
4439 // vector_reduce_s{min,max}(zext(<n x i1>))
4440 // -->
4441 // zext(vector_reduce_{and,or}(<n x i1>))
4442 Value *Arg = II->getArgOperand(0);
4443 Value *Vect;
4444
4445 if (Value *NewOp =
4446 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4447 replaceUse(II->getOperandUse(0), NewOp);
4448 return II;
4449 }
4450
4451 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4452 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4453 if (VTy->getElementType() == Builder.getInt1Ty()) {
4454 Instruction::CastOps ExtOpc = Instruction::CastOps::CastOpsEnd;
4455 if (Arg != Vect)
4456 ExtOpc = cast<CastInst>(Arg)->getOpcode();
4457 Value *Res = ((IID == Intrinsic::vector_reduce_smin) ==
4458 (ExtOpc == Instruction::CastOps::ZExt))
4459 ? Builder.CreateAndReduce(Vect)
4460 : Builder.CreateOrReduce(Vect);
4461 if (Arg != Vect)
4462 Res = Builder.CreateCast(ExtOpc, Res, II->getType());
4463 return replaceInstUsesWith(CI, Res);
4464 }
4465 }
4466 }
4467 [[fallthrough]];
4468 }
4469 case Intrinsic::vector_reduce_fmax:
4470 case Intrinsic::vector_reduce_fmin:
4471 case Intrinsic::vector_reduce_fadd:
4472 case Intrinsic::vector_reduce_fmul: {
4473 bool CanReorderLanes = (IID != Intrinsic::vector_reduce_fadd &&
4474 IID != Intrinsic::vector_reduce_fmul) ||
4475 II->hasAllowReassoc();
4476 const unsigned ArgIdx = (IID == Intrinsic::vector_reduce_fadd ||
4477 IID == Intrinsic::vector_reduce_fmul)
4478 ? 1
4479 : 0;
4480 Value *Arg = II->getArgOperand(ArgIdx);
4481 if (Value *NewOp = simplifyReductionOperand(Arg, CanReorderLanes)) {
4482 replaceUse(II->getOperandUse(ArgIdx), NewOp);
4483 return nullptr;
4484 }
4485 break;
4486 }
4487 case Intrinsic::is_fpclass: {
4488 if (Instruction *I = foldIntrinsicIsFPClass(*II))
4489 return I;
4490 break;
4491 }
4492 case Intrinsic::threadlocal_address: {
4493 Align MinAlign = getKnownAlignment(II->getArgOperand(0), DL, II, &AC, &DT);
4494 MaybeAlign Align = II->getRetAlign();
4495 if (MinAlign > Align.valueOrOne()) {
4496 II->addRetAttr(Attribute::getWithAlignment(II->getContext(), MinAlign));
4497 return II;
4498 }
4499 break;
4500 }
4501 case Intrinsic::fptoui_sat:
4502 case Intrinsic::fptosi_sat:
4503 if (Instruction *I = foldItoFPtoI(*II))
4504 return I;
4505 break;
4506 case Intrinsic::frexp: {
4507 // frexp(frexp(x).fract) -> { frexp(x).fract, 0 }: the fraction operand is
4508 // already normalized, so the first result is idempotent and the second is
4509 // zero.
4510 if (match(II->getArgOperand(0),
4512 Value *Res = Builder.CreateInsertValue(PoisonValue::get(II->getType()),
4513 II->getArgOperand(0), 0);
4514 Res = Builder.CreateInsertValue(
4515 Res, Constant::getNullValue(II->getType()->getStructElementType(1)),
4516 1);
4517 return replaceInstUsesWith(*II, Res);
4518 }
4519 break;
4520 }
4521 case Intrinsic::get_active_lane_mask: {
4522 const APInt *Op0, *Op1;
4523 if (match(II->getOperand(0), m_StrictlyPositive(Op0)) &&
4524 match(II->getOperand(1), m_APInt(Op1))) {
4525 Type *OpTy = II->getOperand(0)->getType();
4526 return replaceInstUsesWith(
4527 *II, Builder.CreateIntrinsic(
4528 II->getType(), Intrinsic::get_active_lane_mask,
4529 {Constant::getNullValue(OpTy),
4530 ConstantInt::get(OpTy, Op1->usub_sat(*Op0))}));
4531 }
4532 break;
4533 }
4534 case Intrinsic::experimental_get_vector_length: {
4535 // get.vector.length(Cnt, MaxLanes) --> Cnt when Cnt <= MaxLanes
4536 unsigned BitWidth =
4537 std::max(II->getArgOperand(0)->getType()->getScalarSizeInBits(),
4538 II->getType()->getScalarSizeInBits());
4539 ConstantRange Cnt =
4540 computeConstantRangeIncludingKnownBits(II->getArgOperand(0), false,
4541 SQ.getWithInstruction(II))
4543 ConstantRange MaxLanes = cast<ConstantInt>(II->getArgOperand(1))
4544 ->getValue()
4545 .zextOrTrunc(Cnt.getBitWidth());
4546 if (cast<ConstantInt>(II->getArgOperand(2))->isOne())
4547 MaxLanes = MaxLanes.multiply(
4548 getVScaleRange(II->getFunction(), Cnt.getBitWidth()));
4549
4550 if (Cnt.icmp(CmpInst::ICMP_ULE, MaxLanes))
4551 return replaceInstUsesWith(
4552 *II, Builder.CreateZExtOrTrunc(II->getArgOperand(0), II->getType()));
4553 return nullptr;
4554 }
4555 default: {
4556 // Handle target specific intrinsics
4557 std::optional<Instruction *> V = targetInstCombineIntrinsic(*II);
4558 if (V)
4559 return *V;
4560 break;
4561 }
4562 }
4563
4564 // Try to fold intrinsic into select/phi operands. This is legal if:
4565 // * The intrinsic is speculatable.
4566 // * The operand is one of the following:
4567 // - a phi.
4568 // - a select with a scalar condition.
4569 // - a select with a vector condition and II is not a cross lane operation.
4571 for (Value *Op : II->args()) {
4572 if (auto *Sel = dyn_cast<SelectInst>(Op)) {
4573 bool IsVectorCond = Sel->getCondition()->getType()->isVectorTy();
4574 if (IsVectorCond &&
4575 (!isNotCrossLaneOperation(II) || !II->getType()->isVectorTy()))
4576 continue;
4577 // Don't replace a scalar select with a more expensive vector select if
4578 // we can't simplify both arms of the select.
4579 bool SimplifyBothArms =
4580 !Op->getType()->isVectorTy() && II->getType()->isVectorTy();
4582 *II, Sel, /*FoldWithMultiUse=*/false, SimplifyBothArms))
4583 return R;
4584 }
4585 if (auto *Phi = dyn_cast<PHINode>(Op))
4586 if (Instruction *R = foldOpIntoPhi(*II, Phi))
4587 return R;
4588 }
4589 }
4590
4592 return Shuf;
4593
4595 return replaceInstUsesWith(*II, Reverse);
4596
4598 return replaceInstUsesWith(*II, Res);
4599
4600 // Some intrinsics (like experimental_gc_statepoint) can be used in invoke
4601 // context, so it is handled in visitCallBase and we should trigger it.
4602 return visitCallBase(*II);
4603}
4604
4605// Fence instruction simplification
4607 auto *NFI = dyn_cast<FenceInst>(FI.getNextNode());
4608 // This check is solely here to handle arbitrary target-dependent syncscopes.
4609 // TODO: Can remove if does not matter in practice.
4610 if (NFI && FI.isIdenticalTo(NFI))
4611 return eraseInstFromFunction(FI);
4612
4613 // Returns true if FI1 is identical or stronger fence than FI2.
4614 auto isIdenticalOrStrongerFence = [](FenceInst *FI1, FenceInst *FI2) {
4615 auto FI1SyncScope = FI1->getSyncScopeID();
4616 // Consider same scope, where scope is global or single-thread.
4617 if (FI1SyncScope != FI2->getSyncScopeID() ||
4618 (FI1SyncScope != SyncScope::System &&
4619 FI1SyncScope != SyncScope::SingleThread))
4620 return false;
4621
4622 return isAtLeastOrStrongerThan(FI1->getOrdering(), FI2->getOrdering());
4623 };
4624 if (NFI && isIdenticalOrStrongerFence(NFI, &FI))
4625 return eraseInstFromFunction(FI);
4626
4627 if (auto *PFI = dyn_cast_or_null<FenceInst>(FI.getPrevNode()))
4628 if (isIdenticalOrStrongerFence(PFI, &FI))
4629 return eraseInstFromFunction(FI);
4630 return nullptr;
4631}
4632
4633// InvokeInst simplification
4635 return visitCallBase(II);
4636}
4637
4638// CallBrInst simplification
4640 return visitCallBase(CBI);
4641}
4642
4643// A simple parser for format string specifiers for the purposes of the
4644// modular-format attribute. In the case of malformed format strings this might
4645// under or over report the specifiers present, but such cases are undefined
4646// behavior.
4648 Bitset<256> Specifiers;
4649 for (size_t I = 0; I < FormatStr.size(); ++I) {
4650 if (FormatStr[I] != '%')
4651 continue;
4652
4653 // Check for escaped '%'.
4654 if (I + 1 < FormatStr.size() && FormatStr[I + 1] == '%') {
4655 ++I; // Skip the second '%'.
4656 continue;
4657 }
4658
4659 // Scan past allowed prefix characters.
4660 size_t J =
4661 FormatStr.find_first_not_of("0123456789-+ #0$.*'hlLjztqwvI", I + 1);
4662 if (J == StringRef::npos)
4663 break;
4664
4665 Specifiers.set(static_cast<unsigned char>(FormatStr[J]));
4666 I = J; // Resume search from after the specifier.
4667 }
4668 return Specifiers;
4669}
4670
4671static bool isAspectNeeded(StringRef Aspect, CallInst *CI,
4672 std::optional<unsigned> FirstArgIdx,
4673 const std::optional<Bitset<256>> &Specifiers) {
4674 if (Aspect == "float") {
4675 if (Specifiers) {
4676 static constexpr Bitset<256> FloatSpecifiers{'f', 'F', 'e', 'E',
4677 'g', 'G', 'a', 'A'};
4678 return (*Specifiers & FloatSpecifiers).any();
4679 }
4680 // Fallback to type-based check for dynamic format string.
4681 if (!FirstArgIdx)
4682 return true;
4683 return llvm::any_of(
4684 llvm::make_range(std::next(CI->arg_begin(), *FirstArgIdx),
4685 CI->arg_end()),
4686 [](Value *V) { return V->getType()->isFloatingPointTy(); });
4687 }
4688 if (Aspect == "fixed") {
4689 if (Specifiers) {
4690 static constexpr Bitset<256> FixedSpecifiers{'r', 'R', 'k', 'K'};
4691 return (*Specifiers & FixedSpecifiers).any();
4692 }
4693 // Fallback for fixed-point: assume needed if format is dynamic.
4694 return true;
4695 }
4696 // Unknown aspects are always considered to be needed.
4697 return true;
4698}
4699
4700static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M,
4701 IRBuilderBase &B) {
4702 SmallString<20> Name = ImplName;
4703 Name += '_';
4704 Name += Aspect;
4705 LLVMContext &Ctx = M->getContext();
4706 Function *RelocNoneFn =
4707 Intrinsic::getOrInsertDeclaration(M, Intrinsic::reloc_none);
4708 B.CreateCall(RelocNoneFn,
4709 {MetadataAsValue::get(Ctx, MDString::get(Ctx, Name))});
4710}
4711
4713 if (!CI->hasFnAttr("modular-format"))
4714 return nullptr;
4715
4717 llvm::split(CI->getFnAttr("modular-format").getValueAsString(), ','));
4718 if (Args.size() < 5)
4719 return nullptr;
4720
4721 StringRef FormatIdxStr = Args[1];
4722 StringRef FirstArgIdxStr = Args[2];
4723 StringRef FnName = Args[3];
4724 StringRef ImplName = Args[4];
4726
4727 unsigned FormatIdx;
4728 std::optional<unsigned> FirstArgIdx;
4729 [[maybe_unused]] bool Error;
4730 Error = FormatIdxStr.getAsInteger(10, FormatIdx);
4731 assert(!Error && "invalid format arg index");
4732 --FormatIdx; // 1-based to 0-based
4733
4734 FirstArgIdx.emplace();
4735 Error = FirstArgIdxStr.getAsInteger(10, *FirstArgIdx);
4736 assert(!Error && "invalid first arg index");
4737 if (*FirstArgIdx > 0)
4738 --*FirstArgIdx; // 1-based to 0-based
4739 else
4740 FirstArgIdx.reset();
4741
4742 if (AllAspects.empty())
4743 return nullptr;
4744
4745 Value *FormatVal = CI->getArgOperand(FormatIdx);
4746 StringRef FormatStr;
4747
4748 std::optional<Bitset<256>> Specifiers;
4749 if (getConstantStringInfo(FormatVal, FormatStr))
4750 Specifiers = parseFormatStringSpecifiers(FormatStr);
4751
4752 SmallVector<StringRef> NeededAspects;
4753 for (StringRef Aspect : AllAspects)
4754 if (isAspectNeeded(Aspect, CI, FirstArgIdx, Specifiers))
4755 NeededAspects.push_back(Aspect);
4756
4757 if (NeededAspects.size() == AllAspects.size())
4758 return nullptr;
4759
4760 Module *M = CI->getModule();
4761 LLVMContext &Ctx = M->getContext();
4762 Function *Callee = CI->getCalledFunction();
4763 FunctionCallee ModularFn = M->getOrInsertFunction(
4764 FnName, Callee->getFunctionType(),
4765 Callee->getAttributes().removeFnAttribute(Ctx, "modular-format"));
4766 CallInst *New = cast<CallInst>(CI->clone());
4767 New->setCalledFunction(ModularFn);
4768 New->removeFnAttr("modular-format");
4769 B.Insert(New);
4770
4771 llvm::sort(NeededAspects);
4772 for (StringRef Request : NeededAspects)
4773 referenceAspect(Request, ImplName, M, B);
4774
4775 return New;
4776}
4777
4778Instruction *InstCombinerImpl::tryOptimizeCall(CallInst *CI) {
4779 if (!CI->getCalledFunction()) return nullptr;
4780
4781 // Skip optimizing notail and musttail calls so
4782 // LibCallSimplifier::optimizeCall doesn't have to preserve those invariants.
4783 // LibCallSimplifier::optimizeCall should try to preserve tail calls though.
4784 if (CI->isMustTailCall() || CI->isNoTailCall())
4785 return nullptr;
4786
4787 auto InstCombineRAUW = [this](Instruction *From, Value *With) {
4788 replaceInstUsesWith(*From, With);
4789 };
4790 auto InstCombineErase = [this](Instruction *I) {
4792 };
4793 LibCallSimplifier Simplifier(DL, &TLI, &DT, &DC, &AC, ORE, BFI, PSI,
4794 InstCombineRAUW, InstCombineErase);
4795 if (Value *With = Simplifier.optimizeCall(CI, Builder)) {
4796 ++NumSimplified;
4797 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4798 }
4799 if (Value *With = optimizeModularFormat(CI, Builder)) {
4800 ++NumSimplified;
4801 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4802 }
4803
4804 return nullptr;
4805}
4806
4808 // Strip off at most one level of pointer casts, looking for an alloca. This
4809 // is good enough in practice and simpler than handling any number of casts.
4810 Value *Underlying = TrampMem->stripPointerCasts();
4811 if (Underlying != TrampMem &&
4812 (!Underlying->hasOneUse() || Underlying->user_back() != TrampMem))
4813 return nullptr;
4814 if (!isa<AllocaInst>(Underlying))
4815 return nullptr;
4816
4817 IntrinsicInst *InitTrampoline = nullptr;
4818 for (User *U : TrampMem->users()) {
4820 if (!II)
4821 return nullptr;
4822 if (II->getIntrinsicID() == Intrinsic::init_trampoline) {
4823 if (InitTrampoline)
4824 // More than one init_trampoline writes to this value. Give up.
4825 return nullptr;
4826 InitTrampoline = II;
4827 continue;
4828 }
4829 if (II->getIntrinsicID() == Intrinsic::adjust_trampoline)
4830 // Allow any number of calls to adjust.trampoline.
4831 continue;
4832 return nullptr;
4833 }
4834
4835 // No call to init.trampoline found.
4836 if (!InitTrampoline)
4837 return nullptr;
4838
4839 // Check that the alloca is being used in the expected way.
4840 if (InitTrampoline->getOperand(0) != TrampMem)
4841 return nullptr;
4842
4843 return InitTrampoline;
4844}
4845
4847 Value *TrampMem) {
4848 // Visit all the previous instructions in the basic block, and try to find a
4849 // init.trampoline which has a direct path to the adjust.trampoline.
4850 for (BasicBlock::iterator I = AdjustTramp->getIterator(),
4851 E = AdjustTramp->getParent()->begin();
4852 I != E;) {
4853 Instruction *Inst = &*--I;
4855 if (II->getIntrinsicID() == Intrinsic::init_trampoline &&
4856 II->getOperand(0) == TrampMem)
4857 return II;
4858 if (Inst->mayWriteToMemory())
4859 return nullptr;
4860 }
4861 return nullptr;
4862}
4863
4864// Given a call to llvm.adjust.trampoline, find and return the corresponding
4865// call to llvm.init.trampoline if the call to the trampoline can be optimized
4866// to a direct call to a function. Otherwise return NULL.
4868 Callee = Callee->stripPointerCasts();
4869 IntrinsicInst *AdjustTramp = dyn_cast<IntrinsicInst>(Callee);
4870 if (!AdjustTramp ||
4871 AdjustTramp->getIntrinsicID() != Intrinsic::adjust_trampoline)
4872 return nullptr;
4873
4874 Value *TrampMem = AdjustTramp->getOperand(0);
4875
4877 return IT;
4878 if (IntrinsicInst *IT = findInitTrampolineFromBB(AdjustTramp, TrampMem))
4879 return IT;
4880 return nullptr;
4881}
4882
4883Instruction *InstCombinerImpl::foldPtrAuthIntrinsicCallee(CallBase &Call) {
4884 const Value *Callee = Call.getCalledOperand();
4885 const auto *IPC = dyn_cast<IntToPtrInst>(Callee);
4886 if (!IPC || !IPC->isNoopCast(DL))
4887 return nullptr;
4888
4889 const auto *II = dyn_cast<IntrinsicInst>(IPC->getOperand(0));
4890 if (!II)
4891 return nullptr;
4892
4893 Intrinsic::ID IIID = II->getIntrinsicID();
4894 if (IIID != Intrinsic::ptrauth_resign && IIID != Intrinsic::ptrauth_sign)
4895 return nullptr;
4896
4897 // Isolate the ptrauth bundle from the others.
4898 std::optional<OperandBundleUse> PtrAuthBundleOrNone;
4900 for (unsigned BI = 0, BE = Call.getNumOperandBundles(); BI != BE; ++BI) {
4901 OperandBundleUse Bundle = Call.getOperandBundleAt(BI);
4902 if (Bundle.getTagID() == LLVMContext::OB_ptrauth)
4903 PtrAuthBundleOrNone = Bundle;
4904 else
4905 NewBundles.emplace_back(Bundle);
4906 }
4907
4908 if (!PtrAuthBundleOrNone)
4909 return nullptr;
4910
4911 Value *NewCallee = nullptr;
4912 switch (IIID) {
4913 // call(ptrauth.resign(p)), ["ptrauth"()] -> call p, ["ptrauth"()]
4914 // assuming the call bundle and the sign operands match.
4915 case Intrinsic::ptrauth_resign: {
4916 // Resign result key should match bundle.
4917 if (II->getOperand(3) != PtrAuthBundleOrNone->Inputs[0])
4918 return nullptr;
4919 // Resign result discriminator should match bundle.
4920 if (II->getOperand(4) != PtrAuthBundleOrNone->Inputs[1])
4921 return nullptr;
4922
4923 // Resign input (auth) key should also match: we can't change the key on
4924 // the new call we're generating, because we don't know what keys are valid.
4925 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4926 return nullptr;
4927
4928 Value *NewBundleOps[] = {II->getOperand(1), II->getOperand(2)};
4929 NewBundles.emplace_back("ptrauth", NewBundleOps);
4930 NewCallee = II->getOperand(0);
4931 break;
4932 }
4933
4934 // call(ptrauth.sign(p)), ["ptrauth"()] -> call p
4935 // assuming the call bundle and the sign operands match.
4936 // Non-ptrauth indirect calls are undesirable, but so is ptrauth.sign.
4937 case Intrinsic::ptrauth_sign: {
4938 // Sign key should match bundle.
4939 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4940 return nullptr;
4941 // Sign discriminator should match bundle.
4942 if (II->getOperand(2) != PtrAuthBundleOrNone->Inputs[1])
4943 return nullptr;
4944 NewCallee = II->getOperand(0);
4945 break;
4946 }
4947 default:
4948 llvm_unreachable("unexpected intrinsic ID");
4949 }
4950
4951 if (!NewCallee)
4952 return nullptr;
4953
4954 NewCallee = Builder.CreateBitOrPointerCast(NewCallee, Callee->getType());
4955 CallBase *NewCall = CallBase::Create(&Call, NewBundles);
4956 NewCall->setCalledOperand(NewCallee);
4957 return NewCall;
4958}
4959
4960Instruction *InstCombinerImpl::foldPtrAuthConstantCallee(CallBase &Call) {
4962 if (!CPA)
4963 return nullptr;
4964
4965 auto *CalleeF = dyn_cast<Function>(CPA->getPointer());
4966 // If the ptrauth constant isn't based on a function pointer, bail out.
4967 if (!CalleeF)
4968 return nullptr;
4969
4970 // Inspect the call ptrauth bundle to check it matches the ptrauth constant.
4972 if (!PAB)
4973 return nullptr;
4974
4975 auto *Key = cast<ConstantInt>(PAB->Inputs[0]);
4976 Value *Discriminator = PAB->Inputs[1];
4977
4978 // If the bundle doesn't match, this is probably going to fail to auth.
4979 if (!CPA->isKnownCompatibleWith(Key, Discriminator, DL))
4980 return nullptr;
4981
4982 // If the bundle matches the constant, proceed in making this a direct call.
4984 NewCall->setCalledOperand(CalleeF);
4985 return NewCall;
4986}
4987
4988bool InstCombinerImpl::annotateAnyAllocSite(CallBase &Call,
4989 const TargetLibraryInfo *TLI) {
4990 // Note: We only handle cases which can't be driven from generic attributes
4991 // here. So, for example, nonnull and noalias (which are common properties
4992 // of some allocation functions) are expected to be handled via annotation
4993 // of the respective allocator declaration with generic attributes.
4994 bool Changed = false;
4995
4996 if (!Call.getType()->isPointerTy())
4997 return Changed;
4998
4999 std::optional<APInt> Size = getAllocSize(&Call, TLI);
5000 if (Size && *Size != 0) {
5001 // TODO: We really should just emit deref_or_null here and then
5002 // let the generic inference code combine that with nonnull.
5003 if (Call.hasRetAttr(Attribute::NonNull)) {
5004 Changed = !Call.hasRetAttr(Attribute::Dereferenceable);
5006 Call.getContext(), Size->getLimitedValue()));
5007 } else {
5008 Changed = !Call.hasRetAttr(Attribute::DereferenceableOrNull);
5010 Call.getContext(), Size->getLimitedValue()));
5011 }
5012 }
5013
5014 // Add alignment attribute if alignment is a power of two constant.
5016 if (!Alignment)
5017 return Changed;
5018
5019 ConstantInt *AlignOpC = dyn_cast<ConstantInt>(Alignment);
5020 if (AlignOpC && AlignOpC->getValue().ult(llvm::Value::MaximumAlignment)) {
5021 uint64_t AlignmentVal = AlignOpC->getZExtValue();
5022 if (llvm::isPowerOf2_64(AlignmentVal)) {
5023 Align ExistingAlign = Call.getRetAlign().valueOrOne();
5024 Align NewAlign = Align(AlignmentVal);
5025 if (NewAlign > ExistingAlign) {
5028 Changed = true;
5029 }
5030 }
5031 }
5032 return Changed;
5033}
5034
5035/// Improvements for call, callbr and invoke instructions.
5036Instruction *InstCombinerImpl::visitCallBase(CallBase &Call) {
5037 bool Changed = annotateAnyAllocSite(Call, &TLI);
5038
5039 // Mark any parameters that are known to be non-null with the nonnull
5040 // attribute. This is helpful for inlining calls to functions with null
5041 // checks on their arguments.
5042 SmallVector<unsigned, 4> ArgNos;
5043 unsigned ArgNo = 0;
5044
5045 for (Value *V : Call.args()) {
5046 if (V->getType()->isPointerTy()) {
5047 // Simplify the nonnull operand if the parameter is known to be nonnull.
5048 // Otherwise, try to infer nonnull for it.
5049 bool HasDereferenceable = Call.getParamDereferenceableBytes(ArgNo) > 0;
5050 if (Call.paramHasAttr(ArgNo, Attribute::NonNull) ||
5051 (HasDereferenceable &&
5053 V->getType()->getPointerAddressSpace()))) {
5054 if (Value *Res = simplifyNonNullOperand(V, HasDereferenceable)) {
5055 replaceOperand(Call, ArgNo, Res);
5056 Changed = true;
5057 }
5058 } else if (isKnownNonZero(V,
5059 getSimplifyQuery().getWithInstruction(&Call))) {
5060 ArgNos.push_back(ArgNo);
5061 }
5062 }
5063 ArgNo++;
5064 }
5065
5066 assert(ArgNo == Call.arg_size() && "Call arguments not processed correctly.");
5067
5068 if (!ArgNos.empty()) {
5069 AttributeList AS = Call.getAttributes();
5070 LLVMContext &Ctx = Call.getContext();
5071 AS = AS.addParamAttribute(Ctx, ArgNos,
5072 Attribute::get(Ctx, Attribute::NonNull));
5073 Call.setAttributes(AS);
5074 Changed = true;
5075 }
5076
5077 // If the callee is a pointer to a function, attempt to move any casts to the
5078 // arguments of the call/callbr/invoke.
5080 Function *CalleeF = dyn_cast<Function>(Callee);
5081 if ((!CalleeF || CalleeF->getFunctionType() != Call.getFunctionType()) &&
5082 transformConstExprCastCall(Call))
5083 return nullptr;
5084
5085 if (CalleeF) {
5086 // Remove the convergent attr on calls when the callee is not convergent.
5087 if (Call.isConvergent() && !CalleeF->isConvergent() &&
5088 !CalleeF->isIntrinsic()) {
5089 LLVM_DEBUG(dbgs() << "Removing convergent attr from instr " << Call
5090 << "\n");
5092 return &Call;
5093 }
5094
5095 // If the call and callee calling conventions don't match, and neither one
5096 // of the calling conventions is compatible with C calling convention
5097 // this call must be unreachable, as the call is undefined.
5098 if ((CalleeF->getCallingConv() != Call.getCallingConv() &&
5099 !(CalleeF->getCallingConv() == llvm::CallingConv::C &&
5103 // Only do this for calls to a function with a body. A prototype may
5104 // not actually end up matching the implementation's calling conv for a
5105 // variety of reasons (e.g. it may be written in assembly).
5106 !CalleeF->isDeclaration()) {
5107 Instruction *OldCall = &Call;
5109 // If OldCall does not return void then replaceInstUsesWith poison.
5110 // This allows ValueHandlers and custom metadata to adjust itself.
5111 if (!OldCall->getType()->isVoidTy())
5112 replaceInstUsesWith(*OldCall, PoisonValue::get(OldCall->getType()));
5113 if (isa<CallInst>(OldCall))
5114 return eraseInstFromFunction(*OldCall);
5115
5116 // We cannot remove an invoke or a callbr, because it would change thexi
5117 // CFG, just change the callee to a null pointer.
5118 cast<CallBase>(OldCall)->setCalledFunction(
5119 CalleeF->getFunctionType(),
5120 Constant::getNullValue(CalleeF->getType()));
5121 return nullptr;
5122 }
5123 }
5124
5125 // Calling a null function pointer is undefined if a null address isn't
5126 // dereferenceable.
5127 if ((isa<ConstantPointerNull>(Callee) &&
5129 isa<UndefValue>(Callee)) {
5130 // If Call does not return void then replaceInstUsesWith poison.
5131 // This allows ValueHandlers and custom metadata to adjust itself.
5132 if (!Call.getType()->isVoidTy())
5134
5135 if (Call.isTerminator()) {
5136 // Can't remove an invoke or callbr because we cannot change the CFG.
5137 return nullptr;
5138 }
5139
5140 // This instruction is not reachable, just remove it.
5143 }
5144
5145 if (IntrinsicInst *II = findInitTrampoline(Callee))
5146 return transformCallThroughTrampoline(Call, *II);
5147
5148 // Combine calls involving pointer authentication intrinsics.
5149 if (Instruction *NewCall = foldPtrAuthIntrinsicCallee(Call))
5150 return NewCall;
5151
5152 // Combine calls to ptrauth constants.
5153 if (Instruction *NewCall = foldPtrAuthConstantCallee(Call))
5154 return NewCall;
5155
5156 if (isa<InlineAsm>(Callee) && !Call.doesNotThrow()) {
5157 InlineAsm *IA = cast<InlineAsm>(Callee);
5158 if (!IA->canThrow()) {
5159 // Normal inline asm calls cannot throw - mark them
5160 // 'nounwind'.
5162 Changed = true;
5163 }
5164 }
5165
5166 // Try to optimize the call if possible, we require DataLayout for most of
5167 // this. None of these calls are seen as possibly dead so go ahead and
5168 // delete the instruction now.
5169 if (CallInst *CI = dyn_cast<CallInst>(&Call)) {
5170 Instruction *I = tryOptimizeCall(CI);
5171 // If we changed something return the result, etc. Otherwise let
5172 // the fallthrough check.
5173 if (I) return eraseInstFromFunction(*I);
5174 }
5175
5176 if (!Call.use_empty() && !Call.isMustTailCall())
5177 if (Value *ReturnedArg = Call.getReturnedArgOperand()) {
5178 Type *CallTy = Call.getType();
5179 Type *RetArgTy = ReturnedArg->getType();
5180 if (RetArgTy->canLosslesslyBitCastTo(CallTy))
5181 return replaceInstUsesWith(
5182 Call, Builder.CreateBitOrPointerCast(ReturnedArg, CallTy));
5183 }
5184
5185 // Drop unnecessary callee_type metadata from calls that were converted
5186 // into direct calls.
5187 if (Call.getMetadata(LLVMContext::MD_callee_type) && !Call.isIndirectCall()) {
5188 Call.setMetadata(LLVMContext::MD_callee_type, nullptr);
5189 Changed = true;
5190 }
5191
5192 // Drop unnecessary kcfi operand bundles from calls that were converted
5193 // into direct calls.
5195 if (Bundle && !Call.isIndirectCall()) {
5196 DEBUG_WITH_TYPE(DEBUG_TYPE "-kcfi", {
5197 if (CalleeF) {
5198 ConstantInt *FunctionType = nullptr;
5199 ConstantInt *ExpectedType = cast<ConstantInt>(Bundle->Inputs[0]);
5200
5201 if (MDNode *MD = CalleeF->getMetadata(LLVMContext::MD_kcfi_type))
5202 FunctionType = mdconst::extract<ConstantInt>(MD->getOperand(0));
5203
5204 if (FunctionType &&
5205 FunctionType->getZExtValue() != ExpectedType->getZExtValue())
5206 dbgs() << Call.getModule()->getName()
5207 << ": warning: kcfi: " << Call.getCaller()->getName()
5208 << ": call to " << CalleeF->getName()
5209 << " using a mismatching function pointer type\n";
5210 }
5211 });
5212
5214 }
5215
5216 if (isRemovableAlloc(&Call, &TLI))
5217 return visitAllocSite(Call);
5218
5219 // Handle intrinsics which can be used in both call and invoke context.
5220 switch (Call.getIntrinsicID()) {
5221 case Intrinsic::experimental_gc_statepoint: {
5222 GCStatepointInst &GCSP = *cast<GCStatepointInst>(&Call);
5223 SmallPtrSet<Value *, 32> LiveGcValues;
5224 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5225 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5226
5227 // Remove the relocation if unused.
5228 if (GCR.use_empty()) {
5230 continue;
5231 }
5232
5233 Value *DerivedPtr = GCR.getDerivedPtr();
5234 Value *BasePtr = GCR.getBasePtr();
5235
5236 // Undef is undef, even after relocation.
5237 if (isa<UndefValue>(DerivedPtr) || isa<UndefValue>(BasePtr)) {
5240 continue;
5241 }
5242
5243 if (auto *PT = dyn_cast<PointerType>(GCR.getType())) {
5244 // The relocation of null will be null for most any collector.
5245 // TODO: provide a hook for this in GCStrategy. There might be some
5246 // weird collector this property does not hold for.
5247 if (isa<ConstantPointerNull>(DerivedPtr)) {
5248 // Use null-pointer of gc_relocate's type to replace it.
5251 continue;
5252 }
5253
5254 // isKnownNonNull -> nonnull attribute
5255 if (!GCR.hasRetAttr(Attribute::NonNull) &&
5256 isKnownNonZero(DerivedPtr,
5257 getSimplifyQuery().getWithInstruction(&Call))) {
5258 GCR.addRetAttr(Attribute::NonNull);
5259 // We discovered new fact, re-check users.
5260 Worklist.pushUsersToWorkList(GCR);
5261 }
5262 }
5263
5264 // If we have two copies of the same pointer in the statepoint argument
5265 // list, canonicalize to one. This may let us common gc.relocates.
5266 if (GCR.getBasePtr() == GCR.getDerivedPtr() &&
5267 GCR.getBasePtrIndex() != GCR.getDerivedPtrIndex()) {
5268 auto *OpIntTy = GCR.getOperand(2)->getType();
5269 GCR.setOperand(2, ConstantInt::get(OpIntTy, GCR.getBasePtrIndex()));
5270 }
5271
5272 // TODO: bitcast(relocate(p)) -> relocate(bitcast(p))
5273 // Canonicalize on the type from the uses to the defs
5274
5275 // TODO: relocate((gep p, C, C2, ...)) -> gep(relocate(p), C, C2, ...)
5276 LiveGcValues.insert(BasePtr);
5277 LiveGcValues.insert(DerivedPtr);
5278 }
5279 std::optional<OperandBundleUse> Bundle =
5281 unsigned NumOfGCLives = LiveGcValues.size();
5282 if (!Bundle || NumOfGCLives == Bundle->Inputs.size())
5283 break;
5284 // We can reduce the size of gc live bundle.
5285 DenseMap<Value *, unsigned> Val2Idx;
5286 std::vector<Value *> NewLiveGc;
5287 for (Value *V : Bundle->Inputs) {
5288 auto [It, Inserted] = Val2Idx.try_emplace(V);
5289 if (!Inserted)
5290 continue;
5291 if (LiveGcValues.count(V)) {
5292 It->second = NewLiveGc.size();
5293 NewLiveGc.push_back(V);
5294 } else
5295 It->second = NumOfGCLives;
5296 }
5297 // Update all gc.relocates
5298 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5299 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5300 Value *BasePtr = GCR.getBasePtr();
5301 assert(Val2Idx.count(BasePtr) && Val2Idx[BasePtr] != NumOfGCLives &&
5302 "Missed live gc for base pointer");
5303 auto *OpIntTy1 = GCR.getOperand(1)->getType();
5304 GCR.setOperand(1, ConstantInt::get(OpIntTy1, Val2Idx[BasePtr]));
5305 Value *DerivedPtr = GCR.getDerivedPtr();
5306 assert(Val2Idx.count(DerivedPtr) && Val2Idx[DerivedPtr] != NumOfGCLives &&
5307 "Missed live gc for derived pointer");
5308 auto *OpIntTy2 = GCR.getOperand(2)->getType();
5309 GCR.setOperand(2, ConstantInt::get(OpIntTy2, Val2Idx[DerivedPtr]));
5310 }
5311 // Create new statepoint instruction.
5312 OperandBundleDef NewBundle("gc-live", std::move(NewLiveGc));
5313 return CallBase::Create(&Call, NewBundle);
5314 }
5315 default: { break; }
5316 }
5317
5318 return Changed ? &Call : nullptr;
5319}
5320
5321/// If the callee is a constexpr cast of a function, attempt to move the cast to
5322/// the arguments of the call/invoke.
5323/// CallBrInst is not supported.
5324bool InstCombinerImpl::transformConstExprCastCall(CallBase &Call) {
5325 auto *Callee =
5327 if (!Callee)
5328 return false;
5329
5331 "CallBr's don't have a single point after a def to insert at");
5332
5333 // Don't perform the transform for declarations, which may not be fully
5334 // accurate. For example, void @foo() is commonly used as a placeholder for
5335 // unknown prototypes.
5336 if (Callee->isDeclaration())
5337 return false;
5338
5339 // If this is a call to a thunk function, don't remove the cast. Thunks are
5340 // used to transparently forward all incoming parameters and outgoing return
5341 // values, so it's important to leave the cast in place.
5342 if (Callee->hasFnAttribute("thunk"))
5343 return false;
5344
5345 // If this is a call to a naked function, the assembly might be
5346 // using an argument, or otherwise rely on the frame layout,
5347 // the function prototype will mismatch.
5348 if (Callee->hasFnAttribute(Attribute::Naked))
5349 return false;
5350
5351 // If this is a musttail call, the callee's prototype must match the caller's
5352 // prototype with the exception of pointee types. The code below doesn't
5353 // implement that, so we can't do this transform.
5354 // TODO: Do the transform if it only requires adding pointer casts.
5355 if (Call.isMustTailCall())
5356 return false;
5357
5359 const AttributeList &CallerPAL = Call.getAttributes();
5360
5361 // Okay, this is a cast from a function to a different type. Unless doing so
5362 // would cause a type conversion of one of our arguments, change this call to
5363 // be a direct call with arguments casted to the appropriate types.
5364 FunctionType *FT = Callee->getFunctionType();
5365 Type *OldRetTy = Caller->getType();
5366 Type *NewRetTy = FT->getReturnType();
5367
5368 // Check to see if we are changing the return type...
5369 if (OldRetTy != NewRetTy) {
5370
5371 if (NewRetTy->isStructTy())
5372 return false; // TODO: Handle multiple return values.
5373
5374 if (!CastInst::isBitOrNoopPointerCastable(NewRetTy, OldRetTy, DL)) {
5375 if (!Caller->use_empty())
5376 return false; // Cannot transform this return value.
5377 }
5378
5379 if (!CallerPAL.isEmpty() && !Caller->use_empty()) {
5380 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5381 if (RAttrs.overlaps(AttributeFuncs::typeIncompatible(
5382 NewRetTy, CallerPAL.getRetAttrs())))
5383 return false; // Attribute not compatible with transformed value.
5384 }
5385
5386 // If the callbase is an invoke instruction, and the return value is
5387 // used by a PHI node in a successor, we cannot change the return type of
5388 // the call because there is no place to put the cast instruction (without
5389 // breaking the critical edge). Bail out in this case.
5390 if (!Caller->use_empty()) {
5391 BasicBlock *PhisNotSupportedBlock = nullptr;
5392 if (auto *II = dyn_cast<InvokeInst>(Caller))
5393 PhisNotSupportedBlock = II->getNormalDest();
5394 if (PhisNotSupportedBlock)
5395 for (User *U : Caller->users())
5396 if (PHINode *PN = dyn_cast<PHINode>(U))
5397 if (PN->getParent() == PhisNotSupportedBlock)
5398 return false;
5399 }
5400 }
5401
5402 unsigned NumActualArgs = Call.arg_size();
5403 unsigned NumCommonArgs = std::min(FT->getNumParams(), NumActualArgs);
5404
5405 // Prevent us turning:
5406 // declare void @takes_i32_inalloca(i32* inalloca)
5407 // call void bitcast (void (i32*)* @takes_i32_inalloca to void (i32)*)(i32 0)
5408 //
5409 // into:
5410 // call void @takes_i32_inalloca(i32* null)
5411 //
5412 // Similarly, avoid folding away bitcasts of byval calls.
5413 if (Callee->getAttributes().hasAttrSomewhere(Attribute::InAlloca) ||
5414 Callee->getAttributes().hasAttrSomewhere(Attribute::Preallocated))
5415 return false;
5416
5417 auto AI = Call.arg_begin();
5418 for (unsigned i = 0, e = NumCommonArgs; i != e; ++i, ++AI) {
5419 Type *ParamTy = FT->getParamType(i);
5420 Type *ActTy = (*AI)->getType();
5421
5422 if (!CastInst::isBitOrNoopPointerCastable(ActTy, ParamTy, DL))
5423 return false; // Cannot transform this parameter value.
5424
5425 // Check if there are any incompatible attributes we cannot drop safely.
5426 if (AttrBuilder(FT->getContext(), CallerPAL.getParamAttrs(i))
5427 .overlaps(AttributeFuncs::typeIncompatible(
5428 ParamTy, CallerPAL.getParamAttrs(i),
5429 AttributeFuncs::ASK_UNSAFE_TO_DROP)))
5430 return false; // Attribute not compatible with transformed value.
5431
5432 if (Call.isInAllocaArgument(i) ||
5433 CallerPAL.hasParamAttr(i, Attribute::Preallocated))
5434 return false; // Cannot transform to and from inalloca/preallocated.
5435
5436 if (CallerPAL.hasParamAttr(i, Attribute::SwiftError))
5437 return false;
5438
5439 if (CallerPAL.hasParamAttr(i, Attribute::ByVal) !=
5440 Callee->getAttributes().hasParamAttr(i, Attribute::ByVal))
5441 return false; // Cannot transform to or from byval.
5442 }
5443
5444 if (FT->getNumParams() < NumActualArgs && FT->isVarArg() &&
5445 !CallerPAL.isEmpty()) {
5446 // In this case we have more arguments than the new function type, but we
5447 // won't be dropping them. Check that these extra arguments have attributes
5448 // that are compatible with being a vararg call argument.
5449 unsigned SRetIdx;
5450 if (CallerPAL.hasAttrSomewhere(Attribute::StructRet, &SRetIdx) &&
5451 SRetIdx - AttributeList::FirstArgIndex >= FT->getNumParams())
5452 return false;
5453 }
5454
5455 // Okay, we decided that this is a safe thing to do: go ahead and start
5456 // inserting cast instructions as necessary.
5457 SmallVector<Value *, 8> Args;
5459 Args.reserve(NumActualArgs);
5460 ArgAttrs.reserve(NumActualArgs);
5461
5462 // Get any return attributes.
5463 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5464
5465 // If the return value is not being used, the type may not be compatible
5466 // with the existing attributes. Wipe out any problematic attributes.
5467 RAttrs.remove(
5468 AttributeFuncs::typeIncompatible(NewRetTy, CallerPAL.getRetAttrs()));
5469
5470 LLVMContext &Ctx = Call.getContext();
5471 AI = Call.arg_begin();
5472 for (unsigned i = 0; i != NumCommonArgs; ++i, ++AI) {
5473 Type *ParamTy = FT->getParamType(i);
5474
5475 Value *NewArg = *AI;
5476 if ((*AI)->getType() != ParamTy)
5477 NewArg = Builder.CreateBitOrPointerCast(*AI, ParamTy);
5478 Args.push_back(NewArg);
5479
5480 // Add any parameter attributes except the ones incompatible with the new
5481 // type. Note that we made sure all incompatible ones are safe to drop.
5482 AttributeMask IncompatibleAttrs = AttributeFuncs::typeIncompatible(
5483 ParamTy, CallerPAL.getParamAttrs(i), AttributeFuncs::ASK_SAFE_TO_DROP);
5484 ArgAttrs.push_back(
5485 CallerPAL.getParamAttrs(i).removeAttributes(Ctx, IncompatibleAttrs));
5486 }
5487
5488 // If the function takes more arguments than the call was taking, add them
5489 // now.
5490 for (unsigned i = NumCommonArgs; i != FT->getNumParams(); ++i) {
5491 Args.push_back(Constant::getNullValue(FT->getParamType(i)));
5492 ArgAttrs.push_back(AttributeSet());
5493 }
5494
5495 // If we are removing arguments to the function, emit an obnoxious warning.
5496 if (FT->getNumParams() < NumActualArgs) {
5497 // TODO: if (!FT->isVarArg()) this call may be unreachable. PR14722
5498 if (FT->isVarArg()) {
5499 // Add all of the arguments in their promoted form to the arg list.
5500 for (unsigned i = FT->getNumParams(); i != NumActualArgs; ++i, ++AI) {
5501 Type *PTy = getPromotedType((*AI)->getType());
5502 Value *NewArg = *AI;
5503 if (PTy != (*AI)->getType()) {
5504 // Must promote to pass through va_arg area!
5505 Instruction::CastOps opcode =
5506 CastInst::getCastOpcode(*AI, false, PTy, false);
5507 NewArg = Builder.CreateCast(opcode, *AI, PTy);
5508 }
5509 Args.push_back(NewArg);
5510
5511 // Add any parameter attributes.
5512 ArgAttrs.push_back(CallerPAL.getParamAttrs(i));
5513 }
5514 }
5515 }
5516
5517 AttributeSet FnAttrs = CallerPAL.getFnAttrs();
5518
5519 if (NewRetTy->isVoidTy())
5520 Caller->setName(""); // Void type should not have a name.
5521
5522 assert((ArgAttrs.size() == FT->getNumParams() || FT->isVarArg()) &&
5523 "missing argument attributes");
5524 AttributeList NewCallerPAL = AttributeList::get(
5525 Ctx, FnAttrs, AttributeSet::get(Ctx, RAttrs), ArgAttrs);
5526
5528 Call.getOperandBundlesAsDefs(OpBundles);
5529
5530 CallBase *NewCall;
5531 if (InvokeInst *II = dyn_cast<InvokeInst>(Caller)) {
5532 NewCall = Builder.CreateInvoke(Callee, II->getNormalDest(),
5533 II->getUnwindDest(), Args, OpBundles);
5534 } else {
5535 NewCall = Builder.CreateCall(Callee, Args, OpBundles);
5536 cast<CallInst>(NewCall)->setTailCallKind(
5537 cast<CallInst>(Caller)->getTailCallKind());
5538 }
5539 NewCall->takeName(Caller);
5541 NewCall->setAttributes(NewCallerPAL);
5542
5543 // Preserve prof metadata if any.
5544 NewCall->copyMetadata(*Caller, {LLVMContext::MD_prof});
5545
5546 // Insert a cast of the return type as necessary.
5547 Instruction *NC = NewCall;
5548 Value *NV = NC;
5549 if (OldRetTy != NV->getType() && !Caller->use_empty()) {
5550 assert(!NV->getType()->isVoidTy());
5552 NC->setDebugLoc(Caller->getDebugLoc());
5553
5554 auto OptInsertPt = NewCall->getInsertionPointAfterDef();
5555 assert(OptInsertPt && "No place to insert cast");
5556 InsertNewInstBefore(NC, *OptInsertPt);
5557 Worklist.pushUsersToWorkList(*Caller);
5558 }
5559
5560 if (!Caller->use_empty())
5561 replaceInstUsesWith(*Caller, NV);
5562 else if (Caller->hasValueHandle()) {
5563 if (OldRetTy == NV->getType())
5565 else
5566 // We cannot call ValueIsRAUWd with a different type, and the
5567 // actual tracked value will disappear.
5569 }
5570
5571 eraseInstFromFunction(*Caller);
5572 return true;
5573}
5574
5575/// Turn a call to a function created by init_trampoline / adjust_trampoline
5576/// intrinsic pair into a direct call to the underlying function.
5578InstCombinerImpl::transformCallThroughTrampoline(CallBase &Call,
5579 IntrinsicInst &Tramp) {
5580 FunctionType *FTy = Call.getFunctionType();
5581 AttributeList Attrs = Call.getAttributes();
5582
5583 // If the call already has the 'nest' attribute somewhere then give up -
5584 // otherwise 'nest' would occur twice after splicing in the chain.
5585 if (Attrs.hasAttrSomewhere(Attribute::Nest))
5586 return nullptr;
5587
5589 FunctionType *NestFTy = NestF->getFunctionType();
5590
5591 AttributeList NestAttrs = NestF->getAttributes();
5592 if (!NestAttrs.isEmpty()) {
5593 unsigned NestArgNo = 0;
5594 Type *NestTy = nullptr;
5595 AttributeSet NestAttr;
5596
5597 // Look for a parameter marked with the 'nest' attribute.
5598 for (FunctionType::param_iterator I = NestFTy->param_begin(),
5599 E = NestFTy->param_end();
5600 I != E; ++NestArgNo, ++I) {
5601 AttributeSet AS = NestAttrs.getParamAttrs(NestArgNo);
5602 if (AS.hasAttribute(Attribute::Nest)) {
5603 // Record the parameter type and any other attributes.
5604 NestTy = *I;
5605 NestAttr = AS;
5606 break;
5607 }
5608 }
5609
5610 if (NestTy) {
5611 std::vector<Value*> NewArgs;
5612 std::vector<AttributeSet> NewArgAttrs;
5613 NewArgs.reserve(Call.arg_size() + 1);
5614 NewArgAttrs.reserve(Call.arg_size());
5615
5616 // Insert the nest argument into the call argument list, which may
5617 // mean appending it. Likewise for attributes.
5618
5619 {
5620 unsigned ArgNo = 0;
5621 auto I = Call.arg_begin(), E = Call.arg_end();
5622 do {
5623 if (ArgNo == NestArgNo) {
5624 // Add the chain argument and attributes.
5625 Value *NestVal = Tramp.getArgOperand(2);
5626 if (NestVal->getType() != NestTy)
5627 NestVal = Builder.CreateBitCast(NestVal, NestTy, "nest");
5628 NewArgs.push_back(NestVal);
5629 NewArgAttrs.push_back(NestAttr);
5630 }
5631
5632 if (I == E)
5633 break;
5634
5635 // Add the original argument and attributes.
5636 NewArgs.push_back(*I);
5637 NewArgAttrs.push_back(Attrs.getParamAttrs(ArgNo));
5638
5639 ++ArgNo;
5640 ++I;
5641 } while (true);
5642 }
5643
5644 // The trampoline may have been bitcast to a bogus type (FTy).
5645 // Handle this by synthesizing a new function type, equal to FTy
5646 // with the chain parameter inserted.
5647
5648 std::vector<Type*> NewTypes;
5649 NewTypes.reserve(FTy->getNumParams()+1);
5650
5651 // Insert the chain's type into the list of parameter types, which may
5652 // mean appending it.
5653 {
5654 unsigned ArgNo = 0;
5655 FunctionType::param_iterator I = FTy->param_begin(),
5656 E = FTy->param_end();
5657
5658 do {
5659 if (ArgNo == NestArgNo)
5660 // Add the chain's type.
5661 NewTypes.push_back(NestTy);
5662
5663 if (I == E)
5664 break;
5665
5666 // Add the original type.
5667 NewTypes.push_back(*I);
5668
5669 ++ArgNo;
5670 ++I;
5671 } while (true);
5672 }
5673
5674 // Replace the trampoline call with a direct call. Let the generic
5675 // code sort out any function type mismatches.
5676 FunctionType *NewFTy =
5677 FunctionType::get(FTy->getReturnType(), NewTypes, FTy->isVarArg());
5678 AttributeList NewPAL =
5679 AttributeList::get(FTy->getContext(), Attrs.getFnAttrs(),
5680 Attrs.getRetAttrs(), NewArgAttrs);
5681
5683 Call.getOperandBundlesAsDefs(OpBundles);
5684
5685 Instruction *NewCaller;
5686 if (InvokeInst *II = dyn_cast<InvokeInst>(&Call)) {
5687 NewCaller = InvokeInst::Create(NewFTy, NestF, II->getNormalDest(),
5688 II->getUnwindDest(), NewArgs, OpBundles);
5689 cast<InvokeInst>(NewCaller)->setCallingConv(II->getCallingConv());
5690 cast<InvokeInst>(NewCaller)->setAttributes(NewPAL);
5691 } else if (CallBrInst *CBI = dyn_cast<CallBrInst>(&Call)) {
5692 NewCaller =
5693 CallBrInst::Create(NewFTy, NestF, CBI->getDefaultDest(),
5694 CBI->getIndirectDests(), NewArgs, OpBundles);
5695 cast<CallBrInst>(NewCaller)->setCallingConv(CBI->getCallingConv());
5696 cast<CallBrInst>(NewCaller)->setAttributes(NewPAL);
5697 } else {
5698 NewCaller = CallInst::Create(NewFTy, NestF, NewArgs, OpBundles);
5699 cast<CallInst>(NewCaller)->setTailCallKind(
5700 cast<CallInst>(Call).getTailCallKind());
5701 cast<CallInst>(NewCaller)->setCallingConv(
5702 cast<CallInst>(Call).getCallingConv());
5703 cast<CallInst>(NewCaller)->setAttributes(NewPAL);
5704 }
5705 NewCaller->setDebugLoc(Call.getDebugLoc());
5706
5707 return NewCaller;
5708 }
5709 }
5710
5711 // Replace the trampoline call with a direct call. Since there is no 'nest'
5712 // parameter, there is no need to adjust the argument list. Let the generic
5713 // code sort out any function type mismatches.
5714 Call.setCalledFunction(FTy, NestF);
5715 return &Call;
5716}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
@ Scaled
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
BitTracker BT
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static SDValue foldBitOrderCrossLogicOp(SDNode *N, SelectionDAG &DAG)
#define Check(C,...)
#define DEBUG_TYPE
Hexagon Common GEP
#define _
IRTranslator LLVM IR MI
static Type * getPromotedType(Type *Ty)
Return the specified type promoted as it would be to pass though a va_arg area.
static Instruction * createOverflowTuple(IntrinsicInst *II, Value *Result, Constant *Overflow)
Creates a result tuple for an overflow intrinsic II with a given Result and a constant Overflow value...
static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M, IRBuilderBase &B)
static IntrinsicInst * findInitTrampolineFromAlloca(Value *TrampMem)
static bool removeTriviallyEmptyRange(IntrinsicInst &EndI, InstCombinerImpl &IC, std::function< bool(const IntrinsicInst &)> IsStart)
static bool inputDenormalIsDAZ(const Function &F, const Type *Ty)
static Instruction * reassociateMinMaxWithConstantInOperand(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If this min/max has a matching min/max operand with a constant, try to push the constant operand into...
static bool isIdempotentBinaryIntrinsic(Intrinsic::ID IID)
Helper to match idempotent binary intrinsics, namely, intrinsics where f(f(x, y), y) == f(x,...
static bool signBitMustBeTheSame(Value *Op0, Value *Op1, const SimplifyQuery &SQ)
Return true if two values Op0 and Op1 are known to have the same sign.
static Value * optimizeModularFormat(CallInst *CI, IRBuilderBase &B)
static Instruction * moveAddAfterMinMax(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0.
static Instruction * simplifyInvariantGroupIntrinsic(IntrinsicInst &II, InstCombinerImpl &IC)
This function transforms launder.invariant.group and strip.invariant.group like: launder(launder(x)) ...
static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E, unsigned NumOperands)
static std::optional< bool > getKnownSign(Value *Op, const SimplifyQuery &SQ)
static cl::opt< unsigned > GuardWideningWindow("instcombine-guard-widening-window", cl::init(3), cl::desc("How wide an instruction window to bypass looking for " "another guard"))
static bool hasUndefSource(AnyMemTransferInst *MI)
Recognize a memcpy/memmove from a trivially otherwise unused alloca.
static Instruction * factorizeMinMaxTree(IntrinsicInst *II)
Reduce a sequence of min/max intrinsics with a common operand.
static Instruction * foldClampRangeOfTwo(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If we have a clamp pattern like max (min X, 42), 41 – where the output can only be one of two possibl...
static Value * simplifyReductionOperand(Value *Arg, bool CanReorderLanes)
static IntrinsicInst * findInitTrampolineFromBB(IntrinsicInst *AdjustTramp, Value *TrampMem)
static bool isAspectNeeded(StringRef Aspect, CallInst *CI, std::optional< unsigned > FirstArgIdx, const std::optional< Bitset< 256 > > &Specifiers)
static Value * foldIntrinsicUsingDistributiveLaws(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
static std::optional< bool > getKnownSignOrZero(Value *Op, const SimplifyQuery &SQ)
static Value * foldMinimumOverTrailingOrLeadingZeroCount(Value *I0, Value *I1, const DataLayout &DL, InstCombiner::BuilderTy &Builder)
Fold an unsigned minimum of trailing or leading zero bits counts: umin(cttz(CtOp1,...
static bool rightDistributesOverLeft(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "(X ROp Y) LOp Z" is always equal to "(X LOp Z) ROp (Y LOp Z)".
static Value * foldIdempotentBinaryIntrinsicRecurrence(InstCombinerImpl &IC, IntrinsicInst *II)
Attempt to simplify value-accumulating recurrences of kind: umax.acc = phi i8 [ umax,...
static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy)
static Instruction * foldCtpop(IntrinsicInst &II, InstCombinerImpl &IC)
static Instruction * simplifyNeonTbl(IntrinsicInst &II, InstCombiner &IC, bool IsExtension)
Convert tbl/tbx intrinsics to shufflevector if the mask is constant, and at most two source operands ...
static Instruction * foldCttzCtlz(IntrinsicInst &II, InstCombinerImpl &IC)
static IntrinsicInst * findInitTrampoline(Value *Callee)
static Value * foldCmpIntrinsicOfExtended(IntrinsicInst *II, InstCombiner::BuilderTy &Builder, const DataLayout &DL)
Fold an scmp/ucmp intrinsic whose operands are extended from a narrower type: scmp (sext X),...
static Bitset< 256 > parseFormatStringSpecifiers(StringRef FormatStr)
static FCmpInst::Predicate fpclassTestIsFCmp0(FPClassTest Mask, const Function &F, Type *Ty)
static bool leftDistributesOverRight(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "X LOp (Y ROp Z)" is always equal to "(X LOp Y) ROp (X LOp Z)".
static Value * reassociateMinMaxWithConstants(IntrinsicInst *II, IRBuilderBase &Builder, const SimplifyQuery &SQ)
If this min/max has a constant operand and an operand that is a matching min/max with a constant oper...
static Value * foldSinAndCosToSinCos(IntrinsicInst *II, IRBuilderBase &B, InstCombinerImpl &IC)
static CallInst * canonicalizeConstantArg0ToArg1(CallInst &Call)
static Instruction * foldNeonShift(IntrinsicInst *II, InstCombinerImpl &IC)
This file provides internal interfaces used to implement the InstCombine.
This file provides the interface for the instcombine pass implementation.
static bool inputDenormalIsIEEE(DenormalMode Mode)
Return true if it's possible to assume IEEE treatment of input denormals in F for Val.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
This file implements the SmallBitVector class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
static LLVM_ABI bool semanticsHasInf(const fltSemantics &)
Definition APFloat.cpp:287
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:353
static LLVM_ABI bool hasSignBitInMSB(const fltSemantics &)
Definition APFloat.cpp:300
bool isNegative() const
Definition APFloat.h:1575
void clearSign()
Definition APFloat.h:1394
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1184
bool isZero() const
Definition APFloat.h:1571
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
Definition APFloat.h:1234
static APFloat getSmallest(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) finite number in the given semantics.
Definition APFloat.h:1244
bool isInfinity() const
Definition APFloat.h:1572
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:231
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:226
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1206
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1984
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1187
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:377
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1693
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1509
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1116
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1964
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1971
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:647
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:216
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1085
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:507
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2072
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:331
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:303
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:197
std::optional< int64_t > trySExtValue() const
Get sign extended value if possible.
Definition APInt.h:1595
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1977
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
This class represents any memset intrinsic.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:407
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
static LLVM_ABI AttributeSet get(LLVMContext &C, const AttrBuilder &B)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
static LLVM_ABI Attribute getWithDereferenceableOrNullBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:446
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI bool isSigned() const
Whether the intrinsic is signed or unsigned.
LLVM_ABI Instruction::BinaryOps getBinaryOp() const
Returns the binary operation underlying the intrinsic.
static BinaryOperator * CreateFAddFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:271
static LLVM_ABI BinaryOperator * CreateNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Helper functions to construct and inspect unary operations (NEG and NOT) via binary operators SUB and...
static BinaryOperator * CreateNSW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:314
static LLVM_ABI BinaryOperator * CreateNot(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static BinaryOperator * CreateNUW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:329
static BinaryOperator * CreateFMulFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:279
static BinaryOperator * CreateFDivFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:283
static BinaryOperator * CreateFSubFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:275
static LLVM_ABI BinaryOperator * CreateNSWNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
This is a constexpr reimplementation of a subset of std::bitset.
Definition Bitset.h:30
constexpr bool any() const
Definition Bitset.h:113
constexpr Bitset & set()
Definition Bitset.h:81
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
void setDoesNotThrow()
MaybeAlign getRetAlign() const
Extract the alignment of the return value.
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
OperandBundleUse getOperandBundleAt(unsigned Index) const
Return the operand bundle at a specific index.
std::optional< OperandBundleUse > getOperandBundle(StringRef Name) const
Return an operand bundle by name, if present.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
bool isInAllocaArgument(unsigned ArgNo) const
Determine whether this argument is passed in an alloca.
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
bool hasRetAttr(Attribute::AttrKind Kind) const
Determine whether the return value has the given attribute.
unsigned getNumOperandBundles() const
Return the number of operand bundles associated with this User.
uint64_t getParamDereferenceableBytes(unsigned i) const
Extract the number of dereferenceable bytes for a call or parameter (0=unknown).
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
LLVM_ABI bool isIndirectCall() const
Return true if the callsite is an indirect call.
static LLVM_ABI CallBase * removeOperandBundleAt(CallBase *CB, size_t Offset, InsertPosition InsertPtr=nullptr)
void setNotConvergent()
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Attribute getFnAttr(StringRef Kind) const
Get the attribute of a given kind for the function.
bool doesNotThrow() const
Determine if the call cannot unwind.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
bool isConvergent() const
Determine if the invoke is convergent.
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
Value * getReturnedArgOperand() const
If one of the arguments has the 'returned' attribute, returns its operand value.
static LLVM_ABI CallBase * Create(CallBase *CB, ArrayRef< OperandBundleDef > Bundles, InsertPosition InsertPt=nullptr)
Create a clone of CB with a different set of operand bundles and insert it before InsertPt.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
static LLVM_ABI CallBase * removeOperandBundle(CallBase *CB, uint32_t ID, InsertPosition InsertPt=nullptr)
Create a clone of CB with operand bundle ID removed.
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
LLVM_ABI Function * getCaller()
Helper to get the caller (the parent function).
CallBr instruction, tracking function calls that may not return control but instead transfer it to a ...
static CallBrInst * Create(FunctionType *Ty, Value *Func, BasicBlock *DefaultDest, ArrayRef< BasicBlock * > IndirectDests, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This class represents a function call, abstracting a target machine's calling convention.
bool isNoTailCall() const
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
bool isMustTailCall() const
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI CastInst * CreateIntegerCast(Value *S, Type *Ty, bool isSigned, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a ZExt, BitCast, or Trunc for int -> int casts.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static LLVM_ABI CastInst * CreateBitOrPointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, a PtrToInt, or an IntToPTr cast instruction.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getNonStrictPredicate() const
For example, SGT -> SGE, SLT -> SLE, ULT -> ULE, UGT -> UGE.
Definition InstrTypes.h:934
Predicate getUnorderedPredicate() const
Definition InstrTypes.h:874
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getLimitedValue(uint64_t Limit=~0ULL) const
getLimitedValue - If the value is smaller than the specified limit, return it, otherwise return the l...
Definition Constants.h:269
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI ConstantPtrAuth * get(Constant *Ptr, ConstantInt *Key, ConstantInt *Disc, Constant *AddrDisc, Constant *DeactivationSymbol)
Return a pointer signed with the specified parameters.
This class represents a range of values.
LLVM_ABI ConstantRange zextOrTrunc(uint32_t BitWidth) const
Make this range have the bit width given by BitWidth.
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
uint32_t getBitWidth() const
Get the bit width of this ConstantRange.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Record of a variable value-assignment, aka a non instruction representation of the dbg....
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:299
unsigned size() const
Definition DenseMap.h:172
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
Definition DenseMap.h:219
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Definition DenseMap.h:214
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static FMFSource intersect(Value *A, Value *B)
Intersect the FMF from two instructions.
Definition IRBuilder.h:107
This class represents an extension of floating point types.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
An instruction for ordering other memory operations.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this fence instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this fence instruction.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Type::subtype_iterator param_iterator
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
bool isConvergent() const
Determine if the call is convergent.
Definition Function.h:592
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:211
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:272
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:328
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:576
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
Definition Function.h:251
LLVM_ABI Value * getBasePtr() const
unsigned getBasePtrIndex() const
The index into the associate statepoint's argument list which contains the base pointer of the pointe...
LLVM_ABI Value * getDerivedPtr() const
unsigned getDerivedPtrIndex() const
The index into the associate statepoint's argument list which contains the pointer whose relocation t...
std::vector< const GCRelocateInst * > getGCRelocates() const
Get list of all gc reloactes linked to this statepoint May contain several relocations for the same b...
Definition Statepoint.h:206
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this GlobalObject.
LLVM_ABI bool isDeclaration() const
Return true if the primary definition of this global value is outside of the current translation unit...
Definition Globals.cpp:408
PointerType * getType() const
Global values are always pointers.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
LLVM_ABI Value * CreateLaunderInvariantGroup(Value *Ptr)
Create a launder.invariant.group intrinsic call.
ConstantInt * getTrue()
Get the constant value for i1 true.
Definition IRBuilder.h:457
LLVM_ABI Value * CreateBinaryIntrinsic(Intrinsic::ID ID, Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 2 operands which is mangled on the first type.
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1449
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2131
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Definition IRBuilder.h:2694
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:462
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2495
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2258
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
LLVM_ABI Value * CreateStripInvariantGroup(Value *Ptr)
Create a strip.invariant.group intrinsic call.
static InsertValueInst * Create(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Instruction * foldOpIntoPhi(Instruction &I, PHINode *PN, bool AllowMultipleUses=false)
Given a binary operator, cast instruction, or select which has a PHI node as operand #0,...
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
bool SimplifyDemandedBits(Instruction *I, unsigned Op, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0) override
This form of SimplifyDemandedBits simplifies the specified instruction operand if possible,...
Instruction * FoldOpIntoSelect(Instruction &Op, SelectInst *SI, bool FoldWithMultiUse=false, bool SimplifyBothArms=false)
Given an instruction with a select as one operand and a constant as the other operand,...
Instruction * SimplifyAnyMemSet(AnyMemSetInst *MI)
Instruction * foldItoFPtoI(FPToIntTy &FI)
fpto{s/u}i.sat --> X or zext(X) or sext(X) or trunc(X) This is safe if the intermediate type has enou...
Instruction * visitFree(CallInst &FI, Value *FreedOp)
Instruction * visitCallBrInst(CallBrInst &CBI)
Instruction * eraseInstFromFunction(Instruction &I) override
Combiner aware instruction erasure.
Value * foldReversedIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are reverses, try to pull the reverse after the intrinsic.
Value * tryGetLog2(Value *Op, bool AssumeNonZero)
Instruction * visitFenceInst(FenceInst &FI)
Instruction * foldShuffledIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are unary shuffles with the same mask, try to shuffle after the int...
Instruction * visitInvokeInst(InvokeInst &II)
bool SimplifyDemandedInstructionBits(Instruction &Inst)
Tries to simplify operands to an integer instruction based on its demanded bits.
void CreateNonTerminatorUnreachable(Instruction *InsertAt)
Create and insert the idiom we use to indicate a block is unreachable without having to rewrite the C...
Instruction * visitVAEndInst(VAEndInst &I)
Instruction * matchBSwapOrBitReverse(Instruction &I, bool MatchBSwaps, bool MatchBitReversals)
Given an initial instruction, check to see if it is the root of a bswap/bitreverse idiom.
Constant * unshuffleConstant(ArrayRef< int > ShMask, Constant *C, VectorType *NewCTy)
Find a constant NewC that has property: shuffle(NewC, poison, ShMask) = C for lanes that select NewC.
Instruction * visitAllocSite(Instruction &FI)
Instruction * SimplifyAnyMemTransfer(AnyMemTransferInst *MI)
OverflowResult computeOverflow(Instruction::BinaryOps BinaryOp, bool IsSigned, Value *LHS, Value *RHS, Instruction *CxtI) const
Instruction * visitCallInst(CallInst &CI)
CallInst simplification.
The core instruction combiner logic.
SimplifyQuery SQ
const DataLayout & getDataLayout() const
unsigned ComputeMaxSignificantBits(const Value *Op, const Instruction *CxtI=nullptr, unsigned Depth=0) const
bool isFreeToInvert(Value *V, bool WillInvertAllUses, bool &DoesConsume)
Return true if the specified value is free to invert (apply ~ to).
DominatorTree & getDominatorTree() const
BlockFrequencyInfo * BFI
TargetLibraryInfo & TLI
Instruction * InsertNewInstBefore(Instruction *New, BasicBlock::iterator Old)
Inserts an instruction New before instruction Old.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
const DataLayout & DL
DomConditionCache DC
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CxtI, unsigned Depth=0) const
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
LLVM_ABI std::optional< Instruction * > targetInstCombineIntrinsic(IntrinsicInst &II)
AssumptionCache & AC
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
bool MaskedValueIsZero(const Value *V, const APInt &Mask, const Instruction *CxtI=nullptr, unsigned Depth=0) const
DominatorTree & DT
ProfileSummaryInfo * PSI
OptimizationRemarkEmitter & ORE
Value * getFreelyInverted(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume)
const SimplifyQuery & getSimplifyQuery() const
bool isKnownToBeAPowerOfTwo(const Value *V, bool OrZero=false, const Instruction *CxtI=nullptr, unsigned Depth=0)
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI bool mayWriteToMemory() const LLVM_READONLY
Return true if this instruction may modify memory.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
bool isTerminator() const
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI std::optional< InstListType::iterator > getInsertionPointAfterDef()
Get the first insertion point at which the result of this instruction is defined.
LLVM_ABI bool isIdenticalTo(const Instruction *I) const LLVM_READONLY
Return true if the specified instruction is exactly identical to the current one.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
Invoke instruction.
static InvokeInst * Create(FunctionType *Ty, Value *Func, BasicBlock *IfNormal, BasicBlock *IfException, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
Metadata node.
Definition Metadata.h:1069
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1567
static LLVM_ABI MDNode * getMostGenericFPMath(MDNode *A, MDNode *B)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:615
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:111
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
ICmpInst::Predicate getPredicate() const
Returns the comparison predicate underlying the intrinsic.
bool isSigned() const
Whether the intrinsic is signed or unsigned.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
StringRef getName() const
Get a short "name" for the module.
Definition Module.h:311
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
Utility class for integer operators which may exhibit overflow - Add, Sub, Mul, and Shl.
Definition Operator.h:78
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
bool isCommutative() const
Return true if the instruction is commutative.
Definition Operator.h:130
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Represents a saturating add/sub intrinsic.
This class represents the LLVM 'select' instruction.
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
This instruction constructs a fixed permutation of two input vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
size_type size() const
Definition SmallPtrSet.h:99
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setVolatile(bool V)
Specify whether this is a volatile store or not.
void setAlignment(Align Align)
void setOrdering(AtomicOrdering Ordering)
Sets the ordering constraint of this store instruction.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static constexpr size_t npos
Definition StringRef.h:58
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
LLVM_ABI size_t find_first_not_of(char C, size_t From=0) const
Find the first character in the string that is not C or npos if not found.
Class to represent struct types.
static LLVM_ABI bool isCallingConvCCompatible(CallBase *CI)
Returns true if call site / callee has cdecl-compatible calling conventions.
Provides information about what library functions are available for the current target.
This class represents a truncation of integer types.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:263
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI bool canLosslesslyBitCastTo(Type *Ty) const
Return true if this type could be converted with a lossless BitCast to type 'Ty'.
Definition Type.cpp:153
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:276
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:106
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
static UnaryOperator * CreateWithCopiedFlags(UnaryOps Opc, Value *V, Instruction *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:148
static UnaryOperator * CreateFNegFMF(Value *Op, Instruction *FMFSource, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:156
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:36
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
This represents the llvm.va_end intrinsic.
static LLVM_ABI void ValueIsDeleted(Value *V)
Definition Value.cpp:1272
static LLVM_ABI void ValueIsRAUWd(Value *Old, Value *New)
Definition Value.cpp:1325
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
static constexpr uint64_t MaximumAlignment
Definition Value.h:799
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:439
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
iterator_range< user_iterator > users()
Definition Value.h:426
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:713
bool use_empty() const
Definition Value.h:346
static constexpr unsigned MaxAlignmentExponent
The maximum alignment for instructions.
Definition Value.h:798
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:81
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
auto m_PosZeroFP()
Matches a floating-point positive zero.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
auto m_PtrToIntOrAddr(const OpTy &Op)
Matches PtrToInt or PtrToAddr.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
BinaryOp_match< LHS, RHS, Instruction::Xor > m_Xor(const LHS &L, const RHS &R)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWSub(const LHS &L, const RHS &R)
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
auto m_UMin(const Opnd0 &Op0, const Opnd1 &Op1)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
OverflowingBinaryOp_match< cst_pred_ty< is_zero_int >, ValTy, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWNeg(const ValTy &V)
Matches a 'Neg' as 'sub nsw 0, V'.
auto m_SMax(const Opnd0 &Op0, const Opnd1 &Op1)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
cstfp_pred_ty< is_neg_zero_fp > m_NegZeroFP()
Match a floating-point negative zero.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_UMax(const Opnd0 &Op0, const Opnd1 &Op1)
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_CopySign(const Opnd0 &Op0, const Opnd1 &Op1)
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
BinOpPred_match< LHS, RHS, is_logical_shift_op > m_LogicalShift(const LHS &L, const RHS &R)
Matches logical shift operations.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
match_combine_or< match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > >, OpTy > m_ZExtOrSExtOrSelf(const OpTy &Op)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_strictlypositive > m_StrictlyPositive()
Match an integer or vector of strictly positive values.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
auto m_FShl(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
cst_pred_ty< is_negated_power2 > m_NegatedPower2()
Match a integer or vector negated power-of-2.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
cst_pred_ty< custom_checkfn< APInt > > m_CheckedInt(function_ref< bool(const APInt &)> CheckFn)
Match an integer or vector where CheckFn(ele) for each element is true.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_c_MaxOrMin(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWSub(const LHS &L, const RHS &R)
auto m_SMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_FAbs(const Opnd0 &Op0)
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoSignedWrap >, DisjointOr_match< LHS, RHS > > m_NSWAddLike(const LHS &L, const RHS &R)
Match either "add nsw" or "or disjoint".
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
Exact_match< T > m_Exact(const T &SubPattern)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
auto m_UnOp()
Match an arbitrary unary operation and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_MaxOrMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
BinaryOp_match< LHS, RHS, Instruction::SRem > m_SRem(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
auto m_VecReverse(const Opnd0 &Op0)
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoUnsignedWrap >, DisjointOr_match< LHS, RHS > > m_NUWAddLike(const LHS &L, const RHS &R)
Match either "add nuw" or "or disjoint".
BinOpPred_match< LHS, RHS, is_bitwiselogic_op > m_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations.
ElementWiseBitCast_match< OpTy > m_ElementWiseBitCast(const OpTy &Op)
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
auto m_FShr(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
Definition DebugInfo.h:205
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:668
constexpr double e
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Intrinsic::ID getInverseMinMaxIntrinsic(Intrinsic::ID MinMaxID)
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:578
@ NeverOverflows
Never overflows.
@ AlwaysOverflowsHigh
Always overflows in the direction of signed/unsigned max value.
@ AlwaysOverflowsLow
Always overflows in the direction of signed/unsigned min value.
@ MayOverflow
May or may not overflow.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
LLVM_ABI Value * simplifyFMulInst(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for an FMul, fold the result or return null.
LLVM_ABI bool isValidAssumeForContext(const Instruction *I, const Instruction *CxtI, const DominatorTree *DT=nullptr, bool AllowEphemerals=false)
Return true if it is valid to use the assumptions provided by an assume intrinsic,...
LLVM_ABI APInt possiblyDemandedEltsInMask(Value *Mask)
Given a mask vector of the form <Y x i1>, return an APInt (of bitwidth Y) for each lane which may be ...
BundleAttr getBundleAttrFromOBU(OperandBundleUse OBU)
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool isRemovableAlloc(const CallBase *V, const TargetLibraryInfo *TLI)
Return true if this is a call to an allocation function that does not have side effects that we are r...
LLVM_ABI bool getConstantStringInfo(const Value *V, StringRef &Str, bool TrimAtNul=true)
This function computes the length of a null-terminated C string pointed to by V.
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
Definition MathExtras.h:224
LLVM_ABI Value * lowerObjectSizeCall(IntrinsicInst *ObjectSize, const DataLayout &DL, const TargetLibraryInfo *TLI, bool MustSucceed)
Try to turn a call to @llvm.objectsize into an integer value of the given Type.
LLVM_ABI AssumeSeparateStorageInfo getAssumeSeparateStorageInfo(OperandBundleUse)
LLVM_ABI Value * getAllocAlignment(const CallBase *V, const TargetLibraryInfo *TLI)
Gets the alignment argument for an aligned_alloc-like function, using either built-in knowledge based...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1793
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
Definition MathExtras.h:352
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
Align getKnownAlignment(Value *V, const DataLayout &DL, const Instruction *CxtI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr)
Try to infer an alignment for the specified pointer.
Definition Local.h:240
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1748
SelectPatternFlavor
Specific patterns of select instructions we can match.
@ SPF_ABS
Floating point maxnum.
@ SPF_NABS
Absolute value.
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isModSet(const ModRefInfo MRI)
Definition ModRef.h:49
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1636
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1779
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1693
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI SelectPatternResult matchSelectPattern(Value *V, Value *&LHS, Value *&RHS, Instruction::CastOps *CastOp=nullptr, unsigned Depth=0)
Pattern match integer [SU]MIN, [SU]MAX and ABS idioms, returning the kind and providing the out param...
LLVM_ABI bool matchSimpleBinaryIntrinsicRecurrence(const IntrinsicInst *I, PHINode *&P, Value *&Init, Value *&OtherOp)
Attempt to match a simple value-accumulating recurrence of the form: llvm.intrinsic....
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1777
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1753
bool isAtLeastOrStrongerThan(AtomicOrdering AO, AtomicOrdering Other)
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
iterator_range< SplittingIterator > split(StringRef Str, StringRef Separator)
Split the specified string over a separator and return a range-compatible iterable over its partition...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool isNotCrossLaneOperation(const Instruction *I)
Return true if the instruction doesn't potentially cross vector lanes.
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI bool isKnownNonZero(const Value *V, const SimplifyQuery &Q, unsigned Depth=0)
Return true if the given value is known to be non-zero when defined.
constexpr int PoisonMaskElem
@ Mod
The access may modify the value stored in memory.
Definition ModRef.h:34
LLVM_ABI Value * simplifyFMAFMul(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for the multiplication of a FMA, fold the result or return null.
@ Other
Any other memory.
Definition ModRef.h:68
LLVM_ABI Value * simplifyConstrainedFPCall(CallBase *Call, const SimplifyQuery &Q)
Given a constrained FP intrinsic call, tries to compute its simplified version.
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1729
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI AssumeNonNullInfo getAssumeNonNullInfo(OperandBundleUse)
@ Add
Sum of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
bool isSafeToSpeculativelyExecuteWithVariableReplaced(const Instruction *I, bool IgnoreUBImplyingAttrs=true)
Don't use information from its non-constant operands.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI Value * getFreedOperand(const CallBase *CB, const TargetLibraryInfo *TLI)
If this if a call to a free function, return the freed operand.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
Definition MathExtras.h:233
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
LLVM_ABI std::optional< APInt > getAllocSize(const CallBase *CB, const TargetLibraryInfo *TLI, function_ref< const Value *(const Value *)> Mapper=[](const Value *V) { return V;})
Return the size of the requested allocation.
LLVM_ABI AssumeAlignInfo getAssumeAlignInfo(OperandBundleUse)
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool maskContainsAllOneOrUndef(Value *Mask)
Given a mask vector of i1, Return true if any of the elements of this predicate mask are known to be ...
LLVM_ABI std::optional< bool > isImpliedByDomCondition(const Value *Cond, const Instruction *ContextI, const DataLayout &DL)
Return the boolean condition value in the context of the given instruction if it is known based on do...
LLVM_ABI bool isDereferenceablePointer(const Value *V, Type *Ty, const SimplifyQuery &Q, bool IgnoreFree=false)
Equivalent to isDereferenceableAndAlignedPointer with an alignment of 1.
Definition Loads.cpp:264
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1766
LLVM_ABI bool isKnownNegation(const Value *X, const Value *Y, bool NeedNSW=false, bool AllowPoison=true)
Return true if the two given values are negation.
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1806
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
LLVM_ABI AssumeDereferenceableInfo getAssumeDereferenceableInfo(OperandBundleUse)
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI AssumeNoUndefInfo getAssumeNoUndefInfo(OperandBundleUse)
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI std::optional< bool > computeKnownFPSignBit(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Return false if we can prove that the specified FP value's sign bit is 0.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define NC
Definition regutils.h:42
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
Definition Metadata.h:763
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
@ IEEE
IEEE-754 denormal numbers preserved.
Matching combinators.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Align valueOrOne() const
For convenience, returns a valid alignment or 1 if undefined.
Definition Alignment.h:130
uint32_t getTagID() const
Return the tag of this operand bundle as an integer.
ArrayRef< Use > Inputs
SelectPatternFlavor Flavor
const DataLayout & DL
const Instruction * CxtI
SimplifyQuery getWithInstruction(const Instruction *I) const