LLVM 24.0.0git
ConstantFolding.cpp
Go to the documentation of this file.
1//===-- ConstantFolding.cpp - Fold instructions into constants ------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines routines for folding instructions into constants.
10//
11// Also, to supplement the basic IR ConstantExpr simplifications,
12// this file defines some additional folding routines that can make use of
13// DataLayout information. These functions cannot go in IR due to library
14// dependency issues.
15//
16//===----------------------------------------------------------------------===//
17
19#include "llvm/ADT/APFloat.h"
20#include "llvm/ADT/APInt.h"
21#include "llvm/ADT/APSInt.h"
22#include "llvm/ADT/ArrayRef.h"
23#include "llvm/ADT/DenseMap.h"
24#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/StringRef.h"
32#include "llvm/Config/config.h"
33#include "llvm/IR/Constant.h"
35#include "llvm/IR/Constants.h"
36#include "llvm/IR/DataLayout.h"
38#include "llvm/IR/Function.h"
39#include "llvm/IR/GlobalValue.h"
41#include "llvm/IR/InstrTypes.h"
42#include "llvm/IR/Instruction.h"
45#include "llvm/IR/Intrinsics.h"
46#include "llvm/IR/IntrinsicsAArch64.h"
47#include "llvm/IR/IntrinsicsAMDGPU.h"
48#include "llvm/IR/IntrinsicsARM.h"
49#include "llvm/IR/IntrinsicsNVPTX.h"
50#include "llvm/IR/IntrinsicsWebAssembly.h"
51#include "llvm/IR/IntrinsicsX86.h"
53#include "llvm/IR/Operator.h"
54#include "llvm/IR/Type.h"
55#include "llvm/IR/Value.h"
59#include <cassert>
60#include <cerrno>
61#include <cfenv>
62#include <cmath>
63#include <cstdint>
64
65using namespace llvm;
66
68 "disable-fp-call-folding",
69 cl::desc("Disable constant-folding of FP intrinsics and libcalls."),
70 cl::init(false), cl::Hidden);
71
72namespace {
73
74//===----------------------------------------------------------------------===//
75// Constant Folding internal helper functions
76//===----------------------------------------------------------------------===//
77
78static Constant *foldConstVectorToAPInt(APInt &Result, Type *DestTy,
79 Constant *C, Type *SrcEltTy,
80 unsigned NumSrcElts,
81 const DataLayout &DL) {
82 // Now that we know that the input value is a vector of integers, just shift
83 // and insert them into our result.
84 unsigned BitShift = DL.getTypeSizeInBits(SrcEltTy);
85 for (unsigned i = 0; i != NumSrcElts; ++i) {
86 Constant *Element;
87 if (DL.isLittleEndian())
88 Element = C->getAggregateElement(NumSrcElts - i - 1);
89 else
90 Element = C->getAggregateElement(i);
91
92 if (isa_and_nonnull<UndefValue>(Element)) {
93 Result <<= BitShift;
94 continue;
95 }
96
97 auto *ElementCI = dyn_cast_or_null<ConstantInt>(Element);
98 if (!ElementCI)
99 return ConstantExpr::getBitCast(C, DestTy);
100
101 Result <<= BitShift;
102 Result |= ElementCI->getValue().zext(Result.getBitWidth());
103 }
104
105 return nullptr;
106}
107
108/// Check whether folding this bitcast into a byte vector would mix poison and
109/// non-poison bits in the same output lane. While integer types track poison on
110/// a per-value basis, byte types track it on a per-bit basis. However,
111/// `ConstantByte` cannot represent values with both poison and non-poison bits.
112///
113/// Source elements are grouped by the output lane they map to. Returns true if
114/// any group contains both poison and non-poison elements.
115static bool foldMixesPoisonBits(Constant *C, unsigned NumSrcElt,
116 unsigned NumDstElt) {
117 // If element counts don't divide evenly, bail out if a poison source element
118 // might span multiple destination lanes.
119 if (NumSrcElt % NumDstElt != 0)
120 return C->containsPoisonElement();
121 unsigned Ratio = NumSrcElt / NumDstElt;
122 for (unsigned i = 0; i != NumSrcElt; i += Ratio) {
123 bool HasPoison = false;
124 bool HasNonPoison = false;
125 for (unsigned j = 0; j != Ratio; ++j) {
126 Constant *Src = C->getAggregateElement(i + j);
127 // Conservatively bail out.
128 if (!Src)
129 return true;
130 if (isa<PoisonValue>(Src))
131 HasPoison = true;
132 else
133 HasNonPoison = true;
134 }
135 if (HasPoison && HasNonPoison)
136 return true;
137 }
138 return false;
139}
140
141/// Track which destination lanes of a bitcast are produced from poison bytes.
142/// A destination lane is marked if any source element mapped to it is poison.
143/// Returns false if an aggregate element cannot be inspected. The caller should
144/// bail out of folding.
145static bool computePoisonDstLanes(Constant *C, unsigned NumSrcElt,
146 unsigned NumDstElt,
147 SmallBitVector &PoisonDstElts) {
148 // If element counts don't divide evenly, bail out if a poison source element
149 // might span multiple destination lanes.
150 if ((NumDstElt < NumSrcElt ? NumSrcElt % NumDstElt : NumDstElt % NumSrcElt))
151 return !C->containsPoisonElement();
152 if (NumDstElt < NumSrcElt) {
153 unsigned Ratio = NumSrcElt / NumDstElt;
154 for (unsigned i = 0; i != NumDstElt; ++i) {
155 for (unsigned j = 0; j != Ratio; ++j) {
156 Constant *Src = C->getAggregateElement(i * Ratio + j);
157 if (!Src)
158 return false;
159 if (isa<PoisonValue>(Src)) {
160 PoisonDstElts[i] = true;
161 break;
162 }
163 }
164 }
165 } else {
166 unsigned Ratio = NumDstElt / NumSrcElt;
167 for (unsigned i = 0; i != NumSrcElt; ++i) {
168 Constant *Src = C->getAggregateElement(i);
169 if (!Src)
170 return false;
171 if (isa<PoisonValue>(Src))
172 PoisonDstElts.set(i * Ratio, (i + 1) * Ratio);
173 }
174 }
175 return true;
176}
177
178/// Constant fold bitcast, symbolically evaluating it with DataLayout.
179/// This always returns a non-null constant, but it may be a
180/// ConstantExpr if unfoldable.
181Constant *FoldBitCast(Constant *C, Type *DestTy, const DataLayout &DL) {
182 assert(CastInst::castIsValid(Instruction::BitCast, C, DestTy) &&
183 "Invalid constantexpr bitcast!");
184
185 // Catch the obvious splat cases.
186 if (Constant *Res = ConstantFoldLoadFromUniformValue(C, DestTy, DL))
187 return Res;
188
189 if (auto *VTy = dyn_cast<VectorType>(C->getType())) {
190 // Handle a vector->scalar integer/fp cast.
191 if (isa<IntegerType>(DestTy) || DestTy->isFloatingPointTy()) {
192 unsigned NumSrcElts = cast<FixedVectorType>(VTy)->getNumElements();
193 Type *SrcEltTy = VTy->getElementType();
194
195 // Bitcasting a byte containing any poison bit to an integer or fp type
196 // yields poison.
197 if (SrcEltTy->isByteTy() && C->containsPoisonElement())
198 return PoisonValue::get(DestTy);
199
200 // If the vector is a vector of floating point or bytes, convert it to a
201 // vector of int to simplify things.
202 if (SrcEltTy->isFloatingPointTy() || SrcEltTy->isByteTy()) {
203 unsigned Width = SrcEltTy->getPrimitiveSizeInBits();
204 auto *SrcIVTy = FixedVectorType::get(
205 IntegerType::get(C->getContext(), Width), NumSrcElts);
206 // Ask IR to do the conversion now that #elts line up.
207 C = ConstantExpr::getBitCast(C, SrcIVTy);
208 }
209
210 APInt Result(DL.getTypeSizeInBits(DestTy), 0);
211 if (Constant *CE = foldConstVectorToAPInt(Result, DestTy, C,
212 SrcEltTy, NumSrcElts, DL))
213 return CE;
214
215 if (isa<IntegerType>(DestTy))
216 return ConstantInt::get(DestTy, Result);
217
218 APFloat FP(DestTy->getFltSemantics(), Result);
219 return ConstantFP::get(DestTy->getContext(), FP);
220 }
221 }
222
223 // The code below only handles casts to vectors currently.
224 auto *DestVTy = dyn_cast<VectorType>(DestTy);
225 if (!DestVTy)
226 return ConstantExpr::getBitCast(C, DestTy);
227
228 // If this is a scalar -> vector cast, convert the input into a <1 x scalar>
229 // vector so the code below can handle it uniformly.
230 if (!isa<VectorType>(C->getType()) &&
232 Constant *Ops = C; // don't take the address of C!
233 return FoldBitCast(ConstantVector::get(Ops), DestTy, DL);
234 }
235
236 // Some of what follows may extend to cover scalable vectors but the current
237 // implementation is fixed length specific.
238 if (!isa<FixedVectorType>(C->getType()))
239 return ConstantExpr::getBitCast(C, DestTy);
240
241 // If this is a bitcast from constant vector -> vector, fold it.
244 return ConstantExpr::getBitCast(C, DestTy);
245
246 // If the element types match, IR can fold it.
247 unsigned NumDstElt = cast<FixedVectorType>(DestVTy)->getNumElements();
248 unsigned NumSrcElt = cast<FixedVectorType>(C->getType())->getNumElements();
249 if (NumDstElt == NumSrcElt)
250 return ConstantExpr::getBitCast(C, DestTy);
251
252 Type *SrcEltTy = cast<VectorType>(C->getType())->getElementType();
253 Type *DstEltTy = DestVTy->getElementType();
254
255 // Otherwise, we're changing the number of elements in a vector, which
256 // requires endianness information to do the right thing. For example,
257 // bitcast (<2 x i64> <i64 0, i64 1> to <4 x i32>)
258 // folds to (little endian):
259 // <4 x i32> <i32 0, i32 0, i32 1, i32 0>
260 // and to (big endian):
261 // <4 x i32> <i32 0, i32 0, i32 0, i32 1>
262
263 // First thing is first. We only want to think about integer here, so if
264 // we have something in FP form, recast it as integer.
265 if (DstEltTy->isFloatingPointTy()) {
266 // Fold to an vector of integers with same size as our FP type.
267 unsigned FPWidth = DstEltTy->getPrimitiveSizeInBits();
268 auto *DestIVTy = FixedVectorType::get(
269 IntegerType::get(C->getContext(), FPWidth), NumDstElt);
270 // Recursively handle this integer conversion, if possible.
271 C = FoldBitCast(C, DestIVTy, DL);
272
273 // Finally, IR can handle this now that #elts line up.
274 return ConstantExpr::getBitCast(C, DestTy);
275 }
276
277 // Handle byte destination type by folding through integers.
278 if (DstEltTy->isByteTy()) {
279 // When combining elements into larger byte values, bail out if the fold
280 // mixes poison and non-poison bits in the same destination element. Byte
281 // types track poison per bit, and no constant value can represent that.
282 if (NumDstElt < NumSrcElt && foldMixesPoisonBits(C, NumSrcElt, NumDstElt))
283 return ConstantExpr::getBitCast(C, DestTy);
284
285 // Fold to a vector of integers with same size as the byte type.
286 unsigned ByteWidth = DstEltTy->getPrimitiveSizeInBits();
287 auto *DestIVTy = FixedVectorType::get(
288 IntegerType::get(C->getContext(), ByteWidth), NumDstElt);
289 C = FoldBitCast(C, DestIVTy, DL);
290 return ConstantExpr::getBitCast(C, DestTy);
291 }
292
293 // Okay, we know the destination is integer, if the input is FP, convert
294 // it to integer first.
295 if (SrcEltTy->isFloatingPointTy()) {
296 unsigned FPWidth = SrcEltTy->getPrimitiveSizeInBits();
297 auto *SrcIVTy = FixedVectorType::get(
298 IntegerType::get(C->getContext(), FPWidth), NumSrcElt);
299 // Ask IR to do the conversion now that #elts line up.
300 C = ConstantExpr::getBitCast(C, SrcIVTy);
301 assert((isa<ConstantVector>(C) || // FIXME: Remove ConstantVector.
303 "Constant folding cannot fail for plain fp->int bitcast!");
304 }
305
306 // Handle byte source type by folding through integers. Byte types track
307 // poison per bit, so any poison bit makes the destination lane poison.
308 // Record which destination lanes contain poison bits, before the generic
309 // fold below refines them to undef/zero, so they can be restored.
310 SmallBitVector PoisonDstElts(NumDstElt);
311 if (SrcEltTy->isByteTy()) {
312 if (!computePoisonDstLanes(C, NumSrcElt, NumDstElt, PoisonDstElts))
313 return ConstantExpr::getBitCast(C, DestTy);
314
315 unsigned ByteWidth = SrcEltTy->getPrimitiveSizeInBits();
316 auto *SrcIVTy = FixedVectorType::get(
317 IntegerType::get(C->getContext(), ByteWidth), NumSrcElt);
318 // Ask IR to do the conversion now that #elts line up.
319 C = ConstantExpr::getBitCast(C, SrcIVTy);
320 assert((isa<ConstantVector>(C) || // FIXME: Remove ConstantVector.
322 "Constant folding cannot fail for plain byte->int bitcast!");
323 }
324
325 // Now we know that the input and output vectors are both integer vectors
326 // of the same size, and that their #elements is not the same.
327 // Use data buffer for easy non-integer element ratio vectors handling,
328 // For example: <4 x i24> to <3 x i32>.
329 bool isLittleEndian = DL.isLittleEndian();
330 unsigned SrcBitSize = SrcEltTy->getPrimitiveSizeInBits();
331 unsigned DstBitSize = DstEltTy->getPrimitiveSizeInBits();
333 unsigned SrcElt = 0;
334
335 APInt Buffer(2 * std::max(SrcBitSize, DstBitSize), 0);
336 APInt UndefMask(Buffer.getBitWidth(), 0);
337 APInt PoisonMask(Buffer.getBitWidth(), 0);
338 unsigned BufferBitSize = 0;
339
340 while (Result.size() != NumDstElt) {
341 // Load SrcElts into Buffer.
342 while (BufferBitSize < DstBitSize) {
343 Constant *Element = C->getAggregateElement(SrcElt++);
344 if (!Element) // Reject constantexpr elements
345 return ConstantExpr::getBitCast(C, DestTy);
346
347 // Shift Buffer & Masks to fit next SrcElt.
348 if (!isLittleEndian) {
349 Buffer <<= SrcBitSize;
350 UndefMask <<= SrcBitSize;
351 PoisonMask <<= SrcBitSize;
352 }
353
354 APInt SrcValue;
355 unsigned BitPosition = isLittleEndian ? BufferBitSize : 0;
356 if (isa<UndefValue>(Element)) {
357 // Set masks fragments bits.
358 UndefMask.setBits(BitPosition, BitPosition + SrcBitSize);
359 if (isa<PoisonValue>(Element))
360 PoisonMask.setBits(BitPosition, BitPosition + SrcBitSize);
361 SrcValue = APInt::getZero(SrcBitSize);
362 } else {
363 auto *Src = dyn_cast<ConstantInt>(Element);
364 if (!Src)
365 return ConstantExpr::getBitCast(C, DestTy);
366 SrcValue = Src->getValue();
367 }
368
369 // Insert src element bits into Buffer on correct position.
370 Buffer.insertBits(SrcValue, BitPosition);
371 BufferBitSize += SrcBitSize;
372 }
373
374 // Create DstElts from Buffer.
375 while (BufferBitSize >= DstBitSize) {
376 unsigned ShiftAmt = isLittleEndian ? 0 : BufferBitSize - DstBitSize;
377 // Emit undef/poison, if all undef mask fragment bits are set.
378 if (UndefMask.extractBits(DstBitSize, ShiftAmt).isAllOnes()) {
379 // Push poison, if any bit in poison mask fragment is set.
380 if (!PoisonMask.extractBits(DstBitSize, ShiftAmt).isZero()) {
381 Result.push_back(PoisonValue::get(DstEltTy));
382 } else {
383 Result.push_back(UndefValue::get(DstEltTy));
384 }
385 } else {
386 // Create and push DstElt.
387 APInt Elt = Buffer.extractBits(DstBitSize, ShiftAmt);
388 Result.push_back(ConstantInt::get(DstEltTy, Elt));
389 }
390
391 // Shift unused Buffer fragment to lower bits.
392 if (isLittleEndian) {
393 Buffer.lshrInPlace(DstBitSize);
394 UndefMask.lshrInPlace(DstBitSize);
395 PoisonMask.lshrInPlace(DstBitSize);
396 }
397 BufferBitSize -= DstBitSize;
398 }
399 }
400
401 // Restore destination lanes whose source bytes contained poison bits.
402 for (unsigned I : PoisonDstElts.set_bits())
403 Result[I] = PoisonValue::get(DstEltTy);
404
405 return ConstantVector::get(Result);
406}
407
408} // end anonymous namespace
409
410/// If this constant is a constant offset from a global, return the global and
411/// the constant. Because of constantexprs, this function is recursive.
413 APInt &Offset, const DataLayout &DL,
414 DSOLocalEquivalent **DSOEquiv) {
415 if (DSOEquiv)
416 *DSOEquiv = nullptr;
417
418 // Trivial case, constant is the global.
419 if ((GV = dyn_cast<GlobalValue>(C))) {
420 unsigned BitWidth = DL.getIndexTypeSizeInBits(GV->getType());
421 Offset = APInt(BitWidth, 0);
422 return true;
423 }
424
425 if (auto *FoundDSOEquiv = dyn_cast<DSOLocalEquivalent>(C)) {
426 if (DSOEquiv)
427 *DSOEquiv = FoundDSOEquiv;
428 GV = FoundDSOEquiv->getGlobalValue();
429 unsigned BitWidth = DL.getIndexTypeSizeInBits(GV->getType());
430 Offset = APInt(BitWidth, 0);
431 return true;
432 }
433
434 // Otherwise, if this isn't a constant expr, bail out.
435 auto *CE = dyn_cast<ConstantExpr>(C);
436 if (!CE) return false;
437
438 // Look through ptr->int and ptr->ptr casts.
439 if (CE->getOpcode() == Instruction::PtrToInt ||
440 CE->getOpcode() == Instruction::PtrToAddr)
441 return IsConstantOffsetFromGlobal(CE->getOperand(0), GV, Offset, DL,
442 DSOEquiv);
443
444 // i32* getelementptr ([5 x i32]* @a, i32 0, i32 5)
445 auto *GEP = dyn_cast<GEPOperator>(CE);
446 if (!GEP)
447 return false;
448
449 unsigned BitWidth = DL.getIndexTypeSizeInBits(GEP->getType());
450 APInt TmpOffset(BitWidth, 0);
451
452 // If the base isn't a global+constant, we aren't either.
453 if (!IsConstantOffsetFromGlobal(CE->getOperand(0), GV, TmpOffset, DL,
454 DSOEquiv))
455 return false;
456
457 // Otherwise, add any offset that our operands provide.
458 if (!GEP->accumulateConstantOffset(DL, TmpOffset))
459 return false;
460
461 Offset = TmpOffset;
462 return true;
463}
464
466 const DataLayout &DL) {
467 do {
468 Type *SrcTy = C->getType();
469 if (SrcTy == DestTy)
470 return C;
471
472 TypeSize DestSize = DL.getTypeSizeInBits(DestTy);
473 TypeSize SrcSize = DL.getTypeSizeInBits(SrcTy);
474 if (!TypeSize::isKnownGE(SrcSize, DestSize))
475 return nullptr;
476
477 // Catch the obvious splat cases (since all-zeros can coerce non-integral
478 // pointers legally).
479 if (Constant *Res = ConstantFoldLoadFromUniformValue(C, DestTy, DL))
480 return Res;
481
482 // If the type sizes are the same and a cast is legal, just directly
483 // cast the constant.
484 // But be careful not to coerce non-integral pointers illegally.
485 if (SrcSize == DestSize &&
486 DL.isNonIntegralPointerType(SrcTy->getScalarType()) ==
487 DL.isNonIntegralPointerType(DestTy->getScalarType())) {
488 Instruction::CastOps Cast = Instruction::BitCast;
489 // If we are going from a pointer to int or vice versa, we spell the cast
490 // differently.
491 if (SrcTy->isIntegerTy() && DestTy->isPointerTy())
492 Cast = Instruction::IntToPtr;
493 else if (SrcTy->isPointerTy() && DestTy->isIntegerTy())
494 Cast = Instruction::PtrToInt;
495
496 if (CastInst::castIsValid(Cast, C, DestTy))
497 return ConstantFoldCastOperand(Cast, C, DestTy, DL);
498 }
499
500 // If this isn't an aggregate type, there is nothing we can do to drill down
501 // and find a bitcastable constant.
502 if (!SrcTy->isAggregateType() && !SrcTy->isVectorTy())
503 return nullptr;
504
505 // We're simulating a load through a pointer that was bitcast to point to
506 // a different type, so we can try to walk down through the initial
507 // elements of an aggregate to see if some part of the aggregate is
508 // castable to implement the "load" semantic model.
509 if (SrcTy->isStructTy()) {
510 // Struct types might have leading zero-length elements like [0 x i32],
511 // which are certainly not what we are looking for, so skip them.
512 unsigned Elem = 0;
513 Constant *ElemC;
514 do {
515 ElemC = C->getAggregateElement(Elem++);
516 } while (ElemC && DL.getTypeSizeInBits(ElemC->getType()).isZero());
517 C = ElemC;
518 } else {
519 // For non-byte-sized vector elements, the first element is not
520 // necessarily located at the vector base address.
521 if (auto *VT = dyn_cast<VectorType>(SrcTy))
522 if (!DL.typeSizeEqualsStoreSize(VT->getElementType()))
523 return nullptr;
524
525 C = C->getAggregateElement(0u);
526 }
527 } while (C);
528
529 return nullptr;
530}
531
532namespace {
533
534/// Recursive helper to read bits out of global. C is the constant being copied
535/// out of. ByteOffset is an offset into C. CurPtr is the pointer to copy
536/// results into and BytesLeft is the number of bytes left in
537/// the CurPtr buffer. DL is the DataLayout. When IsByteLoad is true, do not
538/// unwrap inttoptr constant expressions. The caller would reconstruct those
539/// bits as a ConstantByte, dropping the pointer's provenance.
540bool ReadDataFromGlobal(Constant *C, uint64_t ByteOffset, unsigned char *CurPtr,
541 unsigned BytesLeft, const DataLayout &DL,
542 bool IsByteLoad = false) {
543 assert(ByteOffset <= DL.getTypeAllocSize(C->getType()) &&
544 "Out of range access");
545
546 // Reading type padding, return zero.
547 if (ByteOffset >= DL.getTypeStoreSize(C->getType()))
548 return true;
549
550 // If this element is zero or undefined, we can just return since *CurPtr is
551 // zero initialized.
553 return true;
554
555 auto *CI = dyn_cast<ConstantInt>(C);
556 if (CI && CI->getType()->isIntegerTy()) {
557 if ((CI->getBitWidth() & 7) != 0)
558 return false;
559 const APInt &Val = CI->getValue();
560 unsigned IntBytes = unsigned(CI->getBitWidth()/8);
561
562 for (unsigned i = 0; i != BytesLeft && ByteOffset != IntBytes; ++i) {
563 unsigned n = ByteOffset;
564 if (!DL.isLittleEndian())
565 n = IntBytes - n - 1;
566 CurPtr[i] = Val.extractBits(8, n * 8).getZExtValue();
567 ++ByteOffset;
568 }
569 return true;
570 }
571
572 auto *CFP = dyn_cast<ConstantFP>(C);
573 if (CFP && CFP->getType()->isFloatingPointTy()) {
574 if (CFP->getType()->isDoubleTy()) {
575 C = FoldBitCast(C, Type::getInt64Ty(C->getContext()), DL);
576 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
577 IsByteLoad);
578 }
579 if (CFP->getType()->isFloatTy()){
580 C = FoldBitCast(C, Type::getInt32Ty(C->getContext()), DL);
581 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
582 IsByteLoad);
583 }
584 if (CFP->getType()->isHalfTy()){
585 C = FoldBitCast(C, Type::getInt16Ty(C->getContext()), DL);
586 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
587 IsByteLoad);
588 }
589 return false;
590 }
591
592 if (auto *CS = dyn_cast<ConstantStruct>(C)) {
593 const StructLayout *SL = DL.getStructLayout(CS->getType());
594 unsigned Index = SL->getElementContainingOffset(ByteOffset);
595 uint64_t CurEltOffset = SL->getElementOffset(Index);
596 ByteOffset -= CurEltOffset;
597
598 while (true) {
599 // If the element access is to the element itself and not to tail padding,
600 // read the bytes from the element.
601 uint64_t EltSize = DL.getTypeAllocSize(CS->getOperand(Index)->getType());
602
603 if (ByteOffset < EltSize &&
604 !ReadDataFromGlobal(CS->getOperand(Index), ByteOffset, CurPtr,
605 BytesLeft, DL, IsByteLoad))
606 return false;
607
608 ++Index;
609
610 // Check to see if we read from the last struct element, if so we're done.
611 if (Index == CS->getType()->getNumElements())
612 return true;
613
614 // If we read all of the bytes we needed from this element we're done.
615 uint64_t NextEltOffset = SL->getElementOffset(Index);
616
617 if (BytesLeft <= NextEltOffset - CurEltOffset - ByteOffset)
618 return true;
619
620 // Move to the next element of the struct.
621 CurPtr += NextEltOffset - CurEltOffset - ByteOffset;
622 BytesLeft -= NextEltOffset - CurEltOffset - ByteOffset;
623 ByteOffset = 0;
624 CurEltOffset = NextEltOffset;
625 }
626 // not reached.
627 }
628
632 uint64_t NumElts, EltSize;
633 Type *EltTy;
634 if (auto *AT = dyn_cast<ArrayType>(C->getType())) {
635 NumElts = AT->getNumElements();
636 EltTy = AT->getElementType();
637 EltSize = DL.getTypeAllocSize(EltTy);
638 } else {
639 NumElts = cast<FixedVectorType>(C->getType())->getNumElements();
640 EltTy = cast<FixedVectorType>(C->getType())->getElementType();
641 // TODO: For non-byte-sized vectors, current implementation assumes there is
642 // padding to the next byte boundary between elements.
643 if (!DL.typeSizeEqualsStoreSize(EltTy))
644 return false;
645
646 EltSize = DL.getTypeStoreSize(EltTy);
647 }
648 uint64_t Index = ByteOffset / EltSize;
649 uint64_t Offset = ByteOffset - Index * EltSize;
650
651 for (; Index != NumElts; ++Index) {
652 if (!ReadDataFromGlobal(C->getAggregateElement(Index), Offset, CurPtr,
653 BytesLeft, DL, IsByteLoad))
654 return false;
655
656 uint64_t BytesWritten = EltSize - Offset;
657 assert(BytesWritten <= EltSize && "Not indexing into this element?");
658 if (BytesWritten >= BytesLeft)
659 return true;
660
661 Offset = 0;
662 BytesLeft -= BytesWritten;
663 CurPtr += BytesWritten;
664 }
665 return true;
666 }
667
668 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
669 if (CE->getOpcode() == Instruction::IntToPtr &&
670 CE->getOperand(0)->getType() == DL.getIntPtrType(CE->getType())) {
671 // Folding byte loads through the integer operand would rebuild the result
672 // as a `ConstantByte`, dropping the pointer's provenance.
673 if (IsByteLoad)
674 return false;
675 return ReadDataFromGlobal(CE->getOperand(0), ByteOffset, CurPtr,
676 BytesLeft, DL, IsByteLoad);
677 }
678 }
679
680 // Otherwise, unknown initializer type.
681 return false;
682}
683
684/// OrigLoadTy is the original type being loaded, while LoadTy is the type
685/// currently being folded (which may be integer type mapped from OrigLoadTy).
686Constant *FoldReinterpretLoadFromConst(Constant *C, Type *LoadTy,
687 Type *OrigLoadTy, int64_t Offset,
688 const DataLayout &DL) {
689 // Bail out early. Not expect to load from scalable global variable.
690 if (isa<ScalableVectorType>(LoadTy))
691 return nullptr;
692
693 auto *IntType = dyn_cast<IntegerType>(LoadTy);
694
695 // If this isn't an integer load we can't fold it directly.
696 if (!IntType) {
697 // If this is a non-integer load, we can try folding it as an int load and
698 // then bitcast the result. This can be useful for union cases. Note
699 // that address spaces don't matter here since we're not going to result in
700 // an actual new load.
701 if (!LoadTy->isFloatingPointTy() && !LoadTy->isPointerTy() &&
702 !LoadTy->isByteTy() && !LoadTy->isVectorTy())
703 return nullptr;
704
705 Type *MapTy = Type::getIntNTy(C->getContext(),
706 DL.getTypeSizeInBits(LoadTy).getFixedValue());
707 if (Constant *Res =
708 FoldReinterpretLoadFromConst(C, MapTy, OrigLoadTy, Offset, DL)) {
709 if (Res->isNullValue() && !LoadTy->isX86_AMXTy())
710 // Materializing a zero can be done trivially without a bitcast
711 return Constant::getNullValue(LoadTy);
712 Type *CastTy = LoadTy->isPtrOrPtrVectorTy() ? DL.getIntPtrType(LoadTy) : LoadTy;
713 Res = FoldBitCast(Res, CastTy, DL);
714 if (LoadTy->isPtrOrPtrVectorTy()) {
715 // For vector of pointer, we needed to first convert to a vector of integer, then do vector inttoptr
716 if (Res->isNullValue() && !LoadTy->isX86_AMXTy())
717 return Constant::getNullValue(LoadTy);
718 if (DL.isNonIntegralPointerType(LoadTy->getScalarType()))
719 // Be careful not to replace a load of an addrspace value with an inttoptr here
720 return nullptr;
721 Res = ConstantExpr::getIntToPtr(Res, LoadTy);
722 }
723 return Res;
724 }
725 return nullptr;
726 }
727
728 unsigned BytesLoaded = (IntType->getBitWidth() + 7) / 8;
729 // Allow folding of large type loads (e.g. <16 x double>).
730 if (BytesLoaded > 128 || BytesLoaded == 0)
731 return nullptr;
732
733 // For scalar integer load, use smaller limit to avoid regression during
734 // memcmp expansion. Codegen may generate inefficient string operations.
735 if (BytesLoaded > 32 && OrigLoadTy->isIntegerTy())
736 return nullptr;
737
738 // If we're not accessing anything in this constant, the result is undefined.
739 if (Offset <= -1 * static_cast<int64_t>(BytesLoaded))
740 return PoisonValue::get(IntType);
741
742 // TODO: We should be able to support scalable types.
743 TypeSize InitializerSize = DL.getTypeAllocSize(C->getType());
744 if (InitializerSize.isScalable())
745 return nullptr;
746
747 // If we're not accessing anything in this constant, the result is undefined.
748 if (Offset >= (int64_t)InitializerSize.getFixedValue())
749 return PoisonValue::get(IntType);
750
751 SmallVector<unsigned char, 64> RawBytes(BytesLoaded);
752 unsigned char *CurPtr = RawBytes.data();
753 unsigned BytesLeft = BytesLoaded;
754
755 // If we're loading off the beginning of the global, some bytes may be valid.
756 if (Offset < 0) {
757 CurPtr += -Offset;
758 BytesLeft += Offset;
759 Offset = 0;
760 }
761
762 if (!ReadDataFromGlobal(C, Offset, CurPtr, BytesLeft, DL,
763 /*IsByteLoad=*/OrigLoadTy->isByteOrByteVectorTy()))
764 return nullptr;
765
766 APInt ResultVal = APInt(IntType->getBitWidth(), 0);
767 if (DL.isLittleEndian()) {
768 ResultVal = RawBytes[BytesLoaded - 1];
769 for (unsigned i = 1; i != BytesLoaded; ++i) {
770 ResultVal <<= 8;
771 ResultVal |= RawBytes[BytesLoaded - 1 - i];
772 }
773 } else {
774 ResultVal = RawBytes[0];
775 for (unsigned i = 1; i != BytesLoaded; ++i) {
776 ResultVal <<= 8;
777 ResultVal |= RawBytes[i];
778 }
779 }
780
781 return ConstantInt::get(IntType->getContext(), ResultVal);
782}
783
784} // anonymous namespace
785
786// If GV is a constant with an initializer read its representation starting
787// at Offset and return it as a constant array of unsigned char. Otherwise
788// return null.
790 uint64_t Offset) {
791 if (!GV->isConstant() || !GV->hasDefinitiveInitializer())
792 return nullptr;
793
794 const DataLayout &DL = GV->getDataLayout();
795 Constant *Init = const_cast<Constant *>(GV->getInitializer());
796 TypeSize InitSize = DL.getTypeAllocSize(Init->getType());
797 if (InitSize < Offset)
798 return nullptr;
799
800 uint64_t NBytes = InitSize - Offset;
801 if (NBytes > UINT16_MAX)
802 // Bail for large initializers in excess of 64K to avoid allocating
803 // too much memory.
804 // Offset is assumed to be less than or equal than InitSize (this
805 // is enforced in ReadDataFromGlobal).
806 return nullptr;
807
808 SmallVector<unsigned char, 256> RawBytes(static_cast<size_t>(NBytes));
809 unsigned char *CurPtr = RawBytes.data();
810
811 if (!ReadDataFromGlobal(Init, Offset, CurPtr, NBytes, DL))
812 return nullptr;
813
814 return ConstantDataArray::get(GV->getContext(), RawBytes);
815}
816
817/// If this Offset points exactly to the start of an aggregate element, return
818/// that element, otherwise return nullptr.
820 const DataLayout &DL) {
821 if (Offset.isZero())
822 return Base;
823
825 return nullptr;
826
827 Type *ElemTy = Base->getType();
828 SmallVector<APInt> Indices = DL.getGEPIndicesForOffset(ElemTy, Offset);
829 if (!Offset.isZero() || !Indices[0].isZero())
830 return nullptr;
831
832 Constant *C = Base;
833 for (const APInt &Index : drop_begin(Indices)) {
834 if (Index.isNegative() || Index.getActiveBits() >= 32)
835 return nullptr;
836
837 C = C->getAggregateElement(Index.getZExtValue());
838 if (!C)
839 return nullptr;
840 }
841
842 return C;
843}
844
846 const APInt &Offset,
847 const DataLayout &DL) {
848 if (Constant *AtOffset = getConstantAtOffset(C, Offset, DL))
849 if (Constant *Result = ConstantFoldLoadThroughBitcast(AtOffset, Ty, DL))
850 return Result;
851
852 // Explicitly check for out-of-bounds access, so we return poison even if the
853 // constant is a uniform value.
854 TypeSize Size = DL.getTypeAllocSize(C->getType());
855 if (!Size.isScalable() && Offset.sge(Size.getFixedValue()))
856 return PoisonValue::get(Ty);
857
858 // Try an offset-independent fold of a uniform value.
859 if (Constant *Result = ConstantFoldLoadFromUniformValue(C, Ty, DL))
860 return Result;
861
862 // Try hard to fold loads from bitcasted strange and non-type-safe things.
863 if (Offset.getSignificantBits() <= 64)
864 if (Constant *Result =
865 FoldReinterpretLoadFromConst(C, Ty, Ty, Offset.getSExtValue(), DL))
866 return Result;
867
868 return nullptr;
869}
870
875
878 const DataLayout &DL) {
879 // We can only fold loads from constant globals with a definitive initializer.
880 // Check this upfront, to skip expensive offset calculations.
882 if (!GV || !GV->isConstant() || !GV->hasDefinitiveInitializer())
883 return nullptr;
884
885 C = cast<Constant>(C->stripAndAccumulateConstantOffsets(
886 DL, Offset, /* AllowNonInbounds */ true));
887
888 if (C == GV)
889 if (Constant *Result = ConstantFoldLoadFromConst(GV->getInitializer(), Ty,
890 Offset, DL))
891 return Result;
892
893 // If this load comes from anywhere in a uniform constant global, the value
894 // is always the same, regardless of the loaded offset.
895 return ConstantFoldLoadFromUniformValue(GV->getInitializer(), Ty, DL);
896}
897
899 const DataLayout &DL) {
900 APInt Offset(DL.getIndexTypeSizeInBits(C->getType()), 0);
901 return ConstantFoldLoadFromConstPtr(C, Ty, std::move(Offset), DL);
902}
903
905 const DataLayout &DL) {
906 if (isa<PoisonValue>(C))
907 return PoisonValue::get(Ty);
908 if (isa<UndefValue>(C))
909 return UndefValue::get(Ty);
910 // If padding is needed when storing C to memory, then it isn't considered as
911 // uniform.
912 if (!DL.typeSizeEqualsStoreSize(C->getType()))
913 return nullptr;
914 if (C->isNullValue() && !Ty->isX86_AMXTy())
915 return Constant::getNullValue(Ty);
916 if (C->isAllOnesValue() &&
917 (Ty->isIntOrIntVectorTy() || Ty->isByteOrByteVectorTy() ||
918 Ty->isFPOrFPVectorTy()))
919 return Constant::getAllOnesValue(Ty);
920 return nullptr;
921}
922
923namespace {
924
925/// One of Op0/Op1 is a constant expression.
926/// Attempt to symbolically evaluate the result of a binary operator merging
927/// these together. If target data info is available, it is provided as DL,
928/// otherwise DL is null.
929Constant *SymbolicallyEvaluateBinop(unsigned Opc, Constant *Op0, Constant *Op1,
930 const DataLayout &DL) {
931 // SROA
932
933 // Fold (and 0xffffffff00000000, (shl x, 32)) -> shl.
934 // Fold (lshr (or X, Y), 32) -> (lshr [X/Y], 32) if one doesn't contribute
935 // bits.
936
937 if (Opc == Instruction::And) {
938 KnownBits Known0 = computeKnownBits(Op0, DL);
939 KnownBits Known1 = computeKnownBits(Op1, DL);
940 if ((Known1.One | Known0.Zero).isAllOnes()) {
941 // All the bits of Op0 that the 'and' could be masking are already zero.
942 return Op0;
943 }
944 if ((Known0.One | Known1.Zero).isAllOnes()) {
945 // All the bits of Op1 that the 'and' could be masking are already zero.
946 return Op1;
947 }
948
949 Known0 &= Known1;
950 if (Known0.isConstant())
951 return ConstantInt::get(Op0->getType(), Known0.getConstant());
952 }
953
954 // If the constant expr is something like &A[123] - &A[4].f, fold this into a
955 // constant. This happens frequently when iterating over a global array.
956 if (Opc == Instruction::Sub) {
957 GlobalValue *GV1, *GV2;
958 APInt Offs1, Offs2;
959
960 if (IsConstantOffsetFromGlobal(Op0, GV1, Offs1, DL))
961 if (IsConstantOffsetFromGlobal(Op1, GV2, Offs2, DL) && GV1 == GV2) {
962 unsigned OpSize = DL.getTypeSizeInBits(Op0->getType());
963
964 // (&GV+C1) - (&GV+C2) -> C1-C2, pointer arithmetic cannot overflow.
965 // PtrToInt may change the bitwidth so we have convert to the right size
966 // first.
967 return ConstantInt::get(Op0->getType(), Offs1.zextOrTrunc(OpSize) -
968 Offs2.zextOrTrunc(OpSize));
969 }
970 }
971
972 return nullptr;
973}
974
975/// If array indices are not pointer-sized integers, explicitly cast them so
976/// that they aren't implicitly casted by the getelementptr.
977Constant *CastGEPIndices(Type *SrcElemTy, ArrayRef<Constant *> Ops,
978 Type *ResultTy, GEPNoWrapFlags NW,
979 std::optional<ConstantRange> InRange,
980 const DataLayout &DL, const TargetLibraryInfo *TLI) {
981 Type *IntIdxTy = DL.getIndexType(ResultTy);
982 Type *IntIdxScalarTy = IntIdxTy->getScalarType();
983
984 bool Any = false;
986 for (unsigned i = 1, e = Ops.size(); i != e; ++i) {
987 if ((i == 1 ||
989 SrcElemTy, Ops.slice(1, i - 1)))) &&
990 Ops[i]->getType()->getScalarType() != IntIdxScalarTy) {
991 Any = true;
992 Type *NewType =
993 Ops[i]->getType()->isVectorTy() ? IntIdxTy : IntIdxScalarTy;
995 CastInst::getCastOpcode(Ops[i], true, NewType, true), Ops[i], NewType,
996 DL);
997 if (!NewIdx)
998 return nullptr;
999 NewIdxs.push_back(NewIdx);
1000 } else
1001 NewIdxs.push_back(Ops[i]);
1002 }
1003
1004 if (!Any)
1005 return nullptr;
1006
1007 Constant *C =
1008 ConstantExpr::getGetElementPtr(SrcElemTy, Ops[0], NewIdxs, NW, InRange);
1009 return ConstantFoldConstant(C, DL, TLI);
1010}
1011
1012/// If we can symbolically evaluate the GEP constant expression, do so.
1013Constant *SymbolicallyEvaluateGEP(const GEPOperator *GEP,
1015 const DataLayout &DL,
1016 const TargetLibraryInfo *TLI) {
1017 Type *SrcElemTy = GEP->getSourceElementType();
1018 Type *ResTy = GEP->getType();
1019 if (!SrcElemTy->isSized() || isa<ScalableVectorType>(SrcElemTy))
1020 return nullptr;
1021
1022 if (Constant *C = CastGEPIndices(SrcElemTy, Ops, ResTy, GEP->getNoWrapFlags(),
1023 GEP->getInRange(), DL, TLI))
1024 return C;
1025
1026 Constant *Ptr = Ops[0];
1027 if (!Ptr->getType()->isPointerTy())
1028 return nullptr;
1029
1030 Type *IntIdxTy = DL.getIndexType(Ptr->getType());
1031
1032 for (unsigned i = 1, e = Ops.size(); i != e; ++i)
1033 if (!isa<ConstantInt>(Ops[i]) || !Ops[i]->getType()->isIntegerTy())
1034 return nullptr;
1035
1036 unsigned BitWidth = DL.getTypeSizeInBits(IntIdxTy);
1037 APInt Offset = APInt(
1038 BitWidth,
1039 DL.getIndexedOffsetInType(
1040 SrcElemTy, ArrayRef((Value *const *)Ops.data() + 1, Ops.size() - 1)),
1041 /*isSigned=*/true, /*implicitTrunc=*/true);
1042
1043 std::optional<ConstantRange> InRange = GEP->getInRange();
1044 if (InRange)
1045 InRange = InRange->sextOrTrunc(BitWidth);
1046
1047 // If this is a GEP of a GEP, fold it all into a single GEP.
1048 GEPNoWrapFlags NW = GEP->getNoWrapFlags();
1049 bool Overflow = false;
1050 while (auto *GEP = dyn_cast<GEPOperator>(Ptr)) {
1051 NW &= GEP->getNoWrapFlags();
1052
1053 SmallVector<Value *, 4> NestedOps(llvm::drop_begin(GEP->operands()));
1054
1055 // Do not try the incorporate the sub-GEP if some index is not a number.
1056 bool AllConstantInt = true;
1057 for (Value *NestedOp : NestedOps)
1058 if (!isa<ConstantInt>(NestedOp)) {
1059 AllConstantInt = false;
1060 break;
1061 }
1062 if (!AllConstantInt)
1063 break;
1064
1065 // Adjust inrange offset and intersect inrange attributes
1066 if (auto GEPRange = GEP->getInRange()) {
1067 auto AdjustedGEPRange = GEPRange->sextOrTrunc(BitWidth).subtract(Offset);
1068 InRange =
1069 InRange ? InRange->intersectWith(AdjustedGEPRange) : AdjustedGEPRange;
1070 }
1071
1072 Ptr = cast<Constant>(GEP->getOperand(0));
1073 SrcElemTy = GEP->getSourceElementType();
1074 Offset = Offset.sadd_ov(
1075 APInt(BitWidth, DL.getIndexedOffsetInType(SrcElemTy, NestedOps),
1076 /*isSigned=*/true, /*implicitTrunc=*/true),
1077 Overflow);
1078 }
1079
1080 // Preserving nusw (without inbounds) also requires that the offset
1081 // additions did not overflow.
1082 if (NW.hasNoUnsignedSignedWrap() && !NW.isInBounds() && Overflow)
1084
1085 // If the base value for this address is a literal integer value, fold the
1086 // getelementptr to the resulting integer value casted to the pointer type.
1087 APInt BaseIntVal(DL.getPointerTypeSizeInBits(Ptr->getType()), 0);
1088 if (auto *CE = dyn_cast<ConstantExpr>(Ptr)) {
1089 if (CE->getOpcode() == Instruction::IntToPtr) {
1090 if (auto *Base = dyn_cast<ConstantInt>(CE->getOperand(0)))
1091 BaseIntVal = Base->getValue().zextOrTrunc(BaseIntVal.getBitWidth());
1092 }
1093 }
1094
1095 if ((Ptr->isNullValue() || BaseIntVal != 0) &&
1096 !DL.mustNotIntroduceIntToPtr(Ptr->getType())) {
1097
1098 // If the index size is smaller than the pointer size, add to the low
1099 // bits only.
1100 BaseIntVal.insertBits(BaseIntVal.trunc(BitWidth) + Offset, 0);
1101 Constant *C = ConstantInt::get(Ptr->getContext(), BaseIntVal);
1102 return ConstantExpr::getIntToPtr(C, ResTy);
1103 }
1104
1105 // Try to infer inbounds for GEPs of globals.
1106 if (!NW.isInBounds() && Offset.isNonNegative()) {
1107 bool CanBeNull;
1108 uint64_t DerefBytes = Ptr->getPointerDereferenceableBytes(
1109 DL, CanBeNull, /*CanBeFreed=*/nullptr);
1110 if (DerefBytes != 0 && !CanBeNull && Offset.sle(DerefBytes))
1112 }
1113
1114 // nusw + nneg -> nuw
1115 if (NW.hasNoUnsignedSignedWrap() && Offset.isNonNegative())
1117
1118 // Otherwise canonicalize this to a single ptradd.
1119 LLVMContext &Ctx = Ptr->getContext();
1120 return ConstantExpr::getPtrAdd(Ptr, ConstantInt::get(Ctx, Offset), NW,
1121 InRange);
1122}
1123
1124/// Attempt to constant fold an instruction with the
1125/// specified opcode and operands. If successful, the constant result is
1126/// returned, if not, null is returned. Note that this function can fail when
1127/// attempting to fold instructions like loads and stores, which have no
1128/// constant expression form.
1129Constant *ConstantFoldInstOperandsImpl(const Value *InstOrCE, unsigned Opcode,
1131 const DataLayout &DL,
1132 const TargetLibraryInfo *TLI,
1133 bool AllowNonDeterministic) {
1134 Type *DestTy = InstOrCE->getType();
1135
1136 if (Instruction::isUnaryOp(Opcode))
1137 return ConstantFoldUnaryOpOperand(Opcode, Ops[0], DL);
1138
1139 if (Instruction::isBinaryOp(Opcode)) {
1140 switch (Opcode) {
1141 default:
1142 break;
1143 case Instruction::FAdd:
1144 case Instruction::FSub:
1145 case Instruction::FMul:
1146 case Instruction::FDiv:
1147 case Instruction::FRem:
1148 // Handle floating point instructions separately to account for denormals
1149 // TODO: If a constant expression is being folded rather than an
1150 // instruction, denormals will not be flushed/treated as zero
1151 if (const auto *I = dyn_cast<Instruction>(InstOrCE)) {
1152 return ConstantFoldFPInstOperands(Opcode, Ops[0], Ops[1], DL, I,
1153 AllowNonDeterministic);
1154 }
1155 }
1156 return ConstantFoldBinaryOpOperands(Opcode, Ops[0], Ops[1], DL);
1157 }
1158
1159 if (Instruction::isCast(Opcode))
1160 return ConstantFoldCastOperand(Opcode, Ops[0], DestTy, DL);
1161
1162 if (auto *GEP = dyn_cast<GEPOperator>(InstOrCE)) {
1163 Type *SrcElemTy = GEP->getSourceElementType();
1165 return nullptr;
1166
1167 if (Constant *C = SymbolicallyEvaluateGEP(GEP, Ops, DL, TLI))
1168 return C;
1169
1170 return ConstantExpr::getGetElementPtr(SrcElemTy, Ops[0], Ops.slice(1),
1171 GEP->getNoWrapFlags(),
1172 GEP->getInRange());
1173 }
1174
1175 if (auto *CE = dyn_cast<ConstantExpr>(InstOrCE))
1176 return CE->getWithOperands(Ops);
1177
1178 switch (Opcode) {
1179 default: return nullptr;
1180 case Instruction::ICmp:
1181 case Instruction::FCmp: {
1182 auto *C = cast<CmpInst>(InstOrCE);
1183 return ConstantFoldCompareInstOperands(C->getPredicate(), Ops[0], Ops[1],
1184 DL, TLI, C);
1185 }
1186 case Instruction::Freeze:
1187 return isGuaranteedNotToBeUndefOrPoison(Ops[0]) ? Ops[0] : nullptr;
1188 case Instruction::Call:
1189 if (auto *F = dyn_cast<Function>(Ops.back())) {
1190 const auto *Call = cast<CallBase>(InstOrCE);
1192 return ConstantFoldCall(Call, F, Ops.slice(0, Ops.size() - 1), TLI,
1193 AllowNonDeterministic);
1194 }
1195 return nullptr;
1196 case Instruction::Select:
1197 return ConstantFoldSelectInstruction(Ops[0], Ops[1], Ops[2]);
1198 case Instruction::ExtractElement:
1200 case Instruction::ExtractValue:
1202 Ops[0], cast<ExtractValueInst>(InstOrCE)->getIndices());
1203 case Instruction::InsertElement:
1204 return ConstantExpr::getInsertElement(Ops[0], Ops[1], Ops[2]);
1205 case Instruction::InsertValue:
1207 Ops[0], Ops[1], cast<InsertValueInst>(InstOrCE)->getIndices());
1208 case Instruction::ShuffleVector:
1210 Ops[0], Ops[1], cast<ShuffleVectorInst>(InstOrCE)->getShuffleMask());
1211 case Instruction::Load: {
1212 const auto *LI = dyn_cast<LoadInst>(InstOrCE);
1213 if (LI->isVolatile())
1214 return nullptr;
1215 return ConstantFoldLoadFromConstPtr(Ops[0], LI->getType(), DL);
1216 }
1217 }
1218}
1219
1220} // end anonymous namespace
1221
1222//===----------------------------------------------------------------------===//
1223// Constant Folding public APIs
1224//===----------------------------------------------------------------------===//
1225
1226namespace {
1227
1228Constant *
1229ConstantFoldConstantImpl(const Constant *C, const DataLayout &DL,
1230 const TargetLibraryInfo *TLI,
1233 return const_cast<Constant *>(C);
1234
1236 for (const Use &OldU : C->operands()) {
1237 Constant *OldC = cast<Constant>(&OldU);
1238 Constant *NewC = OldC;
1239 // Recursively fold the ConstantExpr's operands. If we have already folded
1240 // a ConstantExpr, we don't have to process it again.
1241 if (isa<ConstantVector>(OldC) || isa<ConstantExpr>(OldC)) {
1242 auto It = FoldedOps.find(OldC);
1243 if (It == FoldedOps.end()) {
1244 NewC = ConstantFoldConstantImpl(OldC, DL, TLI, FoldedOps);
1245 FoldedOps.insert({OldC, NewC});
1246 } else {
1247 NewC = It->second;
1248 }
1249 }
1250 Ops.push_back(NewC);
1251 }
1252
1253 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1254 if (Constant *Res = ConstantFoldInstOperandsImpl(
1255 CE, CE->getOpcode(), Ops, DL, TLI, /*AllowNonDeterministic=*/true))
1256 return Res;
1257 return const_cast<Constant *>(C);
1258 }
1259
1261 return ConstantVector::get(Ops);
1262}
1263
1264} // end anonymous namespace
1265
1267 const DataLayout &DL,
1268 const TargetLibraryInfo *TLI) {
1269 // Handle PHI nodes quickly here...
1270 if (auto *PN = dyn_cast<PHINode>(I)) {
1271 Constant *CommonValue = nullptr;
1272
1274 for (Value *Incoming : PN->incoming_values()) {
1275 // If the incoming value is undef then skip it. Note that while we could
1276 // skip the value if it is equal to the phi node itself we choose not to
1277 // because that would break the rule that constant folding only applies if
1278 // all operands are constants.
1279 if (isa<UndefValue>(Incoming))
1280 continue;
1281 // If the incoming value is not a constant, then give up.
1282 auto *C = dyn_cast<Constant>(Incoming);
1283 if (!C)
1284 return nullptr;
1285 // Fold the PHI's operands.
1286 C = ConstantFoldConstantImpl(C, DL, TLI, FoldedOps);
1287 // If the incoming value is a different constant to
1288 // the one we saw previously, then give up.
1289 if (CommonValue && C != CommonValue)
1290 return nullptr;
1291 CommonValue = C;
1292 }
1293
1294 // If we reach here, all incoming values are the same constant or undef.
1295 return CommonValue ? CommonValue : UndefValue::get(PN->getType());
1296 }
1297
1298 // Scan the operand list, checking to see if they are all constants, if so,
1299 // hand off to ConstantFoldInstOperandsImpl.
1300 if (!all_of(I->operands(), [](const Use &U) { return isa<Constant>(U); }))
1301 return nullptr;
1302
1305 for (const Use &OpU : I->operands()) {
1306 auto *Op = cast<Constant>(&OpU);
1307 // Fold the Instruction's operands.
1308 Op = ConstantFoldConstantImpl(Op, DL, TLI, FoldedOps);
1309 Ops.push_back(Op);
1310 }
1311
1312 return ConstantFoldInstOperands(I, Ops, DL, TLI);
1313}
1314
1316 const TargetLibraryInfo *TLI) {
1318 return ConstantFoldConstantImpl(C, DL, TLI, FoldedOps);
1319}
1320
1323 const DataLayout &DL,
1324 const TargetLibraryInfo *TLI,
1325 bool AllowNonDeterministic) {
1326 return ConstantFoldInstOperandsImpl(I, I->getOpcode(), Ops, DL, TLI,
1327 AllowNonDeterministic);
1328}
1329
1331 unsigned IntPredicate, Constant *Ops0, Constant *Ops1, const DataLayout &DL,
1332 const TargetLibraryInfo *TLI, const Instruction *I) {
1333 CmpInst::Predicate Predicate = (CmpInst::Predicate)IntPredicate;
1334 // fold: icmp (inttoptr x), null -> icmp x, 0
1335 // fold: icmp null, (inttoptr x) -> icmp 0, x
1336 // fold: icmp (ptrtoint x), 0 -> icmp x, null
1337 // fold: icmp 0, (ptrtoint x) -> icmp null, x
1338 // fold: icmp (inttoptr x), (inttoptr y) -> icmp trunc/zext x, trunc/zext y
1339 // fold: icmp (ptrtoint x), (ptrtoint y) -> icmp x, y
1340 //
1341 // FIXME: The following comment is out of data and the DataLayout is here now.
1342 // ConstantExpr::getCompare cannot do this, because it doesn't have DL
1343 // around to know if bit truncation is happening.
1344 if (auto *CE0 = dyn_cast<ConstantExpr>(Ops0)) {
1345 if (Ops1->isNullValue()) {
1346 if (CE0->getOpcode() == Instruction::IntToPtr) {
1347 Type *IntPtrTy = DL.getIntPtrType(CE0->getType());
1348 // Convert the integer value to the right size to ensure we get the
1349 // proper extension or truncation.
1350 if (Constant *C = ConstantFoldIntegerCast(CE0->getOperand(0), IntPtrTy,
1351 /*IsSigned*/ false, DL)) {
1352 Constant *Null = Constant::getNullValue(C->getType());
1353 return ConstantFoldCompareInstOperands(Predicate, C, Null, DL, TLI);
1354 }
1355 }
1356
1357 // icmp only compares the address part of the pointer, so only do this
1358 // transform if the integer size matches the address size.
1359 if (CE0->getOpcode() == Instruction::PtrToInt ||
1360 CE0->getOpcode() == Instruction::PtrToAddr) {
1361 Type *AddrTy = DL.getAddressType(CE0->getOperand(0)->getType());
1362 if (CE0->getType() == AddrTy) {
1363 Constant *C = CE0->getOperand(0);
1364 Constant *Null = Constant::getNullValue(C->getType());
1365 return ConstantFoldCompareInstOperands(Predicate, C, Null, DL, TLI);
1366 }
1367 }
1368 }
1369
1370 if (auto *CE1 = dyn_cast<ConstantExpr>(Ops1)) {
1371 if (CE0->getOpcode() == CE1->getOpcode()) {
1372 if (CE0->getOpcode() == Instruction::IntToPtr) {
1373 Type *IntPtrTy = DL.getIntPtrType(CE0->getType());
1374
1375 // Convert the integer value to the right size to ensure we get the
1376 // proper extension or truncation.
1377 Constant *C0 = ConstantFoldIntegerCast(CE0->getOperand(0), IntPtrTy,
1378 /*IsSigned*/ false, DL);
1379 Constant *C1 = ConstantFoldIntegerCast(CE1->getOperand(0), IntPtrTy,
1380 /*IsSigned*/ false, DL);
1381 if (C0 && C1)
1382 return ConstantFoldCompareInstOperands(Predicate, C0, C1, DL, TLI);
1383 }
1384
1385 // icmp only compares the address part of the pointer, so only do this
1386 // transform if the integer size matches the address size.
1387 if (CE0->getOpcode() == Instruction::PtrToInt ||
1388 CE0->getOpcode() == Instruction::PtrToAddr) {
1389 Type *AddrTy = DL.getAddressType(CE0->getOperand(0)->getType());
1390 if (CE0->getType() == AddrTy &&
1391 CE0->getOperand(0)->getType() == CE1->getOperand(0)->getType()) {
1393 Predicate, CE0->getOperand(0), CE1->getOperand(0), DL, TLI);
1394 }
1395 }
1396 }
1397 }
1398
1399 // Convert pointer comparison (base+offset1) pred (base+offset2) into
1400 // offset1 pred offset2, for the case where the offset is inbounds. This
1401 // only works for equality and unsigned comparison, as inbounds permits
1402 // crossing the sign boundary. However, the offset comparison itself is
1403 // signed.
1404 if (Ops0->getType()->isPointerTy() && !ICmpInst::isSigned(Predicate)) {
1405 unsigned IndexWidth = DL.getIndexTypeSizeInBits(Ops0->getType());
1406 APInt Offset0(IndexWidth, 0);
1407 bool IsEqPred = ICmpInst::isEquality(Predicate);
1408 Value *Stripped0 = Ops0->stripAndAccumulateConstantOffsets(
1409 DL, Offset0, /*AllowNonInbounds=*/IsEqPred,
1410 /*AllowInvariantGroup=*/false, /*ExternalAnalysis=*/nullptr,
1411 /*LookThroughIntToPtr=*/IsEqPred);
1412 APInt Offset1(IndexWidth, 0);
1413 Value *Stripped1 = Ops1->stripAndAccumulateConstantOffsets(
1414 DL, Offset1, /*AllowNonInbounds=*/IsEqPred,
1415 /*AllowInvariantGroup=*/false, /*ExternalAnalysis=*/nullptr,
1416 /*LookThroughIntToPtr=*/IsEqPred);
1417 if (Stripped0 == Stripped1)
1418 return ConstantInt::getBool(
1419 Ops0->getContext(),
1420 ICmpInst::compare(Offset0, Offset1,
1421 ICmpInst::getSignedPredicate(Predicate)));
1422 }
1423 } else if (isa<ConstantExpr>(Ops1)) {
1424 // If RHS is a constant expression, but the left side isn't, swap the
1425 // operands and try again.
1426 Predicate = ICmpInst::getSwappedPredicate(Predicate);
1427 return ConstantFoldCompareInstOperands(Predicate, Ops1, Ops0, DL, TLI);
1428 }
1429
1430 if (CmpInst::isFPPredicate(Predicate)) {
1431 // Flush any denormal constant float input according to denormal handling
1432 // mode.
1433 Ops0 = FlushFPConstant(Ops0, I, /*IsOutput=*/false);
1434 if (!Ops0)
1435 return nullptr;
1436 Ops1 = FlushFPConstant(Ops1, I, /*IsOutput=*/false);
1437 if (!Ops1)
1438 return nullptr;
1439 }
1440
1441 return ConstantFoldCompareInstruction(Predicate, Ops0, Ops1);
1442}
1443
1445 const DataLayout &DL) {
1447
1448 return ConstantFoldUnaryInstruction(Opcode, Op);
1449}
1450
1452 Constant *RHS,
1453 const DataLayout &DL) {
1455 if (isa<ConstantExpr>(LHS) || isa<ConstantExpr>(RHS))
1456 if (Constant *C = SymbolicallyEvaluateBinop(Opcode, LHS, RHS, DL))
1457 return C;
1458
1460 return ConstantExpr::get(Opcode, LHS, RHS);
1461 return ConstantFoldBinaryInstruction(Opcode, LHS, RHS);
1462}
1463
1466 switch (Mode) {
1468 return nullptr;
1469 case DenormalMode::IEEE:
1470 return ConstantFP::get(Ty, APF);
1472 return ConstantFP::get(
1473 Ty, APFloat::getZero(APF.getSemantics(), APF.isNegative()));
1475 return ConstantFP::get(Ty, APFloat::getZero(APF.getSemantics(), false));
1476 default:
1477 break;
1478 }
1479
1480 llvm_unreachable("unknown denormal mode");
1481}
1482
1483/// Return the denormal mode that can be assumed when executing a floating point
1484/// operation at \p CtxI.
1486 if (!CtxI || !CtxI->getParent() || !CtxI->getFunction())
1487 return DenormalMode::getDynamic();
1488 return CtxI->getFunction()->getDenormalMode(
1489 Ty->getScalarType()->getFltSemantics());
1490}
1491
1493 const Instruction *Inst,
1494 bool IsOutput) {
1495 const APFloat &APF = CFP->getValueAPF();
1496 if (!APF.isDenormal())
1497 return CFP;
1498
1500 return flushDenormalConstant(CFP->getType(), APF,
1501 IsOutput ? Mode.Output : Mode.Input);
1502}
1503
1505 bool IsOutput) {
1506 if (ConstantFP *CFP = dyn_cast<ConstantFP>(Operand))
1507 return flushDenormalConstantFP(CFP, Inst, IsOutput);
1508
1510 return Operand;
1511
1512 Type *Ty = Operand->getType();
1513 VectorType *VecTy = dyn_cast<VectorType>(Ty);
1514 if (VecTy) {
1515 if (auto *Splat = dyn_cast_or_null<ConstantFP>(Operand->getSplatValue())) {
1516 ConstantFP *Folded = flushDenormalConstantFP(Splat, Inst, IsOutput);
1517 if (!Folded)
1518 return nullptr;
1519 return ConstantVector::getSplat(VecTy->getElementCount(), Folded);
1520 }
1521
1522 Ty = VecTy->getElementType();
1523 }
1524
1525 if (isa<ConstantExpr>(Operand))
1526 return Operand;
1527
1528 if (const auto *CV = dyn_cast<ConstantVector>(Operand)) {
1530 for (unsigned i = 0, e = CV->getNumOperands(); i != e; ++i) {
1531 Constant *Element = CV->getAggregateElement(i);
1532 if (isa<UndefValue>(Element)) {
1533 NewElts.push_back(Element);
1534 continue;
1535 }
1536
1537 ConstantFP *CFP = dyn_cast<ConstantFP>(Element);
1538 if (!CFP)
1539 return nullptr;
1540
1541 ConstantFP *Folded = flushDenormalConstantFP(CFP, Inst, IsOutput);
1542 if (!Folded)
1543 return nullptr;
1544 NewElts.push_back(Folded);
1545 }
1546
1547 return ConstantVector::get(NewElts);
1548 }
1549
1550 if (const auto *CDV = dyn_cast<ConstantDataVector>(Operand)) {
1552 for (unsigned I = 0, E = CDV->getNumElements(); I < E; ++I) {
1553 const APFloat &Elt = CDV->getElementAsAPFloat(I);
1554 if (!Elt.isDenormal()) {
1555 NewElts.push_back(ConstantFP::get(Ty, Elt));
1556 } else {
1557 DenormalMode Mode = getInstrDenormalMode(Inst, Ty);
1558 ConstantFP *Folded =
1559 flushDenormalConstant(Ty, Elt, IsOutput ? Mode.Output : Mode.Input);
1560 if (!Folded)
1561 return nullptr;
1562 NewElts.push_back(Folded);
1563 }
1564 }
1565
1566 return ConstantVector::get(NewElts);
1567 }
1568
1569 return nullptr;
1570}
1571
1573 Constant *RHS, const DataLayout &DL,
1574 const Instruction *I,
1575 bool AllowNonDeterministic) {
1576 if (Instruction::isBinaryOp(Opcode)) {
1577 // Flush denormal inputs if needed.
1578 Constant *Op0 = FlushFPConstant(LHS, I, /* IsOutput */ false);
1579 if (!Op0)
1580 return nullptr;
1581 Constant *Op1 = FlushFPConstant(RHS, I, /* IsOutput */ false);
1582 if (!Op1)
1583 return nullptr;
1584
1585 // If nsz or an algebraic FMF flag is set, the result of the FP operation
1586 // may change due to future optimization. Don't constant fold them if
1587 // non-deterministic results are not allowed.
1588 if (!AllowNonDeterministic)
1590 if (FP->hasNoSignedZeros() || FP->hasAllowReassoc() ||
1591 FP->hasAllowContract() || FP->hasAllowReciprocal())
1592 return nullptr;
1593
1594 // Calculate constant result.
1595 Constant *C = ConstantFoldBinaryOpOperands(Opcode, Op0, Op1, DL);
1596 if (!C)
1597 return nullptr;
1598
1599 // Flush denormal output if needed.
1600 C = FlushFPConstant(C, I, /* IsOutput */ true);
1601 if (!C)
1602 return nullptr;
1603
1604 // The precise NaN value is non-deterministic.
1605 if (!AllowNonDeterministic && C->isNaN())
1606 return nullptr;
1607
1608 return C;
1609 }
1610 // If instruction lacks a parent/function and the denormal mode cannot be
1611 // determined, use the default (IEEE).
1612 return ConstantFoldBinaryOpOperands(Opcode, LHS, RHS, DL);
1613}
1614
1616 Type *DestTy, const DataLayout &DL) {
1617 assert(Instruction::isCast(Opcode));
1618
1619 if (auto *CE = dyn_cast<ConstantExpr>(C))
1620 if (CE->isCast())
1621 if (unsigned NewOp = CastInst::isEliminableCastPair(
1622 Instruction::CastOps(CE->getOpcode()),
1623 Instruction::CastOps(Opcode), CE->getOperand(0)->getType(),
1624 C->getType(), DestTy, &DL))
1625 return ConstantFoldCastOperand(NewOp, CE->getOperand(0), DestTy, DL);
1626
1627 switch (Opcode) {
1628 default:
1629 llvm_unreachable("Missing case");
1630 case Instruction::PtrToAddr:
1631 case Instruction::PtrToInt:
1632 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1633 Constant *FoldedValue = nullptr;
1634 // If the input is an inttoptr, eliminate the pair. This requires knowing
1635 // the width of a pointer, so it can't be done in ConstantExpr::getCast.
1636 if (CE->getOpcode() == Instruction::IntToPtr) {
1637 // zext/trunc the inttoptr to pointer/address size.
1638 Type *MidTy = Opcode == Instruction::PtrToInt
1639 ? DL.getAddressType(CE->getType())
1640 : DL.getIntPtrType(CE->getType());
1641 FoldedValue = ConstantFoldIntegerCast(CE->getOperand(0), MidTy,
1642 /*IsSigned=*/false, DL);
1643 } else if (auto *GEP = dyn_cast<GEPOperator>(CE)) {
1644 // If we have GEP, we can perform the following folds:
1645 // (ptrtoint/ptrtoaddr (gep null, x)) -> x
1646 // (ptrtoint/ptrtoaddr (gep (gep null, x), y) -> x + y, etc.
1647 unsigned BitWidth = DL.getIndexTypeSizeInBits(GEP->getType());
1648 APInt BaseOffset(BitWidth, 0);
1649 auto *Base = cast<Constant>(GEP->stripAndAccumulateConstantOffsets(
1650 DL, BaseOffset, /*AllowNonInbounds=*/true));
1651 if (Base->isNullValue()) {
1652 FoldedValue = ConstantInt::get(CE->getContext(), BaseOffset);
1653 } else {
1654 // ptrtoint/ptrtoaddr (gep i8, Ptr, (sub 0, V))
1655 // -> sub (ptrtoint/ptrtoaddr Ptr), V
1656 if (GEP->getNumIndices() == 1 &&
1657 GEP->getSourceElementType()->isIntegerTy(8)) {
1658 auto *Ptr = cast<Constant>(GEP->getPointerOperand());
1659 auto *Sub = dyn_cast<ConstantExpr>(GEP->getOperand(1));
1660 Type *IntIdxTy = DL.getIndexType(Ptr->getType());
1661 if (Sub && Sub->getType() == IntIdxTy &&
1662 Sub->getOpcode() == Instruction::Sub &&
1663 Sub->getOperand(0)->isNullValue())
1664 FoldedValue = ConstantExpr::getSub(
1665 ConstantExpr::getCast(Opcode, Ptr, IntIdxTy),
1666 Sub->getOperand(1));
1667 }
1668 }
1669 }
1670 if (FoldedValue) {
1671 // Do a zext or trunc to get to the ptrtoint/ptrtoaddr dest size.
1672 return ConstantFoldIntegerCast(FoldedValue, DestTy, /*IsSigned=*/false,
1673 DL);
1674 }
1675 }
1676 break;
1677 case Instruction::IntToPtr:
1678 // If the input is a ptrtoint, turn the pair into a ptr to ptr bitcast if
1679 // the int size is >= the ptr size and the address spaces are the same.
1680 // This requires knowing the width of a pointer, so it can't be done in
1681 // ConstantExpr::getCast.
1682 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1683 if (CE->getOpcode() == Instruction::PtrToInt) {
1684 Constant *SrcPtr = CE->getOperand(0);
1685 unsigned SrcPtrSize = DL.getPointerTypeSizeInBits(SrcPtr->getType());
1686 unsigned MidIntSize = CE->getType()->getScalarSizeInBits();
1687
1688 if (MidIntSize >= SrcPtrSize) {
1689 unsigned SrcAS = SrcPtr->getType()->getPointerAddressSpace();
1690 if (SrcAS == DestTy->getPointerAddressSpace())
1691 return FoldBitCast(CE->getOperand(0), DestTy, DL);
1692 }
1693 }
1694 }
1695 break;
1696 case Instruction::Trunc:
1697 case Instruction::ZExt:
1698 case Instruction::SExt:
1699 case Instruction::FPTrunc:
1700 case Instruction::FPExt:
1701 case Instruction::UIToFP:
1702 case Instruction::SIToFP:
1703 case Instruction::FPToUI:
1704 case Instruction::FPToSI:
1705 case Instruction::AddrSpaceCast:
1706 break;
1707 case Instruction::BitCast:
1708 return FoldBitCast(C, DestTy, DL);
1709 }
1710
1712 return ConstantExpr::getCast(Opcode, C, DestTy);
1713 return ConstantFoldCastInstruction(Opcode, C, DestTy);
1714}
1715
1717 bool IsSigned, const DataLayout &DL) {
1718 Type *SrcTy = C->getType();
1719 if (SrcTy == DestTy)
1720 return C;
1721 if (SrcTy->getScalarSizeInBits() > DestTy->getScalarSizeInBits())
1722 return ConstantFoldCastOperand(Instruction::Trunc, C, DestTy, DL);
1723 if (IsSigned)
1724 return ConstantFoldCastOperand(Instruction::SExt, C, DestTy, DL);
1725 return ConstantFoldCastOperand(Instruction::ZExt, C, DestTy, DL);
1726}
1727
1728//===----------------------------------------------------------------------===//
1729// Constant Folding for Calls
1730//
1731
1732/// Returns true if the intrinsic can be constant folded, given \p IsStrictFP.
1733static bool canConstantFoldIntrinsic(Intrinsic::ID ID, bool IsStrictFP) {
1734 switch (ID) {
1735 // Operations that do not operate floating-point numbers and do not depend on
1736 // FP environment can be folded even in strictfp functions.
1737 case Intrinsic::bswap:
1738 case Intrinsic::ctpop:
1739 case Intrinsic::ctlz:
1740 case Intrinsic::cttz:
1741 case Intrinsic::fshl:
1742 case Intrinsic::fshr:
1743 case Intrinsic::clmul:
1744 case Intrinsic::pdep:
1745 case Intrinsic::pext:
1746 case Intrinsic::launder_invariant_group:
1747 case Intrinsic::strip_invariant_group:
1748 case Intrinsic::masked_load:
1749 case Intrinsic::get_active_lane_mask:
1750 case Intrinsic::abs:
1751 case Intrinsic::smax:
1752 case Intrinsic::smin:
1753 case Intrinsic::umax:
1754 case Intrinsic::umin:
1755 case Intrinsic::scmp:
1756 case Intrinsic::ucmp:
1757 case Intrinsic::sadd_with_overflow:
1758 case Intrinsic::uadd_with_overflow:
1759 case Intrinsic::ssub_with_overflow:
1760 case Intrinsic::usub_with_overflow:
1761 case Intrinsic::smul_with_overflow:
1762 case Intrinsic::umul_with_overflow:
1763 case Intrinsic::sadd_sat:
1764 case Intrinsic::uadd_sat:
1765 case Intrinsic::ssub_sat:
1766 case Intrinsic::usub_sat:
1767 case Intrinsic::smul_fix:
1768 case Intrinsic::smul_fix_sat:
1769 case Intrinsic::bitreverse:
1770 case Intrinsic::is_constant:
1771 case Intrinsic::vector_reduce_add:
1772 case Intrinsic::vector_reduce_mul:
1773 case Intrinsic::vector_reduce_and:
1774 case Intrinsic::vector_reduce_or:
1775 case Intrinsic::vector_reduce_xor:
1776 case Intrinsic::vector_reduce_smin:
1777 case Intrinsic::vector_reduce_smax:
1778 case Intrinsic::vector_reduce_umin:
1779 case Intrinsic::vector_reduce_umax:
1780 case Intrinsic::vector_extract:
1781 case Intrinsic::vector_insert:
1782 case Intrinsic::vector_interleave2:
1783 case Intrinsic::vector_interleave3:
1784 case Intrinsic::vector_interleave4:
1785 case Intrinsic::vector_interleave5:
1786 case Intrinsic::vector_interleave6:
1787 case Intrinsic::vector_interleave7:
1788 case Intrinsic::vector_interleave8:
1789 case Intrinsic::vector_deinterleave2:
1790 case Intrinsic::vector_deinterleave3:
1791 case Intrinsic::vector_deinterleave4:
1792 case Intrinsic::vector_deinterleave5:
1793 case Intrinsic::vector_deinterleave6:
1794 case Intrinsic::vector_deinterleave7:
1795 case Intrinsic::vector_deinterleave8:
1796 // Target intrinsics
1797 case Intrinsic::amdgcn_perm:
1798 case Intrinsic::amdgcn_wave_reduce_umin:
1799 case Intrinsic::amdgcn_wave_reduce_umax:
1800 case Intrinsic::amdgcn_wave_reduce_max:
1801 case Intrinsic::amdgcn_wave_reduce_min:
1802 case Intrinsic::amdgcn_wave_reduce_and:
1803 case Intrinsic::amdgcn_wave_reduce_or:
1804 case Intrinsic::amdgcn_s_wqm:
1805 case Intrinsic::amdgcn_s_quadmask:
1806 case Intrinsic::amdgcn_s_bitreplicate:
1807 case Intrinsic::arm_mve_vctp8:
1808 case Intrinsic::arm_mve_vctp16:
1809 case Intrinsic::arm_mve_vctp32:
1810 case Intrinsic::arm_mve_vctp64:
1811 case Intrinsic::aarch64_sve_convert_from_svbool:
1812 case Intrinsic::wasm_alltrue:
1813 case Intrinsic::wasm_anytrue:
1814 case Intrinsic::wasm_dot:
1815 // WebAssembly float semantics are always known
1816 case Intrinsic::wasm_trunc_signed:
1817 case Intrinsic::wasm_trunc_unsigned:
1818 return true;
1819
1820 // Floating point operations cannot be folded in strictfp functions in
1821 // general case. They can be folded if FP environment is known to compiler.
1822 case Intrinsic::minnum:
1823 case Intrinsic::maxnum:
1824 case Intrinsic::minimum:
1825 case Intrinsic::maximum:
1826 case Intrinsic::minimumnum:
1827 case Intrinsic::maximumnum:
1828 case Intrinsic::log:
1829 case Intrinsic::log2:
1830 case Intrinsic::log10:
1831 case Intrinsic::exp:
1832 case Intrinsic::exp2:
1833 case Intrinsic::exp10:
1834 case Intrinsic::sqrt:
1835 case Intrinsic::sin:
1836 case Intrinsic::cos:
1837 case Intrinsic::sincos:
1838 case Intrinsic::sinh:
1839 case Intrinsic::cosh:
1840 case Intrinsic::atan:
1841 case Intrinsic::pow:
1842 case Intrinsic::powi:
1843 case Intrinsic::ldexp:
1844 case Intrinsic::fma:
1845 case Intrinsic::fmuladd:
1846 case Intrinsic::frexp:
1847 case Intrinsic::fptoui_sat:
1848 case Intrinsic::fptosi_sat:
1849 case Intrinsic::amdgcn_cos:
1850 case Intrinsic::amdgcn_cubeid:
1851 case Intrinsic::amdgcn_cubema:
1852 case Intrinsic::amdgcn_cubesc:
1853 case Intrinsic::amdgcn_cubetc:
1854 case Intrinsic::amdgcn_fmul_legacy:
1855 case Intrinsic::amdgcn_fma_legacy:
1856 case Intrinsic::amdgcn_fract:
1857 case Intrinsic::amdgcn_sin:
1858 // The intrinsics below depend on rounding mode in MXCSR.
1859 case Intrinsic::x86_sse_cvtss2si:
1860 case Intrinsic::x86_sse_cvtss2si64:
1861 case Intrinsic::x86_sse_cvttss2si:
1862 case Intrinsic::x86_sse_cvttss2si64:
1863 case Intrinsic::x86_sse2_cvtsd2si:
1864 case Intrinsic::x86_sse2_cvtsd2si64:
1865 case Intrinsic::x86_sse2_cvttsd2si:
1866 case Intrinsic::x86_sse2_cvttsd2si64:
1867 case Intrinsic::x86_avx512_vcvtss2si32:
1868 case Intrinsic::x86_avx512_vcvtss2si64:
1869 case Intrinsic::x86_avx512_cvttss2si:
1870 case Intrinsic::x86_avx512_cvttss2si64:
1871 case Intrinsic::x86_avx512_vcvtsd2si32:
1872 case Intrinsic::x86_avx512_vcvtsd2si64:
1873 case Intrinsic::x86_avx512_cvttsd2si:
1874 case Intrinsic::x86_avx512_cvttsd2si64:
1875 case Intrinsic::x86_avx512_vcvtss2usi32:
1876 case Intrinsic::x86_avx512_vcvtss2usi64:
1877 case Intrinsic::x86_avx512_cvttss2usi:
1878 case Intrinsic::x86_avx512_cvttss2usi64:
1879 case Intrinsic::x86_avx512_vcvtsd2usi32:
1880 case Intrinsic::x86_avx512_vcvtsd2usi64:
1881 case Intrinsic::x86_avx512_cvttsd2usi:
1882 case Intrinsic::x86_avx512_cvttsd2usi64:
1883
1884 // NVVM FMax intrinsics
1885 case Intrinsic::nvvm_fmax_d:
1886 case Intrinsic::nvvm_fmax_f:
1887 case Intrinsic::nvvm_fmax_ftz_f:
1888 case Intrinsic::nvvm_fmax_ftz_nan_f:
1889 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
1890 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
1891 case Intrinsic::nvvm_fmax_nan_f:
1892 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
1893 case Intrinsic::nvvm_fmax_xorsign_abs_f:
1894
1895 // NVVM FMin intrinsics
1896 case Intrinsic::nvvm_fmin_d:
1897 case Intrinsic::nvvm_fmin_f:
1898 case Intrinsic::nvvm_fmin_ftz_f:
1899 case Intrinsic::nvvm_fmin_ftz_nan_f:
1900 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
1901 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
1902 case Intrinsic::nvvm_fmin_nan_f:
1903 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
1904 case Intrinsic::nvvm_fmin_xorsign_abs_f:
1905
1906 // NVVM float/double to int32/uint32 conversion intrinsics
1907 case Intrinsic::nvvm_f2i_rm:
1908 case Intrinsic::nvvm_f2i_rn:
1909 case Intrinsic::nvvm_f2i_rp:
1910 case Intrinsic::nvvm_f2i_rz:
1911 case Intrinsic::nvvm_f2i_rm_ftz:
1912 case Intrinsic::nvvm_f2i_rn_ftz:
1913 case Intrinsic::nvvm_f2i_rp_ftz:
1914 case Intrinsic::nvvm_f2i_rz_ftz:
1915 case Intrinsic::nvvm_f2ui_rm:
1916 case Intrinsic::nvvm_f2ui_rn:
1917 case Intrinsic::nvvm_f2ui_rp:
1918 case Intrinsic::nvvm_f2ui_rz:
1919 case Intrinsic::nvvm_f2ui_rm_ftz:
1920 case Intrinsic::nvvm_f2ui_rn_ftz:
1921 case Intrinsic::nvvm_f2ui_rp_ftz:
1922 case Intrinsic::nvvm_f2ui_rz_ftz:
1923 case Intrinsic::nvvm_d2i_rm:
1924 case Intrinsic::nvvm_d2i_rn:
1925 case Intrinsic::nvvm_d2i_rp:
1926 case Intrinsic::nvvm_d2i_rz:
1927 case Intrinsic::nvvm_d2ui_rm:
1928 case Intrinsic::nvvm_d2ui_rn:
1929 case Intrinsic::nvvm_d2ui_rp:
1930 case Intrinsic::nvvm_d2ui_rz:
1931
1932 // NVVM float/double to int64/uint64 conversion intrinsics
1933 case Intrinsic::nvvm_f2ll_rm:
1934 case Intrinsic::nvvm_f2ll_rn:
1935 case Intrinsic::nvvm_f2ll_rp:
1936 case Intrinsic::nvvm_f2ll_rz:
1937 case Intrinsic::nvvm_f2ll_rm_ftz:
1938 case Intrinsic::nvvm_f2ll_rn_ftz:
1939 case Intrinsic::nvvm_f2ll_rp_ftz:
1940 case Intrinsic::nvvm_f2ll_rz_ftz:
1941 case Intrinsic::nvvm_f2ull_rm:
1942 case Intrinsic::nvvm_f2ull_rn:
1943 case Intrinsic::nvvm_f2ull_rp:
1944 case Intrinsic::nvvm_f2ull_rz:
1945 case Intrinsic::nvvm_f2ull_rm_ftz:
1946 case Intrinsic::nvvm_f2ull_rn_ftz:
1947 case Intrinsic::nvvm_f2ull_rp_ftz:
1948 case Intrinsic::nvvm_f2ull_rz_ftz:
1949 case Intrinsic::nvvm_d2ll_rm:
1950 case Intrinsic::nvvm_d2ll_rn:
1951 case Intrinsic::nvvm_d2ll_rp:
1952 case Intrinsic::nvvm_d2ll_rz:
1953 case Intrinsic::nvvm_d2ull_rm:
1954 case Intrinsic::nvvm_d2ull_rn:
1955 case Intrinsic::nvvm_d2ull_rp:
1956 case Intrinsic::nvvm_d2ull_rz:
1957
1958 // NVVM math intrinsics:
1959 case Intrinsic::nvvm_ceil_d:
1960 case Intrinsic::nvvm_ceil_f:
1961 case Intrinsic::nvvm_ceil_ftz_f:
1962
1963 case Intrinsic::nvvm_fabs:
1964 case Intrinsic::nvvm_fabs_ftz:
1965
1966 case Intrinsic::nvvm_floor_d:
1967 case Intrinsic::nvvm_floor_f:
1968 case Intrinsic::nvvm_floor_ftz_f:
1969
1970 case Intrinsic::nvvm_rcp_rm_d:
1971 case Intrinsic::nvvm_rcp_rm_f:
1972 case Intrinsic::nvvm_rcp_rm_ftz_f:
1973 case Intrinsic::nvvm_rcp_rn_d:
1974 case Intrinsic::nvvm_rcp_rn_f:
1975 case Intrinsic::nvvm_rcp_rn_ftz_f:
1976 case Intrinsic::nvvm_rcp_rp_d:
1977 case Intrinsic::nvvm_rcp_rp_f:
1978 case Intrinsic::nvvm_rcp_rp_ftz_f:
1979 case Intrinsic::nvvm_rcp_rz_d:
1980 case Intrinsic::nvvm_rcp_rz_f:
1981 case Intrinsic::nvvm_rcp_rz_ftz_f:
1982
1983 case Intrinsic::nvvm_round_d:
1984 case Intrinsic::nvvm_round_f:
1985 case Intrinsic::nvvm_round_ftz_f:
1986
1987 case Intrinsic::nvvm_saturate_d:
1988 case Intrinsic::nvvm_saturate_f:
1989 case Intrinsic::nvvm_saturate_ftz_f:
1990
1991 case Intrinsic::nvvm_sqrt_f:
1992 case Intrinsic::nvvm_sqrt_rn_d:
1993 case Intrinsic::nvvm_sqrt_rn_f:
1994 case Intrinsic::nvvm_sqrt_rn_ftz_f:
1995 return !IsStrictFP;
1996
1997 // NVVM add intrinsics with explicit rounding modes
1998 case Intrinsic::nvvm_add_rm_d:
1999 case Intrinsic::nvvm_add_rn_d:
2000 case Intrinsic::nvvm_add_rp_d:
2001 case Intrinsic::nvvm_add_rz_d:
2002 case Intrinsic::nvvm_add_rm_f:
2003 case Intrinsic::nvvm_add_rn_f:
2004 case Intrinsic::nvvm_add_rp_f:
2005 case Intrinsic::nvvm_add_rz_f:
2006 case Intrinsic::nvvm_add_rm_ftz_f:
2007 case Intrinsic::nvvm_add_rn_ftz_f:
2008 case Intrinsic::nvvm_add_rp_ftz_f:
2009 case Intrinsic::nvvm_add_rz_ftz_f:
2010
2011 // NVVM div intrinsics with explicit rounding modes
2012 case Intrinsic::nvvm_div_rm_d:
2013 case Intrinsic::nvvm_div_rn_d:
2014 case Intrinsic::nvvm_div_rp_d:
2015 case Intrinsic::nvvm_div_rz_d:
2016 case Intrinsic::nvvm_div_rm_f:
2017 case Intrinsic::nvvm_div_rn_f:
2018 case Intrinsic::nvvm_div_rp_f:
2019 case Intrinsic::nvvm_div_rz_f:
2020 case Intrinsic::nvvm_div_rm_ftz_f:
2021 case Intrinsic::nvvm_div_rn_ftz_f:
2022 case Intrinsic::nvvm_div_rp_ftz_f:
2023 case Intrinsic::nvvm_div_rz_ftz_f:
2024
2025 // NVVM mul intrinsics with explicit rounding modes
2026 case Intrinsic::nvvm_mul_rm_d:
2027 case Intrinsic::nvvm_mul_rn_d:
2028 case Intrinsic::nvvm_mul_rp_d:
2029 case Intrinsic::nvvm_mul_rz_d:
2030 case Intrinsic::nvvm_mul_rm_f:
2031 case Intrinsic::nvvm_mul_rn_f:
2032 case Intrinsic::nvvm_mul_rp_f:
2033 case Intrinsic::nvvm_mul_rz_f:
2034 case Intrinsic::nvvm_mul_rm_ftz_f:
2035 case Intrinsic::nvvm_mul_rn_ftz_f:
2036 case Intrinsic::nvvm_mul_rp_ftz_f:
2037 case Intrinsic::nvvm_mul_rz_ftz_f:
2038
2039 // NVVM fma intrinsics with explicit rounding modes
2040 case Intrinsic::nvvm_fma_rm_d:
2041 case Intrinsic::nvvm_fma_rn_d:
2042 case Intrinsic::nvvm_fma_rp_d:
2043 case Intrinsic::nvvm_fma_rz_d:
2044 case Intrinsic::nvvm_fma_rm_f:
2045 case Intrinsic::nvvm_fma_rn_f:
2046 case Intrinsic::nvvm_fma_rp_f:
2047 case Intrinsic::nvvm_fma_rz_f:
2048 case Intrinsic::nvvm_fma_rm_ftz_f:
2049 case Intrinsic::nvvm_fma_rn_ftz_f:
2050 case Intrinsic::nvvm_fma_rp_ftz_f:
2051 case Intrinsic::nvvm_fma_rz_ftz_f:
2052
2053 // Sign operations are actually bitwise operations, they do not raise
2054 // exceptions even for SNANs.
2055 case Intrinsic::fabs:
2056 case Intrinsic::copysign:
2057 case Intrinsic::is_fpclass:
2058 // Non-constrained variants of rounding operations means default FP
2059 // environment, they can be folded in any case.
2060 case Intrinsic::ceil:
2061 case Intrinsic::floor:
2062 case Intrinsic::round:
2063 case Intrinsic::roundeven:
2064 case Intrinsic::trunc:
2065 case Intrinsic::nearbyint:
2066 case Intrinsic::rint:
2067 case Intrinsic::canonicalize:
2068
2069 // Constrained intrinsics can be folded if FP environment is known
2070 // to compiler.
2071 case Intrinsic::experimental_constrained_fma:
2072 case Intrinsic::experimental_constrained_fmuladd:
2073 case Intrinsic::experimental_constrained_fadd:
2074 case Intrinsic::experimental_constrained_fsub:
2075 case Intrinsic::experimental_constrained_fmul:
2076 case Intrinsic::experimental_constrained_fdiv:
2077 case Intrinsic::experimental_constrained_frem:
2078 case Intrinsic::experimental_constrained_ceil:
2079 case Intrinsic::experimental_constrained_floor:
2080 case Intrinsic::experimental_constrained_round:
2081 case Intrinsic::experimental_constrained_roundeven:
2082 case Intrinsic::experimental_constrained_trunc:
2083 case Intrinsic::experimental_constrained_nearbyint:
2084 case Intrinsic::experimental_constrained_rint:
2085 case Intrinsic::experimental_constrained_fcmp:
2086 case Intrinsic::experimental_constrained_fcmps:
2087
2088 case Intrinsic::experimental_cttz_elts:
2089 return true;
2090 default:
2091 return false;
2092 }
2093}
2094
2095/// Given a function's return type and its operands, determine if any of them of
2096/// of floating-point type.
2098 return RetTy->isFloatingPointTy() || any_of(Ops, [](Value *V) {
2099 return V->getType()->isFloatingPointTy();
2100 });
2101}
2102
2104 if (Call->isNoBuiltin())
2105 return false;
2106 if (Call->getFunctionType() != F->getFunctionType())
2107 return false;
2108
2109 // Allow FP calls (both libcalls and intrinsics) to avoid being folded.
2110 // This can be useful for GPU targets or in cross-compilation scenarios
2111 // when the exact target FP behaviour is required, and the host compiler's
2112 // behaviour may be slightly different from the device's run-time behaviour.
2115 F->getReturnType(),
2116 ArrayRef<Value *>((Value *const *)(F->arg_begin()), F->arg_size())))
2117 return false;
2118
2119 if (F->getIntrinsicID() != Intrinsic::not_intrinsic)
2120 return canConstantFoldIntrinsic(F->getIntrinsicID(), Call->isStrictFP());
2121
2122 if (!F->hasName() || Call->isStrictFP())
2123 return false;
2124
2125 // In these cases, the check of the length is required. We don't want to
2126 // return true for a name like "cos\0blah" which strcmp would return equal to
2127 // "cos", but has length 8.
2128 StringRef Name = F->getName();
2129 switch (Name[0]) {
2130 default:
2131 return false;
2132 // clang-format off
2133 case 'a':
2134 return Name == "acos" || Name == "acosf" ||
2135 Name == "asin" || Name == "asinf" ||
2136 Name == "atan" || Name == "atanf" ||
2137 Name == "atan2" || Name == "atan2f";
2138 case 'c':
2139 return Name == "ceil" || Name == "ceilf" ||
2140 Name == "cos" || Name == "cosf" ||
2141 Name == "cosh" || Name == "coshf";
2142 case 'e':
2143 return Name == "exp" || Name == "expf" || Name == "exp2" ||
2144 Name == "exp2f" || Name == "erf" || Name == "erff";
2145 case 'f':
2146 return Name == "fabs" || Name == "fabsf" ||
2147 Name == "floor" || Name == "floorf" ||
2148 Name == "fmod" || Name == "fmodf";
2149 case 'i':
2150 return Name == "ilogb" || Name == "ilogbf";
2151 case 'l':
2152 return Name == "log" || Name == "logf" || Name == "logl" ||
2153 Name == "log2" || Name == "log2f" || Name == "log10" ||
2154 Name == "log10f" || Name == "logb" || Name == "logbf" ||
2155 Name == "log1p" || Name == "log1pf";
2156 case 'n':
2157 return Name == "nearbyint" || Name == "nearbyintf" || Name == "nextafter" ||
2158 Name == "nextafterf" || Name == "nexttoward" ||
2159 Name == "nexttowardf";
2160 case 'p':
2161 return Name == "pow" || Name == "powf";
2162 case 'r':
2163 return Name == "remainder" || Name == "remainderf" ||
2164 Name == "rint" || Name == "rintf" ||
2165 Name == "round" || Name == "roundf" ||
2166 Name == "roundeven" || Name == "roundevenf";
2167 case 's':
2168 return Name == "sin" || Name == "sinf" ||
2169 Name == "sinh" || Name == "sinhf" ||
2170 Name == "sqrt" || Name == "sqrtf";
2171 case 't':
2172 return Name == "tan" || Name == "tanf" ||
2173 Name == "tanh" || Name == "tanhf" ||
2174 Name == "trunc" || Name == "truncf";
2175 case '_':
2176 // Check for various function names that get used for the math functions
2177 // when the header files are preprocessed with the macro
2178 // __FINITE_MATH_ONLY__ enabled.
2179 // The '12' here is the length of the shortest name that can match.
2180 // We need to check the size before looking at Name[1] and Name[2]
2181 // so we may as well check a limit that will eliminate mismatches.
2182 if (Name.size() < 12 || Name[1] != '_')
2183 return false;
2184 switch (Name[2]) {
2185 default:
2186 return false;
2187 case 'a':
2188 return Name == "__acos_finite" || Name == "__acosf_finite" ||
2189 Name == "__asin_finite" || Name == "__asinf_finite" ||
2190 Name == "__atan2_finite" || Name == "__atan2f_finite";
2191 case 'c':
2192 return Name == "__cosh_finite" || Name == "__coshf_finite";
2193 case 'e':
2194 return Name == "__exp_finite" || Name == "__expf_finite" ||
2195 Name == "__exp2_finite" || Name == "__exp2f_finite";
2196 case 'l':
2197 return Name == "__log_finite" || Name == "__logf_finite" ||
2198 Name == "__log10_finite" || Name == "__log10f_finite";
2199 case 'p':
2200 return Name == "__pow_finite" || Name == "__powf_finite";
2201 case 's':
2202 return Name == "__sinh_finite" || Name == "__sinhf_finite";
2203 }
2204 // clang-format on
2205 }
2206}
2207
2208namespace {
2209
2210Constant *GetConstantFoldFPValue(double V, Type *Ty) {
2211 if (Ty->isHalfTy() || Ty->isFloatTy() || Ty->isBFloatTy()) {
2212 APFloat APF(V);
2213 bool unused;
2214 APF.convert(Ty->getFltSemantics(), APFloat::rmNearestTiesToEven, &unused);
2215 return ConstantFP::get(Ty->getContext(), APF);
2216 }
2217 if (Ty->isDoubleTy())
2218 return ConstantFP::get(Ty->getContext(), APFloat(V));
2219 llvm_unreachable("Can only constant fold half/float/double/bfloat");
2220}
2221
2222#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2223Constant *GetConstantFoldFPValue128(float128 V, Type *Ty) {
2224 if (Ty->isFP128Ty())
2225 return ConstantFP::get(Ty, V);
2226 llvm_unreachable("Can only constant fold fp128");
2227}
2228#endif
2229
2230/// Clear the floating-point exception state.
2231inline void llvm_fenv_clearexcept() {
2232#if HAVE_DECL_FE_ALL_EXCEPT
2233 feclearexcept(FE_ALL_EXCEPT);
2234#endif
2235 errno = 0;
2236}
2237
2238/// Test if a floating-point exception was raised.
2239inline bool llvm_fenv_testexcept() {
2240 int errno_val = errno;
2241 if (errno_val == ERANGE || errno_val == EDOM)
2242 return true;
2243#if HAVE_DECL_FE_ALL_EXCEPT && HAVE_DECL_FE_INEXACT
2244 if (fetestexcept(FE_ALL_EXCEPT & ~FE_INEXACT))
2245 return true;
2246#endif
2247 return false;
2248}
2249
2250static APFloat FTZPreserveSign(const APFloat &V) {
2251 if (V.isDenormal())
2252 return APFloat::getZero(V.getSemantics(), V.isNegative());
2253 return V;
2254}
2255
2256static APFloat FlushToPositiveZero(const APFloat &V) {
2257 if (V.isDenormal())
2258 return APFloat::getZero(V.getSemantics(), false);
2259 return V;
2260}
2261
2262static APFloat FlushWithDenormKind(const APFloat &V,
2263 DenormalMode::DenormalModeKind DenormKind) {
2266 switch (DenormKind) {
2268 return V;
2270 return FTZPreserveSign(V);
2272 return FlushToPositiveZero(V);
2273 default:
2274 llvm_unreachable("Invalid denormal mode!");
2275 }
2276}
2277
2278Constant *ConstantFoldFP(double (*NativeFP)(double), const APFloat &V, Type *Ty,
2279 DenormalMode DenormMode = DenormalMode::getIEEE()) {
2280 if (!DenormMode.isValid() ||
2281 DenormMode.Input == DenormalMode::DenormalModeKind::Dynamic ||
2282 DenormMode.Output == DenormalMode::DenormalModeKind::Dynamic)
2283 return nullptr;
2284
2285 llvm_fenv_clearexcept();
2286 auto Input = FlushWithDenormKind(V, DenormMode.Input);
2287 double Result = NativeFP(Input.convertToDouble());
2288 if (llvm_fenv_testexcept()) {
2289 llvm_fenv_clearexcept();
2290 return nullptr;
2291 }
2292
2293 Constant *Output = GetConstantFoldFPValue(Result, Ty);
2294 if (DenormMode.Output == DenormalMode::DenormalModeKind::IEEE)
2295 return Output;
2296 const auto *CFP = static_cast<ConstantFP *>(Output);
2297 const auto Res = FlushWithDenormKind(CFP->getValueAPF(), DenormMode.Output);
2298 return ConstantFP::get(Ty->getContext(), Res);
2299}
2300
2301#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2302Constant *ConstantFoldFP128(float128 (*NativeFP)(float128), const APFloat &V,
2303 Type *Ty) {
2304 llvm_fenv_clearexcept();
2305 float128 Result = NativeFP(V.convertToQuad());
2306 if (llvm_fenv_testexcept()) {
2307 llvm_fenv_clearexcept();
2308 return nullptr;
2309 }
2310
2311 return GetConstantFoldFPValue128(Result, Ty);
2312}
2313#endif
2314
2315Constant *ConstantFoldBinaryFP(double (*NativeFP)(double, double),
2316 const APFloat &V, const APFloat &W, Type *Ty) {
2317 llvm_fenv_clearexcept();
2318 double Result = NativeFP(V.convertToDouble(), W.convertToDouble());
2319 if (llvm_fenv_testexcept()) {
2320 llvm_fenv_clearexcept();
2321 return nullptr;
2322 }
2323
2324 return GetConstantFoldFPValue(Result, Ty);
2325}
2326
2327Constant *constantFoldVectorReduce(Intrinsic::ID IID, Constant *Op) {
2328 auto *OpVT = cast<VectorType>(Op->getType());
2329
2330 // This is the same as the underlying binops - poison propagates.
2331 if (Op->containsPoisonElement())
2332 return PoisonValue::get(OpVT->getElementType());
2333
2334 // Shortcut non-accumulating reductions.
2335 if (Constant *SplatVal = Op->getSplatValue()) {
2336 switch (IID) {
2337 case Intrinsic::vector_reduce_and:
2338 case Intrinsic::vector_reduce_or:
2339 case Intrinsic::vector_reduce_smin:
2340 case Intrinsic::vector_reduce_smax:
2341 case Intrinsic::vector_reduce_umin:
2342 case Intrinsic::vector_reduce_umax:
2343 return SplatVal;
2344 case Intrinsic::vector_reduce_add:
2345 if (SplatVal->isNullValue())
2346 return SplatVal;
2347 break;
2348 case Intrinsic::vector_reduce_mul:
2349 if (SplatVal->isNullValue() || SplatVal->isOneValue())
2350 return SplatVal;
2351 break;
2352 case Intrinsic::vector_reduce_xor:
2353 if (SplatVal->isNullValue())
2354 return SplatVal;
2355 if (OpVT->getElementCount().isKnownMultipleOf(2))
2356 return Constant::getNullValue(OpVT->getElementType());
2357 break;
2358 }
2359 }
2360
2362 if (!VT)
2363 return nullptr;
2364
2365 auto *EltC = dyn_cast_or_null<ConstantInt>(Op->getAggregateElement(0U));
2366 if (!EltC)
2367 return nullptr;
2368
2369 APInt Acc = EltC->getValue();
2370 for (unsigned I = 1, E = VT->getNumElements(); I != E; I++) {
2371 if (!(EltC = dyn_cast_or_null<ConstantInt>(Op->getAggregateElement(I))))
2372 return nullptr;
2373 const APInt &X = EltC->getValue();
2374 switch (IID) {
2375 case Intrinsic::vector_reduce_add:
2376 Acc = Acc + X;
2377 break;
2378 case Intrinsic::vector_reduce_mul:
2379 Acc = Acc * X;
2380 break;
2381 case Intrinsic::vector_reduce_and:
2382 Acc = Acc & X;
2383 break;
2384 case Intrinsic::vector_reduce_or:
2385 Acc = Acc | X;
2386 break;
2387 case Intrinsic::vector_reduce_xor:
2388 Acc = Acc ^ X;
2389 break;
2390 case Intrinsic::vector_reduce_smin:
2391 Acc = APIntOps::smin(Acc, X);
2392 break;
2393 case Intrinsic::vector_reduce_smax:
2394 Acc = APIntOps::smax(Acc, X);
2395 break;
2396 case Intrinsic::vector_reduce_umin:
2397 Acc = APIntOps::umin(Acc, X);
2398 break;
2399 case Intrinsic::vector_reduce_umax:
2400 Acc = APIntOps::umax(Acc, X);
2401 break;
2402 }
2403 }
2404
2405 return ConstantInt::get(Op->getContext(), Acc);
2406}
2407
2408/// Attempt to fold an SSE floating point to integer conversion of a constant
2409/// floating point. If roundTowardZero is false, the default IEEE rounding is
2410/// used (toward nearest, ties to even). This matches the behavior of the
2411/// non-truncating SSE instructions in the default rounding mode. The desired
2412/// integer type Ty is used to select how many bits are available for the
2413/// result. Returns null if the conversion cannot be performed, otherwise
2414/// returns the Constant value resulting from the conversion.
2415Constant *ConstantFoldSSEConvertToInt(const APFloat &Val, bool roundTowardZero,
2416 Type *Ty, bool IsSigned) {
2417 // All of these conversion intrinsics form an integer of at most 64bits.
2418 unsigned ResultWidth = Ty->getIntegerBitWidth();
2419 assert(ResultWidth <= 64 &&
2420 "Can only constant fold conversions to 64 and 32 bit ints");
2421
2422 uint64_t UIntVal;
2423 bool isExact = false;
2427 Val.convertToInteger(MutableArrayRef(UIntVal), ResultWidth,
2428 IsSigned, mode, &isExact);
2429 if (status != APFloat::opOK &&
2430 (!roundTowardZero || status != APFloat::opInexact))
2431 return nullptr;
2432 return ConstantInt::get(Ty, UIntVal, IsSigned);
2433}
2434
2435double getValueAsDouble(ConstantFP *Op) {
2436 Type *Ty = Op->getType();
2437
2438 if (Ty->isBFloatTy() || Ty->isHalfTy() || Ty->isFloatTy() || Ty->isDoubleTy())
2439 return Op->getValueAPF().convertToDouble();
2440
2441 bool unused;
2442 APFloat APF = Op->getValueAPF();
2444 return APF.convertToDouble();
2445}
2446
2447static bool getConstIntOrUndef(Value *Op, const APInt *&C) {
2448 if (auto *CI = dyn_cast<ConstantInt>(Op)) {
2449 C = &CI->getValue();
2450 return true;
2451 }
2452 if (isa<UndefValue>(Op)) {
2453 C = nullptr;
2454 return true;
2455 }
2456 return false;
2457}
2458
2459/// Checks if the given intrinsic call, which evaluates to constant, is allowed
2460/// to be folded.
2461///
2462/// \param CI Constrained intrinsic call.
2463/// \param St Exception flags raised during constant evaluation.
2464static bool mayFoldConstrained(ConstrainedFPIntrinsic *CI,
2465 APFloat::opStatus St) {
2466 std::optional<RoundingMode> ORM = CI->getRoundingMode();
2467 std::optional<fp::ExceptionBehavior> EB = CI->getExceptionBehavior();
2468
2469 // If the operation does not change exception status flags, it is safe
2470 // to fold.
2471 if (St == APFloat::opStatus::opOK)
2472 return true;
2473
2474 // If evaluation raised FP exception, the result can depend on rounding
2475 // mode. If the latter is unknown, folding is not possible.
2476 if (ORM == RoundingMode::Dynamic)
2477 return false;
2478
2479 // If FP exceptions are ignored, fold the call, even if such exception is
2480 // raised.
2481 if (EB && *EB != fp::ExceptionBehavior::ebStrict)
2482 return true;
2483
2484 // Leave the calculation for runtime so that exception flags be correctly set
2485 // in hardware.
2486 return false;
2487}
2488
2489/// Returns the rounding mode that should be used for constant evaluation.
2490static RoundingMode
2491getEvaluationRoundingMode(const ConstrainedFPIntrinsic *CI) {
2492 std::optional<RoundingMode> ORM = CI->getRoundingMode();
2493 if (!ORM || *ORM == RoundingMode::Dynamic)
2494 // Even if the rounding mode is unknown, try evaluating the operation.
2495 // If it does not raise inexact exception, rounding was not applied,
2496 // so the result is exact and does not depend on rounding mode. Whether
2497 // other FP exceptions are raised, it does not depend on rounding mode.
2499 return *ORM;
2500}
2501
2502/// Try to constant fold llvm.canonicalize for the given caller and value.
2503static Constant *constantFoldCanonicalize(const Type *Ty, const APFloat &Src,
2504 const Function *CtxF = nullptr) {
2505 // Zero, positive and negative, is always OK to fold.
2506 if (Src.isZero()) {
2507 // Get a fresh 0, since ppc_fp128 does have non-canonical zeros.
2508 return ConstantFP::get(
2509 Ty->getContext(),
2510 APFloat::getZero(Src.getSemantics(), Src.isNegative()));
2511 }
2512
2513 if (!Ty->isIEEELikeFPTy())
2514 return nullptr;
2515
2516 // Zero is always canonical and the sign must be preserved.
2517 //
2518 // Denorms and nans may have special encodings, but it should be OK to fold a
2519 // totally average number.
2520 if (Src.isNormal() || Src.isInfinity())
2521 return ConstantFP::get(Ty->getContext(), Src);
2522
2523 if (Src.isDenormal() && CtxF) {
2524 DenormalMode DenormMode = CtxF->getDenormalMode(Src.getSemantics());
2525
2526 if (DenormMode == DenormalMode::getIEEE())
2527 return ConstantFP::get(Ty->getContext(), Src);
2528
2529 if (DenormMode.Input == DenormalMode::Dynamic)
2530 return nullptr;
2531
2532 // If we know if either input or output is flushed, we can fold.
2533 if ((DenormMode.Input == DenormalMode::Dynamic &&
2534 DenormMode.Output == DenormalMode::IEEE) ||
2535 (DenormMode.Input == DenormalMode::IEEE &&
2536 DenormMode.Output == DenormalMode::Dynamic))
2537 return nullptr;
2538
2539 bool IsPositive =
2540 (!Src.isNegative() || DenormMode.Input == DenormalMode::PositiveZero ||
2541 (DenormMode.Output == DenormalMode::PositiveZero &&
2542 DenormMode.Input == DenormalMode::IEEE));
2543
2544 return ConstantFP::get(Ty->getContext(),
2545 APFloat::getZero(Src.getSemantics(), !IsPositive));
2546 }
2547
2548 return nullptr;
2549}
2550
2551static Constant *ConstantFoldScalarCall1(StringRef Name,
2552 Intrinsic::ID IntrinsicID, Type *Ty,
2554 const TargetLibraryInfo *TLI = nullptr,
2555 const CallBase *Call = nullptr) {
2556 assert(Operands.size() == 1 && "Wrong number of operands.");
2557
2558 if (IntrinsicID == Intrinsic::is_constant) {
2559 // We know we have a "Constant" argument. But we want to only
2560 // return true for manifest constants, not those that depend on
2561 // constants with unknowable values, e.g. GlobalValue or BlockAddress.
2562 if (Operands[0]->isManifestConstant())
2563 return ConstantInt::getTrue(Ty->getContext());
2564 return nullptr;
2565 }
2566
2567 if (isa<UndefValue>(Operands[0])) {
2568 // cosine(arg) is between -1 and 1. cosine(invalid arg) is NaN.
2569 // ctpop() is between 0 and bitwidth, pick 0 for undef.
2570 // fptoui.sat and fptosi.sat can always fold to zero (for a zero input).
2571 if (IntrinsicID == Intrinsic::cos ||
2572 IntrinsicID == Intrinsic::ctpop ||
2573 IntrinsicID == Intrinsic::fptoui_sat ||
2574 IntrinsicID == Intrinsic::fptosi_sat ||
2575 IntrinsicID == Intrinsic::canonicalize)
2576 return Constant::getNullValue(Ty);
2577 if (IntrinsicID == Intrinsic::bswap ||
2578 IntrinsicID == Intrinsic::bitreverse ||
2579 IntrinsicID == Intrinsic::launder_invariant_group ||
2580 IntrinsicID == Intrinsic::strip_invariant_group)
2581 return Operands[0];
2582 }
2583
2585 // launder(null) == null == strip(null) iff in addrspace 0
2586 if (IntrinsicID == Intrinsic::launder_invariant_group ||
2587 IntrinsicID == Intrinsic::strip_invariant_group) {
2588 // If instruction is not yet put in a basic block (e.g. when cloning
2589 // a function during inlining), Call's caller may not be available.
2590 // So check Call's BB first before querying Call->getCaller.
2591 const Function *Caller =
2592 Call && Call->getParent() ? Call->getCaller() : nullptr;
2593 if (Caller &&
2595 Caller, Operands[0]->getType()->getPointerAddressSpace())) {
2596 return Operands[0];
2597 }
2598 return nullptr;
2599 }
2600 }
2601
2602 if (auto *Op = dyn_cast<ConstantFP>(Operands[0])) {
2603 APFloat U = Op->getValueAPF();
2604
2605 if (IntrinsicID == Intrinsic::wasm_trunc_signed ||
2606 IntrinsicID == Intrinsic::wasm_trunc_unsigned) {
2607 bool Signed = IntrinsicID == Intrinsic::wasm_trunc_signed;
2608
2609 if (U.isNaN())
2610 return nullptr;
2611
2612 unsigned Width = Ty->getIntegerBitWidth();
2613 APSInt Int(Width, !Signed);
2614 bool IsExact = false;
2616 U.convertToInteger(Int, APFloat::rmTowardZero, &IsExact);
2617
2619 return ConstantInt::get(Ty, Int);
2620
2621 return nullptr;
2622 }
2623
2624 if (IntrinsicID == Intrinsic::fptoui_sat ||
2625 IntrinsicID == Intrinsic::fptosi_sat) {
2626 // convertToInteger() already has the desired saturation semantics.
2627 APSInt Int(Ty->getIntegerBitWidth(),
2628 IntrinsicID == Intrinsic::fptoui_sat);
2629 bool IsExact;
2630 U.convertToInteger(Int, APFloat::rmTowardZero, &IsExact);
2631 return ConstantInt::get(Ty, Int);
2632 }
2633
2634 if (IntrinsicID == Intrinsic::canonicalize) {
2635 const Function *CtxF =
2636 Call && Call->getParent() ? Call->getFunction() : nullptr;
2637 return constantFoldCanonicalize(Ty, U, CtxF);
2638 }
2639
2640#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2641 if (Ty->isFP128Ty()) {
2642 if (IntrinsicID == Intrinsic::log) {
2643 float128 Result = logf128(Op->getValueAPF().convertToQuad());
2644 return GetConstantFoldFPValue128(Result, Ty);
2645 }
2646
2647 if (TLI && TLI->getLibFunc(Name) == LibFunc_logl &&
2648 TLI->has(LibFunc_logl))
2649 return ConstantFoldFP128(logf128, Op->getValueAPF(), Ty);
2650 }
2651#endif
2652
2653 if (!Ty->isHalfTy() && !Ty->isFloatTy() && !Ty->isDoubleTy() &&
2654 !Ty->isIntegerTy() && !Ty->isBFloatTy())
2655 return nullptr;
2656
2657 // Use internal versions of these intrinsics.
2658
2659 if (IntrinsicID == Intrinsic::nearbyint || IntrinsicID == Intrinsic::rint ||
2660 IntrinsicID == Intrinsic::roundeven) {
2661 U.roundToIntegral(APFloat::rmNearestTiesToEven);
2662 return ConstantFP::get(Ty, U);
2663 }
2664
2665 if (IntrinsicID == Intrinsic::round) {
2666 U.roundToIntegral(APFloat::rmNearestTiesToAway);
2667 return ConstantFP::get(Ty, U);
2668 }
2669
2670 if (IntrinsicID == Intrinsic::roundeven) {
2671 U.roundToIntegral(APFloat::rmNearestTiesToEven);
2672 return ConstantFP::get(Ty, U);
2673 }
2674
2675 if (IntrinsicID == Intrinsic::ceil) {
2676 U.roundToIntegral(APFloat::rmTowardPositive);
2677 return ConstantFP::get(Ty, U);
2678 }
2679
2680 if (IntrinsicID == Intrinsic::floor) {
2681 U.roundToIntegral(APFloat::rmTowardNegative);
2682 return ConstantFP::get(Ty, U);
2683 }
2684
2685 if (IntrinsicID == Intrinsic::trunc) {
2686 U.roundToIntegral(APFloat::rmTowardZero);
2687 return ConstantFP::get(Ty, U);
2688 }
2689
2690 if (IntrinsicID == Intrinsic::fabs) {
2691 U.clearSign();
2692 return ConstantFP::get(Ty, U);
2693 }
2694
2695 if (IntrinsicID == Intrinsic::amdgcn_fract) {
2696 // The v_fract instruction behaves like the OpenCL spec, which defines
2697 // fract(x) as fmin(x - floor(x), 0x1.fffffep-1f): "The min() operator is
2698 // there to prevent fract(-small) from returning 1.0. It returns the
2699 // largest positive floating-point number less than 1.0."
2700 APFloat FloorU(U);
2701 FloorU.roundToIntegral(APFloat::rmTowardNegative);
2702 APFloat FractU(U - FloorU);
2703 APFloat AlmostOne(U.getSemantics(), 1);
2704 AlmostOne.next(/*nextDown*/ true);
2705 return ConstantFP::get(Ty, minimum(FractU, AlmostOne));
2706 }
2707
2708 // Rounding operations (floor, trunc, ceil, round and nearbyint) do not
2709 // raise FP exceptions, unless the argument is signaling NaN.
2710
2712 std::optional<APFloat::roundingMode> RM;
2713 switch (IntrinsicID) {
2714 default:
2715 break;
2716 case Intrinsic::experimental_constrained_nearbyint:
2717 case Intrinsic::experimental_constrained_rint: {
2718 RM = CI->getRoundingMode();
2719 if (!RM || *RM == RoundingMode::Dynamic)
2720 return nullptr;
2721 break;
2722 }
2723 case Intrinsic::experimental_constrained_round:
2725 break;
2726 case Intrinsic::experimental_constrained_ceil:
2728 break;
2729 case Intrinsic::experimental_constrained_floor:
2731 break;
2732 case Intrinsic::experimental_constrained_trunc:
2734 break;
2735 }
2736 if (RM) {
2737 if (U.isFinite()) {
2738 APFloat::opStatus St = U.roundToIntegral(*RM);
2739 if (IntrinsicID == Intrinsic::experimental_constrained_rint &&
2740 St == APFloat::opInexact) {
2741 std::optional<fp::ExceptionBehavior> EB =
2743 if (EB == fp::ebStrict)
2744 return nullptr;
2745 }
2746 } else if (U.isSignaling()) {
2747 std::optional<fp::ExceptionBehavior> EB = CI->getExceptionBehavior();
2748 if (EB && *EB != fp::ebIgnore)
2749 return nullptr;
2750 U = APFloat::getQNaN(U.getSemantics());
2751 }
2752 return ConstantFP::get(Ty, U);
2753 }
2754 }
2755
2756 // NVVM float/double to signed/unsigned int32/int64 conversions:
2757 switch (IntrinsicID) {
2758 // f2i
2759 case Intrinsic::nvvm_f2i_rm:
2760 case Intrinsic::nvvm_f2i_rn:
2761 case Intrinsic::nvvm_f2i_rp:
2762 case Intrinsic::nvvm_f2i_rz:
2763 case Intrinsic::nvvm_f2i_rm_ftz:
2764 case Intrinsic::nvvm_f2i_rn_ftz:
2765 case Intrinsic::nvvm_f2i_rp_ftz:
2766 case Intrinsic::nvvm_f2i_rz_ftz:
2767 // f2ui
2768 case Intrinsic::nvvm_f2ui_rm:
2769 case Intrinsic::nvvm_f2ui_rn:
2770 case Intrinsic::nvvm_f2ui_rp:
2771 case Intrinsic::nvvm_f2ui_rz:
2772 case Intrinsic::nvvm_f2ui_rm_ftz:
2773 case Intrinsic::nvvm_f2ui_rn_ftz:
2774 case Intrinsic::nvvm_f2ui_rp_ftz:
2775 case Intrinsic::nvvm_f2ui_rz_ftz:
2776 // d2i
2777 case Intrinsic::nvvm_d2i_rm:
2778 case Intrinsic::nvvm_d2i_rn:
2779 case Intrinsic::nvvm_d2i_rp:
2780 case Intrinsic::nvvm_d2i_rz:
2781 // d2ui
2782 case Intrinsic::nvvm_d2ui_rm:
2783 case Intrinsic::nvvm_d2ui_rn:
2784 case Intrinsic::nvvm_d2ui_rp:
2785 case Intrinsic::nvvm_d2ui_rz:
2786 // f2ll
2787 case Intrinsic::nvvm_f2ll_rm:
2788 case Intrinsic::nvvm_f2ll_rn:
2789 case Intrinsic::nvvm_f2ll_rp:
2790 case Intrinsic::nvvm_f2ll_rz:
2791 case Intrinsic::nvvm_f2ll_rm_ftz:
2792 case Intrinsic::nvvm_f2ll_rn_ftz:
2793 case Intrinsic::nvvm_f2ll_rp_ftz:
2794 case Intrinsic::nvvm_f2ll_rz_ftz:
2795 // f2ull
2796 case Intrinsic::nvvm_f2ull_rm:
2797 case Intrinsic::nvvm_f2ull_rn:
2798 case Intrinsic::nvvm_f2ull_rp:
2799 case Intrinsic::nvvm_f2ull_rz:
2800 case Intrinsic::nvvm_f2ull_rm_ftz:
2801 case Intrinsic::nvvm_f2ull_rn_ftz:
2802 case Intrinsic::nvvm_f2ull_rp_ftz:
2803 case Intrinsic::nvvm_f2ull_rz_ftz:
2804 // d2ll
2805 case Intrinsic::nvvm_d2ll_rm:
2806 case Intrinsic::nvvm_d2ll_rn:
2807 case Intrinsic::nvvm_d2ll_rp:
2808 case Intrinsic::nvvm_d2ll_rz:
2809 // d2ull
2810 case Intrinsic::nvvm_d2ull_rm:
2811 case Intrinsic::nvvm_d2ull_rn:
2812 case Intrinsic::nvvm_d2ull_rp:
2813 case Intrinsic::nvvm_d2ull_rz: {
2814 // In float-to-integer conversion, NaN inputs are converted to 0.
2815 if (U.isNaN()) {
2816 // In float-to-integer conversion, NaN inputs are converted to 0
2817 // when the source and destination bitwidths are both less than 64.
2818 if (nvvm::FPToIntegerIntrinsicNaNZero(IntrinsicID))
2819 return ConstantInt::get(Ty, 0);
2820
2821 // Otherwise, the most significant bit is set.
2822 unsigned BitWidth = Ty->getIntegerBitWidth();
2823 uint64_t Val = 1ULL << (BitWidth - 1);
2824 return ConstantInt::get(Ty, APInt(BitWidth, Val, /*IsSigned=*/false));
2825 }
2826
2827 APFloat::roundingMode RMode =
2829 bool IsFTZ = nvvm::FPToIntegerIntrinsicShouldFTZ(IntrinsicID);
2830 bool IsSigned = nvvm::FPToIntegerIntrinsicResultIsSigned(IntrinsicID);
2831
2832 APSInt ResInt(Ty->getIntegerBitWidth(), !IsSigned);
2833 auto FloatToRound = IsFTZ ? FTZPreserveSign(U) : U;
2834
2835 // Return max/min value for integers if the result is +/-inf or
2836 // is too large to fit in the result's integer bitwidth.
2837 bool IsExact = false;
2838 FloatToRound.convertToInteger(ResInt, RMode, &IsExact);
2839 return ConstantInt::get(Ty, ResInt);
2840 }
2841 }
2842
2843 /// We only fold functions with finite arguments. Folding NaN and inf is
2844 /// likely to be aborted with an exception anyway, and some host libms
2845 /// have known errors raising exceptions.
2846 if (!U.isFinite())
2847 return nullptr;
2848
2849 /// Currently APFloat versions of these functions do not exist, so we use
2850 /// the host native double versions. Float versions are not called
2851 /// directly but for all these it is true (float)(f((double)arg)) ==
2852 /// f(arg). Long double not supported yet.
2853 const APFloat &APF = Op->getValueAPF();
2854
2855 switch (IntrinsicID) {
2856 default: break;
2857 case Intrinsic::log:
2858 if (U.isZero())
2859 return ConstantFP::getInfinity(Ty, true);
2860 if (U.isNegative())
2861 return ConstantFP::getNaN(Ty);
2862 if (U.isOne())
2863 return ConstantFP::getZero(Ty);
2864 return ConstantFoldFP(log, APF, Ty);
2865 case Intrinsic::log2:
2866 if (U.isZero())
2867 return ConstantFP::getInfinity(Ty, true);
2868 if (U.isNegative())
2869 return ConstantFP::getNaN(Ty);
2870 if (U.isOne())
2871 return ConstantFP::getZero(Ty);
2872 // TODO: What about hosts that lack a C99 library?
2873 return ConstantFoldFP(log2, APF, Ty);
2874 case Intrinsic::log10:
2875 if (U.isZero())
2876 return ConstantFP::getInfinity(Ty, true);
2877 if (U.isNegative())
2878 return ConstantFP::getNaN(Ty);
2879 if (U.isOne())
2880 return ConstantFP::getZero(Ty);
2881 // TODO: What about hosts that lack a C99 library?
2882 return ConstantFoldFP(log10, APF, Ty);
2883 case Intrinsic::exp:
2884 return ConstantFoldFP(exp, APF, Ty);
2885 case Intrinsic::exp2:
2886 // Fold exp2(x) as pow(2, x), in case the host lacks a C99 library.
2887 return ConstantFoldBinaryFP(pow, APFloat(2.0), APF, Ty);
2888 case Intrinsic::exp10:
2889 // Fold exp10(x) as pow(10, x), in case the host lacks a C99 library.
2890 return ConstantFoldBinaryFP(pow, APFloat(10.0), APF, Ty);
2891 case Intrinsic::sin:
2892 return ConstantFoldFP(sin, APF, Ty);
2893 case Intrinsic::cos:
2894 return ConstantFoldFP(cos, APF, Ty);
2895 case Intrinsic::sinh:
2896 return ConstantFoldFP(sinh, APF, Ty);
2897 case Intrinsic::cosh:
2898 return ConstantFoldFP(cosh, APF, Ty);
2899 case Intrinsic::atan:
2900 // Implement optional behavior from C's Annex F for +/-0.0.
2901 if (U.isZero())
2902 return ConstantFP::get(Ty, U);
2903 return ConstantFoldFP(atan, APF, Ty);
2904 case Intrinsic::sqrt:
2905 return ConstantFoldFP(sqrt, APF, Ty);
2906
2907 // NVVM Intrinsics:
2908 case Intrinsic::nvvm_ceil_ftz_f:
2909 case Intrinsic::nvvm_ceil_f:
2910 case Intrinsic::nvvm_ceil_d:
2911 return ConstantFoldFP(
2912 ceil, APF, Ty,
2914 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2915
2916 case Intrinsic::nvvm_fabs_ftz:
2917 case Intrinsic::nvvm_fabs:
2918 return ConstantFoldFP(
2919 fabs, APF, Ty,
2921 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2922
2923 case Intrinsic::nvvm_floor_ftz_f:
2924 case Intrinsic::nvvm_floor_f:
2925 case Intrinsic::nvvm_floor_d:
2926 return ConstantFoldFP(
2927 floor, APF, Ty,
2929 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2930
2931 case Intrinsic::nvvm_rcp_rm_ftz_f:
2932 case Intrinsic::nvvm_rcp_rn_ftz_f:
2933 case Intrinsic::nvvm_rcp_rp_ftz_f:
2934 case Intrinsic::nvvm_rcp_rz_ftz_f:
2935 case Intrinsic::nvvm_rcp_rm_d:
2936 case Intrinsic::nvvm_rcp_rm_f:
2937 case Intrinsic::nvvm_rcp_rn_d:
2938 case Intrinsic::nvvm_rcp_rn_f:
2939 case Intrinsic::nvvm_rcp_rp_d:
2940 case Intrinsic::nvvm_rcp_rp_f:
2941 case Intrinsic::nvvm_rcp_rz_d:
2942 case Intrinsic::nvvm_rcp_rz_f: {
2943 APFloat::roundingMode RoundMode = nvvm::GetRCPRoundingMode(IntrinsicID);
2944 bool IsFTZ = nvvm::RCPShouldFTZ(IntrinsicID);
2945
2946 auto Denominator = IsFTZ ? FTZPreserveSign(APF) : APF;
2948 APFloat::opStatus Status = Res.divide(Denominator, RoundMode);
2949
2951 if (IsFTZ)
2952 Res = FTZPreserveSign(Res);
2953 return ConstantFP::get(Ty, Res);
2954 }
2955 return nullptr;
2956 }
2957
2958 case Intrinsic::nvvm_round_ftz_f:
2959 case Intrinsic::nvvm_round_f:
2960 case Intrinsic::nvvm_round_d: {
2961 // nvvm_round is lowered to PTX cvt.rni, which will round to nearest
2962 // integer, choosing even integer if source is equidistant between two
2963 // integers, so the semantics are closer to "rint" rather than "round".
2964 bool IsFTZ = nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID);
2965 auto V = IsFTZ ? FTZPreserveSign(APF) : APF;
2967 return ConstantFP::get(Ty, V);
2968 }
2969
2970 case Intrinsic::nvvm_saturate_ftz_f:
2971 case Intrinsic::nvvm_saturate_d:
2972 case Intrinsic::nvvm_saturate_f: {
2973 bool IsFTZ = nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID);
2974 auto V = IsFTZ ? FTZPreserveSign(APF) : APF;
2975 if (V.isNegative() || V.isZero() || V.isNaN())
2976 return ConstantFP::getZero(Ty);
2978 if (V > One)
2979 return ConstantFP::get(Ty, One);
2980 return ConstantFP::get(Ty, APF);
2981 }
2982
2983 case Intrinsic::nvvm_sqrt_rn_ftz_f:
2984 case Intrinsic::nvvm_sqrt_f:
2985 case Intrinsic::nvvm_sqrt_rn_d:
2986 case Intrinsic::nvvm_sqrt_rn_f:
2987 if (APF.isNegative())
2988 return nullptr;
2989 return ConstantFoldFP(
2990 sqrt, APF, Ty,
2992 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2993
2994 // AMDGCN Intrinsics:
2995 case Intrinsic::amdgcn_cos:
2996 case Intrinsic::amdgcn_sin: {
2997 double V = getValueAsDouble(Op);
2998 if (V < -256.0 || V > 256.0)
2999 // The gfx8 and gfx9 architectures handle arguments outside the range
3000 // [-256, 256] differently. This should be a rare case so bail out
3001 // rather than trying to handle the difference.
3002 return nullptr;
3003 bool IsCos = IntrinsicID == Intrinsic::amdgcn_cos;
3004 double V4 = V * 4.0;
3005 if (V4 == floor(V4)) {
3006 // Force exact results for quarter-integer inputs.
3007 const double SinVals[4] = { 0.0, 1.0, 0.0, -1.0 };
3008 V = SinVals[((int)V4 + (IsCos ? 1 : 0)) & 3];
3009 } else {
3010 if (IsCos)
3011 V = cos(V * 2.0 * numbers::pi);
3012 else
3013 V = sin(V * 2.0 * numbers::pi);
3014 }
3015 return GetConstantFoldFPValue(V, Ty);
3016 }
3017 }
3018
3019 if (!TLI)
3020 return nullptr;
3021
3022 LibFunc Func = TLI->getLibFunc(Name);
3023 if (Func == NotLibFunc)
3024 return nullptr;
3025
3026 switch (Func) {
3027 default:
3028 break;
3029 case LibFunc_acos:
3030 case LibFunc_acosf:
3031 case LibFunc_acos_finite:
3032 case LibFunc_acosf_finite:
3033 if (TLI->has(Func))
3034 return ConstantFoldFP(acos, APF, Ty);
3035 break;
3036 case LibFunc_asin:
3037 case LibFunc_asinf:
3038 case LibFunc_asin_finite:
3039 case LibFunc_asinf_finite:
3040 if (TLI->has(Func))
3041 return ConstantFoldFP(asin, APF, Ty);
3042 break;
3043 case LibFunc_atan:
3044 case LibFunc_atanf:
3045 // Implement optional behavior from C's Annex F for +/-0.0.
3046 if (U.isZero())
3047 return ConstantFP::get(Ty, U);
3048 if (TLI->has(Func))
3049 return ConstantFoldFP(atan, APF, Ty);
3050 break;
3051 case LibFunc_ceil:
3052 case LibFunc_ceilf:
3053 if (TLI->has(Func)) {
3054 U.roundToIntegral(APFloat::rmTowardPositive);
3055 return ConstantFP::get(Ty, U);
3056 }
3057 break;
3058 case LibFunc_cos:
3059 case LibFunc_cosf:
3060 if (TLI->has(Func))
3061 return ConstantFoldFP(cos, APF, Ty);
3062 break;
3063 case LibFunc_cosh:
3064 case LibFunc_coshf:
3065 case LibFunc_cosh_finite:
3066 case LibFunc_coshf_finite:
3067 if (TLI->has(Func))
3068 return ConstantFoldFP(cosh, APF, Ty);
3069 break;
3070 case LibFunc_exp:
3071 case LibFunc_expf:
3072 case LibFunc_exp_finite:
3073 case LibFunc_expf_finite:
3074 if (TLI->has(Func))
3075 return ConstantFoldFP(exp, APF, Ty);
3076 break;
3077 case LibFunc_exp2:
3078 case LibFunc_exp2f:
3079 case LibFunc_exp2_finite:
3080 case LibFunc_exp2f_finite:
3081 if (TLI->has(Func))
3082 // Fold exp2(x) as pow(2, x), in case the host lacks a C99 library.
3083 return ConstantFoldBinaryFP(pow, APFloat(2.0), APF, Ty);
3084 break;
3085 case LibFunc_fabs:
3086 case LibFunc_fabsf:
3087 if (TLI->has(Func)) {
3088 U.clearSign();
3089 return ConstantFP::get(Ty, U);
3090 }
3091 break;
3092 case LibFunc_floor:
3093 case LibFunc_floorf:
3094 if (TLI->has(Func)) {
3095 U.roundToIntegral(APFloat::rmTowardNegative);
3096 return ConstantFP::get(Ty, U);
3097 }
3098 break;
3099 case LibFunc_log:
3100 case LibFunc_logf:
3101 case LibFunc_log_finite:
3102 case LibFunc_logf_finite:
3103 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3104 return ConstantFoldFP(log, APF, Ty);
3105 break;
3106 case LibFunc_log2:
3107 case LibFunc_log2f:
3108 case LibFunc_log2_finite:
3109 case LibFunc_log2f_finite:
3110 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3111 // TODO: What about hosts that lack a C99 library?
3112 return ConstantFoldFP(log2, APF, Ty);
3113 break;
3114 case LibFunc_log10:
3115 case LibFunc_log10f:
3116 case LibFunc_log10_finite:
3117 case LibFunc_log10f_finite:
3118 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3119 // TODO: What about hosts that lack a C99 library?
3120 return ConstantFoldFP(log10, APF, Ty);
3121 break;
3122 case LibFunc_ilogb:
3123 case LibFunc_ilogbf:
3124 if (!APF.isZero() && TLI->has(Func))
3125 return ConstantInt::get(Ty, ilogb(APF), true);
3126 break;
3127 case LibFunc_logb:
3128 case LibFunc_logbf:
3129 if (!APF.isZero() && TLI->has(Func))
3130 return ConstantFoldFP(logb, APF, Ty);
3131 break;
3132 case LibFunc_log1p:
3133 case LibFunc_log1pf:
3134 // Implement optional behavior from C's Annex F for +/-0.0.
3135 if (U.isZero())
3136 return ConstantFP::get(Ty, U);
3137 if (APF > APFloat::getOne(APF.getSemantics(), true) && TLI->has(Func))
3138 return ConstantFoldFP(log1p, APF, Ty);
3139 break;
3140 case LibFunc_logl:
3141 return nullptr;
3142 case LibFunc_erf:
3143 case LibFunc_erff:
3144 if (TLI->has(Func))
3145 return ConstantFoldFP(erf, APF, Ty);
3146 break;
3147 case LibFunc_nearbyint:
3148 case LibFunc_nearbyintf:
3149 case LibFunc_rint:
3150 case LibFunc_rintf:
3151 case LibFunc_roundeven:
3152 case LibFunc_roundevenf:
3153 if (TLI->has(Func)) {
3154 U.roundToIntegral(APFloat::rmNearestTiesToEven);
3155 return ConstantFP::get(Ty, U);
3156 }
3157 break;
3158 case LibFunc_round:
3159 case LibFunc_roundf:
3160 if (TLI->has(Func)) {
3161 U.roundToIntegral(APFloat::rmNearestTiesToAway);
3162 return ConstantFP::get(Ty, U);
3163 }
3164 break;
3165 case LibFunc_sin:
3166 case LibFunc_sinf:
3167 if (TLI->has(Func))
3168 return ConstantFoldFP(sin, APF, Ty);
3169 break;
3170 case LibFunc_sinh:
3171 case LibFunc_sinhf:
3172 case LibFunc_sinh_finite:
3173 case LibFunc_sinhf_finite:
3174 if (TLI->has(Func))
3175 return ConstantFoldFP(sinh, APF, Ty);
3176 break;
3177 case LibFunc_sqrt:
3178 case LibFunc_sqrtf:
3179 if (!APF.isNegative() && TLI->has(Func))
3180 return ConstantFoldFP(sqrt, APF, Ty);
3181 break;
3182 case LibFunc_tan:
3183 case LibFunc_tanf:
3184 if (TLI->has(Func))
3185 return ConstantFoldFP(tan, APF, Ty);
3186 break;
3187 case LibFunc_tanh:
3188 case LibFunc_tanhf:
3189 if (TLI->has(Func))
3190 return ConstantFoldFP(tanh, APF, Ty);
3191 break;
3192 case LibFunc_trunc:
3193 case LibFunc_truncf:
3194 if (TLI->has(Func)) {
3195 U.roundToIntegral(APFloat::rmTowardZero);
3196 return ConstantFP::get(Ty, U);
3197 }
3198 break;
3199 }
3200 return nullptr;
3201 }
3202
3203 if (auto *Op = dyn_cast<ConstantInt>(Operands[0])) {
3204 switch (IntrinsicID) {
3205 case Intrinsic::bswap:
3206 return ConstantInt::get(Ty->getContext(), Op->getValue().byteSwap());
3207 case Intrinsic::ctpop:
3208 return ConstantInt::get(Ty, Op->getValue().popcount());
3209 case Intrinsic::bitreverse:
3210 return ConstantInt::get(Ty->getContext(), Op->getValue().reverseBits());
3211 case Intrinsic::amdgcn_s_wqm: {
3212 uint64_t Val = Op->getZExtValue();
3213 Val |= (Val & 0x5555555555555555ULL) << 1 |
3214 ((Val >> 1) & 0x5555555555555555ULL);
3215 Val |= (Val & 0x3333333333333333ULL) << 2 |
3216 ((Val >> 2) & 0x3333333333333333ULL);
3217 return ConstantInt::get(Ty, Val);
3218 }
3219
3220 case Intrinsic::amdgcn_s_quadmask: {
3221 uint64_t Val = Op->getZExtValue();
3222 uint64_t QuadMask = 0;
3223 for (unsigned I = 0; I < Op->getBitWidth() / 4; ++I, Val >>= 4) {
3224 if (!(Val & 0xF))
3225 continue;
3226
3227 QuadMask |= (1ULL << I);
3228 }
3229 return ConstantInt::get(Ty, QuadMask);
3230 }
3231
3232 case Intrinsic::amdgcn_s_bitreplicate: {
3233 uint64_t Val = Op->getZExtValue();
3234 Val = (Val & 0x000000000000FFFFULL) | (Val & 0x00000000FFFF0000ULL) << 16;
3235 Val = (Val & 0x000000FF000000FFULL) | (Val & 0x0000FF000000FF00ULL) << 8;
3236 Val = (Val & 0x000F000F000F000FULL) | (Val & 0x00F000F000F000F0ULL) << 4;
3237 Val = (Val & 0x0303030303030303ULL) | (Val & 0x0C0C0C0C0C0C0C0CULL) << 2;
3238 Val = (Val & 0x1111111111111111ULL) | (Val & 0x2222222222222222ULL) << 1;
3239 Val = Val | Val << 1;
3240 return ConstantInt::get(Ty, Val);
3241 }
3242 }
3243 }
3244
3245 if (Operands[0]->getType()->isVectorTy()) {
3246 auto *Op = cast<Constant>(Operands[0]);
3247 switch (IntrinsicID) {
3248 default: break;
3249 case Intrinsic::vector_reduce_add:
3250 case Intrinsic::vector_reduce_mul:
3251 case Intrinsic::vector_reduce_and:
3252 case Intrinsic::vector_reduce_or:
3253 case Intrinsic::vector_reduce_xor:
3254 case Intrinsic::vector_reduce_smin:
3255 case Intrinsic::vector_reduce_smax:
3256 case Intrinsic::vector_reduce_umin:
3257 case Intrinsic::vector_reduce_umax:
3258 if (Constant *C = constantFoldVectorReduce(IntrinsicID, Operands[0]))
3259 return C;
3260 break;
3261 case Intrinsic::x86_sse_cvtss2si:
3262 case Intrinsic::x86_sse_cvtss2si64:
3263 case Intrinsic::x86_sse2_cvtsd2si:
3264 case Intrinsic::x86_sse2_cvtsd2si64:
3265 if (ConstantFP *FPOp =
3266 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3267 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3268 /*roundTowardZero=*/false, Ty,
3269 /*IsSigned*/true);
3270 break;
3271 case Intrinsic::x86_sse_cvttss2si:
3272 case Intrinsic::x86_sse_cvttss2si64:
3273 case Intrinsic::x86_sse2_cvttsd2si:
3274 case Intrinsic::x86_sse2_cvttsd2si64:
3275 if (ConstantFP *FPOp =
3276 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3277 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3278 /*roundTowardZero=*/true, Ty,
3279 /*IsSigned*/true);
3280 break;
3281
3282 case Intrinsic::wasm_anytrue:
3283 return Op->isNullValue() ? ConstantInt::get(Ty, 0)
3284 : ConstantInt::get(Ty, 1);
3285
3286 case Intrinsic::wasm_alltrue:
3287 // Check each element individually
3288 unsigned E = cast<FixedVectorType>(Op->getType())->getNumElements();
3289 for (unsigned I = 0; I != E; ++I) {
3290 Constant *Elt = Op->getAggregateElement(I);
3291 // Return false as soon as we find a non-true element.
3292 if (Elt && Elt->isNullValue())
3293 return ConstantInt::get(Ty, 0);
3294 // Bail as soon as we find an element we cannot prove to be true.
3295 if (!Elt || !isa<ConstantInt>(Elt))
3296 return nullptr;
3297 }
3298
3299 return ConstantInt::get(Ty, 1);
3300 }
3301 }
3302
3303 return nullptr;
3304}
3305
3306static Constant *evaluateCompare(const APFloat &Op1, const APFloat &Op2,
3310 FCmpInst::Predicate Cond = FCmp->getPredicate();
3311 if (FCmp->isSignaling()) {
3312 if (Op1.isNaN() || Op2.isNaN())
3314 } else {
3315 if (Op1.isSignaling() || Op2.isSignaling())
3317 }
3318 bool Result = FCmpInst::compare(Op1, Op2, Cond);
3319 if (mayFoldConstrained(const_cast<ConstrainedFPCmpIntrinsic *>(FCmp), St))
3320 return ConstantInt::get(Call->getType()->getScalarType(), Result);
3321 return nullptr;
3322}
3323
3324static Constant *ConstantFoldNextToward(const APFloat &Op0, const APFloat &Op1,
3325 const Type *RetTy) {
3326 assert(RetTy != nullptr);
3327 bool LosesInfo;
3328
3329 if (Op1.isSignaling())
3330 return nullptr;
3331 if (Op1.isNaN()) {
3332 APFloat Ret(Op1);
3333 Ret.convert(RetTy->getFltSemantics(), detail::rmNearestTiesToEven,
3334 &LosesInfo);
3335 return ConstantFP::get(RetTy->getContext(), Ret);
3336 }
3337
3338 // Recall that the second argument of nexttoward is always a long double,
3339 // so we may need to promote the first argument for comparisons to be valid.
3340 APFloat PromotedOp0(Op0);
3341 PromotedOp0.convert(Op1.getSemantics(), detail::rmNearestTiesToEven,
3342 &LosesInfo);
3343 assert(!LosesInfo && "Unexpected lossy promotion");
3344 const APFloat::cmpResult Result = PromotedOp0.compare(Op1);
3345
3346 // When equal, the standard says we must return the second argument.
3347 // This allows nice behavior such as nexttoward(0.0, -0.0) = -0.0 and
3348 // nexttoward(-0.0, 0.0) = 0.0
3349 if (Result == detail::cmpEqual) {
3350 APFloat Ret(Op1);
3351 Ret.convert(RetTy->getFltSemantics(), detail::rmNearestTiesToEven,
3352 &LosesInfo);
3353 return ConstantFP::get(RetTy->getContext(), Ret);
3354 }
3355
3356 APFloat Next(Op0);
3357 Next.next(/*nextDown=*/Result == APFloat::cmpGreaterThan);
3358 if (Next.isZero() || Next.isDenormal() || Next.isSignaling())
3359 return nullptr;
3360 return ConstantFP::get(RetTy->getContext(), Next);
3361}
3362
3363static Constant *ConstantFoldLibCall2(StringRef Name, Type *Ty,
3365 const TargetLibraryInfo *TLI = nullptr) {
3366 if (!TLI)
3367 return nullptr;
3368
3369 LibFunc Func = TLI->getLibFunc(Name);
3370 if (Func == NotLibFunc)
3371 return nullptr;
3372
3373 const auto *Op1 = dyn_cast<ConstantFP>(Operands[0]);
3374 if (!Op1)
3375 return nullptr;
3376
3377 const auto *Op2 = dyn_cast<ConstantFP>(Operands[1]);
3378 if (!Op2)
3379 return nullptr;
3380
3381 const APFloat &Op1V = Op1->getValueAPF();
3382 const APFloat &Op2V = Op2->getValueAPF();
3383
3384 switch (Func) {
3385 default:
3386 break;
3387 case LibFunc_pow:
3388 case LibFunc_powf:
3389 case LibFunc_pow_finite:
3390 case LibFunc_powf_finite:
3391 if (TLI->has(Func))
3392 return ConstantFoldBinaryFP(pow, Op1V, Op2V, Ty);
3393 break;
3394 case LibFunc_fmod:
3395 case LibFunc_fmodf:
3396 if (TLI->has(Func)) {
3397 APFloat V = Op1->getValueAPF();
3398 if (APFloat::opStatus::opOK == V.mod(Op2->getValueAPF()))
3399 return ConstantFP::get(Ty, V);
3400 }
3401 break;
3402 case LibFunc_remainder:
3403 case LibFunc_remainderf:
3404 if (TLI->has(Func)) {
3405 APFloat V = Op1->getValueAPF();
3406 if (APFloat::opStatus::opOK == V.remainder(Op2->getValueAPF()))
3407 return ConstantFP::get(Ty, V);
3408 }
3409 break;
3410 case LibFunc_atan2:
3411 case LibFunc_atan2f:
3412 // atan2(+/-0.0, +/-0.0) is known to raise an exception on some libm
3413 // (Solaris), so we do not assume a known result for that.
3414 if (Op1V.isZero() && Op2V.isZero())
3415 return nullptr;
3416 [[fallthrough]];
3417 case LibFunc_atan2_finite:
3418 case LibFunc_atan2f_finite:
3419 if (TLI->has(Func))
3420 return ConstantFoldBinaryFP(atan2, Op1V, Op2V, Ty);
3421 break;
3422 case LibFunc_nextafter:
3423 case LibFunc_nextafterf:
3424 case LibFunc_nexttoward:
3425 case LibFunc_nexttowardf:
3426 if (TLI->has(Func))
3427 return ConstantFoldNextToward(Op1V, Op2V, Ty);
3428 break;
3429 }
3430
3431 return nullptr;
3432}
3433
3434static Constant *ConstantFoldIntrinsicCall2(Intrinsic::ID IntrinsicID, Type *Ty,
3436 const CallBase *Call = nullptr) {
3437 assert(Operands.size() == 2 && "Wrong number of operands.");
3438
3439 if (Ty->isFloatingPointTy()) {
3440 // TODO: We should have undef handling for all of the FP intrinsics that
3441 // are attempted to be folded in this function.
3442 bool IsOp0Undef = isa<UndefValue>(Operands[0]);
3443 bool IsOp1Undef = isa<UndefValue>(Operands[1]);
3444 switch (IntrinsicID) {
3445 case Intrinsic::maxnum:
3446 case Intrinsic::minnum:
3447 case Intrinsic::maximum:
3448 case Intrinsic::minimum:
3449 case Intrinsic::maximumnum:
3450 case Intrinsic::minimumnum:
3451 case Intrinsic::nvvm_fmax_d:
3452 case Intrinsic::nvvm_fmin_d:
3453 // If one argument is undef, return the other argument.
3454 if (IsOp0Undef)
3455 return Operands[1];
3456 if (IsOp1Undef)
3457 return Operands[0];
3458 break;
3459
3460 case Intrinsic::nvvm_fmax_f:
3461 case Intrinsic::nvvm_fmax_ftz_f:
3462 case Intrinsic::nvvm_fmax_ftz_nan_f:
3463 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3464 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3465 case Intrinsic::nvvm_fmax_nan_f:
3466 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3467 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3468
3469 case Intrinsic::nvvm_fmin_f:
3470 case Intrinsic::nvvm_fmin_ftz_f:
3471 case Intrinsic::nvvm_fmin_ftz_nan_f:
3472 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
3473 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
3474 case Intrinsic::nvvm_fmin_nan_f:
3475 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
3476 case Intrinsic::nvvm_fmin_xorsign_abs_f:
3477 // If one arg is undef, the other arg can be returned only if it is
3478 // constant, as we may need to flush it to sign-preserving zero or
3479 // canonicalize the NaN.
3480 if (!IsOp0Undef && !IsOp1Undef)
3481 break;
3482 if (auto *Op = dyn_cast<ConstantFP>(Operands[IsOp0Undef ? 1 : 0])) {
3483 if (Op->isNaN()) {
3484 APInt NVCanonicalNaN(32, 0x7fffffff);
3485 return ConstantFP::get(
3486 Ty, APFloat(Ty->getFltSemantics(), NVCanonicalNaN));
3487 }
3488 if (nvvm::FMinFMaxShouldFTZ(IntrinsicID))
3489 return ConstantFP::get(Ty, FTZPreserveSign(Op->getValueAPF()));
3490 else
3491 return Op;
3492 }
3493 break;
3494 }
3495 }
3496
3497 if (const auto *Op1 = dyn_cast<ConstantFP>(Operands[0])) {
3498 const APFloat &Op1V = Op1->getValueAPF();
3499
3500 if (const auto *Op2 = dyn_cast<ConstantFP>(Operands[1])) {
3501 if (Op2->getType() != Op1->getType())
3502 return nullptr;
3503 const APFloat &Op2V = Op2->getValueAPF();
3504
3505 if (const auto *ConstrIntr =
3507 RoundingMode RM = getEvaluationRoundingMode(ConstrIntr);
3508 APFloat Res = Op1V;
3510 switch (IntrinsicID) {
3511 default:
3512 return nullptr;
3513 case Intrinsic::experimental_constrained_fadd:
3514 St = Res.add(Op2V, RM);
3515 break;
3516 case Intrinsic::experimental_constrained_fsub:
3517 St = Res.subtract(Op2V, RM);
3518 break;
3519 case Intrinsic::experimental_constrained_fmul:
3520 St = Res.multiply(Op2V, RM);
3521 break;
3522 case Intrinsic::experimental_constrained_fdiv:
3523 St = Res.divide(Op2V, RM);
3524 break;
3525 case Intrinsic::experimental_constrained_frem:
3526 St = Res.mod(Op2V);
3527 break;
3528 case Intrinsic::experimental_constrained_fcmp:
3529 case Intrinsic::experimental_constrained_fcmps:
3530 return evaluateCompare(Op1V, Op2V, ConstrIntr);
3531 }
3532 if (mayFoldConstrained(const_cast<ConstrainedFPIntrinsic *>(ConstrIntr),
3533 St))
3534 return ConstantFP::get(Ty, Res);
3535 return nullptr;
3536 }
3537
3538 switch (IntrinsicID) {
3539 default:
3540 break;
3541 case Intrinsic::copysign:
3542 return ConstantFP::get(Ty, APFloat::copySign(Op1V, Op2V));
3543 case Intrinsic::minnum:
3544 return ConstantFP::get(Ty, minnum(Op1V, Op2V));
3545 case Intrinsic::maxnum:
3546 return ConstantFP::get(Ty, maxnum(Op1V, Op2V));
3547 case Intrinsic::minimum:
3548 return ConstantFP::get(Ty, minimum(Op1V, Op2V));
3549 case Intrinsic::maximum:
3550 return ConstantFP::get(Ty, maximum(Op1V, Op2V));
3551 case Intrinsic::minimumnum:
3552 return ConstantFP::get(Ty, minimumnum(Op1V, Op2V));
3553 case Intrinsic::maximumnum:
3554 return ConstantFP::get(Ty, maximumnum(Op1V, Op2V));
3555
3556 case Intrinsic::nvvm_fmax_d:
3557 case Intrinsic::nvvm_fmax_f:
3558 case Intrinsic::nvvm_fmax_ftz_f:
3559 case Intrinsic::nvvm_fmax_ftz_nan_f:
3560 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3561 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3562 case Intrinsic::nvvm_fmax_nan_f:
3563 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3564 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3565
3566 case Intrinsic::nvvm_fmin_d:
3567 case Intrinsic::nvvm_fmin_f:
3568 case Intrinsic::nvvm_fmin_ftz_f:
3569 case Intrinsic::nvvm_fmin_ftz_nan_f:
3570 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
3571 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
3572 case Intrinsic::nvvm_fmin_nan_f:
3573 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
3574 case Intrinsic::nvvm_fmin_xorsign_abs_f: {
3575
3576 bool ShouldCanonicalizeNaNs = !(IntrinsicID == Intrinsic::nvvm_fmax_d ||
3577 IntrinsicID == Intrinsic::nvvm_fmin_d);
3578 bool IsFTZ = nvvm::FMinFMaxShouldFTZ(IntrinsicID);
3579 bool IsNaNPropagating = nvvm::FMinFMaxPropagatesNaNs(IntrinsicID);
3580 bool IsXorSignAbs = nvvm::FMinFMaxIsXorSignAbs(IntrinsicID);
3581
3582 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3583 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3584
3585 bool XorSign = false;
3586 if (IsXorSignAbs) {
3587 XorSign = A.isNegative() ^ B.isNegative();
3588 A = abs(A);
3589 B = abs(B);
3590 }
3591
3592 bool IsFMax = false;
3593 switch (IntrinsicID) {
3594 case Intrinsic::nvvm_fmax_d:
3595 case Intrinsic::nvvm_fmax_f:
3596 case Intrinsic::nvvm_fmax_ftz_f:
3597 case Intrinsic::nvvm_fmax_ftz_nan_f:
3598 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3599 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3600 case Intrinsic::nvvm_fmax_nan_f:
3601 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3602 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3603 IsFMax = true;
3604 break;
3605 }
3606 APFloat Res =
3607 IsFMax ? (IsNaNPropagating ? maximum(A, B) : maximumnum(A, B))
3608 : (IsNaNPropagating ? minimum(A, B) : minimumnum(A, B));
3609
3610 if (ShouldCanonicalizeNaNs && Res.isNaN()) {
3611 APFloat NVCanonicalNaN(Res.getSemantics(), APInt(32, 0x7fffffff));
3612 return ConstantFP::get(Ty, NVCanonicalNaN);
3613 }
3614
3615 if (IsXorSignAbs && XorSign != Res.isNegative())
3616 Res.changeSign();
3617
3618 return ConstantFP::get(Ty, Res);
3619 }
3620
3621 case Intrinsic::nvvm_add_rm_f:
3622 case Intrinsic::nvvm_add_rn_f:
3623 case Intrinsic::nvvm_add_rp_f:
3624 case Intrinsic::nvvm_add_rz_f:
3625 case Intrinsic::nvvm_add_rm_d:
3626 case Intrinsic::nvvm_add_rn_d:
3627 case Intrinsic::nvvm_add_rp_d:
3628 case Intrinsic::nvvm_add_rz_d:
3629 case Intrinsic::nvvm_add_rm_ftz_f:
3630 case Intrinsic::nvvm_add_rn_ftz_f:
3631 case Intrinsic::nvvm_add_rp_ftz_f:
3632 case Intrinsic::nvvm_add_rz_ftz_f: {
3633
3634 bool IsFTZ = nvvm::FAddShouldFTZ(IntrinsicID);
3635 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3636 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3637
3638 APFloat::roundingMode RoundMode =
3639 nvvm::GetFAddRoundingMode(IntrinsicID);
3640
3641 APFloat Res = A;
3642 APFloat::opStatus Status = Res.add(B, RoundMode);
3643
3644 if (!Res.isNaN() &&
3646 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3647 return ConstantFP::get(Ty, Res);
3648 }
3649 return nullptr;
3650 }
3651
3652 case Intrinsic::nvvm_mul_rm_f:
3653 case Intrinsic::nvvm_mul_rn_f:
3654 case Intrinsic::nvvm_mul_rp_f:
3655 case Intrinsic::nvvm_mul_rz_f:
3656 case Intrinsic::nvvm_mul_rm_d:
3657 case Intrinsic::nvvm_mul_rn_d:
3658 case Intrinsic::nvvm_mul_rp_d:
3659 case Intrinsic::nvvm_mul_rz_d:
3660 case Intrinsic::nvvm_mul_rm_ftz_f:
3661 case Intrinsic::nvvm_mul_rn_ftz_f:
3662 case Intrinsic::nvvm_mul_rp_ftz_f:
3663 case Intrinsic::nvvm_mul_rz_ftz_f: {
3664
3665 bool IsFTZ = nvvm::FMulShouldFTZ(IntrinsicID);
3666 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3667 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3668
3669 APFloat::roundingMode RoundMode =
3670 nvvm::GetFMulRoundingMode(IntrinsicID);
3671
3672 APFloat Res = A;
3673 APFloat::opStatus Status = Res.multiply(B, RoundMode);
3674
3675 if (!Res.isNaN() &&
3677 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3678 return ConstantFP::get(Ty, Res);
3679 }
3680 return nullptr;
3681 }
3682
3683 case Intrinsic::nvvm_div_rm_f:
3684 case Intrinsic::nvvm_div_rn_f:
3685 case Intrinsic::nvvm_div_rp_f:
3686 case Intrinsic::nvvm_div_rz_f:
3687 case Intrinsic::nvvm_div_rm_d:
3688 case Intrinsic::nvvm_div_rn_d:
3689 case Intrinsic::nvvm_div_rp_d:
3690 case Intrinsic::nvvm_div_rz_d:
3691 case Intrinsic::nvvm_div_rm_ftz_f:
3692 case Intrinsic::nvvm_div_rn_ftz_f:
3693 case Intrinsic::nvvm_div_rp_ftz_f:
3694 case Intrinsic::nvvm_div_rz_ftz_f: {
3695 bool IsFTZ = nvvm::FDivShouldFTZ(IntrinsicID);
3696 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3697 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3698 APFloat::roundingMode RoundMode =
3699 nvvm::GetFDivRoundingMode(IntrinsicID);
3700
3701 APFloat Res = A;
3702 APFloat::opStatus Status = Res.divide(B, RoundMode);
3703 if (!Res.isNaN() &&
3705 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3706 return ConstantFP::get(Ty, Res);
3707 }
3708 return nullptr;
3709 }
3710 }
3711
3712 if (!Ty->isHalfTy() && !Ty->isFloatTy() && !Ty->isDoubleTy())
3713 return nullptr;
3714
3715 switch (IntrinsicID) {
3716 default:
3717 break;
3718 case Intrinsic::pow:
3719 return ConstantFoldBinaryFP(pow, Op1V, Op2V, Ty);
3720 case Intrinsic::amdgcn_fmul_legacy:
3721 // The legacy behaviour is that multiplying +/- 0.0 by anything, even
3722 // NaN or infinity, gives +0.0.
3723 if (Op1V.isZero() || Op2V.isZero())
3724 return ConstantFP::getZero(Ty);
3725 return ConstantFP::get(Ty, Op1V * Op2V);
3726 }
3727
3728 } else if (auto *Op2C = dyn_cast<ConstantInt>(Operands[1])) {
3729 switch (IntrinsicID) {
3730 case Intrinsic::ldexp: {
3731 // APFloat::scalbn takes the exponent as `int`. Clamp wider integer
3732 // exponents into [INT_MIN, INT_MAX] so values still saturate the
3733 // result to +/-inf or +/-0.
3734 APInt Exp = Op2C->getValue();
3735 Exp = Exp.getBitWidth() < 32 ? Exp.sext(32) : Exp.truncSSat(32);
3736 return ConstantFP::get(
3737 Ty->getContext(),
3738 scalbn(Op1V, Exp.getSExtValue(), APFloat::rmNearestTiesToEven));
3739 }
3740 case Intrinsic::is_fpclass: {
3741 FPClassTest Mask = static_cast<FPClassTest>(Op2C->getZExtValue());
3742 bool Result =
3743 ((Mask & fcSNan) && Op1V.isNaN() && Op1V.isSignaling()) ||
3744 ((Mask & fcQNan) && Op1V.isNaN() && !Op1V.isSignaling()) ||
3745 ((Mask & fcNegInf) && Op1V.isNegInfinity()) ||
3746 ((Mask & fcNegNormal) && Op1V.isNormal() && Op1V.isNegative()) ||
3747 ((Mask & fcNegSubnormal) && Op1V.isDenormal() && Op1V.isNegative()) ||
3748 ((Mask & fcNegZero) && Op1V.isZero() && Op1V.isNegative()) ||
3749 ((Mask & fcPosZero) && Op1V.isZero() && !Op1V.isNegative()) ||
3750 ((Mask & fcPosSubnormal) && Op1V.isDenormal() && !Op1V.isNegative()) ||
3751 ((Mask & fcPosNormal) && Op1V.isNormal() && !Op1V.isNegative()) ||
3752 ((Mask & fcPosInf) && Op1V.isPosInfinity());
3753 return ConstantInt::get(Ty, Result);
3754 }
3755 case Intrinsic::powi: {
3756 // Square-and-multiply using the operand's own semantics, matching
3757 // the multiply sequence ExpandPowI builds in SelectionDAG.
3758 int Exp = static_cast<int>(Op2C->getSExtValue());
3759 unsigned UExp = static_cast<unsigned>(Exp);
3760 if (Exp < 0)
3761 UExp = -UExp;
3762 const fltSemantics &Semantics = Op1V.getSemantics();
3763 APFloat Res = APFloat::getOne(Semantics);
3764 APFloat CurSquare = Op1V;
3765 while (UExp) {
3766 if (UExp & 1)
3767 Res = Res * CurSquare;
3768 CurSquare = CurSquare * CurSquare;
3769 UExp >>= 1;
3770 }
3771 if (Exp < 0)
3772 Res = APFloat::getOne(Semantics) / Res;
3773 return ConstantFP::get(Ty, Res);
3774 }
3775 default:
3776 break;
3777 }
3778 }
3779 return nullptr;
3780 }
3781
3782 if (Operands[0]->getType()->isIntegerTy() &&
3783 Operands[1]->getType()->isIntegerTy()) {
3784 const APInt *C0, *C1;
3785 if (!getConstIntOrUndef(Operands[0], C0) ||
3786 !getConstIntOrUndef(Operands[1], C1))
3787 return nullptr;
3788
3789 switch (IntrinsicID) {
3790 default: break;
3791 case Intrinsic::smax:
3792 case Intrinsic::smin:
3793 case Intrinsic::umax:
3794 case Intrinsic::umin:
3795 if (!C0 || !C1)
3796 return MinMaxIntrinsic::getSaturationPoint(IntrinsicID, Ty);
3797 return ConstantInt::get(
3798 Ty, ICmpInst::compare(*C0, *C1,
3799 MinMaxIntrinsic::getPredicate(IntrinsicID))
3800 ? *C0
3801 : *C1);
3802
3803 case Intrinsic::scmp:
3804 case Intrinsic::ucmp:
3805 if (!C0 || !C1)
3806 return ConstantInt::get(Ty, 0);
3807
3808 int Res;
3809 if (IntrinsicID == Intrinsic::scmp)
3810 Res = C0->sgt(*C1) ? 1 : C0->slt(*C1) ? -1 : 0;
3811 else
3812 Res = C0->ugt(*C1) ? 1 : C0->ult(*C1) ? -1 : 0;
3813 return ConstantInt::get(Ty, Res, /*IsSigned=*/true);
3814
3815 case Intrinsic::usub_with_overflow:
3816 case Intrinsic::ssub_with_overflow:
3817 // X - undef -> { 0, false }
3818 // undef - X -> { 0, false }
3819 if (!C0 || !C1)
3820 return Constant::getNullValue(Ty);
3821 [[fallthrough]];
3822 case Intrinsic::uadd_with_overflow:
3823 case Intrinsic::sadd_with_overflow:
3824 // X + undef -> { -1, false }
3825 // undef + x -> { -1, false }
3826 if (!C0 || !C1) {
3827 return ConstantStruct::get(
3828 cast<StructType>(Ty),
3829 {Constant::getAllOnesValue(Ty->getStructElementType(0)),
3830 Constant::getNullValue(Ty->getStructElementType(1))});
3831 }
3832 [[fallthrough]];
3833 case Intrinsic::smul_with_overflow:
3834 case Intrinsic::umul_with_overflow: {
3835 // undef * X -> { 0, false }
3836 // X * undef -> { 0, false }
3837 if (!C0 || !C1)
3838 return Constant::getNullValue(Ty);
3839
3840 APInt Res;
3841 bool Overflow;
3842 switch (IntrinsicID) {
3843 default: llvm_unreachable("Invalid case");
3844 case Intrinsic::sadd_with_overflow:
3845 Res = C0->sadd_ov(*C1, Overflow);
3846 break;
3847 case Intrinsic::uadd_with_overflow:
3848 Res = C0->uadd_ov(*C1, Overflow);
3849 break;
3850 case Intrinsic::ssub_with_overflow:
3851 Res = C0->ssub_ov(*C1, Overflow);
3852 break;
3853 case Intrinsic::usub_with_overflow:
3854 Res = C0->usub_ov(*C1, Overflow);
3855 break;
3856 case Intrinsic::smul_with_overflow:
3857 Res = C0->smul_ov(*C1, Overflow);
3858 break;
3859 case Intrinsic::umul_with_overflow:
3860 Res = C0->umul_ov(*C1, Overflow);
3861 break;
3862 }
3863 Constant *Ops[] = {
3864 ConstantInt::get(Ty->getContext(), Res),
3865 ConstantInt::get(Type::getInt1Ty(Ty->getContext()), Overflow)
3866 };
3868 }
3869 case Intrinsic::uadd_sat:
3870 case Intrinsic::sadd_sat:
3871 if (!C0 || !C1)
3872 return Constant::getAllOnesValue(Ty);
3873 if (IntrinsicID == Intrinsic::uadd_sat)
3874 return ConstantInt::get(Ty, C0->uadd_sat(*C1));
3875 else
3876 return ConstantInt::get(Ty, C0->sadd_sat(*C1));
3877 case Intrinsic::usub_sat:
3878 case Intrinsic::ssub_sat:
3879 if (!C0 || !C1)
3880 return Constant::getNullValue(Ty);
3881 if (IntrinsicID == Intrinsic::usub_sat)
3882 return ConstantInt::get(Ty, C0->usub_sat(*C1));
3883 else
3884 return ConstantInt::get(Ty, C0->ssub_sat(*C1));
3885 case Intrinsic::cttz:
3886 case Intrinsic::ctlz:
3887 assert(C1 && "Must be constant int");
3888
3889 // cttz(0, 1) and ctlz(0, 1) are poison.
3890 if (C1->isOne() && (!C0 || C0->isZero()))
3891 return PoisonValue::get(Ty);
3892 if (!C0)
3893 return Constant::getNullValue(Ty);
3894 if (IntrinsicID == Intrinsic::cttz)
3895 return ConstantInt::get(Ty, C0->countr_zero());
3896 else
3897 return ConstantInt::get(Ty, C0->countl_zero());
3898
3899 case Intrinsic::abs:
3900 assert(C1 && "Must be constant int");
3901 assert((C1->isOne() || C1->isZero()) && "Must be 0 or 1");
3902
3903 // Undef or minimum val operand with poison min --> poison
3904 if (C1->isOne() && (!C0 || C0->isMinSignedValue()))
3905 return PoisonValue::get(Ty);
3906
3907 // Undef operand with no poison min --> 0 (sign bit must be clear)
3908 if (!C0)
3909 return Constant::getNullValue(Ty);
3910
3911 return ConstantInt::get(Ty, C0->abs());
3912 case Intrinsic::clmul:
3913 if (!C0 || !C1)
3914 return Constant::getNullValue(Ty);
3915 return ConstantInt::get(Ty, APIntOps::clmul(*C0, *C1));
3916 case Intrinsic::pdep:
3917 if (!C0 || !C1)
3918 return Constant::getNullValue(Ty);
3919 return ConstantInt::get(Ty, APIntOps::pdep(*C0, *C1));
3920 case Intrinsic::pext:
3921 if (!C0 || !C1)
3922 return Constant::getNullValue(Ty);
3923 return ConstantInt::get(Ty, APIntOps::pext(*C0, *C1));
3924 case Intrinsic::amdgcn_wave_reduce_umin:
3925 case Intrinsic::amdgcn_wave_reduce_umax:
3926 case Intrinsic::amdgcn_wave_reduce_max:
3927 case Intrinsic::amdgcn_wave_reduce_min:
3928 case Intrinsic::amdgcn_wave_reduce_and:
3929 case Intrinsic::amdgcn_wave_reduce_or:
3930 return Operands[0];
3931 }
3932
3933 return nullptr;
3934 }
3935
3936 // Support ConstantVector in case we have an Undef in the top.
3937 if ((isa<ConstantVector>(Operands[0]) ||
3939 // Check for default rounding mode.
3940 // FIXME: Support other rounding modes?
3942 cast<ConstantInt>(Operands[1])->getValue() == 4) {
3943 auto *Op = cast<Constant>(Operands[0]);
3944 switch (IntrinsicID) {
3945 default: break;
3946 case Intrinsic::x86_avx512_vcvtss2si32:
3947 case Intrinsic::x86_avx512_vcvtss2si64:
3948 case Intrinsic::x86_avx512_vcvtsd2si32:
3949 case Intrinsic::x86_avx512_vcvtsd2si64:
3950 if (ConstantFP *FPOp =
3951 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3952 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3953 /*roundTowardZero=*/false, Ty,
3954 /*IsSigned*/true);
3955 break;
3956 case Intrinsic::x86_avx512_vcvtss2usi32:
3957 case Intrinsic::x86_avx512_vcvtss2usi64:
3958 case Intrinsic::x86_avx512_vcvtsd2usi32:
3959 case Intrinsic::x86_avx512_vcvtsd2usi64:
3960 if (ConstantFP *FPOp =
3961 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3962 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3963 /*roundTowardZero=*/false, Ty,
3964 /*IsSigned*/false);
3965 break;
3966 case Intrinsic::x86_avx512_cvttss2si:
3967 case Intrinsic::x86_avx512_cvttss2si64:
3968 case Intrinsic::x86_avx512_cvttsd2si:
3969 case Intrinsic::x86_avx512_cvttsd2si64:
3970 if (ConstantFP *FPOp =
3971 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3972 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3973 /*roundTowardZero=*/true, Ty,
3974 /*IsSigned*/true);
3975 break;
3976 case Intrinsic::x86_avx512_cvttss2usi:
3977 case Intrinsic::x86_avx512_cvttss2usi64:
3978 case Intrinsic::x86_avx512_cvttsd2usi:
3979 case Intrinsic::x86_avx512_cvttsd2usi64:
3980 if (ConstantFP *FPOp =
3981 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3982 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3983 /*roundTowardZero=*/true, Ty,
3984 /*IsSigned*/false);
3985 break;
3986 }
3987 }
3988
3989 if (IntrinsicID == Intrinsic::experimental_cttz_elts) {
3990 auto *FVTy = dyn_cast<FixedVectorType>(Operands[0]->getType());
3991 bool ZeroIsPoison = cast<ConstantInt>(Operands[1])->isOne();
3992 if (!FVTy)
3993 return nullptr;
3994 unsigned Width = Ty->getIntegerBitWidth();
3995 if (APInt::getMaxValue(Width).ult(FVTy->getNumElements()))
3996 return PoisonValue::get(Ty);
3997 for (unsigned I = 0; I < FVTy->getNumElements(); ++I) {
3998 Constant *Elt = Operands[0]->getAggregateElement(I);
3999 if (!Elt)
4000 return nullptr;
4001 if (isa<UndefValue>(Elt) || Elt->isNullValue())
4002 continue;
4003 return ConstantInt::get(Ty, I);
4004 }
4005 if (ZeroIsPoison)
4006 return PoisonValue::get(Ty);
4007 return ConstantInt::get(Ty, FVTy->getNumElements());
4008 }
4009 return nullptr;
4010}
4011
4012static APFloat ConstantFoldAMDGCNCubeIntrinsic(Intrinsic::ID IntrinsicID,
4013 const APFloat &S0,
4014 const APFloat &S1,
4015 const APFloat &S2) {
4016 unsigned ID;
4017 const fltSemantics &Sem = S0.getSemantics();
4018 APFloat MA(Sem), SC(Sem), TC(Sem);
4019 if (abs(S2) >= abs(S0) && abs(S2) >= abs(S1)) {
4020 if (S2.isNegative() && S2.isNonZero() && !S2.isNaN()) {
4021 // S2 < 0
4022 ID = 5;
4023 SC = -S0;
4024 } else {
4025 ID = 4;
4026 SC = S0;
4027 }
4028 MA = S2;
4029 TC = -S1;
4030 } else if (abs(S1) >= abs(S0)) {
4031 if (S1.isNegative() && S1.isNonZero() && !S1.isNaN()) {
4032 // S1 < 0
4033 ID = 3;
4034 TC = -S2;
4035 } else {
4036 ID = 2;
4037 TC = S2;
4038 }
4039 MA = S1;
4040 SC = S0;
4041 } else {
4042 if (S0.isNegative() && S0.isNonZero() && !S0.isNaN()) {
4043 // S0 < 0
4044 ID = 1;
4045 SC = S2;
4046 } else {
4047 ID = 0;
4048 SC = -S2;
4049 }
4050 MA = S0;
4051 TC = -S1;
4052 }
4053 switch (IntrinsicID) {
4054 default:
4055 llvm_unreachable("unhandled amdgcn cube intrinsic");
4056 case Intrinsic::amdgcn_cubeid:
4057 return APFloat(Sem, ID);
4058 case Intrinsic::amdgcn_cubema:
4059 return MA + MA;
4060 case Intrinsic::amdgcn_cubesc:
4061 return SC;
4062 case Intrinsic::amdgcn_cubetc:
4063 return TC;
4064 }
4065}
4066
4067static Constant *ConstantFoldAMDGCNPermIntrinsic(ArrayRef<Constant *> Operands,
4068 Type *Ty) {
4069 const APInt *C0, *C1, *C2;
4070 if (!getConstIntOrUndef(Operands[0], C0) ||
4071 !getConstIntOrUndef(Operands[1], C1) ||
4072 !getConstIntOrUndef(Operands[2], C2))
4073 return nullptr;
4074
4075 if (!C2)
4076 return UndefValue::get(Ty);
4077
4078 APInt Val(32, 0);
4079 unsigned NumUndefBytes = 0;
4080 for (unsigned I = 0; I < 32; I += 8) {
4081 unsigned Sel = C2->extractBitsAsZExtValue(8, I);
4082 unsigned B = 0;
4083
4084 if (Sel >= 13)
4085 B = 0xff;
4086 else if (Sel == 12)
4087 B = 0x00;
4088 else {
4089 const APInt *Src = ((Sel & 10) == 10 || (Sel & 12) == 4) ? C0 : C1;
4090 if (!Src)
4091 ++NumUndefBytes;
4092 else if (Sel < 8)
4093 B = Src->extractBitsAsZExtValue(8, (Sel & 3) * 8);
4094 else
4095 B = Src->extractBitsAsZExtValue(1, (Sel & 1) ? 31 : 15) * 0xff;
4096 }
4097
4098 Val.insertBits(B, I, 8);
4099 }
4100
4101 if (NumUndefBytes == 4)
4102 return UndefValue::get(Ty);
4103
4104 return ConstantInt::get(Ty, Val);
4105}
4106
4107static Constant *ConstantFoldScalarCall3(StringRef Name,
4108 Intrinsic::ID IntrinsicID, Type *Ty,
4110 const TargetLibraryInfo *TLI = nullptr,
4111 const CallBase *Call = nullptr) {
4112 assert(Operands.size() == 3 && "Wrong number of operands.");
4113
4114 if (const auto *Op1 = dyn_cast<ConstantFP>(Operands[0])) {
4115 if (const auto *Op2 = dyn_cast<ConstantFP>(Operands[1])) {
4116 if (const auto *Op3 = dyn_cast<ConstantFP>(Operands[2])) {
4117 const APFloat &C1 = Op1->getValueAPF();
4118 const APFloat &C2 = Op2->getValueAPF();
4119 const APFloat &C3 = Op3->getValueAPF();
4120
4121 if (const auto *ConstrIntr =
4123 RoundingMode RM = getEvaluationRoundingMode(ConstrIntr);
4124 APFloat Res = C1;
4126 switch (IntrinsicID) {
4127 default:
4128 return nullptr;
4129 case Intrinsic::experimental_constrained_fma:
4130 case Intrinsic::experimental_constrained_fmuladd:
4131 St = Res.fusedMultiplyAdd(C2, C3, RM);
4132 break;
4133 }
4134 if (mayFoldConstrained(
4135 const_cast<ConstrainedFPIntrinsic *>(ConstrIntr), St))
4136 return ConstantFP::get(Ty, Res);
4137 return nullptr;
4138 }
4139
4140 switch (IntrinsicID) {
4141 default: break;
4142 case Intrinsic::amdgcn_fma_legacy: {
4143 // The legacy behaviour is that multiplying +/- 0.0 by anything, even
4144 // NaN or infinity, gives +0.0.
4145 if (C1.isZero() || C2.isZero()) {
4146 // It's tempting to just return C3 here, but that would give the
4147 // wrong result if C3 was -0.0.
4148 return ConstantFP::get(Ty, APFloat(0.0f) + C3);
4149 }
4150 [[fallthrough]];
4151 }
4152 case Intrinsic::fma:
4153 case Intrinsic::fmuladd: {
4154 APFloat V = C1;
4156 return ConstantFP::get(Ty, V);
4157 }
4158
4159 case Intrinsic::nvvm_fma_rm_f:
4160 case Intrinsic::nvvm_fma_rn_f:
4161 case Intrinsic::nvvm_fma_rp_f:
4162 case Intrinsic::nvvm_fma_rz_f:
4163 case Intrinsic::nvvm_fma_rm_d:
4164 case Intrinsic::nvvm_fma_rn_d:
4165 case Intrinsic::nvvm_fma_rp_d:
4166 case Intrinsic::nvvm_fma_rz_d:
4167 case Intrinsic::nvvm_fma_rm_ftz_f:
4168 case Intrinsic::nvvm_fma_rn_ftz_f:
4169 case Intrinsic::nvvm_fma_rp_ftz_f:
4170 case Intrinsic::nvvm_fma_rz_ftz_f: {
4171 bool IsFTZ = nvvm::FMAShouldFTZ(IntrinsicID);
4172 APFloat A = IsFTZ ? FTZPreserveSign(C1) : C1;
4173 APFloat B = IsFTZ ? FTZPreserveSign(C2) : C2;
4174 APFloat C = IsFTZ ? FTZPreserveSign(C3) : C3;
4175
4176 APFloat::roundingMode RoundMode =
4177 nvvm::GetFMARoundingMode(IntrinsicID);
4178
4179 APFloat Res = A;
4180 APFloat::opStatus Status = Res.fusedMultiplyAdd(B, C, RoundMode);
4181
4182 if (!Res.isNaN() &&
4184 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
4185 return ConstantFP::get(Ty, Res);
4186 }
4187 return nullptr;
4188 }
4189
4190 case Intrinsic::amdgcn_cubeid:
4191 case Intrinsic::amdgcn_cubema:
4192 case Intrinsic::amdgcn_cubesc:
4193 case Intrinsic::amdgcn_cubetc: {
4194 APFloat V = ConstantFoldAMDGCNCubeIntrinsic(IntrinsicID, C1, C2, C3);
4195 return ConstantFP::get(Ty, V);
4196 }
4197 }
4198 }
4199 }
4200 }
4201
4202 if (IntrinsicID == Intrinsic::smul_fix ||
4203 IntrinsicID == Intrinsic::smul_fix_sat) {
4204 const APInt *C0, *C1;
4205 if (!getConstIntOrUndef(Operands[0], C0) ||
4206 !getConstIntOrUndef(Operands[1], C1))
4207 return nullptr;
4208
4209 // undef * C -> 0
4210 // C * undef -> 0
4211 if (!C0 || !C1)
4212 return Constant::getNullValue(Ty);
4213
4214 // This code performs rounding towards negative infinity in case the result
4215 // cannot be represented exactly for the given scale. Targets that do care
4216 // about rounding should use a target hook for specifying how rounding
4217 // should be done, and provide their own folding to be consistent with
4218 // rounding. This is the same approach as used by
4219 // DAGTypeLegalizer::ExpandIntRes_MULFIX.
4220 unsigned Scale = cast<ConstantInt>(Operands[2])->getZExtValue();
4221 unsigned Width = C0->getBitWidth();
4222 assert(Scale < Width && "Illegal scale.");
4223 unsigned ExtendedWidth = Width * 2;
4224 APInt Product =
4225 (C0->sext(ExtendedWidth) * C1->sext(ExtendedWidth)).ashr(Scale);
4226 if (IntrinsicID == Intrinsic::smul_fix_sat) {
4227 APInt Max = APInt::getSignedMaxValue(Width).sext(ExtendedWidth);
4228 APInt Min = APInt::getSignedMinValue(Width).sext(ExtendedWidth);
4229 Product = APIntOps::smin(Product, Max);
4230 Product = APIntOps::smax(Product, Min);
4231 }
4232 return ConstantInt::get(Ty->getContext(), Product.sextOrTrunc(Width));
4233 }
4234
4235 if (IntrinsicID == Intrinsic::fshl || IntrinsicID == Intrinsic::fshr) {
4236 const APInt *C0, *C1, *C2;
4237 if (!getConstIntOrUndef(Operands[0], C0) ||
4238 !getConstIntOrUndef(Operands[1], C1) ||
4239 !getConstIntOrUndef(Operands[2], C2))
4240 return nullptr;
4241
4242 bool IsRight = IntrinsicID == Intrinsic::fshr;
4243 if (!C2)
4244 return Operands[IsRight ? 1 : 0];
4245 if (!C0 && !C1)
4246 return UndefValue::get(Ty);
4247
4248 // The shift amount is interpreted as modulo the bitwidth. If the shift
4249 // amount is effectively 0, avoid UB due to oversized inverse shift below.
4250 unsigned BitWidth = C2->getBitWidth();
4251 unsigned ShAmt = C2->urem(BitWidth);
4252 if (!ShAmt)
4253 return Operands[IsRight ? 1 : 0];
4254
4255 // (C0 << ShlAmt) | (C1 >> LshrAmt)
4256 unsigned LshrAmt = IsRight ? ShAmt : BitWidth - ShAmt;
4257 unsigned ShlAmt = !IsRight ? ShAmt : BitWidth - ShAmt;
4258 if (!C0)
4259 return ConstantInt::get(Ty, C1->lshr(LshrAmt));
4260 if (!C1)
4261 return ConstantInt::get(Ty, C0->shl(ShlAmt));
4262 return ConstantInt::get(Ty, C0->shl(ShlAmt) | C1->lshr(LshrAmt));
4263 }
4264
4265 if (IntrinsicID == Intrinsic::amdgcn_perm)
4266 return ConstantFoldAMDGCNPermIntrinsic(Operands, Ty);
4267
4268 return nullptr;
4269}
4270
4271static Constant *ConstantFoldScalarCall(StringRef Name,
4272 Intrinsic::ID IntrinsicID, Type *Ty,
4274 const TargetLibraryInfo *TLI = nullptr,
4275 const CallBase *Call = nullptr) {
4276 if (IntrinsicID != Intrinsic::not_intrinsic &&
4278 intrinsicPropagatesPoison(IntrinsicID))
4279 return PoisonValue::get(Ty);
4280
4281 if (Operands.size() == 1)
4282 return ConstantFoldScalarCall1(Name, IntrinsicID, Ty, Operands, TLI, Call);
4283
4284 if (Operands.size() == 2) {
4285 if (Constant *FoldedLibCall =
4286 ConstantFoldLibCall2(Name, Ty, Operands, TLI)) {
4287 return FoldedLibCall;
4288 }
4289 return ConstantFoldIntrinsicCall2(IntrinsicID, Ty, Operands, Call);
4290 }
4291
4292 if (Operands.size() == 3)
4293 return ConstantFoldScalarCall3(Name, IntrinsicID, Ty, Operands, TLI, Call);
4294
4295 return nullptr;
4296}
4297
4298static Constant *ConstantFoldFixedVectorCall(
4299 StringRef Name, Intrinsic::ID IntrinsicID, FixedVectorType *FVTy,
4301 const TargetLibraryInfo *TLI = nullptr, const CallBase *Call = nullptr) {
4304 Type *Ty = FVTy->getElementType();
4305
4306 switch (IntrinsicID) {
4307 case Intrinsic::masked_load: {
4308 auto *SrcPtr = Operands[0];
4309 auto *Mask = Operands[1];
4310 auto *Passthru = Operands[2];
4311
4312 Constant *VecData = ConstantFoldLoadFromConstPtr(SrcPtr, FVTy, DL);
4313
4314 SmallVector<Constant *, 32> NewElements;
4315 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4316 auto *MaskElt = Mask->getAggregateElement(I);
4317 if (!MaskElt)
4318 break;
4319 auto *PassthruElt = Passthru->getAggregateElement(I);
4320 auto *VecElt = VecData ? VecData->getAggregateElement(I) : nullptr;
4321 if (isa<UndefValue>(MaskElt)) {
4322 if (PassthruElt)
4323 NewElements.push_back(PassthruElt);
4324 else if (VecElt)
4325 NewElements.push_back(VecElt);
4326 else
4327 return nullptr;
4328 }
4329 if (MaskElt->isNullValue()) {
4330 if (!PassthruElt)
4331 return nullptr;
4332 NewElements.push_back(PassthruElt);
4333 } else if (MaskElt->isOneValue()) {
4334 if (!VecElt)
4335 return nullptr;
4336 NewElements.push_back(VecElt);
4337 } else {
4338 return nullptr;
4339 }
4340 }
4341 if (NewElements.size() != FVTy->getNumElements())
4342 return nullptr;
4343 return ConstantVector::get(NewElements);
4344 }
4345 case Intrinsic::arm_mve_vctp8:
4346 case Intrinsic::arm_mve_vctp16:
4347 case Intrinsic::arm_mve_vctp32:
4348 case Intrinsic::arm_mve_vctp64: {
4349 if (auto *Op = dyn_cast<ConstantInt>(Operands[0])) {
4350 unsigned Lanes = FVTy->getNumElements();
4351 uint64_t Limit = Op->getZExtValue();
4352
4354 for (unsigned i = 0; i < Lanes; i++) {
4355 if (i < Limit)
4357 else
4359 }
4360 return ConstantVector::get(NCs);
4361 }
4362 return nullptr;
4363 }
4364 case Intrinsic::get_active_lane_mask: {
4365 auto *Op0 = dyn_cast<ConstantInt>(Operands[0]);
4366 auto *Op1 = dyn_cast<ConstantInt>(Operands[1]);
4367 if (Op0 && Op1) {
4368 unsigned Lanes = FVTy->getNumElements();
4369 APInt Base = Op0->getValue();
4370 APInt Limit = Op1->getValue();
4371
4373 for (unsigned I = 0; I < Lanes; I++) {
4374 bool Overflow;
4375 if (Base.uadd_ov(APInt(Base.getBitWidth(), I), Overflow).ult(Limit) &&
4376 !Overflow)
4378 else
4380 }
4381 return ConstantVector::get(NCs);
4382 }
4383 return nullptr;
4384 }
4385 case Intrinsic::vector_extract: {
4386 auto *Idx = dyn_cast<ConstantInt>(Operands[1]);
4387 Constant *Vec = Operands[0];
4388 if (!Idx || !isa<FixedVectorType>(Vec->getType()))
4389 return nullptr;
4390
4391 unsigned NumElements = FVTy->getNumElements();
4392 unsigned VecNumElements =
4393 cast<FixedVectorType>(Vec->getType())->getNumElements();
4394 unsigned StartingIndex = Idx->getZExtValue();
4395
4396 // Extracting entire vector is nop
4397 if (NumElements == VecNumElements && StartingIndex == 0)
4398 return Vec;
4399
4400 for (unsigned I = StartingIndex, E = StartingIndex + NumElements; I < E;
4401 ++I) {
4402 Constant *Elt = Vec->getAggregateElement(I);
4403 if (!Elt)
4404 return nullptr;
4405 Result[I - StartingIndex] = Elt;
4406 }
4407
4408 return ConstantVector::get(Result);
4409 }
4410 case Intrinsic::vector_insert: {
4411 Constant *Vec = Operands[0];
4412 Constant *SubVec = Operands[1];
4413 auto *Idx = dyn_cast<ConstantInt>(Operands[2]);
4414 if (!Idx || !isa<FixedVectorType>(Vec->getType()))
4415 return nullptr;
4416
4417 unsigned SubVecNumElements =
4418 cast<FixedVectorType>(SubVec->getType())->getNumElements();
4419 unsigned VecNumElements =
4420 cast<FixedVectorType>(Vec->getType())->getNumElements();
4421 unsigned IdxN = Idx->getZExtValue();
4422 // Replacing entire vector with a subvec is nop
4423 if (SubVecNumElements == VecNumElements && IdxN == 0)
4424 return SubVec;
4425
4426 for (unsigned I = 0; I < VecNumElements; ++I) {
4427 Constant *Elt;
4428 if (I < IdxN + SubVecNumElements)
4429 Elt = SubVec->getAggregateElement(I - IdxN);
4430 else
4431 Elt = Vec->getAggregateElement(I);
4432 if (!Elt)
4433 return nullptr;
4434 Result[I] = Elt;
4435 }
4436 return ConstantVector::get(Result);
4437 }
4438 case Intrinsic::vector_interleave2:
4439 case Intrinsic::vector_interleave3:
4440 case Intrinsic::vector_interleave4:
4441 case Intrinsic::vector_interleave5:
4442 case Intrinsic::vector_interleave6:
4443 case Intrinsic::vector_interleave7:
4444 case Intrinsic::vector_interleave8: {
4445 unsigned NumElements =
4446 cast<FixedVectorType>(Operands[0]->getType())->getNumElements();
4447 unsigned NumOperands = Operands.size();
4448 for (unsigned I = 0; I < NumElements; ++I) {
4449 for (unsigned J = 0; J < NumOperands; ++J) {
4450 Constant *Elt = Operands[J]->getAggregateElement(I);
4451 if (!Elt)
4452 return nullptr;
4453 Result[NumOperands * I + J] = Elt;
4454 }
4455 }
4456 return ConstantVector::get(Result);
4457 }
4458 case Intrinsic::wasm_dot: {
4459 unsigned NumElements =
4460 cast<FixedVectorType>(Operands[0]->getType())->getNumElements();
4461
4462 assert(NumElements == 8 && Result.size() == 4 &&
4463 "wasm dot takes i16x8 and produces i32x4");
4464 assert(Ty->isIntegerTy());
4465 int32_t MulVector[8];
4466
4467 for (unsigned I = 0; I < NumElements; ++I) {
4468 ConstantInt *Elt0 =
4469 dyn_cast<ConstantInt>(Operands[0]->getAggregateElement(I));
4470 ConstantInt *Elt1 =
4471 dyn_cast<ConstantInt>(Operands[1]->getAggregateElement(I));
4472
4473 if (!Elt0 || !Elt1)
4474 return nullptr;
4475
4476 MulVector[I] = Elt0->getSExtValue() * Elt1->getSExtValue();
4477 }
4478 for (unsigned I = 0; I < Result.size(); I++) {
4479 int64_t IAdd = (int64_t)MulVector[I * 2] + (int64_t)MulVector[I * 2 + 1];
4480 Result[I] = ConstantInt::getSigned(Ty, IAdd, /*ImplicitTrunc=*/true);
4481 }
4482
4483 return ConstantVector::get(Result);
4484 }
4485 default:
4486 break;
4487 }
4488
4489 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4490 // Gather a column of constants.
4491 for (unsigned J = 0, JE = Operands.size(); J != JE; ++J) {
4492 // Some intrinsics use a scalar type for certain arguments.
4493 if (isVectorIntrinsicWithScalarOpAtArg(IntrinsicID, J, /*TTI=*/nullptr)) {
4494 Lane[J] = Operands[J];
4495 continue;
4496 }
4497
4498 Constant *Agg = Operands[J]->getAggregateElement(I);
4499 if (!Agg)
4500 return nullptr;
4501
4502 Lane[J] = Agg;
4503 }
4504
4505 // Use the regular scalar folding to simplify this column.
4506 Constant *Folded =
4507 ConstantFoldScalarCall(Name, IntrinsicID, Ty, Lane, TLI, Call);
4508 if (!Folded)
4509 return nullptr;
4510 Result[I] = Folded;
4511 }
4512
4513 return ConstantVector::get(Result);
4514}
4515
4516static Constant *ConstantFoldScalableVectorCall(
4517 StringRef Name, Intrinsic::ID IntrinsicID, ScalableVectorType *SVTy,
4519 const TargetLibraryInfo *TLI, const CallBase *Call) {
4520 switch (IntrinsicID) {
4521 case Intrinsic::aarch64_sve_convert_from_svbool: {
4522 Constant *Src = Operands[0];
4523 if (!Src->isNullValue())
4524 break;
4525
4526 return ConstantInt::getFalse(SVTy);
4527 }
4528 case Intrinsic::get_active_lane_mask: {
4529 auto *Op0 = dyn_cast<ConstantInt>(Operands[0]);
4530 auto *Op1 = dyn_cast<ConstantInt>(Operands[1]);
4531 if (Op0 && Op1 && Op0->getValue().uge(Op1->getValue()))
4532 return ConstantVector::getNullValue(SVTy);
4533 break;
4534 }
4535 case Intrinsic::vector_interleave2:
4536 case Intrinsic::vector_interleave3:
4537 case Intrinsic::vector_interleave4:
4538 case Intrinsic::vector_interleave5:
4539 case Intrinsic::vector_interleave6:
4540 case Intrinsic::vector_interleave7:
4541 case Intrinsic::vector_interleave8: {
4542 Constant *SplatVal = Operands[0]->getSplatValue();
4543 if (!SplatVal)
4544 return nullptr;
4545
4547 return nullptr;
4548
4549 return ConstantVector::getSplat(SVTy->getElementCount(), SplatVal);
4550 }
4551 default:
4552 break;
4553 }
4554
4555 // If trivially vectorizable, try folding it via the scalar call if all
4556 // operands are splats.
4557
4558 // TODO: ConstantFoldFixedVectorCall should probably check this too?
4559 if (!isTriviallyVectorizable(IntrinsicID))
4560 return nullptr;
4561
4563 for (auto [I, Op] : enumerate(Operands)) {
4564 if (isVectorIntrinsicWithScalarOpAtArg(IntrinsicID, I, /*TTI=*/nullptr)) {
4565 SplatOps.push_back(Op);
4566 continue;
4567 }
4568 Constant *Splat = Op->getSplatValue();
4569 if (!Splat)
4570 return nullptr;
4571 SplatOps.push_back(Splat);
4572 }
4573 Constant *Folded = ConstantFoldScalarCall(
4574 Name, IntrinsicID, SVTy->getElementType(), SplatOps, TLI, Call);
4575 if (!Folded)
4576 return nullptr;
4577 return ConstantVector::getSplat(SVTy->getElementCount(), Folded);
4578}
4579
4580static std::pair<Constant *, Constant *>
4581ConstantFoldScalarFrexpCall(Constant *Op, Type *IntTy) {
4582 auto *ConstFP = dyn_cast<ConstantFP>(Op);
4583 if (!ConstFP)
4584 return {};
4585
4586 const APFloat &U = ConstFP->getValueAPF();
4587 int FrexpExp;
4588 APFloat FrexpMant = frexp(U, FrexpExp, APFloat::rmNearestTiesToEven);
4589 Constant *Result0 = ConstantFP::get(ConstFP->getType(), FrexpMant);
4590
4591 // The exponent is an "unspecified value" for inf/nan. We use zero to avoid
4592 // using undef.
4593 Constant *Result1 = FrexpMant.isFinite()
4594 ? ConstantInt::getSigned(IntTy, FrexpExp)
4595 : ConstantInt::getNullValue(IntTy);
4596 return {Result0, Result1};
4597}
4598
4599/// Handle intrinsics that return tuples, which may be tuples of vectors.
4600static Constant *
4601ConstantFoldStructCall(StringRef Name, Intrinsic::ID IntrinsicID,
4603 const DataLayout &DL, const TargetLibraryInfo *TLI,
4604 const CallBase *Call) {
4605
4606 switch (IntrinsicID) {
4607 case Intrinsic::frexp: {
4608 Type *Ty0 = StTy->getContainedType(0);
4609 Type *Ty1 = StTy->getContainedType(1)->getScalarType();
4610
4611 if (auto *FVTy0 = dyn_cast<FixedVectorType>(Ty0)) {
4612 SmallVector<Constant *, 4> Results0(FVTy0->getNumElements());
4613 SmallVector<Constant *, 4> Results1(FVTy0->getNumElements());
4614
4615 for (unsigned I = 0, E = FVTy0->getNumElements(); I != E; ++I) {
4616 Constant *Lane = Operands[0]->getAggregateElement(I);
4617 std::tie(Results0[I], Results1[I]) =
4618 ConstantFoldScalarFrexpCall(Lane, Ty1);
4619 if (!Results0[I])
4620 return nullptr;
4621 }
4622
4623 return ConstantStruct::get(StTy, ConstantVector::get(Results0),
4624 ConstantVector::get(Results1));
4625 }
4626
4627 auto [Result0, Result1] = ConstantFoldScalarFrexpCall(Operands[0], Ty1);
4628 if (!Result0)
4629 return nullptr;
4630 return ConstantStruct::get(StTy, Result0, Result1);
4631 }
4632 case Intrinsic::sincos: {
4633 Type *Ty = StTy->getContainedType(0);
4634 Type *TyScalar = Ty->getScalarType();
4635
4636 auto ConstantFoldScalarSincosCall =
4637 [&](Constant *Op) -> std::pair<Constant *, Constant *> {
4638 Constant *SinResult =
4639 ConstantFoldScalarCall(Name, Intrinsic::sin, TyScalar, Op, TLI, Call);
4640 Constant *CosResult =
4641 ConstantFoldScalarCall(Name, Intrinsic::cos, TyScalar, Op, TLI, Call);
4642 return std::make_pair(SinResult, CosResult);
4643 };
4644
4645 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty)) {
4646 SmallVector<Constant *> SinResults(FVTy->getNumElements());
4647 SmallVector<Constant *> CosResults(FVTy->getNumElements());
4648
4649 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4650 Constant *Lane = Operands[0]->getAggregateElement(I);
4651 std::tie(SinResults[I], CosResults[I]) =
4652 ConstantFoldScalarSincosCall(Lane);
4653 if (!SinResults[I] || !CosResults[I])
4654 return nullptr;
4655 }
4656
4657 return ConstantStruct::get(StTy, ConstantVector::get(SinResults),
4658 ConstantVector::get(CosResults));
4659 }
4660
4661 if (!Ty->isFloatingPointTy())
4662 return nullptr;
4663
4664 auto [SinResult, CosResult] = ConstantFoldScalarSincosCall(Operands[0]);
4665 if (!SinResult || !CosResult)
4666 return nullptr;
4667 return ConstantStruct::get(StTy, SinResult, CosResult);
4668 }
4669 case Intrinsic::vector_deinterleave2:
4670 case Intrinsic::vector_deinterleave3:
4671 case Intrinsic::vector_deinterleave4:
4672 case Intrinsic::vector_deinterleave5:
4673 case Intrinsic::vector_deinterleave6:
4674 case Intrinsic::vector_deinterleave7:
4675 case Intrinsic::vector_deinterleave8: {
4676 unsigned NumResults = StTy->getNumElements();
4677 auto *Vec = Operands[0];
4678 auto *VecTy = cast<VectorType>(Vec->getType());
4679
4680 ElementCount ResultEC =
4681 VecTy->getElementCount().divideCoefficientBy(NumResults);
4682
4683 if (auto *EltC = Vec->getSplatValue()) {
4684 auto *ResultVec = ConstantVector::getSplat(ResultEC, EltC);
4685 SmallVector<Constant *, 8> Results(NumResults, ResultVec);
4686 return ConstantStruct::get(StTy, Results);
4687 }
4688
4689 if (!ResultEC.isFixed())
4690 return nullptr;
4691
4692 unsigned NumElements = ResultEC.getFixedValue();
4694 SmallVector<Constant *> Elements(NumElements);
4695 for (unsigned I = 0; I != NumResults; ++I) {
4696 for (unsigned J = 0; J != NumElements; ++J) {
4697 Constant *Elt = Vec->getAggregateElement(J * NumResults + I);
4698 if (!Elt)
4699 return nullptr;
4700 Elements[J] = Elt;
4701 }
4702 Results[I] = ConstantVector::get(Elements);
4703 }
4704 return ConstantStruct::get(StTy, Results);
4705 }
4706 default:
4707 // TODO: Constant folding of vector intrinsics that fall through here does
4708 // not work (e.g. overflow intrinsics)
4709 return ConstantFoldScalarCall(Name, IntrinsicID, StTy, Operands, TLI, Call);
4710 }
4711
4712 return nullptr;
4713}
4714
4715} // end anonymous namespace
4716
4719 const DataLayout &DL, Function *CxtF) {
4720 // In the absence of CxtF, assume strictfp conservatively.
4721 if (!canConstantFoldIntrinsic(ID, CxtF ? CxtF->isStrictFP() : true) ||
4724 Ty, ArrayRef<Value *>((Value *const *)Ops.data(), Ops.size()))))
4725 return nullptr;
4726 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
4727 return ConstantFoldFixedVectorCall("", ID, FVTy, Ops, DL);
4728 return ConstantFoldScalarCall("", ID, Ty, Ops);
4729}
4730
4733 const TargetLibraryInfo *TLI,
4734 bool AllowNonDeterministic) {
4735 if (Call->isNoBuiltin())
4736 return nullptr;
4737 if (!F->hasName())
4738 return nullptr;
4739
4740 // If this is not an intrinsic and not recognized as a library call, bail out.
4741 Intrinsic::ID IID = F->getIntrinsicID();
4742 if (IID == Intrinsic::not_intrinsic) {
4743 if (!TLI)
4744 return nullptr;
4745 if (TLI->getLibFunc(*F) == NotLibFunc)
4746 return nullptr;
4747 }
4748
4749 // Conservatively assume that floating-point libcalls may be
4750 // non-deterministic.
4751 Type *Ty = F->getReturnType();
4752 if (!AllowNonDeterministic && Ty->isFPOrFPVectorTy())
4753 return nullptr;
4754
4755 StringRef Name = F->getName();
4756 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
4757 return ConstantFoldFixedVectorCall(
4758 Name, IID, FVTy, Operands, F->getDataLayout(), TLI, Call);
4759
4760 if (auto *SVTy = dyn_cast<ScalableVectorType>(Ty))
4761 return ConstantFoldScalableVectorCall(
4762 Name, IID, SVTy, Operands, F->getDataLayout(), TLI, Call);
4763
4764 if (auto *StTy = dyn_cast<StructType>(Ty))
4765 return ConstantFoldStructCall(Name, IID, StTy, Operands,
4766 F->getDataLayout(), TLI, Call);
4767
4768 // TODO: If this is a library function, we already discovered that above,
4769 // so we should pass the LibFunc, not the name (and it might be better
4770 // still to separate intrinsic handling from libcalls).
4771 return ConstantFoldScalarCall(Name, IID, Ty, Operands, TLI, Call);
4772}
4773
4775 const TargetLibraryInfo *TLI) {
4776 // FIXME: Refactor this code; this duplicates logic in LibCallsShrinkWrap
4777 // (and to some extent ConstantFoldScalarCall).
4778 if (Call->isNoBuiltin() || Call->isStrictFP())
4779 return false;
4780 Function *F = Call->getCalledFunction();
4781 if (!F)
4782 return false;
4783
4784 if (!TLI)
4785 return false;
4786
4787 LibFunc Func = TLI->getLibFunc(*F);
4788 if (Func == NotLibFunc)
4789 return false;
4790
4791 if (Call->arg_size() == 1) {
4792 if (ConstantFP *OpC = dyn_cast<ConstantFP>(Call->getArgOperand(0))) {
4793 const APFloat &Op = OpC->getValueAPF();
4794 switch (Func) {
4795 case LibFunc_logl:
4796 case LibFunc_log:
4797 case LibFunc_logf:
4798 case LibFunc_log2l:
4799 case LibFunc_log2:
4800 case LibFunc_log2f:
4801 case LibFunc_log10l:
4802 case LibFunc_log10:
4803 case LibFunc_log10f:
4804 return Op.isNaN() || (!Op.isZero() && !Op.isNegative());
4805
4806 case LibFunc_ilogb:
4807 return !Op.isNaN() && !Op.isZero() && !Op.isInfinity();
4808
4809 case LibFunc_expl:
4810 case LibFunc_exp:
4811 case LibFunc_expf:
4812 // FIXME: These boundaries are slightly conservative.
4813 if (OpC->getType()->isDoubleTy())
4814 return !(Op < APFloat(-745.0) || Op > APFloat(709.0));
4815 if (OpC->getType()->isFloatTy())
4816 return !(Op < APFloat(-103.0f) || Op > APFloat(88.0f));
4817 break;
4818
4819 case LibFunc_exp2l:
4820 case LibFunc_exp2:
4821 case LibFunc_exp2f:
4822 // FIXME: These boundaries are slightly conservative.
4823 if (OpC->getType()->isDoubleTy())
4824 return !(Op < APFloat(-1074.0) || Op > APFloat(1023.0));
4825 if (OpC->getType()->isFloatTy())
4826 return !(Op < APFloat(-149.0f) || Op > APFloat(127.0f));
4827 break;
4828
4829 case LibFunc_sinl:
4830 case LibFunc_sin:
4831 case LibFunc_sinf:
4832 case LibFunc_cosl:
4833 case LibFunc_cos:
4834 case LibFunc_cosf:
4835 return !Op.isInfinity();
4836
4837 case LibFunc_tanl:
4838 case LibFunc_tan:
4839 case LibFunc_tanf: {
4840 // FIXME: Stop using the host math library.
4841 // FIXME: The computation isn't done in the right precision.
4842 Type *Ty = OpC->getType();
4843 if (Ty->isDoubleTy() || Ty->isFloatTy() || Ty->isHalfTy())
4844 return ConstantFoldFP(tan, OpC->getValueAPF(), Ty) != nullptr;
4845 break;
4846 }
4847
4848 case LibFunc_atan:
4849 case LibFunc_atanf:
4850 case LibFunc_atanl:
4851 // Per POSIX, this MAY fail if Op is denormal. We choose not failing.
4852 return true;
4853
4854 case LibFunc_asinl:
4855 case LibFunc_asin:
4856 case LibFunc_asinf:
4857 case LibFunc_acosl:
4858 case LibFunc_acos:
4859 case LibFunc_acosf:
4860 return !(Op < APFloat::getOne(Op.getSemantics(), true) ||
4861 Op > APFloat::getOne(Op.getSemantics()));
4862
4863 case LibFunc_sinh:
4864 case LibFunc_cosh:
4865 case LibFunc_sinhf:
4866 case LibFunc_coshf:
4867 case LibFunc_sinhl:
4868 case LibFunc_coshl:
4869 // FIXME: These boundaries are slightly conservative.
4870 if (OpC->getType()->isDoubleTy())
4871 return !(Op < APFloat(-710.0) || Op > APFloat(710.0));
4872 if (OpC->getType()->isFloatTy())
4873 return !(Op < APFloat(-89.0f) || Op > APFloat(89.0f));
4874 break;
4875
4876 case LibFunc_sqrtl:
4877 case LibFunc_sqrt:
4878 case LibFunc_sqrtf:
4879 return Op.isNaN() || Op.isZero() || !Op.isNegative();
4880
4881 // FIXME: Add more functions: sqrt_finite, atanh, expm1, log1p,
4882 // maybe others?
4883 default:
4884 break;
4885 }
4886 }
4887 }
4888
4889 if (Call->arg_size() == 2) {
4890 ConstantFP *Op0C = dyn_cast<ConstantFP>(Call->getArgOperand(0));
4891 ConstantFP *Op1C = dyn_cast<ConstantFP>(Call->getArgOperand(1));
4892 if (Op0C && Op1C) {
4893 const APFloat &Op0 = Op0C->getValueAPF();
4894 const APFloat &Op1 = Op1C->getValueAPF();
4895
4896 switch (Func) {
4897 case LibFunc_powl:
4898 case LibFunc_pow:
4899 case LibFunc_powf: {
4900 // FIXME: Stop using the host math library.
4901 // FIXME: The computation isn't done in the right precision.
4902 Type *Ty = Op0C->getType();
4903 if (Ty->isDoubleTy() || Ty->isFloatTy() || Ty->isHalfTy()) {
4904 if (Ty == Op1C->getType())
4905 return ConstantFoldBinaryFP(pow, Op0, Op1, Ty) != nullptr;
4906 }
4907 break;
4908 }
4909
4910 case LibFunc_fmodl:
4911 case LibFunc_fmod:
4912 case LibFunc_fmodf:
4913 case LibFunc_remainderl:
4914 case LibFunc_remainder:
4915 case LibFunc_remainderf:
4916 return Op0.isNaN() || Op1.isNaN() ||
4917 (!Op0.isInfinity() && !Op1.isZero());
4918
4919 case LibFunc_atan2:
4920 case LibFunc_atan2f:
4921 case LibFunc_atan2l:
4922 // Although IEEE-754 says atan2(+/-0.0, +/-0.0) are well-defined, and
4923 // GLIBC and MSVC do not appear to raise an error on those, we
4924 // cannot rely on that behavior. POSIX and C11 say that a domain error
4925 // may occur, so allow for that possibility.
4926 return !Op0.isZero() || !Op1.isZero();
4927
4928 case LibFunc_nextafter:
4929 case LibFunc_nextafterf:
4930 case LibFunc_nextafterl:
4931 case LibFunc_nexttoward:
4932 case LibFunc_nexttowardf:
4933 case LibFunc_nexttowardl: {
4934 return ConstantFoldNextToward(Op0, Op1, F->getReturnType()) != nullptr;
4935 }
4936 default:
4937 break;
4938 }
4939 }
4940 }
4941
4942 return false;
4943}
4944
4946 unsigned CastOp, const DataLayout &DL,
4947 PreservedCastFlags *Flags) {
4948 switch (CastOp) {
4949 case Instruction::BitCast:
4950 // Bitcast is always lossless.
4951 return ConstantFoldCastOperand(Instruction::BitCast, C, InvCastTo, DL);
4952 case Instruction::Trunc: {
4953 auto *ZExtC = ConstantFoldCastOperand(Instruction::ZExt, C, InvCastTo, DL);
4954 if (Flags) {
4955 // Truncation back on ZExt value is always NUW.
4956 Flags->NUW = true;
4957 // Test positivity of C.
4958 auto *SExtC =
4959 ConstantFoldCastOperand(Instruction::SExt, C, InvCastTo, DL);
4960 Flags->NSW = ZExtC == SExtC;
4961 }
4962 return ZExtC;
4963 }
4964 case Instruction::SExt:
4965 case Instruction::ZExt: {
4966 auto *InvC = ConstantExpr::getTrunc(C, InvCastTo);
4967 auto *CastInvC = ConstantFoldCastOperand(CastOp, InvC, C->getType(), DL);
4968 // Must satisfy CastOp(InvC) == C.
4969 if (!CastInvC || CastInvC != C)
4970 return nullptr;
4971 if (Flags && CastOp == Instruction::ZExt) {
4972 auto *SExtInvC =
4973 ConstantFoldCastOperand(Instruction::SExt, InvC, C->getType(), DL);
4974 // Test positivity of InvC.
4975 Flags->NNeg = CastInvC == SExtInvC;
4976 }
4977 return InvC;
4978 }
4979 case Instruction::FPExt: {
4980 Constant *InvC =
4981 ConstantFoldCastOperand(Instruction::FPTrunc, C, InvCastTo, DL);
4982 if (InvC) {
4983 Constant *CastInvC =
4984 ConstantFoldCastOperand(CastOp, InvC, C->getType(), DL);
4985 if (CastInvC == C)
4986 return InvC;
4987 }
4988 return nullptr;
4989 }
4990 default:
4991 return nullptr;
4992 }
4993}
4994
4996 const DataLayout &DL,
4997 PreservedCastFlags *Flags) {
4998 return getLosslessInvCast(C, DestTy, Instruction::ZExt, DL, Flags);
4999}
5000
5002 const DataLayout &DL,
5003 PreservedCastFlags *Flags) {
5004 return getLosslessInvCast(C, DestTy, Instruction::SExt, DL, Flags);
5005}
5006
5007void TargetFolder::anchor() {}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
constexpr LLT S1
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static Constant * FoldBitCast(Constant *V, Type *DestTy)
static ConstantFP * flushDenormalConstant(Type *Ty, const APFloat &APF, DenormalMode::DenormalModeKind Mode)
Constant * getConstantAtOffset(Constant *Base, APInt Offset, const DataLayout &DL)
If this Offset points exactly to the start of an aggregate element, return that element,...
static cl::opt< bool > DisableFPCallFolding("disable-fp-call-folding", cl::desc("Disable constant-folding of FP intrinsics and libcalls."), cl::init(false), cl::Hidden)
static bool canConstantFoldIntrinsic(Intrinsic::ID ID, bool IsStrictFP)
Returns true if the intrinsic can be constant folded, given IsStrictFP.
static ConstantFP * flushDenormalConstantFP(ConstantFP *CFP, const Instruction *Inst, bool IsOutput)
static bool anyTypeContainsFP(Type *RetTy, ArrayRef< Value * > Ops)
Given a function's return type and its operands, determine if any of them of of floating-point type.
static DenormalMode getInstrDenormalMode(const Instruction *CtxI, Type *Ty)
Return the denormal mode that can be assumed when executing a floating point operation at CtxI.
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file defines the DenseMap class.
Hexagon Common GEP
amode Optimize addressing mode
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static bool InRange(int64_t Value, unsigned short Shift, int LBound, int HBound)
This file contains the definitions of the enumerations and flags associated with NVVM Intrinsics,...
if(PassOpts->AAPipeline)
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
This file implements the SmallBitVector class.
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
The Input class is used to parse a yaml document into in-memory structs and vectors.
cmpResult
IEEE-754R 5.11: Floating Point Comparison Relations.
Definition APFloat.h:343
static constexpr roundingMode rmTowardZero
Definition APFloat.h:357
llvm::RoundingMode roundingMode
IEEE-754R 4.3: Rounding-direction attributes.
Definition APFloat.h:351
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmTowardNegative
Definition APFloat.h:356
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:353
static constexpr roundingMode rmTowardPositive
Definition APFloat.h:355
static constexpr roundingMode rmNearestTiesToAway
Definition APFloat.h:358
opStatus
IEEE-754R 7: Default exception handling.
Definition APFloat.h:369
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
Definition APFloat.h:1216
opStatus divide(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1304
void copySign(const APFloat &RHS)
Definition APFloat.h:1398
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Definition APFloat.cpp:5946
opStatus subtract(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1286
bool isNegative() const
Definition APFloat.h:1575
LLVM_ABI double convertToDouble() const
Converts this APFloat to host double value.
Definition APFloat.cpp:6005
bool isPosInfinity() const
Definition APFloat.h:1588
bool isNormal() const
Definition APFloat.h:1579
bool isDenormal() const
Definition APFloat.h:1576
opStatus add(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1277
const fltSemantics & getSemantics() const
Definition APFloat.h:1583
bool isNonZero() const
Definition APFloat.h:1584
bool isFinite() const
Definition APFloat.h:1580
bool isNaN() const
Definition APFloat.h:1573
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1184
opStatus multiply(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1295
bool isSignaling() const
Definition APFloat.h:1577
opStatus fusedMultiplyAdd(const APFloat &Multiplicand, const APFloat &Addend, roundingMode RM)
Definition APFloat.h:1331
bool isZero() const
Definition APFloat.h:1571
opStatus convertToInteger(MutableArrayRef< integerPart > Input, unsigned int Width, bool IsSigned, roundingMode RM, bool *IsExact) const
Definition APFloat.h:1428
opStatus mod(const APFloat &RHS)
Definition APFloat.h:1322
bool isNegInfinity() const
Definition APFloat.h:1589
opStatus roundToIntegral(roundingMode RM)
Definition APFloat.h:1344
void changeSign()
Definition APFloat.h:1393
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1175
bool isInfinity() const
Definition APFloat.h:1572
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt umul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:2007
LLVM_ABI APInt usub_sat(const APInt &RHS) const
Definition APInt.cpp:2091
bool isMinSignedValue() const
Determine if this is the smallest signed value.
Definition APInt.h:420
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1561
LLVM_ABI uint64_t extractBitsAsZExtValue(unsigned numBits, unsigned bitPosition) const
Definition APInt.cpp:516
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1077
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:203
APInt abs() const
Get the absolute value.
Definition APInt.h:1816
LLVM_ABI APInt sadd_sat(const APInt &RHS) const
Definition APInt.cpp:2062
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1206
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1984
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1187
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:377
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1693
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1509
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1116
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
Definition APInt.h:206
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1964
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1971
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1660
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1619
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:216
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1085
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2072
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:830
LLVM_ABI APInt smul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1996
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
Definition APInt.cpp:1029
APInt shl(unsigned shiftAmt) const
Left-shift function.
Definition APInt.h:876
bool slt(const APInt &RHS) const
Signed less than comparison.
Definition APInt.h:1135
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:197
LLVM_ABI APInt extractBits(unsigned numBits, unsigned bitPosition) const
Return an APInt with the extracted bits [bitPosition,bitPosition+numBits).
Definition APInt.cpp:478
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1977
bool isOne() const
Determine if this is a value of 1.
Definition APInt.h:386
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:854
LLVM_ABI APInt ssub_sat(const APInt &RHS) const
Definition APInt.cpp:2081
An arbitrary precision integer that knows its signedness.
Definition APSInt.h:24
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI unsigned isEliminableCastPair(Instruction::CastOps firstOpcode, Instruction::CastOps secondOpcode, Type *SrcTy, Type *MidTy, Type *DstTy, const DataLayout *DL)
Determine how a pair of casts can be eliminated, if they can be at all.
static LLVM_ABI bool castIsValid(Instruction::CastOps op, Type *SrcTy, Type *DstTy)
This method can be used to determine if a cast from SrcTy to DstTy using Opcode op is valid or not.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
bool isSigned() const
Definition InstrTypes.h:993
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
Definition Constants.h:878
static LLVM_ABI Constant * getIntToPtr(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getExtractElement(Constant *Vec, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI bool isDesirableCastOp(unsigned Opcode)
Whether creating a constant expression for this cast is desirable.
static LLVM_ABI Constant * getCast(unsigned ops, Constant *C, Type *Ty, bool OnlyIfReduced=false)
Convenience function for getting a Cast operation.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static Constant * getPtrAdd(Constant *Ptr, Constant *Offset, GEPNoWrapFlags NW=GEPNoWrapFlags::none(), std::optional< ConstantRange > InRange=std::nullopt, Type *OnlyIfReduced=nullptr)
Create a getelementptr i8, ptr, offset constant expression.
Definition Constants.h:1497
static LLVM_ABI Constant * getInsertElement(Constant *Vec, Constant *Elt, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI Constant * getShuffleVector(Constant *V1, Constant *V2, ArrayRef< int > Mask, Type *OnlyIfReducedTy=nullptr)
static bool isSupportedGetElementPtr(const Type *SrcElemTy)
Whether creating a constant expression for this getelementptr type is supported.
Definition Constants.h:1598
static LLVM_ABI Constant * get(unsigned Opcode, Constant *C1, Constant *C2, unsigned Flags=0, Type *OnlyIfReducedTy=nullptr)
get - Return a binary or shift operator constant expression, folding if possible.
static LLVM_ABI bool isDesirableBinOp(unsigned Opcode)
Whether creating a constant expression for this binary operator is desirable.
static Constant * getGetElementPtr(Type *Ty, Constant *C, ArrayRef< Constant * > IdxList, GEPNoWrapFlags NW=GEPNoWrapFlags::none(), std::optional< ConstantRange > InRange=std::nullopt, Type *OnlyIfReducedTy=nullptr)
Getelementptr form.
Definition Constants.h:1470
static LLVM_ABI Constant * getBitCast(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getTrunc(Constant *C, Type *Ty, bool OnlyIfReduced=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
const APFloat & getValueAPF() const
Definition Constants.h:463
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getNaN(Type *Ty, bool Negative=false, uint64_t Payload=0)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
Constrained floating point compare intrinsics.
This is the common base class for constrained floating point intrinsics.
LLVM_ABI std::optional< fp::ExceptionBehavior > getExceptionBehavior() const
LLVM_ABI std::optional< RoundingMode > getRoundingMode() const
Wrapper for a function that represents a value that functionally represents the original function.
Definition Constants.h:1143
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:223
iterator end()
Definition DenseMap.h:141
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:284
static LLVM_ABI bool compare(const APFloat &LHS, const APFloat &RHS, FCmpInst::Predicate Pred)
Return result of LHS Pred RHS comparison.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:867
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
Definition Function.cpp:803
bool isStrictFP() const
Determine if the function has strict floating point sematics.
Definition Function.h:636
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags inBounds()
GEPNoWrapFlags withoutNoUnsignedSignedWrap() const
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedSignedWrap() const
bool isInBounds() const
static LLVM_ABI Type * getIndexedType(Type *Ty, ArrayRef< Value * > IdxList)
Returns the result type of a getelementptr with the given source element type and indexes.
PointerType * getType() const
Global values are always pointers.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this global belongs to.
Definition Globals.cpp:205
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
bool isConstant() const
If the value is a global constant, its value is immutable throughout the runtime execution of the pro...
bool hasDefinitiveInitializer() const
hasDefinitiveInitializer - Whether the global variable has an initializer, and any other instances of...
static LLVM_ABI bool compare(const APInt &LHS, const APInt &RHS, ICmpInst::Predicate Pred)
Return result of LHS Pred RHS comparison.
Predicate getSignedPredicate() const
For example, EQ->EQ, SLE->SLE, UGT->SGT, etc.
bool isEquality() const
Return true if this predicate is either EQ or NE.
bool isCast() const
bool isBinaryOp() const
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
bool isUnaryOp() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
static APInt getSaturationPoint(Intrinsic::ID ID, unsigned numBits)
Min/max intrinsics are monotonic, they operate on a fixed-bitwidth values, so there is a certain thre...
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Class to represent scalable SIMD vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
iterator_range< const_set_bits_iterator > set_bits() const
void push_back(const T &Elt)
pointer data()
Return a pointer to the vector's buffer, even if empty().
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Used to lazily calculate structure layout information for a target machine, based on the DataLayout s...
Definition DataLayout.h:743
LLVM_ABI unsigned getElementContainingOffset(uint64_t FixedOffset) const
Given a valid byte offset into the structure, returns the structure index that contains it.
TypeSize getElementOffset(unsigned Idx) const
Definition DataLayout.h:774
Class to represent struct types.
unsigned getNumElements() const
Random access to the elements.
Provides information about what library functions are available for the current target.
bool has(LibFunc F) const
Tests whether a library function is available.
LibFunc getLibFunc(StringRef funcName) const
Searches for a particular function name.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:310
bool isByteTy() const
True if this is an instance of ByteType.
Definition Type.h:242
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:288
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
bool isByteOrByteVectorTy() const
Return true if this is a byte type or a vector of byte types.
Definition Type.h:248
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:308
bool isSized(SmallPtrSetImpl< Type * > *Visited=nullptr) const
Return true if it makes sense to take the size of this type.
Definition Type.h:326
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:306
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:285
bool isX86_AMXTy() const
Return true if this is X86 AMX.
Definition Type.h:202
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:313
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
Definition Type.h:397
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:106
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
LLVM_ABI const Value * stripAndAccumulateConstantOffsets(const DataLayout &DL, APInt &Offset, bool AllowNonInbounds, bool AllowInvariantGroup=false, function_ref< bool(Value &Value, APInt &Offset)> ExternalAnalysis=nullptr, bool LookThroughIntToPtr=false) const
Accumulate the constant offset this value has compared to a base pointer.
LLVM_ABI uint64_t getPointerDereferenceableBytes(const DataLayout &DL, bool &CanBeNull, bool *CanBeFreed) const
Returns the number of bytes known to be dereferenceable for the pointer value.
Definition Value.cpp:918
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
static constexpr bool isKnownGE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:237
const ParentTy * getParent() const
Definition ilist_node.h:34
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt pext(const APInt &Val, const APInt &Mask)
Perform a "compress" operation, also known as pext or bext.
Definition APInt.cpp:3243
const APInt & smin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be signed.
Definition APInt.h:2275
const APInt & smax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be signed.
Definition APInt.h:2280
LLVM_ABI APInt clmul(const APInt &LHS, const APInt &RHS)
Perform a carry-less multiply, also known as XOR multiplication, and return low-bits.
Definition APInt.cpp:3223
const APInt & umin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be unsigned.
Definition APInt.h:2285
LLVM_ABI APInt pdep(const APInt &Val, const APInt &Mask)
Perform an "expand" operation, also known as pdep or bdep.
Definition APInt.cpp:3253
const APInt & umax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be unsigned.
Definition APInt.h:2290
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ CE
Windows NT (Windows on ARM)
Definition MCAsmInfo.h:51
initializer< Ty > init(const Ty &Val)
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:446
static constexpr cmpResult cmpEqual
Definition APFloat.h:454
@ ebStrict
This corresponds to "fpexcept.strict".
Definition FPEnv.h:42
@ ebIgnore
This corresponds to "fpexcept.ignore".
Definition FPEnv.h:40
constexpr double pi
APFloat::roundingMode GetFMARoundingMode(Intrinsic::ID IntrinsicID)
DenormalMode GetNVVMDenormMode(bool ShouldFTZ)
bool FPToIntegerIntrinsicNaNZero(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFDivRoundingMode(Intrinsic::ID IntrinsicID)
bool FPToIntegerIntrinsicResultIsSigned(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFPToIntegerRoundingMode(Intrinsic::ID IntrinsicID)
bool RCPShouldFTZ(Intrinsic::ID IntrinsicID)
bool FPToIntegerIntrinsicShouldFTZ(Intrinsic::ID IntrinsicID)
bool FDivShouldFTZ(Intrinsic::ID IntrinsicID)
bool FAddShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMinFMaxIsXorSignAbs(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFMulRoundingMode(Intrinsic::ID IntrinsicID)
bool UnaryMathIntrinsicShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMinFMaxShouldFTZ(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFAddRoundingMode(Intrinsic::ID IntrinsicID)
bool FMAShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMulShouldFTZ(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetRCPRoundingMode(Intrinsic::ID IntrinsicID)
bool FMinFMaxPropagatesNaNs(Intrinsic::ID IntrinsicID)
NodeAddr< FuncNode * > Func
Definition RDFGraph.h:393
LLVM_ABI std::error_code status(const Twine &path, file_status &result, bool follow=true)
Get file status as if by POSIX stat().
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:315
@ Offset
Definition DWP.cpp:578
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
LLVM_ABI Constant * ConstantFoldLoadThroughBitcast(Constant *C, Type *DestTy, const DataLayout &DL)
ConstantFoldLoadThroughBitcast - try to cast constant to destination type returning null if unsuccess...
static double log2(double V)
LLVM_ABI Constant * ConstantFoldSelectInstruction(Constant *Cond, Constant *V1, Constant *V2)
Attempt to constant fold a select instruction with the specified operands.
LLVM_ABI Constant * ConstantFoldFPInstOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL, const Instruction *I, bool AllowNonDeterministic=true)
Attempt to constant fold a floating point binary operation with the specified operands,...
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
LLVM_ABI bool canConstantFoldCallTo(const CallBase *Call, const Function *F)
canConstantFoldCallTo - Return true if its even possible to fold a call to the specified function.
unsigned getPointerAddressSpace(const Type *T)
Definition SPIRVUtils.h:395
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI Constant * ConstantFoldInstruction(const Instruction *I, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldInstruction - Try to constant fold the specified instruction.
APFloat abs(APFloat X)
Returns the absolute value of the argument.
Definition APFloat.h:1713
LLVM_ABI Constant * ConstantFoldCompareInstruction(CmpInst::Predicate Predicate, Constant *C1, Constant *C2)
LLVM_ABI Constant * ConstantFoldUnaryInstruction(unsigned Opcode, Constant *V)
LLVM_ABI bool IsConstantOffsetFromGlobal(Constant *C, GlobalValue *&GV, APInt &Offset, const DataLayout &DL, DSOLocalEquivalent **DSOEquiv=nullptr)
If this constant is a constant offset from a global, return the global and the constant.
LLVM_ABI bool isMathLibCallNoop(const CallBase *Call, const TargetLibraryInfo *TLI)
Check whether the given call has no side-effects.
LLVM_ABI Constant * ReadByteArrayFromGlobal(const GlobalVariable *GV, uint64_t Offset)
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1793
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
int ilogb(const APFloat &Arg)
Returns the exponent of the internal representation of the APFloat.
Definition APFloat.h:1684
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
LLVM_ABI Constant * ConstantFoldCall(const CallBase *Call, Function *F, ArrayRef< Constant * > Operands, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldCall - Attempt to constant fold a call to the specified function with the specified argum...
APFloat frexp(const APFloat &X, int &Exp, APFloat::roundingMode RM)
Equivalent of C standard library function.
Definition APFloat.h:1705
LLVM_ABI Constant * ConstantFoldExtractValueInstruction(Constant *Agg, ArrayRef< unsigned > Idxs)
Attempt to constant fold an extractvalue instruction with the specified operands and indices.
LLVM_ABI Constant * ConstantFoldConstant(const Constant *C, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldConstant - Fold the constant using the specified DataLayout.
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1748
LLVM_ABI Constant * ConstantFoldLoadFromUniformValue(Constant *C, Type *Ty, const DataLayout &DL)
If C is a uniform value where all bits are the same (either all zero, all ones, all undef or all pois...
LLVM_ABI Constant * ConstantFoldUnaryOpOperand(unsigned Opcode, Constant *Op, const DataLayout &DL)
Attempt to constant fold a unary operation with the specified operand.
LLVM_ABI Constant * FlushFPConstant(Constant *Operand, const Instruction *I, bool IsOutput)
Attempt to flush float point constant according to denormal mode set in the instruction's parent func...
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_READONLY LLVM_ABI std::optional< APFloat > exp(const APFloat &X, RoundingMode RM=APFloat::rmNearestTiesToEven, APFloat::opStatus *Status=nullptr)
Implement IEEE 754-2019 exp functions.
Definition APFloat.cpp:6165
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1779
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1693
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI Constant * ConstantFoldCastOperand(unsigned Opcode, Constant *C, Type *DestTy, const DataLayout &DL)
Attempt to constant fold a cast with the specified operand.
LLVM_ABI Constant * ConstantFoldLoadFromConst(Constant *C, Type *Ty, const APInt &Offset, const DataLayout &DL)
Extract value of C at the given Offset reinterpreted as Ty.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool intrinsicPropagatesPoison(Intrinsic::ID IID)
Return whether this intrinsic propagates poison for all operands.
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
LLVM_ABI Constant * ConstantFoldIntrinsic(Intrinsic::ID ID, ArrayRef< Constant * > Ops, Type *Ty, const DataLayout &DL, Function *CxtF=nullptr)
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1729
@ Sub
Subtraction of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
IntPtrTy
Definition InstrProf.h:82
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ NearestTiesToEven
roundTiesToEven.
@ Dynamic
Denotes mode unknown at compile time.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Definition STLExtras.h:2166
LLVM_ABI Constant * ConstantFoldCastInstruction(unsigned opcode, Constant *V, Type *DestTy)
LLVM_ABI Constant * ConstantFoldInsertValueInstruction(Constant *Agg, Constant *Val, ArrayRef< unsigned > Idxs)
Attempt to constant fold an insertvalue instruction with the specified operands and indices.
LLVM_ABI Constant * ConstantFoldLoadFromConstPtr(Constant *C, Type *Ty, APInt Offset, const DataLayout &DL)
Return the value that a load from C with offset Offset would produce if it is constant and determinab...
LLVM_ABI Constant * ConstantFoldInstOperands(const Instruction *I, ArrayRef< Constant * > Ops, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldInstOperands - Attempt to constant fold an instruction with the specified operands.
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1766
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1806
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
LLVM_ABI Constant * ConstantFoldIntegerCast(Constant *C, Type *DestTy, bool IsSigned, const DataLayout &DL)
Constant fold a zext, sext or trunc, depending on IsSigned and whether the DestTy is wider or narrowe...
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
LLVM_ABI Constant * ConstantFoldBinaryInstruction(unsigned Opcode, Constant *V1, Constant *V2)
Represent subnormal handling kind for floating point instruction inputs and outputs.
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
DenormalModeKind
Represent handled modes for denormal (aka subnormal) modes in the floating point environment.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
@ PositiveZero
Denormals are flushed to positive zero.
@ Dynamic
Denormals have unknown treatment.
@ IEEE
IEEE-754 denormal numbers preserved.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
static constexpr DenormalMode getDynamic()
static constexpr DenormalMode getIEEE()
bool isConstant() const
Returns true if we know the value of all bits.
Definition KnownBits.h:54
const APInt & getConstant() const
Returns the value when all bits have a known value.
Definition KnownBits.h:58