LLVM 24.0.0git
ConstantFolding.cpp
Go to the documentation of this file.
1//===-- ConstantFolding.cpp - Fold instructions into constants ------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines routines for folding instructions into constants.
10//
11// Also, to supplement the basic IR ConstantExpr simplifications,
12// this file defines some additional folding routines that can make use of
13// DataLayout information. These functions cannot go in IR due to library
14// dependency issues.
15//
16//===----------------------------------------------------------------------===//
17
19#include "llvm/ADT/APFloat.h"
20#include "llvm/ADT/APInt.h"
21#include "llvm/ADT/APSInt.h"
22#include "llvm/ADT/ArrayRef.h"
23#include "llvm/ADT/DenseMap.h"
24#include "llvm/ADT/STLExtras.h"
27#include "llvm/ADT/StringRef.h"
32#include "llvm/Config/config.h"
33#include "llvm/IR/Constant.h"
35#include "llvm/IR/Constants.h"
36#include "llvm/IR/DataLayout.h"
38#include "llvm/IR/Function.h"
39#include "llvm/IR/GlobalValue.h"
41#include "llvm/IR/InstrTypes.h"
42#include "llvm/IR/Instruction.h"
45#include "llvm/IR/Intrinsics.h"
46#include "llvm/IR/IntrinsicsAArch64.h"
47#include "llvm/IR/IntrinsicsAMDGPU.h"
48#include "llvm/IR/IntrinsicsARM.h"
49#include "llvm/IR/IntrinsicsNVPTX.h"
50#include "llvm/IR/IntrinsicsWebAssembly.h"
51#include "llvm/IR/IntrinsicsX86.h"
53#include "llvm/IR/Operator.h"
54#include "llvm/IR/Type.h"
55#include "llvm/IR/Value.h"
56#include "llvm/Support/CRC.h"
60#include <cassert>
61#include <cerrno>
62#include <cfenv>
63#include <cmath>
64#include <cstdint>
65
66using namespace llvm;
67
69 "disable-fp-call-folding",
70 cl::desc("Disable constant-folding of FP intrinsics and libcalls."),
71 cl::init(false), cl::Hidden);
72
73namespace {
74
75//===----------------------------------------------------------------------===//
76// Constant Folding internal helper functions
77//===----------------------------------------------------------------------===//
78
79static Constant *foldConstVectorToAPInt(APInt &Result, Type *DestTy,
80 Constant *C, Type *SrcEltTy,
81 unsigned NumSrcElts,
82 const DataLayout &DL) {
83 // Now that we know that the input value is a vector of integers, just shift
84 // and insert them into our result.
85 unsigned BitShift = DL.getTypeSizeInBits(SrcEltTy);
86 for (unsigned i = 0; i != NumSrcElts; ++i) {
87 Constant *Element;
88 if (DL.isLittleEndian())
89 Element = C->getAggregateElement(NumSrcElts - i - 1);
90 else
91 Element = C->getAggregateElement(i);
92
93 if (isa_and_nonnull<UndefValue>(Element)) {
94 Result <<= BitShift;
95 continue;
96 }
97
98 auto *ElementCI = dyn_cast_or_null<ConstantInt>(Element);
99 if (!ElementCI)
100 return ConstantExpr::getBitCast(C, DestTy);
101
102 Result <<= BitShift;
103 Result |= ElementCI->getValue().zext(Result.getBitWidth());
104 }
105
106 return nullptr;
107}
108
109/// Check whether folding this bitcast into a byte vector would mix poison and
110/// non-poison bits in the same output lane. While integer types track poison on
111/// a per-value basis, byte types track it on a per-bit basis. However,
112/// `ConstantByte` cannot represent values with both poison and non-poison bits.
113///
114/// Source elements are grouped by the output lane they map to. Returns true if
115/// any group contains both poison and non-poison elements.
116static bool foldMixesPoisonBits(Constant *C, unsigned NumSrcElt,
117 unsigned NumDstElt) {
118 // If element counts don't divide evenly, bail out if a poison source element
119 // might span multiple destination lanes.
120 if (NumSrcElt % NumDstElt != 0)
121 return C->containsPoisonElement();
122 unsigned Ratio = NumSrcElt / NumDstElt;
123 for (unsigned i = 0; i != NumSrcElt; i += Ratio) {
124 bool HasPoison = false;
125 bool HasNonPoison = false;
126 for (unsigned j = 0; j != Ratio; ++j) {
127 Constant *Src = C->getAggregateElement(i + j);
128 // Conservatively bail out.
129 if (!Src)
130 return true;
131 if (isa<PoisonValue>(Src))
132 HasPoison = true;
133 else
134 HasNonPoison = true;
135 }
136 if (HasPoison && HasNonPoison)
137 return true;
138 }
139 return false;
140}
141
142/// Track which destination lanes of a bitcast are produced from poison bytes.
143/// A destination lane is marked if any source element mapped to it is poison.
144/// Returns false if an aggregate element cannot be inspected. The caller should
145/// bail out of folding.
146static bool computePoisonDstLanes(Constant *C, unsigned NumSrcElt,
147 unsigned NumDstElt,
148 SmallBitVector &PoisonDstElts) {
149 // If element counts don't divide evenly, bail out if a poison source element
150 // might span multiple destination lanes.
151 if ((NumDstElt < NumSrcElt ? NumSrcElt % NumDstElt : NumDstElt % NumSrcElt))
152 return !C->containsPoisonElement();
153 if (NumDstElt < NumSrcElt) {
154 unsigned Ratio = NumSrcElt / NumDstElt;
155 for (unsigned i = 0; i != NumDstElt; ++i) {
156 for (unsigned j = 0; j != Ratio; ++j) {
157 Constant *Src = C->getAggregateElement(i * Ratio + j);
158 if (!Src)
159 return false;
160 if (isa<PoisonValue>(Src)) {
161 PoisonDstElts[i] = true;
162 break;
163 }
164 }
165 }
166 } else {
167 unsigned Ratio = NumDstElt / NumSrcElt;
168 for (unsigned i = 0; i != NumSrcElt; ++i) {
169 Constant *Src = C->getAggregateElement(i);
170 if (!Src)
171 return false;
172 if (isa<PoisonValue>(Src))
173 PoisonDstElts.set(i * Ratio, (i + 1) * Ratio);
174 }
175 }
176 return true;
177}
178
179/// Constant fold bitcast, symbolically evaluating it with DataLayout.
180/// This always returns a non-null constant, but it may be a
181/// ConstantExpr if unfoldable.
182Constant *FoldBitCast(Constant *C, Type *DestTy, const DataLayout &DL) {
183 assert(CastInst::castIsValid(Instruction::BitCast, C, DestTy) &&
184 "Invalid constantexpr bitcast!");
185
186 // Catch the obvious splat cases.
187 if (Constant *Res = ConstantFoldLoadFromUniformValue(C, DestTy, DL))
188 return Res;
189
190 if (auto *VTy = dyn_cast<VectorType>(C->getType())) {
191 // Handle a vector->scalar integer/fp cast.
192 if (isa<IntegerType>(DestTy) || DestTy->isFloatingPointTy()) {
193 unsigned NumSrcElts = cast<FixedVectorType>(VTy)->getNumElements();
194 Type *SrcEltTy = VTy->getElementType();
195
196 // Bitcasting a byte containing any poison bit to an integer or fp type
197 // yields poison.
198 if (SrcEltTy->isByteTy() && C->containsPoisonElement())
199 return PoisonValue::get(DestTy);
200
201 // If the vector is a vector of floating point or bytes, convert it to a
202 // vector of int to simplify things.
203 if (SrcEltTy->isFloatingPointTy() || SrcEltTy->isByteTy()) {
204 unsigned Width = SrcEltTy->getPrimitiveSizeInBits();
205 auto *SrcIVTy = FixedVectorType::get(
206 IntegerType::get(C->getContext(), Width), NumSrcElts);
207 // Ask IR to do the conversion now that #elts line up.
208 C = ConstantExpr::getBitCast(C, SrcIVTy);
209 }
210
211 APInt Result(DL.getTypeSizeInBits(DestTy), 0);
212 if (Constant *CE = foldConstVectorToAPInt(Result, DestTy, C,
213 SrcEltTy, NumSrcElts, DL))
214 return CE;
215
216 if (isa<IntegerType>(DestTy))
217 return ConstantInt::get(DestTy, Result);
218
219 APFloat FP(DestTy->getFltSemantics(), Result);
220 return ConstantFP::get(DestTy->getContext(), FP);
221 }
222 }
223
224 // The code below only handles casts to vectors currently.
225 auto *DestVTy = dyn_cast<VectorType>(DestTy);
226 if (!DestVTy)
227 return ConstantExpr::getBitCast(C, DestTy);
228
229 // If this is a scalar -> vector cast, convert the input into a <1 x scalar>
230 // vector so the code below can handle it uniformly.
231 if (!isa<VectorType>(C->getType()) &&
233 Constant *Ops = C; // don't take the address of C!
234 return FoldBitCast(ConstantVector::get(Ops), DestTy, DL);
235 }
236
237 // Some of what follows may extend to cover scalable vectors but the current
238 // implementation is fixed length specific.
239 if (!isa<FixedVectorType>(C->getType()))
240 return ConstantExpr::getBitCast(C, DestTy);
241
242 // If this is a bitcast from constant vector -> vector, fold it.
245 return ConstantExpr::getBitCast(C, DestTy);
246
247 // If the element types match, IR can fold it.
248 unsigned NumDstElt = cast<FixedVectorType>(DestVTy)->getNumElements();
249 unsigned NumSrcElt = cast<FixedVectorType>(C->getType())->getNumElements();
250 if (NumDstElt == NumSrcElt)
251 return ConstantExpr::getBitCast(C, DestTy);
252
253 Type *SrcEltTy = cast<VectorType>(C->getType())->getElementType();
254 Type *DstEltTy = DestVTy->getElementType();
255
256 // Otherwise, we're changing the number of elements in a vector, which
257 // requires endianness information to do the right thing. For example,
258 // bitcast (<2 x i64> <i64 0, i64 1> to <4 x i32>)
259 // folds to (little endian):
260 // <4 x i32> <i32 0, i32 0, i32 1, i32 0>
261 // and to (big endian):
262 // <4 x i32> <i32 0, i32 0, i32 0, i32 1>
263
264 // First thing is first. We only want to think about integer here, so if
265 // we have something in FP form, recast it as integer.
266 if (DstEltTy->isFloatingPointTy()) {
267 // Fold to an vector of integers with same size as our FP type.
268 unsigned FPWidth = DstEltTy->getPrimitiveSizeInBits();
269 auto *DestIVTy = FixedVectorType::get(
270 IntegerType::get(C->getContext(), FPWidth), NumDstElt);
271 // Recursively handle this integer conversion, if possible.
272 C = FoldBitCast(C, DestIVTy, DL);
273
274 // Finally, IR can handle this now that #elts line up.
275 return ConstantExpr::getBitCast(C, DestTy);
276 }
277
278 // Handle byte destination type by folding through integers.
279 if (DstEltTy->isByteTy()) {
280 // When combining elements into larger byte values, bail out if the fold
281 // mixes poison and non-poison bits in the same destination element. Byte
282 // types track poison per bit, and no constant value can represent that.
283 if (NumDstElt < NumSrcElt && foldMixesPoisonBits(C, NumSrcElt, NumDstElt))
284 return ConstantExpr::getBitCast(C, DestTy);
285
286 // Fold to a vector of integers with same size as the byte type.
287 unsigned ByteWidth = DstEltTy->getPrimitiveSizeInBits();
288 auto *DestIVTy = FixedVectorType::get(
289 IntegerType::get(C->getContext(), ByteWidth), NumDstElt);
290 C = FoldBitCast(C, DestIVTy, DL);
291 return ConstantExpr::getBitCast(C, DestTy);
292 }
293
294 // Okay, we know the destination is integer, if the input is FP, convert
295 // it to integer first.
296 if (SrcEltTy->isFloatingPointTy()) {
297 unsigned FPWidth = SrcEltTy->getPrimitiveSizeInBits();
298 auto *SrcIVTy = FixedVectorType::get(
299 IntegerType::get(C->getContext(), FPWidth), NumSrcElt);
300 // Ask IR to do the conversion now that #elts line up.
301 C = ConstantExpr::getBitCast(C, SrcIVTy);
302 assert((isa<ConstantVector>(C) || // FIXME: Remove ConstantVector.
304 "Constant folding cannot fail for plain fp->int bitcast!");
305 }
306
307 // Handle byte source type by folding through integers. Byte types track
308 // poison per bit, so any poison bit makes the destination lane poison.
309 // Record which destination lanes contain poison bits, before the generic
310 // fold below refines them to undef/zero, so they can be restored.
311 SmallBitVector PoisonDstElts(NumDstElt);
312 if (SrcEltTy->isByteTy()) {
313 if (!computePoisonDstLanes(C, NumSrcElt, NumDstElt, PoisonDstElts))
314 return ConstantExpr::getBitCast(C, DestTy);
315
316 unsigned ByteWidth = SrcEltTy->getPrimitiveSizeInBits();
317 auto *SrcIVTy = FixedVectorType::get(
318 IntegerType::get(C->getContext(), ByteWidth), NumSrcElt);
319 // Ask IR to do the conversion now that #elts line up.
320 C = ConstantExpr::getBitCast(C, SrcIVTy);
321 assert((isa<ConstantVector>(C) || // FIXME: Remove ConstantVector.
323 "Constant folding cannot fail for plain byte->int bitcast!");
324 }
325
326 // Now we know that the input and output vectors are both integer vectors
327 // of the same size, and that their #elements is not the same.
328 // Use data buffer for easy non-integer element ratio vectors handling,
329 // For example: <4 x i24> to <3 x i32>.
330 bool isLittleEndian = DL.isLittleEndian();
331 unsigned SrcBitSize = SrcEltTy->getPrimitiveSizeInBits();
332 unsigned DstBitSize = DstEltTy->getPrimitiveSizeInBits();
334 unsigned SrcElt = 0;
335
336 APInt Buffer(2 * std::max(SrcBitSize, DstBitSize), 0);
337 APInt UndefMask(Buffer.getBitWidth(), 0);
338 APInt PoisonMask(Buffer.getBitWidth(), 0);
339 unsigned BufferBitSize = 0;
340
341 while (Result.size() != NumDstElt) {
342 // Load SrcElts into Buffer.
343 while (BufferBitSize < DstBitSize) {
344 Constant *Element = C->getAggregateElement(SrcElt++);
345 if (!Element) // Reject constantexpr elements
346 return ConstantExpr::getBitCast(C, DestTy);
347
348 // Shift Buffer & Masks to fit next SrcElt.
349 if (!isLittleEndian) {
350 Buffer <<= SrcBitSize;
351 UndefMask <<= SrcBitSize;
352 PoisonMask <<= SrcBitSize;
353 }
354
355 APInt SrcValue;
356 unsigned BitPosition = isLittleEndian ? BufferBitSize : 0;
357 if (isa<UndefValue>(Element)) {
358 // Set masks fragments bits.
359 UndefMask.setBits(BitPosition, BitPosition + SrcBitSize);
360 if (isa<PoisonValue>(Element))
361 PoisonMask.setBits(BitPosition, BitPosition + SrcBitSize);
362 SrcValue = APInt::getZero(SrcBitSize);
363 } else {
364 auto *Src = dyn_cast<ConstantInt>(Element);
365 if (!Src)
366 return ConstantExpr::getBitCast(C, DestTy);
367 SrcValue = Src->getValue();
368 }
369
370 // Insert src element bits into Buffer on correct position.
371 Buffer.insertBits(SrcValue, BitPosition);
372 BufferBitSize += SrcBitSize;
373 }
374
375 // Create DstElts from Buffer.
376 while (BufferBitSize >= DstBitSize) {
377 unsigned ShiftAmt = isLittleEndian ? 0 : BufferBitSize - DstBitSize;
378 // Emit undef/poison, if all undef mask fragment bits are set.
379 if (UndefMask.extractBits(DstBitSize, ShiftAmt).isAllOnes()) {
380 // Push poison, if any bit in poison mask fragment is set.
381 if (!PoisonMask.extractBits(DstBitSize, ShiftAmt).isZero()) {
382 Result.push_back(PoisonValue::get(DstEltTy));
383 } else {
384 Result.push_back(UndefValue::get(DstEltTy));
385 }
386 } else {
387 // Create and push DstElt.
388 APInt Elt = Buffer.extractBits(DstBitSize, ShiftAmt);
389 Result.push_back(ConstantInt::get(DstEltTy, Elt));
390 }
391
392 // Shift unused Buffer fragment to lower bits.
393 if (isLittleEndian) {
394 Buffer.lshrInPlace(DstBitSize);
395 UndefMask.lshrInPlace(DstBitSize);
396 PoisonMask.lshrInPlace(DstBitSize);
397 }
398 BufferBitSize -= DstBitSize;
399 }
400 }
401
402 // Restore destination lanes whose source bytes contained poison bits.
403 for (unsigned I : PoisonDstElts.set_bits())
404 Result[I] = PoisonValue::get(DstEltTy);
405
406 return ConstantVector::get(Result);
407}
408
409} // end anonymous namespace
410
411/// If this constant is a constant offset from a global, return the global and
412/// the constant. Because of constantexprs, this function is recursive.
414 APInt &Offset, const DataLayout &DL,
415 DSOLocalEquivalent **DSOEquiv) {
416 if (DSOEquiv)
417 *DSOEquiv = nullptr;
418
419 // Trivial case, constant is the global.
420 if ((GV = dyn_cast<GlobalValue>(C))) {
421 unsigned BitWidth = DL.getIndexTypeSizeInBits(GV->getType());
422 Offset = APInt(BitWidth, 0);
423 return true;
424 }
425
426 if (auto *FoundDSOEquiv = dyn_cast<DSOLocalEquivalent>(C)) {
427 if (DSOEquiv)
428 *DSOEquiv = FoundDSOEquiv;
429 GV = FoundDSOEquiv->getGlobalValue();
430 unsigned BitWidth = DL.getIndexTypeSizeInBits(GV->getType());
431 Offset = APInt(BitWidth, 0);
432 return true;
433 }
434
435 // Otherwise, if this isn't a constant expr, bail out.
436 auto *CE = dyn_cast<ConstantExpr>(C);
437 if (!CE) return false;
438
439 // Look through ptr->int and ptr->ptr casts.
440 if (CE->getOpcode() == Instruction::PtrToInt ||
441 CE->getOpcode() == Instruction::PtrToAddr)
442 return IsConstantOffsetFromGlobal(CE->getOperand(0), GV, Offset, DL,
443 DSOEquiv);
444
445 // i32* getelementptr ([5 x i32]* @a, i32 0, i32 5)
446 auto *GEP = dyn_cast<GEPOperator>(CE);
447 if (!GEP)
448 return false;
449
450 unsigned BitWidth = DL.getIndexTypeSizeInBits(GEP->getType());
451 APInt TmpOffset(BitWidth, 0);
452
453 // If the base isn't a global+constant, we aren't either.
454 if (!IsConstantOffsetFromGlobal(CE->getOperand(0), GV, TmpOffset, DL,
455 DSOEquiv))
456 return false;
457
458 // Otherwise, add any offset that our operands provide.
459 if (!GEP->accumulateConstantOffset(DL, TmpOffset))
460 return false;
461
462 Offset = TmpOffset;
463 return true;
464}
465
467 const DataLayout &DL) {
468 do {
469 Type *SrcTy = C->getType();
470 if (SrcTy == DestTy)
471 return C;
472
473 TypeSize DestSize = DL.getTypeSizeInBits(DestTy);
474 TypeSize SrcSize = DL.getTypeSizeInBits(SrcTy);
475 if (!TypeSize::isKnownGE(SrcSize, DestSize))
476 return nullptr;
477
478 // Catch the obvious splat cases (since all-zeros can coerce non-integral
479 // pointers legally).
480 if (Constant *Res = ConstantFoldLoadFromUniformValue(C, DestTy, DL))
481 return Res;
482
483 // If the type sizes are the same and a cast is legal, just directly
484 // cast the constant.
485 // But be careful not to coerce non-integral pointers illegally.
486 if (SrcSize == DestSize &&
487 DL.isNonIntegralPointerType(SrcTy->getScalarType()) ==
488 DL.isNonIntegralPointerType(DestTy->getScalarType())) {
489 Instruction::CastOps Cast = Instruction::BitCast;
490 // If we are going from a pointer to int or vice versa, we spell the cast
491 // differently.
492 if (SrcTy->isIntegerTy() && DestTy->isPointerTy())
493 Cast = Instruction::IntToPtr;
494 else if (SrcTy->isPointerTy() && DestTy->isIntegerTy())
495 Cast = Instruction::PtrToInt;
496
497 if (CastInst::castIsValid(Cast, C, DestTy))
498 return ConstantFoldCastOperand(Cast, C, DestTy, DL);
499 }
500
501 // If this isn't an aggregate type, there is nothing we can do to drill down
502 // and find a bitcastable constant.
503 if (!SrcTy->isAggregateType() && !SrcTy->isVectorTy())
504 return nullptr;
505
506 // We're simulating a load through a pointer that was bitcast to point to
507 // a different type, so we can try to walk down through the initial
508 // elements of an aggregate to see if some part of the aggregate is
509 // castable to implement the "load" semantic model.
510 if (SrcTy->isStructTy()) {
511 // Struct types might have leading zero-length elements like [0 x i32],
512 // which are certainly not what we are looking for, so skip them.
513 unsigned Elem = 0;
514 Constant *ElemC;
515 do {
516 ElemC = C->getAggregateElement(Elem++);
517 } while (ElemC && DL.getTypeSizeInBits(ElemC->getType()).isZero());
518 C = ElemC;
519 } else {
520 // For non-byte-sized vector elements, the first element is not
521 // necessarily located at the vector base address.
522 if (auto *VT = dyn_cast<VectorType>(SrcTy))
523 if (!DL.typeSizeEqualsStoreSize(VT->getElementType()))
524 return nullptr;
525
526 C = C->getAggregateElement(0u);
527 }
528 } while (C);
529
530 return nullptr;
531}
532
533namespace {
534
535/// Recursive helper to read bits out of global. C is the constant being copied
536/// out of. ByteOffset is an offset into C. CurPtr is the pointer to copy
537/// results into and BytesLeft is the number of bytes left in
538/// the CurPtr buffer. DL is the DataLayout. When IsByteLoad is true, do not
539/// unwrap inttoptr constant expressions. The caller would reconstruct those
540/// bits as a ConstantByte, dropping the pointer's provenance.
541bool ReadDataFromGlobal(Constant *C, uint64_t ByteOffset, unsigned char *CurPtr,
542 unsigned BytesLeft, const DataLayout &DL,
543 bool IsByteLoad = false) {
544 assert(ByteOffset <= DL.getTypeAllocSize(C->getType()) &&
545 "Out of range access");
546
547 // Reading type padding, return zero.
548 if (ByteOffset >= DL.getTypeStoreSize(C->getType()))
549 return true;
550
551 // If this element is zero or undefined, we can just return since *CurPtr is
552 // zero initialized.
554 return true;
555
556 auto *CI = dyn_cast<ConstantInt>(C);
557 if (CI && CI->getType()->isIntegerTy()) {
558 if ((CI->getBitWidth() & 7) != 0)
559 return false;
560 const APInt &Val = CI->getValue();
561 unsigned IntBytes = unsigned(CI->getBitWidth()/8);
562
563 for (unsigned i = 0; i != BytesLeft && ByteOffset != IntBytes; ++i) {
564 unsigned n = ByteOffset;
565 if (!DL.isLittleEndian())
566 n = IntBytes - n - 1;
567 CurPtr[i] = Val.extractBits(8, n * 8).getZExtValue();
568 ++ByteOffset;
569 }
570 return true;
571 }
572
573 auto *CFP = dyn_cast<ConstantFP>(C);
574 if (CFP && CFP->getType()->isFloatingPointTy()) {
575 if (CFP->getType()->isDoubleTy()) {
576 C = FoldBitCast(C, Type::getInt64Ty(C->getContext()), DL);
577 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
578 IsByteLoad);
579 }
580 if (CFP->getType()->isFloatTy()){
581 C = FoldBitCast(C, Type::getInt32Ty(C->getContext()), DL);
582 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
583 IsByteLoad);
584 }
585 if (CFP->getType()->isHalfTy()){
586 C = FoldBitCast(C, Type::getInt16Ty(C->getContext()), DL);
587 return ReadDataFromGlobal(C, ByteOffset, CurPtr, BytesLeft, DL,
588 IsByteLoad);
589 }
590 return false;
591 }
592
593 if (auto *CS = dyn_cast<ConstantStruct>(C)) {
594 const StructLayout *SL = DL.getStructLayout(CS->getType());
595 unsigned Index = SL->getElementContainingOffset(ByteOffset);
596 uint64_t CurEltOffset = SL->getElementOffset(Index);
597 ByteOffset -= CurEltOffset;
598
599 while (true) {
600 // If the element access is to the element itself and not to tail padding,
601 // read the bytes from the element.
602 uint64_t EltSize = DL.getTypeAllocSize(CS->getOperand(Index)->getType());
603
604 if (ByteOffset < EltSize &&
605 !ReadDataFromGlobal(CS->getOperand(Index), ByteOffset, CurPtr,
606 BytesLeft, DL, IsByteLoad))
607 return false;
608
609 ++Index;
610
611 // Check to see if we read from the last struct element, if so we're done.
612 if (Index == CS->getType()->getNumElements())
613 return true;
614
615 // If we read all of the bytes we needed from this element we're done.
616 uint64_t NextEltOffset = SL->getElementOffset(Index);
617
618 if (BytesLeft <= NextEltOffset - CurEltOffset - ByteOffset)
619 return true;
620
621 // Move to the next element of the struct.
622 CurPtr += NextEltOffset - CurEltOffset - ByteOffset;
623 BytesLeft -= NextEltOffset - CurEltOffset - ByteOffset;
624 ByteOffset = 0;
625 CurEltOffset = NextEltOffset;
626 }
627 // not reached.
628 }
629
633 uint64_t NumElts, EltSize;
634 Type *EltTy;
635 if (auto *AT = dyn_cast<ArrayType>(C->getType())) {
636 NumElts = AT->getNumElements();
637 EltTy = AT->getElementType();
638 EltSize = DL.getTypeAllocSize(EltTy);
639 } else {
640 NumElts = cast<FixedVectorType>(C->getType())->getNumElements();
641 EltTy = cast<FixedVectorType>(C->getType())->getElementType();
642 // TODO: For non-byte-sized vectors, current implementation assumes there is
643 // padding to the next byte boundary between elements.
644 if (!DL.typeSizeEqualsStoreSize(EltTy))
645 return false;
646
647 EltSize = DL.getTypeStoreSize(EltTy);
648 }
649 uint64_t Index = ByteOffset / EltSize;
650 uint64_t Offset = ByteOffset - Index * EltSize;
651
652 for (; Index != NumElts; ++Index) {
653 if (!ReadDataFromGlobal(C->getAggregateElement(Index), Offset, CurPtr,
654 BytesLeft, DL, IsByteLoad))
655 return false;
656
657 uint64_t BytesWritten = EltSize - Offset;
658 assert(BytesWritten <= EltSize && "Not indexing into this element?");
659 if (BytesWritten >= BytesLeft)
660 return true;
661
662 Offset = 0;
663 BytesLeft -= BytesWritten;
664 CurPtr += BytesWritten;
665 }
666 return true;
667 }
668
669 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
670 if (CE->getOpcode() == Instruction::IntToPtr &&
671 CE->getOperand(0)->getType() == DL.getIntPtrType(CE->getType())) {
672 // Folding byte loads through the integer operand would rebuild the result
673 // as a `ConstantByte`, dropping the pointer's provenance.
674 if (IsByteLoad)
675 return false;
676 return ReadDataFromGlobal(CE->getOperand(0), ByteOffset, CurPtr,
677 BytesLeft, DL, IsByteLoad);
678 }
679 }
680
681 // Otherwise, unknown initializer type.
682 return false;
683}
684
685/// OrigLoadTy is the original type being loaded, while LoadTy is the type
686/// currently being folded (which may be integer type mapped from OrigLoadTy).
687Constant *FoldReinterpretLoadFromConst(Constant *C, Type *LoadTy,
688 Type *OrigLoadTy, int64_t Offset,
689 const DataLayout &DL) {
690 // Bail out early. Not expect to load from scalable global variable.
691 if (isa<ScalableVectorType>(LoadTy))
692 return nullptr;
693
694 auto *IntType = dyn_cast<IntegerType>(LoadTy);
695
696 // If this isn't an integer load we can't fold it directly.
697 if (!IntType) {
698 // If this is a non-integer load, we can try folding it as an int load and
699 // then bitcast the result. This can be useful for union cases. Note
700 // that address spaces don't matter here since we're not going to result in
701 // an actual new load.
702 if (!LoadTy->isFloatingPointTy() && !LoadTy->isPointerTy() &&
703 !LoadTy->isByteTy() && !LoadTy->isVectorTy())
704 return nullptr;
705
706 Type *MapTy = Type::getIntNTy(C->getContext(),
707 DL.getTypeSizeInBits(LoadTy).getFixedValue());
708 if (Constant *Res =
709 FoldReinterpretLoadFromConst(C, MapTy, OrigLoadTy, Offset, DL)) {
710 if (Res->isNullValue() && !LoadTy->isX86_AMXTy())
711 // Materializing a zero can be done trivially without a bitcast
712 return Constant::getNullValue(LoadTy);
713 Type *CastTy = LoadTy->isPtrOrPtrVectorTy() ? DL.getIntPtrType(LoadTy) : LoadTy;
714 Res = FoldBitCast(Res, CastTy, DL);
715 if (LoadTy->isPtrOrPtrVectorTy()) {
716 // For vector of pointer, we needed to first convert to a vector of integer, then do vector inttoptr
717 if (Res->isNullValue() && !LoadTy->isX86_AMXTy())
718 return Constant::getNullValue(LoadTy);
719 if (DL.isNonIntegralPointerType(LoadTy->getScalarType()))
720 // Be careful not to replace a load of an addrspace value with an inttoptr here
721 return nullptr;
722 Res = ConstantExpr::getIntToPtr(Res, LoadTy);
723 }
724 return Res;
725 }
726 return nullptr;
727 }
728
729 unsigned BytesLoaded = (IntType->getBitWidth() + 7) / 8;
730 // Allow folding of large type loads (e.g. <16 x double>).
731 if (BytesLoaded > 128 || BytesLoaded == 0)
732 return nullptr;
733
734 // For scalar integer load, use smaller limit to avoid regression during
735 // memcmp expansion. Codegen may generate inefficient string operations.
736 if (BytesLoaded > 32 && OrigLoadTy->isIntegerTy())
737 return nullptr;
738
739 // If we're not accessing anything in this constant, the result is undefined.
740 if (Offset <= -1 * static_cast<int64_t>(BytesLoaded))
741 return PoisonValue::get(IntType);
742
743 // TODO: We should be able to support scalable types.
744 TypeSize InitializerSize = DL.getTypeAllocSize(C->getType());
745 if (InitializerSize.isScalable())
746 return nullptr;
747
748 // If we're not accessing anything in this constant, the result is undefined.
749 if (Offset >= (int64_t)InitializerSize.getFixedValue())
750 return PoisonValue::get(IntType);
751
752 SmallVector<unsigned char, 64> RawBytes(BytesLoaded);
753 unsigned char *CurPtr = RawBytes.data();
754 unsigned BytesLeft = BytesLoaded;
755
756 // If we're loading off the beginning of the global, some bytes may be valid.
757 if (Offset < 0) {
758 CurPtr += -Offset;
759 BytesLeft += Offset;
760 Offset = 0;
761 }
762
763 if (!ReadDataFromGlobal(C, Offset, CurPtr, BytesLeft, DL,
764 /*IsByteLoad=*/OrigLoadTy->isByteOrByteVectorTy()))
765 return nullptr;
766
767 APInt ResultVal = APInt(IntType->getBitWidth(), 0);
768 if (DL.isLittleEndian()) {
769 ResultVal = RawBytes[BytesLoaded - 1];
770 for (unsigned i = 1; i != BytesLoaded; ++i) {
771 ResultVal <<= 8;
772 ResultVal |= RawBytes[BytesLoaded - 1 - i];
773 }
774 } else {
775 ResultVal = RawBytes[0];
776 for (unsigned i = 1; i != BytesLoaded; ++i) {
777 ResultVal <<= 8;
778 ResultVal |= RawBytes[i];
779 }
780 }
781
782 return ConstantInt::get(IntType->getContext(), ResultVal);
783}
784
785} // anonymous namespace
786
787// If GV is a constant with an initializer read its representation starting
788// at Offset and return it as a constant array of unsigned char. Otherwise
789// return null.
791 uint64_t Offset) {
792 if (!GV->isConstant() || !GV->hasDefinitiveInitializer())
793 return nullptr;
794
795 const DataLayout &DL = GV->getDataLayout();
796 Constant *Init = const_cast<Constant *>(GV->getInitializer());
797 TypeSize InitSize = DL.getTypeAllocSize(Init->getType());
798 if (InitSize < Offset)
799 return nullptr;
800
801 uint64_t NBytes = InitSize - Offset;
802 if (NBytes > UINT16_MAX)
803 // Bail for large initializers in excess of 64K to avoid allocating
804 // too much memory.
805 // Offset is assumed to be less than or equal than InitSize (this
806 // is enforced in ReadDataFromGlobal).
807 return nullptr;
808
809 SmallVector<unsigned char, 256> RawBytes(static_cast<size_t>(NBytes));
810 unsigned char *CurPtr = RawBytes.data();
811
812 if (!ReadDataFromGlobal(Init, Offset, CurPtr, NBytes, DL))
813 return nullptr;
814
815 return ConstantDataArray::get(GV->getContext(), RawBytes);
816}
817
818/// If this Offset points exactly to the start of an aggregate element, return
819/// that element, otherwise return nullptr.
821 const DataLayout &DL) {
822 if (Offset.isZero())
823 return Base;
824
826 return nullptr;
827
828 Type *ElemTy = Base->getType();
829 SmallVector<APInt> Indices = DL.getGEPIndicesForOffset(ElemTy, Offset);
830 if (!Offset.isZero() || !Indices[0].isZero())
831 return nullptr;
832
833 Constant *C = Base;
834 for (const APInt &Index : drop_begin(Indices)) {
835 if (Index.isNegative() || Index.getActiveBits() >= 32)
836 return nullptr;
837
838 C = C->getAggregateElement(Index.getZExtValue());
839 if (!C)
840 return nullptr;
841 }
842
843 return C;
844}
845
847 const APInt &Offset,
848 const DataLayout &DL) {
849 if (Constant *AtOffset = getConstantAtOffset(C, Offset, DL))
850 if (Constant *Result = ConstantFoldLoadThroughBitcast(AtOffset, Ty, DL))
851 return Result;
852
853 // Explicitly check for out-of-bounds access, so we return poison even if the
854 // constant is a uniform value.
855 TypeSize Size = DL.getTypeAllocSize(C->getType());
856 if (!Size.isScalable() && Offset.sge(Size.getFixedValue()))
857 return PoisonValue::get(Ty);
858
859 // Try an offset-independent fold of a uniform value.
860 if (Constant *Result = ConstantFoldLoadFromUniformValue(C, Ty, DL))
861 return Result;
862
863 // Try hard to fold loads from bitcasted strange and non-type-safe things.
864 if (Offset.getSignificantBits() <= 64)
865 if (Constant *Result =
866 FoldReinterpretLoadFromConst(C, Ty, Ty, Offset.getSExtValue(), DL))
867 return Result;
868
869 return nullptr;
870}
871
876
879 const DataLayout &DL) {
880 // We can only fold loads from constant globals with a definitive initializer.
881 // Check this upfront, to skip expensive offset calculations.
883 if (!GV || !GV->isConstant() || !GV->hasDefinitiveInitializer())
884 return nullptr;
885
886 C = cast<Constant>(C->stripAndAccumulateConstantOffsets(
887 DL, Offset, /* AllowNonInbounds */ true));
888
889 if (C == GV)
890 if (Constant *Result = ConstantFoldLoadFromConst(GV->getInitializer(), Ty,
891 Offset, DL))
892 return Result;
893
894 // If this load comes from anywhere in a uniform constant global, the value
895 // is always the same, regardless of the loaded offset.
896 return ConstantFoldLoadFromUniformValue(GV->getInitializer(), Ty, DL);
897}
898
900 const DataLayout &DL) {
901 APInt Offset(DL.getIndexTypeSizeInBits(C->getType()), 0);
902 return ConstantFoldLoadFromConstPtr(C, Ty, std::move(Offset), DL);
903}
904
906 const DataLayout &DL) {
907 if (isa<PoisonValue>(C))
908 return PoisonValue::get(Ty);
909 if (isa<UndefValue>(C))
910 return UndefValue::get(Ty);
911 // If padding is needed when storing C to memory, then it isn't considered as
912 // uniform.
913 if (!DL.typeSizeEqualsStoreSize(C->getType()))
914 return nullptr;
915 if (C->isNullValue() && !Ty->isX86_AMXTy())
916 return Constant::getNullValue(Ty);
917 if (C->isAllOnesValue() &&
918 (Ty->isIntOrIntVectorTy() || Ty->isByteOrByteVectorTy() ||
919 Ty->isFPOrFPVectorTy()))
920 return Constant::getAllOnesValue(Ty);
921 return nullptr;
922}
923
924namespace {
925
926/// One of Op0/Op1 is a constant expression.
927/// Attempt to symbolically evaluate the result of a binary operator merging
928/// these together. If target data info is available, it is provided as DL,
929/// otherwise DL is null.
930Constant *SymbolicallyEvaluateBinop(unsigned Opc, Constant *Op0, Constant *Op1,
931 const DataLayout &DL) {
932 // SROA
933
934 // Fold (and 0xffffffff00000000, (shl x, 32)) -> shl.
935 // Fold (lshr (or X, Y), 32) -> (lshr [X/Y], 32) if one doesn't contribute
936 // bits.
937
938 if (Opc == Instruction::And) {
939 KnownBits Known0 = computeKnownBits(Op0, DL);
940 KnownBits Known1 = computeKnownBits(Op1, DL);
941 if ((Known1.One | Known0.Zero).isAllOnes()) {
942 // All the bits of Op0 that the 'and' could be masking are already zero.
943 return Op0;
944 }
945 if ((Known0.One | Known1.Zero).isAllOnes()) {
946 // All the bits of Op1 that the 'and' could be masking are already zero.
947 return Op1;
948 }
949
950 Known0 &= Known1;
951 if (Known0.isConstant())
952 return ConstantInt::get(Op0->getType(), Known0.getConstant());
953 }
954
955 // If the constant expr is something like &A[123] - &A[4].f, fold this into a
956 // constant. This happens frequently when iterating over a global array.
957 if (Opc == Instruction::Sub) {
958 GlobalValue *GV1, *GV2;
959 APInt Offs1, Offs2;
960
961 if (IsConstantOffsetFromGlobal(Op0, GV1, Offs1, DL))
962 if (IsConstantOffsetFromGlobal(Op1, GV2, Offs2, DL) && GV1 == GV2) {
963 unsigned OpSize = DL.getTypeSizeInBits(Op0->getType());
964
965 // (&GV+C1) - (&GV+C2) -> C1-C2, pointer arithmetic cannot overflow.
966 // PtrToInt may change the bitwidth so we have convert to the right size
967 // first.
968 return ConstantInt::get(Op0->getType(), Offs1.zextOrTrunc(OpSize) -
969 Offs2.zextOrTrunc(OpSize));
970 }
971 }
972
973 return nullptr;
974}
975
976/// If array indices are not pointer-sized integers, explicitly cast them so
977/// that they aren't implicitly casted by the getelementptr.
978Constant *CastGEPIndices(Type *SrcElemTy, ArrayRef<Constant *> Ops,
979 Type *ResultTy, GEPNoWrapFlags NW,
980 std::optional<ConstantRange> InRange,
981 const DataLayout &DL, const TargetLibraryInfo *TLI) {
982 Type *IntIdxTy = DL.getIndexType(ResultTy);
983 Type *IntIdxScalarTy = IntIdxTy->getScalarType();
984
985 bool Any = false;
987 for (unsigned i = 1, e = Ops.size(); i != e; ++i) {
988 if ((i == 1 ||
990 SrcElemTy, Ops.slice(1, i - 1)))) &&
991 Ops[i]->getType()->getScalarType() != IntIdxScalarTy) {
992 Any = true;
993 Type *NewType =
994 Ops[i]->getType()->isVectorTy() ? IntIdxTy : IntIdxScalarTy;
996 CastInst::getCastOpcode(Ops[i], true, NewType, true), Ops[i], NewType,
997 DL);
998 if (!NewIdx)
999 return nullptr;
1000 NewIdxs.push_back(NewIdx);
1001 } else
1002 NewIdxs.push_back(Ops[i]);
1003 }
1004
1005 if (!Any)
1006 return nullptr;
1007
1008 Constant *C = ConstantExpr::getGetElementPtr(DL, SrcElemTy, Ops[0], NewIdxs,
1009 NW, InRange);
1010 if (!C)
1011 return nullptr;
1012 return ConstantFoldConstant(C, DL, TLI);
1013}
1014
1015/// If we can symbolically evaluate the GEP constant expression, do so.
1016Constant *SymbolicallyEvaluateGEP(const GEPOperator *GEP,
1018 const DataLayout &DL,
1019 const TargetLibraryInfo *TLI) {
1020 Type *SrcElemTy = GEP->getSourceElementType();
1021 Type *ResTy = GEP->getType();
1022 if (!SrcElemTy->isSized() || isa<ScalableVectorType>(SrcElemTy))
1023 return nullptr;
1024
1025 if (Constant *C = CastGEPIndices(SrcElemTy, Ops, ResTy, GEP->getNoWrapFlags(),
1026 GEP->getInRange(), DL, TLI))
1027 return C;
1028
1029 Constant *Ptr = Ops[0];
1030 if (!Ptr->getType()->isPointerTy())
1031 return nullptr;
1032
1033 Type *IntIdxTy = DL.getIndexType(Ptr->getType());
1034
1035 for (unsigned i = 1, e = Ops.size(); i != e; ++i)
1036 if (!isa<ConstantInt>(Ops[i]) || !Ops[i]->getType()->isIntegerTy())
1037 return nullptr;
1038
1039 unsigned BitWidth = DL.getTypeSizeInBits(IntIdxTy);
1040 APInt Offset = APInt(
1041 BitWidth,
1042 DL.getIndexedOffsetInType(
1043 SrcElemTy, ArrayRef((Value *const *)Ops.data() + 1, Ops.size() - 1)),
1044 /*isSigned=*/true, /*implicitTrunc=*/true);
1045
1046 std::optional<ConstantRange> InRange = GEP->getInRange();
1047 if (InRange)
1048 InRange = InRange->sextOrTrunc(BitWidth);
1049
1050 // If this is a GEP of a GEP, fold it all into a single GEP.
1051 GEPNoWrapFlags NW = GEP->getNoWrapFlags();
1052 bool Overflow = false;
1053 while (auto *GEP = dyn_cast<GEPOperator>(Ptr)) {
1054 NW &= GEP->getNoWrapFlags();
1055
1056 SmallVector<Value *, 4> NestedOps(llvm::drop_begin(GEP->operands()));
1057
1058 // Do not try the incorporate the sub-GEP if some index is not a number.
1059 bool AllConstantInt = true;
1060 for (Value *NestedOp : NestedOps)
1061 if (!isa<ConstantInt>(NestedOp)) {
1062 AllConstantInt = false;
1063 break;
1064 }
1065 if (!AllConstantInt)
1066 break;
1067
1068 // Adjust inrange offset and intersect inrange attributes
1069 if (auto GEPRange = GEP->getInRange()) {
1070 auto AdjustedGEPRange = GEPRange->sextOrTrunc(BitWidth).subtract(Offset);
1071 InRange =
1072 InRange ? InRange->intersectWith(AdjustedGEPRange) : AdjustedGEPRange;
1073 }
1074
1075 Ptr = cast<Constant>(GEP->getOperand(0));
1076 SrcElemTy = GEP->getSourceElementType();
1077 Offset = Offset.sadd_ov(
1078 APInt(BitWidth, DL.getIndexedOffsetInType(SrcElemTy, NestedOps),
1079 /*isSigned=*/true, /*implicitTrunc=*/true),
1080 Overflow);
1081 }
1082
1083 // Preserving nusw (without inbounds) also requires that the offset
1084 // additions did not overflow.
1085 if (NW.hasNoUnsignedSignedWrap() && !NW.isInBounds() && Overflow)
1087
1088 // If the base value for this address is a literal integer value, fold the
1089 // getelementptr to the resulting integer value casted to the pointer type.
1090 APInt BaseIntVal(DL.getPointerTypeSizeInBits(Ptr->getType()), 0);
1091 if (auto *CE = dyn_cast<ConstantExpr>(Ptr)) {
1092 if (CE->getOpcode() == Instruction::IntToPtr) {
1093 if (auto *Base = dyn_cast<ConstantInt>(CE->getOperand(0)))
1094 BaseIntVal = Base->getValue().zextOrTrunc(BaseIntVal.getBitWidth());
1095 }
1096 }
1097
1098 if ((Ptr->isNullValue() || BaseIntVal != 0) &&
1099 !DL.mustNotIntroduceIntToPtr(Ptr->getType())) {
1100
1101 // If the index size is smaller than the pointer size, add to the low
1102 // bits only.
1103 BaseIntVal.insertBits(BaseIntVal.trunc(BitWidth) + Offset, 0);
1104 Constant *C = ConstantInt::get(Ptr->getContext(), BaseIntVal);
1105 return ConstantExpr::getIntToPtr(C, ResTy);
1106 }
1107
1108 // Try to infer inbounds for GEPs of globals.
1109 if (!NW.isInBounds() && Offset.isNonNegative()) {
1110 bool CanBeNull;
1111 uint64_t DerefBytes = Ptr->getPointerDereferenceableBytes(
1112 DL, CanBeNull, /*CanBeFreed=*/nullptr);
1113 if (DerefBytes != 0 && !CanBeNull && Offset.sle(DerefBytes))
1115 }
1116
1117 // nusw + nneg -> nuw
1118 if (NW.hasNoUnsignedSignedWrap() && Offset.isNonNegative())
1120
1121 // Otherwise canonicalize this to a single ptradd.
1122 LLVMContext &Ctx = Ptr->getContext();
1123 return ConstantExpr::getPtrAdd(Ptr, ConstantInt::get(Ctx, Offset), NW,
1124 InRange);
1125}
1126
1127/// Attempt to constant fold an instruction with the
1128/// specified opcode and operands. If successful, the constant result is
1129/// returned, if not, null is returned. Note that this function can fail when
1130/// attempting to fold instructions like loads and stores, which have no
1131/// constant expression form.
1132Constant *ConstantFoldInstOperandsImpl(const Value *InstOrCE, unsigned Opcode,
1134 const DataLayout &DL,
1135 const TargetLibraryInfo *TLI,
1136 bool AllowNonDeterministic) {
1137 Type *DestTy = InstOrCE->getType();
1138
1139 if (Instruction::isUnaryOp(Opcode))
1140 return ConstantFoldUnaryOpOperand(Opcode, Ops[0], DL);
1141
1142 if (Instruction::isBinaryOp(Opcode)) {
1143 switch (Opcode) {
1144 default:
1145 break;
1146 case Instruction::FAdd:
1147 case Instruction::FSub:
1148 case Instruction::FMul:
1149 case Instruction::FDiv:
1150 case Instruction::FRem:
1151 // Handle floating point instructions separately to account for denormals
1152 // TODO: If a constant expression is being folded rather than an
1153 // instruction, denormals will not be flushed/treated as zero
1154 if (const auto *I = dyn_cast<Instruction>(InstOrCE)) {
1155 return ConstantFoldFPInstOperands(Opcode, Ops[0], Ops[1], DL, I,
1156 AllowNonDeterministic);
1157 }
1158 }
1159 return ConstantFoldBinaryOpOperands(Opcode, Ops[0], Ops[1], DL);
1160 }
1161
1162 if (Instruction::isCast(Opcode))
1163 return ConstantFoldCastOperand(Opcode, Ops[0], DestTy, DL);
1164
1165 if (auto *GEP = dyn_cast<GEPOperator>(InstOrCE)) {
1166 Type *SrcElemTy = GEP->getSourceElementType();
1168 return nullptr;
1169
1170 if (Constant *C = SymbolicallyEvaluateGEP(GEP, Ops, DL, TLI))
1171 return C;
1172
1173 return ConstantExpr::getGetElementPtr(DL, SrcElemTy, Ops[0], Ops.slice(1),
1174 GEP->getNoWrapFlags(),
1175 GEP->getInRange());
1176 }
1177
1178 if (auto *CE = dyn_cast<ConstantExpr>(InstOrCE))
1179 return CE->getWithOperands(Ops);
1180
1181 switch (Opcode) {
1182 default: return nullptr;
1183 case Instruction::ICmp:
1184 case Instruction::FCmp: {
1185 auto *C = cast<CmpInst>(InstOrCE);
1186 return ConstantFoldCompareInstOperands(C->getPredicate(), Ops[0], Ops[1],
1187 DL, TLI, C->getFunction());
1188 }
1189 case Instruction::Freeze:
1190 return isGuaranteedNotToBeUndefOrPoison(Ops[0]) ? Ops[0] : nullptr;
1191 case Instruction::Call:
1192 if (auto *F = dyn_cast<Function>(Ops.back())) {
1193 const auto *Call = cast<CallBase>(InstOrCE);
1194 if (canConstantFoldCallTo(Call, F, TLI))
1195 return ConstantFoldCall(Call, F, Ops.slice(0, Ops.size() - 1), TLI,
1196 AllowNonDeterministic);
1197 }
1198 return nullptr;
1199 case Instruction::Select:
1200 return ConstantFoldSelectInstruction(Ops[0], Ops[1], Ops[2]);
1201 case Instruction::ExtractElement:
1203 case Instruction::ExtractValue:
1205 Ops[0], cast<ExtractValueInst>(InstOrCE)->getIndices());
1206 case Instruction::InsertElement:
1207 return ConstantExpr::getInsertElement(Ops[0], Ops[1], Ops[2]);
1208 case Instruction::InsertValue:
1210 Ops[0], Ops[1], cast<InsertValueInst>(InstOrCE)->getIndices());
1211 case Instruction::ShuffleVector:
1213 Ops[0], Ops[1], cast<ShuffleVectorInst>(InstOrCE)->getShuffleMask());
1214 case Instruction::Load: {
1215 const auto *LI = dyn_cast<LoadInst>(InstOrCE);
1216 if (LI->isVolatile())
1217 return nullptr;
1218 return ConstantFoldLoadFromConstPtr(Ops[0], LI->getType(), DL);
1219 }
1220 }
1221}
1222
1223} // end anonymous namespace
1224
1225//===----------------------------------------------------------------------===//
1226// Constant Folding public APIs
1227//===----------------------------------------------------------------------===//
1228
1229namespace {
1230
1231Constant *
1232ConstantFoldConstantImpl(const Constant *C, const DataLayout &DL,
1233 const TargetLibraryInfo *TLI,
1236 return const_cast<Constant *>(C);
1237
1239 for (const Use &OldU : C->operands()) {
1240 Constant *OldC = cast<Constant>(&OldU);
1241 Constant *NewC = OldC;
1242 // Recursively fold the ConstantExpr's operands. If we have already folded
1243 // a ConstantExpr, we don't have to process it again.
1244 if (isa<ConstantVector>(OldC) || isa<ConstantExpr>(OldC)) {
1245 auto It = FoldedOps.find(OldC);
1246 if (It == FoldedOps.end()) {
1247 NewC = ConstantFoldConstantImpl(OldC, DL, TLI, FoldedOps);
1248 FoldedOps.insert({OldC, NewC});
1249 } else {
1250 NewC = It->second;
1251 }
1252 }
1253 Ops.push_back(NewC);
1254 }
1255
1256 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1257 if (Constant *Res = ConstantFoldInstOperandsImpl(
1258 CE, CE->getOpcode(), Ops, DL, TLI, /*AllowNonDeterministic=*/true))
1259 return Res;
1260 return const_cast<Constant *>(C);
1261 }
1262
1264 return ConstantVector::get(Ops);
1265}
1266
1267} // end anonymous namespace
1268
1270 const DataLayout &DL,
1271 const TargetLibraryInfo *TLI) {
1272 // Handle PHI nodes quickly here...
1273 if (auto *PN = dyn_cast<PHINode>(I)) {
1274 Constant *CommonValue = nullptr;
1275
1277 for (Value *Incoming : PN->incoming_values()) {
1278 // If the incoming value is undef then skip it. Note that while we could
1279 // skip the value if it is equal to the phi node itself we choose not to
1280 // because that would break the rule that constant folding only applies if
1281 // all operands are constants.
1282 if (isa<UndefValue>(Incoming))
1283 continue;
1284 // If the incoming value is not a constant, then give up.
1285 auto *C = dyn_cast<Constant>(Incoming);
1286 if (!C)
1287 return nullptr;
1288 // Fold the PHI's operands.
1289 C = ConstantFoldConstantImpl(C, DL, TLI, FoldedOps);
1290 // If the incoming value is a different constant to
1291 // the one we saw previously, then give up.
1292 if (CommonValue && C != CommonValue)
1293 return nullptr;
1294 CommonValue = C;
1295 }
1296
1297 // If we reach here, all incoming values are the same constant or undef.
1298 return CommonValue ? CommonValue : UndefValue::get(PN->getType());
1299 }
1300
1301 // Scan the operand list, checking to see if they are all constants, if so,
1302 // hand off to ConstantFoldInstOperandsImpl.
1303 if (!all_of(I->operands(), [](const Use &U) { return isa<Constant>(U); }))
1304 return nullptr;
1305
1308 for (const Use &OpU : I->operands()) {
1309 auto *Op = cast<Constant>(&OpU);
1310 // Fold the Instruction's operands.
1311 Op = ConstantFoldConstantImpl(Op, DL, TLI, FoldedOps);
1312 Ops.push_back(Op);
1313 }
1314
1315 return ConstantFoldInstOperands(I, Ops, DL, TLI);
1316}
1317
1319 const TargetLibraryInfo *TLI) {
1321 return ConstantFoldConstantImpl(C, DL, TLI, FoldedOps);
1322}
1323
1326 const DataLayout &DL,
1327 const TargetLibraryInfo *TLI,
1328 bool AllowNonDeterministic) {
1329 return ConstantFoldInstOperandsImpl(I, I->getOpcode(), Ops, DL, TLI,
1330 AllowNonDeterministic);
1331}
1332
1334 Constant *Ops0, Constant *Ops1,
1335 const DataLayout &DL,
1336 const TargetLibraryInfo *TLI,
1337 const Function *CtxF) {
1338 CmpInst::Predicate Predicate = (CmpInst::Predicate)IntPredicate;
1339 // fold: icmp (inttoptr x), null -> icmp x, 0
1340 // fold: icmp null, (inttoptr x) -> icmp 0, x
1341 // fold: icmp (ptrtoint x), 0 -> icmp x, null
1342 // fold: icmp 0, (ptrtoint x) -> icmp null, x
1343 // fold: icmp (inttoptr x), (inttoptr y) -> icmp trunc/zext x, trunc/zext y
1344 // fold: icmp (ptrtoint x), (ptrtoint y) -> icmp x, y
1345 //
1346 // FIXME: The following comment is out of data and the DataLayout is here now.
1347 // ConstantExpr::getCompare cannot do this, because it doesn't have DL
1348 // around to know if bit truncation is happening.
1349 if (auto *CE0 = dyn_cast<ConstantExpr>(Ops0)) {
1350 if (Ops1->isNullValue()) {
1351 if (CE0->getOpcode() == Instruction::IntToPtr) {
1352 Type *IntPtrTy = DL.getIntPtrType(CE0->getType());
1353 // Convert the integer value to the right size to ensure we get the
1354 // proper extension or truncation.
1355 if (Constant *C = ConstantFoldIntegerCast(CE0->getOperand(0), IntPtrTy,
1356 /*IsSigned*/ false, DL)) {
1357 Constant *Null = Constant::getNullValue(C->getType());
1358 return ConstantFoldCompareInstOperands(Predicate, C, Null, DL, TLI);
1359 }
1360 }
1361
1362 // icmp only compares the address part of the pointer, so only do this
1363 // transform if the integer size matches the address size.
1364 if (CE0->getOpcode() == Instruction::PtrToInt ||
1365 CE0->getOpcode() == Instruction::PtrToAddr) {
1366 Type *AddrTy = DL.getAddressType(CE0->getOperand(0)->getType());
1367 if (CE0->getType() == AddrTy) {
1368 Constant *C = CE0->getOperand(0);
1369 Constant *Null = Constant::getNullValue(C->getType());
1370 return ConstantFoldCompareInstOperands(Predicate, C, Null, DL, TLI);
1371 }
1372 }
1373 }
1374
1375 if (auto *CE1 = dyn_cast<ConstantExpr>(Ops1)) {
1376 if (CE0->getOpcode() == CE1->getOpcode()) {
1377 if (CE0->getOpcode() == Instruction::IntToPtr) {
1378 Type *IntPtrTy = DL.getIntPtrType(CE0->getType());
1379
1380 // Convert the integer value to the right size to ensure we get the
1381 // proper extension or truncation.
1382 Constant *C0 = ConstantFoldIntegerCast(CE0->getOperand(0), IntPtrTy,
1383 /*IsSigned*/ false, DL);
1384 Constant *C1 = ConstantFoldIntegerCast(CE1->getOperand(0), IntPtrTy,
1385 /*IsSigned*/ false, DL);
1386 if (C0 && C1)
1387 return ConstantFoldCompareInstOperands(Predicate, C0, C1, DL, TLI);
1388 }
1389
1390 // icmp only compares the address part of the pointer, so only do this
1391 // transform if the integer size matches the address size.
1392 if (CE0->getOpcode() == Instruction::PtrToInt ||
1393 CE0->getOpcode() == Instruction::PtrToAddr) {
1394 Type *AddrTy = DL.getAddressType(CE0->getOperand(0)->getType());
1395 if (CE0->getType() == AddrTy &&
1396 CE0->getOperand(0)->getType() == CE1->getOperand(0)->getType()) {
1398 Predicate, CE0->getOperand(0), CE1->getOperand(0), DL, TLI);
1399 }
1400 }
1401 }
1402 }
1403
1404 // Convert pointer comparison (base+offset1) pred (base+offset2) into
1405 // offset1 pred offset2, for the case where the offset is inbounds. This
1406 // only works for equality and unsigned comparison, as inbounds permits
1407 // crossing the sign boundary. However, the offset comparison itself is
1408 // signed.
1409 if (Ops0->getType()->isPointerTy() && !ICmpInst::isSigned(Predicate)) {
1410 unsigned IndexWidth = DL.getIndexTypeSizeInBits(Ops0->getType());
1411 APInt Offset0(IndexWidth, 0);
1412 bool IsEqPred = ICmpInst::isEquality(Predicate);
1413 Value *Stripped0 = Ops0->stripAndAccumulateConstantOffsets(
1414 DL, Offset0, /*AllowNonInbounds=*/IsEqPred,
1415 /*AllowInvariantGroup=*/false, /*ExternalAnalysis=*/nullptr,
1416 /*LookThroughIntToPtr=*/IsEqPred);
1417 APInt Offset1(IndexWidth, 0);
1418 Value *Stripped1 = Ops1->stripAndAccumulateConstantOffsets(
1419 DL, Offset1, /*AllowNonInbounds=*/IsEqPred,
1420 /*AllowInvariantGroup=*/false, /*ExternalAnalysis=*/nullptr,
1421 /*LookThroughIntToPtr=*/IsEqPred);
1422 if (Stripped0 == Stripped1)
1423 return ConstantInt::getBool(
1424 Ops0->getContext(),
1425 ICmpInst::compare(Offset0, Offset1,
1426 ICmpInst::getSignedPredicate(Predicate)));
1427 }
1428 } else if (isa<ConstantExpr>(Ops1)) {
1429 // If RHS is a constant expression, but the left side isn't, swap the
1430 // operands and try again.
1431 Predicate = ICmpInst::getSwappedPredicate(Predicate);
1432 return ConstantFoldCompareInstOperands(Predicate, Ops1, Ops0, DL, TLI);
1433 }
1434
1435 if (CmpInst::isFPPredicate(Predicate)) {
1436 // Flush any denormal constant float input according to denormal handling
1437 // mode.
1438 Ops0 = FlushFPConstant(Ops0, CtxF, /*IsOutput=*/false);
1439 if (!Ops0)
1440 return nullptr;
1441 Ops1 = FlushFPConstant(Ops1, CtxF, /*IsOutput=*/false);
1442 if (!Ops1)
1443 return nullptr;
1444 }
1445
1446 return ConstantFoldCompareInstruction(Predicate, Ops0, Ops1);
1447}
1448
1450 const DataLayout &DL) {
1452
1453 return ConstantFoldUnaryInstruction(Opcode, Op);
1454}
1455
1457 Constant *RHS,
1458 const DataLayout &DL) {
1460 if (isa<ConstantExpr>(LHS) || isa<ConstantExpr>(RHS))
1461 if (Constant *C = SymbolicallyEvaluateBinop(Opcode, LHS, RHS, DL))
1462 return C;
1463
1465 return ConstantExpr::get(Opcode, LHS, RHS);
1466 return ConstantFoldBinaryInstruction(Opcode, LHS, RHS);
1467}
1468
1471 switch (Mode) {
1473 return nullptr;
1474 case DenormalMode::IEEE:
1475 return ConstantFP::get(Ty, APF);
1477 return ConstantFP::get(
1478 Ty, APFloat::getZero(APF.getSemantics(), APF.isNegative()));
1480 return ConstantFP::get(Ty, APFloat::getZero(APF.getSemantics(), false));
1481 default:
1482 break;
1483 }
1484
1485 llvm_unreachable("unknown denormal mode");
1486}
1487
1488/// Return the denormal mode that can be assumed when executing a floating point
1489/// operation at \p CtxI.
1491 if (!CtxF)
1492 return DenormalMode::getDynamic();
1493 return CtxF->getDenormalMode(Ty->getScalarType()->getFltSemantics());
1494}
1495
1496static ConstantFP *
1497flushDenormalConstantFP(ConstantFP *CFP, const Function *CtxF, bool IsOutput) {
1498 const APFloat &APF = CFP->getValueAPF();
1499 if (!APF.isDenormal())
1500 return CFP;
1501
1503 return flushDenormalConstant(CFP->getType(), APF,
1504 IsOutput ? Mode.Output : Mode.Input);
1505}
1506
1508 bool IsOutput) {
1509 if (ConstantFP *CFP = dyn_cast<ConstantFP>(Operand))
1510 return flushDenormalConstantFP(CFP, CtxF, IsOutput);
1511
1513 return Operand;
1514
1515 Type *Ty = Operand->getType();
1516 VectorType *VecTy = dyn_cast<VectorType>(Ty);
1517 if (VecTy) {
1518 if (auto *Splat = dyn_cast_or_null<ConstantFP>(Operand->getSplatValue())) {
1519 ConstantFP *Folded = flushDenormalConstantFP(Splat, CtxF, IsOutput);
1520 if (!Folded)
1521 return nullptr;
1522 return ConstantVector::getSplat(VecTy->getElementCount(), Folded);
1523 }
1524
1525 Ty = VecTy->getElementType();
1526 }
1527
1528 if (isa<ConstantExpr>(Operand))
1529 return Operand;
1530
1531 if (const auto *CV = dyn_cast<ConstantVector>(Operand)) {
1533 for (unsigned i = 0, e = CV->getNumOperands(); i != e; ++i) {
1534 Constant *Element = CV->getAggregateElement(i);
1535 if (isa<UndefValue>(Element)) {
1536 NewElts.push_back(Element);
1537 continue;
1538 }
1539
1540 ConstantFP *CFP = dyn_cast<ConstantFP>(Element);
1541 if (!CFP)
1542 return nullptr;
1543
1544 ConstantFP *Folded = flushDenormalConstantFP(CFP, CtxF, IsOutput);
1545 if (!Folded)
1546 return nullptr;
1547 NewElts.push_back(Folded);
1548 }
1549
1550 return ConstantVector::get(NewElts);
1551 }
1552
1553 if (const auto *CDV = dyn_cast<ConstantDataVector>(Operand)) {
1555 for (unsigned I = 0, E = CDV->getNumElements(); I < E; ++I) {
1556 const APFloat &Elt = CDV->getElementAsAPFloat(I);
1557 if (!Elt.isDenormal()) {
1558 NewElts.push_back(ConstantFP::get(Ty, Elt));
1559 } else {
1560 DenormalMode Mode = getInstrDenormalMode(CtxF, Ty);
1561 ConstantFP *Folded =
1562 flushDenormalConstant(Ty, Elt, IsOutput ? Mode.Output : Mode.Input);
1563 if (!Folded)
1564 return nullptr;
1565 NewElts.push_back(Folded);
1566 }
1567 }
1568
1569 return ConstantVector::get(NewElts);
1570 }
1571
1572 return nullptr;
1573}
1574
1576 Constant *RHS, const DataLayout &DL,
1577 const Instruction *I,
1578 bool AllowNonDeterministic) {
1579 if (Instruction::isBinaryOp(Opcode)) {
1580 // Flush denormal inputs if needed.
1581 Constant *Op0 =
1582 FlushFPConstant(LHS, I->getFunction(), /* IsOutput */ false);
1583 if (!Op0)
1584 return nullptr;
1585 Constant *Op1 =
1586 FlushFPConstant(RHS, I->getFunction(), /* IsOutput */ false);
1587 if (!Op1)
1588 return nullptr;
1589
1590 // If nsz or an algebraic FMF flag is set, the result of the FP operation
1591 // may change due to future optimization. Don't constant fold them if
1592 // non-deterministic results are not allowed.
1593 if (!AllowNonDeterministic)
1595 if (FP->hasNoSignedZeros() || FP->hasAllowReassoc() ||
1596 FP->hasAllowContract() || FP->hasAllowReciprocal())
1597 return nullptr;
1598
1599 // Calculate constant result.
1600 Constant *C = ConstantFoldBinaryOpOperands(Opcode, Op0, Op1, DL);
1601 if (!C)
1602 return nullptr;
1603
1604 // Flush denormal output if needed.
1605 C = FlushFPConstant(C, I->getFunction(), /* IsOutput */ true);
1606 if (!C)
1607 return nullptr;
1608
1609 // The precise NaN value is non-deterministic.
1610 if (!AllowNonDeterministic && C->isNaN())
1611 return nullptr;
1612
1613 return C;
1614 }
1615 // If instruction lacks a parent/function and the denormal mode cannot be
1616 // determined, use the default (IEEE).
1617 return ConstantFoldBinaryOpOperands(Opcode, LHS, RHS, DL);
1618}
1619
1621 Type *DestTy, const DataLayout &DL) {
1622 assert(Instruction::isCast(Opcode));
1623
1624 if (auto *CE = dyn_cast<ConstantExpr>(C))
1625 if (CE->isCast())
1626 if (unsigned NewOp = CastInst::isEliminableCastPair(
1627 Instruction::CastOps(CE->getOpcode()),
1628 Instruction::CastOps(Opcode), CE->getOperand(0)->getType(),
1629 C->getType(), DestTy, &DL))
1630 return ConstantFoldCastOperand(NewOp, CE->getOperand(0), DestTy, DL);
1631
1632 switch (Opcode) {
1633 default:
1634 llvm_unreachable("Missing case");
1635 case Instruction::PtrToAddr:
1636 case Instruction::PtrToInt:
1637 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1638 Constant *FoldedValue = nullptr;
1639 // If the input is an inttoptr, eliminate the pair. This requires knowing
1640 // the width of a pointer, so it can't be done in ConstantExpr::getCast.
1641 if (CE->getOpcode() == Instruction::IntToPtr) {
1642 // zext/trunc the inttoptr to pointer/address size.
1643 Type *MidTy = Opcode == Instruction::PtrToInt
1644 ? DL.getAddressType(CE->getType())
1645 : DL.getIntPtrType(CE->getType());
1646 FoldedValue = ConstantFoldIntegerCast(CE->getOperand(0), MidTy,
1647 /*IsSigned=*/false, DL);
1648 } else if (auto *GEP = dyn_cast<GEPOperator>(CE)) {
1649 // If we have GEP, we can perform the following folds:
1650 // (ptrtoint/ptrtoaddr (gep null, x)) -> x
1651 // (ptrtoint/ptrtoaddr (gep (gep null, x), y) -> x + y, etc.
1652 unsigned BitWidth = DL.getIndexTypeSizeInBits(GEP->getType());
1653 APInt BaseOffset(BitWidth, 0);
1654 auto *Base = cast<Constant>(GEP->stripAndAccumulateConstantOffsets(
1655 DL, BaseOffset, /*AllowNonInbounds=*/true));
1656 if (Base->isNullValue()) {
1657 FoldedValue = ConstantInt::get(CE->getContext(), BaseOffset);
1658 } else {
1659 // ptrtoint/ptrtoaddr (gep i8, Ptr, (sub 0, V))
1660 // -> sub (ptrtoint/ptrtoaddr Ptr), V
1661 if (GEP->getNumIndices() == 1 &&
1662 GEP->getSourceElementType()->isIntegerTy(8)) {
1663 auto *Ptr = cast<Constant>(GEP->getPointerOperand());
1664 auto *Sub = dyn_cast<ConstantExpr>(GEP->getOperand(1));
1665 Type *IntIdxTy = DL.getIndexType(Ptr->getType());
1666 if (Sub && Sub->getType() == IntIdxTy &&
1667 Sub->getOpcode() == Instruction::Sub &&
1668 Sub->getOperand(0)->isNullValue())
1669 FoldedValue = ConstantExpr::getSub(
1670 ConstantExpr::getCast(Opcode, Ptr, IntIdxTy),
1671 Sub->getOperand(1));
1672 }
1673 }
1674 }
1675 if (FoldedValue) {
1676 // Do a zext or trunc to get to the ptrtoint/ptrtoaddr dest size.
1677 return ConstantFoldIntegerCast(FoldedValue, DestTy, /*IsSigned=*/false,
1678 DL);
1679 }
1680 }
1681 break;
1682 case Instruction::IntToPtr:
1683 // If the input is a ptrtoint, turn the pair into a ptr to ptr bitcast if
1684 // the int size is >= the ptr size and the address spaces are the same.
1685 // This requires knowing the width of a pointer, so it can't be done in
1686 // ConstantExpr::getCast.
1687 if (auto *CE = dyn_cast<ConstantExpr>(C)) {
1688 if (CE->getOpcode() == Instruction::PtrToInt) {
1689 Constant *SrcPtr = CE->getOperand(0);
1690 unsigned SrcPtrSize = DL.getPointerTypeSizeInBits(SrcPtr->getType());
1691 unsigned MidIntSize = CE->getType()->getScalarSizeInBits();
1692
1693 if (MidIntSize >= SrcPtrSize) {
1694 unsigned SrcAS = SrcPtr->getType()->getPointerAddressSpace();
1695 if (SrcAS == DestTy->getPointerAddressSpace())
1696 return FoldBitCast(CE->getOperand(0), DestTy, DL);
1697 }
1698 }
1699 }
1700 break;
1701 case Instruction::Trunc:
1702 case Instruction::ZExt:
1703 case Instruction::SExt:
1704 case Instruction::FPTrunc:
1705 case Instruction::FPExt:
1706 case Instruction::UIToFP:
1707 case Instruction::SIToFP:
1708 case Instruction::FPToUI:
1709 case Instruction::FPToSI:
1710 case Instruction::AddrSpaceCast:
1711 break;
1712 case Instruction::BitCast:
1713 return FoldBitCast(C, DestTy, DL);
1714 }
1715
1717 return ConstantExpr::getCast(Opcode, C, DestTy);
1718 return ConstantFoldCastInstruction(Opcode, C, DestTy);
1719}
1720
1722 bool IsSigned, const DataLayout &DL) {
1723 Type *SrcTy = C->getType();
1724 if (SrcTy == DestTy)
1725 return C;
1726 if (SrcTy->getScalarSizeInBits() > DestTy->getScalarSizeInBits())
1727 return ConstantFoldCastOperand(Instruction::Trunc, C, DestTy, DL);
1728 if (IsSigned)
1729 return ConstantFoldCastOperand(Instruction::SExt, C, DestTy, DL);
1730 return ConstantFoldCastOperand(Instruction::ZExt, C, DestTy, DL);
1731}
1732
1733//===----------------------------------------------------------------------===//
1734// Constant Folding for Calls
1735//
1736
1737/// Returns true if the intrinsic can be constant folded, given \p IsStrictFP.
1738static bool canConstantFoldIntrinsic(Intrinsic::ID ID, bool IsStrictFP) {
1739 switch (ID) {
1740 // Operations that do not operate floating-point numbers and do not depend on
1741 // FP environment can be folded even in strictfp functions.
1742 case Intrinsic::bswap:
1743 case Intrinsic::ctpop:
1744 case Intrinsic::ctlz:
1745 case Intrinsic::cttz:
1746 case Intrinsic::fshl:
1747 case Intrinsic::fshr:
1748 case Intrinsic::clmul:
1749 case Intrinsic::pdep:
1750 case Intrinsic::pext:
1751 case Intrinsic::launder_invariant_group:
1752 case Intrinsic::masked_load:
1753 case Intrinsic::get_active_lane_mask:
1754 case Intrinsic::abs:
1755 case Intrinsic::smax:
1756 case Intrinsic::smin:
1757 case Intrinsic::umax:
1758 case Intrinsic::umin:
1759 case Intrinsic::scmp:
1760 case Intrinsic::ucmp:
1761 case Intrinsic::sadd_with_overflow:
1762 case Intrinsic::uadd_with_overflow:
1763 case Intrinsic::ssub_with_overflow:
1764 case Intrinsic::usub_with_overflow:
1765 case Intrinsic::smul_with_overflow:
1766 case Intrinsic::umul_with_overflow:
1767 case Intrinsic::smulh:
1768 case Intrinsic::umulh:
1769 case Intrinsic::sadd_sat:
1770 case Intrinsic::uadd_sat:
1771 case Intrinsic::ssub_sat:
1772 case Intrinsic::usub_sat:
1773 case Intrinsic::smul_fix:
1774 case Intrinsic::smul_fix_sat:
1775 case Intrinsic::bitreverse:
1776 case Intrinsic::is_constant:
1777 case Intrinsic::vector_reduce_add:
1778 case Intrinsic::vector_reduce_mul:
1779 case Intrinsic::vector_reduce_and:
1780 case Intrinsic::vector_reduce_or:
1781 case Intrinsic::vector_reduce_xor:
1782 case Intrinsic::vector_reduce_smin:
1783 case Intrinsic::vector_reduce_smax:
1784 case Intrinsic::vector_reduce_umin:
1785 case Intrinsic::vector_reduce_umax:
1786 case Intrinsic::vector_partial_reduce_add:
1787 case Intrinsic::vector_extract:
1788 case Intrinsic::vector_insert:
1789 case Intrinsic::vector_interleave2:
1790 case Intrinsic::vector_interleave3:
1791 case Intrinsic::vector_interleave4:
1792 case Intrinsic::vector_interleave5:
1793 case Intrinsic::vector_interleave6:
1794 case Intrinsic::vector_interleave7:
1795 case Intrinsic::vector_interleave8:
1796 case Intrinsic::vector_deinterleave2:
1797 case Intrinsic::vector_deinterleave3:
1798 case Intrinsic::vector_deinterleave4:
1799 case Intrinsic::vector_deinterleave5:
1800 case Intrinsic::vector_deinterleave6:
1801 case Intrinsic::vector_deinterleave7:
1802 case Intrinsic::vector_deinterleave8:
1803 // Target intrinsics
1804 case Intrinsic::amdgcn_perm:
1805 case Intrinsic::amdgcn_wave_reduce_umin:
1806 case Intrinsic::amdgcn_wave_reduce_umax:
1807 case Intrinsic::amdgcn_wave_reduce_max:
1808 case Intrinsic::amdgcn_wave_reduce_min:
1809 case Intrinsic::amdgcn_wave_reduce_and:
1810 case Intrinsic::amdgcn_wave_reduce_or:
1811 case Intrinsic::amdgcn_wave_reduce_xor:
1812 case Intrinsic::amdgcn_wave_reduce_add:
1813 case Intrinsic::amdgcn_wave_reduce_sub:
1814 case Intrinsic::amdgcn_s_wqm:
1815 case Intrinsic::amdgcn_s_quadmask:
1816 case Intrinsic::amdgcn_s_bitreplicate:
1817 case Intrinsic::arm_mve_vctp8:
1818 case Intrinsic::arm_mve_vctp16:
1819 case Intrinsic::arm_mve_vctp32:
1820 case Intrinsic::arm_mve_vctp64:
1821 case Intrinsic::aarch64_crc32b:
1822 case Intrinsic::aarch64_crc32h:
1823 case Intrinsic::aarch64_crc32w:
1824 case Intrinsic::aarch64_crc32x:
1825 case Intrinsic::aarch64_crc32cb:
1826 case Intrinsic::aarch64_crc32ch:
1827 case Intrinsic::aarch64_crc32cw:
1828 case Intrinsic::aarch64_crc32cx:
1829 case Intrinsic::aarch64_sve_convert_from_svbool:
1830 case Intrinsic::wasm_alltrue:
1831 case Intrinsic::wasm_anytrue:
1832 case Intrinsic::wasm_dot:
1833 // WebAssembly float semantics are always known
1834 case Intrinsic::wasm_trunc_signed:
1835 case Intrinsic::wasm_trunc_unsigned:
1836 case Intrinsic::x86_sse42_crc32_32_8:
1837 case Intrinsic::x86_sse42_crc32_32_16:
1838 case Intrinsic::x86_sse42_crc32_32_32:
1839 case Intrinsic::x86_sse42_crc32_64_64:
1840 return true;
1841
1842 // Floating point operations cannot be folded in strictfp functions in
1843 // general case. They can be folded if FP environment is known to compiler.
1844 case Intrinsic::minnum:
1845 case Intrinsic::maxnum:
1846 case Intrinsic::minimum:
1847 case Intrinsic::maximum:
1848 case Intrinsic::minimumnum:
1849 case Intrinsic::maximumnum:
1850 case Intrinsic::log:
1851 case Intrinsic::log2:
1852 case Intrinsic::log10:
1853 case Intrinsic::exp:
1854 case Intrinsic::exp2:
1855 case Intrinsic::exp10:
1856 case Intrinsic::sqrt:
1857 case Intrinsic::sin:
1858 case Intrinsic::cos:
1859 case Intrinsic::sincos:
1860 case Intrinsic::sinh:
1861 case Intrinsic::cosh:
1862 case Intrinsic::atan:
1863 case Intrinsic::pow:
1864 case Intrinsic::powi:
1865 case Intrinsic::ldexp:
1866 case Intrinsic::fma:
1867 case Intrinsic::fmuladd:
1868 case Intrinsic::frexp:
1869 case Intrinsic::fptoui_sat:
1870 case Intrinsic::fptosi_sat:
1871 case Intrinsic::amdgcn_cos:
1872 case Intrinsic::amdgcn_cubeid:
1873 case Intrinsic::amdgcn_cubema:
1874 case Intrinsic::amdgcn_cubesc:
1875 case Intrinsic::amdgcn_cubetc:
1876 case Intrinsic::amdgcn_fmul_legacy:
1877 case Intrinsic::amdgcn_fma_legacy:
1878 case Intrinsic::amdgcn_fract:
1879 case Intrinsic::amdgcn_sin:
1880 // The intrinsics below depend on rounding mode in MXCSR.
1881 case Intrinsic::x86_sse_cvtss2si:
1882 case Intrinsic::x86_sse_cvtss2si64:
1883 case Intrinsic::x86_sse_cvttss2si:
1884 case Intrinsic::x86_sse_cvttss2si64:
1885 case Intrinsic::x86_sse2_cvtsd2si:
1886 case Intrinsic::x86_sse2_cvtsd2si64:
1887 case Intrinsic::x86_sse2_cvttsd2si:
1888 case Intrinsic::x86_sse2_cvttsd2si64:
1889 case Intrinsic::x86_avx512_vcvtss2si32:
1890 case Intrinsic::x86_avx512_vcvtss2si64:
1891 case Intrinsic::x86_avx512_cvttss2si:
1892 case Intrinsic::x86_avx512_cvttss2si64:
1893 case Intrinsic::x86_avx512_vcvtsd2si32:
1894 case Intrinsic::x86_avx512_vcvtsd2si64:
1895 case Intrinsic::x86_avx512_cvttsd2si:
1896 case Intrinsic::x86_avx512_cvttsd2si64:
1897 case Intrinsic::x86_avx512_vcvtss2usi32:
1898 case Intrinsic::x86_avx512_vcvtss2usi64:
1899 case Intrinsic::x86_avx512_cvttss2usi:
1900 case Intrinsic::x86_avx512_cvttss2usi64:
1901 case Intrinsic::x86_avx512_vcvtsd2usi32:
1902 case Intrinsic::x86_avx512_vcvtsd2usi64:
1903 case Intrinsic::x86_avx512_cvttsd2usi:
1904 case Intrinsic::x86_avx512_cvttsd2usi64:
1905
1906 // NVVM FMax intrinsics
1907 case Intrinsic::nvvm_fmax_d:
1908 case Intrinsic::nvvm_fmax_f:
1909 case Intrinsic::nvvm_fmax_ftz_f:
1910 case Intrinsic::nvvm_fmax_ftz_nan_f:
1911 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
1912 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
1913 case Intrinsic::nvvm_fmax_nan_f:
1914 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
1915 case Intrinsic::nvvm_fmax_xorsign_abs_f:
1916
1917 // NVVM FMin intrinsics
1918 case Intrinsic::nvvm_fmin_d:
1919 case Intrinsic::nvvm_fmin_f:
1920 case Intrinsic::nvvm_fmin_ftz_f:
1921 case Intrinsic::nvvm_fmin_ftz_nan_f:
1922 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
1923 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
1924 case Intrinsic::nvvm_fmin_nan_f:
1925 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
1926 case Intrinsic::nvvm_fmin_xorsign_abs_f:
1927
1928 // NVVM float/double to int32/uint32 conversion intrinsics
1929 case Intrinsic::nvvm_f2i_rm:
1930 case Intrinsic::nvvm_f2i_rn:
1931 case Intrinsic::nvvm_f2i_rp:
1932 case Intrinsic::nvvm_f2i_rz:
1933 case Intrinsic::nvvm_f2i_rm_ftz:
1934 case Intrinsic::nvvm_f2i_rn_ftz:
1935 case Intrinsic::nvvm_f2i_rp_ftz:
1936 case Intrinsic::nvvm_f2i_rz_ftz:
1937 case Intrinsic::nvvm_f2ui_rm:
1938 case Intrinsic::nvvm_f2ui_rn:
1939 case Intrinsic::nvvm_f2ui_rp:
1940 case Intrinsic::nvvm_f2ui_rz:
1941 case Intrinsic::nvvm_f2ui_rm_ftz:
1942 case Intrinsic::nvvm_f2ui_rn_ftz:
1943 case Intrinsic::nvvm_f2ui_rp_ftz:
1944 case Intrinsic::nvvm_f2ui_rz_ftz:
1945 case Intrinsic::nvvm_d2i_rm:
1946 case Intrinsic::nvvm_d2i_rn:
1947 case Intrinsic::nvvm_d2i_rp:
1948 case Intrinsic::nvvm_d2i_rz:
1949 case Intrinsic::nvvm_d2ui_rm:
1950 case Intrinsic::nvvm_d2ui_rn:
1951 case Intrinsic::nvvm_d2ui_rp:
1952 case Intrinsic::nvvm_d2ui_rz:
1953
1954 // NVVM float/double to int64/uint64 conversion intrinsics
1955 case Intrinsic::nvvm_f2ll_rm:
1956 case Intrinsic::nvvm_f2ll_rn:
1957 case Intrinsic::nvvm_f2ll_rp:
1958 case Intrinsic::nvvm_f2ll_rz:
1959 case Intrinsic::nvvm_f2ll_rm_ftz:
1960 case Intrinsic::nvvm_f2ll_rn_ftz:
1961 case Intrinsic::nvvm_f2ll_rp_ftz:
1962 case Intrinsic::nvvm_f2ll_rz_ftz:
1963 case Intrinsic::nvvm_f2ull_rm:
1964 case Intrinsic::nvvm_f2ull_rn:
1965 case Intrinsic::nvvm_f2ull_rp:
1966 case Intrinsic::nvvm_f2ull_rz:
1967 case Intrinsic::nvvm_f2ull_rm_ftz:
1968 case Intrinsic::nvvm_f2ull_rn_ftz:
1969 case Intrinsic::nvvm_f2ull_rp_ftz:
1970 case Intrinsic::nvvm_f2ull_rz_ftz:
1971 case Intrinsic::nvvm_d2ll_rm:
1972 case Intrinsic::nvvm_d2ll_rn:
1973 case Intrinsic::nvvm_d2ll_rp:
1974 case Intrinsic::nvvm_d2ll_rz:
1975 case Intrinsic::nvvm_d2ull_rm:
1976 case Intrinsic::nvvm_d2ull_rn:
1977 case Intrinsic::nvvm_d2ull_rp:
1978 case Intrinsic::nvvm_d2ull_rz:
1979
1980 // NVVM math intrinsics:
1981 case Intrinsic::nvvm_ceil_d:
1982 case Intrinsic::nvvm_ceil_f:
1983 case Intrinsic::nvvm_ceil_ftz_f:
1984
1985 case Intrinsic::nvvm_fabs:
1986 case Intrinsic::nvvm_fabs_ftz:
1987
1988 case Intrinsic::nvvm_floor_d:
1989 case Intrinsic::nvvm_floor_f:
1990 case Intrinsic::nvvm_floor_ftz_f:
1991
1992 case Intrinsic::nvvm_rcp_rm_d:
1993 case Intrinsic::nvvm_rcp_rm_f:
1994 case Intrinsic::nvvm_rcp_rm_ftz_f:
1995 case Intrinsic::nvvm_rcp_rn_d:
1996 case Intrinsic::nvvm_rcp_rn_f:
1997 case Intrinsic::nvvm_rcp_rn_ftz_f:
1998 case Intrinsic::nvvm_rcp_rp_d:
1999 case Intrinsic::nvvm_rcp_rp_f:
2000 case Intrinsic::nvvm_rcp_rp_ftz_f:
2001 case Intrinsic::nvvm_rcp_rz_d:
2002 case Intrinsic::nvvm_rcp_rz_f:
2003 case Intrinsic::nvvm_rcp_rz_ftz_f:
2004
2005 case Intrinsic::nvvm_round_d:
2006 case Intrinsic::nvvm_round_f:
2007 case Intrinsic::nvvm_round_ftz_f:
2008
2009 case Intrinsic::nvvm_saturate_d:
2010 case Intrinsic::nvvm_saturate_f:
2011 case Intrinsic::nvvm_saturate_ftz_f:
2012
2013 case Intrinsic::nvvm_sqrt_f:
2014 case Intrinsic::nvvm_sqrt_rn_d:
2015 case Intrinsic::nvvm_sqrt_rn_f:
2016 case Intrinsic::nvvm_sqrt_rn_ftz_f:
2017 return !IsStrictFP;
2018
2019 // NVVM fadd/fmul intrinsics with explicit rounding modes
2020 case Intrinsic::nvvm_fadd:
2021 case Intrinsic::nvvm_fadd_ftz:
2022 case Intrinsic::nvvm_fmul:
2023 case Intrinsic::nvvm_fmul_ftz:
2024
2025 // NVVM div intrinsics with explicit rounding modes
2026 case Intrinsic::nvvm_div_rm_d:
2027 case Intrinsic::nvvm_div_rn_d:
2028 case Intrinsic::nvvm_div_rp_d:
2029 case Intrinsic::nvvm_div_rz_d:
2030 case Intrinsic::nvvm_div_rm_f:
2031 case Intrinsic::nvvm_div_rn_f:
2032 case Intrinsic::nvvm_div_rp_f:
2033 case Intrinsic::nvvm_div_rz_f:
2034 case Intrinsic::nvvm_div_rm_ftz_f:
2035 case Intrinsic::nvvm_div_rn_ftz_f:
2036 case Intrinsic::nvvm_div_rp_ftz_f:
2037 case Intrinsic::nvvm_div_rz_ftz_f:
2038
2039 // NVVM fma intrinsics with explicit rounding modes
2040 case Intrinsic::nvvm_fma_rm_d:
2041 case Intrinsic::nvvm_fma_rn_d:
2042 case Intrinsic::nvvm_fma_rp_d:
2043 case Intrinsic::nvvm_fma_rz_d:
2044 case Intrinsic::nvvm_fma_rm_f:
2045 case Intrinsic::nvvm_fma_rn_f:
2046 case Intrinsic::nvvm_fma_rp_f:
2047 case Intrinsic::nvvm_fma_rz_f:
2048 case Intrinsic::nvvm_fma_rm_ftz_f:
2049 case Intrinsic::nvvm_fma_rn_ftz_f:
2050 case Intrinsic::nvvm_fma_rp_ftz_f:
2051 case Intrinsic::nvvm_fma_rz_ftz_f:
2052
2053 // Sign operations are actually bitwise operations, they do not raise
2054 // exceptions even for SNANs.
2055 case Intrinsic::fabs:
2056 case Intrinsic::copysign:
2057 case Intrinsic::is_fpclass:
2058 // Non-constrained variants of rounding operations means default FP
2059 // environment, they can be folded in any case.
2060 case Intrinsic::ceil:
2061 case Intrinsic::floor:
2062 case Intrinsic::round:
2063 case Intrinsic::roundeven:
2064 case Intrinsic::trunc:
2065 case Intrinsic::nearbyint:
2066 case Intrinsic::rint:
2067 case Intrinsic::canonicalize:
2068
2069 // Constrained intrinsics can be folded if FP environment is known
2070 // to compiler.
2071 case Intrinsic::experimental_constrained_fma:
2072 case Intrinsic::experimental_constrained_fmuladd:
2073 case Intrinsic::experimental_constrained_fadd:
2074 case Intrinsic::experimental_constrained_fsub:
2075 case Intrinsic::experimental_constrained_fmul:
2076 case Intrinsic::experimental_constrained_fdiv:
2077 case Intrinsic::experimental_constrained_frem:
2078 case Intrinsic::experimental_constrained_ceil:
2079 case Intrinsic::experimental_constrained_floor:
2080 case Intrinsic::experimental_constrained_round:
2081 case Intrinsic::experimental_constrained_roundeven:
2082 case Intrinsic::experimental_constrained_trunc:
2083 case Intrinsic::experimental_constrained_nearbyint:
2084 case Intrinsic::experimental_constrained_rint:
2085 case Intrinsic::experimental_constrained_fcmp:
2086 case Intrinsic::experimental_constrained_fcmps:
2087
2088 case Intrinsic::experimental_cttz_elts:
2089 return true;
2090 default:
2091 return false;
2092 }
2093}
2094
2095/// Given a function's return type and its operands, determine if any of them of
2096/// of floating-point type.
2098 return RetTy->isFloatingPointTy() || any_of(Ops, [](Value *V) {
2099 return V->getType()->isFloatingPointTy();
2100 });
2101}
2102
2104 const TargetLibraryInfo *TLI) {
2105 if (Call->isNoBuiltin())
2106 return false;
2107 if (Call->getFunctionType() != F->getFunctionType())
2108 return false;
2109
2110 // Allow FP calls (both libcalls and intrinsics) to avoid being folded.
2111 // This can be useful for GPU targets or in cross-compilation scenarios
2112 // when the exact target FP behaviour is required, and the host compiler's
2113 // behaviour may be slightly different from the device's run-time behaviour.
2116 F->getReturnType(),
2117 ArrayRef<Value *>((Value *const *)(F->arg_begin()), F->arg_size())))
2118 return false;
2119
2120 if (F->getIntrinsicID() != Intrinsic::not_intrinsic)
2121 return canConstantFoldIntrinsic(F->getIntrinsicID(), Call->isStrictFP());
2122
2123 if (!TLI || Call->isStrictFP())
2124 return false;
2125
2126 LibFunc Func = TLI->getLibFunc(*F);
2127 if (Func == NotLibFunc)
2128 return false;
2129
2130 switch (Func) {
2131 case LibFunc_acos:
2132 case LibFunc_acosf:
2133 case LibFunc_acos_finite:
2134 case LibFunc_acosf_finite:
2135 case LibFunc_asin:
2136 case LibFunc_asinf:
2137 case LibFunc_asin_finite:
2138 case LibFunc_asinf_finite:
2139 case LibFunc_atan:
2140 case LibFunc_atanf:
2141 case LibFunc_atan2:
2142 case LibFunc_atan2f:
2143 case LibFunc_atan2_finite:
2144 case LibFunc_atan2f_finite:
2145 case LibFunc_ceil:
2146 case LibFunc_ceilf:
2147 case LibFunc_cosh:
2148 case LibFunc_coshf:
2149 case LibFunc_cosh_finite:
2150 case LibFunc_coshf_finite:
2151 case LibFunc_cos:
2152 case LibFunc_cosf:
2153 case LibFunc_erf:
2154 case LibFunc_erff:
2155 case LibFunc_exp:
2156 case LibFunc_expf:
2157 case LibFunc_exp_finite:
2158 case LibFunc_expf_finite:
2159 case LibFunc_exp2:
2160 case LibFunc_exp2f:
2161 case LibFunc_exp2_finite:
2162 case LibFunc_exp2f_finite:
2163 case LibFunc_fabs:
2164 case LibFunc_fabsf:
2165 case LibFunc_floor:
2166 case LibFunc_floorf:
2167 case LibFunc_fmod:
2168 case LibFunc_fmodf:
2169 case LibFunc_ilogb:
2170 case LibFunc_ilogbf:
2171 case LibFunc_log:
2172 case LibFunc_logf:
2173 case LibFunc_log_finite:
2174 case LibFunc_logf_finite:
2175 case LibFunc_logb:
2176 case LibFunc_logbf:
2177 case LibFunc_logl:
2178 case LibFunc_log2:
2179 case LibFunc_log2f:
2180 case LibFunc_log2_finite:
2181 case LibFunc_log2f_finite:
2182 case LibFunc_log10:
2183 case LibFunc_log10f:
2184 case LibFunc_log10_finite:
2185 case LibFunc_log10f_finite:
2186 case LibFunc_log1p:
2187 case LibFunc_log1pf:
2188 case LibFunc_nearbyint:
2189 case LibFunc_nearbyintf:
2190 case LibFunc_nextafter:
2191 case LibFunc_nextafterf:
2192 case LibFunc_nexttoward:
2193 case LibFunc_nexttowardf:
2194 case LibFunc_pow:
2195 case LibFunc_powf:
2196 case LibFunc_pow_finite:
2197 case LibFunc_powf_finite:
2198 case LibFunc_remainder:
2199 case LibFunc_remainderf:
2200 case LibFunc_rint:
2201 case LibFunc_rintf:
2202 case LibFunc_round:
2203 case LibFunc_roundf:
2204 case LibFunc_roundeven:
2205 case LibFunc_roundevenf:
2206 case LibFunc_sin:
2207 case LibFunc_sinf:
2208 case LibFunc_sinh:
2209 case LibFunc_sinhf:
2210 case LibFunc_sinh_finite:
2211 case LibFunc_sinhf_finite:
2212 case LibFunc_sqrt:
2213 case LibFunc_sqrtf:
2214 case LibFunc_tan:
2215 case LibFunc_tanf:
2216 case LibFunc_tanh:
2217 case LibFunc_tanhf:
2218 case LibFunc_trunc:
2219 case LibFunc_truncf:
2220 return true;
2221 default:
2222 return false;
2223 }
2224}
2225
2226namespace {
2227
2228Constant *GetConstantFoldFPValue(double V, Type *Ty) {
2229 if (Ty->isHalfTy() || Ty->isFloatTy() || Ty->isBFloatTy()) {
2230 APFloat APF(V);
2231 bool unused;
2232 APF.convert(Ty->getFltSemantics(), APFloat::rmNearestTiesToEven, &unused);
2233 return ConstantFP::get(Ty->getContext(), APF);
2234 }
2235 if (Ty->isDoubleTy())
2236 return ConstantFP::get(Ty->getContext(), APFloat(V));
2237 llvm_unreachable("Can only constant fold half/float/double/bfloat");
2238}
2239
2240#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2241Constant *GetConstantFoldFPValue128(float128 V, Type *Ty) {
2242 if (Ty->isFP128Ty())
2243 return ConstantFP::get(Ty, V);
2244 llvm_unreachable("Can only constant fold fp128");
2245}
2246#endif
2247
2248/// Clear the floating-point exception state.
2249inline void llvm_fenv_clearexcept() {
2250#if defined(FE_ALL_EXCEPT)
2251 feclearexcept(FE_ALL_EXCEPT);
2252#endif
2253 errno = 0;
2254}
2255
2256/// Test if a floating-point exception was raised.
2257inline bool llvm_fenv_testexcept() {
2258 int errno_val = errno;
2259 if (errno_val == ERANGE || errno_val == EDOM)
2260 return true;
2261#if defined(FE_ALL_EXCEPT) && defined(FE_INEXACT)
2262 if (fetestexcept(FE_ALL_EXCEPT & ~FE_INEXACT))
2263 return true;
2264#endif
2265 return false;
2266}
2267
2268static APFloat FTZPreserveSign(const APFloat &V) {
2269 if (V.isDenormal())
2270 return APFloat::getZero(V.getSemantics(), V.isNegative());
2271 return V;
2272}
2273
2274static APFloat FlushToPositiveZero(const APFloat &V) {
2275 if (V.isDenormal())
2276 return APFloat::getZero(V.getSemantics(), false);
2277 return V;
2278}
2279
2280static APFloat FlushWithDenormKind(const APFloat &V,
2281 DenormalMode::DenormalModeKind DenormKind) {
2284 switch (DenormKind) {
2286 return V;
2288 return FTZPreserveSign(V);
2290 return FlushToPositiveZero(V);
2291 default:
2292 llvm_unreachable("Invalid denormal mode!");
2293 }
2294}
2295
2296Constant *ConstantFoldFP(double (*NativeFP)(double), const APFloat &V, Type *Ty,
2297 DenormalMode DenormMode = DenormalMode::getIEEE()) {
2298 if (!DenormMode.isValid() ||
2299 DenormMode.Input == DenormalMode::DenormalModeKind::Dynamic ||
2300 DenormMode.Output == DenormalMode::DenormalModeKind::Dynamic)
2301 return nullptr;
2302
2303 llvm_fenv_clearexcept();
2304 auto Input = FlushWithDenormKind(V, DenormMode.Input);
2305 double Result = NativeFP(Input.convertToDouble());
2306 if (llvm_fenv_testexcept()) {
2307 llvm_fenv_clearexcept();
2308 return nullptr;
2309 }
2310
2311 Constant *Output = GetConstantFoldFPValue(Result, Ty);
2312 if (DenormMode.Output == DenormalMode::DenormalModeKind::IEEE)
2313 return Output;
2314 const auto *CFP = static_cast<ConstantFP *>(Output);
2315 const auto Res = FlushWithDenormKind(CFP->getValueAPF(), DenormMode.Output);
2316 return ConstantFP::get(Ty->getContext(), Res);
2317}
2318
2319#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2320Constant *ConstantFoldFP128(float128 (*NativeFP)(float128), const APFloat &V,
2321 Type *Ty) {
2322 llvm_fenv_clearexcept();
2323 float128 Result = NativeFP(V.convertToQuad());
2324 if (llvm_fenv_testexcept()) {
2325 llvm_fenv_clearexcept();
2326 return nullptr;
2327 }
2328
2329 return GetConstantFoldFPValue128(Result, Ty);
2330}
2331#endif
2332
2333Constant *ConstantFoldBinaryFP(double (*NativeFP)(double, double),
2334 const APFloat &V, const APFloat &W, Type *Ty) {
2335 llvm_fenv_clearexcept();
2336 double Result = NativeFP(V.convertToDouble(), W.convertToDouble());
2337 if (llvm_fenv_testexcept()) {
2338 llvm_fenv_clearexcept();
2339 return nullptr;
2340 }
2341
2342 return GetConstantFoldFPValue(Result, Ty);
2343}
2344
2345Constant *constantFoldVectorReduce(Intrinsic::ID IID, Constant *Op) {
2346 auto *OpVT = cast<VectorType>(Op->getType());
2347
2348 // This is the same as the underlying binops - poison propagates.
2349 if (Op->containsPoisonElement())
2350 return PoisonValue::get(OpVT->getElementType());
2351
2352 // Shortcut non-accumulating reductions.
2353 if (Constant *SplatVal = Op->getSplatValue()) {
2354 switch (IID) {
2355 case Intrinsic::vector_reduce_and:
2356 case Intrinsic::vector_reduce_or:
2357 case Intrinsic::vector_reduce_smin:
2358 case Intrinsic::vector_reduce_smax:
2359 case Intrinsic::vector_reduce_umin:
2360 case Intrinsic::vector_reduce_umax:
2361 return SplatVal;
2362 case Intrinsic::vector_reduce_add:
2363 if (SplatVal->isNullValue())
2364 return SplatVal;
2365 break;
2366 case Intrinsic::vector_reduce_mul:
2367 if (SplatVal->isNullValue() || SplatVal->isOneValue())
2368 return SplatVal;
2369 break;
2370 case Intrinsic::vector_reduce_xor:
2371 if (SplatVal->isNullValue())
2372 return SplatVal;
2373 if (OpVT->getElementCount().isKnownMultipleOf(2))
2374 return Constant::getNullValue(OpVT->getElementType());
2375 break;
2376 }
2377 }
2378
2380 if (!VT)
2381 return nullptr;
2382
2383 auto *EltC = dyn_cast_or_null<ConstantInt>(Op->getAggregateElement(0U));
2384 if (!EltC)
2385 return nullptr;
2386
2387 APInt Acc = EltC->getValue();
2388 for (unsigned I = 1, E = VT->getNumElements(); I != E; I++) {
2389 if (!(EltC = dyn_cast_or_null<ConstantInt>(Op->getAggregateElement(I))))
2390 return nullptr;
2391 const APInt &X = EltC->getValue();
2392 switch (IID) {
2393 case Intrinsic::vector_reduce_add:
2394 Acc = Acc + X;
2395 break;
2396 case Intrinsic::vector_reduce_mul:
2397 Acc = Acc * X;
2398 break;
2399 case Intrinsic::vector_reduce_and:
2400 Acc = Acc & X;
2401 break;
2402 case Intrinsic::vector_reduce_or:
2403 Acc = Acc | X;
2404 break;
2405 case Intrinsic::vector_reduce_xor:
2406 Acc = Acc ^ X;
2407 break;
2408 case Intrinsic::vector_reduce_smin:
2409 Acc = APIntOps::smin(Acc, X);
2410 break;
2411 case Intrinsic::vector_reduce_smax:
2412 Acc = APIntOps::smax(Acc, X);
2413 break;
2414 case Intrinsic::vector_reduce_umin:
2415 Acc = APIntOps::umin(Acc, X);
2416 break;
2417 case Intrinsic::vector_reduce_umax:
2418 Acc = APIntOps::umax(Acc, X);
2419 break;
2420 }
2421 }
2422
2423 return ConstantInt::get(Op->getContext(), Acc);
2424}
2425
2426/// Fold a vector partial reduction add using the deterministic grouping
2427/// chosen by TargetLowering::expandPartialReduceMLA. Although the
2428/// LangRef leaves the grouping unspecified, input element I is accumulated
2429/// into result lane I % NumAccElts, with each accumulator element seeding
2430/// its corresponding result lane. Returns nullptr if any element cannot be
2431/// folded.
2432static Constant *constantFoldVectorPartialReduceAdd(Constant *Acc,
2433 Constant *Input,
2434 const DataLayout &DL) {
2435 auto *AccTy = cast<FixedVectorType>(Acc->getType());
2436 // A fixed result type does not guarantee a fixed input type.
2437 auto *InputTy = dyn_cast<FixedVectorType>(Input->getType());
2438 if (!InputTy)
2439 return nullptr;
2440
2441 unsigned NumAccElts = AccTy->getNumElements();
2442 unsigned NumInputElts = InputTy->getNumElements();
2443
2444 SmallVector<Constant *> ResultElts(NumAccElts);
2445 for (unsigned I = 0; I < NumAccElts; ++I) {
2446 ResultElts[I] = Acc->getAggregateElement(I);
2447 if (!ResultElts[I])
2448 return nullptr;
2449 }
2450
2451 for (unsigned I = 0; I < NumInputElts; ++I) {
2452 Constant *InputElt = Input->getAggregateElement(I);
2453 if (!InputElt)
2454 return nullptr;
2455
2456 unsigned ResultIdx = I % NumAccElts;
2458 Instruction::Add, ResultElts[ResultIdx], InputElt, DL);
2459 if (!Folded)
2460 return nullptr;
2461
2462 ResultElts[ResultIdx] = Folded;
2463 }
2464
2465 return ConstantVector::get(ResultElts);
2466}
2467
2468/// Attempt to fold an SSE floating point to integer conversion of a constant
2469/// floating point. If roundTowardZero is false, the default IEEE rounding is
2470/// used (toward nearest, ties to even). This matches the behavior of the
2471/// non-truncating SSE instructions in the default rounding mode. The desired
2472/// integer type Ty is used to select how many bits are available for the
2473/// result. Returns null if the conversion cannot be performed, otherwise
2474/// returns the Constant value resulting from the conversion.
2475Constant *ConstantFoldSSEConvertToInt(const APFloat &Val, bool roundTowardZero,
2476 Type *Ty, bool IsSigned) {
2477 // All of these conversion intrinsics form an integer of at most 64bits.
2478 unsigned ResultWidth = Ty->getIntegerBitWidth();
2479 assert(ResultWidth <= 64 &&
2480 "Can only constant fold conversions to 64 and 32 bit ints");
2481
2482 uint64_t UIntVal;
2483 bool isExact = false;
2487 Val.convertToInteger(MutableArrayRef(UIntVal), ResultWidth,
2488 IsSigned, mode, &isExact);
2489 if (status != APFloat::opOK &&
2490 (!roundTowardZero || status != APFloat::opInexact))
2491 return nullptr;
2492 return ConstantInt::get(Ty, UIntVal, IsSigned);
2493}
2494
2495double getValueAsDouble(ConstantFP *Op) {
2496 Type *Ty = Op->getType();
2497
2498 if (Ty->isBFloatTy() || Ty->isHalfTy() || Ty->isFloatTy() || Ty->isDoubleTy())
2499 return Op->getValueAPF().convertToDouble();
2500
2501 bool unused;
2502 APFloat APF = Op->getValueAPF();
2504 return APF.convertToDouble();
2505}
2506
2507static bool getConstIntOrUndef(Value *Op, const APInt *&C) {
2508 if (auto *CI = dyn_cast<ConstantInt>(Op)) {
2509 C = &CI->getValue();
2510 return true;
2511 }
2512 if (isa<UndefValue>(Op)) {
2513 C = nullptr;
2514 return true;
2515 }
2516 return false;
2517}
2518
2519/// Checks if the given intrinsic call, which evaluates to constant, is allowed
2520/// to be folded.
2521///
2522/// \param CI Constrained intrinsic call.
2523/// \param St Exception flags raised during constant evaluation.
2524static bool mayFoldConstrained(ConstrainedFPIntrinsic *CI,
2525 APFloat::opStatus St) {
2526 std::optional<RoundingMode> ORM = CI->getRoundingMode();
2527 std::optional<fp::ExceptionBehavior> EB = CI->getExceptionBehavior();
2528
2529 // If the operation does not change exception status flags, it is safe
2530 // to fold.
2531 if (St == APFloat::opStatus::opOK)
2532 return true;
2533
2534 // If evaluation raised FP exception, the result can depend on rounding
2535 // mode. If the latter is unknown, folding is not possible.
2536 if (ORM == RoundingMode::Dynamic)
2537 return false;
2538
2539 // If FP exceptions are ignored, fold the call, even if such exception is
2540 // raised.
2541 if (EB && *EB != fp::ExceptionBehavior::ebStrict)
2542 return true;
2543
2544 // Leave the calculation for runtime so that exception flags be correctly set
2545 // in hardware.
2546 return false;
2547}
2548
2549/// Returns the rounding mode that should be used for constant evaluation.
2550static RoundingMode
2551getEvaluationRoundingMode(const ConstrainedFPIntrinsic *CI) {
2552 std::optional<RoundingMode> ORM = CI->getRoundingMode();
2553 if (!ORM || *ORM == RoundingMode::Dynamic)
2554 // Even if the rounding mode is unknown, try evaluating the operation.
2555 // If it does not raise inexact exception, rounding was not applied,
2556 // so the result is exact and does not depend on rounding mode. Whether
2557 // other FP exceptions are raised, it does not depend on rounding mode.
2559 return *ORM;
2560}
2561
2562/// Try to constant fold llvm.canonicalize for the given caller and value.
2563static Constant *constantFoldCanonicalize(const Type *Ty, const APFloat &Src,
2564 const Function *CtxF = nullptr) {
2565 // Zero, positive and negative, is always OK to fold.
2566 if (Src.isZero()) {
2567 // Get a fresh 0, since ppc_fp128 does have non-canonical zeros.
2568 return ConstantFP::get(
2569 Ty->getContext(),
2570 APFloat::getZero(Src.getSemantics(), Src.isNegative()));
2571 }
2572
2573 if (!Ty->isIEEELikeFPTy())
2574 return nullptr;
2575
2576 // Zero is always canonical and the sign must be preserved.
2577 //
2578 // Denorms and nans may have special encodings, but it should be OK to fold a
2579 // totally average number.
2580 if (Src.isNormal() || Src.isInfinity())
2581 return ConstantFP::get(Ty->getContext(), Src);
2582
2583 if (Src.isDenormal() && CtxF) {
2584 DenormalMode DenormMode = CtxF->getDenormalMode(Src.getSemantics());
2585
2586 if (DenormMode == DenormalMode::getIEEE())
2587 return ConstantFP::get(Ty->getContext(), Src);
2588
2589 if (DenormMode.Input == DenormalMode::Dynamic)
2590 return nullptr;
2591
2592 // If we know if either input or output is flushed, we can fold.
2593 if ((DenormMode.Input == DenormalMode::Dynamic &&
2594 DenormMode.Output == DenormalMode::IEEE) ||
2595 (DenormMode.Input == DenormalMode::IEEE &&
2596 DenormMode.Output == DenormalMode::Dynamic))
2597 return nullptr;
2598
2599 bool IsPositive =
2600 (!Src.isNegative() || DenormMode.Input == DenormalMode::PositiveZero ||
2601 (DenormMode.Output == DenormalMode::PositiveZero &&
2602 DenormMode.Input == DenormalMode::IEEE));
2603
2604 return ConstantFP::get(Ty->getContext(),
2605 APFloat::getZero(Src.getSemantics(), !IsPositive));
2606 }
2607
2608 return nullptr;
2609}
2610
2611static Constant *ConstantFoldScalarCall1(StringRef Name,
2612 Intrinsic::ID IntrinsicID, Type *Ty,
2614 const TargetLibraryInfo *TLI = nullptr,
2615 const CallBase *Call = nullptr) {
2616 assert(Operands.size() == 1 && "Wrong number of operands.");
2617
2618 if (IntrinsicID == Intrinsic::is_constant) {
2619 // We know we have a "Constant" argument. But we want to only
2620 // return true for manifest constants, not those that depend on
2621 // constants with unknowable values, e.g. GlobalValue or BlockAddress.
2622 if (Operands[0]->isManifestConstant())
2623 return ConstantInt::getTrue(Ty->getContext());
2624 return nullptr;
2625 }
2626
2627 if (isa<UndefValue>(Operands[0])) {
2628 // cosine(arg) is between -1 and 1. cosine(invalid arg) is NaN.
2629 // ctpop() is between 0 and bitwidth, pick 0 for undef.
2630 // fptoui.sat and fptosi.sat can always fold to zero (for a zero input).
2631 if (IntrinsicID == Intrinsic::cos ||
2632 IntrinsicID == Intrinsic::ctpop ||
2633 IntrinsicID == Intrinsic::fptoui_sat ||
2634 IntrinsicID == Intrinsic::fptosi_sat ||
2635 IntrinsicID == Intrinsic::canonicalize)
2636 return Constant::getNullValue(Ty);
2637 if (IntrinsicID == Intrinsic::bswap ||
2638 IntrinsicID == Intrinsic::bitreverse ||
2639 IntrinsicID == Intrinsic::launder_invariant_group)
2640 return Operands[0];
2641 }
2642
2644 // launder(null) == null iff in addrspace 0
2645 if (IntrinsicID == Intrinsic::launder_invariant_group) {
2646 // If instruction is not yet put in a basic block (e.g. when cloning
2647 // a function during inlining), Call's caller may not be available.
2648 // So check Call's BB first before querying Call->getCaller.
2649 const Function *Caller =
2650 Call && Call->getParent() ? Call->getCaller() : nullptr;
2651 if (Caller &&
2653 Caller, Operands[0]->getType()->getPointerAddressSpace())) {
2654 return Operands[0];
2655 }
2656 return nullptr;
2657 }
2658 }
2659
2660 if (auto *Op = dyn_cast<ConstantFP>(Operands[0])) {
2661 APFloat U = Op->getValueAPF();
2662
2663 if (IntrinsicID == Intrinsic::wasm_trunc_signed ||
2664 IntrinsicID == Intrinsic::wasm_trunc_unsigned) {
2665 bool Signed = IntrinsicID == Intrinsic::wasm_trunc_signed;
2666
2667 if (U.isNaN())
2668 return nullptr;
2669
2670 unsigned Width = Ty->getIntegerBitWidth();
2671 APSInt Int(Width, !Signed);
2672 bool IsExact = false;
2674 U.convertToInteger(Int, APFloat::rmTowardZero, &IsExact);
2675
2677 return ConstantInt::get(Ty, Int);
2678
2679 return nullptr;
2680 }
2681
2682 if (IntrinsicID == Intrinsic::fptoui_sat ||
2683 IntrinsicID == Intrinsic::fptosi_sat) {
2684 // convertToInteger() already has the desired saturation semantics.
2685 APSInt Int(Ty->getIntegerBitWidth(),
2686 IntrinsicID == Intrinsic::fptoui_sat);
2687 bool IsExact;
2688 U.convertToInteger(Int, APFloat::rmTowardZero, &IsExact);
2689 return ConstantInt::get(Ty, Int);
2690 }
2691
2692 if (IntrinsicID == Intrinsic::canonicalize) {
2693 const Function *CtxF =
2694 Call && Call->getParent() ? Call->getFunction() : nullptr;
2695 return constantFoldCanonicalize(Ty, U, CtxF);
2696 }
2697
2698#if defined(HAS_IEE754_FLOAT128) && defined(HAS_LOGF128)
2699 if (Ty->isFP128Ty()) {
2700 if (IntrinsicID == Intrinsic::log) {
2701 float128 Result = logf128(Op->getValueAPF().convertToQuad());
2702 return GetConstantFoldFPValue128(Result, Ty);
2703 }
2704
2705 if (TLI && TLI->getLibFunc(Name) == LibFunc_logl &&
2706 TLI->has(LibFunc_logl))
2707 return ConstantFoldFP128(logf128, Op->getValueAPF(), Ty);
2708 }
2709#endif
2710
2711 if (!Ty->isHalfTy() && !Ty->isFloatTy() && !Ty->isDoubleTy() &&
2712 !Ty->isIntegerTy() && !Ty->isBFloatTy())
2713 return nullptr;
2714
2715 // Use internal versions of these intrinsics.
2716
2717 if (IntrinsicID == Intrinsic::nearbyint || IntrinsicID == Intrinsic::rint ||
2718 IntrinsicID == Intrinsic::roundeven) {
2719 U.roundToIntegral(APFloat::rmNearestTiesToEven);
2720 return ConstantFP::get(Ty, U);
2721 }
2722
2723 if (IntrinsicID == Intrinsic::round) {
2724 U.roundToIntegral(APFloat::rmNearestTiesToAway);
2725 return ConstantFP::get(Ty, U);
2726 }
2727
2728 if (IntrinsicID == Intrinsic::roundeven) {
2729 U.roundToIntegral(APFloat::rmNearestTiesToEven);
2730 return ConstantFP::get(Ty, U);
2731 }
2732
2733 if (IntrinsicID == Intrinsic::ceil) {
2734 U.roundToIntegral(APFloat::rmTowardPositive);
2735 return ConstantFP::get(Ty, U);
2736 }
2737
2738 if (IntrinsicID == Intrinsic::floor) {
2739 U.roundToIntegral(APFloat::rmTowardNegative);
2740 return ConstantFP::get(Ty, U);
2741 }
2742
2743 if (IntrinsicID == Intrinsic::trunc) {
2744 U.roundToIntegral(APFloat::rmTowardZero);
2745 return ConstantFP::get(Ty, U);
2746 }
2747
2748 if (IntrinsicID == Intrinsic::fabs) {
2749 U.clearSign();
2750 return ConstantFP::get(Ty, U);
2751 }
2752
2753 if (IntrinsicID == Intrinsic::amdgcn_fract) {
2754 // The v_fract instruction behaves like the OpenCL spec, which defines
2755 // fract(x) as fmin(x - floor(x), 0x1.fffffep-1f): "The min() operator is
2756 // there to prevent fract(-small) from returning 1.0. It returns the
2757 // largest positive floating-point number less than 1.0."
2758 APFloat FloorU(U);
2759 FloorU.roundToIntegral(APFloat::rmTowardNegative);
2760 APFloat FractU(U - FloorU);
2761 APFloat AlmostOne(U.getSemantics(), 1);
2762 AlmostOne.next(/*nextDown*/ true);
2763 return ConstantFP::get(Ty, minimum(FractU, AlmostOne));
2764 }
2765
2766 // Rounding operations (floor, trunc, ceil, round and nearbyint) do not
2767 // raise FP exceptions, unless the argument is signaling NaN.
2768
2770 std::optional<APFloat::roundingMode> RM;
2771 switch (IntrinsicID) {
2772 default:
2773 break;
2774 case Intrinsic::experimental_constrained_nearbyint:
2775 case Intrinsic::experimental_constrained_rint: {
2776 RM = CI->getRoundingMode();
2777 if (!RM || *RM == RoundingMode::Dynamic)
2778 return nullptr;
2779 break;
2780 }
2781 case Intrinsic::experimental_constrained_round:
2783 break;
2784 case Intrinsic::experimental_constrained_ceil:
2786 break;
2787 case Intrinsic::experimental_constrained_floor:
2789 break;
2790 case Intrinsic::experimental_constrained_trunc:
2792 break;
2793 }
2794 if (RM) {
2795 if (U.isFinite()) {
2796 APFloat::opStatus St = U.roundToIntegral(*RM);
2797 if (IntrinsicID == Intrinsic::experimental_constrained_rint &&
2798 St == APFloat::opInexact) {
2799 std::optional<fp::ExceptionBehavior> EB =
2801 if (EB == fp::ebStrict)
2802 return nullptr;
2803 }
2804 } else if (U.isSignaling()) {
2805 std::optional<fp::ExceptionBehavior> EB = CI->getExceptionBehavior();
2806 if (EB && *EB != fp::ebIgnore)
2807 return nullptr;
2808 U = APFloat::getQNaN(U.getSemantics());
2809 }
2810 return ConstantFP::get(Ty, U);
2811 }
2812 }
2813
2814 // NVVM float/double to signed/unsigned int32/int64 conversions:
2815 switch (IntrinsicID) {
2816 // f2i
2817 case Intrinsic::nvvm_f2i_rm:
2818 case Intrinsic::nvvm_f2i_rn:
2819 case Intrinsic::nvvm_f2i_rp:
2820 case Intrinsic::nvvm_f2i_rz:
2821 case Intrinsic::nvvm_f2i_rm_ftz:
2822 case Intrinsic::nvvm_f2i_rn_ftz:
2823 case Intrinsic::nvvm_f2i_rp_ftz:
2824 case Intrinsic::nvvm_f2i_rz_ftz:
2825 // f2ui
2826 case Intrinsic::nvvm_f2ui_rm:
2827 case Intrinsic::nvvm_f2ui_rn:
2828 case Intrinsic::nvvm_f2ui_rp:
2829 case Intrinsic::nvvm_f2ui_rz:
2830 case Intrinsic::nvvm_f2ui_rm_ftz:
2831 case Intrinsic::nvvm_f2ui_rn_ftz:
2832 case Intrinsic::nvvm_f2ui_rp_ftz:
2833 case Intrinsic::nvvm_f2ui_rz_ftz:
2834 // d2i
2835 case Intrinsic::nvvm_d2i_rm:
2836 case Intrinsic::nvvm_d2i_rn:
2837 case Intrinsic::nvvm_d2i_rp:
2838 case Intrinsic::nvvm_d2i_rz:
2839 // d2ui
2840 case Intrinsic::nvvm_d2ui_rm:
2841 case Intrinsic::nvvm_d2ui_rn:
2842 case Intrinsic::nvvm_d2ui_rp:
2843 case Intrinsic::nvvm_d2ui_rz:
2844 // f2ll
2845 case Intrinsic::nvvm_f2ll_rm:
2846 case Intrinsic::nvvm_f2ll_rn:
2847 case Intrinsic::nvvm_f2ll_rp:
2848 case Intrinsic::nvvm_f2ll_rz:
2849 case Intrinsic::nvvm_f2ll_rm_ftz:
2850 case Intrinsic::nvvm_f2ll_rn_ftz:
2851 case Intrinsic::nvvm_f2ll_rp_ftz:
2852 case Intrinsic::nvvm_f2ll_rz_ftz:
2853 // f2ull
2854 case Intrinsic::nvvm_f2ull_rm:
2855 case Intrinsic::nvvm_f2ull_rn:
2856 case Intrinsic::nvvm_f2ull_rp:
2857 case Intrinsic::nvvm_f2ull_rz:
2858 case Intrinsic::nvvm_f2ull_rm_ftz:
2859 case Intrinsic::nvvm_f2ull_rn_ftz:
2860 case Intrinsic::nvvm_f2ull_rp_ftz:
2861 case Intrinsic::nvvm_f2ull_rz_ftz:
2862 // d2ll
2863 case Intrinsic::nvvm_d2ll_rm:
2864 case Intrinsic::nvvm_d2ll_rn:
2865 case Intrinsic::nvvm_d2ll_rp:
2866 case Intrinsic::nvvm_d2ll_rz:
2867 // d2ull
2868 case Intrinsic::nvvm_d2ull_rm:
2869 case Intrinsic::nvvm_d2ull_rn:
2870 case Intrinsic::nvvm_d2ull_rp:
2871 case Intrinsic::nvvm_d2ull_rz: {
2872 // In float-to-integer conversion, NaN inputs are converted to 0.
2873 if (U.isNaN()) {
2874 // In float-to-integer conversion, NaN inputs are converted to 0
2875 // when the source and destination bitwidths are both less than 64.
2876 if (nvvm::FPToIntegerIntrinsicNaNZero(IntrinsicID))
2877 return ConstantInt::get(Ty, 0);
2878
2879 // Otherwise, the most significant bit is set.
2880 unsigned BitWidth = Ty->getIntegerBitWidth();
2881 uint64_t Val = 1ULL << (BitWidth - 1);
2882 return ConstantInt::get(Ty, APInt(BitWidth, Val, /*IsSigned=*/false));
2883 }
2884
2885 APFloat::roundingMode RMode =
2887 bool IsFTZ = nvvm::FPToIntegerIntrinsicShouldFTZ(IntrinsicID);
2888 bool IsSigned = nvvm::FPToIntegerIntrinsicResultIsSigned(IntrinsicID);
2889
2890 APSInt ResInt(Ty->getIntegerBitWidth(), !IsSigned);
2891 auto FloatToRound = IsFTZ ? FTZPreserveSign(U) : U;
2892
2893 // Return max/min value for integers if the result is +/-inf or
2894 // is too large to fit in the result's integer bitwidth.
2895 bool IsExact = false;
2896 FloatToRound.convertToInteger(ResInt, RMode, &IsExact);
2897 return ConstantInt::get(Ty, ResInt);
2898 }
2899 }
2900
2901 /// We only fold functions with finite arguments. Folding NaN and inf is
2902 /// likely to be aborted with an exception anyway, and some host libms
2903 /// have known errors raising exceptions.
2904 if (!U.isFinite())
2905 return nullptr;
2906
2907 /// Currently APFloat versions of these functions do not exist, so we use
2908 /// the host native double versions. Float versions are not called
2909 /// directly but for all these it is true (float)(f((double)arg)) ==
2910 /// f(arg). Long double not supported yet.
2911 const APFloat &APF = Op->getValueAPF();
2912
2913 switch (IntrinsicID) {
2914 default: break;
2915 case Intrinsic::log:
2916 if (U.isZero())
2917 return ConstantFP::getInfinity(Ty, true);
2918 if (U.isNegative())
2919 return ConstantFP::getNaN(Ty);
2920 if (U.isOne())
2921 return ConstantFP::getZero(Ty);
2922 return ConstantFoldFP(log, APF, Ty);
2923 case Intrinsic::log2:
2924 if (U.isZero())
2925 return ConstantFP::getInfinity(Ty, true);
2926 if (U.isNegative())
2927 return ConstantFP::getNaN(Ty);
2928 if (U.isOne())
2929 return ConstantFP::getZero(Ty);
2930 // TODO: What about hosts that lack a C99 library?
2931 return ConstantFoldFP(log2, APF, Ty);
2932 case Intrinsic::log10:
2933 if (U.isZero())
2934 return ConstantFP::getInfinity(Ty, true);
2935 if (U.isNegative())
2936 return ConstantFP::getNaN(Ty);
2937 if (U.isOne())
2938 return ConstantFP::getZero(Ty);
2939 // TODO: What about hosts that lack a C99 library?
2940 return ConstantFoldFP(log10, APF, Ty);
2941 case Intrinsic::exp:
2942 return ConstantFoldFP(exp, APF, Ty);
2943 case Intrinsic::exp2:
2944 // Fold exp2(x) as pow(2, x), in case the host lacks a C99 library.
2945 return ConstantFoldBinaryFP(pow, APFloat(2.0), APF, Ty);
2946 case Intrinsic::exp10:
2947 // Fold exp10(x) as pow(10, x), in case the host lacks a C99 library.
2948 return ConstantFoldBinaryFP(pow, APFloat(10.0), APF, Ty);
2949 case Intrinsic::sin:
2950 return ConstantFoldFP(sin, APF, Ty);
2951 case Intrinsic::cos:
2952 return ConstantFoldFP(cos, APF, Ty);
2953 case Intrinsic::sinh:
2954 return ConstantFoldFP(sinh, APF, Ty);
2955 case Intrinsic::cosh:
2956 return ConstantFoldFP(cosh, APF, Ty);
2957 case Intrinsic::atan:
2958 // Implement optional behavior from C's Annex F for +/-0.0.
2959 if (U.isZero())
2960 return ConstantFP::get(Ty, U);
2961 return ConstantFoldFP(atan, APF, Ty);
2962 case Intrinsic::sqrt:
2963 return ConstantFoldFP(sqrt, APF, Ty);
2964
2965 // NVVM Intrinsics:
2966 case Intrinsic::nvvm_ceil_ftz_f:
2967 case Intrinsic::nvvm_ceil_f:
2968 case Intrinsic::nvvm_ceil_d:
2969 return ConstantFoldFP(
2970 ceil, APF, Ty,
2972 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2973
2974 case Intrinsic::nvvm_fabs_ftz:
2975 case Intrinsic::nvvm_fabs:
2976 return ConstantFoldFP(
2977 fabs, APF, Ty,
2979 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2980
2981 case Intrinsic::nvvm_floor_ftz_f:
2982 case Intrinsic::nvvm_floor_f:
2983 case Intrinsic::nvvm_floor_d:
2984 return ConstantFoldFP(
2985 floor, APF, Ty,
2987 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
2988
2989 case Intrinsic::nvvm_rcp_rm_ftz_f:
2990 case Intrinsic::nvvm_rcp_rn_ftz_f:
2991 case Intrinsic::nvvm_rcp_rp_ftz_f:
2992 case Intrinsic::nvvm_rcp_rz_ftz_f:
2993 case Intrinsic::nvvm_rcp_rm_d:
2994 case Intrinsic::nvvm_rcp_rm_f:
2995 case Intrinsic::nvvm_rcp_rn_d:
2996 case Intrinsic::nvvm_rcp_rn_f:
2997 case Intrinsic::nvvm_rcp_rp_d:
2998 case Intrinsic::nvvm_rcp_rp_f:
2999 case Intrinsic::nvvm_rcp_rz_d:
3000 case Intrinsic::nvvm_rcp_rz_f: {
3001 APFloat::roundingMode RoundMode = nvvm::GetRCPRoundingMode(IntrinsicID);
3002 bool IsFTZ = nvvm::RCPShouldFTZ(IntrinsicID);
3003
3004 auto Denominator = IsFTZ ? FTZPreserveSign(APF) : APF;
3006 APFloat::opStatus Status = Res.divide(Denominator, RoundMode);
3007
3009 if (IsFTZ)
3010 Res = FTZPreserveSign(Res);
3011 return ConstantFP::get(Ty, Res);
3012 }
3013 return nullptr;
3014 }
3015
3016 case Intrinsic::nvvm_round_ftz_f:
3017 case Intrinsic::nvvm_round_f:
3018 case Intrinsic::nvvm_round_d: {
3019 // nvvm_round is lowered to PTX cvt.rni, which will round to nearest
3020 // integer, choosing even integer if source is equidistant between two
3021 // integers, so the semantics are closer to "rint" rather than "round".
3022 bool IsFTZ = nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID);
3023 auto V = IsFTZ ? FTZPreserveSign(APF) : APF;
3025 return ConstantFP::get(Ty, V);
3026 }
3027
3028 case Intrinsic::nvvm_saturate_ftz_f:
3029 case Intrinsic::nvvm_saturate_d:
3030 case Intrinsic::nvvm_saturate_f: {
3031 bool IsFTZ = nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID);
3032 auto V = IsFTZ ? FTZPreserveSign(APF) : APF;
3033 if (V.isNegative() || V.isZero() || V.isNaN())
3034 return ConstantFP::getZero(Ty);
3036 if (V > One)
3037 return ConstantFP::get(Ty, One);
3038 return ConstantFP::get(Ty, APF);
3039 }
3040
3041 case Intrinsic::nvvm_sqrt_rn_ftz_f:
3042 case Intrinsic::nvvm_sqrt_f:
3043 case Intrinsic::nvvm_sqrt_rn_d:
3044 case Intrinsic::nvvm_sqrt_rn_f:
3045 if (APF.isNegative())
3046 return nullptr;
3047 return ConstantFoldFP(
3048 sqrt, APF, Ty,
3050 nvvm::UnaryMathIntrinsicShouldFTZ(IntrinsicID)));
3051
3052 // AMDGCN Intrinsics:
3053 case Intrinsic::amdgcn_cos:
3054 case Intrinsic::amdgcn_sin: {
3055 double V = getValueAsDouble(Op);
3056 if (V < -256.0 || V > 256.0)
3057 // The gfx8 and gfx9 architectures handle arguments outside the range
3058 // [-256, 256] differently. This should be a rare case so bail out
3059 // rather than trying to handle the difference.
3060 return nullptr;
3061 bool IsCos = IntrinsicID == Intrinsic::amdgcn_cos;
3062 double V4 = V * 4.0;
3063 if (V4 == floor(V4)) {
3064 // Force exact results for quarter-integer inputs.
3065 const double SinVals[4] = { 0.0, 1.0, 0.0, -1.0 };
3066 V = SinVals[((int)V4 + (IsCos ? 1 : 0)) & 3];
3067 } else {
3068 if (IsCos)
3069 V = cos(V * 2.0 * numbers::pi);
3070 else
3071 V = sin(V * 2.0 * numbers::pi);
3072 }
3073 return GetConstantFoldFPValue(V, Ty);
3074 }
3075 }
3076
3077 if (!TLI)
3078 return nullptr;
3079
3080 LibFunc Func = TLI->getLibFunc(Name);
3081 if (Func == NotLibFunc)
3082 return nullptr;
3083
3084 switch (Func) {
3085 default:
3086 break;
3087 case LibFunc_acos:
3088 case LibFunc_acosf:
3089 case LibFunc_acos_finite:
3090 case LibFunc_acosf_finite:
3091 if (TLI->has(Func))
3092 return ConstantFoldFP(acos, APF, Ty);
3093 break;
3094 case LibFunc_asin:
3095 case LibFunc_asinf:
3096 case LibFunc_asin_finite:
3097 case LibFunc_asinf_finite:
3098 if (TLI->has(Func))
3099 return ConstantFoldFP(asin, APF, Ty);
3100 break;
3101 case LibFunc_atan:
3102 case LibFunc_atanf:
3103 // Implement optional behavior from C's Annex F for +/-0.0.
3104 if (U.isZero())
3105 return ConstantFP::get(Ty, U);
3106 if (TLI->has(Func))
3107 return ConstantFoldFP(atan, APF, Ty);
3108 break;
3109 case LibFunc_ceil:
3110 case LibFunc_ceilf:
3111 if (TLI->has(Func)) {
3112 U.roundToIntegral(APFloat::rmTowardPositive);
3113 return ConstantFP::get(Ty, U);
3114 }
3115 break;
3116 case LibFunc_cos:
3117 case LibFunc_cosf:
3118 if (TLI->has(Func))
3119 return ConstantFoldFP(cos, APF, Ty);
3120 break;
3121 case LibFunc_cosh:
3122 case LibFunc_coshf:
3123 case LibFunc_cosh_finite:
3124 case LibFunc_coshf_finite:
3125 if (TLI->has(Func))
3126 return ConstantFoldFP(cosh, APF, Ty);
3127 break;
3128 case LibFunc_exp:
3129 case LibFunc_expf:
3130 case LibFunc_exp_finite:
3131 case LibFunc_expf_finite:
3132 if (TLI->has(Func))
3133 return ConstantFoldFP(exp, APF, Ty);
3134 break;
3135 case LibFunc_exp2:
3136 case LibFunc_exp2f:
3137 case LibFunc_exp2_finite:
3138 case LibFunc_exp2f_finite:
3139 if (TLI->has(Func))
3140 // Fold exp2(x) as pow(2, x), in case the host lacks a C99 library.
3141 return ConstantFoldBinaryFP(pow, APFloat(2.0), APF, Ty);
3142 break;
3143 case LibFunc_fabs:
3144 case LibFunc_fabsf:
3145 if (TLI->has(Func)) {
3146 U.clearSign();
3147 return ConstantFP::get(Ty, U);
3148 }
3149 break;
3150 case LibFunc_floor:
3151 case LibFunc_floorf:
3152 if (TLI->has(Func)) {
3153 U.roundToIntegral(APFloat::rmTowardNegative);
3154 return ConstantFP::get(Ty, U);
3155 }
3156 break;
3157 case LibFunc_log:
3158 case LibFunc_logf:
3159 case LibFunc_log_finite:
3160 case LibFunc_logf_finite:
3161 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3162 return ConstantFoldFP(log, APF, Ty);
3163 break;
3164 case LibFunc_log2:
3165 case LibFunc_log2f:
3166 case LibFunc_log2_finite:
3167 case LibFunc_log2f_finite:
3168 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3169 // TODO: What about hosts that lack a C99 library?
3170 return ConstantFoldFP(log2, APF, Ty);
3171 break;
3172 case LibFunc_log10:
3173 case LibFunc_log10f:
3174 case LibFunc_log10_finite:
3175 case LibFunc_log10f_finite:
3176 if (!APF.isNegative() && !APF.isZero() && TLI->has(Func))
3177 // TODO: What about hosts that lack a C99 library?
3178 return ConstantFoldFP(log10, APF, Ty);
3179 break;
3180 case LibFunc_ilogb:
3181 case LibFunc_ilogbf:
3182 if (!APF.isZero() && TLI->has(Func))
3183 return ConstantInt::get(Ty, ilogb(APF), true);
3184 break;
3185 case LibFunc_logb:
3186 case LibFunc_logbf:
3187 if (!APF.isZero() && TLI->has(Func))
3188 return ConstantFoldFP(logb, APF, Ty);
3189 break;
3190 case LibFunc_log1p:
3191 case LibFunc_log1pf:
3192 // Implement optional behavior from C's Annex F for +/-0.0.
3193 if (U.isZero())
3194 return ConstantFP::get(Ty, U);
3195 if (APF > APFloat::getOne(APF.getSemantics(), true) && TLI->has(Func))
3196 return ConstantFoldFP(log1p, APF, Ty);
3197 break;
3198 case LibFunc_logl:
3199 return nullptr;
3200 case LibFunc_erf:
3201 case LibFunc_erff:
3202 if (TLI->has(Func))
3203 return ConstantFoldFP(erf, APF, Ty);
3204 break;
3205 case LibFunc_nearbyint:
3206 case LibFunc_nearbyintf:
3207 case LibFunc_rint:
3208 case LibFunc_rintf:
3209 case LibFunc_roundeven:
3210 case LibFunc_roundevenf:
3211 if (TLI->has(Func)) {
3212 U.roundToIntegral(APFloat::rmNearestTiesToEven);
3213 return ConstantFP::get(Ty, U);
3214 }
3215 break;
3216 case LibFunc_round:
3217 case LibFunc_roundf:
3218 if (TLI->has(Func)) {
3219 U.roundToIntegral(APFloat::rmNearestTiesToAway);
3220 return ConstantFP::get(Ty, U);
3221 }
3222 break;
3223 case LibFunc_sin:
3224 case LibFunc_sinf:
3225 if (TLI->has(Func))
3226 return ConstantFoldFP(sin, APF, Ty);
3227 break;
3228 case LibFunc_sinh:
3229 case LibFunc_sinhf:
3230 case LibFunc_sinh_finite:
3231 case LibFunc_sinhf_finite:
3232 if (TLI->has(Func))
3233 return ConstantFoldFP(sinh, APF, Ty);
3234 break;
3235 case LibFunc_sqrt:
3236 case LibFunc_sqrtf:
3237 if (!APF.isNegative() && TLI->has(Func))
3238 return ConstantFoldFP(sqrt, APF, Ty);
3239 break;
3240 case LibFunc_tan:
3241 case LibFunc_tanf:
3242 if (TLI->has(Func))
3243 return ConstantFoldFP(tan, APF, Ty);
3244 break;
3245 case LibFunc_tanh:
3246 case LibFunc_tanhf:
3247 if (TLI->has(Func))
3248 return ConstantFoldFP(tanh, APF, Ty);
3249 break;
3250 case LibFunc_trunc:
3251 case LibFunc_truncf:
3252 if (TLI->has(Func)) {
3253 U.roundToIntegral(APFloat::rmTowardZero);
3254 return ConstantFP::get(Ty, U);
3255 }
3256 break;
3257 }
3258 return nullptr;
3259 }
3260
3261 if (auto *Op = dyn_cast<ConstantInt>(Operands[0])) {
3262 switch (IntrinsicID) {
3263 case Intrinsic::bswap:
3264 return ConstantInt::get(Ty->getContext(), Op->getValue().byteSwap());
3265 case Intrinsic::ctpop:
3266 return ConstantInt::get(Ty, Op->getValue().popcount());
3267 case Intrinsic::bitreverse:
3268 return ConstantInt::get(Ty->getContext(), Op->getValue().reverseBits());
3269 case Intrinsic::amdgcn_s_wqm: {
3270 uint64_t Val = Op->getZExtValue();
3271 Val |= (Val & 0x5555555555555555ULL) << 1 |
3272 ((Val >> 1) & 0x5555555555555555ULL);
3273 Val |= (Val & 0x3333333333333333ULL) << 2 |
3274 ((Val >> 2) & 0x3333333333333333ULL);
3275 return ConstantInt::get(Ty, Val);
3276 }
3277
3278 case Intrinsic::amdgcn_s_quadmask: {
3279 uint64_t Val = Op->getZExtValue();
3280 uint64_t QuadMask = 0;
3281 for (unsigned I = 0; I < Op->getBitWidth() / 4; ++I, Val >>= 4) {
3282 if (!(Val & 0xF))
3283 continue;
3284
3285 QuadMask |= (1ULL << I);
3286 }
3287 return ConstantInt::get(Ty, QuadMask);
3288 }
3289
3290 case Intrinsic::amdgcn_s_bitreplicate: {
3291 uint64_t Val = Op->getZExtValue();
3292 Val = (Val & 0x000000000000FFFFULL) | (Val & 0x00000000FFFF0000ULL) << 16;
3293 Val = (Val & 0x000000FF000000FFULL) | (Val & 0x0000FF000000FF00ULL) << 8;
3294 Val = (Val & 0x000F000F000F000FULL) | (Val & 0x00F000F000F000F0ULL) << 4;
3295 Val = (Val & 0x0303030303030303ULL) | (Val & 0x0C0C0C0C0C0C0C0CULL) << 2;
3296 Val = (Val & 0x1111111111111111ULL) | (Val & 0x2222222222222222ULL) << 1;
3297 Val = Val | Val << 1;
3298 return ConstantInt::get(Ty, Val);
3299 }
3300 }
3301 }
3302
3303 if (Operands[0]->getType()->isVectorTy()) {
3304 auto *Op = cast<Constant>(Operands[0]);
3305 switch (IntrinsicID) {
3306 default: break;
3307 case Intrinsic::vector_reduce_add:
3308 case Intrinsic::vector_reduce_mul:
3309 case Intrinsic::vector_reduce_and:
3310 case Intrinsic::vector_reduce_or:
3311 case Intrinsic::vector_reduce_xor:
3312 case Intrinsic::vector_reduce_smin:
3313 case Intrinsic::vector_reduce_smax:
3314 case Intrinsic::vector_reduce_umin:
3315 case Intrinsic::vector_reduce_umax:
3316 if (Constant *C = constantFoldVectorReduce(IntrinsicID, Operands[0]))
3317 return C;
3318 break;
3319 case Intrinsic::x86_sse_cvtss2si:
3320 case Intrinsic::x86_sse_cvtss2si64:
3321 case Intrinsic::x86_sse2_cvtsd2si:
3322 case Intrinsic::x86_sse2_cvtsd2si64:
3323 if (ConstantFP *FPOp =
3324 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3325 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3326 /*roundTowardZero=*/false, Ty,
3327 /*IsSigned*/true);
3328 break;
3329 case Intrinsic::x86_sse_cvttss2si:
3330 case Intrinsic::x86_sse_cvttss2si64:
3331 case Intrinsic::x86_sse2_cvttsd2si:
3332 case Intrinsic::x86_sse2_cvttsd2si64:
3333 if (ConstantFP *FPOp =
3334 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3335 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3336 /*roundTowardZero=*/true, Ty,
3337 /*IsSigned*/true);
3338 break;
3339
3340 case Intrinsic::wasm_anytrue:
3341 return Op->isNullValue() ? ConstantInt::get(Ty, 0)
3342 : ConstantInt::get(Ty, 1);
3343
3344 case Intrinsic::wasm_alltrue:
3345 // Check each element individually
3346 unsigned E = cast<FixedVectorType>(Op->getType())->getNumElements();
3347 for (unsigned I = 0; I != E; ++I) {
3348 Constant *Elt = Op->getAggregateElement(I);
3349 // Return false as soon as we find a non-true element.
3350 if (Elt && Elt->isNullValue())
3351 return ConstantInt::get(Ty, 0);
3352 // Bail as soon as we find an element we cannot prove to be true.
3353 if (!Elt || !isa<ConstantInt>(Elt))
3354 return nullptr;
3355 }
3356
3357 return ConstantInt::get(Ty, 1);
3358 }
3359 }
3360
3361 return nullptr;
3362}
3363
3364static Constant *evaluateCompare(const APFloat &Op1, const APFloat &Op2,
3368 FCmpInst::Predicate Cond = FCmp->getPredicate();
3369 if (FCmp->isSignaling()) {
3370 if (Op1.isNaN() || Op2.isNaN())
3372 } else {
3373 if (Op1.isSignaling() || Op2.isSignaling())
3375 }
3376 bool Result = FCmpInst::compare(Op1, Op2, Cond);
3377 if (mayFoldConstrained(const_cast<ConstrainedFPCmpIntrinsic *>(FCmp), St))
3378 return ConstantInt::get(Call->getType()->getScalarType(), Result);
3379 return nullptr;
3380}
3381
3382static Constant *ConstantFoldNextToward(const APFloat &Op0, const APFloat &Op1,
3383 const Type *RetTy) {
3384 assert(RetTy != nullptr);
3385 bool LosesInfo;
3386
3387 if (Op1.isSignaling())
3388 return nullptr;
3389 if (Op1.isNaN()) {
3390 APFloat Ret(Op1);
3391 Ret.convert(RetTy->getFltSemantics(), detail::rmNearestTiesToEven,
3392 &LosesInfo);
3393 return ConstantFP::get(RetTy->getContext(), Ret);
3394 }
3395
3396 // Recall that the second argument of nexttoward is always a long double,
3397 // so we may need to promote the first argument for comparisons to be valid.
3398 APFloat PromotedOp0(Op0);
3399 PromotedOp0.convert(Op1.getSemantics(), detail::rmNearestTiesToEven,
3400 &LosesInfo);
3401 assert(!LosesInfo && "Unexpected lossy promotion");
3402 const APFloat::cmpResult Result = PromotedOp0.compare(Op1);
3403
3404 // When equal, the standard says we must return the second argument.
3405 // This allows nice behavior such as nexttoward(0.0, -0.0) = -0.0 and
3406 // nexttoward(-0.0, 0.0) = 0.0
3407 if (Result == detail::cmpEqual) {
3408 APFloat Ret(Op1);
3409 Ret.convert(RetTy->getFltSemantics(), detail::rmNearestTiesToEven,
3410 &LosesInfo);
3411 return ConstantFP::get(RetTy->getContext(), Ret);
3412 }
3413
3414 APFloat Next(Op0);
3415 Next.next(/*nextDown=*/Result == APFloat::cmpGreaterThan);
3416 if (Next.isZero() || Next.isDenormal() || Next.isSignaling())
3417 return nullptr;
3418 return ConstantFP::get(RetTy->getContext(), Next);
3419}
3420
3421static Constant *ConstantFoldLibCall2(StringRef Name, Type *Ty,
3423 const TargetLibraryInfo *TLI = nullptr) {
3424 if (!TLI)
3425 return nullptr;
3426
3427 LibFunc Func = TLI->getLibFunc(Name);
3428 if (Func == NotLibFunc)
3429 return nullptr;
3430
3431 const auto *Op1 = dyn_cast<ConstantFP>(Operands[0]);
3432 if (!Op1)
3433 return nullptr;
3434
3435 const auto *Op2 = dyn_cast<ConstantFP>(Operands[1]);
3436 if (!Op2)
3437 return nullptr;
3438
3439 const APFloat &Op1V = Op1->getValueAPF();
3440 const APFloat &Op2V = Op2->getValueAPF();
3441
3442 switch (Func) {
3443 default:
3444 break;
3445 case LibFunc_pow:
3446 case LibFunc_powf:
3447 case LibFunc_pow_finite:
3448 case LibFunc_powf_finite:
3449 if (TLI->has(Func))
3450 return ConstantFoldBinaryFP(pow, Op1V, Op2V, Ty);
3451 break;
3452 case LibFunc_fmod:
3453 case LibFunc_fmodf:
3454 if (TLI->has(Func)) {
3455 APFloat V = Op1->getValueAPF();
3456 if (APFloat::opStatus::opOK == V.mod(Op2->getValueAPF()))
3457 return ConstantFP::get(Ty, V);
3458 }
3459 break;
3460 case LibFunc_remainder:
3461 case LibFunc_remainderf:
3462 if (TLI->has(Func)) {
3463 APFloat V = Op1->getValueAPF();
3464 if (APFloat::opStatus::opOK == V.remainder(Op2->getValueAPF()))
3465 return ConstantFP::get(Ty, V);
3466 }
3467 break;
3468 case LibFunc_atan2:
3469 case LibFunc_atan2f:
3470 // atan2(+/-0.0, +/-0.0) is known to raise an exception on some libm
3471 // (Solaris), so we do not assume a known result for that.
3472 if (Op1V.isZero() && Op2V.isZero())
3473 return nullptr;
3474 [[fallthrough]];
3475 case LibFunc_atan2_finite:
3476 case LibFunc_atan2f_finite:
3477 if (TLI->has(Func))
3478 return ConstantFoldBinaryFP(atan2, Op1V, Op2V, Ty);
3479 break;
3480 case LibFunc_nextafter:
3481 case LibFunc_nextafterf:
3482 case LibFunc_nexttoward:
3483 case LibFunc_nexttowardf:
3484 if (TLI->has(Func))
3485 return ConstantFoldNextToward(Op1V, Op2V, Ty);
3486 break;
3487 }
3488
3489 return nullptr;
3490}
3491
3492static Constant *ConstantFoldCRC32(Type *Ty, const APInt *CrcArg,
3493 const APInt *DataArg, unsigned DataBytes,
3494 uint32_t Poly) {
3495 if (!CrcArg || !DataArg)
3496 return nullptr;
3497 uint32_t Crc = CrcArg->getZExtValue();
3498 uint64_t Data = DataArg->getZExtValue();
3499 uint32_t Result = calculateReflectedCRC32(Crc, Data, DataBytes, Poly);
3500 return ConstantInt::get(Ty, Result);
3501}
3502
3503static Constant *ConstantFoldIntrinsicCall2(Intrinsic::ID IntrinsicID, Type *Ty,
3505 const CallBase *Call = nullptr) {
3506 assert(Operands.size() == 2 && "Wrong number of operands.");
3507
3508 if (Ty->isFloatingPointTy()) {
3509 // TODO: We should have undef handling for all of the FP intrinsics that
3510 // are attempted to be folded in this function.
3511 bool IsOp0Undef = isa<UndefValue>(Operands[0]);
3512 bool IsOp1Undef = isa<UndefValue>(Operands[1]);
3513 switch (IntrinsicID) {
3514 case Intrinsic::maxnum:
3515 case Intrinsic::minnum:
3516 case Intrinsic::maximum:
3517 case Intrinsic::minimum:
3518 case Intrinsic::maximumnum:
3519 case Intrinsic::minimumnum:
3520 case Intrinsic::nvvm_fmax_d:
3521 case Intrinsic::nvvm_fmin_d:
3522 // If one argument is undef, return the other argument.
3523 if (IsOp0Undef)
3524 return Operands[1];
3525 if (IsOp1Undef)
3526 return Operands[0];
3527 break;
3528
3529 case Intrinsic::nvvm_fmax_f:
3530 case Intrinsic::nvvm_fmax_ftz_f:
3531 case Intrinsic::nvvm_fmax_ftz_nan_f:
3532 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3533 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3534 case Intrinsic::nvvm_fmax_nan_f:
3535 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3536 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3537
3538 case Intrinsic::nvvm_fmin_f:
3539 case Intrinsic::nvvm_fmin_ftz_f:
3540 case Intrinsic::nvvm_fmin_ftz_nan_f:
3541 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
3542 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
3543 case Intrinsic::nvvm_fmin_nan_f:
3544 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
3545 case Intrinsic::nvvm_fmin_xorsign_abs_f:
3546 // If one arg is undef, the other arg can be returned only if it is
3547 // constant, as we may need to flush it to sign-preserving zero or
3548 // canonicalize the NaN.
3549 if (!IsOp0Undef && !IsOp1Undef)
3550 break;
3551 if (auto *Op = dyn_cast<ConstantFP>(Operands[IsOp0Undef ? 1 : 0])) {
3552 if (Op->isNaN()) {
3553 APInt NVCanonicalNaN(32, 0x7fffffff);
3554 return ConstantFP::get(
3555 Ty, APFloat(Ty->getFltSemantics(), NVCanonicalNaN));
3556 }
3557 if (nvvm::FMinFMaxShouldFTZ(IntrinsicID))
3558 return ConstantFP::get(Ty, FTZPreserveSign(Op->getValueAPF()));
3559 else
3560 return Op;
3561 }
3562 break;
3563 }
3564 }
3565
3566 if (const auto *Op1 = dyn_cast<ConstantFP>(Operands[0])) {
3567 const APFloat &Op1V = Op1->getValueAPF();
3568
3569 if (const auto *Op2 = dyn_cast<ConstantFP>(Operands[1])) {
3570 if (Op2->getType() != Op1->getType())
3571 return nullptr;
3572 const APFloat &Op2V = Op2->getValueAPF();
3573
3574 if (const auto *ConstrIntr =
3576 RoundingMode RM = getEvaluationRoundingMode(ConstrIntr);
3577 APFloat Res = Op1V;
3579 switch (IntrinsicID) {
3580 default:
3581 return nullptr;
3582 case Intrinsic::experimental_constrained_fadd:
3583 St = Res.add(Op2V, RM);
3584 break;
3585 case Intrinsic::experimental_constrained_fsub:
3586 St = Res.subtract(Op2V, RM);
3587 break;
3588 case Intrinsic::experimental_constrained_fmul:
3589 St = Res.multiply(Op2V, RM);
3590 break;
3591 case Intrinsic::experimental_constrained_fdiv:
3592 St = Res.divide(Op2V, RM);
3593 break;
3594 case Intrinsic::experimental_constrained_frem:
3595 St = Res.mod(Op2V);
3596 break;
3597 case Intrinsic::experimental_constrained_fcmp:
3598 case Intrinsic::experimental_constrained_fcmps:
3599 return evaluateCompare(Op1V, Op2V, ConstrIntr);
3600 }
3601 if (mayFoldConstrained(const_cast<ConstrainedFPIntrinsic *>(ConstrIntr),
3602 St))
3603 return ConstantFP::get(Ty, Res);
3604 return nullptr;
3605 }
3606
3607 switch (IntrinsicID) {
3608 default:
3609 break;
3610 case Intrinsic::copysign:
3611 return ConstantFP::get(Ty, APFloat::copySign(Op1V, Op2V));
3612 case Intrinsic::minnum:
3613 return ConstantFP::get(Ty, minnum(Op1V, Op2V));
3614 case Intrinsic::maxnum:
3615 return ConstantFP::get(Ty, maxnum(Op1V, Op2V));
3616 case Intrinsic::minimum:
3617 return ConstantFP::get(Ty, minimum(Op1V, Op2V));
3618 case Intrinsic::maximum:
3619 return ConstantFP::get(Ty, maximum(Op1V, Op2V));
3620 case Intrinsic::minimumnum:
3621 return ConstantFP::get(Ty, minimumnum(Op1V, Op2V));
3622 case Intrinsic::maximumnum:
3623 return ConstantFP::get(Ty, maximumnum(Op1V, Op2V));
3624
3625 case Intrinsic::nvvm_fmax_d:
3626 case Intrinsic::nvvm_fmax_f:
3627 case Intrinsic::nvvm_fmax_ftz_f:
3628 case Intrinsic::nvvm_fmax_ftz_nan_f:
3629 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3630 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3631 case Intrinsic::nvvm_fmax_nan_f:
3632 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3633 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3634
3635 case Intrinsic::nvvm_fmin_d:
3636 case Intrinsic::nvvm_fmin_f:
3637 case Intrinsic::nvvm_fmin_ftz_f:
3638 case Intrinsic::nvvm_fmin_ftz_nan_f:
3639 case Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_f:
3640 case Intrinsic::nvvm_fmin_ftz_xorsign_abs_f:
3641 case Intrinsic::nvvm_fmin_nan_f:
3642 case Intrinsic::nvvm_fmin_nan_xorsign_abs_f:
3643 case Intrinsic::nvvm_fmin_xorsign_abs_f: {
3644
3645 bool ShouldCanonicalizeNaNs = !(IntrinsicID == Intrinsic::nvvm_fmax_d ||
3646 IntrinsicID == Intrinsic::nvvm_fmin_d);
3647 bool IsFTZ = nvvm::FMinFMaxShouldFTZ(IntrinsicID);
3648 bool IsNaNPropagating = nvvm::FMinFMaxPropagatesNaNs(IntrinsicID);
3649 bool IsXorSignAbs = nvvm::FMinFMaxIsXorSignAbs(IntrinsicID);
3650
3651 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3652 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3653
3654 bool XorSign = false;
3655 if (IsXorSignAbs) {
3656 XorSign = A.isNegative() ^ B.isNegative();
3657 A = abs(A);
3658 B = abs(B);
3659 }
3660
3661 bool IsFMax = false;
3662 switch (IntrinsicID) {
3663 case Intrinsic::nvvm_fmax_d:
3664 case Intrinsic::nvvm_fmax_f:
3665 case Intrinsic::nvvm_fmax_ftz_f:
3666 case Intrinsic::nvvm_fmax_ftz_nan_f:
3667 case Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_f:
3668 case Intrinsic::nvvm_fmax_ftz_xorsign_abs_f:
3669 case Intrinsic::nvvm_fmax_nan_f:
3670 case Intrinsic::nvvm_fmax_nan_xorsign_abs_f:
3671 case Intrinsic::nvvm_fmax_xorsign_abs_f:
3672 IsFMax = true;
3673 break;
3674 }
3675 APFloat Res =
3676 IsFMax ? (IsNaNPropagating ? maximum(A, B) : maximumnum(A, B))
3677 : (IsNaNPropagating ? minimum(A, B) : minimumnum(A, B));
3678
3679 if (ShouldCanonicalizeNaNs && Res.isNaN()) {
3680 APFloat NVCanonicalNaN(Res.getSemantics(), APInt(32, 0x7fffffff));
3681 return ConstantFP::get(Ty, NVCanonicalNaN);
3682 }
3683
3684 if (IsXorSignAbs && XorSign != Res.isNegative())
3685 Res.changeSign();
3686
3687 return ConstantFP::get(Ty, Res);
3688 }
3689
3690 case Intrinsic::nvvm_div_rm_f:
3691 case Intrinsic::nvvm_div_rn_f:
3692 case Intrinsic::nvvm_div_rp_f:
3693 case Intrinsic::nvvm_div_rz_f:
3694 case Intrinsic::nvvm_div_rm_d:
3695 case Intrinsic::nvvm_div_rn_d:
3696 case Intrinsic::nvvm_div_rp_d:
3697 case Intrinsic::nvvm_div_rz_d:
3698 case Intrinsic::nvvm_div_rm_ftz_f:
3699 case Intrinsic::nvvm_div_rn_ftz_f:
3700 case Intrinsic::nvvm_div_rp_ftz_f:
3701 case Intrinsic::nvvm_div_rz_ftz_f: {
3702 bool IsFTZ = nvvm::FDivShouldFTZ(IntrinsicID);
3703 APFloat A = IsFTZ ? FTZPreserveSign(Op1V) : Op1V;
3704 APFloat B = IsFTZ ? FTZPreserveSign(Op2V) : Op2V;
3705 APFloat::roundingMode RoundMode =
3706 nvvm::GetFDivRoundingMode(IntrinsicID);
3707
3708 APFloat Res = A;
3709 APFloat::opStatus Status = Res.divide(B, RoundMode);
3710 if (!Res.isNaN() &&
3712 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
3713 return ConstantFP::get(Ty, Res);
3714 }
3715 return nullptr;
3716 }
3717 }
3718
3719 if (!Ty->isHalfTy() && !Ty->isFloatTy() && !Ty->isDoubleTy())
3720 return nullptr;
3721
3722 switch (IntrinsicID) {
3723 default:
3724 break;
3725 case Intrinsic::pow:
3726 return ConstantFoldBinaryFP(pow, Op1V, Op2V, Ty);
3727 case Intrinsic::amdgcn_fmul_legacy:
3728 // The legacy behaviour is that multiplying +/- 0.0 by anything, even
3729 // NaN or infinity, gives +0.0.
3730 if (Op1V.isZero() || Op2V.isZero())
3731 return ConstantFP::getZero(Ty);
3732 return ConstantFP::get(Ty, Op1V * Op2V);
3733 }
3734
3735 } else if (auto *Op2C = dyn_cast<ConstantInt>(Operands[1])) {
3736 switch (IntrinsicID) {
3737 case Intrinsic::ldexp: {
3738 // APFloat::scalbn takes the exponent as `int`. Clamp wider integer
3739 // exponents into [INT_MIN, INT_MAX] so values still saturate the
3740 // result to +/-inf or +/-0.
3741 APInt Exp = Op2C->getValue();
3742 Exp = Exp.getBitWidth() < 32 ? Exp.sext(32) : Exp.truncSSat(32);
3743 return ConstantFP::get(
3744 Ty->getContext(),
3745 scalbn(Op1V, Exp.getSExtValue(), APFloat::rmNearestTiesToEven));
3746 }
3747 case Intrinsic::is_fpclass: {
3748 FPClassTest Mask = static_cast<FPClassTest>(Op2C->getZExtValue());
3749 bool Result =
3750 ((Mask & fcSNan) && Op1V.isNaN() && Op1V.isSignaling()) ||
3751 ((Mask & fcQNan) && Op1V.isNaN() && !Op1V.isSignaling()) ||
3752 ((Mask & fcNegInf) && Op1V.isNegInfinity()) ||
3753 ((Mask & fcNegNormal) && Op1V.isNormal() && Op1V.isNegative()) ||
3754 ((Mask & fcNegSubnormal) && Op1V.isDenormal() && Op1V.isNegative()) ||
3755 ((Mask & fcNegZero) && Op1V.isZero() && Op1V.isNegative()) ||
3756 ((Mask & fcPosZero) && Op1V.isZero() && !Op1V.isNegative()) ||
3757 ((Mask & fcPosSubnormal) && Op1V.isDenormal() && !Op1V.isNegative()) ||
3758 ((Mask & fcPosNormal) && Op1V.isNormal() && !Op1V.isNegative()) ||
3759 ((Mask & fcPosInf) && Op1V.isPosInfinity());
3760 return ConstantInt::get(Ty, Result);
3761 }
3762 case Intrinsic::powi: {
3763 // Square-and-multiply using the operand's own semantics, matching
3764 // the multiply sequence ExpandPowI builds in SelectionDAG.
3765 int Exp = static_cast<int>(Op2C->getSExtValue());
3766 unsigned UExp = static_cast<unsigned>(Exp);
3767 if (Exp < 0)
3768 UExp = -UExp;
3769 const fltSemantics &Semantics = Op1V.getSemantics();
3770 APFloat Res = APFloat::getOne(Semantics);
3771 APFloat CurSquare = Op1V;
3772 while (UExp) {
3773 if (UExp & 1)
3774 Res = Res * CurSquare;
3775 CurSquare = CurSquare * CurSquare;
3776 UExp >>= 1;
3777 }
3778 if (Exp < 0)
3779 Res = APFloat::getOne(Semantics) / Res;
3780 return ConstantFP::get(Ty, Res);
3781 }
3782 default:
3783 break;
3784 }
3785 }
3786 return nullptr;
3787 }
3788
3789 if (Operands[0]->getType()->isIntegerTy() &&
3790 Operands[1]->getType()->isIntegerTy()) {
3791 const APInt *C0, *C1;
3792 if (!getConstIntOrUndef(Operands[0], C0) ||
3793 !getConstIntOrUndef(Operands[1], C1))
3794 return nullptr;
3795
3796 switch (IntrinsicID) {
3797 default: break;
3798 case Intrinsic::smax:
3799 case Intrinsic::smin:
3800 case Intrinsic::umax:
3801 case Intrinsic::umin:
3802 if (!C0 || !C1)
3803 return MinMaxIntrinsic::getSaturationPoint(IntrinsicID, Ty);
3804 return ConstantInt::get(
3805 Ty, ICmpInst::compare(*C0, *C1,
3806 MinMaxIntrinsic::getPredicate(IntrinsicID))
3807 ? *C0
3808 : *C1);
3809
3810 case Intrinsic::scmp:
3811 case Intrinsic::ucmp:
3812 if (!C0 || !C1)
3813 return ConstantInt::get(Ty, 0);
3814
3815 int Res;
3816 if (IntrinsicID == Intrinsic::scmp)
3817 Res = C0->sgt(*C1) ? 1 : C0->slt(*C1) ? -1 : 0;
3818 else
3819 Res = C0->ugt(*C1) ? 1 : C0->ult(*C1) ? -1 : 0;
3820 return ConstantInt::get(Ty, Res, /*IsSigned=*/true);
3821
3822 case Intrinsic::usub_with_overflow:
3823 case Intrinsic::ssub_with_overflow:
3824 // X - undef -> { 0, false }
3825 // undef - X -> { 0, false }
3826 if (!C0 || !C1)
3827 return Constant::getNullValue(Ty);
3828 [[fallthrough]];
3829 case Intrinsic::uadd_with_overflow:
3830 case Intrinsic::sadd_with_overflow:
3831 // X + undef -> { -1, false }
3832 // undef + x -> { -1, false }
3833 if (!C0 || !C1) {
3834 return ConstantStruct::get(
3835 cast<StructType>(Ty),
3836 {Constant::getAllOnesValue(Ty->getStructElementType(0)),
3837 Constant::getNullValue(Ty->getStructElementType(1))});
3838 }
3839 [[fallthrough]];
3840 case Intrinsic::smul_with_overflow:
3841 case Intrinsic::umul_with_overflow: {
3842 // undef * X -> { 0, false }
3843 // X * undef -> { 0, false }
3844 if (!C0 || !C1)
3845 return Constant::getNullValue(Ty);
3846
3847 APInt Res;
3848 bool Overflow;
3849 switch (IntrinsicID) {
3850 default: llvm_unreachable("Invalid case");
3851 case Intrinsic::sadd_with_overflow:
3852 Res = C0->sadd_ov(*C1, Overflow);
3853 break;
3854 case Intrinsic::uadd_with_overflow:
3855 Res = C0->uadd_ov(*C1, Overflow);
3856 break;
3857 case Intrinsic::ssub_with_overflow:
3858 Res = C0->ssub_ov(*C1, Overflow);
3859 break;
3860 case Intrinsic::usub_with_overflow:
3861 Res = C0->usub_ov(*C1, Overflow);
3862 break;
3863 case Intrinsic::smul_with_overflow:
3864 Res = C0->smul_ov(*C1, Overflow);
3865 break;
3866 case Intrinsic::umul_with_overflow:
3867 Res = C0->umul_ov(*C1, Overflow);
3868 break;
3869 }
3870 Constant *Ops[] = {
3871 ConstantInt::get(Ty->getContext(), Res),
3872 ConstantInt::get(Type::getInt1Ty(Ty->getContext()), Overflow)
3873 };
3875 }
3876 case Intrinsic::uadd_sat:
3877 case Intrinsic::sadd_sat:
3878 if (!C0 || !C1)
3879 return Constant::getAllOnesValue(Ty);
3880 if (IntrinsicID == Intrinsic::uadd_sat)
3881 return ConstantInt::get(Ty, C0->uadd_sat(*C1));
3882 else
3883 return ConstantInt::get(Ty, C0->sadd_sat(*C1));
3884 case Intrinsic::usub_sat:
3885 case Intrinsic::ssub_sat:
3886 if (!C0 || !C1)
3887 return Constant::getNullValue(Ty);
3888 if (IntrinsicID == Intrinsic::usub_sat)
3889 return ConstantInt::get(Ty, C0->usub_sat(*C1));
3890 else
3891 return ConstantInt::get(Ty, C0->ssub_sat(*C1));
3892 case Intrinsic::cttz:
3893 case Intrinsic::ctlz:
3894 assert(C1 && "Must be constant int");
3895
3896 // cttz(0, 1) and ctlz(0, 1) are poison.
3897 if (C1->isOne() && (!C0 || C0->isZero()))
3898 return PoisonValue::get(Ty);
3899 if (!C0)
3900 return Constant::getNullValue(Ty);
3901 if (IntrinsicID == Intrinsic::cttz)
3902 return ConstantInt::get(Ty, C0->countr_zero());
3903 else
3904 return ConstantInt::get(Ty, C0->countl_zero());
3905
3906 case Intrinsic::abs:
3907 assert(C1 && "Must be constant int");
3908 assert((C1->isOne() || C1->isZero()) && "Must be 0 or 1");
3909
3910 // Undef or minimum val operand with poison min --> poison
3911 if (C1->isOne() && (!C0 || C0->isMinSignedValue()))
3912 return PoisonValue::get(Ty);
3913
3914 // Undef operand with no poison min --> 0 (sign bit must be clear)
3915 if (!C0)
3916 return Constant::getNullValue(Ty);
3917
3918 return ConstantInt::get(Ty, C0->abs());
3919 case Intrinsic::clmul:
3920 if (!C0 || !C1)
3921 return Constant::getNullValue(Ty);
3922 return ConstantInt::get(Ty, APIntOps::clmul(*C0, *C1));
3923 case Intrinsic::pdep:
3924 if (!C0 || !C1)
3925 return Constant::getNullValue(Ty);
3926 return ConstantInt::get(Ty, APIntOps::pdep(*C0, *C1));
3927 case Intrinsic::pext:
3928 if (!C0 || !C1)
3929 return Constant::getNullValue(Ty);
3930 return ConstantInt::get(Ty, APIntOps::pext(*C0, *C1));
3931 case Intrinsic::smulh:
3932 if (!C0 || !C1)
3933 return Constant::getNullValue(Ty);
3934 return ConstantInt::get(Ty, APIntOps::mulhs(*C0, *C1));
3935 case Intrinsic::umulh:
3936 if (!C0 || !C1)
3937 return Constant::getNullValue(Ty);
3938 return ConstantInt::get(Ty, APIntOps::mulhu(*C0, *C1));
3939 case Intrinsic::amdgcn_wave_reduce_add:
3940 case Intrinsic::amdgcn_wave_reduce_sub:
3941 case Intrinsic::amdgcn_wave_reduce_xor: {
3942 if (C0 && C0->isZero())
3943 return Constant::getNullValue(Ty);
3944 return nullptr;
3945 }
3946 case Intrinsic::amdgcn_wave_reduce_umin:
3947 case Intrinsic::amdgcn_wave_reduce_umax:
3948 case Intrinsic::amdgcn_wave_reduce_max:
3949 case Intrinsic::amdgcn_wave_reduce_min:
3950 case Intrinsic::amdgcn_wave_reduce_and:
3951 case Intrinsic::amdgcn_wave_reduce_or:
3952 return Operands[0];
3953 case Intrinsic::aarch64_crc32b:
3954 return ConstantFoldCRC32(Ty, C0, C1, 1, 0xEDB88320);
3955 case Intrinsic::aarch64_crc32h:
3956 return ConstantFoldCRC32(Ty, C0, C1, 2, 0xEDB88320);
3957 case Intrinsic::aarch64_crc32w:
3958 return ConstantFoldCRC32(Ty, C0, C1, 4, 0xEDB88320);
3959 case Intrinsic::aarch64_crc32x:
3960 return ConstantFoldCRC32(Ty, C0, C1, 8, 0xEDB88320);
3961 case Intrinsic::aarch64_crc32cb:
3962 case Intrinsic::x86_sse42_crc32_32_8:
3963 return ConstantFoldCRC32(Ty, C0, C1, 1, 0x82F63B78);
3964 case Intrinsic::aarch64_crc32ch:
3965 case Intrinsic::x86_sse42_crc32_32_16:
3966 return ConstantFoldCRC32(Ty, C0, C1, 2, 0x82F63B78);
3967 case Intrinsic::aarch64_crc32cw:
3968 case Intrinsic::x86_sse42_crc32_32_32:
3969 return ConstantFoldCRC32(Ty, C0, C1, 4, 0x82F63B78);
3970 case Intrinsic::aarch64_crc32cx:
3971 case Intrinsic::x86_sse42_crc32_64_64:
3972 return ConstantFoldCRC32(Ty, C0, C1, 8, 0x82F63B78);
3973 }
3974
3975 return nullptr;
3976 }
3977
3978 // Support ConstantVector in case we have an Undef in the top.
3979 if ((isa<ConstantVector>(Operands[0]) ||
3981 // Check for default rounding mode.
3982 // FIXME: Support other rounding modes?
3984 cast<ConstantInt>(Operands[1])->getValue() == 4) {
3985 auto *Op = cast<Constant>(Operands[0]);
3986 switch (IntrinsicID) {
3987 default: break;
3988 case Intrinsic::x86_avx512_vcvtss2si32:
3989 case Intrinsic::x86_avx512_vcvtss2si64:
3990 case Intrinsic::x86_avx512_vcvtsd2si32:
3991 case Intrinsic::x86_avx512_vcvtsd2si64:
3992 if (ConstantFP *FPOp =
3993 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
3994 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
3995 /*roundTowardZero=*/false, Ty,
3996 /*IsSigned*/true);
3997 break;
3998 case Intrinsic::x86_avx512_vcvtss2usi32:
3999 case Intrinsic::x86_avx512_vcvtss2usi64:
4000 case Intrinsic::x86_avx512_vcvtsd2usi32:
4001 case Intrinsic::x86_avx512_vcvtsd2usi64:
4002 if (ConstantFP *FPOp =
4003 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
4004 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
4005 /*roundTowardZero=*/false, Ty,
4006 /*IsSigned*/false);
4007 break;
4008 case Intrinsic::x86_avx512_cvttss2si:
4009 case Intrinsic::x86_avx512_cvttss2si64:
4010 case Intrinsic::x86_avx512_cvttsd2si:
4011 case Intrinsic::x86_avx512_cvttsd2si64:
4012 if (ConstantFP *FPOp =
4013 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
4014 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
4015 /*roundTowardZero=*/true, Ty,
4016 /*IsSigned*/true);
4017 break;
4018 case Intrinsic::x86_avx512_cvttss2usi:
4019 case Intrinsic::x86_avx512_cvttss2usi64:
4020 case Intrinsic::x86_avx512_cvttsd2usi:
4021 case Intrinsic::x86_avx512_cvttsd2usi64:
4022 if (ConstantFP *FPOp =
4023 dyn_cast_or_null<ConstantFP>(Op->getAggregateElement(0U)))
4024 return ConstantFoldSSEConvertToInt(FPOp->getValueAPF(),
4025 /*roundTowardZero=*/true, Ty,
4026 /*IsSigned*/false);
4027 break;
4028 }
4029 }
4030
4031 if (IntrinsicID == Intrinsic::experimental_cttz_elts) {
4032 auto *FVTy = dyn_cast<FixedVectorType>(Operands[0]->getType());
4033 bool ZeroIsPoison = cast<ConstantInt>(Operands[1])->isOne();
4034 if (!FVTy)
4035 return nullptr;
4036 unsigned Width = Ty->getIntegerBitWidth();
4037 if (APInt::getMaxValue(Width).ult(FVTy->getNumElements()) ||
4038 Operands[0]->containsPoisonElement())
4039 return PoisonValue::get(Ty);
4040 for (unsigned I = 0; I < FVTy->getNumElements(); ++I) {
4041 Constant *Elt = Operands[0]->getAggregateElement(I);
4042 if (!Elt)
4043 return nullptr;
4044 if (isa<UndefValue>(Elt) || Elt->isNullValue())
4045 continue;
4046 return ConstantInt::get(Ty, I);
4047 }
4048 if (ZeroIsPoison)
4049 return PoisonValue::get(Ty);
4050 return ConstantInt::get(Ty, FVTy->getNumElements());
4051 }
4052 return nullptr;
4053}
4054
4055static APFloat ConstantFoldAMDGCNCubeIntrinsic(Intrinsic::ID IntrinsicID,
4056 const APFloat &S0,
4057 const APFloat &S1,
4058 const APFloat &S2) {
4059 unsigned ID;
4060 const fltSemantics &Sem = S0.getSemantics();
4061 APFloat MA(Sem), SC(Sem), TC(Sem);
4062 if (abs(S2) >= abs(S0) && abs(S2) >= abs(S1)) {
4063 if (S2.isNegative() && S2.isNonZero() && !S2.isNaN()) {
4064 // S2 < 0
4065 ID = 5;
4066 SC = -S0;
4067 } else {
4068 ID = 4;
4069 SC = S0;
4070 }
4071 MA = S2;
4072 TC = -S1;
4073 } else if (abs(S1) >= abs(S0)) {
4074 if (S1.isNegative() && S1.isNonZero() && !S1.isNaN()) {
4075 // S1 < 0
4076 ID = 3;
4077 TC = -S2;
4078 } else {
4079 ID = 2;
4080 TC = S2;
4081 }
4082 MA = S1;
4083 SC = S0;
4084 } else {
4085 if (S0.isNegative() && S0.isNonZero() && !S0.isNaN()) {
4086 // S0 < 0
4087 ID = 1;
4088 SC = S2;
4089 } else {
4090 ID = 0;
4091 SC = -S2;
4092 }
4093 MA = S0;
4094 TC = -S1;
4095 }
4096 switch (IntrinsicID) {
4097 default:
4098 llvm_unreachable("unhandled amdgcn cube intrinsic");
4099 case Intrinsic::amdgcn_cubeid:
4100 return APFloat(Sem, ID);
4101 case Intrinsic::amdgcn_cubema:
4102 return MA + MA;
4103 case Intrinsic::amdgcn_cubesc:
4104 return SC;
4105 case Intrinsic::amdgcn_cubetc:
4106 return TC;
4107 }
4108}
4109
4110static Constant *ConstantFoldAMDGCNPermIntrinsic(ArrayRef<Constant *> Operands,
4111 Type *Ty) {
4112 const APInt *C0, *C1, *C2;
4113 if (!getConstIntOrUndef(Operands[0], C0) ||
4114 !getConstIntOrUndef(Operands[1], C1) ||
4115 !getConstIntOrUndef(Operands[2], C2))
4116 return nullptr;
4117
4118 if (!C2)
4119 return UndefValue::get(Ty);
4120
4121 APInt Val(32, 0);
4122 unsigned NumUndefBytes = 0;
4123 for (unsigned I = 0; I < 32; I += 8) {
4124 unsigned Sel = C2->extractBitsAsZExtValue(8, I);
4125 unsigned B = 0;
4126
4127 if (Sel >= 13)
4128 B = 0xff;
4129 else if (Sel == 12)
4130 B = 0x00;
4131 else {
4132 const APInt *Src = ((Sel & 10) == 10 || (Sel & 12) == 4) ? C0 : C1;
4133 if (!Src)
4134 ++NumUndefBytes;
4135 else if (Sel < 8)
4136 B = Src->extractBitsAsZExtValue(8, (Sel & 3) * 8);
4137 else
4138 B = Src->extractBitsAsZExtValue(1, (Sel & 1) ? 31 : 15) * 0xff;
4139 }
4140
4141 Val.insertBits(B, I, 8);
4142 }
4143
4144 if (NumUndefBytes == 4)
4145 return UndefValue::get(Ty);
4146
4147 return ConstantInt::get(Ty, Val);
4148}
4149
4150static Constant *ConstantFoldScalarCall3(StringRef Name,
4151 Intrinsic::ID IntrinsicID, Type *Ty,
4153 const TargetLibraryInfo *TLI = nullptr,
4154 const CallBase *Call = nullptr) {
4155 assert(Operands.size() == 3 && "Wrong number of operands.");
4156
4157 if (const auto *Op1 = dyn_cast<ConstantFP>(Operands[0])) {
4158 if (const auto *Op2 = dyn_cast<ConstantFP>(Operands[1])) {
4159 if (const auto *Op3 = dyn_cast<ConstantFP>(Operands[2])) {
4160 const APFloat &C1 = Op1->getValueAPF();
4161 const APFloat &C2 = Op2->getValueAPF();
4162 const APFloat &C3 = Op3->getValueAPF();
4163
4164 if (const auto *ConstrIntr =
4166 RoundingMode RM = getEvaluationRoundingMode(ConstrIntr);
4167 APFloat Res = C1;
4169 switch (IntrinsicID) {
4170 default:
4171 return nullptr;
4172 case Intrinsic::experimental_constrained_fma:
4173 case Intrinsic::experimental_constrained_fmuladd:
4174 St = Res.fusedMultiplyAdd(C2, C3, RM);
4175 break;
4176 }
4177 if (mayFoldConstrained(
4178 const_cast<ConstrainedFPIntrinsic *>(ConstrIntr), St))
4179 return ConstantFP::get(Ty, Res);
4180 return nullptr;
4181 }
4182
4183 switch (IntrinsicID) {
4184 default: break;
4185 case Intrinsic::amdgcn_fma_legacy: {
4186 // The legacy behaviour is that multiplying +/- 0.0 by anything, even
4187 // NaN or infinity, gives +0.0.
4188 if (C1.isZero() || C2.isZero()) {
4189 // It's tempting to just return C3 here, but that would give the
4190 // wrong result if C3 was -0.0.
4191 return ConstantFP::get(Ty, APFloat(0.0f) + C3);
4192 }
4193 [[fallthrough]];
4194 }
4195 case Intrinsic::fma:
4196 case Intrinsic::fmuladd: {
4197 APFloat V = C1;
4199 return ConstantFP::get(Ty, V);
4200 }
4201
4202 case Intrinsic::nvvm_fma_rm_f:
4203 case Intrinsic::nvvm_fma_rn_f:
4204 case Intrinsic::nvvm_fma_rp_f:
4205 case Intrinsic::nvvm_fma_rz_f:
4206 case Intrinsic::nvvm_fma_rm_d:
4207 case Intrinsic::nvvm_fma_rn_d:
4208 case Intrinsic::nvvm_fma_rp_d:
4209 case Intrinsic::nvvm_fma_rz_d:
4210 case Intrinsic::nvvm_fma_rm_ftz_f:
4211 case Intrinsic::nvvm_fma_rn_ftz_f:
4212 case Intrinsic::nvvm_fma_rp_ftz_f:
4213 case Intrinsic::nvvm_fma_rz_ftz_f: {
4214 bool IsFTZ = nvvm::FMAShouldFTZ(IntrinsicID);
4215 APFloat A = IsFTZ ? FTZPreserveSign(C1) : C1;
4216 APFloat B = IsFTZ ? FTZPreserveSign(C2) : C2;
4217 APFloat C = IsFTZ ? FTZPreserveSign(C3) : C3;
4218
4219 APFloat::roundingMode RoundMode =
4220 nvvm::GetFMARoundingMode(IntrinsicID);
4221
4222 APFloat Res = A;
4223 APFloat::opStatus Status = Res.fusedMultiplyAdd(B, C, RoundMode);
4224
4225 if (!Res.isNaN() &&
4227 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
4228 return ConstantFP::get(Ty, Res);
4229 }
4230 return nullptr;
4231 }
4232
4233 case Intrinsic::amdgcn_cubeid:
4234 case Intrinsic::amdgcn_cubema:
4235 case Intrinsic::amdgcn_cubesc:
4236 case Intrinsic::amdgcn_cubetc: {
4237 APFloat V = ConstantFoldAMDGCNCubeIntrinsic(IntrinsicID, C1, C2, C3);
4238 return ConstantFP::get(Ty, V);
4239 }
4240 }
4241 }
4242
4243 // TODO: Add constant folding for the _sat variants.
4244 const bool IsFAdd = IntrinsicID == Intrinsic::nvvm_fadd ||
4245 IntrinsicID == Intrinsic::nvvm_fadd_ftz;
4246 const bool IsFMul = IntrinsicID == Intrinsic::nvvm_fmul ||
4247 IntrinsicID == Intrinsic::nvvm_fmul_ftz;
4248 if (IsFAdd || IsFMul) {
4249 bool IsFTZ = IntrinsicID == Intrinsic::nvvm_fadd_ftz ||
4250 IntrinsicID == Intrinsic::nvvm_fmul_ftz;
4251 APFloat A =
4252 IsFTZ ? FTZPreserveSign(Op1->getValueAPF()) : Op1->getValueAPF();
4253 APFloat B =
4254 IsFTZ ? FTZPreserveSign(Op2->getValueAPF()) : Op2->getValueAPF();
4255
4256 APFloat::roundingMode RoundMode =
4258
4259 APFloat Res = A;
4261 IsFAdd ? Res.add(B, RoundMode) : Res.multiply(B, RoundMode);
4262
4263 if (!Res.isNaN() &&
4265 Res = IsFTZ ? FTZPreserveSign(Res) : Res;
4266 return ConstantFP::get(Ty, Res);
4267 }
4268 return nullptr;
4269 }
4270 }
4271 }
4272
4273 if (IntrinsicID == Intrinsic::smul_fix ||
4274 IntrinsicID == Intrinsic::smul_fix_sat) {
4275 const APInt *C0, *C1;
4276 if (!getConstIntOrUndef(Operands[0], C0) ||
4277 !getConstIntOrUndef(Operands[1], C1))
4278 return nullptr;
4279
4280 // undef * C -> 0
4281 // C * undef -> 0
4282 if (!C0 || !C1)
4283 return Constant::getNullValue(Ty);
4284
4285 // This code performs rounding towards negative infinity in case the result
4286 // cannot be represented exactly for the given scale. Targets that do care
4287 // about rounding should use a target hook for specifying how rounding
4288 // should be done, and provide their own folding to be consistent with
4289 // rounding. This is the same approach as used by
4290 // DAGTypeLegalizer::ExpandIntRes_MULFIX.
4291 unsigned Scale = cast<ConstantInt>(Operands[2])->getZExtValue();
4292 unsigned Width = C0->getBitWidth();
4293 assert(Scale < Width && "Illegal scale.");
4294 unsigned ExtendedWidth = Width * 2;
4295 APInt Product =
4296 (C0->sext(ExtendedWidth) * C1->sext(ExtendedWidth)).ashr(Scale);
4297 if (IntrinsicID == Intrinsic::smul_fix_sat) {
4298 APInt Max = APInt::getSignedMaxValue(Width).sext(ExtendedWidth);
4299 APInt Min = APInt::getSignedMinValue(Width).sext(ExtendedWidth);
4300 Product = APIntOps::smin(Product, Max);
4301 Product = APIntOps::smax(Product, Min);
4302 }
4303 return ConstantInt::get(Ty->getContext(), Product.sextOrTrunc(Width));
4304 }
4305
4306 if (IntrinsicID == Intrinsic::fshl || IntrinsicID == Intrinsic::fshr) {
4307 const APInt *C0, *C1, *C2;
4308 if (!getConstIntOrUndef(Operands[0], C0) ||
4309 !getConstIntOrUndef(Operands[1], C1) ||
4310 !getConstIntOrUndef(Operands[2], C2))
4311 return nullptr;
4312
4313 bool IsRight = IntrinsicID == Intrinsic::fshr;
4314 if (!C2)
4315 return Operands[IsRight ? 1 : 0];
4316 if (!C0 && !C1)
4317 return UndefValue::get(Ty);
4318
4319 // The shift amount is interpreted as modulo the bitwidth. If the shift
4320 // amount is effectively 0, avoid UB due to oversized inverse shift below.
4321 unsigned BitWidth = C2->getBitWidth();
4322 unsigned ShAmt = C2->urem(BitWidth);
4323 if (!ShAmt)
4324 return Operands[IsRight ? 1 : 0];
4325
4326 // (C0 << ShlAmt) | (C1 >> LshrAmt)
4327 unsigned LshrAmt = IsRight ? ShAmt : BitWidth - ShAmt;
4328 unsigned ShlAmt = !IsRight ? ShAmt : BitWidth - ShAmt;
4329 if (!C0)
4330 return ConstantInt::get(Ty, C1->lshr(LshrAmt));
4331 if (!C1)
4332 return ConstantInt::get(Ty, C0->shl(ShlAmt));
4333 return ConstantInt::get(Ty, C0->shl(ShlAmt) | C1->lshr(LshrAmt));
4334 }
4335
4336 if (IntrinsicID == Intrinsic::amdgcn_perm)
4337 return ConstantFoldAMDGCNPermIntrinsic(Operands, Ty);
4338
4339 return nullptr;
4340}
4341
4342static Constant *ConstantFoldScalarCall(StringRef Name,
4343 Intrinsic::ID IntrinsicID, Type *Ty,
4345 const TargetLibraryInfo *TLI = nullptr,
4346 const CallBase *Call = nullptr) {
4347 if (IntrinsicID != Intrinsic::not_intrinsic &&
4349 intrinsicPropagatesPoison(IntrinsicID))
4350 return PoisonValue::get(Ty);
4351
4352 if (Operands.size() == 1)
4353 return ConstantFoldScalarCall1(Name, IntrinsicID, Ty, Operands, TLI, Call);
4354
4355 if (Operands.size() == 2) {
4356 if (Constant *FoldedLibCall =
4357 ConstantFoldLibCall2(Name, Ty, Operands, TLI)) {
4358 return FoldedLibCall;
4359 }
4360 return ConstantFoldIntrinsicCall2(IntrinsicID, Ty, Operands, Call);
4361 }
4362
4363 if (Operands.size() == 3)
4364 return ConstantFoldScalarCall3(Name, IntrinsicID, Ty, Operands, TLI, Call);
4365
4366 return nullptr;
4367}
4368
4369static Constant *ConstantFoldFixedVectorCall(
4370 StringRef Name, Intrinsic::ID IntrinsicID, FixedVectorType *FVTy,
4372 const TargetLibraryInfo *TLI = nullptr, const CallBase *Call = nullptr) {
4375 Type *Ty = FVTy->getElementType();
4376
4377 switch (IntrinsicID) {
4378 case Intrinsic::masked_load: {
4379 auto *SrcPtr = Operands[0];
4380 auto *Mask = Operands[1];
4381 auto *Passthru = Operands[2];
4382
4383 Constant *VecData = ConstantFoldLoadFromConstPtr(SrcPtr, FVTy, DL);
4384
4385 SmallVector<Constant *, 32> NewElements;
4386 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4387 auto *MaskElt = Mask->getAggregateElement(I);
4388 if (!MaskElt)
4389 break;
4390 auto *PassthruElt = Passthru->getAggregateElement(I);
4391 auto *VecElt = VecData ? VecData->getAggregateElement(I) : nullptr;
4392 if (isa<UndefValue>(MaskElt)) {
4393 if (PassthruElt)
4394 NewElements.push_back(PassthruElt);
4395 else if (VecElt)
4396 NewElements.push_back(VecElt);
4397 else
4398 return nullptr;
4399 }
4400 if (MaskElt->isNullValue()) {
4401 if (!PassthruElt)
4402 return nullptr;
4403 NewElements.push_back(PassthruElt);
4404 } else if (MaskElt->isOneValue()) {
4405 if (!VecElt)
4406 return nullptr;
4407 NewElements.push_back(VecElt);
4408 } else {
4409 return nullptr;
4410 }
4411 }
4412 if (NewElements.size() != FVTy->getNumElements())
4413 return nullptr;
4414 return ConstantVector::get(NewElements);
4415 }
4416 case Intrinsic::arm_mve_vctp8:
4417 case Intrinsic::arm_mve_vctp16:
4418 case Intrinsic::arm_mve_vctp32:
4419 case Intrinsic::arm_mve_vctp64: {
4420 if (auto *Op = dyn_cast<ConstantInt>(Operands[0])) {
4421 unsigned Lanes = FVTy->getNumElements();
4422 uint64_t Limit = Op->getZExtValue();
4423
4425 for (unsigned i = 0; i < Lanes; i++) {
4426 if (i < Limit)
4428 else
4430 }
4431 return ConstantVector::get(NCs);
4432 }
4433 return nullptr;
4434 }
4435 case Intrinsic::get_active_lane_mask: {
4436 auto *Op0 = dyn_cast<ConstantInt>(Operands[0]);
4437 auto *Op1 = dyn_cast<ConstantInt>(Operands[1]);
4438 if (Op0 && Op1) {
4439 unsigned Lanes = FVTy->getNumElements();
4440 APInt Base = Op0->getValue();
4441 APInt Limit = Op1->getValue();
4442
4444 for (unsigned I = 0; I < Lanes; I++) {
4445 bool Overflow;
4446 if (Base.uadd_ov(APInt(Base.getBitWidth(), I), Overflow).ult(Limit) &&
4447 !Overflow)
4449 else
4451 }
4452 return ConstantVector::get(NCs);
4453 }
4454 return nullptr;
4455 }
4456 case Intrinsic::vector_extract: {
4458 Constant *Vec = Operands[0];
4459 if (!Idx || !isa<FixedVectorType>(Vec->getType()))
4460 return nullptr;
4461
4462 unsigned NumElements = FVTy->getNumElements();
4463 unsigned VecNumElements =
4464 cast<FixedVectorType>(Vec->getType())->getNumElements();
4465 unsigned StartingIndex = Idx->getZExtValue();
4466
4467 // Extracting entire vector is nop
4468 if (NumElements == VecNumElements && StartingIndex == 0)
4469 return Vec;
4470
4471 for (unsigned I = StartingIndex, E = StartingIndex + NumElements; I < E;
4472 ++I) {
4473 Constant *Elt = Vec->getAggregateElement(I);
4474 if (!Elt)
4475 return nullptr;
4476 Result[I - StartingIndex] = Elt;
4477 }
4478
4479 return ConstantVector::get(Result);
4480 }
4481 case Intrinsic::vector_insert: {
4482 Constant *Vec = Operands[0];
4483 Constant *SubVec = Operands[1];
4485 if (!Idx || !isa<FixedVectorType>(Vec->getType()))
4486 return nullptr;
4487
4488 unsigned SubVecNumElements =
4489 cast<FixedVectorType>(SubVec->getType())->getNumElements();
4490 unsigned VecNumElements =
4491 cast<FixedVectorType>(Vec->getType())->getNumElements();
4492 unsigned IdxN = Idx->getZExtValue();
4493 // Replacing entire vector with a subvec is nop
4494 if (SubVecNumElements == VecNumElements && IdxN == 0)
4495 return SubVec;
4496
4497 for (unsigned I = 0; I < VecNumElements; ++I) {
4498 Constant *Elt;
4499 if (I < IdxN + SubVecNumElements)
4500 Elt = SubVec->getAggregateElement(I - IdxN);
4501 else
4502 Elt = Vec->getAggregateElement(I);
4503 if (!Elt)
4504 return nullptr;
4505 Result[I] = Elt;
4506 }
4507 return ConstantVector::get(Result);
4508 }
4509 case Intrinsic::vector_interleave2:
4510 case Intrinsic::vector_interleave3:
4511 case Intrinsic::vector_interleave4:
4512 case Intrinsic::vector_interleave5:
4513 case Intrinsic::vector_interleave6:
4514 case Intrinsic::vector_interleave7:
4515 case Intrinsic::vector_interleave8: {
4516 unsigned NumElements =
4517 cast<FixedVectorType>(Operands[0]->getType())->getNumElements();
4518 unsigned NumOperands = Operands.size();
4519 for (unsigned I = 0; I < NumElements; ++I) {
4520 for (unsigned J = 0; J < NumOperands; ++J) {
4521 Constant *Elt = Operands[J]->getAggregateElement(I);
4522 if (!Elt)
4523 return nullptr;
4524 Result[NumOperands * I + J] = Elt;
4525 }
4526 }
4527 return ConstantVector::get(Result);
4528 }
4529 case Intrinsic::vector_partial_reduce_add:
4530 return constantFoldVectorPartialReduceAdd(Operands[0], Operands[1], DL);
4531 case Intrinsic::wasm_dot: {
4532 unsigned NumElements =
4533 cast<FixedVectorType>(Operands[0]->getType())->getNumElements();
4534
4535 assert(NumElements == 8 && Result.size() == 4 &&
4536 "wasm dot takes i16x8 and produces i32x4");
4537 assert(Ty->isIntegerTy());
4538 int32_t MulVector[8];
4539
4540 for (unsigned I = 0; I < NumElements; ++I) {
4541 ConstantInt *Elt0 =
4542 dyn_cast<ConstantInt>(Operands[0]->getAggregateElement(I));
4543 ConstantInt *Elt1 =
4544 dyn_cast<ConstantInt>(Operands[1]->getAggregateElement(I));
4545
4546 if (!Elt0 || !Elt1)
4547 return nullptr;
4548
4549 MulVector[I] = Elt0->getSExtValue() * Elt1->getSExtValue();
4550 }
4551 for (unsigned I = 0; I < Result.size(); I++) {
4552 int64_t IAdd = (int64_t)MulVector[I * 2] + (int64_t)MulVector[I * 2 + 1];
4553 Result[I] = ConstantInt::getSigned(Ty, IAdd, /*ImplicitTrunc=*/true);
4554 }
4555
4556 return ConstantVector::get(Result);
4557 }
4558 case Intrinsic::nvvm_fadd:
4559 case Intrinsic::nvvm_fadd_ftz:
4560 case Intrinsic::nvvm_fmul:
4561 case Intrinsic::nvvm_fmul_ftz:
4562 // The rounding mode operand is a scalar, so the lane-wise folding below
4563 // does not apply.
4564 // TODO: Fold these by passing the rounding mode through to every lane.
4565 return nullptr;
4566 default:
4567 break;
4568 }
4569
4570 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4571 // Gather a column of constants.
4572 for (unsigned J = 0, JE = Operands.size(); J != JE; ++J) {
4573 // Some intrinsics use a scalar type for certain arguments.
4574 if (isVectorIntrinsicWithScalarOpAtArg(IntrinsicID, J, /*TTI=*/nullptr)) {
4575 Lane[J] = Operands[J];
4576 continue;
4577 }
4578
4579 Constant *Agg = Operands[J]->getAggregateElement(I);
4580 if (!Agg)
4581 return nullptr;
4582
4583 Lane[J] = Agg;
4584 }
4585
4586 // Use the regular scalar folding to simplify this column.
4587 Constant *Folded =
4588 ConstantFoldScalarCall(Name, IntrinsicID, Ty, Lane, TLI, Call);
4589 if (!Folded)
4590 return nullptr;
4591 Result[I] = Folded;
4592 }
4593
4594 return ConstantVector::get(Result);
4595}
4596
4597static Constant *ConstantFoldScalableVectorCall(
4598 StringRef Name, Intrinsic::ID IntrinsicID, ScalableVectorType *SVTy,
4600 const TargetLibraryInfo *TLI, const CallBase *Call) {
4601 switch (IntrinsicID) {
4602 case Intrinsic::aarch64_sve_convert_from_svbool: {
4603 Constant *Src = Operands[0];
4604 if (!Src->isNullValue())
4605 break;
4606
4607 return ConstantInt::getFalse(SVTy);
4608 }
4609 case Intrinsic::get_active_lane_mask: {
4610 auto *Op0 = dyn_cast<ConstantInt>(Operands[0]);
4611 auto *Op1 = dyn_cast<ConstantInt>(Operands[1]);
4612 if (Op0 && Op1 && Op0->getValue().uge(Op1->getValue()))
4613 return ConstantVector::getNullValue(SVTy);
4614 break;
4615 }
4616 case Intrinsic::vector_interleave2:
4617 case Intrinsic::vector_interleave3:
4618 case Intrinsic::vector_interleave4:
4619 case Intrinsic::vector_interleave5:
4620 case Intrinsic::vector_interleave6:
4621 case Intrinsic::vector_interleave7:
4622 case Intrinsic::vector_interleave8: {
4623 Constant *SplatVal = Operands[0]->getSplatValue();
4624 if (!SplatVal)
4625 return nullptr;
4626
4628 return nullptr;
4629
4630 return ConstantVector::getSplat(SVTy->getElementCount(), SplatVal);
4631 }
4632 default:
4633 break;
4634 }
4635
4636 // If trivially vectorizable, try folding it via the scalar call if all
4637 // operands are splats.
4638
4639 // TODO: ConstantFoldFixedVectorCall should probably check this too?
4640 if (!isTriviallyVectorizable(IntrinsicID))
4641 return nullptr;
4642
4644 for (auto [I, Op] : enumerate(Operands)) {
4645 if (isVectorIntrinsicWithScalarOpAtArg(IntrinsicID, I, /*TTI=*/nullptr)) {
4646 SplatOps.push_back(Op);
4647 continue;
4648 }
4649 Constant *Splat = Op->getSplatValue();
4650 if (!Splat)
4651 return nullptr;
4652 SplatOps.push_back(Splat);
4653 }
4654 Constant *Folded = ConstantFoldScalarCall(
4655 Name, IntrinsicID, SVTy->getElementType(), SplatOps, TLI, Call);
4656 if (!Folded)
4657 return nullptr;
4658 return ConstantVector::getSplat(SVTy->getElementCount(), Folded);
4659}
4660
4661static std::pair<Constant *, Constant *>
4662ConstantFoldScalarFrexpCall(Constant *Op, Type *IntTy) {
4663 auto *ConstFP = dyn_cast<ConstantFP>(Op);
4664 if (!ConstFP)
4665 return {};
4666
4667 const APFloat &U = ConstFP->getValueAPF();
4668 int FrexpExp;
4669 APFloat FrexpMant = frexp(U, FrexpExp, APFloat::rmNearestTiesToEven);
4670 Constant *Result0 = ConstantFP::get(ConstFP->getType(), FrexpMant);
4671
4672 // The exponent is an "unspecified value" for inf/nan. We use zero to avoid
4673 // using undef.
4674 Constant *Result1 = FrexpMant.isFinite()
4675 ? ConstantInt::getSigned(IntTy, FrexpExp)
4676 : ConstantInt::getNullValue(IntTy);
4677 return {Result0, Result1};
4678}
4679
4680/// Handle intrinsics that return tuples, which may be tuples of vectors.
4681static Constant *
4682ConstantFoldStructCall(StringRef Name, Intrinsic::ID IntrinsicID,
4684 const DataLayout &DL, const TargetLibraryInfo *TLI,
4685 const CallBase *Call) {
4686
4687 switch (IntrinsicID) {
4688 case Intrinsic::frexp: {
4689 Type *Ty0 = StTy->getContainedType(0);
4690 Type *Ty1 = StTy->getContainedType(1)->getScalarType();
4691
4692 if (auto *FVTy0 = dyn_cast<FixedVectorType>(Ty0)) {
4693 SmallVector<Constant *, 4> Results0(FVTy0->getNumElements());
4694 SmallVector<Constant *, 4> Results1(FVTy0->getNumElements());
4695
4696 for (unsigned I = 0, E = FVTy0->getNumElements(); I != E; ++I) {
4697 Constant *Lane = Operands[0]->getAggregateElement(I);
4698 std::tie(Results0[I], Results1[I]) =
4699 ConstantFoldScalarFrexpCall(Lane, Ty1);
4700 if (!Results0[I])
4701 return nullptr;
4702 }
4703
4704 return ConstantStruct::get(StTy, ConstantVector::get(Results0),
4705 ConstantVector::get(Results1));
4706 }
4707
4708 auto [Result0, Result1] = ConstantFoldScalarFrexpCall(Operands[0], Ty1);
4709 if (!Result0)
4710 return nullptr;
4711 return ConstantStruct::get(StTy, Result0, Result1);
4712 }
4713 case Intrinsic::sincos: {
4714 Type *Ty = StTy->getContainedType(0);
4715 Type *TyScalar = Ty->getScalarType();
4716
4717 auto ConstantFoldScalarSincosCall =
4718 [&](Constant *Op) -> std::pair<Constant *, Constant *> {
4719 Constant *SinResult =
4720 ConstantFoldScalarCall(Name, Intrinsic::sin, TyScalar, Op, TLI, Call);
4721 Constant *CosResult =
4722 ConstantFoldScalarCall(Name, Intrinsic::cos, TyScalar, Op, TLI, Call);
4723 return std::make_pair(SinResult, CosResult);
4724 };
4725
4726 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty)) {
4727 SmallVector<Constant *> SinResults(FVTy->getNumElements());
4728 SmallVector<Constant *> CosResults(FVTy->getNumElements());
4729
4730 for (unsigned I = 0, E = FVTy->getNumElements(); I != E; ++I) {
4731 Constant *Lane = Operands[0]->getAggregateElement(I);
4732 std::tie(SinResults[I], CosResults[I]) =
4733 ConstantFoldScalarSincosCall(Lane);
4734 if (!SinResults[I] || !CosResults[I])
4735 return nullptr;
4736 }
4737
4738 return ConstantStruct::get(StTy, ConstantVector::get(SinResults),
4739 ConstantVector::get(CosResults));
4740 }
4741
4742 if (!Ty->isFloatingPointTy())
4743 return nullptr;
4744
4745 auto [SinResult, CosResult] = ConstantFoldScalarSincosCall(Operands[0]);
4746 if (!SinResult || !CosResult)
4747 return nullptr;
4748 return ConstantStruct::get(StTy, SinResult, CosResult);
4749 }
4750 case Intrinsic::vector_deinterleave2:
4751 case Intrinsic::vector_deinterleave3:
4752 case Intrinsic::vector_deinterleave4:
4753 case Intrinsic::vector_deinterleave5:
4754 case Intrinsic::vector_deinterleave6:
4755 case Intrinsic::vector_deinterleave7:
4756 case Intrinsic::vector_deinterleave8: {
4757 unsigned NumResults = StTy->getNumElements();
4758 auto *Vec = Operands[0];
4759 auto *VecTy = cast<VectorType>(Vec->getType());
4760
4761 ElementCount ResultEC =
4762 VecTy->getElementCount().divideCoefficientBy(NumResults);
4763
4764 if (auto *EltC = Vec->getSplatValue()) {
4765 auto *ResultVec = ConstantVector::getSplat(ResultEC, EltC);
4766 SmallVector<Constant *, 8> Results(NumResults, ResultVec);
4767 return ConstantStruct::get(StTy, Results);
4768 }
4769
4770 if (!ResultEC.isFixed())
4771 return nullptr;
4772
4773 unsigned NumElements = ResultEC.getFixedValue();
4775 SmallVector<Constant *> Elements(NumElements);
4776 for (unsigned I = 0; I != NumResults; ++I) {
4777 for (unsigned J = 0; J != NumElements; ++J) {
4778 Constant *Elt = Vec->getAggregateElement(J * NumResults + I);
4779 if (!Elt)
4780 return nullptr;
4781 Elements[J] = Elt;
4782 }
4783 Results[I] = ConstantVector::get(Elements);
4784 }
4785 return ConstantStruct::get(StTy, Results);
4786 }
4787 default:
4788 // TODO: Constant folding of vector intrinsics that fall through here does
4789 // not work (e.g. overflow intrinsics)
4790 return ConstantFoldScalarCall(Name, IntrinsicID, StTy, Operands, TLI, Call);
4791 }
4792
4793 return nullptr;
4794}
4795
4796} // end anonymous namespace
4797
4800 const DataLayout &DL,
4801 const Function *CtxF) {
4802 // In the absence of CtxF, assume strictfp conservatively.
4803 if (!canConstantFoldIntrinsic(ID, CtxF ? CtxF->isStrictFP() : true) ||
4806 Ty, ArrayRef<Value *>((Value *const *)Ops.data(), Ops.size()))))
4807 return nullptr;
4808 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
4809 return ConstantFoldFixedVectorCall("", ID, FVTy, Ops, DL);
4810 return ConstantFoldScalarCall("", ID, Ty, Ops);
4811}
4812
4815 const TargetLibraryInfo *TLI,
4816 bool AllowNonDeterministic) {
4817 if (Call->isNoBuiltin())
4818 return nullptr;
4819 if (!F->hasName())
4820 return nullptr;
4821
4822 // If this is not an intrinsic and not recognized as a library call, bail out.
4823 Intrinsic::ID IID = F->getIntrinsicID();
4824 if (IID == Intrinsic::not_intrinsic) {
4825 if (!TLI)
4826 return nullptr;
4827 if (TLI->getLibFunc(*F) == NotLibFunc)
4828 return nullptr;
4829 }
4830
4831 // Conservatively assume that floating-point libcalls may be
4832 // non-deterministic.
4833 Type *Ty = F->getReturnType();
4834 if (!AllowNonDeterministic && Ty->isFPOrFPVectorTy())
4835 return nullptr;
4836
4837 StringRef Name = F->getName();
4838 if (auto *FVTy = dyn_cast<FixedVectorType>(Ty))
4839 return ConstantFoldFixedVectorCall(
4840 Name, IID, FVTy, Operands, F->getDataLayout(), TLI, Call);
4841
4842 if (auto *SVTy = dyn_cast<ScalableVectorType>(Ty))
4843 return ConstantFoldScalableVectorCall(
4844 Name, IID, SVTy, Operands, F->getDataLayout(), TLI, Call);
4845
4846 if (auto *StTy = dyn_cast<StructType>(Ty))
4847 return ConstantFoldStructCall(Name, IID, StTy, Operands,
4848 F->getDataLayout(), TLI, Call);
4849
4850 // TODO: If this is a library function, we already discovered that above,
4851 // so we should pass the LibFunc, not the name (and it might be better
4852 // still to separate intrinsic handling from libcalls).
4853 return ConstantFoldScalarCall(Name, IID, Ty, Operands, TLI, Call);
4854}
4855
4857 const TargetLibraryInfo *TLI) {
4858 // FIXME: Refactor this code; this duplicates logic in LibCallsShrinkWrap
4859 // (and to some extent ConstantFoldScalarCall).
4860 if (Call->isNoBuiltin() || Call->isStrictFP())
4861 return false;
4862 Function *F = Call->getCalledFunction();
4863 if (!F)
4864 return false;
4865
4866 if (!TLI)
4867 return false;
4868
4869 LibFunc Func = TLI->getLibFunc(*F);
4870 if (Func == NotLibFunc)
4871 return false;
4872
4873 if (Call->arg_size() == 1) {
4874 if (ConstantFP *OpC = dyn_cast<ConstantFP>(Call->getArgOperand(0))) {
4875 const APFloat &Op = OpC->getValueAPF();
4876 switch (Func) {
4877 case LibFunc_logl:
4878 case LibFunc_log:
4879 case LibFunc_logf:
4880 case LibFunc_log2l:
4881 case LibFunc_log2:
4882 case LibFunc_log2f:
4883 case LibFunc_log10l:
4884 case LibFunc_log10:
4885 case LibFunc_log10f:
4886 return Op.isNaN() || (!Op.isZero() && !Op.isNegative());
4887
4888 case LibFunc_ilogb:
4889 return !Op.isNaN() && !Op.isZero() && !Op.isInfinity();
4890
4891 case LibFunc_expl:
4892 case LibFunc_exp:
4893 case LibFunc_expf:
4894 // FIXME: These boundaries are slightly conservative.
4895 if (OpC->getType()->isDoubleTy())
4896 return !(Op < APFloat(-745.0) || Op > APFloat(709.0));
4897 if (OpC->getType()->isFloatTy())
4898 return !(Op < APFloat(-103.0f) || Op > APFloat(88.0f));
4899 break;
4900
4901 case LibFunc_exp2l:
4902 case LibFunc_exp2:
4903 case LibFunc_exp2f:
4904 // FIXME: These boundaries are slightly conservative.
4905 if (OpC->getType()->isDoubleTy())
4906 return !(Op < APFloat(-1074.0) || Op > APFloat(1023.0));
4907 if (OpC->getType()->isFloatTy())
4908 return !(Op < APFloat(-149.0f) || Op > APFloat(127.0f));
4909 break;
4910
4911 case LibFunc_sinl:
4912 case LibFunc_sin:
4913 case LibFunc_sinf:
4914 case LibFunc_cosl:
4915 case LibFunc_cos:
4916 case LibFunc_cosf:
4917 return !Op.isInfinity();
4918
4919 case LibFunc_tanl:
4920 case LibFunc_tan:
4921 case LibFunc_tanf: {
4922 // FIXME: Stop using the host math library.
4923 // FIXME: The computation isn't done in the right precision.
4924 Type *Ty = OpC->getType();
4925 if (Ty->isDoubleTy() || Ty->isFloatTy() || Ty->isHalfTy())
4926 return ConstantFoldFP(tan, OpC->getValueAPF(), Ty) != nullptr;
4927 break;
4928 }
4929
4930 case LibFunc_atan:
4931 case LibFunc_atanf:
4932 case LibFunc_atanl:
4933 // Per POSIX, this MAY fail if Op is denormal. We choose not failing.
4934 return true;
4935
4936 case LibFunc_asinl:
4937 case LibFunc_asin:
4938 case LibFunc_asinf:
4939 case LibFunc_acosl:
4940 case LibFunc_acos:
4941 case LibFunc_acosf:
4942 return !(Op < APFloat::getOne(Op.getSemantics(), true) ||
4943 Op > APFloat::getOne(Op.getSemantics()));
4944
4945 case LibFunc_sinh:
4946 case LibFunc_cosh:
4947 case LibFunc_sinhf:
4948 case LibFunc_coshf:
4949 case LibFunc_sinhl:
4950 case LibFunc_coshl:
4951 // FIXME: These boundaries are slightly conservative.
4952 if (OpC->getType()->isDoubleTy())
4953 return !(Op < APFloat(-710.0) || Op > APFloat(710.0));
4954 if (OpC->getType()->isFloatTy())
4955 return !(Op < APFloat(-89.0f) || Op > APFloat(89.0f));
4956 break;
4957
4958 case LibFunc_sqrtl:
4959 case LibFunc_sqrt:
4960 case LibFunc_sqrtf:
4961 return Op.isNaN() || Op.isZero() || !Op.isNegative();
4962
4963 // FIXME: Add more functions: sqrt_finite, atanh, expm1, log1p,
4964 // maybe others?
4965 default:
4966 break;
4967 }
4968 }
4969 }
4970
4971 if (Call->arg_size() == 2) {
4972 ConstantFP *Op0C = dyn_cast<ConstantFP>(Call->getArgOperand(0));
4973 ConstantFP *Op1C = dyn_cast<ConstantFP>(Call->getArgOperand(1));
4974 if (Op0C && Op1C) {
4975 const APFloat &Op0 = Op0C->getValueAPF();
4976 const APFloat &Op1 = Op1C->getValueAPF();
4977
4978 switch (Func) {
4979 case LibFunc_powl:
4980 case LibFunc_pow:
4981 case LibFunc_powf: {
4982 // FIXME: Stop using the host math library.
4983 // FIXME: The computation isn't done in the right precision.
4984 Type *Ty = Op0C->getType();
4985 if (Ty->isDoubleTy() || Ty->isFloatTy() || Ty->isHalfTy()) {
4986 if (Ty == Op1C->getType())
4987 return ConstantFoldBinaryFP(pow, Op0, Op1, Ty) != nullptr;
4988 }
4989 break;
4990 }
4991
4992 case LibFunc_fmodl:
4993 case LibFunc_fmod:
4994 case LibFunc_fmodf:
4995 case LibFunc_remainderl:
4996 case LibFunc_remainder:
4997 case LibFunc_remainderf:
4998 return Op0.isNaN() || Op1.isNaN() ||
4999 (!Op0.isInfinity() && !Op1.isZero());
5000
5001 case LibFunc_atan2:
5002 case LibFunc_atan2f:
5003 case LibFunc_atan2l:
5004 // Although IEEE-754 says atan2(+/-0.0, +/-0.0) are well-defined, and
5005 // GLIBC and MSVC do not appear to raise an error on those, we
5006 // cannot rely on that behavior. POSIX and C11 say that a domain error
5007 // may occur, so allow for that possibility.
5008 return !Op0.isZero() || !Op1.isZero();
5009
5010 case LibFunc_nextafter:
5011 case LibFunc_nextafterf:
5012 case LibFunc_nextafterl:
5013 case LibFunc_nexttoward:
5014 case LibFunc_nexttowardf:
5015 case LibFunc_nexttowardl: {
5016 return ConstantFoldNextToward(Op0, Op1, F->getReturnType()) != nullptr;
5017 }
5018 default:
5019 break;
5020 }
5021 }
5022 }
5023
5024 return false;
5025}
5026
5028 unsigned CastOp, const DataLayout &DL,
5029 PreservedCastFlags *Flags) {
5030 switch (CastOp) {
5031 case Instruction::BitCast:
5032 // Bitcast is always lossless.
5033 return ConstantFoldCastOperand(Instruction::BitCast, C, InvCastTo, DL);
5034 case Instruction::Trunc: {
5035 auto *ZExtC = ConstantFoldCastOperand(Instruction::ZExt, C, InvCastTo, DL);
5036 if (Flags) {
5037 // Truncation back on ZExt value is always NUW.
5038 Flags->NUW = true;
5039 // Test positivity of C.
5040 auto *SExtC =
5041 ConstantFoldCastOperand(Instruction::SExt, C, InvCastTo, DL);
5042 Flags->NSW = ZExtC == SExtC;
5043 }
5044 return ZExtC;
5045 }
5046 case Instruction::SExt:
5047 case Instruction::ZExt: {
5048 auto *InvC = ConstantExpr::getTrunc(C, InvCastTo);
5049 auto *CastInvC = ConstantFoldCastOperand(CastOp, InvC, C->getType(), DL);
5050 // Must satisfy CastOp(InvC) == C.
5051 if (!CastInvC || CastInvC != C)
5052 return nullptr;
5053 if (Flags && CastOp == Instruction::ZExt) {
5054 auto *SExtInvC =
5055 ConstantFoldCastOperand(Instruction::SExt, InvC, C->getType(), DL);
5056 // Test positivity of InvC.
5057 Flags->NNeg = CastInvC == SExtInvC;
5058 }
5059 return InvC;
5060 }
5061 case Instruction::FPExt: {
5062 Constant *InvC =
5063 ConstantFoldCastOperand(Instruction::FPTrunc, C, InvCastTo, DL);
5064 if (InvC) {
5065 Constant *CastInvC =
5066 ConstantFoldCastOperand(CastOp, InvC, C->getType(), DL);
5067 if (CastInvC == C)
5068 return InvC;
5069 }
5070 return nullptr;
5071 }
5072 default:
5073 return nullptr;
5074 }
5075}
5076
5078 const DataLayout &DL,
5079 PreservedCastFlags *Flags) {
5080 return getLosslessInvCast(C, DestTy, Instruction::ZExt, DL, Flags);
5081}
5082
5084 const DataLayout &DL,
5085 PreservedCastFlags *Flags) {
5086 return getLosslessInvCast(C, DestTy, Instruction::SExt, DL, Flags);
5087}
5088
5089void TargetFolder::anchor() {}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
constexpr LLT S1
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static Constant * FoldBitCast(Constant *V, Type *DestTy)
static ConstantFP * flushDenormalConstant(Type *Ty, const APFloat &APF, DenormalMode::DenormalModeKind Mode)
Constant * getConstantAtOffset(Constant *Base, APInt Offset, const DataLayout &DL)
If this Offset points exactly to the start of an aggregate element, return that element,...
static cl::opt< bool > DisableFPCallFolding("disable-fp-call-folding", cl::desc("Disable constant-folding of FP intrinsics and libcalls."), cl::init(false), cl::Hidden)
static bool canConstantFoldIntrinsic(Intrinsic::ID ID, bool IsStrictFP)
Returns true if the intrinsic can be constant folded, given IsStrictFP.
static ConstantFP * flushDenormalConstantFP(ConstantFP *CFP, const Function *CtxF, bool IsOutput)
static bool anyTypeContainsFP(Type *RetTy, ArrayRef< Value * > Ops)
Given a function's return type and its operands, determine if any of them of of floating-point type.
static DenormalMode getInstrDenormalMode(const Function *CtxF, Type *Ty)
Return the denormal mode that can be assumed when executing a floating point operation at CtxI.
This file contains the declarations for the subclasses of Constant, which represent the different fla...
This file defines the DenseMap class.
Hexagon Common GEP
amode Optimize addressing mode
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static bool InRange(int64_t Value, unsigned short Shift, int LBound, int HBound)
This file contains the definitions of the enumerations and flags associated with NVVM Intrinsics,...
if(PassOpts->AAPipeline)
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
SI Fold Operands
This file contains some templates that are useful if you are working with the STL at all.
This file implements the SmallBitVector class.
This file defines the SmallVector class.
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
The Input class is used to parse a yaml document into in-memory structs and vectors.
cmpResult
IEEE-754R 5.11: Floating Point Comparison Relations.
Definition APFloat.h:351
static constexpr roundingMode rmTowardZero
Definition APFloat.h:365
llvm::RoundingMode roundingMode
IEEE-754R 4.3: Rounding-direction attributes.
Definition APFloat.h:359
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmTowardNegative
Definition APFloat.h:364
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static constexpr roundingMode rmTowardPositive
Definition APFloat.h:363
static constexpr roundingMode rmNearestTiesToAway
Definition APFloat.h:366
opStatus
IEEE-754R 7: Default exception handling.
Definition APFloat.h:377
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
Definition APFloat.h:1224
opStatus divide(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1312
void copySign(const APFloat &RHS)
Definition APFloat.h:1406
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Definition APFloat.cpp:6034
opStatus subtract(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1294
bool isNegative() const
Definition APFloat.h:1583
LLVM_ABI double convertToDouble() const
Converts this APFloat to host double value.
Definition APFloat.cpp:6093
bool isPosInfinity() const
Definition APFloat.h:1596
bool isNormal() const
Definition APFloat.h:1587
bool isDenormal() const
Definition APFloat.h:1584
opStatus add(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1285
const fltSemantics & getSemantics() const
Definition APFloat.h:1591
bool isNonZero() const
Definition APFloat.h:1592
bool isFinite() const
Definition APFloat.h:1588
bool isNaN() const
Definition APFloat.h:1581
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
opStatus multiply(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1303
bool isSignaling() const
Definition APFloat.h:1585
opStatus fusedMultiplyAdd(const APFloat &Multiplicand, const APFloat &Addend, roundingMode RM)
Definition APFloat.h:1339
bool isZero() const
Definition APFloat.h:1579
opStatus convertToInteger(MutableArrayRef< integerPart > Input, unsigned int Width, bool IsSigned, roundingMode RM, bool *IsExact) const
Definition APFloat.h:1436
opStatus mod(const APFloat &RHS)
Definition APFloat.h:1330
bool isNegInfinity() const
Definition APFloat.h:1597
opStatus roundToIntegral(roundingMode RM)
Definition APFloat.h:1352
void changeSign()
Definition APFloat.h:1401
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1183
bool isInfinity() const
Definition APFloat.h:1580
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt umul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:2009
LLVM_ABI APInt usub_sat(const APInt &RHS) const
Definition APInt.cpp:2093
bool isMinSignedValue() const
Determine if this is the smallest signed value.
Definition APInt.h:419
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
LLVM_ABI uint64_t extractBitsAsZExtValue(unsigned numBits, unsigned bitPosition) const
Definition APInt.cpp:517
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:202
APInt abs() const
Get the absolute value.
Definition APInt.h:1815
LLVM_ABI APInt sadd_sat(const APInt &RHS) const
Definition APInt.cpp:2064
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1205
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1986
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1186
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1695
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1115
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
Definition APInt.h:205
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1966
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1973
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1618
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:215
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1086
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2074
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:829
LLVM_ABI APInt smul_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1998
LLVM_ABI APInt sext(unsigned width) const
Sign extend to a new width.
Definition APInt.cpp:1030
APInt shl(unsigned shiftAmt) const
Left-shift function.
Definition APInt.h:875
bool slt(const APInt &RHS) const
Signed less than comparison.
Definition APInt.h:1134
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
LLVM_ABI APInt extractBits(unsigned numBits, unsigned bitPosition) const
Return an APInt with the extracted bits [bitPosition,bitPosition+numBits).
Definition APInt.cpp:478
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1979
bool isOne() const
Determine if this is a value of 1.
Definition APInt.h:385
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
LLVM_ABI APInt ssub_sat(const APInt &RHS) const
Definition APInt.cpp:2083
An arbitrary precision integer that knows its signedness.
Definition APSInt.h:24
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI unsigned isEliminableCastPair(Instruction::CastOps firstOpcode, Instruction::CastOps secondOpcode, Type *SrcTy, Type *MidTy, Type *DstTy, const DataLayout *DL)
Determine how a pair of casts can be eliminated, if they can be at all.
static LLVM_ABI bool castIsValid(Instruction::CastOps op, Type *SrcTy, Type *DstTy)
This method can be used to determine if a cast from SrcTy to DstTy using Opcode op is valid or not.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
bool isSigned() const
Definition InstrTypes.h:993
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
static bool isFPPredicate(Predicate P)
Definition InstrTypes.h:833
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
Definition Constants.h:878
static LLVM_ABI Constant * getIntToPtr(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getExtractElement(Constant *Vec, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_ABI bool isDesirableCastOp(unsigned Opcode)
Whether creating a constant expression for this cast is desirable.
static LLVM_ABI Constant * getCast(unsigned ops, Constant *C, Type *Ty, bool OnlyIfReduced=false)
Convenience function for getting a Cast operation.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static Constant * getPtrAdd(Constant *Ptr, Constant *Offset, GEPNoWrapFlags NW=GEPNoWrapFlags::none(), std::optional< ConstantRange > InRange=std::nullopt, Type *OnlyIfReduced=nullptr)
Create a getelementptr i8, ptr, offset constant expression.
Definition Constants.h:1518
static LLVM_ABI Constant * getInsertElement(Constant *Vec, Constant *Elt, Constant *Idx, Type *OnlyIfReducedTy=nullptr)
static LLVM_SUPPRESS_DEPRECATED_DECLARATIONS_PUSH Constant * getGetElementPtr(Type *Ty, Constant *C, ArrayRef< Constant * > IdxList, GEPNoWrapFlags NW=GEPNoWrapFlags::none(), std::optional< ConstantRange > InRange=std::nullopt, Type *OnlyIfReducedTy=nullptr)
Getelementptr form.
Definition Constants.h:1477
static LLVM_ABI Constant * getShuffleVector(Constant *V1, Constant *V2, ArrayRef< int > Mask, Type *OnlyIfReducedTy=nullptr)
static bool isSupportedGetElementPtr(const Type *SrcElemTy)
Whether creating a constant expression for this getelementptr type is supported.
Definition Constants.h:1624
static LLVM_ABI Constant * get(unsigned Opcode, Constant *C1, Constant *C2, unsigned Flags=0, Type *OnlyIfReducedTy=nullptr)
get - Return a binary or shift operator constant expression, folding if possible.
static LLVM_ABI bool isDesirableBinOp(unsigned Opcode)
Whether creating a constant expression for this binary operator is desirable.
static LLVM_ABI Constant * getBitCast(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getTrunc(Constant *C, Type *Ty, bool OnlyIfReduced=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
const APFloat & getValueAPF() const
Definition Constants.h:463
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getNaN(Type *Ty, bool Negative=false, uint64_t Payload=0)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
int64_t getSExtValue() const
Return the constant as a 64-bit integer value after it has been sign extended as appropriate for the ...
Definition Constants.h:174
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
LLVM_ABI Constant * getSplatValue(bool AllowPoison=false) const
If all elements of the vector constant have the same value, return that value.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
Constrained floating point compare intrinsics.
This is the common base class for constrained floating point intrinsics.
LLVM_ABI std::optional< fp::ExceptionBehavior > getExceptionBehavior() const
LLVM_ABI std::optional< RoundingMode > getRoundingMode() const
Wrapper for a function that represents a value that functionally represents the original function.
Definition Constants.h:1143
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:767
iterator end()
Definition DenseMap.h:687
std::pair< iterator, bool > insert(const std::pair< KeyT, ValueT > &KV)
Definition DenseMap.h:828
static LLVM_ABI bool compare(const APFloat &LHS, const APFloat &RHS, FCmpInst::Predicate Pred)
Return result of LHS Pred RHS comparison.
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
Definition Function.cpp:810
bool isStrictFP() const
Determine if the function has strict floating point sematics.
Definition Function.h:637
Represents flags for the getelementptr instruction/expression.
static GEPNoWrapFlags inBounds()
GEPNoWrapFlags withoutNoUnsignedSignedWrap() const
static GEPNoWrapFlags noUnsignedWrap()
bool hasNoUnsignedSignedWrap() const
bool isInBounds() const
static LLVM_ABI Type * getIndexedType(Type *Ty, ArrayRef< Value * > IdxList)
Returns the result type of a getelementptr with the given source element type and indexes.
PointerType * getType() const
Global values are always pointers.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this global belongs to.
Definition Globals.cpp:205
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
bool isConstant() const
If the value is a global constant, its value is immutable throughout the runtime execution of the pro...
bool hasDefinitiveInitializer() const
hasDefinitiveInitializer - Whether the global variable has an initializer, and any other instances of...
static LLVM_ABI bool compare(const APInt &LHS, const APInt &RHS, ICmpInst::Predicate Pred)
Return result of LHS Pred RHS comparison.
Predicate getSignedPredicate() const
For example, EQ->EQ, SLE->SLE, UGT->SGT, etc.
bool isEquality() const
Return true if this predicate is either EQ or NE.
bool isCast() const
bool isBinaryOp() const
bool isUnaryOp() const
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
static APInt getSaturationPoint(Intrinsic::ID ID, unsigned numBits)
Min/max intrinsics are monotonic, they operate on a fixed-bitwidth values, so there is a certain thre...
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Class to represent scalable SIMD vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
iterator_range< const_set_bits_iterator > set_bits() const
void push_back(const T &Elt)
pointer data()
Return a pointer to the vector's buffer, even if empty().
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
Used to lazily calculate structure layout information for a target machine, based on the DataLayout s...
Definition DataLayout.h:743
LLVM_ABI unsigned getElementContainingOffset(uint64_t FixedOffset) const
Given a valid byte offset into the structure, returns the structure index that contains it.
TypeSize getElementOffset(unsigned Idx) const
Definition DataLayout.h:774
Class to represent struct types.
unsigned getNumElements() const
Random access to the elements.
Provides information about what library functions are available for the current target.
bool has(LibFunc F) const
Tests whether a library function is available.
LibFunc getLibFunc(StringRef funcName) const
Searches for a particular function name.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isByteTy() const
True if this is an instance of ByteType.
Definition Type.h:237
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
bool isSized() const
Return true if it makes sense to take the size of this type.
Definition Type.h:321
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
bool isByteOrByteVectorTy() const
Return true if this is a byte type or a vector of byte types.
Definition Type.h:243
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:298
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
static LLVM_ABI IntegerType * getInt1Ty(LLVMContext &C)
Definition Type.cpp:296
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:280
bool isX86_AMXTy() const
Return true if this is X86 AMX.
Definition Type.h:202
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
Type * getContainedType(unsigned i) const
This method is used to implement the type iterator (defined at the end of the file).
Definition Type.h:392
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:96
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
LLVM_ABI const Value * stripAndAccumulateConstantOffsets(const DataLayout &DL, APInt &Offset, bool AllowNonInbounds, bool AllowInvariantGroup=false, function_ref< bool(Value &Value, APInt &Offset)> ExternalAnalysis=nullptr, bool LookThroughIntToPtr=false) const
Accumulate the constant offset this value has compared to a base pointer.
LLVM_ABI uint64_t getPointerDereferenceableBytes(const DataLayout &DL, bool &CanBeNull, bool *CanBeFreed) const
Returns the number of bytes known to be dereferenceable for the pointer value.
Definition Value.cpp:918
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isScalable() const
Returns whether the quantity is scaled by a runtime quantity (vscale).
Definition TypeSize.h:168
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
static constexpr bool isKnownGE(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:237
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI APInt mulhu(const APInt &C1, const APInt &C2)
Performs (2*N)-bit multiplication on zero-extended operands.
Definition APInt.cpp:3165
LLVM_ABI APInt pext(const APInt &Val, const APInt &Mask)
Perform a "compress" operation, also known as pext or bext.
Definition APInt.cpp:3245
LLVM_ABI APInt mulhs(const APInt &C1, const APInt &C2)
Performs (2*N)-bit multiplication on sign-extended operands.
Definition APInt.cpp:3157
const APInt & smin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be signed.
Definition APInt.h:2274
const APInt & smax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be signed.
Definition APInt.h:2279
LLVM_ABI APInt clmul(const APInt &LHS, const APInt &RHS)
Perform a carry-less multiply, also known as XOR multiplication, and return low-bits.
Definition APInt.cpp:3225
const APInt & umin(const APInt &A, const APInt &B)
Determine the smaller of two APInts considered to be unsigned.
Definition APInt.h:2284
LLVM_ABI APInt pdep(const APInt &Val, const APInt &Mask)
Perform an "expand" operation, also known as pdep or bdep.
Definition APInt.cpp:3255
const APInt & umax(const APInt &A, const APInt &B)
Determine the larger of two APInts considered to be unsigned.
Definition APInt.h:2289
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ CE
Windows NT (Windows on ARM)
Definition MCAsmInfo.h:51
initializer< Ty > init(const Ty &Val)
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:454
static constexpr cmpResult cmpEqual
Definition APFloat.h:462
@ ebStrict
This corresponds to "fpexcept.strict".
Definition FPEnv.h:42
@ ebIgnore
This corresponds to "fpexcept.ignore".
Definition FPEnv.h:40
constexpr double pi
APFloat::roundingMode GetRoundingModeFromImmArg(const Value *ImmArgVal)
APFloat::roundingMode GetFMARoundingMode(Intrinsic::ID IntrinsicID)
DenormalMode GetNVVMDenormMode(bool ShouldFTZ)
bool FPToIntegerIntrinsicNaNZero(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFDivRoundingMode(Intrinsic::ID IntrinsicID)
bool FPToIntegerIntrinsicResultIsSigned(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetFPToIntegerRoundingMode(Intrinsic::ID IntrinsicID)
bool RCPShouldFTZ(Intrinsic::ID IntrinsicID)
bool FPToIntegerIntrinsicShouldFTZ(Intrinsic::ID IntrinsicID)
bool FDivShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMinFMaxIsXorSignAbs(Intrinsic::ID IntrinsicID)
bool UnaryMathIntrinsicShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMinFMaxShouldFTZ(Intrinsic::ID IntrinsicID)
bool FMAShouldFTZ(Intrinsic::ID IntrinsicID)
APFloat::roundingMode GetRCPRoundingMode(Intrinsic::ID IntrinsicID)
bool FMinFMaxPropagatesNaNs(Intrinsic::ID IntrinsicID)
NodeAddr< FuncNode * > Func
Definition RDFGraph.h:393
LLVM_ABI std::error_code status(const Twine &path, file_status &result, bool follow=true)
Get file status as if by POSIX stat().
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI Constant * ConstantFoldLoadThroughBitcast(Constant *C, Type *DestTy, const DataLayout &DL)
ConstantFoldLoadThroughBitcast - try to cast constant to destination type returning null if unsuccess...
static double log2(double V)
LLVM_ABI Constant * ConstantFoldSelectInstruction(Constant *Cond, Constant *V1, Constant *V2)
Attempt to constant fold a select instruction with the specified operands.
LLVM_ABI Constant * ConstantFoldFPInstOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL, const Instruction *I, bool AllowNonDeterministic=true)
Attempt to constant fold a floating point binary operation with the specified operands,...
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
unsigned getPointerAddressSpace(const Type *T)
Definition SPIRVUtils.h:395
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI Constant * ConstantFoldInstruction(const Instruction *I, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldInstruction - Try to constant fold the specified instruction.
APFloat abs(APFloat X)
Returns the absolute value of the argument.
Definition APFloat.h:1721
LLVM_ABI Constant * ConstantFoldCompareInstruction(CmpInst::Predicate Predicate, Constant *C1, Constant *C2)
LLVM_ABI Constant * ConstantFoldUnaryInstruction(unsigned Opcode, Constant *V)
LLVM_ABI bool IsConstantOffsetFromGlobal(Constant *C, GlobalValue *&GV, APInt &Offset, const DataLayout &DL, DSOLocalEquivalent **DSOEquiv=nullptr)
If this constant is a constant offset from a global, return the global and the constant.
LLVM_ABI bool isMathLibCallNoop(const CallBase *Call, const TargetLibraryInfo *TLI)
Check whether the given call has no side-effects.
LLVM_ABI Constant * ReadByteArrayFromGlobal(const GlobalVariable *GV, uint64_t Offset)
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1801
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
int ilogb(const APFloat &Arg)
Returns the exponent of the internal representation of the APFloat.
Definition APFloat.h:1692
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
LLVM_ABI Constant * ConstantFoldCall(const CallBase *Call, Function *F, ArrayRef< Constant * > Operands, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldCall - Attempt to constant fold a call to the specified function with the specified argum...
LLVM_ABI bool canConstantFoldCallTo(const CallBase *Call, const Function *F, const TargetLibraryInfo *TLI=nullptr)
canConstantFoldCallTo - Return true if its even possible to fold a call to the specified function.
APFloat frexp(const APFloat &X, int &Exp, APFloat::roundingMode RM)
Equivalent of C standard library function.
Definition APFloat.h:1713
LLVM_ABI Constant * ConstantFoldIntrinsic(Intrinsic::ID ID, ArrayRef< Constant * > Ops, Type *Ty, const DataLayout &DL, const Function *CtxF=nullptr)
LLVM_ABI Constant * ConstantFoldExtractValueInstruction(Constant *Agg, ArrayRef< unsigned > Idxs)
Attempt to constant fold an extractvalue instruction with the specified operands and indices.
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Function *CtxF=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
LLVM_ABI Constant * ConstantFoldConstant(const Constant *C, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr)
ConstantFoldConstant - Fold the constant using the specified DataLayout.
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1756
LLVM_ABI Constant * ConstantFoldLoadFromUniformValue(Constant *C, Type *Ty, const DataLayout &DL)
If C is a uniform value where all bits are the same (either all zero, all ones, all undef or all pois...
LLVM_ABI Constant * ConstantFoldUnaryOpOperand(unsigned Opcode, Constant *Op, const DataLayout &DL)
Attempt to constant fold a unary operation with the specified operand.
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_READONLY LLVM_ABI std::optional< APFloat > exp(const APFloat &X, RoundingMode RM=APFloat::rmNearestTiesToEven, APFloat::opStatus *Status=nullptr)
Implement IEEE 754-2019 exp functions.
Definition APFloat.cpp:6253
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1787
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1701
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI Constant * ConstantFoldCastOperand(unsigned Opcode, Constant *C, Type *DestTy, const DataLayout &DL)
Attempt to constant fold a cast with the specified operand.
LLVM_ABI Constant * ConstantFoldLoadFromConst(Constant *C, Type *Ty, const APInt &Offset, const DataLayout &DL)
Extract value of C at the given Offset reinterpreted as Ty.
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool intrinsicPropagatesPoison(Intrinsic::ID IID)
Return whether this intrinsic propagates poison for all operands.
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI Constant * FlushFPConstant(Constant *Operand, const Function *CtxF, bool IsOutput)
Attempt to flush float point constant according to denormal mode set in the instruction's parent func...
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1737
@ Sub
Subtraction of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
IntPtrTy
Definition InstrProf.h:82
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ NearestTiesToEven
roundTiesToEven.
@ Dynamic
Denotes mode unknown at compile time.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
Definition STLExtras.h:2182
LLVM_ABI Constant * ConstantFoldCastInstruction(unsigned opcode, Constant *V, Type *DestTy)
LLVM_ABI Constant * ConstantFoldInsertValueInstruction(Constant *Agg, Constant *Val, ArrayRef< unsigned > Idxs)
Attempt to constant fold an insertvalue instruction with the specified operands and indices.
LLVM_ABI Constant * ConstantFoldLoadFromConstPtr(Constant *C, Type *Ty, APInt Offset, const DataLayout &DL)
Return the value that a load from C with offset Offset would produce if it is constant and determinab...
constexpr uint32_t calculateReflectedCRC32(uint32_t Crc, uint64_t Data, unsigned DataBytes, uint32_t Poly)
Definition CRC.h:34
LLVM_ABI Constant * ConstantFoldInstOperands(const Instruction *I, ArrayRef< Constant * > Ops, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldInstOperands - Attempt to constant fold an instruction with the specified operands.
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1774
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1814
LLVM_ABI Constant * ConstantFoldIntegerCast(Constant *C, Type *DestTy, bool IsSigned, const DataLayout &DL)
Constant fold a zext, sext or trunc, depending on IsSigned and whether the DestTy is wider or narrowe...
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
constexpr detail::IsaCheckPredicate< Types... > IsaPred
Function object wrapper for the llvm::isa type check.
Definition Casting.h:866
LLVM_ABI Constant * ConstantFoldBinaryInstruction(unsigned Opcode, Constant *V1, Constant *V2)
Represent subnormal handling kind for floating point instruction inputs and outputs.
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
DenormalModeKind
Represent handled modes for denormal (aka subnormal) modes in the floating point environment.
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
@ PositiveZero
Denormals are flushed to positive zero.
@ Dynamic
Denormals have unknown treatment.
@ IEEE
IEEE-754 denormal numbers preserved.
DenormalModeKind Output
Denormal flushing mode for floating point instruction results in the default floating point environme...
static constexpr DenormalMode getDynamic()
static constexpr DenormalMode getIEEE()
bool isConstant() const
Returns true if we know the value of all bits.
Definition KnownBits.h:54
const APInt & getConstant() const
Returns the value when all bits have a known value.
Definition KnownBits.h:58