LLVM 24.0.0git
InstCombineCalls.cpp
Go to the documentation of this file.
1//===- InstCombineCalls.cpp -----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the visitCall, visitInvoke, and visitCallBr functions.
10//
11//===----------------------------------------------------------------------===//
12
13#include "InstCombineInternal.h"
14#include "llvm/ADT/APFloat.h"
15#include "llvm/ADT/APInt.h"
16#include "llvm/ADT/APSInt.h"
17#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/Bitset.h"
22#include "llvm/ADT/Statistic.h"
28#include "llvm/Analysis/Loads.h"
33#include "llvm/IR/Attributes.h"
34#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
39#include "llvm/IR/DebugInfo.h"
41#include "llvm/IR/Function.h"
43#include "llvm/IR/InlineAsm.h"
44#include "llvm/IR/InstrTypes.h"
45#include "llvm/IR/Instruction.h"
48#include "llvm/IR/Intrinsics.h"
49#include "llvm/IR/IntrinsicsAArch64.h"
50#include "llvm/IR/IntrinsicsAMDGPU.h"
51#include "llvm/IR/IntrinsicsARM.h"
52#include "llvm/IR/IntrinsicsHexagon.h"
53#include "llvm/IR/LLVMContext.h"
54#include "llvm/IR/Metadata.h"
56#include "llvm/IR/Statepoint.h"
57#include "llvm/IR/Type.h"
58#include "llvm/IR/User.h"
59#include "llvm/IR/Value.h"
60#include "llvm/IR/ValueHandle.h"
65#include "llvm/Support/Debug.h"
76#include <algorithm>
77#include <cassert>
78#include <cstdint>
79#include <optional>
80#include <utility>
81#include <vector>
82
83#define DEBUG_TYPE "instcombine"
85
86using namespace llvm;
87using namespace PatternMatch;
88
89STATISTIC(NumSimplified, "Number of library calls simplified");
90
92 "instcombine-guard-widening-window",
93 cl::init(3),
94 cl::desc("How wide an instruction window to bypass looking for "
95 "another guard"));
96
97/// Return the specified type promoted as it would be to pass though a va_arg
98/// area.
100 if (IntegerType* ITy = dyn_cast<IntegerType>(Ty)) {
101 if (ITy->getBitWidth() < 32)
102 return Type::getInt32Ty(Ty->getContext());
103 }
104 return Ty;
105}
106
107/// Recognize a memcpy/memmove from a trivially otherwise unused alloca.
108/// TODO: This should probably be integrated with visitAllocSites, but that
109/// requires a deeper change to allow either unread or unwritten objects.
111 auto *Src = MI->getRawSource();
112 while (isa<GetElementPtrInst>(Src)) {
113 if (!Src->hasOneUse())
114 return false;
115 Src = cast<Instruction>(Src)->getOperand(0);
116 }
117 return isa<AllocaInst>(Src) && Src->hasOneUse();
118}
119
121 Align DstAlign = getKnownAlignment(MI->getRawDest(), DL, MI, &AC, &DT);
122 MaybeAlign CopyDstAlign = MI->getDestAlign();
123 if (!CopyDstAlign || *CopyDstAlign < DstAlign) {
124 MI->setDestAlignment(DstAlign);
125 return MI;
126 }
127
128 Align SrcAlign = getKnownAlignment(MI->getRawSource(), DL, MI, &AC, &DT);
129 MaybeAlign CopySrcAlign = MI->getSourceAlign();
130 if (!CopySrcAlign || *CopySrcAlign < SrcAlign) {
131 MI->setSourceAlignment(SrcAlign);
132 return MI;
133 }
134
135 // If we have a store to a location which is known constant, we can conclude
136 // that the store must be storing the constant value (else the memory
137 // wouldn't be constant), and this must be a noop.
138 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
139 // Set the size of the copy to 0, it will be deleted on the next iteration.
140 MI->setLength((uint64_t)0);
141 return MI;
142 }
143
144 // If the source is provably undef, the memcpy/memmove doesn't do anything
145 // (unless the transfer is volatile).
146 if (hasUndefSource(MI) && !MI->isVolatile()) {
147 // Set the size of the copy to 0, it will be deleted on the next iteration.
148 MI->setLength((uint64_t)0);
149 return MI;
150 }
151
152 // If MemCpyInst length is 1/2/4/8 bytes then replace memcpy with
153 // load/store.
154 ConstantInt *MemOpLength = dyn_cast<ConstantInt>(MI->getLength());
155 if (!MemOpLength) return nullptr;
156
157 // Source and destination pointer types are always "i8*" for intrinsic. See
158 // if the size is something we can handle with a single primitive load/store.
159 // A single load+store correctly handles overlapping memory in the memmove
160 // case.
161 uint64_t Size = MemOpLength->getLimitedValue();
162 assert(Size && "0-sized memory transferring should be removed already.");
163
164 if (Size > 8 || (Size&(Size-1)))
165 return nullptr; // If not 1/2/4/8 bytes, exit.
166
167 // If it is an atomic and alignment is less than the size then we will
168 // introduce the unaligned memory access which will be later transformed
169 // into libcall in CodeGen. This is not evident performance gain so disable
170 // it now.
171 if (MI->isAtomic())
172 if (*CopyDstAlign < Size || *CopySrcAlign < Size)
173 return nullptr;
174
175 // Use an integer load+store unless we can find something better.
176 IntegerType* IntType = IntegerType::get(MI->getContext(), Size<<3);
177
178 // If the memcpy has metadata describing the members, see if we can get the
179 // TBAA, scope and noalias tags describing our copy.
180 AAMDNodes AACopyMD = MI->getAAMetadata().adjustForAccess(Size);
181
182 Value *Src = MI->getArgOperand(1);
183 Value *Dest = MI->getArgOperand(0);
184 LoadInst *L = Builder.CreateLoad(IntType, Src);
185 // Alignment from the mem intrinsic will be better, so use it.
186 L->setAlignment(*CopySrcAlign);
187 L->setAAMetadata(AACopyMD);
188 MDNode *LoopMemParallelMD =
189 MI->getMetadata(LLVMContext::MD_mem_parallel_loop_access);
190 if (LoopMemParallelMD)
191 L->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
192 MDNode *AccessGroupMD = MI->getMetadata(LLVMContext::MD_access_group);
193 if (AccessGroupMD)
194 L->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
195
196 StoreInst *S = Builder.CreateStore(L, Dest);
197 // Alignment from the mem intrinsic will be better, so use it.
198 S->setAlignment(*CopyDstAlign);
199 S->setAAMetadata(AACopyMD);
200 if (LoopMemParallelMD)
201 S->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
202 if (AccessGroupMD)
203 S->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
204 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
205
206 if (auto *MT = dyn_cast<MemTransferInst>(MI)) {
207 // non-atomics can be volatile
208 L->setVolatile(MT->isVolatile());
209 S->setVolatile(MT->isVolatile());
210 }
211 if (MI->isAtomic()) {
212 // atomics have to be unordered
213 L->setOrdering(AtomicOrdering::Unordered);
215 }
216
217 // Set the size of the copy to 0, it will be deleted on the next iteration.
218 MI->setLength((uint64_t)0);
219 return MI;
220}
221
223 const Align KnownAlignment =
224 getKnownAlignment(MI->getDest(), DL, MI, &AC, &DT);
225 MaybeAlign MemSetAlign = MI->getDestAlign();
226 if (!MemSetAlign || *MemSetAlign < KnownAlignment) {
227 MI->setDestAlignment(KnownAlignment);
228 return MI;
229 }
230
231 // If we have a store to a location which is known constant, we can conclude
232 // that the store must be storing the constant value (else the memory
233 // wouldn't be constant), and this must be a noop.
234 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
235 // Set the size of the copy to 0, it will be deleted on the next iteration.
236 MI->setLength((uint64_t)0);
237 return MI;
238 }
239
240 // Remove memset with an undef value.
241 // FIXME: This is technically incorrect because it might overwrite a poison
242 // value. Change to PoisonValue once #52930 is resolved.
243 if (isa<UndefValue>(MI->getValue())) {
244 // Set the size of the copy to 0, it will be deleted on the next iteration.
245 MI->setLength((uint64_t)0);
246 return MI;
247 }
248
249 // Extract the length and validate the fill type.
250 ConstantInt *LenC = dyn_cast<ConstantInt>(MI->getLength());
251 Value *Fill = MI->getValue();
252 if (!LenC || !Fill->getType()->isIntegerTy(8))
253 return nullptr;
254 const uint64_t Len = LenC->getLimitedValue();
255 assert(Len && "0-sized memory setting should be removed already.");
256 const Align Alignment = MI->getDestAlign().valueOrOne();
257
258 // If it is an atomic and alignment is less than the size then we will
259 // introduce the unaligned memory access which will be later transformed
260 // into libcall in CodeGen. This is not evident performance gain so disable
261 // it now.
262 if (MI->isAtomic() && Alignment < Len)
263 return nullptr;
264
265 // memset(s,c,n) -> store s, c (for n=1,2,4,8)
266 if (Len <= 8 && isPowerOf2_32((uint32_t)Len)) {
267 Value *Dest = MI->getDest();
268
269 // Extract the fill value and store. A one-byte memset does not need
270 // replication so a nonconstant i8 fill can be stored directly.
271 Value *FillVal;
272 if (auto *FillC = dyn_cast<ConstantInt>(Fill))
273 FillVal = ConstantInt::get(MI->getContext(),
274 APInt::getSplat(Len * 8, FillC->getValue()));
275 else if (Len == 1)
276 FillVal = Fill;
277 else
278 return nullptr;
279
280 StoreInst *S = Builder.CreateStore(FillVal, Dest, MI->isVolatile());
281 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
282 for (DbgVariableRecord *DbgAssign : at::getDVRAssignmentMarkers(S)) {
283 if (llvm::is_contained(DbgAssign->location_ops(), Fill))
284 DbgAssign->replaceVariableLocationOp(Fill, FillVal);
285 }
286
287 S->setAlignment(Alignment);
288 if (MI->isAtomic())
290
291 // Set the size of the copy to 0, it will be deleted on the next iteration.
292 MI->setLength((uint64_t)0);
293 return MI;
294 }
295
296 return nullptr;
297}
298
299// TODO, Obvious Missing Transforms:
300// * Narrow width by halfs excluding zero/undef lanes
301Value *InstCombinerImpl::simplifyMaskedLoad(IntrinsicInst &II) {
302 Value *LoadPtr = II.getArgOperand(0);
303 const Align Alignment = II.getParamAlign(0).valueOrOne();
304 Value *Mask = II.getArgOperand(1);
305
306 // If the mask is all ones or poison, this is a plain vector load of the 1st
307 // argument.
308 if (match(Mask, m_AllOnesOrPoison())) {
309 LoadInst *L = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
310 "unmaskedload");
311 L->copyMetadata(II);
312 return L;
313 }
314
315 // If we can unconditionally load from this address, replace with a
316 // load/select idiom.
317 if (isDereferenceablePointer(LoadPtr, II.getType(),
319 LoadInst *LI = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
320 "unmaskedload");
321 LI->copyMetadata(II);
322 return Builder.CreateSelect(II.getArgOperand(1), LI, II.getArgOperand(2));
323 }
324
325 return nullptr;
326}
327
328// TODO, Obvious Missing Transforms:
329// * Single constant active lane -> store
330// * Narrow width by halfs excluding zero/undef lanes
331Instruction *InstCombinerImpl::simplifyMaskedStore(IntrinsicInst &II) {
332 Value *StorePtr = II.getArgOperand(1);
333 Align Alignment = II.getParamAlign(1).valueOrOne();
334 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
335 if (!ConstMask)
336 return nullptr;
337
338 // If the mask is all zeros or poison, this instruction does nothing.
339 if (match(ConstMask, m_ZeroOrPoison()))
341
342 // If the mask is all ones or poison, this is a plain vector store of the 1st
343 // argument.
344 if (match(ConstMask, m_AllOnesOrPoison())) {
345 StoreInst *S =
346 new StoreInst(II.getArgOperand(0), StorePtr, false, Alignment);
347 S->copyMetadata(II);
348 return S;
349 }
350
351 if (isa<ScalableVectorType>(ConstMask->getType()))
352 return nullptr;
353
354 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
355 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
356 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
357 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
358 PoisonElts))
359 return replaceOperand(II, 0, V);
360
361 return nullptr;
362}
363
364// TODO, Obvious Missing Transforms:
365// * Single constant active lane load -> load
366// * Dereferenceable address & few lanes -> scalarize speculative load/selects
367// * Adjacent vector addresses -> masked.load
368// * Narrow width by halfs excluding zero/undef lanes
369// * Vector incrementing address -> vector masked load
370Instruction *InstCombinerImpl::simplifyMaskedGather(IntrinsicInst &II) {
371 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(1));
372 if (!ConstMask)
373 return nullptr;
374
375 // Vector splat address w/known mask -> scalar load
376 // Fold the gather to load the source vector first lane
377 // because it is reloading the same value each time
378 if (ConstMask->isAllOnesValue())
379 if (auto *SplatPtr = getSplatValue(II.getArgOperand(0))) {
380 auto *VecTy = cast<VectorType>(II.getType());
381 const Align Alignment = II.getParamAlign(0).valueOrOne();
382 LoadInst *L = Builder.CreateAlignedLoad(VecTy->getElementType(), SplatPtr,
383 Alignment, "load.scalar");
384 Value *Shuf =
385 Builder.CreateVectorSplat(VecTy->getElementCount(), L, "broadcast");
387 }
388
389 return nullptr;
390}
391
392// TODO, Obvious Missing Transforms:
393// * Single constant active lane -> store
394// * Adjacent vector addresses -> masked.store
395// * Narrow store width by halfs excluding zero/undef lanes
396// * Vector incrementing address -> vector masked store
397Instruction *InstCombinerImpl::simplifyMaskedScatter(IntrinsicInst &II) {
398 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
399 if (!ConstMask)
400 return nullptr;
401
402 // If the mask is all zeros or poison, a scatter does nothing.
403 if (match(ConstMask, m_ZeroOrPoison()))
405
406 // Vector splat address -> scalar store
407 if (auto *SplatPtr = getSplatValue(II.getArgOperand(1))) {
408 // scatter(splat(value), splat(ptr), non-zero-mask) -> store value, ptr
409 if (auto *SplatValue = getSplatValue(II.getArgOperand(0))) {
410 if (maskContainsAllOneOrUndef(ConstMask)) {
411 Align Alignment = II.getParamAlign(1).valueOrOne();
412 StoreInst *S = new StoreInst(SplatValue, SplatPtr, /*IsVolatile=*/false,
413 Alignment);
414 S->copyMetadata(II);
415 return S;
416 }
417 }
418 // scatter(vector, splat(ptr), splat(true)) -> store extract(vector,
419 // lastlane), ptr
420 if (ConstMask->isAllOnesValue()) {
421 Align Alignment = II.getParamAlign(1).valueOrOne();
422 VectorType *WideLoadTy = cast<VectorType>(II.getArgOperand(1)->getType());
423 ElementCount VF = WideLoadTy->getElementCount();
424 Value *RunTimeVF = Builder.CreateElementCount(Builder.getInt32Ty(), VF);
425 Value *LastLane = Builder.CreateSub(RunTimeVF, Builder.getInt32(1));
426 Value *Extract =
427 Builder.CreateExtractElement(II.getArgOperand(0), LastLane);
428 StoreInst *S =
429 new StoreInst(Extract, SplatPtr, /*IsVolatile=*/false, Alignment);
430 S->copyMetadata(II);
431 return S;
432 }
433 }
434 if (isa<ScalableVectorType>(ConstMask->getType()))
435 return nullptr;
436
437 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
438 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
439 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
440 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
441 PoisonElts))
442 return replaceOperand(II, 0, V);
443 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(1), DemandedElts,
444 PoisonElts))
445 return replaceOperand(II, 1, V);
446
447 return nullptr;
448}
449
450/// This function transforms launder.invariant.group like:
451/// launder(launder(%x)) -> launder(%x) (the result is not the argument)
452/// This is legal because it preserves the most recent information about
453/// the presence or absence of invariant.group.
455 InstCombinerImpl &IC) {
456 auto *Arg = II.getArgOperand(0);
457 auto *StrippedArg = Arg->stripPointerCasts();
458 auto *StrippedInvariantGroupsArg = StrippedArg;
459 while (auto *Intr = dyn_cast<IntrinsicInst>(StrippedInvariantGroupsArg)) {
460 if (Intr->getIntrinsicID() != Intrinsic::launder_invariant_group)
461 break;
462 StrippedInvariantGroupsArg = Intr->getArgOperand(0)->stripPointerCasts();
463 }
464 if (StrippedArg == StrippedInvariantGroupsArg)
465 return nullptr; // No launders to remove.
466
467 Value *Result =
468 IC.Builder.CreateLaunderInvariantGroup(StrippedInvariantGroupsArg);
469 if (Result->getType()->getPointerAddressSpace() !=
470 II.getType()->getPointerAddressSpace())
471 Result = IC.Builder.CreateAddrSpaceCast(Result, II.getType());
472
473 return cast<Instruction>(Result);
474}
475
477 assert((II.getIntrinsicID() == Intrinsic::cttz ||
478 II.getIntrinsicID() == Intrinsic::ctlz) &&
479 "Expected cttz or ctlz intrinsic");
480 bool IsTZ = II.getIntrinsicID() == Intrinsic::cttz;
481 Value *Op0 = II.getArgOperand(0);
482 Value *Op1 = II.getArgOperand(1);
483 Value *X;
484 // ctlz(bitreverse(x)) -> cttz(x)
485 // cttz(bitreverse(x)) -> ctlz(x)
486 if (match(Op0, m_BitReverse(m_Value(X)))) {
487 Intrinsic::ID ID = IsTZ ? Intrinsic::ctlz : Intrinsic::cttz;
488 Function *F =
489 Intrinsic::getOrInsertDeclaration(II.getModule(), ID, II.getType());
490 return CallInst::Create(F, {X, II.getArgOperand(1)});
491 }
492
493 if (II.getType()->isIntOrIntVectorTy(1)) {
494 // ctlz/cttz i1 Op0 --> not Op0
495 if (match(Op1, m_Zero()))
496 return BinaryOperator::CreateNot(Op0);
497 // If zero is poison, then the input can be assumed to be "true", so the
498 // instruction simplifies to "false".
499 assert(match(Op1, m_One()) && "Expected ctlz/cttz operand to be 0 or 1");
500 return IC.replaceInstUsesWith(II, ConstantInt::getNullValue(II.getType()));
501 }
502
503 // If ctlz/cttz is only used as a shift amount, set is_zero_poison to true.
504 if (II.hasOneUse() && match(Op1, m_Zero()) &&
505 match(II.user_back(), m_Shift(m_Value(), m_Specific(&II))))
506 return CallInst::Create(II.getCalledFunction(),
507 {Op0, IC.Builder.getTrue()});
508
509 Constant *C;
510
511 if (IsTZ) {
512 // cttz(-x) -> cttz(x)
513 if (match(Op0, m_Neg(m_Value(X))))
514 return CallInst::Create(II.getCalledFunction(), {X, Op1});
515
516 // cttz(-x & x) -> cttz(x)
517 if (match(Op0, m_c_And(m_Neg(m_Value(X)), m_Deferred(X))))
518 return CallInst::Create(II.getCalledFunction(), {X, Op1});
519
520 // cttz(mul(X, OddC)) -> cttz(X)
521 if (match(Op0, m_Mul(m_Value(X),
522 m_CheckedInt([](const APInt &C) { return C[0]; }))))
523 return CallInst::Create(II.getCalledFunction(), {X, Op1});
524
525 // cttz(sext(x)) -> cttz(zext(x))
526 if (match(Op0, m_OneUse(m_SExt(m_Value(X))))) {
527 auto *Zext = IC.Builder.CreateZExt(X, II.getType());
528 auto *CttzZext =
529 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, Zext, Op1);
530 return IC.replaceInstUsesWith(II, CttzZext);
531 }
532
533 // Zext doesn't change the number of trailing zeros, so narrow:
534 // cttz(zext(x)) -> zext(cttz(x)) if the 'ZeroIsPoison' parameter is 'true'.
535 if (match(Op0, m_OneUse(m_ZExt(m_Value(X)))) && match(Op1, m_One())) {
536 auto *Cttz = IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, X,
537 IC.Builder.getTrue());
538 auto *ZextCttz = IC.Builder.CreateZExt(Cttz, II.getType());
539 return IC.replaceInstUsesWith(II, ZextCttz);
540 }
541
542 // cttz(abs(x)) -> cttz(x)
543 // cttz(nabs(x)) -> cttz(x)
544 Value *Y;
546 if (SPF == SPF_ABS || SPF == SPF_NABS)
547 return CallInst::Create(II.getCalledFunction(), {X, Op1});
548
550 return CallInst::Create(II.getCalledFunction(), {X, Op1});
551
552 // cttz(shl(%const, %val), 1) --> add(cttz(%const, 1), %val)
553 if (match(Op0, m_Shl(m_ImmConstant(C), m_Value(X))) &&
554 match(Op1, m_One())) {
555 Value *ConstCttz =
556 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
557 return BinaryOperator::CreateAdd(ConstCttz, X);
558 }
559
560 // cttz(lshr exact (%const, %val), 1) --> sub(cttz(%const, 1), %val)
561 if (match(Op0, m_Exact(m_LShr(m_ImmConstant(C), m_Value(X)))) &&
562 match(Op1, m_One())) {
563 Value *ConstCttz =
564 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
565 return BinaryOperator::CreateSub(ConstCttz, X);
566 }
567
568 // cttz(add(lshr(UINT_MAX, %val), 1)) --> sub(width, %val)
569 if (match(Op0, m_Add(m_LShr(m_AllOnes(), m_Value(X)), m_One()))) {
570 Value *Width =
571 ConstantInt::get(II.getType(), II.getType()->getScalarSizeInBits());
572 return BinaryOperator::CreateSub(Width, X);
573 }
574 } else {
575 // ctlz(lshr(%const, %val), 1) --> add(ctlz(%const, 1), %val)
576 if (match(Op0, m_LShr(m_ImmConstant(C), m_Value(X))) &&
577 match(Op1, m_One())) {
578 Value *ConstCtlz =
579 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
580 return BinaryOperator::CreateAdd(ConstCtlz, X);
581 }
582
583 // ctlz(shl nuw (%const, %val), 1) --> sub(ctlz(%const, 1), %val)
584 if (match(Op0, m_NUWShl(m_ImmConstant(C), m_Value(X))) &&
585 match(Op1, m_One())) {
586 Value *ConstCtlz =
587 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
588 return BinaryOperator::CreateSub(ConstCtlz, X);
589 }
590
591 // ctlz(~x & (x - 1)) -> bitwidth - cttz(x, false)
592 if (Op0->hasOneUse() &&
593 match(Op0,
595 Type *Ty = II.getType();
596 unsigned BitWidth = Ty->getScalarSizeInBits();
597 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
598 {X, IC.Builder.getFalse()});
599 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
600 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
601 }
602 }
603
604 // cttz(Pow2) -> Log2(Pow2)
605 // ctlz(Pow2) -> BitWidth - 1 - Log2(Pow2)
606 if (auto *R = IC.tryGetLog2(Op0, match(Op1, m_One()))) {
607 if (IsTZ)
608 return IC.replaceInstUsesWith(II, R);
609 BinaryOperator *BO = BinaryOperator::CreateSub(
610 ConstantInt::get(R->getType(), R->getType()->getScalarSizeInBits() - 1),
611 R);
612 BO->setHasNoSignedWrap();
614 return BO;
615 }
616
618
619 // Create a mask for bits above (ctlz) or below (cttz) the first known one.
620 unsigned PossibleZeros = IsTZ ? Known.countMaxTrailingZeros()
621 : Known.countMaxLeadingZeros();
622 unsigned DefiniteZeros = IsTZ ? Known.countMinTrailingZeros()
623 : Known.countMinLeadingZeros();
624
625 // If all bits above (ctlz) or below (cttz) the first known one are known
626 // zero, this value is constant.
627 // FIXME: This should be in InstSimplify because we're replacing an
628 // instruction with a constant.
629 if (PossibleZeros == DefiniteZeros) {
630 auto *C = ConstantInt::get(Op0->getType(), DefiniteZeros);
631 return IC.replaceInstUsesWith(II, C);
632 }
633
634 // If the input to cttz/ctlz is known to be non-zero,
635 // then change the 'ZeroIsPoison' parameter to 'true'
636 // because we know the zero behavior can't affect the result.
637 if (!Known.One.isZero() ||
639 if (!match(II.getArgOperand(1), m_One()))
640 return CallInst::Create(II.getCalledFunction(),
641 {Op0, IC.Builder.getTrue()});
642 }
643
644 // Add range attribute since known bits can't completely reflect what we know.
645 unsigned BitWidth = Op0->getType()->getScalarSizeInBits();
646 if (BitWidth != 1 && !II.hasRetAttr(Attribute::Range) &&
647 !II.getMetadata(LLVMContext::MD_range)) {
648 ConstantRange Range(APInt(BitWidth, DefiniteZeros),
649 APInt(BitWidth, PossibleZeros + 1));
650 II.addRangeRetAttr(Range);
651 return &II;
652 }
653
654 return nullptr;
655}
656
658 assert(II.getIntrinsicID() == Intrinsic::ctpop &&
659 "Expected ctpop intrinsic");
660 Type *Ty = II.getType();
661 unsigned BitWidth = Ty->getScalarSizeInBits();
662 Value *Op0 = II.getArgOperand(0);
663 Value *X, *Y;
664
665 // ctpop(bitreverse(x)) -> ctpop(x)
666 // ctpop(bswap(x)) -> ctpop(x)
667 if (match(Op0, m_BitReverse(m_Value(X))) || match(Op0, m_BSwap(m_Value(X))))
668 return CallInst::Create(II.getCalledFunction(), X);
669
670 // ctpop(rot(x)) -> ctpop(x)
671 if ((match(Op0, m_FShl(m_Value(X), m_Value(Y), m_Value())) ||
672 match(Op0, m_FShr(m_Value(X), m_Value(Y), m_Value()))) &&
673 X == Y)
674 return CallInst::Create(II.getCalledFunction(), X);
675
676 // ctpop(x | -x) -> bitwidth - cttz(x, false)
677 if (Op0->hasOneUse() &&
678 match(Op0, m_c_Or(m_Value(X), m_Neg(m_Deferred(X))))) {
679 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
680 {X, IC.Builder.getFalse()});
681 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
682 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
683 }
684
685 // ctpop(~x & (x - 1)) -> cttz(x, false)
686 if (match(Op0,
688 Function *F =
689 Intrinsic::getOrInsertDeclaration(II.getModule(), Intrinsic::cttz, Ty);
690 return CallInst::Create(F, {X, IC.Builder.getFalse()});
691 }
692
693 // Zext doesn't change the number of set bits, so narrow:
694 // ctpop (zext X) --> zext (ctpop X)
695 if (match(Op0, m_OneUse(m_ZExt(m_Value(X))))) {
696 Value *NarrowPop = IC.Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, X);
697 return CastInst::Create(Instruction::ZExt, NarrowPop, Ty);
698 }
699
701 IC.computeKnownBits(Op0, Known, &II);
702
703 // If all bits are zero except for exactly one fixed bit, then the result
704 // must be 0 or 1, and we can get that answer by shifting to LSB:
705 // ctpop (X & 32) --> (X & 32) >> 5
706 // TODO: Investigate removing this as its likely unnecessary given the below
707 // `isKnownToBeAPowerOfTwo` check.
708 if ((~Known.Zero).isPowerOf2())
709 return BinaryOperator::CreateLShr(
710 Op0, ConstantInt::get(Ty, (~Known.Zero).exactLogBase2()));
711
712 // More generally we can also handle non-constant power of 2 patterns such as
713 // shl/shr(Pow2, X), (X & -X), etc... by transforming:
714 // ctpop(Pow2OrZero) --> icmp ne X, 0
715 if (IC.isKnownToBeAPowerOfTwo(Op0, /* OrZero */ true))
716 return CastInst::Create(Instruction::ZExt,
719 Ty);
720
721 // Add range attribute since known bits can't completely reflect what we know.
722 if (BitWidth != 1) {
723 ConstantRange OldRange =
724 II.getRange().value_or(ConstantRange::getFull(BitWidth));
725
726 unsigned Lower = Known.countMinPopulation();
727 unsigned Upper = Known.countMaxPopulation() + 1;
728
729 if (Lower == 0 && OldRange.contains(APInt::getZero(BitWidth)) &&
731 Lower = 1;
732
734 Range = Range.intersectWith(OldRange, ConstantRange::Unsigned);
735
736 if (Range != OldRange) {
737 II.addRangeRetAttr(Range);
738 return &II;
739 }
740 }
741
742 return nullptr;
743}
744
745/// Convert `tbl`/`tbx` intrinsics to shufflevector if the mask is constant, and
746/// at most two source operands are actually referenced.
748 bool IsExtension) {
749 // Bail out if the mask is not a constant.
750 auto *C = dyn_cast<Constant>(II.getArgOperand(II.arg_size() - 1));
751 if (!C)
752 return nullptr;
753
754 auto *RetTy = cast<FixedVectorType>(II.getType());
755 unsigned NumIndexes = RetTy->getNumElements();
756
757 // Only perform this transformation for <8 x i8> and <16 x i8> vector types.
758 if (!RetTy->getElementType()->isIntegerTy(8) ||
759 (NumIndexes != 8 && NumIndexes != 16))
760 return nullptr;
761
762 // For tbx instructions, the first argument is the "fallback" vector, which
763 // has the same length as the mask and return type.
764 unsigned int StartIndex = (unsigned)IsExtension;
765 auto *SourceTy =
766 cast<FixedVectorType>(II.getArgOperand(StartIndex)->getType());
767 // Note that the element count of each source vector does *not* need to be the
768 // same as the element count of the return type and mask! All source vectors
769 // must have the same element count as each other, though.
770 unsigned NumElementsPerSource = SourceTy->getNumElements();
771
772 // There are no tbl/tbx intrinsics for which the destination size exceeds the
773 // source size. However, our definitions of the intrinsics, at least in
774 // IntrinsicsAArch64.td, allow for arbitrary destination vector sizes, so it
775 // *could* technically happen.
776 if (NumIndexes > NumElementsPerSource)
777 return nullptr;
778
779 // The tbl/tbx intrinsics take several source operands followed by a mask
780 // operand.
781 unsigned int NumSourceOperands = II.arg_size() - 1 - (unsigned)IsExtension;
782
783 // Map input operands to shuffle indices. This also helpfully deduplicates the
784 // input arguments, in case the same value is passed as an argument multiple
785 // times.
786 SmallDenseMap<Value *, unsigned, 2> ValueToShuffleSlot;
787 Value *ShuffleOperands[2] = {PoisonValue::get(SourceTy),
788 PoisonValue::get(SourceTy)};
789
790 int Indexes[16];
791 for (unsigned I = 0; I < NumIndexes; ++I) {
792 Constant *COp = C->getAggregateElement(I);
793
794 if (!COp || (!isa<UndefValue>(COp) && !isa<ConstantInt>(COp)))
795 return nullptr;
796
797 if (isa<UndefValue>(COp)) {
798 Indexes[I] = -1;
799 continue;
800 }
801
802 uint64_t Index = cast<ConstantInt>(COp)->getZExtValue();
803 // The index of the input argument that this index references (0 = first
804 // source argument, etc).
805 unsigned SourceOperandIndex = Index / NumElementsPerSource;
806 // The index of the element at that source operand.
807 unsigned SourceOperandElementIndex = Index % NumElementsPerSource;
808
809 Value *SourceOperand;
810 if (SourceOperandIndex >= NumSourceOperands) {
811 // This index is out of bounds. Map it to index into either the fallback
812 // vector (tbx) or vector of zeroes (tbl).
813 SourceOperandIndex = NumSourceOperands;
814 if (IsExtension) {
815 // For out-of-bounds indices in tbx, choose the `I`th element of the
816 // fallback.
817 SourceOperand = II.getArgOperand(0);
818 SourceOperandElementIndex = I;
819 } else {
820 // Otherwise, choose some element from the dummy vector of zeroes (we'll
821 // always choose the first).
822 SourceOperand = Constant::getNullValue(SourceTy);
823 SourceOperandElementIndex = 0;
824 }
825 } else {
826 SourceOperand = II.getArgOperand(SourceOperandIndex + StartIndex);
827 }
828
829 // The source operand may be the fallback vector, which may not have the
830 // same number of elements as the source vector. In that case, we *could*
831 // choose to extend its length with another shufflevector, but it's simpler
832 // to just bail instead.
833 if (cast<FixedVectorType>(SourceOperand->getType())->getNumElements() !=
834 NumElementsPerSource)
835 return nullptr;
836
837 // We now know the source operand referenced by this index. Make it a
838 // shufflevector operand, if it isn't already.
839 unsigned NumSlots = ValueToShuffleSlot.size();
840 // This shuffle references more than two sources, and hence cannot be
841 // represented as a shufflevector.
842 if (NumSlots == 2 && !ValueToShuffleSlot.contains(SourceOperand))
843 return nullptr;
844
845 auto [It, Inserted] =
846 ValueToShuffleSlot.try_emplace(SourceOperand, NumSlots);
847 if (Inserted)
848 ShuffleOperands[It->getSecond()] = SourceOperand;
849
850 unsigned RemappedIndex =
851 (It->getSecond() * NumElementsPerSource) + SourceOperandElementIndex;
852 Indexes[I] = RemappedIndex;
853 }
854
856 ShuffleOperands[0], ShuffleOperands[1], ArrayRef(Indexes, NumIndexes));
857 return IC.replaceInstUsesWith(II, Shuf);
858}
859
860// Returns true iff the 2 intrinsics have the same operands, limiting the
861// comparison to the first NumOperands.
862static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E,
863 unsigned NumOperands) {
864 assert(I.arg_size() >= NumOperands && "Not enough operands");
865 assert(E.arg_size() >= NumOperands && "Not enough operands");
866 for (unsigned i = 0; i < NumOperands; i++)
867 if (I.getArgOperand(i) != E.getArgOperand(i))
868 return false;
869 return true;
870}
871
872// Remove trivially empty start/end intrinsic ranges, i.e. a start
873// immediately followed by an end (ignoring debuginfo or other
874// start/end intrinsics in between). As this handles only the most trivial
875// cases, tracking the nesting level is not needed:
876//
877// call @llvm.foo.start(i1 0)
878// call @llvm.foo.start(i1 0) ; This one won't be skipped: it will be removed
879// call @llvm.foo.end(i1 0)
880// call @llvm.foo.end(i1 0) ; &I
881static bool
883 std::function<bool(const IntrinsicInst &)> IsStart) {
884 // We start from the end intrinsic and scan backwards, so that InstCombine
885 // has already processed (and potentially removed) all the instructions
886 // before the end intrinsic.
887 BasicBlock::reverse_iterator BI(EndI), BE(EndI.getParent()->rend());
888 for (; BI != BE; ++BI) {
889 if (auto *I = dyn_cast<IntrinsicInst>(&*BI)) {
890 if (I->isDebugOrPseudoInst() ||
891 I->getIntrinsicID() == EndI.getIntrinsicID())
892 continue;
893 if (IsStart(*I)) {
894 if (haveSameOperands(EndI, *I, EndI.arg_size())) {
896 IC.eraseInstFromFunction(EndI);
897 return true;
898 }
899 // Skip start intrinsics that don't pair with this end intrinsic.
900 continue;
901 }
902 }
903 break;
904 }
905
906 return false;
907}
908
910 removeTriviallyEmptyRange(I, *this, [&I](const IntrinsicInst &II) {
911 // Bail out on the case where the source va_list of a va_copy is destroyed
912 // immediately by a follow-up va_end.
913 return II.getIntrinsicID() == Intrinsic::vastart ||
914 (II.getIntrinsicID() == Intrinsic::vacopy &&
915 I.getArgOperand(0) != II.getArgOperand(1));
916 });
917 return nullptr;
918}
919
921 assert(Call.arg_size() > 1 && "Need at least 2 args to swap");
922 Value *Arg0 = Call.getArgOperand(0), *Arg1 = Call.getArgOperand(1);
923 if (isa<Constant>(Arg0) && !isa<Constant>(Arg1)) {
924 Call.setArgOperand(0, Arg1);
925 Call.setArgOperand(1, Arg0);
926 AttributeList CallAttr = Call.getAttributes();
927 AttributeSet LHSAttr = CallAttr.getParamAttrs(0);
928 AttributeSet RHSAttr = CallAttr.getParamAttrs(1);
929 LLVMContext &Ctx = Call.getContext();
930 Call.setAttributes(CallAttr
931 .setAttributesAtIndex(
932 Ctx, AttributeList::FirstArgIndex + 0, RHSAttr)
933 .setAttributesAtIndex(
934 Ctx, AttributeList::FirstArgIndex + 1, LHSAttr));
935 return &Call;
936 }
937 return nullptr;
938}
939
940/// Creates a result tuple for an overflow intrinsic \p II with a given
941/// \p Result and a constant \p Overflow value.
943 Constant *Overflow) {
944 Constant *V[] = {PoisonValue::get(Result->getType()), Overflow};
945 StructType *ST = cast<StructType>(II->getType());
946 Constant *Struct = ConstantStruct::get(ST, V);
947 return InsertValueInst::Create(Struct, Result, 0);
948}
949
951InstCombinerImpl::foldIntrinsicWithOverflowCommon(IntrinsicInst *II) {
952 WithOverflowInst *WO = cast<WithOverflowInst>(II);
953 Value *OperationResult = nullptr;
954 Constant *OverflowResult = nullptr;
955 if (OptimizeOverflowCheck(WO->getBinaryOp(), WO->isSigned(), WO->getLHS(),
956 WO->getRHS(), *WO, OperationResult, OverflowResult))
957 return createOverflowTuple(WO, OperationResult, OverflowResult);
958
959 // See whether we can optimize the overflow check with assumption information.
960 for (User *U : WO->users()) {
961 if (!match(U, m_ExtractValue<1>(m_Value())))
962 continue;
963
964 for (auto &AssumeVH : AC.assumptionsFor(U)) {
965 if (!AssumeVH)
966 continue;
967 CallInst *I = cast<CallInst>(AssumeVH);
968 if (!match(I->getArgOperand(0), m_Not(m_Specific(U))))
969 continue;
970 if (!isValidAssumeForContext(I, II, /*DT=*/nullptr,
971 /*AllowEphemerals=*/true))
972 continue;
973 Value *Result =
974 Builder.CreateBinOp(WO->getBinaryOp(), WO->getLHS(), WO->getRHS());
975 Result->takeName(WO);
976 if (auto *Inst = dyn_cast<Instruction>(Result)) {
977 if (WO->isSigned())
978 Inst->setHasNoSignedWrap();
979 else
980 Inst->setHasNoUnsignedWrap();
981 }
982 return createOverflowTuple(WO, Result,
983 ConstantInt::getFalse(U->getType()));
984 }
985 }
986
987 return nullptr;
988}
989
990static bool inputDenormalIsIEEE(const Function &F, const Type *Ty) {
991 Ty = Ty->getScalarType();
992 return F.getDenormalMode(Ty->getFltSemantics()).Input == DenormalMode::IEEE;
993}
994
995static bool inputDenormalIsDAZ(const Function &F, const Type *Ty) {
996 Ty = Ty->getScalarType();
997 return F.getDenormalMode(Ty->getFltSemantics()).inputsAreZero();
998}
999
1000/// \returns the compare predicate type if the test performed by
1001/// llvm.is.fpclass(x, \p Mask) is equivalent to fcmp o__ x, 0.0 with the
1002/// floating-point environment assumed for \p F for type \p Ty
1004 const Function &F, Type *Ty) {
1005 switch (static_cast<unsigned>(Mask)) {
1006 case fcZero:
1007 if (inputDenormalIsIEEE(F, Ty))
1008 return FCmpInst::FCMP_OEQ;
1009 break;
1010 case fcZero | fcSubnormal:
1011 if (inputDenormalIsDAZ(F, Ty))
1012 return FCmpInst::FCMP_OEQ;
1013 break;
1014 case fcPositive | fcNegZero:
1015 if (inputDenormalIsIEEE(F, Ty))
1016 return FCmpInst::FCMP_OGE;
1017 break;
1019 if (inputDenormalIsDAZ(F, Ty))
1020 return FCmpInst::FCMP_OGE;
1021 break;
1023 if (inputDenormalIsIEEE(F, Ty))
1024 return FCmpInst::FCMP_OGT;
1025 break;
1026 case fcNegative | fcPosZero:
1027 if (inputDenormalIsIEEE(F, Ty))
1028 return FCmpInst::FCMP_OLE;
1029 break;
1031 if (inputDenormalIsDAZ(F, Ty))
1032 return FCmpInst::FCMP_OLE;
1033 break;
1035 if (inputDenormalIsIEEE(F, Ty))
1036 return FCmpInst::FCMP_OLT;
1037 break;
1038 case fcPosNormal | fcPosInf:
1039 if (inputDenormalIsDAZ(F, Ty))
1040 return FCmpInst::FCMP_OGT;
1041 break;
1042 case fcNegNormal | fcNegInf:
1043 if (inputDenormalIsDAZ(F, Ty))
1044 return FCmpInst::FCMP_OLT;
1045 break;
1046 case ~fcZero & ~fcNan:
1047 if (inputDenormalIsIEEE(F, Ty))
1048 return FCmpInst::FCMP_ONE;
1049 break;
1050 case ~(fcZero | fcSubnormal) & ~fcNan:
1051 if (inputDenormalIsDAZ(F, Ty))
1052 return FCmpInst::FCMP_ONE;
1053 break;
1054 default:
1055 break;
1056 }
1057
1059}
1060
1061Instruction *InstCombinerImpl::foldIntrinsicIsFPClass(IntrinsicInst &II) {
1062 Value *Src0 = II.getArgOperand(0);
1063 Value *Src1 = II.getArgOperand(1);
1064 const ConstantInt *CMask = cast<ConstantInt>(Src1);
1065 FPClassTest Mask = static_cast<FPClassTest>(CMask->getZExtValue());
1066 const bool IsUnordered = (Mask & fcNan) == fcNan;
1067 const bool IsOrdered = (Mask & fcNan) == fcNone;
1068 const FPClassTest OrderedMask = Mask & ~fcNan;
1069 const FPClassTest OrderedInvertedMask = ~OrderedMask & ~fcNan;
1070
1071 const bool IsStrict =
1072 II.getFunction()->getAttributes().hasFnAttr(Attribute::StrictFP);
1073
1074 Value *FNegSrc;
1075 // is.fpclass (fneg x), mask -> is.fpclass x, (fneg mask)
1076 if (match(Src0, m_FNeg(m_Value(FNegSrc))))
1077 return CallInst::Create(
1078 II.getCalledFunction(),
1079 {FNegSrc, ConstantInt::get(Src1->getType(), fneg(Mask))});
1080
1081 Value *FAbsSrc;
1082 if (match(Src0, m_FAbs(m_Value(FAbsSrc))))
1083 return CallInst::Create(
1084 II.getCalledFunction(),
1085 {FAbsSrc, ConstantInt::get(Src1->getType(), inverse_fabs(Mask))});
1086
1087 if ((OrderedMask == fcInf || OrderedInvertedMask == fcInf) &&
1088 (IsOrdered || IsUnordered) && !IsStrict) {
1089 // is.fpclass(x, fcInf) -> fcmp oeq fabs(x), +inf
1090 // is.fpclass(x, ~fcInf) -> fcmp one fabs(x), +inf
1091 // is.fpclass(x, fcInf|fcNan) -> fcmp ueq fabs(x), +inf
1092 // is.fpclass(x, ~(fcInf|fcNan)) -> fcmp une fabs(x), +inf
1094 FCmpInst::Predicate Pred =
1095 IsUnordered ? FCmpInst::FCMP_UEQ : FCmpInst::FCMP_OEQ;
1096 if (OrderedInvertedMask == fcInf)
1097 Pred = IsUnordered ? FCmpInst::FCMP_UNE : FCmpInst::FCMP_ONE;
1098
1099 Value *Fabs = Builder.CreateFAbs(Src0);
1100 Value *CmpInf = Builder.CreateFCmp(Pred, Fabs, Inf);
1101 CmpInf->takeName(&II);
1102 return replaceInstUsesWith(II, CmpInf);
1103 }
1104
1105 if ((OrderedMask == fcPosInf || OrderedMask == fcNegInf) &&
1106 (IsOrdered || IsUnordered) && !IsStrict) {
1107 // is.fpclass(x, fcPosInf) -> fcmp oeq x, +inf
1108 // is.fpclass(x, fcNegInf) -> fcmp oeq x, -inf
1109 // is.fpclass(x, fcPosInf|fcNan) -> fcmp ueq x, +inf
1110 // is.fpclass(x, fcNegInf|fcNan) -> fcmp ueq x, -inf
1111 Constant *Inf =
1112 ConstantFP::getInfinity(Src0->getType(), OrderedMask == fcNegInf);
1113 Value *EqInf = IsUnordered ? Builder.CreateFCmpUEQ(Src0, Inf)
1114 : Builder.CreateFCmpOEQ(Src0, Inf);
1115
1116 EqInf->takeName(&II);
1117 return replaceInstUsesWith(II, EqInf);
1118 }
1119
1120 if ((OrderedInvertedMask == fcPosInf || OrderedInvertedMask == fcNegInf) &&
1121 (IsOrdered || IsUnordered) && !IsStrict) {
1122 // is.fpclass(x, ~fcPosInf) -> fcmp one x, +inf
1123 // is.fpclass(x, ~fcNegInf) -> fcmp one x, -inf
1124 // is.fpclass(x, ~fcPosInf|fcNan) -> fcmp une x, +inf
1125 // is.fpclass(x, ~fcNegInf|fcNan) -> fcmp une x, -inf
1127 OrderedInvertedMask == fcNegInf);
1128 Value *NeInf = IsUnordered ? Builder.CreateFCmpUNE(Src0, Inf)
1129 : Builder.CreateFCmpONE(Src0, Inf);
1130 NeInf->takeName(&II);
1131 return replaceInstUsesWith(II, NeInf);
1132 }
1133
1134 if (Mask == fcNan && !IsStrict) {
1135 // Equivalent of isnan. Replace with standard fcmp if we don't care about FP
1136 // exceptions.
1137 Value *IsNan =
1138 Builder.CreateFCmpUNO(Src0, ConstantFP::getZero(Src0->getType()));
1139 IsNan->takeName(&II);
1140 return replaceInstUsesWith(II, IsNan);
1141 }
1142
1143 if (Mask == (~fcNan & fcAllFlags) && !IsStrict) {
1144 // Equivalent of !isnan. Replace with standard fcmp.
1145 Value *FCmp =
1146 Builder.CreateFCmpORD(Src0, ConstantFP::getZero(Src0->getType()));
1147 FCmp->takeName(&II);
1148 return replaceInstUsesWith(II, FCmp);
1149 }
1150
1152
1153 // Try to replace with an fcmp with 0
1154 //
1155 // is.fpclass(x, fcZero) -> fcmp oeq x, 0.0
1156 // is.fpclass(x, fcZero | fcNan) -> fcmp ueq x, 0.0
1157 // is.fpclass(x, ~fcZero & ~fcNan) -> fcmp one x, 0.0
1158 // is.fpclass(x, ~fcZero) -> fcmp une x, 0.0
1159 //
1160 // is.fpclass(x, fcPosSubnormal | fcPosNormal | fcPosInf) -> fcmp ogt x, 0.0
1161 // is.fpclass(x, fcPositive | fcNegZero) -> fcmp oge x, 0.0
1162 //
1163 // is.fpclass(x, fcNegSubnormal | fcNegNormal | fcNegInf) -> fcmp olt x, 0.0
1164 // is.fpclass(x, fcNegative | fcPosZero) -> fcmp ole x, 0.0
1165 //
1166 if (!IsStrict && (IsOrdered || IsUnordered) &&
1167 (PredType = fpclassTestIsFCmp0(OrderedMask, *II.getFunction(),
1168 Src0->getType())) !=
1171 // Equivalent of == 0.
1172 Value *FCmp = Builder.CreateFCmp(
1173 IsUnordered ? FCmpInst::getUnorderedPredicate(PredType) : PredType,
1174 Src0, Zero);
1175
1176 FCmp->takeName(&II);
1177 return replaceInstUsesWith(II, FCmp);
1178 }
1179
1180 KnownFPClass Known =
1181 computeKnownFPClass(Src0, Mask, SQ.getWithInstruction(&II));
1182
1183 // If none of the tests which can return false are possible, fold to true.
1184 // fp_class (nnan x), ~(qnan|snan) -> true
1185 // fp_class (ninf x), ~(ninf|pinf) -> true
1186 if (Known.isKnownAlways(Mask))
1187 return replaceInstUsesWith(II, ConstantInt::get(II.getType(), true));
1188
1189 // Clear test bits we know must be false from the source value.
1190 // fp_class (nnan x), qnan|snan|other -> fp_class (nnan x), other
1191 // fp_class (ninf x), ninf|pinf|other -> fp_class (ninf x), other
1192 if ((Mask & Known.getKnownFPClasses()) != Mask) {
1193 II.setArgOperand(
1194 1, ConstantInt::get(Src1->getType(), Mask & Known.getKnownFPClasses()));
1195 return &II;
1196 }
1197
1198 return nullptr;
1199}
1200
1201static std::optional<bool> getKnownSign(Value *Op, const SimplifyQuery &SQ) {
1203 if (Known.isNonNegative())
1204 return false;
1205 if (Known.isNegative())
1206 return true;
1207
1208 Value *X, *Y;
1209 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1211
1212 return std::nullopt;
1213}
1214
1215static std::optional<bool> getKnownSignOrZero(Value *Op,
1216 const SimplifyQuery &SQ) {
1217 if (std::optional<bool> Sign = getKnownSign(Op, SQ))
1218 return Sign;
1219
1220 Value *X, *Y;
1221 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1223
1224 return std::nullopt;
1225}
1226
1227/// Return true if two values \p Op0 and \p Op1 are known to have the same sign.
1228static bool signBitMustBeTheSame(Value *Op0, Value *Op1,
1229 const SimplifyQuery &SQ) {
1230 std::optional<bool> Known1 = getKnownSign(Op1, SQ);
1231 if (!Known1)
1232 return false;
1233 std::optional<bool> Known0 = getKnownSign(Op0, SQ);
1234 if (!Known0)
1235 return false;
1236 return *Known0 == *Known1;
1237}
1238
1239// Determines if ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b)) is safe.
1240//
1241// This is true if, when the add saturates, the resulting ldexp is guaranteed to
1242// produce 0 or inf.
1243static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy) {
1244 const fltSemantics &FltSem = FpTy->getScalarType()->getFltSemantics();
1245 if (!APFloat::semanticsHasInf(FltSem))
1246 return false;
1247
1248 // Cap ExpBits at 32 because scalbn takes an int. This is sufficient for any
1249 // reasonable fp type (for example, `double` only has 11 exponent bits).
1250 unsigned ExpBits = std::min(ExpTy->getScalarSizeInBits(), 32u);
1251 int SignedMax = static_cast<int>(maxIntN(ExpBits));
1252 int SignedMin = static_cast<int>(minIntN(ExpBits));
1253 APFloat ScaledUp = scalbn(APFloat::getSmallest(FltSem), SignedMax,
1255 APFloat ScaledDown = scalbn(APFloat::getLargest(FltSem), SignedMin,
1257 return ScaledUp.isInfinity() && ScaledDown.isZero();
1258}
1259
1260/// Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0. This
1261/// can trigger other combines.
1263 InstCombiner::BuilderTy &Builder) {
1264 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1265 assert((MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin ||
1266 MinMaxID == Intrinsic::umax || MinMaxID == Intrinsic::umin) &&
1267 "Expected a min or max intrinsic");
1268
1269 // TODO: Match vectors with undef elements, but undef may not propagate.
1270 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
1271 Value *X;
1272 const APInt *C0, *C1;
1273 if (!match(Op0, m_OneUse(m_Add(m_Value(X), m_APInt(C0)))) ||
1274 !match(Op1, m_APInt(C1)))
1275 return nullptr;
1276
1277 // Check for necessary no-wrap and overflow constraints.
1278 bool IsSigned = MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin;
1279 auto *Add = cast<BinaryOperator>(Op0);
1280 if ((IsSigned && !Add->hasNoSignedWrap()) ||
1281 (!IsSigned && !Add->hasNoUnsignedWrap()))
1282 return nullptr;
1283
1284 // If the constant difference overflows, then instsimplify should reduce the
1285 // min/max to the add or C1.
1286 bool Overflow;
1287 APInt CDiff =
1288 IsSigned ? C1->ssub_ov(*C0, Overflow) : C1->usub_ov(*C0, Overflow);
1289 assert(!Overflow && "Expected simplify of min/max");
1290
1291 // min/max (add X, C0), C1 --> add (min/max X, C1 - C0), C0
1292 // Note: the "mismatched" no-overflow setting does not propagate.
1293 Constant *NewMinMaxC = ConstantInt::get(II->getType(), CDiff);
1294 Value *NewMinMax = Builder.CreateBinaryIntrinsic(MinMaxID, X, NewMinMaxC);
1295 return IsSigned ? BinaryOperator::CreateNSWAdd(NewMinMax, Add->getOperand(1))
1296 : BinaryOperator::CreateNUWAdd(NewMinMax, Add->getOperand(1));
1297}
1298/// Match a sadd_sat or ssub_sat which is using min/max to clamp the value.
1299Instruction *InstCombinerImpl::matchSAddSubSat(IntrinsicInst &MinMax1) {
1300 Type *Ty = MinMax1.getType();
1301
1302 // We are looking for a tree of:
1303 // max(INT_MIN, min(INT_MAX, add(sext(A), sext(B))))
1304 // Where the min and max could be reversed
1305 Instruction *MinMax2;
1306 BinaryOperator *AddSub;
1307 const APInt *MinValue, *MaxValue;
1308 if (match(&MinMax1, m_SMin(m_Instruction(MinMax2), m_APInt(MaxValue)))) {
1309 if (!match(MinMax2, m_SMax(m_BinOp(AddSub), m_APInt(MinValue))))
1310 return nullptr;
1311 } else if (match(&MinMax1,
1312 m_SMax(m_Instruction(MinMax2), m_APInt(MinValue)))) {
1313 if (!match(MinMax2, m_SMin(m_BinOp(AddSub), m_APInt(MaxValue))))
1314 return nullptr;
1315 } else
1316 return nullptr;
1317
1318 // Check that the constants clamp a saturate, and that the new type would be
1319 // sensible to convert to.
1320 if (!(*MaxValue + 1).isPowerOf2() || -*MinValue != *MaxValue + 1)
1321 return nullptr;
1322 // In what bitwidth can this be treated as saturating arithmetics?
1323 unsigned NewBitWidth = (*MaxValue + 1).logBase2() + 1;
1324 // FIXME: This isn't quite right for vectors, but using the scalar type is a
1325 // good first approximation for what should be done there.
1326 if (!shouldChangeType(Ty->getScalarType()->getIntegerBitWidth(), NewBitWidth))
1327 return nullptr;
1328
1329 // Also make sure that the inner min/max and the add/sub have one use.
1330 if (!MinMax2->hasOneUse() || !AddSub->hasOneUse())
1331 return nullptr;
1332
1333 // Create the new type (which can be a vector type)
1334 Type *NewTy = Ty->getWithNewBitWidth(NewBitWidth);
1335
1336 Intrinsic::ID IntrinsicID;
1337 if (AddSub->getOpcode() == Instruction::Add)
1338 IntrinsicID = Intrinsic::sadd_sat;
1339 else if (AddSub->getOpcode() == Instruction::Sub)
1340 IntrinsicID = Intrinsic::ssub_sat;
1341 else
1342 return nullptr;
1343
1344 // The two operands of the add/sub must be nsw-truncatable to the NewTy. This
1345 // is usually achieved via a sext from a smaller type.
1346 if (ComputeMaxSignificantBits(AddSub->getOperand(0), AddSub) > NewBitWidth ||
1347 ComputeMaxSignificantBits(AddSub->getOperand(1), AddSub) > NewBitWidth)
1348 return nullptr;
1349
1350 // Finally create and return the sat intrinsic, truncated to the new type
1351 Value *AT = Builder.CreateTrunc(AddSub->getOperand(0), NewTy);
1352 Value *BT = Builder.CreateTrunc(AddSub->getOperand(1), NewTy);
1353 Value *Sat = Builder.CreateIntrinsic(IntrinsicID, NewTy, {AT, BT});
1354 return CastInst::Create(Instruction::SExt, Sat, Ty);
1355}
1356
1357
1358/// If we have a clamp pattern like max (min X, 42), 41 -- where the output
1359/// can only be one of two possible constant values -- turn that into a select
1360/// of constants.
1362 InstCombiner::BuilderTy &Builder) {
1363 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
1364 Value *X;
1365 const APInt *C0, *C1;
1366 if (!match(I1, m_APInt(C1)) || !I0->hasOneUse())
1367 return nullptr;
1368
1370 switch (II->getIntrinsicID()) {
1371 case Intrinsic::smax:
1372 if (match(I0, m_SMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1373 Pred = ICmpInst::ICMP_SGT;
1374 break;
1375 case Intrinsic::smin:
1376 if (match(I0, m_SMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1377 Pred = ICmpInst::ICMP_SLT;
1378 break;
1379 case Intrinsic::umax:
1380 if (match(I0, m_UMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1381 Pred = ICmpInst::ICMP_UGT;
1382 break;
1383 case Intrinsic::umin:
1384 if (match(I0, m_UMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1385 Pred = ICmpInst::ICMP_ULT;
1386 break;
1387 default:
1388 llvm_unreachable("Expected min/max intrinsic");
1389 }
1390 if (Pred == CmpInst::BAD_ICMP_PREDICATE)
1391 return nullptr;
1392
1393 // max (min X, 42), 41 --> X > 41 ? 42 : 41
1394 // min (max X, 42), 43 --> X < 43 ? 42 : 43
1395 Value *Cmp = Builder.CreateICmp(Pred, X, I1);
1396 return SelectInst::Create(Cmp, ConstantInt::get(II->getType(), *C0), I1);
1397}
1398
1399/// If this min/max has a constant operand and an operand that is a matching
1400/// min/max with a constant operand, constant-fold the 2 constant operands.
1402 IRBuilderBase &Builder,
1403 const SimplifyQuery &SQ) {
1404 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1405 auto *LHS = dyn_cast<MinMaxIntrinsic>(II->getArgOperand(0));
1406 if (!LHS)
1407 return nullptr;
1408
1409 Constant *C0, *C1;
1410 if (!match(LHS->getArgOperand(1), m_ImmConstant(C0)) ||
1411 !match(II->getArgOperand(1), m_ImmConstant(C1)))
1412 return nullptr;
1413
1414 // max (max X, C0), C1 --> max X, (max C0, C1)
1415 // min (min X, C0), C1 --> min X, (min C0, C1)
1416 // umax (smax X, nneg C0), nneg C1 --> smax X, (umax C0, C1)
1417 // smin (umin X, nneg C0), nneg C1 --> umin X, (smin C0, C1)
1418 Intrinsic::ID InnerMinMaxID = LHS->getIntrinsicID();
1419 if (InnerMinMaxID != MinMaxID &&
1420 !(((MinMaxID == Intrinsic::umax && InnerMinMaxID == Intrinsic::smax) ||
1421 (MinMaxID == Intrinsic::smin && InnerMinMaxID == Intrinsic::umin)) &&
1422 isKnownNonNegative(C0, SQ) && isKnownNonNegative(C1, SQ)))
1423 return nullptr;
1424
1426 Value *CondC = Builder.CreateICmp(Pred, C0, C1);
1427 Value *NewC = Builder.CreateSelect(CondC, C0, C1);
1428 return Builder.CreateIntrinsic(InnerMinMaxID, II->getType(),
1429 {LHS->getArgOperand(0), NewC});
1430}
1431
1432/// If this min/max has a matching min/max operand with a constant, try to push
1433/// the constant operand into this instruction. This can enable more folds.
1434static Instruction *
1436 InstCombiner::BuilderTy &Builder) {
1437 // Match and capture a min/max operand candidate.
1438 Value *X, *Y;
1439 Constant *C;
1440 Instruction *Inner;
1442 m_Instruction(Inner),
1444 m_Value(Y))))
1445 return nullptr;
1446
1447 // The inner op must match. Check for constants to avoid infinite loops.
1448 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1449 auto *InnerMM = dyn_cast<IntrinsicInst>(Inner);
1450 if (!InnerMM || InnerMM->getIntrinsicID() != MinMaxID ||
1452 return nullptr;
1453
1454 // max (max X, C), Y --> max (max X, Y), C
1456 MinMaxID, II->getType());
1457 Value *NewInner = Builder.CreateBinaryIntrinsic(MinMaxID, X, Y);
1458 NewInner->takeName(Inner);
1459 return CallInst::Create(MinMax, {NewInner, C});
1460}
1461
1462/// Reduce a sequence of min/max intrinsics with a common operand.
1464 // Match 3 of the same min/max ops. Example: umin(umin(), umin()).
1465 auto *LHS = dyn_cast<IntrinsicInst>(II->getArgOperand(0));
1466 auto *RHS = dyn_cast<IntrinsicInst>(II->getArgOperand(1));
1467 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1468 if (!LHS || !RHS || LHS->getIntrinsicID() != MinMaxID ||
1469 RHS->getIntrinsicID() != MinMaxID ||
1470 (!LHS->hasOneUse() && !RHS->hasOneUse()))
1471 return nullptr;
1472
1473 Value *A = LHS->getArgOperand(0);
1474 Value *B = LHS->getArgOperand(1);
1475 Value *C = RHS->getArgOperand(0);
1476 Value *D = RHS->getArgOperand(1);
1477
1478 // Look for a common operand.
1479 Value *MinMaxOp = nullptr;
1480 Value *ThirdOp = nullptr;
1481 if (LHS->hasOneUse()) {
1482 // If the LHS is only used in this chain and the RHS is used outside of it,
1483 // reuse the RHS min/max because that will eliminate the LHS.
1484 if (D == A || C == A) {
1485 // min(min(a, b), min(c, a)) --> min(min(c, a), b)
1486 // min(min(a, b), min(a, d)) --> min(min(a, d), b)
1487 MinMaxOp = RHS;
1488 ThirdOp = B;
1489 } else if (D == B || C == B) {
1490 // min(min(a, b), min(c, b)) --> min(min(c, b), a)
1491 // min(min(a, b), min(b, d)) --> min(min(b, d), a)
1492 MinMaxOp = RHS;
1493 ThirdOp = A;
1494 }
1495 } else {
1496 assert(RHS->hasOneUse() && "Expected one-use operand");
1497 // Reuse the LHS. This will eliminate the RHS.
1498 if (D == A || D == B) {
1499 // min(min(a, b), min(c, a)) --> min(min(a, b), c)
1500 // min(min(a, b), min(c, b)) --> min(min(a, b), c)
1501 MinMaxOp = LHS;
1502 ThirdOp = C;
1503 } else if (C == A || C == B) {
1504 // min(min(a, b), min(b, d)) --> min(min(a, b), d)
1505 // min(min(a, b), min(c, b)) --> min(min(a, b), d)
1506 MinMaxOp = LHS;
1507 ThirdOp = D;
1508 }
1509 }
1510
1511 if (!MinMaxOp || !ThirdOp)
1512 return nullptr;
1513
1514 Module *Mod = II->getModule();
1515 Function *MinMax =
1516 Intrinsic::getOrInsertDeclaration(Mod, MinMaxID, II->getType());
1517 return CallInst::Create(MinMax, { MinMaxOp, ThirdOp });
1518}
1519
1520/// If all arguments of the intrinsic are unary shuffles with the same mask,
1521/// try to shuffle after the intrinsic.
1524 if (!II->getType()->isVectorTy() ||
1525 !isTriviallyVectorizable(II->getIntrinsicID()) ||
1526 !II->getCalledFunction()->isSpeculatable())
1527 return nullptr;
1528
1529 Value *X;
1530 Constant *C;
1531 ArrayRef<int> Mask;
1532 auto *NonConstArg = find_if_not(II->args(), [&II](Use &Arg) {
1533 return isa<Constant>(Arg.get()) ||
1534 isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1535 Arg.getOperandNo(), nullptr);
1536 });
1537 if (!NonConstArg ||
1538 !match(NonConstArg, m_Shuffle(m_Value(X), m_Poison(), m_Mask(Mask))))
1539 return nullptr;
1540
1541 // At least 1 operand must be a shuffle with 1 use because we are creating 2
1542 // instructions.
1543 if (none_of(II->args(), match_fn(m_OneUse(m_Shuffle(m_Value(), m_Value())))))
1544 return nullptr;
1545
1546 // See if all arguments are shuffled with the same mask.
1548 Type *SrcTy = X->getType();
1549 for (Use &Arg : II->args()) {
1550 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1551 Arg.getOperandNo(), nullptr))
1552 NewArgs.push_back(Arg);
1553 else if (match(&Arg,
1554 m_Shuffle(m_Value(X), m_Poison(), m_SpecificMask(Mask))) &&
1555 X->getType() == SrcTy)
1556 NewArgs.push_back(X);
1557 else if (match(&Arg, m_ImmConstant(C))) {
1558 // If it's a constant, try find the constant that would be shuffled to C.
1559 if (Constant *ShuffledC =
1560 unshuffleConstant(Mask, C, cast<VectorType>(SrcTy)))
1561 NewArgs.push_back(ShuffledC);
1562 else
1563 return nullptr;
1564 } else
1565 return nullptr;
1566 }
1567
1568 // intrinsic (shuf X, M), (shuf Y, M), ... --> shuf (intrinsic X, Y, ...), M
1569 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1570 // Result type might be a different vector width.
1571 // TODO: Check that the result type isn't widened?
1572 VectorType *ResTy =
1573 VectorType::get(II->getType()->getScalarType(), cast<VectorType>(SrcTy));
1574 Value *NewIntrinsic =
1575 Builder.CreateIntrinsic(ResTy, II->getIntrinsicID(), NewArgs, FPI);
1576 return new ShuffleVectorInst(NewIntrinsic, Mask);
1577}
1578
1579/// If all arguments of the intrinsic are reverses, try to pull the reverse
1580/// after the intrinsic.
1582 if (!II->getType()->isVectorTy() ||
1583 !isTriviallyVectorizable(II->getIntrinsicID()))
1584 return nullptr;
1585
1586 // At least 1 operand must be a reverse with 1 use because we are creating 2
1587 // instructions.
1588 if (none_of(II->args(), [](Value *V) {
1589 return match(V, m_OneUse(m_VecReverse(m_Value())));
1590 }))
1591 return nullptr;
1592
1593 Value *X;
1594 Constant *C;
1595 SmallVector<Value *> NewArgs;
1596 for (Use &Arg : II->args()) {
1597 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1598 Arg.getOperandNo(), nullptr))
1599 NewArgs.push_back(Arg);
1600 else if (match(&Arg, m_VecReverse(m_Value(X))))
1601 NewArgs.push_back(X);
1602 else if (isSplatValue(Arg))
1603 NewArgs.push_back(Arg);
1604 else if (match(&Arg, m_ImmConstant(C)))
1605 NewArgs.push_back(Builder.CreateVectorReverse(C));
1606 else
1607 return nullptr;
1608 }
1609
1610 // intrinsic (reverse X), (reverse Y), ... --> reverse (intrinsic X, Y, ...)
1611 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1612 Value *NewIntrinsic = Builder.CreateIntrinsic(
1613 II->getType(), II->getIntrinsicID(), NewArgs, FPI);
1614 return Builder.CreateVectorReverse(NewIntrinsic);
1615}
1616
1617/// Fold the following cases and accepts bswap and bitreverse intrinsics:
1618/// bswap(logic_op(bswap(x), y)) --> logic_op(x, bswap(y))
1619/// bswap(logic_op(bswap(x), bswap(y))) --> logic_op(x, y) (ignores multiuse)
1620template <Intrinsic::ID IntrID>
1622 InstCombiner::BuilderTy &Builder) {
1623 static_assert(IntrID == Intrinsic::bswap || IntrID == Intrinsic::bitreverse,
1624 "This helper only supports BSWAP and BITREVERSE intrinsics");
1625
1626 Value *X, *Y;
1627 // Find bitwise logic op. Check that it is a BinaryOperator explicitly so we
1628 // don't match ConstantExpr that aren't meaningful for this transform.
1631 Value *OldReorderX, *OldReorderY;
1633
1634 // If both X and Y are bswap/bitreverse, the transform reduces the number
1635 // of instructions even if there's multiuse.
1636 // If only one operand is bswap/bitreverse, we need to ensure the operand
1637 // have only one use.
1638 if (match(X, m_Intrinsic<IntrID>(m_Value(OldReorderX))) &&
1639 match(Y, m_Intrinsic<IntrID>(m_Value(OldReorderY)))) {
1640 return BinaryOperator::Create(Op, OldReorderX, OldReorderY);
1641 }
1642
1643 if (match(X, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderX))))) {
1644 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, Y);
1645 return BinaryOperator::Create(Op, OldReorderX, NewReorder);
1646 }
1647
1648 if (match(Y, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderY))))) {
1649 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, X);
1650 return BinaryOperator::Create(Op, NewReorder, OldReorderY);
1651 }
1652 }
1653 return nullptr;
1654}
1655
1656/// Helper to match idempotent binary intrinsics, namely, intrinsics where
1657/// `f(f(x, y), y) == f(x, y)` holds.
1659 switch (IID) {
1660 case Intrinsic::smax:
1661 case Intrinsic::smin:
1662 case Intrinsic::umax:
1663 case Intrinsic::umin:
1664 case Intrinsic::maximum:
1665 case Intrinsic::minimum:
1666 case Intrinsic::maximumnum:
1667 case Intrinsic::minimumnum:
1668 case Intrinsic::maxnum:
1669 case Intrinsic::minnum:
1670 return true;
1671 default:
1672 return false;
1673 }
1674}
1675
1676/// Attempt to simplify value-accumulating recurrences of kind:
1677/// %umax.acc = phi i8 [ %umax, %backedge ], [ %a, %entry ]
1678/// %umax = call i8 @llvm.umax.i8(i8 %umax.acc, i8 %b)
1679/// And let the idempotent binary intrinsic be hoisted, when the operands are
1680/// known to be loop-invariant.
1682 IntrinsicInst *II) {
1683 PHINode *PN;
1684 Value *Init, *OtherOp;
1685
1686 // A binary intrinsic recurrence with loop-invariant operands is equivalent to
1687 // `call @llvm.binary.intrinsic(Init, OtherOp)`.
1688 auto IID = II->getIntrinsicID();
1689 if (!isIdempotentBinaryIntrinsic(IID) ||
1691 !IC.getDominatorTree().dominates(OtherOp, PN))
1692 return nullptr;
1693
1694 auto *InvariantBinaryInst =
1695 IC.Builder.CreateBinaryIntrinsic(IID, Init, OtherOp);
1696 if (isa<FPMathOperator>(InvariantBinaryInst))
1697 cast<Instruction>(InvariantBinaryInst)->copyFastMathFlags(II);
1698 return InvariantBinaryInst;
1699}
1700
1701static Value *simplifyReductionOperand(Value *Arg, bool CanReorderLanes) {
1702 if (!CanReorderLanes)
1703 return nullptr;
1704
1705 Value *V;
1706 if (match(Arg, m_VecReverse(m_Value(V))))
1707 return V;
1708
1709 ArrayRef<int> Mask;
1710 if (!isa<FixedVectorType>(Arg->getType()) ||
1711 !match(Arg, m_Shuffle(m_Value(V), m_Undef(), m_Mask(Mask))) ||
1712 !cast<ShuffleVectorInst>(Arg)->isSingleSource())
1713 return nullptr;
1714
1715 int Sz = Mask.size();
1716 SmallBitVector UsedIndices(Sz);
1717 for (int Idx : Mask) {
1718 if (Idx == PoisonMaskElem || UsedIndices.test(Idx))
1719 return nullptr;
1720 UsedIndices.set(Idx);
1721 }
1722
1723 // Can remove shuffle iff just shuffled elements, no repeats, undefs, or
1724 // other changes.
1725 return UsedIndices.all() ? V : nullptr;
1726}
1727
1728/// Fold an unsigned minimum of trailing or leading zero bits counts:
1729/// umin(cttz(CtOp1, ZeroUndef), ConstOp) --> cttz(CtOp1 | (1 << ConstOp))
1730/// umin(ctlz(CtOp1, ZeroUndef), ConstOp) --> ctlz(CtOp1 | (SignedMin
1731/// >> ConstOp))
1732/// umin(cttz(CtOp1), cttz(CtOp2)) --> cttz(CtOp1 | CtOp2)
1733/// umin(ctlz(CtOp1), ctlz(CtOp2)) --> ctlz(CtOp1 | CtOp2)
1734template <Intrinsic::ID IntrID>
1735static Value *
1737 const DataLayout &DL,
1738 InstCombiner::BuilderTy &Builder) {
1739 static_assert(IntrID == Intrinsic::cttz || IntrID == Intrinsic::ctlz,
1740 "This helper only supports cttz and ctlz intrinsics");
1741
1742 Value *CtOp1, *CtOp2;
1743 Value *ZeroUndef1, *ZeroUndef2;
1744 if (!match(I0, m_OneUse(
1745 m_Intrinsic<IntrID>(m_Value(CtOp1), m_Value(ZeroUndef1)))))
1746 return nullptr;
1747
1748 if (match(I1,
1749 m_OneUse(m_Intrinsic<IntrID>(m_Value(CtOp2), m_Value(ZeroUndef2)))))
1750 return Builder.CreateBinaryIntrinsic(
1751 IntrID, Builder.CreateOr(CtOp1, CtOp2),
1752 Builder.CreateOr(ZeroUndef1, ZeroUndef2));
1753
1754 unsigned BitWidth = I1->getType()->getScalarSizeInBits();
1755 auto LessBitWidth = [BitWidth](auto &C) { return C.ult(BitWidth); };
1756 if (!match(I1, m_CheckedInt(LessBitWidth)))
1757 // We have a constant >= BitWidth (which can be handled by CVP)
1758 // or a non-splat vector with elements < and >= BitWidth
1759 return nullptr;
1760
1761 Type *Ty = I1->getType();
1763 IntrID == Intrinsic::cttz ? Instruction::Shl : Instruction::LShr,
1764 IntrID == Intrinsic::cttz
1765 ? ConstantInt::get(Ty, 1)
1766 : ConstantInt::get(Ty, APInt::getSignedMinValue(BitWidth)),
1767 cast<Constant>(I1), DL);
1768 return Builder.CreateBinaryIntrinsic(
1769 IntrID, Builder.CreateOr(CtOp1, NewConst),
1770 ConstantInt::getTrue(ZeroUndef1->getType()));
1771}
1772
1773/// Return whether "X LOp (Y ROp Z)" is always equal to
1774/// "(X LOp Y) ROp (X LOp Z)".
1776 bool HasNSW, Intrinsic::ID ROp) {
1777 switch (ROp) {
1778 case Intrinsic::umax:
1779 case Intrinsic::umin:
1780 if (HasNUW && LOp == Instruction::Add)
1781 return true;
1782 if (HasNUW && LOp == Instruction::Shl)
1783 return true;
1784 return false;
1785 case Intrinsic::smax:
1786 case Intrinsic::smin:
1787 return HasNSW && LOp == Instruction::Add;
1788 default:
1789 return false;
1790 }
1791}
1792
1793/// Return whether "(X ROp Y) LOp Z" is always equal to
1794/// "(X LOp Z) ROp (Y LOp Z)".
1796 bool HasNSW, Intrinsic::ID ROp) {
1797 if (Instruction::isCommutative(LOp) || LOp == Instruction::Shl)
1798 return leftDistributesOverRight(LOp, HasNUW, HasNSW, ROp);
1799 switch (ROp) {
1800 case Intrinsic::umax:
1801 case Intrinsic::umin:
1802 return HasNUW && LOp == Instruction::Sub;
1803 case Intrinsic::smax:
1804 case Intrinsic::smin:
1805 return HasNSW && LOp == Instruction::Sub;
1806 default:
1807 return false;
1808 }
1809}
1810
1811// Attempts to factorise a common term
1812// in an instruction that has the form "(A op' B) op (C op' D)
1813// where op is an intrinsic and op' is a binop
1814static Value *
1816 InstCombiner::BuilderTy &Builder) {
1817 Value *LHS = II->getOperand(0), *RHS = II->getOperand(1);
1818 Intrinsic::ID TopLevelOpcode = II->getIntrinsicID();
1819
1822
1823 if (!Op0 || !Op1)
1824 return nullptr;
1825
1826 if (Op0->getOpcode() != Op1->getOpcode())
1827 return nullptr;
1828
1829 if (!Op0->hasOneUse() || !Op1->hasOneUse())
1830 return nullptr;
1831
1832 Instruction::BinaryOps InnerOpcode =
1833 static_cast<Instruction::BinaryOps>(Op0->getOpcode());
1834 bool HasNUW = Op0->hasNoUnsignedWrap() && Op1->hasNoUnsignedWrap();
1835 bool HasNSW = Op0->hasNoSignedWrap() && Op1->hasNoSignedWrap();
1836
1837 Value *A = Op0->getOperand(0);
1838 Value *B = Op0->getOperand(1);
1839 Value *C = Op1->getOperand(0);
1840 Value *D = Op1->getOperand(1);
1841
1842 // Attempts to swap variables such that A equals C or B equals D,
1843 // if the inner operation is commutative.
1844 if (Op0->isCommutative() && A != C && B != D) {
1845 if (A == D || B == C)
1846 std::swap(C, D);
1847 else
1848 return nullptr;
1849 }
1850
1851 if (A == C &&
1852 leftDistributesOverRight(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1853 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, B, D);
1854 return Builder.CreateNoWrapBinOp(InnerOpcode, A, NewIntrinsic, HasNUW,
1855 HasNSW);
1856 }
1857 if (B == D &&
1858 rightDistributesOverLeft(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1859 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, A, C);
1860 return Builder.CreateNoWrapBinOp(InnerOpcode, NewIntrinsic, B, HasNUW,
1861 HasNSW);
1862 }
1863 return nullptr;
1864}
1865
1867 Value *Arg0 = II->getArgOperand(0);
1868 auto *ShiftConst = dyn_cast<Constant>(II->getArgOperand(1));
1869 if (!ShiftConst)
1870 return nullptr;
1871
1872 int ElemBits = Arg0->getType()->getScalarSizeInBits();
1873 bool AllPositive = true;
1874 bool AllNegative = true;
1875
1876 auto Check = [&](Constant *C) -> bool {
1877 if (auto *CI = dyn_cast_or_null<ConstantInt>(C)) {
1878 const APInt &V = CI->getValue();
1879 if (V.isNonNegative()) {
1880 AllNegative = false;
1881 return AllPositive && V.ult(ElemBits);
1882 }
1883 AllPositive = false;
1884 return AllNegative && V.sgt(-ElemBits);
1885 }
1886 return false;
1887 };
1888
1889 if (auto *VTy = dyn_cast<FixedVectorType>(Arg0->getType())) {
1890 for (unsigned I = 0, E = VTy->getNumElements(); I < E; ++I) {
1891 if (!Check(ShiftConst->getAggregateElement(I)))
1892 return nullptr;
1893 }
1894
1895 } else if (!Check(ShiftConst))
1896 return nullptr;
1897
1898 IRBuilderBase &B = IC.Builder;
1899 if (AllPositive)
1900 return IC.replaceInstUsesWith(*II, B.CreateShl(Arg0, ShiftConst));
1901
1902 Value *NegAmt = B.CreateNeg(ShiftConst);
1903 Intrinsic::ID IID = II->getIntrinsicID();
1904 const bool IsSigned =
1905 IID == Intrinsic::arm_neon_vshifts || IID == Intrinsic::aarch64_neon_sshl;
1906 Value *Result =
1907 IsSigned ? B.CreateAShr(Arg0, NegAmt) : B.CreateLShr(Arg0, NegAmt);
1908 return IC.replaceInstUsesWith(*II, Result);
1909}
1910
1911// If II is llvm.sin(x) or llvm.cos(x), and there is a matching
1912// llvm.cos(x) or llvm.sin(x) using the same argument, combine them
1913// into a single llvm.sincos(x) call. Returns the result for II
1914// extracted from sincos, or nullptr if no match is found.
1916 InstCombinerImpl &IC) {
1917 Intrinsic::ID IID = II->getIntrinsicID();
1918 bool IsSin = IID == Intrinsic::sin;
1919 Intrinsic::ID MatchID = IsSin ? Intrinsic::cos : Intrinsic::sin;
1920
1921 Value *Arg = II->getArgOperand(0);
1922
1923 // Don't bother looking through uses of constants.
1924 if (isa<Constant>(Arg))
1925 return nullptr;
1926
1927 // Look for a matching cos/sin intrinsic with the same argument.
1928 IntrinsicInst *Match = nullptr;
1929 for (User *U : Arg->users()) {
1930 if (auto *Cand = dyn_cast<IntrinsicInst>(U)) {
1931 if (Cand != II && !Cand->use_empty() &&
1932 Cand->getIntrinsicID() == MatchID) {
1933 Match = Cand;
1934 break;
1935 }
1936 }
1937 }
1938
1939 if (!Match)
1940 return nullptr;
1941
1942 // Insert sincos right after the argument definition.
1944 if (auto *ArgInst = dyn_cast<Instruction>(Arg)) {
1945 std::optional<BasicBlock::iterator> InsertPt =
1946 ArgInst->getInsertionPointAfterDef();
1947 if (!InsertPt)
1948 return nullptr;
1949 B.SetInsertPoint(*InsertPt);
1950 } else {
1951 BasicBlock &EntryBB = II->getFunction()->getEntryBlock();
1952 B.SetInsertPoint(&EntryBB, EntryBB.begin());
1953 }
1954
1956 II->getModule(), Intrinsic::sincos, Arg->getType());
1957 CallInst *SinCos = B.CreateCall(SinCosFunc, Arg, "sincos");
1958 // Intersect fast-math flags from the two calls.
1959 SinCos->setFastMathFlags(II->getFastMathFlags() & Match->getFastMathFlags());
1960 // Propagate the most-generic fpmath metadata from the two original calls.
1962 II->getMetadata(LLVMContext::MD_fpmath),
1963 Match->getMetadata(LLVMContext::MD_fpmath)))
1964 SinCos->setMetadata(LLVMContext::MD_fpmath, MD);
1965 Value *Sin = B.CreateExtractValue(SinCos, 0, "sin");
1966 Value *Cos = B.CreateExtractValue(SinCos, 1, "cos");
1967
1968 // Replace the matching call and erase it.
1969 IC.replaceInstUsesWith(*Match, IsSin ? Cos : Sin);
1970 IC.eraseInstFromFunction(*Match);
1971 return IsSin ? Sin : Cos;
1972}
1973
1974/// Fold an scmp/ucmp intrinsic whose operands are extended from a narrower
1975/// type:
1976/// scmp (sext X), (sext Y) --> scmp X, Y
1977/// scmp (zext X), (zext Y) --> ucmp X, Y
1978/// ucmp (ext X), (ext Y) --> ucmp X, Y
1979/// Both operands must use the same extend opcode and source type. A constant
1980/// operand is narrowed instead, if truncating and re-extending it gives back
1981/// the same constant.
1983 InstCombiner::BuilderTy &Builder,
1984 const DataLayout &DL) {
1985 // scmp/ucmp are not commutative, so the extend may be on either side.
1986 unsigned ExtIdx = 0;
1987 Value *X;
1988 if (!match(II->getArgOperand(0), m_ZExtOrSExt(m_Value(X)))) {
1989 ExtIdx = 1;
1990 if (!match(II->getArgOperand(1), m_ZExtOrSExt(m_Value(X))))
1991 return nullptr;
1992 }
1993
1994 auto CastOpc = static_cast<Instruction::CastOps>(
1995 cast<Operator>(II->getArgOperand(ExtIdx))->getOpcode());
1996 Type *NarrowTy = X->getType();
1997
1998 // The other operand must be the same kind of extend from the same type, or a
1999 // constant that can be narrowed losslessly.
2000 Value *OtherOp = II->getArgOperand(1 - ExtIdx);
2001 Value *Y;
2002 Constant *WideC;
2003 if (match(OtherOp, m_ZExtOrSExt(m_Value(Y)))) {
2004 if (cast<Operator>(OtherOp)->getOpcode() != CastOpc ||
2005 Y->getType() != NarrowTy)
2006 return nullptr;
2007 } else if (match(OtherOp, m_ImmConstant(WideC))) {
2008 Y = getLosslessInvCast(WideC, NarrowTy, CastOpc, DL);
2009 if (!Y)
2010 return nullptr;
2011 } else {
2012 return nullptr;
2013 }
2014
2015 // Both extends preserve the unsigned order, so an unsigned compare of the
2016 // narrow operands is always equivalent. The signed order is only preserved by
2017 // sext; zero extended values are non-negative, so a signed compare of those
2018 // is an unsigned compare of the narrow operands.
2019 Intrinsic::ID NewIID =
2020 II->getIntrinsicID() == Intrinsic::scmp && CastOpc == Instruction::SExt
2021 ? Intrinsic::scmp
2022 : Intrinsic::ucmp;
2023 if (ExtIdx != 0)
2024 std::swap(X, Y);
2025 return Builder.CreateIntrinsic(II->getType(), NewIID, {X, Y});
2026}
2027
2028/// CallInst simplification. This mostly only handles folding of intrinsic
2029/// instructions. For normal calls, it allows visitCallBase to do the heavy
2030/// lifting.
2032 // Don't try to simplify calls without uses. It will not do anything useful,
2033 // but will result in the following folds being skipped.
2034 if (!CI.use_empty()) {
2035 SmallVector<Value *, 8> Args(CI.args());
2036 if (Value *V = simplifyCall(&CI, CI.getCalledOperand(), Args,
2037 SQ.getWithInstruction(&CI)))
2038 return replaceInstUsesWith(CI, V);
2039 }
2040
2041 if (Value *FreedOp = getFreedOperand(&CI, &TLI))
2042 return visitFree(CI, FreedOp);
2043
2044 // If the caller function (i.e. us, the function that contains this CallInst)
2045 // is nounwind, mark the call as nounwind, even if the callee isn't.
2046 if (CI.getFunction()->doesNotThrow() && !CI.doesNotThrow()) {
2047 CI.setDoesNotThrow();
2048 return &CI;
2049 }
2050
2052 if (!II)
2053 return visitCallBase(CI);
2054
2055 // Intrinsics cannot occur in an invoke or a callbr, so handle them here
2056 // instead of in visitCallBase.
2057 if (auto *MI = dyn_cast<AnyMemIntrinsic>(II)) {
2058 if (auto NumBytes = MI->getLengthInBytes()) {
2059 // memmove/cpy/set of zero bytes is a noop.
2060 if (NumBytes->isZero())
2061 return eraseInstFromFunction(CI);
2062
2063 // For atomic unordered mem intrinsics if len is not a positive or
2064 // not a multiple of element size then behavior is undefined.
2065 if (MI->isAtomic() &&
2066 (NumBytes->isNegative() ||
2067 (NumBytes->getZExtValue() % MI->getElementSizeInBytes() != 0))) {
2069 assert(MI->getType()->isVoidTy() &&
2070 "non void atomic unordered mem intrinsic");
2071 return eraseInstFromFunction(*MI);
2072 }
2073 }
2074
2075 // No other transformations apply to volatile transfers.
2076 if (MI->isVolatile())
2077 return nullptr;
2078
2080 // memmove(x,x,size) -> noop.
2081 if (MTI->getSource() == MTI->getDest())
2082 return eraseInstFromFunction(CI);
2083 }
2084
2085 auto IsPointerUndefined = [MI](Value *Ptr) {
2086 return isa<ConstantPointerNull>(Ptr) &&
2088 MI->getFunction(),
2089 cast<PointerType>(Ptr->getType())->getAddressSpace());
2090 };
2091 bool SrcIsUndefined = false;
2092 // If we can determine a pointer alignment that is bigger than currently
2093 // set, update the alignment.
2094 if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
2096 return I;
2097 SrcIsUndefined = IsPointerUndefined(MTI->getRawSource());
2098 } else if (auto *MSI = dyn_cast<AnyMemSetInst>(MI)) {
2099 if (Instruction *I = SimplifyAnyMemSet(MSI))
2100 return I;
2101 }
2102
2103 // If src/dest is null, this memory intrinsic must be a noop.
2104 if (SrcIsUndefined || IsPointerUndefined(MI->getRawDest())) {
2105 Builder.CreateAssumption(Builder.CreateIsNull(MI->getLength()));
2106 return eraseInstFromFunction(CI);
2107 }
2108
2109 // If we have a memmove and the source operation is a constant global,
2110 // then the source and dest pointers can't alias, so we can change this
2111 // into a call to memcpy.
2112 if (auto *MMI = dyn_cast<AnyMemMoveInst>(MI)) {
2113 if (GlobalVariable *GVSrc = dyn_cast<GlobalVariable>(MMI->getSource()))
2114 if (GVSrc->isConstant()) {
2115 Module *M = CI.getModule();
2116 Intrinsic::ID MemCpyID =
2117 MMI->isAtomic()
2118 ? Intrinsic::memcpy_element_unordered_atomic
2119 : Intrinsic::memcpy;
2120 Type *Tys[3] = { CI.getArgOperand(0)->getType(),
2121 CI.getArgOperand(1)->getType(),
2122 CI.getArgOperand(2)->getType() };
2124 Intrinsic::getOrInsertDeclaration(M, MemCpyID, Tys));
2125 return II;
2126 }
2127 }
2128 }
2129
2130 // For fixed width vector result intrinsics, use the generic demanded vector
2131 // support.
2132 if (auto *IIFVTy = dyn_cast<FixedVectorType>(II->getType())) {
2133 auto VWidth = IIFVTy->getNumElements();
2134 APInt PoisonElts(VWidth, 0);
2135 APInt AllOnesEltMask(APInt::getAllOnes(VWidth));
2136 if (Value *V = SimplifyDemandedVectorElts(II, AllOnesEltMask, PoisonElts)) {
2137 if (V != II)
2138 return replaceInstUsesWith(*II, V);
2139 return II;
2140 }
2141 }
2142
2143 if (II->isCommutative()) {
2144 if (auto Pair = matchSymmetricPair(II->getOperand(0), II->getOperand(1))) {
2145 replaceOperand(*II, 0, Pair->first);
2146 replaceOperand(*II, 1, Pair->second);
2147 II->dropPoisonGeneratingAnnotations();
2148 II->dropUBImplyingAttrsAndMetadata();
2149 return II;
2150 }
2151
2152 if (CallInst *NewCall = canonicalizeConstantArg0ToArg1(CI))
2153 return NewCall;
2154 }
2155
2156 // Unused constrained FP intrinsic calls may have declared side effect, which
2157 // prevents it from being removed. In some cases however the side effect is
2158 // actually absent. To detect this case, call SimplifyConstrainedFPCall. If it
2159 // returns a replacement, the call may be removed.
2160 if (CI.use_empty() && isa<ConstrainedFPIntrinsic>(CI)) {
2161 if (simplifyConstrainedFPCall(&CI, SQ.getWithInstruction(&CI)))
2162 return eraseInstFromFunction(CI);
2163 }
2164
2165 Intrinsic::ID IID = II->getIntrinsicID();
2166 switch (IID) {
2167 case Intrinsic::objectsize: {
2168 SmallVector<Instruction *> InsertedInstructions;
2169 if (Value *V = lowerObjectSizeCall(II, DL, &TLI, AA, /*MustSucceed=*/false,
2170 &InsertedInstructions)) {
2171 for (Instruction *Inserted : InsertedInstructions)
2172 Worklist.add(Inserted);
2173 return replaceInstUsesWith(CI, V);
2174 }
2175 return nullptr;
2176 }
2177 case Intrinsic::abs: {
2178 Value *IIOperand = II->getArgOperand(0);
2179 bool IntMinIsPoison = cast<Constant>(II->getArgOperand(1))->isOneValue();
2180
2181 // abs(-x) -> abs(x)
2182 Value *X;
2183 if (match(IIOperand, m_Neg(m_Value(X))))
2184 return CallInst::Create(
2185 II->getCalledFunction(),
2186 {X,
2187 Builder.getInt1(IntMinIsPoison ||
2188 cast<Instruction>(IIOperand)->hasNoSignedWrap())});
2189
2190 if (match(IIOperand, m_c_Select(m_Neg(m_Value(X)), m_Deferred(X))))
2191 return CallInst::Create(II->getCalledFunction(),
2192 {X, II->getArgOperand(1)});
2193
2194 Value *Y;
2195 // abs(a * abs(b)) -> abs(a * b)
2196 if (match(IIOperand,
2199 bool NSW =
2200 cast<Instruction>(IIOperand)->hasNoSignedWrap() && IntMinIsPoison;
2201 auto *XY = NSW ? Builder.CreateNSWMul(X, Y) : Builder.CreateMul(X, Y);
2202 return CallInst::Create(II->getCalledFunction(),
2203 {XY, II->getArgOperand(1)});
2204 }
2205
2206 if (std::optional<bool> Known =
2207 getKnownSignOrZero(IIOperand, SQ.getWithInstruction(II))) {
2208 // abs(x) -> x if x >= 0 (include abs(x-y) --> x - y where x >= y)
2209 // abs(x) -> x if x > 0 (include abs(x-y) --> x - y where x > y)
2210 if (!*Known)
2211 return replaceInstUsesWith(*II, IIOperand);
2212
2213 // abs(x) -> -x if x < 0
2214 // abs(x) -> -x if x < = 0 (include abs(x-y) --> y - x where x <= y)
2215 if (IntMinIsPoison)
2216 return BinaryOperator::CreateNSWNeg(IIOperand);
2217 return BinaryOperator::CreateNeg(IIOperand);
2218 }
2219
2220 // abs (sext X) --> zext (abs X*)
2221 // Clear the IsIntMin (nsw) bit on the abs to allow narrowing.
2222 if (match(IIOperand, m_OneUse(m_SExt(m_Value(X))))) {
2223 Value *NarrowAbs =
2224 Builder.CreateBinaryIntrinsic(Intrinsic::abs, X, Builder.getFalse());
2225 return CastInst::Create(Instruction::ZExt, NarrowAbs, II->getType());
2226 }
2227
2228 // Match a complicated way to check if a number is odd/even:
2229 // abs (srem X, 2) --> and X, 1
2230 const APInt *C;
2231 if (match(IIOperand, m_SRem(m_Value(X), m_APInt(C))) && *C == 2)
2232 return BinaryOperator::CreateAnd(X, ConstantInt::get(II->getType(), 1));
2233
2234 break;
2235 }
2236 case Intrinsic::umin: {
2237 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2238 // umin(x, 1) == zext(x != 0)
2239 if (match(I1, m_One())) {
2240 assert(II->getType()->getScalarSizeInBits() != 1 &&
2241 "Expected simplify of umin with max constant");
2242 Value *Zero = Constant::getNullValue(I0->getType());
2243 Value *Cmp = Builder.CreateICmpNE(I0, Zero);
2244 return CastInst::Create(Instruction::ZExt, Cmp, II->getType());
2245 }
2246 // umin(cttz(x), const) --> cttz(x | (1 << const))
2247 if (Value *FoldedCttz =
2249 I0, I1, DL, Builder))
2250 return replaceInstUsesWith(*II, FoldedCttz);
2251 // umin(ctlz(x), const) --> ctlz(x | (SignedMin >> const))
2252 if (Value *FoldedCtlz =
2254 I0, I1, DL, Builder))
2255 return replaceInstUsesWith(*II, FoldedCtlz);
2256 [[fallthrough]];
2257 }
2258 case Intrinsic::umax: {
2259 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2260 Value *X, *Y;
2261 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_ZExt(m_Value(Y))) &&
2262 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2263 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2264 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2265 }
2266 Constant *C;
2267 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2268 I0->hasOneUse()) {
2269 if (Constant *NarrowC = getLosslessUnsignedTrunc(C, X->getType(), DL)) {
2270 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2271 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2272 }
2273 }
2274 // If C is not 0:
2275 // umax(nuw_shl(x, C), x + 1) -> x == 0 ? 1 : nuw_shl(x, C)
2276 // If C is not 0 or 1:
2277 // umax(nuw_mul(x, C), x + 1) -> x == 0 ? 1 : nuw_mul(x, C)
2278 auto foldMaxMulShift = [&](Value *A, Value *B) -> Instruction * {
2279 const APInt *C;
2280 Value *X;
2281 if (!match(A, m_NUWShl(m_Value(X), m_APInt(C))) &&
2282 !(match(A, m_NUWMul(m_Value(X), m_APInt(C))) && !C->isOne()))
2283 return nullptr;
2284 if (C->isZero())
2285 return nullptr;
2286 if (!match(B, m_OneUse(m_Add(m_Specific(X), m_One()))))
2287 return nullptr;
2288
2289 Value *Cmp = Builder.CreateICmpEQ(X, ConstantInt::get(X->getType(), 0));
2290 Value *NewSelect = nullptr;
2291 NewSelect = Builder.CreateSelectWithUnknownProfile(
2292 Cmp, ConstantInt::get(X->getType(), 1), A, DEBUG_TYPE);
2293 return replaceInstUsesWith(*II, NewSelect);
2294 };
2295
2296 if (IID == Intrinsic::umax) {
2297 if (Instruction *I = foldMaxMulShift(I0, I1))
2298 return I;
2299 if (Instruction *I = foldMaxMulShift(I1, I0))
2300 return I;
2301 }
2302
2303 // If both operands of unsigned min/max are sign-extended, it is still ok
2304 // to narrow the operation.
2305 [[fallthrough]];
2306 }
2307 case Intrinsic::smax:
2308 case Intrinsic::smin: {
2309 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2310 Value *X, *Y;
2311 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_SExt(m_Value(Y))) &&
2312 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2313 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2314 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2315 }
2316
2317 Constant *C;
2318 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2319 I0->hasOneUse()) {
2320 if (Constant *NarrowC = getLosslessSignedTrunc(C, X->getType(), DL)) {
2321 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2322 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2323 }
2324 }
2325
2326 // smax(smin(X, MinC), MaxC) -> smin(smax(X, MaxC), MinC) if MinC s>= MaxC
2327 // umax(umin(X, MinC), MaxC) -> umin(umax(X, MaxC), MinC) if MinC u>= MaxC
2328 const APInt *MinC, *MaxC;
2329 auto CreateCanonicalClampForm = [&](bool IsSigned) {
2330 auto MaxIID = IsSigned ? Intrinsic::smax : Intrinsic::umax;
2331 auto MinIID = IsSigned ? Intrinsic::smin : Intrinsic::umin;
2332 Value *NewMax = Builder.CreateBinaryIntrinsic(
2333 MaxIID, X, ConstantInt::get(X->getType(), *MaxC));
2334 return replaceInstUsesWith(
2335 *II, Builder.CreateBinaryIntrinsic(
2336 MinIID, NewMax, ConstantInt::get(X->getType(), *MinC)));
2337 };
2338 if (IID == Intrinsic::smax &&
2340 m_APInt(MinC)))) &&
2341 match(I1, m_APInt(MaxC)) && MinC->sgt(*MaxC))
2342 return CreateCanonicalClampForm(true);
2343 if (IID == Intrinsic::umax &&
2345 m_APInt(MinC)))) &&
2346 match(I1, m_APInt(MaxC)) && MinC->ugt(*MaxC))
2347 return CreateCanonicalClampForm(false);
2348
2349 // umin(i1 X, i1 Y) -> and i1 X, Y
2350 // smax(i1 X, i1 Y) -> and i1 X, Y
2351 if ((IID == Intrinsic::umin || IID == Intrinsic::smax) &&
2352 II->getType()->isIntOrIntVectorTy(1)) {
2353 return BinaryOperator::CreateAnd(I0, I1);
2354 }
2355
2356 // umax(i1 X, i1 Y) -> or i1 X, Y
2357 // smin(i1 X, i1 Y) -> or i1 X, Y
2358 if ((IID == Intrinsic::umax || IID == Intrinsic::smin) &&
2359 II->getType()->isIntOrIntVectorTy(1)) {
2360 return BinaryOperator::CreateOr(I0, I1);
2361 }
2362
2363 // smin(smax(X, -1), 1) -> scmp(X, 0)
2364 // smax(smin(X, 1), -1) -> scmp(X, 0)
2365 // At this point, smax(smin(X, 1), -1) is changed to smin(smax(X, -1)
2366 // And i1's have been changed to and/ors
2367 // So we only need to check for smin
2368 if (IID == Intrinsic::smin) {
2369 if (match(I0, m_OneUse(m_SMax(m_Value(X), m_AllOnes()))) &&
2370 match(I1, m_One())) {
2371 Value *Zero = ConstantInt::get(X->getType(), 0);
2372 return replaceInstUsesWith(
2373 CI,
2374 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {X, Zero}));
2375 }
2376 }
2377
2378 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2379 // smax (neg nsw X), (neg nsw Y) --> neg nsw (smin X, Y)
2380 // smin (neg nsw X), (neg nsw Y) --> neg nsw (smax X, Y)
2381 // TODO: Canonicalize neg after min/max if I1 is constant.
2382 if (match(I0, m_NSWNeg(m_Value(X))) && match(I1, m_NSWNeg(m_Value(Y))) &&
2383 (I0->hasOneUse() || I1->hasOneUse())) {
2385 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, X, Y);
2386 return BinaryOperator::CreateNSWNeg(InvMaxMin);
2387 }
2388 }
2389
2390 // (umax X, (xor X, Pow2))
2391 // -> (or X, Pow2)
2392 // (umin X, (xor X, Pow2))
2393 // -> (and X, ~Pow2)
2394 // (smax X, (xor X, Pos_Pow2))
2395 // -> (or X, Pos_Pow2)
2396 // (smin X, (xor X, Pos_Pow2))
2397 // -> (and X, ~Pos_Pow2)
2398 // (smax X, (xor X, Neg_Pow2))
2399 // -> (and X, ~Neg_Pow2)
2400 // (smin X, (xor X, Neg_Pow2))
2401 // -> (or X, Neg_Pow2)
2402 if ((match(I0, m_c_Xor(m_Specific(I1), m_Value(X))) ||
2403 match(I1, m_c_Xor(m_Specific(I0), m_Value(X)))) &&
2404 isKnownToBeAPowerOfTwo(X, /* OrZero */ true)) {
2405 bool UseOr = IID == Intrinsic::smax || IID == Intrinsic::umax;
2406 bool UseAndN = IID == Intrinsic::smin || IID == Intrinsic::umin;
2407
2408 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2409 auto KnownSign = getKnownSign(X, SQ.getWithInstruction(II));
2410 if (KnownSign == std::nullopt) {
2411 UseOr = false;
2412 UseAndN = false;
2413 } else if (*KnownSign /* true is Signed. */) {
2414 UseOr ^= true;
2415 UseAndN ^= true;
2416 Type *Ty = I0->getType();
2417 // Negative power of 2 must be IntMin. It's possible to be able to
2418 // prove negative / power of 2 without actually having known bits, so
2419 // just get the value by hand.
2421 Ty, APInt::getSignedMinValue(Ty->getScalarSizeInBits()));
2422 }
2423 }
2424 if (UseOr)
2425 return BinaryOperator::CreateOr(I0, X);
2426 else if (UseAndN)
2427 return BinaryOperator::CreateAnd(I0, Builder.CreateNot(X));
2428 }
2429
2430 // If we can eliminate ~A and Y is free to invert:
2431 // max ~A, Y --> ~(min A, ~Y)
2432 //
2433 // Examples:
2434 // max ~A, ~Y --> ~(min A, Y)
2435 // max ~A, C --> ~(min A, ~C)
2436 // max ~A, (max ~Y, ~Z) --> ~min( A, (min Y, Z))
2437 auto moveNotAfterMinMax = [&](Value *X, Value *Y) -> Instruction * {
2438 Value *A;
2439 if (match(X, m_OneUse(m_Not(m_Value(A)))) &&
2440 !isFreeToInvert(A, A->hasOneUse())) {
2441 if (Value *NotY = getFreelyInverted(Y, Y->hasOneUse(), &Builder)) {
2443 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, A, NotY);
2444 return BinaryOperator::CreateNot(InvMaxMin);
2445 }
2446 }
2447 return nullptr;
2448 };
2449
2450 if (Instruction *I = moveNotAfterMinMax(I0, I1))
2451 return I;
2452 if (Instruction *I = moveNotAfterMinMax(I1, I0))
2453 return I;
2454
2456 return I;
2457
2458 // minmax (X & NegPow2C, Y & NegPow2C) --> minmax(X, Y) & NegPow2C
2459 const APInt *RHSC;
2460 if (match(I0, m_OneUse(m_And(m_Value(X), m_NegatedPower2(RHSC)))) &&
2461 match(I1, m_OneUse(m_And(m_Value(Y), m_SpecificInt(*RHSC)))))
2462 return BinaryOperator::CreateAnd(Builder.CreateBinaryIntrinsic(IID, X, Y),
2463 ConstantInt::get(II->getType(), *RHSC));
2464
2465 // smax(X, -X) --> abs(X)
2466 // smin(X, -X) --> -abs(X)
2467 // umax(X, -X) --> -abs(X)
2468 // umin(X, -X) --> abs(X)
2469 if (isKnownNegation(I0, I1)) {
2470 // We can choose either operand as the input to abs(), but if we can
2471 // eliminate the only use of a value, that's better for subsequent
2472 // transforms/analysis.
2473 if (I0->hasOneUse() && !I1->hasOneUse())
2474 std::swap(I0, I1);
2475
2476 // This is some variant of abs(). See if we can propagate 'nsw' to the abs
2477 // operation and potentially its negation.
2478 bool IntMinIsPoison = isKnownNegation(I0, I1, /* NeedNSW */ true);
2479 Value *Abs = Builder.CreateBinaryIntrinsic(
2480 Intrinsic::abs, I0,
2481 ConstantInt::getBool(II->getContext(), IntMinIsPoison));
2482
2483 // We don't have a "nabs" intrinsic, so negate if needed based on the
2484 // max/min operation.
2485 if (IID == Intrinsic::smin || IID == Intrinsic::umax)
2486 Abs = Builder.CreateNeg(Abs, "nabs", IntMinIsPoison);
2487 return replaceInstUsesWith(CI, Abs);
2488 }
2489
2491 return Sel;
2492
2493 if (Instruction *SAdd = matchSAddSubSat(*II))
2494 return SAdd;
2495
2496 if (Value *NewMinMax = reassociateMinMaxWithConstants(II, Builder, SQ))
2497 return replaceInstUsesWith(*II, NewMinMax);
2498
2500 return R;
2501
2502 if (Instruction *NewMinMax = factorizeMinMaxTree(II))
2503 return NewMinMax;
2504
2505 // Try to fold minmax with constant RHS based on range information
2506 if (match(I1, m_APIntAllowPoison(RHSC))) {
2507 ICmpInst::Predicate Pred =
2509 bool IsSigned = MinMaxIntrinsic::isSigned(IID);
2511 I0, IsSigned, SQ.getWithInstruction(II));
2512 if (!LHS_CR.isFullSet()) {
2513 if (LHS_CR.icmp(Pred, *RHSC))
2514 return replaceInstUsesWith(*II, I0);
2515 if (LHS_CR.icmp(ICmpInst::getSwappedPredicate(Pred), *RHSC))
2516 return replaceInstUsesWith(*II,
2517 ConstantInt::get(II->getType(), *RHSC));
2518 }
2519 }
2520
2522 return replaceInstUsesWith(*II, V);
2523
2524 break;
2525 }
2526 case Intrinsic::scmp:
2527 case Intrinsic::ucmp: {
2529 return replaceInstUsesWith(CI, V);
2530
2531 if (IID == Intrinsic::ucmp)
2532 break;
2533
2534 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2535
2536 // scmp(X, 0) -> sext_or_trunc(X) if X is known to be one of -1, 0, 1.
2537 if (match(I1, m_Zero())) {
2538 ConstantRange Range = computeConstantRange(I0, /*ForSigned=*/true,
2539 SQ.getWithInstruction(II));
2540 if (Range.getSignedMin().sge(-1) && Range.getSignedMax().sle(1))
2541 return replaceInstUsesWith(
2542 CI, Builder.CreateSExtOrTrunc(I0, II->getType()));
2543 }
2544 Value *LHS, *RHS;
2545 if (match(I0, m_NSWSub(m_Value(LHS), m_Value(RHS))) && match(I1, m_Zero()))
2546 return replaceInstUsesWith(
2547 CI,
2548 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {LHS, RHS}));
2549 break;
2550 }
2551 case Intrinsic::bitreverse: {
2552 Value *IIOperand = II->getArgOperand(0);
2553 // bitrev (zext i1 X to ?) --> X ? SignBitC : 0
2554 Value *X;
2555 if (match(IIOperand, m_ZExt(m_Value(X))) &&
2556 X->getType()->isIntOrIntVectorTy(1)) {
2557 Type *Ty = II->getType();
2558 APInt SignBit = APInt::getSignMask(Ty->getScalarSizeInBits());
2559 return SelectInst::Create(X, ConstantInt::get(Ty, SignBit),
2561 }
2562
2563 if (Instruction *crossLogicOpFold =
2565 return crossLogicOpFold;
2566
2567 break;
2568 }
2569 case Intrinsic::bswap: {
2570 Value *IIOperand = II->getArgOperand(0);
2571
2572 // Try to canonicalize bswap-of-logical-shift-by-8-bit-multiple as
2573 // inverse-shift-of-bswap:
2574 // bswap (shl X, Y) --> lshr (bswap X), Y
2575 // bswap (lshr X, Y) --> shl (bswap X), Y
2576 Value *X, *Y;
2577 if (match(IIOperand, m_OneUse(m_LogicalShift(m_Value(X), m_Value(Y))))) {
2578 unsigned BitWidth = IIOperand->getType()->getScalarSizeInBits();
2580 Value *NewSwap = Builder.CreateUnaryIntrinsic(Intrinsic::bswap, X);
2581 BinaryOperator::BinaryOps InverseShift =
2582 cast<BinaryOperator>(IIOperand)->getOpcode() == Instruction::Shl
2583 ? Instruction::LShr
2584 : Instruction::Shl;
2585 return BinaryOperator::Create(InverseShift, NewSwap, Y);
2586 }
2587 }
2588
2589 KnownBits Known = computeKnownBits(IIOperand, II);
2590 uint64_t LZ = alignDown(Known.countMinLeadingZeros(), 8);
2591 uint64_t TZ = alignDown(Known.countMinTrailingZeros(), 8);
2592 unsigned BW = Known.getBitWidth();
2593
2594 // bswap(x) -> shift(x) if x has exactly one "active byte"
2595 if (BW - LZ - TZ == 8) {
2596 assert(LZ != TZ && "active byte cannot be in the middle");
2597 if (LZ > TZ) // -> shl(x) if the "active byte" is in the low part of x
2598 return BinaryOperator::CreateNUWShl(
2599 IIOperand, ConstantInt::get(IIOperand->getType(), LZ - TZ));
2600 // -> lshr(x) if the "active byte" is in the high part of x
2601 return BinaryOperator::CreateExactLShr(
2602 IIOperand, ConstantInt::get(IIOperand->getType(), TZ - LZ));
2603 }
2604
2605 // bswap(trunc(bswap(x))) -> trunc(lshr(x, c))
2606 if (match(IIOperand, m_Trunc(m_BSwap(m_Value(X))))) {
2607 unsigned C = X->getType()->getScalarSizeInBits() - BW;
2608 Value *CV = ConstantInt::get(X->getType(), C);
2609 Value *V = Builder.CreateLShr(X, CV);
2610 return new TruncInst(V, IIOperand->getType());
2611 }
2612
2613 if (Instruction *crossLogicOpFold =
2615 return crossLogicOpFold;
2616 }
2617
2618 // Try to fold into bitreverse if bswap is the root of the expression tree.
2619 if (Instruction *BitOp = matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ false,
2620 /*MatchBitReversals*/ true))
2621 return BitOp;
2622 break;
2623 }
2624 case Intrinsic::masked_load:
2625 if (Value *SimplifiedMaskedOp = simplifyMaskedLoad(*II))
2626 return replaceInstUsesWith(CI, SimplifiedMaskedOp);
2627 break;
2628 case Intrinsic::masked_store:
2629 return simplifyMaskedStore(*II);
2630 case Intrinsic::masked_gather:
2631 return simplifyMaskedGather(*II);
2632 case Intrinsic::masked_scatter:
2633 return simplifyMaskedScatter(*II);
2634 case Intrinsic::launder_invariant_group:
2635 if (auto *SkippedBarrier = simplifyInvariantGroupIntrinsic(*II, *this))
2636 return replaceInstUsesWith(*II, SkippedBarrier);
2637 break;
2638 case Intrinsic::powi: {
2639 if (ConstantInt *Power = dyn_cast<ConstantInt>(II->getArgOperand(1))) {
2640 // 0 and 1 are handled in instsimplify
2641 // powi(x, -1) -> 1/x
2642 if (Power->isMinusOne())
2643 return BinaryOperator::CreateFDivFMF(ConstantFP::get(CI.getType(), 1.0),
2644 II->getArgOperand(0), II);
2645 // powi(x, 2) -> x*x
2646 if (Power->equalsInt(2))
2647 return BinaryOperator::CreateFMulFMF(II->getArgOperand(0),
2648 II->getArgOperand(0), II);
2649
2650 if (!Power->getValue()[0]) {
2651 Value *X;
2652 // If power is even:
2653 // powi(-x, p) -> powi(x, p)
2654 // powi(fabs(x), p) -> powi(x, p)
2655 // powi(copysign(x, y), p) -> powi(x, p)
2656 if (match(II->getArgOperand(0), m_FNeg(m_Value(X))) ||
2657 match(II->getArgOperand(0), m_FAbs(m_Value(X))) ||
2658 match(II->getArgOperand(0),
2660 return CallInst::Create(II->getCalledFunction(), {X, Power});
2661 }
2662 }
2663 if (ConstantFP *Base = dyn_cast<ConstantFP>(II->getArgOperand(0))) {
2664 Value *Exp = II->getArgOperand(1);
2665 Type *Ty = Base->getType();
2666 // powi(2.0, p) -> ldexp(1.0, p)
2667 if (II->hasApproxFunc() && Base->isExactlyValue(2.0)) {
2668 ConstantFP *One = ConstantFP::get(Ty, 1.0);
2669 if (auto *VTy = dyn_cast<VectorType>(Ty))
2670 Exp = Builder.CreateVectorSplat(VTy->getElementCount(), Exp);
2671 Value *Ldexp = Builder.CreateLdexp(One, Exp, II);
2672 return replaceInstUsesWith(*II, Ldexp);
2673 }
2674 }
2675 break;
2676 }
2677
2678 case Intrinsic::cttz:
2679 case Intrinsic::ctlz:
2680 if (auto *I = foldCttzCtlz(*II, *this))
2681 return I;
2682 break;
2683
2684 case Intrinsic::ctpop:
2685 if (auto *I = foldCtpop(*II, *this))
2686 return I;
2687 break;
2688
2689 case Intrinsic::fshl:
2690 case Intrinsic::fshr: {
2691 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
2692 Type *Ty = II->getType();
2693 unsigned BitWidth = Ty->getScalarSizeInBits();
2694 Constant *ShAmtC;
2695 if (match(II->getArgOperand(2), m_ImmConstant(ShAmtC))) {
2696 // Canonicalize a shift amount constant operand to modulo the bit-width.
2697 Constant *WidthC = ConstantInt::get(Ty, BitWidth);
2698 Constant *ModuloC =
2699 ConstantFoldBinaryOpOperands(Instruction::URem, ShAmtC, WidthC, DL);
2700 if (!ModuloC)
2701 return nullptr;
2702 if (ModuloC != ShAmtC)
2703 return CallInst::Create(II->getCalledFunction(), {Op0, Op1, ModuloC});
2704
2706 ShAmtC, DL),
2707 m_One()) &&
2708 "Shift amount expected to be modulo bitwidth");
2709
2710 // Canonicalize funnel shift right by constant to funnel shift left. This
2711 // is not entirely arbitrary. For historical reasons, the backend may
2712 // recognize rotate left patterns but miss rotate right patterns.
2713 if (IID == Intrinsic::fshr) {
2714 // fshr X, Y, C --> fshl X, Y, (BitWidth - C) if C is not zero.
2715 if (!isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2716 return nullptr;
2717
2718 Constant *LeftShiftC = ConstantExpr::getSub(WidthC, ShAmtC);
2719 Module *Mod = II->getModule();
2720 Function *Fshl =
2721 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::fshl, Ty);
2722 return CallInst::Create(Fshl, { Op0, Op1, LeftShiftC });
2723 }
2724 assert(IID == Intrinsic::fshl &&
2725 "All funnel shifts by simple constants should go left");
2726
2727 // fshl(X, 0, C) --> shl X, C
2728 // fshl(X, undef, C) --> shl X, C
2729 if (match(Op1, m_ZeroInt()) || match(Op1, m_Undef()))
2730 return BinaryOperator::CreateShl(Op0, ShAmtC);
2731
2732 // fshl(0, X, C) --> lshr X, (BW-C)
2733 // fshl(undef, X, C) --> lshr X, (BW-C)
2734 // Similar to fshr -> fshl fold above, this is only valid if C is not zero
2735 if ((match(Op0, m_ZeroInt()) || match(Op0, m_Undef())) &&
2736 isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2737 return BinaryOperator::CreateLShr(Op1,
2738 ConstantExpr::getSub(WidthC, ShAmtC));
2739
2740 // fshl i16 X, X, 8 --> bswap i16 X (reduce to more-specific form)
2741 if (Op0 == Op1 && BitWidth == 16 && match(ShAmtC, m_SpecificInt(8))) {
2742 Module *Mod = II->getModule();
2743 Function *Bswap =
2744 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::bswap, Ty);
2745 return CallInst::Create(Bswap, { Op0 });
2746 }
2747 if (Instruction *BitOp =
2748 matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ true,
2749 /*MatchBitReversals*/ true))
2750 return BitOp;
2751
2752 // R = fshl(X, X, C2)
2753 // fshl(R, R, C1) --> fshl(X, X, (C1 + C2) % bitsize)
2754 Value *InnerOp;
2755 const APInt *ShAmtInnerC, *ShAmtOuterC;
2756 if (match(Op0, m_FShl(m_Value(InnerOp), m_Deferred(InnerOp),
2757 m_APInt(ShAmtInnerC))) &&
2758 match(ShAmtC, m_APInt(ShAmtOuterC)) && Op0 == Op1) {
2759 APInt Sum = *ShAmtOuterC + *ShAmtInnerC;
2760 APInt Modulo = Sum.urem(APInt(Sum.getBitWidth(), BitWidth));
2761 if (Modulo.isZero())
2762 return replaceInstUsesWith(*II, InnerOp);
2763 Constant *ModuloC = ConstantInt::get(Ty, Modulo);
2765 {InnerOp, InnerOp, ModuloC});
2766 }
2767 }
2768
2769 // fshl(X, X, Neg(Y)) --> fshr(X, X, Y)
2770 // fshr(X, X, Neg(Y)) --> fshl(X, X, Y)
2771 // if BitWidth is a power-of-2
2772 Value *Y;
2773 if (Op0 == Op1 && isPowerOf2_32(BitWidth) &&
2774 match(II->getArgOperand(2), m_Neg(m_Value(Y)))) {
2775 Module *Mod = II->getModule();
2777 Mod, IID == Intrinsic::fshl ? Intrinsic::fshr : Intrinsic::fshl, Ty);
2778 return CallInst::Create(OppositeShift, {Op0, Op1, Y});
2779 }
2780
2781 // fshl(X, 0, Y) --> shl(X, and(Y, BitWidth - 1)) if bitwidth is a
2782 // power-of-2
2783 if (IID == Intrinsic::fshl && isPowerOf2_32(BitWidth) &&
2784 match(Op1, m_ZeroInt())) {
2785 Value *Op2 = II->getArgOperand(2);
2786 Value *And = Builder.CreateAnd(Op2, ConstantInt::get(Ty, BitWidth - 1));
2787 return BinaryOperator::CreateShl(Op0, And);
2788 }
2789
2790 // Left or right might be masked.
2792 return &CI;
2793
2794 // The shift amount (operand 2) of a funnel shift is modulo the bitwidth,
2795 // so only the low bits of the shift amount are demanded if the bitwidth is
2796 // a power-of-2.
2797 if (!isPowerOf2_32(BitWidth))
2798 break;
2800 KnownBits Op2Known(BitWidth);
2801 if (SimplifyDemandedBits(II, 2, Op2Demanded, Op2Known))
2802 return &CI;
2803 break;
2804 }
2805 case Intrinsic::pdep: {
2806 const APInt *MaskC;
2807 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2808 unsigned MaskIdx, MaskLen;
2809 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2810 // any single contiguous sequence of 1s anywhere in the mask simply
2811 // describes a subset of the input bits shifted to the appropriate
2812 // position. Replace with the straight forward IR.
2813 Value *Input = II->getArgOperand(0);
2814 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2815 Value *Shifted = Builder.CreateShl(Input, ShiftAmt);
2816 Value *Masked = Builder.CreateAnd(Shifted, II->getArgOperand(1));
2817 return replaceInstUsesWith(*II, Masked);
2818 }
2819 }
2820 break;
2821 }
2822 case Intrinsic::pext: {
2823 const APInt *MaskC;
2824 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2825 unsigned MaskIdx, MaskLen;
2826 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2827 // any single contiguous sequence of 1s anywhere in the mask simply
2828 // describes a subset of the input bits shifted to the appropriate
2829 // position. Replace with the straight forward IR.
2830 Value *Input = II->getArgOperand(0);
2831 Value *Masked = Builder.CreateAnd(Input, II->getArgOperand(1));
2832 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2833 Value *Shifted = Builder.CreateLShr(Masked, ShiftAmt);
2834 return replaceInstUsesWith(*II, Shifted);
2835 }
2836 }
2837 break;
2838 }
2839 case Intrinsic::ptrmask: {
2840 unsigned BitWidth = DL.getPointerTypeSizeInBits(II->getType());
2843 return II;
2844
2845 Value *InnerPtr, *InnerMask;
2846 bool Changed = false;
2847 // Combine:
2848 // (ptrmask (ptrmask p, A), B)
2849 // -> (ptrmask p, (and A, B))
2850 if (match(II->getArgOperand(0),
2852 m_Value(InnerMask))))) {
2853 assert(II->getArgOperand(1)->getType() == InnerMask->getType() &&
2854 "Mask types must match");
2855 // TODO: If InnerMask == Op1, we could copy attributes from inner
2856 // callsite -> outer callsite.
2857 Value *NewMask = Builder.CreateAnd(II->getArgOperand(1), InnerMask);
2858 replaceOperand(CI, 0, InnerPtr);
2859 replaceOperand(CI, 1, NewMask);
2860 Changed = true;
2861 }
2862
2863 // See if we can deduce non-null.
2864 if (!CI.hasRetAttr(Attribute::NonNull) &&
2865 (Known.isNonZero() ||
2866 isKnownNonZero(II, getSimplifyQuery().getWithInstruction(II)))) {
2867 CI.addRetAttr(Attribute::NonNull);
2868 Changed = true;
2869 }
2870
2871 unsigned NewAlignmentLog =
2873 std::min(BitWidth - 1, Known.countMinTrailingZeros()));
2874 // Known bits will capture if we had alignment information associated with
2875 // the pointer argument.
2876 if (NewAlignmentLog > Log2(CI.getRetAlign().valueOrOne())) {
2878 CI.getContext(), Align(uint64_t(1) << NewAlignmentLog)));
2879 Changed = true;
2880 }
2881 if (Changed)
2882 return &CI;
2883 break;
2884 }
2885 case Intrinsic::uadd_with_overflow:
2886 case Intrinsic::sadd_with_overflow: {
2887 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2888 return I;
2889
2890 // Given 2 constant operands whose sum does not overflow:
2891 // uaddo (X +nuw C0), C1 -> uaddo X, C0 + C1
2892 // saddo (X +nsw C0), C1 -> saddo X, C0 + C1
2893 Value *X;
2894 const APInt *C0, *C1;
2895 Value *Arg0 = II->getArgOperand(0);
2896 Value *Arg1 = II->getArgOperand(1);
2897 bool IsSigned = IID == Intrinsic::sadd_with_overflow;
2898 bool HasNWAdd = IsSigned
2899 ? match(Arg0, m_NSWAddLike(m_Value(X), m_APInt(C0)))
2900 : match(Arg0, m_NUWAddLike(m_Value(X), m_APInt(C0)));
2901 if (HasNWAdd && match(Arg1, m_APInt(C1))) {
2902 bool Overflow;
2903 APInt NewC =
2904 IsSigned ? C1->sadd_ov(*C0, Overflow) : C1->uadd_ov(*C0, Overflow);
2905 if (!Overflow)
2906 return replaceInstUsesWith(
2907 *II, Builder.CreateBinaryIntrinsic(
2908 IID, X, ConstantInt::get(Arg1->getType(), NewC)));
2909 }
2910 break;
2911 }
2912
2913 case Intrinsic::umul_with_overflow:
2914 case Intrinsic::smul_with_overflow:
2915 case Intrinsic::usub_with_overflow:
2916 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2917 return I;
2918 break;
2919
2920 case Intrinsic::ssub_with_overflow: {
2921 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2922 return I;
2923
2924 Constant *C;
2925 Value *Arg0 = II->getArgOperand(0);
2926 Value *Arg1 = II->getArgOperand(1);
2927 // Given a constant C that is not the minimum signed value
2928 // for an integer of a given bit width:
2929 //
2930 // ssubo X, C -> saddo X, -C
2931 if (match(Arg1, m_Constant(C)) && C->isNotMinSignedValue()) {
2932 Value *NegVal = ConstantExpr::getNeg(C);
2933 // Build a saddo call that is equivalent to the discovered
2934 // ssubo call.
2935 return replaceInstUsesWith(
2936 *II, Builder.CreateBinaryIntrinsic(Intrinsic::sadd_with_overflow,
2937 Arg0, NegVal));
2938 }
2939
2940 break;
2941 }
2942
2943 case Intrinsic::uadd_sat:
2944 case Intrinsic::sadd_sat:
2945 case Intrinsic::usub_sat:
2946 case Intrinsic::ssub_sat: {
2948 Type *Ty = SI->getType();
2949 Value *Arg0 = SI->getLHS();
2950 Value *Arg1 = SI->getRHS();
2951
2952 // Make use of known overflow information.
2953 OverflowResult OR = computeOverflow(SI->getBinaryOp(), SI->isSigned(),
2954 Arg0, Arg1, SI);
2955 switch (OR) {
2957 break;
2959 if (SI->isSigned())
2960 return BinaryOperator::CreateNSW(SI->getBinaryOp(), Arg0, Arg1);
2961 else
2962 return BinaryOperator::CreateNUW(SI->getBinaryOp(), Arg0, Arg1);
2964 unsigned BitWidth = Ty->getScalarSizeInBits();
2965 APInt Min = APSInt::getMinValue(BitWidth, !SI->isSigned());
2966 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Min));
2967 }
2969 unsigned BitWidth = Ty->getScalarSizeInBits();
2970 APInt Max = APSInt::getMaxValue(BitWidth, !SI->isSigned());
2971 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Max));
2972 }
2973 }
2974
2975 // usub_sat((sub nuw C, A), C1) -> usub_sat(usub_sat(C, C1), A)
2976 // which after that:
2977 // usub_sat((sub nuw C, A), C1) -> usub_sat(C - C1, A) if C1 u< C
2978 // usub_sat((sub nuw C, A), C1) -> 0 otherwise
2979 Constant *C, *C1;
2980 Value *A;
2981 if (IID == Intrinsic::usub_sat &&
2982 match(Arg0, m_NUWSub(m_ImmConstant(C), m_Value(A))) &&
2983 match(Arg1, m_ImmConstant(C1))) {
2984 auto *NewC = Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, C, C1);
2985 auto *NewSub =
2986 Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, NewC, A);
2987 return replaceInstUsesWith(*SI, NewSub);
2988 }
2989
2990 // ssub.sat(X, C) -> sadd.sat(X, -C) if C != MIN
2991 if (IID == Intrinsic::ssub_sat && match(Arg1, m_Constant(C)) &&
2992 C->isNotMinSignedValue()) {
2993 Value *NegVal = ConstantExpr::getNeg(C);
2994 return replaceInstUsesWith(
2995 *II, Builder.CreateBinaryIntrinsic(
2996 Intrinsic::sadd_sat, Arg0, NegVal));
2997 }
2998
2999 // sat(sat(X + Val2) + Val) -> sat(X + (Val+Val2))
3000 // sat(sat(X - Val2) - Val) -> sat(X - (Val+Val2))
3001 // if Val and Val2 have the same sign
3002 if (auto *Other = dyn_cast<IntrinsicInst>(Arg0)) {
3003 Value *X;
3004 const APInt *Val, *Val2;
3005 APInt NewVal;
3006 bool IsUnsigned =
3007 IID == Intrinsic::uadd_sat || IID == Intrinsic::usub_sat;
3008 if (Other->getIntrinsicID() == IID &&
3009 match(Arg1, m_APInt(Val)) &&
3010 match(Other->getArgOperand(0), m_Value(X)) &&
3011 match(Other->getArgOperand(1), m_APInt(Val2))) {
3012 if (IsUnsigned)
3013 NewVal = Val->uadd_sat(*Val2);
3014 else if (Val->isNonNegative() == Val2->isNonNegative()) {
3015 bool Overflow;
3016 NewVal = Val->sadd_ov(*Val2, Overflow);
3017 if (Overflow) {
3018 // Both adds together may add more than SignedMaxValue
3019 // without saturating the final result.
3020 break;
3021 }
3022 } else {
3023 // Cannot fold saturated addition with different signs.
3024 break;
3025 }
3026
3027 return replaceInstUsesWith(
3028 *II, Builder.CreateBinaryIntrinsic(
3029 IID, X, ConstantInt::get(II->getType(), NewVal)));
3030 }
3031 }
3032 break;
3033 }
3034
3035 case Intrinsic::minnum:
3036 case Intrinsic::maxnum:
3037 case Intrinsic::minimumnum:
3038 case Intrinsic::maximumnum:
3039 case Intrinsic::minimum:
3040 case Intrinsic::maximum: {
3041 Value *Arg0 = II->getArgOperand(0);
3042 Value *Arg1 = II->getArgOperand(1);
3043 Value *X, *Y;
3044 if (match(Arg0, m_FNeg(m_Value(X))) && match(Arg1, m_FNeg(m_Value(Y))) &&
3045 (Arg0->hasOneUse() || Arg1->hasOneUse())) {
3046 // If both operands are negated, invert the call and negate the result:
3047 // min(-X, -Y) --> -(max(X, Y))
3048 // max(-X, -Y) --> -(min(X, Y))
3049 Intrinsic::ID NewIID;
3050 switch (IID) {
3051 case Intrinsic::maxnum:
3052 NewIID = Intrinsic::minnum;
3053 break;
3054 case Intrinsic::minnum:
3055 NewIID = Intrinsic::maxnum;
3056 break;
3057 case Intrinsic::maximumnum:
3058 NewIID = Intrinsic::minimumnum;
3059 break;
3060 case Intrinsic::minimumnum:
3061 NewIID = Intrinsic::maximumnum;
3062 break;
3063 case Intrinsic::maximum:
3064 NewIID = Intrinsic::minimum;
3065 break;
3066 case Intrinsic::minimum:
3067 NewIID = Intrinsic::maximum;
3068 break;
3069 default:
3070 llvm_unreachable("unexpected intrinsic ID");
3071 }
3072 Value *NewCall = Builder.CreateBinaryIntrinsic(NewIID, X, Y, II);
3073 Instruction *FNeg = UnaryOperator::CreateFNeg(NewCall);
3074 FNeg->copyIRFlags(II);
3075 return FNeg;
3076 }
3077
3078 // m(m(X, C2), C1) -> m(X, C)
3079 const APFloat *C1, *C2;
3080 if (auto *M = dyn_cast<IntrinsicInst>(Arg0)) {
3081 if (M->getIntrinsicID() == IID && match(Arg1, m_APFloat(C1)) &&
3082 ((match(M->getArgOperand(0), m_Value(X)) &&
3083 match(M->getArgOperand(1), m_APFloat(C2))) ||
3084 (match(M->getArgOperand(1), m_Value(X)) &&
3085 match(M->getArgOperand(0), m_APFloat(C2))))) {
3086 APFloat Res(0.0);
3087 switch (IID) {
3088 case Intrinsic::maxnum:
3089 Res = maxnum(*C1, *C2);
3090 break;
3091 case Intrinsic::minnum:
3092 Res = minnum(*C1, *C2);
3093 break;
3094 case Intrinsic::maximumnum:
3095 Res = maximumnum(*C1, *C2);
3096 break;
3097 case Intrinsic::minimumnum:
3098 Res = minimumnum(*C1, *C2);
3099 break;
3100 case Intrinsic::maximum:
3101 Res = maximum(*C1, *C2);
3102 break;
3103 case Intrinsic::minimum:
3104 Res = minimum(*C1, *C2);
3105 break;
3106 default:
3107 llvm_unreachable("unexpected intrinsic ID");
3108 }
3109 // TODO: Conservatively intersecting FMF. If Res == C2, the transform
3110 // was a simplification (so Arg0 and its original flags could
3111 // propagate?)
3112 Value *V = Builder.CreateBinaryIntrinsic(
3113 IID, X, ConstantFP::get(Arg0->getType(), Res),
3115 return replaceInstUsesWith(*II, V);
3116 }
3117 }
3118
3119 // m((fpext X), (fpext Y)) -> fpext (m(X, Y))
3120 if (match(Arg0, m_FPExt(m_Value(X))) && match(Arg1, m_FPExt(m_Value(Y))) &&
3121 (Arg0->hasOneUse() || Arg1->hasOneUse()) &&
3122 X->getType() == Y->getType()) {
3123 Value *NewCall =
3124 Builder.CreateBinaryIntrinsic(IID, X, Y, II, II->getName());
3125 return new FPExtInst(NewCall, II->getType());
3126 }
3127
3128 // m(fpext X, C) -> fpext m(X, TruncC) if C can be losslessly truncated.
3129 Constant *C;
3130 if (match(Arg0, m_OneUse(m_FPExt(m_Value(X)))) &&
3131 match(Arg1, m_ImmConstant(C))) {
3132 if (Constant *TruncC =
3133 getLosslessInvCast(C, X->getType(), Instruction::FPExt, DL)) {
3134 Value *NewCall =
3135 Builder.CreateBinaryIntrinsic(IID, X, TruncC, II, II->getName());
3136 return new FPExtInst(NewCall, II->getType());
3137 }
3138 }
3139
3140 // max X, -X --> fabs X
3141 // min X, -X --> -(fabs X)
3142 // TODO: Remove one-use limitation? That is obviously better for max,
3143 // hence why we don't check for one-use for that. However,
3144 // it would be an extra instruction for min (fnabs), but
3145 // that is still likely better for analysis and codegen.
3146 auto IsMinMaxOrXNegX = [IID, &X](Value *Op0, Value *Op1) {
3147 if (match(Op0, m_FNeg(m_Value(X))) && match(Op1, m_Specific(X)))
3148 return Op0->hasOneUse() ||
3149 (IID != Intrinsic::minimum && IID != Intrinsic::minnum &&
3150 IID != Intrinsic::minimumnum);
3151 return false;
3152 };
3153
3154 if (IsMinMaxOrXNegX(Arg0, Arg1) || IsMinMaxOrXNegX(Arg1, Arg0)) {
3155 Value *R = Builder.CreateFAbs(X, II);
3156 if (IID == Intrinsic::minimum || IID == Intrinsic::minnum ||
3157 IID == Intrinsic::minimumnum)
3158 R = Builder.CreateFNegFMF(R, II);
3159 return replaceInstUsesWith(*II, R);
3160 }
3161
3162 break;
3163 }
3164 case Intrinsic::matrix_multiply: {
3165 // Optimize negation in matrix multiplication.
3166
3167 // -A * -B -> A * B
3168 Value *A, *B;
3169 if (match(II->getArgOperand(0), m_FNeg(m_Value(A))) &&
3170 match(II->getArgOperand(1), m_FNeg(m_Value(B)))) {
3171 replaceOperand(*II, 0, A);
3172 replaceOperand(*II, 1, B);
3173 return II;
3174 }
3175
3176 Value *Op0 = II->getOperand(0);
3177 Value *Op1 = II->getOperand(1);
3178 Value *OpNotNeg, *NegatedOp;
3179 unsigned NegatedOpArg, OtherOpArg;
3180 if (match(Op0, m_FNeg(m_Value(OpNotNeg)))) {
3181 NegatedOp = Op0;
3182 NegatedOpArg = 0;
3183 OtherOpArg = 1;
3184 } else if (match(Op1, m_FNeg(m_Value(OpNotNeg)))) {
3185 NegatedOp = Op1;
3186 NegatedOpArg = 1;
3187 OtherOpArg = 0;
3188 } else
3189 // Multiplication doesn't have a negated operand.
3190 break;
3191
3192 // Only optimize if the negated operand has only one use.
3193 if (!NegatedOp->hasOneUse())
3194 break;
3195
3196 Value *OtherOp = II->getOperand(OtherOpArg);
3197 VectorType *RetTy = cast<VectorType>(II->getType());
3198 VectorType *NegatedOpTy = cast<VectorType>(NegatedOp->getType());
3199 VectorType *OtherOpTy = cast<VectorType>(OtherOp->getType());
3200 ElementCount NegatedCount = NegatedOpTy->getElementCount();
3201 ElementCount OtherCount = OtherOpTy->getElementCount();
3202 ElementCount RetCount = RetTy->getElementCount();
3203 // (-A) * B -> A * (-B), if it is cheaper to negate B and vice versa.
3204 if (ElementCount::isKnownGT(NegatedCount, OtherCount) &&
3205 ElementCount::isKnownLT(OtherCount, RetCount)) {
3206 Value *InverseOtherOp = Builder.CreateFNeg(OtherOp);
3207 replaceOperand(*II, NegatedOpArg, OpNotNeg);
3208 replaceOperand(*II, OtherOpArg, InverseOtherOp);
3209 return II;
3210 }
3211 // (-A) * B -> -(A * B), if it is cheaper to negate the result
3212 if (ElementCount::isKnownGT(NegatedCount, RetCount)) {
3213 SmallVector<Value *, 5> NewArgs(II->args());
3214 NewArgs[NegatedOpArg] = OpNotNeg;
3215 Value *NewMul = Builder.CreateIntrinsic(II->getType(), IID, NewArgs, II);
3216 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(NewMul, II));
3217 }
3218 break;
3219 }
3220 case Intrinsic::fmuladd: {
3221 // Try to simplify the underlying FMul.
3222 if (Value *V =
3223 simplifyFMulInst(II->getArgOperand(0), II->getArgOperand(1),
3224 II->getFastMathFlags(), SQ.getWithInstruction(II)))
3225 return BinaryOperator::CreateFAddFMF(V, II->getArgOperand(2),
3226 II->getFastMathFlags());
3227
3228 [[fallthrough]];
3229 }
3230 case Intrinsic::fma: {
3231 // fma fneg(x), fneg(y), z -> fma x, y, z
3232 Value *Src0 = II->getArgOperand(0);
3233 Value *Src1 = II->getArgOperand(1);
3234 Value *Src2 = II->getArgOperand(2);
3235 Value *X, *Y;
3236 if (match(Src0, m_FNeg(m_Value(X))) && match(Src1, m_FNeg(m_Value(Y))))
3237 return replaceInstUsesWith(
3238 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, Y, Src2}, II));
3239
3240 // fma fabs(x), fabs(x), z -> fma x, x, z
3241 if (match(Src0, m_FAbs(m_Value(X))) && match(Src1, m_FAbs(m_Specific(X))))
3242 return replaceInstUsesWith(
3243 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, X, Src2}, II));
3244
3245 // Try to simplify the underlying FMul. We can only apply simplifications
3246 // that do not require rounding.
3247 if (Value *V = simplifyFMAFMul(Src0, Src1, II->getFastMathFlags(),
3248 SQ.getWithInstruction(II)))
3249 return BinaryOperator::CreateFAddFMF(V, Src2, II->getFastMathFlags());
3250
3251 // fma x, y, 0 -> fmul x, y
3252 // This is always valid for -0.0, but requires nsz for +0.0 as
3253 // -0.0 + 0.0 = 0.0, which would not be the same as the fmul on its own.
3254 if (match(Src2, m_NegZeroFP()) ||
3255 (match(Src2, m_PosZeroFP()) && II->getFastMathFlags().noSignedZeros()))
3256 return BinaryOperator::CreateFMulFMF(Src0, Src1, II);
3257
3258 // fma x, -1.0, y -> fsub y, x
3259 if (match(Src1, m_SpecificFP(-1.0)))
3260 return BinaryOperator::CreateFSubFMF(Src2, Src0, II);
3261
3262 break;
3263 }
3264 case Intrinsic::copysign: {
3265 Value *Mag = II->getArgOperand(0), *Sign = II->getArgOperand(1);
3266 if (std::optional<bool> KnownSignBit = computeKnownFPSignBit(
3267 Sign, getSimplifyQuery().getWithInstruction(II))) {
3268 if (*KnownSignBit) {
3269 // If we know that the sign argument is negative, reduce to FNABS:
3270 // copysign Mag, -Sign --> fneg (fabs Mag)
3271 Value *Fabs = Builder.CreateFAbs(Mag, II);
3272 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(Fabs, II));
3273 }
3274
3275 // If we know that the sign argument is positive, reduce to FABS:
3276 // copysign Mag, +Sign --> fabs Mag
3277 Value *Fabs = Builder.CreateFAbs(Mag, II);
3278 return replaceInstUsesWith(*II, Fabs);
3279 }
3280
3281 // Propagate sign argument through nested calls:
3282 // copysign Mag, (copysign ?, X) --> copysign Mag, X
3283 Value *X;
3285 Value *CopySign =
3286 Builder.CreateCopySign(Mag, X, FMFSource::intersect(II, Sign));
3287 return replaceInstUsesWith(*II, CopySign);
3288 }
3289
3290 // Clear sign-bit of constant magnitude:
3291 // copysign -MagC, X --> copysign MagC, X
3292 // TODO: Support constant folding for fabs
3293 const APFloat *MagC;
3294 if (match(Mag, m_APFloat(MagC)) && MagC->isNegative()) {
3295 APFloat PosMagC = *MagC;
3296 PosMagC.clearSign();
3297 return replaceInstUsesWith(
3298 *II, Builder.CreateCopySign(ConstantFP::get(Mag->getType(), PosMagC),
3299 Sign, II));
3300 }
3301
3302 // Peek through changes of magnitude's sign-bit. This call rewrites those:
3303 // copysign (fabs X), Sign --> copysign X, Sign
3304 // copysign (fneg X), Sign --> copysign X, Sign
3305 if (match(Mag, m_FAbs(m_Value(X))) || match(Mag, m_FNeg(m_Value(X))))
3306 return replaceInstUsesWith(*II, Builder.CreateCopySign(X, Sign, II));
3307
3308 // copysign(floor(fabs(X)), X) --> copysign(trunc(X), X)
3309 // copysign ignores the sign bit of its magnitude argument (implicit fabs),
3310 // so replacing floor(fabs(X)) with trunc(X) is correct for all inputs
3311 // including NaN without requiring nnan. The m_FAbs match also ensures
3312 // the floor argument is non-negative, so floor == trunc.
3313 Value *FAbsArg;
3314 if (match(Mag, m_Intrinsic<Intrinsic::floor>(m_FAbs(m_Value(FAbsArg)))) &&
3315 FAbsArg == Sign) {
3316 Value *Trunc = Builder.CreateUnaryIntrinsic(Intrinsic::trunc, Sign, II);
3317 return replaceInstUsesWith(*II, Builder.CreateCopySign(Trunc, Sign, II));
3318 }
3319
3320 Type *SignEltTy = Sign->getType()->getScalarType();
3321
3322 Value *CastSrc;
3323 if (match(Sign,
3325 CastSrc->getType()->isIntOrIntVectorTy() &&
3329 APInt::getSignMask(Known.getBitWidth()), Known,
3330 SQ))
3331 return II;
3332 }
3333
3334 break;
3335 }
3336 case Intrinsic::fabs: {
3337 Value *Cond, *TVal, *FVal;
3338 Value *Arg = II->getArgOperand(0);
3339 Value *X;
3340 // fabs (-X) --> fabs (X)
3341 if (match(Arg, m_FNeg(m_Value(X)))) {
3342 Value *Fabs = Builder.CreateFAbs(X, II);
3343 return replaceInstUsesWith(CI, Fabs);
3344 }
3345
3346 if (match(Arg, m_Select(m_Value(Cond), m_Value(TVal), m_Value(FVal)))) {
3347 // fabs (select Cond, TrueC, FalseC) --> select Cond, AbsT, AbsF
3348 if (Arg->hasOneUse() ? (isa<Constant>(TVal) || isa<Constant>(FVal))
3349 : (isa<Constant>(TVal) && isa<Constant>(FVal))) {
3350 CallInst *AbsT = Builder.CreateCall(II->getCalledFunction(), {TVal});
3351 CallInst *AbsF = Builder.CreateCall(II->getCalledFunction(), {FVal});
3352 SelectInst *SI = SelectInst::Create(Cond, AbsT, AbsF);
3353 SI->setFastMathFlags(II->getFastMathFlags() |
3354 cast<SelectInst>(Arg)->getFastMathFlags());
3355 // Can't copy nsz to select, as even with the nsz flag the fabs result
3356 // always has the sign bit unset.
3357 SI->setHasNoSignedZeros(false);
3358 return SI;
3359 }
3360 // fabs (select Cond, -FVal, FVal) --> fabs FVal
3361 if (match(TVal, m_FNeg(m_Specific(FVal))))
3362 return replaceInstUsesWith(*II, Builder.CreateFAbs(FVal, II));
3363 // fabs (select Cond, TVal, -TVal) --> fabs TVal
3364 if (match(FVal, m_FNeg(m_Specific(TVal))))
3365 return replaceInstUsesWith(*II, Builder.CreateFAbs(TVal, II));
3366 }
3367
3368 Value *Magnitude, *Sign;
3369 if (match(II->getArgOperand(0),
3370 m_CopySign(m_Value(Magnitude), m_Value(Sign)))) {
3371 // fabs (copysign x, y) -> (fabs x)
3372 Value *AbsSign = Builder.CreateFAbs(Magnitude, II);
3373 return replaceInstUsesWith(*II, AbsSign);
3374 }
3375
3376 [[fallthrough]];
3377 }
3378 case Intrinsic::ceil:
3379 case Intrinsic::floor:
3380 case Intrinsic::round:
3381 case Intrinsic::roundeven:
3382 case Intrinsic::nearbyint:
3383 case Intrinsic::rint:
3384 case Intrinsic::trunc: {
3385 Value *ExtSrc;
3386 if (match(II->getArgOperand(0), m_OneUse(m_FPExt(m_Value(ExtSrc))))) {
3387 // Narrow the call: intrinsic (fpext x) -> fpext (intrinsic x)
3388 Value *NarrowII = Builder.CreateUnaryIntrinsic(IID, ExtSrc, II);
3389 return new FPExtInst(NarrowII, II->getType());
3390 }
3391 break;
3392 }
3393 case Intrinsic::cos:
3394 case Intrinsic::amdgcn_cos:
3395 case Intrinsic::cosh: {
3396 Value *X, *Sign;
3397 Value *Src = II->getArgOperand(0);
3398 if (match(Src, m_FNeg(m_Value(X))) || match(Src, m_FAbs(m_Value(X))) ||
3399 match(Src, m_CopySign(m_Value(X), m_Value(Sign)))) {
3400 // f(-x) --> f(x)
3401 // f(fabs(x)) --> f(x)
3402 // f(copysign(x, y)) --> f(x)
3403 // for f in {cos, cosh}
3404 return replaceInstUsesWith(*II, Builder.CreateUnaryIntrinsic(IID, X, II));
3405 }
3406 if (IID == Intrinsic::cos) {
3407 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3408 return replaceInstUsesWith(*II, Result);
3409 }
3410 break;
3411 }
3412 case Intrinsic::sin:
3413 case Intrinsic::amdgcn_sin:
3414 case Intrinsic::sinh:
3415 case Intrinsic::tan:
3416 case Intrinsic::tanh: {
3417 Value *X;
3418 if (match(II->getArgOperand(0), m_OneUse(m_FNeg(m_Value(X))))) {
3419 // f(-x) --> -f(x)
3420 // for f in {sin, sinh, tan, tanh}
3421 Value *NewFunc = Builder.CreateUnaryIntrinsic(IID, X, II);
3422 return UnaryOperator::CreateFNegFMF(NewFunc, II);
3423 }
3424 if (IID == Intrinsic::sin) {
3425 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3426 return replaceInstUsesWith(*II, Result);
3427 }
3428 break;
3429 }
3430 case Intrinsic::ldexp: {
3431 Value *Src = II->getArgOperand(0);
3432 Value *Exp = II->getArgOperand(1);
3433
3434 // ldexp(x, K) -> fmul x, 2^K
3435 uint64_t ConstExp;
3436 if (match(Exp, m_ConstantInt(ConstExp))) {
3437 const fltSemantics &FPTy =
3438 Src->getType()->getScalarType()->getFltSemantics();
3439
3440 APFloat Scaled = scalbn(APFloat::getOne(FPTy), static_cast<int>(ConstExp),
3442 if (!Scaled.isZero() && !Scaled.isInfinity()) {
3443 // Skip overflow and underflow cases.
3444 Constant *FPConst = ConstantFP::get(Src->getType(), Scaled);
3445 return BinaryOperator::CreateFMulFMF(Src, FPConst, II);
3446 }
3447 }
3448
3449 // ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b))
3450 //
3451 // A danger is if the first ldexp would overflow to infinity or underflow to
3452 // zero, but the combined exponent avoids it.
3453 //
3454 // We ignore this with reassoc, or if we know both exponents have the same
3455 // sign (since then we'd just double down on the over/underflow which would
3456 // occur anyway).
3457 //
3458 // ldexp can take arbitrary integer types, so we also need to ensure that
3459 // our exponent type is wide enough so that if sadd.sat(a, b) saturates,
3460 // then ldexp at the saturated exponent saturates to inf or zero as well.
3461 //
3462 // TODO: Could do better if we had range tracking for the input value
3463 // exponent. Also could broaden sign check to cover == 0 case.
3464 Value *InnerSrc;
3465 Value *InnerExp;
3467 m_Value(InnerSrc), m_Value(InnerExp)))) &&
3468 Exp->getType() == InnerExp->getType()) {
3469 FastMathFlags FMF = II->getFastMathFlags();
3470 FastMathFlags InnerFlags = cast<FPMathOperator>(Src)->getFastMathFlags();
3471
3472 if (ldexpSaturatingAddIsSafe(II->getType(), Exp->getType()) &&
3473 ((FMF.allowReassoc() && InnerFlags.allowReassoc()) ||
3474 signBitMustBeTheSame(Exp, InnerExp, SQ.getWithInstruction(II)))) {
3475 Value *NewExp =
3476 Builder.CreateBinaryIntrinsic(Intrinsic::sadd_sat, InnerExp, Exp);
3477 return replaceInstUsesWith(
3478 *II, Builder.CreateLdexp(InnerSrc, NewExp, FMF | InnerFlags));
3479 }
3480 }
3481
3482 // ldexp(x, zext(i1 y)) -> fmul x, (select y, 2.0, 1.0)
3483 // ldexp(x, sext(i1 y)) -> fmul x, (select y, 0.5, 1.0)
3484 Value *ExtSrc;
3485 if (match(Exp, m_ZExt(m_Value(ExtSrc))) &&
3486 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3487 Value *Select =
3488 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 2.0),
3489 ConstantFP::get(II->getType(), 1.0));
3491 }
3492 if (match(Exp, m_SExt(m_Value(ExtSrc))) &&
3493 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3494 Value *Select =
3495 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 0.5),
3496 ConstantFP::get(II->getType(), 1.0));
3498 }
3499
3500 // ldexp(x, c ? exp : 0) -> c ? ldexp(x, exp) : x
3501 // ldexp(x, c ? 0 : exp) -> c ? x : ldexp(x, exp)
3502 ///
3503 // TODO: If we cared, should insert a canonicalize for x
3504 Value *SelectCond, *SelectLHS, *SelectRHS;
3505 if (match(II->getArgOperand(1),
3506 m_OneUse(m_Select(m_Value(SelectCond), m_Value(SelectLHS),
3507 m_Value(SelectRHS))))) {
3508 Value *NewLdexp = nullptr;
3509 Value *Select = nullptr;
3510 if (match(SelectRHS, m_ZeroInt())) {
3511 NewLdexp = Builder.CreateLdexp(Src, SelectLHS, II);
3512 Select = Builder.CreateSelect(SelectCond, NewLdexp, Src);
3513 } else if (match(SelectLHS, m_ZeroInt())) {
3514 NewLdexp = Builder.CreateLdexp(Src, SelectRHS, II);
3515 Select = Builder.CreateSelect(SelectCond, Src, NewLdexp);
3516 }
3517
3518 if (NewLdexp) {
3519 Select->takeName(II);
3520 return replaceInstUsesWith(*II, Select);
3521 }
3522 }
3523
3524 break;
3525 }
3526 case Intrinsic::ptrauth_auth:
3527 case Intrinsic::ptrauth_resign: {
3528 // (sign|resign) + (auth|resign) can be folded by omitting the middle
3529 // sign+auth component if the key and discriminator match.
3530 bool NeedSign = II->getIntrinsicID() == Intrinsic::ptrauth_resign;
3531 Value *Ptr = II->getArgOperand(0);
3532 Value *Key = II->getArgOperand(1);
3533 Value *Disc = II->getArgOperand(2);
3534 Value *DS = nullptr;
3535 if (auto Bundle = II->getOperandBundle(LLVMContext::OB_deactivation_symbol))
3536 DS = Bundle->Inputs[0];
3537
3538 // AuthKey will be the key we need to end up authenticating against in
3539 // whatever we replace this sequence with.
3540 Value *AuthKey = nullptr, *AuthDisc = nullptr, *BasePtr;
3541 if (const auto *CI = dyn_cast<CallBase>(Ptr)) {
3542 Value *OtherDS = nullptr;
3543 if (auto Bundle =
3545 OtherDS = Bundle->Inputs[0];
3546 if (DS != OtherDS)
3547 break;
3548
3549 if (CI->getIntrinsicID() == Intrinsic::ptrauth_sign) {
3550 if (CI->getArgOperand(1) != Key || CI->getArgOperand(2) != Disc)
3551 break;
3552 } else if (CI->getIntrinsicID() == Intrinsic::ptrauth_resign) {
3553 // The resign intrinsic does not support deactivation symbols.
3554 assert(!DS);
3555 if (CI->getArgOperand(3) != Key || CI->getArgOperand(4) != Disc)
3556 break;
3557 AuthKey = CI->getArgOperand(1);
3558 AuthDisc = CI->getArgOperand(2);
3559 } else
3560 break;
3561 BasePtr = CI->getArgOperand(0);
3562 } else if (const auto *PtrToInt = dyn_cast<PtrToIntOperator>(Ptr)) {
3563 // ptrauth constants are equivalent to a call to @llvm.ptrauth.sign for
3564 // our purposes, so check for that too.
3565 const auto *CPA = dyn_cast<ConstantPtrAuth>(PtrToInt->getOperand(0));
3566 if (!CPA || DS || !CPA->isKnownCompatibleWith(Key, Disc, DL))
3567 break;
3568
3569 // resign(ptrauth(p,ks,ds),ks,ds,kr,dr) -> ptrauth(p,kr,dr)
3570 if (NeedSign && isa<ConstantInt>(II->getArgOperand(4))) {
3571 auto *SignKey = cast<ConstantInt>(II->getArgOperand(3));
3572 auto *SignDisc = cast<ConstantInt>(II->getArgOperand(4));
3573 auto *Null = ConstantPointerNull::get(Builder.getPtrTy());
3574 auto *NewCPA = ConstantPtrAuth::get(CPA->getPointer(), SignKey,
3575 SignDisc, /*AddrDisc=*/Null,
3576 /*DeactivationSymbol=*/Null);
3578 *II, ConstantExpr::getPointerCast(NewCPA, II->getType()));
3579 return eraseInstFromFunction(*II);
3580 }
3581
3582 // auth(ptrauth(p,k,d),k,d) -> p
3583 BasePtr = Builder.CreatePtrToInt(CPA->getPointer(), II->getType());
3584 } else
3585 break;
3586
3587 unsigned NewIntrin;
3588 if (AuthKey && NeedSign) {
3589 // resign(0,1) + resign(1,2) = resign(0, 2)
3590 NewIntrin = Intrinsic::ptrauth_resign;
3591 } else if (AuthKey) {
3592 // resign(0,1) + auth(1) = auth(0)
3593 NewIntrin = Intrinsic::ptrauth_auth;
3594 } else if (NeedSign) {
3595 // sign(0) + resign(0, 1) = sign(1)
3596 NewIntrin = Intrinsic::ptrauth_sign;
3597 } else {
3598 // sign(0) + auth(0) = nop
3599 replaceInstUsesWith(*II, BasePtr);
3600 return eraseInstFromFunction(*II);
3601 }
3602
3603 SmallVector<Value *, 4> CallArgs;
3604 CallArgs.push_back(BasePtr);
3605 if (AuthKey) {
3606 CallArgs.push_back(AuthKey);
3607 CallArgs.push_back(AuthDisc);
3608 }
3609
3610 if (NeedSign) {
3611 CallArgs.push_back(II->getArgOperand(3));
3612 CallArgs.push_back(II->getArgOperand(4));
3613 }
3614
3615 std::vector<OperandBundleDef> Bundles;
3616 if (DS)
3617 Bundles.push_back(OperandBundleDef("deactivation-symbol", DS));
3618
3619 Function *NewFn =
3620 Intrinsic::getOrInsertDeclaration(II->getModule(), NewIntrin);
3621 return CallInst::Create(NewFn, CallArgs, Bundles);
3622 }
3623 case Intrinsic::arm_neon_vtbl1:
3624 case Intrinsic::arm_neon_vtbl2:
3625 case Intrinsic::arm_neon_vtbl3:
3626 case Intrinsic::arm_neon_vtbl4:
3627 case Intrinsic::aarch64_neon_tbl1:
3628 case Intrinsic::aarch64_neon_tbl2:
3629 case Intrinsic::aarch64_neon_tbl3:
3630 case Intrinsic::aarch64_neon_tbl4:
3631 return simplifyNeonTbl(*II, *this, /*IsExtension=*/false);
3632 case Intrinsic::arm_neon_vtbx1:
3633 case Intrinsic::arm_neon_vtbx2:
3634 case Intrinsic::arm_neon_vtbx3:
3635 case Intrinsic::arm_neon_vtbx4:
3636 case Intrinsic::aarch64_neon_tbx1:
3637 case Intrinsic::aarch64_neon_tbx2:
3638 case Intrinsic::aarch64_neon_tbx3:
3639 case Intrinsic::aarch64_neon_tbx4:
3640 return simplifyNeonTbl(*II, *this, /*IsExtension=*/true);
3641
3642 case Intrinsic::arm_neon_vmulls:
3643 case Intrinsic::arm_neon_vmullu:
3644 case Intrinsic::aarch64_neon_smull:
3645 case Intrinsic::aarch64_neon_umull: {
3646 Value *Arg0 = II->getArgOperand(0);
3647 Value *Arg1 = II->getArgOperand(1);
3648
3649 // Handle mul by zero first:
3651 return replaceInstUsesWith(CI, ConstantAggregateZero::get(II->getType()));
3652 }
3653
3654 // Check for constant LHS & RHS - in this case we just simplify.
3655 bool Zext = (IID == Intrinsic::arm_neon_vmullu ||
3656 IID == Intrinsic::aarch64_neon_umull);
3657 VectorType *NewVT = cast<VectorType>(II->getType());
3658 if (Constant *CV0 = dyn_cast<Constant>(Arg0)) {
3659 if (Constant *CV1 = dyn_cast<Constant>(Arg1)) {
3660 Value *V0 = Builder.CreateIntCast(CV0, NewVT, /*isSigned=*/!Zext);
3661 Value *V1 = Builder.CreateIntCast(CV1, NewVT, /*isSigned=*/!Zext);
3662 return replaceInstUsesWith(CI, Builder.CreateMul(V0, V1));
3663 }
3664
3665 // Couldn't simplify - canonicalize constant to the RHS.
3666 std::swap(Arg0, Arg1);
3667 }
3668
3669 // Handle mul by one:
3670 if (Constant *CV1 = dyn_cast<Constant>(Arg1))
3671 if (ConstantInt *Splat =
3672 dyn_cast_or_null<ConstantInt>(CV1->getSplatValue()))
3673 if (Splat->isOne())
3674 return CastInst::CreateIntegerCast(Arg0, II->getType(),
3675 /*isSigned=*/!Zext);
3676
3677 break;
3678 }
3679 case Intrinsic::arm_neon_aesd:
3680 case Intrinsic::arm_neon_aese:
3681 case Intrinsic::aarch64_crypto_aesd:
3682 case Intrinsic::aarch64_crypto_aese:
3683 case Intrinsic::aarch64_sve_aesd:
3684 case Intrinsic::aarch64_sve_aese: {
3685 Value *DataArg = II->getArgOperand(0);
3686 Value *KeyArg = II->getArgOperand(1);
3687
3688 // Accept zero on either operand.
3689 if (!match(KeyArg, m_ZeroInt()))
3690 std::swap(KeyArg, DataArg);
3691
3692 // Try to use the builtin XOR in AESE and AESD to eliminate a prior XOR
3693 Value *Data, *Key;
3694 if (match(KeyArg, m_ZeroInt()) &&
3695 match(DataArg, m_Xor(m_Value(Data), m_Value(Key)))) {
3696 replaceOperand(*II, 0, Data);
3697 replaceOperand(*II, 1, Key);
3698 return II;
3699 }
3700 break;
3701 }
3702 case Intrinsic::arm_neon_vshifts:
3703 case Intrinsic::arm_neon_vshiftu:
3704 case Intrinsic::aarch64_neon_sshl:
3705 case Intrinsic::aarch64_neon_ushl:
3706 return foldNeonShift(II, *this);
3707 case Intrinsic::hexagon_V6_vandvrt:
3708 case Intrinsic::hexagon_V6_vandvrt_128B: {
3709 // Simplify Q -> V -> Q conversion.
3710 if (auto Op0 = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3711 Intrinsic::ID ID0 = Op0->getIntrinsicID();
3712 if (ID0 != Intrinsic::hexagon_V6_vandqrt &&
3713 ID0 != Intrinsic::hexagon_V6_vandqrt_128B)
3714 break;
3715 Value *Bytes = Op0->getArgOperand(1), *Mask = II->getArgOperand(1);
3716 uint64_t Bytes1 = computeKnownBits(Bytes, Op0).One.getZExtValue();
3717 uint64_t Mask1 = computeKnownBits(Mask, II).One.getZExtValue();
3718 // Check if every byte has common bits in Bytes and Mask.
3719 uint64_t C = Bytes1 & Mask1;
3720 if ((C & 0xFF) && (C & 0xFF00) && (C & 0xFF0000) && (C & 0xFF000000))
3721 return replaceInstUsesWith(*II, Op0->getArgOperand(0));
3722 }
3723 break;
3724 }
3725 case Intrinsic::stackrestore: {
3726 enum class ClassifyResult {
3727 None,
3728 Alloca,
3729 StackRestore,
3730 CallWithSideEffects,
3731 };
3732 auto Classify = [](const Instruction *I) {
3733 if (isa<AllocaInst>(I))
3734 return ClassifyResult::Alloca;
3735
3736 if (auto *CI = dyn_cast<CallInst>(I)) {
3737 if (auto *II = dyn_cast<IntrinsicInst>(CI)) {
3738 if (II->getIntrinsicID() == Intrinsic::stackrestore)
3739 return ClassifyResult::StackRestore;
3740
3741 if (II->mayHaveSideEffects())
3742 return ClassifyResult::CallWithSideEffects;
3743 } else {
3744 // Consider all non-intrinsic calls to be side effects
3745 return ClassifyResult::CallWithSideEffects;
3746 }
3747 }
3748
3749 return ClassifyResult::None;
3750 };
3751
3752 // If the stacksave and the stackrestore are in the same BB, and there is
3753 // no intervening call, alloca, or stackrestore of a different stacksave,
3754 // remove the restore. This can happen when variable allocas are DCE'd.
3755 if (IntrinsicInst *SS = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3756 if (SS->getIntrinsicID() == Intrinsic::stacksave &&
3757 SS->getParent() == II->getParent()) {
3758 BasicBlock::iterator BI(SS);
3759 bool CannotRemove = false;
3760 for (++BI; &*BI != II; ++BI) {
3761 switch (Classify(&*BI)) {
3762 case ClassifyResult::None:
3763 // So far so good, look at next instructions.
3764 break;
3765
3766 case ClassifyResult::StackRestore:
3767 // If we found an intervening stackrestore for a different
3768 // stacksave, we can't remove the stackrestore. Otherwise, continue.
3769 if (cast<IntrinsicInst>(*BI).getArgOperand(0) != SS)
3770 CannotRemove = true;
3771 break;
3772
3773 case ClassifyResult::Alloca:
3774 case ClassifyResult::CallWithSideEffects:
3775 // If we found an alloca, a non-intrinsic call, or an intrinsic
3776 // call with side effects, we can't remove the stackrestore.
3777 CannotRemove = true;
3778 break;
3779 }
3780 if (CannotRemove)
3781 break;
3782 }
3783
3784 if (!CannotRemove)
3785 return eraseInstFromFunction(CI);
3786 }
3787 }
3788
3789 // Scan down this block to see if there is another stack restore in the
3790 // same block without an intervening call/alloca.
3792 Instruction *TI = II->getParent()->getTerminator();
3793 bool CannotRemove = false;
3794 for (++BI; &*BI != TI; ++BI) {
3795 switch (Classify(&*BI)) {
3796 case ClassifyResult::None:
3797 // So far so good, look at next instructions.
3798 break;
3799
3800 case ClassifyResult::StackRestore:
3801 // If there is a stackrestore below this one, remove this one.
3802 return eraseInstFromFunction(CI);
3803
3804 case ClassifyResult::Alloca:
3805 case ClassifyResult::CallWithSideEffects:
3806 // If we found an alloca, a non-intrinsic call, or an intrinsic call
3807 // with side effects (such as llvm.stacksave and llvm.read_register),
3808 // we can't remove the stack restore.
3809 CannotRemove = true;
3810 break;
3811 }
3812 if (CannotRemove)
3813 break;
3814 }
3815
3816 // If the stack restore is in a return, resume, or unwind block and if there
3817 // are no allocas or calls between the restore and the return, nuke the
3818 // restore.
3819 if (!CannotRemove && (isa<ReturnInst>(TI) || isa<ResumeInst>(TI)))
3820 return eraseInstFromFunction(CI);
3821 break;
3822 }
3823 case Intrinsic::lifetime_end:
3824 // Asan needs to poison memory to detect invalid access which is possible
3825 // even for empty lifetime range.
3826 if (II->getFunction()->hasFnAttribute(Attribute::SanitizeAddress) ||
3827 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemory) ||
3828 II->getFunction()->hasFnAttribute(Attribute::SanitizeHWAddress) ||
3829 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag))
3830 break;
3831
3832 if (removeTriviallyEmptyRange(*II, *this, [](const IntrinsicInst &I) {
3833 return I.getIntrinsicID() == Intrinsic::lifetime_start;
3834 }))
3835 return nullptr;
3836 break;
3837 case Intrinsic::assume: {
3838 for (auto [Idx, OBU] : llvm::enumerate(II->operand_bundles())) {
3839 auto RemoveBundle = [&, Idx = Idx]() -> Instruction * {
3840 if (II->getNumOperandBundles() == 1)
3841 return eraseInstFromFunction(*II);
3843 };
3844
3845 switch (getBundleAttrFromOBU(OBU)) {
3846 case BundleAttr::None:
3847 llvm_unreachable("Unexpected Attribute");
3848 case BundleAttr::Align: {
3849 // Try to remove redundant alignment assumptions.
3850 auto [Ptr, _, OffsetPtr, Alignment, Offset] = getAssumeAlignInfo(OBU);
3851
3852 if (!Alignment)
3853 break;
3854
3855 // Remove align 1 and non-power-of-two bundles; they don't add any
3856 // useful information.
3857 if (*Alignment == 1 || !isPowerOf2_64(*Alignment))
3858 return RemoveBundle();
3859
3860 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3861 GEP &&
3862 GEP->getMaxPreservedAlignment(getDataLayout()) >= *Alignment) {
3863 Builder.CreateAlignmentAssumption(
3864 getDataLayout(), GEP->getPointerOperand(), *Alignment,
3865 OffsetPtr ? const_cast<Value *>(OffsetPtr->get()) : nullptr);
3866 return RemoveBundle();
3867 }
3868
3869 if (!Offset)
3870 break;
3871
3872 Value *BasePtr;
3873 const APInt *PtrOffset;
3874 if (match(Ptr.get(), m_PtrAdd(m_Value(BasePtr), m_APInt(PtrOffset)))) {
3875 auto PtrOffsetVal =
3876 PtrOffset->sextOrTrunc(DL.getIndexTypeSizeInBits(Ptr->getType()))
3877 .trySExtValue();
3878 if (!PtrOffsetVal)
3879 break;
3880 Builder.CreateAlignmentAssumption(
3881 DL, BasePtr, *Alignment,
3882 Builder.getInt64(*Offset - *PtrOffsetVal));
3883 return RemoveBundle();
3884 }
3885
3886 // Don't try to remove align assumptions for pointers derived from
3887 // arguments. We might lose information if the function gets inline and
3888 // the align argument attribute disappears.
3889 Value *UO = getUnderlyingObject(Ptr);
3890 if (!UO || isa<Argument>(UO))
3891 break;
3892
3893 // Compute known bits for the pointer and drop the assume if the
3894 // known alignment isn't increased by it.
3895 auto AlignMask = (*Alignment - 1);
3896 if (KnownBits KB = computeKnownBits(Ptr, II);
3897 (KB.Zero & AlignMask) == (~*Offset & AlignMask) &&
3898 (KB.One & AlignMask) == (*Offset & AlignMask))
3899 return RemoveBundle();
3900 break;
3901 }
3902
3903 case BundleAttr::Dereferenceable: {
3904 auto [Ptr, _, Count] = getAssumeDereferenceableInfo(OBU);
3905
3906 if (!Count)
3907 break;
3908
3909 if (*Count == 0 ||
3911 getSimplifyQuery().getWithInstruction(II)))
3912 return RemoveBundle();
3913
3914 break;
3915 }
3916
3917 case BundleAttr::Ignore:
3918 return RemoveBundle();
3919
3920 case BundleAttr::NonNull: {
3921 auto [Ptr] = llvm::getAssumeNonNullInfo(OBU);
3922
3923 // Drop assume if we can prove nonnull without it
3924 if (isKnownNonZero(Ptr, getSimplifyQuery().getWithInstruction(II)))
3925 return RemoveBundle();
3926
3927 // Fold the assume into metadata if it's valid at the load
3928 if (auto *LI = dyn_cast<LoadInst>(Ptr);
3929 LI &&
3930 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3931 MDNode *MD = MDNode::get(II->getContext(), {});
3932 LI->setMetadata(LLVMContext::MD_nonnull, MD);
3933 LI->setMetadata(LLVMContext::MD_noundef, MD);
3934 return RemoveBundle();
3935 }
3936
3937 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3938 GEP && GEP->isInBounds() &&
3939 !NullPointerIsDefined(II->getFunction(),
3940 Ptr->getType()->getPointerAddressSpace())) {
3941 Builder.CreateNonnullAssumption(GEP->stripInBoundsOffsets());
3942 return RemoveBundle();
3943 }
3944
3945 // TODO: apply nonnull return attributes to calls and invokes
3946 break;
3947 }
3948
3949 case BundleAttr::NoUndef: {
3950 auto [Val] = getAssumeNoUndefInfo(OBU);
3951
3953 return RemoveBundle();
3954
3955 if (auto *LI = dyn_cast<LoadInst>(Val);
3956 LI &&
3957 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3958 LI->setMetadata(LLVMContext::MD_noundef,
3959 MDNode::get(II->getContext(), {}));
3960 return RemoveBundle();
3961 }
3962
3963 } break;
3964
3965 case BundleAttr::SeparateStorage: {
3966 auto [Ptr1, Ptr2] = getAssumeSeparateStorageInfo(OBU);
3967 // Separate storage assumptions apply to the underlying allocations, not
3968 // any particular pointer within them. When evaluating the hints for AA
3969 // purposes we getUnderlyingObject them; by precomputing the answers
3970 // here we can avoid having to do so repeatedly there.
3971 auto MaybeSimplifyHint = [&](const Use &U) {
3972 Value *Hint = U.get();
3973 // Not having a limit is safe because InstCombine removes unreachable
3974 // code.
3975 Value *UnderlyingObject = getUnderlyingObject(Hint, /*MaxLookup*/ 0);
3976 if (Hint != UnderlyingObject)
3977 replaceUse(const_cast<Use &>(U), UnderlyingObject);
3978 };
3979 MaybeSimplifyHint(Ptr1);
3980 MaybeSimplifyHint(Ptr2);
3981 } break;
3982
3983 // TODO: Drop these assumes when they are redundant
3984 case BundleAttr::DereferenceableOrNull:
3985 break;
3986
3987 // This cannot be simplified
3988 case BundleAttr::Cold:
3989 break;
3990 }
3991 }
3992
3993 // If the assume has operand bundles, the folds below will never work, so
3994 // don't bother trying.
3995 if (II->hasOperandBundles())
3996 break;
3997
3998 Value *IIOperand = II->getArgOperand(0);
3999
4000 // Canonicalize assume(a && b) -> assume(a); assume(b);
4001 // Note: New assumption intrinsics created here are registered by
4002 // the InstCombineIRInserter object.
4003 Value *A, *B;
4004 if (match(IIOperand, m_LogicalAnd(m_Value(A), m_Value(B)))) {
4005 Builder.CreateAssumption(A);
4006 Builder.CreateAssumption(B);
4007 return eraseInstFromFunction(*II);
4008 }
4009 // assume(!(a || b)) -> assume(!a); assume(!b);
4010 if (match(IIOperand, m_Not(m_LogicalOr(m_Value(A), m_Value(B))))) {
4011 Builder.CreateAssumption(Builder.CreateNot(A));
4012 Builder.CreateAssumption(Builder.CreateNot(B));
4013 return eraseInstFromFunction(*II);
4014 }
4015
4016 // Convert nonnull assume like:
4017 // %A = icmp ne i32* %PTR, null
4018 // call void @llvm.assume(i1 %A)
4019 // into
4020 // call void @llvm.assume(i1 true) [ "nonnull"(i32* %PTR) ]
4021 if (match(
4022 IIOperand,
4025 m_Zero())))) &&
4026 A->getType()->isPointerTy()) {
4027 Builder.CreateNonnullAssumption(A);
4028 return eraseInstFromFunction(*II);
4029 }
4030
4031 // Convert alignment assume like:
4032 // %B = ptrtoint ptr %A to i64
4033 // %C = and i64 %B, Constant
4034 // %D = icmp eq i64 %C, 0
4035 // call void @llvm.assume(i1 %D)
4036 // into
4037 // call void @llvm.assume(i1 true) [ "align"(ptr [[A]], i64 Constant + 1)]
4038 uint64_t AlignMask = 1;
4039 if ((match(IIOperand, m_Not(m_Trunc(m_Value(A)))) ||
4040 match(IIOperand,
4042 m_And(m_Value(A), m_ConstantInt(AlignMask)),
4043 m_Zero())))) {
4044 if (isPowerOf2_64(AlignMask + 1) &&
4046 Builder.CreateAlignmentAssumption(getDataLayout(), A, AlignMask + 1);
4047 return eraseInstFromFunction(*II);
4048 }
4049 }
4050
4051 // Remove assumes on true/false
4052 if (auto *CI = dyn_cast<ConstantInt>(IIOperand);
4053 CI || isa<UndefValue, PoisonValue>(IIOperand)) {
4054 if (!CI || CI->isZero())
4056 return eraseInstFromFunction(*II);
4057 }
4058
4059 // Update the cache of affected values for this assumption (we might be
4060 // here because we just simplified the condition).
4061 AC.updateAffectedValues(cast<AssumeInst>(II));
4062 break;
4063 }
4064 case Intrinsic::experimental_guard: {
4065 // Is this guard followed by another guard? We scan forward over a small
4066 // fixed window of instructions to handle common cases with conditions
4067 // computed between guards.
4068 Instruction *NextInst = II->getNextNode();
4069 for (unsigned i = 0; i < GuardWideningWindow; i++) {
4070 // Note: Using context-free form to avoid compile time blow up
4071 if (!isSafeToSpeculativelyExecute(NextInst))
4072 break;
4073 NextInst = NextInst->getNextNode();
4074 }
4075 Value *NextCond = nullptr;
4076 if (match(NextInst,
4078 Value *CurrCond = II->getArgOperand(0);
4079
4080 // Remove a guard that it is immediately preceded by an identical guard.
4081 // Otherwise canonicalize guard(a); guard(b) -> guard(a & b).
4082 if (CurrCond != NextCond) {
4083 Instruction *MoveI = II->getNextNode();
4084 while (MoveI != NextInst) {
4085 auto *Temp = MoveI;
4086 MoveI = MoveI->getNextNode();
4087 Temp->moveBefore(II->getIterator());
4088 }
4089 replaceOperand(*II, 0, Builder.CreateAnd(CurrCond, NextCond));
4090 }
4091 eraseInstFromFunction(*NextInst);
4092 return II;
4093 }
4094 break;
4095 }
4096 case Intrinsic::vector_insert: {
4097 Value *Vec = II->getArgOperand(0);
4098 Value *SubVec = II->getArgOperand(1);
4099 Value *Idx = II->getArgOperand(2);
4100 auto *DstTy = dyn_cast<FixedVectorType>(II->getType());
4101 auto *VecTy = dyn_cast<FixedVectorType>(Vec->getType());
4102 auto *SubVecTy = dyn_cast<FixedVectorType>(SubVec->getType());
4103
4104 // Only canonicalize if the destination vector, Vec, and SubVec are all
4105 // fixed vectors.
4106 if (DstTy && VecTy && SubVecTy) {
4107 unsigned DstNumElts = DstTy->getNumElements();
4108 unsigned VecNumElts = VecTy->getNumElements();
4109 unsigned SubVecNumElts = SubVecTy->getNumElements();
4110 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4111
4112 // An insert that entirely overwrites Vec with SubVec is a nop.
4113 if (VecNumElts == SubVecNumElts)
4114 return replaceInstUsesWith(CI, SubVec);
4115
4116 // Widen SubVec into a vector of the same width as Vec, since
4117 // shufflevector requires the two input vectors to be the same width.
4118 // Elements beyond the bounds of SubVec within the widened vector are
4119 // undefined.
4120 SmallVector<int, 8> WidenMask;
4121 unsigned i;
4122 for (i = 0; i != SubVecNumElts; ++i)
4123 WidenMask.push_back(i);
4124 for (; i != VecNumElts; ++i)
4125 WidenMask.push_back(PoisonMaskElem);
4126
4127 Value *WidenShuffle = Builder.CreateShuffleVector(SubVec, WidenMask);
4128
4130 for (unsigned i = 0; i != IdxN; ++i)
4131 Mask.push_back(i);
4132 for (unsigned i = DstNumElts; i != DstNumElts + SubVecNumElts; ++i)
4133 Mask.push_back(i);
4134 for (unsigned i = IdxN + SubVecNumElts; i != DstNumElts; ++i)
4135 Mask.push_back(i);
4136
4137 Value *Shuffle = Builder.CreateShuffleVector(Vec, WidenShuffle, Mask);
4138 return replaceInstUsesWith(CI, Shuffle);
4139 }
4140 break;
4141 }
4142 case Intrinsic::vector_extract: {
4143 Value *Vec = II->getArgOperand(0);
4144 Value *Idx = II->getArgOperand(1);
4145
4146 Type *ReturnType = II->getType();
4147 // (extract_vector (insert_vector InsertTuple, InsertValue, InsertIdx),
4148 // ExtractIdx)
4149 unsigned ExtractIdx = cast<ConstantInt>(Idx)->getZExtValue();
4150 Value *InsertTuple, *InsertIdx, *InsertValue;
4152 m_Value(InsertValue),
4153 m_Value(InsertIdx))) &&
4154 InsertValue->getType() == ReturnType) {
4155 unsigned Index = cast<ConstantInt>(InsertIdx)->getZExtValue();
4156 // Case where we get the same index right after setting it.
4157 // extract.vector(insert.vector(InsertTuple, InsertValue, Idx), Idx) -->
4158 // InsertValue
4159 if (ExtractIdx == Index)
4160 return replaceInstUsesWith(CI, InsertValue);
4161 // If we are getting a different index than what was set in the
4162 // insert.vector intrinsic. We can just set the input tuple to the one up
4163 // in the chain. extract.vector(insert.vector(InsertTuple, InsertValue,
4164 // InsertIndex), ExtractIndex)
4165 // --> extract.vector(InsertTuple, ExtractIndex)
4166 else
4167 return replaceOperand(CI, 0, InsertTuple);
4168 }
4169
4170 ConstantInt *ALMUpperBound;
4172 m_Value(), m_ConstantInt(ALMUpperBound)))) {
4173 const auto &Attrs = II->getFunction()->getAttributes().getFnAttrs();
4174 unsigned VScaleMin = Attrs.getVScaleRangeMin();
4175 unsigned ScaleFactor =
4176 cast<VectorType>(ReturnType)->isScalableTy() ? VScaleMin : 1;
4177 if (ExtractIdx * ScaleFactor >= ALMUpperBound->getZExtValue())
4178 return replaceInstUsesWith(CI,
4179 ConstantVector::getNullValue(ReturnType));
4180 }
4181
4182 auto *DstTy = dyn_cast<VectorType>(ReturnType);
4183 auto *VecTy = dyn_cast<VectorType>(Vec->getType());
4184
4185 if (DstTy && VecTy) {
4186 auto DstEltCnt = DstTy->getElementCount();
4187 auto VecEltCnt = VecTy->getElementCount();
4188 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4189
4190 // Extracting the entirety of Vec is a nop.
4191 if (DstEltCnt == VecTy->getElementCount()) {
4192 replaceInstUsesWith(CI, Vec);
4193 return eraseInstFromFunction(CI);
4194 }
4195
4196 // Only canonicalize to shufflevector if the destination vector and
4197 // Vec are fixed vectors.
4198 if (VecEltCnt.isScalable() || DstEltCnt.isScalable())
4199 break;
4200
4202 for (unsigned i = 0; i != DstEltCnt.getKnownMinValue(); ++i)
4203 Mask.push_back(IdxN + i);
4204
4205 Value *Shuffle = Builder.CreateShuffleVector(Vec, Mask);
4206 return replaceInstUsesWith(CI, Shuffle);
4207 }
4208 break;
4209 }
4210 case Intrinsic::experimental_vp_reverse: {
4211 Value *X;
4212 Value *Vec = II->getArgOperand(0);
4213 Value *Mask = II->getArgOperand(1);
4214 if (!match(Mask, m_AllOnes()))
4215 break;
4216 Value *EVL = II->getArgOperand(2);
4217 // TODO: Canonicalize experimental.vp.reverse after unop/binops?
4218 // rev(unop rev(X)) --> unop X
4219 if (match(Vec,
4221 m_Value(X), m_AllOnes(), m_Specific(EVL)))))) {
4222 auto *OldUnOp = cast<UnaryOperator>(Vec);
4224 OldUnOp->getOpcode(), X, OldUnOp, OldUnOp->getName(),
4225 II->getIterator());
4226 return replaceInstUsesWith(CI, NewUnOp);
4227 }
4228 break;
4229 }
4230 case Intrinsic::vector_reduce_or:
4231 case Intrinsic::vector_reduce_and: {
4232 // Canonicalize logical or/and reductions:
4233 // Or reduction for i1 is represented as:
4234 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4235 // %res = cmp ne iReduxWidth %val, 0
4236 // And reduction for i1 is represented as:
4237 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4238 // %res = cmp eq iReduxWidth %val, 11111
4239 Value *Arg = II->getArgOperand(0);
4240 Value *Vect;
4241
4242 if (Value *NewOp =
4243 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4244 replaceUse(II->getOperandUse(0), NewOp);
4245 return II;
4246 }
4247
4248 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4249 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4250 if (FTy->getElementType() == Builder.getInt1Ty()) {
4251 Value *Res = Builder.CreateBitCast(
4252 Vect, Builder.getIntNTy(FTy->getNumElements()));
4253 if (IID == Intrinsic::vector_reduce_and) {
4254 Res = Builder.CreateICmpEQ(
4256 } else {
4257 assert(IID == Intrinsic::vector_reduce_or &&
4258 "Expected or reduction.");
4259 Res = Builder.CreateIsNotNull(Res);
4260 }
4261 if (Arg != Vect)
4262 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4263 II->getType());
4264 return replaceInstUsesWith(CI, Res);
4265 }
4266 }
4267 [[fallthrough]];
4268 }
4269 case Intrinsic::vector_reduce_add: {
4270 if (IID == Intrinsic::vector_reduce_add) {
4271 // Convert vector_reduce_add(ZExt(<n x i1>)) to
4272 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4273 // Convert vector_reduce_add(SExt(<n x i1>)) to
4274 // -ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4275 // Convert vector_reduce_add(<n x i1>) to
4276 // Trunc(ctpop(bitcast <n x i1> to in)).
4277 Value *Arg = II->getArgOperand(0);
4278 Value *Vect;
4279
4280 if (Value *NewOp =
4281 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4282 replaceUse(II->getOperandUse(0), NewOp);
4283 return II;
4284 }
4285
4286 // vector.reduce.add.vNiM(splat(%x)) -> mul(%x, N)
4287 if (Value *Splat = getSplatValue(Arg)) {
4288 ElementCount VecToReduceCount =
4289 cast<VectorType>(Arg->getType())->getElementCount();
4290 if (VecToReduceCount.isFixed()) {
4291 unsigned VectorSize = VecToReduceCount.getFixedValue();
4292 return BinaryOperator::CreateMul(
4293 Splat,
4294 ConstantInt::get(Splat->getType(), VectorSize, /*IsSigned=*/false,
4295 /*ImplicitTrunc=*/true));
4296 }
4297 }
4298
4299 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4300 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4301 if (FTy->getElementType() == Builder.getInt1Ty()) {
4302 Value *V = Builder.CreateBitCast(
4303 Vect, Builder.getIntNTy(FTy->getNumElements()));
4304 Value *Res = Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
4305 Res = Builder.CreateZExtOrTrunc(Res, II->getType());
4306 if (Arg != Vect &&
4307 cast<Instruction>(Arg)->getOpcode() == Instruction::SExt)
4308 Res = Builder.CreateNeg(Res);
4309 return replaceInstUsesWith(CI, Res);
4310 }
4311 }
4312 }
4313 [[fallthrough]];
4314 }
4315 case Intrinsic::vector_reduce_xor: {
4316 if (IID == Intrinsic::vector_reduce_xor) {
4317 // Exclusive disjunction reduction over the vector with
4318 // (potentially-extended) i1 element type is actually a
4319 // (potentially-extended) arithmetic `add` reduction over the original
4320 // non-extended value:
4321 // vector_reduce_xor(?ext(<n x i1>))
4322 // -->
4323 // ?ext(vector_reduce_add(<n x i1>))
4324 Value *Arg = II->getArgOperand(0);
4325 Value *Vect;
4326
4327 if (Value *NewOp =
4328 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4329 replaceUse(II->getOperandUse(0), NewOp);
4330 return II;
4331 }
4332
4333 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4334 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4335 if (VTy->getElementType() == Builder.getInt1Ty()) {
4336 Value *Res = Builder.CreateAddReduce(Vect);
4337 if (Arg != Vect)
4338 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4339 II->getType());
4340 return replaceInstUsesWith(CI, Res);
4341 }
4342 }
4343 }
4344 [[fallthrough]];
4345 }
4346 case Intrinsic::vector_reduce_mul: {
4347 if (IID == Intrinsic::vector_reduce_mul) {
4348 Value *Arg = II->getArgOperand(0);
4349
4350 if (Value *NewOp =
4351 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4352 replaceUse(II->getOperandUse(0), NewOp);
4353 return II;
4354 }
4355
4356 // vector_reduce_mul(zext(<n x i1>)), or
4357 // vector_reduce_mul(sext(<n x i1>)) (if n is even) -->
4358 // zext(vector_reduce_and(<n x i1>)).
4359 // (The sext case doesn't work if n is odd because multiplying an odd
4360 // number of -1's produces -1, not 1.)
4361 Value *Vect;
4362 bool IsZext = match(Arg, m_ZExt(m_Value(Vect))) &&
4363 Vect->getType()->isIntOrIntVectorTy(1);
4364 bool IsSext =
4365 match(Arg, m_SExt(m_Value(Vect))) &&
4366 Vect->getType()->isIntOrIntVectorTy(1) &&
4367 cast<VectorType>(Vect->getType())->getElementCount().isKnownEven();
4368 if (IsZext || IsSext) {
4369 Value *Res = Builder.CreateAndReduce(Vect);
4370 return CastInst::Create(Instruction::ZExt, Res, II->getType());
4371 }
4372
4373 // vector_reduce_mul(<n x i1>) --> vector_reduce_and(<n x i1>)
4374 if (Arg->getType()->isIntOrIntVectorTy(1))
4375 return replaceInstUsesWith(CI, Builder.CreateAndReduce(Arg));
4376 }
4377 [[fallthrough]];
4378 }
4379 case Intrinsic::vector_reduce_umin:
4380 case Intrinsic::vector_reduce_umax: {
4381 if (IID == Intrinsic::vector_reduce_umin ||
4382 IID == Intrinsic::vector_reduce_umax) {
4383 // UMin/UMax reduction over the vector with (potentially-extended)
4384 // i1 element type is actually a (potentially-extended)
4385 // logical `and`/`or` reduction over the original non-extended value:
4386 // vector_reduce_u{min,max}(?ext(<n x i1>))
4387 // -->
4388 // ?ext(vector_reduce_{and,or}(<n x i1>))
4389 Value *Arg = II->getArgOperand(0);
4390 Value *Vect;
4391
4392 if (Value *NewOp =
4393 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4394 replaceUse(II->getOperandUse(0), NewOp);
4395 return II;
4396 }
4397
4398 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4399 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4400 if (VTy->getElementType() == Builder.getInt1Ty()) {
4401 Value *Res = IID == Intrinsic::vector_reduce_umin
4402 ? Builder.CreateAndReduce(Vect)
4403 : Builder.CreateOrReduce(Vect);
4404 if (Arg != Vect)
4405 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4406 II->getType());
4407 return replaceInstUsesWith(CI, Res);
4408 }
4409 }
4410 }
4411 [[fallthrough]];
4412 }
4413 case Intrinsic::vector_reduce_smin:
4414 case Intrinsic::vector_reduce_smax: {
4415 if (IID == Intrinsic::vector_reduce_smin ||
4416 IID == Intrinsic::vector_reduce_smax) {
4417 // SMin/SMax reduction over the vector with (potentially-extended)
4418 // i1 element type is actually a (potentially-extended)
4419 // logical `and`/`or` reduction over the original non-extended value:
4420 // vector_reduce_s{min,max}(<n x i1>)
4421 // -->
4422 // vector_reduce_{or,and}(<n x i1>)
4423 // and
4424 // vector_reduce_s{min,max}(sext(<n x i1>))
4425 // -->
4426 // sext(vector_reduce_{or,and}(<n x i1>))
4427 // and
4428 // vector_reduce_s{min,max}(zext(<n x i1>))
4429 // -->
4430 // zext(vector_reduce_{and,or}(<n x i1>))
4431 Value *Arg = II->getArgOperand(0);
4432 Value *Vect;
4433
4434 if (Value *NewOp =
4435 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4436 replaceUse(II->getOperandUse(0), NewOp);
4437 return II;
4438 }
4439
4440 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4441 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4442 if (VTy->getElementType() == Builder.getInt1Ty()) {
4443 Instruction::CastOps ExtOpc = Instruction::CastOps::CastOpsEnd;
4444 if (Arg != Vect)
4445 ExtOpc = cast<CastInst>(Arg)->getOpcode();
4446 Value *Res = ((IID == Intrinsic::vector_reduce_smin) ==
4447 (ExtOpc == Instruction::CastOps::ZExt))
4448 ? Builder.CreateAndReduce(Vect)
4449 : Builder.CreateOrReduce(Vect);
4450 if (Arg != Vect)
4451 Res = Builder.CreateCast(ExtOpc, Res, II->getType());
4452 return replaceInstUsesWith(CI, Res);
4453 }
4454 }
4455 }
4456 [[fallthrough]];
4457 }
4458 case Intrinsic::vector_reduce_fmax:
4459 case Intrinsic::vector_reduce_fmin:
4460 case Intrinsic::vector_reduce_fadd:
4461 case Intrinsic::vector_reduce_fmul: {
4462 bool CanReorderLanes = (IID != Intrinsic::vector_reduce_fadd &&
4463 IID != Intrinsic::vector_reduce_fmul) ||
4464 II->hasAllowReassoc();
4465 const unsigned ArgIdx = (IID == Intrinsic::vector_reduce_fadd ||
4466 IID == Intrinsic::vector_reduce_fmul)
4467 ? 1
4468 : 0;
4469 Value *Arg = II->getArgOperand(ArgIdx);
4470 if (Value *NewOp = simplifyReductionOperand(Arg, CanReorderLanes)) {
4471 replaceUse(II->getOperandUse(ArgIdx), NewOp);
4472 return nullptr;
4473 }
4474 break;
4475 }
4476 case Intrinsic::is_fpclass: {
4477 if (Instruction *I = foldIntrinsicIsFPClass(*II))
4478 return I;
4479 break;
4480 }
4481 case Intrinsic::threadlocal_address: {
4482 Align MinAlign = getKnownAlignment(II->getArgOperand(0), DL, II, &AC, &DT);
4483 MaybeAlign Align = II->getRetAlign();
4484 if (MinAlign > Align.valueOrOne()) {
4485 II->addRetAttr(Attribute::getWithAlignment(II->getContext(), MinAlign));
4486 return II;
4487 }
4488 break;
4489 }
4490 case Intrinsic::fptoui_sat:
4491 case Intrinsic::fptosi_sat:
4492 if (Instruction *I = foldItoFPtoI(*II))
4493 return I;
4494 break;
4495 case Intrinsic::frexp: {
4496 // frexp(frexp(x).fract) -> { frexp(x).fract, 0 }: the fraction operand is
4497 // already normalized, so the first result is idempotent and the second is
4498 // zero.
4499 if (match(II->getArgOperand(0),
4501 Value *Res = Builder.CreateInsertValue(PoisonValue::get(II->getType()),
4502 II->getArgOperand(0), 0);
4503 Res = Builder.CreateInsertValue(
4504 Res, Constant::getNullValue(II->getType()->getStructElementType(1)),
4505 1);
4506 return replaceInstUsesWith(*II, Res);
4507 }
4508 break;
4509 }
4510 case Intrinsic::get_active_lane_mask: {
4511 const APInt *Op0, *Op1;
4512 if (match(II->getOperand(0), m_StrictlyPositive(Op0)) &&
4513 match(II->getOperand(1), m_APInt(Op1))) {
4514 Type *OpTy = II->getOperand(0)->getType();
4515 return replaceInstUsesWith(
4516 *II, Builder.CreateIntrinsic(
4517 II->getType(), Intrinsic::get_active_lane_mask,
4518 {Constant::getNullValue(OpTy),
4519 ConstantInt::get(OpTy, Op1->usub_sat(*Op0))}));
4520 }
4521 break;
4522 }
4523 case Intrinsic::experimental_get_vector_length: {
4524 // get.vector.length(Cnt, MaxLanes) --> Cnt when Cnt <= MaxLanes
4525 unsigned BitWidth =
4526 std::max(II->getArgOperand(0)->getType()->getScalarSizeInBits(),
4527 II->getType()->getScalarSizeInBits());
4528 ConstantRange Cnt =
4529 computeConstantRangeIncludingKnownBits(II->getArgOperand(0), false,
4530 SQ.getWithInstruction(II))
4532 ConstantRange MaxLanes = cast<ConstantInt>(II->getArgOperand(1))
4533 ->getValue()
4534 .zextOrTrunc(Cnt.getBitWidth());
4535 if (cast<ConstantInt>(II->getArgOperand(2))->isOne())
4536 MaxLanes = MaxLanes.multiply(
4537 getVScaleRange(II->getFunction(), Cnt.getBitWidth()));
4538
4539 if (Cnt.icmp(CmpInst::ICMP_ULE, MaxLanes))
4540 return replaceInstUsesWith(
4541 *II, Builder.CreateZExtOrTrunc(II->getArgOperand(0), II->getType()));
4542 return nullptr;
4543 }
4544 default: {
4545 // Handle target specific intrinsics
4546 std::optional<Instruction *> V = targetInstCombineIntrinsic(*II);
4547 if (V)
4548 return *V;
4549 break;
4550 }
4551 }
4552
4553 // Try to fold intrinsic into select/phi operands. This is legal if:
4554 // * The intrinsic is speculatable.
4555 // * The operand is one of the following:
4556 // - a phi.
4557 // - a select with a scalar condition.
4558 // - a select with a vector condition and II is not a cross lane operation.
4560 for (Value *Op : II->args()) {
4561 if (auto *Sel = dyn_cast<SelectInst>(Op)) {
4562 bool IsVectorCond = Sel->getCondition()->getType()->isVectorTy();
4563 if (IsVectorCond &&
4564 (!isNotCrossLaneOperation(II) || !II->getType()->isVectorTy()))
4565 continue;
4566 // Don't replace a scalar select with a more expensive vector select if
4567 // we can't simplify both arms of the select.
4568 bool SimplifyBothArms =
4569 !Op->getType()->isVectorTy() && II->getType()->isVectorTy();
4571 *II, Sel, /*FoldWithMultiUse=*/false, SimplifyBothArms))
4572 return R;
4573 }
4574 if (auto *Phi = dyn_cast<PHINode>(Op))
4575 if (Instruction *R = foldOpIntoPhi(*II, Phi))
4576 return R;
4577 }
4578 }
4579
4581 return Shuf;
4582
4584 return replaceInstUsesWith(*II, Reverse);
4585
4587 return replaceInstUsesWith(*II, Res);
4588
4589 // Some intrinsics (like experimental_gc_statepoint) can be used in invoke
4590 // context, so it is handled in visitCallBase and we should trigger it.
4591 return visitCallBase(*II);
4592}
4593
4594// Fence instruction simplification
4596 auto *NFI = dyn_cast<FenceInst>(FI.getNextNode());
4597 // This check is solely here to handle arbitrary target-dependent syncscopes.
4598 // TODO: Can remove if does not matter in practice.
4599 if (NFI && FI.isIdenticalTo(NFI))
4600 return eraseInstFromFunction(FI);
4601
4602 // Returns true if FI1 is identical or stronger fence than FI2.
4603 auto isIdenticalOrStrongerFence = [](FenceInst *FI1, FenceInst *FI2) {
4604 auto FI1SyncScope = FI1->getSyncScopeID();
4605 // Consider same scope, where scope is global or single-thread.
4606 if (FI1SyncScope != FI2->getSyncScopeID() ||
4607 (FI1SyncScope != SyncScope::System &&
4608 FI1SyncScope != SyncScope::SingleThread))
4609 return false;
4610
4611 return isAtLeastOrStrongerThan(FI1->getOrdering(), FI2->getOrdering());
4612 };
4613 if (NFI && isIdenticalOrStrongerFence(NFI, &FI))
4614 return eraseInstFromFunction(FI);
4615
4616 if (auto *PFI = dyn_cast_or_null<FenceInst>(FI.getPrevNode()))
4617 if (isIdenticalOrStrongerFence(PFI, &FI))
4618 return eraseInstFromFunction(FI);
4619 return nullptr;
4620}
4621
4622// InvokeInst simplification
4624 return visitCallBase(II);
4625}
4626
4627// CallBrInst simplification
4629 return visitCallBase(CBI);
4630}
4631
4632// A simple parser for format string specifiers for the purposes of the
4633// modular-format attribute. In the case of malformed format strings this might
4634// under or over report the specifiers present, but such cases are undefined
4635// behavior.
4637 Bitset<256> Specifiers;
4638 for (size_t I = 0; I < FormatStr.size(); ++I) {
4639 if (FormatStr[I] != '%')
4640 continue;
4641
4642 // Check for escaped '%'.
4643 if (I + 1 < FormatStr.size() && FormatStr[I + 1] == '%') {
4644 ++I; // Skip the second '%'.
4645 continue;
4646 }
4647
4648 // Scan past allowed prefix characters.
4649 size_t J =
4650 FormatStr.find_first_not_of("0123456789-+ #0$.*'hlLjztqwvI", I + 1);
4651 if (J == StringRef::npos)
4652 break;
4653
4654 Specifiers.set(static_cast<unsigned char>(FormatStr[J]));
4655 I = J; // Resume search from after the specifier.
4656 }
4657 return Specifiers;
4658}
4659
4660static bool isAspectNeeded(StringRef Aspect, CallInst *CI,
4661 std::optional<unsigned> FirstArgIdx,
4662 const std::optional<Bitset<256>> &Specifiers) {
4663 if (Aspect == "float") {
4664 if (Specifiers) {
4665 static constexpr Bitset<256> FloatSpecifiers{'f', 'F', 'e', 'E',
4666 'g', 'G', 'a', 'A'};
4667 return (*Specifiers & FloatSpecifiers).any();
4668 }
4669 // Fallback to type-based check for dynamic format string.
4670 if (!FirstArgIdx)
4671 return true;
4672 return llvm::any_of(
4673 llvm::make_range(std::next(CI->arg_begin(), *FirstArgIdx),
4674 CI->arg_end()),
4675 [](Value *V) { return V->getType()->isFloatingPointTy(); });
4676 }
4677 if (Aspect == "fixed") {
4678 if (Specifiers) {
4679 static constexpr Bitset<256> FixedSpecifiers{'r', 'R', 'k', 'K'};
4680 return (*Specifiers & FixedSpecifiers).any();
4681 }
4682 // Fallback for fixed-point: assume needed if format is dynamic.
4683 return true;
4684 }
4685 // Unknown aspects are always considered to be needed.
4686 return true;
4687}
4688
4689static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M,
4690 IRBuilderBase &B) {
4691 SmallString<20> Name = ImplName;
4692 Name += '_';
4693 Name += Aspect;
4694 LLVMContext &Ctx = M->getContext();
4695 Function *RelocNoneFn =
4696 Intrinsic::getOrInsertDeclaration(M, Intrinsic::reloc_none);
4697 B.CreateCall(RelocNoneFn,
4698 {MetadataAsValue::get(Ctx, MDString::get(Ctx, Name))});
4699}
4700
4702 if (!CI->hasFnAttr("modular-format"))
4703 return nullptr;
4704
4706 llvm::split(CI->getFnAttr("modular-format").getValueAsString(), ','));
4707 if (Args.size() < 5)
4708 return nullptr;
4709
4710 StringRef FormatIdxStr = Args[1];
4711 StringRef FirstArgIdxStr = Args[2];
4712 StringRef FnName = Args[3];
4713 StringRef ImplName = Args[4];
4715
4716 unsigned FormatIdx;
4717 std::optional<unsigned> FirstArgIdx;
4718 [[maybe_unused]] bool Error;
4719 Error = FormatIdxStr.getAsInteger(10, FormatIdx);
4720 assert(!Error && "invalid format arg index");
4721 --FormatIdx; // 1-based to 0-based
4722
4723 FirstArgIdx.emplace();
4724 Error = FirstArgIdxStr.getAsInteger(10, *FirstArgIdx);
4725 assert(!Error && "invalid first arg index");
4726 if (*FirstArgIdx > 0)
4727 --*FirstArgIdx; // 1-based to 0-based
4728 else
4729 FirstArgIdx.reset();
4730
4731 if (AllAspects.empty())
4732 return nullptr;
4733
4734 Value *FormatVal = CI->getArgOperand(FormatIdx);
4735 StringRef FormatStr;
4736
4737 std::optional<Bitset<256>> Specifiers;
4738 if (getConstantStringInfo(FormatVal, FormatStr))
4739 Specifiers = parseFormatStringSpecifiers(FormatStr);
4740
4741 SmallVector<StringRef> NeededAspects;
4742 for (StringRef Aspect : AllAspects)
4743 if (isAspectNeeded(Aspect, CI, FirstArgIdx, Specifiers))
4744 NeededAspects.push_back(Aspect);
4745
4746 if (NeededAspects.size() == AllAspects.size())
4747 return nullptr;
4748
4749 Module *M = CI->getModule();
4750 LLVMContext &Ctx = M->getContext();
4751 Function *Callee = CI->getCalledFunction();
4752 FunctionCallee ModularFn = M->getOrInsertFunction(
4753 FnName, Callee->getFunctionType(),
4754 Callee->getAttributes().removeFnAttribute(Ctx, "modular-format"));
4755 CallInst *New = cast<CallInst>(CI->clone());
4756 New->setCalledFunction(ModularFn);
4757 New->removeFnAttr("modular-format");
4758 B.Insert(New);
4759
4760 llvm::sort(NeededAspects);
4761 for (StringRef Request : NeededAspects)
4762 referenceAspect(Request, ImplName, M, B);
4763
4764 return New;
4765}
4766
4767Instruction *InstCombinerImpl::tryOptimizeCall(CallInst *CI) {
4768 if (!CI->getCalledFunction()) return nullptr;
4769
4770 // Skip optimizing notail and musttail calls so
4771 // LibCallSimplifier::optimizeCall doesn't have to preserve those invariants.
4772 // LibCallSimplifier::optimizeCall should try to preserve tail calls though.
4773 if (CI->isMustTailCall() || CI->isNoTailCall())
4774 return nullptr;
4775
4776 auto InstCombineRAUW = [this](Instruction *From, Value *With) {
4777 replaceInstUsesWith(*From, With);
4778 };
4779 auto InstCombineErase = [this](Instruction *I) {
4781 };
4782 LibCallSimplifier Simplifier(DL, &TLI, &DT, &DC, &AC, ORE, BFI, PSI,
4783 InstCombineRAUW, InstCombineErase);
4784 if (Value *With = Simplifier.optimizeCall(CI, Builder)) {
4785 ++NumSimplified;
4786 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4787 }
4788 if (Value *With = optimizeModularFormat(CI, Builder)) {
4789 ++NumSimplified;
4790 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4791 }
4792
4793 return nullptr;
4794}
4795
4797 // Strip off at most one level of pointer casts, looking for an alloca. This
4798 // is good enough in practice and simpler than handling any number of casts.
4799 Value *Underlying = TrampMem->stripPointerCasts();
4800 if (Underlying != TrampMem &&
4801 (!Underlying->hasOneUse() || Underlying->user_back() != TrampMem))
4802 return nullptr;
4803 if (!isa<AllocaInst>(Underlying))
4804 return nullptr;
4805
4806 IntrinsicInst *InitTrampoline = nullptr;
4807 for (User *U : TrampMem->users()) {
4809 if (!II)
4810 return nullptr;
4811 if (II->getIntrinsicID() == Intrinsic::init_trampoline) {
4812 if (InitTrampoline)
4813 // More than one init_trampoline writes to this value. Give up.
4814 return nullptr;
4815 InitTrampoline = II;
4816 continue;
4817 }
4818 if (II->getIntrinsicID() == Intrinsic::adjust_trampoline)
4819 // Allow any number of calls to adjust.trampoline.
4820 continue;
4821 return nullptr;
4822 }
4823
4824 // No call to init.trampoline found.
4825 if (!InitTrampoline)
4826 return nullptr;
4827
4828 // Check that the alloca is being used in the expected way.
4829 if (InitTrampoline->getOperand(0) != TrampMem)
4830 return nullptr;
4831
4832 return InitTrampoline;
4833}
4834
4836 Value *TrampMem) {
4837 // Visit all the previous instructions in the basic block, and try to find a
4838 // init.trampoline which has a direct path to the adjust.trampoline.
4839 for (BasicBlock::iterator I = AdjustTramp->getIterator(),
4840 E = AdjustTramp->getParent()->begin();
4841 I != E;) {
4842 Instruction *Inst = &*--I;
4844 if (II->getIntrinsicID() == Intrinsic::init_trampoline &&
4845 II->getOperand(0) == TrampMem)
4846 return II;
4847 if (Inst->mayWriteToMemory())
4848 return nullptr;
4849 }
4850 return nullptr;
4851}
4852
4853// Given a call to llvm.adjust.trampoline, find and return the corresponding
4854// call to llvm.init.trampoline if the call to the trampoline can be optimized
4855// to a direct call to a function. Otherwise return NULL.
4857 Callee = Callee->stripPointerCasts();
4858 IntrinsicInst *AdjustTramp = dyn_cast<IntrinsicInst>(Callee);
4859 if (!AdjustTramp ||
4860 AdjustTramp->getIntrinsicID() != Intrinsic::adjust_trampoline)
4861 return nullptr;
4862
4863 Value *TrampMem = AdjustTramp->getOperand(0);
4864
4866 return IT;
4867 if (IntrinsicInst *IT = findInitTrampolineFromBB(AdjustTramp, TrampMem))
4868 return IT;
4869 return nullptr;
4870}
4871
4872Instruction *InstCombinerImpl::foldPtrAuthIntrinsicCallee(CallBase &Call) {
4873 const Value *Callee = Call.getCalledOperand();
4874 const auto *IPC = dyn_cast<IntToPtrInst>(Callee);
4875 if (!IPC || !IPC->isNoopCast(DL))
4876 return nullptr;
4877
4878 const auto *II = dyn_cast<IntrinsicInst>(IPC->getOperand(0));
4879 if (!II)
4880 return nullptr;
4881
4882 Intrinsic::ID IIID = II->getIntrinsicID();
4883 if (IIID != Intrinsic::ptrauth_resign && IIID != Intrinsic::ptrauth_sign)
4884 return nullptr;
4885
4886 // Isolate the ptrauth bundle from the others.
4887 std::optional<OperandBundleUse> PtrAuthBundleOrNone;
4889 for (unsigned BI = 0, BE = Call.getNumOperandBundles(); BI != BE; ++BI) {
4890 OperandBundleUse Bundle = Call.getOperandBundleAt(BI);
4891 if (Bundle.getTagID() == LLVMContext::OB_ptrauth)
4892 PtrAuthBundleOrNone = Bundle;
4893 else
4894 NewBundles.emplace_back(Bundle);
4895 }
4896
4897 if (!PtrAuthBundleOrNone)
4898 return nullptr;
4899
4900 Value *NewCallee = nullptr;
4901 switch (IIID) {
4902 // call(ptrauth.resign(p)), ["ptrauth"()] -> call p, ["ptrauth"()]
4903 // assuming the call bundle and the sign operands match.
4904 case Intrinsic::ptrauth_resign: {
4905 // Resign result key should match bundle.
4906 if (II->getOperand(3) != PtrAuthBundleOrNone->Inputs[0])
4907 return nullptr;
4908 // Resign result discriminator should match bundle.
4909 if (II->getOperand(4) != PtrAuthBundleOrNone->Inputs[1])
4910 return nullptr;
4911
4912 // Resign input (auth) key should also match: we can't change the key on
4913 // the new call we're generating, because we don't know what keys are valid.
4914 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4915 return nullptr;
4916
4917 Value *NewBundleOps[] = {II->getOperand(1), II->getOperand(2)};
4918 NewBundles.emplace_back("ptrauth", NewBundleOps);
4919 NewCallee = II->getOperand(0);
4920 break;
4921 }
4922
4923 // call(ptrauth.sign(p)), ["ptrauth"()] -> call p
4924 // assuming the call bundle and the sign operands match.
4925 // Non-ptrauth indirect calls are undesirable, but so is ptrauth.sign.
4926 case Intrinsic::ptrauth_sign: {
4927 // Sign key should match bundle.
4928 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4929 return nullptr;
4930 // Sign discriminator should match bundle.
4931 if (II->getOperand(2) != PtrAuthBundleOrNone->Inputs[1])
4932 return nullptr;
4933 NewCallee = II->getOperand(0);
4934 break;
4935 }
4936 default:
4937 llvm_unreachable("unexpected intrinsic ID");
4938 }
4939
4940 if (!NewCallee)
4941 return nullptr;
4942
4943 NewCallee = Builder.CreateBitOrPointerCast(NewCallee, Callee->getType());
4944 CallBase *NewCall = CallBase::Create(&Call, NewBundles);
4945 NewCall->setCalledOperand(NewCallee);
4946 return NewCall;
4947}
4948
4949Instruction *InstCombinerImpl::foldPtrAuthConstantCallee(CallBase &Call) {
4951 if (!CPA)
4952 return nullptr;
4953
4954 auto *CalleeF = dyn_cast<Function>(CPA->getPointer());
4955 // If the ptrauth constant isn't based on a function pointer, bail out.
4956 if (!CalleeF)
4957 return nullptr;
4958
4959 // Inspect the call ptrauth bundle to check it matches the ptrauth constant.
4961 if (!PAB)
4962 return nullptr;
4963
4964 auto *Key = cast<ConstantInt>(PAB->Inputs[0]);
4965 Value *Discriminator = PAB->Inputs[1];
4966
4967 // If the bundle doesn't match, this is probably going to fail to auth.
4968 if (!CPA->isKnownCompatibleWith(Key, Discriminator, DL))
4969 return nullptr;
4970
4971 // If the bundle matches the constant, proceed in making this a direct call.
4973 NewCall->setCalledOperand(CalleeF);
4974 return NewCall;
4975}
4976
4977bool InstCombinerImpl::annotateAnyAllocSite(CallBase &Call,
4978 const TargetLibraryInfo *TLI) {
4979 // Note: We only handle cases which can't be driven from generic attributes
4980 // here. So, for example, nonnull and noalias (which are common properties
4981 // of some allocation functions) are expected to be handled via annotation
4982 // of the respective allocator declaration with generic attributes.
4983 bool Changed = false;
4984
4985 if (!Call.getType()->isPointerTy())
4986 return Changed;
4987
4988 std::optional<APInt> Size = getAllocSize(&Call, TLI);
4989 if (Size && *Size != 0) {
4990 // TODO: We really should just emit deref_or_null here and then
4991 // let the generic inference code combine that with nonnull.
4992 if (Call.hasRetAttr(Attribute::NonNull)) {
4993 Changed = !Call.hasRetAttr(Attribute::Dereferenceable);
4995 Call.getContext(), Size->getLimitedValue()));
4996 } else {
4997 Changed = !Call.hasRetAttr(Attribute::DereferenceableOrNull);
4999 Call.getContext(), Size->getLimitedValue()));
5000 }
5001 }
5002
5003 // Add alignment attribute if alignment is a power of two constant.
5005 if (!Alignment)
5006 return Changed;
5007
5008 ConstantInt *AlignOpC = dyn_cast<ConstantInt>(Alignment);
5009 if (AlignOpC && AlignOpC->getValue().ult(llvm::Value::MaximumAlignment)) {
5010 uint64_t AlignmentVal = AlignOpC->getZExtValue();
5011 if (llvm::isPowerOf2_64(AlignmentVal)) {
5012 Align ExistingAlign = Call.getRetAlign().valueOrOne();
5013 Align NewAlign = Align(AlignmentVal);
5014 if (NewAlign > ExistingAlign) {
5017 Changed = true;
5018 }
5019 }
5020 }
5021 return Changed;
5022}
5023
5024/// Improvements for call, callbr and invoke instructions.
5025Instruction *InstCombinerImpl::visitCallBase(CallBase &Call) {
5026 bool Changed = annotateAnyAllocSite(Call, &TLI);
5027
5028 // Mark any parameters that are known to be non-null with the nonnull
5029 // attribute. This is helpful for inlining calls to functions with null
5030 // checks on their arguments.
5031 SmallVector<unsigned, 4> ArgNos;
5032 unsigned ArgNo = 0;
5033
5034 for (Value *V : Call.args()) {
5035 if (V->getType()->isPointerTy()) {
5036 // Simplify the nonnull operand if the parameter is known to be nonnull.
5037 // Otherwise, try to infer nonnull for it.
5038 bool HasDereferenceable = Call.getParamDereferenceableBytes(ArgNo) > 0;
5039 if (Call.paramHasAttr(ArgNo, Attribute::NonNull) ||
5040 (HasDereferenceable &&
5042 V->getType()->getPointerAddressSpace()))) {
5043 if (Value *Res = simplifyNonNullOperand(V, HasDereferenceable)) {
5044 replaceOperand(Call, ArgNo, Res);
5045 Changed = true;
5046 }
5047 } else if (isKnownNonZero(V,
5048 getSimplifyQuery().getWithInstruction(&Call))) {
5049 ArgNos.push_back(ArgNo);
5050 }
5051 }
5052 ArgNo++;
5053 }
5054
5055 assert(ArgNo == Call.arg_size() && "Call arguments not processed correctly.");
5056
5057 if (!ArgNos.empty()) {
5058 AttributeList AS = Call.getAttributes();
5059 LLVMContext &Ctx = Call.getContext();
5060 AS = AS.addParamAttribute(Ctx, ArgNos,
5061 Attribute::get(Ctx, Attribute::NonNull));
5062 Call.setAttributes(AS);
5063 Changed = true;
5064 }
5065
5066 // If the callee is a pointer to a function, attempt to move any casts to the
5067 // arguments of the call/callbr/invoke.
5069 Function *CalleeF = dyn_cast<Function>(Callee);
5070 if ((!CalleeF || CalleeF->getFunctionType() != Call.getFunctionType()) &&
5071 transformConstExprCastCall(Call))
5072 return nullptr;
5073
5074 if (CalleeF) {
5075 // Remove the convergent attr on calls when the callee is not convergent.
5076 if (Call.isConvergent() && !CalleeF->isConvergent() &&
5077 !CalleeF->isIntrinsic()) {
5078 LLVM_DEBUG(dbgs() << "Removing convergent attr from instr " << Call
5079 << "\n");
5081 return &Call;
5082 }
5083
5084 // If the call and callee calling conventions don't match, and neither one
5085 // of the calling conventions is compatible with C calling convention
5086 // this call must be unreachable, as the call is undefined.
5087 if ((CalleeF->getCallingConv() != Call.getCallingConv() &&
5088 !(CalleeF->getCallingConv() == llvm::CallingConv::C &&
5092 // Only do this for calls to a function with a body. A prototype may
5093 // not actually end up matching the implementation's calling conv for a
5094 // variety of reasons (e.g. it may be written in assembly).
5095 !CalleeF->isDeclaration()) {
5096 Instruction *OldCall = &Call;
5098 // If OldCall does not return void then replaceInstUsesWith poison.
5099 // This allows ValueHandlers and custom metadata to adjust itself.
5100 if (!OldCall->getType()->isVoidTy())
5101 replaceInstUsesWith(*OldCall, PoisonValue::get(OldCall->getType()));
5102 if (isa<CallInst>(OldCall))
5103 return eraseInstFromFunction(*OldCall);
5104
5105 // We cannot remove an invoke or a callbr, because it would change thexi
5106 // CFG, just change the callee to a null pointer.
5107 cast<CallBase>(OldCall)->setCalledFunction(
5108 CalleeF->getFunctionType(),
5109 Constant::getNullValue(CalleeF->getType()));
5110 return nullptr;
5111 }
5112 }
5113
5114 // Calling a null function pointer is undefined if a null address isn't
5115 // dereferenceable.
5116 if ((isa<ConstantPointerNull>(Callee) &&
5118 isa<UndefValue>(Callee)) {
5119 // If Call does not return void then replaceInstUsesWith poison.
5120 // This allows ValueHandlers and custom metadata to adjust itself.
5121 if (!Call.getType()->isVoidTy())
5123
5124 if (Call.isTerminator()) {
5125 // Can't remove an invoke or callbr because we cannot change the CFG.
5126 return nullptr;
5127 }
5128
5129 // This instruction is not reachable, just remove it.
5132 }
5133
5134 if (IntrinsicInst *II = findInitTrampoline(Callee))
5135 return transformCallThroughTrampoline(Call, *II);
5136
5137 // Combine calls involving pointer authentication intrinsics.
5138 if (Instruction *NewCall = foldPtrAuthIntrinsicCallee(Call))
5139 return NewCall;
5140
5141 // Combine calls to ptrauth constants.
5142 if (Instruction *NewCall = foldPtrAuthConstantCallee(Call))
5143 return NewCall;
5144
5145 if (isa<InlineAsm>(Callee) && !Call.doesNotThrow()) {
5146 InlineAsm *IA = cast<InlineAsm>(Callee);
5147 if (!IA->canThrow()) {
5148 // Normal inline asm calls cannot throw - mark them
5149 // 'nounwind'.
5151 Changed = true;
5152 }
5153 }
5154
5155 // Try to optimize the call if possible, we require DataLayout for most of
5156 // this. None of these calls are seen as possibly dead so go ahead and
5157 // delete the instruction now.
5158 if (CallInst *CI = dyn_cast<CallInst>(&Call)) {
5159 Instruction *I = tryOptimizeCall(CI);
5160 // If we changed something return the result, etc. Otherwise let
5161 // the fallthrough check.
5162 if (I) return eraseInstFromFunction(*I);
5163 }
5164
5165 if (!Call.use_empty() && !Call.isMustTailCall())
5166 if (Value *ReturnedArg = Call.getReturnedArgOperand()) {
5167 Type *CallTy = Call.getType();
5168 Type *RetArgTy = ReturnedArg->getType();
5169 if (RetArgTy->canLosslesslyBitCastTo(CallTy))
5170 return replaceInstUsesWith(
5171 Call, Builder.CreateBitOrPointerCast(ReturnedArg, CallTy));
5172 }
5173
5174 // Drop unnecessary callee_type metadata from calls that were converted
5175 // into direct calls.
5176 if (Call.getMetadata(LLVMContext::MD_callee_type) && !Call.isIndirectCall()) {
5177 Call.setMetadata(LLVMContext::MD_callee_type, nullptr);
5178 Changed = true;
5179 }
5180
5181 // Drop unnecessary kcfi operand bundles from calls that were converted
5182 // into direct calls.
5184 if (Bundle && !Call.isIndirectCall()) {
5185 DEBUG_WITH_TYPE(DEBUG_TYPE "-kcfi", {
5186 if (CalleeF) {
5187 ConstantInt *FunctionType = nullptr;
5188 ConstantInt *ExpectedType = cast<ConstantInt>(Bundle->Inputs[0]);
5189
5190 if (MDNode *MD = CalleeF->getMetadata(LLVMContext::MD_kcfi_type))
5191 FunctionType = mdconst::extract<ConstantInt>(MD->getOperand(0));
5192
5193 if (FunctionType &&
5194 FunctionType->getZExtValue() != ExpectedType->getZExtValue())
5195 dbgs() << Call.getModule()->getName()
5196 << ": warning: kcfi: " << Call.getCaller()->getName()
5197 << ": call to " << CalleeF->getName()
5198 << " using a mismatching function pointer type\n";
5199 }
5200 });
5201
5203 }
5204
5205 if (isRemovableAlloc(&Call, &TLI))
5206 return visitAllocSite(Call);
5207
5208 // Handle intrinsics which can be used in both call and invoke context.
5209 switch (Call.getIntrinsicID()) {
5210 case Intrinsic::experimental_gc_statepoint: {
5211 GCStatepointInst &GCSP = *cast<GCStatepointInst>(&Call);
5212 SmallPtrSet<Value *, 32> LiveGcValues;
5213 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5214 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5215
5216 // Remove the relocation if unused.
5217 if (GCR.use_empty()) {
5219 continue;
5220 }
5221
5222 Value *DerivedPtr = GCR.getDerivedPtr();
5223 Value *BasePtr = GCR.getBasePtr();
5224
5225 // Undef is undef, even after relocation.
5226 if (isa<UndefValue>(DerivedPtr) || isa<UndefValue>(BasePtr)) {
5229 continue;
5230 }
5231
5232 if (auto *PT = dyn_cast<PointerType>(GCR.getType())) {
5233 // The relocation of null will be null for most any collector.
5234 // TODO: provide a hook for this in GCStrategy. There might be some
5235 // weird collector this property does not hold for.
5236 if (isa<ConstantPointerNull>(DerivedPtr)) {
5237 // Use null-pointer of gc_relocate's type to replace it.
5240 continue;
5241 }
5242
5243 // isKnownNonNull -> nonnull attribute
5244 if (!GCR.hasRetAttr(Attribute::NonNull) &&
5245 isKnownNonZero(DerivedPtr,
5246 getSimplifyQuery().getWithInstruction(&Call))) {
5247 GCR.addRetAttr(Attribute::NonNull);
5248 // We discovered new fact, re-check users.
5249 Worklist.pushUsersToWorkList(GCR);
5250 }
5251 }
5252
5253 // If we have two copies of the same pointer in the statepoint argument
5254 // list, canonicalize to one. This may let us common gc.relocates.
5255 if (GCR.getBasePtr() == GCR.getDerivedPtr() &&
5256 GCR.getBasePtrIndex() != GCR.getDerivedPtrIndex()) {
5257 auto *OpIntTy = GCR.getOperand(2)->getType();
5258 GCR.setOperand(2, ConstantInt::get(OpIntTy, GCR.getBasePtrIndex()));
5259 }
5260
5261 // TODO: bitcast(relocate(p)) -> relocate(bitcast(p))
5262 // Canonicalize on the type from the uses to the defs
5263
5264 // TODO: relocate((gep p, C, C2, ...)) -> gep(relocate(p), C, C2, ...)
5265 LiveGcValues.insert(BasePtr);
5266 LiveGcValues.insert(DerivedPtr);
5267 }
5268 std::optional<OperandBundleUse> Bundle =
5270 unsigned NumOfGCLives = LiveGcValues.size();
5271 if (!Bundle || NumOfGCLives == Bundle->Inputs.size())
5272 break;
5273 // We can reduce the size of gc live bundle.
5274 DenseMap<Value *, unsigned> Val2Idx;
5275 std::vector<Value *> NewLiveGc;
5276 for (Value *V : Bundle->Inputs) {
5277 auto [It, Inserted] = Val2Idx.try_emplace(V);
5278 if (!Inserted)
5279 continue;
5280 if (LiveGcValues.count(V)) {
5281 It->second = NewLiveGc.size();
5282 NewLiveGc.push_back(V);
5283 } else
5284 It->second = NumOfGCLives;
5285 }
5286 // Update all gc.relocates
5287 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5288 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5289 Value *BasePtr = GCR.getBasePtr();
5290 assert(Val2Idx.count(BasePtr) && Val2Idx[BasePtr] != NumOfGCLives &&
5291 "Missed live gc for base pointer");
5292 auto *OpIntTy1 = GCR.getOperand(1)->getType();
5293 GCR.setOperand(1, ConstantInt::get(OpIntTy1, Val2Idx[BasePtr]));
5294 Value *DerivedPtr = GCR.getDerivedPtr();
5295 assert(Val2Idx.count(DerivedPtr) && Val2Idx[DerivedPtr] != NumOfGCLives &&
5296 "Missed live gc for derived pointer");
5297 auto *OpIntTy2 = GCR.getOperand(2)->getType();
5298 GCR.setOperand(2, ConstantInt::get(OpIntTy2, Val2Idx[DerivedPtr]));
5299 }
5300 // Create new statepoint instruction.
5301 OperandBundleDef NewBundle("gc-live", std::move(NewLiveGc));
5302 return CallBase::Create(&Call, NewBundle);
5303 }
5304 default: { break; }
5305 }
5306
5307 return Changed ? &Call : nullptr;
5308}
5309
5310/// If the callee is a constexpr cast of a function, attempt to move the cast to
5311/// the arguments of the call/invoke.
5312/// CallBrInst is not supported.
5313bool InstCombinerImpl::transformConstExprCastCall(CallBase &Call) {
5314 auto *Callee =
5316 if (!Callee)
5317 return false;
5318
5320 "CallBr's don't have a single point after a def to insert at");
5321
5322 // Don't perform the transform for declarations, which may not be fully
5323 // accurate. For example, void @foo() is commonly used as a placeholder for
5324 // unknown prototypes.
5325 if (Callee->isDeclaration())
5326 return false;
5327
5328 // If this is a call to a thunk function, don't remove the cast. Thunks are
5329 // used to transparently forward all incoming parameters and outgoing return
5330 // values, so it's important to leave the cast in place.
5331 if (Callee->hasFnAttribute("thunk"))
5332 return false;
5333
5334 // If this is a call to a naked function, the assembly might be
5335 // using an argument, or otherwise rely on the frame layout,
5336 // the function prototype will mismatch.
5337 if (Callee->hasFnAttribute(Attribute::Naked))
5338 return false;
5339
5340 // If this is a musttail call, the callee's prototype must match the caller's
5341 // prototype with the exception of pointee types. The code below doesn't
5342 // implement that, so we can't do this transform.
5343 // TODO: Do the transform if it only requires adding pointer casts.
5344 if (Call.isMustTailCall())
5345 return false;
5346
5348 const AttributeList &CallerPAL = Call.getAttributes();
5349
5350 // Okay, this is a cast from a function to a different type. Unless doing so
5351 // would cause a type conversion of one of our arguments, change this call to
5352 // be a direct call with arguments casted to the appropriate types.
5353 FunctionType *FT = Callee->getFunctionType();
5354 Type *OldRetTy = Caller->getType();
5355 Type *NewRetTy = FT->getReturnType();
5356
5357 // Check to see if we are changing the return type...
5358 if (OldRetTy != NewRetTy) {
5359
5360 if (NewRetTy->isStructTy())
5361 return false; // TODO: Handle multiple return values.
5362
5363 if (!CastInst::isBitOrNoopPointerCastable(NewRetTy, OldRetTy, DL)) {
5364 if (!Caller->use_empty())
5365 return false; // Cannot transform this return value.
5366 }
5367
5368 if (!CallerPAL.isEmpty() && !Caller->use_empty()) {
5369 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5370 if (RAttrs.overlaps(AttributeFuncs::typeIncompatible(
5371 NewRetTy, CallerPAL.getRetAttrs())))
5372 return false; // Attribute not compatible with transformed value.
5373 }
5374
5375 // If the callbase is an invoke instruction, and the return value is
5376 // used by a PHI node in a successor, we cannot change the return type of
5377 // the call because there is no place to put the cast instruction (without
5378 // breaking the critical edge). Bail out in this case.
5379 if (!Caller->use_empty()) {
5380 BasicBlock *PhisNotSupportedBlock = nullptr;
5381 if (auto *II = dyn_cast<InvokeInst>(Caller))
5382 PhisNotSupportedBlock = II->getNormalDest();
5383 if (PhisNotSupportedBlock)
5384 for (User *U : Caller->users())
5385 if (PHINode *PN = dyn_cast<PHINode>(U))
5386 if (PN->getParent() == PhisNotSupportedBlock)
5387 return false;
5388 }
5389 }
5390
5391 unsigned NumActualArgs = Call.arg_size();
5392 unsigned NumCommonArgs = std::min(FT->getNumParams(), NumActualArgs);
5393
5394 // Prevent us turning:
5395 // declare void @takes_i32_inalloca(i32* inalloca)
5396 // call void bitcast (void (i32*)* @takes_i32_inalloca to void (i32)*)(i32 0)
5397 //
5398 // into:
5399 // call void @takes_i32_inalloca(i32* null)
5400 //
5401 // Similarly, avoid folding away bitcasts of byval calls.
5402 if (Callee->getAttributes().hasAttrSomewhere(Attribute::InAlloca) ||
5403 Callee->getAttributes().hasAttrSomewhere(Attribute::Preallocated))
5404 return false;
5405
5406 auto AI = Call.arg_begin();
5407 for (unsigned i = 0, e = NumCommonArgs; i != e; ++i, ++AI) {
5408 Type *ParamTy = FT->getParamType(i);
5409 Type *ActTy = (*AI)->getType();
5410
5411 if (!CastInst::isBitOrNoopPointerCastable(ActTy, ParamTy, DL))
5412 return false; // Cannot transform this parameter value.
5413
5414 // Check if there are any incompatible attributes we cannot drop safely.
5415 if (AttrBuilder(FT->getContext(), CallerPAL.getParamAttrs(i))
5416 .overlaps(AttributeFuncs::typeIncompatible(
5417 ParamTy, CallerPAL.getParamAttrs(i),
5418 AttributeFuncs::ASK_UNSAFE_TO_DROP)))
5419 return false; // Attribute not compatible with transformed value.
5420
5421 if (Call.isInAllocaArgument(i) ||
5422 CallerPAL.hasParamAttr(i, Attribute::Preallocated))
5423 return false; // Cannot transform to and from inalloca/preallocated.
5424
5425 if (CallerPAL.hasParamAttr(i, Attribute::SwiftError))
5426 return false;
5427
5428 if (CallerPAL.hasParamAttr(i, Attribute::ByVal) !=
5429 Callee->getAttributes().hasParamAttr(i, Attribute::ByVal))
5430 return false; // Cannot transform to or from byval.
5431 }
5432
5433 if (FT->getNumParams() < NumActualArgs && FT->isVarArg() &&
5434 !CallerPAL.isEmpty()) {
5435 // In this case we have more arguments than the new function type, but we
5436 // won't be dropping them. Check that these extra arguments have attributes
5437 // that are compatible with being a vararg call argument.
5438 unsigned SRetIdx;
5439 if (CallerPAL.hasAttrSomewhere(Attribute::StructRet, &SRetIdx) &&
5440 SRetIdx - AttributeList::FirstArgIndex >= FT->getNumParams())
5441 return false;
5442 }
5443
5444 // Okay, we decided that this is a safe thing to do: go ahead and start
5445 // inserting cast instructions as necessary.
5446 SmallVector<Value *, 8> Args;
5448 Args.reserve(NumActualArgs);
5449 ArgAttrs.reserve(NumActualArgs);
5450
5451 // Get any return attributes.
5452 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5453
5454 // If the return value is not being used, the type may not be compatible
5455 // with the existing attributes. Wipe out any problematic attributes.
5456 RAttrs.remove(
5457 AttributeFuncs::typeIncompatible(NewRetTy, CallerPAL.getRetAttrs()));
5458
5459 LLVMContext &Ctx = Call.getContext();
5460 AI = Call.arg_begin();
5461 for (unsigned i = 0; i != NumCommonArgs; ++i, ++AI) {
5462 Type *ParamTy = FT->getParamType(i);
5463
5464 Value *NewArg = *AI;
5465 if ((*AI)->getType() != ParamTy)
5466 NewArg = Builder.CreateBitOrPointerCast(*AI, ParamTy);
5467 Args.push_back(NewArg);
5468
5469 // Add any parameter attributes except the ones incompatible with the new
5470 // type. Note that we made sure all incompatible ones are safe to drop.
5471 AttributeMask IncompatibleAttrs = AttributeFuncs::typeIncompatible(
5472 ParamTy, CallerPAL.getParamAttrs(i), AttributeFuncs::ASK_SAFE_TO_DROP);
5473 ArgAttrs.push_back(
5474 CallerPAL.getParamAttrs(i).removeAttributes(Ctx, IncompatibleAttrs));
5475 }
5476
5477 // If the function takes more arguments than the call was taking, add them
5478 // now.
5479 for (unsigned i = NumCommonArgs; i != FT->getNumParams(); ++i) {
5480 Args.push_back(Constant::getNullValue(FT->getParamType(i)));
5481 ArgAttrs.push_back(AttributeSet());
5482 }
5483
5484 // If we are removing arguments to the function, emit an obnoxious warning.
5485 if (FT->getNumParams() < NumActualArgs) {
5486 // TODO: if (!FT->isVarArg()) this call may be unreachable. PR14722
5487 if (FT->isVarArg()) {
5488 // Add all of the arguments in their promoted form to the arg list.
5489 for (unsigned i = FT->getNumParams(); i != NumActualArgs; ++i, ++AI) {
5490 Type *PTy = getPromotedType((*AI)->getType());
5491 Value *NewArg = *AI;
5492 if (PTy != (*AI)->getType()) {
5493 // Must promote to pass through va_arg area!
5494 Instruction::CastOps opcode =
5495 CastInst::getCastOpcode(*AI, false, PTy, false);
5496 NewArg = Builder.CreateCast(opcode, *AI, PTy);
5497 }
5498 Args.push_back(NewArg);
5499
5500 // Add any parameter attributes.
5501 ArgAttrs.push_back(CallerPAL.getParamAttrs(i));
5502 }
5503 }
5504 }
5505
5506 AttributeSet FnAttrs = CallerPAL.getFnAttrs();
5507
5508 if (NewRetTy->isVoidTy())
5509 Caller->setName(""); // Void type should not have a name.
5510
5511 assert((ArgAttrs.size() == FT->getNumParams() || FT->isVarArg()) &&
5512 "missing argument attributes");
5513 AttributeList NewCallerPAL = AttributeList::get(
5514 Ctx, FnAttrs, AttributeSet::get(Ctx, RAttrs), ArgAttrs);
5515
5517 Call.getOperandBundlesAsDefs(OpBundles);
5518
5519 CallBase *NewCall;
5520 if (InvokeInst *II = dyn_cast<InvokeInst>(Caller)) {
5521 NewCall = Builder.CreateInvoke(Callee, II->getNormalDest(),
5522 II->getUnwindDest(), Args, OpBundles);
5523 } else {
5524 NewCall = Builder.CreateCall(Callee, Args, OpBundles);
5525 cast<CallInst>(NewCall)->setTailCallKind(
5526 cast<CallInst>(Caller)->getTailCallKind());
5527 }
5528 NewCall->takeName(Caller);
5530 NewCall->setAttributes(NewCallerPAL);
5531
5532 // Preserve prof metadata if any.
5533 NewCall->copyMetadata(*Caller, {LLVMContext::MD_prof});
5534
5535 // Insert a cast of the return type as necessary.
5536 Instruction *NC = NewCall;
5537 Value *NV = NC;
5538 if (OldRetTy != NV->getType() && !Caller->use_empty()) {
5539 assert(!NV->getType()->isVoidTy());
5541 NC->setDebugLoc(Caller->getDebugLoc());
5542
5543 auto OptInsertPt = NewCall->getInsertionPointAfterDef();
5544 assert(OptInsertPt && "No place to insert cast");
5545 InsertNewInstBefore(NC, *OptInsertPt);
5546 Worklist.pushUsersToWorkList(*Caller);
5547 }
5548
5549 if (!Caller->use_empty())
5550 replaceInstUsesWith(*Caller, NV);
5551 else if (Caller->hasValueHandle()) {
5552 if (OldRetTy == NV->getType())
5554 else
5555 // We cannot call ValueIsRAUWd with a different type, and the
5556 // actual tracked value will disappear.
5558 }
5559
5560 eraseInstFromFunction(*Caller);
5561 return true;
5562}
5563
5564/// Turn a call to a function created by init_trampoline / adjust_trampoline
5565/// intrinsic pair into a direct call to the underlying function.
5567InstCombinerImpl::transformCallThroughTrampoline(CallBase &Call,
5568 IntrinsicInst &Tramp) {
5569 FunctionType *FTy = Call.getFunctionType();
5570 AttributeList Attrs = Call.getAttributes();
5571
5572 // If the call already has the 'nest' attribute somewhere then give up -
5573 // otherwise 'nest' would occur twice after splicing in the chain.
5574 if (Attrs.hasAttrSomewhere(Attribute::Nest))
5575 return nullptr;
5576
5578 FunctionType *NestFTy = NestF->getFunctionType();
5579
5580 AttributeList NestAttrs = NestF->getAttributes();
5581 if (!NestAttrs.isEmpty()) {
5582 unsigned NestArgNo = 0;
5583 Type *NestTy = nullptr;
5584 AttributeSet NestAttr;
5585
5586 // Look for a parameter marked with the 'nest' attribute.
5587 for (FunctionType::param_iterator I = NestFTy->param_begin(),
5588 E = NestFTy->param_end();
5589 I != E; ++NestArgNo, ++I) {
5590 AttributeSet AS = NestAttrs.getParamAttrs(NestArgNo);
5591 if (AS.hasAttribute(Attribute::Nest)) {
5592 // Record the parameter type and any other attributes.
5593 NestTy = *I;
5594 NestAttr = AS;
5595 break;
5596 }
5597 }
5598
5599 if (NestTy) {
5600 std::vector<Value*> NewArgs;
5601 std::vector<AttributeSet> NewArgAttrs;
5602 NewArgs.reserve(Call.arg_size() + 1);
5603 NewArgAttrs.reserve(Call.arg_size());
5604
5605 // Insert the nest argument into the call argument list, which may
5606 // mean appending it. Likewise for attributes.
5607
5608 {
5609 unsigned ArgNo = 0;
5610 auto I = Call.arg_begin(), E = Call.arg_end();
5611 do {
5612 if (ArgNo == NestArgNo) {
5613 // Add the chain argument and attributes.
5614 Value *NestVal = Tramp.getArgOperand(2);
5615 if (NestVal->getType() != NestTy)
5616 NestVal = Builder.CreateBitCast(NestVal, NestTy, "nest");
5617 NewArgs.push_back(NestVal);
5618 NewArgAttrs.push_back(NestAttr);
5619 }
5620
5621 if (I == E)
5622 break;
5623
5624 // Add the original argument and attributes.
5625 NewArgs.push_back(*I);
5626 NewArgAttrs.push_back(Attrs.getParamAttrs(ArgNo));
5627
5628 ++ArgNo;
5629 ++I;
5630 } while (true);
5631 }
5632
5633 // The trampoline may have been bitcast to a bogus type (FTy).
5634 // Handle this by synthesizing a new function type, equal to FTy
5635 // with the chain parameter inserted.
5636
5637 std::vector<Type*> NewTypes;
5638 NewTypes.reserve(FTy->getNumParams()+1);
5639
5640 // Insert the chain's type into the list of parameter types, which may
5641 // mean appending it.
5642 {
5643 unsigned ArgNo = 0;
5644 FunctionType::param_iterator I = FTy->param_begin(),
5645 E = FTy->param_end();
5646
5647 do {
5648 if (ArgNo == NestArgNo)
5649 // Add the chain's type.
5650 NewTypes.push_back(NestTy);
5651
5652 if (I == E)
5653 break;
5654
5655 // Add the original type.
5656 NewTypes.push_back(*I);
5657
5658 ++ArgNo;
5659 ++I;
5660 } while (true);
5661 }
5662
5663 // Replace the trampoline call with a direct call. Let the generic
5664 // code sort out any function type mismatches.
5665 FunctionType *NewFTy =
5666 FunctionType::get(FTy->getReturnType(), NewTypes, FTy->isVarArg());
5667 AttributeList NewPAL =
5668 AttributeList::get(FTy->getContext(), Attrs.getFnAttrs(),
5669 Attrs.getRetAttrs(), NewArgAttrs);
5670
5672 Call.getOperandBundlesAsDefs(OpBundles);
5673
5674 Instruction *NewCaller;
5675 if (InvokeInst *II = dyn_cast<InvokeInst>(&Call)) {
5676 NewCaller = InvokeInst::Create(NewFTy, NestF, II->getNormalDest(),
5677 II->getUnwindDest(), NewArgs, OpBundles);
5678 cast<InvokeInst>(NewCaller)->setCallingConv(II->getCallingConv());
5679 cast<InvokeInst>(NewCaller)->setAttributes(NewPAL);
5680 } else if (CallBrInst *CBI = dyn_cast<CallBrInst>(&Call)) {
5681 NewCaller =
5682 CallBrInst::Create(NewFTy, NestF, CBI->getDefaultDest(),
5683 CBI->getIndirectDests(), NewArgs, OpBundles);
5684 cast<CallBrInst>(NewCaller)->setCallingConv(CBI->getCallingConv());
5685 cast<CallBrInst>(NewCaller)->setAttributes(NewPAL);
5686 } else {
5687 NewCaller = CallInst::Create(NewFTy, NestF, NewArgs, OpBundles);
5688 cast<CallInst>(NewCaller)->setTailCallKind(
5689 cast<CallInst>(Call).getTailCallKind());
5690 cast<CallInst>(NewCaller)->setCallingConv(
5691 cast<CallInst>(Call).getCallingConv());
5692 cast<CallInst>(NewCaller)->setAttributes(NewPAL);
5693 }
5694 NewCaller->setDebugLoc(Call.getDebugLoc());
5695
5696 return NewCaller;
5697 }
5698 }
5699
5700 // Replace the trampoline call with a direct call. Since there is no 'nest'
5701 // parameter, there is no need to adjust the argument list. Let the generic
5702 // code sort out any function type mismatches.
5703 Call.setCalledFunction(FTy, NestF);
5704 return &Call;
5705}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
@ Scaled
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
BitTracker BT
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static SDValue foldBitOrderCrossLogicOp(SDNode *N, SelectionDAG &DAG)
#define Check(C,...)
#define DEBUG_TYPE
Hexagon Common GEP
#define _
IRTranslator LLVM IR MI
static Type * getPromotedType(Type *Ty)
Return the specified type promoted as it would be to pass though a va_arg area.
static Instruction * createOverflowTuple(IntrinsicInst *II, Value *Result, Constant *Overflow)
Creates a result tuple for an overflow intrinsic II with a given Result and a constant Overflow value...
static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M, IRBuilderBase &B)
static IntrinsicInst * findInitTrampolineFromAlloca(Value *TrampMem)
static bool removeTriviallyEmptyRange(IntrinsicInst &EndI, InstCombinerImpl &IC, std::function< bool(const IntrinsicInst &)> IsStart)
static bool inputDenormalIsDAZ(const Function &F, const Type *Ty)
static Instruction * reassociateMinMaxWithConstantInOperand(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If this min/max has a matching min/max operand with a constant, try to push the constant operand into...
static bool isIdempotentBinaryIntrinsic(Intrinsic::ID IID)
Helper to match idempotent binary intrinsics, namely, intrinsics where f(f(x, y), y) == f(x,...
static bool signBitMustBeTheSame(Value *Op0, Value *Op1, const SimplifyQuery &SQ)
Return true if two values Op0 and Op1 are known to have the same sign.
static Value * optimizeModularFormat(CallInst *CI, IRBuilderBase &B)
static Instruction * moveAddAfterMinMax(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0.
static Instruction * simplifyInvariantGroupIntrinsic(IntrinsicInst &II, InstCombinerImpl &IC)
This function transforms launder.invariant.group like: launder(launder(x)) -> launder(x) (the result ...
static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E, unsigned NumOperands)
static std::optional< bool > getKnownSign(Value *Op, const SimplifyQuery &SQ)
static cl::opt< unsigned > GuardWideningWindow("instcombine-guard-widening-window", cl::init(3), cl::desc("How wide an instruction window to bypass looking for " "another guard"))
static bool hasUndefSource(AnyMemTransferInst *MI)
Recognize a memcpy/memmove from a trivially otherwise unused alloca.
static Instruction * factorizeMinMaxTree(IntrinsicInst *II)
Reduce a sequence of min/max intrinsics with a common operand.
static Instruction * foldClampRangeOfTwo(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If we have a clamp pattern like max (min X, 42), 41 – where the output can only be one of two possibl...
static Value * simplifyReductionOperand(Value *Arg, bool CanReorderLanes)
static IntrinsicInst * findInitTrampolineFromBB(IntrinsicInst *AdjustTramp, Value *TrampMem)
static bool isAspectNeeded(StringRef Aspect, CallInst *CI, std::optional< unsigned > FirstArgIdx, const std::optional< Bitset< 256 > > &Specifiers)
static Value * foldIntrinsicUsingDistributiveLaws(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
static std::optional< bool > getKnownSignOrZero(Value *Op, const SimplifyQuery &SQ)
static Value * foldMinimumOverTrailingOrLeadingZeroCount(Value *I0, Value *I1, const DataLayout &DL, InstCombiner::BuilderTy &Builder)
Fold an unsigned minimum of trailing or leading zero bits counts: umin(cttz(CtOp1,...
static bool rightDistributesOverLeft(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "(X ROp Y) LOp Z" is always equal to "(X LOp Z) ROp (Y LOp Z)".
static Value * foldIdempotentBinaryIntrinsicRecurrence(InstCombinerImpl &IC, IntrinsicInst *II)
Attempt to simplify value-accumulating recurrences of kind: umax.acc = phi i8 [ umax,...
static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy)
static Instruction * foldCtpop(IntrinsicInst &II, InstCombinerImpl &IC)
static Instruction * simplifyNeonTbl(IntrinsicInst &II, InstCombiner &IC, bool IsExtension)
Convert tbl/tbx intrinsics to shufflevector if the mask is constant, and at most two source operands ...
static Instruction * foldCttzCtlz(IntrinsicInst &II, InstCombinerImpl &IC)
static IntrinsicInst * findInitTrampoline(Value *Callee)
static Value * foldCmpIntrinsicOfExtended(IntrinsicInst *II, InstCombiner::BuilderTy &Builder, const DataLayout &DL)
Fold an scmp/ucmp intrinsic whose operands are extended from a narrower type: scmp (sext X),...
static Bitset< 256 > parseFormatStringSpecifiers(StringRef FormatStr)
static FCmpInst::Predicate fpclassTestIsFCmp0(FPClassTest Mask, const Function &F, Type *Ty)
static bool leftDistributesOverRight(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "X LOp (Y ROp Z)" is always equal to "(X LOp Y) ROp (X LOp Z)".
static Value * reassociateMinMaxWithConstants(IntrinsicInst *II, IRBuilderBase &Builder, const SimplifyQuery &SQ)
If this min/max has a constant operand and an operand that is a matching min/max with a constant oper...
static Value * foldSinAndCosToSinCos(IntrinsicInst *II, IRBuilderBase &B, InstCombinerImpl &IC)
static CallInst * canonicalizeConstantArg0ToArg1(CallInst &Call)
static Instruction * foldNeonShift(IntrinsicInst *II, InstCombinerImpl &IC)
This file provides internal interfaces used to implement the InstCombine.
This file provides the interface for the instcombine pass implementation.
static bool inputDenormalIsIEEE(DenormalMode Mode)
Return true if it's possible to assume IEEE treatment of input denormals in F for Val.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
const SmallVectorImpl< MachineOperand > & Cond
This file implements the SmallBitVector class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
static LLVM_ABI bool semanticsHasInf(const fltSemantics &)
Definition APFloat.cpp:362
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static LLVM_ABI bool hasSignBitInMSB(const fltSemantics &)
Definition APFloat.cpp:375
bool isNegative() const
Definition APFloat.h:1583
void clearSign()
Definition APFloat.h:1402
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
bool isZero() const
Definition APFloat.h:1579
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
Definition APFloat.h:1242
static APFloat getSmallest(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) finite number in the given semantics.
Definition APFloat.h:1252
bool isInfinity() const
Definition APFloat.h:1580
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:225
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1205
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1986
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1186
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1695
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1115
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1966
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1973
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:215
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1086
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:506
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2074
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:330
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
std::optional< int64_t > trySExtValue() const
Get sign extended value if possible.
Definition APInt.h:1594
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1979
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
This class represents any memset intrinsic.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
static LLVM_ABI AttributeSet get(LLVMContext &C, const AttrBuilder &B)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
static LLVM_ABI Attribute getWithDereferenceableOrNullBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:446
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI bool isSigned() const
Whether the intrinsic is signed or unsigned.
LLVM_ABI Instruction::BinaryOps getBinaryOp() const
Returns the binary operation underlying the intrinsic.
static BinaryOperator * CreateFAddFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:271
static LLVM_ABI BinaryOperator * CreateNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Helper functions to construct and inspect unary operations (NEG and NOT) via binary operators SUB and...
static BinaryOperator * CreateNSW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:314
static LLVM_ABI BinaryOperator * CreateNot(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static BinaryOperator * CreateNUW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:329
static BinaryOperator * CreateFMulFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:279
static BinaryOperator * CreateFDivFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:283
static BinaryOperator * CreateFSubFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:275
static LLVM_ABI BinaryOperator * CreateNSWNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
This is a constexpr reimplementation of a subset of std::bitset.
Definition Bitset.h:30
constexpr bool any() const
Definition Bitset.h:113
constexpr Bitset & set()
Definition Bitset.h:81
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
void setDoesNotThrow()
MaybeAlign getRetAlign() const
Extract the alignment of the return value.
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
OperandBundleUse getOperandBundleAt(unsigned Index) const
Return the operand bundle at a specific index.
std::optional< OperandBundleUse > getOperandBundle(StringRef Name) const
Return an operand bundle by name, if present.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
bool isInAllocaArgument(unsigned ArgNo) const
Determine whether this argument is passed in an alloca.
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
bool hasRetAttr(Attribute::AttrKind Kind) const
Determine whether the return value has the given attribute.
unsigned getNumOperandBundles() const
Return the number of operand bundles associated with this User.
uint64_t getParamDereferenceableBytes(unsigned i) const
Extract the number of dereferenceable bytes for a call or parameter (0=unknown).
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
LLVM_ABI bool isIndirectCall() const
Return true if the callsite is an indirect call.
static LLVM_ABI CallBase * removeOperandBundleAt(CallBase *CB, size_t Offset, InsertPosition InsertPtr=nullptr)
void setNotConvergent()
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Attribute getFnAttr(StringRef Kind) const
Get the attribute of a given kind for the function.
bool doesNotThrow() const
Determine if the call cannot unwind.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
bool isConvergent() const
Determine if the invoke is convergent.
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
Value * getReturnedArgOperand() const
If one of the arguments has the 'returned' attribute, returns its operand value.
static LLVM_ABI CallBase * Create(CallBase *CB, ArrayRef< OperandBundleDef > Bundles, InsertPosition InsertPt=nullptr)
Create a clone of CB with a different set of operand bundles and insert it before InsertPt.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
static LLVM_ABI CallBase * removeOperandBundle(CallBase *CB, uint32_t ID, InsertPosition InsertPt=nullptr)
Create a clone of CB with operand bundle ID removed.
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
LLVM_ABI Function * getCaller()
Helper to get the caller (the parent function).
CallBr instruction, tracking function calls that may not return control but instead transfer it to a ...
static CallBrInst * Create(FunctionType *Ty, Value *Func, BasicBlock *DefaultDest, ArrayRef< BasicBlock * > IndirectDests, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This class represents a function call, abstracting a target machine's calling convention.
bool isNoTailCall() const
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
bool isMustTailCall() const
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI CastInst * CreateIntegerCast(Value *S, Type *Ty, bool isSigned, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a ZExt, BitCast, or Trunc for int -> int casts.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static LLVM_ABI CastInst * CreateBitOrPointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, a PtrToInt, or an IntToPTr cast instruction.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getNonStrictPredicate() const
For example, SGT -> SGE, SLT -> SLE, ULT -> ULE, UGT -> UGE.
Definition InstrTypes.h:934
Predicate getUnorderedPredicate() const
Definition InstrTypes.h:874
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getLimitedValue(uint64_t Limit=~0ULL) const
getLimitedValue - If the value is smaller than the specified limit, return it, otherwise return the l...
Definition Constants.h:269
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI ConstantPtrAuth * get(Constant *Ptr, ConstantInt *Key, ConstantInt *Disc, Constant *AddrDisc, Constant *DeactivationSymbol)
Return a pointer signed with the specified parameters.
This class represents a range of values.
LLVM_ABI ConstantRange zextOrTrunc(uint32_t BitWidth) const
Make this range have the bit width given by BitWidth.
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
uint32_t getBitWidth() const
Get the bit width of this ConstantRange.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Record of a variable value-assignment, aka a non instruction representation of the dbg....
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:348
unsigned size() const
Definition DenseMap.h:207
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
Definition DenseMap.h:254
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Definition DenseMap.h:249
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static FMFSource intersect(Value *A, Value *B)
Intersect the FMF from two instructions.
Definition IRBuilder.h:107
This class represents an extension of floating point types.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
An instruction for ordering other memory operations.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this fence instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this fence instruction.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Type::subtype_iterator param_iterator
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
bool isConvergent() const
Determine if the call is convergent.
Definition Function.h:593
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:329
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:577
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
Definition Function.h:252
LLVM_ABI Value * getBasePtr() const
unsigned getBasePtrIndex() const
The index into the associate statepoint's argument list which contains the base pointer of the pointe...
LLVM_ABI Value * getDerivedPtr() const
unsigned getDerivedPtrIndex() const
The index into the associate statepoint's argument list which contains the pointer whose relocation t...
std::vector< const GCRelocateInst * > getGCRelocates() const
Get list of all gc reloactes linked to this statepoint May contain several relocations for the same b...
Definition Statepoint.h:206
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this GlobalObject.
LLVM_ABI bool isDeclaration() const
Return true if the primary definition of this global value is outside of the current translation unit...
Definition Globals.cpp:408
PointerType * getType() const
Global values are always pointers.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNull=false)
Definition IRBuilder.h:2256
LLVM_ABI Value * CreateLaunderInvariantGroup(Value *Ptr)
Create a launder.invariant.group intrinsic call.
ConstantInt * getTrue()
Get the constant value for i1 true.
Definition IRBuilder.h:457
LLVM_ABI Value * CreateBinaryIntrinsic(Intrinsic::ID ID, Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 2 operands which is mangled on the first type.
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1447
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2129
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Definition IRBuilder.h:2699
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:462
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2500
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
static InsertValueInst * Create(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Instruction * foldOpIntoPhi(Instruction &I, PHINode *PN, bool AllowMultipleUses=false)
Given a binary operator, cast instruction, or select which has a PHI node as operand #0,...
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
bool SimplifyDemandedBits(Instruction *I, unsigned Op, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0) override
This form of SimplifyDemandedBits simplifies the specified instruction operand if possible,...
Instruction * FoldOpIntoSelect(Instruction &Op, SelectInst *SI, bool FoldWithMultiUse=false, bool SimplifyBothArms=false)
Given an instruction with a select as one operand and a constant as the other operand,...
Instruction * SimplifyAnyMemSet(AnyMemSetInst *MI)
Instruction * foldItoFPtoI(FPToIntTy &FI)
fpto{s/u}i.sat --> X or zext(X) or sext(X) or trunc(X) This is safe if the intermediate type has enou...
Instruction * visitFree(CallInst &FI, Value *FreedOp)
Instruction * visitCallBrInst(CallBrInst &CBI)
Instruction * eraseInstFromFunction(Instruction &I) override
Combiner aware instruction erasure.
Value * foldReversedIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are reverses, try to pull the reverse after the intrinsic.
Value * tryGetLog2(Value *Op, bool AssumeNonZero)
Instruction * visitFenceInst(FenceInst &FI)
Instruction * foldShuffledIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are unary shuffles with the same mask, try to shuffle after the int...
Instruction * visitInvokeInst(InvokeInst &II)
bool SimplifyDemandedInstructionBits(Instruction &Inst)
Tries to simplify operands to an integer instruction based on its demanded bits.
void CreateNonTerminatorUnreachable(Instruction *InsertAt)
Create and insert the idiom we use to indicate a block is unreachable without having to rewrite the C...
Instruction * visitVAEndInst(VAEndInst &I)
Instruction * matchBSwapOrBitReverse(Instruction &I, bool MatchBSwaps, bool MatchBitReversals)
Given an initial instruction, check to see if it is the root of a bswap/bitreverse idiom.
Constant * unshuffleConstant(ArrayRef< int > ShMask, Constant *C, VectorType *NewCTy)
Find a constant NewC that has property: shuffle(NewC, poison, ShMask) = C for lanes that select NewC.
Instruction * visitAllocSite(Instruction &FI)
Instruction * SimplifyAnyMemTransfer(AnyMemTransferInst *MI)
OverflowResult computeOverflow(Instruction::BinaryOps BinaryOp, bool IsSigned, Value *LHS, Value *RHS, Instruction *CxtI) const
Instruction * visitCallInst(CallInst &CI)
CallInst simplification.
The core instruction combiner logic.
SimplifyQuery SQ
const DataLayout & getDataLayout() const
unsigned ComputeMaxSignificantBits(const Value *Op, const Instruction *CxtI=nullptr, unsigned Depth=0) const
bool isFreeToInvert(Value *V, bool WillInvertAllUses, bool &DoesConsume)
Return true if the specified value is free to invert (apply ~ to).
DominatorTree & getDominatorTree() const
BlockFrequencyInfo * BFI
TargetLibraryInfo & TLI
Instruction * InsertNewInstBefore(Instruction *New, BasicBlock::iterator Old)
Inserts an instruction New before instruction Old.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
const DataLayout & DL
DomConditionCache DC
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CxtI, unsigned Depth=0) const
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
LLVM_ABI std::optional< Instruction * > targetInstCombineIntrinsic(IntrinsicInst &II)
AssumptionCache & AC
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
bool MaskedValueIsZero(const Value *V, const APInt &Mask, const Instruction *CxtI=nullptr, unsigned Depth=0) const
DominatorTree & DT
ProfileSummaryInfo * PSI
OptimizationRemarkEmitter & ORE
Value * getFreelyInverted(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume)
const SimplifyQuery & getSimplifyQuery() const
bool isKnownToBeAPowerOfTwo(const Value *V, bool OrZero=false, const Instruction *CxtI=nullptr, unsigned Depth=0)
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI bool mayWriteToMemory() const LLVM_READONLY
Return true if this instruction may modify memory.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
bool isTerminator() const
iterator_range< user_iterator > users()
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI std::optional< InstListType::iterator > getInsertionPointAfterDef()
Get the first insertion point at which the result of this instruction is defined.
LLVM_ABI bool isIdenticalTo(const Instruction *I) const LLVM_READONLY
Return true if the specified instruction is exactly identical to the current one.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:338
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
Invoke instruction.
static InvokeInst * Create(FunctionType *Ty, Value *Func, BasicBlock *IfNormal, BasicBlock *IfException, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
Metadata node.
Definition Metadata.h:1081
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
static LLVM_ABI MDNode * getMostGenericFPMath(MDNode *A, MDNode *B)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:597
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:107
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
ICmpInst::Predicate getPredicate() const
Returns the comparison predicate underlying the intrinsic.
bool isSigned() const
Whether the intrinsic is signed or unsigned.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
StringRef getName() const
Get a short "name" for the module.
Definition Module.h:316
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
Utility class for integer operators which may exhibit overflow - Add, Sub, Mul, and Shl.
Definition Operator.h:78
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
bool isCommutative() const
Return true if the instruction is commutative.
Definition Operator.h:130
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Represents a saturating add/sub intrinsic.
This class represents the LLVM 'select' instruction.
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
This instruction constructs a fixed permutation of two input vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
size_type size() const
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setVolatile(bool V)
Specify whether this is a volatile store or not.
void setAlignment(Align Align)
void setOrdering(AtomicOrdering Ordering)
Sets the ordering constraint of this store instruction.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static constexpr size_t npos
Definition StringRef.h:58
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
LLVM_ABI size_t find_first_not_of(char C, size_t From=0) const
Find the first character in the string that is not C or npos if not found.
Class to represent struct types.
static LLVM_ABI bool isCallingConvCCompatible(CallBase *CI)
Returns true if call site / callee has cdecl-compatible calling conventions.
Provides information about what library functions are available for the current target.
This class represents a truncation of integer types.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:258
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
LLVM_ABI bool canLosslesslyBitCastTo(Type *Ty) const
Return true if this type could be converted with a lossless BitCast to type 'Ty'.
Definition Type.cpp:143
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:271
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:96
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
static UnaryOperator * CreateWithCopiedFlags(UnaryOps Opc, Value *V, Instruction *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:148
static UnaryOperator * CreateFNegFMF(Value *Op, Instruction *FMFSource, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:156
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:35
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
This represents the llvm.va_end intrinsic.
static LLVM_ABI void ValueIsDeleted(Value *V)
Definition Value.cpp:1272
static LLVM_ABI void ValueIsRAUWd(Value *Old, Value *New)
Definition Value.cpp:1325
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
static constexpr uint64_t MaximumAlignment
Definition Value.h:801
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
iterator_range< user_iterator > users()
Definition Value.h:428
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:712
bool use_empty() const
Definition Value.h:348
static constexpr unsigned MaxAlignmentExponent
The maximum alignment for instructions.
Definition Value.h:800
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:81
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
auto m_PosZeroFP()
Matches a floating-point positive zero.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
auto m_PtrToIntOrAddr(const OpTy &Op)
Matches PtrToInt or PtrToAddr.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
BinaryOp_match< LHS, RHS, Instruction::Xor > m_Xor(const LHS &L, const RHS &R)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWSub(const LHS &L, const RHS &R)
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
auto m_UMin(const Opnd0 &Op0, const Opnd1 &Op1)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
OverflowingBinaryOp_match< cst_pred_ty< is_zero_int >, ValTy, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWNeg(const ValTy &V)
Matches a 'Neg' as 'sub nsw 0, V'.
auto m_SMax(const Opnd0 &Op0, const Opnd1 &Op1)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
cstfp_pred_ty< is_neg_zero_fp > m_NegZeroFP()
Match a floating-point negative zero.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_UMax(const Opnd0 &Op0, const Opnd1 &Op1)
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_CopySign(const Opnd0 &Op0, const Opnd1 &Op1)
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
BinOpPred_match< LHS, RHS, is_logical_shift_op > m_LogicalShift(const LHS &L, const RHS &R)
Matches logical shift operations.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
match_combine_or< match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > >, OpTy > m_ZExtOrSExtOrSelf(const OpTy &Op)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_strictlypositive > m_StrictlyPositive()
Match an integer or vector of strictly positive values.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
auto m_FShl(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
cst_pred_ty< is_negated_power2 > m_NegatedPower2()
Match a integer or vector negated power-of-2.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
cst_pred_ty< custom_checkfn< APInt > > m_CheckedInt(function_ref< bool(const APInt &)> CheckFn)
Match an integer or vector where CheckFn(ele) for each element is true.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_c_MaxOrMin(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWSub(const LHS &L, const RHS &R)
auto m_SMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_FAbs(const Opnd0 &Op0)
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoSignedWrap >, DisjointOr_match< LHS, RHS > > m_NSWAddLike(const LHS &L, const RHS &R)
Match either "add nsw" or "or disjoint".
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
Exact_match< T > m_Exact(const T &SubPattern)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
auto m_UnOp()
Match an arbitrary unary operation and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_MaxOrMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
BinaryOp_match< LHS, RHS, Instruction::SRem > m_SRem(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
auto m_VecReverse(const Opnd0 &Op0)
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoUnsignedWrap >, DisjointOr_match< LHS, RHS > > m_NUWAddLike(const LHS &L, const RHS &R)
Match either "add nuw" or "or disjoint".
BinOpPred_match< LHS, RHS, is_bitwiselogic_op > m_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations.
ElementWiseBitCast_match< OpTy > m_ElementWiseBitCast(const OpTy &Op)
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
auto m_FShr(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
Definition DebugInfo.h:212
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
constexpr double e
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Intrinsic::ID getInverseMinMaxIntrinsic(Intrinsic::ID MinMaxID)
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
@ NeverOverflows
Never overflows.
@ AlwaysOverflowsHigh
Always overflows in the direction of signed/unsigned max value.
@ AlwaysOverflowsLow
Always overflows in the direction of signed/unsigned min value.
@ MayOverflow
May or may not overflow.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
LLVM_ABI Value * simplifyFMulInst(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for an FMul, fold the result or return null.
LLVM_ABI bool isValidAssumeForContext(const Instruction *I, const Instruction *CxtI, const DominatorTree *DT=nullptr, bool AllowEphemerals=false)
Return true if it is valid to use the assumptions provided by an assume intrinsic,...
LLVM_ABI APInt possiblyDemandedEltsInMask(Value *Mask)
Given a mask vector of the form <Y x i1>, return an APInt (of bitwidth Y) for each lane which may be ...
BundleAttr getBundleAttrFromOBU(OperandBundleUse OBU)
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2570
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool isRemovableAlloc(const CallBase *V, const TargetLibraryInfo *TLI)
Return true if this is a call to an allocation function that does not have side effects that we are r...
LLVM_ABI bool getConstantStringInfo(const Value *V, StringRef &Str, bool TrimAtNul=true)
This function computes the length of a null-terminated C string pointed to by V.
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
Definition MathExtras.h:224
LLVM_ABI Value * lowerObjectSizeCall(IntrinsicInst *ObjectSize, const DataLayout &DL, const TargetLibraryInfo *TLI, bool MustSucceed)
Try to turn a call to @llvm.objectsize into an integer value of the given Type.
LLVM_ABI AssumeSeparateStorageInfo getAssumeSeparateStorageInfo(OperandBundleUse)
LLVM_ABI Value * getAllocAlignment(const CallBase *V, const TargetLibraryInfo *TLI)
Gets the alignment argument for an aligned_alloc-like function, using either built-in knowledge based...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1801
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
Definition MathExtras.h:352
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
Align getKnownAlignment(Value *V, const DataLayout &DL, const Instruction *CxtI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr)
Try to infer an alignment for the specified pointer.
Definition Local.h:240
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1756
SelectPatternFlavor
Specific patterns of select instructions we can match.
@ SPF_ABS
Floating point maxnum.
@ SPF_NABS
Absolute value.
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isModSet(const ModRefInfo MRI)
Definition ModRef.h:49
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1787
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1701
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI SelectPatternResult matchSelectPattern(Value *V, Value *&LHS, Value *&RHS, Instruction::CastOps *CastOp=nullptr, unsigned Depth=0)
Pattern match integer [SU]MIN, [SU]MAX and ABS idioms, returning the kind and providing the out param...
LLVM_ABI bool matchSimpleBinaryIntrinsicRecurrence(const IntrinsicInst *I, PHINode *&P, Value *&Init, Value *&OtherOp)
Attempt to match a simple value-accumulating recurrence of the form: llvm.intrinsic....
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1793
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
bool isAtLeastOrStrongerThan(AtomicOrdering AO, AtomicOrdering Other)
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
iterator_range< SplittingIterator > split(StringRef Str, StringRef Separator)
Split the specified string over a separator and return a range-compatible iterable over its partition...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth, bool MustPreserveProvenance=false)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool isNotCrossLaneOperation(const Instruction *I)
Return true if the instruction doesn't potentially cross vector lanes.
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI bool isKnownNonZero(const Value *V, const SimplifyQuery &Q, unsigned Depth=0)
Return true if the given value is known to be non-zero when defined.
constexpr int PoisonMaskElem
@ Mod
The access may modify the value stored in memory.
Definition ModRef.h:34
LLVM_ABI Value * simplifyFMAFMul(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for the multiplication of a FMA, fold the result or return null.
@ Other
Any other memory.
Definition ModRef.h:68
LLVM_ABI Value * simplifyConstrainedFPCall(CallBase *Call, const SimplifyQuery &Q)
Given a constrained FP intrinsic call, tries to compute its simplified version.
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1737
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI AssumeNonNullInfo getAssumeNonNullInfo(OperandBundleUse)
@ Add
Sum of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
bool isSafeToSpeculativelyExecuteWithVariableReplaced(const Instruction *I, bool IgnoreUBImplyingAttrs=true)
Don't use information from its non-constant operands.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI Value * getFreedOperand(const CallBase *CB, const TargetLibraryInfo *TLI)
If this if a call to a free function, return the freed operand.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
Definition MathExtras.h:233
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
LLVM_ABI std::optional< APInt > getAllocSize(const CallBase *CB, const TargetLibraryInfo *TLI, function_ref< const Value *(const Value *)> Mapper=[](const Value *V) { return V;})
Return the size of the requested allocation.
LLVM_ABI AssumeAlignInfo getAssumeAlignInfo(OperandBundleUse)
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool maskContainsAllOneOrUndef(Value *Mask)
Given a mask vector of i1, Return true if any of the elements of this predicate mask are known to be ...
LLVM_ABI std::optional< bool > isImpliedByDomCondition(const Value *Cond, const Instruction *ContextI, const DataLayout &DL)
Return the boolean condition value in the context of the given instruction if it is known based on do...
LLVM_ABI bool isDereferenceablePointer(const Value *V, Type *Ty, const SimplifyQuery &Q, bool IgnoreFree=false)
Equivalent to isDereferenceableAndAlignedPointer with an alignment of 1.
Definition Loads.cpp:264
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1774
LLVM_ABI bool isKnownNegation(const Value *X, const Value *Y, bool NeedNSW=false, bool AllowPoison=true)
Return true if the two given values are negation.
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1814
LLVM_ABI AssumeDereferenceableInfo getAssumeDereferenceableInfo(OperandBundleUse)
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI AssumeNoUndefInfo getAssumeNoUndefInfo(OperandBundleUse)
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI std::optional< bool > computeKnownFPSignBit(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Return false if we can prove that the specified FP value's sign bit is 0.
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Function *CxtF=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define NC
Definition regutils.h:42
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
Definition Metadata.h:774
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
@ IEEE
IEEE-754 denormal numbers preserved.
Matching combinators.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Align valueOrOne() const
For convenience, returns a valid alignment or 1 if undefined.
Definition Alignment.h:130
uint32_t getTagID() const
Return the tag of this operand bundle as an integer.
ArrayRef< Use > Inputs
SelectPatternFlavor Flavor
const DataLayout & DL
const Instruction * CxtI
SimplifyQuery getWithInstruction(const Instruction *I) const