LLVM 24.0.0git
InstCombineCalls.cpp
Go to the documentation of this file.
1//===- InstCombineCalls.cpp -----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the visitCall, visitInvoke, and visitCallBr functions.
10//
11//===----------------------------------------------------------------------===//
12
13#include "InstCombineInternal.h"
14#include "llvm/ADT/APFloat.h"
15#include "llvm/ADT/APInt.h"
16#include "llvm/ADT/APSInt.h"
17#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/Bitset.h"
22#include "llvm/ADT/Statistic.h"
28#include "llvm/Analysis/Loads.h"
33#include "llvm/IR/Attributes.h"
34#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
39#include "llvm/IR/DebugInfo.h"
41#include "llvm/IR/Function.h"
43#include "llvm/IR/InlineAsm.h"
44#include "llvm/IR/InstrTypes.h"
45#include "llvm/IR/Instruction.h"
48#include "llvm/IR/Intrinsics.h"
49#include "llvm/IR/IntrinsicsAArch64.h"
50#include "llvm/IR/IntrinsicsAMDGPU.h"
51#include "llvm/IR/IntrinsicsARM.h"
52#include "llvm/IR/IntrinsicsHexagon.h"
53#include "llvm/IR/LLVMContext.h"
54#include "llvm/IR/Metadata.h"
56#include "llvm/IR/Statepoint.h"
57#include "llvm/IR/Type.h"
58#include "llvm/IR/User.h"
59#include "llvm/IR/Value.h"
60#include "llvm/IR/ValueHandle.h"
65#include "llvm/Support/Debug.h"
76#include <algorithm>
77#include <cassert>
78#include <cstdint>
79#include <optional>
80#include <utility>
81#include <vector>
82
83#define DEBUG_TYPE "instcombine"
85
86using namespace llvm;
87using namespace PatternMatch;
88
89STATISTIC(NumSimplified, "Number of library calls simplified");
90
92 "instcombine-guard-widening-window",
93 cl::init(3),
94 cl::desc("How wide an instruction window to bypass looking for "
95 "another guard"));
96
97/// Return the specified type promoted as it would be to pass though a va_arg
98/// area.
100 if (IntegerType* ITy = dyn_cast<IntegerType>(Ty)) {
101 if (ITy->getBitWidth() < 32)
102 return Type::getInt32Ty(Ty->getContext());
103 }
104 return Ty;
105}
106
107/// Recognize a memcpy/memmove from a trivially otherwise unused alloca.
108/// TODO: This should probably be integrated with visitAllocSites, but that
109/// requires a deeper change to allow either unread or unwritten objects.
111 auto *Src = MI->getRawSource();
112 while (isa<GetElementPtrInst>(Src)) {
113 if (!Src->hasOneUse())
114 return false;
115 Src = cast<Instruction>(Src)->getOperand(0);
116 }
117 return isa<AllocaInst>(Src) && Src->hasOneUse();
118}
119
121 Align DstAlign = getKnownAlignment(MI->getRawDest(), DL, MI, &AC, &DT);
122 MaybeAlign CopyDstAlign = MI->getDestAlign();
123 if (!CopyDstAlign || *CopyDstAlign < DstAlign) {
124 MI->setDestAlignment(DstAlign);
125 return MI;
126 }
127
128 Align SrcAlign = getKnownAlignment(MI->getRawSource(), DL, MI, &AC, &DT);
129 MaybeAlign CopySrcAlign = MI->getSourceAlign();
130 if (!CopySrcAlign || *CopySrcAlign < SrcAlign) {
131 MI->setSourceAlignment(SrcAlign);
132 return MI;
133 }
134
135 // If we have a store to a location which is known constant, we can conclude
136 // that the store must be storing the constant value (else the memory
137 // wouldn't be constant), and this must be a noop.
138 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
139 // Set the size of the copy to 0, it will be deleted on the next iteration.
140 MI->setLength((uint64_t)0);
141 return MI;
142 }
143
144 // If the source is provably undef, the memcpy/memmove doesn't do anything
145 // (unless the transfer is volatile).
146 if (hasUndefSource(MI) && !MI->isVolatile()) {
147 // Set the size of the copy to 0, it will be deleted on the next iteration.
148 MI->setLength((uint64_t)0);
149 return MI;
150 }
151
152 // If MemCpyInst length is 1/2/4/8 bytes then replace memcpy with
153 // load/store.
154 ConstantInt *MemOpLength = dyn_cast<ConstantInt>(MI->getLength());
155 if (!MemOpLength) return nullptr;
156
157 // Source and destination pointer types are always "i8*" for intrinsic. See
158 // if the size is something we can handle with a single primitive load/store.
159 // A single load+store correctly handles overlapping memory in the memmove
160 // case.
161 uint64_t Size = MemOpLength->getLimitedValue();
162 assert(Size && "0-sized memory transferring should be removed already.");
163
164 if (Size > 8 || (Size&(Size-1)))
165 return nullptr; // If not 1/2/4/8 bytes, exit.
166
167 // If it is an atomic and alignment is less than the size then we will
168 // introduce the unaligned memory access which will be later transformed
169 // into libcall in CodeGen. This is not evident performance gain so disable
170 // it now.
171 if (MI->isAtomic())
172 if (*CopyDstAlign < Size || *CopySrcAlign < Size)
173 return nullptr;
174
175 // Use an integer load+store unless we can find something better.
176 IntegerType* IntType = IntegerType::get(MI->getContext(), Size<<3);
177
178 // If the memcpy has metadata describing the members, see if we can get the
179 // TBAA, scope and noalias tags describing our copy.
180 AAMDNodes AACopyMD = MI->getAAMetadata().adjustForAccess(Size);
181
182 Value *Src = MI->getArgOperand(1);
183 Value *Dest = MI->getArgOperand(0);
184 LoadInst *L = Builder.CreateLoad(IntType, Src);
185 // Alignment from the mem intrinsic will be better, so use it.
186 L->setAlignment(*CopySrcAlign);
187 L->setAAMetadata(AACopyMD);
188 MDNode *LoopMemParallelMD =
189 MI->getMetadata(LLVMContext::MD_mem_parallel_loop_access);
190 if (LoopMemParallelMD)
191 L->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
192 MDNode *AccessGroupMD = MI->getMetadata(LLVMContext::MD_access_group);
193 if (AccessGroupMD)
194 L->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
195
196 StoreInst *S = Builder.CreateStore(L, Dest);
197 // Alignment from the mem intrinsic will be better, so use it.
198 S->setAlignment(*CopyDstAlign);
199 S->setAAMetadata(AACopyMD);
200 if (LoopMemParallelMD)
201 S->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
202 if (AccessGroupMD)
203 S->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
204 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
205
206 if (auto *MT = dyn_cast<MemTransferInst>(MI)) {
207 // non-atomics can be volatile
208 L->setVolatile(MT->isVolatile());
209 S->setVolatile(MT->isVolatile());
210 }
211 if (MI->isAtomic()) {
212 // atomics have to be unordered
213 L->setOrdering(AtomicOrdering::Unordered);
215 }
216
217 // Set the size of the copy to 0, it will be deleted on the next iteration.
218 MI->setLength((uint64_t)0);
219 return MI;
220}
221
223 const Align KnownAlignment =
224 getKnownAlignment(MI->getDest(), DL, MI, &AC, &DT);
225 MaybeAlign MemSetAlign = MI->getDestAlign();
226 if (!MemSetAlign || *MemSetAlign < KnownAlignment) {
227 MI->setDestAlignment(KnownAlignment);
228 return MI;
229 }
230
231 // If we have a store to a location which is known constant, we can conclude
232 // that the store must be storing the constant value (else the memory
233 // wouldn't be constant), and this must be a noop.
234 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
235 // Set the size of the copy to 0, it will be deleted on the next iteration.
236 MI->setLength((uint64_t)0);
237 return MI;
238 }
239
240 // Remove memset with an undef value.
241 // FIXME: This is technically incorrect because it might overwrite a poison
242 // value. Change to PoisonValue once #52930 is resolved.
243 if (isa<UndefValue>(MI->getValue())) {
244 // Set the size of the copy to 0, it will be deleted on the next iteration.
245 MI->setLength((uint64_t)0);
246 return MI;
247 }
248
249 // Extract the length and validate the fill type.
250 ConstantInt *LenC = dyn_cast<ConstantInt>(MI->getLength());
251 Value *Fill = MI->getValue();
252 if (!LenC || !Fill->getType()->isIntegerTy(8))
253 return nullptr;
254 const uint64_t Len = LenC->getLimitedValue();
255 assert(Len && "0-sized memory setting should be removed already.");
256 const Align Alignment = MI->getDestAlign().valueOrOne();
257
258 // If it is an atomic and alignment is less than the size then we will
259 // introduce the unaligned memory access which will be later transformed
260 // into libcall in CodeGen. This is not evident performance gain so disable
261 // it now.
262 if (MI->isAtomic() && Alignment < Len)
263 return nullptr;
264
265 // memset(s,c,n) -> store s, c (for n=1,2,4,8)
266 if (Len <= 8 && isPowerOf2_32((uint32_t)Len)) {
267 Value *Dest = MI->getDest();
268
269 // Extract the fill value and store. A one-byte memset does not need
270 // replication so a nonconstant i8 fill can be stored directly.
271 Value *FillVal;
272 if (auto *FillC = dyn_cast<ConstantInt>(Fill))
273 FillVal = ConstantInt::get(MI->getContext(),
274 APInt::getSplat(Len * 8, FillC->getValue()));
275 else if (Len == 1)
276 FillVal = Fill;
277 else
278 return nullptr;
279
280 StoreInst *S = Builder.CreateStore(FillVal, Dest, MI->isVolatile());
281 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
282 for (DbgVariableRecord *DbgAssign : at::getDVRAssignmentMarkers(S)) {
283 if (llvm::is_contained(DbgAssign->location_ops(), Fill))
284 DbgAssign->replaceVariableLocationOp(Fill, FillVal);
285 }
286
287 S->setAlignment(Alignment);
288 if (MI->isAtomic())
290
291 // Set the size of the copy to 0, it will be deleted on the next iteration.
292 MI->setLength((uint64_t)0);
293 return MI;
294 }
295
296 return nullptr;
297}
298
299// TODO, Obvious Missing Transforms:
300// * Narrow width by halfs excluding zero/undef lanes
301Value *InstCombinerImpl::simplifyMaskedLoad(IntrinsicInst &II) {
302 Value *LoadPtr = II.getArgOperand(0);
303 const Align Alignment = II.getParamAlign(0).valueOrOne();
304 Value *Mask = II.getArgOperand(1);
305
306 // If the mask is all ones or poison, this is a plain vector load of the 1st
307 // argument.
308 if (match(Mask, m_AllOnesOrPoison())) {
309 LoadInst *L = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
310 "unmaskedload");
311 L->copyMetadata(II);
312 return L;
313 }
314
315 // If we can unconditionally load from this address, replace with a
316 // load/select idiom.
317 if (isDereferenceablePointer(LoadPtr, II.getType(),
319 LoadInst *LI = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
320 "unmaskedload");
321 LI->copyMetadata(II);
322 return Builder.CreateSelect(II.getArgOperand(1), LI, II.getArgOperand(2));
323 }
324
325 return nullptr;
326}
327
328// TODO, Obvious Missing Transforms:
329// * Single constant active lane -> store
330// * Narrow width by halfs excluding zero/undef lanes
331Instruction *InstCombinerImpl::simplifyMaskedStore(IntrinsicInst &II) {
332 Value *StorePtr = II.getArgOperand(1);
333 Align Alignment = II.getParamAlign(1).valueOrOne();
334 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
335 if (!ConstMask)
336 return nullptr;
337
338 // If the mask is all zeros or poison, this instruction does nothing.
339 if (match(ConstMask, m_ZeroOrPoison()))
341
342 // If the mask is all ones or poison, this is a plain vector store of the 1st
343 // argument.
344 if (match(ConstMask, m_AllOnesOrPoison())) {
345 StoreInst *S =
346 new StoreInst(II.getArgOperand(0), StorePtr, false, Alignment);
347 S->copyMetadata(II);
348 return S;
349 }
350
351 if (isa<ScalableVectorType>(ConstMask->getType()))
352 return nullptr;
353
354 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
355 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
356 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
357 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
358 PoisonElts))
359 return replaceOperand(II, 0, V);
360
361 return nullptr;
362}
363
364// TODO, Obvious Missing Transforms:
365// * Single constant active lane load -> load
366// * Dereferenceable address & few lanes -> scalarize speculative load/selects
367// * Adjacent vector addresses -> masked.load
368// * Narrow width by halfs excluding zero/undef lanes
369// * Vector incrementing address -> vector masked load
370Instruction *InstCombinerImpl::simplifyMaskedGather(IntrinsicInst &II) {
371 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(1));
372 if (!ConstMask)
373 return nullptr;
374
375 // Vector splat address w/known mask -> scalar load
376 // Fold the gather to load the source vector first lane
377 // because it is reloading the same value each time
378 if (ConstMask->isAllOnesValue())
379 if (auto *SplatPtr = getSplatValue(II.getArgOperand(0))) {
380 auto *VecTy = cast<VectorType>(II.getType());
381 const Align Alignment = II.getParamAlign(0).valueOrOne();
382 LoadInst *L = Builder.CreateAlignedLoad(VecTy->getElementType(), SplatPtr,
383 Alignment, "load.scalar");
384 Value *Shuf =
385 Builder.CreateVectorSplat(VecTy->getElementCount(), L, "broadcast");
387 }
388
389 return nullptr;
390}
391
392// TODO, Obvious Missing Transforms:
393// * Single constant active lane -> store
394// * Adjacent vector addresses -> masked.store
395// * Narrow store width by halfs excluding zero/undef lanes
396// * Vector incrementing address -> vector masked store
397Instruction *InstCombinerImpl::simplifyMaskedScatter(IntrinsicInst &II) {
398 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
399 if (!ConstMask)
400 return nullptr;
401
402 // If the mask is all zeros or poison, a scatter does nothing.
403 if (match(ConstMask, m_ZeroOrPoison()))
405
406 // Vector splat address -> scalar store
407 if (auto *SplatPtr = getSplatValue(II.getArgOperand(1))) {
408 // scatter(splat(value), splat(ptr), non-zero-mask) -> store value, ptr
409 if (auto *SplatValue = getSplatValue(II.getArgOperand(0))) {
410 if (maskContainsAllOneOrUndef(ConstMask)) {
411 Align Alignment = II.getParamAlign(1).valueOrOne();
412 StoreInst *S = new StoreInst(SplatValue, SplatPtr, /*IsVolatile=*/false,
413 Alignment);
414 S->copyMetadata(II);
415 return S;
416 }
417 }
418 // scatter(vector, splat(ptr), splat(true)) -> store extract(vector,
419 // lastlane), ptr
420 if (ConstMask->isAllOnesValue()) {
421 Align Alignment = II.getParamAlign(1).valueOrOne();
422 VectorType *WideLoadTy = cast<VectorType>(II.getArgOperand(1)->getType());
423 ElementCount VF = WideLoadTy->getElementCount();
424 Value *RunTimeVF = Builder.CreateElementCount(Builder.getInt32Ty(), VF);
425 Value *LastLane = Builder.CreateSub(RunTimeVF, Builder.getInt32(1));
426 Value *Extract =
427 Builder.CreateExtractElement(II.getArgOperand(0), LastLane);
428 StoreInst *S =
429 new StoreInst(Extract, SplatPtr, /*IsVolatile=*/false, Alignment);
430 S->copyMetadata(II);
431 return S;
432 }
433 }
434 if (isa<ScalableVectorType>(ConstMask->getType()))
435 return nullptr;
436
437 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
438 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
439 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
440 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
441 PoisonElts))
442 return replaceOperand(II, 0, V);
443 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(1), DemandedElts,
444 PoisonElts))
445 return replaceOperand(II, 1, V);
446
447 return nullptr;
448}
449
450/// This function transforms launder.invariant.group and strip.invariant.group
451/// like:
452/// launder(launder(%x)) -> launder(%x) (the result is not the argument)
453/// launder(strip(%x)) -> launder(%x)
454/// strip(strip(%x)) -> strip(%x) (the result is not the argument)
455/// strip(launder(%x)) -> strip(%x)
456/// This is legal because it preserves the most recent information about
457/// the presence or absence of invariant.group.
459 InstCombinerImpl &IC) {
460 auto *Arg = II.getArgOperand(0);
461 auto *StrippedArg = Arg->stripPointerCasts();
462 auto *StrippedInvariantGroupsArg = StrippedArg;
463 while (auto *Intr = dyn_cast<IntrinsicInst>(StrippedInvariantGroupsArg)) {
464 if (Intr->getIntrinsicID() != Intrinsic::launder_invariant_group &&
465 Intr->getIntrinsicID() != Intrinsic::strip_invariant_group)
466 break;
467 StrippedInvariantGroupsArg = Intr->getArgOperand(0)->stripPointerCasts();
468 }
469 if (StrippedArg == StrippedInvariantGroupsArg)
470 return nullptr; // No launders/strips to remove.
471
472 Value *Result = nullptr;
473
474 if (II.getIntrinsicID() == Intrinsic::launder_invariant_group)
475 Result = IC.Builder.CreateLaunderInvariantGroup(StrippedInvariantGroupsArg);
476 else if (II.getIntrinsicID() == Intrinsic::strip_invariant_group)
477 Result = IC.Builder.CreateStripInvariantGroup(StrippedInvariantGroupsArg);
478 else
480 "simplifyInvariantGroupIntrinsic only handles launder and strip");
481 if (Result->getType()->getPointerAddressSpace() !=
482 II.getType()->getPointerAddressSpace())
483 Result = IC.Builder.CreateAddrSpaceCast(Result, II.getType());
484
485 return cast<Instruction>(Result);
486}
487
489 assert((II.getIntrinsicID() == Intrinsic::cttz ||
490 II.getIntrinsicID() == Intrinsic::ctlz) &&
491 "Expected cttz or ctlz intrinsic");
492 bool IsTZ = II.getIntrinsicID() == Intrinsic::cttz;
493 Value *Op0 = II.getArgOperand(0);
494 Value *Op1 = II.getArgOperand(1);
495 Value *X;
496 // ctlz(bitreverse(x)) -> cttz(x)
497 // cttz(bitreverse(x)) -> ctlz(x)
498 if (match(Op0, m_BitReverse(m_Value(X)))) {
499 Intrinsic::ID ID = IsTZ ? Intrinsic::ctlz : Intrinsic::cttz;
500 Function *F =
501 Intrinsic::getOrInsertDeclaration(II.getModule(), ID, II.getType());
502 return CallInst::Create(F, {X, II.getArgOperand(1)});
503 }
504
505 if (II.getType()->isIntOrIntVectorTy(1)) {
506 // ctlz/cttz i1 Op0 --> not Op0
507 if (match(Op1, m_Zero()))
508 return BinaryOperator::CreateNot(Op0);
509 // If zero is poison, then the input can be assumed to be "true", so the
510 // instruction simplifies to "false".
511 assert(match(Op1, m_One()) && "Expected ctlz/cttz operand to be 0 or 1");
512 return IC.replaceInstUsesWith(II, ConstantInt::getNullValue(II.getType()));
513 }
514
515 // If ctlz/cttz is only used as a shift amount, set is_zero_poison to true.
516 if (II.hasOneUse() && match(Op1, m_Zero()) &&
517 match(II.user_back(), m_Shift(m_Value(), m_Specific(&II))))
518 return CallInst::Create(II.getCalledFunction(),
519 {Op0, IC.Builder.getTrue()});
520
521 Constant *C;
522
523 if (IsTZ) {
524 // cttz(-x) -> cttz(x)
525 if (match(Op0, m_Neg(m_Value(X))))
526 return CallInst::Create(II.getCalledFunction(), {X, Op1});
527
528 // cttz(-x & x) -> cttz(x)
529 if (match(Op0, m_c_And(m_Neg(m_Value(X)), m_Deferred(X))))
530 return CallInst::Create(II.getCalledFunction(), {X, Op1});
531
532 // cttz(mul(X, OddC)) -> cttz(X)
533 if (match(Op0, m_Mul(m_Value(X),
534 m_CheckedInt([](const APInt &C) { return C[0]; }))))
535 return CallInst::Create(II.getCalledFunction(), {X, Op1});
536
537 // cttz(sext(x)) -> cttz(zext(x))
538 if (match(Op0, m_OneUse(m_SExt(m_Value(X))))) {
539 auto *Zext = IC.Builder.CreateZExt(X, II.getType());
540 auto *CttzZext =
541 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, Zext, Op1);
542 return IC.replaceInstUsesWith(II, CttzZext);
543 }
544
545 // Zext doesn't change the number of trailing zeros, so narrow:
546 // cttz(zext(x)) -> zext(cttz(x)) if the 'ZeroIsPoison' parameter is 'true'.
547 if (match(Op0, m_OneUse(m_ZExt(m_Value(X)))) && match(Op1, m_One())) {
548 auto *Cttz = IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, X,
549 IC.Builder.getTrue());
550 auto *ZextCttz = IC.Builder.CreateZExt(Cttz, II.getType());
551 return IC.replaceInstUsesWith(II, ZextCttz);
552 }
553
554 // cttz(abs(x)) -> cttz(x)
555 // cttz(nabs(x)) -> cttz(x)
556 Value *Y;
558 if (SPF == SPF_ABS || SPF == SPF_NABS)
559 return CallInst::Create(II.getCalledFunction(), {X, Op1});
560
562 return CallInst::Create(II.getCalledFunction(), {X, Op1});
563
564 // cttz(shl(%const, %val), 1) --> add(cttz(%const, 1), %val)
565 if (match(Op0, m_Shl(m_ImmConstant(C), m_Value(X))) &&
566 match(Op1, m_One())) {
567 Value *ConstCttz =
568 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
569 return BinaryOperator::CreateAdd(ConstCttz, X);
570 }
571
572 // cttz(lshr exact (%const, %val), 1) --> sub(cttz(%const, 1), %val)
573 if (match(Op0, m_Exact(m_LShr(m_ImmConstant(C), m_Value(X)))) &&
574 match(Op1, m_One())) {
575 Value *ConstCttz =
576 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
577 return BinaryOperator::CreateSub(ConstCttz, X);
578 }
579
580 // cttz(add(lshr(UINT_MAX, %val), 1)) --> sub(width, %val)
581 if (match(Op0, m_Add(m_LShr(m_AllOnes(), m_Value(X)), m_One()))) {
582 Value *Width =
583 ConstantInt::get(II.getType(), II.getType()->getScalarSizeInBits());
584 return BinaryOperator::CreateSub(Width, X);
585 }
586 } else {
587 // ctlz(lshr(%const, %val), 1) --> add(ctlz(%const, 1), %val)
588 if (match(Op0, m_LShr(m_ImmConstant(C), m_Value(X))) &&
589 match(Op1, m_One())) {
590 Value *ConstCtlz =
591 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
592 return BinaryOperator::CreateAdd(ConstCtlz, X);
593 }
594
595 // ctlz(shl nuw (%const, %val), 1) --> sub(ctlz(%const, 1), %val)
596 if (match(Op0, m_NUWShl(m_ImmConstant(C), m_Value(X))) &&
597 match(Op1, m_One())) {
598 Value *ConstCtlz =
599 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
600 return BinaryOperator::CreateSub(ConstCtlz, X);
601 }
602
603 // ctlz(~x & (x - 1)) -> bitwidth - cttz(x, false)
604 if (Op0->hasOneUse() &&
605 match(Op0,
607 Type *Ty = II.getType();
608 unsigned BitWidth = Ty->getScalarSizeInBits();
609 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
610 {X, IC.Builder.getFalse()});
611 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
612 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
613 }
614 }
615
616 // cttz(Pow2) -> Log2(Pow2)
617 // ctlz(Pow2) -> BitWidth - 1 - Log2(Pow2)
618 if (auto *R = IC.tryGetLog2(Op0, match(Op1, m_One()))) {
619 if (IsTZ)
620 return IC.replaceInstUsesWith(II, R);
621 BinaryOperator *BO = BinaryOperator::CreateSub(
622 ConstantInt::get(R->getType(), R->getType()->getScalarSizeInBits() - 1),
623 R);
624 BO->setHasNoSignedWrap();
626 return BO;
627 }
628
630
631 // Create a mask for bits above (ctlz) or below (cttz) the first known one.
632 unsigned PossibleZeros = IsTZ ? Known.countMaxTrailingZeros()
633 : Known.countMaxLeadingZeros();
634 unsigned DefiniteZeros = IsTZ ? Known.countMinTrailingZeros()
635 : Known.countMinLeadingZeros();
636
637 // If all bits above (ctlz) or below (cttz) the first known one are known
638 // zero, this value is constant.
639 // FIXME: This should be in InstSimplify because we're replacing an
640 // instruction with a constant.
641 if (PossibleZeros == DefiniteZeros) {
642 auto *C = ConstantInt::get(Op0->getType(), DefiniteZeros);
643 return IC.replaceInstUsesWith(II, C);
644 }
645
646 // If the input to cttz/ctlz is known to be non-zero,
647 // then change the 'ZeroIsPoison' parameter to 'true'
648 // because we know the zero behavior can't affect the result.
649 if (!Known.One.isZero() ||
651 if (!match(II.getArgOperand(1), m_One()))
652 return CallInst::Create(II.getCalledFunction(),
653 {Op0, IC.Builder.getTrue()});
654 }
655
656 // Add range attribute since known bits can't completely reflect what we know.
657 unsigned BitWidth = Op0->getType()->getScalarSizeInBits();
658 if (BitWidth != 1 && !II.hasRetAttr(Attribute::Range) &&
659 !II.getMetadata(LLVMContext::MD_range)) {
660 ConstantRange Range(APInt(BitWidth, DefiniteZeros),
661 APInt(BitWidth, PossibleZeros + 1));
662 II.addRangeRetAttr(Range);
663 return &II;
664 }
665
666 return nullptr;
667}
668
670 assert(II.getIntrinsicID() == Intrinsic::ctpop &&
671 "Expected ctpop intrinsic");
672 Type *Ty = II.getType();
673 unsigned BitWidth = Ty->getScalarSizeInBits();
674 Value *Op0 = II.getArgOperand(0);
675 Value *X, *Y;
676
677 // ctpop(bitreverse(x)) -> ctpop(x)
678 // ctpop(bswap(x)) -> ctpop(x)
679 if (match(Op0, m_BitReverse(m_Value(X))) || match(Op0, m_BSwap(m_Value(X))))
680 return CallInst::Create(II.getCalledFunction(), X);
681
682 // ctpop(rot(x)) -> ctpop(x)
683 if ((match(Op0, m_FShl(m_Value(X), m_Value(Y), m_Value())) ||
684 match(Op0, m_FShr(m_Value(X), m_Value(Y), m_Value()))) &&
685 X == Y)
686 return CallInst::Create(II.getCalledFunction(), X);
687
688 // ctpop(x | -x) -> bitwidth - cttz(x, false)
689 if (Op0->hasOneUse() &&
690 match(Op0, m_c_Or(m_Value(X), m_Neg(m_Deferred(X))))) {
691 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
692 {X, IC.Builder.getFalse()});
693 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
694 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
695 }
696
697 // ctpop(~x & (x - 1)) -> cttz(x, false)
698 if (match(Op0,
700 Function *F =
701 Intrinsic::getOrInsertDeclaration(II.getModule(), Intrinsic::cttz, Ty);
702 return CallInst::Create(F, {X, IC.Builder.getFalse()});
703 }
704
705 // Zext doesn't change the number of set bits, so narrow:
706 // ctpop (zext X) --> zext (ctpop X)
707 if (match(Op0, m_OneUse(m_ZExt(m_Value(X))))) {
708 Value *NarrowPop = IC.Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, X);
709 return CastInst::Create(Instruction::ZExt, NarrowPop, Ty);
710 }
711
713 IC.computeKnownBits(Op0, Known, &II);
714
715 // If all bits are zero except for exactly one fixed bit, then the result
716 // must be 0 or 1, and we can get that answer by shifting to LSB:
717 // ctpop (X & 32) --> (X & 32) >> 5
718 // TODO: Investigate removing this as its likely unnecessary given the below
719 // `isKnownToBeAPowerOfTwo` check.
720 if ((~Known.Zero).isPowerOf2())
721 return BinaryOperator::CreateLShr(
722 Op0, ConstantInt::get(Ty, (~Known.Zero).exactLogBase2()));
723
724 // More generally we can also handle non-constant power of 2 patterns such as
725 // shl/shr(Pow2, X), (X & -X), etc... by transforming:
726 // ctpop(Pow2OrZero) --> icmp ne X, 0
727 if (IC.isKnownToBeAPowerOfTwo(Op0, /* OrZero */ true))
728 return CastInst::Create(Instruction::ZExt,
731 Ty);
732
733 // Add range attribute since known bits can't completely reflect what we know.
734 if (BitWidth != 1) {
735 ConstantRange OldRange =
736 II.getRange().value_or(ConstantRange::getFull(BitWidth));
737
738 unsigned Lower = Known.countMinPopulation();
739 unsigned Upper = Known.countMaxPopulation() + 1;
740
741 if (Lower == 0 && OldRange.contains(APInt::getZero(BitWidth)) &&
743 Lower = 1;
744
746 Range = Range.intersectWith(OldRange, ConstantRange::Unsigned);
747
748 if (Range != OldRange) {
749 II.addRangeRetAttr(Range);
750 return &II;
751 }
752 }
753
754 return nullptr;
755}
756
757/// Convert `tbl`/`tbx` intrinsics to shufflevector if the mask is constant, and
758/// at most two source operands are actually referenced.
760 bool IsExtension) {
761 // Bail out if the mask is not a constant.
762 auto *C = dyn_cast<Constant>(II.getArgOperand(II.arg_size() - 1));
763 if (!C)
764 return nullptr;
765
766 auto *RetTy = cast<FixedVectorType>(II.getType());
767 unsigned NumIndexes = RetTy->getNumElements();
768
769 // Only perform this transformation for <8 x i8> and <16 x i8> vector types.
770 if (!RetTy->getElementType()->isIntegerTy(8) ||
771 (NumIndexes != 8 && NumIndexes != 16))
772 return nullptr;
773
774 // For tbx instructions, the first argument is the "fallback" vector, which
775 // has the same length as the mask and return type.
776 unsigned int StartIndex = (unsigned)IsExtension;
777 auto *SourceTy =
778 cast<FixedVectorType>(II.getArgOperand(StartIndex)->getType());
779 // Note that the element count of each source vector does *not* need to be the
780 // same as the element count of the return type and mask! All source vectors
781 // must have the same element count as each other, though.
782 unsigned NumElementsPerSource = SourceTy->getNumElements();
783
784 // There are no tbl/tbx intrinsics for which the destination size exceeds the
785 // source size. However, our definitions of the intrinsics, at least in
786 // IntrinsicsAArch64.td, allow for arbitrary destination vector sizes, so it
787 // *could* technically happen.
788 if (NumIndexes > NumElementsPerSource)
789 return nullptr;
790
791 // The tbl/tbx intrinsics take several source operands followed by a mask
792 // operand.
793 unsigned int NumSourceOperands = II.arg_size() - 1 - (unsigned)IsExtension;
794
795 // Map input operands to shuffle indices. This also helpfully deduplicates the
796 // input arguments, in case the same value is passed as an argument multiple
797 // times.
798 SmallDenseMap<Value *, unsigned, 2> ValueToShuffleSlot;
799 Value *ShuffleOperands[2] = {PoisonValue::get(SourceTy),
800 PoisonValue::get(SourceTy)};
801
802 int Indexes[16];
803 for (unsigned I = 0; I < NumIndexes; ++I) {
804 Constant *COp = C->getAggregateElement(I);
805
806 if (!COp || (!isa<UndefValue>(COp) && !isa<ConstantInt>(COp)))
807 return nullptr;
808
809 if (isa<UndefValue>(COp)) {
810 Indexes[I] = -1;
811 continue;
812 }
813
814 uint64_t Index = cast<ConstantInt>(COp)->getZExtValue();
815 // The index of the input argument that this index references (0 = first
816 // source argument, etc).
817 unsigned SourceOperandIndex = Index / NumElementsPerSource;
818 // The index of the element at that source operand.
819 unsigned SourceOperandElementIndex = Index % NumElementsPerSource;
820
821 Value *SourceOperand;
822 if (SourceOperandIndex >= NumSourceOperands) {
823 // This index is out of bounds. Map it to index into either the fallback
824 // vector (tbx) or vector of zeroes (tbl).
825 SourceOperandIndex = NumSourceOperands;
826 if (IsExtension) {
827 // For out-of-bounds indices in tbx, choose the `I`th element of the
828 // fallback.
829 SourceOperand = II.getArgOperand(0);
830 SourceOperandElementIndex = I;
831 } else {
832 // Otherwise, choose some element from the dummy vector of zeroes (we'll
833 // always choose the first).
834 SourceOperand = Constant::getNullValue(SourceTy);
835 SourceOperandElementIndex = 0;
836 }
837 } else {
838 SourceOperand = II.getArgOperand(SourceOperandIndex + StartIndex);
839 }
840
841 // The source operand may be the fallback vector, which may not have the
842 // same number of elements as the source vector. In that case, we *could*
843 // choose to extend its length with another shufflevector, but it's simpler
844 // to just bail instead.
845 if (cast<FixedVectorType>(SourceOperand->getType())->getNumElements() !=
846 NumElementsPerSource)
847 return nullptr;
848
849 // We now know the source operand referenced by this index. Make it a
850 // shufflevector operand, if it isn't already.
851 unsigned NumSlots = ValueToShuffleSlot.size();
852 // This shuffle references more than two sources, and hence cannot be
853 // represented as a shufflevector.
854 if (NumSlots == 2 && !ValueToShuffleSlot.contains(SourceOperand))
855 return nullptr;
856
857 auto [It, Inserted] =
858 ValueToShuffleSlot.try_emplace(SourceOperand, NumSlots);
859 if (Inserted)
860 ShuffleOperands[It->getSecond()] = SourceOperand;
861
862 unsigned RemappedIndex =
863 (It->getSecond() * NumElementsPerSource) + SourceOperandElementIndex;
864 Indexes[I] = RemappedIndex;
865 }
866
868 ShuffleOperands[0], ShuffleOperands[1], ArrayRef(Indexes, NumIndexes));
869 return IC.replaceInstUsesWith(II, Shuf);
870}
871
872// Returns true iff the 2 intrinsics have the same operands, limiting the
873// comparison to the first NumOperands.
874static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E,
875 unsigned NumOperands) {
876 assert(I.arg_size() >= NumOperands && "Not enough operands");
877 assert(E.arg_size() >= NumOperands && "Not enough operands");
878 for (unsigned i = 0; i < NumOperands; i++)
879 if (I.getArgOperand(i) != E.getArgOperand(i))
880 return false;
881 return true;
882}
883
884// Remove trivially empty start/end intrinsic ranges, i.e. a start
885// immediately followed by an end (ignoring debuginfo or other
886// start/end intrinsics in between). As this handles only the most trivial
887// cases, tracking the nesting level is not needed:
888//
889// call @llvm.foo.start(i1 0)
890// call @llvm.foo.start(i1 0) ; This one won't be skipped: it will be removed
891// call @llvm.foo.end(i1 0)
892// call @llvm.foo.end(i1 0) ; &I
893static bool
895 std::function<bool(const IntrinsicInst &)> IsStart) {
896 // We start from the end intrinsic and scan backwards, so that InstCombine
897 // has already processed (and potentially removed) all the instructions
898 // before the end intrinsic.
899 BasicBlock::reverse_iterator BI(EndI), BE(EndI.getParent()->rend());
900 for (; BI != BE; ++BI) {
901 if (auto *I = dyn_cast<IntrinsicInst>(&*BI)) {
902 if (I->isDebugOrPseudoInst() ||
903 I->getIntrinsicID() == EndI.getIntrinsicID())
904 continue;
905 if (IsStart(*I)) {
906 if (haveSameOperands(EndI, *I, EndI.arg_size())) {
908 IC.eraseInstFromFunction(EndI);
909 return true;
910 }
911 // Skip start intrinsics that don't pair with this end intrinsic.
912 continue;
913 }
914 }
915 break;
916 }
917
918 return false;
919}
920
922 removeTriviallyEmptyRange(I, *this, [&I](const IntrinsicInst &II) {
923 // Bail out on the case where the source va_list of a va_copy is destroyed
924 // immediately by a follow-up va_end.
925 return II.getIntrinsicID() == Intrinsic::vastart ||
926 (II.getIntrinsicID() == Intrinsic::vacopy &&
927 I.getArgOperand(0) != II.getArgOperand(1));
928 });
929 return nullptr;
930}
931
933 assert(Call.arg_size() > 1 && "Need at least 2 args to swap");
934 Value *Arg0 = Call.getArgOperand(0), *Arg1 = Call.getArgOperand(1);
935 if (isa<Constant>(Arg0) && !isa<Constant>(Arg1)) {
936 Call.setArgOperand(0, Arg1);
937 Call.setArgOperand(1, Arg0);
938 AttributeList CallAttr = Call.getAttributes();
939 AttributeSet LHSAttr = CallAttr.getParamAttrs(0);
940 AttributeSet RHSAttr = CallAttr.getParamAttrs(1);
941 LLVMContext &Ctx = Call.getContext();
942 Call.setAttributes(CallAttr
943 .setAttributesAtIndex(
944 Ctx, AttributeList::FirstArgIndex + 0, RHSAttr)
945 .setAttributesAtIndex(
946 Ctx, AttributeList::FirstArgIndex + 1, LHSAttr));
947 return &Call;
948 }
949 return nullptr;
950}
951
952/// Creates a result tuple for an overflow intrinsic \p II with a given
953/// \p Result and a constant \p Overflow value.
955 Constant *Overflow) {
956 Constant *V[] = {PoisonValue::get(Result->getType()), Overflow};
957 StructType *ST = cast<StructType>(II->getType());
958 Constant *Struct = ConstantStruct::get(ST, V);
959 return InsertValueInst::Create(Struct, Result, 0);
960}
961
963InstCombinerImpl::foldIntrinsicWithOverflowCommon(IntrinsicInst *II) {
964 WithOverflowInst *WO = cast<WithOverflowInst>(II);
965 Value *OperationResult = nullptr;
966 Constant *OverflowResult = nullptr;
967 if (OptimizeOverflowCheck(WO->getBinaryOp(), WO->isSigned(), WO->getLHS(),
968 WO->getRHS(), *WO, OperationResult, OverflowResult))
969 return createOverflowTuple(WO, OperationResult, OverflowResult);
970
971 // See whether we can optimize the overflow check with assumption information.
972 for (User *U : WO->users()) {
973 if (!match(U, m_ExtractValue<1>(m_Value())))
974 continue;
975
976 for (auto &AssumeVH : AC.assumptionsFor(U)) {
977 if (!AssumeVH)
978 continue;
979 CallInst *I = cast<CallInst>(AssumeVH);
980 if (!match(I->getArgOperand(0), m_Not(m_Specific(U))))
981 continue;
982 if (!isValidAssumeForContext(I, II, /*DT=*/nullptr,
983 /*AllowEphemerals=*/true))
984 continue;
985 Value *Result =
986 Builder.CreateBinOp(WO->getBinaryOp(), WO->getLHS(), WO->getRHS());
987 Result->takeName(WO);
988 if (auto *Inst = dyn_cast<Instruction>(Result)) {
989 if (WO->isSigned())
990 Inst->setHasNoSignedWrap();
991 else
992 Inst->setHasNoUnsignedWrap();
993 }
994 return createOverflowTuple(WO, Result,
995 ConstantInt::getFalse(U->getType()));
996 }
997 }
998
999 return nullptr;
1000}
1001
1002static bool inputDenormalIsIEEE(const Function &F, const Type *Ty) {
1003 Ty = Ty->getScalarType();
1004 return F.getDenormalMode(Ty->getFltSemantics()).Input == DenormalMode::IEEE;
1005}
1006
1007static bool inputDenormalIsDAZ(const Function &F, const Type *Ty) {
1008 Ty = Ty->getScalarType();
1009 return F.getDenormalMode(Ty->getFltSemantics()).inputsAreZero();
1010}
1011
1012/// \returns the compare predicate type if the test performed by
1013/// llvm.is.fpclass(x, \p Mask) is equivalent to fcmp o__ x, 0.0 with the
1014/// floating-point environment assumed for \p F for type \p Ty
1016 const Function &F, Type *Ty) {
1017 switch (static_cast<unsigned>(Mask)) {
1018 case fcZero:
1019 if (inputDenormalIsIEEE(F, Ty))
1020 return FCmpInst::FCMP_OEQ;
1021 break;
1022 case fcZero | fcSubnormal:
1023 if (inputDenormalIsDAZ(F, Ty))
1024 return FCmpInst::FCMP_OEQ;
1025 break;
1026 case fcPositive | fcNegZero:
1027 if (inputDenormalIsIEEE(F, Ty))
1028 return FCmpInst::FCMP_OGE;
1029 break;
1031 if (inputDenormalIsDAZ(F, Ty))
1032 return FCmpInst::FCMP_OGE;
1033 break;
1035 if (inputDenormalIsIEEE(F, Ty))
1036 return FCmpInst::FCMP_OGT;
1037 break;
1038 case fcNegative | fcPosZero:
1039 if (inputDenormalIsIEEE(F, Ty))
1040 return FCmpInst::FCMP_OLE;
1041 break;
1043 if (inputDenormalIsDAZ(F, Ty))
1044 return FCmpInst::FCMP_OLE;
1045 break;
1047 if (inputDenormalIsIEEE(F, Ty))
1048 return FCmpInst::FCMP_OLT;
1049 break;
1050 case fcPosNormal | fcPosInf:
1051 if (inputDenormalIsDAZ(F, Ty))
1052 return FCmpInst::FCMP_OGT;
1053 break;
1054 case fcNegNormal | fcNegInf:
1055 if (inputDenormalIsDAZ(F, Ty))
1056 return FCmpInst::FCMP_OLT;
1057 break;
1058 case ~fcZero & ~fcNan:
1059 if (inputDenormalIsIEEE(F, Ty))
1060 return FCmpInst::FCMP_ONE;
1061 break;
1062 case ~(fcZero | fcSubnormal) & ~fcNan:
1063 if (inputDenormalIsDAZ(F, Ty))
1064 return FCmpInst::FCMP_ONE;
1065 break;
1066 default:
1067 break;
1068 }
1069
1071}
1072
1073Instruction *InstCombinerImpl::foldIntrinsicIsFPClass(IntrinsicInst &II) {
1074 Value *Src0 = II.getArgOperand(0);
1075 Value *Src1 = II.getArgOperand(1);
1076 const ConstantInt *CMask = cast<ConstantInt>(Src1);
1077 FPClassTest Mask = static_cast<FPClassTest>(CMask->getZExtValue());
1078 const bool IsUnordered = (Mask & fcNan) == fcNan;
1079 const bool IsOrdered = (Mask & fcNan) == fcNone;
1080 const FPClassTest OrderedMask = Mask & ~fcNan;
1081 const FPClassTest OrderedInvertedMask = ~OrderedMask & ~fcNan;
1082
1083 const bool IsStrict =
1084 II.getFunction()->getAttributes().hasFnAttr(Attribute::StrictFP);
1085
1086 Value *FNegSrc;
1087 // is.fpclass (fneg x), mask -> is.fpclass x, (fneg mask)
1088 if (match(Src0, m_FNeg(m_Value(FNegSrc))))
1089 return CallInst::Create(
1090 II.getCalledFunction(),
1091 {FNegSrc, ConstantInt::get(Src1->getType(), fneg(Mask))});
1092
1093 Value *FAbsSrc;
1094 if (match(Src0, m_FAbs(m_Value(FAbsSrc))))
1095 return CallInst::Create(
1096 II.getCalledFunction(),
1097 {FAbsSrc, ConstantInt::get(Src1->getType(), inverse_fabs(Mask))});
1098
1099 if ((OrderedMask == fcInf || OrderedInvertedMask == fcInf) &&
1100 (IsOrdered || IsUnordered) && !IsStrict) {
1101 // is.fpclass(x, fcInf) -> fcmp oeq fabs(x), +inf
1102 // is.fpclass(x, ~fcInf) -> fcmp one fabs(x), +inf
1103 // is.fpclass(x, fcInf|fcNan) -> fcmp ueq fabs(x), +inf
1104 // is.fpclass(x, ~(fcInf|fcNan)) -> fcmp une fabs(x), +inf
1106 FCmpInst::Predicate Pred =
1107 IsUnordered ? FCmpInst::FCMP_UEQ : FCmpInst::FCMP_OEQ;
1108 if (OrderedInvertedMask == fcInf)
1109 Pred = IsUnordered ? FCmpInst::FCMP_UNE : FCmpInst::FCMP_ONE;
1110
1111 Value *Fabs = Builder.CreateFAbs(Src0);
1112 Value *CmpInf = Builder.CreateFCmp(Pred, Fabs, Inf);
1113 CmpInf->takeName(&II);
1114 return replaceInstUsesWith(II, CmpInf);
1115 }
1116
1117 if ((OrderedMask == fcPosInf || OrderedMask == fcNegInf) &&
1118 (IsOrdered || IsUnordered) && !IsStrict) {
1119 // is.fpclass(x, fcPosInf) -> fcmp oeq x, +inf
1120 // is.fpclass(x, fcNegInf) -> fcmp oeq x, -inf
1121 // is.fpclass(x, fcPosInf|fcNan) -> fcmp ueq x, +inf
1122 // is.fpclass(x, fcNegInf|fcNan) -> fcmp ueq x, -inf
1123 Constant *Inf =
1124 ConstantFP::getInfinity(Src0->getType(), OrderedMask == fcNegInf);
1125 Value *EqInf = IsUnordered ? Builder.CreateFCmpUEQ(Src0, Inf)
1126 : Builder.CreateFCmpOEQ(Src0, Inf);
1127
1128 EqInf->takeName(&II);
1129 return replaceInstUsesWith(II, EqInf);
1130 }
1131
1132 if ((OrderedInvertedMask == fcPosInf || OrderedInvertedMask == fcNegInf) &&
1133 (IsOrdered || IsUnordered) && !IsStrict) {
1134 // is.fpclass(x, ~fcPosInf) -> fcmp one x, +inf
1135 // is.fpclass(x, ~fcNegInf) -> fcmp one x, -inf
1136 // is.fpclass(x, ~fcPosInf|fcNan) -> fcmp une x, +inf
1137 // is.fpclass(x, ~fcNegInf|fcNan) -> fcmp une x, -inf
1139 OrderedInvertedMask == fcNegInf);
1140 Value *NeInf = IsUnordered ? Builder.CreateFCmpUNE(Src0, Inf)
1141 : Builder.CreateFCmpONE(Src0, Inf);
1142 NeInf->takeName(&II);
1143 return replaceInstUsesWith(II, NeInf);
1144 }
1145
1146 if (Mask == fcNan && !IsStrict) {
1147 // Equivalent of isnan. Replace with standard fcmp if we don't care about FP
1148 // exceptions.
1149 Value *IsNan =
1150 Builder.CreateFCmpUNO(Src0, ConstantFP::getZero(Src0->getType()));
1151 IsNan->takeName(&II);
1152 return replaceInstUsesWith(II, IsNan);
1153 }
1154
1155 if (Mask == (~fcNan & fcAllFlags) && !IsStrict) {
1156 // Equivalent of !isnan. Replace with standard fcmp.
1157 Value *FCmp =
1158 Builder.CreateFCmpORD(Src0, ConstantFP::getZero(Src0->getType()));
1159 FCmp->takeName(&II);
1160 return replaceInstUsesWith(II, FCmp);
1161 }
1162
1164
1165 // Try to replace with an fcmp with 0
1166 //
1167 // is.fpclass(x, fcZero) -> fcmp oeq x, 0.0
1168 // is.fpclass(x, fcZero | fcNan) -> fcmp ueq x, 0.0
1169 // is.fpclass(x, ~fcZero & ~fcNan) -> fcmp one x, 0.0
1170 // is.fpclass(x, ~fcZero) -> fcmp une x, 0.0
1171 //
1172 // is.fpclass(x, fcPosSubnormal | fcPosNormal | fcPosInf) -> fcmp ogt x, 0.0
1173 // is.fpclass(x, fcPositive | fcNegZero) -> fcmp oge x, 0.0
1174 //
1175 // is.fpclass(x, fcNegSubnormal | fcNegNormal | fcNegInf) -> fcmp olt x, 0.0
1176 // is.fpclass(x, fcNegative | fcPosZero) -> fcmp ole x, 0.0
1177 //
1178 if (!IsStrict && (IsOrdered || IsUnordered) &&
1179 (PredType = fpclassTestIsFCmp0(OrderedMask, *II.getFunction(),
1180 Src0->getType())) !=
1183 // Equivalent of == 0.
1184 Value *FCmp = Builder.CreateFCmp(
1185 IsUnordered ? FCmpInst::getUnorderedPredicate(PredType) : PredType,
1186 Src0, Zero);
1187
1188 FCmp->takeName(&II);
1189 return replaceInstUsesWith(II, FCmp);
1190 }
1191
1192 KnownFPClass Known =
1193 computeKnownFPClass(Src0, Mask, SQ.getWithInstruction(&II));
1194
1195 // If none of the tests which can return false are possible, fold to true.
1196 // fp_class (nnan x), ~(qnan|snan) -> true
1197 // fp_class (ninf x), ~(ninf|pinf) -> true
1198 if (Known.isKnownAlways(Mask))
1199 return replaceInstUsesWith(II, ConstantInt::get(II.getType(), true));
1200
1201 // Clear test bits we know must be false from the source value.
1202 // fp_class (nnan x), qnan|snan|other -> fp_class (nnan x), other
1203 // fp_class (ninf x), ninf|pinf|other -> fp_class (ninf x), other
1204 if ((Mask & Known.KnownFPClasses) != Mask) {
1205 II.setArgOperand(
1206 1, ConstantInt::get(Src1->getType(), Mask & Known.KnownFPClasses));
1207 return &II;
1208 }
1209
1210 return nullptr;
1211}
1212
1213static std::optional<bool> getKnownSign(Value *Op, const SimplifyQuery &SQ) {
1215 if (Known.isNonNegative())
1216 return false;
1217 if (Known.isNegative())
1218 return true;
1219
1220 Value *X, *Y;
1221 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1223
1224 return std::nullopt;
1225}
1226
1227static std::optional<bool> getKnownSignOrZero(Value *Op,
1228 const SimplifyQuery &SQ) {
1229 if (std::optional<bool> Sign = getKnownSign(Op, SQ))
1230 return Sign;
1231
1232 Value *X, *Y;
1233 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1235
1236 return std::nullopt;
1237}
1238
1239/// Return true if two values \p Op0 and \p Op1 are known to have the same sign.
1240static bool signBitMustBeTheSame(Value *Op0, Value *Op1,
1241 const SimplifyQuery &SQ) {
1242 std::optional<bool> Known1 = getKnownSign(Op1, SQ);
1243 if (!Known1)
1244 return false;
1245 std::optional<bool> Known0 = getKnownSign(Op0, SQ);
1246 if (!Known0)
1247 return false;
1248 return *Known0 == *Known1;
1249}
1250
1251// Determines if ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b)) is safe.
1252//
1253// This is true if, when the add saturates, the resulting ldexp is guaranteed to
1254// produce 0 or inf.
1255static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy) {
1256 const fltSemantics &FltSem = FpTy->getScalarType()->getFltSemantics();
1257 if (!APFloat::semanticsHasInf(FltSem))
1258 return false;
1259
1260 // Cap ExpBits at 32 because scalbn takes an int. This is sufficient for any
1261 // reasonable fp type (for example, `double` only has 11 exponent bits).
1262 unsigned ExpBits = std::min(ExpTy->getScalarSizeInBits(), 32u);
1263 int SignedMax = static_cast<int>(maxIntN(ExpBits));
1264 int SignedMin = static_cast<int>(minIntN(ExpBits));
1265 APFloat ScaledUp = scalbn(APFloat::getSmallest(FltSem), SignedMax,
1267 APFloat ScaledDown = scalbn(APFloat::getLargest(FltSem), SignedMin,
1269 return ScaledUp.isInfinity() && ScaledDown.isZero();
1270}
1271
1272/// Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0. This
1273/// can trigger other combines.
1275 InstCombiner::BuilderTy &Builder) {
1276 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1277 assert((MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin ||
1278 MinMaxID == Intrinsic::umax || MinMaxID == Intrinsic::umin) &&
1279 "Expected a min or max intrinsic");
1280
1281 // TODO: Match vectors with undef elements, but undef may not propagate.
1282 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
1283 Value *X;
1284 const APInt *C0, *C1;
1285 if (!match(Op0, m_OneUse(m_Add(m_Value(X), m_APInt(C0)))) ||
1286 !match(Op1, m_APInt(C1)))
1287 return nullptr;
1288
1289 // Check for necessary no-wrap and overflow constraints.
1290 bool IsSigned = MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin;
1291 auto *Add = cast<BinaryOperator>(Op0);
1292 if ((IsSigned && !Add->hasNoSignedWrap()) ||
1293 (!IsSigned && !Add->hasNoUnsignedWrap()))
1294 return nullptr;
1295
1296 // If the constant difference overflows, then instsimplify should reduce the
1297 // min/max to the add or C1.
1298 bool Overflow;
1299 APInt CDiff =
1300 IsSigned ? C1->ssub_ov(*C0, Overflow) : C1->usub_ov(*C0, Overflow);
1301 assert(!Overflow && "Expected simplify of min/max");
1302
1303 // min/max (add X, C0), C1 --> add (min/max X, C1 - C0), C0
1304 // Note: the "mismatched" no-overflow setting does not propagate.
1305 Constant *NewMinMaxC = ConstantInt::get(II->getType(), CDiff);
1306 Value *NewMinMax = Builder.CreateBinaryIntrinsic(MinMaxID, X, NewMinMaxC);
1307 return IsSigned ? BinaryOperator::CreateNSWAdd(NewMinMax, Add->getOperand(1))
1308 : BinaryOperator::CreateNUWAdd(NewMinMax, Add->getOperand(1));
1309}
1310/// Match a sadd_sat or ssub_sat which is using min/max to clamp the value.
1311Instruction *InstCombinerImpl::matchSAddSubSat(IntrinsicInst &MinMax1) {
1312 Type *Ty = MinMax1.getType();
1313
1314 // We are looking for a tree of:
1315 // max(INT_MIN, min(INT_MAX, add(sext(A), sext(B))))
1316 // Where the min and max could be reversed
1317 Instruction *MinMax2;
1318 BinaryOperator *AddSub;
1319 const APInt *MinValue, *MaxValue;
1320 if (match(&MinMax1, m_SMin(m_Instruction(MinMax2), m_APInt(MaxValue)))) {
1321 if (!match(MinMax2, m_SMax(m_BinOp(AddSub), m_APInt(MinValue))))
1322 return nullptr;
1323 } else if (match(&MinMax1,
1324 m_SMax(m_Instruction(MinMax2), m_APInt(MinValue)))) {
1325 if (!match(MinMax2, m_SMin(m_BinOp(AddSub), m_APInt(MaxValue))))
1326 return nullptr;
1327 } else
1328 return nullptr;
1329
1330 // Check that the constants clamp a saturate, and that the new type would be
1331 // sensible to convert to.
1332 if (!(*MaxValue + 1).isPowerOf2() || -*MinValue != *MaxValue + 1)
1333 return nullptr;
1334 // In what bitwidth can this be treated as saturating arithmetics?
1335 unsigned NewBitWidth = (*MaxValue + 1).logBase2() + 1;
1336 // FIXME: This isn't quite right for vectors, but using the scalar type is a
1337 // good first approximation for what should be done there.
1338 if (!shouldChangeType(Ty->getScalarType()->getIntegerBitWidth(), NewBitWidth))
1339 return nullptr;
1340
1341 // Also make sure that the inner min/max and the add/sub have one use.
1342 if (!MinMax2->hasOneUse() || !AddSub->hasOneUse())
1343 return nullptr;
1344
1345 // Create the new type (which can be a vector type)
1346 Type *NewTy = Ty->getWithNewBitWidth(NewBitWidth);
1347
1348 Intrinsic::ID IntrinsicID;
1349 if (AddSub->getOpcode() == Instruction::Add)
1350 IntrinsicID = Intrinsic::sadd_sat;
1351 else if (AddSub->getOpcode() == Instruction::Sub)
1352 IntrinsicID = Intrinsic::ssub_sat;
1353 else
1354 return nullptr;
1355
1356 // The two operands of the add/sub must be nsw-truncatable to the NewTy. This
1357 // is usually achieved via a sext from a smaller type.
1358 if (ComputeMaxSignificantBits(AddSub->getOperand(0), AddSub) > NewBitWidth ||
1359 ComputeMaxSignificantBits(AddSub->getOperand(1), AddSub) > NewBitWidth)
1360 return nullptr;
1361
1362 // Finally create and return the sat intrinsic, truncated to the new type
1363 Value *AT = Builder.CreateTrunc(AddSub->getOperand(0), NewTy);
1364 Value *BT = Builder.CreateTrunc(AddSub->getOperand(1), NewTy);
1365 Value *Sat = Builder.CreateIntrinsic(IntrinsicID, NewTy, {AT, BT});
1366 return CastInst::Create(Instruction::SExt, Sat, Ty);
1367}
1368
1369
1370/// If we have a clamp pattern like max (min X, 42), 41 -- where the output
1371/// can only be one of two possible constant values -- turn that into a select
1372/// of constants.
1374 InstCombiner::BuilderTy &Builder) {
1375 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
1376 Value *X;
1377 const APInt *C0, *C1;
1378 if (!match(I1, m_APInt(C1)) || !I0->hasOneUse())
1379 return nullptr;
1380
1382 switch (II->getIntrinsicID()) {
1383 case Intrinsic::smax:
1384 if (match(I0, m_SMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1385 Pred = ICmpInst::ICMP_SGT;
1386 break;
1387 case Intrinsic::smin:
1388 if (match(I0, m_SMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1389 Pred = ICmpInst::ICMP_SLT;
1390 break;
1391 case Intrinsic::umax:
1392 if (match(I0, m_UMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1393 Pred = ICmpInst::ICMP_UGT;
1394 break;
1395 case Intrinsic::umin:
1396 if (match(I0, m_UMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1397 Pred = ICmpInst::ICMP_ULT;
1398 break;
1399 default:
1400 llvm_unreachable("Expected min/max intrinsic");
1401 }
1402 if (Pred == CmpInst::BAD_ICMP_PREDICATE)
1403 return nullptr;
1404
1405 // max (min X, 42), 41 --> X > 41 ? 42 : 41
1406 // min (max X, 42), 43 --> X < 43 ? 42 : 43
1407 Value *Cmp = Builder.CreateICmp(Pred, X, I1);
1408 return SelectInst::Create(Cmp, ConstantInt::get(II->getType(), *C0), I1);
1409}
1410
1411/// If this min/max has a constant operand and an operand that is a matching
1412/// min/max with a constant operand, constant-fold the 2 constant operands.
1414 IRBuilderBase &Builder,
1415 const SimplifyQuery &SQ) {
1416 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1417 auto *LHS = dyn_cast<MinMaxIntrinsic>(II->getArgOperand(0));
1418 if (!LHS)
1419 return nullptr;
1420
1421 Constant *C0, *C1;
1422 if (!match(LHS->getArgOperand(1), m_ImmConstant(C0)) ||
1423 !match(II->getArgOperand(1), m_ImmConstant(C1)))
1424 return nullptr;
1425
1426 // max (max X, C0), C1 --> max X, (max C0, C1)
1427 // min (min X, C0), C1 --> min X, (min C0, C1)
1428 // umax (smax X, nneg C0), nneg C1 --> smax X, (umax C0, C1)
1429 // smin (umin X, nneg C0), nneg C1 --> umin X, (smin C0, C1)
1430 Intrinsic::ID InnerMinMaxID = LHS->getIntrinsicID();
1431 if (InnerMinMaxID != MinMaxID &&
1432 !(((MinMaxID == Intrinsic::umax && InnerMinMaxID == Intrinsic::smax) ||
1433 (MinMaxID == Intrinsic::smin && InnerMinMaxID == Intrinsic::umin)) &&
1434 isKnownNonNegative(C0, SQ) && isKnownNonNegative(C1, SQ)))
1435 return nullptr;
1436
1438 Value *CondC = Builder.CreateICmp(Pred, C0, C1);
1439 Value *NewC = Builder.CreateSelect(CondC, C0, C1);
1440 return Builder.CreateIntrinsic(InnerMinMaxID, II->getType(),
1441 {LHS->getArgOperand(0), NewC});
1442}
1443
1444/// If this min/max has a matching min/max operand with a constant, try to push
1445/// the constant operand into this instruction. This can enable more folds.
1446static Instruction *
1448 InstCombiner::BuilderTy &Builder) {
1449 // Match and capture a min/max operand candidate.
1450 Value *X, *Y;
1451 Constant *C;
1452 Instruction *Inner;
1454 m_Instruction(Inner),
1456 m_Value(Y))))
1457 return nullptr;
1458
1459 // The inner op must match. Check for constants to avoid infinite loops.
1460 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1461 auto *InnerMM = dyn_cast<IntrinsicInst>(Inner);
1462 if (!InnerMM || InnerMM->getIntrinsicID() != MinMaxID ||
1464 return nullptr;
1465
1466 // max (max X, C), Y --> max (max X, Y), C
1468 MinMaxID, II->getType());
1469 Value *NewInner = Builder.CreateBinaryIntrinsic(MinMaxID, X, Y);
1470 NewInner->takeName(Inner);
1471 return CallInst::Create(MinMax, {NewInner, C});
1472}
1473
1474/// Reduce a sequence of min/max intrinsics with a common operand.
1476 // Match 3 of the same min/max ops. Example: umin(umin(), umin()).
1477 auto *LHS = dyn_cast<IntrinsicInst>(II->getArgOperand(0));
1478 auto *RHS = dyn_cast<IntrinsicInst>(II->getArgOperand(1));
1479 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1480 if (!LHS || !RHS || LHS->getIntrinsicID() != MinMaxID ||
1481 RHS->getIntrinsicID() != MinMaxID ||
1482 (!LHS->hasOneUse() && !RHS->hasOneUse()))
1483 return nullptr;
1484
1485 Value *A = LHS->getArgOperand(0);
1486 Value *B = LHS->getArgOperand(1);
1487 Value *C = RHS->getArgOperand(0);
1488 Value *D = RHS->getArgOperand(1);
1489
1490 // Look for a common operand.
1491 Value *MinMaxOp = nullptr;
1492 Value *ThirdOp = nullptr;
1493 if (LHS->hasOneUse()) {
1494 // If the LHS is only used in this chain and the RHS is used outside of it,
1495 // reuse the RHS min/max because that will eliminate the LHS.
1496 if (D == A || C == A) {
1497 // min(min(a, b), min(c, a)) --> min(min(c, a), b)
1498 // min(min(a, b), min(a, d)) --> min(min(a, d), b)
1499 MinMaxOp = RHS;
1500 ThirdOp = B;
1501 } else if (D == B || C == B) {
1502 // min(min(a, b), min(c, b)) --> min(min(c, b), a)
1503 // min(min(a, b), min(b, d)) --> min(min(b, d), a)
1504 MinMaxOp = RHS;
1505 ThirdOp = A;
1506 }
1507 } else {
1508 assert(RHS->hasOneUse() && "Expected one-use operand");
1509 // Reuse the LHS. This will eliminate the RHS.
1510 if (D == A || D == B) {
1511 // min(min(a, b), min(c, a)) --> min(min(a, b), c)
1512 // min(min(a, b), min(c, b)) --> min(min(a, b), c)
1513 MinMaxOp = LHS;
1514 ThirdOp = C;
1515 } else if (C == A || C == B) {
1516 // min(min(a, b), min(b, d)) --> min(min(a, b), d)
1517 // min(min(a, b), min(c, b)) --> min(min(a, b), d)
1518 MinMaxOp = LHS;
1519 ThirdOp = D;
1520 }
1521 }
1522
1523 if (!MinMaxOp || !ThirdOp)
1524 return nullptr;
1525
1526 Module *Mod = II->getModule();
1527 Function *MinMax =
1528 Intrinsic::getOrInsertDeclaration(Mod, MinMaxID, II->getType());
1529 return CallInst::Create(MinMax, { MinMaxOp, ThirdOp });
1530}
1531
1532/// If all arguments of the intrinsic are unary shuffles with the same mask,
1533/// try to shuffle after the intrinsic.
1536 if (!II->getType()->isVectorTy() ||
1537 !isTriviallyVectorizable(II->getIntrinsicID()) ||
1538 !II->getCalledFunction()->isSpeculatable())
1539 return nullptr;
1540
1541 Value *X;
1542 Constant *C;
1543 ArrayRef<int> Mask;
1544 auto *NonConstArg = find_if_not(II->args(), [&II](Use &Arg) {
1545 return isa<Constant>(Arg.get()) ||
1546 isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1547 Arg.getOperandNo(), nullptr);
1548 });
1549 if (!NonConstArg ||
1550 !match(NonConstArg, m_Shuffle(m_Value(X), m_Poison(), m_Mask(Mask))))
1551 return nullptr;
1552
1553 // At least 1 operand must be a shuffle with 1 use because we are creating 2
1554 // instructions.
1555 if (none_of(II->args(), match_fn(m_OneUse(m_Shuffle(m_Value(), m_Value())))))
1556 return nullptr;
1557
1558 // See if all arguments are shuffled with the same mask.
1560 Type *SrcTy = X->getType();
1561 for (Use &Arg : II->args()) {
1562 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1563 Arg.getOperandNo(), nullptr))
1564 NewArgs.push_back(Arg);
1565 else if (match(&Arg,
1566 m_Shuffle(m_Value(X), m_Poison(), m_SpecificMask(Mask))) &&
1567 X->getType() == SrcTy)
1568 NewArgs.push_back(X);
1569 else if (match(&Arg, m_ImmConstant(C))) {
1570 // If it's a constant, try find the constant that would be shuffled to C.
1571 if (Constant *ShuffledC =
1572 unshuffleConstant(Mask, C, cast<VectorType>(SrcTy)))
1573 NewArgs.push_back(ShuffledC);
1574 else
1575 return nullptr;
1576 } else
1577 return nullptr;
1578 }
1579
1580 // intrinsic (shuf X, M), (shuf Y, M), ... --> shuf (intrinsic X, Y, ...), M
1581 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1582 // Result type might be a different vector width.
1583 // TODO: Check that the result type isn't widened?
1584 VectorType *ResTy =
1585 VectorType::get(II->getType()->getScalarType(), cast<VectorType>(SrcTy));
1586 Value *NewIntrinsic =
1587 Builder.CreateIntrinsic(ResTy, II->getIntrinsicID(), NewArgs, FPI);
1588 return new ShuffleVectorInst(NewIntrinsic, Mask);
1589}
1590
1591/// If all arguments of the intrinsic are reverses, try to pull the reverse
1592/// after the intrinsic.
1594 if (!II->getType()->isVectorTy() ||
1595 !isTriviallyVectorizable(II->getIntrinsicID()))
1596 return nullptr;
1597
1598 // At least 1 operand must be a reverse with 1 use because we are creating 2
1599 // instructions.
1600 if (none_of(II->args(), [](Value *V) {
1601 return match(V, m_OneUse(m_VecReverse(m_Value())));
1602 }))
1603 return nullptr;
1604
1605 Value *X;
1606 Constant *C;
1607 SmallVector<Value *> NewArgs;
1608 for (Use &Arg : II->args()) {
1609 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1610 Arg.getOperandNo(), nullptr))
1611 NewArgs.push_back(Arg);
1612 else if (match(&Arg, m_VecReverse(m_Value(X))))
1613 NewArgs.push_back(X);
1614 else if (isSplatValue(Arg))
1615 NewArgs.push_back(Arg);
1616 else if (match(&Arg, m_ImmConstant(C)))
1617 NewArgs.push_back(Builder.CreateVectorReverse(C));
1618 else
1619 return nullptr;
1620 }
1621
1622 // intrinsic (reverse X), (reverse Y), ... --> reverse (intrinsic X, Y, ...)
1623 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1624 Value *NewIntrinsic = Builder.CreateIntrinsic(
1625 II->getType(), II->getIntrinsicID(), NewArgs, FPI);
1626 return Builder.CreateVectorReverse(NewIntrinsic);
1627}
1628
1629/// Fold the following cases and accepts bswap and bitreverse intrinsics:
1630/// bswap(logic_op(bswap(x), y)) --> logic_op(x, bswap(y))
1631/// bswap(logic_op(bswap(x), bswap(y))) --> logic_op(x, y) (ignores multiuse)
1632template <Intrinsic::ID IntrID>
1634 InstCombiner::BuilderTy &Builder) {
1635 static_assert(IntrID == Intrinsic::bswap || IntrID == Intrinsic::bitreverse,
1636 "This helper only supports BSWAP and BITREVERSE intrinsics");
1637
1638 Value *X, *Y;
1639 // Find bitwise logic op. Check that it is a BinaryOperator explicitly so we
1640 // don't match ConstantExpr that aren't meaningful for this transform.
1643 Value *OldReorderX, *OldReorderY;
1645
1646 // If both X and Y are bswap/bitreverse, the transform reduces the number
1647 // of instructions even if there's multiuse.
1648 // If only one operand is bswap/bitreverse, we need to ensure the operand
1649 // have only one use.
1650 if (match(X, m_Intrinsic<IntrID>(m_Value(OldReorderX))) &&
1651 match(Y, m_Intrinsic<IntrID>(m_Value(OldReorderY)))) {
1652 return BinaryOperator::Create(Op, OldReorderX, OldReorderY);
1653 }
1654
1655 if (match(X, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderX))))) {
1656 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, Y);
1657 return BinaryOperator::Create(Op, OldReorderX, NewReorder);
1658 }
1659
1660 if (match(Y, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderY))))) {
1661 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, X);
1662 return BinaryOperator::Create(Op, NewReorder, OldReorderY);
1663 }
1664 }
1665 return nullptr;
1666}
1667
1668/// Helper to match idempotent binary intrinsics, namely, intrinsics where
1669/// `f(f(x, y), y) == f(x, y)` holds.
1671 switch (IID) {
1672 case Intrinsic::smax:
1673 case Intrinsic::smin:
1674 case Intrinsic::umax:
1675 case Intrinsic::umin:
1676 case Intrinsic::maximum:
1677 case Intrinsic::minimum:
1678 case Intrinsic::maximumnum:
1679 case Intrinsic::minimumnum:
1680 case Intrinsic::maxnum:
1681 case Intrinsic::minnum:
1682 return true;
1683 default:
1684 return false;
1685 }
1686}
1687
1688/// Attempt to simplify value-accumulating recurrences of kind:
1689/// %umax.acc = phi i8 [ %umax, %backedge ], [ %a, %entry ]
1690/// %umax = call i8 @llvm.umax.i8(i8 %umax.acc, i8 %b)
1691/// And let the idempotent binary intrinsic be hoisted, when the operands are
1692/// known to be loop-invariant.
1694 IntrinsicInst *II) {
1695 PHINode *PN;
1696 Value *Init, *OtherOp;
1697
1698 // A binary intrinsic recurrence with loop-invariant operands is equivalent to
1699 // `call @llvm.binary.intrinsic(Init, OtherOp)`.
1700 auto IID = II->getIntrinsicID();
1701 if (!isIdempotentBinaryIntrinsic(IID) ||
1703 !IC.getDominatorTree().dominates(OtherOp, PN))
1704 return nullptr;
1705
1706 auto *InvariantBinaryInst =
1707 IC.Builder.CreateBinaryIntrinsic(IID, Init, OtherOp);
1708 if (isa<FPMathOperator>(InvariantBinaryInst))
1709 cast<Instruction>(InvariantBinaryInst)->copyFastMathFlags(II);
1710 return InvariantBinaryInst;
1711}
1712
1713static Value *simplifyReductionOperand(Value *Arg, bool CanReorderLanes) {
1714 if (!CanReorderLanes)
1715 return nullptr;
1716
1717 Value *V;
1718 if (match(Arg, m_VecReverse(m_Value(V))))
1719 return V;
1720
1721 ArrayRef<int> Mask;
1722 if (!isa<FixedVectorType>(Arg->getType()) ||
1723 !match(Arg, m_Shuffle(m_Value(V), m_Undef(), m_Mask(Mask))) ||
1724 !cast<ShuffleVectorInst>(Arg)->isSingleSource())
1725 return nullptr;
1726
1727 int Sz = Mask.size();
1728 SmallBitVector UsedIndices(Sz);
1729 for (int Idx : Mask) {
1730 if (Idx == PoisonMaskElem || UsedIndices.test(Idx))
1731 return nullptr;
1732 UsedIndices.set(Idx);
1733 }
1734
1735 // Can remove shuffle iff just shuffled elements, no repeats, undefs, or
1736 // other changes.
1737 return UsedIndices.all() ? V : nullptr;
1738}
1739
1740/// Fold an unsigned minimum of trailing or leading zero bits counts:
1741/// umin(cttz(CtOp1, ZeroUndef), ConstOp) --> cttz(CtOp1 | (1 << ConstOp))
1742/// umin(ctlz(CtOp1, ZeroUndef), ConstOp) --> ctlz(CtOp1 | (SignedMin
1743/// >> ConstOp))
1744/// umin(cttz(CtOp1), cttz(CtOp2)) --> cttz(CtOp1 | CtOp2)
1745/// umin(ctlz(CtOp1), ctlz(CtOp2)) --> ctlz(CtOp1 | CtOp2)
1746template <Intrinsic::ID IntrID>
1747static Value *
1749 const DataLayout &DL,
1750 InstCombiner::BuilderTy &Builder) {
1751 static_assert(IntrID == Intrinsic::cttz || IntrID == Intrinsic::ctlz,
1752 "This helper only supports cttz and ctlz intrinsics");
1753
1754 Value *CtOp1, *CtOp2;
1755 Value *ZeroUndef1, *ZeroUndef2;
1756 if (!match(I0, m_OneUse(
1757 m_Intrinsic<IntrID>(m_Value(CtOp1), m_Value(ZeroUndef1)))))
1758 return nullptr;
1759
1760 if (match(I1,
1761 m_OneUse(m_Intrinsic<IntrID>(m_Value(CtOp2), m_Value(ZeroUndef2)))))
1762 return Builder.CreateBinaryIntrinsic(
1763 IntrID, Builder.CreateOr(CtOp1, CtOp2),
1764 Builder.CreateOr(ZeroUndef1, ZeroUndef2));
1765
1766 unsigned BitWidth = I1->getType()->getScalarSizeInBits();
1767 auto LessBitWidth = [BitWidth](auto &C) { return C.ult(BitWidth); };
1768 if (!match(I1, m_CheckedInt(LessBitWidth)))
1769 // We have a constant >= BitWidth (which can be handled by CVP)
1770 // or a non-splat vector with elements < and >= BitWidth
1771 return nullptr;
1772
1773 Type *Ty = I1->getType();
1775 IntrID == Intrinsic::cttz ? Instruction::Shl : Instruction::LShr,
1776 IntrID == Intrinsic::cttz
1777 ? ConstantInt::get(Ty, 1)
1778 : ConstantInt::get(Ty, APInt::getSignedMinValue(BitWidth)),
1779 cast<Constant>(I1), DL);
1780 return Builder.CreateBinaryIntrinsic(
1781 IntrID, Builder.CreateOr(CtOp1, NewConst),
1782 ConstantInt::getTrue(ZeroUndef1->getType()));
1783}
1784
1785/// Return whether "X LOp (Y ROp Z)" is always equal to
1786/// "(X LOp Y) ROp (X LOp Z)".
1788 bool HasNSW, Intrinsic::ID ROp) {
1789 switch (ROp) {
1790 case Intrinsic::umax:
1791 case Intrinsic::umin:
1792 if (HasNUW && LOp == Instruction::Add)
1793 return true;
1794 if (HasNUW && LOp == Instruction::Shl)
1795 return true;
1796 return false;
1797 case Intrinsic::smax:
1798 case Intrinsic::smin:
1799 return HasNSW && LOp == Instruction::Add;
1800 default:
1801 return false;
1802 }
1803}
1804
1805/// Return whether "(X ROp Y) LOp Z" is always equal to
1806/// "(X LOp Z) ROp (Y LOp Z)".
1808 bool HasNSW, Intrinsic::ID ROp) {
1809 if (Instruction::isCommutative(LOp) || LOp == Instruction::Shl)
1810 return leftDistributesOverRight(LOp, HasNUW, HasNSW, ROp);
1811 switch (ROp) {
1812 case Intrinsic::umax:
1813 case Intrinsic::umin:
1814 return HasNUW && LOp == Instruction::Sub;
1815 case Intrinsic::smax:
1816 case Intrinsic::smin:
1817 return HasNSW && LOp == Instruction::Sub;
1818 default:
1819 return false;
1820 }
1821}
1822
1823// Attempts to factorise a common term
1824// in an instruction that has the form "(A op' B) op (C op' D)
1825// where op is an intrinsic and op' is a binop
1826static Value *
1828 InstCombiner::BuilderTy &Builder) {
1829 Value *LHS = II->getOperand(0), *RHS = II->getOperand(1);
1830 Intrinsic::ID TopLevelOpcode = II->getIntrinsicID();
1831
1834
1835 if (!Op0 || !Op1)
1836 return nullptr;
1837
1838 if (Op0->getOpcode() != Op1->getOpcode())
1839 return nullptr;
1840
1841 if (!Op0->hasOneUse() || !Op1->hasOneUse())
1842 return nullptr;
1843
1844 Instruction::BinaryOps InnerOpcode =
1845 static_cast<Instruction::BinaryOps>(Op0->getOpcode());
1846 bool HasNUW = Op0->hasNoUnsignedWrap() && Op1->hasNoUnsignedWrap();
1847 bool HasNSW = Op0->hasNoSignedWrap() && Op1->hasNoSignedWrap();
1848
1849 Value *A = Op0->getOperand(0);
1850 Value *B = Op0->getOperand(1);
1851 Value *C = Op1->getOperand(0);
1852 Value *D = Op1->getOperand(1);
1853
1854 // Attempts to swap variables such that A equals C or B equals D,
1855 // if the inner operation is commutative.
1856 if (Op0->isCommutative() && A != C && B != D) {
1857 if (A == D || B == C)
1858 std::swap(C, D);
1859 else
1860 return nullptr;
1861 }
1862
1863 if (A == C &&
1864 leftDistributesOverRight(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1865 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, B, D);
1866 return Builder.CreateNoWrapBinOp(InnerOpcode, A, NewIntrinsic, HasNUW,
1867 HasNSW);
1868 }
1869 if (B == D &&
1870 rightDistributesOverLeft(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1871 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, A, C);
1872 return Builder.CreateNoWrapBinOp(InnerOpcode, NewIntrinsic, B, HasNUW,
1873 HasNSW);
1874 }
1875 return nullptr;
1876}
1877
1879 Value *Arg0 = II->getArgOperand(0);
1880 auto *ShiftConst = dyn_cast<Constant>(II->getArgOperand(1));
1881 if (!ShiftConst)
1882 return nullptr;
1883
1884 int ElemBits = Arg0->getType()->getScalarSizeInBits();
1885 bool AllPositive = true;
1886 bool AllNegative = true;
1887
1888 auto Check = [&](Constant *C) -> bool {
1889 if (auto *CI = dyn_cast_or_null<ConstantInt>(C)) {
1890 const APInt &V = CI->getValue();
1891 if (V.isNonNegative()) {
1892 AllNegative = false;
1893 return AllPositive && V.ult(ElemBits);
1894 }
1895 AllPositive = false;
1896 return AllNegative && V.sgt(-ElemBits);
1897 }
1898 return false;
1899 };
1900
1901 if (auto *VTy = dyn_cast<FixedVectorType>(Arg0->getType())) {
1902 for (unsigned I = 0, E = VTy->getNumElements(); I < E; ++I) {
1903 if (!Check(ShiftConst->getAggregateElement(I)))
1904 return nullptr;
1905 }
1906
1907 } else if (!Check(ShiftConst))
1908 return nullptr;
1909
1910 IRBuilderBase &B = IC.Builder;
1911 if (AllPositive)
1912 return IC.replaceInstUsesWith(*II, B.CreateShl(Arg0, ShiftConst));
1913
1914 Value *NegAmt = B.CreateNeg(ShiftConst);
1915 Intrinsic::ID IID = II->getIntrinsicID();
1916 const bool IsSigned =
1917 IID == Intrinsic::arm_neon_vshifts || IID == Intrinsic::aarch64_neon_sshl;
1918 Value *Result =
1919 IsSigned ? B.CreateAShr(Arg0, NegAmt) : B.CreateLShr(Arg0, NegAmt);
1920 return IC.replaceInstUsesWith(*II, Result);
1921}
1922
1923// If II is llvm.sin(x) or llvm.cos(x), and there is a matching
1924// llvm.cos(x) or llvm.sin(x) using the same argument, combine them
1925// into a single llvm.sincos(x) call. Returns the result for II
1926// extracted from sincos, or nullptr if no match is found.
1928 InstCombinerImpl &IC) {
1929 Intrinsic::ID IID = II->getIntrinsicID();
1930 bool IsSin = IID == Intrinsic::sin;
1931 Intrinsic::ID MatchID = IsSin ? Intrinsic::cos : Intrinsic::sin;
1932
1933 Value *Arg = II->getArgOperand(0);
1934
1935 // Don't bother looking through uses of constants.
1936 if (isa<Constant>(Arg))
1937 return nullptr;
1938
1939 // Look for a matching cos/sin intrinsic with the same argument.
1940 IntrinsicInst *Match = nullptr;
1941 for (User *U : Arg->users()) {
1942 if (auto *Cand = dyn_cast<IntrinsicInst>(U)) {
1943 if (Cand != II && !Cand->use_empty() &&
1944 Cand->getIntrinsicID() == MatchID) {
1945 Match = Cand;
1946 break;
1947 }
1948 }
1949 }
1950
1951 if (!Match)
1952 return nullptr;
1953
1954 // Insert sincos right after the argument definition.
1956 if (auto *ArgInst = dyn_cast<Instruction>(Arg)) {
1957 std::optional<BasicBlock::iterator> InsertPt =
1958 ArgInst->getInsertionPointAfterDef();
1959 if (!InsertPt)
1960 return nullptr;
1961 B.SetInsertPoint(*InsertPt);
1962 } else {
1963 BasicBlock &EntryBB = II->getFunction()->getEntryBlock();
1964 B.SetInsertPoint(&EntryBB, EntryBB.begin());
1965 }
1966
1968 II->getModule(), Intrinsic::sincos, Arg->getType());
1969 CallInst *SinCos = B.CreateCall(SinCosFunc, Arg, "sincos");
1970 // Intersect fast-math flags from the two calls.
1971 SinCos->setFastMathFlags(II->getFastMathFlags() & Match->getFastMathFlags());
1972 // Propagate the most-generic fpmath metadata from the two original calls.
1974 II->getMetadata(LLVMContext::MD_fpmath),
1975 Match->getMetadata(LLVMContext::MD_fpmath)))
1976 SinCos->setMetadata(LLVMContext::MD_fpmath, MD);
1977 Value *Sin = B.CreateExtractValue(SinCos, 0, "sin");
1978 Value *Cos = B.CreateExtractValue(SinCos, 1, "cos");
1979
1980 // Replace the matching call and erase it.
1981 IC.replaceInstUsesWith(*Match, IsSin ? Cos : Sin);
1982 IC.eraseInstFromFunction(*Match);
1983 return IsSin ? Sin : Cos;
1984}
1985
1986/// Fold an scmp/ucmp intrinsic whose operands are extended from a narrower
1987/// type:
1988/// scmp (sext X), (sext Y) --> scmp X, Y
1989/// scmp (zext X), (zext Y) --> ucmp X, Y
1990/// ucmp (ext X), (ext Y) --> ucmp X, Y
1991/// Both operands must use the same extend opcode and source type. A constant
1992/// operand is narrowed instead, if truncating and re-extending it gives back
1993/// the same constant.
1995 InstCombiner::BuilderTy &Builder,
1996 const DataLayout &DL) {
1997 // scmp/ucmp are not commutative, so the extend may be on either side.
1998 unsigned ExtIdx = 0;
1999 Value *X;
2000 if (!match(II->getArgOperand(0), m_ZExtOrSExt(m_Value(X)))) {
2001 ExtIdx = 1;
2002 if (!match(II->getArgOperand(1), m_ZExtOrSExt(m_Value(X))))
2003 return nullptr;
2004 }
2005
2006 auto CastOpc = static_cast<Instruction::CastOps>(
2007 cast<Operator>(II->getArgOperand(ExtIdx))->getOpcode());
2008 Type *NarrowTy = X->getType();
2009
2010 // The other operand must be the same kind of extend from the same type, or a
2011 // constant that can be narrowed losslessly.
2012 Value *OtherOp = II->getArgOperand(1 - ExtIdx);
2013 Value *Y;
2014 Constant *WideC;
2015 if (match(OtherOp, m_ZExtOrSExt(m_Value(Y)))) {
2016 if (cast<Operator>(OtherOp)->getOpcode() != CastOpc ||
2017 Y->getType() != NarrowTy)
2018 return nullptr;
2019 } else if (match(OtherOp, m_ImmConstant(WideC))) {
2020 Y = getLosslessInvCast(WideC, NarrowTy, CastOpc, DL);
2021 if (!Y)
2022 return nullptr;
2023 } else {
2024 return nullptr;
2025 }
2026
2027 // Both extends preserve the unsigned order, so an unsigned compare of the
2028 // narrow operands is always equivalent. The signed order is only preserved by
2029 // sext; zero extended values are non-negative, so a signed compare of those
2030 // is an unsigned compare of the narrow operands.
2031 Intrinsic::ID NewIID =
2032 II->getIntrinsicID() == Intrinsic::scmp && CastOpc == Instruction::SExt
2033 ? Intrinsic::scmp
2034 : Intrinsic::ucmp;
2035 if (ExtIdx != 0)
2036 std::swap(X, Y);
2037 return Builder.CreateIntrinsic(II->getType(), NewIID, {X, Y});
2038}
2039
2040/// CallInst simplification. This mostly only handles folding of intrinsic
2041/// instructions. For normal calls, it allows visitCallBase to do the heavy
2042/// lifting.
2044 // Don't try to simplify calls without uses. It will not do anything useful,
2045 // but will result in the following folds being skipped.
2046 if (!CI.use_empty()) {
2047 SmallVector<Value *, 8> Args(CI.args());
2048 if (Value *V = simplifyCall(&CI, CI.getCalledOperand(), Args,
2049 SQ.getWithInstruction(&CI)))
2050 return replaceInstUsesWith(CI, V);
2051 }
2052
2053 if (Value *FreedOp = getFreedOperand(&CI, &TLI))
2054 return visitFree(CI, FreedOp);
2055
2056 // If the caller function (i.e. us, the function that contains this CallInst)
2057 // is nounwind, mark the call as nounwind, even if the callee isn't.
2058 if (CI.getFunction()->doesNotThrow() && !CI.doesNotThrow()) {
2059 CI.setDoesNotThrow();
2060 return &CI;
2061 }
2062
2064 if (!II)
2065 return visitCallBase(CI);
2066
2067 // Intrinsics cannot occur in an invoke or a callbr, so handle them here
2068 // instead of in visitCallBase.
2069 if (auto *MI = dyn_cast<AnyMemIntrinsic>(II)) {
2070 if (auto NumBytes = MI->getLengthInBytes()) {
2071 // memmove/cpy/set of zero bytes is a noop.
2072 if (NumBytes->isZero())
2073 return eraseInstFromFunction(CI);
2074
2075 // For atomic unordered mem intrinsics if len is not a positive or
2076 // not a multiple of element size then behavior is undefined.
2077 if (MI->isAtomic() &&
2078 (NumBytes->isNegative() ||
2079 (NumBytes->getZExtValue() % MI->getElementSizeInBytes() != 0))) {
2081 assert(MI->getType()->isVoidTy() &&
2082 "non void atomic unordered mem intrinsic");
2083 return eraseInstFromFunction(*MI);
2084 }
2085 }
2086
2087 // No other transformations apply to volatile transfers.
2088 if (MI->isVolatile())
2089 return nullptr;
2090
2092 // memmove(x,x,size) -> noop.
2093 if (MTI->getSource() == MTI->getDest())
2094 return eraseInstFromFunction(CI);
2095 }
2096
2097 auto IsPointerUndefined = [MI](Value *Ptr) {
2098 return isa<ConstantPointerNull>(Ptr) &&
2100 MI->getFunction(),
2101 cast<PointerType>(Ptr->getType())->getAddressSpace());
2102 };
2103 bool SrcIsUndefined = false;
2104 // If we can determine a pointer alignment that is bigger than currently
2105 // set, update the alignment.
2106 if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
2108 return I;
2109 SrcIsUndefined = IsPointerUndefined(MTI->getRawSource());
2110 } else if (auto *MSI = dyn_cast<AnyMemSetInst>(MI)) {
2111 if (Instruction *I = SimplifyAnyMemSet(MSI))
2112 return I;
2113 }
2114
2115 // If src/dest is null, this memory intrinsic must be a noop.
2116 if (SrcIsUndefined || IsPointerUndefined(MI->getRawDest())) {
2117 Builder.CreateAssumption(Builder.CreateIsNull(MI->getLength()));
2118 return eraseInstFromFunction(CI);
2119 }
2120
2121 // If we have a memmove and the source operation is a constant global,
2122 // then the source and dest pointers can't alias, so we can change this
2123 // into a call to memcpy.
2124 if (auto *MMI = dyn_cast<AnyMemMoveInst>(MI)) {
2125 if (GlobalVariable *GVSrc = dyn_cast<GlobalVariable>(MMI->getSource()))
2126 if (GVSrc->isConstant()) {
2127 Module *M = CI.getModule();
2128 Intrinsic::ID MemCpyID =
2129 MMI->isAtomic()
2130 ? Intrinsic::memcpy_element_unordered_atomic
2131 : Intrinsic::memcpy;
2132 Type *Tys[3] = { CI.getArgOperand(0)->getType(),
2133 CI.getArgOperand(1)->getType(),
2134 CI.getArgOperand(2)->getType() };
2136 Intrinsic::getOrInsertDeclaration(M, MemCpyID, Tys));
2137 return II;
2138 }
2139 }
2140 }
2141
2142 // For fixed width vector result intrinsics, use the generic demanded vector
2143 // support.
2144 if (auto *IIFVTy = dyn_cast<FixedVectorType>(II->getType())) {
2145 auto VWidth = IIFVTy->getNumElements();
2146 APInt PoisonElts(VWidth, 0);
2147 APInt AllOnesEltMask(APInt::getAllOnes(VWidth));
2148 if (Value *V = SimplifyDemandedVectorElts(II, AllOnesEltMask, PoisonElts)) {
2149 if (V != II)
2150 return replaceInstUsesWith(*II, V);
2151 return II;
2152 }
2153 }
2154
2155 if (II->isCommutative()) {
2156 if (auto Pair = matchSymmetricPair(II->getOperand(0), II->getOperand(1))) {
2157 replaceOperand(*II, 0, Pair->first);
2158 replaceOperand(*II, 1, Pair->second);
2159 II->dropPoisonGeneratingAnnotations();
2160 II->dropUBImplyingAttrsAndMetadata();
2161 return II;
2162 }
2163
2164 if (CallInst *NewCall = canonicalizeConstantArg0ToArg1(CI))
2165 return NewCall;
2166 }
2167
2168 // Unused constrained FP intrinsic calls may have declared side effect, which
2169 // prevents it from being removed. In some cases however the side effect is
2170 // actually absent. To detect this case, call SimplifyConstrainedFPCall. If it
2171 // returns a replacement, the call may be removed.
2172 if (CI.use_empty() && isa<ConstrainedFPIntrinsic>(CI)) {
2173 if (simplifyConstrainedFPCall(&CI, SQ.getWithInstruction(&CI)))
2174 return eraseInstFromFunction(CI);
2175 }
2176
2177 Intrinsic::ID IID = II->getIntrinsicID();
2178 switch (IID) {
2179 case Intrinsic::objectsize: {
2180 SmallVector<Instruction *> InsertedInstructions;
2181 if (Value *V = lowerObjectSizeCall(II, DL, &TLI, AA, /*MustSucceed=*/false,
2182 &InsertedInstructions)) {
2183 for (Instruction *Inserted : InsertedInstructions)
2184 Worklist.add(Inserted);
2185 return replaceInstUsesWith(CI, V);
2186 }
2187 return nullptr;
2188 }
2189 case Intrinsic::abs: {
2190 Value *IIOperand = II->getArgOperand(0);
2191 bool IntMinIsPoison = cast<Constant>(II->getArgOperand(1))->isOneValue();
2192
2193 // abs(-x) -> abs(x)
2194 Value *X;
2195 if (match(IIOperand, m_Neg(m_Value(X))))
2196 return CallInst::Create(
2197 II->getCalledFunction(),
2198 {X,
2199 Builder.getInt1(IntMinIsPoison ||
2200 cast<Instruction>(IIOperand)->hasNoSignedWrap())});
2201
2202 if (match(IIOperand, m_c_Select(m_Neg(m_Value(X)), m_Deferred(X))))
2203 return CallInst::Create(II->getCalledFunction(),
2204 {X, II->getArgOperand(1)});
2205
2206 Value *Y;
2207 // abs(a * abs(b)) -> abs(a * b)
2208 if (match(IIOperand,
2211 bool NSW =
2212 cast<Instruction>(IIOperand)->hasNoSignedWrap() && IntMinIsPoison;
2213 auto *XY = NSW ? Builder.CreateNSWMul(X, Y) : Builder.CreateMul(X, Y);
2214 return CallInst::Create(II->getCalledFunction(),
2215 {XY, II->getArgOperand(1)});
2216 }
2217
2218 if (std::optional<bool> Known =
2219 getKnownSignOrZero(IIOperand, SQ.getWithInstruction(II))) {
2220 // abs(x) -> x if x >= 0 (include abs(x-y) --> x - y where x >= y)
2221 // abs(x) -> x if x > 0 (include abs(x-y) --> x - y where x > y)
2222 if (!*Known)
2223 return replaceInstUsesWith(*II, IIOperand);
2224
2225 // abs(x) -> -x if x < 0
2226 // abs(x) -> -x if x < = 0 (include abs(x-y) --> y - x where x <= y)
2227 if (IntMinIsPoison)
2228 return BinaryOperator::CreateNSWNeg(IIOperand);
2229 return BinaryOperator::CreateNeg(IIOperand);
2230 }
2231
2232 // abs (sext X) --> zext (abs X*)
2233 // Clear the IsIntMin (nsw) bit on the abs to allow narrowing.
2234 if (match(IIOperand, m_OneUse(m_SExt(m_Value(X))))) {
2235 Value *NarrowAbs =
2236 Builder.CreateBinaryIntrinsic(Intrinsic::abs, X, Builder.getFalse());
2237 return CastInst::Create(Instruction::ZExt, NarrowAbs, II->getType());
2238 }
2239
2240 // Match a complicated way to check if a number is odd/even:
2241 // abs (srem X, 2) --> and X, 1
2242 const APInt *C;
2243 if (match(IIOperand, m_SRem(m_Value(X), m_APInt(C))) && *C == 2)
2244 return BinaryOperator::CreateAnd(X, ConstantInt::get(II->getType(), 1));
2245
2246 break;
2247 }
2248 case Intrinsic::umin: {
2249 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2250 // umin(x, 1) == zext(x != 0)
2251 if (match(I1, m_One())) {
2252 assert(II->getType()->getScalarSizeInBits() != 1 &&
2253 "Expected simplify of umin with max constant");
2254 Value *Zero = Constant::getNullValue(I0->getType());
2255 Value *Cmp = Builder.CreateICmpNE(I0, Zero);
2256 return CastInst::Create(Instruction::ZExt, Cmp, II->getType());
2257 }
2258 // umin(cttz(x), const) --> cttz(x | (1 << const))
2259 if (Value *FoldedCttz =
2261 I0, I1, DL, Builder))
2262 return replaceInstUsesWith(*II, FoldedCttz);
2263 // umin(ctlz(x), const) --> ctlz(x | (SignedMin >> const))
2264 if (Value *FoldedCtlz =
2266 I0, I1, DL, Builder))
2267 return replaceInstUsesWith(*II, FoldedCtlz);
2268 [[fallthrough]];
2269 }
2270 case Intrinsic::umax: {
2271 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2272 Value *X, *Y;
2273 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_ZExt(m_Value(Y))) &&
2274 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2275 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2276 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2277 }
2278 Constant *C;
2279 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2280 I0->hasOneUse()) {
2281 if (Constant *NarrowC = getLosslessUnsignedTrunc(C, X->getType(), DL)) {
2282 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2283 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2284 }
2285 }
2286 // If C is not 0:
2287 // umax(nuw_shl(x, C), x + 1) -> x == 0 ? 1 : nuw_shl(x, C)
2288 // If C is not 0 or 1:
2289 // umax(nuw_mul(x, C), x + 1) -> x == 0 ? 1 : nuw_mul(x, C)
2290 auto foldMaxMulShift = [&](Value *A, Value *B) -> Instruction * {
2291 const APInt *C;
2292 Value *X;
2293 if (!match(A, m_NUWShl(m_Value(X), m_APInt(C))) &&
2294 !(match(A, m_NUWMul(m_Value(X), m_APInt(C))) && !C->isOne()))
2295 return nullptr;
2296 if (C->isZero())
2297 return nullptr;
2298 if (!match(B, m_OneUse(m_Add(m_Specific(X), m_One()))))
2299 return nullptr;
2300
2301 Value *Cmp = Builder.CreateICmpEQ(X, ConstantInt::get(X->getType(), 0));
2302 Value *NewSelect = nullptr;
2303 NewSelect = Builder.CreateSelectWithUnknownProfile(
2304 Cmp, ConstantInt::get(X->getType(), 1), A, DEBUG_TYPE);
2305 return replaceInstUsesWith(*II, NewSelect);
2306 };
2307
2308 if (IID == Intrinsic::umax) {
2309 if (Instruction *I = foldMaxMulShift(I0, I1))
2310 return I;
2311 if (Instruction *I = foldMaxMulShift(I1, I0))
2312 return I;
2313 }
2314
2315 // If both operands of unsigned min/max are sign-extended, it is still ok
2316 // to narrow the operation.
2317 [[fallthrough]];
2318 }
2319 case Intrinsic::smax:
2320 case Intrinsic::smin: {
2321 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2322 Value *X, *Y;
2323 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_SExt(m_Value(Y))) &&
2324 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2325 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2326 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2327 }
2328
2329 Constant *C;
2330 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2331 I0->hasOneUse()) {
2332 if (Constant *NarrowC = getLosslessSignedTrunc(C, X->getType(), DL)) {
2333 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2334 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2335 }
2336 }
2337
2338 // smax(smin(X, MinC), MaxC) -> smin(smax(X, MaxC), MinC) if MinC s>= MaxC
2339 // umax(umin(X, MinC), MaxC) -> umin(umax(X, MaxC), MinC) if MinC u>= MaxC
2340 const APInt *MinC, *MaxC;
2341 auto CreateCanonicalClampForm = [&](bool IsSigned) {
2342 auto MaxIID = IsSigned ? Intrinsic::smax : Intrinsic::umax;
2343 auto MinIID = IsSigned ? Intrinsic::smin : Intrinsic::umin;
2344 Value *NewMax = Builder.CreateBinaryIntrinsic(
2345 MaxIID, X, ConstantInt::get(X->getType(), *MaxC));
2346 return replaceInstUsesWith(
2347 *II, Builder.CreateBinaryIntrinsic(
2348 MinIID, NewMax, ConstantInt::get(X->getType(), *MinC)));
2349 };
2350 if (IID == Intrinsic::smax &&
2352 m_APInt(MinC)))) &&
2353 match(I1, m_APInt(MaxC)) && MinC->sgt(*MaxC))
2354 return CreateCanonicalClampForm(true);
2355 if (IID == Intrinsic::umax &&
2357 m_APInt(MinC)))) &&
2358 match(I1, m_APInt(MaxC)) && MinC->ugt(*MaxC))
2359 return CreateCanonicalClampForm(false);
2360
2361 // umin(i1 X, i1 Y) -> and i1 X, Y
2362 // smax(i1 X, i1 Y) -> and i1 X, Y
2363 if ((IID == Intrinsic::umin || IID == Intrinsic::smax) &&
2364 II->getType()->isIntOrIntVectorTy(1)) {
2365 return BinaryOperator::CreateAnd(I0, I1);
2366 }
2367
2368 // umax(i1 X, i1 Y) -> or i1 X, Y
2369 // smin(i1 X, i1 Y) -> or i1 X, Y
2370 if ((IID == Intrinsic::umax || IID == Intrinsic::smin) &&
2371 II->getType()->isIntOrIntVectorTy(1)) {
2372 return BinaryOperator::CreateOr(I0, I1);
2373 }
2374
2375 // smin(smax(X, -1), 1) -> scmp(X, 0)
2376 // smax(smin(X, 1), -1) -> scmp(X, 0)
2377 // At this point, smax(smin(X, 1), -1) is changed to smin(smax(X, -1)
2378 // And i1's have been changed to and/ors
2379 // So we only need to check for smin
2380 if (IID == Intrinsic::smin) {
2381 if (match(I0, m_OneUse(m_SMax(m_Value(X), m_AllOnes()))) &&
2382 match(I1, m_One())) {
2383 Value *Zero = ConstantInt::get(X->getType(), 0);
2384 return replaceInstUsesWith(
2385 CI,
2386 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {X, Zero}));
2387 }
2388 }
2389
2390 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2391 // smax (neg nsw X), (neg nsw Y) --> neg nsw (smin X, Y)
2392 // smin (neg nsw X), (neg nsw Y) --> neg nsw (smax X, Y)
2393 // TODO: Canonicalize neg after min/max if I1 is constant.
2394 if (match(I0, m_NSWNeg(m_Value(X))) && match(I1, m_NSWNeg(m_Value(Y))) &&
2395 (I0->hasOneUse() || I1->hasOneUse())) {
2397 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, X, Y);
2398 return BinaryOperator::CreateNSWNeg(InvMaxMin);
2399 }
2400 }
2401
2402 // (umax X, (xor X, Pow2))
2403 // -> (or X, Pow2)
2404 // (umin X, (xor X, Pow2))
2405 // -> (and X, ~Pow2)
2406 // (smax X, (xor X, Pos_Pow2))
2407 // -> (or X, Pos_Pow2)
2408 // (smin X, (xor X, Pos_Pow2))
2409 // -> (and X, ~Pos_Pow2)
2410 // (smax X, (xor X, Neg_Pow2))
2411 // -> (and X, ~Neg_Pow2)
2412 // (smin X, (xor X, Neg_Pow2))
2413 // -> (or X, Neg_Pow2)
2414 if ((match(I0, m_c_Xor(m_Specific(I1), m_Value(X))) ||
2415 match(I1, m_c_Xor(m_Specific(I0), m_Value(X)))) &&
2416 isKnownToBeAPowerOfTwo(X, /* OrZero */ true)) {
2417 bool UseOr = IID == Intrinsic::smax || IID == Intrinsic::umax;
2418 bool UseAndN = IID == Intrinsic::smin || IID == Intrinsic::umin;
2419
2420 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2421 auto KnownSign = getKnownSign(X, SQ.getWithInstruction(II));
2422 if (KnownSign == std::nullopt) {
2423 UseOr = false;
2424 UseAndN = false;
2425 } else if (*KnownSign /* true is Signed. */) {
2426 UseOr ^= true;
2427 UseAndN ^= true;
2428 Type *Ty = I0->getType();
2429 // Negative power of 2 must be IntMin. It's possible to be able to
2430 // prove negative / power of 2 without actually having known bits, so
2431 // just get the value by hand.
2433 Ty, APInt::getSignedMinValue(Ty->getScalarSizeInBits()));
2434 }
2435 }
2436 if (UseOr)
2437 return BinaryOperator::CreateOr(I0, X);
2438 else if (UseAndN)
2439 return BinaryOperator::CreateAnd(I0, Builder.CreateNot(X));
2440 }
2441
2442 // If we can eliminate ~A and Y is free to invert:
2443 // max ~A, Y --> ~(min A, ~Y)
2444 //
2445 // Examples:
2446 // max ~A, ~Y --> ~(min A, Y)
2447 // max ~A, C --> ~(min A, ~C)
2448 // max ~A, (max ~Y, ~Z) --> ~min( A, (min Y, Z))
2449 auto moveNotAfterMinMax = [&](Value *X, Value *Y) -> Instruction * {
2450 Value *A;
2451 if (match(X, m_OneUse(m_Not(m_Value(A)))) &&
2452 !isFreeToInvert(A, A->hasOneUse())) {
2453 if (Value *NotY = getFreelyInverted(Y, Y->hasOneUse(), &Builder)) {
2455 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, A, NotY);
2456 return BinaryOperator::CreateNot(InvMaxMin);
2457 }
2458 }
2459 return nullptr;
2460 };
2461
2462 if (Instruction *I = moveNotAfterMinMax(I0, I1))
2463 return I;
2464 if (Instruction *I = moveNotAfterMinMax(I1, I0))
2465 return I;
2466
2468 return I;
2469
2470 // minmax (X & NegPow2C, Y & NegPow2C) --> minmax(X, Y) & NegPow2C
2471 const APInt *RHSC;
2472 if (match(I0, m_OneUse(m_And(m_Value(X), m_NegatedPower2(RHSC)))) &&
2473 match(I1, m_OneUse(m_And(m_Value(Y), m_SpecificInt(*RHSC)))))
2474 return BinaryOperator::CreateAnd(Builder.CreateBinaryIntrinsic(IID, X, Y),
2475 ConstantInt::get(II->getType(), *RHSC));
2476
2477 // smax(X, -X) --> abs(X)
2478 // smin(X, -X) --> -abs(X)
2479 // umax(X, -X) --> -abs(X)
2480 // umin(X, -X) --> abs(X)
2481 if (isKnownNegation(I0, I1)) {
2482 // We can choose either operand as the input to abs(), but if we can
2483 // eliminate the only use of a value, that's better for subsequent
2484 // transforms/analysis.
2485 if (I0->hasOneUse() && !I1->hasOneUse())
2486 std::swap(I0, I1);
2487
2488 // This is some variant of abs(). See if we can propagate 'nsw' to the abs
2489 // operation and potentially its negation.
2490 bool IntMinIsPoison = isKnownNegation(I0, I1, /* NeedNSW */ true);
2491 Value *Abs = Builder.CreateBinaryIntrinsic(
2492 Intrinsic::abs, I0,
2493 ConstantInt::getBool(II->getContext(), IntMinIsPoison));
2494
2495 // We don't have a "nabs" intrinsic, so negate if needed based on the
2496 // max/min operation.
2497 if (IID == Intrinsic::smin || IID == Intrinsic::umax)
2498 Abs = Builder.CreateNeg(Abs, "nabs", IntMinIsPoison);
2499 return replaceInstUsesWith(CI, Abs);
2500 }
2501
2503 return Sel;
2504
2505 if (Instruction *SAdd = matchSAddSubSat(*II))
2506 return SAdd;
2507
2508 if (Value *NewMinMax = reassociateMinMaxWithConstants(II, Builder, SQ))
2509 return replaceInstUsesWith(*II, NewMinMax);
2510
2512 return R;
2513
2514 if (Instruction *NewMinMax = factorizeMinMaxTree(II))
2515 return NewMinMax;
2516
2517 // Try to fold minmax with constant RHS based on range information
2518 if (match(I1, m_APIntAllowPoison(RHSC))) {
2519 ICmpInst::Predicate Pred =
2521 bool IsSigned = MinMaxIntrinsic::isSigned(IID);
2523 I0, IsSigned, SQ.getWithInstruction(II));
2524 if (!LHS_CR.isFullSet()) {
2525 if (LHS_CR.icmp(Pred, *RHSC))
2526 return replaceInstUsesWith(*II, I0);
2527 if (LHS_CR.icmp(ICmpInst::getSwappedPredicate(Pred), *RHSC))
2528 return replaceInstUsesWith(*II,
2529 ConstantInt::get(II->getType(), *RHSC));
2530 }
2531 }
2532
2534 return replaceInstUsesWith(*II, V);
2535
2536 break;
2537 }
2538 case Intrinsic::scmp:
2539 case Intrinsic::ucmp: {
2541 return replaceInstUsesWith(CI, V);
2542
2543 if (IID == Intrinsic::ucmp)
2544 break;
2545
2546 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2547
2548 // scmp(X, 0) -> sext_or_trunc(X) if X is known to be one of -1, 0, 1.
2549 if (match(I1, m_Zero())) {
2550 ConstantRange Range = computeConstantRange(I0, /*ForSigned=*/true,
2551 SQ.getWithInstruction(II));
2552 if (Range.getSignedMin().sge(-1) && Range.getSignedMax().sle(1))
2553 return replaceInstUsesWith(
2554 CI, Builder.CreateSExtOrTrunc(I0, II->getType()));
2555 }
2556 Value *LHS, *RHS;
2557 if (match(I0, m_NSWSub(m_Value(LHS), m_Value(RHS))) && match(I1, m_Zero()))
2558 return replaceInstUsesWith(
2559 CI,
2560 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {LHS, RHS}));
2561 break;
2562 }
2563 case Intrinsic::bitreverse: {
2564 Value *IIOperand = II->getArgOperand(0);
2565 // bitrev (zext i1 X to ?) --> X ? SignBitC : 0
2566 Value *X;
2567 if (match(IIOperand, m_ZExt(m_Value(X))) &&
2568 X->getType()->isIntOrIntVectorTy(1)) {
2569 Type *Ty = II->getType();
2570 APInt SignBit = APInt::getSignMask(Ty->getScalarSizeInBits());
2571 return SelectInst::Create(X, ConstantInt::get(Ty, SignBit),
2573 }
2574
2575 if (Instruction *crossLogicOpFold =
2577 return crossLogicOpFold;
2578
2579 break;
2580 }
2581 case Intrinsic::bswap: {
2582 Value *IIOperand = II->getArgOperand(0);
2583
2584 // Try to canonicalize bswap-of-logical-shift-by-8-bit-multiple as
2585 // inverse-shift-of-bswap:
2586 // bswap (shl X, Y) --> lshr (bswap X), Y
2587 // bswap (lshr X, Y) --> shl (bswap X), Y
2588 Value *X, *Y;
2589 if (match(IIOperand, m_OneUse(m_LogicalShift(m_Value(X), m_Value(Y))))) {
2590 unsigned BitWidth = IIOperand->getType()->getScalarSizeInBits();
2592 Value *NewSwap = Builder.CreateUnaryIntrinsic(Intrinsic::bswap, X);
2593 BinaryOperator::BinaryOps InverseShift =
2594 cast<BinaryOperator>(IIOperand)->getOpcode() == Instruction::Shl
2595 ? Instruction::LShr
2596 : Instruction::Shl;
2597 return BinaryOperator::Create(InverseShift, NewSwap, Y);
2598 }
2599 }
2600
2601 KnownBits Known = computeKnownBits(IIOperand, II);
2602 uint64_t LZ = alignDown(Known.countMinLeadingZeros(), 8);
2603 uint64_t TZ = alignDown(Known.countMinTrailingZeros(), 8);
2604 unsigned BW = Known.getBitWidth();
2605
2606 // bswap(x) -> shift(x) if x has exactly one "active byte"
2607 if (BW - LZ - TZ == 8) {
2608 assert(LZ != TZ && "active byte cannot be in the middle");
2609 if (LZ > TZ) // -> shl(x) if the "active byte" is in the low part of x
2610 return BinaryOperator::CreateNUWShl(
2611 IIOperand, ConstantInt::get(IIOperand->getType(), LZ - TZ));
2612 // -> lshr(x) if the "active byte" is in the high part of x
2613 return BinaryOperator::CreateExactLShr(
2614 IIOperand, ConstantInt::get(IIOperand->getType(), TZ - LZ));
2615 }
2616
2617 // bswap(trunc(bswap(x))) -> trunc(lshr(x, c))
2618 if (match(IIOperand, m_Trunc(m_BSwap(m_Value(X))))) {
2619 unsigned C = X->getType()->getScalarSizeInBits() - BW;
2620 Value *CV = ConstantInt::get(X->getType(), C);
2621 Value *V = Builder.CreateLShr(X, CV);
2622 return new TruncInst(V, IIOperand->getType());
2623 }
2624
2625 if (Instruction *crossLogicOpFold =
2627 return crossLogicOpFold;
2628 }
2629
2630 // Try to fold into bitreverse if bswap is the root of the expression tree.
2631 if (Instruction *BitOp = matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ false,
2632 /*MatchBitReversals*/ true))
2633 return BitOp;
2634 break;
2635 }
2636 case Intrinsic::masked_load:
2637 if (Value *SimplifiedMaskedOp = simplifyMaskedLoad(*II))
2638 return replaceInstUsesWith(CI, SimplifiedMaskedOp);
2639 break;
2640 case Intrinsic::masked_store:
2641 return simplifyMaskedStore(*II);
2642 case Intrinsic::masked_gather:
2643 return simplifyMaskedGather(*II);
2644 case Intrinsic::masked_scatter:
2645 return simplifyMaskedScatter(*II);
2646 case Intrinsic::launder_invariant_group:
2647 case Intrinsic::strip_invariant_group:
2648 if (auto *SkippedBarrier = simplifyInvariantGroupIntrinsic(*II, *this))
2649 return replaceInstUsesWith(*II, SkippedBarrier);
2650 break;
2651 case Intrinsic::powi: {
2652 if (ConstantInt *Power = dyn_cast<ConstantInt>(II->getArgOperand(1))) {
2653 // 0 and 1 are handled in instsimplify
2654 // powi(x, -1) -> 1/x
2655 if (Power->isMinusOne())
2656 return BinaryOperator::CreateFDivFMF(ConstantFP::get(CI.getType(), 1.0),
2657 II->getArgOperand(0), II);
2658 // powi(x, 2) -> x*x
2659 if (Power->equalsInt(2))
2660 return BinaryOperator::CreateFMulFMF(II->getArgOperand(0),
2661 II->getArgOperand(0), II);
2662
2663 if (!Power->getValue()[0]) {
2664 Value *X;
2665 // If power is even:
2666 // powi(-x, p) -> powi(x, p)
2667 // powi(fabs(x), p) -> powi(x, p)
2668 // powi(copysign(x, y), p) -> powi(x, p)
2669 if (match(II->getArgOperand(0), m_FNeg(m_Value(X))) ||
2670 match(II->getArgOperand(0), m_FAbs(m_Value(X))) ||
2671 match(II->getArgOperand(0),
2673 return CallInst::Create(II->getCalledFunction(), {X, Power});
2674 }
2675 }
2676 if (ConstantFP *Base = dyn_cast<ConstantFP>(II->getArgOperand(0))) {
2677 Value *Exp = II->getArgOperand(1);
2678 Type *Ty = Base->getType();
2679 // powi(2.0, p) -> ldexp(1.0, p)
2680 if (II->hasApproxFunc() && Base->isExactlyValue(2.0)) {
2681 ConstantFP *One = ConstantFP::get(Ty, 1.0);
2682 if (auto *VTy = dyn_cast<VectorType>(Ty))
2683 Exp = Builder.CreateVectorSplat(VTy->getElementCount(), Exp);
2684 Value *Ldexp = Builder.CreateLdexp(One, Exp, II);
2685 return replaceInstUsesWith(*II, Ldexp);
2686 }
2687 }
2688 break;
2689 }
2690
2691 case Intrinsic::cttz:
2692 case Intrinsic::ctlz:
2693 if (auto *I = foldCttzCtlz(*II, *this))
2694 return I;
2695 break;
2696
2697 case Intrinsic::ctpop:
2698 if (auto *I = foldCtpop(*II, *this))
2699 return I;
2700 break;
2701
2702 case Intrinsic::fshl:
2703 case Intrinsic::fshr: {
2704 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
2705 Type *Ty = II->getType();
2706 unsigned BitWidth = Ty->getScalarSizeInBits();
2707 Constant *ShAmtC;
2708 if (match(II->getArgOperand(2), m_ImmConstant(ShAmtC))) {
2709 // Canonicalize a shift amount constant operand to modulo the bit-width.
2710 Constant *WidthC = ConstantInt::get(Ty, BitWidth);
2711 Constant *ModuloC =
2712 ConstantFoldBinaryOpOperands(Instruction::URem, ShAmtC, WidthC, DL);
2713 if (!ModuloC)
2714 return nullptr;
2715 if (ModuloC != ShAmtC)
2716 return CallInst::Create(II->getCalledFunction(), {Op0, Op1, ModuloC});
2717
2719 ShAmtC, DL),
2720 m_One()) &&
2721 "Shift amount expected to be modulo bitwidth");
2722
2723 // Canonicalize funnel shift right by constant to funnel shift left. This
2724 // is not entirely arbitrary. For historical reasons, the backend may
2725 // recognize rotate left patterns but miss rotate right patterns.
2726 if (IID == Intrinsic::fshr) {
2727 // fshr X, Y, C --> fshl X, Y, (BitWidth - C) if C is not zero.
2728 if (!isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2729 return nullptr;
2730
2731 Constant *LeftShiftC = ConstantExpr::getSub(WidthC, ShAmtC);
2732 Module *Mod = II->getModule();
2733 Function *Fshl =
2734 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::fshl, Ty);
2735 return CallInst::Create(Fshl, { Op0, Op1, LeftShiftC });
2736 }
2737 assert(IID == Intrinsic::fshl &&
2738 "All funnel shifts by simple constants should go left");
2739
2740 // fshl(X, 0, C) --> shl X, C
2741 // fshl(X, undef, C) --> shl X, C
2742 if (match(Op1, m_ZeroInt()) || match(Op1, m_Undef()))
2743 return BinaryOperator::CreateShl(Op0, ShAmtC);
2744
2745 // fshl(0, X, C) --> lshr X, (BW-C)
2746 // fshl(undef, X, C) --> lshr X, (BW-C)
2747 // Similar to fshr -> fshl fold above, this is only valid if C is not zero
2748 if ((match(Op0, m_ZeroInt()) || match(Op0, m_Undef())) &&
2749 isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2750 return BinaryOperator::CreateLShr(Op1,
2751 ConstantExpr::getSub(WidthC, ShAmtC));
2752
2753 // fshl i16 X, X, 8 --> bswap i16 X (reduce to more-specific form)
2754 if (Op0 == Op1 && BitWidth == 16 && match(ShAmtC, m_SpecificInt(8))) {
2755 Module *Mod = II->getModule();
2756 Function *Bswap =
2757 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::bswap, Ty);
2758 return CallInst::Create(Bswap, { Op0 });
2759 }
2760 if (Instruction *BitOp =
2761 matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ true,
2762 /*MatchBitReversals*/ true))
2763 return BitOp;
2764
2765 // R = fshl(X, X, C2)
2766 // fshl(R, R, C1) --> fshl(X, X, (C1 + C2) % bitsize)
2767 Value *InnerOp;
2768 const APInt *ShAmtInnerC, *ShAmtOuterC;
2769 if (match(Op0, m_FShl(m_Value(InnerOp), m_Deferred(InnerOp),
2770 m_APInt(ShAmtInnerC))) &&
2771 match(ShAmtC, m_APInt(ShAmtOuterC)) && Op0 == Op1) {
2772 APInt Sum = *ShAmtOuterC + *ShAmtInnerC;
2773 APInt Modulo = Sum.urem(APInt(Sum.getBitWidth(), BitWidth));
2774 if (Modulo.isZero())
2775 return replaceInstUsesWith(*II, InnerOp);
2776 Constant *ModuloC = ConstantInt::get(Ty, Modulo);
2778 {InnerOp, InnerOp, ModuloC});
2779 }
2780 }
2781
2782 // fshl(X, X, Neg(Y)) --> fshr(X, X, Y)
2783 // fshr(X, X, Neg(Y)) --> fshl(X, X, Y)
2784 // if BitWidth is a power-of-2
2785 Value *Y;
2786 if (Op0 == Op1 && isPowerOf2_32(BitWidth) &&
2787 match(II->getArgOperand(2), m_Neg(m_Value(Y)))) {
2788 Module *Mod = II->getModule();
2790 Mod, IID == Intrinsic::fshl ? Intrinsic::fshr : Intrinsic::fshl, Ty);
2791 return CallInst::Create(OppositeShift, {Op0, Op1, Y});
2792 }
2793
2794 // fshl(X, 0, Y) --> shl(X, and(Y, BitWidth - 1)) if bitwidth is a
2795 // power-of-2
2796 if (IID == Intrinsic::fshl && isPowerOf2_32(BitWidth) &&
2797 match(Op1, m_ZeroInt())) {
2798 Value *Op2 = II->getArgOperand(2);
2799 Value *And = Builder.CreateAnd(Op2, ConstantInt::get(Ty, BitWidth - 1));
2800 return BinaryOperator::CreateShl(Op0, And);
2801 }
2802
2803 // Left or right might be masked.
2805 return &CI;
2806
2807 // The shift amount (operand 2) of a funnel shift is modulo the bitwidth,
2808 // so only the low bits of the shift amount are demanded if the bitwidth is
2809 // a power-of-2.
2810 if (!isPowerOf2_32(BitWidth))
2811 break;
2813 KnownBits Op2Known(BitWidth);
2814 if (SimplifyDemandedBits(II, 2, Op2Demanded, Op2Known))
2815 return &CI;
2816 break;
2817 }
2818 case Intrinsic::pdep: {
2819 const APInt *MaskC;
2820 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2821 unsigned MaskIdx, MaskLen;
2822 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2823 // any single contiguous sequence of 1s anywhere in the mask simply
2824 // describes a subset of the input bits shifted to the appropriate
2825 // position. Replace with the straight forward IR.
2826 Value *Input = II->getArgOperand(0);
2827 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2828 Value *Shifted = Builder.CreateShl(Input, ShiftAmt);
2829 Value *Masked = Builder.CreateAnd(Shifted, II->getArgOperand(1));
2830 return replaceInstUsesWith(*II, Masked);
2831 }
2832 }
2833 break;
2834 }
2835 case Intrinsic::pext: {
2836 const APInt *MaskC;
2837 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2838 unsigned MaskIdx, MaskLen;
2839 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2840 // any single contiguous sequence of 1s anywhere in the mask simply
2841 // describes a subset of the input bits shifted to the appropriate
2842 // position. Replace with the straight forward IR.
2843 Value *Input = II->getArgOperand(0);
2844 Value *Masked = Builder.CreateAnd(Input, II->getArgOperand(1));
2845 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2846 Value *Shifted = Builder.CreateLShr(Masked, ShiftAmt);
2847 return replaceInstUsesWith(*II, Shifted);
2848 }
2849 }
2850 break;
2851 }
2852 case Intrinsic::ptrmask: {
2853 unsigned BitWidth = DL.getPointerTypeSizeInBits(II->getType());
2856 return II;
2857
2858 Value *InnerPtr, *InnerMask;
2859 bool Changed = false;
2860 // Combine:
2861 // (ptrmask (ptrmask p, A), B)
2862 // -> (ptrmask p, (and A, B))
2863 if (match(II->getArgOperand(0),
2865 m_Value(InnerMask))))) {
2866 assert(II->getArgOperand(1)->getType() == InnerMask->getType() &&
2867 "Mask types must match");
2868 // TODO: If InnerMask == Op1, we could copy attributes from inner
2869 // callsite -> outer callsite.
2870 Value *NewMask = Builder.CreateAnd(II->getArgOperand(1), InnerMask);
2871 replaceOperand(CI, 0, InnerPtr);
2872 replaceOperand(CI, 1, NewMask);
2873 Changed = true;
2874 }
2875
2876 // See if we can deduce non-null.
2877 if (!CI.hasRetAttr(Attribute::NonNull) &&
2878 (Known.isNonZero() ||
2879 isKnownNonZero(II, getSimplifyQuery().getWithInstruction(II)))) {
2880 CI.addRetAttr(Attribute::NonNull);
2881 Changed = true;
2882 }
2883
2884 unsigned NewAlignmentLog =
2886 std::min(BitWidth - 1, Known.countMinTrailingZeros()));
2887 // Known bits will capture if we had alignment information associated with
2888 // the pointer argument.
2889 if (NewAlignmentLog > Log2(CI.getRetAlign().valueOrOne())) {
2891 CI.getContext(), Align(uint64_t(1) << NewAlignmentLog)));
2892 Changed = true;
2893 }
2894 if (Changed)
2895 return &CI;
2896 break;
2897 }
2898 case Intrinsic::uadd_with_overflow:
2899 case Intrinsic::sadd_with_overflow: {
2900 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2901 return I;
2902
2903 // Given 2 constant operands whose sum does not overflow:
2904 // uaddo (X +nuw C0), C1 -> uaddo X, C0 + C1
2905 // saddo (X +nsw C0), C1 -> saddo X, C0 + C1
2906 Value *X;
2907 const APInt *C0, *C1;
2908 Value *Arg0 = II->getArgOperand(0);
2909 Value *Arg1 = II->getArgOperand(1);
2910 bool IsSigned = IID == Intrinsic::sadd_with_overflow;
2911 bool HasNWAdd = IsSigned
2912 ? match(Arg0, m_NSWAddLike(m_Value(X), m_APInt(C0)))
2913 : match(Arg0, m_NUWAddLike(m_Value(X), m_APInt(C0)));
2914 if (HasNWAdd && match(Arg1, m_APInt(C1))) {
2915 bool Overflow;
2916 APInt NewC =
2917 IsSigned ? C1->sadd_ov(*C0, Overflow) : C1->uadd_ov(*C0, Overflow);
2918 if (!Overflow)
2919 return replaceInstUsesWith(
2920 *II, Builder.CreateBinaryIntrinsic(
2921 IID, X, ConstantInt::get(Arg1->getType(), NewC)));
2922 }
2923 break;
2924 }
2925
2926 case Intrinsic::umul_with_overflow:
2927 case Intrinsic::smul_with_overflow:
2928 case Intrinsic::usub_with_overflow:
2929 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2930 return I;
2931 break;
2932
2933 case Intrinsic::ssub_with_overflow: {
2934 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2935 return I;
2936
2937 Constant *C;
2938 Value *Arg0 = II->getArgOperand(0);
2939 Value *Arg1 = II->getArgOperand(1);
2940 // Given a constant C that is not the minimum signed value
2941 // for an integer of a given bit width:
2942 //
2943 // ssubo X, C -> saddo X, -C
2944 if (match(Arg1, m_Constant(C)) && C->isNotMinSignedValue()) {
2945 Value *NegVal = ConstantExpr::getNeg(C);
2946 // Build a saddo call that is equivalent to the discovered
2947 // ssubo call.
2948 return replaceInstUsesWith(
2949 *II, Builder.CreateBinaryIntrinsic(Intrinsic::sadd_with_overflow,
2950 Arg0, NegVal));
2951 }
2952
2953 break;
2954 }
2955
2956 case Intrinsic::uadd_sat:
2957 case Intrinsic::sadd_sat:
2958 case Intrinsic::usub_sat:
2959 case Intrinsic::ssub_sat: {
2961 Type *Ty = SI->getType();
2962 Value *Arg0 = SI->getLHS();
2963 Value *Arg1 = SI->getRHS();
2964
2965 // Make use of known overflow information.
2966 OverflowResult OR = computeOverflow(SI->getBinaryOp(), SI->isSigned(),
2967 Arg0, Arg1, SI);
2968 switch (OR) {
2970 break;
2972 if (SI->isSigned())
2973 return BinaryOperator::CreateNSW(SI->getBinaryOp(), Arg0, Arg1);
2974 else
2975 return BinaryOperator::CreateNUW(SI->getBinaryOp(), Arg0, Arg1);
2977 unsigned BitWidth = Ty->getScalarSizeInBits();
2978 APInt Min = APSInt::getMinValue(BitWidth, !SI->isSigned());
2979 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Min));
2980 }
2982 unsigned BitWidth = Ty->getScalarSizeInBits();
2983 APInt Max = APSInt::getMaxValue(BitWidth, !SI->isSigned());
2984 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Max));
2985 }
2986 }
2987
2988 // usub_sat((sub nuw C, A), C1) -> usub_sat(usub_sat(C, C1), A)
2989 // which after that:
2990 // usub_sat((sub nuw C, A), C1) -> usub_sat(C - C1, A) if C1 u< C
2991 // usub_sat((sub nuw C, A), C1) -> 0 otherwise
2992 Constant *C, *C1;
2993 Value *A;
2994 if (IID == Intrinsic::usub_sat &&
2995 match(Arg0, m_NUWSub(m_ImmConstant(C), m_Value(A))) &&
2996 match(Arg1, m_ImmConstant(C1))) {
2997 auto *NewC = Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, C, C1);
2998 auto *NewSub =
2999 Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, NewC, A);
3000 return replaceInstUsesWith(*SI, NewSub);
3001 }
3002
3003 // ssub.sat(X, C) -> sadd.sat(X, -C) if C != MIN
3004 if (IID == Intrinsic::ssub_sat && match(Arg1, m_Constant(C)) &&
3005 C->isNotMinSignedValue()) {
3006 Value *NegVal = ConstantExpr::getNeg(C);
3007 return replaceInstUsesWith(
3008 *II, Builder.CreateBinaryIntrinsic(
3009 Intrinsic::sadd_sat, Arg0, NegVal));
3010 }
3011
3012 // sat(sat(X + Val2) + Val) -> sat(X + (Val+Val2))
3013 // sat(sat(X - Val2) - Val) -> sat(X - (Val+Val2))
3014 // if Val and Val2 have the same sign
3015 if (auto *Other = dyn_cast<IntrinsicInst>(Arg0)) {
3016 Value *X;
3017 const APInt *Val, *Val2;
3018 APInt NewVal;
3019 bool IsUnsigned =
3020 IID == Intrinsic::uadd_sat || IID == Intrinsic::usub_sat;
3021 if (Other->getIntrinsicID() == IID &&
3022 match(Arg1, m_APInt(Val)) &&
3023 match(Other->getArgOperand(0), m_Value(X)) &&
3024 match(Other->getArgOperand(1), m_APInt(Val2))) {
3025 if (IsUnsigned)
3026 NewVal = Val->uadd_sat(*Val2);
3027 else if (Val->isNonNegative() == Val2->isNonNegative()) {
3028 bool Overflow;
3029 NewVal = Val->sadd_ov(*Val2, Overflow);
3030 if (Overflow) {
3031 // Both adds together may add more than SignedMaxValue
3032 // without saturating the final result.
3033 break;
3034 }
3035 } else {
3036 // Cannot fold saturated addition with different signs.
3037 break;
3038 }
3039
3040 return replaceInstUsesWith(
3041 *II, Builder.CreateBinaryIntrinsic(
3042 IID, X, ConstantInt::get(II->getType(), NewVal)));
3043 }
3044 }
3045 break;
3046 }
3047
3048 case Intrinsic::minnum:
3049 case Intrinsic::maxnum:
3050 case Intrinsic::minimumnum:
3051 case Intrinsic::maximumnum:
3052 case Intrinsic::minimum:
3053 case Intrinsic::maximum: {
3054 Value *Arg0 = II->getArgOperand(0);
3055 Value *Arg1 = II->getArgOperand(1);
3056 Value *X, *Y;
3057 if (match(Arg0, m_FNeg(m_Value(X))) && match(Arg1, m_FNeg(m_Value(Y))) &&
3058 (Arg0->hasOneUse() || Arg1->hasOneUse())) {
3059 // If both operands are negated, invert the call and negate the result:
3060 // min(-X, -Y) --> -(max(X, Y))
3061 // max(-X, -Y) --> -(min(X, Y))
3062 Intrinsic::ID NewIID;
3063 switch (IID) {
3064 case Intrinsic::maxnum:
3065 NewIID = Intrinsic::minnum;
3066 break;
3067 case Intrinsic::minnum:
3068 NewIID = Intrinsic::maxnum;
3069 break;
3070 case Intrinsic::maximumnum:
3071 NewIID = Intrinsic::minimumnum;
3072 break;
3073 case Intrinsic::minimumnum:
3074 NewIID = Intrinsic::maximumnum;
3075 break;
3076 case Intrinsic::maximum:
3077 NewIID = Intrinsic::minimum;
3078 break;
3079 case Intrinsic::minimum:
3080 NewIID = Intrinsic::maximum;
3081 break;
3082 default:
3083 llvm_unreachable("unexpected intrinsic ID");
3084 }
3085 Value *NewCall = Builder.CreateBinaryIntrinsic(NewIID, X, Y, II);
3086 Instruction *FNeg = UnaryOperator::CreateFNeg(NewCall);
3087 FNeg->copyIRFlags(II);
3088 return FNeg;
3089 }
3090
3091 // m(m(X, C2), C1) -> m(X, C)
3092 const APFloat *C1, *C2;
3093 if (auto *M = dyn_cast<IntrinsicInst>(Arg0)) {
3094 if (M->getIntrinsicID() == IID && match(Arg1, m_APFloat(C1)) &&
3095 ((match(M->getArgOperand(0), m_Value(X)) &&
3096 match(M->getArgOperand(1), m_APFloat(C2))) ||
3097 (match(M->getArgOperand(1), m_Value(X)) &&
3098 match(M->getArgOperand(0), m_APFloat(C2))))) {
3099 APFloat Res(0.0);
3100 switch (IID) {
3101 case Intrinsic::maxnum:
3102 Res = maxnum(*C1, *C2);
3103 break;
3104 case Intrinsic::minnum:
3105 Res = minnum(*C1, *C2);
3106 break;
3107 case Intrinsic::maximumnum:
3108 Res = maximumnum(*C1, *C2);
3109 break;
3110 case Intrinsic::minimumnum:
3111 Res = minimumnum(*C1, *C2);
3112 break;
3113 case Intrinsic::maximum:
3114 Res = maximum(*C1, *C2);
3115 break;
3116 case Intrinsic::minimum:
3117 Res = minimum(*C1, *C2);
3118 break;
3119 default:
3120 llvm_unreachable("unexpected intrinsic ID");
3121 }
3122 // TODO: Conservatively intersecting FMF. If Res == C2, the transform
3123 // was a simplification (so Arg0 and its original flags could
3124 // propagate?)
3125 Value *V = Builder.CreateBinaryIntrinsic(
3126 IID, X, ConstantFP::get(Arg0->getType(), Res),
3128 return replaceInstUsesWith(*II, V);
3129 }
3130 }
3131
3132 // m((fpext X), (fpext Y)) -> fpext (m(X, Y))
3133 if (match(Arg0, m_FPExt(m_Value(X))) && match(Arg1, m_FPExt(m_Value(Y))) &&
3134 (Arg0->hasOneUse() || Arg1->hasOneUse()) &&
3135 X->getType() == Y->getType()) {
3136 Value *NewCall =
3137 Builder.CreateBinaryIntrinsic(IID, X, Y, II, II->getName());
3138 return new FPExtInst(NewCall, II->getType());
3139 }
3140
3141 // m(fpext X, C) -> fpext m(X, TruncC) if C can be losslessly truncated.
3142 Constant *C;
3143 if (match(Arg0, m_OneUse(m_FPExt(m_Value(X)))) &&
3144 match(Arg1, m_ImmConstant(C))) {
3145 if (Constant *TruncC =
3146 getLosslessInvCast(C, X->getType(), Instruction::FPExt, DL)) {
3147 Value *NewCall =
3148 Builder.CreateBinaryIntrinsic(IID, X, TruncC, II, II->getName());
3149 return new FPExtInst(NewCall, II->getType());
3150 }
3151 }
3152
3153 // max X, -X --> fabs X
3154 // min X, -X --> -(fabs X)
3155 // TODO: Remove one-use limitation? That is obviously better for max,
3156 // hence why we don't check for one-use for that. However,
3157 // it would be an extra instruction for min (fnabs), but
3158 // that is still likely better for analysis and codegen.
3159 auto IsMinMaxOrXNegX = [IID, &X](Value *Op0, Value *Op1) {
3160 if (match(Op0, m_FNeg(m_Value(X))) && match(Op1, m_Specific(X)))
3161 return Op0->hasOneUse() ||
3162 (IID != Intrinsic::minimum && IID != Intrinsic::minnum &&
3163 IID != Intrinsic::minimumnum);
3164 return false;
3165 };
3166
3167 if (IsMinMaxOrXNegX(Arg0, Arg1) || IsMinMaxOrXNegX(Arg1, Arg0)) {
3168 Value *R = Builder.CreateFAbs(X, II);
3169 if (IID == Intrinsic::minimum || IID == Intrinsic::minnum ||
3170 IID == Intrinsic::minimumnum)
3171 R = Builder.CreateFNegFMF(R, II);
3172 return replaceInstUsesWith(*II, R);
3173 }
3174
3175 break;
3176 }
3177 case Intrinsic::matrix_multiply: {
3178 // Optimize negation in matrix multiplication.
3179
3180 // -A * -B -> A * B
3181 Value *A, *B;
3182 if (match(II->getArgOperand(0), m_FNeg(m_Value(A))) &&
3183 match(II->getArgOperand(1), m_FNeg(m_Value(B)))) {
3184 replaceOperand(*II, 0, A);
3185 replaceOperand(*II, 1, B);
3186 return II;
3187 }
3188
3189 Value *Op0 = II->getOperand(0);
3190 Value *Op1 = II->getOperand(1);
3191 Value *OpNotNeg, *NegatedOp;
3192 unsigned NegatedOpArg, OtherOpArg;
3193 if (match(Op0, m_FNeg(m_Value(OpNotNeg)))) {
3194 NegatedOp = Op0;
3195 NegatedOpArg = 0;
3196 OtherOpArg = 1;
3197 } else if (match(Op1, m_FNeg(m_Value(OpNotNeg)))) {
3198 NegatedOp = Op1;
3199 NegatedOpArg = 1;
3200 OtherOpArg = 0;
3201 } else
3202 // Multiplication doesn't have a negated operand.
3203 break;
3204
3205 // Only optimize if the negated operand has only one use.
3206 if (!NegatedOp->hasOneUse())
3207 break;
3208
3209 Value *OtherOp = II->getOperand(OtherOpArg);
3210 VectorType *RetTy = cast<VectorType>(II->getType());
3211 VectorType *NegatedOpTy = cast<VectorType>(NegatedOp->getType());
3212 VectorType *OtherOpTy = cast<VectorType>(OtherOp->getType());
3213 ElementCount NegatedCount = NegatedOpTy->getElementCount();
3214 ElementCount OtherCount = OtherOpTy->getElementCount();
3215 ElementCount RetCount = RetTy->getElementCount();
3216 // (-A) * B -> A * (-B), if it is cheaper to negate B and vice versa.
3217 if (ElementCount::isKnownGT(NegatedCount, OtherCount) &&
3218 ElementCount::isKnownLT(OtherCount, RetCount)) {
3219 Value *InverseOtherOp = Builder.CreateFNeg(OtherOp);
3220 replaceOperand(*II, NegatedOpArg, OpNotNeg);
3221 replaceOperand(*II, OtherOpArg, InverseOtherOp);
3222 return II;
3223 }
3224 // (-A) * B -> -(A * B), if it is cheaper to negate the result
3225 if (ElementCount::isKnownGT(NegatedCount, RetCount)) {
3226 SmallVector<Value *, 5> NewArgs(II->args());
3227 NewArgs[NegatedOpArg] = OpNotNeg;
3228 Value *NewMul = Builder.CreateIntrinsic(II->getType(), IID, NewArgs, II);
3229 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(NewMul, II));
3230 }
3231 break;
3232 }
3233 case Intrinsic::fmuladd: {
3234 // Try to simplify the underlying FMul.
3235 if (Value *V =
3236 simplifyFMulInst(II->getArgOperand(0), II->getArgOperand(1),
3237 II->getFastMathFlags(), SQ.getWithInstruction(II)))
3238 return BinaryOperator::CreateFAddFMF(V, II->getArgOperand(2),
3239 II->getFastMathFlags());
3240
3241 [[fallthrough]];
3242 }
3243 case Intrinsic::fma: {
3244 // fma fneg(x), fneg(y), z -> fma x, y, z
3245 Value *Src0 = II->getArgOperand(0);
3246 Value *Src1 = II->getArgOperand(1);
3247 Value *Src2 = II->getArgOperand(2);
3248 Value *X, *Y;
3249 if (match(Src0, m_FNeg(m_Value(X))) && match(Src1, m_FNeg(m_Value(Y))))
3250 return replaceInstUsesWith(
3251 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, Y, Src2}, II));
3252
3253 // fma fabs(x), fabs(x), z -> fma x, x, z
3254 if (match(Src0, m_FAbs(m_Value(X))) && match(Src1, m_FAbs(m_Specific(X))))
3255 return replaceInstUsesWith(
3256 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, X, Src2}, II));
3257
3258 // Try to simplify the underlying FMul. We can only apply simplifications
3259 // that do not require rounding.
3260 if (Value *V = simplifyFMAFMul(Src0, Src1, II->getFastMathFlags(),
3261 SQ.getWithInstruction(II)))
3262 return BinaryOperator::CreateFAddFMF(V, Src2, II->getFastMathFlags());
3263
3264 // fma x, y, 0 -> fmul x, y
3265 // This is always valid for -0.0, but requires nsz for +0.0 as
3266 // -0.0 + 0.0 = 0.0, which would not be the same as the fmul on its own.
3267 if (match(Src2, m_NegZeroFP()) ||
3268 (match(Src2, m_PosZeroFP()) && II->getFastMathFlags().noSignedZeros()))
3269 return BinaryOperator::CreateFMulFMF(Src0, Src1, II);
3270
3271 // fma x, -1.0, y -> fsub y, x
3272 if (match(Src1, m_SpecificFP(-1.0)))
3273 return BinaryOperator::CreateFSubFMF(Src2, Src0, II);
3274
3275 break;
3276 }
3277 case Intrinsic::copysign: {
3278 Value *Mag = II->getArgOperand(0), *Sign = II->getArgOperand(1);
3279 if (std::optional<bool> KnownSignBit = computeKnownFPSignBit(
3280 Sign, getSimplifyQuery().getWithInstruction(II))) {
3281 if (*KnownSignBit) {
3282 // If we know that the sign argument is negative, reduce to FNABS:
3283 // copysign Mag, -Sign --> fneg (fabs Mag)
3284 Value *Fabs = Builder.CreateFAbs(Mag, II);
3285 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(Fabs, II));
3286 }
3287
3288 // If we know that the sign argument is positive, reduce to FABS:
3289 // copysign Mag, +Sign --> fabs Mag
3290 Value *Fabs = Builder.CreateFAbs(Mag, II);
3291 return replaceInstUsesWith(*II, Fabs);
3292 }
3293
3294 // Propagate sign argument through nested calls:
3295 // copysign Mag, (copysign ?, X) --> copysign Mag, X
3296 Value *X;
3298 Value *CopySign =
3299 Builder.CreateCopySign(Mag, X, FMFSource::intersect(II, Sign));
3300 return replaceInstUsesWith(*II, CopySign);
3301 }
3302
3303 // Clear sign-bit of constant magnitude:
3304 // copysign -MagC, X --> copysign MagC, X
3305 // TODO: Support constant folding for fabs
3306 const APFloat *MagC;
3307 if (match(Mag, m_APFloat(MagC)) && MagC->isNegative()) {
3308 APFloat PosMagC = *MagC;
3309 PosMagC.clearSign();
3310 return replaceInstUsesWith(
3311 *II, Builder.CreateCopySign(ConstantFP::get(Mag->getType(), PosMagC),
3312 Sign, II));
3313 }
3314
3315 // Peek through changes of magnitude's sign-bit. This call rewrites those:
3316 // copysign (fabs X), Sign --> copysign X, Sign
3317 // copysign (fneg X), Sign --> copysign X, Sign
3318 if (match(Mag, m_FAbs(m_Value(X))) || match(Mag, m_FNeg(m_Value(X))))
3319 return replaceInstUsesWith(*II, Builder.CreateCopySign(X, Sign, II));
3320
3321 // copysign(floor(fabs(X)), X) --> copysign(trunc(X), X)
3322 // copysign ignores the sign bit of its magnitude argument (implicit fabs),
3323 // so replacing floor(fabs(X)) with trunc(X) is correct for all inputs
3324 // including NaN without requiring nnan. The m_FAbs match also ensures
3325 // the floor argument is non-negative, so floor == trunc.
3326 Value *FAbsArg;
3327 if (match(Mag, m_Intrinsic<Intrinsic::floor>(m_FAbs(m_Value(FAbsArg)))) &&
3328 FAbsArg == Sign) {
3329 Value *Trunc = Builder.CreateUnaryIntrinsic(Intrinsic::trunc, Sign, II);
3330 return replaceInstUsesWith(*II, Builder.CreateCopySign(Trunc, Sign, II));
3331 }
3332
3333 Type *SignEltTy = Sign->getType()->getScalarType();
3334
3335 Value *CastSrc;
3336 if (match(Sign,
3338 CastSrc->getType()->isIntOrIntVectorTy() &&
3342 APInt::getSignMask(Known.getBitWidth()), Known,
3343 SQ))
3344 return II;
3345 }
3346
3347 break;
3348 }
3349 case Intrinsic::fabs: {
3350 Value *Cond, *TVal, *FVal;
3351 Value *Arg = II->getArgOperand(0);
3352 Value *X;
3353 // fabs (-X) --> fabs (X)
3354 if (match(Arg, m_FNeg(m_Value(X)))) {
3355 Value *Fabs = Builder.CreateFAbs(X, II);
3356 return replaceInstUsesWith(CI, Fabs);
3357 }
3358
3359 if (match(Arg, m_Select(m_Value(Cond), m_Value(TVal), m_Value(FVal)))) {
3360 // fabs (select Cond, TrueC, FalseC) --> select Cond, AbsT, AbsF
3361 if (Arg->hasOneUse() ? (isa<Constant>(TVal) || isa<Constant>(FVal))
3362 : (isa<Constant>(TVal) && isa<Constant>(FVal))) {
3363 CallInst *AbsT = Builder.CreateCall(II->getCalledFunction(), {TVal});
3364 CallInst *AbsF = Builder.CreateCall(II->getCalledFunction(), {FVal});
3365 SelectInst *SI = SelectInst::Create(Cond, AbsT, AbsF);
3366 SI->setFastMathFlags(II->getFastMathFlags() |
3367 cast<SelectInst>(Arg)->getFastMathFlags());
3368 // Can't copy nsz to select, as even with the nsz flag the fabs result
3369 // always has the sign bit unset.
3370 SI->setHasNoSignedZeros(false);
3371 return SI;
3372 }
3373 // fabs (select Cond, -FVal, FVal) --> fabs FVal
3374 if (match(TVal, m_FNeg(m_Specific(FVal))))
3375 return replaceInstUsesWith(*II, Builder.CreateFAbs(FVal, II));
3376 // fabs (select Cond, TVal, -TVal) --> fabs TVal
3377 if (match(FVal, m_FNeg(m_Specific(TVal))))
3378 return replaceInstUsesWith(*II, Builder.CreateFAbs(TVal, II));
3379 }
3380
3381 Value *Magnitude, *Sign;
3382 if (match(II->getArgOperand(0),
3383 m_CopySign(m_Value(Magnitude), m_Value(Sign)))) {
3384 // fabs (copysign x, y) -> (fabs x)
3385 Value *AbsSign = Builder.CreateFAbs(Magnitude, II);
3386 return replaceInstUsesWith(*II, AbsSign);
3387 }
3388
3389 [[fallthrough]];
3390 }
3391 case Intrinsic::ceil:
3392 case Intrinsic::floor:
3393 case Intrinsic::round:
3394 case Intrinsic::roundeven:
3395 case Intrinsic::nearbyint:
3396 case Intrinsic::rint:
3397 case Intrinsic::trunc: {
3398 Value *ExtSrc;
3399 if (match(II->getArgOperand(0), m_OneUse(m_FPExt(m_Value(ExtSrc))))) {
3400 // Narrow the call: intrinsic (fpext x) -> fpext (intrinsic x)
3401 Value *NarrowII = Builder.CreateUnaryIntrinsic(IID, ExtSrc, II);
3402 return new FPExtInst(NarrowII, II->getType());
3403 }
3404 break;
3405 }
3406 case Intrinsic::cos:
3407 case Intrinsic::amdgcn_cos:
3408 case Intrinsic::cosh: {
3409 Value *X, *Sign;
3410 Value *Src = II->getArgOperand(0);
3411 if (match(Src, m_FNeg(m_Value(X))) || match(Src, m_FAbs(m_Value(X))) ||
3412 match(Src, m_CopySign(m_Value(X), m_Value(Sign)))) {
3413 // f(-x) --> f(x)
3414 // f(fabs(x)) --> f(x)
3415 // f(copysign(x, y)) --> f(x)
3416 // for f in {cos, cosh}
3417 return replaceInstUsesWith(*II, Builder.CreateUnaryIntrinsic(IID, X, II));
3418 }
3419 if (IID == Intrinsic::cos) {
3420 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3421 return replaceInstUsesWith(*II, Result);
3422 }
3423 break;
3424 }
3425 case Intrinsic::sin:
3426 case Intrinsic::amdgcn_sin:
3427 case Intrinsic::sinh:
3428 case Intrinsic::tan:
3429 case Intrinsic::tanh: {
3430 Value *X;
3431 if (match(II->getArgOperand(0), m_OneUse(m_FNeg(m_Value(X))))) {
3432 // f(-x) --> -f(x)
3433 // for f in {sin, sinh, tan, tanh}
3434 Value *NewFunc = Builder.CreateUnaryIntrinsic(IID, X, II);
3435 return UnaryOperator::CreateFNegFMF(NewFunc, II);
3436 }
3437 if (IID == Intrinsic::sin) {
3438 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3439 return replaceInstUsesWith(*II, Result);
3440 }
3441 break;
3442 }
3443 case Intrinsic::ldexp: {
3444 Value *Src = II->getArgOperand(0);
3445 Value *Exp = II->getArgOperand(1);
3446
3447 // ldexp(x, K) -> fmul x, 2^K
3448 uint64_t ConstExp;
3449 if (match(Exp, m_ConstantInt(ConstExp))) {
3450 const fltSemantics &FPTy =
3451 Src->getType()->getScalarType()->getFltSemantics();
3452
3453 APFloat Scaled = scalbn(APFloat::getOne(FPTy), static_cast<int>(ConstExp),
3455 if (!Scaled.isZero() && !Scaled.isInfinity()) {
3456 // Skip overflow and underflow cases.
3457 Constant *FPConst = ConstantFP::get(Src->getType(), Scaled);
3458 return BinaryOperator::CreateFMulFMF(Src, FPConst, II);
3459 }
3460 }
3461
3462 // ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b))
3463 //
3464 // A danger is if the first ldexp would overflow to infinity or underflow to
3465 // zero, but the combined exponent avoids it.
3466 //
3467 // We ignore this with reassoc, or if we know both exponents have the same
3468 // sign (since then we'd just double down on the over/underflow which would
3469 // occur anyway).
3470 //
3471 // ldexp can take arbitrary integer types, so we also need to ensure that
3472 // our exponent type is wide enough so that if sadd.sat(a, b) saturates,
3473 // then ldexp at the saturated exponent saturates to inf or zero as well.
3474 //
3475 // TODO: Could do better if we had range tracking for the input value
3476 // exponent. Also could broaden sign check to cover == 0 case.
3477 Value *InnerSrc;
3478 Value *InnerExp;
3480 m_Value(InnerSrc), m_Value(InnerExp)))) &&
3481 Exp->getType() == InnerExp->getType()) {
3482 FastMathFlags FMF = II->getFastMathFlags();
3483 FastMathFlags InnerFlags = cast<FPMathOperator>(Src)->getFastMathFlags();
3484
3485 if (ldexpSaturatingAddIsSafe(II->getType(), Exp->getType()) &&
3486 ((FMF.allowReassoc() && InnerFlags.allowReassoc()) ||
3487 signBitMustBeTheSame(Exp, InnerExp, SQ.getWithInstruction(II)))) {
3488 Value *NewExp =
3489 Builder.CreateBinaryIntrinsic(Intrinsic::sadd_sat, InnerExp, Exp);
3490 return replaceInstUsesWith(
3491 *II, Builder.CreateLdexp(InnerSrc, NewExp, FMF | InnerFlags));
3492 }
3493 }
3494
3495 // ldexp(x, zext(i1 y)) -> fmul x, (select y, 2.0, 1.0)
3496 // ldexp(x, sext(i1 y)) -> fmul x, (select y, 0.5, 1.0)
3497 Value *ExtSrc;
3498 if (match(Exp, m_ZExt(m_Value(ExtSrc))) &&
3499 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3500 Value *Select =
3501 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 2.0),
3502 ConstantFP::get(II->getType(), 1.0));
3504 }
3505 if (match(Exp, m_SExt(m_Value(ExtSrc))) &&
3506 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3507 Value *Select =
3508 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 0.5),
3509 ConstantFP::get(II->getType(), 1.0));
3511 }
3512
3513 // ldexp(x, c ? exp : 0) -> c ? ldexp(x, exp) : x
3514 // ldexp(x, c ? 0 : exp) -> c ? x : ldexp(x, exp)
3515 ///
3516 // TODO: If we cared, should insert a canonicalize for x
3517 Value *SelectCond, *SelectLHS, *SelectRHS;
3518 if (match(II->getArgOperand(1),
3519 m_OneUse(m_Select(m_Value(SelectCond), m_Value(SelectLHS),
3520 m_Value(SelectRHS))))) {
3521 Value *NewLdexp = nullptr;
3522 Value *Select = nullptr;
3523 if (match(SelectRHS, m_ZeroInt())) {
3524 NewLdexp = Builder.CreateLdexp(Src, SelectLHS, II);
3525 Select = Builder.CreateSelect(SelectCond, NewLdexp, Src);
3526 } else if (match(SelectLHS, m_ZeroInt())) {
3527 NewLdexp = Builder.CreateLdexp(Src, SelectRHS, II);
3528 Select = Builder.CreateSelect(SelectCond, Src, NewLdexp);
3529 }
3530
3531 if (NewLdexp) {
3532 Select->takeName(II);
3533 return replaceInstUsesWith(*II, Select);
3534 }
3535 }
3536
3537 break;
3538 }
3539 case Intrinsic::ptrauth_auth:
3540 case Intrinsic::ptrauth_resign: {
3541 // (sign|resign) + (auth|resign) can be folded by omitting the middle
3542 // sign+auth component if the key and discriminator match.
3543 bool NeedSign = II->getIntrinsicID() == Intrinsic::ptrauth_resign;
3544 Value *Ptr = II->getArgOperand(0);
3545 Value *Key = II->getArgOperand(1);
3546 Value *Disc = II->getArgOperand(2);
3547 Value *DS = nullptr;
3548 if (auto Bundle = II->getOperandBundle(LLVMContext::OB_deactivation_symbol))
3549 DS = Bundle->Inputs[0];
3550
3551 // AuthKey will be the key we need to end up authenticating against in
3552 // whatever we replace this sequence with.
3553 Value *AuthKey = nullptr, *AuthDisc = nullptr, *BasePtr;
3554 if (const auto *CI = dyn_cast<CallBase>(Ptr)) {
3555 Value *OtherDS = nullptr;
3556 if (auto Bundle =
3558 OtherDS = Bundle->Inputs[0];
3559 if (DS != OtherDS)
3560 break;
3561
3562 if (CI->getIntrinsicID() == Intrinsic::ptrauth_sign) {
3563 if (CI->getArgOperand(1) != Key || CI->getArgOperand(2) != Disc)
3564 break;
3565 } else if (CI->getIntrinsicID() == Intrinsic::ptrauth_resign) {
3566 // The resign intrinsic does not support deactivation symbols.
3567 assert(!DS);
3568 if (CI->getArgOperand(3) != Key || CI->getArgOperand(4) != Disc)
3569 break;
3570 AuthKey = CI->getArgOperand(1);
3571 AuthDisc = CI->getArgOperand(2);
3572 } else
3573 break;
3574 BasePtr = CI->getArgOperand(0);
3575 } else if (const auto *PtrToInt = dyn_cast<PtrToIntOperator>(Ptr)) {
3576 // ptrauth constants are equivalent to a call to @llvm.ptrauth.sign for
3577 // our purposes, so check for that too.
3578 const auto *CPA = dyn_cast<ConstantPtrAuth>(PtrToInt->getOperand(0));
3579 if (!CPA || DS || !CPA->isKnownCompatibleWith(Key, Disc, DL))
3580 break;
3581
3582 // resign(ptrauth(p,ks,ds),ks,ds,kr,dr) -> ptrauth(p,kr,dr)
3583 if (NeedSign && isa<ConstantInt>(II->getArgOperand(4))) {
3584 auto *SignKey = cast<ConstantInt>(II->getArgOperand(3));
3585 auto *SignDisc = cast<ConstantInt>(II->getArgOperand(4));
3586 auto *Null = ConstantPointerNull::get(Builder.getPtrTy());
3587 auto *NewCPA = ConstantPtrAuth::get(CPA->getPointer(), SignKey,
3588 SignDisc, /*AddrDisc=*/Null,
3589 /*DeactivationSymbol=*/Null);
3591 *II, ConstantExpr::getPointerCast(NewCPA, II->getType()));
3592 return eraseInstFromFunction(*II);
3593 }
3594
3595 // auth(ptrauth(p,k,d),k,d) -> p
3596 BasePtr = Builder.CreatePtrToInt(CPA->getPointer(), II->getType());
3597 } else
3598 break;
3599
3600 unsigned NewIntrin;
3601 if (AuthKey && NeedSign) {
3602 // resign(0,1) + resign(1,2) = resign(0, 2)
3603 NewIntrin = Intrinsic::ptrauth_resign;
3604 } else if (AuthKey) {
3605 // resign(0,1) + auth(1) = auth(0)
3606 NewIntrin = Intrinsic::ptrauth_auth;
3607 } else if (NeedSign) {
3608 // sign(0) + resign(0, 1) = sign(1)
3609 NewIntrin = Intrinsic::ptrauth_sign;
3610 } else {
3611 // sign(0) + auth(0) = nop
3612 replaceInstUsesWith(*II, BasePtr);
3613 return eraseInstFromFunction(*II);
3614 }
3615
3616 SmallVector<Value *, 4> CallArgs;
3617 CallArgs.push_back(BasePtr);
3618 if (AuthKey) {
3619 CallArgs.push_back(AuthKey);
3620 CallArgs.push_back(AuthDisc);
3621 }
3622
3623 if (NeedSign) {
3624 CallArgs.push_back(II->getArgOperand(3));
3625 CallArgs.push_back(II->getArgOperand(4));
3626 }
3627
3628 std::vector<OperandBundleDef> Bundles;
3629 if (DS)
3630 Bundles.push_back(OperandBundleDef("deactivation-symbol", DS));
3631
3632 Function *NewFn =
3633 Intrinsic::getOrInsertDeclaration(II->getModule(), NewIntrin);
3634 return CallInst::Create(NewFn, CallArgs, Bundles);
3635 }
3636 case Intrinsic::arm_neon_vtbl1:
3637 case Intrinsic::arm_neon_vtbl2:
3638 case Intrinsic::arm_neon_vtbl3:
3639 case Intrinsic::arm_neon_vtbl4:
3640 case Intrinsic::aarch64_neon_tbl1:
3641 case Intrinsic::aarch64_neon_tbl2:
3642 case Intrinsic::aarch64_neon_tbl3:
3643 case Intrinsic::aarch64_neon_tbl4:
3644 return simplifyNeonTbl(*II, *this, /*IsExtension=*/false);
3645 case Intrinsic::arm_neon_vtbx1:
3646 case Intrinsic::arm_neon_vtbx2:
3647 case Intrinsic::arm_neon_vtbx3:
3648 case Intrinsic::arm_neon_vtbx4:
3649 case Intrinsic::aarch64_neon_tbx1:
3650 case Intrinsic::aarch64_neon_tbx2:
3651 case Intrinsic::aarch64_neon_tbx3:
3652 case Intrinsic::aarch64_neon_tbx4:
3653 return simplifyNeonTbl(*II, *this, /*IsExtension=*/true);
3654
3655 case Intrinsic::arm_neon_vmulls:
3656 case Intrinsic::arm_neon_vmullu:
3657 case Intrinsic::aarch64_neon_smull:
3658 case Intrinsic::aarch64_neon_umull: {
3659 Value *Arg0 = II->getArgOperand(0);
3660 Value *Arg1 = II->getArgOperand(1);
3661
3662 // Handle mul by zero first:
3664 return replaceInstUsesWith(CI, ConstantAggregateZero::get(II->getType()));
3665 }
3666
3667 // Check for constant LHS & RHS - in this case we just simplify.
3668 bool Zext = (IID == Intrinsic::arm_neon_vmullu ||
3669 IID == Intrinsic::aarch64_neon_umull);
3670 VectorType *NewVT = cast<VectorType>(II->getType());
3671 if (Constant *CV0 = dyn_cast<Constant>(Arg0)) {
3672 if (Constant *CV1 = dyn_cast<Constant>(Arg1)) {
3673 Value *V0 = Builder.CreateIntCast(CV0, NewVT, /*isSigned=*/!Zext);
3674 Value *V1 = Builder.CreateIntCast(CV1, NewVT, /*isSigned=*/!Zext);
3675 return replaceInstUsesWith(CI, Builder.CreateMul(V0, V1));
3676 }
3677
3678 // Couldn't simplify - canonicalize constant to the RHS.
3679 std::swap(Arg0, Arg1);
3680 }
3681
3682 // Handle mul by one:
3683 if (Constant *CV1 = dyn_cast<Constant>(Arg1))
3684 if (ConstantInt *Splat =
3685 dyn_cast_or_null<ConstantInt>(CV1->getSplatValue()))
3686 if (Splat->isOne())
3687 return CastInst::CreateIntegerCast(Arg0, II->getType(),
3688 /*isSigned=*/!Zext);
3689
3690 break;
3691 }
3692 case Intrinsic::arm_neon_aesd:
3693 case Intrinsic::arm_neon_aese:
3694 case Intrinsic::aarch64_crypto_aesd:
3695 case Intrinsic::aarch64_crypto_aese:
3696 case Intrinsic::aarch64_sve_aesd:
3697 case Intrinsic::aarch64_sve_aese: {
3698 Value *DataArg = II->getArgOperand(0);
3699 Value *KeyArg = II->getArgOperand(1);
3700
3701 // Accept zero on either operand.
3702 if (!match(KeyArg, m_ZeroInt()))
3703 std::swap(KeyArg, DataArg);
3704
3705 // Try to use the builtin XOR in AESE and AESD to eliminate a prior XOR
3706 Value *Data, *Key;
3707 if (match(KeyArg, m_ZeroInt()) &&
3708 match(DataArg, m_Xor(m_Value(Data), m_Value(Key)))) {
3709 replaceOperand(*II, 0, Data);
3710 replaceOperand(*II, 1, Key);
3711 return II;
3712 }
3713 break;
3714 }
3715 case Intrinsic::arm_neon_vshifts:
3716 case Intrinsic::arm_neon_vshiftu:
3717 case Intrinsic::aarch64_neon_sshl:
3718 case Intrinsic::aarch64_neon_ushl:
3719 return foldNeonShift(II, *this);
3720 case Intrinsic::hexagon_V6_vandvrt:
3721 case Intrinsic::hexagon_V6_vandvrt_128B: {
3722 // Simplify Q -> V -> Q conversion.
3723 if (auto Op0 = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3724 Intrinsic::ID ID0 = Op0->getIntrinsicID();
3725 if (ID0 != Intrinsic::hexagon_V6_vandqrt &&
3726 ID0 != Intrinsic::hexagon_V6_vandqrt_128B)
3727 break;
3728 Value *Bytes = Op0->getArgOperand(1), *Mask = II->getArgOperand(1);
3729 uint64_t Bytes1 = computeKnownBits(Bytes, Op0).One.getZExtValue();
3730 uint64_t Mask1 = computeKnownBits(Mask, II).One.getZExtValue();
3731 // Check if every byte has common bits in Bytes and Mask.
3732 uint64_t C = Bytes1 & Mask1;
3733 if ((C & 0xFF) && (C & 0xFF00) && (C & 0xFF0000) && (C & 0xFF000000))
3734 return replaceInstUsesWith(*II, Op0->getArgOperand(0));
3735 }
3736 break;
3737 }
3738 case Intrinsic::stackrestore: {
3739 enum class ClassifyResult {
3740 None,
3741 Alloca,
3742 StackRestore,
3743 CallWithSideEffects,
3744 };
3745 auto Classify = [](const Instruction *I) {
3746 if (isa<AllocaInst>(I))
3747 return ClassifyResult::Alloca;
3748
3749 if (auto *CI = dyn_cast<CallInst>(I)) {
3750 if (auto *II = dyn_cast<IntrinsicInst>(CI)) {
3751 if (II->getIntrinsicID() == Intrinsic::stackrestore)
3752 return ClassifyResult::StackRestore;
3753
3754 if (II->mayHaveSideEffects())
3755 return ClassifyResult::CallWithSideEffects;
3756 } else {
3757 // Consider all non-intrinsic calls to be side effects
3758 return ClassifyResult::CallWithSideEffects;
3759 }
3760 }
3761
3762 return ClassifyResult::None;
3763 };
3764
3765 // If the stacksave and the stackrestore are in the same BB, and there is
3766 // no intervening call, alloca, or stackrestore of a different stacksave,
3767 // remove the restore. This can happen when variable allocas are DCE'd.
3768 if (IntrinsicInst *SS = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3769 if (SS->getIntrinsicID() == Intrinsic::stacksave &&
3770 SS->getParent() == II->getParent()) {
3771 BasicBlock::iterator BI(SS);
3772 bool CannotRemove = false;
3773 for (++BI; &*BI != II; ++BI) {
3774 switch (Classify(&*BI)) {
3775 case ClassifyResult::None:
3776 // So far so good, look at next instructions.
3777 break;
3778
3779 case ClassifyResult::StackRestore:
3780 // If we found an intervening stackrestore for a different
3781 // stacksave, we can't remove the stackrestore. Otherwise, continue.
3782 if (cast<IntrinsicInst>(*BI).getArgOperand(0) != SS)
3783 CannotRemove = true;
3784 break;
3785
3786 case ClassifyResult::Alloca:
3787 case ClassifyResult::CallWithSideEffects:
3788 // If we found an alloca, a non-intrinsic call, or an intrinsic
3789 // call with side effects, we can't remove the stackrestore.
3790 CannotRemove = true;
3791 break;
3792 }
3793 if (CannotRemove)
3794 break;
3795 }
3796
3797 if (!CannotRemove)
3798 return eraseInstFromFunction(CI);
3799 }
3800 }
3801
3802 // Scan down this block to see if there is another stack restore in the
3803 // same block without an intervening call/alloca.
3805 Instruction *TI = II->getParent()->getTerminator();
3806 bool CannotRemove = false;
3807 for (++BI; &*BI != TI; ++BI) {
3808 switch (Classify(&*BI)) {
3809 case ClassifyResult::None:
3810 // So far so good, look at next instructions.
3811 break;
3812
3813 case ClassifyResult::StackRestore:
3814 // If there is a stackrestore below this one, remove this one.
3815 return eraseInstFromFunction(CI);
3816
3817 case ClassifyResult::Alloca:
3818 case ClassifyResult::CallWithSideEffects:
3819 // If we found an alloca, a non-intrinsic call, or an intrinsic call
3820 // with side effects (such as llvm.stacksave and llvm.read_register),
3821 // we can't remove the stack restore.
3822 CannotRemove = true;
3823 break;
3824 }
3825 if (CannotRemove)
3826 break;
3827 }
3828
3829 // If the stack restore is in a return, resume, or unwind block and if there
3830 // are no allocas or calls between the restore and the return, nuke the
3831 // restore.
3832 if (!CannotRemove && (isa<ReturnInst>(TI) || isa<ResumeInst>(TI)))
3833 return eraseInstFromFunction(CI);
3834 break;
3835 }
3836 case Intrinsic::lifetime_end:
3837 // Asan needs to poison memory to detect invalid access which is possible
3838 // even for empty lifetime range.
3839 if (II->getFunction()->hasFnAttribute(Attribute::SanitizeAddress) ||
3840 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemory) ||
3841 II->getFunction()->hasFnAttribute(Attribute::SanitizeHWAddress) ||
3842 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag))
3843 break;
3844
3845 if (removeTriviallyEmptyRange(*II, *this, [](const IntrinsicInst &I) {
3846 return I.getIntrinsicID() == Intrinsic::lifetime_start;
3847 }))
3848 return nullptr;
3849 break;
3850 case Intrinsic::assume: {
3851 for (auto [Idx, OBU] : llvm::enumerate(II->operand_bundles())) {
3852 auto RemoveBundle = [&, Idx = Idx]() -> Instruction * {
3853 if (II->getNumOperandBundles() == 1)
3854 return eraseInstFromFunction(*II);
3856 };
3857
3858 switch (getBundleAttrFromOBU(OBU)) {
3859 case BundleAttr::None:
3860 llvm_unreachable("Unexpected Attribute");
3861 case BundleAttr::Align: {
3862 // Try to remove redundant alignment assumptions.
3863 auto [Ptr, _, OffsetPtr, Alignment, Offset] = getAssumeAlignInfo(OBU);
3864
3865 if (!Alignment)
3866 break;
3867
3868 // Remove align 1 and non-power-of-two bundles; they don't add any
3869 // useful information.
3870 if (*Alignment == 1 || !isPowerOf2_64(*Alignment))
3871 return RemoveBundle();
3872
3873 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3874 GEP &&
3875 GEP->getMaxPreservedAlignment(getDataLayout()) >= *Alignment) {
3876 Builder.CreateAlignmentAssumption(
3877 getDataLayout(), GEP->getPointerOperand(), *Alignment,
3878 OffsetPtr ? const_cast<Value *>(OffsetPtr->get()) : nullptr);
3879 return RemoveBundle();
3880 }
3881
3882 if (!Offset)
3883 break;
3884
3885 Value *BasePtr;
3886 const APInt *PtrOffset;
3887 if (match(Ptr.get(), m_PtrAdd(m_Value(BasePtr), m_APInt(PtrOffset)))) {
3888 auto PtrOffsetVal =
3889 PtrOffset->sextOrTrunc(DL.getIndexTypeSizeInBits(Ptr->getType()))
3890 .trySExtValue();
3891 if (!PtrOffsetVal)
3892 break;
3893 Builder.CreateAlignmentAssumption(
3894 DL, BasePtr, *Alignment,
3895 Builder.getInt64(*Offset - *PtrOffsetVal));
3896 return RemoveBundle();
3897 }
3898
3899 // Don't try to remove align assumptions for pointers derived from
3900 // arguments. We might lose information if the function gets inline and
3901 // the align argument attribute disappears.
3902 Value *UO = getUnderlyingObject(Ptr);
3903 if (!UO || isa<Argument>(UO))
3904 break;
3905
3906 // Compute known bits for the pointer and drop the assume if the
3907 // known alignment isn't increased by it.
3908 auto AlignMask = (*Alignment - 1);
3909 if (KnownBits KB = computeKnownBits(Ptr, II);
3910 (KB.Zero & AlignMask) == (~*Offset & AlignMask) &&
3911 (KB.One & AlignMask) == (*Offset & AlignMask))
3912 return RemoveBundle();
3913 break;
3914 }
3915
3916 case BundleAttr::Dereferenceable: {
3917 auto [Ptr, _, Count] = getAssumeDereferenceableInfo(OBU);
3918
3919 if (!Count)
3920 break;
3921
3922 if (*Count == 0 ||
3924 getSimplifyQuery().getWithInstruction(II)))
3925 return RemoveBundle();
3926
3927 break;
3928 }
3929
3930 case BundleAttr::Ignore:
3931 return RemoveBundle();
3932
3933 case BundleAttr::NonNull: {
3934 auto [Ptr] = llvm::getAssumeNonNullInfo(OBU);
3935
3936 // Drop assume if we can prove nonnull without it
3937 if (isKnownNonZero(Ptr, getSimplifyQuery().getWithInstruction(II)))
3938 return RemoveBundle();
3939
3940 // Fold the assume into metadata if it's valid at the load
3941 if (auto *LI = dyn_cast<LoadInst>(Ptr);
3942 LI &&
3943 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3944 MDNode *MD = MDNode::get(II->getContext(), {});
3945 LI->setMetadata(LLVMContext::MD_nonnull, MD);
3946 LI->setMetadata(LLVMContext::MD_noundef, MD);
3947 return RemoveBundle();
3948 }
3949
3950 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3951 GEP && GEP->isInBounds() &&
3952 !NullPointerIsDefined(II->getFunction(),
3953 Ptr->getType()->getPointerAddressSpace())) {
3954 Builder.CreateNonnullAssumption(GEP->stripInBoundsOffsets());
3955 return RemoveBundle();
3956 }
3957
3958 // TODO: apply nonnull return attributes to calls and invokes
3959 break;
3960 }
3961
3962 case BundleAttr::NoUndef: {
3963 auto [Val] = getAssumeNoUndefInfo(OBU);
3964
3966 return RemoveBundle();
3967
3968 if (auto *LI = dyn_cast<LoadInst>(Val);
3969 LI &&
3970 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3971 LI->setMetadata(LLVMContext::MD_noundef,
3972 MDNode::get(II->getContext(), {}));
3973 return RemoveBundle();
3974 }
3975
3976 } break;
3977
3978 case BundleAttr::SeparateStorage: {
3979 auto [Ptr1, Ptr2] = getAssumeSeparateStorageInfo(OBU);
3980 // Separate storage assumptions apply to the underlying allocations, not
3981 // any particular pointer within them. When evaluating the hints for AA
3982 // purposes we getUnderlyingObject them; by precomputing the answers
3983 // here we can avoid having to do so repeatedly there.
3984 auto MaybeSimplifyHint = [&](const Use &U) {
3985 Value *Hint = U.get();
3986 // Not having a limit is safe because InstCombine removes unreachable
3987 // code.
3988 Value *UnderlyingObject = getUnderlyingObject(Hint, /*MaxLookup*/ 0);
3989 if (Hint != UnderlyingObject)
3990 replaceUse(const_cast<Use &>(U), UnderlyingObject);
3991 };
3992 MaybeSimplifyHint(Ptr1);
3993 MaybeSimplifyHint(Ptr2);
3994 } break;
3995
3996 // TODO: Drop these assumes when they are redundant
3997 case BundleAttr::DereferenceableOrNull:
3998 break;
3999
4000 // This cannot be simplified
4001 case BundleAttr::Cold:
4002 break;
4003 }
4004 }
4005
4006 // If the assume has operand bundles, the folds below will never work, so
4007 // don't bother trying.
4008 if (II->hasOperandBundles())
4009 break;
4010
4011 Value *IIOperand = II->getArgOperand(0);
4012
4013 // Canonicalize assume(a && b) -> assume(a); assume(b);
4014 // Note: New assumption intrinsics created here are registered by
4015 // the InstCombineIRInserter object.
4016 Value *A, *B;
4017 if (match(IIOperand, m_LogicalAnd(m_Value(A), m_Value(B)))) {
4018 Builder.CreateAssumption(A);
4019 Builder.CreateAssumption(B);
4020 return eraseInstFromFunction(*II);
4021 }
4022 // assume(!(a || b)) -> assume(!a); assume(!b);
4023 if (match(IIOperand, m_Not(m_LogicalOr(m_Value(A), m_Value(B))))) {
4024 Builder.CreateAssumption(Builder.CreateNot(A));
4025 Builder.CreateAssumption(Builder.CreateNot(B));
4026 return eraseInstFromFunction(*II);
4027 }
4028
4029 // Convert nonnull assume like:
4030 // %A = icmp ne i32* %PTR, null
4031 // call void @llvm.assume(i1 %A)
4032 // into
4033 // call void @llvm.assume(i1 true) [ "nonnull"(i32* %PTR) ]
4034 if (match(
4035 IIOperand,
4038 m_Zero())))) &&
4039 A->getType()->isPointerTy()) {
4040 Builder.CreateNonnullAssumption(A);
4041 return eraseInstFromFunction(*II);
4042 }
4043
4044 // Convert alignment assume like:
4045 // %B = ptrtoint ptr %A to i64
4046 // %C = and i64 %B, Constant
4047 // %D = icmp eq i64 %C, 0
4048 // call void @llvm.assume(i1 %D)
4049 // into
4050 // call void @llvm.assume(i1 true) [ "align"(ptr [[A]], i64 Constant + 1)]
4051 uint64_t AlignMask = 1;
4052 if ((match(IIOperand, m_Not(m_Trunc(m_Value(A)))) ||
4053 match(IIOperand,
4055 m_And(m_Value(A), m_ConstantInt(AlignMask)),
4056 m_Zero())))) {
4057 if (isPowerOf2_64(AlignMask + 1) &&
4059 Builder.CreateAlignmentAssumption(getDataLayout(), A, AlignMask + 1);
4060 return eraseInstFromFunction(*II);
4061 }
4062 }
4063
4064 // Remove assumes on true/false
4065 if (auto *CI = dyn_cast<ConstantInt>(IIOperand);
4066 CI || isa<UndefValue, PoisonValue>(IIOperand)) {
4067 if (!CI || CI->isZero())
4069 return eraseInstFromFunction(*II);
4070 }
4071
4072 // Update the cache of affected values for this assumption (we might be
4073 // here because we just simplified the condition).
4074 AC.updateAffectedValues(cast<AssumeInst>(II));
4075 break;
4076 }
4077 case Intrinsic::experimental_guard: {
4078 // Is this guard followed by another guard? We scan forward over a small
4079 // fixed window of instructions to handle common cases with conditions
4080 // computed between guards.
4081 Instruction *NextInst = II->getNextNode();
4082 for (unsigned i = 0; i < GuardWideningWindow; i++) {
4083 // Note: Using context-free form to avoid compile time blow up
4084 if (!isSafeToSpeculativelyExecute(NextInst))
4085 break;
4086 NextInst = NextInst->getNextNode();
4087 }
4088 Value *NextCond = nullptr;
4089 if (match(NextInst,
4091 Value *CurrCond = II->getArgOperand(0);
4092
4093 // Remove a guard that it is immediately preceded by an identical guard.
4094 // Otherwise canonicalize guard(a); guard(b) -> guard(a & b).
4095 if (CurrCond != NextCond) {
4096 Instruction *MoveI = II->getNextNode();
4097 while (MoveI != NextInst) {
4098 auto *Temp = MoveI;
4099 MoveI = MoveI->getNextNode();
4100 Temp->moveBefore(II->getIterator());
4101 }
4102 replaceOperand(*II, 0, Builder.CreateAnd(CurrCond, NextCond));
4103 }
4104 eraseInstFromFunction(*NextInst);
4105 return II;
4106 }
4107 break;
4108 }
4109 case Intrinsic::vector_insert: {
4110 Value *Vec = II->getArgOperand(0);
4111 Value *SubVec = II->getArgOperand(1);
4112 Value *Idx = II->getArgOperand(2);
4113 auto *DstTy = dyn_cast<FixedVectorType>(II->getType());
4114 auto *VecTy = dyn_cast<FixedVectorType>(Vec->getType());
4115 auto *SubVecTy = dyn_cast<FixedVectorType>(SubVec->getType());
4116
4117 // Only canonicalize if the destination vector, Vec, and SubVec are all
4118 // fixed vectors.
4119 if (DstTy && VecTy && SubVecTy) {
4120 unsigned DstNumElts = DstTy->getNumElements();
4121 unsigned VecNumElts = VecTy->getNumElements();
4122 unsigned SubVecNumElts = SubVecTy->getNumElements();
4123 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4124
4125 // An insert that entirely overwrites Vec with SubVec is a nop.
4126 if (VecNumElts == SubVecNumElts)
4127 return replaceInstUsesWith(CI, SubVec);
4128
4129 // Widen SubVec into a vector of the same width as Vec, since
4130 // shufflevector requires the two input vectors to be the same width.
4131 // Elements beyond the bounds of SubVec within the widened vector are
4132 // undefined.
4133 SmallVector<int, 8> WidenMask;
4134 unsigned i;
4135 for (i = 0; i != SubVecNumElts; ++i)
4136 WidenMask.push_back(i);
4137 for (; i != VecNumElts; ++i)
4138 WidenMask.push_back(PoisonMaskElem);
4139
4140 Value *WidenShuffle = Builder.CreateShuffleVector(SubVec, WidenMask);
4141
4143 for (unsigned i = 0; i != IdxN; ++i)
4144 Mask.push_back(i);
4145 for (unsigned i = DstNumElts; i != DstNumElts + SubVecNumElts; ++i)
4146 Mask.push_back(i);
4147 for (unsigned i = IdxN + SubVecNumElts; i != DstNumElts; ++i)
4148 Mask.push_back(i);
4149
4150 Value *Shuffle = Builder.CreateShuffleVector(Vec, WidenShuffle, Mask);
4151 return replaceInstUsesWith(CI, Shuffle);
4152 }
4153 break;
4154 }
4155 case Intrinsic::vector_extract: {
4156 Value *Vec = II->getArgOperand(0);
4157 Value *Idx = II->getArgOperand(1);
4158
4159 Type *ReturnType = II->getType();
4160 // (extract_vector (insert_vector InsertTuple, InsertValue, InsertIdx),
4161 // ExtractIdx)
4162 unsigned ExtractIdx = cast<ConstantInt>(Idx)->getZExtValue();
4163 Value *InsertTuple, *InsertIdx, *InsertValue;
4165 m_Value(InsertValue),
4166 m_Value(InsertIdx))) &&
4167 InsertValue->getType() == ReturnType) {
4168 unsigned Index = cast<ConstantInt>(InsertIdx)->getZExtValue();
4169 // Case where we get the same index right after setting it.
4170 // extract.vector(insert.vector(InsertTuple, InsertValue, Idx), Idx) -->
4171 // InsertValue
4172 if (ExtractIdx == Index)
4173 return replaceInstUsesWith(CI, InsertValue);
4174 // If we are getting a different index than what was set in the
4175 // insert.vector intrinsic. We can just set the input tuple to the one up
4176 // in the chain. extract.vector(insert.vector(InsertTuple, InsertValue,
4177 // InsertIndex), ExtractIndex)
4178 // --> extract.vector(InsertTuple, ExtractIndex)
4179 else
4180 return replaceOperand(CI, 0, InsertTuple);
4181 }
4182
4183 ConstantInt *ALMUpperBound;
4185 m_Value(), m_ConstantInt(ALMUpperBound)))) {
4186 const auto &Attrs = II->getFunction()->getAttributes().getFnAttrs();
4187 unsigned VScaleMin = Attrs.getVScaleRangeMin();
4188 unsigned ScaleFactor =
4189 cast<VectorType>(ReturnType)->isScalableTy() ? VScaleMin : 1;
4190 if (ExtractIdx * ScaleFactor >= ALMUpperBound->getZExtValue())
4191 return replaceInstUsesWith(CI,
4192 ConstantVector::getNullValue(ReturnType));
4193 }
4194
4195 auto *DstTy = dyn_cast<VectorType>(ReturnType);
4196 auto *VecTy = dyn_cast<VectorType>(Vec->getType());
4197
4198 if (DstTy && VecTy) {
4199 auto DstEltCnt = DstTy->getElementCount();
4200 auto VecEltCnt = VecTy->getElementCount();
4201 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4202
4203 // Extracting the entirety of Vec is a nop.
4204 if (DstEltCnt == VecTy->getElementCount()) {
4205 replaceInstUsesWith(CI, Vec);
4206 return eraseInstFromFunction(CI);
4207 }
4208
4209 // Only canonicalize to shufflevector if the destination vector and
4210 // Vec are fixed vectors.
4211 if (VecEltCnt.isScalable() || DstEltCnt.isScalable())
4212 break;
4213
4215 for (unsigned i = 0; i != DstEltCnt.getKnownMinValue(); ++i)
4216 Mask.push_back(IdxN + i);
4217
4218 Value *Shuffle = Builder.CreateShuffleVector(Vec, Mask);
4219 return replaceInstUsesWith(CI, Shuffle);
4220 }
4221 break;
4222 }
4223 case Intrinsic::experimental_vp_reverse: {
4224 Value *X;
4225 Value *Vec = II->getArgOperand(0);
4226 Value *Mask = II->getArgOperand(1);
4227 if (!match(Mask, m_AllOnes()))
4228 break;
4229 Value *EVL = II->getArgOperand(2);
4230 // TODO: Canonicalize experimental.vp.reverse after unop/binops?
4231 // rev(unop rev(X)) --> unop X
4232 if (match(Vec,
4234 m_Value(X), m_AllOnes(), m_Specific(EVL)))))) {
4235 auto *OldUnOp = cast<UnaryOperator>(Vec);
4237 OldUnOp->getOpcode(), X, OldUnOp, OldUnOp->getName(),
4238 II->getIterator());
4239 return replaceInstUsesWith(CI, NewUnOp);
4240 }
4241 break;
4242 }
4243 case Intrinsic::vector_reduce_or:
4244 case Intrinsic::vector_reduce_and: {
4245 // Canonicalize logical or/and reductions:
4246 // Or reduction for i1 is represented as:
4247 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4248 // %res = cmp ne iReduxWidth %val, 0
4249 // And reduction for i1 is represented as:
4250 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4251 // %res = cmp eq iReduxWidth %val, 11111
4252 Value *Arg = II->getArgOperand(0);
4253 Value *Vect;
4254
4255 if (Value *NewOp =
4256 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4257 replaceUse(II->getOperandUse(0), NewOp);
4258 return II;
4259 }
4260
4261 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4262 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4263 if (FTy->getElementType() == Builder.getInt1Ty()) {
4264 Value *Res = Builder.CreateBitCast(
4265 Vect, Builder.getIntNTy(FTy->getNumElements()));
4266 if (IID == Intrinsic::vector_reduce_and) {
4267 Res = Builder.CreateICmpEQ(
4269 } else {
4270 assert(IID == Intrinsic::vector_reduce_or &&
4271 "Expected or reduction.");
4272 Res = Builder.CreateIsNotNull(Res);
4273 }
4274 if (Arg != Vect)
4275 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4276 II->getType());
4277 return replaceInstUsesWith(CI, Res);
4278 }
4279 }
4280 [[fallthrough]];
4281 }
4282 case Intrinsic::vector_reduce_add: {
4283 if (IID == Intrinsic::vector_reduce_add) {
4284 // Convert vector_reduce_add(ZExt(<n x i1>)) to
4285 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4286 // Convert vector_reduce_add(SExt(<n x i1>)) to
4287 // -ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4288 // Convert vector_reduce_add(<n x i1>) to
4289 // Trunc(ctpop(bitcast <n x i1> to in)).
4290 Value *Arg = II->getArgOperand(0);
4291 Value *Vect;
4292
4293 if (Value *NewOp =
4294 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4295 replaceUse(II->getOperandUse(0), NewOp);
4296 return II;
4297 }
4298
4299 // vector.reduce.add.vNiM(splat(%x)) -> mul(%x, N)
4300 if (Value *Splat = getSplatValue(Arg)) {
4301 ElementCount VecToReduceCount =
4302 cast<VectorType>(Arg->getType())->getElementCount();
4303 if (VecToReduceCount.isFixed()) {
4304 unsigned VectorSize = VecToReduceCount.getFixedValue();
4305 return BinaryOperator::CreateMul(
4306 Splat,
4307 ConstantInt::get(Splat->getType(), VectorSize, /*IsSigned=*/false,
4308 /*ImplicitTrunc=*/true));
4309 }
4310 }
4311
4312 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4313 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4314 if (FTy->getElementType() == Builder.getInt1Ty()) {
4315 Value *V = Builder.CreateBitCast(
4316 Vect, Builder.getIntNTy(FTy->getNumElements()));
4317 Value *Res = Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
4318 Res = Builder.CreateZExtOrTrunc(Res, II->getType());
4319 if (Arg != Vect &&
4320 cast<Instruction>(Arg)->getOpcode() == Instruction::SExt)
4321 Res = Builder.CreateNeg(Res);
4322 return replaceInstUsesWith(CI, Res);
4323 }
4324 }
4325 }
4326 [[fallthrough]];
4327 }
4328 case Intrinsic::vector_reduce_xor: {
4329 if (IID == Intrinsic::vector_reduce_xor) {
4330 // Exclusive disjunction reduction over the vector with
4331 // (potentially-extended) i1 element type is actually a
4332 // (potentially-extended) arithmetic `add` reduction over the original
4333 // non-extended value:
4334 // vector_reduce_xor(?ext(<n x i1>))
4335 // -->
4336 // ?ext(vector_reduce_add(<n x i1>))
4337 Value *Arg = II->getArgOperand(0);
4338 Value *Vect;
4339
4340 if (Value *NewOp =
4341 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4342 replaceUse(II->getOperandUse(0), NewOp);
4343 return II;
4344 }
4345
4346 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4347 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4348 if (VTy->getElementType() == Builder.getInt1Ty()) {
4349 Value *Res = Builder.CreateAddReduce(Vect);
4350 if (Arg != Vect)
4351 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4352 II->getType());
4353 return replaceInstUsesWith(CI, Res);
4354 }
4355 }
4356 }
4357 [[fallthrough]];
4358 }
4359 case Intrinsic::vector_reduce_mul: {
4360 if (IID == Intrinsic::vector_reduce_mul) {
4361 Value *Arg = II->getArgOperand(0);
4362
4363 if (Value *NewOp =
4364 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4365 replaceUse(II->getOperandUse(0), NewOp);
4366 return II;
4367 }
4368
4369 // vector_reduce_mul(zext(<n x i1>)), or
4370 // vector_reduce_mul(sext(<n x i1>)) (if n is even) -->
4371 // zext(vector_reduce_and(<n x i1>)).
4372 // (The sext case doesn't work if n is odd because multiplying an odd
4373 // number of -1's produces -1, not 1.)
4374 Value *Vect;
4375 bool IsZext = match(Arg, m_ZExt(m_Value(Vect))) &&
4376 Vect->getType()->isIntOrIntVectorTy(1);
4377 bool IsSext =
4378 match(Arg, m_SExt(m_Value(Vect))) &&
4379 Vect->getType()->isIntOrIntVectorTy(1) &&
4380 cast<VectorType>(Vect->getType())->getElementCount().isKnownEven();
4381 if (IsZext || IsSext) {
4382 Value *Res = Builder.CreateAndReduce(Vect);
4383 return CastInst::Create(Instruction::ZExt, Res, II->getType());
4384 }
4385
4386 // vector_reduce_mul(<n x i1>) --> vector_reduce_and(<n x i1>)
4387 if (Arg->getType()->isIntOrIntVectorTy(1))
4388 return replaceInstUsesWith(CI, Builder.CreateAndReduce(Arg));
4389 }
4390 [[fallthrough]];
4391 }
4392 case Intrinsic::vector_reduce_umin:
4393 case Intrinsic::vector_reduce_umax: {
4394 if (IID == Intrinsic::vector_reduce_umin ||
4395 IID == Intrinsic::vector_reduce_umax) {
4396 // UMin/UMax reduction over the vector with (potentially-extended)
4397 // i1 element type is actually a (potentially-extended)
4398 // logical `and`/`or` reduction over the original non-extended value:
4399 // vector_reduce_u{min,max}(?ext(<n x i1>))
4400 // -->
4401 // ?ext(vector_reduce_{and,or}(<n x i1>))
4402 Value *Arg = II->getArgOperand(0);
4403 Value *Vect;
4404
4405 if (Value *NewOp =
4406 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4407 replaceUse(II->getOperandUse(0), NewOp);
4408 return II;
4409 }
4410
4411 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4412 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4413 if (VTy->getElementType() == Builder.getInt1Ty()) {
4414 Value *Res = IID == Intrinsic::vector_reduce_umin
4415 ? Builder.CreateAndReduce(Vect)
4416 : Builder.CreateOrReduce(Vect);
4417 if (Arg != Vect)
4418 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4419 II->getType());
4420 return replaceInstUsesWith(CI, Res);
4421 }
4422 }
4423 }
4424 [[fallthrough]];
4425 }
4426 case Intrinsic::vector_reduce_smin:
4427 case Intrinsic::vector_reduce_smax: {
4428 if (IID == Intrinsic::vector_reduce_smin ||
4429 IID == Intrinsic::vector_reduce_smax) {
4430 // SMin/SMax reduction over the vector with (potentially-extended)
4431 // i1 element type is actually a (potentially-extended)
4432 // logical `and`/`or` reduction over the original non-extended value:
4433 // vector_reduce_s{min,max}(<n x i1>)
4434 // -->
4435 // vector_reduce_{or,and}(<n x i1>)
4436 // and
4437 // vector_reduce_s{min,max}(sext(<n x i1>))
4438 // -->
4439 // sext(vector_reduce_{or,and}(<n x i1>))
4440 // and
4441 // vector_reduce_s{min,max}(zext(<n x i1>))
4442 // -->
4443 // zext(vector_reduce_{and,or}(<n x i1>))
4444 Value *Arg = II->getArgOperand(0);
4445 Value *Vect;
4446
4447 if (Value *NewOp =
4448 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4449 replaceUse(II->getOperandUse(0), NewOp);
4450 return II;
4451 }
4452
4453 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4454 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4455 if (VTy->getElementType() == Builder.getInt1Ty()) {
4456 Instruction::CastOps ExtOpc = Instruction::CastOps::CastOpsEnd;
4457 if (Arg != Vect)
4458 ExtOpc = cast<CastInst>(Arg)->getOpcode();
4459 Value *Res = ((IID == Intrinsic::vector_reduce_smin) ==
4460 (ExtOpc == Instruction::CastOps::ZExt))
4461 ? Builder.CreateAndReduce(Vect)
4462 : Builder.CreateOrReduce(Vect);
4463 if (Arg != Vect)
4464 Res = Builder.CreateCast(ExtOpc, Res, II->getType());
4465 return replaceInstUsesWith(CI, Res);
4466 }
4467 }
4468 }
4469 [[fallthrough]];
4470 }
4471 case Intrinsic::vector_reduce_fmax:
4472 case Intrinsic::vector_reduce_fmin:
4473 case Intrinsic::vector_reduce_fadd:
4474 case Intrinsic::vector_reduce_fmul: {
4475 bool CanReorderLanes = (IID != Intrinsic::vector_reduce_fadd &&
4476 IID != Intrinsic::vector_reduce_fmul) ||
4477 II->hasAllowReassoc();
4478 const unsigned ArgIdx = (IID == Intrinsic::vector_reduce_fadd ||
4479 IID == Intrinsic::vector_reduce_fmul)
4480 ? 1
4481 : 0;
4482 Value *Arg = II->getArgOperand(ArgIdx);
4483 if (Value *NewOp = simplifyReductionOperand(Arg, CanReorderLanes)) {
4484 replaceUse(II->getOperandUse(ArgIdx), NewOp);
4485 return nullptr;
4486 }
4487 break;
4488 }
4489 case Intrinsic::is_fpclass: {
4490 if (Instruction *I = foldIntrinsicIsFPClass(*II))
4491 return I;
4492 break;
4493 }
4494 case Intrinsic::threadlocal_address: {
4495 Align MinAlign = getKnownAlignment(II->getArgOperand(0), DL, II, &AC, &DT);
4496 MaybeAlign Align = II->getRetAlign();
4497 if (MinAlign > Align.valueOrOne()) {
4498 II->addRetAttr(Attribute::getWithAlignment(II->getContext(), MinAlign));
4499 return II;
4500 }
4501 break;
4502 }
4503 case Intrinsic::fptoui_sat:
4504 case Intrinsic::fptosi_sat:
4505 if (Instruction *I = foldItoFPtoI(*II))
4506 return I;
4507 break;
4508 case Intrinsic::frexp: {
4509 // frexp(frexp(x).fract) -> { frexp(x).fract, 0 }: the fraction operand is
4510 // already normalized, so the first result is idempotent and the second is
4511 // zero.
4512 if (match(II->getArgOperand(0),
4514 Value *Res = Builder.CreateInsertValue(PoisonValue::get(II->getType()),
4515 II->getArgOperand(0), 0);
4516 Res = Builder.CreateInsertValue(
4517 Res, Constant::getNullValue(II->getType()->getStructElementType(1)),
4518 1);
4519 return replaceInstUsesWith(*II, Res);
4520 }
4521 break;
4522 }
4523 case Intrinsic::get_active_lane_mask: {
4524 const APInt *Op0, *Op1;
4525 if (match(II->getOperand(0), m_StrictlyPositive(Op0)) &&
4526 match(II->getOperand(1), m_APInt(Op1))) {
4527 Type *OpTy = II->getOperand(0)->getType();
4528 return replaceInstUsesWith(
4529 *II, Builder.CreateIntrinsic(
4530 II->getType(), Intrinsic::get_active_lane_mask,
4531 {Constant::getNullValue(OpTy),
4532 ConstantInt::get(OpTy, Op1->usub_sat(*Op0))}));
4533 }
4534 break;
4535 }
4536 case Intrinsic::experimental_get_vector_length: {
4537 // get.vector.length(Cnt, MaxLanes) --> Cnt when Cnt <= MaxLanes
4538 unsigned BitWidth =
4539 std::max(II->getArgOperand(0)->getType()->getScalarSizeInBits(),
4540 II->getType()->getScalarSizeInBits());
4541 ConstantRange Cnt =
4542 computeConstantRangeIncludingKnownBits(II->getArgOperand(0), false,
4543 SQ.getWithInstruction(II))
4545 ConstantRange MaxLanes = cast<ConstantInt>(II->getArgOperand(1))
4546 ->getValue()
4547 .zextOrTrunc(Cnt.getBitWidth());
4548 if (cast<ConstantInt>(II->getArgOperand(2))->isOne())
4549 MaxLanes = MaxLanes.multiply(
4550 getVScaleRange(II->getFunction(), Cnt.getBitWidth()));
4551
4552 if (Cnt.icmp(CmpInst::ICMP_ULE, MaxLanes))
4553 return replaceInstUsesWith(
4554 *II, Builder.CreateZExtOrTrunc(II->getArgOperand(0), II->getType()));
4555 return nullptr;
4556 }
4557 default: {
4558 // Handle target specific intrinsics
4559 std::optional<Instruction *> V = targetInstCombineIntrinsic(*II);
4560 if (V)
4561 return *V;
4562 break;
4563 }
4564 }
4565
4566 // Try to fold intrinsic into select/phi operands. This is legal if:
4567 // * The intrinsic is speculatable.
4568 // * The operand is one of the following:
4569 // - a phi.
4570 // - a select with a scalar condition.
4571 // - a select with a vector condition and II is not a cross lane operation.
4573 for (Value *Op : II->args()) {
4574 if (auto *Sel = dyn_cast<SelectInst>(Op)) {
4575 bool IsVectorCond = Sel->getCondition()->getType()->isVectorTy();
4576 if (IsVectorCond &&
4577 (!isNotCrossLaneOperation(II) || !II->getType()->isVectorTy()))
4578 continue;
4579 // Don't replace a scalar select with a more expensive vector select if
4580 // we can't simplify both arms of the select.
4581 bool SimplifyBothArms =
4582 !Op->getType()->isVectorTy() && II->getType()->isVectorTy();
4584 *II, Sel, /*FoldWithMultiUse=*/false, SimplifyBothArms))
4585 return R;
4586 }
4587 if (auto *Phi = dyn_cast<PHINode>(Op))
4588 if (Instruction *R = foldOpIntoPhi(*II, Phi))
4589 return R;
4590 }
4591 }
4592
4594 return Shuf;
4595
4597 return replaceInstUsesWith(*II, Reverse);
4598
4600 return replaceInstUsesWith(*II, Res);
4601
4602 // Some intrinsics (like experimental_gc_statepoint) can be used in invoke
4603 // context, so it is handled in visitCallBase and we should trigger it.
4604 return visitCallBase(*II);
4605}
4606
4607// Fence instruction simplification
4609 auto *NFI = dyn_cast<FenceInst>(FI.getNextNode());
4610 // This check is solely here to handle arbitrary target-dependent syncscopes.
4611 // TODO: Can remove if does not matter in practice.
4612 if (NFI && FI.isIdenticalTo(NFI))
4613 return eraseInstFromFunction(FI);
4614
4615 // Returns true if FI1 is identical or stronger fence than FI2.
4616 auto isIdenticalOrStrongerFence = [](FenceInst *FI1, FenceInst *FI2) {
4617 auto FI1SyncScope = FI1->getSyncScopeID();
4618 // Consider same scope, where scope is global or single-thread.
4619 if (FI1SyncScope != FI2->getSyncScopeID() ||
4620 (FI1SyncScope != SyncScope::System &&
4621 FI1SyncScope != SyncScope::SingleThread))
4622 return false;
4623
4624 return isAtLeastOrStrongerThan(FI1->getOrdering(), FI2->getOrdering());
4625 };
4626 if (NFI && isIdenticalOrStrongerFence(NFI, &FI))
4627 return eraseInstFromFunction(FI);
4628
4629 if (auto *PFI = dyn_cast_or_null<FenceInst>(FI.getPrevNode()))
4630 if (isIdenticalOrStrongerFence(PFI, &FI))
4631 return eraseInstFromFunction(FI);
4632 return nullptr;
4633}
4634
4635// InvokeInst simplification
4637 return visitCallBase(II);
4638}
4639
4640// CallBrInst simplification
4642 return visitCallBase(CBI);
4643}
4644
4645// A simple parser for format string specifiers for the purposes of the
4646// modular-format attribute. In the case of malformed format strings this might
4647// under or over report the specifiers present, but such cases are undefined
4648// behavior.
4650 Bitset<256> Specifiers;
4651 for (size_t I = 0; I < FormatStr.size(); ++I) {
4652 if (FormatStr[I] != '%')
4653 continue;
4654
4655 // Check for escaped '%'.
4656 if (I + 1 < FormatStr.size() && FormatStr[I + 1] == '%') {
4657 ++I; // Skip the second '%'.
4658 continue;
4659 }
4660
4661 // Scan past allowed prefix characters.
4662 size_t J =
4663 FormatStr.find_first_not_of("0123456789-+ #0$.*'hlLjztqwvI", I + 1);
4664 if (J == StringRef::npos)
4665 break;
4666
4667 Specifiers.set(static_cast<unsigned char>(FormatStr[J]));
4668 I = J; // Resume search from after the specifier.
4669 }
4670 return Specifiers;
4671}
4672
4673static bool isAspectNeeded(StringRef Aspect, CallInst *CI,
4674 std::optional<unsigned> FirstArgIdx,
4675 const std::optional<Bitset<256>> &Specifiers) {
4676 if (Aspect == "float") {
4677 if (Specifiers) {
4678 static constexpr Bitset<256> FloatSpecifiers{'f', 'F', 'e', 'E',
4679 'g', 'G', 'a', 'A'};
4680 return (*Specifiers & FloatSpecifiers).any();
4681 }
4682 // Fallback to type-based check for dynamic format string.
4683 if (!FirstArgIdx)
4684 return true;
4685 return llvm::any_of(
4686 llvm::make_range(std::next(CI->arg_begin(), *FirstArgIdx),
4687 CI->arg_end()),
4688 [](Value *V) { return V->getType()->isFloatingPointTy(); });
4689 }
4690 if (Aspect == "fixed") {
4691 if (Specifiers) {
4692 static constexpr Bitset<256> FixedSpecifiers{'r', 'R', 'k', 'K'};
4693 return (*Specifiers & FixedSpecifiers).any();
4694 }
4695 // Fallback for fixed-point: assume needed if format is dynamic.
4696 return true;
4697 }
4698 // Unknown aspects are always considered to be needed.
4699 return true;
4700}
4701
4702static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M,
4703 IRBuilderBase &B) {
4704 SmallString<20> Name = ImplName;
4705 Name += '_';
4706 Name += Aspect;
4707 LLVMContext &Ctx = M->getContext();
4708 Function *RelocNoneFn =
4709 Intrinsic::getOrInsertDeclaration(M, Intrinsic::reloc_none);
4710 B.CreateCall(RelocNoneFn,
4711 {MetadataAsValue::get(Ctx, MDString::get(Ctx, Name))});
4712}
4713
4715 if (!CI->hasFnAttr("modular-format"))
4716 return nullptr;
4717
4719 llvm::split(CI->getFnAttr("modular-format").getValueAsString(), ','));
4720 if (Args.size() < 5)
4721 return nullptr;
4722
4723 StringRef FormatIdxStr = Args[1];
4724 StringRef FirstArgIdxStr = Args[2];
4725 StringRef FnName = Args[3];
4726 StringRef ImplName = Args[4];
4728
4729 unsigned FormatIdx;
4730 std::optional<unsigned> FirstArgIdx;
4731 [[maybe_unused]] bool Error;
4732 Error = FormatIdxStr.getAsInteger(10, FormatIdx);
4733 assert(!Error && "invalid format arg index");
4734 --FormatIdx; // 1-based to 0-based
4735
4736 FirstArgIdx.emplace();
4737 Error = FirstArgIdxStr.getAsInteger(10, *FirstArgIdx);
4738 assert(!Error && "invalid first arg index");
4739 if (*FirstArgIdx > 0)
4740 --*FirstArgIdx; // 1-based to 0-based
4741 else
4742 FirstArgIdx.reset();
4743
4744 if (AllAspects.empty())
4745 return nullptr;
4746
4747 Value *FormatVal = CI->getArgOperand(FormatIdx);
4748 StringRef FormatStr;
4749
4750 std::optional<Bitset<256>> Specifiers;
4751 if (getConstantStringInfo(FormatVal, FormatStr))
4752 Specifiers = parseFormatStringSpecifiers(FormatStr);
4753
4754 SmallVector<StringRef> NeededAspects;
4755 for (StringRef Aspect : AllAspects)
4756 if (isAspectNeeded(Aspect, CI, FirstArgIdx, Specifiers))
4757 NeededAspects.push_back(Aspect);
4758
4759 if (NeededAspects.size() == AllAspects.size())
4760 return nullptr;
4761
4762 Module *M = CI->getModule();
4763 LLVMContext &Ctx = M->getContext();
4764 Function *Callee = CI->getCalledFunction();
4765 FunctionCallee ModularFn = M->getOrInsertFunction(
4766 FnName, Callee->getFunctionType(),
4767 Callee->getAttributes().removeFnAttribute(Ctx, "modular-format"));
4768 CallInst *New = cast<CallInst>(CI->clone());
4769 New->setCalledFunction(ModularFn);
4770 New->removeFnAttr("modular-format");
4771 B.Insert(New);
4772
4773 llvm::sort(NeededAspects);
4774 for (StringRef Request : NeededAspects)
4775 referenceAspect(Request, ImplName, M, B);
4776
4777 return New;
4778}
4779
4780Instruction *InstCombinerImpl::tryOptimizeCall(CallInst *CI) {
4781 if (!CI->getCalledFunction()) return nullptr;
4782
4783 // Skip optimizing notail and musttail calls so
4784 // LibCallSimplifier::optimizeCall doesn't have to preserve those invariants.
4785 // LibCallSimplifier::optimizeCall should try to preserve tail calls though.
4786 if (CI->isMustTailCall() || CI->isNoTailCall())
4787 return nullptr;
4788
4789 auto InstCombineRAUW = [this](Instruction *From, Value *With) {
4790 replaceInstUsesWith(*From, With);
4791 };
4792 auto InstCombineErase = [this](Instruction *I) {
4794 };
4795 LibCallSimplifier Simplifier(DL, &TLI, &DT, &DC, &AC, ORE, BFI, PSI,
4796 InstCombineRAUW, InstCombineErase);
4797 if (Value *With = Simplifier.optimizeCall(CI, Builder)) {
4798 ++NumSimplified;
4799 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4800 }
4801 if (Value *With = optimizeModularFormat(CI, Builder)) {
4802 ++NumSimplified;
4803 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4804 }
4805
4806 return nullptr;
4807}
4808
4810 // Strip off at most one level of pointer casts, looking for an alloca. This
4811 // is good enough in practice and simpler than handling any number of casts.
4812 Value *Underlying = TrampMem->stripPointerCasts();
4813 if (Underlying != TrampMem &&
4814 (!Underlying->hasOneUse() || Underlying->user_back() != TrampMem))
4815 return nullptr;
4816 if (!isa<AllocaInst>(Underlying))
4817 return nullptr;
4818
4819 IntrinsicInst *InitTrampoline = nullptr;
4820 for (User *U : TrampMem->users()) {
4822 if (!II)
4823 return nullptr;
4824 if (II->getIntrinsicID() == Intrinsic::init_trampoline) {
4825 if (InitTrampoline)
4826 // More than one init_trampoline writes to this value. Give up.
4827 return nullptr;
4828 InitTrampoline = II;
4829 continue;
4830 }
4831 if (II->getIntrinsicID() == Intrinsic::adjust_trampoline)
4832 // Allow any number of calls to adjust.trampoline.
4833 continue;
4834 return nullptr;
4835 }
4836
4837 // No call to init.trampoline found.
4838 if (!InitTrampoline)
4839 return nullptr;
4840
4841 // Check that the alloca is being used in the expected way.
4842 if (InitTrampoline->getOperand(0) != TrampMem)
4843 return nullptr;
4844
4845 return InitTrampoline;
4846}
4847
4849 Value *TrampMem) {
4850 // Visit all the previous instructions in the basic block, and try to find a
4851 // init.trampoline which has a direct path to the adjust.trampoline.
4852 for (BasicBlock::iterator I = AdjustTramp->getIterator(),
4853 E = AdjustTramp->getParent()->begin();
4854 I != E;) {
4855 Instruction *Inst = &*--I;
4857 if (II->getIntrinsicID() == Intrinsic::init_trampoline &&
4858 II->getOperand(0) == TrampMem)
4859 return II;
4860 if (Inst->mayWriteToMemory())
4861 return nullptr;
4862 }
4863 return nullptr;
4864}
4865
4866// Given a call to llvm.adjust.trampoline, find and return the corresponding
4867// call to llvm.init.trampoline if the call to the trampoline can be optimized
4868// to a direct call to a function. Otherwise return NULL.
4870 Callee = Callee->stripPointerCasts();
4871 IntrinsicInst *AdjustTramp = dyn_cast<IntrinsicInst>(Callee);
4872 if (!AdjustTramp ||
4873 AdjustTramp->getIntrinsicID() != Intrinsic::adjust_trampoline)
4874 return nullptr;
4875
4876 Value *TrampMem = AdjustTramp->getOperand(0);
4877
4879 return IT;
4880 if (IntrinsicInst *IT = findInitTrampolineFromBB(AdjustTramp, TrampMem))
4881 return IT;
4882 return nullptr;
4883}
4884
4885Instruction *InstCombinerImpl::foldPtrAuthIntrinsicCallee(CallBase &Call) {
4886 const Value *Callee = Call.getCalledOperand();
4887 const auto *IPC = dyn_cast<IntToPtrInst>(Callee);
4888 if (!IPC || !IPC->isNoopCast(DL))
4889 return nullptr;
4890
4891 const auto *II = dyn_cast<IntrinsicInst>(IPC->getOperand(0));
4892 if (!II)
4893 return nullptr;
4894
4895 Intrinsic::ID IIID = II->getIntrinsicID();
4896 if (IIID != Intrinsic::ptrauth_resign && IIID != Intrinsic::ptrauth_sign)
4897 return nullptr;
4898
4899 // Isolate the ptrauth bundle from the others.
4900 std::optional<OperandBundleUse> PtrAuthBundleOrNone;
4902 for (unsigned BI = 0, BE = Call.getNumOperandBundles(); BI != BE; ++BI) {
4903 OperandBundleUse Bundle = Call.getOperandBundleAt(BI);
4904 if (Bundle.getTagID() == LLVMContext::OB_ptrauth)
4905 PtrAuthBundleOrNone = Bundle;
4906 else
4907 NewBundles.emplace_back(Bundle);
4908 }
4909
4910 if (!PtrAuthBundleOrNone)
4911 return nullptr;
4912
4913 Value *NewCallee = nullptr;
4914 switch (IIID) {
4915 // call(ptrauth.resign(p)), ["ptrauth"()] -> call p, ["ptrauth"()]
4916 // assuming the call bundle and the sign operands match.
4917 case Intrinsic::ptrauth_resign: {
4918 // Resign result key should match bundle.
4919 if (II->getOperand(3) != PtrAuthBundleOrNone->Inputs[0])
4920 return nullptr;
4921 // Resign result discriminator should match bundle.
4922 if (II->getOperand(4) != PtrAuthBundleOrNone->Inputs[1])
4923 return nullptr;
4924
4925 // Resign input (auth) key should also match: we can't change the key on
4926 // the new call we're generating, because we don't know what keys are valid.
4927 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4928 return nullptr;
4929
4930 Value *NewBundleOps[] = {II->getOperand(1), II->getOperand(2)};
4931 NewBundles.emplace_back("ptrauth", NewBundleOps);
4932 NewCallee = II->getOperand(0);
4933 break;
4934 }
4935
4936 // call(ptrauth.sign(p)), ["ptrauth"()] -> call p
4937 // assuming the call bundle and the sign operands match.
4938 // Non-ptrauth indirect calls are undesirable, but so is ptrauth.sign.
4939 case Intrinsic::ptrauth_sign: {
4940 // Sign key should match bundle.
4941 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4942 return nullptr;
4943 // Sign discriminator should match bundle.
4944 if (II->getOperand(2) != PtrAuthBundleOrNone->Inputs[1])
4945 return nullptr;
4946 NewCallee = II->getOperand(0);
4947 break;
4948 }
4949 default:
4950 llvm_unreachable("unexpected intrinsic ID");
4951 }
4952
4953 if (!NewCallee)
4954 return nullptr;
4955
4956 NewCallee = Builder.CreateBitOrPointerCast(NewCallee, Callee->getType());
4957 CallBase *NewCall = CallBase::Create(&Call, NewBundles);
4958 NewCall->setCalledOperand(NewCallee);
4959 return NewCall;
4960}
4961
4962Instruction *InstCombinerImpl::foldPtrAuthConstantCallee(CallBase &Call) {
4964 if (!CPA)
4965 return nullptr;
4966
4967 auto *CalleeF = dyn_cast<Function>(CPA->getPointer());
4968 // If the ptrauth constant isn't based on a function pointer, bail out.
4969 if (!CalleeF)
4970 return nullptr;
4971
4972 // Inspect the call ptrauth bundle to check it matches the ptrauth constant.
4974 if (!PAB)
4975 return nullptr;
4976
4977 auto *Key = cast<ConstantInt>(PAB->Inputs[0]);
4978 Value *Discriminator = PAB->Inputs[1];
4979
4980 // If the bundle doesn't match, this is probably going to fail to auth.
4981 if (!CPA->isKnownCompatibleWith(Key, Discriminator, DL))
4982 return nullptr;
4983
4984 // If the bundle matches the constant, proceed in making this a direct call.
4986 NewCall->setCalledOperand(CalleeF);
4987 return NewCall;
4988}
4989
4990bool InstCombinerImpl::annotateAnyAllocSite(CallBase &Call,
4991 const TargetLibraryInfo *TLI) {
4992 // Note: We only handle cases which can't be driven from generic attributes
4993 // here. So, for example, nonnull and noalias (which are common properties
4994 // of some allocation functions) are expected to be handled via annotation
4995 // of the respective allocator declaration with generic attributes.
4996 bool Changed = false;
4997
4998 if (!Call.getType()->isPointerTy())
4999 return Changed;
5000
5001 std::optional<APInt> Size = getAllocSize(&Call, TLI);
5002 if (Size && *Size != 0) {
5003 // TODO: We really should just emit deref_or_null here and then
5004 // let the generic inference code combine that with nonnull.
5005 if (Call.hasRetAttr(Attribute::NonNull)) {
5006 Changed = !Call.hasRetAttr(Attribute::Dereferenceable);
5008 Call.getContext(), Size->getLimitedValue()));
5009 } else {
5010 Changed = !Call.hasRetAttr(Attribute::DereferenceableOrNull);
5012 Call.getContext(), Size->getLimitedValue()));
5013 }
5014 }
5015
5016 // Add alignment attribute if alignment is a power of two constant.
5018 if (!Alignment)
5019 return Changed;
5020
5021 ConstantInt *AlignOpC = dyn_cast<ConstantInt>(Alignment);
5022 if (AlignOpC && AlignOpC->getValue().ult(llvm::Value::MaximumAlignment)) {
5023 uint64_t AlignmentVal = AlignOpC->getZExtValue();
5024 if (llvm::isPowerOf2_64(AlignmentVal)) {
5025 Align ExistingAlign = Call.getRetAlign().valueOrOne();
5026 Align NewAlign = Align(AlignmentVal);
5027 if (NewAlign > ExistingAlign) {
5030 Changed = true;
5031 }
5032 }
5033 }
5034 return Changed;
5035}
5036
5037/// Improvements for call, callbr and invoke instructions.
5038Instruction *InstCombinerImpl::visitCallBase(CallBase &Call) {
5039 bool Changed = annotateAnyAllocSite(Call, &TLI);
5040
5041 // Mark any parameters that are known to be non-null with the nonnull
5042 // attribute. This is helpful for inlining calls to functions with null
5043 // checks on their arguments.
5044 SmallVector<unsigned, 4> ArgNos;
5045 unsigned ArgNo = 0;
5046
5047 for (Value *V : Call.args()) {
5048 if (V->getType()->isPointerTy()) {
5049 // Simplify the nonnull operand if the parameter is known to be nonnull.
5050 // Otherwise, try to infer nonnull for it.
5051 bool HasDereferenceable = Call.getParamDereferenceableBytes(ArgNo) > 0;
5052 if (Call.paramHasAttr(ArgNo, Attribute::NonNull) ||
5053 (HasDereferenceable &&
5055 V->getType()->getPointerAddressSpace()))) {
5056 if (Value *Res = simplifyNonNullOperand(V, HasDereferenceable)) {
5057 replaceOperand(Call, ArgNo, Res);
5058 Changed = true;
5059 }
5060 } else if (isKnownNonZero(V,
5061 getSimplifyQuery().getWithInstruction(&Call))) {
5062 ArgNos.push_back(ArgNo);
5063 }
5064 }
5065 ArgNo++;
5066 }
5067
5068 assert(ArgNo == Call.arg_size() && "Call arguments not processed correctly.");
5069
5070 if (!ArgNos.empty()) {
5071 AttributeList AS = Call.getAttributes();
5072 LLVMContext &Ctx = Call.getContext();
5073 AS = AS.addParamAttribute(Ctx, ArgNos,
5074 Attribute::get(Ctx, Attribute::NonNull));
5075 Call.setAttributes(AS);
5076 Changed = true;
5077 }
5078
5079 // If the callee is a pointer to a function, attempt to move any casts to the
5080 // arguments of the call/callbr/invoke.
5082 Function *CalleeF = dyn_cast<Function>(Callee);
5083 if ((!CalleeF || CalleeF->getFunctionType() != Call.getFunctionType()) &&
5084 transformConstExprCastCall(Call))
5085 return nullptr;
5086
5087 if (CalleeF) {
5088 // Remove the convergent attr on calls when the callee is not convergent.
5089 if (Call.isConvergent() && !CalleeF->isConvergent() &&
5090 !CalleeF->isIntrinsic()) {
5091 LLVM_DEBUG(dbgs() << "Removing convergent attr from instr " << Call
5092 << "\n");
5094 return &Call;
5095 }
5096
5097 // If the call and callee calling conventions don't match, and neither one
5098 // of the calling conventions is compatible with C calling convention
5099 // this call must be unreachable, as the call is undefined.
5100 if ((CalleeF->getCallingConv() != Call.getCallingConv() &&
5101 !(CalleeF->getCallingConv() == llvm::CallingConv::C &&
5105 // Only do this for calls to a function with a body. A prototype may
5106 // not actually end up matching the implementation's calling conv for a
5107 // variety of reasons (e.g. it may be written in assembly).
5108 !CalleeF->isDeclaration()) {
5109 Instruction *OldCall = &Call;
5111 // If OldCall does not return void then replaceInstUsesWith poison.
5112 // This allows ValueHandlers and custom metadata to adjust itself.
5113 if (!OldCall->getType()->isVoidTy())
5114 replaceInstUsesWith(*OldCall, PoisonValue::get(OldCall->getType()));
5115 if (isa<CallInst>(OldCall))
5116 return eraseInstFromFunction(*OldCall);
5117
5118 // We cannot remove an invoke or a callbr, because it would change thexi
5119 // CFG, just change the callee to a null pointer.
5120 cast<CallBase>(OldCall)->setCalledFunction(
5121 CalleeF->getFunctionType(),
5122 Constant::getNullValue(CalleeF->getType()));
5123 return nullptr;
5124 }
5125 }
5126
5127 // Calling a null function pointer is undefined if a null address isn't
5128 // dereferenceable.
5129 if ((isa<ConstantPointerNull>(Callee) &&
5131 isa<UndefValue>(Callee)) {
5132 // If Call does not return void then replaceInstUsesWith poison.
5133 // This allows ValueHandlers and custom metadata to adjust itself.
5134 if (!Call.getType()->isVoidTy())
5136
5137 if (Call.isTerminator()) {
5138 // Can't remove an invoke or callbr because we cannot change the CFG.
5139 return nullptr;
5140 }
5141
5142 // This instruction is not reachable, just remove it.
5145 }
5146
5147 if (IntrinsicInst *II = findInitTrampoline(Callee))
5148 return transformCallThroughTrampoline(Call, *II);
5149
5150 // Combine calls involving pointer authentication intrinsics.
5151 if (Instruction *NewCall = foldPtrAuthIntrinsicCallee(Call))
5152 return NewCall;
5153
5154 // Combine calls to ptrauth constants.
5155 if (Instruction *NewCall = foldPtrAuthConstantCallee(Call))
5156 return NewCall;
5157
5158 if (isa<InlineAsm>(Callee) && !Call.doesNotThrow()) {
5159 InlineAsm *IA = cast<InlineAsm>(Callee);
5160 if (!IA->canThrow()) {
5161 // Normal inline asm calls cannot throw - mark them
5162 // 'nounwind'.
5164 Changed = true;
5165 }
5166 }
5167
5168 // Try to optimize the call if possible, we require DataLayout for most of
5169 // this. None of these calls are seen as possibly dead so go ahead and
5170 // delete the instruction now.
5171 if (CallInst *CI = dyn_cast<CallInst>(&Call)) {
5172 Instruction *I = tryOptimizeCall(CI);
5173 // If we changed something return the result, etc. Otherwise let
5174 // the fallthrough check.
5175 if (I) return eraseInstFromFunction(*I);
5176 }
5177
5178 if (!Call.use_empty() && !Call.isMustTailCall())
5179 if (Value *ReturnedArg = Call.getReturnedArgOperand()) {
5180 Type *CallTy = Call.getType();
5181 Type *RetArgTy = ReturnedArg->getType();
5182 if (RetArgTy->canLosslesslyBitCastTo(CallTy))
5183 return replaceInstUsesWith(
5184 Call, Builder.CreateBitOrPointerCast(ReturnedArg, CallTy));
5185 }
5186
5187 // Drop unnecessary callee_type metadata from calls that were converted
5188 // into direct calls.
5189 if (Call.getMetadata(LLVMContext::MD_callee_type) && !Call.isIndirectCall()) {
5190 Call.setMetadata(LLVMContext::MD_callee_type, nullptr);
5191 Changed = true;
5192 }
5193
5194 // Drop unnecessary kcfi operand bundles from calls that were converted
5195 // into direct calls.
5197 if (Bundle && !Call.isIndirectCall()) {
5198 DEBUG_WITH_TYPE(DEBUG_TYPE "-kcfi", {
5199 if (CalleeF) {
5200 ConstantInt *FunctionType = nullptr;
5201 ConstantInt *ExpectedType = cast<ConstantInt>(Bundle->Inputs[0]);
5202
5203 if (MDNode *MD = CalleeF->getMetadata(LLVMContext::MD_kcfi_type))
5204 FunctionType = mdconst::extract<ConstantInt>(MD->getOperand(0));
5205
5206 if (FunctionType &&
5207 FunctionType->getZExtValue() != ExpectedType->getZExtValue())
5208 dbgs() << Call.getModule()->getName()
5209 << ": warning: kcfi: " << Call.getCaller()->getName()
5210 << ": call to " << CalleeF->getName()
5211 << " using a mismatching function pointer type\n";
5212 }
5213 });
5214
5216 }
5217
5218 if (isRemovableAlloc(&Call, &TLI))
5219 return visitAllocSite(Call);
5220
5221 // Handle intrinsics which can be used in both call and invoke context.
5222 switch (Call.getIntrinsicID()) {
5223 case Intrinsic::experimental_gc_statepoint: {
5224 GCStatepointInst &GCSP = *cast<GCStatepointInst>(&Call);
5225 SmallPtrSet<Value *, 32> LiveGcValues;
5226 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5227 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5228
5229 // Remove the relocation if unused.
5230 if (GCR.use_empty()) {
5232 continue;
5233 }
5234
5235 Value *DerivedPtr = GCR.getDerivedPtr();
5236 Value *BasePtr = GCR.getBasePtr();
5237
5238 // Undef is undef, even after relocation.
5239 if (isa<UndefValue>(DerivedPtr) || isa<UndefValue>(BasePtr)) {
5242 continue;
5243 }
5244
5245 if (auto *PT = dyn_cast<PointerType>(GCR.getType())) {
5246 // The relocation of null will be null for most any collector.
5247 // TODO: provide a hook for this in GCStrategy. There might be some
5248 // weird collector this property does not hold for.
5249 if (isa<ConstantPointerNull>(DerivedPtr)) {
5250 // Use null-pointer of gc_relocate's type to replace it.
5253 continue;
5254 }
5255
5256 // isKnownNonNull -> nonnull attribute
5257 if (!GCR.hasRetAttr(Attribute::NonNull) &&
5258 isKnownNonZero(DerivedPtr,
5259 getSimplifyQuery().getWithInstruction(&Call))) {
5260 GCR.addRetAttr(Attribute::NonNull);
5261 // We discovered new fact, re-check users.
5262 Worklist.pushUsersToWorkList(GCR);
5263 }
5264 }
5265
5266 // If we have two copies of the same pointer in the statepoint argument
5267 // list, canonicalize to one. This may let us common gc.relocates.
5268 if (GCR.getBasePtr() == GCR.getDerivedPtr() &&
5269 GCR.getBasePtrIndex() != GCR.getDerivedPtrIndex()) {
5270 auto *OpIntTy = GCR.getOperand(2)->getType();
5271 GCR.setOperand(2, ConstantInt::get(OpIntTy, GCR.getBasePtrIndex()));
5272 }
5273
5274 // TODO: bitcast(relocate(p)) -> relocate(bitcast(p))
5275 // Canonicalize on the type from the uses to the defs
5276
5277 // TODO: relocate((gep p, C, C2, ...)) -> gep(relocate(p), C, C2, ...)
5278 LiveGcValues.insert(BasePtr);
5279 LiveGcValues.insert(DerivedPtr);
5280 }
5281 std::optional<OperandBundleUse> Bundle =
5283 unsigned NumOfGCLives = LiveGcValues.size();
5284 if (!Bundle || NumOfGCLives == Bundle->Inputs.size())
5285 break;
5286 // We can reduce the size of gc live bundle.
5287 DenseMap<Value *, unsigned> Val2Idx;
5288 std::vector<Value *> NewLiveGc;
5289 for (Value *V : Bundle->Inputs) {
5290 auto [It, Inserted] = Val2Idx.try_emplace(V);
5291 if (!Inserted)
5292 continue;
5293 if (LiveGcValues.count(V)) {
5294 It->second = NewLiveGc.size();
5295 NewLiveGc.push_back(V);
5296 } else
5297 It->second = NumOfGCLives;
5298 }
5299 // Update all gc.relocates
5300 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5301 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5302 Value *BasePtr = GCR.getBasePtr();
5303 assert(Val2Idx.count(BasePtr) && Val2Idx[BasePtr] != NumOfGCLives &&
5304 "Missed live gc for base pointer");
5305 auto *OpIntTy1 = GCR.getOperand(1)->getType();
5306 GCR.setOperand(1, ConstantInt::get(OpIntTy1, Val2Idx[BasePtr]));
5307 Value *DerivedPtr = GCR.getDerivedPtr();
5308 assert(Val2Idx.count(DerivedPtr) && Val2Idx[DerivedPtr] != NumOfGCLives &&
5309 "Missed live gc for derived pointer");
5310 auto *OpIntTy2 = GCR.getOperand(2)->getType();
5311 GCR.setOperand(2, ConstantInt::get(OpIntTy2, Val2Idx[DerivedPtr]));
5312 }
5313 // Create new statepoint instruction.
5314 OperandBundleDef NewBundle("gc-live", std::move(NewLiveGc));
5315 return CallBase::Create(&Call, NewBundle);
5316 }
5317 default: { break; }
5318 }
5319
5320 return Changed ? &Call : nullptr;
5321}
5322
5323/// If the callee is a constexpr cast of a function, attempt to move the cast to
5324/// the arguments of the call/invoke.
5325/// CallBrInst is not supported.
5326bool InstCombinerImpl::transformConstExprCastCall(CallBase &Call) {
5327 auto *Callee =
5329 if (!Callee)
5330 return false;
5331
5333 "CallBr's don't have a single point after a def to insert at");
5334
5335 // Don't perform the transform for declarations, which may not be fully
5336 // accurate. For example, void @foo() is commonly used as a placeholder for
5337 // unknown prototypes.
5338 if (Callee->isDeclaration())
5339 return false;
5340
5341 // If this is a call to a thunk function, don't remove the cast. Thunks are
5342 // used to transparently forward all incoming parameters and outgoing return
5343 // values, so it's important to leave the cast in place.
5344 if (Callee->hasFnAttribute("thunk"))
5345 return false;
5346
5347 // If this is a call to a naked function, the assembly might be
5348 // using an argument, or otherwise rely on the frame layout,
5349 // the function prototype will mismatch.
5350 if (Callee->hasFnAttribute(Attribute::Naked))
5351 return false;
5352
5353 // If this is a musttail call, the callee's prototype must match the caller's
5354 // prototype with the exception of pointee types. The code below doesn't
5355 // implement that, so we can't do this transform.
5356 // TODO: Do the transform if it only requires adding pointer casts.
5357 if (Call.isMustTailCall())
5358 return false;
5359
5361 const AttributeList &CallerPAL = Call.getAttributes();
5362
5363 // Okay, this is a cast from a function to a different type. Unless doing so
5364 // would cause a type conversion of one of our arguments, change this call to
5365 // be a direct call with arguments casted to the appropriate types.
5366 FunctionType *FT = Callee->getFunctionType();
5367 Type *OldRetTy = Caller->getType();
5368 Type *NewRetTy = FT->getReturnType();
5369
5370 // Check to see if we are changing the return type...
5371 if (OldRetTy != NewRetTy) {
5372
5373 if (NewRetTy->isStructTy())
5374 return false; // TODO: Handle multiple return values.
5375
5376 if (!CastInst::isBitOrNoopPointerCastable(NewRetTy, OldRetTy, DL)) {
5377 if (!Caller->use_empty())
5378 return false; // Cannot transform this return value.
5379 }
5380
5381 if (!CallerPAL.isEmpty() && !Caller->use_empty()) {
5382 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5383 if (RAttrs.overlaps(AttributeFuncs::typeIncompatible(
5384 NewRetTy, CallerPAL.getRetAttrs())))
5385 return false; // Attribute not compatible with transformed value.
5386 }
5387
5388 // If the callbase is an invoke instruction, and the return value is
5389 // used by a PHI node in a successor, we cannot change the return type of
5390 // the call because there is no place to put the cast instruction (without
5391 // breaking the critical edge). Bail out in this case.
5392 if (!Caller->use_empty()) {
5393 BasicBlock *PhisNotSupportedBlock = nullptr;
5394 if (auto *II = dyn_cast<InvokeInst>(Caller))
5395 PhisNotSupportedBlock = II->getNormalDest();
5396 if (PhisNotSupportedBlock)
5397 for (User *U : Caller->users())
5398 if (PHINode *PN = dyn_cast<PHINode>(U))
5399 if (PN->getParent() == PhisNotSupportedBlock)
5400 return false;
5401 }
5402 }
5403
5404 unsigned NumActualArgs = Call.arg_size();
5405 unsigned NumCommonArgs = std::min(FT->getNumParams(), NumActualArgs);
5406
5407 // Prevent us turning:
5408 // declare void @takes_i32_inalloca(i32* inalloca)
5409 // call void bitcast (void (i32*)* @takes_i32_inalloca to void (i32)*)(i32 0)
5410 //
5411 // into:
5412 // call void @takes_i32_inalloca(i32* null)
5413 //
5414 // Similarly, avoid folding away bitcasts of byval calls.
5415 if (Callee->getAttributes().hasAttrSomewhere(Attribute::InAlloca) ||
5416 Callee->getAttributes().hasAttrSomewhere(Attribute::Preallocated))
5417 return false;
5418
5419 auto AI = Call.arg_begin();
5420 for (unsigned i = 0, e = NumCommonArgs; i != e; ++i, ++AI) {
5421 Type *ParamTy = FT->getParamType(i);
5422 Type *ActTy = (*AI)->getType();
5423
5424 if (!CastInst::isBitOrNoopPointerCastable(ActTy, ParamTy, DL))
5425 return false; // Cannot transform this parameter value.
5426
5427 // Check if there are any incompatible attributes we cannot drop safely.
5428 if (AttrBuilder(FT->getContext(), CallerPAL.getParamAttrs(i))
5429 .overlaps(AttributeFuncs::typeIncompatible(
5430 ParamTy, CallerPAL.getParamAttrs(i),
5431 AttributeFuncs::ASK_UNSAFE_TO_DROP)))
5432 return false; // Attribute not compatible with transformed value.
5433
5434 if (Call.isInAllocaArgument(i) ||
5435 CallerPAL.hasParamAttr(i, Attribute::Preallocated))
5436 return false; // Cannot transform to and from inalloca/preallocated.
5437
5438 if (CallerPAL.hasParamAttr(i, Attribute::SwiftError))
5439 return false;
5440
5441 if (CallerPAL.hasParamAttr(i, Attribute::ByVal) !=
5442 Callee->getAttributes().hasParamAttr(i, Attribute::ByVal))
5443 return false; // Cannot transform to or from byval.
5444 }
5445
5446 if (FT->getNumParams() < NumActualArgs && FT->isVarArg() &&
5447 !CallerPAL.isEmpty()) {
5448 // In this case we have more arguments than the new function type, but we
5449 // won't be dropping them. Check that these extra arguments have attributes
5450 // that are compatible with being a vararg call argument.
5451 unsigned SRetIdx;
5452 if (CallerPAL.hasAttrSomewhere(Attribute::StructRet, &SRetIdx) &&
5453 SRetIdx - AttributeList::FirstArgIndex >= FT->getNumParams())
5454 return false;
5455 }
5456
5457 // Okay, we decided that this is a safe thing to do: go ahead and start
5458 // inserting cast instructions as necessary.
5459 SmallVector<Value *, 8> Args;
5461 Args.reserve(NumActualArgs);
5462 ArgAttrs.reserve(NumActualArgs);
5463
5464 // Get any return attributes.
5465 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5466
5467 // If the return value is not being used, the type may not be compatible
5468 // with the existing attributes. Wipe out any problematic attributes.
5469 RAttrs.remove(
5470 AttributeFuncs::typeIncompatible(NewRetTy, CallerPAL.getRetAttrs()));
5471
5472 LLVMContext &Ctx = Call.getContext();
5473 AI = Call.arg_begin();
5474 for (unsigned i = 0; i != NumCommonArgs; ++i, ++AI) {
5475 Type *ParamTy = FT->getParamType(i);
5476
5477 Value *NewArg = *AI;
5478 if ((*AI)->getType() != ParamTy)
5479 NewArg = Builder.CreateBitOrPointerCast(*AI, ParamTy);
5480 Args.push_back(NewArg);
5481
5482 // Add any parameter attributes except the ones incompatible with the new
5483 // type. Note that we made sure all incompatible ones are safe to drop.
5484 AttributeMask IncompatibleAttrs = AttributeFuncs::typeIncompatible(
5485 ParamTy, CallerPAL.getParamAttrs(i), AttributeFuncs::ASK_SAFE_TO_DROP);
5486 ArgAttrs.push_back(
5487 CallerPAL.getParamAttrs(i).removeAttributes(Ctx, IncompatibleAttrs));
5488 }
5489
5490 // If the function takes more arguments than the call was taking, add them
5491 // now.
5492 for (unsigned i = NumCommonArgs; i != FT->getNumParams(); ++i) {
5493 Args.push_back(Constant::getNullValue(FT->getParamType(i)));
5494 ArgAttrs.push_back(AttributeSet());
5495 }
5496
5497 // If we are removing arguments to the function, emit an obnoxious warning.
5498 if (FT->getNumParams() < NumActualArgs) {
5499 // TODO: if (!FT->isVarArg()) this call may be unreachable. PR14722
5500 if (FT->isVarArg()) {
5501 // Add all of the arguments in their promoted form to the arg list.
5502 for (unsigned i = FT->getNumParams(); i != NumActualArgs; ++i, ++AI) {
5503 Type *PTy = getPromotedType((*AI)->getType());
5504 Value *NewArg = *AI;
5505 if (PTy != (*AI)->getType()) {
5506 // Must promote to pass through va_arg area!
5507 Instruction::CastOps opcode =
5508 CastInst::getCastOpcode(*AI, false, PTy, false);
5509 NewArg = Builder.CreateCast(opcode, *AI, PTy);
5510 }
5511 Args.push_back(NewArg);
5512
5513 // Add any parameter attributes.
5514 ArgAttrs.push_back(CallerPAL.getParamAttrs(i));
5515 }
5516 }
5517 }
5518
5519 AttributeSet FnAttrs = CallerPAL.getFnAttrs();
5520
5521 if (NewRetTy->isVoidTy())
5522 Caller->setName(""); // Void type should not have a name.
5523
5524 assert((ArgAttrs.size() == FT->getNumParams() || FT->isVarArg()) &&
5525 "missing argument attributes");
5526 AttributeList NewCallerPAL = AttributeList::get(
5527 Ctx, FnAttrs, AttributeSet::get(Ctx, RAttrs), ArgAttrs);
5528
5530 Call.getOperandBundlesAsDefs(OpBundles);
5531
5532 CallBase *NewCall;
5533 if (InvokeInst *II = dyn_cast<InvokeInst>(Caller)) {
5534 NewCall = Builder.CreateInvoke(Callee, II->getNormalDest(),
5535 II->getUnwindDest(), Args, OpBundles);
5536 } else {
5537 NewCall = Builder.CreateCall(Callee, Args, OpBundles);
5538 cast<CallInst>(NewCall)->setTailCallKind(
5539 cast<CallInst>(Caller)->getTailCallKind());
5540 }
5541 NewCall->takeName(Caller);
5543 NewCall->setAttributes(NewCallerPAL);
5544
5545 // Preserve prof metadata if any.
5546 NewCall->copyMetadata(*Caller, {LLVMContext::MD_prof});
5547
5548 // Insert a cast of the return type as necessary.
5549 Instruction *NC = NewCall;
5550 Value *NV = NC;
5551 if (OldRetTy != NV->getType() && !Caller->use_empty()) {
5552 assert(!NV->getType()->isVoidTy());
5554 NC->setDebugLoc(Caller->getDebugLoc());
5555
5556 auto OptInsertPt = NewCall->getInsertionPointAfterDef();
5557 assert(OptInsertPt && "No place to insert cast");
5558 InsertNewInstBefore(NC, *OptInsertPt);
5559 Worklist.pushUsersToWorkList(*Caller);
5560 }
5561
5562 if (!Caller->use_empty())
5563 replaceInstUsesWith(*Caller, NV);
5564 else if (Caller->hasValueHandle()) {
5565 if (OldRetTy == NV->getType())
5567 else
5568 // We cannot call ValueIsRAUWd with a different type, and the
5569 // actual tracked value will disappear.
5571 }
5572
5573 eraseInstFromFunction(*Caller);
5574 return true;
5575}
5576
5577/// Turn a call to a function created by init_trampoline / adjust_trampoline
5578/// intrinsic pair into a direct call to the underlying function.
5580InstCombinerImpl::transformCallThroughTrampoline(CallBase &Call,
5581 IntrinsicInst &Tramp) {
5582 FunctionType *FTy = Call.getFunctionType();
5583 AttributeList Attrs = Call.getAttributes();
5584
5585 // If the call already has the 'nest' attribute somewhere then give up -
5586 // otherwise 'nest' would occur twice after splicing in the chain.
5587 if (Attrs.hasAttrSomewhere(Attribute::Nest))
5588 return nullptr;
5589
5591 FunctionType *NestFTy = NestF->getFunctionType();
5592
5593 AttributeList NestAttrs = NestF->getAttributes();
5594 if (!NestAttrs.isEmpty()) {
5595 unsigned NestArgNo = 0;
5596 Type *NestTy = nullptr;
5597 AttributeSet NestAttr;
5598
5599 // Look for a parameter marked with the 'nest' attribute.
5600 for (FunctionType::param_iterator I = NestFTy->param_begin(),
5601 E = NestFTy->param_end();
5602 I != E; ++NestArgNo, ++I) {
5603 AttributeSet AS = NestAttrs.getParamAttrs(NestArgNo);
5604 if (AS.hasAttribute(Attribute::Nest)) {
5605 // Record the parameter type and any other attributes.
5606 NestTy = *I;
5607 NestAttr = AS;
5608 break;
5609 }
5610 }
5611
5612 if (NestTy) {
5613 std::vector<Value*> NewArgs;
5614 std::vector<AttributeSet> NewArgAttrs;
5615 NewArgs.reserve(Call.arg_size() + 1);
5616 NewArgAttrs.reserve(Call.arg_size());
5617
5618 // Insert the nest argument into the call argument list, which may
5619 // mean appending it. Likewise for attributes.
5620
5621 {
5622 unsigned ArgNo = 0;
5623 auto I = Call.arg_begin(), E = Call.arg_end();
5624 do {
5625 if (ArgNo == NestArgNo) {
5626 // Add the chain argument and attributes.
5627 Value *NestVal = Tramp.getArgOperand(2);
5628 if (NestVal->getType() != NestTy)
5629 NestVal = Builder.CreateBitCast(NestVal, NestTy, "nest");
5630 NewArgs.push_back(NestVal);
5631 NewArgAttrs.push_back(NestAttr);
5632 }
5633
5634 if (I == E)
5635 break;
5636
5637 // Add the original argument and attributes.
5638 NewArgs.push_back(*I);
5639 NewArgAttrs.push_back(Attrs.getParamAttrs(ArgNo));
5640
5641 ++ArgNo;
5642 ++I;
5643 } while (true);
5644 }
5645
5646 // The trampoline may have been bitcast to a bogus type (FTy).
5647 // Handle this by synthesizing a new function type, equal to FTy
5648 // with the chain parameter inserted.
5649
5650 std::vector<Type*> NewTypes;
5651 NewTypes.reserve(FTy->getNumParams()+1);
5652
5653 // Insert the chain's type into the list of parameter types, which may
5654 // mean appending it.
5655 {
5656 unsigned ArgNo = 0;
5657 FunctionType::param_iterator I = FTy->param_begin(),
5658 E = FTy->param_end();
5659
5660 do {
5661 if (ArgNo == NestArgNo)
5662 // Add the chain's type.
5663 NewTypes.push_back(NestTy);
5664
5665 if (I == E)
5666 break;
5667
5668 // Add the original type.
5669 NewTypes.push_back(*I);
5670
5671 ++ArgNo;
5672 ++I;
5673 } while (true);
5674 }
5675
5676 // Replace the trampoline call with a direct call. Let the generic
5677 // code sort out any function type mismatches.
5678 FunctionType *NewFTy =
5679 FunctionType::get(FTy->getReturnType(), NewTypes, FTy->isVarArg());
5680 AttributeList NewPAL =
5681 AttributeList::get(FTy->getContext(), Attrs.getFnAttrs(),
5682 Attrs.getRetAttrs(), NewArgAttrs);
5683
5685 Call.getOperandBundlesAsDefs(OpBundles);
5686
5687 Instruction *NewCaller;
5688 if (InvokeInst *II = dyn_cast<InvokeInst>(&Call)) {
5689 NewCaller = InvokeInst::Create(NewFTy, NestF, II->getNormalDest(),
5690 II->getUnwindDest(), NewArgs, OpBundles);
5691 cast<InvokeInst>(NewCaller)->setCallingConv(II->getCallingConv());
5692 cast<InvokeInst>(NewCaller)->setAttributes(NewPAL);
5693 } else if (CallBrInst *CBI = dyn_cast<CallBrInst>(&Call)) {
5694 NewCaller =
5695 CallBrInst::Create(NewFTy, NestF, CBI->getDefaultDest(),
5696 CBI->getIndirectDests(), NewArgs, OpBundles);
5697 cast<CallBrInst>(NewCaller)->setCallingConv(CBI->getCallingConv());
5698 cast<CallBrInst>(NewCaller)->setAttributes(NewPAL);
5699 } else {
5700 NewCaller = CallInst::Create(NewFTy, NestF, NewArgs, OpBundles);
5701 cast<CallInst>(NewCaller)->setTailCallKind(
5702 cast<CallInst>(Call).getTailCallKind());
5703 cast<CallInst>(NewCaller)->setCallingConv(
5704 cast<CallInst>(Call).getCallingConv());
5705 cast<CallInst>(NewCaller)->setAttributes(NewPAL);
5706 }
5707 NewCaller->setDebugLoc(Call.getDebugLoc());
5708
5709 return NewCaller;
5710 }
5711 }
5712
5713 // Replace the trampoline call with a direct call. Since there is no 'nest'
5714 // parameter, there is no need to adjust the argument list. Let the generic
5715 // code sort out any function type mismatches.
5716 Call.setCalledFunction(FTy, NestF);
5717 return &Call;
5718}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
@ Scaled
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
BitTracker BT
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static SDValue foldBitOrderCrossLogicOp(SDNode *N, SelectionDAG &DAG)
#define Check(C,...)
#define DEBUG_TYPE
Hexagon Common GEP
#define _
IRTranslator LLVM IR MI
static Type * getPromotedType(Type *Ty)
Return the specified type promoted as it would be to pass though a va_arg area.
static Instruction * createOverflowTuple(IntrinsicInst *II, Value *Result, Constant *Overflow)
Creates a result tuple for an overflow intrinsic II with a given Result and a constant Overflow value...
static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M, IRBuilderBase &B)
static IntrinsicInst * findInitTrampolineFromAlloca(Value *TrampMem)
static bool removeTriviallyEmptyRange(IntrinsicInst &EndI, InstCombinerImpl &IC, std::function< bool(const IntrinsicInst &)> IsStart)
static bool inputDenormalIsDAZ(const Function &F, const Type *Ty)
static Instruction * reassociateMinMaxWithConstantInOperand(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If this min/max has a matching min/max operand with a constant, try to push the constant operand into...
static bool isIdempotentBinaryIntrinsic(Intrinsic::ID IID)
Helper to match idempotent binary intrinsics, namely, intrinsics where f(f(x, y), y) == f(x,...
static bool signBitMustBeTheSame(Value *Op0, Value *Op1, const SimplifyQuery &SQ)
Return true if two values Op0 and Op1 are known to have the same sign.
static Value * optimizeModularFormat(CallInst *CI, IRBuilderBase &B)
static Instruction * moveAddAfterMinMax(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0.
static Instruction * simplifyInvariantGroupIntrinsic(IntrinsicInst &II, InstCombinerImpl &IC)
This function transforms launder.invariant.group and strip.invariant.group like: launder(launder(x)) ...
static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E, unsigned NumOperands)
static std::optional< bool > getKnownSign(Value *Op, const SimplifyQuery &SQ)
static cl::opt< unsigned > GuardWideningWindow("instcombine-guard-widening-window", cl::init(3), cl::desc("How wide an instruction window to bypass looking for " "another guard"))
static bool hasUndefSource(AnyMemTransferInst *MI)
Recognize a memcpy/memmove from a trivially otherwise unused alloca.
static Instruction * factorizeMinMaxTree(IntrinsicInst *II)
Reduce a sequence of min/max intrinsics with a common operand.
static Instruction * foldClampRangeOfTwo(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If we have a clamp pattern like max (min X, 42), 41 – where the output can only be one of two possibl...
static Value * simplifyReductionOperand(Value *Arg, bool CanReorderLanes)
static IntrinsicInst * findInitTrampolineFromBB(IntrinsicInst *AdjustTramp, Value *TrampMem)
static bool isAspectNeeded(StringRef Aspect, CallInst *CI, std::optional< unsigned > FirstArgIdx, const std::optional< Bitset< 256 > > &Specifiers)
static Value * foldIntrinsicUsingDistributiveLaws(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
static std::optional< bool > getKnownSignOrZero(Value *Op, const SimplifyQuery &SQ)
static Value * foldMinimumOverTrailingOrLeadingZeroCount(Value *I0, Value *I1, const DataLayout &DL, InstCombiner::BuilderTy &Builder)
Fold an unsigned minimum of trailing or leading zero bits counts: umin(cttz(CtOp1,...
static bool rightDistributesOverLeft(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "(X ROp Y) LOp Z" is always equal to "(X LOp Z) ROp (Y LOp Z)".
static Value * foldIdempotentBinaryIntrinsicRecurrence(InstCombinerImpl &IC, IntrinsicInst *II)
Attempt to simplify value-accumulating recurrences of kind: umax.acc = phi i8 [ umax,...
static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy)
static Instruction * foldCtpop(IntrinsicInst &II, InstCombinerImpl &IC)
static Instruction * simplifyNeonTbl(IntrinsicInst &II, InstCombiner &IC, bool IsExtension)
Convert tbl/tbx intrinsics to shufflevector if the mask is constant, and at most two source operands ...
static Instruction * foldCttzCtlz(IntrinsicInst &II, InstCombinerImpl &IC)
static IntrinsicInst * findInitTrampoline(Value *Callee)
static Value * foldCmpIntrinsicOfExtended(IntrinsicInst *II, InstCombiner::BuilderTy &Builder, const DataLayout &DL)
Fold an scmp/ucmp intrinsic whose operands are extended from a narrower type: scmp (sext X),...
static Bitset< 256 > parseFormatStringSpecifiers(StringRef FormatStr)
static FCmpInst::Predicate fpclassTestIsFCmp0(FPClassTest Mask, const Function &F, Type *Ty)
static bool leftDistributesOverRight(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "X LOp (Y ROp Z)" is always equal to "(X LOp Y) ROp (X LOp Z)".
static Value * reassociateMinMaxWithConstants(IntrinsicInst *II, IRBuilderBase &Builder, const SimplifyQuery &SQ)
If this min/max has a constant operand and an operand that is a matching min/max with a constant oper...
static Value * foldSinAndCosToSinCos(IntrinsicInst *II, IRBuilderBase &B, InstCombinerImpl &IC)
static CallInst * canonicalizeConstantArg0ToArg1(CallInst &Call)
static Instruction * foldNeonShift(IntrinsicInst *II, InstCombinerImpl &IC)
This file provides internal interfaces used to implement the InstCombine.
This file provides the interface for the instcombine pass implementation.
static bool inputDenormalIsIEEE(DenormalMode Mode)
Return true if it's possible to assume IEEE treatment of input denormals in F for Val.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
const SmallVectorImpl< MachineOperand > & Cond
This file implements the SmallBitVector class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
static LLVM_ABI bool semanticsHasInf(const fltSemantics &)
Definition APFloat.cpp:351
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static LLVM_ABI bool hasSignBitInMSB(const fltSemantics &)
Definition APFloat.cpp:364
bool isNegative() const
Definition APFloat.h:1583
void clearSign()
Definition APFloat.h:1402
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
bool isZero() const
Definition APFloat.h:1579
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
Definition APFloat.h:1242
static APFloat getSmallest(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) finite number in the given semantics.
Definition APFloat.h:1252
bool isInfinity() const
Definition APFloat.h:1580
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:231
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:226
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1206
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1984
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1187
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:377
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1693
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1509
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1116
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1964
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1971
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:647
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:216
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1085
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:507
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2072
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:331
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:303
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:197
std::optional< int64_t > trySExtValue() const
Get sign extended value if possible.
Definition APInt.h:1595
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1977
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
This class represents any memset intrinsic.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:407
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
static LLVM_ABI AttributeSet get(LLVMContext &C, const AttrBuilder &B)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
static LLVM_ABI Attribute getWithDereferenceableOrNullBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:446
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI bool isSigned() const
Whether the intrinsic is signed or unsigned.
LLVM_ABI Instruction::BinaryOps getBinaryOp() const
Returns the binary operation underlying the intrinsic.
static BinaryOperator * CreateFAddFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:271
static LLVM_ABI BinaryOperator * CreateNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Helper functions to construct and inspect unary operations (NEG and NOT) via binary operators SUB and...
static BinaryOperator * CreateNSW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:314
static LLVM_ABI BinaryOperator * CreateNot(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static BinaryOperator * CreateNUW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:329
static BinaryOperator * CreateFMulFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:279
static BinaryOperator * CreateFDivFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:283
static BinaryOperator * CreateFSubFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:275
static LLVM_ABI BinaryOperator * CreateNSWNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
This is a constexpr reimplementation of a subset of std::bitset.
Definition Bitset.h:30
constexpr bool any() const
Definition Bitset.h:113
constexpr Bitset & set()
Definition Bitset.h:81
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
void setDoesNotThrow()
MaybeAlign getRetAlign() const
Extract the alignment of the return value.
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
OperandBundleUse getOperandBundleAt(unsigned Index) const
Return the operand bundle at a specific index.
std::optional< OperandBundleUse > getOperandBundle(StringRef Name) const
Return an operand bundle by name, if present.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
bool isInAllocaArgument(unsigned ArgNo) const
Determine whether this argument is passed in an alloca.
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
bool hasRetAttr(Attribute::AttrKind Kind) const
Determine whether the return value has the given attribute.
unsigned getNumOperandBundles() const
Return the number of operand bundles associated with this User.
uint64_t getParamDereferenceableBytes(unsigned i) const
Extract the number of dereferenceable bytes for a call or parameter (0=unknown).
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
LLVM_ABI bool isIndirectCall() const
Return true if the callsite is an indirect call.
static LLVM_ABI CallBase * removeOperandBundleAt(CallBase *CB, size_t Offset, InsertPosition InsertPtr=nullptr)
void setNotConvergent()
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Attribute getFnAttr(StringRef Kind) const
Get the attribute of a given kind for the function.
bool doesNotThrow() const
Determine if the call cannot unwind.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
bool isConvergent() const
Determine if the invoke is convergent.
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
Value * getReturnedArgOperand() const
If one of the arguments has the 'returned' attribute, returns its operand value.
static LLVM_ABI CallBase * Create(CallBase *CB, ArrayRef< OperandBundleDef > Bundles, InsertPosition InsertPt=nullptr)
Create a clone of CB with a different set of operand bundles and insert it before InsertPt.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
static LLVM_ABI CallBase * removeOperandBundle(CallBase *CB, uint32_t ID, InsertPosition InsertPt=nullptr)
Create a clone of CB with operand bundle ID removed.
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
LLVM_ABI Function * getCaller()
Helper to get the caller (the parent function).
CallBr instruction, tracking function calls that may not return control but instead transfer it to a ...
static CallBrInst * Create(FunctionType *Ty, Value *Func, BasicBlock *DefaultDest, ArrayRef< BasicBlock * > IndirectDests, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This class represents a function call, abstracting a target machine's calling convention.
bool isNoTailCall() const
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
bool isMustTailCall() const
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI CastInst * CreateIntegerCast(Value *S, Type *Ty, bool isSigned, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a ZExt, BitCast, or Trunc for int -> int casts.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static LLVM_ABI CastInst * CreateBitOrPointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, a PtrToInt, or an IntToPTr cast instruction.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getNonStrictPredicate() const
For example, SGT -> SGE, SLT -> SLE, ULT -> ULE, UGT -> UGE.
Definition InstrTypes.h:934
Predicate getUnorderedPredicate() const
Definition InstrTypes.h:874
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getLimitedValue(uint64_t Limit=~0ULL) const
getLimitedValue - If the value is smaller than the specified limit, return it, otherwise return the l...
Definition Constants.h:269
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI ConstantPtrAuth * get(Constant *Ptr, ConstantInt *Key, ConstantInt *Disc, Constant *AddrDisc, Constant *DeactivationSymbol)
Return a pointer signed with the specified parameters.
This class represents a range of values.
LLVM_ABI ConstantRange zextOrTrunc(uint32_t BitWidth) const
Make this range have the bit width given by BitWidth.
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
uint32_t getBitWidth() const
Get the bit width of this ConstantRange.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Record of a variable value-assignment, aka a non instruction representation of the dbg....
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:299
unsigned size() const
Definition DenseMap.h:172
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
Definition DenseMap.h:219
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Definition DenseMap.h:214
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static FMFSource intersect(Value *A, Value *B)
Intersect the FMF from two instructions.
Definition IRBuilder.h:107
This class represents an extension of floating point types.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
An instruction for ordering other memory operations.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this fence instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this fence instruction.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Type::subtype_iterator param_iterator
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
bool isConvergent() const
Determine if the call is convergent.
Definition Function.h:593
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:329
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:577
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
Definition Function.h:252
LLVM_ABI Value * getBasePtr() const
unsigned getBasePtrIndex() const
The index into the associate statepoint's argument list which contains the base pointer of the pointe...
LLVM_ABI Value * getDerivedPtr() const
unsigned getDerivedPtrIndex() const
The index into the associate statepoint's argument list which contains the pointer whose relocation t...
std::vector< const GCRelocateInst * > getGCRelocates() const
Get list of all gc reloactes linked to this statepoint May contain several relocations for the same b...
Definition Statepoint.h:206
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this GlobalObject.
LLVM_ABI bool isDeclaration() const
Return true if the primary definition of this global value is outside of the current translation unit...
Definition Globals.cpp:408
PointerType * getType() const
Global values are always pointers.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNull=false)
Definition IRBuilder.h:2258
LLVM_ABI Value * CreateLaunderInvariantGroup(Value *Ptr)
Create a launder.invariant.group intrinsic call.
ConstantInt * getTrue()
Get the constant value for i1 true.
Definition IRBuilder.h:457
LLVM_ABI Value * CreateBinaryIntrinsic(Intrinsic::ID ID, Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 2 operands which is mangled on the first type.
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1449
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2131
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Definition IRBuilder.h:2701
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:462
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2502
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
LLVM_ABI Value * CreateStripInvariantGroup(Value *Ptr)
Create a strip.invariant.group intrinsic call.
static InsertValueInst * Create(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Instruction * foldOpIntoPhi(Instruction &I, PHINode *PN, bool AllowMultipleUses=false)
Given a binary operator, cast instruction, or select which has a PHI node as operand #0,...
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
bool SimplifyDemandedBits(Instruction *I, unsigned Op, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0) override
This form of SimplifyDemandedBits simplifies the specified instruction operand if possible,...
Instruction * FoldOpIntoSelect(Instruction &Op, SelectInst *SI, bool FoldWithMultiUse=false, bool SimplifyBothArms=false)
Given an instruction with a select as one operand and a constant as the other operand,...
Instruction * SimplifyAnyMemSet(AnyMemSetInst *MI)
Instruction * foldItoFPtoI(FPToIntTy &FI)
fpto{s/u}i.sat --> X or zext(X) or sext(X) or trunc(X) This is safe if the intermediate type has enou...
Instruction * visitFree(CallInst &FI, Value *FreedOp)
Instruction * visitCallBrInst(CallBrInst &CBI)
Instruction * eraseInstFromFunction(Instruction &I) override
Combiner aware instruction erasure.
Value * foldReversedIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are reverses, try to pull the reverse after the intrinsic.
Value * tryGetLog2(Value *Op, bool AssumeNonZero)
Instruction * visitFenceInst(FenceInst &FI)
Instruction * foldShuffledIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are unary shuffles with the same mask, try to shuffle after the int...
Instruction * visitInvokeInst(InvokeInst &II)
bool SimplifyDemandedInstructionBits(Instruction &Inst)
Tries to simplify operands to an integer instruction based on its demanded bits.
void CreateNonTerminatorUnreachable(Instruction *InsertAt)
Create and insert the idiom we use to indicate a block is unreachable without having to rewrite the C...
Instruction * visitVAEndInst(VAEndInst &I)
Instruction * matchBSwapOrBitReverse(Instruction &I, bool MatchBSwaps, bool MatchBitReversals)
Given an initial instruction, check to see if it is the root of a bswap/bitreverse idiom.
Constant * unshuffleConstant(ArrayRef< int > ShMask, Constant *C, VectorType *NewCTy)
Find a constant NewC that has property: shuffle(NewC, poison, ShMask) = C for lanes that select NewC.
Instruction * visitAllocSite(Instruction &FI)
Instruction * SimplifyAnyMemTransfer(AnyMemTransferInst *MI)
OverflowResult computeOverflow(Instruction::BinaryOps BinaryOp, bool IsSigned, Value *LHS, Value *RHS, Instruction *CxtI) const
Instruction * visitCallInst(CallInst &CI)
CallInst simplification.
The core instruction combiner logic.
SimplifyQuery SQ
const DataLayout & getDataLayout() const
unsigned ComputeMaxSignificantBits(const Value *Op, const Instruction *CxtI=nullptr, unsigned Depth=0) const
bool isFreeToInvert(Value *V, bool WillInvertAllUses, bool &DoesConsume)
Return true if the specified value is free to invert (apply ~ to).
DominatorTree & getDominatorTree() const
BlockFrequencyInfo * BFI
TargetLibraryInfo & TLI
Instruction * InsertNewInstBefore(Instruction *New, BasicBlock::iterator Old)
Inserts an instruction New before instruction Old.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
const DataLayout & DL
DomConditionCache DC
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CxtI, unsigned Depth=0) const
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
LLVM_ABI std::optional< Instruction * > targetInstCombineIntrinsic(IntrinsicInst &II)
AssumptionCache & AC
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
bool MaskedValueIsZero(const Value *V, const APInt &Mask, const Instruction *CxtI=nullptr, unsigned Depth=0) const
DominatorTree & DT
ProfileSummaryInfo * PSI
OptimizationRemarkEmitter & ORE
Value * getFreelyInverted(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume)
const SimplifyQuery & getSimplifyQuery() const
bool isKnownToBeAPowerOfTwo(const Value *V, bool OrZero=false, const Instruction *CxtI=nullptr, unsigned Depth=0)
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI bool mayWriteToMemory() const LLVM_READONLY
Return true if this instruction may modify memory.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
bool isTerminator() const
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI std::optional< InstListType::iterator > getInsertionPointAfterDef()
Get the first insertion point at which the result of this instruction is defined.
LLVM_ABI bool isIdenticalTo(const Instruction *I) const LLVM_READONLY
Return true if the specified instruction is exactly identical to the current one.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
Invoke instruction.
static InvokeInst * Create(FunctionType *Ty, Value *Func, BasicBlock *IfNormal, BasicBlock *IfException, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
Metadata node.
Definition Metadata.h:1069
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1567
static LLVM_ABI MDNode * getMostGenericFPMath(MDNode *A, MDNode *B)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:615
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:111
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
ICmpInst::Predicate getPredicate() const
Returns the comparison predicate underlying the intrinsic.
bool isSigned() const
Whether the intrinsic is signed or unsigned.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
StringRef getName() const
Get a short "name" for the module.
Definition Module.h:316
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
Utility class for integer operators which may exhibit overflow - Add, Sub, Mul, and Shl.
Definition Operator.h:78
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
bool isCommutative() const
Return true if the instruction is commutative.
Definition Operator.h:130
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Represents a saturating add/sub intrinsic.
This class represents the LLVM 'select' instruction.
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
This instruction constructs a fixed permutation of two input vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
size_type size() const
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setVolatile(bool V)
Specify whether this is a volatile store or not.
void setAlignment(Align Align)
void setOrdering(AtomicOrdering Ordering)
Sets the ordering constraint of this store instruction.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static constexpr size_t npos
Definition StringRef.h:58
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
LLVM_ABI size_t find_first_not_of(char C, size_t From=0) const
Find the first character in the string that is not C or npos if not found.
Class to represent struct types.
static LLVM_ABI bool isCallingConvCCompatible(CallBase *CI)
Returns true if call site / callee has cdecl-compatible calling conventions.
Provides information about what library functions are available for the current target.
This class represents a truncation of integer types.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:263
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI bool canLosslesslyBitCastTo(Type *Ty) const
Return true if this type could be converted with a lossless BitCast to type 'Ty'.
Definition Type.cpp:153
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:276
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:106
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
static UnaryOperator * CreateWithCopiedFlags(UnaryOps Opc, Value *V, Instruction *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:148
static UnaryOperator * CreateFNegFMF(Value *Op, Instruction *FMFSource, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:156
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:35
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
This represents the llvm.va_end intrinsic.
static LLVM_ABI void ValueIsDeleted(Value *V)
Definition Value.cpp:1272
static LLVM_ABI void ValueIsRAUWd(Value *Old, Value *New)
Definition Value.cpp:1325
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
static constexpr uint64_t MaximumAlignment
Definition Value.h:799
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:439
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
iterator_range< user_iterator > users()
Definition Value.h:426
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:713
bool use_empty() const
Definition Value.h:346
static constexpr unsigned MaxAlignmentExponent
The maximum alignment for instructions.
Definition Value.h:798
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:81
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
auto m_PosZeroFP()
Matches a floating-point positive zero.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
AllOnesConstantMatch m_AllOnes()
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_or< Ty... > m_CombineOr(const Ty &...Ps)
Combine pattern matchers matching any of Ps patterns.
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
auto m_PtrToIntOrAddr(const OpTy &Op)
Matches PtrToInt or PtrToAddr.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
BinaryOp_match< LHS, RHS, Instruction::Xor > m_Xor(const LHS &L, const RHS &R)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWSub(const LHS &L, const RHS &R)
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
auto m_UMin(const Opnd0 &Op0, const Opnd1 &Op1)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
OverflowingBinaryOp_match< cst_pred_ty< is_zero_int >, ValTy, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWNeg(const ValTy &V)
Matches a 'Neg' as 'sub nsw 0, V'.
auto m_SMax(const Opnd0 &Op0, const Opnd1 &Op1)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
cstfp_pred_ty< is_neg_zero_fp > m_NegZeroFP()
Match a floating-point negative zero.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_UMax(const Opnd0 &Op0, const Opnd1 &Op1)
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_CopySign(const Opnd0 &Op0, const Opnd1 &Op1)
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
BinOpPred_match< LHS, RHS, is_logical_shift_op > m_LogicalShift(const LHS &L, const RHS &R)
Matches logical shift operations.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
auto m_Constant()
Match an arbitrary Constant and ignore it.
match_combine_or< match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > >, OpTy > m_ZExtOrSExtOrSelf(const OpTy &Op)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_strictlypositive > m_StrictlyPositive()
Match an integer or vector of strictly positive values.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
auto m_FShl(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
cst_pred_ty< is_negated_power2 > m_NegatedPower2()
Match a integer or vector negated power-of-2.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
cst_pred_ty< custom_checkfn< APInt > > m_CheckedInt(function_ref< bool(const APInt &)> CheckFn)
Match an integer or vector where CheckFn(ele) for each element is true.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_c_MaxOrMin(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWSub(const LHS &L, const RHS &R)
auto m_SMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_FAbs(const Opnd0 &Op0)
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoSignedWrap >, DisjointOr_match< LHS, RHS > > m_NSWAddLike(const LHS &L, const RHS &R)
Match either "add nsw" or "or disjoint".
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
Exact_match< T > m_Exact(const T &SubPattern)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
auto m_UnOp()
Match an arbitrary unary operation and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_MaxOrMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
BinaryOp_match< LHS, RHS, Instruction::SRem > m_SRem(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
auto m_VecReverse(const Opnd0 &Op0)
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoUnsignedWrap >, DisjointOr_match< LHS, RHS > > m_NUWAddLike(const LHS &L, const RHS &R)
Match either "add nuw" or "or disjoint".
BinOpPred_match< LHS, RHS, is_bitwiselogic_op > m_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations.
ElementWiseBitCast_match< OpTy > m_ElementWiseBitCast(const OpTy &Op)
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
auto m_FShr(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
Definition DebugInfo.h:205
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:668
constexpr double e
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Intrinsic::ID getInverseMinMaxIntrinsic(Intrinsic::ID MinMaxID)
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
@ NeverOverflows
Never overflows.
@ AlwaysOverflowsHigh
Always overflows in the direction of signed/unsigned max value.
@ AlwaysOverflowsLow
Always overflows in the direction of signed/unsigned min value.
@ MayOverflow
May or may not overflow.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
LLVM_ABI Value * simplifyFMulInst(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for an FMul, fold the result or return null.
LLVM_ABI bool isValidAssumeForContext(const Instruction *I, const Instruction *CxtI, const DominatorTree *DT=nullptr, bool AllowEphemerals=false)
Return true if it is valid to use the assumptions provided by an assume intrinsic,...
LLVM_ABI APInt possiblyDemandedEltsInMask(Value *Mask)
Given a mask vector of the form <Y x i1>, return an APInt (of bitwidth Y) for each lane which may be ...
BundleAttr getBundleAttrFromOBU(OperandBundleUse OBU)
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool isRemovableAlloc(const CallBase *V, const TargetLibraryInfo *TLI)
Return true if this is a call to an allocation function that does not have side effects that we are r...
LLVM_ABI bool getConstantStringInfo(const Value *V, StringRef &Str, bool TrimAtNul=true)
This function computes the length of a null-terminated C string pointed to by V.
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
Definition MathExtras.h:224
LLVM_ABI Value * lowerObjectSizeCall(IntrinsicInst *ObjectSize, const DataLayout &DL, const TargetLibraryInfo *TLI, bool MustSucceed)
Try to turn a call to @llvm.objectsize into an integer value of the given Type.
LLVM_ABI AssumeSeparateStorageInfo getAssumeSeparateStorageInfo(OperandBundleUse)
LLVM_ABI Value * getAllocAlignment(const CallBase *V, const TargetLibraryInfo *TLI)
Gets the alignment argument for an aligned_alloc-like function, using either built-in knowledge based...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1801
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
Definition MathExtras.h:352
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
Align getKnownAlignment(Value *V, const DataLayout &DL, const Instruction *CxtI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr)
Try to infer an alignment for the specified pointer.
Definition Local.h:240
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1756
SelectPatternFlavor
Specific patterns of select instructions we can match.
@ SPF_ABS
Floating point maxnum.
@ SPF_NABS
Absolute value.
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isModSet(const ModRefInfo MRI)
Definition ModRef.h:49
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1636
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1787
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1701
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI SelectPatternResult matchSelectPattern(Value *V, Value *&LHS, Value *&RHS, Instruction::CastOps *CastOp=nullptr, unsigned Depth=0)
Pattern match integer [SU]MIN, [SU]MAX and ABS idioms, returning the kind and providing the out param...
LLVM_ABI bool matchSimpleBinaryIntrinsicRecurrence(const IntrinsicInst *I, PHINode *&P, Value *&Init, Value *&OtherOp)
Attempt to match a simple value-accumulating recurrence of the form: llvm.intrinsic....
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1777
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1753
bool isAtLeastOrStrongerThan(AtomicOrdering AO, AtomicOrdering Other)
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
iterator_range< SplittingIterator > split(StringRef Str, StringRef Separator)
Split the specified string over a separator and return a range-compatible iterable over its partition...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool isNotCrossLaneOperation(const Instruction *I)
Return true if the instruction doesn't potentially cross vector lanes.
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI bool isKnownNonZero(const Value *V, const SimplifyQuery &Q, unsigned Depth=0)
Return true if the given value is known to be non-zero when defined.
constexpr int PoisonMaskElem
@ Mod
The access may modify the value stored in memory.
Definition ModRef.h:34
LLVM_ABI Value * simplifyFMAFMul(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for the multiplication of a FMA, fold the result or return null.
@ Other
Any other memory.
Definition ModRef.h:68
LLVM_ABI Value * simplifyConstrainedFPCall(CallBase *Call, const SimplifyQuery &Q)
Given a constrained FP intrinsic call, tries to compute its simplified version.
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1737
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI AssumeNonNullInfo getAssumeNonNullInfo(OperandBundleUse)
@ Add
Sum of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
bool isSafeToSpeculativelyExecuteWithVariableReplaced(const Instruction *I, bool IgnoreUBImplyingAttrs=true)
Don't use information from its non-constant operands.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI Value * getFreedOperand(const CallBase *CB, const TargetLibraryInfo *TLI)
If this if a call to a free function, return the freed operand.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
Definition MathExtras.h:233
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
LLVM_ABI std::optional< APInt > getAllocSize(const CallBase *CB, const TargetLibraryInfo *TLI, function_ref< const Value *(const Value *)> Mapper=[](const Value *V) { return V;})
Return the size of the requested allocation.
LLVM_ABI AssumeAlignInfo getAssumeAlignInfo(OperandBundleUse)
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool maskContainsAllOneOrUndef(Value *Mask)
Given a mask vector of i1, Return true if any of the elements of this predicate mask are known to be ...
LLVM_ABI std::optional< bool > isImpliedByDomCondition(const Value *Cond, const Instruction *ContextI, const DataLayout &DL)
Return the boolean condition value in the context of the given instruction if it is known based on do...
LLVM_ABI bool isDereferenceablePointer(const Value *V, Type *Ty, const SimplifyQuery &Q, bool IgnoreFree=false)
Equivalent to isDereferenceableAndAlignedPointer with an alignment of 1.
Definition Loads.cpp:264
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1774
LLVM_ABI bool isKnownNegation(const Value *X, const Value *Y, bool NeedNSW=false, bool AllowPoison=true)
Return true if the two given values are negation.
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1814
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
LLVM_ABI AssumeDereferenceableInfo getAssumeDereferenceableInfo(OperandBundleUse)
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI AssumeNoUndefInfo getAssumeNoUndefInfo(OperandBundleUse)
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI std::optional< bool > computeKnownFPSignBit(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Return false if we can prove that the specified FP value's sign bit is 0.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define NC
Definition regutils.h:42
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
Definition Metadata.h:763
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
@ IEEE
IEEE-754 denormal numbers preserved.
Matching combinators.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Align valueOrOne() const
For convenience, returns a valid alignment or 1 if undefined.
Definition Alignment.h:130
uint32_t getTagID() const
Return the tag of this operand bundle as an integer.
ArrayRef< Use > Inputs
SelectPatternFlavor Flavor
const DataLayout & DL
const Instruction * CxtI
SimplifyQuery getWithInstruction(const Instruction *I) const