LLVM 24.0.0git
InstCombineCalls.cpp
Go to the documentation of this file.
1//===- InstCombineCalls.cpp -----------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the visitCall, visitInvoke, and visitCallBr functions.
10//
11//===----------------------------------------------------------------------===//
12
13#include "InstCombineInternal.h"
14#include "llvm/ADT/APFloat.h"
15#include "llvm/ADT/APInt.h"
16#include "llvm/ADT/APSInt.h"
17#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/Bitset.h"
22#include "llvm/ADT/Statistic.h"
28#include "llvm/Analysis/Loads.h"
33#include "llvm/IR/Attributes.h"
34#include "llvm/IR/BasicBlock.h"
36#include "llvm/IR/Constant.h"
37#include "llvm/IR/Constants.h"
38#include "llvm/IR/DataLayout.h"
39#include "llvm/IR/DebugInfo.h"
41#include "llvm/IR/Function.h"
43#include "llvm/IR/InlineAsm.h"
44#include "llvm/IR/InstrTypes.h"
45#include "llvm/IR/Instruction.h"
48#include "llvm/IR/Intrinsics.h"
49#include "llvm/IR/IntrinsicsAArch64.h"
50#include "llvm/IR/IntrinsicsAMDGPU.h"
51#include "llvm/IR/IntrinsicsARM.h"
52#include "llvm/IR/IntrinsicsHexagon.h"
53#include "llvm/IR/LLVMContext.h"
54#include "llvm/IR/Metadata.h"
57#include "llvm/IR/Statepoint.h"
58#include "llvm/IR/Type.h"
59#include "llvm/IR/User.h"
60#include "llvm/IR/Value.h"
61#include "llvm/IR/ValueHandle.h"
66#include "llvm/Support/Debug.h"
77#include <algorithm>
78#include <cassert>
79#include <cstdint>
80#include <optional>
81#include <utility>
82#include <vector>
83
84#define DEBUG_TYPE "instcombine"
86
87using namespace llvm;
88using namespace PatternMatch;
89
90STATISTIC(NumSimplified, "Number of library calls simplified");
91
93 "instcombine-guard-widening-window",
94 cl::init(3),
95 cl::desc("How wide an instruction window to bypass looking for "
96 "another guard"));
97
98/// Return the specified type promoted as it would be to pass though a va_arg
99/// area.
101 if (IntegerType* ITy = dyn_cast<IntegerType>(Ty)) {
102 if (ITy->getBitWidth() < 32)
103 return Type::getInt32Ty(Ty->getContext());
104 }
105 return Ty;
106}
107
108/// Recognize a memcpy/memmove from a trivially otherwise unused alloca.
109/// TODO: This should probably be integrated with visitAllocSites, but that
110/// requires a deeper change to allow either unread or unwritten objects.
112 auto *Src = MI->getRawSource();
113 while (isa<GetElementPtrInst>(Src)) {
114 if (!Src->hasOneUse())
115 return false;
116 Src = cast<Instruction>(Src)->getOperand(0);
117 }
118 return isa<AllocaInst>(Src) && Src->hasOneUse();
119}
120
122 Align DstAlign = getKnownAlignment(MI->getRawDest(), DL, MI, &AC, &DT);
123 MaybeAlign CopyDstAlign = MI->getDestAlign();
124 if (!CopyDstAlign || *CopyDstAlign < DstAlign) {
125 MI->setDestAlignment(DstAlign);
126 return MI;
127 }
128
129 Align SrcAlign = getKnownAlignment(MI->getRawSource(), DL, MI, &AC, &DT);
130 MaybeAlign CopySrcAlign = MI->getSourceAlign();
131 if (!CopySrcAlign || *CopySrcAlign < SrcAlign) {
132 MI->setSourceAlignment(SrcAlign);
133 return MI;
134 }
135
136 // If we have a store to a location which is known constant, we can conclude
137 // that the store must be storing the constant value (else the memory
138 // wouldn't be constant), and this must be a noop.
139 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
140 // Set the size of the copy to 0, it will be deleted on the next iteration.
141 MI->setLength((uint64_t)0);
142 return MI;
143 }
144
145 // If the source is provably undef, the memcpy/memmove doesn't do anything
146 // (unless the transfer is volatile).
147 if (hasUndefSource(MI) && !MI->isVolatile()) {
148 // Set the size of the copy to 0, it will be deleted on the next iteration.
149 MI->setLength((uint64_t)0);
150 return MI;
151 }
152
153 // If MemCpyInst length is 1/2/4/8 bytes then replace memcpy with
154 // load/store.
155 ConstantInt *MemOpLength = dyn_cast<ConstantInt>(MI->getLength());
156 if (!MemOpLength) return nullptr;
157
158 // Source and destination pointer types are always "i8*" for intrinsic. See
159 // if the size is something we can handle with a single primitive load/store.
160 // A single load+store correctly handles overlapping memory in the memmove
161 // case.
162 uint64_t Size = MemOpLength->getLimitedValue();
163 assert(Size && "0-sized memory transferring should be removed already.");
164
165 if (Size > 8 || (Size&(Size-1)))
166 return nullptr; // If not 1/2/4/8 bytes, exit.
167
168 // If it is an atomic and alignment is less than the size then we will
169 // introduce the unaligned memory access which will be later transformed
170 // into libcall in CodeGen. This is not evident performance gain so disable
171 // it now.
172 if (MI->isAtomic())
173 if (*CopyDstAlign < Size || *CopySrcAlign < Size)
174 return nullptr;
175
176 // Use an integer load+store unless we can find something better.
177 IntegerType* IntType = IntegerType::get(MI->getContext(), Size<<3);
178
179 // If the memcpy has metadata describing the members, see if we can get the
180 // TBAA, scope and noalias tags describing our copy.
181 AAMDNodes AACopyMD = MI->getAAMetadata().adjustForAccess(Size);
182
183 Value *Src = MI->getArgOperand(1);
184 Value *Dest = MI->getArgOperand(0);
185 LoadInst *L = Builder.CreateLoad(IntType, Src);
186 // Alignment from the mem intrinsic will be better, so use it.
187 L->setAlignment(*CopySrcAlign);
188 L->setAAMetadata(AACopyMD);
189 MDNode *LoopMemParallelMD =
190 MI->getMetadata(LLVMContext::MD_mem_parallel_loop_access);
191 if (LoopMemParallelMD)
192 L->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
193 MDNode *AccessGroupMD = MI->getMetadata(LLVMContext::MD_access_group);
194 if (AccessGroupMD)
195 L->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
196
197 StoreInst *S = Builder.CreateStore(L, Dest);
198 // Alignment from the mem intrinsic will be better, so use it.
199 S->setAlignment(*CopyDstAlign);
200 S->setAAMetadata(AACopyMD);
201 if (LoopMemParallelMD)
202 S->setMetadata(LLVMContext::MD_mem_parallel_loop_access, LoopMemParallelMD);
203 if (AccessGroupMD)
204 S->setMetadata(LLVMContext::MD_access_group, AccessGroupMD);
205 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
206
207 if (auto *MT = dyn_cast<MemTransferInst>(MI)) {
208 // non-atomics can be volatile
209 L->setVolatile(MT->isVolatile());
210 S->setVolatile(MT->isVolatile());
211 }
212 if (MI->isAtomic()) {
213 // atomics have to be unordered
214 L->setOrdering(AtomicOrdering::Unordered);
216 }
217
218 // Set the size of the copy to 0, it will be deleted on the next iteration.
219 MI->setLength((uint64_t)0);
220 return MI;
221}
222
224 const Align KnownAlignment =
225 getKnownAlignment(MI->getDest(), DL, MI, &AC, &DT);
226 MaybeAlign MemSetAlign = MI->getDestAlign();
227 if (!MemSetAlign || *MemSetAlign < KnownAlignment) {
228 MI->setDestAlignment(KnownAlignment);
229 return MI;
230 }
231
232 // If we have a store to a location which is known constant, we can conclude
233 // that the store must be storing the constant value (else the memory
234 // wouldn't be constant), and this must be a noop.
235 if (!isModSet(AA->getModRefInfoMask(MI->getDest()))) {
236 // Set the size of the copy to 0, it will be deleted on the next iteration.
237 MI->setLength((uint64_t)0);
238 return MI;
239 }
240
241 // Remove memset with an undef value.
242 // FIXME: This is technically incorrect because it might overwrite a poison
243 // value. Change to PoisonValue once #52930 is resolved.
244 if (isa<UndefValue>(MI->getValue())) {
245 // Set the size of the copy to 0, it will be deleted on the next iteration.
246 MI->setLength((uint64_t)0);
247 return MI;
248 }
249
250 // Extract the length and alignment and fill if they are constant.
251 ConstantInt *LenC = dyn_cast<ConstantInt>(MI->getLength());
252 ConstantInt *FillC = dyn_cast<ConstantInt>(MI->getValue());
253 if (!LenC || !FillC || !FillC->getType()->isIntegerTy(8))
254 return nullptr;
255 const uint64_t Len = LenC->getLimitedValue();
256 assert(Len && "0-sized memory setting should be removed already.");
257 const Align Alignment = MI->getDestAlign().valueOrOne();
258
259 // If it is an atomic and alignment is less than the size then we will
260 // introduce the unaligned memory access which will be later transformed
261 // into libcall in CodeGen. This is not evident performance gain so disable
262 // it now.
263 if (MI->isAtomic() && Alignment < Len)
264 return nullptr;
265
266 // memset(s,c,n) -> store s, c (for n=1,2,4,8)
267 if (Len <= 8 && isPowerOf2_32((uint32_t)Len)) {
268 Value *Dest = MI->getDest();
269
270 // Extract the fill value and store.
271 Constant *FillVal = ConstantInt::get(
272 MI->getContext(), APInt::getSplat(Len * 8, FillC->getValue()));
273 StoreInst *S = Builder.CreateStore(FillVal, Dest, MI->isVolatile());
274 S->copyMetadata(*MI, LLVMContext::MD_DIAssignID);
275 for (DbgVariableRecord *DbgAssign : at::getDVRAssignmentMarkers(S)) {
276 if (llvm::is_contained(DbgAssign->location_ops(), FillC))
277 DbgAssign->replaceVariableLocationOp(FillC, FillVal);
278 }
279
280 S->setAlignment(Alignment);
281 if (MI->isAtomic())
283
284 // Set the size of the copy to 0, it will be deleted on the next iteration.
285 MI->setLength((uint64_t)0);
286 return MI;
287 }
288
289 return nullptr;
290}
291
292// TODO, Obvious Missing Transforms:
293// * Narrow width by halfs excluding zero/undef lanes
294Value *InstCombinerImpl::simplifyMaskedLoad(IntrinsicInst &II) {
295 Value *LoadPtr = II.getArgOperand(0);
296 const Align Alignment = II.getParamAlign(0).valueOrOne();
297 Value *Mask = II.getArgOperand(1);
298
299 // If the mask is all ones or poison, this is a plain vector load of the 1st
300 // argument.
301 if (match(Mask, m_AllOnesOrPoison())) {
302 LoadInst *L = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
303 "unmaskedload");
304 L->copyMetadata(II);
305 return L;
306 }
307
308 // If we can unconditionally load from this address, replace with a
309 // load/select idiom.
310 if (isDereferenceablePointer(LoadPtr, II.getType(),
312 LoadInst *LI = Builder.CreateAlignedLoad(II.getType(), LoadPtr, Alignment,
313 "unmaskedload");
314 LI->copyMetadata(II);
315 return Builder.CreateSelect(II.getArgOperand(1), LI, II.getArgOperand(2));
316 }
317
318 return nullptr;
319}
320
321// TODO, Obvious Missing Transforms:
322// * Single constant active lane -> store
323// * Narrow width by halfs excluding zero/undef lanes
324Instruction *InstCombinerImpl::simplifyMaskedStore(IntrinsicInst &II) {
325 Value *StorePtr = II.getArgOperand(1);
326 Align Alignment = II.getParamAlign(1).valueOrOne();
327 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
328 if (!ConstMask)
329 return nullptr;
330
331 // If the mask is all zeros or poison, this instruction does nothing.
332 if (match(ConstMask, m_ZeroOrPoison()))
334
335 // If the mask is all ones or poison, this is a plain vector store of the 1st
336 // argument.
337 if (match(ConstMask, m_AllOnesOrPoison())) {
338 StoreInst *S =
339 new StoreInst(II.getArgOperand(0), StorePtr, false, Alignment);
340 S->copyMetadata(II);
341 return S;
342 }
343
344 if (isa<ScalableVectorType>(ConstMask->getType()))
345 return nullptr;
346
347 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
348 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
349 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
350 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
351 PoisonElts))
352 return replaceOperand(II, 0, V);
353
354 return nullptr;
355}
356
357// TODO, Obvious Missing Transforms:
358// * Single constant active lane load -> load
359// * Dereferenceable address & few lanes -> scalarize speculative load/selects
360// * Adjacent vector addresses -> masked.load
361// * Narrow width by halfs excluding zero/undef lanes
362// * Vector incrementing address -> vector masked load
363Instruction *InstCombinerImpl::simplifyMaskedGather(IntrinsicInst &II) {
364 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(1));
365 if (!ConstMask)
366 return nullptr;
367
368 // Vector splat address w/known mask -> scalar load
369 // Fold the gather to load the source vector first lane
370 // because it is reloading the same value each time
371 if (ConstMask->isAllOnesValue())
372 if (auto *SplatPtr = getSplatValue(II.getArgOperand(0))) {
373 auto *VecTy = cast<VectorType>(II.getType());
374 const Align Alignment = II.getParamAlign(0).valueOrOne();
375 LoadInst *L = Builder.CreateAlignedLoad(VecTy->getElementType(), SplatPtr,
376 Alignment, "load.scalar");
377 Value *Shuf =
378 Builder.CreateVectorSplat(VecTy->getElementCount(), L, "broadcast");
380 }
381
382 return nullptr;
383}
384
385// TODO, Obvious Missing Transforms:
386// * Single constant active lane -> store
387// * Adjacent vector addresses -> masked.store
388// * Narrow store width by halfs excluding zero/undef lanes
389// * Vector incrementing address -> vector masked store
390Instruction *InstCombinerImpl::simplifyMaskedScatter(IntrinsicInst &II) {
391 auto *ConstMask = dyn_cast<Constant>(II.getArgOperand(2));
392 if (!ConstMask)
393 return nullptr;
394
395 // If the mask is all zeros or poison, a scatter does nothing.
396 if (match(ConstMask, m_ZeroOrPoison()))
398
399 // Vector splat address -> scalar store
400 if (auto *SplatPtr = getSplatValue(II.getArgOperand(1))) {
401 // scatter(splat(value), splat(ptr), non-zero-mask) -> store value, ptr
402 if (auto *SplatValue = getSplatValue(II.getArgOperand(0))) {
403 if (maskContainsAllOneOrUndef(ConstMask)) {
404 Align Alignment = II.getParamAlign(1).valueOrOne();
405 StoreInst *S = new StoreInst(SplatValue, SplatPtr, /*IsVolatile=*/false,
406 Alignment);
407 S->copyMetadata(II);
408 return S;
409 }
410 }
411 // scatter(vector, splat(ptr), splat(true)) -> store extract(vector,
412 // lastlane), ptr
413 if (ConstMask->isAllOnesValue()) {
414 Align Alignment = II.getParamAlign(1).valueOrOne();
415 VectorType *WideLoadTy = cast<VectorType>(II.getArgOperand(1)->getType());
416 ElementCount VF = WideLoadTy->getElementCount();
417 Value *RunTimeVF = Builder.CreateElementCount(Builder.getInt32Ty(), VF);
418 Value *LastLane = Builder.CreateSub(RunTimeVF, Builder.getInt32(1));
419 Value *Extract =
420 Builder.CreateExtractElement(II.getArgOperand(0), LastLane);
421 StoreInst *S =
422 new StoreInst(Extract, SplatPtr, /*IsVolatile=*/false, Alignment);
423 S->copyMetadata(II);
424 return S;
425 }
426 }
427 if (isa<ScalableVectorType>(ConstMask->getType()))
428 return nullptr;
429
430 // Use masked off lanes to simplify operands via SimplifyDemandedVectorElts
431 APInt DemandedElts = possiblyDemandedEltsInMask(ConstMask);
432 APInt PoisonElts(DemandedElts.getBitWidth(), 0);
433 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(0), DemandedElts,
434 PoisonElts))
435 return replaceOperand(II, 0, V);
436 if (Value *V = SimplifyDemandedVectorElts(II.getOperand(1), DemandedElts,
437 PoisonElts))
438 return replaceOperand(II, 1, V);
439
440 return nullptr;
441}
442
443/// This function transforms launder.invariant.group and strip.invariant.group
444/// like:
445/// launder(launder(%x)) -> launder(%x) (the result is not the argument)
446/// launder(strip(%x)) -> launder(%x)
447/// strip(strip(%x)) -> strip(%x) (the result is not the argument)
448/// strip(launder(%x)) -> strip(%x)
449/// This is legal because it preserves the most recent information about
450/// the presence or absence of invariant.group.
452 InstCombinerImpl &IC) {
453 auto *Arg = II.getArgOperand(0);
454 auto *StrippedArg = Arg->stripPointerCasts();
455 auto *StrippedInvariantGroupsArg = StrippedArg;
456 while (auto *Intr = dyn_cast<IntrinsicInst>(StrippedInvariantGroupsArg)) {
457 if (Intr->getIntrinsicID() != Intrinsic::launder_invariant_group &&
458 Intr->getIntrinsicID() != Intrinsic::strip_invariant_group)
459 break;
460 StrippedInvariantGroupsArg = Intr->getArgOperand(0)->stripPointerCasts();
461 }
462 if (StrippedArg == StrippedInvariantGroupsArg)
463 return nullptr; // No launders/strips to remove.
464
465 Value *Result = nullptr;
466
467 if (II.getIntrinsicID() == Intrinsic::launder_invariant_group)
468 Result = IC.Builder.CreateLaunderInvariantGroup(StrippedInvariantGroupsArg);
469 else if (II.getIntrinsicID() == Intrinsic::strip_invariant_group)
470 Result = IC.Builder.CreateStripInvariantGroup(StrippedInvariantGroupsArg);
471 else
473 "simplifyInvariantGroupIntrinsic only handles launder and strip");
474 if (Result->getType()->getPointerAddressSpace() !=
475 II.getType()->getPointerAddressSpace())
476 Result = IC.Builder.CreateAddrSpaceCast(Result, II.getType());
477
478 return cast<Instruction>(Result);
479}
480
482 assert((II.getIntrinsicID() == Intrinsic::cttz ||
483 II.getIntrinsicID() == Intrinsic::ctlz) &&
484 "Expected cttz or ctlz intrinsic");
485 bool IsTZ = II.getIntrinsicID() == Intrinsic::cttz;
486 Value *Op0 = II.getArgOperand(0);
487 Value *Op1 = II.getArgOperand(1);
488 Value *X;
489 // ctlz(bitreverse(x)) -> cttz(x)
490 // cttz(bitreverse(x)) -> ctlz(x)
491 if (match(Op0, m_BitReverse(m_Value(X)))) {
492 Intrinsic::ID ID = IsTZ ? Intrinsic::ctlz : Intrinsic::cttz;
493 Function *F =
494 Intrinsic::getOrInsertDeclaration(II.getModule(), ID, II.getType());
495 return CallInst::Create(F, {X, II.getArgOperand(1)});
496 }
497
498 if (II.getType()->isIntOrIntVectorTy(1)) {
499 // ctlz/cttz i1 Op0 --> not Op0
500 if (match(Op1, m_Zero()))
501 return BinaryOperator::CreateNot(Op0);
502 // If zero is poison, then the input can be assumed to be "true", so the
503 // instruction simplifies to "false".
504 assert(match(Op1, m_One()) && "Expected ctlz/cttz operand to be 0 or 1");
505 return IC.replaceInstUsesWith(II, ConstantInt::getNullValue(II.getType()));
506 }
507
508 // If ctlz/cttz is only used as a shift amount, set is_zero_poison to true.
509 if (II.hasOneUse() && match(Op1, m_Zero()) &&
510 match(II.user_back(), m_Shift(m_Value(), m_Specific(&II))))
511 return CallInst::Create(II.getCalledFunction(),
512 {Op0, IC.Builder.getTrue()});
513
514 Constant *C;
515
516 if (IsTZ) {
517 // cttz(-x) -> cttz(x)
518 if (match(Op0, m_Neg(m_Value(X))))
519 return CallInst::Create(II.getCalledFunction(), {X, Op1});
520
521 // cttz(-x & x) -> cttz(x)
522 if (match(Op0, m_c_And(m_Neg(m_Value(X)), m_Deferred(X))))
523 return CallInst::Create(II.getCalledFunction(), {X, Op1});
524
525 // cttz(sext(x)) -> cttz(zext(x))
526 if (match(Op0, m_OneUse(m_SExt(m_Value(X))))) {
527 auto *Zext = IC.Builder.CreateZExt(X, II.getType());
528 auto *CttzZext =
529 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, Zext, Op1);
530 return IC.replaceInstUsesWith(II, CttzZext);
531 }
532
533 // Zext doesn't change the number of trailing zeros, so narrow:
534 // cttz(zext(x)) -> zext(cttz(x)) if the 'ZeroIsPoison' parameter is 'true'.
535 if (match(Op0, m_OneUse(m_ZExt(m_Value(X)))) && match(Op1, m_One())) {
536 auto *Cttz = IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, X,
537 IC.Builder.getTrue());
538 auto *ZextCttz = IC.Builder.CreateZExt(Cttz, II.getType());
539 return IC.replaceInstUsesWith(II, ZextCttz);
540 }
541
542 // cttz(abs(x)) -> cttz(x)
543 // cttz(nabs(x)) -> cttz(x)
544 Value *Y;
546 if (SPF == SPF_ABS || SPF == SPF_NABS)
547 return CallInst::Create(II.getCalledFunction(), {X, Op1});
548
550 return CallInst::Create(II.getCalledFunction(), {X, Op1});
551
552 // cttz(shl(%const, %val), 1) --> add(cttz(%const, 1), %val)
553 if (match(Op0, m_Shl(m_ImmConstant(C), m_Value(X))) &&
554 match(Op1, m_One())) {
555 Value *ConstCttz =
556 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
557 return BinaryOperator::CreateAdd(ConstCttz, X);
558 }
559
560 // cttz(lshr exact (%const, %val), 1) --> sub(cttz(%const, 1), %val)
561 if (match(Op0, m_Exact(m_LShr(m_ImmConstant(C), m_Value(X)))) &&
562 match(Op1, m_One())) {
563 Value *ConstCttz =
564 IC.Builder.CreateBinaryIntrinsic(Intrinsic::cttz, C, Op1);
565 return BinaryOperator::CreateSub(ConstCttz, X);
566 }
567
568 // cttz(add(lshr(UINT_MAX, %val), 1)) --> sub(width, %val)
569 if (match(Op0, m_Add(m_LShr(m_AllOnes(), m_Value(X)), m_One()))) {
570 Value *Width =
571 ConstantInt::get(II.getType(), II.getType()->getScalarSizeInBits());
572 return BinaryOperator::CreateSub(Width, X);
573 }
574 } else {
575 // ctlz(lshr(%const, %val), 1) --> add(ctlz(%const, 1), %val)
576 if (match(Op0, m_LShr(m_ImmConstant(C), m_Value(X))) &&
577 match(Op1, m_One())) {
578 Value *ConstCtlz =
579 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
580 return BinaryOperator::CreateAdd(ConstCtlz, X);
581 }
582
583 // ctlz(shl nuw (%const, %val), 1) --> sub(ctlz(%const, 1), %val)
584 if (match(Op0, m_NUWShl(m_ImmConstant(C), m_Value(X))) &&
585 match(Op1, m_One())) {
586 Value *ConstCtlz =
587 IC.Builder.CreateBinaryIntrinsic(Intrinsic::ctlz, C, Op1);
588 return BinaryOperator::CreateSub(ConstCtlz, X);
589 }
590
591 // ctlz(~x & (x - 1)) -> bitwidth - cttz(x, false)
592 if (Op0->hasOneUse() &&
593 match(Op0,
595 Type *Ty = II.getType();
596 unsigned BitWidth = Ty->getScalarSizeInBits();
597 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
598 {X, IC.Builder.getFalse()});
599 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
600 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
601 }
602 }
603
604 // cttz(Pow2) -> Log2(Pow2)
605 // ctlz(Pow2) -> BitWidth - 1 - Log2(Pow2)
606 if (auto *R = IC.tryGetLog2(Op0, match(Op1, m_One()))) {
607 if (IsTZ)
608 return IC.replaceInstUsesWith(II, R);
609 BinaryOperator *BO = BinaryOperator::CreateSub(
610 ConstantInt::get(R->getType(), R->getType()->getScalarSizeInBits() - 1),
611 R);
612 BO->setHasNoSignedWrap();
614 return BO;
615 }
616
618
619 // Create a mask for bits above (ctlz) or below (cttz) the first known one.
620 unsigned PossibleZeros = IsTZ ? Known.countMaxTrailingZeros()
621 : Known.countMaxLeadingZeros();
622 unsigned DefiniteZeros = IsTZ ? Known.countMinTrailingZeros()
623 : Known.countMinLeadingZeros();
624
625 // If all bits above (ctlz) or below (cttz) the first known one are known
626 // zero, this value is constant.
627 // FIXME: This should be in InstSimplify because we're replacing an
628 // instruction with a constant.
629 if (PossibleZeros == DefiniteZeros) {
630 auto *C = ConstantInt::get(Op0->getType(), DefiniteZeros);
631 return IC.replaceInstUsesWith(II, C);
632 }
633
634 // If the input to cttz/ctlz is known to be non-zero,
635 // then change the 'ZeroIsPoison' parameter to 'true'
636 // because we know the zero behavior can't affect the result.
637 if (!Known.One.isZero() ||
639 if (!match(II.getArgOperand(1), m_One()))
640 return CallInst::Create(II.getCalledFunction(),
641 {Op0, IC.Builder.getTrue()});
642 }
643
644 // Add range attribute since known bits can't completely reflect what we know.
645 unsigned BitWidth = Op0->getType()->getScalarSizeInBits();
646 if (BitWidth != 1 && !II.hasRetAttr(Attribute::Range) &&
647 !II.getMetadata(LLVMContext::MD_range)) {
648 ConstantRange Range(APInt(BitWidth, DefiniteZeros),
649 APInt(BitWidth, PossibleZeros + 1));
650 II.addRangeRetAttr(Range);
651 return &II;
652 }
653
654 return nullptr;
655}
656
658 assert(II.getIntrinsicID() == Intrinsic::ctpop &&
659 "Expected ctpop intrinsic");
660 Type *Ty = II.getType();
661 unsigned BitWidth = Ty->getScalarSizeInBits();
662 Value *Op0 = II.getArgOperand(0);
663 Value *X, *Y;
664
665 // ctpop(bitreverse(x)) -> ctpop(x)
666 // ctpop(bswap(x)) -> ctpop(x)
667 if (match(Op0, m_BitReverse(m_Value(X))) || match(Op0, m_BSwap(m_Value(X))))
668 return CallInst::Create(II.getCalledFunction(), X);
669
670 // ctpop(rot(x)) -> ctpop(x)
671 if ((match(Op0, m_FShl(m_Value(X), m_Value(Y), m_Value())) ||
672 match(Op0, m_FShr(m_Value(X), m_Value(Y), m_Value()))) &&
673 X == Y)
674 return CallInst::Create(II.getCalledFunction(), X);
675
676 // ctpop(x | -x) -> bitwidth - cttz(x, false)
677 if (Op0->hasOneUse() &&
678 match(Op0, m_c_Or(m_Value(X), m_Neg(m_Deferred(X))))) {
679 auto *Cttz = IC.Builder.CreateIntrinsic(Intrinsic::cttz, Ty,
680 {X, IC.Builder.getFalse()});
681 auto *Bw = ConstantInt::get(Ty, APInt(BitWidth, BitWidth));
682 return IC.replaceInstUsesWith(II, IC.Builder.CreateSub(Bw, Cttz));
683 }
684
685 // ctpop(~x & (x - 1)) -> cttz(x, false)
686 if (match(Op0,
688 Function *F =
689 Intrinsic::getOrInsertDeclaration(II.getModule(), Intrinsic::cttz, Ty);
690 return CallInst::Create(F, {X, IC.Builder.getFalse()});
691 }
692
693 // Zext doesn't change the number of set bits, so narrow:
694 // ctpop (zext X) --> zext (ctpop X)
695 if (match(Op0, m_OneUse(m_ZExt(m_Value(X))))) {
696 Value *NarrowPop = IC.Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, X);
697 return CastInst::Create(Instruction::ZExt, NarrowPop, Ty);
698 }
699
701 IC.computeKnownBits(Op0, Known, &II);
702
703 // If all bits are zero except for exactly one fixed bit, then the result
704 // must be 0 or 1, and we can get that answer by shifting to LSB:
705 // ctpop (X & 32) --> (X & 32) >> 5
706 // TODO: Investigate removing this as its likely unnecessary given the below
707 // `isKnownToBeAPowerOfTwo` check.
708 if ((~Known.Zero).isPowerOf2())
709 return BinaryOperator::CreateLShr(
710 Op0, ConstantInt::get(Ty, (~Known.Zero).exactLogBase2()));
711
712 // More generally we can also handle non-constant power of 2 patterns such as
713 // shl/shr(Pow2, X), (X & -X), etc... by transforming:
714 // ctpop(Pow2OrZero) --> icmp ne X, 0
715 if (IC.isKnownToBeAPowerOfTwo(Op0, /* OrZero */ true))
716 return CastInst::Create(Instruction::ZExt,
719 Ty);
720
721 // Add range attribute since known bits can't completely reflect what we know.
722 if (BitWidth != 1) {
723 ConstantRange OldRange =
724 II.getRange().value_or(ConstantRange::getFull(BitWidth));
725
726 unsigned Lower = Known.countMinPopulation();
727 unsigned Upper = Known.countMaxPopulation() + 1;
728
729 if (Lower == 0 && OldRange.contains(APInt::getZero(BitWidth)) &&
731 Lower = 1;
732
734 Range = Range.intersectWith(OldRange, ConstantRange::Unsigned);
735
736 if (Range != OldRange) {
737 II.addRangeRetAttr(Range);
738 return &II;
739 }
740 }
741
742 return nullptr;
743}
744
745/// Convert `tbl`/`tbx` intrinsics to shufflevector if the mask is constant, and
746/// at most two source operands are actually referenced.
748 bool IsExtension) {
749 // Bail out if the mask is not a constant.
750 auto *C = dyn_cast<Constant>(II.getArgOperand(II.arg_size() - 1));
751 if (!C)
752 return nullptr;
753
754 auto *RetTy = cast<FixedVectorType>(II.getType());
755 unsigned NumIndexes = RetTy->getNumElements();
756
757 // Only perform this transformation for <8 x i8> and <16 x i8> vector types.
758 if (!RetTy->getElementType()->isIntegerTy(8) ||
759 (NumIndexes != 8 && NumIndexes != 16))
760 return nullptr;
761
762 // For tbx instructions, the first argument is the "fallback" vector, which
763 // has the same length as the mask and return type.
764 unsigned int StartIndex = (unsigned)IsExtension;
765 auto *SourceTy =
766 cast<FixedVectorType>(II.getArgOperand(StartIndex)->getType());
767 // Note that the element count of each source vector does *not* need to be the
768 // same as the element count of the return type and mask! All source vectors
769 // must have the same element count as each other, though.
770 unsigned NumElementsPerSource = SourceTy->getNumElements();
771
772 // There are no tbl/tbx intrinsics for which the destination size exceeds the
773 // source size. However, our definitions of the intrinsics, at least in
774 // IntrinsicsAArch64.td, allow for arbitrary destination vector sizes, so it
775 // *could* technically happen.
776 if (NumIndexes > NumElementsPerSource)
777 return nullptr;
778
779 // The tbl/tbx intrinsics take several source operands followed by a mask
780 // operand.
781 unsigned int NumSourceOperands = II.arg_size() - 1 - (unsigned)IsExtension;
782
783 // Map input operands to shuffle indices. This also helpfully deduplicates the
784 // input arguments, in case the same value is passed as an argument multiple
785 // times.
786 SmallDenseMap<Value *, unsigned, 2> ValueToShuffleSlot;
787 Value *ShuffleOperands[2] = {PoisonValue::get(SourceTy),
788 PoisonValue::get(SourceTy)};
789
790 int Indexes[16];
791 for (unsigned I = 0; I < NumIndexes; ++I) {
792 Constant *COp = C->getAggregateElement(I);
793
794 if (!COp || (!isa<UndefValue>(COp) && !isa<ConstantInt>(COp)))
795 return nullptr;
796
797 if (isa<UndefValue>(COp)) {
798 Indexes[I] = -1;
799 continue;
800 }
801
802 uint64_t Index = cast<ConstantInt>(COp)->getZExtValue();
803 // The index of the input argument that this index references (0 = first
804 // source argument, etc).
805 unsigned SourceOperandIndex = Index / NumElementsPerSource;
806 // The index of the element at that source operand.
807 unsigned SourceOperandElementIndex = Index % NumElementsPerSource;
808
809 Value *SourceOperand;
810 if (SourceOperandIndex >= NumSourceOperands) {
811 // This index is out of bounds. Map it to index into either the fallback
812 // vector (tbx) or vector of zeroes (tbl).
813 SourceOperandIndex = NumSourceOperands;
814 if (IsExtension) {
815 // For out-of-bounds indices in tbx, choose the `I`th element of the
816 // fallback.
817 SourceOperand = II.getArgOperand(0);
818 SourceOperandElementIndex = I;
819 } else {
820 // Otherwise, choose some element from the dummy vector of zeroes (we'll
821 // always choose the first).
822 SourceOperand = Constant::getNullValue(SourceTy);
823 SourceOperandElementIndex = 0;
824 }
825 } else {
826 SourceOperand = II.getArgOperand(SourceOperandIndex + StartIndex);
827 }
828
829 // The source operand may be the fallback vector, which may not have the
830 // same number of elements as the source vector. In that case, we *could*
831 // choose to extend its length with another shufflevector, but it's simpler
832 // to just bail instead.
833 if (cast<FixedVectorType>(SourceOperand->getType())->getNumElements() !=
834 NumElementsPerSource)
835 return nullptr;
836
837 // We now know the source operand referenced by this index. Make it a
838 // shufflevector operand, if it isn't already.
839 unsigned NumSlots = ValueToShuffleSlot.size();
840 // This shuffle references more than two sources, and hence cannot be
841 // represented as a shufflevector.
842 if (NumSlots == 2 && !ValueToShuffleSlot.contains(SourceOperand))
843 return nullptr;
844
845 auto [It, Inserted] =
846 ValueToShuffleSlot.try_emplace(SourceOperand, NumSlots);
847 if (Inserted)
848 ShuffleOperands[It->getSecond()] = SourceOperand;
849
850 unsigned RemappedIndex =
851 (It->getSecond() * NumElementsPerSource) + SourceOperandElementIndex;
852 Indexes[I] = RemappedIndex;
853 }
854
856 ShuffleOperands[0], ShuffleOperands[1], ArrayRef(Indexes, NumIndexes));
857 return IC.replaceInstUsesWith(II, Shuf);
858}
859
860// Returns true iff the 2 intrinsics have the same operands, limiting the
861// comparison to the first NumOperands.
862static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E,
863 unsigned NumOperands) {
864 assert(I.arg_size() >= NumOperands && "Not enough operands");
865 assert(E.arg_size() >= NumOperands && "Not enough operands");
866 for (unsigned i = 0; i < NumOperands; i++)
867 if (I.getArgOperand(i) != E.getArgOperand(i))
868 return false;
869 return true;
870}
871
872// Remove trivially empty start/end intrinsic ranges, i.e. a start
873// immediately followed by an end (ignoring debuginfo or other
874// start/end intrinsics in between). As this handles only the most trivial
875// cases, tracking the nesting level is not needed:
876//
877// call @llvm.foo.start(i1 0)
878// call @llvm.foo.start(i1 0) ; This one won't be skipped: it will be removed
879// call @llvm.foo.end(i1 0)
880// call @llvm.foo.end(i1 0) ; &I
881static bool
883 std::function<bool(const IntrinsicInst &)> IsStart) {
884 // We start from the end intrinsic and scan backwards, so that InstCombine
885 // has already processed (and potentially removed) all the instructions
886 // before the end intrinsic.
887 BasicBlock::reverse_iterator BI(EndI), BE(EndI.getParent()->rend());
888 for (; BI != BE; ++BI) {
889 if (auto *I = dyn_cast<IntrinsicInst>(&*BI)) {
890 if (I->isDebugOrPseudoInst() ||
891 I->getIntrinsicID() == EndI.getIntrinsicID())
892 continue;
893 if (IsStart(*I)) {
894 if (haveSameOperands(EndI, *I, EndI.arg_size())) {
896 IC.eraseInstFromFunction(EndI);
897 return true;
898 }
899 // Skip start intrinsics that don't pair with this end intrinsic.
900 continue;
901 }
902 }
903 break;
904 }
905
906 return false;
907}
908
910 removeTriviallyEmptyRange(I, *this, [&I](const IntrinsicInst &II) {
911 // Bail out on the case where the source va_list of a va_copy is destroyed
912 // immediately by a follow-up va_end.
913 return II.getIntrinsicID() == Intrinsic::vastart ||
914 (II.getIntrinsicID() == Intrinsic::vacopy &&
915 I.getArgOperand(0) != II.getArgOperand(1));
916 });
917 return nullptr;
918}
919
921 assert(Call.arg_size() > 1 && "Need at least 2 args to swap");
922 Value *Arg0 = Call.getArgOperand(0), *Arg1 = Call.getArgOperand(1);
923 if (isa<Constant>(Arg0) && !isa<Constant>(Arg1)) {
924 Call.setArgOperand(0, Arg1);
925 Call.setArgOperand(1, Arg0);
926 AttributeList CallAttr = Call.getAttributes();
927 AttributeSet LHSAttr = CallAttr.getParamAttrs(0);
928 AttributeSet RHSAttr = CallAttr.getParamAttrs(1);
929 LLVMContext &Ctx = Call.getContext();
930 Call.setAttributes(CallAttr
931 .setAttributesAtIndex(
932 Ctx, AttributeList::FirstArgIndex + 0, RHSAttr)
933 .setAttributesAtIndex(
934 Ctx, AttributeList::FirstArgIndex + 1, LHSAttr));
935 return &Call;
936 }
937 return nullptr;
938}
939
940/// Creates a result tuple for an overflow intrinsic \p II with a given
941/// \p Result and a constant \p Overflow value.
943 Constant *Overflow) {
944 Constant *V[] = {PoisonValue::get(Result->getType()), Overflow};
945 StructType *ST = cast<StructType>(II->getType());
946 Constant *Struct = ConstantStruct::get(ST, V);
947 return InsertValueInst::Create(Struct, Result, 0);
948}
949
951InstCombinerImpl::foldIntrinsicWithOverflowCommon(IntrinsicInst *II) {
952 WithOverflowInst *WO = cast<WithOverflowInst>(II);
953 Value *OperationResult = nullptr;
954 Constant *OverflowResult = nullptr;
955 if (OptimizeOverflowCheck(WO->getBinaryOp(), WO->isSigned(), WO->getLHS(),
956 WO->getRHS(), *WO, OperationResult, OverflowResult))
957 return createOverflowTuple(WO, OperationResult, OverflowResult);
958
959 // See whether we can optimize the overflow check with assumption information.
960 for (User *U : WO->users()) {
961 if (!match(U, m_ExtractValue<1>(m_Value())))
962 continue;
963
964 for (auto &AssumeVH : AC.assumptionsFor(U)) {
965 if (!AssumeVH)
966 continue;
967 CallInst *I = cast<CallInst>(AssumeVH);
968 if (!match(I->getArgOperand(0), m_Not(m_Specific(U))))
969 continue;
970 if (!isValidAssumeForContext(I, II, /*DT=*/nullptr,
971 /*AllowEphemerals=*/true))
972 continue;
973 Value *Result =
974 Builder.CreateBinOp(WO->getBinaryOp(), WO->getLHS(), WO->getRHS());
975 Result->takeName(WO);
976 if (auto *Inst = dyn_cast<Instruction>(Result)) {
977 if (WO->isSigned())
978 Inst->setHasNoSignedWrap();
979 else
980 Inst->setHasNoUnsignedWrap();
981 }
982 return createOverflowTuple(WO, Result,
983 ConstantInt::getFalse(U->getType()));
984 }
985 }
986
987 return nullptr;
988}
989
990static bool inputDenormalIsIEEE(const Function &F, const Type *Ty) {
991 Ty = Ty->getScalarType();
992 return F.getDenormalMode(Ty->getFltSemantics()).Input == DenormalMode::IEEE;
993}
994
995static bool inputDenormalIsDAZ(const Function &F, const Type *Ty) {
996 Ty = Ty->getScalarType();
997 return F.getDenormalMode(Ty->getFltSemantics()).inputsAreZero();
998}
999
1000/// \returns the compare predicate type if the test performed by
1001/// llvm.is.fpclass(x, \p Mask) is equivalent to fcmp o__ x, 0.0 with the
1002/// floating-point environment assumed for \p F for type \p Ty
1004 const Function &F, Type *Ty) {
1005 switch (static_cast<unsigned>(Mask)) {
1006 case fcZero:
1007 if (inputDenormalIsIEEE(F, Ty))
1008 return FCmpInst::FCMP_OEQ;
1009 break;
1010 case fcZero | fcSubnormal:
1011 if (inputDenormalIsDAZ(F, Ty))
1012 return FCmpInst::FCMP_OEQ;
1013 break;
1014 case fcPositive | fcNegZero:
1015 if (inputDenormalIsIEEE(F, Ty))
1016 return FCmpInst::FCMP_OGE;
1017 break;
1019 if (inputDenormalIsDAZ(F, Ty))
1020 return FCmpInst::FCMP_OGE;
1021 break;
1023 if (inputDenormalIsIEEE(F, Ty))
1024 return FCmpInst::FCMP_OGT;
1025 break;
1026 case fcNegative | fcPosZero:
1027 if (inputDenormalIsIEEE(F, Ty))
1028 return FCmpInst::FCMP_OLE;
1029 break;
1031 if (inputDenormalIsDAZ(F, Ty))
1032 return FCmpInst::FCMP_OLE;
1033 break;
1035 if (inputDenormalIsIEEE(F, Ty))
1036 return FCmpInst::FCMP_OLT;
1037 break;
1038 case fcPosNormal | fcPosInf:
1039 if (inputDenormalIsDAZ(F, Ty))
1040 return FCmpInst::FCMP_OGT;
1041 break;
1042 case fcNegNormal | fcNegInf:
1043 if (inputDenormalIsDAZ(F, Ty))
1044 return FCmpInst::FCMP_OLT;
1045 break;
1046 case ~fcZero & ~fcNan:
1047 if (inputDenormalIsIEEE(F, Ty))
1048 return FCmpInst::FCMP_ONE;
1049 break;
1050 case ~(fcZero | fcSubnormal) & ~fcNan:
1051 if (inputDenormalIsDAZ(F, Ty))
1052 return FCmpInst::FCMP_ONE;
1053 break;
1054 default:
1055 break;
1056 }
1057
1059}
1060
1061Instruction *InstCombinerImpl::foldIntrinsicIsFPClass(IntrinsicInst &II) {
1062 Value *Src0 = II.getArgOperand(0);
1063 Value *Src1 = II.getArgOperand(1);
1064 const ConstantInt *CMask = cast<ConstantInt>(Src1);
1065 FPClassTest Mask = static_cast<FPClassTest>(CMask->getZExtValue());
1066 const bool IsUnordered = (Mask & fcNan) == fcNan;
1067 const bool IsOrdered = (Mask & fcNan) == fcNone;
1068 const FPClassTest OrderedMask = Mask & ~fcNan;
1069 const FPClassTest OrderedInvertedMask = ~OrderedMask & ~fcNan;
1070
1071 const bool IsStrict =
1072 II.getFunction()->getAttributes().hasFnAttr(Attribute::StrictFP);
1073
1074 Value *FNegSrc;
1075 // is.fpclass (fneg x), mask -> is.fpclass x, (fneg mask)
1076 if (match(Src0, m_FNeg(m_Value(FNegSrc))))
1077 return CallInst::Create(
1078 II.getCalledFunction(),
1079 {FNegSrc, ConstantInt::get(Src1->getType(), fneg(Mask))});
1080
1081 Value *FAbsSrc;
1082 if (match(Src0, m_FAbs(m_Value(FAbsSrc))))
1083 return CallInst::Create(
1084 II.getCalledFunction(),
1085 {FAbsSrc, ConstantInt::get(Src1->getType(), inverse_fabs(Mask))});
1086
1087 if ((OrderedMask == fcInf || OrderedInvertedMask == fcInf) &&
1088 (IsOrdered || IsUnordered) && !IsStrict) {
1089 // is.fpclass(x, fcInf) -> fcmp oeq fabs(x), +inf
1090 // is.fpclass(x, ~fcInf) -> fcmp one fabs(x), +inf
1091 // is.fpclass(x, fcInf|fcNan) -> fcmp ueq fabs(x), +inf
1092 // is.fpclass(x, ~(fcInf|fcNan)) -> fcmp une fabs(x), +inf
1094 FCmpInst::Predicate Pred =
1095 IsUnordered ? FCmpInst::FCMP_UEQ : FCmpInst::FCMP_OEQ;
1096 if (OrderedInvertedMask == fcInf)
1097 Pred = IsUnordered ? FCmpInst::FCMP_UNE : FCmpInst::FCMP_ONE;
1098
1099 Value *Fabs = Builder.CreateFAbs(Src0);
1100 Value *CmpInf = Builder.CreateFCmp(Pred, Fabs, Inf);
1101 CmpInf->takeName(&II);
1102 return replaceInstUsesWith(II, CmpInf);
1103 }
1104
1105 if ((OrderedMask == fcPosInf || OrderedMask == fcNegInf) &&
1106 (IsOrdered || IsUnordered) && !IsStrict) {
1107 // is.fpclass(x, fcPosInf) -> fcmp oeq x, +inf
1108 // is.fpclass(x, fcNegInf) -> fcmp oeq x, -inf
1109 // is.fpclass(x, fcPosInf|fcNan) -> fcmp ueq x, +inf
1110 // is.fpclass(x, fcNegInf|fcNan) -> fcmp ueq x, -inf
1111 Constant *Inf =
1112 ConstantFP::getInfinity(Src0->getType(), OrderedMask == fcNegInf);
1113 Value *EqInf = IsUnordered ? Builder.CreateFCmpUEQ(Src0, Inf)
1114 : Builder.CreateFCmpOEQ(Src0, Inf);
1115
1116 EqInf->takeName(&II);
1117 return replaceInstUsesWith(II, EqInf);
1118 }
1119
1120 if ((OrderedInvertedMask == fcPosInf || OrderedInvertedMask == fcNegInf) &&
1121 (IsOrdered || IsUnordered) && !IsStrict) {
1122 // is.fpclass(x, ~fcPosInf) -> fcmp one x, +inf
1123 // is.fpclass(x, ~fcNegInf) -> fcmp one x, -inf
1124 // is.fpclass(x, ~fcPosInf|fcNan) -> fcmp une x, +inf
1125 // is.fpclass(x, ~fcNegInf|fcNan) -> fcmp une x, -inf
1127 OrderedInvertedMask == fcNegInf);
1128 Value *NeInf = IsUnordered ? Builder.CreateFCmpUNE(Src0, Inf)
1129 : Builder.CreateFCmpONE(Src0, Inf);
1130 NeInf->takeName(&II);
1131 return replaceInstUsesWith(II, NeInf);
1132 }
1133
1134 if (Mask == fcNan && !IsStrict) {
1135 // Equivalent of isnan. Replace with standard fcmp if we don't care about FP
1136 // exceptions.
1137 Value *IsNan =
1138 Builder.CreateFCmpUNO(Src0, ConstantFP::getZero(Src0->getType()));
1139 IsNan->takeName(&II);
1140 return replaceInstUsesWith(II, IsNan);
1141 }
1142
1143 if (Mask == (~fcNan & fcAllFlags) && !IsStrict) {
1144 // Equivalent of !isnan. Replace with standard fcmp.
1145 Value *FCmp =
1146 Builder.CreateFCmpORD(Src0, ConstantFP::getZero(Src0->getType()));
1147 FCmp->takeName(&II);
1148 return replaceInstUsesWith(II, FCmp);
1149 }
1150
1152
1153 // Try to replace with an fcmp with 0
1154 //
1155 // is.fpclass(x, fcZero) -> fcmp oeq x, 0.0
1156 // is.fpclass(x, fcZero | fcNan) -> fcmp ueq x, 0.0
1157 // is.fpclass(x, ~fcZero & ~fcNan) -> fcmp one x, 0.0
1158 // is.fpclass(x, ~fcZero) -> fcmp une x, 0.0
1159 //
1160 // is.fpclass(x, fcPosSubnormal | fcPosNormal | fcPosInf) -> fcmp ogt x, 0.0
1161 // is.fpclass(x, fcPositive | fcNegZero) -> fcmp oge x, 0.0
1162 //
1163 // is.fpclass(x, fcNegSubnormal | fcNegNormal | fcNegInf) -> fcmp olt x, 0.0
1164 // is.fpclass(x, fcNegative | fcPosZero) -> fcmp ole x, 0.0
1165 //
1166 if (!IsStrict && (IsOrdered || IsUnordered) &&
1167 (PredType = fpclassTestIsFCmp0(OrderedMask, *II.getFunction(),
1168 Src0->getType())) !=
1171 // Equivalent of == 0.
1172 Value *FCmp = Builder.CreateFCmp(
1173 IsUnordered ? FCmpInst::getUnorderedPredicate(PredType) : PredType,
1174 Src0, Zero);
1175
1176 FCmp->takeName(&II);
1177 return replaceInstUsesWith(II, FCmp);
1178 }
1179
1180 KnownFPClass Known =
1181 computeKnownFPClass(Src0, Mask, SQ.getWithInstruction(&II));
1182
1183 // Clear test bits we know must be false from the source value.
1184 // fp_class (nnan x), qnan|snan|other -> fp_class (nnan x), other
1185 // fp_class (ninf x), ninf|pinf|other -> fp_class (ninf x), other
1186 if ((Mask & Known.KnownFPClasses) != Mask) {
1187 II.setArgOperand(
1188 1, ConstantInt::get(Src1->getType(), Mask & Known.KnownFPClasses));
1189 return &II;
1190 }
1191
1192 // If none of the tests which can return false are possible, fold to true.
1193 // fp_class (nnan x), ~(qnan|snan) -> true
1194 // fp_class (ninf x), ~(ninf|pinf) -> true
1195 if (Mask == Known.KnownFPClasses)
1196 return replaceInstUsesWith(II, ConstantInt::get(II.getType(), true));
1197
1198 return nullptr;
1199}
1200
1201static std::optional<bool> getKnownSign(Value *Op, const SimplifyQuery &SQ) {
1203 if (Known.isNonNegative())
1204 return false;
1205 if (Known.isNegative())
1206 return true;
1207
1208 Value *X, *Y;
1209 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1211
1212 return std::nullopt;
1213}
1214
1215static std::optional<bool> getKnownSignOrZero(Value *Op,
1216 const SimplifyQuery &SQ) {
1217 if (std::optional<bool> Sign = getKnownSign(Op, SQ))
1218 return Sign;
1219
1220 Value *X, *Y;
1221 if (match(Op, m_NSWSub(m_Value(X), m_Value(Y))))
1223
1224 return std::nullopt;
1225}
1226
1227/// Return true if two values \p Op0 and \p Op1 are known to have the same sign.
1228static bool signBitMustBeTheSame(Value *Op0, Value *Op1,
1229 const SimplifyQuery &SQ) {
1230 std::optional<bool> Known1 = getKnownSign(Op1, SQ);
1231 if (!Known1)
1232 return false;
1233 std::optional<bool> Known0 = getKnownSign(Op0, SQ);
1234 if (!Known0)
1235 return false;
1236 return *Known0 == *Known1;
1237}
1238
1239// Determines if ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b)) is safe.
1240//
1241// This is true if, when the add saturates, the resulting ldexp is guaranteed to
1242// produce 0 or inf.
1243static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy) {
1244 const fltSemantics &FltSem = FpTy->getScalarType()->getFltSemantics();
1245 if (!APFloat::semanticsHasInf(FltSem))
1246 return false;
1247
1248 // Cap ExpBits at 32 because scalbn takes an int. This is sufficient for any
1249 // reasonable fp type (for example, `double` only has 11 exponent bits).
1250 unsigned ExpBits = std::min(ExpTy->getScalarSizeInBits(), 32u);
1251 int SignedMax = static_cast<int>(maxIntN(ExpBits));
1252 int SignedMin = static_cast<int>(minIntN(ExpBits));
1253 APFloat ScaledUp = scalbn(APFloat::getSmallest(FltSem), SignedMax,
1255 APFloat ScaledDown = scalbn(APFloat::getLargest(FltSem), SignedMin,
1257 return ScaledUp.isInfinity() && ScaledDown.isZero();
1258}
1259
1260/// Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0. This
1261/// can trigger other combines.
1263 InstCombiner::BuilderTy &Builder) {
1264 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1265 assert((MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin ||
1266 MinMaxID == Intrinsic::umax || MinMaxID == Intrinsic::umin) &&
1267 "Expected a min or max intrinsic");
1268
1269 // TODO: Match vectors with undef elements, but undef may not propagate.
1270 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
1271 Value *X;
1272 const APInt *C0, *C1;
1273 if (!match(Op0, m_OneUse(m_Add(m_Value(X), m_APInt(C0)))) ||
1274 !match(Op1, m_APInt(C1)))
1275 return nullptr;
1276
1277 // Check for necessary no-wrap and overflow constraints.
1278 bool IsSigned = MinMaxID == Intrinsic::smax || MinMaxID == Intrinsic::smin;
1279 auto *Add = cast<BinaryOperator>(Op0);
1280 if ((IsSigned && !Add->hasNoSignedWrap()) ||
1281 (!IsSigned && !Add->hasNoUnsignedWrap()))
1282 return nullptr;
1283
1284 // If the constant difference overflows, then instsimplify should reduce the
1285 // min/max to the add or C1.
1286 bool Overflow;
1287 APInt CDiff =
1288 IsSigned ? C1->ssub_ov(*C0, Overflow) : C1->usub_ov(*C0, Overflow);
1289 assert(!Overflow && "Expected simplify of min/max");
1290
1291 // min/max (add X, C0), C1 --> add (min/max X, C1 - C0), C0
1292 // Note: the "mismatched" no-overflow setting does not propagate.
1293 Constant *NewMinMaxC = ConstantInt::get(II->getType(), CDiff);
1294 Value *NewMinMax = Builder.CreateBinaryIntrinsic(MinMaxID, X, NewMinMaxC);
1295 return IsSigned ? BinaryOperator::CreateNSWAdd(NewMinMax, Add->getOperand(1))
1296 : BinaryOperator::CreateNUWAdd(NewMinMax, Add->getOperand(1));
1297}
1298/// Match a sadd_sat or ssub_sat which is using min/max to clamp the value.
1299Instruction *InstCombinerImpl::matchSAddSubSat(IntrinsicInst &MinMax1) {
1300 Type *Ty = MinMax1.getType();
1301
1302 // We are looking for a tree of:
1303 // max(INT_MIN, min(INT_MAX, add(sext(A), sext(B))))
1304 // Where the min and max could be reversed
1305 Instruction *MinMax2;
1306 BinaryOperator *AddSub;
1307 const APInt *MinValue, *MaxValue;
1308 if (match(&MinMax1, m_SMin(m_Instruction(MinMax2), m_APInt(MaxValue)))) {
1309 if (!match(MinMax2, m_SMax(m_BinOp(AddSub), m_APInt(MinValue))))
1310 return nullptr;
1311 } else if (match(&MinMax1,
1312 m_SMax(m_Instruction(MinMax2), m_APInt(MinValue)))) {
1313 if (!match(MinMax2, m_SMin(m_BinOp(AddSub), m_APInt(MaxValue))))
1314 return nullptr;
1315 } else
1316 return nullptr;
1317
1318 // Check that the constants clamp a saturate, and that the new type would be
1319 // sensible to convert to.
1320 if (!(*MaxValue + 1).isPowerOf2() || -*MinValue != *MaxValue + 1)
1321 return nullptr;
1322 // In what bitwidth can this be treated as saturating arithmetics?
1323 unsigned NewBitWidth = (*MaxValue + 1).logBase2() + 1;
1324 // FIXME: This isn't quite right for vectors, but using the scalar type is a
1325 // good first approximation for what should be done there.
1326 if (!shouldChangeType(Ty->getScalarType()->getIntegerBitWidth(), NewBitWidth))
1327 return nullptr;
1328
1329 // Also make sure that the inner min/max and the add/sub have one use.
1330 if (!MinMax2->hasOneUse() || !AddSub->hasOneUse())
1331 return nullptr;
1332
1333 // Create the new type (which can be a vector type)
1334 Type *NewTy = Ty->getWithNewBitWidth(NewBitWidth);
1335
1336 Intrinsic::ID IntrinsicID;
1337 if (AddSub->getOpcode() == Instruction::Add)
1338 IntrinsicID = Intrinsic::sadd_sat;
1339 else if (AddSub->getOpcode() == Instruction::Sub)
1340 IntrinsicID = Intrinsic::ssub_sat;
1341 else
1342 return nullptr;
1343
1344 // The two operands of the add/sub must be nsw-truncatable to the NewTy. This
1345 // is usually achieved via a sext from a smaller type.
1346 if (ComputeMaxSignificantBits(AddSub->getOperand(0), AddSub) > NewBitWidth ||
1347 ComputeMaxSignificantBits(AddSub->getOperand(1), AddSub) > NewBitWidth)
1348 return nullptr;
1349
1350 // Finally create and return the sat intrinsic, truncated to the new type
1351 Value *AT = Builder.CreateTrunc(AddSub->getOperand(0), NewTy);
1352 Value *BT = Builder.CreateTrunc(AddSub->getOperand(1), NewTy);
1353 Value *Sat = Builder.CreateIntrinsic(IntrinsicID, NewTy, {AT, BT});
1354 return CastInst::Create(Instruction::SExt, Sat, Ty);
1355}
1356
1357
1358/// If we have a clamp pattern like max (min X, 42), 41 -- where the output
1359/// can only be one of two possible constant values -- turn that into a select
1360/// of constants.
1362 InstCombiner::BuilderTy &Builder) {
1363 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
1364 Value *X;
1365 const APInt *C0, *C1;
1366 if (!match(I1, m_APInt(C1)) || !I0->hasOneUse())
1367 return nullptr;
1368
1370 switch (II->getIntrinsicID()) {
1371 case Intrinsic::smax:
1372 if (match(I0, m_SMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1373 Pred = ICmpInst::ICMP_SGT;
1374 break;
1375 case Intrinsic::smin:
1376 if (match(I0, m_SMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1377 Pred = ICmpInst::ICMP_SLT;
1378 break;
1379 case Intrinsic::umax:
1380 if (match(I0, m_UMin(m_Value(X), m_APInt(C0))) && *C0 == *C1 + 1)
1381 Pred = ICmpInst::ICMP_UGT;
1382 break;
1383 case Intrinsic::umin:
1384 if (match(I0, m_UMax(m_Value(X), m_APInt(C0))) && *C1 == *C0 + 1)
1385 Pred = ICmpInst::ICMP_ULT;
1386 break;
1387 default:
1388 llvm_unreachable("Expected min/max intrinsic");
1389 }
1390 if (Pred == CmpInst::BAD_ICMP_PREDICATE)
1391 return nullptr;
1392
1393 // max (min X, 42), 41 --> X > 41 ? 42 : 41
1394 // min (max X, 42), 43 --> X < 43 ? 42 : 43
1395 Value *Cmp = Builder.CreateICmp(Pred, X, I1);
1396 return SelectInst::Create(Cmp, ConstantInt::get(II->getType(), *C0), I1);
1397}
1398
1399/// If this min/max has a constant operand and an operand that is a matching
1400/// min/max with a constant operand, constant-fold the 2 constant operands.
1402 IRBuilderBase &Builder,
1403 const SimplifyQuery &SQ) {
1404 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1405 auto *LHS = dyn_cast<MinMaxIntrinsic>(II->getArgOperand(0));
1406 if (!LHS)
1407 return nullptr;
1408
1409 Constant *C0, *C1;
1410 if (!match(LHS->getArgOperand(1), m_ImmConstant(C0)) ||
1411 !match(II->getArgOperand(1), m_ImmConstant(C1)))
1412 return nullptr;
1413
1414 // max (max X, C0), C1 --> max X, (max C0, C1)
1415 // min (min X, C0), C1 --> min X, (min C0, C1)
1416 // umax (smax X, nneg C0), nneg C1 --> smax X, (umax C0, C1)
1417 // smin (umin X, nneg C0), nneg C1 --> umin X, (smin C0, C1)
1418 Intrinsic::ID InnerMinMaxID = LHS->getIntrinsicID();
1419 if (InnerMinMaxID != MinMaxID &&
1420 !(((MinMaxID == Intrinsic::umax && InnerMinMaxID == Intrinsic::smax) ||
1421 (MinMaxID == Intrinsic::smin && InnerMinMaxID == Intrinsic::umin)) &&
1422 isKnownNonNegative(C0, SQ) && isKnownNonNegative(C1, SQ)))
1423 return nullptr;
1424
1426 Value *CondC = Builder.CreateICmp(Pred, C0, C1);
1427 Value *NewC = Builder.CreateSelect(CondC, C0, C1);
1428 return Builder.CreateIntrinsic(InnerMinMaxID, II->getType(),
1429 {LHS->getArgOperand(0), NewC});
1430}
1431
1432/// If this min/max has a matching min/max operand with a constant, try to push
1433/// the constant operand into this instruction. This can enable more folds.
1434static Instruction *
1436 InstCombiner::BuilderTy &Builder) {
1437 // Match and capture a min/max operand candidate.
1438 Value *X, *Y;
1439 Constant *C;
1440 Instruction *Inner;
1442 m_Instruction(Inner),
1444 m_Value(Y))))
1445 return nullptr;
1446
1447 // The inner op must match. Check for constants to avoid infinite loops.
1448 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1449 auto *InnerMM = dyn_cast<IntrinsicInst>(Inner);
1450 if (!InnerMM || InnerMM->getIntrinsicID() != MinMaxID ||
1452 return nullptr;
1453
1454 // max (max X, C), Y --> max (max X, Y), C
1456 MinMaxID, II->getType());
1457 Value *NewInner = Builder.CreateBinaryIntrinsic(MinMaxID, X, Y);
1458 NewInner->takeName(Inner);
1459 return CallInst::Create(MinMax, {NewInner, C});
1460}
1461
1462/// Reduce a sequence of min/max intrinsics with a common operand.
1464 // Match 3 of the same min/max ops. Example: umin(umin(), umin()).
1465 auto *LHS = dyn_cast<IntrinsicInst>(II->getArgOperand(0));
1466 auto *RHS = dyn_cast<IntrinsicInst>(II->getArgOperand(1));
1467 Intrinsic::ID MinMaxID = II->getIntrinsicID();
1468 if (!LHS || !RHS || LHS->getIntrinsicID() != MinMaxID ||
1469 RHS->getIntrinsicID() != MinMaxID ||
1470 (!LHS->hasOneUse() && !RHS->hasOneUse()))
1471 return nullptr;
1472
1473 Value *A = LHS->getArgOperand(0);
1474 Value *B = LHS->getArgOperand(1);
1475 Value *C = RHS->getArgOperand(0);
1476 Value *D = RHS->getArgOperand(1);
1477
1478 // Look for a common operand.
1479 Value *MinMaxOp = nullptr;
1480 Value *ThirdOp = nullptr;
1481 if (LHS->hasOneUse()) {
1482 // If the LHS is only used in this chain and the RHS is used outside of it,
1483 // reuse the RHS min/max because that will eliminate the LHS.
1484 if (D == A || C == A) {
1485 // min(min(a, b), min(c, a)) --> min(min(c, a), b)
1486 // min(min(a, b), min(a, d)) --> min(min(a, d), b)
1487 MinMaxOp = RHS;
1488 ThirdOp = B;
1489 } else if (D == B || C == B) {
1490 // min(min(a, b), min(c, b)) --> min(min(c, b), a)
1491 // min(min(a, b), min(b, d)) --> min(min(b, d), a)
1492 MinMaxOp = RHS;
1493 ThirdOp = A;
1494 }
1495 } else {
1496 assert(RHS->hasOneUse() && "Expected one-use operand");
1497 // Reuse the LHS. This will eliminate the RHS.
1498 if (D == A || D == B) {
1499 // min(min(a, b), min(c, a)) --> min(min(a, b), c)
1500 // min(min(a, b), min(c, b)) --> min(min(a, b), c)
1501 MinMaxOp = LHS;
1502 ThirdOp = C;
1503 } else if (C == A || C == B) {
1504 // min(min(a, b), min(b, d)) --> min(min(a, b), d)
1505 // min(min(a, b), min(c, b)) --> min(min(a, b), d)
1506 MinMaxOp = LHS;
1507 ThirdOp = D;
1508 }
1509 }
1510
1511 if (!MinMaxOp || !ThirdOp)
1512 return nullptr;
1513
1514 Module *Mod = II->getModule();
1515 Function *MinMax =
1516 Intrinsic::getOrInsertDeclaration(Mod, MinMaxID, II->getType());
1517 return CallInst::Create(MinMax, { MinMaxOp, ThirdOp });
1518}
1519
1520/// If all arguments of the intrinsic are unary shuffles with the same mask,
1521/// try to shuffle after the intrinsic.
1524 if (!II->getType()->isVectorTy() ||
1525 !isTriviallyVectorizable(II->getIntrinsicID()) ||
1526 !II->getCalledFunction()->isSpeculatable())
1527 return nullptr;
1528
1529 Value *X;
1530 Constant *C;
1531 ArrayRef<int> Mask;
1532 auto *NonConstArg = find_if_not(II->args(), [&II](Use &Arg) {
1533 return isa<Constant>(Arg.get()) ||
1534 isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1535 Arg.getOperandNo(), nullptr);
1536 });
1537 if (!NonConstArg ||
1538 !match(NonConstArg, m_Shuffle(m_Value(X), m_Poison(), m_Mask(Mask))))
1539 return nullptr;
1540
1541 // At least 1 operand must be a shuffle with 1 use because we are creating 2
1542 // instructions.
1543 if (none_of(II->args(), match_fn(m_OneUse(m_Shuffle(m_Value(), m_Value())))))
1544 return nullptr;
1545
1546 // See if all arguments are shuffled with the same mask.
1548 Type *SrcTy = X->getType();
1549 for (Use &Arg : II->args()) {
1550 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1551 Arg.getOperandNo(), nullptr))
1552 NewArgs.push_back(Arg);
1553 else if (match(&Arg,
1554 m_Shuffle(m_Value(X), m_Poison(), m_SpecificMask(Mask))) &&
1555 X->getType() == SrcTy)
1556 NewArgs.push_back(X);
1557 else if (match(&Arg, m_ImmConstant(C))) {
1558 // If it's a constant, try find the constant that would be shuffled to C.
1559 if (Constant *ShuffledC =
1560 unshuffleConstant(Mask, C, cast<VectorType>(SrcTy)))
1561 NewArgs.push_back(ShuffledC);
1562 else
1563 return nullptr;
1564 } else
1565 return nullptr;
1566 }
1567
1568 // intrinsic (shuf X, M), (shuf Y, M), ... --> shuf (intrinsic X, Y, ...), M
1569 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1570 // Result type might be a different vector width.
1571 // TODO: Check that the result type isn't widened?
1572 VectorType *ResTy =
1573 VectorType::get(II->getType()->getScalarType(), cast<VectorType>(SrcTy));
1574 Value *NewIntrinsic =
1575 Builder.CreateIntrinsic(ResTy, II->getIntrinsicID(), NewArgs, FPI);
1576 return new ShuffleVectorInst(NewIntrinsic, Mask);
1577}
1578
1579/// If all arguments of the intrinsic are reverses, try to pull the reverse
1580/// after the intrinsic.
1582 if (!II->getType()->isVectorTy() ||
1583 !isTriviallyVectorizable(II->getIntrinsicID()))
1584 return nullptr;
1585
1586 // At least 1 operand must be a reverse with 1 use because we are creating 2
1587 // instructions.
1588 if (none_of(II->args(), [](Value *V) {
1589 return match(V, m_OneUse(m_VecReverse(m_Value())));
1590 }))
1591 return nullptr;
1592
1593 Value *X;
1594 Constant *C;
1595 SmallVector<Value *> NewArgs;
1596 for (Use &Arg : II->args()) {
1597 if (isVectorIntrinsicWithScalarOpAtArg(II->getIntrinsicID(),
1598 Arg.getOperandNo(), nullptr))
1599 NewArgs.push_back(Arg);
1600 else if (match(&Arg, m_VecReverse(m_Value(X))))
1601 NewArgs.push_back(X);
1602 else if (isSplatValue(Arg))
1603 NewArgs.push_back(Arg);
1604 else if (match(&Arg, m_ImmConstant(C)))
1605 NewArgs.push_back(Builder.CreateVectorReverse(C));
1606 else
1607 return nullptr;
1608 }
1609
1610 // intrinsic (reverse X), (reverse Y), ... --> reverse (intrinsic X, Y, ...)
1611 Instruction *FPI = isa<FPMathOperator>(II) ? II : nullptr;
1612 Value *NewIntrinsic = Builder.CreateIntrinsic(
1613 II->getType(), II->getIntrinsicID(), NewArgs, FPI);
1614 return Builder.CreateVectorReverse(NewIntrinsic);
1615}
1616
1617/// Fold the following cases and accepts bswap and bitreverse intrinsics:
1618/// bswap(logic_op(bswap(x), y)) --> logic_op(x, bswap(y))
1619/// bswap(logic_op(bswap(x), bswap(y))) --> logic_op(x, y) (ignores multiuse)
1620template <Intrinsic::ID IntrID>
1622 InstCombiner::BuilderTy &Builder) {
1623 static_assert(IntrID == Intrinsic::bswap || IntrID == Intrinsic::bitreverse,
1624 "This helper only supports BSWAP and BITREVERSE intrinsics");
1625
1626 Value *X, *Y;
1627 // Find bitwise logic op. Check that it is a BinaryOperator explicitly so we
1628 // don't match ConstantExpr that aren't meaningful for this transform.
1631 Value *OldReorderX, *OldReorderY;
1633
1634 // If both X and Y are bswap/bitreverse, the transform reduces the number
1635 // of instructions even if there's multiuse.
1636 // If only one operand is bswap/bitreverse, we need to ensure the operand
1637 // have only one use.
1638 if (match(X, m_Intrinsic<IntrID>(m_Value(OldReorderX))) &&
1639 match(Y, m_Intrinsic<IntrID>(m_Value(OldReorderY)))) {
1640 return BinaryOperator::Create(Op, OldReorderX, OldReorderY);
1641 }
1642
1643 if (match(X, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderX))))) {
1644 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, Y);
1645 return BinaryOperator::Create(Op, OldReorderX, NewReorder);
1646 }
1647
1648 if (match(Y, m_OneUse(m_Intrinsic<IntrID>(m_Value(OldReorderY))))) {
1649 Value *NewReorder = Builder.CreateUnaryIntrinsic(IntrID, X);
1650 return BinaryOperator::Create(Op, NewReorder, OldReorderY);
1651 }
1652 }
1653 return nullptr;
1654}
1655
1656/// Helper to match idempotent binary intrinsics, namely, intrinsics where
1657/// `f(f(x, y), y) == f(x, y)` holds.
1659 switch (IID) {
1660 case Intrinsic::smax:
1661 case Intrinsic::smin:
1662 case Intrinsic::umax:
1663 case Intrinsic::umin:
1664 case Intrinsic::maximum:
1665 case Intrinsic::minimum:
1666 case Intrinsic::maximumnum:
1667 case Intrinsic::minimumnum:
1668 case Intrinsic::maxnum:
1669 case Intrinsic::minnum:
1670 return true;
1671 default:
1672 return false;
1673 }
1674}
1675
1676/// Attempt to simplify value-accumulating recurrences of kind:
1677/// %umax.acc = phi i8 [ %umax, %backedge ], [ %a, %entry ]
1678/// %umax = call i8 @llvm.umax.i8(i8 %umax.acc, i8 %b)
1679/// And let the idempotent binary intrinsic be hoisted, when the operands are
1680/// known to be loop-invariant.
1682 IntrinsicInst *II) {
1683 PHINode *PN;
1684 Value *Init, *OtherOp;
1685
1686 // A binary intrinsic recurrence with loop-invariant operands is equivalent to
1687 // `call @llvm.binary.intrinsic(Init, OtherOp)`.
1688 auto IID = II->getIntrinsicID();
1689 if (!isIdempotentBinaryIntrinsic(IID) ||
1691 !IC.getDominatorTree().dominates(OtherOp, PN))
1692 return nullptr;
1693
1694 auto *InvariantBinaryInst =
1695 IC.Builder.CreateBinaryIntrinsic(IID, Init, OtherOp);
1696 if (isa<FPMathOperator>(InvariantBinaryInst))
1697 cast<Instruction>(InvariantBinaryInst)->copyFastMathFlags(II);
1698 return InvariantBinaryInst;
1699}
1700
1701static Value *simplifyReductionOperand(Value *Arg, bool CanReorderLanes) {
1702 if (!CanReorderLanes)
1703 return nullptr;
1704
1705 Value *V;
1706 if (match(Arg, m_VecReverse(m_Value(V))))
1707 return V;
1708
1709 ArrayRef<int> Mask;
1710 if (!isa<FixedVectorType>(Arg->getType()) ||
1711 !match(Arg, m_Shuffle(m_Value(V), m_Undef(), m_Mask(Mask))) ||
1712 !cast<ShuffleVectorInst>(Arg)->isSingleSource())
1713 return nullptr;
1714
1715 int Sz = Mask.size();
1716 SmallBitVector UsedIndices(Sz);
1717 for (int Idx : Mask) {
1718 if (Idx == PoisonMaskElem || UsedIndices.test(Idx))
1719 return nullptr;
1720 UsedIndices.set(Idx);
1721 }
1722
1723 // Can remove shuffle iff just shuffled elements, no repeats, undefs, or
1724 // other changes.
1725 return UsedIndices.all() ? V : nullptr;
1726}
1727
1728/// Fold an unsigned minimum of trailing or leading zero bits counts:
1729/// umin(cttz(CtOp1, ZeroUndef), ConstOp) --> cttz(CtOp1 | (1 << ConstOp))
1730/// umin(ctlz(CtOp1, ZeroUndef), ConstOp) --> ctlz(CtOp1 | (SignedMin
1731/// >> ConstOp))
1732/// umin(cttz(CtOp1), cttz(CtOp2)) --> cttz(CtOp1 | CtOp2)
1733/// umin(ctlz(CtOp1), ctlz(CtOp2)) --> ctlz(CtOp1 | CtOp2)
1734template <Intrinsic::ID IntrID>
1735static Value *
1737 const DataLayout &DL,
1738 InstCombiner::BuilderTy &Builder) {
1739 static_assert(IntrID == Intrinsic::cttz || IntrID == Intrinsic::ctlz,
1740 "This helper only supports cttz and ctlz intrinsics");
1741
1742 Value *CtOp1, *CtOp2;
1743 Value *ZeroUndef1, *ZeroUndef2;
1744 if (!match(I0, m_OneUse(
1745 m_Intrinsic<IntrID>(m_Value(CtOp1), m_Value(ZeroUndef1)))))
1746 return nullptr;
1747
1748 if (match(I1,
1749 m_OneUse(m_Intrinsic<IntrID>(m_Value(CtOp2), m_Value(ZeroUndef2)))))
1750 return Builder.CreateBinaryIntrinsic(
1751 IntrID, Builder.CreateOr(CtOp1, CtOp2),
1752 Builder.CreateOr(ZeroUndef1, ZeroUndef2));
1753
1754 unsigned BitWidth = I1->getType()->getScalarSizeInBits();
1755 auto LessBitWidth = [BitWidth](auto &C) { return C.ult(BitWidth); };
1756 if (!match(I1, m_CheckedInt(LessBitWidth)))
1757 // We have a constant >= BitWidth (which can be handled by CVP)
1758 // or a non-splat vector with elements < and >= BitWidth
1759 return nullptr;
1760
1761 Type *Ty = I1->getType();
1763 IntrID == Intrinsic::cttz ? Instruction::Shl : Instruction::LShr,
1764 IntrID == Intrinsic::cttz
1765 ? ConstantInt::get(Ty, 1)
1766 : ConstantInt::get(Ty, APInt::getSignedMinValue(BitWidth)),
1767 cast<Constant>(I1), DL);
1768 return Builder.CreateBinaryIntrinsic(
1769 IntrID, Builder.CreateOr(CtOp1, NewConst),
1770 ConstantInt::getTrue(ZeroUndef1->getType()));
1771}
1772
1773/// Return whether "X LOp (Y ROp Z)" is always equal to
1774/// "(X LOp Y) ROp (X LOp Z)".
1776 bool HasNSW, Intrinsic::ID ROp) {
1777 switch (ROp) {
1778 case Intrinsic::umax:
1779 case Intrinsic::umin:
1780 if (HasNUW && LOp == Instruction::Add)
1781 return true;
1782 if (HasNUW && LOp == Instruction::Shl)
1783 return true;
1784 return false;
1785 case Intrinsic::smax:
1786 case Intrinsic::smin:
1787 return HasNSW && LOp == Instruction::Add;
1788 default:
1789 return false;
1790 }
1791}
1792
1793/// Return whether "(X ROp Y) LOp Z" is always equal to
1794/// "(X LOp Z) ROp (Y LOp Z)".
1796 bool HasNSW, Intrinsic::ID ROp) {
1797 if (Instruction::isCommutative(LOp) || LOp == Instruction::Shl)
1798 return leftDistributesOverRight(LOp, HasNUW, HasNSW, ROp);
1799 switch (ROp) {
1800 case Intrinsic::umax:
1801 case Intrinsic::umin:
1802 return HasNUW && LOp == Instruction::Sub;
1803 case Intrinsic::smax:
1804 case Intrinsic::smin:
1805 return HasNSW && LOp == Instruction::Sub;
1806 default:
1807 return false;
1808 }
1809}
1810
1811// Attempts to factorise a common term
1812// in an instruction that has the form "(A op' B) op (C op' D)
1813// where op is an intrinsic and op' is a binop
1814static Value *
1816 InstCombiner::BuilderTy &Builder) {
1817 Value *LHS = II->getOperand(0), *RHS = II->getOperand(1);
1818 Intrinsic::ID TopLevelOpcode = II->getIntrinsicID();
1819
1822
1823 if (!Op0 || !Op1)
1824 return nullptr;
1825
1826 if (Op0->getOpcode() != Op1->getOpcode())
1827 return nullptr;
1828
1829 if (!Op0->hasOneUse() || !Op1->hasOneUse())
1830 return nullptr;
1831
1832 Instruction::BinaryOps InnerOpcode =
1833 static_cast<Instruction::BinaryOps>(Op0->getOpcode());
1834 bool HasNUW = Op0->hasNoUnsignedWrap() && Op1->hasNoUnsignedWrap();
1835 bool HasNSW = Op0->hasNoSignedWrap() && Op1->hasNoSignedWrap();
1836
1837 Value *A = Op0->getOperand(0);
1838 Value *B = Op0->getOperand(1);
1839 Value *C = Op1->getOperand(0);
1840 Value *D = Op1->getOperand(1);
1841
1842 // Attempts to swap variables such that A equals C or B equals D,
1843 // if the inner operation is commutative.
1844 if (Op0->isCommutative() && A != C && B != D) {
1845 if (A == D || B == C)
1846 std::swap(C, D);
1847 else
1848 return nullptr;
1849 }
1850
1851 BinaryOperator *NewBinop;
1852 if (A == C &&
1853 leftDistributesOverRight(InnerOpcode, HasNUW, HasNSW, TopLevelOpcode)) {
1854 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, B, D);
1855 NewBinop =
1856 cast<BinaryOperator>(Builder.CreateBinOp(InnerOpcode, A, NewIntrinsic));
1857 } else if (B == D && rightDistributesOverLeft(InnerOpcode, HasNUW, HasNSW,
1858 TopLevelOpcode)) {
1859 Value *NewIntrinsic = Builder.CreateBinaryIntrinsic(TopLevelOpcode, A, C);
1860 NewBinop =
1861 cast<BinaryOperator>(Builder.CreateBinOp(InnerOpcode, NewIntrinsic, B));
1862 } else {
1863 return nullptr;
1864 }
1865
1866 NewBinop->setHasNoUnsignedWrap(HasNUW);
1867 NewBinop->setHasNoSignedWrap(HasNSW);
1868
1869 return NewBinop;
1870}
1871
1873 Value *Arg0 = II->getArgOperand(0);
1874 auto *ShiftConst = dyn_cast<Constant>(II->getArgOperand(1));
1875 if (!ShiftConst)
1876 return nullptr;
1877
1878 int ElemBits = Arg0->getType()->getScalarSizeInBits();
1879 bool AllPositive = true;
1880 bool AllNegative = true;
1881
1882 auto Check = [&](Constant *C) -> bool {
1883 if (auto *CI = dyn_cast_or_null<ConstantInt>(C)) {
1884 const APInt &V = CI->getValue();
1885 if (V.isNonNegative()) {
1886 AllNegative = false;
1887 return AllPositive && V.ult(ElemBits);
1888 }
1889 AllPositive = false;
1890 return AllNegative && V.sgt(-ElemBits);
1891 }
1892 return false;
1893 };
1894
1895 if (auto *VTy = dyn_cast<FixedVectorType>(Arg0->getType())) {
1896 for (unsigned I = 0, E = VTy->getNumElements(); I < E; ++I) {
1897 if (!Check(ShiftConst->getAggregateElement(I)))
1898 return nullptr;
1899 }
1900
1901 } else if (!Check(ShiftConst))
1902 return nullptr;
1903
1904 IRBuilderBase &B = IC.Builder;
1905 if (AllPositive)
1906 return IC.replaceInstUsesWith(*II, B.CreateShl(Arg0, ShiftConst));
1907
1908 Value *NegAmt = B.CreateNeg(ShiftConst);
1909 Intrinsic::ID IID = II->getIntrinsicID();
1910 const bool IsSigned =
1911 IID == Intrinsic::arm_neon_vshifts || IID == Intrinsic::aarch64_neon_sshl;
1912 Value *Result =
1913 IsSigned ? B.CreateAShr(Arg0, NegAmt) : B.CreateLShr(Arg0, NegAmt);
1914 return IC.replaceInstUsesWith(*II, Result);
1915}
1916
1917// If II is llvm.sin(x) or llvm.cos(x), and there is a matching
1918// llvm.cos(x) or llvm.sin(x) using the same argument, combine them
1919// into a single llvm.sincos(x) call. Returns the result for II
1920// extracted from sincos, or nullptr if no match is found.
1922 InstCombinerImpl &IC) {
1923 Intrinsic::ID IID = II->getIntrinsicID();
1924 bool IsSin = IID == Intrinsic::sin;
1925 Intrinsic::ID MatchID = IsSin ? Intrinsic::cos : Intrinsic::sin;
1926
1927 Value *Arg = II->getArgOperand(0);
1928
1929 // Don't bother looking through uses of constants.
1930 if (isa<Constant>(Arg))
1931 return nullptr;
1932
1933 // Look for a matching cos/sin intrinsic with the same argument.
1934 IntrinsicInst *Match = nullptr;
1935 for (User *U : Arg->users()) {
1936 if (auto *Cand = dyn_cast<IntrinsicInst>(U)) {
1937 if (Cand != II && !Cand->use_empty() &&
1938 Cand->getIntrinsicID() == MatchID) {
1939 Match = Cand;
1940 break;
1941 }
1942 }
1943 }
1944
1945 if (!Match)
1946 return nullptr;
1947
1948 // Insert sincos right after the argument definition.
1950 if (auto *ArgInst = dyn_cast<Instruction>(Arg)) {
1951 std::optional<BasicBlock::iterator> InsertPt =
1952 ArgInst->getInsertionPointAfterDef();
1953 if (!InsertPt)
1954 return nullptr;
1955 B.SetInsertPoint(*InsertPt);
1956 } else {
1957 BasicBlock &EntryBB = II->getFunction()->getEntryBlock();
1958 B.SetInsertPoint(&EntryBB, EntryBB.begin());
1959 }
1960
1962 II->getModule(), Intrinsic::sincos, Arg->getType());
1963 CallInst *SinCos = B.CreateCall(SinCosFunc, Arg, "sincos");
1964 // Intersect fast-math flags from the two calls.
1965 SinCos->setFastMathFlags(II->getFastMathFlags() & Match->getFastMathFlags());
1966 // Propagate the most-generic fpmath metadata from the two original calls.
1968 II->getMetadata(LLVMContext::MD_fpmath),
1969 Match->getMetadata(LLVMContext::MD_fpmath)))
1970 SinCos->setMetadata(LLVMContext::MD_fpmath, MD);
1971 Value *Sin = B.CreateExtractValue(SinCos, 0, "sin");
1972 Value *Cos = B.CreateExtractValue(SinCos, 1, "cos");
1973
1974 // Replace the matching call and erase it.
1975 IC.replaceInstUsesWith(*Match, IsSin ? Cos : Sin);
1976 IC.eraseInstFromFunction(*Match);
1977 return IsSin ? Sin : Cos;
1978}
1979
1980/// CallInst simplification. This mostly only handles folding of intrinsic
1981/// instructions. For normal calls, it allows visitCallBase to do the heavy
1982/// lifting.
1984 // Don't try to simplify calls without uses. It will not do anything useful,
1985 // but will result in the following folds being skipped.
1986 if (!CI.use_empty()) {
1987 SmallVector<Value *, 8> Args(CI.args());
1988 if (Value *V = simplifyCall(&CI, CI.getCalledOperand(), Args,
1989 SQ.getWithInstruction(&CI)))
1990 return replaceInstUsesWith(CI, V);
1991 }
1992
1993 if (Value *FreedOp = getFreedOperand(&CI, &TLI))
1994 return visitFree(CI, FreedOp);
1995
1996 // If the caller function (i.e. us, the function that contains this CallInst)
1997 // is nounwind, mark the call as nounwind, even if the callee isn't.
1998 if (CI.getFunction()->doesNotThrow() && !CI.doesNotThrow()) {
1999 CI.setDoesNotThrow();
2000 return &CI;
2001 }
2002
2004 if (!II)
2005 return visitCallBase(CI);
2006
2007 // Intrinsics cannot occur in an invoke or a callbr, so handle them here
2008 // instead of in visitCallBase.
2009 if (auto *MI = dyn_cast<AnyMemIntrinsic>(II)) {
2010 if (auto NumBytes = MI->getLengthInBytes()) {
2011 // memmove/cpy/set of zero bytes is a noop.
2012 if (NumBytes->isZero())
2013 return eraseInstFromFunction(CI);
2014
2015 // For atomic unordered mem intrinsics if len is not a positive or
2016 // not a multiple of element size then behavior is undefined.
2017 if (MI->isAtomic() &&
2018 (NumBytes->isNegative() ||
2019 (NumBytes->getZExtValue() % MI->getElementSizeInBytes() != 0))) {
2021 assert(MI->getType()->isVoidTy() &&
2022 "non void atomic unordered mem intrinsic");
2023 return eraseInstFromFunction(*MI);
2024 }
2025 }
2026
2027 // No other transformations apply to volatile transfers.
2028 if (MI->isVolatile())
2029 return nullptr;
2030
2032 // memmove(x,x,size) -> noop.
2033 if (MTI->getSource() == MTI->getDest())
2034 return eraseInstFromFunction(CI);
2035 }
2036
2037 auto IsPointerUndefined = [MI](Value *Ptr) {
2038 return isa<ConstantPointerNull>(Ptr) &&
2040 MI->getFunction(),
2041 cast<PointerType>(Ptr->getType())->getAddressSpace());
2042 };
2043 bool SrcIsUndefined = false;
2044 // If we can determine a pointer alignment that is bigger than currently
2045 // set, update the alignment.
2046 if (auto *MTI = dyn_cast<AnyMemTransferInst>(MI)) {
2048 return I;
2049 SrcIsUndefined = IsPointerUndefined(MTI->getRawSource());
2050 } else if (auto *MSI = dyn_cast<AnyMemSetInst>(MI)) {
2051 if (Instruction *I = SimplifyAnyMemSet(MSI))
2052 return I;
2053 }
2054
2055 // If src/dest is null, this memory intrinsic must be a noop.
2056 if (SrcIsUndefined || IsPointerUndefined(MI->getRawDest())) {
2057 Builder.CreateAssumption(Builder.CreateIsNull(MI->getLength()));
2058 return eraseInstFromFunction(CI);
2059 }
2060
2061 // If we have a memmove and the source operation is a constant global,
2062 // then the source and dest pointers can't alias, so we can change this
2063 // into a call to memcpy.
2064 if (auto *MMI = dyn_cast<AnyMemMoveInst>(MI)) {
2065 if (GlobalVariable *GVSrc = dyn_cast<GlobalVariable>(MMI->getSource()))
2066 if (GVSrc->isConstant()) {
2067 Module *M = CI.getModule();
2068 Intrinsic::ID MemCpyID =
2069 MMI->isAtomic()
2070 ? Intrinsic::memcpy_element_unordered_atomic
2071 : Intrinsic::memcpy;
2072 Type *Tys[3] = { CI.getArgOperand(0)->getType(),
2073 CI.getArgOperand(1)->getType(),
2074 CI.getArgOperand(2)->getType() };
2076 Intrinsic::getOrInsertDeclaration(M, MemCpyID, Tys));
2077 return II;
2078 }
2079 }
2080 }
2081
2082 // For fixed width vector result intrinsics, use the generic demanded vector
2083 // support.
2084 if (auto *IIFVTy = dyn_cast<FixedVectorType>(II->getType())) {
2085 auto VWidth = IIFVTy->getNumElements();
2086 APInt PoisonElts(VWidth, 0);
2087 APInt AllOnesEltMask(APInt::getAllOnes(VWidth));
2088 if (Value *V = SimplifyDemandedVectorElts(II, AllOnesEltMask, PoisonElts)) {
2089 if (V != II)
2090 return replaceInstUsesWith(*II, V);
2091 return II;
2092 }
2093 }
2094
2095 if (II->isCommutative()) {
2096 if (auto Pair = matchSymmetricPair(II->getOperand(0), II->getOperand(1))) {
2097 replaceOperand(*II, 0, Pair->first);
2098 replaceOperand(*II, 1, Pair->second);
2099 II->dropPoisonGeneratingAnnotations();
2100 II->dropUBImplyingAttrsAndMetadata();
2101 return II;
2102 }
2103
2104 if (CallInst *NewCall = canonicalizeConstantArg0ToArg1(CI))
2105 return NewCall;
2106 }
2107
2108 // Unused constrained FP intrinsic calls may have declared side effect, which
2109 // prevents it from being removed. In some cases however the side effect is
2110 // actually absent. To detect this case, call SimplifyConstrainedFPCall. If it
2111 // returns a replacement, the call may be removed.
2112 if (CI.use_empty() && isa<ConstrainedFPIntrinsic>(CI)) {
2113 if (simplifyConstrainedFPCall(&CI, SQ.getWithInstruction(&CI)))
2114 return eraseInstFromFunction(CI);
2115 }
2116
2117 Intrinsic::ID IID = II->getIntrinsicID();
2118 switch (IID) {
2119 case Intrinsic::objectsize: {
2120 SmallVector<Instruction *> InsertedInstructions;
2121 if (Value *V = lowerObjectSizeCall(II, DL, &TLI, AA, /*MustSucceed=*/false,
2122 &InsertedInstructions)) {
2123 for (Instruction *Inserted : InsertedInstructions)
2124 Worklist.add(Inserted);
2125 return replaceInstUsesWith(CI, V);
2126 }
2127 return nullptr;
2128 }
2129 case Intrinsic::abs: {
2130 Value *IIOperand = II->getArgOperand(0);
2131 bool IntMinIsPoison = cast<Constant>(II->getArgOperand(1))->isOneValue();
2132
2133 // abs(-x) -> abs(x)
2134 Value *X;
2135 if (match(IIOperand, m_Neg(m_Value(X))))
2136 return CallInst::Create(
2137 II->getCalledFunction(),
2138 {X,
2139 Builder.getInt1(IntMinIsPoison ||
2140 cast<Instruction>(IIOperand)->hasNoSignedWrap())});
2141
2142 if (match(IIOperand, m_c_Select(m_Neg(m_Value(X)), m_Deferred(X))))
2143 return CallInst::Create(II->getCalledFunction(),
2144 {X, II->getArgOperand(1)});
2145
2146 Value *Y;
2147 // abs(a * abs(b)) -> abs(a * b)
2148 if (match(IIOperand,
2151 bool NSW =
2152 cast<Instruction>(IIOperand)->hasNoSignedWrap() && IntMinIsPoison;
2153 auto *XY = NSW ? Builder.CreateNSWMul(X, Y) : Builder.CreateMul(X, Y);
2154 return CallInst::Create(II->getCalledFunction(),
2155 {XY, II->getArgOperand(1)});
2156 }
2157
2158 if (std::optional<bool> Known =
2159 getKnownSignOrZero(IIOperand, SQ.getWithInstruction(II))) {
2160 // abs(x) -> x if x >= 0 (include abs(x-y) --> x - y where x >= y)
2161 // abs(x) -> x if x > 0 (include abs(x-y) --> x - y where x > y)
2162 if (!*Known)
2163 return replaceInstUsesWith(*II, IIOperand);
2164
2165 // abs(x) -> -x if x < 0
2166 // abs(x) -> -x if x < = 0 (include abs(x-y) --> y - x where x <= y)
2167 if (IntMinIsPoison)
2168 return BinaryOperator::CreateNSWNeg(IIOperand);
2169 return BinaryOperator::CreateNeg(IIOperand);
2170 }
2171
2172 // abs (sext X) --> zext (abs X*)
2173 // Clear the IsIntMin (nsw) bit on the abs to allow narrowing.
2174 if (match(IIOperand, m_OneUse(m_SExt(m_Value(X))))) {
2175 Value *NarrowAbs =
2176 Builder.CreateBinaryIntrinsic(Intrinsic::abs, X, Builder.getFalse());
2177 return CastInst::Create(Instruction::ZExt, NarrowAbs, II->getType());
2178 }
2179
2180 // Match a complicated way to check if a number is odd/even:
2181 // abs (srem X, 2) --> and X, 1
2182 const APInt *C;
2183 if (match(IIOperand, m_SRem(m_Value(X), m_APInt(C))) && *C == 2)
2184 return BinaryOperator::CreateAnd(X, ConstantInt::get(II->getType(), 1));
2185
2186 break;
2187 }
2188 case Intrinsic::umin: {
2189 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2190 // umin(x, 1) == zext(x != 0)
2191 if (match(I1, m_One())) {
2192 assert(II->getType()->getScalarSizeInBits() != 1 &&
2193 "Expected simplify of umin with max constant");
2194 Value *Zero = Constant::getNullValue(I0->getType());
2195 Value *Cmp = Builder.CreateICmpNE(I0, Zero);
2196 return CastInst::Create(Instruction::ZExt, Cmp, II->getType());
2197 }
2198 // umin(cttz(x), const) --> cttz(x | (1 << const))
2199 if (Value *FoldedCttz =
2201 I0, I1, DL, Builder))
2202 return replaceInstUsesWith(*II, FoldedCttz);
2203 // umin(ctlz(x), const) --> ctlz(x | (SignedMin >> const))
2204 if (Value *FoldedCtlz =
2206 I0, I1, DL, Builder))
2207 return replaceInstUsesWith(*II, FoldedCtlz);
2208 [[fallthrough]];
2209 }
2210 case Intrinsic::umax: {
2211 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2212 Value *X, *Y;
2213 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_ZExt(m_Value(Y))) &&
2214 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2215 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2216 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2217 }
2218 Constant *C;
2219 if (match(I0, m_ZExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2220 I0->hasOneUse()) {
2221 if (Constant *NarrowC = getLosslessUnsignedTrunc(C, X->getType(), DL)) {
2222 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2223 return CastInst::Create(Instruction::ZExt, NarrowMaxMin, II->getType());
2224 }
2225 }
2226 // If C is not 0:
2227 // umax(nuw_shl(x, C), x + 1) -> x == 0 ? 1 : nuw_shl(x, C)
2228 // If C is not 0 or 1:
2229 // umax(nuw_mul(x, C), x + 1) -> x == 0 ? 1 : nuw_mul(x, C)
2230 auto foldMaxMulShift = [&](Value *A, Value *B) -> Instruction * {
2231 const APInt *C;
2232 Value *X;
2233 if (!match(A, m_NUWShl(m_Value(X), m_APInt(C))) &&
2234 !(match(A, m_NUWMul(m_Value(X), m_APInt(C))) && !C->isOne()))
2235 return nullptr;
2236 if (C->isZero())
2237 return nullptr;
2238 if (!match(B, m_OneUse(m_Add(m_Specific(X), m_One()))))
2239 return nullptr;
2240
2241 Value *Cmp = Builder.CreateICmpEQ(X, ConstantInt::get(X->getType(), 0));
2242 Value *NewSelect = nullptr;
2243 NewSelect = Builder.CreateSelectWithUnknownProfile(
2244 Cmp, ConstantInt::get(X->getType(), 1), A, DEBUG_TYPE);
2245 return replaceInstUsesWith(*II, NewSelect);
2246 };
2247
2248 if (IID == Intrinsic::umax) {
2249 if (Instruction *I = foldMaxMulShift(I0, I1))
2250 return I;
2251 if (Instruction *I = foldMaxMulShift(I1, I0))
2252 return I;
2253 }
2254
2255 // If both operands of unsigned min/max are sign-extended, it is still ok
2256 // to narrow the operation.
2257 [[fallthrough]];
2258 }
2259 case Intrinsic::smax:
2260 case Intrinsic::smin: {
2261 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2262 Value *X, *Y;
2263 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_SExt(m_Value(Y))) &&
2264 (I0->hasOneUse() || I1->hasOneUse()) && X->getType() == Y->getType()) {
2265 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, Y);
2266 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2267 }
2268
2269 Constant *C;
2270 if (match(I0, m_SExt(m_Value(X))) && match(I1, m_Constant(C)) &&
2271 I0->hasOneUse()) {
2272 if (Constant *NarrowC = getLosslessSignedTrunc(C, X->getType(), DL)) {
2273 Value *NarrowMaxMin = Builder.CreateBinaryIntrinsic(IID, X, NarrowC);
2274 return CastInst::Create(Instruction::SExt, NarrowMaxMin, II->getType());
2275 }
2276 }
2277
2278 // smax(smin(X, MinC), MaxC) -> smin(smax(X, MaxC), MinC) if MinC s>= MaxC
2279 // umax(umin(X, MinC), MaxC) -> umin(umax(X, MaxC), MinC) if MinC u>= MaxC
2280 const APInt *MinC, *MaxC;
2281 auto CreateCanonicalClampForm = [&](bool IsSigned) {
2282 auto MaxIID = IsSigned ? Intrinsic::smax : Intrinsic::umax;
2283 auto MinIID = IsSigned ? Intrinsic::smin : Intrinsic::umin;
2284 Value *NewMax = Builder.CreateBinaryIntrinsic(
2285 MaxIID, X, ConstantInt::get(X->getType(), *MaxC));
2286 return replaceInstUsesWith(
2287 *II, Builder.CreateBinaryIntrinsic(
2288 MinIID, NewMax, ConstantInt::get(X->getType(), *MinC)));
2289 };
2290 if (IID == Intrinsic::smax &&
2292 m_APInt(MinC)))) &&
2293 match(I1, m_APInt(MaxC)) && MinC->sgt(*MaxC))
2294 return CreateCanonicalClampForm(true);
2295 if (IID == Intrinsic::umax &&
2297 m_APInt(MinC)))) &&
2298 match(I1, m_APInt(MaxC)) && MinC->ugt(*MaxC))
2299 return CreateCanonicalClampForm(false);
2300
2301 // umin(i1 X, i1 Y) -> and i1 X, Y
2302 // smax(i1 X, i1 Y) -> and i1 X, Y
2303 if ((IID == Intrinsic::umin || IID == Intrinsic::smax) &&
2304 II->getType()->isIntOrIntVectorTy(1)) {
2305 return BinaryOperator::CreateAnd(I0, I1);
2306 }
2307
2308 // umax(i1 X, i1 Y) -> or i1 X, Y
2309 // smin(i1 X, i1 Y) -> or i1 X, Y
2310 if ((IID == Intrinsic::umax || IID == Intrinsic::smin) &&
2311 II->getType()->isIntOrIntVectorTy(1)) {
2312 return BinaryOperator::CreateOr(I0, I1);
2313 }
2314
2315 // smin(smax(X, -1), 1) -> scmp(X, 0)
2316 // smax(smin(X, 1), -1) -> scmp(X, 0)
2317 // At this point, smax(smin(X, 1), -1) is changed to smin(smax(X, -1)
2318 // And i1's have been changed to and/ors
2319 // So we only need to check for smin
2320 if (IID == Intrinsic::smin) {
2321 if (match(I0, m_OneUse(m_SMax(m_Value(X), m_AllOnes()))) &&
2322 match(I1, m_One())) {
2323 Value *Zero = ConstantInt::get(X->getType(), 0);
2324 return replaceInstUsesWith(
2325 CI,
2326 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {X, Zero}));
2327 }
2328 }
2329
2330 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2331 // smax (neg nsw X), (neg nsw Y) --> neg nsw (smin X, Y)
2332 // smin (neg nsw X), (neg nsw Y) --> neg nsw (smax X, Y)
2333 // TODO: Canonicalize neg after min/max if I1 is constant.
2334 if (match(I0, m_NSWNeg(m_Value(X))) && match(I1, m_NSWNeg(m_Value(Y))) &&
2335 (I0->hasOneUse() || I1->hasOneUse())) {
2337 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, X, Y);
2338 return BinaryOperator::CreateNSWNeg(InvMaxMin);
2339 }
2340 }
2341
2342 // (umax X, (xor X, Pow2))
2343 // -> (or X, Pow2)
2344 // (umin X, (xor X, Pow2))
2345 // -> (and X, ~Pow2)
2346 // (smax X, (xor X, Pos_Pow2))
2347 // -> (or X, Pos_Pow2)
2348 // (smin X, (xor X, Pos_Pow2))
2349 // -> (and X, ~Pos_Pow2)
2350 // (smax X, (xor X, Neg_Pow2))
2351 // -> (and X, ~Neg_Pow2)
2352 // (smin X, (xor X, Neg_Pow2))
2353 // -> (or X, Neg_Pow2)
2354 if ((match(I0, m_c_Xor(m_Specific(I1), m_Value(X))) ||
2355 match(I1, m_c_Xor(m_Specific(I0), m_Value(X)))) &&
2356 isKnownToBeAPowerOfTwo(X, /* OrZero */ true)) {
2357 bool UseOr = IID == Intrinsic::smax || IID == Intrinsic::umax;
2358 bool UseAndN = IID == Intrinsic::smin || IID == Intrinsic::umin;
2359
2360 if (IID == Intrinsic::smax || IID == Intrinsic::smin) {
2361 auto KnownSign = getKnownSign(X, SQ.getWithInstruction(II));
2362 if (KnownSign == std::nullopt) {
2363 UseOr = false;
2364 UseAndN = false;
2365 } else if (*KnownSign /* true is Signed. */) {
2366 UseOr ^= true;
2367 UseAndN ^= true;
2368 Type *Ty = I0->getType();
2369 // Negative power of 2 must be IntMin. It's possible to be able to
2370 // prove negative / power of 2 without actually having known bits, so
2371 // just get the value by hand.
2373 Ty, APInt::getSignedMinValue(Ty->getScalarSizeInBits()));
2374 }
2375 }
2376 if (UseOr)
2377 return BinaryOperator::CreateOr(I0, X);
2378 else if (UseAndN)
2379 return BinaryOperator::CreateAnd(I0, Builder.CreateNot(X));
2380 }
2381
2382 // If we can eliminate ~A and Y is free to invert:
2383 // max ~A, Y --> ~(min A, ~Y)
2384 //
2385 // Examples:
2386 // max ~A, ~Y --> ~(min A, Y)
2387 // max ~A, C --> ~(min A, ~C)
2388 // max ~A, (max ~Y, ~Z) --> ~min( A, (min Y, Z))
2389 auto moveNotAfterMinMax = [&](Value *X, Value *Y) -> Instruction * {
2390 Value *A;
2391 if (match(X, m_OneUse(m_Not(m_Value(A)))) &&
2392 !isFreeToInvert(A, A->hasOneUse())) {
2393 if (Value *NotY = getFreelyInverted(Y, Y->hasOneUse(), &Builder)) {
2395 Value *InvMaxMin = Builder.CreateBinaryIntrinsic(InvID, A, NotY);
2396 return BinaryOperator::CreateNot(InvMaxMin);
2397 }
2398 }
2399 return nullptr;
2400 };
2401
2402 if (Instruction *I = moveNotAfterMinMax(I0, I1))
2403 return I;
2404 if (Instruction *I = moveNotAfterMinMax(I1, I0))
2405 return I;
2406
2408 return I;
2409
2410 // minmax (X & NegPow2C, Y & NegPow2C) --> minmax(X, Y) & NegPow2C
2411 const APInt *RHSC;
2412 if (match(I0, m_OneUse(m_And(m_Value(X), m_NegatedPower2(RHSC)))) &&
2413 match(I1, m_OneUse(m_And(m_Value(Y), m_SpecificInt(*RHSC)))))
2414 return BinaryOperator::CreateAnd(Builder.CreateBinaryIntrinsic(IID, X, Y),
2415 ConstantInt::get(II->getType(), *RHSC));
2416
2417 // smax(X, -X) --> abs(X)
2418 // smin(X, -X) --> -abs(X)
2419 // umax(X, -X) --> -abs(X)
2420 // umin(X, -X) --> abs(X)
2421 if (isKnownNegation(I0, I1)) {
2422 // We can choose either operand as the input to abs(), but if we can
2423 // eliminate the only use of a value, that's better for subsequent
2424 // transforms/analysis.
2425 if (I0->hasOneUse() && !I1->hasOneUse())
2426 std::swap(I0, I1);
2427
2428 // This is some variant of abs(). See if we can propagate 'nsw' to the abs
2429 // operation and potentially its negation.
2430 bool IntMinIsPoison = isKnownNegation(I0, I1, /* NeedNSW */ true);
2431 Value *Abs = Builder.CreateBinaryIntrinsic(
2432 Intrinsic::abs, I0,
2433 ConstantInt::getBool(II->getContext(), IntMinIsPoison));
2434
2435 // We don't have a "nabs" intrinsic, so negate if needed based on the
2436 // max/min operation.
2437 if (IID == Intrinsic::smin || IID == Intrinsic::umax)
2438 Abs = Builder.CreateNeg(Abs, "nabs", IntMinIsPoison);
2439 return replaceInstUsesWith(CI, Abs);
2440 }
2441
2443 return Sel;
2444
2445 if (Instruction *SAdd = matchSAddSubSat(*II))
2446 return SAdd;
2447
2448 if (Value *NewMinMax = reassociateMinMaxWithConstants(II, Builder, SQ))
2449 return replaceInstUsesWith(*II, NewMinMax);
2450
2452 return R;
2453
2454 if (Instruction *NewMinMax = factorizeMinMaxTree(II))
2455 return NewMinMax;
2456
2457 // Try to fold minmax with constant RHS based on range information
2458 if (match(I1, m_APIntAllowPoison(RHSC))) {
2459 ICmpInst::Predicate Pred =
2461 bool IsSigned = MinMaxIntrinsic::isSigned(IID);
2463 I0, IsSigned, SQ.getWithInstruction(II));
2464 if (!LHS_CR.isFullSet()) {
2465 if (LHS_CR.icmp(Pred, *RHSC))
2466 return replaceInstUsesWith(*II, I0);
2467 if (LHS_CR.icmp(ICmpInst::getSwappedPredicate(Pred), *RHSC))
2468 return replaceInstUsesWith(*II,
2469 ConstantInt::get(II->getType(), *RHSC));
2470 }
2471 }
2472
2474 return replaceInstUsesWith(*II, V);
2475
2476 break;
2477 }
2478 case Intrinsic::scmp: {
2479 Value *I0 = II->getArgOperand(0), *I1 = II->getArgOperand(1);
2480 Value *LHS, *RHS;
2481 if (match(I0, m_NSWSub(m_Value(LHS), m_Value(RHS))) && match(I1, m_Zero()))
2482 return replaceInstUsesWith(
2483 CI,
2484 Builder.CreateIntrinsic(II->getType(), Intrinsic::scmp, {LHS, RHS}));
2485 break;
2486 }
2487 case Intrinsic::bitreverse: {
2488 Value *IIOperand = II->getArgOperand(0);
2489 // bitrev (zext i1 X to ?) --> X ? SignBitC : 0
2490 Value *X;
2491 if (match(IIOperand, m_ZExt(m_Value(X))) &&
2492 X->getType()->isIntOrIntVectorTy(1)) {
2493 Type *Ty = II->getType();
2494 APInt SignBit = APInt::getSignMask(Ty->getScalarSizeInBits());
2495 return SelectInst::Create(X, ConstantInt::get(Ty, SignBit),
2497 }
2498
2499 if (Instruction *crossLogicOpFold =
2501 return crossLogicOpFold;
2502
2503 break;
2504 }
2505 case Intrinsic::bswap: {
2506 Value *IIOperand = II->getArgOperand(0);
2507
2508 // Try to canonicalize bswap-of-logical-shift-by-8-bit-multiple as
2509 // inverse-shift-of-bswap:
2510 // bswap (shl X, Y) --> lshr (bswap X), Y
2511 // bswap (lshr X, Y) --> shl (bswap X), Y
2512 Value *X, *Y;
2513 if (match(IIOperand, m_OneUse(m_LogicalShift(m_Value(X), m_Value(Y))))) {
2514 unsigned BitWidth = IIOperand->getType()->getScalarSizeInBits();
2516 Value *NewSwap = Builder.CreateUnaryIntrinsic(Intrinsic::bswap, X);
2517 BinaryOperator::BinaryOps InverseShift =
2518 cast<BinaryOperator>(IIOperand)->getOpcode() == Instruction::Shl
2519 ? Instruction::LShr
2520 : Instruction::Shl;
2521 return BinaryOperator::Create(InverseShift, NewSwap, Y);
2522 }
2523 }
2524
2525 KnownBits Known = computeKnownBits(IIOperand, II);
2526 uint64_t LZ = alignDown(Known.countMinLeadingZeros(), 8);
2527 uint64_t TZ = alignDown(Known.countMinTrailingZeros(), 8);
2528 unsigned BW = Known.getBitWidth();
2529
2530 // bswap(x) -> shift(x) if x has exactly one "active byte"
2531 if (BW - LZ - TZ == 8) {
2532 assert(LZ != TZ && "active byte cannot be in the middle");
2533 if (LZ > TZ) // -> shl(x) if the "active byte" is in the low part of x
2534 return BinaryOperator::CreateNUWShl(
2535 IIOperand, ConstantInt::get(IIOperand->getType(), LZ - TZ));
2536 // -> lshr(x) if the "active byte" is in the high part of x
2537 return BinaryOperator::CreateExactLShr(
2538 IIOperand, ConstantInt::get(IIOperand->getType(), TZ - LZ));
2539 }
2540
2541 // bswap(trunc(bswap(x))) -> trunc(lshr(x, c))
2542 if (match(IIOperand, m_Trunc(m_BSwap(m_Value(X))))) {
2543 unsigned C = X->getType()->getScalarSizeInBits() - BW;
2544 Value *CV = ConstantInt::get(X->getType(), C);
2545 Value *V = Builder.CreateLShr(X, CV);
2546 return new TruncInst(V, IIOperand->getType());
2547 }
2548
2549 if (Instruction *crossLogicOpFold =
2551 return crossLogicOpFold;
2552 }
2553
2554 // Try to fold into bitreverse if bswap is the root of the expression tree.
2555 if (Instruction *BitOp = matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ false,
2556 /*MatchBitReversals*/ true))
2557 return BitOp;
2558 break;
2559 }
2560 case Intrinsic::masked_load:
2561 if (Value *SimplifiedMaskedOp = simplifyMaskedLoad(*II))
2562 return replaceInstUsesWith(CI, SimplifiedMaskedOp);
2563 break;
2564 case Intrinsic::masked_store:
2565 return simplifyMaskedStore(*II);
2566 case Intrinsic::masked_gather:
2567 return simplifyMaskedGather(*II);
2568 case Intrinsic::masked_scatter:
2569 return simplifyMaskedScatter(*II);
2570 case Intrinsic::launder_invariant_group:
2571 case Intrinsic::strip_invariant_group:
2572 if (auto *SkippedBarrier = simplifyInvariantGroupIntrinsic(*II, *this))
2573 return replaceInstUsesWith(*II, SkippedBarrier);
2574 break;
2575 case Intrinsic::powi: {
2576 if (ConstantInt *Power = dyn_cast<ConstantInt>(II->getArgOperand(1))) {
2577 // 0 and 1 are handled in instsimplify
2578 // powi(x, -1) -> 1/x
2579 if (Power->isMinusOne())
2580 return BinaryOperator::CreateFDivFMF(ConstantFP::get(CI.getType(), 1.0),
2581 II->getArgOperand(0), II);
2582 // powi(x, 2) -> x*x
2583 if (Power->equalsInt(2))
2584 return BinaryOperator::CreateFMulFMF(II->getArgOperand(0),
2585 II->getArgOperand(0), II);
2586
2587 if (!Power->getValue()[0]) {
2588 Value *X;
2589 // If power is even:
2590 // powi(-x, p) -> powi(x, p)
2591 // powi(fabs(x), p) -> powi(x, p)
2592 // powi(copysign(x, y), p) -> powi(x, p)
2593 if (match(II->getArgOperand(0), m_FNeg(m_Value(X))) ||
2594 match(II->getArgOperand(0), m_FAbs(m_Value(X))) ||
2595 match(II->getArgOperand(0),
2597 return CallInst::Create(II->getCalledFunction(), {X, Power});
2598 }
2599 }
2600 if (ConstantFP *Base = dyn_cast<ConstantFP>(II->getArgOperand(0))) {
2601 Value *Exp = II->getArgOperand(1);
2602 Type *Ty = Base->getType();
2603 // powi(2.0, p) -> ldexp(1.0, p)
2604 if (II->hasApproxFunc() && Base->isExactlyValue(2.0)) {
2605 ConstantFP *One = ConstantFP::get(Ty, 1.0);
2606 if (auto *VTy = dyn_cast<VectorType>(Ty))
2607 Exp = Builder.CreateVectorSplat(VTy->getElementCount(), Exp);
2608 Value *Ldexp = Builder.CreateLdexp(One, Exp, II);
2609 return replaceInstUsesWith(*II, Ldexp);
2610 }
2611 }
2612 break;
2613 }
2614
2615 case Intrinsic::cttz:
2616 case Intrinsic::ctlz:
2617 if (auto *I = foldCttzCtlz(*II, *this))
2618 return I;
2619 break;
2620
2621 case Intrinsic::ctpop:
2622 if (auto *I = foldCtpop(*II, *this))
2623 return I;
2624 break;
2625
2626 case Intrinsic::fshl:
2627 case Intrinsic::fshr: {
2628 Value *Op0 = II->getArgOperand(0), *Op1 = II->getArgOperand(1);
2629 Type *Ty = II->getType();
2630 unsigned BitWidth = Ty->getScalarSizeInBits();
2631 Constant *ShAmtC;
2632 if (match(II->getArgOperand(2), m_ImmConstant(ShAmtC))) {
2633 // Canonicalize a shift amount constant operand to modulo the bit-width.
2634 Constant *WidthC = ConstantInt::get(Ty, BitWidth);
2635 Constant *ModuloC =
2636 ConstantFoldBinaryOpOperands(Instruction::URem, ShAmtC, WidthC, DL);
2637 if (!ModuloC)
2638 return nullptr;
2639 if (ModuloC != ShAmtC)
2640 return CallInst::Create(II->getCalledFunction(), {Op0, Op1, ModuloC});
2641
2643 ShAmtC, DL),
2644 m_One()) &&
2645 "Shift amount expected to be modulo bitwidth");
2646
2647 // Canonicalize funnel shift right by constant to funnel shift left. This
2648 // is not entirely arbitrary. For historical reasons, the backend may
2649 // recognize rotate left patterns but miss rotate right patterns.
2650 if (IID == Intrinsic::fshr) {
2651 // fshr X, Y, C --> fshl X, Y, (BitWidth - C) if C is not zero.
2652 if (!isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2653 return nullptr;
2654
2655 Constant *LeftShiftC = ConstantExpr::getSub(WidthC, ShAmtC);
2656 Module *Mod = II->getModule();
2657 Function *Fshl =
2658 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::fshl, Ty);
2659 return CallInst::Create(Fshl, { Op0, Op1, LeftShiftC });
2660 }
2661 assert(IID == Intrinsic::fshl &&
2662 "All funnel shifts by simple constants should go left");
2663
2664 // fshl(X, 0, C) --> shl X, C
2665 // fshl(X, undef, C) --> shl X, C
2666 if (match(Op1, m_ZeroInt()) || match(Op1, m_Undef()))
2667 return BinaryOperator::CreateShl(Op0, ShAmtC);
2668
2669 // fshl(0, X, C) --> lshr X, (BW-C)
2670 // fshl(undef, X, C) --> lshr X, (BW-C)
2671 // Similar to fshr -> fshl fold above, this is only valid if C is not zero
2672 if ((match(Op0, m_ZeroInt()) || match(Op0, m_Undef())) &&
2673 isKnownNonZero(ShAmtC, SQ.getWithInstruction(II)))
2674 return BinaryOperator::CreateLShr(Op1,
2675 ConstantExpr::getSub(WidthC, ShAmtC));
2676
2677 // fshl i16 X, X, 8 --> bswap i16 X (reduce to more-specific form)
2678 if (Op0 == Op1 && BitWidth == 16 && match(ShAmtC, m_SpecificInt(8))) {
2679 Module *Mod = II->getModule();
2680 Function *Bswap =
2681 Intrinsic::getOrInsertDeclaration(Mod, Intrinsic::bswap, Ty);
2682 return CallInst::Create(Bswap, { Op0 });
2683 }
2684 if (Instruction *BitOp =
2685 matchBSwapOrBitReverse(*II, /*MatchBSwaps*/ true,
2686 /*MatchBitReversals*/ true))
2687 return BitOp;
2688
2689 // R = fshl(X, X, C2)
2690 // fshl(R, R, C1) --> fshl(X, X, (C1 + C2) % bitsize)
2691 Value *InnerOp;
2692 const APInt *ShAmtInnerC, *ShAmtOuterC;
2693 if (match(Op0, m_FShl(m_Value(InnerOp), m_Deferred(InnerOp),
2694 m_APInt(ShAmtInnerC))) &&
2695 match(ShAmtC, m_APInt(ShAmtOuterC)) && Op0 == Op1) {
2696 APInt Sum = *ShAmtOuterC + *ShAmtInnerC;
2697 APInt Modulo = Sum.urem(APInt(Sum.getBitWidth(), BitWidth));
2698 if (Modulo.isZero())
2699 return replaceInstUsesWith(*II, InnerOp);
2700 Constant *ModuloC = ConstantInt::get(Ty, Modulo);
2702 {InnerOp, InnerOp, ModuloC});
2703 }
2704 }
2705
2706 // fshl(X, X, Neg(Y)) --> fshr(X, X, Y)
2707 // fshr(X, X, Neg(Y)) --> fshl(X, X, Y)
2708 // if BitWidth is a power-of-2
2709 Value *Y;
2710 if (Op0 == Op1 && isPowerOf2_32(BitWidth) &&
2711 match(II->getArgOperand(2), m_Neg(m_Value(Y)))) {
2712 Module *Mod = II->getModule();
2714 Mod, IID == Intrinsic::fshl ? Intrinsic::fshr : Intrinsic::fshl, Ty);
2715 return CallInst::Create(OppositeShift, {Op0, Op1, Y});
2716 }
2717
2718 // fshl(X, 0, Y) --> shl(X, and(Y, BitWidth - 1)) if bitwidth is a
2719 // power-of-2
2720 if (IID == Intrinsic::fshl && isPowerOf2_32(BitWidth) &&
2721 match(Op1, m_ZeroInt())) {
2722 Value *Op2 = II->getArgOperand(2);
2723 Value *And = Builder.CreateAnd(Op2, ConstantInt::get(Ty, BitWidth - 1));
2724 return BinaryOperator::CreateShl(Op0, And);
2725 }
2726
2727 // Left or right might be masked.
2729 return &CI;
2730
2731 // The shift amount (operand 2) of a funnel shift is modulo the bitwidth,
2732 // so only the low bits of the shift amount are demanded if the bitwidth is
2733 // a power-of-2.
2734 if (!isPowerOf2_32(BitWidth))
2735 break;
2737 KnownBits Op2Known(BitWidth);
2738 if (SimplifyDemandedBits(II, 2, Op2Demanded, Op2Known))
2739 return &CI;
2740 break;
2741 }
2742 case Intrinsic::pdep: {
2743 const APInt *MaskC;
2744 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2745 unsigned MaskIdx, MaskLen;
2746 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2747 // any single contiguous sequence of 1s anywhere in the mask simply
2748 // describes a subset of the input bits shifted to the appropriate
2749 // position. Replace with the straight forward IR.
2750 Value *Input = II->getArgOperand(0);
2751 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2752 Value *Shifted = Builder.CreateShl(Input, ShiftAmt);
2753 Value *Masked = Builder.CreateAnd(Shifted, II->getArgOperand(1));
2754 return replaceInstUsesWith(*II, Masked);
2755 }
2756 }
2757 break;
2758 }
2759 case Intrinsic::pext: {
2760 const APInt *MaskC;
2761 if (match(II->getArgOperand(1), m_APInt(MaskC))) {
2762 unsigned MaskIdx, MaskLen;
2763 if (MaskC->isShiftedMask(MaskIdx, MaskLen)) {
2764 // any single contiguous sequence of 1s anywhere in the mask simply
2765 // describes a subset of the input bits shifted to the appropriate
2766 // position. Replace with the straight forward IR.
2767 Value *Input = II->getArgOperand(0);
2768 Value *Masked = Builder.CreateAnd(Input, II->getArgOperand(1));
2769 Value *ShiftAmt = ConstantInt::get(II->getType(), MaskIdx);
2770 Value *Shifted = Builder.CreateLShr(Masked, ShiftAmt);
2771 return replaceInstUsesWith(*II, Shifted);
2772 }
2773 }
2774 break;
2775 }
2776 case Intrinsic::ptrmask: {
2777 unsigned BitWidth = DL.getPointerTypeSizeInBits(II->getType());
2780 return II;
2781
2782 Value *InnerPtr, *InnerMask;
2783 bool Changed = false;
2784 // Combine:
2785 // (ptrmask (ptrmask p, A), B)
2786 // -> (ptrmask p, (and A, B))
2787 if (match(II->getArgOperand(0),
2789 m_Value(InnerMask))))) {
2790 assert(II->getArgOperand(1)->getType() == InnerMask->getType() &&
2791 "Mask types must match");
2792 // TODO: If InnerMask == Op1, we could copy attributes from inner
2793 // callsite -> outer callsite.
2794 Value *NewMask = Builder.CreateAnd(II->getArgOperand(1), InnerMask);
2795 replaceOperand(CI, 0, InnerPtr);
2796 replaceOperand(CI, 1, NewMask);
2797 Changed = true;
2798 }
2799
2800 // See if we can deduce non-null.
2801 if (!CI.hasRetAttr(Attribute::NonNull) &&
2802 (Known.isNonZero() ||
2803 isKnownNonZero(II, getSimplifyQuery().getWithInstruction(II)))) {
2804 CI.addRetAttr(Attribute::NonNull);
2805 Changed = true;
2806 }
2807
2808 unsigned NewAlignmentLog =
2810 std::min(BitWidth - 1, Known.countMinTrailingZeros()));
2811 // Known bits will capture if we had alignment information associated with
2812 // the pointer argument.
2813 if (NewAlignmentLog > Log2(CI.getRetAlign().valueOrOne())) {
2815 CI.getContext(), Align(uint64_t(1) << NewAlignmentLog)));
2816 Changed = true;
2817 }
2818 if (Changed)
2819 return &CI;
2820 break;
2821 }
2822 case Intrinsic::uadd_with_overflow:
2823 case Intrinsic::sadd_with_overflow: {
2824 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2825 return I;
2826
2827 // Given 2 constant operands whose sum does not overflow:
2828 // uaddo (X +nuw C0), C1 -> uaddo X, C0 + C1
2829 // saddo (X +nsw C0), C1 -> saddo X, C0 + C1
2830 Value *X;
2831 const APInt *C0, *C1;
2832 Value *Arg0 = II->getArgOperand(0);
2833 Value *Arg1 = II->getArgOperand(1);
2834 bool IsSigned = IID == Intrinsic::sadd_with_overflow;
2835 bool HasNWAdd = IsSigned
2836 ? match(Arg0, m_NSWAddLike(m_Value(X), m_APInt(C0)))
2837 : match(Arg0, m_NUWAddLike(m_Value(X), m_APInt(C0)));
2838 if (HasNWAdd && match(Arg1, m_APInt(C1))) {
2839 bool Overflow;
2840 APInt NewC =
2841 IsSigned ? C1->sadd_ov(*C0, Overflow) : C1->uadd_ov(*C0, Overflow);
2842 if (!Overflow)
2843 return replaceInstUsesWith(
2844 *II, Builder.CreateBinaryIntrinsic(
2845 IID, X, ConstantInt::get(Arg1->getType(), NewC)));
2846 }
2847 break;
2848 }
2849
2850 case Intrinsic::umul_with_overflow:
2851 case Intrinsic::smul_with_overflow:
2852 case Intrinsic::usub_with_overflow:
2853 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2854 return I;
2855 break;
2856
2857 case Intrinsic::ssub_with_overflow: {
2858 if (Instruction *I = foldIntrinsicWithOverflowCommon(II))
2859 return I;
2860
2861 Constant *C;
2862 Value *Arg0 = II->getArgOperand(0);
2863 Value *Arg1 = II->getArgOperand(1);
2864 // Given a constant C that is not the minimum signed value
2865 // for an integer of a given bit width:
2866 //
2867 // ssubo X, C -> saddo X, -C
2868 if (match(Arg1, m_Constant(C)) && C->isNotMinSignedValue()) {
2869 Value *NegVal = ConstantExpr::getNeg(C);
2870 // Build a saddo call that is equivalent to the discovered
2871 // ssubo call.
2872 return replaceInstUsesWith(
2873 *II, Builder.CreateBinaryIntrinsic(Intrinsic::sadd_with_overflow,
2874 Arg0, NegVal));
2875 }
2876
2877 break;
2878 }
2879
2880 case Intrinsic::uadd_sat:
2881 case Intrinsic::sadd_sat:
2882 case Intrinsic::usub_sat:
2883 case Intrinsic::ssub_sat: {
2885 Type *Ty = SI->getType();
2886 Value *Arg0 = SI->getLHS();
2887 Value *Arg1 = SI->getRHS();
2888
2889 // Make use of known overflow information.
2890 OverflowResult OR = computeOverflow(SI->getBinaryOp(), SI->isSigned(),
2891 Arg0, Arg1, SI);
2892 switch (OR) {
2894 break;
2896 if (SI->isSigned())
2897 return BinaryOperator::CreateNSW(SI->getBinaryOp(), Arg0, Arg1);
2898 else
2899 return BinaryOperator::CreateNUW(SI->getBinaryOp(), Arg0, Arg1);
2901 unsigned BitWidth = Ty->getScalarSizeInBits();
2902 APInt Min = APSInt::getMinValue(BitWidth, !SI->isSigned());
2903 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Min));
2904 }
2906 unsigned BitWidth = Ty->getScalarSizeInBits();
2907 APInt Max = APSInt::getMaxValue(BitWidth, !SI->isSigned());
2908 return replaceInstUsesWith(*SI, ConstantInt::get(Ty, Max));
2909 }
2910 }
2911
2912 // usub_sat((sub nuw C, A), C1) -> usub_sat(usub_sat(C, C1), A)
2913 // which after that:
2914 // usub_sat((sub nuw C, A), C1) -> usub_sat(C - C1, A) if C1 u< C
2915 // usub_sat((sub nuw C, A), C1) -> 0 otherwise
2916 Constant *C, *C1;
2917 Value *A;
2918 if (IID == Intrinsic::usub_sat &&
2919 match(Arg0, m_NUWSub(m_ImmConstant(C), m_Value(A))) &&
2920 match(Arg1, m_ImmConstant(C1))) {
2921 auto *NewC = Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, C, C1);
2922 auto *NewSub =
2923 Builder.CreateBinaryIntrinsic(Intrinsic::usub_sat, NewC, A);
2924 return replaceInstUsesWith(*SI, NewSub);
2925 }
2926
2927 // ssub.sat(X, C) -> sadd.sat(X, -C) if C != MIN
2928 if (IID == Intrinsic::ssub_sat && match(Arg1, m_Constant(C)) &&
2929 C->isNotMinSignedValue()) {
2930 Value *NegVal = ConstantExpr::getNeg(C);
2931 return replaceInstUsesWith(
2932 *II, Builder.CreateBinaryIntrinsic(
2933 Intrinsic::sadd_sat, Arg0, NegVal));
2934 }
2935
2936 // sat(sat(X + Val2) + Val) -> sat(X + (Val+Val2))
2937 // sat(sat(X - Val2) - Val) -> sat(X - (Val+Val2))
2938 // if Val and Val2 have the same sign
2939 if (auto *Other = dyn_cast<IntrinsicInst>(Arg0)) {
2940 Value *X;
2941 const APInt *Val, *Val2;
2942 APInt NewVal;
2943 bool IsUnsigned =
2944 IID == Intrinsic::uadd_sat || IID == Intrinsic::usub_sat;
2945 if (Other->getIntrinsicID() == IID &&
2946 match(Arg1, m_APInt(Val)) &&
2947 match(Other->getArgOperand(0), m_Value(X)) &&
2948 match(Other->getArgOperand(1), m_APInt(Val2))) {
2949 if (IsUnsigned)
2950 NewVal = Val->uadd_sat(*Val2);
2951 else if (Val->isNonNegative() == Val2->isNonNegative()) {
2952 bool Overflow;
2953 NewVal = Val->sadd_ov(*Val2, Overflow);
2954 if (Overflow) {
2955 // Both adds together may add more than SignedMaxValue
2956 // without saturating the final result.
2957 break;
2958 }
2959 } else {
2960 // Cannot fold saturated addition with different signs.
2961 break;
2962 }
2963
2964 return replaceInstUsesWith(
2965 *II, Builder.CreateBinaryIntrinsic(
2966 IID, X, ConstantInt::get(II->getType(), NewVal)));
2967 }
2968 }
2969 break;
2970 }
2971
2972 case Intrinsic::minnum:
2973 case Intrinsic::maxnum:
2974 case Intrinsic::minimumnum:
2975 case Intrinsic::maximumnum:
2976 case Intrinsic::minimum:
2977 case Intrinsic::maximum: {
2978 Value *Arg0 = II->getArgOperand(0);
2979 Value *Arg1 = II->getArgOperand(1);
2980 Value *X, *Y;
2981 if (match(Arg0, m_FNeg(m_Value(X))) && match(Arg1, m_FNeg(m_Value(Y))) &&
2982 (Arg0->hasOneUse() || Arg1->hasOneUse())) {
2983 // If both operands are negated, invert the call and negate the result:
2984 // min(-X, -Y) --> -(max(X, Y))
2985 // max(-X, -Y) --> -(min(X, Y))
2986 Intrinsic::ID NewIID;
2987 switch (IID) {
2988 case Intrinsic::maxnum:
2989 NewIID = Intrinsic::minnum;
2990 break;
2991 case Intrinsic::minnum:
2992 NewIID = Intrinsic::maxnum;
2993 break;
2994 case Intrinsic::maximumnum:
2995 NewIID = Intrinsic::minimumnum;
2996 break;
2997 case Intrinsic::minimumnum:
2998 NewIID = Intrinsic::maximumnum;
2999 break;
3000 case Intrinsic::maximum:
3001 NewIID = Intrinsic::minimum;
3002 break;
3003 case Intrinsic::minimum:
3004 NewIID = Intrinsic::maximum;
3005 break;
3006 default:
3007 llvm_unreachable("unexpected intrinsic ID");
3008 }
3009 Value *NewCall = Builder.CreateBinaryIntrinsic(NewIID, X, Y, II);
3010 Instruction *FNeg = UnaryOperator::CreateFNeg(NewCall);
3011 FNeg->copyIRFlags(II);
3012 return FNeg;
3013 }
3014
3015 // m(m(X, C2), C1) -> m(X, C)
3016 const APFloat *C1, *C2;
3017 if (auto *M = dyn_cast<IntrinsicInst>(Arg0)) {
3018 if (M->getIntrinsicID() == IID && match(Arg1, m_APFloat(C1)) &&
3019 ((match(M->getArgOperand(0), m_Value(X)) &&
3020 match(M->getArgOperand(1), m_APFloat(C2))) ||
3021 (match(M->getArgOperand(1), m_Value(X)) &&
3022 match(M->getArgOperand(0), m_APFloat(C2))))) {
3023 APFloat Res(0.0);
3024 switch (IID) {
3025 case Intrinsic::maxnum:
3026 Res = maxnum(*C1, *C2);
3027 break;
3028 case Intrinsic::minnum:
3029 Res = minnum(*C1, *C2);
3030 break;
3031 case Intrinsic::maximumnum:
3032 Res = maximumnum(*C1, *C2);
3033 break;
3034 case Intrinsic::minimumnum:
3035 Res = minimumnum(*C1, *C2);
3036 break;
3037 case Intrinsic::maximum:
3038 Res = maximum(*C1, *C2);
3039 break;
3040 case Intrinsic::minimum:
3041 Res = minimum(*C1, *C2);
3042 break;
3043 default:
3044 llvm_unreachable("unexpected intrinsic ID");
3045 }
3046 // TODO: Conservatively intersecting FMF. If Res == C2, the transform
3047 // was a simplification (so Arg0 and its original flags could
3048 // propagate?)
3049 Value *V = Builder.CreateBinaryIntrinsic(
3050 IID, X, ConstantFP::get(Arg0->getType(), Res),
3052 return replaceInstUsesWith(*II, V);
3053 }
3054 }
3055
3056 // m((fpext X), (fpext Y)) -> fpext (m(X, Y))
3057 if (match(Arg0, m_FPExt(m_Value(X))) && match(Arg1, m_FPExt(m_Value(Y))) &&
3058 (Arg0->hasOneUse() || Arg1->hasOneUse()) &&
3059 X->getType() == Y->getType()) {
3060 Value *NewCall =
3061 Builder.CreateBinaryIntrinsic(IID, X, Y, II, II->getName());
3062 return new FPExtInst(NewCall, II->getType());
3063 }
3064
3065 // m(fpext X, C) -> fpext m(X, TruncC) if C can be losslessly truncated.
3066 Constant *C;
3067 if (match(Arg0, m_OneUse(m_FPExt(m_Value(X)))) &&
3068 match(Arg1, m_ImmConstant(C))) {
3069 if (Constant *TruncC =
3070 getLosslessInvCast(C, X->getType(), Instruction::FPExt, DL)) {
3071 Value *NewCall =
3072 Builder.CreateBinaryIntrinsic(IID, X, TruncC, II, II->getName());
3073 return new FPExtInst(NewCall, II->getType());
3074 }
3075 }
3076
3077 // max X, -X --> fabs X
3078 // min X, -X --> -(fabs X)
3079 // TODO: Remove one-use limitation? That is obviously better for max,
3080 // hence why we don't check for one-use for that. However,
3081 // it would be an extra instruction for min (fnabs), but
3082 // that is still likely better for analysis and codegen.
3083 auto IsMinMaxOrXNegX = [IID, &X](Value *Op0, Value *Op1) {
3084 if (match(Op0, m_FNeg(m_Value(X))) && match(Op1, m_Specific(X)))
3085 return Op0->hasOneUse() ||
3086 (IID != Intrinsic::minimum && IID != Intrinsic::minnum &&
3087 IID != Intrinsic::minimumnum);
3088 return false;
3089 };
3090
3091 if (IsMinMaxOrXNegX(Arg0, Arg1) || IsMinMaxOrXNegX(Arg1, Arg0)) {
3092 Value *R = Builder.CreateFAbs(X, II);
3093 if (IID == Intrinsic::minimum || IID == Intrinsic::minnum ||
3094 IID == Intrinsic::minimumnum)
3095 R = Builder.CreateFNegFMF(R, II);
3096 return replaceInstUsesWith(*II, R);
3097 }
3098
3099 break;
3100 }
3101 case Intrinsic::matrix_multiply: {
3102 // Optimize negation in matrix multiplication.
3103
3104 // -A * -B -> A * B
3105 Value *A, *B;
3106 if (match(II->getArgOperand(0), m_FNeg(m_Value(A))) &&
3107 match(II->getArgOperand(1), m_FNeg(m_Value(B)))) {
3108 replaceOperand(*II, 0, A);
3109 replaceOperand(*II, 1, B);
3110 return II;
3111 }
3112
3113 Value *Op0 = II->getOperand(0);
3114 Value *Op1 = II->getOperand(1);
3115 Value *OpNotNeg, *NegatedOp;
3116 unsigned NegatedOpArg, OtherOpArg;
3117 if (match(Op0, m_FNeg(m_Value(OpNotNeg)))) {
3118 NegatedOp = Op0;
3119 NegatedOpArg = 0;
3120 OtherOpArg = 1;
3121 } else if (match(Op1, m_FNeg(m_Value(OpNotNeg)))) {
3122 NegatedOp = Op1;
3123 NegatedOpArg = 1;
3124 OtherOpArg = 0;
3125 } else
3126 // Multiplication doesn't have a negated operand.
3127 break;
3128
3129 // Only optimize if the negated operand has only one use.
3130 if (!NegatedOp->hasOneUse())
3131 break;
3132
3133 Value *OtherOp = II->getOperand(OtherOpArg);
3134 VectorType *RetTy = cast<VectorType>(II->getType());
3135 VectorType *NegatedOpTy = cast<VectorType>(NegatedOp->getType());
3136 VectorType *OtherOpTy = cast<VectorType>(OtherOp->getType());
3137 ElementCount NegatedCount = NegatedOpTy->getElementCount();
3138 ElementCount OtherCount = OtherOpTy->getElementCount();
3139 ElementCount RetCount = RetTy->getElementCount();
3140 // (-A) * B -> A * (-B), if it is cheaper to negate B and vice versa.
3141 if (ElementCount::isKnownGT(NegatedCount, OtherCount) &&
3142 ElementCount::isKnownLT(OtherCount, RetCount)) {
3143 Value *InverseOtherOp = Builder.CreateFNeg(OtherOp);
3144 replaceOperand(*II, NegatedOpArg, OpNotNeg);
3145 replaceOperand(*II, OtherOpArg, InverseOtherOp);
3146 return II;
3147 }
3148 // (-A) * B -> -(A * B), if it is cheaper to negate the result
3149 if (ElementCount::isKnownGT(NegatedCount, RetCount)) {
3150 SmallVector<Value *, 5> NewArgs(II->args());
3151 NewArgs[NegatedOpArg] = OpNotNeg;
3152 Value *NewMul = Builder.CreateIntrinsic(II->getType(), IID, NewArgs, II);
3153 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(NewMul, II));
3154 }
3155 break;
3156 }
3157 case Intrinsic::fmuladd: {
3158 // Try to simplify the underlying FMul.
3159 if (Value *V =
3160 simplifyFMulInst(II->getArgOperand(0), II->getArgOperand(1),
3161 II->getFastMathFlags(), SQ.getWithInstruction(II)))
3162 return BinaryOperator::CreateFAddFMF(V, II->getArgOperand(2),
3163 II->getFastMathFlags());
3164
3165 [[fallthrough]];
3166 }
3167 case Intrinsic::fma: {
3168 // fma fneg(x), fneg(y), z -> fma x, y, z
3169 Value *Src0 = II->getArgOperand(0);
3170 Value *Src1 = II->getArgOperand(1);
3171 Value *Src2 = II->getArgOperand(2);
3172 Value *X, *Y;
3173 if (match(Src0, m_FNeg(m_Value(X))) && match(Src1, m_FNeg(m_Value(Y))))
3174 return replaceInstUsesWith(
3175 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, Y, Src2}, II));
3176
3177 // fma fabs(x), fabs(x), z -> fma x, x, z
3178 if (match(Src0, m_FAbs(m_Value(X))) && match(Src1, m_FAbs(m_Specific(X))))
3179 return replaceInstUsesWith(
3180 *II, Builder.CreateIntrinsic(IID, II->getType(), {X, X, Src2}, II));
3181
3182 // Try to simplify the underlying FMul. We can only apply simplifications
3183 // that do not require rounding.
3184 if (Value *V = simplifyFMAFMul(Src0, Src1, II->getFastMathFlags(),
3185 SQ.getWithInstruction(II)))
3186 return BinaryOperator::CreateFAddFMF(V, Src2, II->getFastMathFlags());
3187
3188 // fma x, y, 0 -> fmul x, y
3189 // This is always valid for -0.0, but requires nsz for +0.0 as
3190 // -0.0 + 0.0 = 0.0, which would not be the same as the fmul on its own.
3191 if (match(Src2, m_NegZeroFP()) ||
3192 (match(Src2, m_PosZeroFP()) && II->getFastMathFlags().noSignedZeros()))
3193 return BinaryOperator::CreateFMulFMF(Src0, Src1, II);
3194
3195 // fma x, -1.0, y -> fsub y, x
3196 if (match(Src1, m_SpecificFP(-1.0)))
3197 return BinaryOperator::CreateFSubFMF(Src2, Src0, II);
3198
3199 break;
3200 }
3201 case Intrinsic::copysign: {
3202 Value *Mag = II->getArgOperand(0), *Sign = II->getArgOperand(1);
3203 if (std::optional<bool> KnownSignBit = computeKnownFPSignBit(
3204 Sign, getSimplifyQuery().getWithInstruction(II))) {
3205 if (*KnownSignBit) {
3206 // If we know that the sign argument is negative, reduce to FNABS:
3207 // copysign Mag, -Sign --> fneg (fabs Mag)
3208 Value *Fabs = Builder.CreateFAbs(Mag, II);
3209 return replaceInstUsesWith(*II, Builder.CreateFNegFMF(Fabs, II));
3210 }
3211
3212 // If we know that the sign argument is positive, reduce to FABS:
3213 // copysign Mag, +Sign --> fabs Mag
3214 Value *Fabs = Builder.CreateFAbs(Mag, II);
3215 return replaceInstUsesWith(*II, Fabs);
3216 }
3217
3218 // Propagate sign argument through nested calls:
3219 // copysign Mag, (copysign ?, X) --> copysign Mag, X
3220 Value *X;
3222 Value *CopySign =
3223 Builder.CreateCopySign(Mag, X, FMFSource::intersect(II, Sign));
3224 return replaceInstUsesWith(*II, CopySign);
3225 }
3226
3227 // Clear sign-bit of constant magnitude:
3228 // copysign -MagC, X --> copysign MagC, X
3229 // TODO: Support constant folding for fabs
3230 const APFloat *MagC;
3231 if (match(Mag, m_APFloat(MagC)) && MagC->isNegative()) {
3232 APFloat PosMagC = *MagC;
3233 PosMagC.clearSign();
3234 return replaceInstUsesWith(
3235 *II, Builder.CreateCopySign(ConstantFP::get(Mag->getType(), PosMagC),
3236 Sign, II));
3237 }
3238
3239 // Peek through changes of magnitude's sign-bit. This call rewrites those:
3240 // copysign (fabs X), Sign --> copysign X, Sign
3241 // copysign (fneg X), Sign --> copysign X, Sign
3242 if (match(Mag, m_FAbs(m_Value(X))) || match(Mag, m_FNeg(m_Value(X))))
3243 return replaceInstUsesWith(*II, Builder.CreateCopySign(X, Sign, II));
3244
3245 // copysign(floor(fabs(X)), X) --> copysign(trunc(X), X)
3246 // copysign ignores the sign bit of its magnitude argument (implicit fabs),
3247 // so replacing floor(fabs(X)) with trunc(X) is correct for all inputs
3248 // including NaN without requiring nnan. The m_FAbs match also ensures
3249 // the floor argument is non-negative, so floor == trunc.
3250 Value *FAbsArg;
3251 if (match(Mag, m_Intrinsic<Intrinsic::floor>(m_FAbs(m_Value(FAbsArg)))) &&
3252 FAbsArg == Sign) {
3253 Value *Trunc = Builder.CreateUnaryIntrinsic(Intrinsic::trunc, Sign, II);
3254 return replaceInstUsesWith(*II, Builder.CreateCopySign(Trunc, Sign, II));
3255 }
3256
3257 Type *SignEltTy = Sign->getType()->getScalarType();
3258
3259 Value *CastSrc;
3260 if (match(Sign,
3262 CastSrc->getType()->isIntOrIntVectorTy() &&
3266 APInt::getSignMask(Known.getBitWidth()), Known,
3267 SQ))
3268 return II;
3269 }
3270
3271 break;
3272 }
3273 case Intrinsic::fabs: {
3274 Value *Cond, *TVal, *FVal;
3275 Value *Arg = II->getArgOperand(0);
3276 Value *X;
3277 // fabs (-X) --> fabs (X)
3278 if (match(Arg, m_FNeg(m_Value(X)))) {
3279 Value *Fabs = Builder.CreateFAbs(X, II);
3280 return replaceInstUsesWith(CI, Fabs);
3281 }
3282
3283 if (match(Arg, m_Select(m_Value(Cond), m_Value(TVal), m_Value(FVal)))) {
3284 // fabs (select Cond, TrueC, FalseC) --> select Cond, AbsT, AbsF
3285 if (Arg->hasOneUse() ? (isa<Constant>(TVal) || isa<Constant>(FVal))
3286 : (isa<Constant>(TVal) && isa<Constant>(FVal))) {
3287 CallInst *AbsT = Builder.CreateCall(II->getCalledFunction(), {TVal});
3288 CallInst *AbsF = Builder.CreateCall(II->getCalledFunction(), {FVal});
3289 SelectInst *SI = SelectInst::Create(Cond, AbsT, AbsF);
3290 SI->setFastMathFlags(II->getFastMathFlags() |
3291 cast<SelectInst>(Arg)->getFastMathFlags());
3292 // Can't copy nsz to select, as even with the nsz flag the fabs result
3293 // always has the sign bit unset.
3294 SI->setHasNoSignedZeros(false);
3295 return SI;
3296 }
3297 // fabs (select Cond, -FVal, FVal) --> fabs FVal
3298 if (match(TVal, m_FNeg(m_Specific(FVal))))
3299 return replaceInstUsesWith(*II, Builder.CreateFAbs(FVal, II));
3300 // fabs (select Cond, TVal, -TVal) --> fabs TVal
3301 if (match(FVal, m_FNeg(m_Specific(TVal))))
3302 return replaceInstUsesWith(*II, Builder.CreateFAbs(TVal, II));
3303 }
3304
3305 Value *Magnitude, *Sign;
3306 if (match(II->getArgOperand(0),
3307 m_CopySign(m_Value(Magnitude), m_Value(Sign)))) {
3308 // fabs (copysign x, y) -> (fabs x)
3309 Value *AbsSign = Builder.CreateFAbs(Magnitude, II);
3310 return replaceInstUsesWith(*II, AbsSign);
3311 }
3312
3313 [[fallthrough]];
3314 }
3315 case Intrinsic::ceil:
3316 case Intrinsic::floor:
3317 case Intrinsic::round:
3318 case Intrinsic::roundeven:
3319 case Intrinsic::nearbyint:
3320 case Intrinsic::rint:
3321 case Intrinsic::trunc: {
3322 Value *ExtSrc;
3323 if (match(II->getArgOperand(0), m_OneUse(m_FPExt(m_Value(ExtSrc))))) {
3324 // Narrow the call: intrinsic (fpext x) -> fpext (intrinsic x)
3325 Value *NarrowII = Builder.CreateUnaryIntrinsic(IID, ExtSrc, II);
3326 return new FPExtInst(NarrowII, II->getType());
3327 }
3328 break;
3329 }
3330 case Intrinsic::cos:
3331 case Intrinsic::amdgcn_cos:
3332 case Intrinsic::cosh: {
3333 Value *X, *Sign;
3334 Value *Src = II->getArgOperand(0);
3335 if (match(Src, m_FNeg(m_Value(X))) || match(Src, m_FAbs(m_Value(X))) ||
3336 match(Src, m_CopySign(m_Value(X), m_Value(Sign)))) {
3337 // f(-x) --> f(x)
3338 // f(fabs(x)) --> f(x)
3339 // f(copysign(x, y)) --> f(x)
3340 // for f in {cos, cosh}
3341 return replaceInstUsesWith(*II, Builder.CreateUnaryIntrinsic(IID, X, II));
3342 }
3343 if (IID == Intrinsic::cos) {
3344 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3345 return replaceInstUsesWith(*II, Result);
3346 }
3347 break;
3348 }
3349 case Intrinsic::sin:
3350 case Intrinsic::amdgcn_sin:
3351 case Intrinsic::sinh:
3352 case Intrinsic::tan:
3353 case Intrinsic::tanh: {
3354 Value *X;
3355 if (match(II->getArgOperand(0), m_OneUse(m_FNeg(m_Value(X))))) {
3356 // f(-x) --> -f(x)
3357 // for f in {sin, sinh, tan, tanh}
3358 Value *NewFunc = Builder.CreateUnaryIntrinsic(IID, X, II);
3359 return UnaryOperator::CreateFNegFMF(NewFunc, II);
3360 }
3361 if (IID == Intrinsic::sin) {
3362 if (Value *Result = foldSinAndCosToSinCos(II, Builder, *this))
3363 return replaceInstUsesWith(*II, Result);
3364 }
3365 break;
3366 }
3367 case Intrinsic::ldexp: {
3368 Value *Src = II->getArgOperand(0);
3369 Value *Exp = II->getArgOperand(1);
3370
3371 // ldexp(x, K) -> fmul x, 2^K
3372 uint64_t ConstExp;
3373 if (match(Exp, m_ConstantInt(ConstExp))) {
3374 const fltSemantics &FPTy =
3375 Src->getType()->getScalarType()->getFltSemantics();
3376
3377 APFloat Scaled = scalbn(APFloat::getOne(FPTy), static_cast<int>(ConstExp),
3379 if (!Scaled.isZero() && !Scaled.isInfinity()) {
3380 // Skip overflow and underflow cases.
3381 Constant *FPConst = ConstantFP::get(Src->getType(), Scaled);
3382 return BinaryOperator::CreateFMulFMF(Src, FPConst, II);
3383 }
3384 }
3385
3386 // ldexp(ldexp(x, a), b) -> ldexp(x, sadd.sat(a, b))
3387 //
3388 // A danger is if the first ldexp would overflow to infinity or underflow to
3389 // zero, but the combined exponent avoids it.
3390 //
3391 // We ignore this with reassoc, or if we know both exponents have the same
3392 // sign (since then we'd just double down on the over/underflow which would
3393 // occur anyway).
3394 //
3395 // ldexp can take arbitrary integer types, so we also need to ensure that
3396 // our exponent type is wide enough so that if sadd.sat(a, b) saturates,
3397 // then ldexp at the saturated exponent saturates to inf or zero as well.
3398 //
3399 // TODO: Could do better if we had range tracking for the input value
3400 // exponent. Also could broaden sign check to cover == 0 case.
3401 Value *InnerSrc;
3402 Value *InnerExp;
3404 m_Value(InnerSrc), m_Value(InnerExp)))) &&
3405 Exp->getType() == InnerExp->getType()) {
3406 FastMathFlags FMF = II->getFastMathFlags();
3407 FastMathFlags InnerFlags = cast<FPMathOperator>(Src)->getFastMathFlags();
3408
3409 if (ldexpSaturatingAddIsSafe(II->getType(), Exp->getType()) &&
3410 ((FMF.allowReassoc() && InnerFlags.allowReassoc()) ||
3411 signBitMustBeTheSame(Exp, InnerExp, SQ.getWithInstruction(II)))) {
3412 Value *NewExp =
3413 Builder.CreateBinaryIntrinsic(Intrinsic::sadd_sat, InnerExp, Exp);
3414 return replaceInstUsesWith(
3415 *II, Builder.CreateLdexp(InnerSrc, NewExp, FMF | InnerFlags));
3416 }
3417 }
3418
3419 // ldexp(x, zext(i1 y)) -> fmul x, (select y, 2.0, 1.0)
3420 // ldexp(x, sext(i1 y)) -> fmul x, (select y, 0.5, 1.0)
3421 Value *ExtSrc;
3422 if (match(Exp, m_ZExt(m_Value(ExtSrc))) &&
3423 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3424 Value *Select =
3425 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 2.0),
3426 ConstantFP::get(II->getType(), 1.0));
3428 }
3429 if (match(Exp, m_SExt(m_Value(ExtSrc))) &&
3430 ExtSrc->getType()->getScalarSizeInBits() == 1) {
3431 Value *Select =
3432 Builder.CreateSelect(ExtSrc, ConstantFP::get(II->getType(), 0.5),
3433 ConstantFP::get(II->getType(), 1.0));
3435 }
3436
3437 // ldexp(x, c ? exp : 0) -> c ? ldexp(x, exp) : x
3438 // ldexp(x, c ? 0 : exp) -> c ? x : ldexp(x, exp)
3439 ///
3440 // TODO: If we cared, should insert a canonicalize for x
3441 Value *SelectCond, *SelectLHS, *SelectRHS;
3442 if (match(II->getArgOperand(1),
3443 m_OneUse(m_Select(m_Value(SelectCond), m_Value(SelectLHS),
3444 m_Value(SelectRHS))))) {
3445 Value *NewLdexp = nullptr;
3446 Value *Select = nullptr;
3447 if (match(SelectRHS, m_ZeroInt())) {
3448 NewLdexp = Builder.CreateLdexp(Src, SelectLHS, II);
3449 Select = Builder.CreateSelect(SelectCond, NewLdexp, Src);
3450 } else if (match(SelectLHS, m_ZeroInt())) {
3451 NewLdexp = Builder.CreateLdexp(Src, SelectRHS, II);
3452 Select = Builder.CreateSelect(SelectCond, Src, NewLdexp);
3453 }
3454
3455 if (NewLdexp) {
3456 Select->takeName(II);
3457 return replaceInstUsesWith(*II, Select);
3458 }
3459 }
3460
3461 break;
3462 }
3463 case Intrinsic::ptrauth_auth:
3464 case Intrinsic::ptrauth_resign: {
3465 // (sign|resign) + (auth|resign) can be folded by omitting the middle
3466 // sign+auth component if the key and discriminator match.
3467 bool NeedSign = II->getIntrinsicID() == Intrinsic::ptrauth_resign;
3468 Value *Ptr = II->getArgOperand(0);
3469 Value *Key = II->getArgOperand(1);
3470 Value *Disc = II->getArgOperand(2);
3471 Value *DS = nullptr;
3472 if (auto Bundle = II->getOperandBundle(LLVMContext::OB_deactivation_symbol))
3473 DS = Bundle->Inputs[0];
3474
3475 // AuthKey will be the key we need to end up authenticating against in
3476 // whatever we replace this sequence with.
3477 Value *AuthKey = nullptr, *AuthDisc = nullptr, *BasePtr;
3478 if (const auto *CI = dyn_cast<CallBase>(Ptr)) {
3479 Value *OtherDS = nullptr;
3480 if (auto Bundle =
3482 OtherDS = Bundle->Inputs[0];
3483 if (DS != OtherDS)
3484 break;
3485
3486 if (CI->getIntrinsicID() == Intrinsic::ptrauth_sign) {
3487 if (CI->getArgOperand(1) != Key || CI->getArgOperand(2) != Disc)
3488 break;
3489 } else if (CI->getIntrinsicID() == Intrinsic::ptrauth_resign) {
3490 // The resign intrinsic does not support deactivation symbols.
3491 assert(!DS);
3492 if (CI->getArgOperand(3) != Key || CI->getArgOperand(4) != Disc)
3493 break;
3494 AuthKey = CI->getArgOperand(1);
3495 AuthDisc = CI->getArgOperand(2);
3496 } else
3497 break;
3498 BasePtr = CI->getArgOperand(0);
3499 } else if (const auto *PtrToInt = dyn_cast<PtrToIntOperator>(Ptr)) {
3500 // ptrauth constants are equivalent to a call to @llvm.ptrauth.sign for
3501 // our purposes, so check for that too.
3502 const auto *CPA = dyn_cast<ConstantPtrAuth>(PtrToInt->getOperand(0));
3503 if (!CPA || DS || !CPA->isKnownCompatibleWith(Key, Disc, DL))
3504 break;
3505
3506 // resign(ptrauth(p,ks,ds),ks,ds,kr,dr) -> ptrauth(p,kr,dr)
3507 if (NeedSign && isa<ConstantInt>(II->getArgOperand(4))) {
3508 auto *SignKey = cast<ConstantInt>(II->getArgOperand(3));
3509 auto *SignDisc = cast<ConstantInt>(II->getArgOperand(4));
3510 auto *Null = ConstantPointerNull::get(Builder.getPtrTy());
3511 auto *NewCPA = ConstantPtrAuth::get(CPA->getPointer(), SignKey,
3512 SignDisc, /*AddrDisc=*/Null,
3513 /*DeactivationSymbol=*/Null);
3515 *II, ConstantExpr::getPointerCast(NewCPA, II->getType()));
3516 return eraseInstFromFunction(*II);
3517 }
3518
3519 // auth(ptrauth(p,k,d),k,d) -> p
3520 BasePtr = Builder.CreatePtrToInt(CPA->getPointer(), II->getType());
3521 } else
3522 break;
3523
3524 unsigned NewIntrin;
3525 if (AuthKey && NeedSign) {
3526 // resign(0,1) + resign(1,2) = resign(0, 2)
3527 NewIntrin = Intrinsic::ptrauth_resign;
3528 } else if (AuthKey) {
3529 // resign(0,1) + auth(1) = auth(0)
3530 NewIntrin = Intrinsic::ptrauth_auth;
3531 } else if (NeedSign) {
3532 // sign(0) + resign(0, 1) = sign(1)
3533 NewIntrin = Intrinsic::ptrauth_sign;
3534 } else {
3535 // sign(0) + auth(0) = nop
3536 replaceInstUsesWith(*II, BasePtr);
3537 return eraseInstFromFunction(*II);
3538 }
3539
3540 SmallVector<Value *, 4> CallArgs;
3541 CallArgs.push_back(BasePtr);
3542 if (AuthKey) {
3543 CallArgs.push_back(AuthKey);
3544 CallArgs.push_back(AuthDisc);
3545 }
3546
3547 if (NeedSign) {
3548 CallArgs.push_back(II->getArgOperand(3));
3549 CallArgs.push_back(II->getArgOperand(4));
3550 }
3551
3552 std::vector<OperandBundleDef> Bundles;
3553 if (DS)
3554 Bundles.push_back(OperandBundleDef("deactivation-symbol", DS));
3555
3556 Function *NewFn =
3557 Intrinsic::getOrInsertDeclaration(II->getModule(), NewIntrin);
3558 return CallInst::Create(NewFn, CallArgs, Bundles);
3559 }
3560 case Intrinsic::arm_neon_vtbl1:
3561 case Intrinsic::arm_neon_vtbl2:
3562 case Intrinsic::arm_neon_vtbl3:
3563 case Intrinsic::arm_neon_vtbl4:
3564 case Intrinsic::aarch64_neon_tbl1:
3565 case Intrinsic::aarch64_neon_tbl2:
3566 case Intrinsic::aarch64_neon_tbl3:
3567 case Intrinsic::aarch64_neon_tbl4:
3568 return simplifyNeonTbl(*II, *this, /*IsExtension=*/false);
3569 case Intrinsic::arm_neon_vtbx1:
3570 case Intrinsic::arm_neon_vtbx2:
3571 case Intrinsic::arm_neon_vtbx3:
3572 case Intrinsic::arm_neon_vtbx4:
3573 case Intrinsic::aarch64_neon_tbx1:
3574 case Intrinsic::aarch64_neon_tbx2:
3575 case Intrinsic::aarch64_neon_tbx3:
3576 case Intrinsic::aarch64_neon_tbx4:
3577 return simplifyNeonTbl(*II, *this, /*IsExtension=*/true);
3578
3579 case Intrinsic::arm_neon_vmulls:
3580 case Intrinsic::arm_neon_vmullu:
3581 case Intrinsic::aarch64_neon_smull:
3582 case Intrinsic::aarch64_neon_umull: {
3583 Value *Arg0 = II->getArgOperand(0);
3584 Value *Arg1 = II->getArgOperand(1);
3585
3586 // Handle mul by zero first:
3588 return replaceInstUsesWith(CI, ConstantAggregateZero::get(II->getType()));
3589 }
3590
3591 // Check for constant LHS & RHS - in this case we just simplify.
3592 bool Zext = (IID == Intrinsic::arm_neon_vmullu ||
3593 IID == Intrinsic::aarch64_neon_umull);
3594 VectorType *NewVT = cast<VectorType>(II->getType());
3595 if (Constant *CV0 = dyn_cast<Constant>(Arg0)) {
3596 if (Constant *CV1 = dyn_cast<Constant>(Arg1)) {
3597 Value *V0 = Builder.CreateIntCast(CV0, NewVT, /*isSigned=*/!Zext);
3598 Value *V1 = Builder.CreateIntCast(CV1, NewVT, /*isSigned=*/!Zext);
3599 return replaceInstUsesWith(CI, Builder.CreateMul(V0, V1));
3600 }
3601
3602 // Couldn't simplify - canonicalize constant to the RHS.
3603 std::swap(Arg0, Arg1);
3604 }
3605
3606 // Handle mul by one:
3607 if (Constant *CV1 = dyn_cast<Constant>(Arg1))
3608 if (ConstantInt *Splat =
3609 dyn_cast_or_null<ConstantInt>(CV1->getSplatValue()))
3610 if (Splat->isOne())
3611 return CastInst::CreateIntegerCast(Arg0, II->getType(),
3612 /*isSigned=*/!Zext);
3613
3614 break;
3615 }
3616 case Intrinsic::arm_neon_aesd:
3617 case Intrinsic::arm_neon_aese:
3618 case Intrinsic::aarch64_crypto_aesd:
3619 case Intrinsic::aarch64_crypto_aese:
3620 case Intrinsic::aarch64_sve_aesd:
3621 case Intrinsic::aarch64_sve_aese: {
3622 Value *DataArg = II->getArgOperand(0);
3623 Value *KeyArg = II->getArgOperand(1);
3624
3625 // Accept zero on either operand.
3626 if (!match(KeyArg, m_ZeroInt()))
3627 std::swap(KeyArg, DataArg);
3628
3629 // Try to use the builtin XOR in AESE and AESD to eliminate a prior XOR
3630 Value *Data, *Key;
3631 if (match(KeyArg, m_ZeroInt()) &&
3632 match(DataArg, m_Xor(m_Value(Data), m_Value(Key)))) {
3633 replaceOperand(*II, 0, Data);
3634 replaceOperand(*II, 1, Key);
3635 return II;
3636 }
3637 break;
3638 }
3639 case Intrinsic::arm_neon_vshifts:
3640 case Intrinsic::arm_neon_vshiftu:
3641 case Intrinsic::aarch64_neon_sshl:
3642 case Intrinsic::aarch64_neon_ushl:
3643 return foldNeonShift(II, *this);
3644 case Intrinsic::hexagon_V6_vandvrt:
3645 case Intrinsic::hexagon_V6_vandvrt_128B: {
3646 // Simplify Q -> V -> Q conversion.
3647 if (auto Op0 = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3648 Intrinsic::ID ID0 = Op0->getIntrinsicID();
3649 if (ID0 != Intrinsic::hexagon_V6_vandqrt &&
3650 ID0 != Intrinsic::hexagon_V6_vandqrt_128B)
3651 break;
3652 Value *Bytes = Op0->getArgOperand(1), *Mask = II->getArgOperand(1);
3653 uint64_t Bytes1 = computeKnownBits(Bytes, Op0).One.getZExtValue();
3654 uint64_t Mask1 = computeKnownBits(Mask, II).One.getZExtValue();
3655 // Check if every byte has common bits in Bytes and Mask.
3656 uint64_t C = Bytes1 & Mask1;
3657 if ((C & 0xFF) && (C & 0xFF00) && (C & 0xFF0000) && (C & 0xFF000000))
3658 return replaceInstUsesWith(*II, Op0->getArgOperand(0));
3659 }
3660 break;
3661 }
3662 case Intrinsic::stackrestore: {
3663 enum class ClassifyResult {
3664 None,
3665 Alloca,
3666 StackRestore,
3667 CallWithSideEffects,
3668 };
3669 auto Classify = [](const Instruction *I) {
3670 if (isa<AllocaInst>(I))
3671 return ClassifyResult::Alloca;
3672
3673 if (auto *CI = dyn_cast<CallInst>(I)) {
3674 if (auto *II = dyn_cast<IntrinsicInst>(CI)) {
3675 if (II->getIntrinsicID() == Intrinsic::stackrestore)
3676 return ClassifyResult::StackRestore;
3677
3678 if (II->mayHaveSideEffects())
3679 return ClassifyResult::CallWithSideEffects;
3680 } else {
3681 // Consider all non-intrinsic calls to be side effects
3682 return ClassifyResult::CallWithSideEffects;
3683 }
3684 }
3685
3686 return ClassifyResult::None;
3687 };
3688
3689 // If the stacksave and the stackrestore are in the same BB, and there is
3690 // no intervening call, alloca, or stackrestore of a different stacksave,
3691 // remove the restore. This can happen when variable allocas are DCE'd.
3692 if (IntrinsicInst *SS = dyn_cast<IntrinsicInst>(II->getArgOperand(0))) {
3693 if (SS->getIntrinsicID() == Intrinsic::stacksave &&
3694 SS->getParent() == II->getParent()) {
3695 BasicBlock::iterator BI(SS);
3696 bool CannotRemove = false;
3697 for (++BI; &*BI != II; ++BI) {
3698 switch (Classify(&*BI)) {
3699 case ClassifyResult::None:
3700 // So far so good, look at next instructions.
3701 break;
3702
3703 case ClassifyResult::StackRestore:
3704 // If we found an intervening stackrestore for a different
3705 // stacksave, we can't remove the stackrestore. Otherwise, continue.
3706 if (cast<IntrinsicInst>(*BI).getArgOperand(0) != SS)
3707 CannotRemove = true;
3708 break;
3709
3710 case ClassifyResult::Alloca:
3711 case ClassifyResult::CallWithSideEffects:
3712 // If we found an alloca, a non-intrinsic call, or an intrinsic
3713 // call with side effects, we can't remove the stackrestore.
3714 CannotRemove = true;
3715 break;
3716 }
3717 if (CannotRemove)
3718 break;
3719 }
3720
3721 if (!CannotRemove)
3722 return eraseInstFromFunction(CI);
3723 }
3724 }
3725
3726 // Scan down this block to see if there is another stack restore in the
3727 // same block without an intervening call/alloca.
3729 Instruction *TI = II->getParent()->getTerminator();
3730 bool CannotRemove = false;
3731 for (++BI; &*BI != TI; ++BI) {
3732 switch (Classify(&*BI)) {
3733 case ClassifyResult::None:
3734 // So far so good, look at next instructions.
3735 break;
3736
3737 case ClassifyResult::StackRestore:
3738 // If there is a stackrestore below this one, remove this one.
3739 return eraseInstFromFunction(CI);
3740
3741 case ClassifyResult::Alloca:
3742 case ClassifyResult::CallWithSideEffects:
3743 // If we found an alloca, a non-intrinsic call, or an intrinsic call
3744 // with side effects (such as llvm.stacksave and llvm.read_register),
3745 // we can't remove the stack restore.
3746 CannotRemove = true;
3747 break;
3748 }
3749 if (CannotRemove)
3750 break;
3751 }
3752
3753 // If the stack restore is in a return, resume, or unwind block and if there
3754 // are no allocas or calls between the restore and the return, nuke the
3755 // restore.
3756 if (!CannotRemove && (isa<ReturnInst>(TI) || isa<ResumeInst>(TI)))
3757 return eraseInstFromFunction(CI);
3758 break;
3759 }
3760 case Intrinsic::lifetime_end:
3761 // Asan needs to poison memory to detect invalid access which is possible
3762 // even for empty lifetime range.
3763 if (II->getFunction()->hasFnAttribute(Attribute::SanitizeAddress) ||
3764 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemory) ||
3765 II->getFunction()->hasFnAttribute(Attribute::SanitizeHWAddress) ||
3766 II->getFunction()->hasFnAttribute(Attribute::SanitizeMemTag))
3767 break;
3768
3769 if (removeTriviallyEmptyRange(*II, *this, [](const IntrinsicInst &I) {
3770 return I.getIntrinsicID() == Intrinsic::lifetime_start;
3771 }))
3772 return nullptr;
3773 break;
3774 case Intrinsic::assume: {
3775 for (auto [Idx, OBU] : llvm::enumerate(II->operand_bundles())) {
3776 auto RemoveBundle = [&, Idx = Idx]() -> Instruction * {
3777 if (II->getNumOperandBundles() == 1)
3778 return eraseInstFromFunction(*II);
3780 };
3781
3782 switch (getBundleAttrFromOBU(OBU)) {
3783 case BundleAttr::None:
3784 llvm_unreachable("Unexpected Attribute");
3785 case BundleAttr::Align: {
3786 // Try to remove redundant alignment assumptions.
3787 auto [Ptr, _, OffsetPtr, Alignment, Offset] = getAssumeAlignInfo(OBU);
3788
3789 if (!Alignment)
3790 break;
3791
3792 // Remove align 1 and non-power-of-two bundles; they don't add any
3793 // useful information.
3794 if (*Alignment == 1 || !isPowerOf2_64(*Alignment))
3795 return RemoveBundle();
3796
3797 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3798 GEP &&
3799 GEP->getMaxPreservedAlignment(getDataLayout()) >= *Alignment) {
3800 Builder.CreateAlignmentAssumption(
3801 getDataLayout(), GEP->getPointerOperand(), *Alignment,
3802 OffsetPtr ? const_cast<Value *>(OffsetPtr->get()) : nullptr);
3803 return RemoveBundle();
3804 }
3805
3806 if (!Offset)
3807 break;
3808
3809 Value *BasePtr;
3810 const APInt *PtrOffset;
3811 if (match(Ptr.get(), m_PtrAdd(m_Value(BasePtr), m_APInt(PtrOffset)))) {
3812 auto PtrOffsetVal =
3813 PtrOffset->sextOrTrunc(DL.getIndexTypeSizeInBits(Ptr->getType()))
3814 .trySExtValue();
3815 if (!PtrOffsetVal)
3816 break;
3817 Builder.CreateAlignmentAssumption(
3818 DL, BasePtr, *Alignment,
3819 Builder.getInt64(*Offset - *PtrOffsetVal));
3820 return RemoveBundle();
3821 }
3822
3823 // Don't try to remove align assumptions for pointers derived from
3824 // arguments. We might lose information if the function gets inline and
3825 // the align argument attribute disappears.
3826 Value *UO = getUnderlyingObject(Ptr);
3827 if (!UO || isa<Argument>(UO))
3828 break;
3829
3830 // Compute known bits for the pointer and drop the assume if the
3831 // known alignment isn't increased by it.
3832 auto AlignMask = (*Alignment - 1);
3833 if (KnownBits KB = computeKnownBits(Ptr, II);
3834 (KB.Zero & AlignMask) == (~*Offset & AlignMask) &&
3835 (KB.One & AlignMask) == (*Offset & AlignMask))
3836 return RemoveBundle();
3837 break;
3838 }
3839
3840 case BundleAttr::Dereferenceable: {
3841 auto [Ptr, _, Count] = getAssumeDereferenceableInfo(OBU);
3842
3843 if (!Count)
3844 break;
3845
3846 if (*Count == 0 ||
3848 getSimplifyQuery().getWithInstruction(II)))
3849 return RemoveBundle();
3850
3851 break;
3852 }
3853
3854 case BundleAttr::Ignore:
3855 return RemoveBundle();
3856
3857 case BundleAttr::NonNull: {
3858 auto [Ptr] = llvm::getAssumeNonNullInfo(OBU);
3859
3860 // Drop assume if we can prove nonnull without it
3861 if (isKnownNonZero(Ptr, getSimplifyQuery().getWithInstruction(II)))
3862 return RemoveBundle();
3863
3864 // Fold the assume into metadata if it's valid at the load
3865 if (auto *LI = dyn_cast<LoadInst>(Ptr);
3866 LI &&
3867 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3868 MDNode *MD = MDNode::get(II->getContext(), {});
3869 LI->setMetadata(LLVMContext::MD_nonnull, MD);
3870 LI->setMetadata(LLVMContext::MD_noundef, MD);
3871 return RemoveBundle();
3872 }
3873
3874 if (auto *GEP = dyn_cast<GEPOperator>(Ptr);
3875 GEP && GEP->isInBounds() &&
3876 !NullPointerIsDefined(II->getFunction(),
3877 Ptr->getType()->getPointerAddressSpace())) {
3878 Builder.CreateNonnullAssumption(GEP->stripInBoundsOffsets());
3879 return RemoveBundle();
3880 }
3881
3882 // TODO: apply nonnull return attributes to calls and invokes
3883 break;
3884 }
3885
3886 case BundleAttr::NoUndef: {
3887 auto [Val] = getAssumeNoUndefInfo(OBU);
3888
3890 return RemoveBundle();
3891
3892 if (auto *LI = dyn_cast<LoadInst>(Val);
3893 LI &&
3894 isValidAssumeForContext(II, LI, &DT, /*AllowEphemerals=*/true)) {
3895 LI->setMetadata(LLVMContext::MD_noundef,
3896 MDNode::get(II->getContext(), {}));
3897 return RemoveBundle();
3898 }
3899
3900 } break;
3901
3902 case BundleAttr::SeparateStorage: {
3903 auto [Ptr1, Ptr2] = getAssumeSeparateStorageInfo(OBU);
3904 // Separate storage assumptions apply to the underlying allocations, not
3905 // any particular pointer within them. When evaluating the hints for AA
3906 // purposes we getUnderlyingObject them; by precomputing the answers
3907 // here we can avoid having to do so repeatedly there.
3908 auto MaybeSimplifyHint = [&](const Use &U) {
3909 Value *Hint = U.get();
3910 // Not having a limit is safe because InstCombine removes unreachable
3911 // code.
3912 Value *UnderlyingObject = getUnderlyingObject(Hint, /*MaxLookup*/ 0);
3913 if (Hint != UnderlyingObject)
3914 replaceUse(const_cast<Use &>(U), UnderlyingObject);
3915 };
3916 MaybeSimplifyHint(Ptr1);
3917 MaybeSimplifyHint(Ptr2);
3918 } break;
3919
3920 // TODO: Drop these assumes when they are redundant
3921 case BundleAttr::DereferenceableOrNull:
3922 break;
3923
3924 // This cannot be simplified
3925 case BundleAttr::Cold:
3926 break;
3927 }
3928 }
3929
3930 // If the assume has operand bundles, the folds below will never work, so
3931 // don't bother trying.
3932 if (II->hasOperandBundles())
3933 break;
3934
3935 Value *IIOperand = II->getArgOperand(0);
3936
3937 // Canonicalize assume(a && b) -> assume(a); assume(b);
3938 // Note: New assumption intrinsics created here are registered by
3939 // the InstCombineIRInserter object.
3940 Value *A, *B;
3941 if (match(IIOperand, m_LogicalAnd(m_Value(A), m_Value(B)))) {
3942 Builder.CreateAssumption(A);
3943 Builder.CreateAssumption(B);
3944 return eraseInstFromFunction(*II);
3945 }
3946 // assume(!(a || b)) -> assume(!a); assume(!b);
3947 if (match(IIOperand, m_Not(m_LogicalOr(m_Value(A), m_Value(B))))) {
3948 Builder.CreateAssumption(Builder.CreateNot(A));
3949 Builder.CreateAssumption(Builder.CreateNot(B));
3950 return eraseInstFromFunction(*II);
3951 }
3952
3953 // Convert nonnull assume like:
3954 // %A = icmp ne i32* %PTR, null
3955 // call void @llvm.assume(i1 %A)
3956 // into
3957 // call void @llvm.assume(i1 true) [ "nonnull"(i32* %PTR) ]
3958 if (match(IIOperand,
3960 A->getType()->isPointerTy()) {
3961 Builder.CreateNonnullAssumption(A);
3962 return eraseInstFromFunction(*II);
3963 }
3964
3965 // Convert alignment assume like:
3966 // %B = ptrtoint ptr %A to i64
3967 // %C = and i64 %B, Constant
3968 // %D = icmp eq i64 %C, 0
3969 // call void @llvm.assume(i1 %D)
3970 // into
3971 // call void @llvm.assume(i1 true) [ "align"(ptr [[A]], i64 Constant + 1)]
3972 uint64_t AlignMask = 1;
3973 if ((match(IIOperand, m_Not(m_Trunc(m_Value(A)))) ||
3974 match(IIOperand,
3976 m_And(m_Value(A), m_ConstantInt(AlignMask)),
3977 m_Zero())))) {
3978 if (isPowerOf2_64(AlignMask + 1) &&
3980 Builder.CreateAlignmentAssumption(getDataLayout(), A, AlignMask + 1);
3981 return eraseInstFromFunction(*II);
3982 }
3983 }
3984
3985 // Remove assumes on true/false
3986 if (auto *CI = dyn_cast<ConstantInt>(IIOperand);
3987 CI || isa<UndefValue, PoisonValue>(IIOperand)) {
3988 if (!CI || CI->isZero())
3990 return eraseInstFromFunction(*II);
3991 }
3992
3993 // Update the cache of affected values for this assumption (we might be
3994 // here because we just simplified the condition).
3995 AC.updateAffectedValues(cast<AssumeInst>(II));
3996 break;
3997 }
3998 case Intrinsic::experimental_guard: {
3999 // Is this guard followed by another guard? We scan forward over a small
4000 // fixed window of instructions to handle common cases with conditions
4001 // computed between guards.
4002 Instruction *NextInst = II->getNextNode();
4003 for (unsigned i = 0; i < GuardWideningWindow; i++) {
4004 // Note: Using context-free form to avoid compile time blow up
4005 if (!isSafeToSpeculativelyExecute(NextInst))
4006 break;
4007 NextInst = NextInst->getNextNode();
4008 }
4009 Value *NextCond = nullptr;
4010 if (match(NextInst,
4012 Value *CurrCond = II->getArgOperand(0);
4013
4014 // Remove a guard that it is immediately preceded by an identical guard.
4015 // Otherwise canonicalize guard(a); guard(b) -> guard(a & b).
4016 if (CurrCond != NextCond) {
4017 Instruction *MoveI = II->getNextNode();
4018 while (MoveI != NextInst) {
4019 auto *Temp = MoveI;
4020 MoveI = MoveI->getNextNode();
4021 Temp->moveBefore(II->getIterator());
4022 }
4023 replaceOperand(*II, 0, Builder.CreateAnd(CurrCond, NextCond));
4024 }
4025 eraseInstFromFunction(*NextInst);
4026 return II;
4027 }
4028 break;
4029 }
4030 case Intrinsic::vector_insert: {
4031 Value *Vec = II->getArgOperand(0);
4032 Value *SubVec = II->getArgOperand(1);
4033 Value *Idx = II->getArgOperand(2);
4034 auto *DstTy = dyn_cast<FixedVectorType>(II->getType());
4035 auto *VecTy = dyn_cast<FixedVectorType>(Vec->getType());
4036 auto *SubVecTy = dyn_cast<FixedVectorType>(SubVec->getType());
4037
4038 // Only canonicalize if the destination vector, Vec, and SubVec are all
4039 // fixed vectors.
4040 if (DstTy && VecTy && SubVecTy) {
4041 unsigned DstNumElts = DstTy->getNumElements();
4042 unsigned VecNumElts = VecTy->getNumElements();
4043 unsigned SubVecNumElts = SubVecTy->getNumElements();
4044 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4045
4046 // An insert that entirely overwrites Vec with SubVec is a nop.
4047 if (VecNumElts == SubVecNumElts)
4048 return replaceInstUsesWith(CI, SubVec);
4049
4050 // Widen SubVec into a vector of the same width as Vec, since
4051 // shufflevector requires the two input vectors to be the same width.
4052 // Elements beyond the bounds of SubVec within the widened vector are
4053 // undefined.
4054 SmallVector<int, 8> WidenMask;
4055 unsigned i;
4056 for (i = 0; i != SubVecNumElts; ++i)
4057 WidenMask.push_back(i);
4058 for (; i != VecNumElts; ++i)
4059 WidenMask.push_back(PoisonMaskElem);
4060
4061 Value *WidenShuffle = Builder.CreateShuffleVector(SubVec, WidenMask);
4062
4064 for (unsigned i = 0; i != IdxN; ++i)
4065 Mask.push_back(i);
4066 for (unsigned i = DstNumElts; i != DstNumElts + SubVecNumElts; ++i)
4067 Mask.push_back(i);
4068 for (unsigned i = IdxN + SubVecNumElts; i != DstNumElts; ++i)
4069 Mask.push_back(i);
4070
4071 Value *Shuffle = Builder.CreateShuffleVector(Vec, WidenShuffle, Mask);
4072 return replaceInstUsesWith(CI, Shuffle);
4073 }
4074 break;
4075 }
4076 case Intrinsic::vector_extract: {
4077 Value *Vec = II->getArgOperand(0);
4078 Value *Idx = II->getArgOperand(1);
4079
4080 Type *ReturnType = II->getType();
4081 // (extract_vector (insert_vector InsertTuple, InsertValue, InsertIdx),
4082 // ExtractIdx)
4083 unsigned ExtractIdx = cast<ConstantInt>(Idx)->getZExtValue();
4084 Value *InsertTuple, *InsertIdx, *InsertValue;
4086 m_Value(InsertValue),
4087 m_Value(InsertIdx))) &&
4088 InsertValue->getType() == ReturnType) {
4089 unsigned Index = cast<ConstantInt>(InsertIdx)->getZExtValue();
4090 // Case where we get the same index right after setting it.
4091 // extract.vector(insert.vector(InsertTuple, InsertValue, Idx), Idx) -->
4092 // InsertValue
4093 if (ExtractIdx == Index)
4094 return replaceInstUsesWith(CI, InsertValue);
4095 // If we are getting a different index than what was set in the
4096 // insert.vector intrinsic. We can just set the input tuple to the one up
4097 // in the chain. extract.vector(insert.vector(InsertTuple, InsertValue,
4098 // InsertIndex), ExtractIndex)
4099 // --> extract.vector(InsertTuple, ExtractIndex)
4100 else
4101 return replaceOperand(CI, 0, InsertTuple);
4102 }
4103
4104 ConstantInt *ALMUpperBound;
4106 m_Value(), m_ConstantInt(ALMUpperBound)))) {
4107 const auto &Attrs = II->getFunction()->getAttributes().getFnAttrs();
4108 unsigned VScaleMin = Attrs.getVScaleRangeMin();
4109 unsigned ScaleFactor =
4110 cast<VectorType>(ReturnType)->isScalableTy() ? VScaleMin : 1;
4111 if (ExtractIdx * ScaleFactor >= ALMUpperBound->getZExtValue())
4112 return replaceInstUsesWith(CI,
4113 ConstantVector::getNullValue(ReturnType));
4114 }
4115
4116 auto *DstTy = dyn_cast<VectorType>(ReturnType);
4117 auto *VecTy = dyn_cast<VectorType>(Vec->getType());
4118
4119 if (DstTy && VecTy) {
4120 auto DstEltCnt = DstTy->getElementCount();
4121 auto VecEltCnt = VecTy->getElementCount();
4122 unsigned IdxN = cast<ConstantInt>(Idx)->getZExtValue();
4123
4124 // Extracting the entirety of Vec is a nop.
4125 if (DstEltCnt == VecTy->getElementCount()) {
4126 replaceInstUsesWith(CI, Vec);
4127 return eraseInstFromFunction(CI);
4128 }
4129
4130 // Only canonicalize to shufflevector if the destination vector and
4131 // Vec are fixed vectors.
4132 if (VecEltCnt.isScalable() || DstEltCnt.isScalable())
4133 break;
4134
4136 for (unsigned i = 0; i != DstEltCnt.getKnownMinValue(); ++i)
4137 Mask.push_back(IdxN + i);
4138
4139 Value *Shuffle = Builder.CreateShuffleVector(Vec, Mask);
4140 return replaceInstUsesWith(CI, Shuffle);
4141 }
4142 break;
4143 }
4144 case Intrinsic::experimental_vp_reverse: {
4145 Value *X;
4146 Value *Vec = II->getArgOperand(0);
4147 Value *Mask = II->getArgOperand(1);
4148 if (!match(Mask, m_AllOnes()))
4149 break;
4150 Value *EVL = II->getArgOperand(2);
4151 // TODO: Canonicalize experimental.vp.reverse after unop/binops?
4152 // rev(unop rev(X)) --> unop X
4153 if (match(Vec,
4155 m_Value(X), m_AllOnes(), m_Specific(EVL)))))) {
4156 auto *OldUnOp = cast<UnaryOperator>(Vec);
4158 OldUnOp->getOpcode(), X, OldUnOp, OldUnOp->getName(),
4159 II->getIterator());
4160 return replaceInstUsesWith(CI, NewUnOp);
4161 }
4162 break;
4163 }
4164 case Intrinsic::vector_reduce_or:
4165 case Intrinsic::vector_reduce_and: {
4166 // Canonicalize logical or/and reductions:
4167 // Or reduction for i1 is represented as:
4168 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4169 // %res = cmp ne iReduxWidth %val, 0
4170 // And reduction for i1 is represented as:
4171 // %val = bitcast <ReduxWidth x i1> to iReduxWidth
4172 // %res = cmp eq iReduxWidth %val, 11111
4173 Value *Arg = II->getArgOperand(0);
4174 Value *Vect;
4175
4176 if (Value *NewOp =
4177 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4178 replaceUse(II->getOperandUse(0), NewOp);
4179 return II;
4180 }
4181
4182 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4183 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4184 if (FTy->getElementType() == Builder.getInt1Ty()) {
4185 Value *Res = Builder.CreateBitCast(
4186 Vect, Builder.getIntNTy(FTy->getNumElements()));
4187 if (IID == Intrinsic::vector_reduce_and) {
4188 Res = Builder.CreateICmpEQ(
4190 } else {
4191 assert(IID == Intrinsic::vector_reduce_or &&
4192 "Expected or reduction.");
4193 Res = Builder.CreateIsNotNull(Res);
4194 }
4195 if (Arg != Vect)
4196 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4197 II->getType());
4198 return replaceInstUsesWith(CI, Res);
4199 }
4200 }
4201 [[fallthrough]];
4202 }
4203 case Intrinsic::vector_reduce_add: {
4204 if (IID == Intrinsic::vector_reduce_add) {
4205 // Convert vector_reduce_add(ZExt(<n x i1>)) to
4206 // ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4207 // Convert vector_reduce_add(SExt(<n x i1>)) to
4208 // -ZExtOrTrunc(ctpop(bitcast <n x i1> to in)).
4209 // Convert vector_reduce_add(<n x i1>) to
4210 // Trunc(ctpop(bitcast <n x i1> to in)).
4211 Value *Arg = II->getArgOperand(0);
4212 Value *Vect;
4213
4214 if (Value *NewOp =
4215 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4216 replaceUse(II->getOperandUse(0), NewOp);
4217 return II;
4218 }
4219
4220 // vector.reduce.add.vNiM(splat(%x)) -> mul(%x, N)
4221 if (Value *Splat = getSplatValue(Arg)) {
4222 ElementCount VecToReduceCount =
4223 cast<VectorType>(Arg->getType())->getElementCount();
4224 if (VecToReduceCount.isFixed()) {
4225 unsigned VectorSize = VecToReduceCount.getFixedValue();
4226 return BinaryOperator::CreateMul(
4227 Splat,
4228 ConstantInt::get(Splat->getType(), VectorSize, /*IsSigned=*/false,
4229 /*ImplicitTrunc=*/true));
4230 }
4231 }
4232
4233 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4234 if (auto *FTy = dyn_cast<FixedVectorType>(Vect->getType()))
4235 if (FTy->getElementType() == Builder.getInt1Ty()) {
4236 Value *V = Builder.CreateBitCast(
4237 Vect, Builder.getIntNTy(FTy->getNumElements()));
4238 Value *Res = Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, V);
4239 Res = Builder.CreateZExtOrTrunc(Res, II->getType());
4240 if (Arg != Vect &&
4241 cast<Instruction>(Arg)->getOpcode() == Instruction::SExt)
4242 Res = Builder.CreateNeg(Res);
4243 return replaceInstUsesWith(CI, Res);
4244 }
4245 }
4246 }
4247 [[fallthrough]];
4248 }
4249 case Intrinsic::vector_reduce_xor: {
4250 if (IID == Intrinsic::vector_reduce_xor) {
4251 // Exclusive disjunction reduction over the vector with
4252 // (potentially-extended) i1 element type is actually a
4253 // (potentially-extended) arithmetic `add` reduction over the original
4254 // non-extended value:
4255 // vector_reduce_xor(?ext(<n x i1>))
4256 // -->
4257 // ?ext(vector_reduce_add(<n x i1>))
4258 Value *Arg = II->getArgOperand(0);
4259 Value *Vect;
4260
4261 if (Value *NewOp =
4262 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4263 replaceUse(II->getOperandUse(0), NewOp);
4264 return II;
4265 }
4266
4267 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4268 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4269 if (VTy->getElementType() == Builder.getInt1Ty()) {
4270 Value *Res = Builder.CreateAddReduce(Vect);
4271 if (Arg != Vect)
4272 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4273 II->getType());
4274 return replaceInstUsesWith(CI, Res);
4275 }
4276 }
4277 }
4278 [[fallthrough]];
4279 }
4280 case Intrinsic::vector_reduce_mul: {
4281 if (IID == Intrinsic::vector_reduce_mul) {
4282 Value *Arg = II->getArgOperand(0);
4283
4284 if (Value *NewOp =
4285 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4286 replaceUse(II->getOperandUse(0), NewOp);
4287 return II;
4288 }
4289
4290 // vector_reduce_mul(zext(<n x i1>)), or
4291 // vector_reduce_mul(sext(<n x i1>)) (if n is even) -->
4292 // zext(vector_reduce_and(<n x i1>)).
4293 // (The sext case doesn't work if n is odd because multiplying an odd
4294 // number of -1's produces -1, not 1.)
4295 Value *Vect;
4296 bool IsZext = match(Arg, m_ZExt(m_Value(Vect))) &&
4297 Vect->getType()->isIntOrIntVectorTy(1);
4298 bool IsSext =
4299 match(Arg, m_SExt(m_Value(Vect))) &&
4300 Vect->getType()->isIntOrIntVectorTy(1) &&
4301 cast<VectorType>(Vect->getType())->getElementCount().isKnownEven();
4302 if (IsZext || IsSext) {
4303 Value *Res = Builder.CreateAndReduce(Vect);
4304 return CastInst::Create(Instruction::ZExt, Res, II->getType());
4305 }
4306
4307 // vector_reduce_mul(<n x i1>) --> vector_reduce_and(<n x i1>)
4308 if (Arg->getType()->isIntOrIntVectorTy(1))
4309 return replaceInstUsesWith(CI, Builder.CreateAndReduce(Arg));
4310 }
4311 [[fallthrough]];
4312 }
4313 case Intrinsic::vector_reduce_umin:
4314 case Intrinsic::vector_reduce_umax: {
4315 if (IID == Intrinsic::vector_reduce_umin ||
4316 IID == Intrinsic::vector_reduce_umax) {
4317 // UMin/UMax reduction over the vector with (potentially-extended)
4318 // i1 element type is actually a (potentially-extended)
4319 // logical `and`/`or` reduction over the original non-extended value:
4320 // vector_reduce_u{min,max}(?ext(<n x i1>))
4321 // -->
4322 // ?ext(vector_reduce_{and,or}(<n x i1>))
4323 Value *Arg = II->getArgOperand(0);
4324 Value *Vect;
4325
4326 if (Value *NewOp =
4327 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4328 replaceUse(II->getOperandUse(0), NewOp);
4329 return II;
4330 }
4331
4332 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4333 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4334 if (VTy->getElementType() == Builder.getInt1Ty()) {
4335 Value *Res = IID == Intrinsic::vector_reduce_umin
4336 ? Builder.CreateAndReduce(Vect)
4337 : Builder.CreateOrReduce(Vect);
4338 if (Arg != Vect)
4339 Res = Builder.CreateCast(cast<CastInst>(Arg)->getOpcode(), Res,
4340 II->getType());
4341 return replaceInstUsesWith(CI, Res);
4342 }
4343 }
4344 }
4345 [[fallthrough]];
4346 }
4347 case Intrinsic::vector_reduce_smin:
4348 case Intrinsic::vector_reduce_smax: {
4349 if (IID == Intrinsic::vector_reduce_smin ||
4350 IID == Intrinsic::vector_reduce_smax) {
4351 // SMin/SMax reduction over the vector with (potentially-extended)
4352 // i1 element type is actually a (potentially-extended)
4353 // logical `and`/`or` reduction over the original non-extended value:
4354 // vector_reduce_s{min,max}(<n x i1>)
4355 // -->
4356 // vector_reduce_{or,and}(<n x i1>)
4357 // and
4358 // vector_reduce_s{min,max}(sext(<n x i1>))
4359 // -->
4360 // sext(vector_reduce_{or,and}(<n x i1>))
4361 // and
4362 // vector_reduce_s{min,max}(zext(<n x i1>))
4363 // -->
4364 // zext(vector_reduce_{and,or}(<n x i1>))
4365 Value *Arg = II->getArgOperand(0);
4366 Value *Vect;
4367
4368 if (Value *NewOp =
4369 simplifyReductionOperand(Arg, /*CanReorderLanes=*/true)) {
4370 replaceUse(II->getOperandUse(0), NewOp);
4371 return II;
4372 }
4373
4374 if (match(Arg, m_ZExtOrSExtOrSelf(m_Value(Vect)))) {
4375 if (auto *VTy = dyn_cast<VectorType>(Vect->getType()))
4376 if (VTy->getElementType() == Builder.getInt1Ty()) {
4377 Instruction::CastOps ExtOpc = Instruction::CastOps::CastOpsEnd;
4378 if (Arg != Vect)
4379 ExtOpc = cast<CastInst>(Arg)->getOpcode();
4380 Value *Res = ((IID == Intrinsic::vector_reduce_smin) ==
4381 (ExtOpc == Instruction::CastOps::ZExt))
4382 ? Builder.CreateAndReduce(Vect)
4383 : Builder.CreateOrReduce(Vect);
4384 if (Arg != Vect)
4385 Res = Builder.CreateCast(ExtOpc, Res, II->getType());
4386 return replaceInstUsesWith(CI, Res);
4387 }
4388 }
4389 }
4390 [[fallthrough]];
4391 }
4392 case Intrinsic::vector_reduce_fmax:
4393 case Intrinsic::vector_reduce_fmin:
4394 case Intrinsic::vector_reduce_fadd:
4395 case Intrinsic::vector_reduce_fmul: {
4396 bool CanReorderLanes = (IID != Intrinsic::vector_reduce_fadd &&
4397 IID != Intrinsic::vector_reduce_fmul) ||
4398 II->hasAllowReassoc();
4399 const unsigned ArgIdx = (IID == Intrinsic::vector_reduce_fadd ||
4400 IID == Intrinsic::vector_reduce_fmul)
4401 ? 1
4402 : 0;
4403 Value *Arg = II->getArgOperand(ArgIdx);
4404 if (Value *NewOp = simplifyReductionOperand(Arg, CanReorderLanes)) {
4405 replaceUse(II->getOperandUse(ArgIdx), NewOp);
4406 return nullptr;
4407 }
4408 break;
4409 }
4410 case Intrinsic::is_fpclass: {
4411 if (Instruction *I = foldIntrinsicIsFPClass(*II))
4412 return I;
4413 break;
4414 }
4415 case Intrinsic::threadlocal_address: {
4416 Align MinAlign = getKnownAlignment(II->getArgOperand(0), DL, II, &AC, &DT);
4417 MaybeAlign Align = II->getRetAlign();
4418 if (MinAlign > Align.valueOrOne()) {
4419 II->addRetAttr(Attribute::getWithAlignment(II->getContext(), MinAlign));
4420 return II;
4421 }
4422 break;
4423 }
4424 case Intrinsic::fptoui_sat:
4425 case Intrinsic::fptosi_sat:
4426 if (Instruction *I = foldItoFPtoI(*II))
4427 return I;
4428 break;
4429 case Intrinsic::frexp: {
4430 // frexp(frexp(x).fract) -> { frexp(x).fract, 0 }: the fraction operand is
4431 // already normalized, so the first result is idempotent and the second is
4432 // zero.
4433 if (match(II->getArgOperand(0),
4435 Value *Res = Builder.CreateInsertValue(PoisonValue::get(II->getType()),
4436 II->getArgOperand(0), 0);
4437 Res = Builder.CreateInsertValue(
4438 Res, Constant::getNullValue(II->getType()->getStructElementType(1)),
4439 1);
4440 return replaceInstUsesWith(*II, Res);
4441 }
4442 break;
4443 }
4444 case Intrinsic::get_active_lane_mask: {
4445 const APInt *Op0, *Op1;
4446 if (match(II->getOperand(0), m_StrictlyPositive(Op0)) &&
4447 match(II->getOperand(1), m_APInt(Op1))) {
4448 Type *OpTy = II->getOperand(0)->getType();
4449 return replaceInstUsesWith(
4450 *II, Builder.CreateIntrinsic(
4451 II->getType(), Intrinsic::get_active_lane_mask,
4452 {Constant::getNullValue(OpTy),
4453 ConstantInt::get(OpTy, Op1->usub_sat(*Op0))}));
4454 }
4455 break;
4456 }
4457 case Intrinsic::experimental_get_vector_length: {
4458 // get.vector.length(Cnt, MaxLanes) --> Cnt when Cnt <= MaxLanes
4459 unsigned BitWidth =
4460 std::max(II->getArgOperand(0)->getType()->getScalarSizeInBits(),
4461 II->getType()->getScalarSizeInBits());
4462 ConstantRange Cnt =
4463 computeConstantRangeIncludingKnownBits(II->getArgOperand(0), false,
4464 SQ.getWithInstruction(II))
4466 ConstantRange MaxLanes = cast<ConstantInt>(II->getArgOperand(1))
4467 ->getValue()
4468 .zextOrTrunc(Cnt.getBitWidth());
4469 if (cast<ConstantInt>(II->getArgOperand(2))->isOne())
4470 MaxLanes = MaxLanes.multiply(
4471 getVScaleRange(II->getFunction(), Cnt.getBitWidth()));
4472
4473 if (Cnt.icmp(CmpInst::ICMP_ULE, MaxLanes))
4474 return replaceInstUsesWith(
4475 *II, Builder.CreateZExtOrTrunc(II->getArgOperand(0), II->getType()));
4476 return nullptr;
4477 }
4478 default: {
4479 // Handle target specific intrinsics
4480 std::optional<Instruction *> V = targetInstCombineIntrinsic(*II);
4481 if (V)
4482 return *V;
4483 break;
4484 }
4485 }
4486
4487 // Try to fold intrinsic into select/phi operands. This is legal if:
4488 // * The intrinsic is speculatable.
4489 // * The operand is one of the following:
4490 // - a phi.
4491 // - a select with a scalar condition.
4492 // - a select with a vector condition and II is not a cross lane operation.
4494 for (Value *Op : II->args()) {
4495 if (auto *Sel = dyn_cast<SelectInst>(Op)) {
4496 bool IsVectorCond = Sel->getCondition()->getType()->isVectorTy();
4497 if (IsVectorCond &&
4498 (!isNotCrossLaneOperation(II) || !II->getType()->isVectorTy()))
4499 continue;
4500 // Don't replace a scalar select with a more expensive vector select if
4501 // we can't simplify both arms of the select.
4502 bool SimplifyBothArms =
4503 !Op->getType()->isVectorTy() && II->getType()->isVectorTy();
4505 *II, Sel, /*FoldWithMultiUse=*/false, SimplifyBothArms))
4506 return R;
4507 }
4508 if (auto *Phi = dyn_cast<PHINode>(Op))
4509 if (Instruction *R = foldOpIntoPhi(*II, Phi))
4510 return R;
4511 }
4512 }
4513
4515 return Shuf;
4516
4518 return replaceInstUsesWith(*II, Reverse);
4519
4521 return replaceInstUsesWith(*II, Res);
4522
4523 // Some intrinsics (like experimental_gc_statepoint) can be used in invoke
4524 // context, so it is handled in visitCallBase and we should trigger it.
4525 return visitCallBase(*II);
4526}
4527
4528// Fence instruction simplification
4530 auto *NFI = dyn_cast<FenceInst>(FI.getNextNode());
4531 // This check is solely here to handle arbitrary target-dependent syncscopes.
4532 // TODO: Can remove if does not matter in practice.
4533 if (NFI && FI.isIdenticalTo(NFI))
4534 return eraseInstFromFunction(FI);
4535
4536 // Returns true if FI1 is identical or stronger fence than FI2.
4537 auto isIdenticalOrStrongerFence = [](FenceInst *FI1, FenceInst *FI2) {
4538 auto FI1SyncScope = FI1->getSyncScopeID();
4539 // Consider same scope, where scope is global or single-thread.
4540 if (FI1SyncScope != FI2->getSyncScopeID() ||
4541 (FI1SyncScope != SyncScope::System &&
4542 FI1SyncScope != SyncScope::SingleThread))
4543 return false;
4544
4545 return isAtLeastOrStrongerThan(FI1->getOrdering(), FI2->getOrdering());
4546 };
4547 if (NFI && isIdenticalOrStrongerFence(NFI, &FI))
4548 return eraseInstFromFunction(FI);
4549
4550 if (auto *PFI = dyn_cast_or_null<FenceInst>(FI.getPrevNode()))
4551 if (isIdenticalOrStrongerFence(PFI, &FI))
4552 return eraseInstFromFunction(FI);
4553 return nullptr;
4554}
4555
4556// InvokeInst simplification
4558 return visitCallBase(II);
4559}
4560
4561// CallBrInst simplification
4563 return visitCallBase(CBI);
4564}
4565
4566// A simple parser for format string specifiers for the purposes of the
4567// modular-format attribute. In the case of malformed format strings this might
4568// under or over report the specifiers present, but such cases are undefined
4569// behavior.
4571 Bitset<256> Specifiers;
4572 for (size_t I = 0; I < FormatStr.size(); ++I) {
4573 if (FormatStr[I] != '%')
4574 continue;
4575
4576 // Check for escaped '%'.
4577 if (I + 1 < FormatStr.size() && FormatStr[I + 1] == '%') {
4578 ++I; // Skip the second '%'.
4579 continue;
4580 }
4581
4582 // Scan past allowed prefix characters.
4583 size_t J =
4584 FormatStr.find_first_not_of("0123456789-+ #0$.*'hlLjztqwvI", I + 1);
4585 if (J == StringRef::npos)
4586 break;
4587
4588 Specifiers.set(static_cast<unsigned char>(FormatStr[J]));
4589 I = J; // Resume search from after the specifier.
4590 }
4591 return Specifiers;
4592}
4593
4594static bool isAspectNeeded(StringRef Aspect, CallInst *CI,
4595 std::optional<unsigned> FirstArgIdx,
4596 const std::optional<Bitset<256>> &Specifiers) {
4597 if (Aspect == "float") {
4598 if (Specifiers) {
4599 static constexpr Bitset<256> FloatSpecifiers{'f', 'F', 'e', 'E',
4600 'g', 'G', 'a', 'A'};
4601 return (*Specifiers & FloatSpecifiers).any();
4602 }
4603 // Fallback to type-based check for dynamic format string.
4604 if (!FirstArgIdx)
4605 return true;
4606 return llvm::any_of(
4607 llvm::make_range(std::next(CI->arg_begin(), *FirstArgIdx),
4608 CI->arg_end()),
4609 [](Value *V) { return V->getType()->isFloatingPointTy(); });
4610 }
4611 if (Aspect == "fixed") {
4612 if (Specifiers) {
4613 static constexpr Bitset<256> FixedSpecifiers{'r', 'R', 'k', 'K'};
4614 return (*Specifiers & FixedSpecifiers).any();
4615 }
4616 // Fallback for fixed-point: assume needed if format is dynamic.
4617 return true;
4618 }
4619 // Unknown aspects are always considered to be needed.
4620 return true;
4621}
4622
4623static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M,
4624 IRBuilderBase &B) {
4625 SmallString<20> Name = ImplName;
4626 Name += '_';
4627 Name += Aspect;
4628 LLVMContext &Ctx = M->getContext();
4629 Function *RelocNoneFn =
4630 Intrinsic::getOrInsertDeclaration(M, Intrinsic::reloc_none);
4631 B.CreateCall(RelocNoneFn,
4632 {MetadataAsValue::get(Ctx, MDString::get(Ctx, Name))});
4633}
4634
4636 if (!CI->hasFnAttr("modular-format"))
4637 return nullptr;
4638
4640 llvm::split(CI->getFnAttr("modular-format").getValueAsString(), ','));
4641 if (Args.size() < 5)
4642 return nullptr;
4643
4644 StringRef FormatIdxStr = Args[1];
4645 StringRef FirstArgIdxStr = Args[2];
4646 StringRef FnName = Args[3];
4647 StringRef ImplName = Args[4];
4649
4650 unsigned FormatIdx;
4651 std::optional<unsigned> FirstArgIdx;
4652 [[maybe_unused]] bool Error;
4653 Error = FormatIdxStr.getAsInteger(10, FormatIdx);
4654 assert(!Error && "invalid format arg index");
4655 --FormatIdx; // 1-based to 0-based
4656
4657 FirstArgIdx.emplace();
4658 Error = FirstArgIdxStr.getAsInteger(10, *FirstArgIdx);
4659 assert(!Error && "invalid first arg index");
4660 if (*FirstArgIdx > 0)
4661 --*FirstArgIdx; // 1-based to 0-based
4662 else
4663 FirstArgIdx.reset();
4664
4665 if (AllAspects.empty())
4666 return nullptr;
4667
4668 Value *FormatVal = CI->getArgOperand(FormatIdx);
4669 StringRef FormatStr;
4670
4671 std::optional<Bitset<256>> Specifiers;
4672 if (getConstantStringInfo(FormatVal, FormatStr))
4673 Specifiers = parseFormatStringSpecifiers(FormatStr);
4674
4675 SmallVector<StringRef> NeededAspects;
4676 for (StringRef Aspect : AllAspects)
4677 if (isAspectNeeded(Aspect, CI, FirstArgIdx, Specifiers))
4678 NeededAspects.push_back(Aspect);
4679
4680 if (NeededAspects.size() == AllAspects.size())
4681 return nullptr;
4682
4683 Module *M = CI->getModule();
4684 LLVMContext &Ctx = M->getContext();
4685 Function *Callee = CI->getCalledFunction();
4686 FunctionCallee ModularFn = M->getOrInsertFunction(
4687 FnName, Callee->getFunctionType(),
4688 Callee->getAttributes().removeFnAttribute(Ctx, "modular-format"));
4689 CallInst *New = cast<CallInst>(CI->clone());
4690 New->setCalledFunction(ModularFn);
4691 New->removeFnAttr("modular-format");
4692 B.Insert(New);
4693
4694 llvm::sort(NeededAspects);
4695 for (StringRef Request : NeededAspects)
4696 referenceAspect(Request, ImplName, M, B);
4697
4698 return New;
4699}
4700
4701Instruction *InstCombinerImpl::tryOptimizeCall(CallInst *CI) {
4702 if (!CI->getCalledFunction()) return nullptr;
4703
4704 // Skip optimizing notail and musttail calls so
4705 // LibCallSimplifier::optimizeCall doesn't have to preserve those invariants.
4706 // LibCallSimplifier::optimizeCall should try to preserve tail calls though.
4707 if (CI->isMustTailCall() || CI->isNoTailCall())
4708 return nullptr;
4709
4710 auto InstCombineRAUW = [this](Instruction *From, Value *With) {
4711 replaceInstUsesWith(*From, With);
4712 };
4713 auto InstCombineErase = [this](Instruction *I) {
4715 };
4716 LibCallSimplifier Simplifier(DL, &TLI, &DT, &DC, &AC, ORE, BFI, PSI,
4717 InstCombineRAUW, InstCombineErase);
4718 if (Value *With = Simplifier.optimizeCall(CI, Builder)) {
4719 ++NumSimplified;
4720 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4721 }
4722 if (Value *With = optimizeModularFormat(CI, Builder)) {
4723 ++NumSimplified;
4724 return CI->use_empty() ? CI : replaceInstUsesWith(*CI, With);
4725 }
4726
4727 return nullptr;
4728}
4729
4731 // Strip off at most one level of pointer casts, looking for an alloca. This
4732 // is good enough in practice and simpler than handling any number of casts.
4733 Value *Underlying = TrampMem->stripPointerCasts();
4734 if (Underlying != TrampMem &&
4735 (!Underlying->hasOneUse() || Underlying->user_back() != TrampMem))
4736 return nullptr;
4737 if (!isa<AllocaInst>(Underlying))
4738 return nullptr;
4739
4740 IntrinsicInst *InitTrampoline = nullptr;
4741 for (User *U : TrampMem->users()) {
4743 if (!II)
4744 return nullptr;
4745 if (II->getIntrinsicID() == Intrinsic::init_trampoline) {
4746 if (InitTrampoline)
4747 // More than one init_trampoline writes to this value. Give up.
4748 return nullptr;
4749 InitTrampoline = II;
4750 continue;
4751 }
4752 if (II->getIntrinsicID() == Intrinsic::adjust_trampoline)
4753 // Allow any number of calls to adjust.trampoline.
4754 continue;
4755 return nullptr;
4756 }
4757
4758 // No call to init.trampoline found.
4759 if (!InitTrampoline)
4760 return nullptr;
4761
4762 // Check that the alloca is being used in the expected way.
4763 if (InitTrampoline->getOperand(0) != TrampMem)
4764 return nullptr;
4765
4766 return InitTrampoline;
4767}
4768
4770 Value *TrampMem) {
4771 // Visit all the previous instructions in the basic block, and try to find a
4772 // init.trampoline which has a direct path to the adjust.trampoline.
4773 for (BasicBlock::iterator I = AdjustTramp->getIterator(),
4774 E = AdjustTramp->getParent()->begin();
4775 I != E;) {
4776 Instruction *Inst = &*--I;
4778 if (II->getIntrinsicID() == Intrinsic::init_trampoline &&
4779 II->getOperand(0) == TrampMem)
4780 return II;
4781 if (Inst->mayWriteToMemory())
4782 return nullptr;
4783 }
4784 return nullptr;
4785}
4786
4787// Given a call to llvm.adjust.trampoline, find and return the corresponding
4788// call to llvm.init.trampoline if the call to the trampoline can be optimized
4789// to a direct call to a function. Otherwise return NULL.
4791 Callee = Callee->stripPointerCasts();
4792 IntrinsicInst *AdjustTramp = dyn_cast<IntrinsicInst>(Callee);
4793 if (!AdjustTramp ||
4794 AdjustTramp->getIntrinsicID() != Intrinsic::adjust_trampoline)
4795 return nullptr;
4796
4797 Value *TrampMem = AdjustTramp->getOperand(0);
4798
4800 return IT;
4801 if (IntrinsicInst *IT = findInitTrampolineFromBB(AdjustTramp, TrampMem))
4802 return IT;
4803 return nullptr;
4804}
4805
4806Instruction *InstCombinerImpl::foldPtrAuthIntrinsicCallee(CallBase &Call) {
4807 const Value *Callee = Call.getCalledOperand();
4808 const auto *IPC = dyn_cast<IntToPtrInst>(Callee);
4809 if (!IPC || !IPC->isNoopCast(DL))
4810 return nullptr;
4811
4812 const auto *II = dyn_cast<IntrinsicInst>(IPC->getOperand(0));
4813 if (!II)
4814 return nullptr;
4815
4816 Intrinsic::ID IIID = II->getIntrinsicID();
4817 if (IIID != Intrinsic::ptrauth_resign && IIID != Intrinsic::ptrauth_sign)
4818 return nullptr;
4819
4820 // Isolate the ptrauth bundle from the others.
4821 std::optional<OperandBundleUse> PtrAuthBundleOrNone;
4823 for (unsigned BI = 0, BE = Call.getNumOperandBundles(); BI != BE; ++BI) {
4824 OperandBundleUse Bundle = Call.getOperandBundleAt(BI);
4825 if (Bundle.getTagID() == LLVMContext::OB_ptrauth)
4826 PtrAuthBundleOrNone = Bundle;
4827 else
4828 NewBundles.emplace_back(Bundle);
4829 }
4830
4831 if (!PtrAuthBundleOrNone)
4832 return nullptr;
4833
4834 Value *NewCallee = nullptr;
4835 switch (IIID) {
4836 // call(ptrauth.resign(p)), ["ptrauth"()] -> call p, ["ptrauth"()]
4837 // assuming the call bundle and the sign operands match.
4838 case Intrinsic::ptrauth_resign: {
4839 // Resign result key should match bundle.
4840 if (II->getOperand(3) != PtrAuthBundleOrNone->Inputs[0])
4841 return nullptr;
4842 // Resign result discriminator should match bundle.
4843 if (II->getOperand(4) != PtrAuthBundleOrNone->Inputs[1])
4844 return nullptr;
4845
4846 // Resign input (auth) key should also match: we can't change the key on
4847 // the new call we're generating, because we don't know what keys are valid.
4848 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4849 return nullptr;
4850
4851 Value *NewBundleOps[] = {II->getOperand(1), II->getOperand(2)};
4852 NewBundles.emplace_back("ptrauth", NewBundleOps);
4853 NewCallee = II->getOperand(0);
4854 break;
4855 }
4856
4857 // call(ptrauth.sign(p)), ["ptrauth"()] -> call p
4858 // assuming the call bundle and the sign operands match.
4859 // Non-ptrauth indirect calls are undesirable, but so is ptrauth.sign.
4860 case Intrinsic::ptrauth_sign: {
4861 // Sign key should match bundle.
4862 if (II->getOperand(1) != PtrAuthBundleOrNone->Inputs[0])
4863 return nullptr;
4864 // Sign discriminator should match bundle.
4865 if (II->getOperand(2) != PtrAuthBundleOrNone->Inputs[1])
4866 return nullptr;
4867 NewCallee = II->getOperand(0);
4868 break;
4869 }
4870 default:
4871 llvm_unreachable("unexpected intrinsic ID");
4872 }
4873
4874 if (!NewCallee)
4875 return nullptr;
4876
4877 NewCallee = Builder.CreateBitOrPointerCast(NewCallee, Callee->getType());
4878 CallBase *NewCall = CallBase::Create(&Call, NewBundles);
4879 NewCall->setCalledOperand(NewCallee);
4880 return NewCall;
4881}
4882
4883Instruction *InstCombinerImpl::foldPtrAuthConstantCallee(CallBase &Call) {
4885 if (!CPA)
4886 return nullptr;
4887
4888 auto *CalleeF = dyn_cast<Function>(CPA->getPointer());
4889 // If the ptrauth constant isn't based on a function pointer, bail out.
4890 if (!CalleeF)
4891 return nullptr;
4892
4893 // Inspect the call ptrauth bundle to check it matches the ptrauth constant.
4895 if (!PAB)
4896 return nullptr;
4897
4898 auto *Key = cast<ConstantInt>(PAB->Inputs[0]);
4899 Value *Discriminator = PAB->Inputs[1];
4900
4901 // If the bundle doesn't match, this is probably going to fail to auth.
4902 if (!CPA->isKnownCompatibleWith(Key, Discriminator, DL))
4903 return nullptr;
4904
4905 // If the bundle matches the constant, proceed in making this a direct call.
4907 NewCall->setCalledOperand(CalleeF);
4908 return NewCall;
4909}
4910
4911bool InstCombinerImpl::annotateAnyAllocSite(CallBase &Call,
4912 const TargetLibraryInfo *TLI) {
4913 // Note: We only handle cases which can't be driven from generic attributes
4914 // here. So, for example, nonnull and noalias (which are common properties
4915 // of some allocation functions) are expected to be handled via annotation
4916 // of the respective allocator declaration with generic attributes.
4917 bool Changed = false;
4918
4919 if (!Call.getType()->isPointerTy())
4920 return Changed;
4921
4922 std::optional<APInt> Size = getAllocSize(&Call, TLI);
4923 if (Size && *Size != 0) {
4924 // TODO: We really should just emit deref_or_null here and then
4925 // let the generic inference code combine that with nonnull.
4926 if (Call.hasRetAttr(Attribute::NonNull)) {
4927 Changed = !Call.hasRetAttr(Attribute::Dereferenceable);
4929 Call.getContext(), Size->getLimitedValue()));
4930 } else {
4931 Changed = !Call.hasRetAttr(Attribute::DereferenceableOrNull);
4933 Call.getContext(), Size->getLimitedValue()));
4934 }
4935 }
4936
4937 // Add alignment attribute if alignment is a power of two constant.
4938 Value *Alignment = getAllocAlignment(&Call, TLI);
4939 if (!Alignment)
4940 return Changed;
4941
4942 ConstantInt *AlignOpC = dyn_cast<ConstantInt>(Alignment);
4943 if (AlignOpC && AlignOpC->getValue().ult(llvm::Value::MaximumAlignment)) {
4944 uint64_t AlignmentVal = AlignOpC->getZExtValue();
4945 if (llvm::isPowerOf2_64(AlignmentVal)) {
4946 Align ExistingAlign = Call.getRetAlign().valueOrOne();
4947 Align NewAlign = Align(AlignmentVal);
4948 if (NewAlign > ExistingAlign) {
4951 Changed = true;
4952 }
4953 }
4954 }
4955 return Changed;
4956}
4957
4958/// Improvements for call, callbr and invoke instructions.
4959Instruction *InstCombinerImpl::visitCallBase(CallBase &Call) {
4960 bool Changed = annotateAnyAllocSite(Call, &TLI);
4961
4962 // Mark any parameters that are known to be non-null with the nonnull
4963 // attribute. This is helpful for inlining calls to functions with null
4964 // checks on their arguments.
4965 SmallVector<unsigned, 4> ArgNos;
4966 unsigned ArgNo = 0;
4967
4968 for (Value *V : Call.args()) {
4969 if (V->getType()->isPointerTy()) {
4970 // Simplify the nonnull operand if the parameter is known to be nonnull.
4971 // Otherwise, try to infer nonnull for it.
4972 bool HasDereferenceable = Call.getParamDereferenceableBytes(ArgNo) > 0;
4973 if (Call.paramHasAttr(ArgNo, Attribute::NonNull) ||
4974 (HasDereferenceable &&
4976 V->getType()->getPointerAddressSpace()))) {
4977 if (Value *Res = simplifyNonNullOperand(V, HasDereferenceable)) {
4978 replaceOperand(Call, ArgNo, Res);
4979 Changed = true;
4980 }
4981 } else if (isKnownNonZero(V,
4982 getSimplifyQuery().getWithInstruction(&Call))) {
4983 ArgNos.push_back(ArgNo);
4984 }
4985 }
4986 ArgNo++;
4987 }
4988
4989 assert(ArgNo == Call.arg_size() && "Call arguments not processed correctly.");
4990
4991 if (!ArgNos.empty()) {
4992 AttributeList AS = Call.getAttributes();
4993 LLVMContext &Ctx = Call.getContext();
4994 AS = AS.addParamAttribute(Ctx, ArgNos,
4995 Attribute::get(Ctx, Attribute::NonNull));
4996 Call.setAttributes(AS);
4997 Changed = true;
4998 }
4999
5000 // If the callee is a pointer to a function, attempt to move any casts to the
5001 // arguments of the call/callbr/invoke.
5003 Function *CalleeF = dyn_cast<Function>(Callee);
5004 if ((!CalleeF || CalleeF->getFunctionType() != Call.getFunctionType()) &&
5005 transformConstExprCastCall(Call))
5006 return nullptr;
5007
5008 if (CalleeF) {
5009 // Remove the convergent attr on calls when the callee is not convergent.
5010 if (Call.isConvergent() && !CalleeF->isConvergent() &&
5011 !CalleeF->isIntrinsic()) {
5012 LLVM_DEBUG(dbgs() << "Removing convergent attr from instr " << Call
5013 << "\n");
5015 return &Call;
5016 }
5017
5018 // If the call and callee calling conventions don't match, and neither one
5019 // of the calling conventions is compatible with C calling convention
5020 // this call must be unreachable, as the call is undefined.
5021 if ((CalleeF->getCallingConv() != Call.getCallingConv() &&
5022 !(CalleeF->getCallingConv() == llvm::CallingConv::C &&
5026 // Only do this for calls to a function with a body. A prototype may
5027 // not actually end up matching the implementation's calling conv for a
5028 // variety of reasons (e.g. it may be written in assembly).
5029 !CalleeF->isDeclaration()) {
5030 Instruction *OldCall = &Call;
5032 // If OldCall does not return void then replaceInstUsesWith poison.
5033 // This allows ValueHandlers and custom metadata to adjust itself.
5034 if (!OldCall->getType()->isVoidTy())
5035 replaceInstUsesWith(*OldCall, PoisonValue::get(OldCall->getType()));
5036 if (isa<CallInst>(OldCall))
5037 return eraseInstFromFunction(*OldCall);
5038
5039 // We cannot remove an invoke or a callbr, because it would change thexi
5040 // CFG, just change the callee to a null pointer.
5041 cast<CallBase>(OldCall)->setCalledFunction(
5042 CalleeF->getFunctionType(),
5043 Constant::getNullValue(CalleeF->getType()));
5044 return nullptr;
5045 }
5046 }
5047
5048 // Calling a null function pointer is undefined if a null address isn't
5049 // dereferenceable.
5050 if ((isa<ConstantPointerNull>(Callee) &&
5052 isa<UndefValue>(Callee)) {
5053 // If Call does not return void then replaceInstUsesWith poison.
5054 // This allows ValueHandlers and custom metadata to adjust itself.
5055 if (!Call.getType()->isVoidTy())
5057
5058 if (Call.isTerminator()) {
5059 // Can't remove an invoke or callbr because we cannot change the CFG.
5060 return nullptr;
5061 }
5062
5063 // This instruction is not reachable, just remove it.
5066 }
5067
5068 if (IntrinsicInst *II = findInitTrampoline(Callee))
5069 return transformCallThroughTrampoline(Call, *II);
5070
5071 // Combine calls involving pointer authentication intrinsics.
5072 if (Instruction *NewCall = foldPtrAuthIntrinsicCallee(Call))
5073 return NewCall;
5074
5075 // Combine calls to ptrauth constants.
5076 if (Instruction *NewCall = foldPtrAuthConstantCallee(Call))
5077 return NewCall;
5078
5079 if (isa<InlineAsm>(Callee) && !Call.doesNotThrow()) {
5080 InlineAsm *IA = cast<InlineAsm>(Callee);
5081 if (!IA->canThrow()) {
5082 // Normal inline asm calls cannot throw - mark them
5083 // 'nounwind'.
5085 Changed = true;
5086 }
5087 }
5088
5089 // Try to optimize the call if possible, we require DataLayout for most of
5090 // this. None of these calls are seen as possibly dead so go ahead and
5091 // delete the instruction now.
5092 if (CallInst *CI = dyn_cast<CallInst>(&Call)) {
5093 Instruction *I = tryOptimizeCall(CI);
5094 // If we changed something return the result, etc. Otherwise let
5095 // the fallthrough check.
5096 if (I) return eraseInstFromFunction(*I);
5097 }
5098
5099 if (!Call.use_empty() && !Call.isMustTailCall())
5100 if (Value *ReturnedArg = Call.getReturnedArgOperand()) {
5101 Type *CallTy = Call.getType();
5102 Type *RetArgTy = ReturnedArg->getType();
5103 if (RetArgTy->canLosslesslyBitCastTo(CallTy))
5104 return replaceInstUsesWith(
5105 Call, Builder.CreateBitOrPointerCast(ReturnedArg, CallTy));
5106 }
5107
5108 // Drop unnecessary callee_type metadata from calls that were converted
5109 // into direct calls.
5110 if (Call.getMetadata(LLVMContext::MD_callee_type) && !Call.isIndirectCall()) {
5111 Call.setMetadata(LLVMContext::MD_callee_type, nullptr);
5112 Changed = true;
5113 }
5114
5115 // Drop unnecessary kcfi operand bundles from calls that were converted
5116 // into direct calls.
5118 if (Bundle && !Call.isIndirectCall()) {
5119 DEBUG_WITH_TYPE(DEBUG_TYPE "-kcfi", {
5120 if (CalleeF) {
5121 ConstantInt *FunctionType = nullptr;
5122 ConstantInt *ExpectedType = cast<ConstantInt>(Bundle->Inputs[0]);
5123
5124 if (MDNode *MD = CalleeF->getMetadata(LLVMContext::MD_kcfi_type))
5125 FunctionType = mdconst::extract<ConstantInt>(MD->getOperand(0));
5126
5127 if (FunctionType &&
5128 FunctionType->getZExtValue() != ExpectedType->getZExtValue())
5129 dbgs() << Call.getModule()->getName()
5130 << ": warning: kcfi: " << Call.getCaller()->getName()
5131 << ": call to " << CalleeF->getName()
5132 << " using a mismatching function pointer type\n";
5133 }
5134 });
5135
5137 }
5138
5139 if (isRemovableAlloc(&Call, &TLI))
5140 return visitAllocSite(Call);
5141
5142 // Handle intrinsics which can be used in both call and invoke context.
5143 switch (Call.getIntrinsicID()) {
5144 case Intrinsic::experimental_gc_statepoint: {
5145 GCStatepointInst &GCSP = *cast<GCStatepointInst>(&Call);
5146 SmallPtrSet<Value *, 32> LiveGcValues;
5147 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5148 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5149
5150 // Remove the relocation if unused.
5151 if (GCR.use_empty()) {
5153 continue;
5154 }
5155
5156 Value *DerivedPtr = GCR.getDerivedPtr();
5157 Value *BasePtr = GCR.getBasePtr();
5158
5159 // Undef is undef, even after relocation.
5160 if (isa<UndefValue>(DerivedPtr) || isa<UndefValue>(BasePtr)) {
5163 continue;
5164 }
5165
5166 if (auto *PT = dyn_cast<PointerType>(GCR.getType())) {
5167 // The relocation of null will be null for most any collector.
5168 // TODO: provide a hook for this in GCStrategy. There might be some
5169 // weird collector this property does not hold for.
5170 if (isa<ConstantPointerNull>(DerivedPtr)) {
5171 // Use null-pointer of gc_relocate's type to replace it.
5174 continue;
5175 }
5176
5177 // isKnownNonNull -> nonnull attribute
5178 if (!GCR.hasRetAttr(Attribute::NonNull) &&
5179 isKnownNonZero(DerivedPtr,
5180 getSimplifyQuery().getWithInstruction(&Call))) {
5181 GCR.addRetAttr(Attribute::NonNull);
5182 // We discovered new fact, re-check users.
5183 Worklist.pushUsersToWorkList(GCR);
5184 }
5185 }
5186
5187 // If we have two copies of the same pointer in the statepoint argument
5188 // list, canonicalize to one. This may let us common gc.relocates.
5189 if (GCR.getBasePtr() == GCR.getDerivedPtr() &&
5190 GCR.getBasePtrIndex() != GCR.getDerivedPtrIndex()) {
5191 auto *OpIntTy = GCR.getOperand(2)->getType();
5192 GCR.setOperand(2, ConstantInt::get(OpIntTy, GCR.getBasePtrIndex()));
5193 }
5194
5195 // TODO: bitcast(relocate(p)) -> relocate(bitcast(p))
5196 // Canonicalize on the type from the uses to the defs
5197
5198 // TODO: relocate((gep p, C, C2, ...)) -> gep(relocate(p), C, C2, ...)
5199 LiveGcValues.insert(BasePtr);
5200 LiveGcValues.insert(DerivedPtr);
5201 }
5202 std::optional<OperandBundleUse> Bundle =
5204 unsigned NumOfGCLives = LiveGcValues.size();
5205 if (!Bundle || NumOfGCLives == Bundle->Inputs.size())
5206 break;
5207 // We can reduce the size of gc live bundle.
5208 DenseMap<Value *, unsigned> Val2Idx;
5209 std::vector<Value *> NewLiveGc;
5210 for (Value *V : Bundle->Inputs) {
5211 auto [It, Inserted] = Val2Idx.try_emplace(V);
5212 if (!Inserted)
5213 continue;
5214 if (LiveGcValues.count(V)) {
5215 It->second = NewLiveGc.size();
5216 NewLiveGc.push_back(V);
5217 } else
5218 It->second = NumOfGCLives;
5219 }
5220 // Update all gc.relocates
5221 for (const GCRelocateInst *Reloc : GCSP.getGCRelocates()) {
5222 GCRelocateInst &GCR = *const_cast<GCRelocateInst *>(Reloc);
5223 Value *BasePtr = GCR.getBasePtr();
5224 assert(Val2Idx.count(BasePtr) && Val2Idx[BasePtr] != NumOfGCLives &&
5225 "Missed live gc for base pointer");
5226 auto *OpIntTy1 = GCR.getOperand(1)->getType();
5227 GCR.setOperand(1, ConstantInt::get(OpIntTy1, Val2Idx[BasePtr]));
5228 Value *DerivedPtr = GCR.getDerivedPtr();
5229 assert(Val2Idx.count(DerivedPtr) && Val2Idx[DerivedPtr] != NumOfGCLives &&
5230 "Missed live gc for derived pointer");
5231 auto *OpIntTy2 = GCR.getOperand(2)->getType();
5232 GCR.setOperand(2, ConstantInt::get(OpIntTy2, Val2Idx[DerivedPtr]));
5233 }
5234 // Create new statepoint instruction.
5235 OperandBundleDef NewBundle("gc-live", std::move(NewLiveGc));
5236 return CallBase::Create(&Call, NewBundle);
5237 }
5238 default: { break; }
5239 }
5240
5241 return Changed ? &Call : nullptr;
5242}
5243
5244/// If the callee is a constexpr cast of a function, attempt to move the cast to
5245/// the arguments of the call/invoke.
5246/// CallBrInst is not supported.
5247bool InstCombinerImpl::transformConstExprCastCall(CallBase &Call) {
5248 auto *Callee =
5250 if (!Callee)
5251 return false;
5252
5254 "CallBr's don't have a single point after a def to insert at");
5255
5256 // Don't perform the transform for declarations, which may not be fully
5257 // accurate. For example, void @foo() is commonly used as a placeholder for
5258 // unknown prototypes.
5259 if (Callee->isDeclaration())
5260 return false;
5261
5262 // If this is a call to a thunk function, don't remove the cast. Thunks are
5263 // used to transparently forward all incoming parameters and outgoing return
5264 // values, so it's important to leave the cast in place.
5265 if (Callee->hasFnAttribute("thunk"))
5266 return false;
5267
5268 // If this is a call to a naked function, the assembly might be
5269 // using an argument, or otherwise rely on the frame layout,
5270 // the function prototype will mismatch.
5271 if (Callee->hasFnAttribute(Attribute::Naked))
5272 return false;
5273
5274 // If this is a musttail call, the callee's prototype must match the caller's
5275 // prototype with the exception of pointee types. The code below doesn't
5276 // implement that, so we can't do this transform.
5277 // TODO: Do the transform if it only requires adding pointer casts.
5278 if (Call.isMustTailCall())
5279 return false;
5280
5282 const AttributeList &CallerPAL = Call.getAttributes();
5283
5284 // Okay, this is a cast from a function to a different type. Unless doing so
5285 // would cause a type conversion of one of our arguments, change this call to
5286 // be a direct call with arguments casted to the appropriate types.
5287 FunctionType *FT = Callee->getFunctionType();
5288 Type *OldRetTy = Caller->getType();
5289 Type *NewRetTy = FT->getReturnType();
5290
5291 // Check to see if we are changing the return type...
5292 if (OldRetTy != NewRetTy) {
5293
5294 if (NewRetTy->isStructTy())
5295 return false; // TODO: Handle multiple return values.
5296
5297 if (!CastInst::isBitOrNoopPointerCastable(NewRetTy, OldRetTy, DL)) {
5298 if (!Caller->use_empty())
5299 return false; // Cannot transform this return value.
5300 }
5301
5302 if (!CallerPAL.isEmpty() && !Caller->use_empty()) {
5303 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5304 if (RAttrs.overlaps(AttributeFuncs::typeIncompatible(
5305 NewRetTy, CallerPAL.getRetAttrs())))
5306 return false; // Attribute not compatible with transformed value.
5307 }
5308
5309 // If the callbase is an invoke instruction, and the return value is
5310 // used by a PHI node in a successor, we cannot change the return type of
5311 // the call because there is no place to put the cast instruction (without
5312 // breaking the critical edge). Bail out in this case.
5313 if (!Caller->use_empty()) {
5314 BasicBlock *PhisNotSupportedBlock = nullptr;
5315 if (auto *II = dyn_cast<InvokeInst>(Caller))
5316 PhisNotSupportedBlock = II->getNormalDest();
5317 if (PhisNotSupportedBlock)
5318 for (User *U : Caller->users())
5319 if (PHINode *PN = dyn_cast<PHINode>(U))
5320 if (PN->getParent() == PhisNotSupportedBlock)
5321 return false;
5322 }
5323 }
5324
5325 unsigned NumActualArgs = Call.arg_size();
5326 unsigned NumCommonArgs = std::min(FT->getNumParams(), NumActualArgs);
5327
5328 // Prevent us turning:
5329 // declare void @takes_i32_inalloca(i32* inalloca)
5330 // call void bitcast (void (i32*)* @takes_i32_inalloca to void (i32)*)(i32 0)
5331 //
5332 // into:
5333 // call void @takes_i32_inalloca(i32* null)
5334 //
5335 // Similarly, avoid folding away bitcasts of byval calls.
5336 if (Callee->getAttributes().hasAttrSomewhere(Attribute::InAlloca) ||
5337 Callee->getAttributes().hasAttrSomewhere(Attribute::Preallocated))
5338 return false;
5339
5340 auto AI = Call.arg_begin();
5341 for (unsigned i = 0, e = NumCommonArgs; i != e; ++i, ++AI) {
5342 Type *ParamTy = FT->getParamType(i);
5343 Type *ActTy = (*AI)->getType();
5344
5345 if (!CastInst::isBitOrNoopPointerCastable(ActTy, ParamTy, DL))
5346 return false; // Cannot transform this parameter value.
5347
5348 // Check if there are any incompatible attributes we cannot drop safely.
5349 if (AttrBuilder(FT->getContext(), CallerPAL.getParamAttrs(i))
5350 .overlaps(AttributeFuncs::typeIncompatible(
5351 ParamTy, CallerPAL.getParamAttrs(i),
5352 AttributeFuncs::ASK_UNSAFE_TO_DROP)))
5353 return false; // Attribute not compatible with transformed value.
5354
5355 if (Call.isInAllocaArgument(i) ||
5356 CallerPAL.hasParamAttr(i, Attribute::Preallocated))
5357 return false; // Cannot transform to and from inalloca/preallocated.
5358
5359 if (CallerPAL.hasParamAttr(i, Attribute::SwiftError))
5360 return false;
5361
5362 if (CallerPAL.hasParamAttr(i, Attribute::ByVal) !=
5363 Callee->getAttributes().hasParamAttr(i, Attribute::ByVal))
5364 return false; // Cannot transform to or from byval.
5365 }
5366
5367 if (FT->getNumParams() < NumActualArgs && FT->isVarArg() &&
5368 !CallerPAL.isEmpty()) {
5369 // In this case we have more arguments than the new function type, but we
5370 // won't be dropping them. Check that these extra arguments have attributes
5371 // that are compatible with being a vararg call argument.
5372 unsigned SRetIdx;
5373 if (CallerPAL.hasAttrSomewhere(Attribute::StructRet, &SRetIdx) &&
5374 SRetIdx - AttributeList::FirstArgIndex >= FT->getNumParams())
5375 return false;
5376 }
5377
5378 // Okay, we decided that this is a safe thing to do: go ahead and start
5379 // inserting cast instructions as necessary.
5380 SmallVector<Value *, 8> Args;
5382 Args.reserve(NumActualArgs);
5383 ArgAttrs.reserve(NumActualArgs);
5384
5385 // Get any return attributes.
5386 AttrBuilder RAttrs(FT->getContext(), CallerPAL.getRetAttrs());
5387
5388 // If the return value is not being used, the type may not be compatible
5389 // with the existing attributes. Wipe out any problematic attributes.
5390 RAttrs.remove(
5391 AttributeFuncs::typeIncompatible(NewRetTy, CallerPAL.getRetAttrs()));
5392
5393 LLVMContext &Ctx = Call.getContext();
5394 AI = Call.arg_begin();
5395 for (unsigned i = 0; i != NumCommonArgs; ++i, ++AI) {
5396 Type *ParamTy = FT->getParamType(i);
5397
5398 Value *NewArg = *AI;
5399 if ((*AI)->getType() != ParamTy)
5400 NewArg = Builder.CreateBitOrPointerCast(*AI, ParamTy);
5401 Args.push_back(NewArg);
5402
5403 // Add any parameter attributes except the ones incompatible with the new
5404 // type. Note that we made sure all incompatible ones are safe to drop.
5405 AttributeMask IncompatibleAttrs = AttributeFuncs::typeIncompatible(
5406 ParamTy, CallerPAL.getParamAttrs(i), AttributeFuncs::ASK_SAFE_TO_DROP);
5407 ArgAttrs.push_back(
5408 CallerPAL.getParamAttrs(i).removeAttributes(Ctx, IncompatibleAttrs));
5409 }
5410
5411 // If the function takes more arguments than the call was taking, add them
5412 // now.
5413 for (unsigned i = NumCommonArgs; i != FT->getNumParams(); ++i) {
5414 Args.push_back(Constant::getNullValue(FT->getParamType(i)));
5415 ArgAttrs.push_back(AttributeSet());
5416 }
5417
5418 // If we are removing arguments to the function, emit an obnoxious warning.
5419 if (FT->getNumParams() < NumActualArgs) {
5420 // TODO: if (!FT->isVarArg()) this call may be unreachable. PR14722
5421 if (FT->isVarArg()) {
5422 // Add all of the arguments in their promoted form to the arg list.
5423 for (unsigned i = FT->getNumParams(); i != NumActualArgs; ++i, ++AI) {
5424 Type *PTy = getPromotedType((*AI)->getType());
5425 Value *NewArg = *AI;
5426 if (PTy != (*AI)->getType()) {
5427 // Must promote to pass through va_arg area!
5428 Instruction::CastOps opcode =
5429 CastInst::getCastOpcode(*AI, false, PTy, false);
5430 NewArg = Builder.CreateCast(opcode, *AI, PTy);
5431 }
5432 Args.push_back(NewArg);
5433
5434 // Add any parameter attributes.
5435 ArgAttrs.push_back(CallerPAL.getParamAttrs(i));
5436 }
5437 }
5438 }
5439
5440 AttributeSet FnAttrs = CallerPAL.getFnAttrs();
5441
5442 if (NewRetTy->isVoidTy())
5443 Caller->setName(""); // Void type should not have a name.
5444
5445 assert((ArgAttrs.size() == FT->getNumParams() || FT->isVarArg()) &&
5446 "missing argument attributes");
5447 AttributeList NewCallerPAL = AttributeList::get(
5448 Ctx, FnAttrs, AttributeSet::get(Ctx, RAttrs), ArgAttrs);
5449
5451 Call.getOperandBundlesAsDefs(OpBundles);
5452
5453 CallBase *NewCall;
5454 if (InvokeInst *II = dyn_cast<InvokeInst>(Caller)) {
5455 NewCall = Builder.CreateInvoke(Callee, II->getNormalDest(),
5456 II->getUnwindDest(), Args, OpBundles);
5457 } else {
5458 NewCall = Builder.CreateCall(Callee, Args, OpBundles);
5459 cast<CallInst>(NewCall)->setTailCallKind(
5460 cast<CallInst>(Caller)->getTailCallKind());
5461 }
5462 NewCall->takeName(Caller);
5464 NewCall->setAttributes(NewCallerPAL);
5465
5466 // Preserve prof metadata if any.
5467 NewCall->copyMetadata(*Caller, {LLVMContext::MD_prof});
5468
5469 // Insert a cast of the return type as necessary.
5470 Instruction *NC = NewCall;
5471 Value *NV = NC;
5472 if (OldRetTy != NV->getType() && !Caller->use_empty()) {
5473 assert(!NV->getType()->isVoidTy());
5475 NC->setDebugLoc(Caller->getDebugLoc());
5476
5477 auto OptInsertPt = NewCall->getInsertionPointAfterDef();
5478 assert(OptInsertPt && "No place to insert cast");
5479 InsertNewInstBefore(NC, *OptInsertPt);
5480 Worklist.pushUsersToWorkList(*Caller);
5481 }
5482
5483 if (!Caller->use_empty())
5484 replaceInstUsesWith(*Caller, NV);
5485 else if (Caller->hasValueHandle()) {
5486 if (OldRetTy == NV->getType())
5488 else
5489 // We cannot call ValueIsRAUWd with a different type, and the
5490 // actual tracked value will disappear.
5492 }
5493
5494 eraseInstFromFunction(*Caller);
5495 return true;
5496}
5497
5498/// Turn a call to a function created by init_trampoline / adjust_trampoline
5499/// intrinsic pair into a direct call to the underlying function.
5501InstCombinerImpl::transformCallThroughTrampoline(CallBase &Call,
5502 IntrinsicInst &Tramp) {
5503 FunctionType *FTy = Call.getFunctionType();
5504 AttributeList Attrs = Call.getAttributes();
5505
5506 // If the call already has the 'nest' attribute somewhere then give up -
5507 // otherwise 'nest' would occur twice after splicing in the chain.
5508 if (Attrs.hasAttrSomewhere(Attribute::Nest))
5509 return nullptr;
5510
5512 FunctionType *NestFTy = NestF->getFunctionType();
5513
5514 AttributeList NestAttrs = NestF->getAttributes();
5515 if (!NestAttrs.isEmpty()) {
5516 unsigned NestArgNo = 0;
5517 Type *NestTy = nullptr;
5518 AttributeSet NestAttr;
5519
5520 // Look for a parameter marked with the 'nest' attribute.
5521 for (FunctionType::param_iterator I = NestFTy->param_begin(),
5522 E = NestFTy->param_end();
5523 I != E; ++NestArgNo, ++I) {
5524 AttributeSet AS = NestAttrs.getParamAttrs(NestArgNo);
5525 if (AS.hasAttribute(Attribute::Nest)) {
5526 // Record the parameter type and any other attributes.
5527 NestTy = *I;
5528 NestAttr = AS;
5529 break;
5530 }
5531 }
5532
5533 if (NestTy) {
5534 std::vector<Value*> NewArgs;
5535 std::vector<AttributeSet> NewArgAttrs;
5536 NewArgs.reserve(Call.arg_size() + 1);
5537 NewArgAttrs.reserve(Call.arg_size());
5538
5539 // Insert the nest argument into the call argument list, which may
5540 // mean appending it. Likewise for attributes.
5541
5542 {
5543 unsigned ArgNo = 0;
5544 auto I = Call.arg_begin(), E = Call.arg_end();
5545 do {
5546 if (ArgNo == NestArgNo) {
5547 // Add the chain argument and attributes.
5548 Value *NestVal = Tramp.getArgOperand(2);
5549 if (NestVal->getType() != NestTy)
5550 NestVal = Builder.CreateBitCast(NestVal, NestTy, "nest");
5551 NewArgs.push_back(NestVal);
5552 NewArgAttrs.push_back(NestAttr);
5553 }
5554
5555 if (I == E)
5556 break;
5557
5558 // Add the original argument and attributes.
5559 NewArgs.push_back(*I);
5560 NewArgAttrs.push_back(Attrs.getParamAttrs(ArgNo));
5561
5562 ++ArgNo;
5563 ++I;
5564 } while (true);
5565 }
5566
5567 // The trampoline may have been bitcast to a bogus type (FTy).
5568 // Handle this by synthesizing a new function type, equal to FTy
5569 // with the chain parameter inserted.
5570
5571 std::vector<Type*> NewTypes;
5572 NewTypes.reserve(FTy->getNumParams()+1);
5573
5574 // Insert the chain's type into the list of parameter types, which may
5575 // mean appending it.
5576 {
5577 unsigned ArgNo = 0;
5578 FunctionType::param_iterator I = FTy->param_begin(),
5579 E = FTy->param_end();
5580
5581 do {
5582 if (ArgNo == NestArgNo)
5583 // Add the chain's type.
5584 NewTypes.push_back(NestTy);
5585
5586 if (I == E)
5587 break;
5588
5589 // Add the original type.
5590 NewTypes.push_back(*I);
5591
5592 ++ArgNo;
5593 ++I;
5594 } while (true);
5595 }
5596
5597 // Replace the trampoline call with a direct call. Let the generic
5598 // code sort out any function type mismatches.
5599 FunctionType *NewFTy =
5600 FunctionType::get(FTy->getReturnType(), NewTypes, FTy->isVarArg());
5601 AttributeList NewPAL =
5602 AttributeList::get(FTy->getContext(), Attrs.getFnAttrs(),
5603 Attrs.getRetAttrs(), NewArgAttrs);
5604
5606 Call.getOperandBundlesAsDefs(OpBundles);
5607
5608 Instruction *NewCaller;
5609 if (InvokeInst *II = dyn_cast<InvokeInst>(&Call)) {
5610 NewCaller = InvokeInst::Create(NewFTy, NestF, II->getNormalDest(),
5611 II->getUnwindDest(), NewArgs, OpBundles);
5612 cast<InvokeInst>(NewCaller)->setCallingConv(II->getCallingConv());
5613 cast<InvokeInst>(NewCaller)->setAttributes(NewPAL);
5614 } else if (CallBrInst *CBI = dyn_cast<CallBrInst>(&Call)) {
5615 NewCaller =
5616 CallBrInst::Create(NewFTy, NestF, CBI->getDefaultDest(),
5617 CBI->getIndirectDests(), NewArgs, OpBundles);
5618 cast<CallBrInst>(NewCaller)->setCallingConv(CBI->getCallingConv());
5619 cast<CallBrInst>(NewCaller)->setAttributes(NewPAL);
5620 } else {
5621 NewCaller = CallInst::Create(NewFTy, NestF, NewArgs, OpBundles);
5622 cast<CallInst>(NewCaller)->setTailCallKind(
5623 cast<CallInst>(Call).getTailCallKind());
5624 cast<CallInst>(NewCaller)->setCallingConv(
5625 cast<CallInst>(Call).getCallingConv());
5626 cast<CallInst>(NewCaller)->setAttributes(NewPAL);
5627 }
5628 NewCaller->setDebugLoc(Call.getDebugLoc());
5629
5630 return NewCaller;
5631 }
5632 }
5633
5634 // Replace the trampoline call with a direct call. Since there is no 'nest'
5635 // parameter, there is no need to adjust the argument list. Let the generic
5636 // code sort out any function type mismatches.
5637 Call.setCalledFunction(FTy, NestF);
5638 return &Call;
5639}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU Register Bank Select
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
@ Scaled
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:856
BitTracker BT
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static SDValue foldBitOrderCrossLogicOp(SDNode *N, SelectionDAG &DAG)
#define Check(C,...)
#define DEBUG_TYPE
Hexagon Common GEP
#define _
IRTranslator LLVM IR MI
static Type * getPromotedType(Type *Ty)
Return the specified type promoted as it would be to pass though a va_arg area.
static Instruction * createOverflowTuple(IntrinsicInst *II, Value *Result, Constant *Overflow)
Creates a result tuple for an overflow intrinsic II with a given Result and a constant Overflow value...
static void referenceAspect(StringRef Aspect, StringRef ImplName, Module *M, IRBuilderBase &B)
static IntrinsicInst * findInitTrampolineFromAlloca(Value *TrampMem)
static bool removeTriviallyEmptyRange(IntrinsicInst &EndI, InstCombinerImpl &IC, std::function< bool(const IntrinsicInst &)> IsStart)
static bool inputDenormalIsDAZ(const Function &F, const Type *Ty)
static Instruction * reassociateMinMaxWithConstantInOperand(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If this min/max has a matching min/max operand with a constant, try to push the constant operand into...
static bool isIdempotentBinaryIntrinsic(Intrinsic::ID IID)
Helper to match idempotent binary intrinsics, namely, intrinsics where f(f(x, y), y) == f(x,...
static bool signBitMustBeTheSame(Value *Op0, Value *Op1, const SimplifyQuery &SQ)
Return true if two values Op0 and Op1 are known to have the same sign.
static Value * optimizeModularFormat(CallInst *CI, IRBuilderBase &B)
static Instruction * moveAddAfterMinMax(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
Try to canonicalize min/max(X + C0, C1) as min/max(X, C1 - C0) + C0.
static Instruction * simplifyInvariantGroupIntrinsic(IntrinsicInst &II, InstCombinerImpl &IC)
This function transforms launder.invariant.group and strip.invariant.group like: launder(launder(x)) ...
static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E, unsigned NumOperands)
static std::optional< bool > getKnownSign(Value *Op, const SimplifyQuery &SQ)
static cl::opt< unsigned > GuardWideningWindow("instcombine-guard-widening-window", cl::init(3), cl::desc("How wide an instruction window to bypass looking for " "another guard"))
static bool hasUndefSource(AnyMemTransferInst *MI)
Recognize a memcpy/memmove from a trivially otherwise unused alloca.
static Instruction * factorizeMinMaxTree(IntrinsicInst *II)
Reduce a sequence of min/max intrinsics with a common operand.
static Instruction * foldClampRangeOfTwo(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
If we have a clamp pattern like max (min X, 42), 41 – where the output can only be one of two possibl...
static Value * simplifyReductionOperand(Value *Arg, bool CanReorderLanes)
static IntrinsicInst * findInitTrampolineFromBB(IntrinsicInst *AdjustTramp, Value *TrampMem)
static bool isAspectNeeded(StringRef Aspect, CallInst *CI, std::optional< unsigned > FirstArgIdx, const std::optional< Bitset< 256 > > &Specifiers)
static Value * foldIntrinsicUsingDistributiveLaws(IntrinsicInst *II, InstCombiner::BuilderTy &Builder)
static std::optional< bool > getKnownSignOrZero(Value *Op, const SimplifyQuery &SQ)
static Value * foldMinimumOverTrailingOrLeadingZeroCount(Value *I0, Value *I1, const DataLayout &DL, InstCombiner::BuilderTy &Builder)
Fold an unsigned minimum of trailing or leading zero bits counts: umin(cttz(CtOp1,...
static bool rightDistributesOverLeft(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "(X ROp Y) LOp Z" is always equal to "(X LOp Z) ROp (Y LOp Z)".
static Value * foldIdempotentBinaryIntrinsicRecurrence(InstCombinerImpl &IC, IntrinsicInst *II)
Attempt to simplify value-accumulating recurrences of kind: umax.acc = phi i8 [ umax,...
static bool ldexpSaturatingAddIsSafe(Type *FpTy, Type *ExpTy)
static Instruction * foldCtpop(IntrinsicInst &II, InstCombinerImpl &IC)
static Instruction * simplifyNeonTbl(IntrinsicInst &II, InstCombiner &IC, bool IsExtension)
Convert tbl/tbx intrinsics to shufflevector if the mask is constant, and at most two source operands ...
static Instruction * foldCttzCtlz(IntrinsicInst &II, InstCombinerImpl &IC)
static IntrinsicInst * findInitTrampoline(Value *Callee)
static Bitset< 256 > parseFormatStringSpecifiers(StringRef FormatStr)
static FCmpInst::Predicate fpclassTestIsFCmp0(FPClassTest Mask, const Function &F, Type *Ty)
static bool leftDistributesOverRight(Instruction::BinaryOps LOp, bool HasNUW, bool HasNSW, Intrinsic::ID ROp)
Return whether "X LOp (Y ROp Z)" is always equal to "(X LOp Y) ROp (X LOp Z)".
static Value * reassociateMinMaxWithConstants(IntrinsicInst *II, IRBuilderBase &Builder, const SimplifyQuery &SQ)
If this min/max has a constant operand and an operand that is a matching min/max with a constant oper...
static Value * foldSinAndCosToSinCos(IntrinsicInst *II, IRBuilderBase &B, InstCombinerImpl &IC)
static CallInst * canonicalizeConstantArg0ToArg1(CallInst &Call)
static Instruction * foldNeonShift(IntrinsicInst *II, InstCombinerImpl &IC)
This file provides internal interfaces used to implement the InstCombine.
This file provides the interface for the instcombine pass implementation.
static bool inputDenormalIsIEEE(DenormalMode Mode)
Return true if it's possible to assume IEEE treatment of input denormals in F for Val.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
static const Function * getCalledFunction(const Value *V)
This file contains the declarations for metadata subclasses.
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
if(auto Err=PB.parsePassPipeline(MPM, Passes)) return wrap(std MPM run * Mod
This file contains the declarations for profiling metadata utility functions.
const SmallVectorImpl< MachineOperand > & Cond
This file implements the SmallBitVector class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define DEBUG_WITH_TYPE(TYPE,...)
DEBUG_WITH_TYPE macro - This macro should be used by passes to emit debug information.
Definition Debug.h:72
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
static LLVM_ABI bool semanticsHasInf(const fltSemantics &)
Definition APFloat.cpp:272
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:345
static LLVM_ABI bool hasSignBitInMSB(const fltSemantics &)
Definition APFloat.cpp:285
bool isNegative() const
Definition APFloat.h:1565
void clearSign()
Definition APFloat.h:1384
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1174
bool isZero() const
Definition APFloat.h:1561
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
Definition APFloat.h:1224
static APFloat getSmallest(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) finite number in the given semantics.
Definition APFloat.h:1234
bool isInfinity() const
Definition APFloat.h:1562
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:235
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:230
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1210
LLVM_ABI APInt usub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1983
bool ugt(const APInt &RHS) const
Unsigned greater than comparison.
Definition APInt.h:1191
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:381
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1692
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1513
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1120
LLVM_ABI APInt sadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1963
LLVM_ABI APInt uadd_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1970
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:652
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:220
LLVM_ABI APInt sextOrTrunc(unsigned width) const
Sign extend or truncate to width.
Definition APInt.cpp:1084
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:511
LLVM_ABI APInt uadd_sat(const APInt &RHS) const
Definition APInt.cpp:2071
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:335
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:307
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:201
std::optional< int64_t > trySExtValue() const
Get sign extended value if possible.
Definition APInt.h:1599
LLVM_ABI APInt ssub_ov(const APInt &RHS, bool &Overflow) const
Definition APInt.cpp:1976
static APSInt getMinValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the minimum integer value with the given bit width and signedness.
Definition APSInt.h:310
static APSInt getMaxValue(uint32_t numBits, bool Unsigned)
Return the APSInt representing the maximum integer value with the given bit width and signedness.
Definition APSInt.h:302
This class represents any memset intrinsic.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:407
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
static LLVM_ABI AttributeSet get(LLVMContext &C, const AttrBuilder &B)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
static LLVM_ABI Attribute getWithDereferenceableOrNullBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI Attribute getWithAlignment(LLVMContext &Context, Align Alignment)
Return a uniquified Attribute object that has the specific alignment set.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator begin()
Instruction iterator methods.
Definition BasicBlock.h:461
InstListType::reverse_iterator reverse_iterator
Definition BasicBlock.h:172
InstListType::iterator iterator
Instruction iterators...
Definition BasicBlock.h:170
LLVM_ABI bool isSigned() const
Whether the intrinsic is signed or unsigned.
LLVM_ABI Instruction::BinaryOps getBinaryOp() const
Returns the binary operation underlying the intrinsic.
static BinaryOperator * CreateFAddFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:271
static LLVM_ABI BinaryOperator * CreateNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Helper functions to construct and inspect unary operations (NEG and NOT) via binary operators SUB and...
static BinaryOperator * CreateNSW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:314
static LLVM_ABI BinaryOperator * CreateNot(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
static LLVM_ABI BinaryOperator * Create(BinaryOps Op, Value *S1, Value *S2, const Twine &Name=Twine(), InsertPosition InsertBefore=nullptr)
Construct a binary instruction, given the opcode and the two operands.
static BinaryOperator * CreateNUW(BinaryOps Opc, Value *V1, Value *V2, const Twine &Name="")
Definition InstrTypes.h:329
static BinaryOperator * CreateFMulFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:279
static BinaryOperator * CreateFDivFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:283
static BinaryOperator * CreateFSubFMF(Value *V1, Value *V2, FastMathFlags FMF, const Twine &Name="")
Definition InstrTypes.h:275
static LLVM_ABI BinaryOperator * CreateNSWNeg(Value *Op, const Twine &Name="", InsertPosition InsertBefore=nullptr)
This is a constexpr reimplementation of a subset of std::bitset.
Definition Bitset.h:30
constexpr bool any() const
Definition Bitset.h:113
constexpr Bitset & set()
Definition Bitset.h:81
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
void setDoesNotThrow()
MaybeAlign getRetAlign() const
Extract the alignment of the return value.
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
OperandBundleUse getOperandBundleAt(unsigned Index) const
Return the operand bundle at a specific index.
std::optional< OperandBundleUse > getOperandBundle(StringRef Name) const
Return an operand bundle by name, if present.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
bool isInAllocaArgument(unsigned ArgNo) const
Determine whether this argument is passed in an alloca.
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
bool hasRetAttr(Attribute::AttrKind Kind) const
Determine whether the return value has the given attribute.
unsigned getNumOperandBundles() const
Return the number of operand bundles associated with this User.
uint64_t getParamDereferenceableBytes(unsigned i) const
Extract the number of dereferenceable bytes for a call or parameter (0=unknown).
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
User::op_iterator arg_begin()
Return the iterator pointing to the beginning of the argument list.
LLVM_ABI bool isIndirectCall() const
Return true if the callsite is an indirect call.
static LLVM_ABI CallBase * removeOperandBundleAt(CallBase *CB, size_t Offset, InsertPosition InsertPtr=nullptr)
void setNotConvergent()
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Attribute getFnAttr(StringRef Kind) const
Get the attribute of a given kind for the function.
bool doesNotThrow() const
Determine if the call cannot unwind.
void addRetAttr(Attribute::AttrKind Kind)
Adds the attribute to the return value.
Value * getArgOperand(unsigned i) const
User::op_iterator arg_end()
Return the iterator pointing to the end of the argument list.
bool isConvergent() const
Determine if the invoke is convergent.
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
Value * getReturnedArgOperand() const
If one of the arguments has the 'returned' attribute, returns its operand value.
static LLVM_ABI CallBase * Create(CallBase *CB, ArrayRef< OperandBundleDef > Bundles, InsertPosition InsertPt=nullptr)
Create a clone of CB with a different set of operand bundles and insert it before InsertPt.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
static LLVM_ABI CallBase * removeOperandBundle(CallBase *CB, uint32_t ID, InsertPosition InsertPt=nullptr)
Create a clone of CB with operand bundle ID removed.
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
LLVM_ABI Function * getCaller()
Helper to get the caller (the parent function).
CallBr instruction, tracking function calls that may not return control but instead transfer it to a ...
static CallBrInst * Create(FunctionType *Ty, Value *Func, BasicBlock *DefaultDest, ArrayRef< BasicBlock * > IndirectDests, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This class represents a function call, abstracting a target machine's calling convention.
bool isNoTailCall() const
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
bool isMustTailCall() const
static LLVM_ABI Instruction::CastOps getCastOpcode(const Value *Val, bool SrcIsSigned, Type *Ty, bool DstIsSigned)
Returns the opcode necessary to cast Val into Ty using usual casting rules.
static LLVM_ABI CastInst * CreateIntegerCast(Value *S, Type *Ty, bool isSigned, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a ZExt, BitCast, or Trunc for int -> int casts.
static LLVM_ABI bool isBitOrNoopPointerCastable(Type *SrcTy, Type *DestTy, const DataLayout &DL)
Check whether a bitcast, inttoptr, or ptrtoint cast between these types is valid and a no-op.
static LLVM_ABI CastInst * CreateBitOrPointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, a PtrToInt, or an IntToPTr cast instruction.
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Predicate getNonStrictPredicate() const
For example, SGT -> SGE, SLT -> SLE, ULT -> ULE, UGT -> UGE.
Definition InstrTypes.h:934
Predicate getUnorderedPredicate() const
Definition InstrTypes.h:874
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getSub(Constant *C1, Constant *C2, bool HasNUW=false, bool HasNSW=false)
static LLVM_ABI Constant * getNeg(Constant *C, bool HasNSW=false)
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getLimitedValue(uint64_t Limit=~0ULL) const
getLimitedValue - If the value is smaller than the specified limit, return it, otherwise return the l...
Definition Constants.h:269
static LLVM_ABI ConstantInt * getTrue(LLVMContext &Context)
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
static LLVM_ABI ConstantInt * getBool(LLVMContext &Context, bool V)
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI ConstantPtrAuth * get(Constant *Ptr, ConstantInt *Key, ConstantInt *Disc, Constant *AddrDisc, Constant *DeactivationSymbol)
Return a pointer signed with the specified parameters.
This class represents a range of values.
LLVM_ABI ConstantRange zextOrTrunc(uint32_t BitWidth) const
Make this range have the bit width given by BitWidth.
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI bool icmp(CmpInst::Predicate Pred, const ConstantRange &Other) const
Does the predicate Pred hold between ranges this and Other?
LLVM_ABI ConstantRange multiply(const ConstantRange &Other, unsigned NoWrapKind=0) const
Return a new range representing the possible values resulting from a multiplication of a value in thi...
LLVM_ABI bool contains(const APInt &Val) const
Return true if the specified value is in the set.
uint32_t getBitWidth() const
Get the bit width of this ConstantRange.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Record of a variable value-assignment, aka a non instruction representation of the dbg....
std::pair< iterator, bool > try_emplace(KeyT &&Key, Ts &&...Args)
Definition DenseMap.h:299
unsigned size() const
Definition DenseMap.h:172
size_type count(const_arg_type_t< KeyT > Val) const
Return 1 if the specified key is in the map, 0 otherwise.
Definition DenseMap.h:219
bool contains(const_arg_type_t< KeyT > Val) const
Return true if the specified key is in the map, false otherwise.
Definition DenseMap.h:214
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static FMFSource intersect(Value *A, Value *B)
Intersect the FMF from two instructions.
Definition IRBuilder.h:107
This class represents an extension of floating point types.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool allowReassoc() const
Flag queries.
Definition FMF.h:64
An instruction for ordering other memory operations.
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this fence instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this fence instruction.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
Type::subtype_iterator param_iterator
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
bool isConvergent() const
Determine if the call is convergent.
Definition Function.h:592
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:211
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:272
AttributeList getAttributes() const
Return the attribute list for this Function.
Definition Function.h:328
bool doesNotThrow() const
Determine if the function cannot unwind.
Definition Function.h:576
bool isIntrinsic() const
isIntrinsic - Returns true if the function's name starts with "llvm.".
Definition Function.h:251
LLVM_ABI Value * getBasePtr() const
unsigned getBasePtrIndex() const
The index into the associate statepoint's argument list which contains the base pointer of the pointe...
LLVM_ABI Value * getDerivedPtr() const
unsigned getDerivedPtrIndex() const
The index into the associate statepoint's argument list which contains the pointer whose relocation t...
std::vector< const GCRelocateInst * > getGCRelocates() const
Get list of all gc reloactes linked to this statepoint May contain several relocations for the same b...
Definition Statepoint.h:206
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this GlobalObject.
LLVM_ABI bool isDeclaration() const
Return true if the primary definition of this global value is outside of the current translation unit...
Definition Globals.cpp:408
PointerType * getType() const
Global values are always pointers.
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
LLVM_ABI Value * CreateLaunderInvariantGroup(Value *Ptr)
Create a launder.invariant.group intrinsic call.
ConstantInt * getTrue()
Get the constant value for i1 true.
Definition IRBuilder.h:457
LLVM_ABI Value * CreateBinaryIntrinsic(Intrinsic::ID ID, Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 2 operands which is mangled on the first type.
Value * CreateSub(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Definition IRBuilder.h:1439
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Definition IRBuilder.h:2121
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
Definition IRBuilder.h:2684
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
ConstantInt * getFalse()
Get the constant value for i1 false.
Definition IRBuilder.h:462
Value * CreateICmp(CmpInst::Predicate P, Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2485
Value * CreateAddrSpaceCast(Value *V, Type *DestTy, const Twine &Name="")
Definition IRBuilder.h:2248
LLVM_ABI Value * CreateUnaryIntrinsic(Intrinsic::ID ID, Value *Op, FMFSource FMFSource={}, const Twine &Name="")
Create a call to intrinsic ID with 1 operand which is mangled on its type.
LLVM_ABI Value * CreateStripInvariantGroup(Value *Ptr)
Create a strip.invariant.group intrinsic call.
static InsertValueInst * Create(Value *Agg, Value *Val, ArrayRef< unsigned > Idxs, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
Instruction * foldOpIntoPhi(Instruction &I, PHINode *PN, bool AllowMultipleUses=false)
Given a binary operator, cast instruction, or select which has a PHI node as operand #0,...
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
bool SimplifyDemandedBits(Instruction *I, unsigned Op, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0) override
This form of SimplifyDemandedBits simplifies the specified instruction operand if possible,...
Instruction * FoldOpIntoSelect(Instruction &Op, SelectInst *SI, bool FoldWithMultiUse=false, bool SimplifyBothArms=false)
Given an instruction with a select as one operand and a constant as the other operand,...
Instruction * SimplifyAnyMemSet(AnyMemSetInst *MI)
Instruction * foldItoFPtoI(FPToIntTy &FI)
fpto{s/u}i.sat --> X or zext(X) or sext(X) or trunc(X) This is safe if the intermediate type has enou...
Instruction * visitFree(CallInst &FI, Value *FreedOp)
Instruction * visitCallBrInst(CallBrInst &CBI)
Instruction * eraseInstFromFunction(Instruction &I) override
Combiner aware instruction erasure.
Value * foldReversedIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are reverses, try to pull the reverse after the intrinsic.
Value * tryGetLog2(Value *Op, bool AssumeNonZero)
Instruction * visitFenceInst(FenceInst &FI)
Instruction * foldShuffledIntrinsicOperands(IntrinsicInst *II)
If all arguments of the intrinsic are unary shuffles with the same mask, try to shuffle after the int...
Instruction * visitInvokeInst(InvokeInst &II)
bool SimplifyDemandedInstructionBits(Instruction &Inst)
Tries to simplify operands to an integer instruction based on its demanded bits.
void CreateNonTerminatorUnreachable(Instruction *InsertAt)
Create and insert the idiom we use to indicate a block is unreachable without having to rewrite the C...
Instruction * visitVAEndInst(VAEndInst &I)
Instruction * matchBSwapOrBitReverse(Instruction &I, bool MatchBSwaps, bool MatchBitReversals)
Given an initial instruction, check to see if it is the root of a bswap/bitreverse idiom.
Constant * unshuffleConstant(ArrayRef< int > ShMask, Constant *C, VectorType *NewCTy)
Find a constant NewC that has property: shuffle(NewC, poison, ShMask) = C for lanes that select NewC.
Instruction * visitAllocSite(Instruction &FI)
Instruction * SimplifyAnyMemTransfer(AnyMemTransferInst *MI)
OverflowResult computeOverflow(Instruction::BinaryOps BinaryOp, bool IsSigned, Value *LHS, Value *RHS, Instruction *CxtI) const
Instruction * visitCallInst(CallInst &CI)
CallInst simplification.
The core instruction combiner logic.
SimplifyQuery SQ
const DataLayout & getDataLayout() const
unsigned ComputeMaxSignificantBits(const Value *Op, const Instruction *CxtI=nullptr, unsigned Depth=0) const
bool isFreeToInvert(Value *V, bool WillInvertAllUses, bool &DoesConsume)
Return true if the specified value is free to invert (apply ~ to).
DominatorTree & getDominatorTree() const
BlockFrequencyInfo * BFI
TargetLibraryInfo & TLI
Instruction * InsertNewInstBefore(Instruction *New, BasicBlock::iterator Old)
Inserts an instruction New before instruction Old.
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
const DataLayout & DL
DomConditionCache DC
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CxtI, unsigned Depth=0) const
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
LLVM_ABI std::optional< Instruction * > targetInstCombineIntrinsic(IntrinsicInst &II)
AssumptionCache & AC
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
bool MaskedValueIsZero(const Value *V, const APInt &Mask, const Instruction *CxtI=nullptr, unsigned Depth=0) const
DominatorTree & DT
ProfileSummaryInfo * PSI
OptimizationRemarkEmitter & ORE
Value * getFreelyInverted(Value *V, bool WillInvertAllUses, BuilderTy *Builder, bool &DoesConsume)
const SimplifyQuery & getSimplifyQuery() const
bool isKnownToBeAPowerOfTwo(const Value *V, bool OrZero=false, const Instruction *CxtI=nullptr, unsigned Depth=0)
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void setHasNoUnsignedWrap(bool b=true)
Set or clear the nuw flag on this instruction, which must be an operator which supports this flag.
LLVM_ABI bool mayWriteToMemory() const LLVM_READONLY
Return true if this instruction may modify memory.
LLVM_ABI void copyIRFlags(const Value *V, bool IncludeWrapFlags=true)
Convenience method to copy supported exact, fast-math, and (optionally) wrapping flags from V to this...
LLVM_ABI void setHasNoSignedWrap(bool b=true)
Set or clear the nsw flag on this instruction, which must be an operator which supports this flag.
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI void setAAMetadata(const AAMDNodes &N)
Sets the AA metadata on this instruction from the AAMDNodes structure.
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void moveBefore(InstListType::iterator InsertPos)
Unlink this instruction from its current basic block and insert it into the basic block that MovePos ...
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
bool isTerminator() const
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
LLVM_ABI std::optional< InstListType::iterator > getInsertionPointAfterDef()
Get the first insertion point at which the result of this instruction is defined.
LLVM_ABI bool isIdenticalTo(const Instruction *I) const LLVM_READONLY
Return true if the specified instruction is exactly identical to the current one.
void setDebugLoc(DebugLoc Loc)
Set the debug location information for this instruction.
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
static LLVM_ABI IntegerType * get(LLVMContext &C, unsigned NumBits)
This static method is the primary way of constructing an IntegerType.
Definition Type.cpp:348
A wrapper class for inspecting calls to intrinsic functions.
Intrinsic::ID getIntrinsicID() const
Return the intrinsic ID of this intrinsic.
Invoke instruction.
static InvokeInst * Create(FunctionType *Ty, Value *Func, BasicBlock *IfNormal, BasicBlock *IfException, ArrayRef< Value * > Args, const Twine &NameStr, InsertPosition InsertBefore=nullptr)
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
An instruction for reading from memory.
Metadata node.
Definition Metadata.h:1069
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1565
static LLVM_ABI MDNode * getMostGenericFPMath(MDNode *A, MDNode *B)
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:614
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:110
static ICmpInst::Predicate getPredicate(Intrinsic::ID ID)
Returns the comparison predicate underlying the intrinsic.
ICmpInst::Predicate getPredicate() const
Returns the comparison predicate underlying the intrinsic.
bool isSigned() const
Whether the intrinsic is signed or unsigned.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
StringRef getName() const
Get a short "name" for the module.
Definition Module.h:311
unsigned getOpcode() const
Return the opcode for this Instruction or ConstantExpr.
Definition Operator.h:43
Utility class for integer operators which may exhibit overflow - Add, Sub, Mul, and Shl.
Definition Operator.h:78
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
bool isCommutative() const
Return true if the instruction is commutative.
Definition Operator.h:130
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
Represents a saturating add/sub intrinsic.
This class represents the LLVM 'select' instruction.
static SelectInst * Create(Value *C, Value *S1, Value *S2, const Twine &NameStr="", InsertPosition InsertBefore=nullptr, const Instruction *MDFrom=nullptr)
This instruction constructs a fixed permutation of two input vectors.
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
bool test(unsigned Idx) const
Returns true if bit Idx is set.
bool all() const
Returns true if all bits are set.
size_type size() const
Definition SmallPtrSet.h:99
size_type count(ConstPtrType Ptr) const
count - Return 1 if the specified pointer is in the set, 0 otherwise.
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
reference emplace_back(ArgTypes &&... Args)
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
void setVolatile(bool V)
Specify whether this is a volatile store or not.
void setAlignment(Align Align)
void setOrdering(AtomicOrdering Ordering)
Sets the ordering constraint of this store instruction.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
static constexpr size_t npos
Definition StringRef.h:58
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
LLVM_ABI size_t find_first_not_of(char C, size_t From=0) const
Find the first character in the string that is not C or npos if not found.
Class to represent struct types.
static LLVM_ABI bool isCallingConvCCompatible(CallBase *CI)
Returns true if call site / callee has cdecl-compatible calling conventions.
Provides information about what library functions are available for the current target.
This class represents a truncation of integer types.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:263
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:282
LLVM_ABI bool canLosslesslyBitCastTo(Type *Ty) const
Return true if this type could be converted with a lossless BitCast to type 'Ty'.
Definition Type.cpp:153
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:368
bool isStructTy() const
True if this is an instance of StructType.
Definition Type.h:276
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:197
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:257
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:106
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
static UnaryOperator * CreateWithCopiedFlags(UnaryOps Opc, Value *V, Instruction *CopyO, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:148
static UnaryOperator * CreateFNegFMF(Value *Op, Instruction *FMFSource, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Definition InstrTypes.h:156
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:36
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
This represents the llvm.va_end intrinsic.
static LLVM_ABI void ValueIsDeleted(Value *V)
Definition Value.cpp:1263
static LLVM_ABI void ValueIsRAUWd(Value *Old, Value *New)
Definition Value.cpp:1316
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:255
static constexpr uint64_t MaximumAlignment
Definition Value.h:799
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:439
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:258
iterator_range< user_iterator > users()
Definition Value.h:426
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:713
bool use_empty() const
Definition Value.h:346
static constexpr unsigned MaxAlignmentExponent
The maximum alignment for instructions.
Definition Value.h:798
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
ElementCount getElementCount() const
Return an ElementCount instance to represent the (possibly scalable) number of elements in the vector...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
static constexpr bool isKnownLT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:216
constexpr bool isFixed() const
Returns true if the quantity is not scaled by vscale.
Definition TypeSize.h:171
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
NodeTy * getNextNode()
Get the next node, or nullptr for the list tail.
Definition ilist_node.h:348
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char Attrs[]
Key for Kernel::Metadata::mAttrs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:81
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
match_combine_and< Ty... > m_CombineAnd(const Ty &...Ps)
Combine pattern matchers matching all of Ps patterns.
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_BSwap(const Opnd0 &Op0)
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
auto m_BitReverse(const Opnd0 &Op0)
auto m_PtrToIntOrAddr(const OpTy &Op)
Matches PtrToInt or PtrToAddr.
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
BinaryOp_match< LHS, RHS, Instruction::And, true > m_c_And(const LHS &L, const RHS &R)
Matches an And with LHS and RHS in either order.
CastInst_match< OpTy, TruncInst > m_Trunc(const OpTy &Op)
Matches Trunc.
BinaryOp_match< LHS, RHS, Instruction::Xor > m_Xor(const LHS &L, const RHS &R)
ap_match< APInt > m_APIntAllowPoison(const APInt *&Res)
Match APInt while allowing poison in splat vector constants.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWSub(const LHS &L, const RHS &R)
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
auto m_UMin(const Opnd0 &Op0, const Opnd1 &Op1)
match_deferred< Value > m_Deferred(Value *const &V)
Like m_Specific(), but works if the specific value to match is determined as part of the same match()...
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
auto match_fn(const Pattern &P)
A match functor that can be used as a UnaryPredicate in functional algorithms like all_of.
OverflowingBinaryOp_match< cst_pred_ty< is_zero_int >, ValTy, Instruction::Sub, OverflowingBinaryOperator::NoSignedWrap > m_NSWNeg(const ValTy &V)
Matches a 'Neg' as 'sub nsw 0, V'.
auto m_SMax(const Opnd0 &Op0, const Opnd1 &Op1)
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
ThreeOps_match< Cond, LHS, RHS, Instruction::Select > m_Select(const Cond &C, const LHS &L, const RHS &R)
Matches SelectInst.
cstfp_pred_ty< is_neg_zero_fp > m_NegZeroFP()
Match a floating-point negative zero.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
auto m_UMax(const Opnd0 &Op0, const Opnd1 &Op1)
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_CopySign(const Opnd0 &Op0, const Opnd1 &Op1)
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
BinOpPred_match< LHS, RHS, is_logical_shift_op > m_LogicalShift(const LHS &L, const RHS &R)
Matches logical shift operations.
auto m_Value()
Match an arbitrary value and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Xor, true > m_c_Xor(const LHS &L, const RHS &R)
Matches an Xor with LHS and RHS in either order.
auto m_Constant()
Match an arbitrary Constant and ignore it.
match_combine_or< match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > >, OpTy > m_ZExtOrSExtOrSelf(const OpTy &Op)
auto m_LogicalOr()
Matches L || R where L and R are arbitrary values.
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
cst_pred_ty< is_strictlypositive > m_StrictlyPositive()
Match an integer or vector of strictly positive values.
ThreeOps_match< decltype(m_Value()), LHS, RHS, Instruction::Select, true > m_c_Select(const LHS &L, const RHS &R)
Match Select(C, LHS, RHS) or Select(C, RHS, LHS)
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
SpecificCmpClass_match< LHS, RHS, ICmpInst > m_SpecificICmp(CmpPredicate MatchPred, const LHS &L, const RHS &R)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
OverflowingBinaryOp_match< LHS, RHS, Instruction::Shl, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWShl(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Mul, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWMul(const LHS &L, const RHS &R)
auto m_FShl(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
cst_pred_ty< is_negated_power2 > m_NegatedPower2()
Match a integer or vector negated power-of-2.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
cst_pred_ty< custom_checkfn< APInt > > m_CheckedInt(function_ref< bool(const APInt &)> CheckFn)
Match an integer or vector where CheckFn(ele) for each element is true.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_c_MaxOrMin(const LHS &L, const RHS &R)
OverflowingBinaryOp_match< LHS, RHS, Instruction::Sub, OverflowingBinaryOperator::NoUnsignedWrap > m_NUWSub(const LHS &L, const RHS &R)
auto m_SMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_FAbs(const Opnd0 &Op0)
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoSignedWrap >, DisjointOr_match< LHS, RHS > > m_NSWAddLike(const LHS &L, const RHS &R)
Match either "add nsw" or "or disjoint".
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
Exact_match< T > m_Exact(const T &SubPattern)
FNeg_match< OpTy > m_FNeg(const OpTy &X)
Match 'fneg X' as 'fsub -0.0, X'.
BinOpPred_match< LHS, RHS, is_shift_op > m_Shift(const LHS &L, const RHS &R)
Matches shift operations.
cstfp_pred_ty< is_pos_zero_fp > m_PosZeroFP()
Match a floating-point positive zero.
auto m_UnOp()
Match an arbitrary unary operation and ignore it.
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_MaxOrMin(const Opnd0 &Op0, const Opnd1 &Op1)
auto m_LogicalAnd()
Matches L && R where L and R are arbitrary values.
BinaryOp_match< LHS, RHS, Instruction::SRem > m_SRem(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
auto m_VecReverse(const Opnd0 &Op0)
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
BinaryOp_match< LHS, RHS, Instruction::Or, true > m_c_Or(const LHS &L, const RHS &R)
Matches an Or with LHS and RHS in either order.
match_combine_or< OverflowingBinaryOp_match< LHS, RHS, Instruction::Add, OverflowingBinaryOperator::NoUnsignedWrap >, DisjointOr_match< LHS, RHS > > m_NUWAddLike(const LHS &L, const RHS &R)
Match either "add nuw" or "or disjoint".
BinOpPred_match< LHS, RHS, is_bitwiselogic_op > m_BitwiseLogic(const LHS &L, const RHS &R)
Matches bitwise logic operations.
ElementWiseBitCast_match< OpTy > m_ElementWiseBitCast(const OpTy &Op)
BinaryOp_match< LHS, RHS, Instruction::Mul, true > m_c_Mul(const LHS &L, const RHS &R)
Matches a Mul with LHS and RHS in either order.
auto m_FShr(const Opnd0 &Op0, const Opnd1 &Op1, const Opnd2 &Op2)
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
SmallVector< DbgVariableRecord * > getDVRAssignmentMarkers(const Instruction *Inst)
Return a range of dbg_assign records for which Inst performs the assignment they encode.
Definition DebugInfo.h:205
initializer< Ty > init(const Ty &Val)
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:668
constexpr double e
DiagnosticInfoOptimizationBase::Argument NV
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Intrinsic::ID getInverseMinMaxIntrinsic(Intrinsic::ID MinMaxID)
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:345
@ Offset
Definition DWP.cpp:578
@ NeverOverflows
Never overflows.
@ AlwaysOverflowsHigh
Always overflows in the direction of signed/unsigned max value.
@ AlwaysOverflowsLow
Always overflows in the direction of signed/unsigned min value.
@ MayOverflow
May or may not overflow.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
LLVM_ABI Value * simplifyFMulInst(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for an FMul, fold the result or return null.
LLVM_ABI bool isValidAssumeForContext(const Instruction *I, const Instruction *CxtI, const DominatorTree *DT=nullptr, bool AllowEphemerals=false)
Return true if it is valid to use the assumptions provided by an assume intrinsic,...
LLVM_ABI APInt possiblyDemandedEltsInMask(Value *Mask)
Given a mask vector of the form <Y x i1>, return an APInt (of bitwidth Y) for each lane which may be ...
BundleAttr getBundleAttrFromOBU(OperandBundleUse OBU)
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
Definition STLExtras.h:2554
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool isRemovableAlloc(const CallBase *V, const TargetLibraryInfo *TLI)
Return true if this is a call to an allocation function that does not have side effects that we are r...
LLVM_ABI bool getConstantStringInfo(const Value *V, StringRef &Str, bool TrimAtNul=true)
This function computes the length of a null-terminated C string pointed to by V.
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
Definition MathExtras.h:224
LLVM_ABI Value * lowerObjectSizeCall(IntrinsicInst *ObjectSize, const DataLayout &DL, const TargetLibraryInfo *TLI, bool MustSucceed)
Try to turn a call to @llvm.objectsize into an integer value of the given Type.
LLVM_ABI AssumeSeparateStorageInfo getAssumeSeparateStorageInfo(OperandBundleUse)
LLVM_ABI Value * getAllocAlignment(const CallBase *V, const TargetLibraryInfo *TLI)
Gets the alignment argument for an aligned_alloc-like function, using either built-in knowledge based...
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
LLVM_READONLY APFloat maximum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximum semantics.
Definition APFloat.h:1783
LLVM_ABI Value * simplifyCall(CallBase *Call, Value *Callee, ArrayRef< Value * > Args, const SimplifyQuery &Q)
Given a callsite, callee, and arguments, fold the result or return null.
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:547
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isSafeToSpeculativelyExecute(const Instruction *I, const Instruction *CtxI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr, const TargetLibraryInfo *TLI=nullptr, bool UseVariableInfo=true, bool IgnoreUBImplyingAttrs=true)
Return true if the instruction does not have any effects besides calculating the result and does not ...
LLVM_ABI Value * getSplatValue(const Value *V)
Get splat value if the input is a splat vector or return nullptr.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
Definition MathExtras.h:358
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
Align getKnownAlignment(Value *V, const DataLayout &DL, const Instruction *CxtI=nullptr, AssumptionCache *AC=nullptr, const DominatorTree *DT=nullptr)
Try to infer an alignment for the specified pointer.
Definition Local.h:254
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
LLVM_ABI bool isSplatValue(const Value *V, int Index=-1, unsigned Depth=0)
Return true if each element of the vector value V is poisoned or equal to every other non-poisoned el...
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
Definition APFloat.h:1738
SelectPatternFlavor
Specific patterns of select instructions we can match.
@ SPF_ABS
Floating point maxnum.
@ SPF_NABS
Absolute value.
LLVM_ABI Constant * getLosslessUnsignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isModSet(const ModRefInfo MRI)
Definition ModRef.h:49
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1636
LLVM_READONLY APFloat minimumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimumNumber semantics.
Definition APFloat.h:1769
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
Definition APFloat.h:1683
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CxtI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
LLVM_ABI SelectPatternResult matchSelectPattern(Value *V, Value *&LHS, Value *&RHS, Instruction::CastOps *CastOp=nullptr, unsigned Depth=0)
Pattern match integer [SU]MIN, [SU]MAX and ABS idioms, returning the kind and providing the out param...
LLVM_ABI bool matchSimpleBinaryIntrinsicRecurrence(const IntrinsicInst *I, PHINode *&P, Value *&Init, Value *&OtherOp)
Attempt to match a simple value-accumulating recurrence of the form: llvm.intrinsic....
LLVM_ABI bool NullPointerIsDefined(const Function *F, unsigned AS=0)
Check whether null pointer dereferencing is considered undefined behavior for a given function or an ...
auto find_if_not(R &&Range, UnaryPredicate P)
Definition STLExtras.h:1777
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1753
bool isAtLeastOrStrongerThan(AtomicOrdering AO, AtomicOrdering Other)
LLVM_ABI Constant * getLosslessSignedTrunc(Constant *C, Type *DestTy, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
LLVM_ABI ConstantRange getVScaleRange(const Function *F, unsigned BitWidth)
Determine the possible constant range of vscale with the given bit width, based on the vscale_range f...
iterator_range< SplittingIterator > split(StringRef Str, StringRef Separator)
Split the specified string over a separator and return a range-compatible iterable over its partition...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI bool isNotCrossLaneOperation(const Instruction *I)
Return true if the instruction doesn't potentially cross vector lanes.
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
LLVM_ABI bool isKnownNonZero(const Value *V, const SimplifyQuery &Q, unsigned Depth=0)
Return true if the given value is known to be non-zero when defined.
constexpr int PoisonMaskElem
@ Mod
The access may modify the value stored in memory.
Definition ModRef.h:34
LLVM_ABI Value * simplifyFMAFMul(Value *LHS, Value *RHS, FastMathFlags FMF, const SimplifyQuery &Q, fp::ExceptionBehavior ExBehavior=fp::ebIgnore, RoundingMode Rounding=RoundingMode::NearestTiesToEven)
Given operands for the multiplication of a FMA, fold the result or return null.
@ Other
Any other memory.
Definition ModRef.h:68
LLVM_ABI Value * simplifyConstrainedFPCall(CallBase *Call, const SimplifyQuery &Q)
Given a constrained FP intrinsic call, tries to compute its simplified version.
LLVM_READONLY APFloat minnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 minNum semantics.
Definition APFloat.h:1719
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI AssumeNonNullInfo getAssumeNonNullInfo(OperandBundleUse)
@ Add
Sum of integers.
LLVM_ABI bool isVectorIntrinsicWithScalarOpAtArg(Intrinsic::ID ID, unsigned ScalarOpdIdx, const TargetTransformInfo *TTI)
Identifies if the vector form of the intrinsic has a scalar operand.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
LLVM_ABI ConstantRange computeConstantRangeIncludingKnownBits(const WithCache< const Value * > &V, bool ForSigned, const SimplifyQuery &SQ)
Combine constant ranges from computeConstantRange() and computeKnownBits().
DWARFExpression::Operation Op
bool isSafeToSpeculativelyExecuteWithVariableReplaced(const Instruction *I, bool IgnoreUBImplyingAttrs=true)
Don't use information from its non-constant operands.
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Return true if this function can prove that V does not have undef bits and is never poison.
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI Value * getFreedOperand(const CallBase *CB, const TargetLibraryInfo *TLI)
If this if a call to a free function, return the freed operand.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
Definition MathExtras.h:233
constexpr unsigned BitWidth
LLVM_ABI Constant * getLosslessInvCast(Constant *C, Type *InvCastTo, unsigned CastOp, const DataLayout &DL, PreservedCastFlags *Flags=nullptr)
Try to cast C to InvC losslessly, satisfying CastOp(InvC) equals C, or CastOp(InvC) is a refined valu...
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1947
LLVM_ABI std::optional< APInt > getAllocSize(const CallBase *CB, const TargetLibraryInfo *TLI, function_ref< const Value *(const Value *)> Mapper=[](const Value *V) { return V;})
Return the size of the requested allocation.
LLVM_ABI AssumeAlignInfo getAssumeAlignInfo(OperandBundleUse)
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool maskContainsAllOneOrUndef(Value *Mask)
Given a mask vector of i1, Return true if any of the elements of this predicate mask are known to be ...
LLVM_ABI std::optional< bool > isImpliedByDomCondition(const Value *Cond, const Instruction *ContextI, const DataLayout &DL)
Return the boolean condition value in the context of the given instruction if it is known based on do...
LLVM_ABI bool isDereferenceablePointer(const Value *V, Type *Ty, const SimplifyQuery &Q, bool IgnoreFree=false)
Equivalent to isDereferenceableAndAlignedPointer with an alignment of 1.
Definition Loads.cpp:264
LLVM_READONLY APFloat minimum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 minimum semantics.
Definition APFloat.h:1756
LLVM_ABI bool isKnownNegation(const Value *X, const Value *Y, bool NeedNSW=false, bool AllowPoison=true)
Return true if the two given values are negation.
LLVM_READONLY APFloat maximumnum(const APFloat &A, const APFloat &B)
Implements IEEE 754-2019 maximumNumber semantics.
Definition APFloat.h:1796
LLVM_ABI const Value * getUnderlyingObject(const Value *V, unsigned MaxLookup=MaxLookupSearchDepth)
This method strips off any GEP address adjustments, pointer casts or llvm.threadlocal....
LLVM_ABI AssumeDereferenceableInfo getAssumeDereferenceableInfo(OperandBundleUse)
LLVM_ABI bool isKnownNonNegative(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Returns true if the give value is known to be non-negative.
LLVM_ABI AssumeNoUndefInfo getAssumeNoUndefInfo(OperandBundleUse)
LLVM_ABI bool isTriviallyVectorizable(Intrinsic::ID ID)
Identify if the intrinsic is trivially vectorizable.
LLVM_ABI std::optional< bool > computeKnownFPSignBit(const Value *V, const SimplifyQuery &SQ, unsigned Depth=0)
Return false if we can prove that the specified FP value's sign bit is 0.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define NC
Definition regutils.h:42
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
Definition Metadata.h:763
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
@ IEEE
IEEE-754 denormal numbers preserved.
Matching combinators.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
Align valueOrOne() const
For convenience, returns a valid alignment or 1 if undefined.
Definition Alignment.h:130
uint32_t getTagID() const
Return the tag of this operand bundle as an integer.
ArrayRef< Use > Inputs
SelectPatternFlavor Flavor
const DataLayout & DL
const Instruction * CxtI
SimplifyQuery getWithInstruction(const Instruction *I) const