LLVM 24.0.0git
X86ISelDAGToDAG.cpp
Go to the documentation of this file.
1//===- X86ISelDAGToDAG.cpp - A DAG pattern matching inst selector for X86 -===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines a DAG pattern matching instruction selector for X86,
10// converting from a legalized dag to a X86 dag.
11//
12//===----------------------------------------------------------------------===//
13
14#include "X86.h"
16#include "X86Subtarget.h"
17#include "X86TargetMachine.h"
18#include "llvm/ADT/Statistic.h"
22#include "llvm/Config/llvm-config.h"
24#include "llvm/IR/Function.h"
26#include "llvm/IR/Intrinsics.h"
27#include "llvm/IR/IntrinsicsX86.h"
28#include "llvm/IR/Module.h"
29#include "llvm/IR/Type.h"
30#include "llvm/Support/Debug.h"
34#include <cstdint>
35#include <optional>
36
37using namespace llvm;
38
39#define DEBUG_TYPE "x86-isel"
40#define PASS_NAME "X86 DAG->DAG Instruction Selection"
41
42STATISTIC(NumLoadMoved, "Number of loads moved below TokenFactor");
43
44//===----------------------------------------------------------------------===//
45// Pattern Matcher Implementation
46//===----------------------------------------------------------------------===//
47
48namespace {
49 /// This corresponds to X86AddressMode, but uses SDValue's instead of register
50 /// numbers for the leaves of the matched tree.
51 struct X86ISelAddressMode {
52 enum {
53 RegBase,
54 FrameIndexBase
55 } BaseType = RegBase;
56
57 // This is really a union, discriminated by BaseType!
58 SDValue Base_Reg;
59 int Base_FrameIndex = 0;
60
61 unsigned Scale = 1;
62 SDValue IndexReg;
63 int32_t Disp = 0;
64 SDValue Segment;
65 const GlobalValue *GV = nullptr;
66 const Constant *CP = nullptr;
67 const BlockAddress *BlockAddr = nullptr;
68 const char *ES = nullptr;
69 MCSymbol *MCSym = nullptr;
70 int JT = -1;
71 Align Alignment; // CP alignment.
72 unsigned char SymbolFlags = X86II::MO_NO_FLAG; // X86II::MO_*
73 bool NegateIndex = false;
74 // True when this address is being matched to be emitted as a LEA rather
75 // than folded into a memory operand. Unlike a memory operand, a LEA turns
76 // the folded arithmetic into real instructions, so it is not profitable to
77 // split an already-materialized (multi-use) value here. (Issue #51707)
78 bool IsForLEA = false;
79
80 X86ISelAddressMode() = default;
81
82 bool hasSymbolicDisplacement() const {
83 return GV != nullptr || CP != nullptr || ES != nullptr ||
84 MCSym != nullptr || JT != -1 || BlockAddr != nullptr;
85 }
86
87 bool hasBaseOrIndexReg() const {
88 return BaseType == FrameIndexBase ||
89 IndexReg.getNode() != nullptr || Base_Reg.getNode() != nullptr;
90 }
91
92 /// Return true if this addressing mode is already RIP-relative.
93 bool isRIPRelative() const {
94 if (BaseType != RegBase) return false;
95 if (RegisterSDNode *RegNode =
96 dyn_cast_or_null<RegisterSDNode>(Base_Reg.getNode()))
97 return RegNode->getReg() == X86::RIP;
98 return false;
99 }
100
101 void setBaseReg(SDValue Reg) {
102 BaseType = RegBase;
103 Base_Reg = Reg;
104 }
105
106#if !defined(NDEBUG) || defined(LLVM_ENABLE_DUMP)
107 void dump(SelectionDAG *DAG = nullptr) {
108 dbgs() << "X86ISelAddressMode " << this << '\n';
109 dbgs() << "Base_Reg ";
110 if (Base_Reg.getNode())
111 Base_Reg.getNode()->dump(DAG);
112 else
113 dbgs() << "nul\n";
114 if (BaseType == FrameIndexBase)
115 dbgs() << " Base.FrameIndex " << Base_FrameIndex << '\n';
116 dbgs() << " Scale " << Scale << '\n'
117 << "IndexReg ";
118 if (NegateIndex)
119 dbgs() << "negate ";
120 if (IndexReg.getNode())
121 IndexReg.getNode()->dump(DAG);
122 else
123 dbgs() << "nul\n";
124 dbgs() << " Disp " << Disp << '\n'
125 << "GV ";
126 if (GV)
127 GV->dump();
128 else
129 dbgs() << "nul";
130 dbgs() << " CP ";
131 if (CP)
132 CP->dump();
133 else
134 dbgs() << "nul";
135 dbgs() << '\n'
136 << "ES ";
137 if (ES)
138 dbgs() << ES;
139 else
140 dbgs() << "nul";
141 dbgs() << " MCSym ";
142 if (MCSym)
143 dbgs() << MCSym;
144 else
145 dbgs() << "nul";
146 dbgs() << " JT" << JT << " Align" << Alignment.value() << '\n';
147 }
148#endif
149 };
150}
151
152namespace {
153 //===--------------------------------------------------------------------===//
154 /// ISel - X86-specific code to select X86 machine instructions for
155 /// SelectionDAG operations.
156 ///
157 class X86DAGToDAGISel final : public SelectionDAGISel {
158 /// Keep a pointer to the X86Subtarget around so that we can
159 /// make the right decision when generating code for different targets.
160 const X86Subtarget *Subtarget;
161
162 /// If true, selector should try to optimize for minimum code size.
163 bool OptForMinSize;
164
165 /// Disable direct TLS access through segment registers.
166 bool IndirectTlsSegRefs;
167
168 public:
169 X86DAGToDAGISel() = delete;
170
171 explicit X86DAGToDAGISel(X86TargetMachine &tm, CodeGenOptLevel OptLevel)
172 : SelectionDAGISel(tm, OptLevel), Subtarget(nullptr),
173 OptForMinSize(false), IndirectTlsSegRefs(false) {}
174
175 bool runOnMachineFunction(MachineFunction &MF) override {
176 // Reset the subtarget each time through.
177 Subtarget = &MF.getSubtarget<X86Subtarget>();
178 IndirectTlsSegRefs = MF.getFunction().hasFnAttribute(
179 "indirect-tls-seg-refs");
180
181 // OptFor[Min]Size are used in pattern predicates that isel is matching.
182 OptForMinSize = MF.getFunction().hasMinSize();
184 }
185
186 void emitFunctionEntryCode() override;
187
188 bool IsProfitableToFold(SDValue N, SDNode *U, SDNode *Root) const override;
189
190 void PreprocessISelDAG() override;
191 void PostprocessISelDAG() override;
192
193// Include the pieces autogenerated from the target description.
194#include "X86GenDAGISel.inc"
195
196 private:
197 void Select(SDNode *N) override;
198
199 bool foldOffsetIntoAddress(uint64_t Offset, X86ISelAddressMode &AM);
200 bool matchLoadInAddress(LoadSDNode *N, X86ISelAddressMode &AM,
201 bool AllowSegmentRegForX32 = false);
202 bool matchWrapper(SDValue N, X86ISelAddressMode &AM);
203 bool matchAddress(SDValue N, X86ISelAddressMode &AM);
204 bool matchVectorAddress(SDValue N, X86ISelAddressMode &AM);
205 bool matchAdd(SDValue &N, X86ISelAddressMode &AM, unsigned Depth);
206 bool hasMaterializingUse(SDValue V) const;
207 SDValue matchIndexRecursively(SDValue N, X86ISelAddressMode &AM,
208 unsigned Depth);
209 bool matchAddressRecursively(SDValue N, X86ISelAddressMode &AM,
210 unsigned Depth);
211 bool matchVectorAddressRecursively(SDValue N, X86ISelAddressMode &AM,
212 unsigned Depth);
213 bool matchAddressBase(SDValue N, X86ISelAddressMode &AM);
214 bool selectAddr(SDNode *Parent, SDValue N, SDValue &Base, SDValue &Scale,
215 SDValue &Index, SDValue &Disp, SDValue &Segment,
216 bool HasNDDM = true);
217 bool selectNDDAddr(SDNode *Parent, SDValue N, SDValue &Base, SDValue &Scale,
218 SDValue &Index, SDValue &Disp, SDValue &Segment);
219 bool selectVectorAddr(MemSDNode *Parent, SDValue BasePtr, SDValue IndexOp,
220 SDValue ScaleOp, SDValue &Base, SDValue &Scale,
221 SDValue &Index, SDValue &Disp, SDValue &Segment);
222 bool selectMOV64Imm32(SDValue N, SDValue &Imm);
223 bool selectLEAAddr(SDValue N, SDValue &Base,
224 SDValue &Scale, SDValue &Index, SDValue &Disp,
225 SDValue &Segment);
226 bool selectLEA64_Addr(SDValue N, SDValue &Base, SDValue &Scale,
227 SDValue &Index, SDValue &Disp, SDValue &Segment);
228 bool selectTLSADDRAddr(SDValue N, SDValue &Base,
229 SDValue &Scale, SDValue &Index, SDValue &Disp,
230 SDValue &Segment);
231 bool selectRelocImm(SDValue N, SDValue &Op);
232
233 bool tryFoldLoad(SDNode *Root, SDNode *P, SDValue N,
234 SDValue &Base, SDValue &Scale,
235 SDValue &Index, SDValue &Disp,
236 SDValue &Segment);
237
238 // Convenience method where P is also root.
239 bool tryFoldLoad(SDNode *P, SDValue N,
240 SDValue &Base, SDValue &Scale,
241 SDValue &Index, SDValue &Disp,
242 SDValue &Segment) {
243 return tryFoldLoad(P, P, N, Base, Scale, Index, Disp, Segment);
244 }
245
246 bool tryFoldBroadcast(SDNode *Root, SDNode *P, SDValue N,
247 SDValue &Base, SDValue &Scale,
248 SDValue &Index, SDValue &Disp,
249 SDValue &Segment);
250
251 bool isProfitableToFormMaskedOp(SDNode *N) const;
252
253 /// Implement addressing mode selection for inline asm expressions.
254 bool SelectInlineAsmMemoryOperand(const SDValue &Op,
255 InlineAsm::ConstraintCode ConstraintID,
256 std::vector<SDValue> &OutOps) override;
257
258 void emitSpecialCodeForMain();
259
260 inline void getAddressOperands(X86ISelAddressMode &AM, const SDLoc &DL,
261 MVT VT, SDValue &Base, SDValue &Scale,
262 SDValue &Index, SDValue &Disp,
263 SDValue &Segment) {
264 if (AM.BaseType == X86ISelAddressMode::FrameIndexBase)
265 Base = CurDAG->getTargetFrameIndex(
266 AM.Base_FrameIndex, TLI->getPointerTy(CurDAG->getDataLayout()));
267 else if (AM.Base_Reg.getNode())
268 Base = AM.Base_Reg;
269 else
270 Base = CurDAG->getRegister(0, VT);
271
272 Scale = getI8Imm(AM.Scale, DL);
273
274#define GET_ND_IF_ENABLED(OPC) (Subtarget->hasNDD() ? OPC##_ND : OPC)
275#define GET_NDM_IF_ENABLED(OPC) \
276 (Subtarget->hasNDD() && Subtarget->hasNDDM() ? OPC##_ND : OPC)
277 // Negate the index if needed.
278 if (AM.NegateIndex) {
279 unsigned NegOpc;
280 switch (VT.SimpleTy) {
281 default:
282 llvm_unreachable("Unsupported VT!");
283 case MVT::i64:
284 NegOpc = GET_ND_IF_ENABLED(X86::NEG64r);
285 break;
286 case MVT::i32:
287 NegOpc = GET_ND_IF_ENABLED(X86::NEG32r);
288 break;
289 case MVT::i16:
290 NegOpc = GET_ND_IF_ENABLED(X86::NEG16r);
291 break;
292 case MVT::i8:
293 NegOpc = GET_ND_IF_ENABLED(X86::NEG8r);
294 break;
295 }
296 SDValue Neg = SDValue(CurDAG->getMachineNode(NegOpc, DL, VT, MVT::i32,
297 AM.IndexReg), 0);
298 AM.IndexReg = Neg;
299 }
300
301 if (AM.IndexReg.getNode())
302 Index = AM.IndexReg;
303 else
304 Index = CurDAG->getRegister(0, VT);
305
306 // These are 32-bit even in 64-bit mode since RIP-relative offset
307 // is 32-bit.
308 if (AM.GV)
309 Disp = CurDAG->getTargetGlobalAddress(AM.GV, SDLoc(),
310 MVT::i32, AM.Disp,
311 AM.SymbolFlags);
312 else if (AM.CP)
313 Disp = CurDAG->getTargetConstantPool(AM.CP, MVT::i32, AM.Alignment,
314 AM.Disp, AM.SymbolFlags);
315 else if (AM.ES) {
316 assert(!AM.Disp && "Non-zero displacement is ignored with ES.");
317 Disp = CurDAG->getTargetExternalSymbol(AM.ES, MVT::i32, AM.SymbolFlags);
318 } else if (AM.MCSym) {
319 assert(!AM.Disp && "Non-zero displacement is ignored with MCSym.");
320 assert(AM.SymbolFlags == 0 && "oo");
321 Disp = CurDAG->getMCSymbol(AM.MCSym, MVT::i32);
322 } else if (AM.JT != -1) {
323 assert(!AM.Disp && "Non-zero displacement is ignored with JT.");
324 Disp = CurDAG->getTargetJumpTable(AM.JT, MVT::i32, AM.SymbolFlags);
325 } else if (AM.BlockAddr)
326 Disp = CurDAG->getTargetBlockAddress(AM.BlockAddr, MVT::i32, AM.Disp,
327 AM.SymbolFlags);
328 else
329 Disp = CurDAG->getSignedTargetConstant(AM.Disp, DL, MVT::i32);
330
331 if (AM.Segment.getNode())
332 Segment = AM.Segment;
333 else
334 Segment = CurDAG->getRegister(0, MVT::i16);
335 }
336
337 // Utility function to determine whether it is AMX SDNode right after
338 // lowering but before ISEL.
339 bool isAMXSDNode(SDNode *N) const {
340 // Check if N is AMX SDNode:
341 // 1. check result type;
342 // 2. check operand type;
343 for (unsigned Idx = 0, E = N->getNumValues(); Idx != E; ++Idx) {
344 if (N->getValueType(Idx) == MVT::x86amx)
345 return true;
346 }
347 for (unsigned Idx = 0, E = N->getNumOperands(); Idx != E; ++Idx) {
348 SDValue Op = N->getOperand(Idx);
349 if (Op.getValueType() == MVT::x86amx)
350 return true;
351 }
352 return false;
353 }
354
355 // Utility function to determine whether we should avoid selecting
356 // immediate forms of instructions for better code size or not.
357 // At a high level, we'd like to avoid such instructions when
358 // we have similar constants used within the same basic block
359 // that can be kept in a register.
360 //
361 bool shouldAvoidImmediateInstFormsForSize(SDNode *N) const {
362 uint32_t UseCount = 0;
363
364 // Do not want to hoist if we're not optimizing for size.
365 // TODO: We'd like to remove this restriction.
366 // See the comment in X86InstrInfo.td for more info.
367 if (!CurDAG->shouldOptForSize())
368 return false;
369
370 // Walk all the users of the immediate.
371 for (const SDNode *User : N->users()) {
372 if (UseCount >= 2)
373 break;
374
375 // This user is already selected. Count it as a legitimate use and
376 // move on.
377 if (User->isMachineOpcode()) {
378 UseCount++;
379 continue;
380 }
381
382 // We want to count stores of immediates as real uses.
383 if (User->getOpcode() == ISD::STORE &&
384 User->getOperand(1).getNode() == N) {
385 UseCount++;
386 continue;
387 }
388
389 // We don't currently match users that have > 2 operands (except
390 // for stores, which are handled above)
391 // Those instruction won't match in ISEL, for now, and would
392 // be counted incorrectly.
393 // This may change in the future as we add additional instruction
394 // types.
395 if (User->getNumOperands() != 2)
396 continue;
397
398 // If this is a sign-extended 8-bit integer immediate used in an ALU
399 // instruction, there is probably an opcode encoding to save space.
401 if (C && isInt<8>(C->getSExtValue()))
402 continue;
403
404 // Immediates that are used for offsets as part of stack
405 // manipulation should be left alone. These are typically
406 // used to indicate SP offsets for argument passing and
407 // will get pulled into stores/pushes (implicitly).
408 if (User->getOpcode() == X86ISD::ADD ||
409 User->getOpcode() == ISD::ADD ||
410 User->getOpcode() == X86ISD::SUB ||
411 User->getOpcode() == ISD::SUB) {
412
413 // Find the other operand of the add/sub.
414 SDValue OtherOp = User->getOperand(0);
415 if (OtherOp.getNode() == N)
416 OtherOp = User->getOperand(1);
417
418 // Don't count if the other operand is SP.
419 RegisterSDNode *RegNode;
420 if (OtherOp->getOpcode() == ISD::CopyFromReg &&
422 OtherOp->getOperand(1).getNode())))
423 if ((RegNode->getReg() == X86::ESP) ||
424 (RegNode->getReg() == X86::RSP))
425 continue;
426 }
427
428 // ... otherwise, count this and move on.
429 UseCount++;
430 }
431
432 // If we have more than 1 use, then recommend for hoisting.
433 return (UseCount > 1);
434 }
435
436 /// Return a target constant with the specified value of type i8.
437 inline SDValue getI8Imm(unsigned Imm, const SDLoc &DL) {
438 return CurDAG->getTargetConstant(Imm, DL, MVT::i8);
439 }
440
441 /// Return a target constant with the specified value, of type i32.
442 inline SDValue getI32Imm(unsigned Imm, const SDLoc &DL) {
443 return CurDAG->getTargetConstant(Imm, DL, MVT::i32);
444 }
445
446 /// Return a target constant with the specified value, of type i64.
447 inline SDValue getI64Imm(uint64_t Imm, const SDLoc &DL) {
448 return CurDAG->getTargetConstant(Imm, DL, MVT::i64);
449 }
450
451 SDValue getExtractVEXTRACTImmediate(SDNode *N, unsigned VecWidth,
452 const SDLoc &DL) {
453 assert((VecWidth == 128 || VecWidth == 256) && "Unexpected vector width");
454 uint64_t Index = N->getConstantOperandVal(1);
455 MVT VecVT = N->getOperand(0).getSimpleValueType();
456 return getI8Imm((Index * VecVT.getScalarSizeInBits()) / VecWidth, DL);
457 }
458
459 SDValue getInsertVINSERTImmediate(SDNode *N, unsigned VecWidth,
460 const SDLoc &DL) {
461 assert((VecWidth == 128 || VecWidth == 256) && "Unexpected vector width");
462 uint64_t Index = N->getConstantOperandVal(2);
463 MVT VecVT = N->getSimpleValueType(0);
464 return getI8Imm((Index * VecVT.getScalarSizeInBits()) / VecWidth, DL);
465 }
466
467 SDValue getPermuteVINSERTCommutedImmediate(SDNode *N, unsigned VecWidth,
468 const SDLoc &DL) {
469 assert(VecWidth == 128 && "Unexpected vector width");
470 uint64_t Index = N->getConstantOperandVal(2);
471 MVT VecVT = N->getSimpleValueType(0);
472 uint64_t InsertIdx = (Index * VecVT.getScalarSizeInBits()) / VecWidth;
473 assert((InsertIdx == 0 || InsertIdx == 1) && "Bad insertf128 index");
474 // vinsert(0,sub,vec) -> [sub0][vec1] -> vperm2x128(0x30,vec,sub)
475 // vinsert(1,sub,vec) -> [vec0][sub0] -> vperm2x128(0x02,vec,sub)
476 return getI8Imm(InsertIdx ? 0x02 : 0x30, DL);
477 }
478
479 SDValue getSBBZero(SDNode *N) {
480 SDLoc dl(N);
481 MVT VT = N->getSimpleValueType(0);
482
483 // Create zero.
484 SDVTList VTs = CurDAG->getVTList(MVT::i32, MVT::i32);
485 SDValue Zero =
486 SDValue(CurDAG->getMachineNode(X86::MOV32r0, dl, VTs, {}), 0);
487 if (VT == MVT::i64) {
488 Zero = SDValue(
489 CurDAG->getMachineNode(
490 TargetOpcode::SUBREG_TO_REG, dl, MVT::i64, Zero,
491 CurDAG->getTargetConstant(X86::sub_32bit, dl, MVT::i32)),
492 0);
493 }
494
495 // Copy flags to the EFLAGS register and glue it to next node.
496 unsigned Opcode = N->getOpcode();
497 assert((Opcode == X86ISD::SBB || Opcode == X86ISD::SETCC_CARRY) &&
498 "Unexpected opcode for SBB materialization");
499 unsigned FlagOpIndex = Opcode == X86ISD::SBB ? 2 : 1;
500 SDValue EFLAGS =
501 CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, X86::EFLAGS,
502 N->getOperand(FlagOpIndex), SDValue());
503
504 // Create a 64-bit instruction if the result is 64-bits otherwise use the
505 // 32-bit version.
506 unsigned Opc = VT == MVT::i64 ? X86::SBB64rr : X86::SBB32rr;
507 MVT SBBVT = VT == MVT::i64 ? MVT::i64 : MVT::i32;
508 VTs = CurDAG->getVTList(SBBVT, MVT::i32);
509 return SDValue(
510 CurDAG->getMachineNode(Opc, dl, VTs,
511 {Zero, Zero, EFLAGS, EFLAGS.getValue(1)}),
512 0);
513 }
514
515 // Helper to detect unneeded and instructions on shift amounts. Called
516 // from PatFrags in tablegen.
517 bool isUnneededShiftMask(SDNode *N, unsigned Width) const {
518 assert(N->getOpcode() == ISD::AND && "Unexpected opcode");
519 const APInt &Val = N->getConstantOperandAPInt(1);
520
521 if (Val.countr_one() >= Width)
522 return true;
523
524 APInt Mask = Val | CurDAG->computeKnownBits(N->getOperand(0)).Zero;
525 return Mask.countr_one() >= Width;
526 }
527
528 // Any instruction that defines a 32-bit result zeroes the upper 32 bits of
529 // the 64-bit register. Truncate can be lowered to EXTRACT_SUBREG.
530 // CopyFromReg may be copying from a truncate. AssertSext/AssertZext/
531 // AssertAlign aren't saying anything about the upper 32 bits. FREEZE may
532 // be coming from a truncate. BitScan fall through values may not zero the
533 // upper bits correctly. Called from the def32 PatLeaf in tablegen.
534 bool isDef32(SDNode *N) const {
535 unsigned Opc = N->getOpcode();
536 return Opc != ISD::TRUNCATE && Opc != TargetOpcode::EXTRACT_SUBREG &&
539 Opc != ISD::FREEZE &&
540 !((Opc == X86ISD::BSF || Opc == X86ISD::BSR) &&
541 !N->getOperand(0).isUndef() &&
542 !isa<ConstantSDNode>(N->getOperand(0)));
543 }
544
545 /// Return an SDNode that returns the value of the global base register.
546 /// Output instructions required to initialize the global base register,
547 /// if necessary.
548 SDNode *getGlobalBaseReg();
549
550 /// Return a reference to the TargetMachine, casted to the target-specific
551 /// type.
552 const X86TargetMachine &getTargetMachine() const {
553 return static_cast<const X86TargetMachine &>(TM);
554 }
555
556 /// Return a reference to the TargetInstrInfo, casted to the target-specific
557 /// type.
558 const X86InstrInfo *getInstrInfo() const {
559 return Subtarget->getInstrInfo();
560 }
561
562 /// Return a condition code of the given SDNode
563 X86::CondCode getCondFromNode(SDNode *N) const;
564
565 /// Address-mode matching performs shift-of-and to and-of-shift
566 /// reassociation in order to expose more scaled addressing
567 /// opportunities.
568 bool ComplexPatternFuncMutatesDAG() const override {
569 return true;
570 }
571
572 bool isSExtAbsoluteSymbolRef(unsigned Width, SDNode *N) const;
573
574 // Indicates we should prefer to use a non-temporal load for this load.
575 bool useNonTemporalLoad(LoadSDNode *N) const {
576 if (!N->isNonTemporal())
577 return false;
578
579 unsigned StoreSize = N->getMemoryVT().getStoreSize();
580
581 if (N->getAlign().value() < StoreSize)
582 return false;
583
584 switch (StoreSize) {
585 default: llvm_unreachable("Unsupported store size");
586 case 4:
587 case 8:
588 return false;
589 case 16:
590 return Subtarget->hasSSE41();
591 case 32:
592 return Subtarget->hasAVX2();
593 case 64:
594 return Subtarget->hasAVX512();
595 }
596 }
597
598 bool foldLoadStoreIntoMemOperand(SDNode *Node);
599 MachineSDNode *matchBEXTRFromAndImm(SDNode *Node);
600 bool matchBitExtract(SDNode *Node);
601 bool shrinkAndImmediate(SDNode *N);
602 bool isMaskZeroExtended(SDNode *N) const;
603 bool tryShiftAmountMod(SDNode *N);
604 bool tryShrinkShlLogicImm(SDNode *N);
605 bool tryVPTERNLOG(SDNode *N);
606 bool matchVPTERNLOG(SDNode *Root, SDNode *ParentA, SDNode *ParentB,
607 SDNode *ParentC, SDValue A, SDValue B, SDValue C,
608 uint8_t Imm);
609 bool tryVPTESTM(SDNode *Root, SDValue Setcc, SDValue Mask);
610 bool tryMatchBitSelect(SDNode *N);
611
612 MachineSDNode *emitPCMPISTR(unsigned ROpc, unsigned MOpc, bool MayFoldLoad,
613 const SDLoc &dl, MVT VT, SDNode *Node);
614 MachineSDNode *emitPCMPESTR(unsigned ROpc, unsigned MOpc, bool MayFoldLoad,
615 const SDLoc &dl, MVT VT, SDNode *Node,
616 SDValue &InGlue);
617
618 bool tryOptimizeRem8Extend(SDNode *N);
619
620 bool onlyUsesZeroFlag(SDValue Flags) const;
621 bool hasNoSignFlagUses(SDValue Flags) const;
622 bool hasNoCarryFlagUses(SDValue Flags) const;
623 bool checkTCRetEnoughRegs(SDNode *N) const;
624 };
625
626 class X86DAGToDAGISelLegacy : public SelectionDAGISelLegacy {
627 public:
628 static char ID;
629 explicit X86DAGToDAGISelLegacy(X86TargetMachine &tm,
630 CodeGenOptLevel OptLevel)
631 : SelectionDAGISelLegacy(
632 ID, std::make_unique<X86DAGToDAGISel>(tm, OptLevel)) {}
633 };
634}
635
636char X86DAGToDAGISelLegacy::ID = 0;
637
638INITIALIZE_PASS(X86DAGToDAGISelLegacy, DEBUG_TYPE, PASS_NAME, false, false)
639
640// Returns true if this masked compare can be implemented legally with this
641// type.
642static bool isLegalMaskCompare(SDNode *N, const X86Subtarget *Subtarget) {
643 unsigned Opcode = N->getOpcode();
644 if (Opcode == X86ISD::CMPM || Opcode == X86ISD::CMPMM ||
645 Opcode == X86ISD::STRICT_CMPM || Opcode == ISD::SETCC ||
646 Opcode == X86ISD::CMPMM_SAE || Opcode == X86ISD::VFPCLASS) {
647 // We can get 256-bit 8 element types here without VLX being enabled. When
648 // this happens we will use 512-bit operations and the mask will not be
649 // zero extended.
650 EVT OpVT = N->getOperand(0).getValueType();
651 // The first operand of X86ISD::STRICT_CMPM is chain, so we need to get the
652 // second operand.
653 if (Opcode == X86ISD::STRICT_CMPM)
654 OpVT = N->getOperand(1).getValueType();
655 if (OpVT.is256BitVector() || OpVT.is128BitVector())
656 return Subtarget->hasVLX();
657
658 return true;
659 }
660 // Scalar opcodes use 128 bit registers, but aren't subject to the VLX check.
661 if (Opcode == X86ISD::VFPCLASSS || Opcode == X86ISD::FSETCCM ||
662 Opcode == X86ISD::FSETCCM_SAE)
663 return true;
664
665 return false;
666}
667
668// Returns true if we can assume the writer of the mask has zero extended it
669// for us.
670bool X86DAGToDAGISel::isMaskZeroExtended(SDNode *N) const {
671 // If this is an AND, check if we have a compare on either side. As long as
672 // one side guarantees the mask is zero extended, the AND will preserve those
673 // zeros.
674 if (N->getOpcode() == ISD::AND)
675 return isLegalMaskCompare(N->getOperand(0).getNode(), Subtarget) ||
676 isLegalMaskCompare(N->getOperand(1).getNode(), Subtarget);
677
678 return isLegalMaskCompare(N, Subtarget);
679}
680
681bool
682X86DAGToDAGISel::IsProfitableToFold(SDValue N, SDNode *U, SDNode *Root) const {
683 if (OptLevel == CodeGenOptLevel::None)
684 return false;
685
686 if (!N.hasOneUse())
687 return false;
688
689 if (N.getOpcode() != ISD::LOAD)
690 return true;
691
692 // Don't fold non-temporal loads if we have an instruction for them.
693 if (useNonTemporalLoad(cast<LoadSDNode>(N)))
694 return false;
695
696 // If N is a load, do additional profitability checks.
697 if (U == Root) {
698 switch (U->getOpcode()) {
699 default: break;
700 case X86ISD::ADD:
701 case X86ISD::ADC:
702 case X86ISD::SUB:
703 case X86ISD::SBB:
704 case X86ISD::AND:
705 case X86ISD::XOR:
706 case X86ISD::OR:
707 case ISD::ADD:
708 case ISD::UADDO_CARRY:
709 case ISD::AND:
710 case ISD::OR:
711 case ISD::XOR: {
712 SDValue Op1 = U->getOperand(1);
713
714 // If the other operand is a 8-bit immediate we should fold the immediate
715 // instead. This reduces code size.
716 // e.g.
717 // movl 4(%esp), %eax
718 // addl $4, %eax
719 // vs.
720 // movl $4, %eax
721 // addl 4(%esp), %eax
722 // The former is 2 bytes shorter. In case where the increment is 1, then
723 // the saving can be 4 bytes (by using incl %eax).
724 if (auto *Imm = dyn_cast<ConstantSDNode>(Op1)) {
725 if (Imm->getAPIntValue().isSignedIntN(8))
726 return false;
727
728 // If this is a 64-bit AND with an immediate that fits in 32-bits,
729 // prefer using the smaller and over folding the load. This is needed to
730 // make sure immediates created by shrinkAndImmediate are always folded.
731 // Ideally we would narrow the load during DAG combine and get the
732 // best of both worlds.
733 if (U->getOpcode() == ISD::AND &&
734 Imm->getAPIntValue().getBitWidth() == 64 &&
735 Imm->getAPIntValue().isIntN(32))
736 return false;
737
738 // If this really a zext_inreg that can be represented with a movzx
739 // instruction, prefer that.
740 // TODO: We could shrink the load and fold if it is non-volatile.
741 if (U->getOpcode() == ISD::AND &&
742 (Imm->getAPIntValue() == UINT8_MAX ||
743 Imm->getAPIntValue() == UINT16_MAX ||
744 Imm->getAPIntValue() == UINT32_MAX))
745 return false;
746
747 // ADD/SUB with can negate the immediate and use the opposite operation
748 // to fit 128 into a sign extended 8 bit immediate.
749 if ((U->getOpcode() == ISD::ADD || U->getOpcode() == ISD::SUB) &&
750 (-Imm->getAPIntValue()).isSignedIntN(8))
751 return false;
752
753 if ((U->getOpcode() == X86ISD::ADD || U->getOpcode() == X86ISD::SUB) &&
754 (-Imm->getAPIntValue()).isSignedIntN(8) &&
755 hasNoCarryFlagUses(SDValue(U, 1)))
756 return false;
757 }
758
759 // If the other operand is a TLS address, we should fold it instead.
760 // This produces
761 // movl %gs:0, %eax
762 // leal i@NTPOFF(%eax), %eax
763 // instead of
764 // movl $i@NTPOFF, %eax
765 // addl %gs:0, %eax
766 // if the block also has an access to a second TLS address this will save
767 // a load.
768 // FIXME: This is probably also true for non-TLS addresses.
769 if (Op1.getOpcode() == X86ISD::Wrapper) {
770 SDValue Val = Op1.getOperand(0);
772 return false;
773 }
774
775 // Don't fold load if this matches the BTS/BTR/BTC patterns.
776 // BTS: (or X, (shl 1, n))
777 // BTR: (and X, (rotl -2, n))
778 // BTC: (xor X, (shl 1, n))
779 if (U->getOpcode() == ISD::OR || U->getOpcode() == ISD::XOR) {
780 if (U->getOperand(0).getOpcode() == ISD::SHL &&
781 isOneConstant(U->getOperand(0).getOperand(0)))
782 return false;
783
784 if (U->getOperand(1).getOpcode() == ISD::SHL &&
785 isOneConstant(U->getOperand(1).getOperand(0)))
786 return false;
787 }
788 if (U->getOpcode() == ISD::AND) {
789 SDValue U0 = U->getOperand(0);
790 SDValue U1 = U->getOperand(1);
791 if (U0.getOpcode() == ISD::ROTL) {
793 if (C && C->getSExtValue() == -2)
794 return false;
795 }
796
797 if (U1.getOpcode() == ISD::ROTL) {
799 if (C && C->getSExtValue() == -2)
800 return false;
801 }
802 }
803
804 break;
805 }
806 case ISD::SHL:
807 case ISD::SRA:
808 case ISD::SRL:
809 // Don't fold a load into a shift by immediate. The BMI2 instructions
810 // support folding a load, but not an immediate. The legacy instructions
811 // support folding an immediate, but can't fold a load. Folding an
812 // immediate is preferable to folding a load.
813 if (isa<ConstantSDNode>(U->getOperand(1)))
814 return false;
815
816 break;
817 }
818 }
819
820 // Prevent folding a load if this can implemented with an insert_subreg or
821 // a move that implicitly zeroes.
822 if (Root->getOpcode() == ISD::INSERT_SUBVECTOR &&
823 isNullConstant(Root->getOperand(2)) &&
824 (Root->getOperand(0).isUndef() ||
826 return false;
827
828 return true;
829}
830
831// Indicates it is profitable to form an AVX512 masked operation. Returning
832// false will favor a masked register-register masked move or vblendm and the
833// operation will be selected separately.
834bool X86DAGToDAGISel::isProfitableToFormMaskedOp(SDNode *N) const {
835 assert(
836 (N->getOpcode() == ISD::VSELECT || N->getOpcode() == X86ISD::SELECTS) &&
837 "Unexpected opcode!");
838
839 // If the operation has additional users, the operation will be duplicated.
840 // Check the use count to prevent that.
841 // FIXME: Are there cheap opcodes we might want to duplicate?
842 return N->getOperand(1).hasOneUse();
843}
844
845/// Replace the original chain operand of the call with
846/// load's chain operand and move load below the call's chain operand.
848 SDValue Call, SDValue OrigChain) {
850 SDValue Chain = OrigChain.getOperand(0);
851 if (Chain.getNode() == Load.getNode())
852 Ops.push_back(Load.getOperand(0));
853 else {
854 assert(Chain.getOpcode() == ISD::TokenFactor &&
855 "Unexpected chain operand");
856 for (unsigned i = 0, e = Chain.getNumOperands(); i != e; ++i)
857 if (Chain.getOperand(i).getNode() == Load.getNode())
858 Ops.push_back(Load.getOperand(0));
859 else
860 Ops.push_back(Chain.getOperand(i));
861 SDValue NewChain =
862 CurDAG->getNode(ISD::TokenFactor, SDLoc(Load), MVT::Other, Ops);
863 Ops.clear();
864 Ops.push_back(NewChain);
865 }
866 Ops.append(OrigChain->op_begin() + 1, OrigChain->op_end());
867 CurDAG->UpdateNodeOperands(OrigChain.getNode(), Ops);
868 CurDAG->UpdateNodeOperands(Load.getNode(), Call.getOperand(0),
869 Load.getOperand(1), Load.getOperand(2));
870
871 Ops.clear();
872 Ops.push_back(SDValue(Load.getNode(), 1));
873 Ops.append(Call->op_begin() + 1, Call->op_end());
874 CurDAG->UpdateNodeOperands(Call.getNode(), Ops);
875}
876
877/// Return true if call address is a load and it can be
878/// moved below CALLSEQ_START and the chains leading up to the call.
879/// Return the CALLSEQ_START by reference as a second output.
880/// In the case of a tail call, there isn't a callseq node between the call
881/// chain and the load.
882static bool isCalleeLoad(SDValue Callee, SDValue &Chain, bool HasCallSeq) {
883 // The transformation is somewhat dangerous if the call's chain was glued to
884 // the call. After MoveBelowOrigChain the load is moved between the call and
885 // the chain, this can create a cycle if the load is not folded. So it is
886 // *really* important that we are sure the load will be folded.
887 if (Callee.getNode() == Chain.getNode() || !Callee.hasOneUse())
888 return false;
889 auto *LD = dyn_cast<LoadSDNode>(Callee.getNode());
890 if (!LD ||
891 !LD->isSimple() ||
892 LD->getAddressingMode() != ISD::UNINDEXED ||
893 LD->getExtensionType() != ISD::NON_EXTLOAD)
894 return false;
895
896 // If the load's outgoing chain has more than one use, we can't (currently)
897 // move the load since we'd most likely create a loop. TODO: Maybe it could
898 // work if moveBelowOrigChain() updated *all* the chain users.
899 if (!Callee.getValue(1).hasOneUse())
900 return false;
901
902 // Now let's find the callseq_start.
903 while (HasCallSeq && Chain.getOpcode() != ISD::CALLSEQ_START) {
904 if (!Chain.hasOneUse())
905 return false;
906 Chain = Chain.getOperand(0);
907 }
908
909 while (true) {
910 if (!Chain.getNumOperands())
911 return false;
912
913 // It's not safe to move the callee (a load) across e.g. a store.
914 // Conservatively abort if the chain contains a node other than the ones
915 // below.
916 switch (Chain.getNode()->getOpcode()) {
918 case ISD::CopyToReg:
919 case ISD::LOAD:
920 break;
921 default:
922 return false;
923 }
924
925 if (Chain.getOperand(0).getNode() == Callee.getNode())
926 return true;
927 if (Chain.getOperand(0).getOpcode() == ISD::TokenFactor &&
928 Chain.getOperand(0).getValue(0).hasOneUse() &&
929 Callee.getValue(1).isOperandOf(Chain.getOperand(0).getNode()) &&
930 Callee.getValue(1).hasOneUse())
931 return true;
932
933 // Look past CopyToRegs. We only walk one path, so the chain mustn't branch.
934 if (Chain.getOperand(0).getOpcode() == ISD::CopyToReg &&
935 Chain.getOperand(0).getValue(0).hasOneUse()) {
936 Chain = Chain.getOperand(0);
937 continue;
938 }
939
940 return false;
941 }
942}
943
944static bool isEndbrImm(uint64_t Imm, unsigned BitWidth) {
945 if (BitWidth > 64 || BitWidth % 8 != 0)
946 return false;
947
948 const unsigned NumBytes = BitWidth / 8;
949 if (NumBytes < 4)
950 return false;
951
952 const uint8_t OptionalPrefixBytes[] = {0x26, 0x2e, 0x36, 0x3e, 0x64,
953 0x65, 0x66, 0x67, 0xf0, 0xf2};
954 uint8_t Bytes[8];
955 for (unsigned I = 0; I != NumBytes; ++I)
956 Bytes[I] = (Imm >> (I * 8)) & 0xFF;
957
958 for (unsigned I = 0; I + 3 < NumBytes; ++I) {
959 if (Bytes[I] != 0xf3)
960 continue;
961
962 unsigned J = I + 1;
963 while (J < NumBytes && llvm::is_contained(OptionalPrefixBytes, Bytes[J]))
964 ++J;
965
966 if (J + 2 < NumBytes && Bytes[J] == 0x0f && Bytes[J + 1] == 0x1e &&
967 (Bytes[J + 2] == 0xfa || Bytes[J + 2] == 0xfb))
968 return true;
969 }
970
971 return false;
972}
973
974static bool needBWI(MVT VT) {
975 return (VT == MVT::v32i16 || VT == MVT::v32f16 || VT == MVT::v64i8);
976}
977
978void X86DAGToDAGISel::PreprocessISelDAG() {
979 bool MadeChange = false;
980 for (SelectionDAG::allnodes_iterator I = CurDAG->allnodes_begin(),
981 E = CurDAG->allnodes_end(); I != E; ) {
982 SDNode *N = &*I++; // Preincrement iterator to avoid invalidation issues.
983
984 // This is for CET enhancement.
985 //
986 // ENDBR32 and ENDBR64 have specific opcodes:
987 // ENDBR32: F3 0F 1E FB
988 // ENDBR64: F3 0F 1E FA
989 // We want to prevent attackers from finding unintended ENDBR32/64 opcode
990 // matches in executable code. Here's an example:
991 // If the compiler had to generate asm for the following code:
992 // a = 0xFA1E0FF3
993 // it could, for example, generate:
994 // mov 0xFA1E0FF3, dword ptr[a]
995 // In such a case, the binary would include a gadget that starts with a
996 // fake ENDBR64 opcode. Split such constants into multiple operations so
997 // the byte sequence does not appear in executable code.
998 if (N->getOpcode() == ISD::Constant) {
999 MVT VT = N->getSimpleValueType(0);
1000 assert(VT.isScalarInteger() &&
1001 "ISD::Constant must have a scalar integer type");
1002 if (!VT.isScalarInteger() || VT.getSizeInBits() > 64)
1003 continue;
1004
1005 uint64_t Imm = cast<ConstantSDNode>(N)->getZExtValue();
1006 if (isEndbrImm(Imm, VT.getSizeInBits())) {
1007 // Check that the cf-protection-branch is enabled.
1008 Metadata *CFProtectionBranch =
1010 "cf-protection-branch");
1011 if (CFProtectionBranch ||
1012 Subtarget->getCLOpts().indirect_branch_tracking) {
1013 SDLoc dl(N);
1014 uint64_t ComplementImm =
1016 SDValue Complement =
1017 CurDAG->getConstant(ComplementImm, dl, VT, false, true);
1018 Complement = CurDAG->getNOT(dl, Complement, VT);
1019 --I;
1020 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), Complement);
1021 ++I;
1022 MadeChange = true;
1023 continue;
1024 }
1025 }
1026 }
1027
1028 // If this is a target specific AND node with no flag usages, turn it back
1029 // into ISD::AND to enable test instruction matching.
1030 if (N->getOpcode() == X86ISD::AND && !N->hasAnyUseOfValue(1)) {
1031 SDValue Res = CurDAG->getNode(ISD::AND, SDLoc(N), N->getValueType(0),
1032 N->getOperand(0), N->getOperand(1));
1033 --I;
1034 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), Res);
1035 ++I;
1036 MadeChange = true;
1037 continue;
1038 }
1039
1040 // Convert vector increment or decrement to sub/add with an all-ones
1041 // constant:
1042 // add X, <1, 1...> --> sub X, <-1, -1...>
1043 // sub X, <1, 1...> --> add X, <-1, -1...>
1044 // The all-ones vector constant can be materialized using a pcmpeq
1045 // instruction that is commonly recognized as an idiom (has no register
1046 // dependency), so that's better/smaller than loading a splat 1 constant.
1047 //
1048 // But don't do this if it would inhibit a potentially profitable load
1049 // folding opportunity for the other operand. That only occurs with the
1050 // intersection of:
1051 // (1) The other operand (op0) is load foldable.
1052 // (2) The op is an add (otherwise, we are *creating* an add and can still
1053 // load fold the other op).
1054 // (3) The target has AVX (otherwise, we have a destructive add and can't
1055 // load fold the other op without killing the constant op).
1056 // (4) The constant 1 vector has multiple uses (so it is profitable to load
1057 // into a register anyway).
1058 auto mayPreventLoadFold = [&]() {
1059 return X86::mayFoldLoad(N->getOperand(0), *Subtarget) &&
1060 N->getOpcode() == ISD::ADD && Subtarget->hasAVX() &&
1061 !N->getOperand(1).hasOneUse();
1062 };
1063 if ((N->getOpcode() == ISD::ADD || N->getOpcode() == ISD::SUB) &&
1064 N->getSimpleValueType(0).isVector() && !mayPreventLoadFold()) {
1065 APInt SplatVal;
1067 peekThroughBitcasts(N->getOperand(0)).getNode()) &&
1068 X86::isConstantSplat(N->getOperand(1), SplatVal) &&
1069 SplatVal.isOne()) {
1070 SDLoc DL(N);
1071
1072 MVT VT = N->getSimpleValueType(0);
1073 unsigned NumElts = VT.getSizeInBits() / 32;
1074 SDValue AllOnes =
1075 CurDAG->getAllOnesConstant(DL, MVT::getVectorVT(MVT::i32, NumElts));
1076 AllOnes = CurDAG->getBitcast(VT, AllOnes);
1077
1078 unsigned NewOpcode = N->getOpcode() == ISD::ADD ? ISD::SUB : ISD::ADD;
1079 SDValue Res =
1080 CurDAG->getNode(NewOpcode, DL, VT, N->getOperand(0), AllOnes);
1081 --I;
1082 CurDAG->ReplaceAllUsesWith(N, Res.getNode());
1083 ++I;
1084 MadeChange = true;
1085 continue;
1086 }
1087 }
1088
1089 switch (N->getOpcode()) {
1090 case X86ISD::VBROADCAST: {
1091 MVT VT = N->getSimpleValueType(0);
1092 // Emulate v32i16/v64i8 broadcast without BWI.
1093 if (!Subtarget->hasBWI() && needBWI(VT)) {
1094 MVT NarrowVT = VT.getHalfNumVectorElementsVT();
1095 SDLoc dl(N);
1096 SDValue NarrowBCast =
1097 CurDAG->getNode(X86ISD::VBROADCAST, dl, NarrowVT, N->getOperand(0));
1098 SDValue Res =
1099 CurDAG->getNode(ISD::INSERT_SUBVECTOR, dl, VT, CurDAG->getUNDEF(VT),
1100 NarrowBCast, CurDAG->getIntPtrConstant(0, dl));
1101 unsigned Index = NarrowVT.getVectorMinNumElements();
1102 Res = CurDAG->getNode(ISD::INSERT_SUBVECTOR, dl, VT, Res, NarrowBCast,
1103 CurDAG->getIntPtrConstant(Index, dl));
1104
1105 --I;
1106 CurDAG->ReplaceAllUsesWith(N, Res.getNode());
1107 ++I;
1108 MadeChange = true;
1109 continue;
1110 }
1111
1112 break;
1113 }
1114 case X86ISD::VBROADCAST_LOAD: {
1115 MVT VT = N->getSimpleValueType(0);
1116 // Emulate v32i16/v64i8 broadcast without BWI.
1117 if (!Subtarget->hasBWI() && needBWI(VT)) {
1118 MVT NarrowVT = VT.getHalfNumVectorElementsVT();
1119 auto *MemNode = cast<MemSDNode>(N);
1120 SDLoc dl(N);
1121 SDVTList VTs = CurDAG->getVTList(NarrowVT, MVT::Other);
1122 SDValue Ops[] = {MemNode->getChain(), MemNode->getBasePtr()};
1123 SDValue NarrowBCast = CurDAG->getMemIntrinsicNode(
1124 X86ISD::VBROADCAST_LOAD, dl, VTs, Ops, MemNode->getMemoryVT(),
1125 MemNode->getMemOperand());
1126 SDValue Res =
1127 CurDAG->getNode(ISD::INSERT_SUBVECTOR, dl, VT, CurDAG->getUNDEF(VT),
1128 NarrowBCast, CurDAG->getIntPtrConstant(0, dl));
1129 unsigned Index = NarrowVT.getVectorMinNumElements();
1130 Res = CurDAG->getNode(ISD::INSERT_SUBVECTOR, dl, VT, Res, NarrowBCast,
1131 CurDAG->getIntPtrConstant(Index, dl));
1132
1133 --I;
1134 SDValue To[] = {Res, NarrowBCast.getValue(1)};
1135 CurDAG->ReplaceAllUsesWith(N, To);
1136 ++I;
1137 MadeChange = true;
1138 continue;
1139 }
1140
1141 break;
1142 }
1143 case ISD::LOAD: {
1144 // If this is a XMM/YMM load of the same lower bits as another YMM/ZMM
1145 // load, then just extract the lower subvector and avoid the second load.
1146 auto *Ld = cast<LoadSDNode>(N);
1147 MVT VT = N->getSimpleValueType(0);
1148 if (!ISD::isNormalLoad(Ld) || !Ld->isSimple() ||
1149 !(VT.is128BitVector() || VT.is256BitVector()))
1150 break;
1151
1152 MVT MaxVT = VT;
1153 SDNode *MaxLd = nullptr;
1154 SDValue Ptr = Ld->getBasePtr();
1155 SDValue Chain = Ld->getChain();
1156 for (SDNode *User : Ptr->users()) {
1157 auto *UserLd = dyn_cast<LoadSDNode>(User);
1158 MVT UserVT = User->getSimpleValueType(0);
1159 if (User != N && UserLd && ISD::isNormalLoad(User) &&
1160 UserLd->getBasePtr() == Ptr && UserLd->getChain() == Chain &&
1161 !User->hasAnyUseOfValue(1) &&
1162 (UserVT.is256BitVector() || UserVT.is512BitVector()) &&
1163 UserVT.getSizeInBits() > VT.getSizeInBits() &&
1164 (!MaxLd || UserVT.getSizeInBits() > MaxVT.getSizeInBits())) {
1165 MaxLd = User;
1166 MaxVT = UserVT;
1167 }
1168 }
1169 if (MaxLd) {
1170 SDLoc dl(N);
1171 unsigned NumSubElts = VT.getSizeInBits() / MaxVT.getScalarSizeInBits();
1172 MVT SubVT = MVT::getVectorVT(MaxVT.getScalarType(), NumSubElts);
1173 SDValue Extract = CurDAG->getNode(ISD::EXTRACT_SUBVECTOR, dl, SubVT,
1174 SDValue(MaxLd, 0),
1175 CurDAG->getIntPtrConstant(0, dl));
1176 SDValue Res = CurDAG->getBitcast(VT, Extract);
1177
1178 --I;
1179 SDValue To[] = {Res, SDValue(MaxLd, 1)};
1180 CurDAG->ReplaceAllUsesWith(N, To);
1181 ++I;
1182 MadeChange = true;
1183 continue;
1184 }
1185 break;
1186 }
1187 case ISD::VSELECT: {
1188 // Replace VSELECT with non-mask conditions with with BLENDV/VPTERNLOG.
1189 EVT EleVT = N->getOperand(0).getValueType().getVectorElementType();
1190 if (EleVT == MVT::i1)
1191 break;
1192
1193 assert(Subtarget->hasSSE41() && "Expected SSE4.1 support!");
1194 assert(N->getValueType(0).getVectorElementType() != MVT::i16 &&
1195 "We can't replace VSELECT with BLENDV in vXi16!");
1196 SDValue R;
1197 if (Subtarget->hasVLX() && CurDAG->ComputeNumSignBits(N->getOperand(0)) ==
1198 EleVT.getSizeInBits()) {
1199 R = CurDAG->getNode(X86ISD::VPTERNLOG, SDLoc(N), N->getValueType(0),
1200 N->getOperand(0), N->getOperand(1), N->getOperand(2),
1201 CurDAG->getTargetConstant(0xCA, SDLoc(N), MVT::i8));
1202 } else {
1203 R = CurDAG->getNode(X86ISD::BLENDV, SDLoc(N), N->getValueType(0),
1204 N->getOperand(0), N->getOperand(1),
1205 N->getOperand(2));
1206 }
1207 --I;
1208 CurDAG->ReplaceAllUsesWith(N, R.getNode());
1209 ++I;
1210 MadeChange = true;
1211 continue;
1212 }
1213 case ISD::FP_ROUND:
1215 case ISD::FP_TO_SINT:
1216 case ISD::FP_TO_UINT:
1219 // Replace vector fp_to_s/uint with their X86 specific equivalent so we
1220 // don't need 2 sets of patterns.
1221 if (!N->getSimpleValueType(0).isVector())
1222 break;
1223
1224 unsigned NewOpc;
1225 switch (N->getOpcode()) {
1226 default: llvm_unreachable("Unexpected opcode!");
1227 case ISD::FP_ROUND: NewOpc = X86ISD::VFPROUND; break;
1228 case ISD::STRICT_FP_ROUND: NewOpc = X86ISD::STRICT_VFPROUND; break;
1229 case ISD::STRICT_FP_TO_SINT: NewOpc = X86ISD::STRICT_CVTTP2SI; break;
1230 case ISD::FP_TO_SINT: NewOpc = X86ISD::CVTTP2SI; break;
1231 case ISD::STRICT_FP_TO_UINT: NewOpc = X86ISD::STRICT_CVTTP2UI; break;
1232 case ISD::FP_TO_UINT: NewOpc = X86ISD::CVTTP2UI; break;
1233 }
1234 SDValue Res;
1235 if (N->isStrictFPOpcode())
1236 Res =
1237 CurDAG->getNode(NewOpc, SDLoc(N), {N->getValueType(0), MVT::Other},
1238 {N->getOperand(0), N->getOperand(1)});
1239 else
1240 Res =
1241 CurDAG->getNode(NewOpc, SDLoc(N), N->getValueType(0),
1242 N->getOperand(0));
1243 --I;
1244 CurDAG->ReplaceAllUsesWith(N, Res.getNode());
1245 ++I;
1246 MadeChange = true;
1247 continue;
1248 }
1249 case ISD::SHL:
1250 case ISD::SRA:
1251 case ISD::SRL: {
1252 // Replace vector shifts with their X86 specific equivalent so we don't
1253 // need 2 sets of patterns.
1254 if (!N->getValueType(0).isVector())
1255 break;
1256
1257 unsigned NewOpc;
1258 switch (N->getOpcode()) {
1259 default: llvm_unreachable("Unexpected opcode!");
1260 case ISD::SHL: NewOpc = X86ISD::VSHLV; break;
1261 case ISD::SRA: NewOpc = X86ISD::VSRAV; break;
1262 case ISD::SRL: NewOpc = X86ISD::VSRLV; break;
1263 }
1264 SDValue Res = CurDAG->getNode(NewOpc, SDLoc(N), N->getValueType(0),
1265 N->getOperand(0), N->getOperand(1));
1266 --I;
1267 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), Res);
1268 ++I;
1269 MadeChange = true;
1270 continue;
1271 }
1272 case ISD::ANY_EXTEND:
1274 // Replace vector any extend with the zero extend equivalents so we don't
1275 // need 2 sets of patterns. Ignore vXi1 extensions.
1276 if (!N->getValueType(0).isVector())
1277 break;
1278
1279 unsigned NewOpc;
1280 if (N->getOperand(0).getScalarValueSizeInBits() == 1) {
1281 assert(N->getOpcode() == ISD::ANY_EXTEND &&
1282 "Unexpected opcode for mask vector!");
1283 NewOpc = ISD::SIGN_EXTEND;
1284 } else {
1285 NewOpc = N->getOpcode() == ISD::ANY_EXTEND
1288 }
1289
1290 SDValue Res = CurDAG->getNode(NewOpc, SDLoc(N), N->getValueType(0),
1291 N->getOperand(0));
1292 --I;
1293 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), Res);
1294 ++I;
1295 MadeChange = true;
1296 continue;
1297 }
1298 case ISD::FCEIL:
1299 case ISD::STRICT_FCEIL:
1300 case ISD::FFLOOR:
1301 case ISD::STRICT_FFLOOR:
1302 case ISD::FTRUNC:
1303 case ISD::STRICT_FTRUNC:
1304 case ISD::FROUNDEVEN:
1306 case ISD::FNEARBYINT:
1308 case ISD::FRINT:
1309 case ISD::STRICT_FRINT: {
1310 // Replace fp rounding with their X86 specific equivalent so we don't
1311 // need 2 sets of patterns.
1312 unsigned Imm;
1313 switch (N->getOpcode()) {
1314 default: llvm_unreachable("Unexpected opcode!");
1315 case ISD::STRICT_FCEIL:
1316 case ISD::FCEIL: Imm = 0xA; break;
1317 case ISD::STRICT_FFLOOR:
1318 case ISD::FFLOOR: Imm = 0x9; break;
1319 case ISD::STRICT_FTRUNC:
1320 case ISD::FTRUNC: Imm = 0xB; break;
1322 case ISD::FROUNDEVEN: Imm = 0x8; break;
1324 case ISD::FNEARBYINT: Imm = 0xC; break;
1325 case ISD::STRICT_FRINT:
1326 case ISD::FRINT: Imm = 0x4; break;
1327 }
1328 SDLoc dl(N);
1329 bool IsStrict = N->isStrictFPOpcode();
1330 SDValue Res;
1331 if (IsStrict)
1332 Res = CurDAG->getNode(X86ISD::STRICT_VRNDSCALE, dl,
1333 {N->getValueType(0), MVT::Other},
1334 {N->getOperand(0), N->getOperand(1),
1335 CurDAG->getTargetConstant(Imm, dl, MVT::i32)});
1336 else
1337 Res = CurDAG->getNode(X86ISD::VRNDSCALE, dl, N->getValueType(0),
1338 N->getOperand(0),
1339 CurDAG->getTargetConstant(Imm, dl, MVT::i32));
1340 --I;
1341 CurDAG->ReplaceAllUsesWith(N, Res.getNode());
1342 ++I;
1343 MadeChange = true;
1344 continue;
1345 }
1346 case X86ISD::FANDN:
1347 case X86ISD::FAND:
1348 case X86ISD::FOR:
1349 case X86ISD::FXOR: {
1350 // Widen scalar fp logic ops to vector to reduce isel patterns.
1351 // FIXME: Can we do this during lowering/combine.
1352 MVT VT = N->getSimpleValueType(0);
1353 if (VT.isVector() || VT == MVT::f128)
1354 break;
1355
1356 MVT VecVT = VT == MVT::f64 ? MVT::v2f64
1357 : VT == MVT::f32 ? MVT::v4f32
1358 : MVT::v8f16;
1359
1360 SDLoc dl(N);
1361 SDValue Op0 = CurDAG->getNode(ISD::SCALAR_TO_VECTOR, dl, VecVT,
1362 N->getOperand(0));
1363 SDValue Op1 = CurDAG->getNode(ISD::SCALAR_TO_VECTOR, dl, VecVT,
1364 N->getOperand(1));
1365
1366 SDValue Res;
1367 if (Subtarget->hasSSE2()) {
1368 EVT IntVT = EVT(VecVT).changeVectorElementTypeToInteger();
1369 Op0 = CurDAG->getNode(ISD::BITCAST, dl, IntVT, Op0);
1370 Op1 = CurDAG->getNode(ISD::BITCAST, dl, IntVT, Op1);
1371 unsigned Opc;
1372 switch (N->getOpcode()) {
1373 default: llvm_unreachable("Unexpected opcode!");
1374 case X86ISD::FANDN: Opc = X86ISD::ANDNP; break;
1375 case X86ISD::FAND: Opc = ISD::AND; break;
1376 case X86ISD::FOR: Opc = ISD::OR; break;
1377 case X86ISD::FXOR: Opc = ISD::XOR; break;
1378 }
1379 Res = CurDAG->getNode(Opc, dl, IntVT, Op0, Op1);
1380 Res = CurDAG->getNode(ISD::BITCAST, dl, VecVT, Res);
1381 } else {
1382 Res = CurDAG->getNode(N->getOpcode(), dl, VecVT, Op0, Op1);
1383 }
1384 Res = CurDAG->getNode(ISD::EXTRACT_VECTOR_ELT, dl, VT, Res,
1385 CurDAG->getIntPtrConstant(0, dl));
1386 --I;
1387 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), Res);
1388 ++I;
1389 MadeChange = true;
1390 continue;
1391 }
1392 }
1393
1394 if (OptLevel != CodeGenOptLevel::None &&
1395 // Only do this when the target can fold the load into the call or
1396 // jmp.
1397 !Subtarget->useIndirectThunkCalls() &&
1398 ((N->getOpcode() == X86ISD::CALL && !Subtarget->slowTwoMemOps() &&
1399 !Subtarget->slowIndirectCall()) ||
1400 (N->getOpcode() == X86ISD::TC_RETURN &&
1401 (Subtarget->is64Bit() ||
1402 !getTargetMachine().isPositionIndependent())))) {
1403 /// Also try moving call address load from outside callseq_start to just
1404 /// before the call to allow it to be folded.
1405 ///
1406 /// [Load chain]
1407 /// ^
1408 /// |
1409 /// [Load]
1410 /// ^ ^
1411 /// | |
1412 /// / \--
1413 /// / |
1414 ///[CALLSEQ_START] |
1415 /// ^ |
1416 /// | |
1417 /// [LOAD/C2Reg] |
1418 /// | |
1419 /// \ /
1420 /// \ /
1421 /// [CALL]
1422 bool HasCallSeq = N->getOpcode() == X86ISD::CALL;
1423 SDValue Chain = N->getOperand(0);
1424 SDValue Load = N->getOperand(1);
1425 if (!isCalleeLoad(Load, Chain, HasCallSeq))
1426 continue;
1427 if (N->getOpcode() == X86ISD::TC_RETURN && !checkTCRetEnoughRegs(N))
1428 continue;
1429 moveBelowOrigChain(CurDAG, Load, SDValue(N, 0), Chain);
1430 ++NumLoadMoved;
1431 MadeChange = true;
1432 continue;
1433 }
1434
1435 // Lower fpround and fpextend nodes that target the FP stack to be store and
1436 // load to the stack. This is a gross hack. We would like to simply mark
1437 // these as being illegal, but when we do that, legalize produces these when
1438 // it expands calls, then expands these in the same legalize pass. We would
1439 // like dag combine to be able to hack on these between the call expansion
1440 // and the node legalization. As such this pass basically does "really
1441 // late" legalization of these inline with the X86 isel pass.
1442 // FIXME: This should only happen when not compiled with -O0.
1443 switch (N->getOpcode()) {
1444 default: continue;
1445 case ISD::FP_ROUND:
1446 case ISD::FP_EXTEND:
1447 {
1448 MVT SrcVT = N->getOperand(0).getSimpleValueType();
1449 MVT DstVT = N->getSimpleValueType(0);
1450
1451 // If any of the sources are vectors, no fp stack involved.
1452 if (SrcVT.isVector() || DstVT.isVector())
1453 continue;
1454
1455 // If the source and destination are SSE registers, then this is a legal
1456 // conversion that should not be lowered.
1457 const X86TargetLowering *X86Lowering =
1458 static_cast<const X86TargetLowering *>(TLI);
1459 bool SrcIsSSE = X86Lowering->isScalarFPTypeInSSEReg(SrcVT);
1460 bool DstIsSSE = X86Lowering->isScalarFPTypeInSSEReg(DstVT);
1461 if (SrcIsSSE && DstIsSSE)
1462 continue;
1463
1464 if (!SrcIsSSE && !DstIsSSE) {
1465 // If this is an FPStack extension, it is a noop.
1466 if (N->getOpcode() == ISD::FP_EXTEND)
1467 continue;
1468 // If this is a value-preserving FPStack truncation, it is a noop.
1469 if (N->getConstantOperandVal(1))
1470 continue;
1471 }
1472
1473 // Here we could have an FP stack truncation or an FPStack <-> SSE convert.
1474 // FPStack has extload and truncstore. SSE can fold direct loads into other
1475 // operations. Based on this, decide what we want to do.
1476 MVT MemVT = (N->getOpcode() == ISD::FP_ROUND) ? DstVT : SrcVT;
1477 SDValue MemTmp = CurDAG->CreateStackTemporary(MemVT);
1478 int SPFI = cast<FrameIndexSDNode>(MemTmp)->getIndex();
1479 MachinePointerInfo MPI =
1480 MachinePointerInfo::getFixedStack(CurDAG->getMachineFunction(), SPFI);
1481 SDLoc dl(N);
1482
1483 // FIXME: optimize the case where the src/dest is a load or store?
1484
1485 SDValue Store = CurDAG->getTruncStore(
1486 CurDAG->getEntryNode(), dl, N->getOperand(0), MemTmp, MPI, MemVT);
1487 SDValue Result = CurDAG->getExtLoad(ISD::EXTLOAD, dl, DstVT, Store,
1488 MemTmp, MPI, MemVT);
1489
1490 // We're about to replace all uses of the FP_ROUND/FP_EXTEND with the
1491 // extload we created. This will cause general havok on the dag because
1492 // anything below the conversion could be folded into other existing nodes.
1493 // To avoid invalidating 'I', back it up to the convert node.
1494 --I;
1495 CurDAG->ReplaceAllUsesOfValueWith(SDValue(N, 0), Result);
1496 break;
1497 }
1498
1499 //The sequence of events for lowering STRICT_FP versions of these nodes requires
1500 //dealing with the chain differently, as there is already a preexisting chain.
1503 {
1504 MVT SrcVT = N->getOperand(1).getSimpleValueType();
1505 MVT DstVT = N->getSimpleValueType(0);
1506
1507 // If any of the sources are vectors, no fp stack involved.
1508 if (SrcVT.isVector() || DstVT.isVector())
1509 continue;
1510
1511 // If the source and destination are SSE registers, then this is a legal
1512 // conversion that should not be lowered.
1513 const X86TargetLowering *X86Lowering =
1514 static_cast<const X86TargetLowering *>(TLI);
1515 bool SrcIsSSE = X86Lowering->isScalarFPTypeInSSEReg(SrcVT);
1516 bool DstIsSSE = X86Lowering->isScalarFPTypeInSSEReg(DstVT);
1517 if (SrcIsSSE && DstIsSSE)
1518 continue;
1519
1520 if (!SrcIsSSE && !DstIsSSE) {
1521 // If this is an FPStack extension, it is a noop.
1522 if (N->getOpcode() == ISD::STRICT_FP_EXTEND)
1523 continue;
1524 // If this is a value-preserving FPStack truncation, it is a noop.
1525 if (N->getConstantOperandVal(2))
1526 continue;
1527 }
1528
1529 // Here we could have an FP stack truncation or an FPStack <-> SSE convert.
1530 // FPStack has extload and truncstore. SSE can fold direct loads into other
1531 // operations. Based on this, decide what we want to do.
1532 MVT MemVT = (N->getOpcode() == ISD::STRICT_FP_ROUND) ? DstVT : SrcVT;
1533 SDValue MemTmp = CurDAG->CreateStackTemporary(MemVT);
1534 int SPFI = cast<FrameIndexSDNode>(MemTmp)->getIndex();
1535 MachinePointerInfo MPI =
1536 MachinePointerInfo::getFixedStack(CurDAG->getMachineFunction(), SPFI);
1537 SDLoc dl(N);
1538
1539 // FIXME: optimize the case where the src/dest is a load or store?
1540
1541 //Since the operation is StrictFP, use the preexisting chain.
1542 SDValue Store, Result;
1543 if (!SrcIsSSE) {
1544 SDVTList VTs = CurDAG->getVTList(MVT::Other);
1545 SDValue Ops[] = {N->getOperand(0), N->getOperand(1), MemTmp};
1546 Store = CurDAG->getMemIntrinsicNode(X86ISD::FST, dl, VTs, Ops, MemVT,
1547 MPI, /*Align*/ std::nullopt,
1549 if (N->getFlags().hasNoFPExcept()) {
1550 SDNodeFlags Flags = Store->getFlags();
1551 Flags.setNoFPExcept(true);
1552 Store->setFlags(Flags);
1553 }
1554 } else {
1555 assert(SrcVT == MemVT && "Unexpected VT!");
1556 Store = CurDAG->getStore(N->getOperand(0), dl, N->getOperand(1), MemTmp,
1557 MPI);
1558 }
1559
1560 if (!DstIsSSE) {
1561 SDVTList VTs = CurDAG->getVTList(DstVT, MVT::Other);
1562 SDValue Ops[] = {Store, MemTmp};
1563 Result = CurDAG->getMemIntrinsicNode(
1564 X86ISD::FLD, dl, VTs, Ops, MemVT, MPI,
1565 /*Align*/ std::nullopt, MachineMemOperand::MOLoad);
1566 if (N->getFlags().hasNoFPExcept()) {
1567 SDNodeFlags Flags = Result->getFlags();
1568 Flags.setNoFPExcept(true);
1569 Result->setFlags(Flags);
1570 }
1571 } else {
1572 assert(DstVT == MemVT && "Unexpected VT!");
1573 Result = CurDAG->getLoad(DstVT, dl, Store, MemTmp, MPI);
1574 }
1575
1576 // We're about to replace all uses of the FP_ROUND/FP_EXTEND with the
1577 // extload we created. This will cause general havok on the dag because
1578 // anything below the conversion could be folded into other existing nodes.
1579 // To avoid invalidating 'I', back it up to the convert node.
1580 --I;
1581 CurDAG->ReplaceAllUsesWith(N, Result.getNode());
1582 break;
1583 }
1584 }
1585
1586
1587 // Now that we did that, the node is dead. Increment the iterator to the
1588 // next node to process, then delete N.
1589 ++I;
1590 MadeChange = true;
1591 }
1592
1593 // Remove any dead nodes that may have been left behind.
1594 if (MadeChange)
1595 CurDAG->RemoveDeadNodes();
1596}
1597
1598// Look for a redundant movzx/movsx that can occur after an 8-bit divrem.
1599bool X86DAGToDAGISel::tryOptimizeRem8Extend(SDNode *N) {
1600 unsigned Opc = N->getMachineOpcode();
1601 if (Opc != X86::MOVZX32rr8 && Opc != X86::MOVSX32rr8 &&
1602 Opc != X86::MOVSX64rr8)
1603 return false;
1604
1605 SDValue N0 = N->getOperand(0);
1606
1607 // We need to be extracting the lower bit of an extend.
1608 if (!N0.isMachineOpcode() ||
1609 N0.getMachineOpcode() != TargetOpcode::EXTRACT_SUBREG ||
1610 N0.getConstantOperandVal(1) != X86::sub_8bit)
1611 return false;
1612
1613 // We're looking for either a movsx or movzx to match the original opcode.
1614 unsigned ExpectedOpc = Opc == X86::MOVZX32rr8 ? X86::MOVZX32rr8_NOREX
1615 : X86::MOVSX32rr8_NOREX;
1616 SDValue N00 = N0.getOperand(0);
1617 if (!N00.isMachineOpcode() || N00.getMachineOpcode() != ExpectedOpc)
1618 return false;
1619
1620 if (Opc == X86::MOVSX64rr8) {
1621 // If we had a sign extend from 8 to 64 bits. We still need to go from 32
1622 // to 64.
1623 MachineSDNode *Extend = CurDAG->getMachineNode(X86::MOVSX64rr32, SDLoc(N),
1624 MVT::i64, N00);
1625 ReplaceUses(N, Extend);
1626 } else {
1627 // Ok we can drop this extend and just use the original extend.
1628 ReplaceUses(N, N00.getNode());
1629 }
1630
1631 return true;
1632}
1633
1634void X86DAGToDAGISel::PostprocessISelDAG() {
1635 // Skip peepholes at -O0.
1636 if (TM.getOptLevel() == CodeGenOptLevel::None)
1637 return;
1638
1639 SelectionDAG::allnodes_iterator Position = CurDAG->allnodes_end();
1640
1641 bool MadeChange = false;
1642 while (Position != CurDAG->allnodes_begin()) {
1643 SDNode *N = &*--Position;
1644 // Skip dead nodes and any non-machine opcodes.
1645 if (N->use_empty() || !N->isMachineOpcode())
1646 continue;
1647
1648 if (tryOptimizeRem8Extend(N)) {
1649 MadeChange = true;
1650 continue;
1651 }
1652
1653 unsigned Opc = N->getMachineOpcode();
1654 switch (Opc) {
1655 default:
1656 continue;
1657 // ANDrr/rm + TESTrr+ -> TESTrr/TESTmr
1658 case X86::TEST8rr:
1659 case X86::TEST16rr:
1660 case X86::TEST32rr:
1661 case X86::TEST64rr:
1662 // ANDrr/rm + CTESTrr -> CTESTrr/CTESTmr
1663 case X86::CTEST8rr:
1664 case X86::CTEST16rr:
1665 case X86::CTEST32rr:
1666 case X86::CTEST64rr: {
1667 auto &Op0 = N->getOperand(0);
1668 if (Op0 != N->getOperand(1) || !Op0->hasNUsesOfValue(2, Op0.getResNo()) ||
1669 !Op0.isMachineOpcode())
1670 continue;
1671 SDValue And = N->getOperand(0);
1672#define CASE_ND(OP) \
1673 case X86::OP: \
1674 case X86::OP##_ND:
1675 switch (And.getMachineOpcode()) {
1676 default:
1677 continue;
1678 CASE_ND(AND8rr)
1679 CASE_ND(AND16rr)
1680 CASE_ND(AND32rr)
1681 CASE_ND(AND64rr) {
1682 if (And->hasAnyUseOfValue(1))
1683 continue;
1684 SmallVector<SDValue> Ops(N->op_values());
1685 Ops[0] = And.getOperand(0);
1686 Ops[1] = And.getOperand(1);
1687 MachineSDNode *Test =
1688 CurDAG->getMachineNode(Opc, SDLoc(N), MVT::i32, Ops);
1689 ReplaceUses(N, Test);
1690 MadeChange = true;
1691 continue;
1692 }
1693 CASE_ND(AND8rm)
1694 CASE_ND(AND16rm)
1695 CASE_ND(AND32rm)
1696 CASE_ND(AND64rm) {
1697 if (And->hasAnyUseOfValue(1))
1698 continue;
1699 unsigned NewOpc;
1700 bool IsCTESTCC = X86::isCTESTCC(Opc);
1701#define FROM_TO(A, B) \
1702 CASE_ND(A) NewOpc = IsCTESTCC ? X86::C##B : X86::B; \
1703 break;
1704 switch (And.getMachineOpcode()) {
1705 FROM_TO(AND8rm, TEST8mr);
1706 FROM_TO(AND16rm, TEST16mr);
1707 FROM_TO(AND32rm, TEST32mr);
1708 FROM_TO(AND64rm, TEST64mr);
1709 }
1710#undef FROM_TO
1711#undef CASE_ND
1712 // Need to swap the memory and register operand.
1713 SmallVector<SDValue> Ops = {And.getOperand(1), And.getOperand(2),
1714 And.getOperand(3), And.getOperand(4),
1715 And.getOperand(5), And.getOperand(0)};
1716 // CC, Cflags.
1717 if (IsCTESTCC) {
1718 Ops.push_back(N->getOperand(2));
1719 Ops.push_back(N->getOperand(3));
1720 }
1721 // Chain of memory load
1722 Ops.push_back(And.getOperand(6));
1723 // Glue
1724 if (IsCTESTCC)
1725 Ops.push_back(N->getOperand(4));
1726
1727 MachineSDNode *Test = CurDAG->getMachineNode(
1728 NewOpc, SDLoc(N), MVT::i32, MVT::Other, Ops);
1729 CurDAG->setNodeMemRefs(
1730 Test, cast<MachineSDNode>(And.getNode())->memoperands());
1731 ReplaceUses(And.getValue(2), SDValue(Test, 1));
1732 ReplaceUses(SDValue(N, 0), SDValue(Test, 0));
1733 MadeChange = true;
1734 continue;
1735 }
1736 }
1737 }
1738 // Look for a KAND+KORTEST and turn it into KTEST if only the zero flag is
1739 // used. We're doing this late so we can prefer to fold the AND into masked
1740 // comparisons. Doing that can be better for the live range of the mask
1741 // register.
1742 case X86::KORTESTBkk:
1743 case X86::KORTESTWkk:
1744 case X86::KORTESTDkk:
1745 case X86::KORTESTQkk: {
1746 SDValue Op0 = N->getOperand(0);
1747 if (Op0 != N->getOperand(1) || !N->isOnlyUserOf(Op0.getNode()) ||
1748 !Op0.isMachineOpcode() || !onlyUsesZeroFlag(SDValue(N, 0)))
1749 continue;
1750#define CASE(A) \
1751 case X86::A: \
1752 break;
1753 switch (Op0.getMachineOpcode()) {
1754 default:
1755 continue;
1756 CASE(KANDBkk)
1757 CASE(KANDWkk)
1758 CASE(KANDDkk)
1759 CASE(KANDQkk)
1760 }
1761 unsigned NewOpc;
1762#define FROM_TO(A, B) \
1763 case X86::A: \
1764 NewOpc = X86::B; \
1765 break;
1766 switch (Opc) {
1767 FROM_TO(KORTESTBkk, KTESTBkk)
1768 FROM_TO(KORTESTWkk, KTESTWkk)
1769 FROM_TO(KORTESTDkk, KTESTDkk)
1770 FROM_TO(KORTESTQkk, KTESTQkk)
1771 }
1772 // KANDW is legal with AVX512F, but KTESTW requires AVX512DQ. The other
1773 // KAND instructions and KTEST use the same ISA feature.
1774 if (NewOpc == X86::KTESTWkk && !Subtarget->hasDQI())
1775 continue;
1776#undef FROM_TO
1777 MachineSDNode *KTest = CurDAG->getMachineNode(
1778 NewOpc, SDLoc(N), MVT::i32, Op0.getOperand(0), Op0.getOperand(1));
1779 ReplaceUses(N, KTest);
1780 MadeChange = true;
1781 continue;
1782 }
1783 // Attempt to remove vectors moves that were inserted to zero upper bits.
1784 case TargetOpcode::SUBREG_TO_REG: {
1785 unsigned SubRegIdx = N->getConstantOperandVal(1);
1786 if (SubRegIdx != X86::sub_xmm && SubRegIdx != X86::sub_ymm)
1787 continue;
1788
1789 SDValue Move = N->getOperand(0);
1790 if (!Move.isMachineOpcode())
1791 continue;
1792
1793 // Make sure its one of the move opcodes we recognize.
1794 switch (Move.getMachineOpcode()) {
1795 default:
1796 continue;
1797 CASE(VMOVAPDrr) CASE(VMOVUPDrr)
1798 CASE(VMOVAPSrr) CASE(VMOVUPSrr)
1799 CASE(VMOVDQArr) CASE(VMOVDQUrr)
1800 CASE(VMOVAPDYrr) CASE(VMOVUPDYrr)
1801 CASE(VMOVAPSYrr) CASE(VMOVUPSYrr)
1802 CASE(VMOVDQAYrr) CASE(VMOVDQUYrr)
1803 CASE(VMOVAPDZ128rr) CASE(VMOVUPDZ128rr)
1804 CASE(VMOVAPSZ128rr) CASE(VMOVUPSZ128rr)
1805 CASE(VMOVDQA32Z128rr) CASE(VMOVDQU32Z128rr)
1806 CASE(VMOVDQA64Z128rr) CASE(VMOVDQU64Z128rr)
1807 CASE(VMOVAPDZ256rr) CASE(VMOVUPDZ256rr)
1808 CASE(VMOVAPSZ256rr) CASE(VMOVUPSZ256rr)
1809 CASE(VMOVDQA32Z256rr) CASE(VMOVDQU32Z256rr)
1810 CASE(VMOVDQA64Z256rr) CASE(VMOVDQU64Z256rr)
1811 }
1812#undef CASE
1813
1814 SDValue In = Move.getOperand(0);
1815 if (!In.isMachineOpcode() ||
1816 In.getMachineOpcode() <= TargetOpcode::GENERIC_OP_END)
1817 continue;
1818
1819 // Make sure the instruction has a VEX, XOP, or EVEX prefix. This covers
1820 // the SHA instructions which use a legacy encoding.
1821 uint64_t TSFlags = getInstrInfo()->get(In.getMachineOpcode()).TSFlags;
1822 if ((TSFlags & X86II::EncodingMask) != X86II::VEX &&
1823 (TSFlags & X86II::EncodingMask) != X86II::EVEX &&
1824 (TSFlags & X86II::EncodingMask) != X86II::XOP)
1825 continue;
1826
1827 // Producing instruction is another vector instruction. We can drop the
1828 // move.
1829 CurDAG->UpdateNodeOperands(N, In, N->getOperand(1));
1830 MadeChange = true;
1831 }
1832 }
1833 }
1834
1835 if (MadeChange)
1836 CurDAG->RemoveDeadNodes();
1837}
1838
1839
1840/// Emit any code that needs to be executed only in the main function.
1841void X86DAGToDAGISel::emitSpecialCodeForMain() {
1842 if (Subtarget->isTargetCygMing()) {
1843 TargetLowering::ArgListTy Args;
1844 auto &DL = CurDAG->getDataLayout();
1845
1846 TargetLowering::CallLoweringInfo CLI(*CurDAG);
1847 CLI.setChain(CurDAG->getRoot())
1848 .setCallee(CallingConv::C, Type::getVoidTy(*CurDAG->getContext()),
1849 CurDAG->getExternalSymbol("__main", TLI->getPointerTy(DL)),
1850 std::move(Args));
1851 const TargetLowering &TLI = CurDAG->getTargetLoweringInfo();
1852 std::pair<SDValue, SDValue> Result = TLI.LowerCallTo(CLI);
1853 CurDAG->setRoot(Result.second);
1854 }
1855}
1856
1857void X86DAGToDAGISel::emitFunctionEntryCode() {
1858 // If this is main, emit special code for main.
1859 const Function &F = MF->getFunction();
1860 if (F.hasExternalLinkage() && F.getName() == "main")
1861 emitSpecialCodeForMain();
1862}
1863
1864static bool isDispSafeForFrameIndexOrRegBase(int64_t Val) {
1865 // We can run into an issue where a frame index or a register base
1866 // includes a displacement that, when added to the explicit displacement,
1867 // will overflow the displacement field. Assuming that the
1868 // displacement fits into a 31-bit integer (which is only slightly more
1869 // aggressive than the current fundamental assumption that it fits into
1870 // a 32-bit integer), a 31-bit disp should always be safe.
1871 return isInt<31>(Val);
1872}
1873
1874bool X86DAGToDAGISel::foldOffsetIntoAddress(uint64_t Offset,
1875 X86ISelAddressMode &AM) {
1876 // We may have already matched a displacement and the caller just added the
1877 // symbolic displacement. So we still need to do the checks even if Offset
1878 // is zero.
1879
1880 int64_t Val = AM.Disp + Offset;
1881
1882 // Cannot combine ExternalSymbol displacements with integer offsets.
1883 if (Val != 0 && (AM.ES || AM.MCSym))
1884 return true;
1885
1886 CodeModel::Model M = TM.getCodeModel();
1887 if (Subtarget->is64Bit()) {
1888 if (Val != 0 &&
1890 AM.hasSymbolicDisplacement()))
1891 return true;
1892 // In addition to the checks required for a register base, check that
1893 // we do not try to use an unsafe Disp with a frame index.
1894 if (AM.BaseType == X86ISelAddressMode::FrameIndexBase &&
1896 return true;
1897 // In ILP32 (x32) mode, pointers are 32 bits and need to be zero-extended to
1898 // 64 bits. Instructions with 32-bit register addresses perform this zero
1899 // extension for us and we can safely ignore the high bits of Offset.
1900 // Instructions with only a 32-bit immediate address do not, though: they
1901 // sign extend instead. This means only address the low 2GB of address space
1902 // is directly addressable, we need indirect addressing for the high 2GB of
1903 // address space.
1904 // TODO: Some of the earlier checks may be relaxed for ILP32 mode as the
1905 // implicit zero extension of instructions would cover up any problem.
1906 // However, we have asserts elsewhere that get triggered if we do, so keep
1907 // the checks for now.
1908 // TODO: We would actually be able to accept these, as well as the same
1909 // addresses in LP64 mode, by adding the EIZ pseudo-register as an operand
1910 // to get an address size override to be emitted. However, this
1911 // pseudo-register is not part of any register class and therefore causes
1912 // MIR verification to fail.
1913 if (Subtarget->isTarget64BitILP32() &&
1914 !isDispSafeForFrameIndexOrRegBase((uint32_t)Val) &&
1915 !AM.hasBaseOrIndexReg())
1916 return true;
1917 } else if (Subtarget->is16Bit()) {
1918 // In 16-bit mode, displacements are limited to [-65535,65535] for FK_Data_2
1919 // fixups of unknown signedness. See X86AsmBackend::applyFixup.
1920 if (Val < -(int64_t)UINT16_MAX || Val > (int64_t)UINT16_MAX)
1921 return true;
1922 } else if (AM.hasBaseOrIndexReg() && !isDispSafeForFrameIndexOrRegBase(Val))
1923 // For 32-bit X86, make sure the displacement still isn't close to the
1924 // expressible limit.
1925 return true;
1926 AM.Disp = Val;
1927 return false;
1928}
1929
1930bool X86DAGToDAGISel::matchLoadInAddress(LoadSDNode *N, X86ISelAddressMode &AM,
1931 bool AllowSegmentRegForX32) {
1932 SDValue Address = N->getOperand(1);
1933
1934 // load gs:0 -> GS segment register.
1935 // load fs:0 -> FS segment register.
1936 //
1937 // This optimization is generally valid because the GNU TLS model defines that
1938 // gs:0 (or fs:0 on X86-64) contains its own address. However, for X86-64 mode
1939 // with 32-bit registers, as we get in ILP32 mode, those registers are first
1940 // zero-extended to 64 bits and then added it to the base address, which gives
1941 // unwanted results when the register holds a negative value.
1942 // For more information see http://people.redhat.com/drepper/tls.pdf
1943 if (isNullConstant(Address) && AM.Segment.getNode() == nullptr &&
1944 !IndirectTlsSegRefs &&
1945 (Subtarget->isTargetGlibc() || Subtarget->isTargetMusl() ||
1946 Subtarget->isTargetAndroid() || Subtarget->isTargetFuchsia())) {
1947 if (Subtarget->isTarget64BitILP32() && !AllowSegmentRegForX32)
1948 return true;
1949 switch (N->getPointerInfo().getAddrSpace()) {
1950 case X86AS::GS:
1951 AM.Segment = CurDAG->getRegister(X86::GS, MVT::i16);
1952 return false;
1953 case X86AS::FS:
1954 AM.Segment = CurDAG->getRegister(X86::FS, MVT::i16);
1955 return false;
1956 // Address space X86AS::SS is not handled here, because it is not used to
1957 // address TLS areas.
1958 }
1959 }
1960
1961 return true;
1962}
1963
1964/// Try to match X86ISD::Wrapper and X86ISD::WrapperRIP nodes into an addressing
1965/// mode. These wrap things that will resolve down into a symbol reference.
1966/// If no match is possible, this returns true, otherwise it returns false.
1967bool X86DAGToDAGISel::matchWrapper(SDValue N, X86ISelAddressMode &AM) {
1968 // If the addressing mode already has a symbol as the displacement, we can
1969 // never match another symbol.
1970 if (AM.hasSymbolicDisplacement())
1971 return true;
1972
1973 bool IsRIPRelTLS = false;
1974 bool IsRIPRel = N.getOpcode() == X86ISD::WrapperRIP;
1975 if (IsRIPRel) {
1976 SDValue Val = N.getOperand(0);
1978 IsRIPRelTLS = true;
1979 }
1980
1981 // We can't use an addressing mode in the 64-bit large code model.
1982 // Global TLS addressing is an exception. In the medium code model,
1983 // we use can use a mode when RIP wrappers are present.
1984 // That signifies access to globals that are known to be "near",
1985 // such as the GOT itself.
1986 CodeModel::Model M = TM.getCodeModel();
1987 if (Subtarget->is64Bit() && M == CodeModel::Large && !IsRIPRelTLS)
1988 return true;
1989
1990 // Base and index reg must be 0 in order to use %rip as base.
1991 if (IsRIPRel && AM.hasBaseOrIndexReg())
1992 return true;
1993
1994 // Make a local copy in case we can't do this fold.
1995 X86ISelAddressMode Backup = AM;
1996
1997 int64_t Offset = 0;
1998 SDValue N0 = N.getOperand(0);
1999 if (auto *G = dyn_cast<GlobalAddressSDNode>(N0)) {
2000 AM.GV = G->getGlobal();
2001 AM.SymbolFlags = G->getTargetFlags();
2002 Offset = G->getOffset();
2003 } else if (auto *CP = dyn_cast<ConstantPoolSDNode>(N0)) {
2004 AM.CP = CP->getConstVal();
2005 AM.Alignment = CP->getAlign();
2006 AM.SymbolFlags = CP->getTargetFlags();
2007 Offset = CP->getOffset();
2008 } else if (auto *S = dyn_cast<ExternalSymbolSDNode>(N0)) {
2009 AM.ES = S->getSymbol();
2010 AM.SymbolFlags = S->getTargetFlags();
2011 } else if (auto *S = dyn_cast<MCSymbolSDNode>(N0)) {
2012 AM.MCSym = S->getMCSymbol();
2013 } else if (auto *J = dyn_cast<JumpTableSDNode>(N0)) {
2014 AM.JT = J->getIndex();
2015 AM.SymbolFlags = J->getTargetFlags();
2016 } else if (auto *BA = dyn_cast<BlockAddressSDNode>(N0)) {
2017 AM.BlockAddr = BA->getBlockAddress();
2018 AM.SymbolFlags = BA->getTargetFlags();
2019 Offset = BA->getOffset();
2020 } else
2021 llvm_unreachable("Unhandled symbol reference node.");
2022
2023 // Can't use an addressing mode with large globals.
2024 if (Subtarget->is64Bit() && !IsRIPRel && AM.GV &&
2025 TM.isLargeGlobalValue(AM.GV)) {
2026 AM = Backup;
2027 return true;
2028 }
2029
2030 if (foldOffsetIntoAddress(Offset, AM)) {
2031 AM = Backup;
2032 return true;
2033 }
2034
2035 if (IsRIPRel)
2036 AM.setBaseReg(CurDAG->getRegister(X86::RIP, MVT::i64));
2037
2038 // Commit the changes now that we know this fold is safe.
2039 return false;
2040}
2041
2042/// Add the specified node to the specified addressing mode, returning true if
2043/// it cannot be done. This just pattern matches for the addressing mode.
2044bool X86DAGToDAGISel::matchAddress(SDValue N, X86ISelAddressMode &AM) {
2045 if (matchAddressRecursively(N, AM, 0))
2046 return true;
2047
2048 // Post-processing: Make a second attempt to fold a load, if we now know
2049 // that there will not be any other register. This is only performed for
2050 // 64-bit ILP32 mode since 32-bit mode and 64-bit LP64 mode will have folded
2051 // any foldable load the first time.
2052 if (Subtarget->isTarget64BitILP32() &&
2053 AM.BaseType == X86ISelAddressMode::RegBase &&
2054 AM.Base_Reg.getNode() != nullptr && AM.IndexReg.getNode() == nullptr) {
2055 SDValue Save_Base_Reg = AM.Base_Reg;
2056 if (auto *LoadN = dyn_cast<LoadSDNode>(Save_Base_Reg)) {
2057 AM.Base_Reg = SDValue();
2058 if (matchLoadInAddress(LoadN, AM, /*AllowSegmentRegForX32=*/true))
2059 AM.Base_Reg = Save_Base_Reg;
2060 }
2061 }
2062
2063 // Post-processing: Convert lea(,%reg,2) to lea(%reg,%reg), which has
2064 // a smaller encoding and avoids a scaled-index. Not valid when the index is
2065 // negated: this copies the index into the base, but only the index is negated
2066 // when the address is emitted, so the result would be index + (-index) - that
2067 // is, zero - rather than (-index) * 2.
2068 if (AM.Scale == 2 && !AM.NegateIndex &&
2069 AM.BaseType == X86ISelAddressMode::RegBase &&
2070 AM.Base_Reg.getNode() == nullptr) {
2071 AM.Base_Reg = AM.IndexReg;
2072 AM.Scale = 1;
2073 }
2074
2075 // Post-processing: Convert foo to foo(%rip), even in non-PIC mode,
2076 // because it has a smaller encoding.
2077 if (TM.getCodeModel() != CodeModel::Large &&
2078 (!AM.GV || !TM.isLargeGlobalValue(AM.GV)) && Subtarget->is64Bit() &&
2079 AM.Scale == 1 && AM.BaseType == X86ISelAddressMode::RegBase &&
2080 AM.Base_Reg.getNode() == nullptr && AM.IndexReg.getNode() == nullptr &&
2081 AM.SymbolFlags == X86II::MO_NO_FLAG && AM.hasSymbolicDisplacement()) {
2082 // However, when GV is a local function symbol and in the same section as
2083 // the current instruction, and AM.Disp is negative and near INT32_MIN,
2084 // referencing GV+Disp generates a relocation referencing the section symbol
2085 // with an even smaller offset, which might underflow. We should bail out if
2086 // the negative offset is too close to INT32_MIN. Actually, we are more
2087 // conservative here, using a smaller magic number also used by
2088 // isOffsetSuitableForCodeModel.
2089 if (isa_and_nonnull<Function>(AM.GV) && AM.Disp < -16 * 1024 * 1024)
2090 return true;
2091
2092 AM.Base_Reg = CurDAG->getRegister(X86::RIP, MVT::i64);
2093 }
2094
2095 return false;
2096}
2097
2098// Returns true if V has a use that materializes it in a register as a value -
2099// a stored value operand or a CopyToReg (a return value, call argument, or a
2100// value that is live out of the block). Such a use means V will be in a
2101// register regardless, so reusing it when forming an LEA is free. Uses where V
2102// is only an address (a load/store pointer, or folded into another address
2103// computation) do not materialize it. This is a more precise replacement for
2104// the !hasOneUse() proxy: an address-only multi-use value is not materialized.
2105bool X86DAGToDAGISel::hasMaterializingUse(SDValue V) const {
2106 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
2107 for (SDUse &U : V->uses()) {
2108 if (U.getResNo() != V.getResNo())
2109 continue;
2110 SDNode *User = U.getUser();
2111 // A return value, call argument, or a value live out of the block.
2112 if (User->getOpcode() == ISD::CopyToReg)
2113 return true;
2114 // A stored value materializes V (V as a store *address* does not).
2115 if (auto *St = dyn_cast<StoreSDNode>(User)) {
2116 if (St->getValue() == V)
2117 return true;
2118 continue;
2119 }
2120 // Selection may already have turned the ISD::STORE into a machine store by
2121 // the time we get here. V materializes it if it is a stored value, i.e. an
2122 // operand that is neither part of the memory reference (the address
2123 // operands) nor the chain/glue. The memory reference is not always the
2124 // first operand, so locate it via the instruction's memory-operand info
2125 // rather than assuming a fixed layout. (No getOperandBias() is needed:
2126 // unlike a MachineInstr, an SDNode's operand list has no leading defs.)
2127 if (!User->isMachineOpcode())
2128 continue;
2129 const MCInstrDesc &Desc = TII->get(User->getMachineOpcode());
2130 if (!Desc.mayStore())
2131 continue;
2132 int MemRefBegin = X86II::getMemoryOperandNo(Desc.TSFlags);
2133 if (MemRefBegin < 0)
2134 continue;
2135 unsigned MemRefEnd = MemRefBegin + X86::AddrNumOperands;
2136 for (unsigned I = 0, E = User->getNumOperands(); I != E; ++I) {
2137 if (I >= static_cast<unsigned>(MemRefBegin) && I < MemRefEnd)
2138 continue; // an address operand
2139 SDValue Opnd = User->getOperand(I);
2140 if (Opnd.getValueType() == MVT::Other || Opnd.getValueType() == MVT::Glue)
2141 continue; // chain / glue
2142 if (Opnd == V)
2143 return true; // a stored value operand
2144 }
2145 }
2146 return false;
2147}
2148
2149bool X86DAGToDAGISel::matchAdd(SDValue &N, X86ISelAddressMode &AM,
2150 unsigned Depth) {
2151 // Add an artificial use to this node so that we can keep track of
2152 // it if it gets CSE'd with a different node.
2153 HandleSDNode Handle(N);
2154
2155 auto IsAddOrAddLike = [&](SDValue V) {
2156 return V.getOpcode() == ISD::ADD || CurDAG->isADDLike(V);
2157 };
2158
2159 // When forming a LEA, avoid splitting an already-materialized value: use the
2160 // operand directly as a base/index register instead. hasMaterializingUse()
2161 // decides whether the operand is genuinely materialized - it has a use that
2162 // puts it in a register as a value. A value used only as an address is not
2163 // materialized, and splitting it there would only add a redundant
2164 // materialization (see the two_ptrs test).
2165 auto SplitsMaterializedValue = [&](SDValue Op) {
2166 if (!AM.IsForLEA || !hasMaterializingUse(Op))
2167 return false;
2168
2169 // add-like: decomposes to base + index (+ disp)
2170 if (IsAddOrAddLike(Op))
2171 return IsAddOrAddLike(Op.getOperand(0)) ||
2172 IsAddOrAddLike(Op.getOperand(1));
2173
2174 // shl by 1/2/3 folds to a scaled index
2175 if (Op.getOpcode() == ISD::SHL)
2176 if (auto *C = dyn_cast<ConstantSDNode>(Op.getOperand(1)))
2177 return C->getZExtValue() >= 1 && C->getZExtValue() <= 3 &&
2178 IsAddOrAddLike(Op.getOperand(0));
2179
2180 return false;
2181 };
2182
2183 // The check is applied here, per add operand, rather than inside
2184 // matchAddressRecursively, so that it only fires when an add directly
2185 // consumes the value. matchAddressRecursively is also entered for the LEA
2186 // root itself and from the SUB case's operand fold.
2187 // Firing there produces worse code.
2188 auto MatchOperand = [&](SDValue Op) {
2189 // The reuse shortcut places Op directly as a base/index register via
2190 // matchAddressBase. That is illegal once AM is already %rip-relative:
2191 // [%rip + disp32] takes no register beyond RIP itself (its implicit base) -
2192 // no additional base and no index - so adding one would form an invalid
2193 // address (folding a RIP-relative global and a materialized value into a
2194 // single LEA, which asserts "Invalid rip-relative address" in the MC
2195 // encoder). matchAddressRecursively correctly refuses to fold a register
2196 // into a %rip-relative address, so fall back to it and let matchAdd keep
2197 // the operands separate.
2198 if (SplitsMaterializedValue(Op) && !AM.isRIPRelative())
2199 return matchAddressBase(Op, AM);
2200 return matchAddressRecursively(Op, AM, Depth + 1);
2201 };
2202
2203 X86ISelAddressMode Backup = AM;
2204 if (!MatchOperand(N.getOperand(0)) &&
2205 !MatchOperand(Handle.getValue().getOperand(1)))
2206 return false;
2207 AM = Backup;
2208
2209 // Try again after commutating the operands.
2210 if (!MatchOperand(Handle.getValue().getOperand(1)) &&
2211 !MatchOperand(Handle.getValue().getOperand(0)))
2212 return false;
2213 AM = Backup;
2214
2215 // If we couldn't fold both operands into the address at the same time,
2216 // see if we can just put each operand into a register and fold at least
2217 // the add.
2218 if (AM.BaseType == X86ISelAddressMode::RegBase &&
2219 !AM.Base_Reg.getNode() &&
2220 !AM.IndexReg.getNode()) {
2221 N = Handle.getValue();
2222 AM.Base_Reg = N.getOperand(0);
2223 AM.IndexReg = N.getOperand(1);
2224 AM.Scale = 1;
2225 return false;
2226 }
2227 N = Handle.getValue();
2228 return true;
2229}
2230
2231// Insert a node into the DAG at least before the Pos node's position. This
2232// will reposition the node as needed, and will assign it a node ID that is <=
2233// the Pos node's ID. Note that this does *not* preserve the uniqueness of node
2234// IDs! The selection DAG must no longer depend on their uniqueness when this
2235// is used.
2236static void insertDAGNode(SelectionDAG &DAG, SDValue Pos, SDValue N) {
2237 if (N->getNodeId() == -1 ||
2240 DAG.RepositionNode(Pos->getIterator(), N.getNode());
2241 // Mark Node as invalid for pruning as after this it may be a successor to a
2242 // selected node but otherwise be in the same position of Pos.
2243 // Conservatively mark it with the same -abs(Id) to assure node id
2244 // invariant is preserved.
2245 N->setNodeId(Pos->getNodeId());
2247 }
2248}
2249
2250// Transform "(X >> (8-C1)) & (0xff << C1)" to "((X >> 8) & 0xff) << C1" if
2251// safe. This allows us to convert the shift and and into an h-register
2252// extract and a scaled index. Returns false if the simplification is
2253// performed.
2255 uint64_t Mask,
2256 SDValue Shift, SDValue X,
2257 X86ISelAddressMode &AM) {
2258 if (Shift.getOpcode() != ISD::SRL ||
2259 !isa<ConstantSDNode>(Shift.getOperand(1)) ||
2260 !Shift.hasOneUse())
2261 return true;
2262
2263 int ScaleLog = 8 - Shift.getConstantOperandVal(1);
2264 if (ScaleLog <= 0 || ScaleLog >= 4 ||
2265 Mask != (0xffu << ScaleLog))
2266 return true;
2267
2268 MVT XVT = X.getSimpleValueType();
2269 MVT VT = N.getSimpleValueType();
2270 SDLoc DL(N);
2271 SDValue Eight = DAG.getConstant(8, DL, MVT::i8);
2272 SDValue NewMask = DAG.getConstant(0xff, DL, XVT);
2273 SDValue Srl = DAG.getNode(ISD::SRL, DL, XVT, X, Eight);
2274 SDValue And = DAG.getNode(ISD::AND, DL, XVT, Srl, NewMask);
2275 SDValue Ext = DAG.getZExtOrTrunc(And, DL, VT);
2276 SDValue ShlCount = DAG.getConstant(ScaleLog, DL, MVT::i8);
2277 SDValue Shl = DAG.getNode(ISD::SHL, DL, VT, Ext, ShlCount);
2278
2279 // Insert the new nodes into the topological ordering. We must do this in
2280 // a valid topological ordering as nothing is going to go back and re-sort
2281 // these nodes. We continually insert before 'N' in sequence as this is
2282 // essentially a pre-flattened and pre-sorted sequence of nodes. There is no
2283 // hierarchy left to express.
2284 insertDAGNode(DAG, N, Eight);
2285 insertDAGNode(DAG, N, NewMask);
2286 insertDAGNode(DAG, N, Srl);
2287 insertDAGNode(DAG, N, And);
2288 insertDAGNode(DAG, N, Ext);
2289 insertDAGNode(DAG, N, ShlCount);
2290 insertDAGNode(DAG, N, Shl);
2291 DAG.ReplaceAllUsesWith(N, Shl);
2292 DAG.RemoveDeadNode(N.getNode());
2293 AM.IndexReg = Ext;
2294 AM.Scale = (1 << ScaleLog);
2295 return false;
2296}
2297
2298// Transforms "(X << C1) & C2" to "(X & (C2>>C1)) << C1" if safe and if this
2299// allows us to fold the shift into this addressing mode. Returns false if the
2300// transform succeeded.
2302 X86ISelAddressMode &AM) {
2303 SDValue Shift = N.getOperand(0);
2304
2305 // Use a signed mask so that shifting right will insert sign bits. These
2306 // bits will be removed when we shift the result left so it doesn't matter
2307 // what we use. This might allow a smaller immediate encoding.
2308 int64_t Mask = cast<ConstantSDNode>(N->getOperand(1))->getSExtValue();
2309
2310 // If we have an any_extend feeding the AND, look through it to see if there
2311 // is a shift behind it. But only if the AND doesn't use the extended bits.
2312 // FIXME: Generalize this to other ANY_EXTEND than i32 to i64?
2313 bool FoundAnyExtend = false;
2314 if (Shift.getOpcode() == ISD::ANY_EXTEND && Shift.hasOneUse() &&
2315 Shift.getOperand(0).getSimpleValueType() == MVT::i32 &&
2316 isUInt<32>(Mask)) {
2317 FoundAnyExtend = true;
2318 Shift = Shift.getOperand(0);
2319 }
2320
2321 if (Shift.getOpcode() != ISD::SHL ||
2323 return true;
2324
2325 SDValue X = Shift.getOperand(0);
2326
2327 // Not likely to be profitable if either the AND or SHIFT node has more
2328 // than one use (unless all uses are for address computation). Besides,
2329 // isel mechanism requires their node ids to be reused.
2330 if (!N.hasOneUse() || !Shift.hasOneUse())
2331 return true;
2332
2333 // Verify that the shift amount is something we can fold.
2334 unsigned ShiftAmt = Shift.getConstantOperandVal(1);
2335 if (ShiftAmt != 1 && ShiftAmt != 2 && ShiftAmt != 3)
2336 return true;
2337
2338 MVT VT = N.getSimpleValueType();
2339 SDLoc DL(N);
2340 if (FoundAnyExtend) {
2341 SDValue NewX = DAG.getNode(ISD::ANY_EXTEND, DL, VT, X);
2342 insertDAGNode(DAG, N, NewX);
2343 X = NewX;
2344 }
2345
2346 SDValue NewMask = DAG.getSignedConstant(Mask >> ShiftAmt, DL, VT);
2347 SDValue NewAnd = DAG.getNode(ISD::AND, DL, VT, X, NewMask);
2348 SDValue NewShift = DAG.getNode(ISD::SHL, DL, VT, NewAnd, Shift.getOperand(1));
2349
2350 // Insert the new nodes into the topological ordering. We must do this in
2351 // a valid topological ordering as nothing is going to go back and re-sort
2352 // these nodes. We continually insert before 'N' in sequence as this is
2353 // essentially a pre-flattened and pre-sorted sequence of nodes. There is no
2354 // hierarchy left to express.
2355 insertDAGNode(DAG, N, NewMask);
2356 insertDAGNode(DAG, N, NewAnd);
2357 insertDAGNode(DAG, N, NewShift);
2358 DAG.ReplaceAllUsesWith(N, NewShift);
2359 DAG.RemoveDeadNode(N.getNode());
2360
2361 AM.Scale = 1 << ShiftAmt;
2362 AM.IndexReg = NewAnd;
2363 return false;
2364}
2365
2366// Implement some heroics to detect shifts of masked values where the mask can
2367// be replaced by extending the shift and undoing that in the addressing mode
2368// scale. Patterns such as (shl (srl x, c1), c2) are canonicalized into (and
2369// (srl x, SHIFT), MASK) by DAGCombines that don't know the shl can be done in
2370// the addressing mode. This results in code such as:
2371//
2372// int f(short *y, int *lookup_table) {
2373// ...
2374// return *y + lookup_table[*y >> 11];
2375// }
2376//
2377// Turning into:
2378// movzwl (%rdi), %eax
2379// movl %eax, %ecx
2380// shrl $11, %ecx
2381// addl (%rsi,%rcx,4), %eax
2382//
2383// Instead of:
2384// movzwl (%rdi), %eax
2385// movl %eax, %ecx
2386// shrl $9, %ecx
2387// andl $124, %rcx
2388// addl (%rsi,%rcx), %eax
2389//
2390// Note that this function assumes the mask is provided as a mask *after* the
2391// value is shifted. The input chain may or may not match that, but computing
2392// such a mask is trivial.
2394 uint64_t Mask,
2395 SDValue Shift, SDValue X,
2396 X86ISelAddressMode &AM) {
2397 if (Shift.getOpcode() != ISD::SRL || !Shift.hasOneUse() ||
2399 return true;
2400
2401 // We need to ensure that mask is a continuous run of bits.
2402 unsigned MaskIdx, MaskLen;
2403 if (!isShiftedMask_64(Mask, MaskIdx, MaskLen))
2404 return true;
2405 unsigned MaskLZ = 64 - (MaskIdx + MaskLen);
2406
2407 unsigned ShiftAmt = Shift.getConstantOperandVal(1);
2408
2409 // The amount of shift we're trying to fit into the addressing mode is taken
2410 // from the shifted mask index (number of trailing zeros of the mask).
2411 unsigned AMShiftAmt = MaskIdx;
2412
2413 // There is nothing we can do here unless the mask is removing some bits.
2414 // Also, the addressing mode can only represent shifts of 1, 2, or 3 bits.
2415 if (AMShiftAmt == 0 || AMShiftAmt > 3) return true;
2416
2417 // Scale the leading zero count down based on the actual size of the value.
2418 // Also scale it down based on the size of the shift.
2419 unsigned ScaleDown = (64 - X.getSimpleValueType().getSizeInBits()) + ShiftAmt;
2420 if (MaskLZ < ScaleDown)
2421 return true;
2422 MaskLZ -= ScaleDown;
2423
2424 // The final check is to ensure that any masked out high bits of X are
2425 // already known to be zero. Otherwise, the mask has a semantic impact
2426 // other than masking out a couple of low bits. Unfortunately, because of
2427 // the mask, zero extensions will be removed from operands in some cases.
2428 // This code works extra hard to look through extensions because we can
2429 // replace them with zero extensions cheaply if necessary.
2430 bool ReplacingAnyExtend = false;
2431 if (X.getOpcode() == ISD::ANY_EXTEND) {
2432 unsigned ExtendBits = X.getSimpleValueType().getSizeInBits() -
2433 X.getOperand(0).getSimpleValueType().getSizeInBits();
2434 // Assume that we'll replace the any-extend with a zero-extend, and
2435 // narrow the search to the extended value.
2436 X = X.getOperand(0);
2437 MaskLZ = ExtendBits > MaskLZ ? 0 : MaskLZ - ExtendBits;
2438 ReplacingAnyExtend = true;
2439 }
2440 APInt MaskedHighBits =
2441 APInt::getHighBitsSet(X.getSimpleValueType().getSizeInBits(), MaskLZ);
2442 if (!DAG.MaskedValueIsZero(X, MaskedHighBits))
2443 return true;
2444
2445 // We've identified a pattern that can be transformed into a single shift
2446 // and an addressing mode. Make it so.
2447 MVT VT = N.getSimpleValueType();
2448 if (ReplacingAnyExtend) {
2449 assert(X.getValueType() != VT);
2450 // We looked through an ANY_EXTEND node, insert a ZERO_EXTEND.
2451 SDValue NewX = DAG.getNode(ISD::ZERO_EXTEND, SDLoc(X), VT, X);
2452 insertDAGNode(DAG, N, NewX);
2453 X = NewX;
2454 }
2455
2456 MVT XVT = X.getSimpleValueType();
2457 SDLoc DL(N);
2458 SDValue NewSRLAmt = DAG.getConstant(ShiftAmt + AMShiftAmt, DL, MVT::i8);
2459 SDValue NewSRL = DAG.getNode(ISD::SRL, DL, XVT, X, NewSRLAmt);
2460 SDValue NewExt = DAG.getZExtOrTrunc(NewSRL, DL, VT);
2461 SDValue NewSHLAmt = DAG.getConstant(AMShiftAmt, DL, MVT::i8);
2462 SDValue NewSHL = DAG.getNode(ISD::SHL, DL, VT, NewExt, NewSHLAmt);
2463
2464 // Insert the new nodes into the topological ordering. We must do this in
2465 // a valid topological ordering as nothing is going to go back and re-sort
2466 // these nodes. We continually insert before 'N' in sequence as this is
2467 // essentially a pre-flattened and pre-sorted sequence of nodes. There is no
2468 // hierarchy left to express.
2469 insertDAGNode(DAG, N, NewSRLAmt);
2470 insertDAGNode(DAG, N, NewSRL);
2471 insertDAGNode(DAG, N, NewExt);
2472 insertDAGNode(DAG, N, NewSHLAmt);
2473 insertDAGNode(DAG, N, NewSHL);
2474 DAG.ReplaceAllUsesWith(N, NewSHL);
2475 DAG.RemoveDeadNode(N.getNode());
2476
2477 AM.Scale = 1 << AMShiftAmt;
2478 AM.IndexReg = NewExt;
2479 return false;
2480}
2481
2482// Transform "(X >> SHIFT) & (MASK << C1)" to
2483// "((X >> (SHIFT + C1)) & (MASK)) << C1". Everything before the SHL will be
2484// matched to a BEXTR later. Returns false if the simplification is performed.
2486 uint64_t Mask,
2487 SDValue Shift, SDValue X,
2488 X86ISelAddressMode &AM,
2489 const X86Subtarget &Subtarget) {
2490 if (Shift.getOpcode() != ISD::SRL ||
2491 !isa<ConstantSDNode>(Shift.getOperand(1)) ||
2492 !Shift.hasOneUse() || !N.hasOneUse())
2493 return true;
2494
2495 // Only do this if BEXTR will be matched by matchBEXTRFromAndImm.
2496 if (!Subtarget.hasTBM() &&
2497 !(Subtarget.hasBMI() && Subtarget.hasFastBEXTR()))
2498 return true;
2499
2500 // We need to ensure that mask is a continuous run of bits.
2501 unsigned MaskIdx, MaskLen;
2502 if (!isShiftedMask_64(Mask, MaskIdx, MaskLen))
2503 return true;
2504
2505 unsigned ShiftAmt = Shift.getConstantOperandVal(1);
2506
2507 // The amount of shift we're trying to fit into the addressing mode is taken
2508 // from the shifted mask index (number of trailing zeros of the mask).
2509 unsigned AMShiftAmt = MaskIdx;
2510
2511 // There is nothing we can do here unless the mask is removing some bits.
2512 // Also, the addressing mode can only represent shifts of 1, 2, or 3 bits.
2513 if (AMShiftAmt == 0 || AMShiftAmt > 3) return true;
2514
2515 MVT XVT = X.getSimpleValueType();
2516 MVT VT = N.getSimpleValueType();
2517 SDLoc DL(N);
2518 SDValue NewSRLAmt = DAG.getConstant(ShiftAmt + AMShiftAmt, DL, MVT::i8);
2519 SDValue NewSRL = DAG.getNode(ISD::SRL, DL, XVT, X, NewSRLAmt);
2520 SDValue NewMask = DAG.getConstant(Mask >> AMShiftAmt, DL, XVT);
2521 SDValue NewAnd = DAG.getNode(ISD::AND, DL, XVT, NewSRL, NewMask);
2522 SDValue NewExt = DAG.getZExtOrTrunc(NewAnd, DL, VT);
2523 SDValue NewSHLAmt = DAG.getConstant(AMShiftAmt, DL, MVT::i8);
2524 SDValue NewSHL = DAG.getNode(ISD::SHL, DL, VT, NewExt, NewSHLAmt);
2525
2526 // Insert the new nodes into the topological ordering. We must do this in
2527 // a valid topological ordering as nothing is going to go back and re-sort
2528 // these nodes. We continually insert before 'N' in sequence as this is
2529 // essentially a pre-flattened and pre-sorted sequence of nodes. There is no
2530 // hierarchy left to express.
2531 insertDAGNode(DAG, N, NewSRLAmt);
2532 insertDAGNode(DAG, N, NewSRL);
2533 insertDAGNode(DAG, N, NewMask);
2534 insertDAGNode(DAG, N, NewAnd);
2535 insertDAGNode(DAG, N, NewExt);
2536 insertDAGNode(DAG, N, NewSHLAmt);
2537 insertDAGNode(DAG, N, NewSHL);
2538 DAG.ReplaceAllUsesWith(N, NewSHL);
2539 DAG.RemoveDeadNode(N.getNode());
2540
2541 AM.Scale = 1 << AMShiftAmt;
2542 AM.IndexReg = NewExt;
2543 return false;
2544}
2545
2546// Attempt to peek further into a scaled index register, collecting additional
2547// extensions / offsets / etc. Returns /p N if we can't peek any further.
2548SDValue X86DAGToDAGISel::matchIndexRecursively(SDValue N,
2549 X86ISelAddressMode &AM,
2550 unsigned Depth) {
2551 assert(AM.IndexReg.getNode() == nullptr && "IndexReg already matched");
2552 assert((AM.Scale == 1 || AM.Scale == 2 || AM.Scale == 4 || AM.Scale == 8) &&
2553 "Illegal index scale");
2554
2555 // Limit recursion.
2557 return N;
2558
2559 EVT VT = N.getValueType();
2560 unsigned Opc = N.getOpcode();
2561
2562 // index: add(x,c) -> index: x, disp + c
2563 if (CurDAG->isBaseWithConstantOffset(N)) {
2564 auto *AddVal = cast<ConstantSDNode>(N.getOperand(1));
2565 uint64_t Offset = (uint64_t)AddVal->getSExtValue() * AM.Scale;
2566 if (!foldOffsetIntoAddress(Offset, AM))
2567 return matchIndexRecursively(N.getOperand(0), AM, Depth + 1);
2568 }
2569
2570 // index: add(x,x) -> index: x, scale * 2
2571 if (Opc == ISD::ADD && N.getOperand(0) == N.getOperand(1)) {
2572 if (AM.Scale <= 4) {
2573 AM.Scale *= 2;
2574 return matchIndexRecursively(N.getOperand(0), AM, Depth + 1);
2575 }
2576 }
2577
2578 // index: shl(x,i) -> index: x, scale * (1 << i)
2579 if (Opc == X86ISD::VSHLI) {
2580 uint64_t ShiftAmt = N.getConstantOperandVal(1);
2581 uint64_t ScaleAmt = 1ULL << ShiftAmt;
2582 if ((AM.Scale * ScaleAmt) <= 8) {
2583 AM.Scale *= ScaleAmt;
2584 return matchIndexRecursively(N.getOperand(0), AM, Depth + 1);
2585 }
2586 }
2587
2588 // index: sext(add_nsw(x,c)) -> index: sext(x), disp + sext(c)
2589 // TODO: call matchIndexRecursively(AddSrc) if we won't corrupt sext?
2590 if (Opc == ISD::SIGN_EXTEND && !VT.isVector() && N.hasOneUse()) {
2591 SDValue Src = N.getOperand(0);
2592 if (Src.getOpcode() == ISD::ADD && Src->getFlags().hasNoSignedWrap() &&
2593 Src.hasOneUse()) {
2594 if (CurDAG->isBaseWithConstantOffset(Src)) {
2595 SDValue AddSrc = Src.getOperand(0);
2596 auto *AddVal = cast<ConstantSDNode>(Src.getOperand(1));
2597 int64_t Offset = AddVal->getSExtValue();
2598 if (!foldOffsetIntoAddress((uint64_t)Offset * AM.Scale, AM)) {
2599 SDLoc DL(N);
2600 SDValue ExtSrc = CurDAG->getNode(Opc, DL, VT, AddSrc);
2601 SDValue ExtVal = CurDAG->getSignedConstant(Offset, DL, VT);
2602 SDValue ExtAdd = CurDAG->getNode(ISD::ADD, DL, VT, ExtSrc, ExtVal);
2603 insertDAGNode(*CurDAG, N, ExtSrc);
2604 insertDAGNode(*CurDAG, N, ExtVal);
2605 insertDAGNode(*CurDAG, N, ExtAdd);
2606 CurDAG->ReplaceAllUsesWith(N, ExtAdd);
2607 CurDAG->RemoveDeadNode(N.getNode());
2608 return ExtSrc;
2609 }
2610 }
2611 }
2612 }
2613
2614 // index: zext(add_nuw(x,c)) -> index: zext(x), disp + zext(c)
2615 // index: zext(addlike(x,c)) -> index: zext(x), disp + zext(c)
2616 // TODO: call matchIndexRecursively(AddSrc) if we won't corrupt sext?
2617 if (Opc == ISD::ZERO_EXTEND && !VT.isVector() && N.hasOneUse()) {
2618 SDValue Src = N.getOperand(0);
2619 unsigned SrcOpc = Src.getOpcode();
2620 if (((SrcOpc == ISD::ADD && Src->getFlags().hasNoUnsignedWrap()) ||
2621 CurDAG->isADDLike(Src, /*NoWrap=*/true)) &&
2622 Src.hasOneUse()) {
2623 if (CurDAG->isBaseWithConstantOffset(Src)) {
2624 SDValue AddSrc = Src.getOperand(0);
2625 uint64_t Offset = Src.getConstantOperandVal(1);
2626 if (!foldOffsetIntoAddress(Offset * AM.Scale, AM)) {
2627 SDLoc DL(N);
2628 SDValue Res;
2629 // If we're also scaling, see if we can use that as well.
2630 if (AddSrc.getOpcode() == ISD::SHL &&
2631 isa<ConstantSDNode>(AddSrc.getOperand(1))) {
2632 SDValue ShVal = AddSrc.getOperand(0);
2633 uint64_t ShAmt = AddSrc.getConstantOperandVal(1);
2634 APInt HiBits =
2636 uint64_t ScaleAmt = 1ULL << ShAmt;
2637 if ((AM.Scale * ScaleAmt) <= 8 &&
2638 (AddSrc->getFlags().hasNoUnsignedWrap() ||
2639 CurDAG->MaskedValueIsZero(ShVal, HiBits))) {
2640 AM.Scale *= ScaleAmt;
2641 SDValue ExtShVal = CurDAG->getNode(Opc, DL, VT, ShVal);
2642 SDValue ExtShift = CurDAG->getNode(ISD::SHL, DL, VT, ExtShVal,
2643 AddSrc.getOperand(1));
2644 insertDAGNode(*CurDAG, N, ExtShVal);
2645 insertDAGNode(*CurDAG, N, ExtShift);
2646 AddSrc = ExtShift;
2647 Res = ExtShVal;
2648 }
2649 }
2650 SDValue ExtSrc = CurDAG->getNode(Opc, DL, VT, AddSrc);
2651 SDValue ExtVal = CurDAG->getConstant(Offset, DL, VT);
2652 SDValue ExtAdd = CurDAG->getNode(SrcOpc, DL, VT, ExtSrc, ExtVal);
2653 insertDAGNode(*CurDAG, N, ExtSrc);
2654 insertDAGNode(*CurDAG, N, ExtVal);
2655 insertDAGNode(*CurDAG, N, ExtAdd);
2656 CurDAG->ReplaceAllUsesWith(N, ExtAdd);
2657 CurDAG->RemoveDeadNode(N.getNode());
2658 return Res ? Res : ExtSrc;
2659 }
2660 }
2661 }
2662 }
2663
2664 // TODO: Handle extensions, shifted masks etc.
2665 return N;
2666}
2667
2668bool X86DAGToDAGISel::matchAddressRecursively(SDValue N, X86ISelAddressMode &AM,
2669 unsigned Depth) {
2670 LLVM_DEBUG({
2671 dbgs() << "MatchAddress: ";
2672 AM.dump(CurDAG);
2673 });
2674 // Limit recursion.
2676 return matchAddressBase(N, AM);
2677
2678 // If this is already a %rip relative address, we can only merge immediates
2679 // into it. Instead of handling this in every case, we handle it here.
2680 // RIP relative addressing: %rip + 32-bit displacement!
2681 if (AM.isRIPRelative()) {
2682 // FIXME: JumpTable and ExternalSymbol address currently don't like
2683 // displacements. It isn't very important, but this should be fixed for
2684 // consistency.
2685 if (!(AM.ES || AM.MCSym) && AM.JT != -1)
2686 return true;
2687
2688 if (auto *Cst = dyn_cast<ConstantSDNode>(N))
2689 if (!foldOffsetIntoAddress(Cst->getSExtValue(), AM))
2690 return false;
2691 return true;
2692 }
2693
2694 switch (N.getOpcode()) {
2695 default: break;
2696 case ISD::LOCAL_RECOVER: {
2697 if (!AM.hasSymbolicDisplacement() && AM.Disp == 0)
2698 if (const auto *ESNode = dyn_cast<MCSymbolSDNode>(N.getOperand(0))) {
2699 // Use the symbol and don't prefix it.
2700 AM.MCSym = ESNode->getMCSymbol();
2701 return false;
2702 }
2703 break;
2704 }
2705 case ISD::Constant: {
2706 uint64_t Val = cast<ConstantSDNode>(N)->getSExtValue();
2707 if (!foldOffsetIntoAddress(Val, AM))
2708 return false;
2709 break;
2710 }
2711
2712 case X86ISD::Wrapper:
2713 case X86ISD::WrapperRIP:
2714 if (!matchWrapper(N, AM))
2715 return false;
2716 break;
2717
2718 case ISD::LOAD:
2719 if (!matchLoadInAddress(cast<LoadSDNode>(N), AM))
2720 return false;
2721 break;
2722
2723 case ISD::FrameIndex:
2724 if (AM.BaseType == X86ISelAddressMode::RegBase &&
2725 AM.Base_Reg.getNode() == nullptr &&
2726 (!Subtarget->is64Bit() || isDispSafeForFrameIndexOrRegBase(AM.Disp))) {
2727 AM.BaseType = X86ISelAddressMode::FrameIndexBase;
2728 AM.Base_FrameIndex = cast<FrameIndexSDNode>(N)->getIndex();
2729 return false;
2730 }
2731 break;
2732
2733 case ISD::SHL:
2734 if (AM.IndexReg.getNode() != nullptr || AM.Scale != 1)
2735 break;
2736
2737 if (auto *CN = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
2738 unsigned Val = CN->getZExtValue();
2739 // Note that we handle x<<1 as (,x,2) rather than (x,x) here so
2740 // that the base operand remains free for further matching. If
2741 // the base doesn't end up getting used, a post-processing step
2742 // in MatchAddress turns (,x,2) into (x,x), which is cheaper.
2743 if (Val == 1 || Val == 2 || Val == 3) {
2744 SDValue ShVal = N.getOperand(0);
2745 AM.Scale = 1 << Val;
2746 AM.IndexReg = matchIndexRecursively(ShVal, AM, Depth + 1);
2747 return false;
2748 }
2749 }
2750 break;
2751
2752 case ISD::SRL: {
2753 // Scale must not be used already.
2754 if (AM.IndexReg.getNode() != nullptr || AM.Scale != 1) break;
2755
2756 // We only handle up to 64-bit values here as those are what matter for
2757 // addressing mode optimizations.
2758 assert(N.getSimpleValueType().getSizeInBits() <= 64 &&
2759 "Unexpected value size!");
2760
2761 SDValue And = N.getOperand(0);
2762 if (And.getOpcode() != ISD::AND) break;
2763 SDValue X = And.getOperand(0);
2764
2765 // The mask used for the transform is expected to be post-shift, but we
2766 // found the shift first so just apply the shift to the mask before passing
2767 // it down.
2768 if (!isa<ConstantSDNode>(N.getOperand(1)) ||
2769 !isa<ConstantSDNode>(And.getOperand(1)))
2770 break;
2771 uint64_t Mask = And.getConstantOperandVal(1) >> N.getConstantOperandVal(1);
2772
2773 // Try to fold the mask and shift into the scale, and return false if we
2774 // succeed.
2775 if (!foldMaskAndShiftToScale(*CurDAG, N, Mask, N, X, AM))
2776 return false;
2777 break;
2778 }
2779
2780 case ISD::SMUL_LOHI:
2781 case ISD::UMUL_LOHI:
2782 // A mul_lohi where we need the low part can be folded as a plain multiply.
2783 if (N.getResNo() != 0) break;
2784 [[fallthrough]];
2785 case ISD::MUL:
2786 case X86ISD::MUL_IMM:
2787 // X*[3,5,9] -> X+X*[2,4,8]
2788 if (AM.BaseType == X86ISelAddressMode::RegBase &&
2789 AM.Base_Reg.getNode() == nullptr &&
2790 AM.IndexReg.getNode() == nullptr) {
2791 if (auto *CN = dyn_cast<ConstantSDNode>(N.getOperand(1)))
2792 if (CN->getZExtValue() == 3 || CN->getZExtValue() == 5 ||
2793 CN->getZExtValue() == 9) {
2794 AM.Scale = unsigned(CN->getZExtValue())-1;
2795
2796 SDValue MulVal = N.getOperand(0);
2797 SDValue Reg;
2798
2799 // Okay, we know that we have a scale by now. However, if the scaled
2800 // value is an add of something and a constant, we can fold the
2801 // constant into the disp field here.
2802 if (MulVal.getNode()->getOpcode() == ISD::ADD && MulVal.hasOneUse() &&
2803 isa<ConstantSDNode>(MulVal.getOperand(1))) {
2804 Reg = MulVal.getOperand(0);
2805 auto *AddVal = cast<ConstantSDNode>(MulVal.getOperand(1));
2806 uint64_t Disp = AddVal->getSExtValue() * CN->getZExtValue();
2807 if (foldOffsetIntoAddress(Disp, AM))
2808 Reg = N.getOperand(0);
2809 } else {
2810 Reg = N.getOperand(0);
2811 }
2812
2813 AM.IndexReg = AM.Base_Reg = Reg;
2814 return false;
2815 }
2816 }
2817 break;
2818
2819 case ISD::SUB: {
2820 // Given A-B, if A can be completely folded into the address leaving the
2821 // index field unused, use -B as the index. This is a win if A has multiple
2822 // parts that can be folded into the address. Also, this saves a mov if the
2823 // base register has other uses, since it avoids a two-address sub
2824 // instruction, however it costs an additional mov if the index register
2825 // has other uses.
2826 // B may itself be a constant shift, in which case the shift folds into
2827 // the scale - see below.
2828
2829 // Add an artificial use to this node so that we can keep track of
2830 // it if it gets CSE'd with a different node.
2831 HandleSDNode Handle(N);
2832
2833 // Test if the LHS of the sub can be folded.
2834 X86ISelAddressMode Backup = AM;
2835 if (matchAddressRecursively(N.getOperand(0), AM, Depth+1)) {
2836 N = Handle.getValue();
2837 AM = Backup;
2838 break;
2839 }
2840 N = Handle.getValue();
2841 // Test if the index field is free for use.
2842 if (AM.IndexReg.getNode() || AM.isRIPRelative()) {
2843 AM = Backup;
2844 break;
2845 }
2846
2847 int Cost = 0;
2848 SDValue RHS = N.getOperand(1);
2849
2850 // A-(B<<C) can use -B as a scaled index for C in [1,3], which folds the
2851 // shift into the address as well as the subtract. When B is not a foldable
2852 // shift, NegScale stays empty and this is the plain A-B fold, which only
2853 // breaks even on instruction count - a-b is mov+sub either way. Absorbing
2854 // the shift saves one:
2855 //
2856 // a - (b << 2) movq %rdi, %rax -> negq %rsi
2857 // shlq $2, %rsi leaq (%rdi,%rsi,4), %rax
2858 // subq %rsi, %rax
2859 //
2860 // That pays for the negate, so drop the cost by one.
2861 std::optional<unsigned> NegScale;
2862 if (RHS.getOpcode() == ISD::SHL && RHS.hasOneUse()) {
2863 if (auto *ShAmt = dyn_cast<ConstantSDNode>(RHS.getOperand(1))) {
2864 uint64_t ShVal = ShAmt->getZExtValue();
2865 if (ShVal >= 1 && ShVal <= 3) {
2866 NegScale = 1u << ShVal;
2867 RHS = RHS.getOperand(0);
2868 --Cost;
2869 }
2870 }
2871 }
2872
2873 // If the RHS involves a register with multiple uses, this
2874 // transformation incurs an extra mov, due to the neg instruction
2875 // clobbering its operand. The CopyFromReg part of that is a guess -
2876 // SelectionDAG is per-block, so uses elsewhere are invisible - and it is
2877 // not applied to a folded shift, where it is wrong often enough to matter.
2878 // The multiple-use part still is; see @y_outlives_lea.
2879 if (!RHS.getNode()->hasOneUse() ||
2880 (!NegScale && RHS.getNode()->getOpcode() == ISD::CopyFromReg) ||
2881 RHS.getNode()->getOpcode() == ISD::TRUNCATE ||
2882 RHS.getNode()->getOpcode() == ISD::ANY_EXTEND ||
2883 (RHS.getNode()->getOpcode() == ISD::ZERO_EXTEND &&
2884 RHS.getOperand(0).getValueType() == MVT::i32))
2885 ++Cost;
2886 // A - (A << C), where the base is itself the value being negated.
2887 bool BaseIsNegatedValue = NegScale &&
2888 AM.BaseType == X86ISelAddressMode::RegBase &&
2889 AM.Base_Reg == RHS;
2890 // If the base is a register with multiple uses, this transformation may
2891 // save a mov - but not for BaseIsNegatedValue, where the baseline emits the
2892 // shift non-destructively into another register and the SUB writes A in
2893 // place, so there is no copy for the LEA to save. The copy the NEG needs
2894 // there is charged by the multiple-use test above.
2895 if (((AM.BaseType == X86ISelAddressMode::RegBase && AM.Base_Reg.getNode() &&
2896 !AM.Base_Reg.getNode()->hasOneUse()) ||
2897 AM.BaseType == X86ISelAddressMode::FrameIndexBase) &&
2898 !BaseIsNegatedValue)
2899 --Cost;
2900 // If the folded LHS was interesting, this transformation saves
2901 // address arithmetic.
2902 if ((AM.hasSymbolicDisplacement() && !Backup.hasSymbolicDisplacement()) +
2903 ((AM.Disp != 0) && (Backup.Disp == 0)) +
2904 (AM.Segment.getNode() && !Backup.Segment.getNode()) >= 2)
2905 --Cost;
2906 // If it doesn't look like it may be an overall win, don't do it.
2907 if (Cost >= 0) {
2908 AM = Backup;
2909 break;
2910 }
2911
2912 // Ok, the transformation is legal and appears profitable. Go for it.
2913 // Negation will be emitted later to avoid creating dangling nodes if this
2914 // was an unprofitable LEA.
2915 AM.IndexReg = RHS;
2916 AM.NegateIndex = true;
2917 AM.Scale = NegScale.value_or(1);
2918 return false;
2919 }
2920
2921 case ISD::OR:
2922 case ISD::XOR:
2923 // See if we can treat the OR/XOR node as an ADD node.
2924 if (!CurDAG->isADDLike(N))
2925 break;
2926 [[fallthrough]];
2927 case ISD::ADD:
2928 if (!matchAdd(N, AM, Depth))
2929 return false;
2930 break;
2931
2932 case ISD::AND: {
2933 // Perform some heroic transforms on an and of a constant-count shift
2934 // with a constant to enable use of the scaled offset field.
2935
2936 // Scale must not be used already.
2937 if (AM.IndexReg.getNode() != nullptr || AM.Scale != 1) break;
2938
2939 // We only handle up to 64-bit values here as those are what matter for
2940 // addressing mode optimizations.
2941 assert(N.getSimpleValueType().getSizeInBits() <= 64 &&
2942 "Unexpected value size!");
2943
2944 if (!isa<ConstantSDNode>(N.getOperand(1)))
2945 break;
2946
2947 if (N.getOperand(0).getOpcode() == ISD::SRL) {
2948 SDValue Shift = N.getOperand(0);
2949 SDValue X = Shift.getOperand(0);
2950
2951 uint64_t Mask = N.getConstantOperandVal(1);
2952
2953 // Try to fold the mask and shift into an extract and scale.
2954 if (!foldMaskAndShiftToExtract(*CurDAG, N, Mask, Shift, X, AM))
2955 return false;
2956
2957 // Try to fold the mask and shift directly into the scale.
2958 if (!foldMaskAndShiftToScale(*CurDAG, N, Mask, Shift, X, AM))
2959 return false;
2960
2961 // Try to fold the mask and shift into BEXTR and scale.
2962 if (!foldMaskedShiftToBEXTR(*CurDAG, N, Mask, Shift, X, AM, *Subtarget))
2963 return false;
2964 }
2965
2966 // Try to swap the mask and shift to place shifts which can be done as
2967 // a scale on the outside of the mask.
2968 if (!foldMaskedShiftToScaledMask(*CurDAG, N, AM))
2969 return false;
2970
2971 break;
2972 }
2973 case ISD::ZERO_EXTEND: {
2974 // Try to widen a zexted shift left to the same size as its use, so we can
2975 // match the shift as a scale factor.
2976 if (AM.IndexReg.getNode() != nullptr || AM.Scale != 1)
2977 break;
2978
2979 SDValue Src = N.getOperand(0);
2980
2981 // See if we can match a zext(addlike(x,c)).
2982 // TODO: Move more ZERO_EXTEND patterns into matchIndexRecursively.
2983 if (Src.getOpcode() == ISD::ADD || Src.getOpcode() == ISD::OR)
2984 if (SDValue Index = matchIndexRecursively(N, AM, Depth + 1))
2985 if (Index != N) {
2986 AM.IndexReg = Index;
2987 return false;
2988 }
2989
2990 // Peek through mask: zext(and(shl(x,c1),c2))
2991 APInt Mask = APInt::getAllOnes(Src.getScalarValueSizeInBits());
2992 if (Src.getOpcode() == ISD::AND && Src.hasOneUse())
2993 if (auto *MaskC = dyn_cast<ConstantSDNode>(Src.getOperand(1))) {
2994 Mask = MaskC->getAPIntValue();
2995 Src = Src.getOperand(0);
2996 }
2997
2998 if (Src.getOpcode() == ISD::SHL && Src.hasOneUse() && N->hasOneUse()) {
2999 // Give up if the shift is not a valid scale factor [1,2,3].
3000 SDValue ShlSrc = Src.getOperand(0);
3001 SDValue ShlAmt = Src.getOperand(1);
3002 auto *ShAmtC = dyn_cast<ConstantSDNode>(ShlAmt);
3003 if (!ShAmtC)
3004 break;
3005 unsigned ShAmtV = ShAmtC->getZExtValue();
3006 if (ShAmtV > 3)
3007 break;
3008
3009 // The narrow shift must only shift out zero bits (it must be 'nuw').
3010 // That makes it safe to widen to the destination type.
3011 APInt HighZeros =
3012 APInt::getHighBitsSet(ShlSrc.getValueSizeInBits(), ShAmtV);
3013 if (!Src->getFlags().hasNoUnsignedWrap() &&
3014 !CurDAG->MaskedValueIsZero(ShlSrc, HighZeros & Mask))
3015 break;
3016
3017 // zext (shl nuw i8 %x, C1) to i32
3018 // --> shl (zext i8 %x to i32), (zext C1)
3019 // zext (and (shl nuw i8 %x, C1), C2) to i32
3020 // --> shl (zext i8 (and %x, C2 >> C1) to i32), (zext C1)
3021 MVT SrcVT = ShlSrc.getSimpleValueType();
3022 MVT VT = N.getSimpleValueType();
3023 SDLoc DL(N);
3024
3025 SDValue Res = ShlSrc;
3026 if (!Mask.isAllOnes()) {
3027 Res = CurDAG->getConstant(Mask.lshr(ShAmtV), DL, SrcVT);
3028 insertDAGNode(*CurDAG, N, Res);
3029 Res = CurDAG->getNode(ISD::AND, DL, SrcVT, ShlSrc, Res);
3030 insertDAGNode(*CurDAG, N, Res);
3031 }
3032 SDValue Zext = CurDAG->getNode(ISD::ZERO_EXTEND, DL, VT, Res);
3033 insertDAGNode(*CurDAG, N, Zext);
3034 SDValue NewShl = CurDAG->getNode(ISD::SHL, DL, VT, Zext, ShlAmt);
3035 insertDAGNode(*CurDAG, N, NewShl);
3036 CurDAG->ReplaceAllUsesWith(N, NewShl);
3037 CurDAG->RemoveDeadNode(N.getNode());
3038
3039 // Convert the shift to scale factor.
3040 AM.Scale = 1 << ShAmtV;
3041 // If matchIndexRecursively is not called here,
3042 // Zext may be replaced by other nodes but later used to call a builder
3043 // method
3044 AM.IndexReg = matchIndexRecursively(Zext, AM, Depth + 1);
3045 return false;
3046 }
3047
3048 if (Src.getOpcode() == ISD::SRL && !Mask.isAllOnes()) {
3049 // Try to fold the mask and shift into an extract and scale.
3050 if (!foldMaskAndShiftToExtract(*CurDAG, N, Mask.getZExtValue(), Src,
3051 Src.getOperand(0), AM))
3052 return false;
3053
3054 // Try to fold the mask and shift directly into the scale.
3055 if (!foldMaskAndShiftToScale(*CurDAG, N, Mask.getZExtValue(), Src,
3056 Src.getOperand(0), AM))
3057 return false;
3058
3059 // Try to fold the mask and shift into BEXTR and scale.
3060 if (!foldMaskedShiftToBEXTR(*CurDAG, N, Mask.getZExtValue(), Src,
3061 Src.getOperand(0), AM, *Subtarget))
3062 return false;
3063 }
3064
3065 break;
3066 }
3067 }
3068
3069 return matchAddressBase(N, AM);
3070}
3071
3072/// Helper for MatchAddress. Add the specified node to the
3073/// specified addressing mode without any further recursion.
3074bool X86DAGToDAGISel::matchAddressBase(SDValue N, X86ISelAddressMode &AM) {
3075 // Is the base register already occupied?
3076 if (AM.BaseType != X86ISelAddressMode::RegBase || AM.Base_Reg.getNode()) {
3077 // If so, check to see if the scale index register is set.
3078 if (!AM.IndexReg.getNode()) {
3079 AM.IndexReg = N;
3080 AM.Scale = 1;
3081 return false;
3082 }
3083
3084 // Otherwise, we cannot select it.
3085 return true;
3086 }
3087
3088 // Default, generate it as a register.
3089 AM.BaseType = X86ISelAddressMode::RegBase;
3090 AM.Base_Reg = N;
3091 return false;
3092}
3093
3094bool X86DAGToDAGISel::matchVectorAddressRecursively(SDValue N,
3095 X86ISelAddressMode &AM,
3096 unsigned Depth) {
3097 LLVM_DEBUG({
3098 dbgs() << "MatchVectorAddress: ";
3099 AM.dump(CurDAG);
3100 });
3101 // Limit recursion.
3103 return matchAddressBase(N, AM);
3104
3105 // TODO: Support other operations.
3106 switch (N.getOpcode()) {
3107 case ISD::Constant: {
3108 uint64_t Val = cast<ConstantSDNode>(N)->getSExtValue();
3109 if (!foldOffsetIntoAddress(Val, AM))
3110 return false;
3111 break;
3112 }
3113 case X86ISD::Wrapper:
3114 if (!matchWrapper(N, AM))
3115 return false;
3116 break;
3117 case ISD::ADD: {
3118 // Add an artificial use to this node so that we can keep track of
3119 // it if it gets CSE'd with a different node.
3120 HandleSDNode Handle(N);
3121
3122 X86ISelAddressMode Backup = AM;
3123 if (!matchVectorAddressRecursively(N.getOperand(0), AM, Depth + 1) &&
3124 !matchVectorAddressRecursively(Handle.getValue().getOperand(1), AM,
3125 Depth + 1))
3126 return false;
3127 AM = Backup;
3128
3129 // Try again after commuting the operands.
3130 if (!matchVectorAddressRecursively(Handle.getValue().getOperand(1), AM,
3131 Depth + 1) &&
3132 !matchVectorAddressRecursively(Handle.getValue().getOperand(0), AM,
3133 Depth + 1))
3134 return false;
3135 AM = Backup;
3136
3137 N = Handle.getValue();
3138 break;
3139 }
3140 }
3141
3142 return matchAddressBase(N, AM);
3143}
3144
3145/// Helper for selectVectorAddr. Handles things that can be folded into a
3146/// gather/scatter address. The index register and scale should have already
3147/// been handled.
3148bool X86DAGToDAGISel::matchVectorAddress(SDValue N, X86ISelAddressMode &AM) {
3149 return matchVectorAddressRecursively(N, AM, 0);
3150}
3151
3152bool X86DAGToDAGISel::selectVectorAddr(MemSDNode *Parent, SDValue BasePtr,
3153 SDValue IndexOp, SDValue ScaleOp,
3154 SDValue &Base, SDValue &Scale,
3155 SDValue &Index, SDValue &Disp,
3156 SDValue &Segment) {
3157 X86ISelAddressMode AM;
3158 AM.Scale = ScaleOp->getAsZExtVal();
3159
3160 // Attempt to match index patterns, as long as we're not relying on implicit
3161 // sign-extension, which is performed BEFORE scale.
3162 if (IndexOp.getScalarValueSizeInBits() == BasePtr.getScalarValueSizeInBits())
3163 AM.IndexReg = matchIndexRecursively(IndexOp, AM, 0);
3164 else
3165 AM.IndexReg = IndexOp;
3166
3167 unsigned AddrSpace = Parent->getPointerInfo().getAddrSpace();
3168 if (AddrSpace == X86AS::GS)
3169 AM.Segment = CurDAG->getRegister(X86::GS, MVT::i16);
3170 if (AddrSpace == X86AS::FS)
3171 AM.Segment = CurDAG->getRegister(X86::FS, MVT::i16);
3172 if (AddrSpace == X86AS::SS)
3173 AM.Segment = CurDAG->getRegister(X86::SS, MVT::i16);
3174
3175 SDLoc DL(BasePtr);
3176 MVT VT = BasePtr.getSimpleValueType();
3177
3178 // Try to match into the base and displacement fields.
3179 if (matchVectorAddress(BasePtr, AM))
3180 return false;
3181
3182 getAddressOperands(AM, DL, VT, Base, Scale, Index, Disp, Segment);
3183 return true;
3184}
3185
3186/// Returns true if it is able to pattern match an addressing mode.
3187/// It returns the operands which make up the maximal addressing mode it can
3188/// match by reference.
3189///
3190/// Parent is the parent node of the addr operand that is being matched. It
3191/// is always a load, store, atomic node, or null. It is only null when
3192/// checking memory operands for inline asm nodes.
3193bool X86DAGToDAGISel::selectAddr(SDNode *Parent, SDValue N, SDValue &Base,
3194 SDValue &Scale, SDValue &Index, SDValue &Disp,
3195 SDValue &Segment, bool HasNDDM) {
3196 X86ISelAddressMode AM;
3197
3198 if (Parent &&
3199 // This list of opcodes are all the nodes that have an "addr:$ptr" operand
3200 // that are not a MemSDNode, and thus don't have proper addrspace info.
3201 Parent->getOpcode() != ISD::INTRINSIC_W_CHAIN && // unaligned loads, fixme
3202 Parent->getOpcode() != ISD::INTRINSIC_VOID && // nontemporal stores
3203 Parent->getOpcode() != X86ISD::TLSCALL && // Fixme
3204 Parent->getOpcode() != X86ISD::ENQCMD && // Fixme
3205 Parent->getOpcode() != X86ISD::ENQCMDS && // Fixme
3206 Parent->getOpcode() != X86ISD::EH_SJLJ_SETJMP && // setjmp
3207 Parent->getOpcode() != X86ISD::EH_SJLJ_LONGJMP) { // longjmp
3208 unsigned AddrSpace =
3209 cast<MemSDNode>(Parent)->getPointerInfo().getAddrSpace();
3210 if (AddrSpace == X86AS::GS)
3211 AM.Segment = CurDAG->getRegister(X86::GS, MVT::i16);
3212 if (AddrSpace == X86AS::FS)
3213 AM.Segment = CurDAG->getRegister(X86::FS, MVT::i16);
3214 if (AddrSpace == X86AS::SS)
3215 AM.Segment = CurDAG->getRegister(X86::SS, MVT::i16);
3216 }
3217
3218 // Save the DL and VT before calling matchAddress, it can invalidate N.
3219 SDLoc DL(N);
3220 MVT VT = N.getSimpleValueType();
3221
3222 if (matchAddress(N, AM))
3223 return false;
3224
3225 if (!HasNDDM && !AM.isRIPRelative())
3226 return false;
3227
3228 getAddressOperands(AM, DL, VT, Base, Scale, Index, Disp, Segment);
3229 return true;
3230}
3231
3232bool X86DAGToDAGISel::selectNDDAddr(SDNode *Parent, SDValue N, SDValue &Base,
3233 SDValue &Scale, SDValue &Index,
3234 SDValue &Disp, SDValue &Segment) {
3235 return selectAddr(Parent, N, Base, Scale, Index, Disp, Segment,
3236 Subtarget->hasNDDM());
3237}
3238
3239bool X86DAGToDAGISel::selectMOV64Imm32(SDValue N, SDValue &Imm) {
3240 // Cannot use 32 bit constants to reference objects in kernel/large code
3241 // model.
3242 if (TM.getCodeModel() == CodeModel::Kernel ||
3243 TM.getCodeModel() == CodeModel::Large)
3244 return false;
3245
3246 // In static codegen with small code model, we can get the address of a label
3247 // into a register with 'movl'
3248 if (N->getOpcode() != X86ISD::Wrapper)
3249 return false;
3250
3251 N = N.getOperand(0);
3252
3253 // At least GNU as does not accept 'movl' for TPOFF relocations.
3254 // FIXME: We could use 'movl' when we know we are targeting MC.
3255 if (N->getOpcode() == ISD::TargetGlobalTLSAddress)
3256 return false;
3257
3258 Imm = N;
3259 // Small/medium code model can reference non-TargetGlobalAddress objects with
3260 // 32 bit constants.
3261 if (N->getOpcode() != ISD::TargetGlobalAddress) {
3262 return TM.getCodeModel() == CodeModel::Small ||
3263 TM.getCodeModel() == CodeModel::Medium;
3264 }
3265
3266 const GlobalValue *GV = cast<GlobalAddressSDNode>(N)->getGlobal();
3267 if (std::optional<ConstantRange> CR = GV->getAbsoluteSymbolRange())
3268 return CR->getUnsignedMax().ult(1ull << 32);
3269
3270 return !TM.isLargeGlobalValue(GV);
3271}
3272
3273bool X86DAGToDAGISel::selectLEA64_Addr(SDValue N, SDValue &Base, SDValue &Scale,
3274 SDValue &Index, SDValue &Disp,
3275 SDValue &Segment) {
3276 // Save the debug loc before calling selectLEAAddr, in case it invalidates N.
3277 SDLoc DL(N);
3278
3279 if (!selectLEAAddr(N, Base, Scale, Index, Disp, Segment))
3280 return false;
3281
3282 EVT BaseType = Base.getValueType();
3283 unsigned SubReg;
3284 if (BaseType == MVT::i8)
3285 SubReg = X86::sub_8bit;
3286 else if (BaseType == MVT::i16)
3287 SubReg = X86::sub_16bit;
3288 else
3289 SubReg = X86::sub_32bit;
3290
3292 if (RN && RN->getReg() == 0)
3293 Base = CurDAG->getRegister(0, MVT::i64);
3294 else if ((BaseType == MVT::i8 || BaseType == MVT::i16 ||
3295 BaseType == MVT::i32) &&
3297 // Base could already be %rip, particularly in the x32 ABI.
3298 SDValue ImplDef = SDValue(CurDAG->getMachineNode(X86::IMPLICIT_DEF, DL,
3299 MVT::i64), 0);
3300 Base = CurDAG->getTargetInsertSubreg(SubReg, DL, MVT::i64, ImplDef, Base);
3301 }
3302
3303 [[maybe_unused]] EVT IndexType = Index.getValueType();
3305 if (RN && RN->getReg() == 0)
3306 Index = CurDAG->getRegister(0, MVT::i64);
3307 else {
3308 assert((IndexType == BaseType) &&
3309 "Expect to be extending 8/16/32-bit registers for use in LEA");
3310 SDValue ImplDef = SDValue(CurDAG->getMachineNode(X86::IMPLICIT_DEF, DL,
3311 MVT::i64), 0);
3312 Index = CurDAG->getTargetInsertSubreg(SubReg, DL, MVT::i64, ImplDef, Index);
3313 }
3314
3315 return true;
3316}
3317
3318/// Calls SelectAddr and determines if the maximal addressing
3319/// mode it matches can be cost effectively emitted as an LEA instruction.
3320bool X86DAGToDAGISel::selectLEAAddr(SDValue N,
3321 SDValue &Base, SDValue &Scale,
3322 SDValue &Index, SDValue &Disp,
3323 SDValue &Segment) {
3324 X86ISelAddressMode AM;
3325 AM.IsForLEA = true;
3326
3327 // Save the DL and VT before calling matchAddress, it can invalidate N.
3328 SDLoc DL(N);
3329 MVT VT = N.getSimpleValueType();
3330
3331 // Set AM.Segment to prevent MatchAddress from using one. LEA doesn't support
3332 // segments.
3333 SDValue Copy = AM.Segment;
3334 SDValue T = CurDAG->getRegister(0, MVT::i32);
3335 AM.Segment = T;
3336 if (matchAddress(N, AM))
3337 return false;
3338 assert (T == AM.Segment);
3339 AM.Segment = Copy;
3340
3341 unsigned Complexity = 0;
3342 if (AM.BaseType == X86ISelAddressMode::RegBase && AM.Base_Reg.getNode())
3343 Complexity = 1;
3344 else if (AM.BaseType == X86ISelAddressMode::FrameIndexBase)
3345 Complexity = 4;
3346
3347 if (AM.IndexReg.getNode())
3348 Complexity++;
3349
3350 // Don't match just leal(,%reg,2). It's cheaper to do addl %reg, %reg, or with
3351 // a simple shift.
3352 if (AM.Scale > 1)
3353 Complexity++;
3354
3355 // FIXME: We are artificially lowering the criteria to turn ADD %reg, $GA
3356 // to a LEA. This is determined with some experimentation but is by no means
3357 // optimal (especially for code size consideration). LEA is nice because of
3358 // its three-address nature. Tweak the cost function again when we can run
3359 // convertToThreeAddress() at register allocation time.
3360 if (AM.hasSymbolicDisplacement()) {
3361 // For X86-64, always use LEA to materialize RIP-relative addresses.
3362 if (Subtarget->is64Bit())
3363 Complexity = 4;
3364 else
3365 Complexity += 2;
3366 }
3367
3368 // Heuristic: try harder to form an LEA from ADD if the operands set flags.
3369 // Unlike ADD, LEA does not affect flags, so we will be less likely to require
3370 // duplicating flag-producing instructions later in the pipeline.
3371 if (N.getOpcode() == ISD::ADD) {
3372 auto isMathWithFlags = [](SDValue V) {
3373 switch (V.getOpcode()) {
3374 case X86ISD::ADD:
3375 case X86ISD::SUB:
3376 case X86ISD::ADC:
3377 case X86ISD::SBB:
3378 case X86ISD::SMUL:
3379 case X86ISD::UMUL:
3380 /* TODO: These opcodes can be added safely, but we may want to justify
3381 their inclusion for different reasons (better for reg-alloc).
3382 case X86ISD::OR:
3383 case X86ISD::XOR:
3384 case X86ISD::AND:
3385 */
3386 // Value 1 is the flag output of the node - verify it's not dead.
3387 return !SDValue(V.getNode(), 1).use_empty();
3388 default:
3389 return false;
3390 }
3391 };
3392 // TODO: We might want to factor in whether there's a load folding
3393 // opportunity for the math op that disappears with LEA.
3394 if (isMathWithFlags(N.getOperand(0)) || isMathWithFlags(N.getOperand(1)))
3395 Complexity++;
3396 }
3397
3398 if (AM.Disp)
3399 Complexity++;
3400
3401 // If it isn't worth using an LEA, reject it.
3402 if (Complexity <= 2)
3403 return false;
3404
3405 getAddressOperands(AM, DL, VT, Base, Scale, Index, Disp, Segment);
3406 return true;
3407}
3408
3409/// This is only run on TargetGlobalTLSAddress nodes.
3410bool X86DAGToDAGISel::selectTLSADDRAddr(SDValue N, SDValue &Base,
3411 SDValue &Scale, SDValue &Index,
3412 SDValue &Disp, SDValue &Segment) {
3413 assert(N.getOpcode() == ISD::TargetGlobalTLSAddress ||
3414 N.getOpcode() == ISD::TargetExternalSymbol);
3415
3416 X86ISelAddressMode AM;
3417 if (auto *GA = dyn_cast<GlobalAddressSDNode>(N)) {
3418 AM.GV = GA->getGlobal();
3419 AM.Disp += GA->getOffset();
3420 AM.SymbolFlags = GA->getTargetFlags();
3421 } else {
3422 auto *SA = cast<ExternalSymbolSDNode>(N);
3423 AM.ES = SA->getSymbol();
3424 AM.SymbolFlags = SA->getTargetFlags();
3425 }
3426
3427 if (Subtarget->is32Bit()) {
3428 AM.Scale = 1;
3429 AM.IndexReg = CurDAG->getRegister(X86::EBX, MVT::i32);
3430 }
3431
3432 MVT VT = N.getSimpleValueType();
3433 getAddressOperands(AM, SDLoc(N), VT, Base, Scale, Index, Disp, Segment);
3434 return true;
3435}
3436
3437bool X86DAGToDAGISel::selectRelocImm(SDValue N, SDValue &Op) {
3438 // Keep track of the original value type and whether this value was
3439 // truncated. If we see a truncation from pointer type to VT that truncates
3440 // bits that are known to be zero, we can use a narrow reference.
3441 EVT VT = N.getValueType();
3442 bool WasTruncated = false;
3443 if (N.getOpcode() == ISD::TRUNCATE) {
3444 WasTruncated = true;
3445 N = N.getOperand(0);
3446 }
3447
3448 if (N.getOpcode() != X86ISD::Wrapper)
3449 return false;
3450
3451 // We can only use non-GlobalValues as immediates if they were not truncated,
3452 // as we do not have any range information. If we have a GlobalValue and the
3453 // address was not truncated, we can select it as an operand directly.
3454 unsigned Opc = N.getOperand(0)->getOpcode();
3455 if (Opc != ISD::TargetGlobalAddress || !WasTruncated) {
3456 Op = N.getOperand(0);
3457 // We can only select the operand directly if we didn't have to look past a
3458 // truncate.
3459 return !WasTruncated;
3460 }
3461
3462 // Check that the global's range fits into VT.
3463 auto *GA = cast<GlobalAddressSDNode>(N.getOperand(0));
3464 std::optional<ConstantRange> CR = GA->getGlobal()->getAbsoluteSymbolRange();
3465 if (!CR || CR->getUnsignedMax().uge(1ull << VT.getSizeInBits()))
3466 return false;
3467
3468 // Okay, we can use a narrow reference.
3469 Op = CurDAG->getTargetGlobalAddress(GA->getGlobal(), SDLoc(N), VT,
3470 GA->getOffset(), GA->getTargetFlags());
3471 return true;
3472}
3473
3474bool X86DAGToDAGISel::tryFoldLoad(SDNode *Root, SDNode *P, SDValue N,
3475 SDValue &Base, SDValue &Scale,
3476 SDValue &Index, SDValue &Disp,
3477 SDValue &Segment) {
3478 assert(Root && P && "Unknown root/parent nodes");
3479 if (!ISD::isNON_EXTLoad(N.getNode()) ||
3480 !IsProfitableToFold(N, P, Root) ||
3481 !IsLegalToFold(N, P, Root, OptLevel))
3482 return false;
3483
3484 return selectAddr(N.getNode(),
3485 N.getOperand(1), Base, Scale, Index, Disp, Segment);
3486}
3487
3488bool X86DAGToDAGISel::tryFoldBroadcast(SDNode *Root, SDNode *P, SDValue N,
3489 SDValue &Base, SDValue &Scale,
3490 SDValue &Index, SDValue &Disp,
3491 SDValue &Segment) {
3492 assert(Root && P && "Unknown root/parent nodes");
3493 if (N->getOpcode() != X86ISD::VBROADCAST_LOAD ||
3494 !IsProfitableToFold(N, P, Root) ||
3495 !IsLegalToFold(N, P, Root, OptLevel))
3496 return false;
3497
3498 return selectAddr(N.getNode(),
3499 N.getOperand(1), Base, Scale, Index, Disp, Segment);
3500}
3501
3502/// Return an SDNode that returns the value of the global base register.
3503/// Output instructions required to initialize the global base register,
3504/// if necessary.
3505SDNode *X86DAGToDAGISel::getGlobalBaseReg() {
3506 Register GlobalBaseReg = getInstrInfo()->getGlobalBaseReg(MF);
3507 auto &DL = MF->getDataLayout();
3508 return CurDAG->getRegister(GlobalBaseReg, TLI->getPointerTy(DL)).getNode();
3509}
3510
3511bool X86DAGToDAGISel::isSExtAbsoluteSymbolRef(unsigned Width, SDNode *N) const {
3512 if (N->getOpcode() == ISD::TRUNCATE)
3513 N = N->getOperand(0).getNode();
3514 if (N->getOpcode() != X86ISD::Wrapper)
3515 return false;
3516
3517 auto *GA = dyn_cast<GlobalAddressSDNode>(N->getOperand(0));
3518 if (!GA)
3519 return false;
3520
3521 auto *GV = GA->getGlobal();
3522 std::optional<ConstantRange> CR = GV->getAbsoluteSymbolRange();
3523 if (CR)
3524 return CR->getSignedMin().sge(-1ull << Width) &&
3525 CR->getSignedMax().slt(1ull << Width);
3526 // In the kernel code model, globals are in the negative 2GB of the address
3527 // space, so globals can be a sign extended 32-bit immediate.
3528 // In other code models, small globals are in the low 2GB of the address
3529 // space, so sign extending them is equivalent to zero extending them.
3530 return TM.getCodeModel() != CodeModel::Large && Width == 32 &&
3531 !TM.isLargeGlobalValue(GV);
3532}
3533
3534X86::CondCode X86DAGToDAGISel::getCondFromNode(SDNode *N) const {
3535 assert(N->isMachineOpcode() && "Unexpected node");
3536 unsigned Opc = N->getMachineOpcode();
3537 const MCInstrDesc &MCID = getInstrInfo()->get(Opc);
3538 int CondNo = X86::getCondSrcNoFromDesc(MCID);
3539 if (CondNo < 0)
3540 return X86::COND_INVALID;
3541
3542 return static_cast<X86::CondCode>(N->getConstantOperandVal(CondNo));
3543}
3544
3545/// Test whether the given X86ISD::CMP node has any users that use a flag
3546/// other than ZF.
3547bool X86DAGToDAGISel::onlyUsesZeroFlag(SDValue Flags) const {
3548 // Examine each user of the node.
3549 for (SDUse &Use : Flags->uses()) {
3550 // Only check things that use the flags.
3551 if (Use.getResNo() != Flags.getResNo())
3552 continue;
3553 SDNode *User = Use.getUser();
3554 // Only examine CopyToReg uses that copy to EFLAGS.
3555 if (User->getOpcode() != ISD::CopyToReg ||
3556 cast<RegisterSDNode>(User->getOperand(1))->getReg() != X86::EFLAGS)
3557 return false;
3558 // Examine each user of the CopyToReg use.
3559 for (SDUse &FlagUse : User->uses()) {
3560 // Only examine the Flag result.
3561 if (FlagUse.getResNo() != 1)
3562 continue;
3563 // Anything unusual: assume conservatively.
3564 if (!FlagUse.getUser()->isMachineOpcode())
3565 return false;
3566 // Examine the condition code of the user.
3567 X86::CondCode CC = getCondFromNode(FlagUse.getUser());
3568
3569 switch (CC) {
3570 // Comparisons which only use the zero flag.
3571 case X86::COND_E: case X86::COND_NE:
3572 continue;
3573 // Anything else: assume conservatively.
3574 default:
3575 return false;
3576 }
3577 }
3578 }
3579 return true;
3580}
3581
3582/// Test whether the given X86ISD::CMP node has any uses which require the SF
3583/// flag to be accurate.
3584bool X86DAGToDAGISel::hasNoSignFlagUses(SDValue Flags) const {
3585 // Examine each user of the node.
3586 for (SDUse &Use : Flags->uses()) {
3587 // Only check things that use the flags.
3588 if (Use.getResNo() != Flags.getResNo())
3589 continue;
3590 SDNode *User = Use.getUser();
3591 // Only examine CopyToReg uses that copy to EFLAGS.
3592 if (User->getOpcode() != ISD::CopyToReg ||
3593 cast<RegisterSDNode>(User->getOperand(1))->getReg() != X86::EFLAGS)
3594 return false;
3595 // Examine each user of the CopyToReg use.
3596 for (SDUse &FlagUse : User->uses()) {
3597 // Only examine the Flag result.
3598 if (FlagUse.getResNo() != 1)
3599 continue;
3600 // Anything unusual: assume conservatively.
3601 if (!FlagUse.getUser()->isMachineOpcode())
3602 return false;
3603 // Examine the condition code of the user.
3604 X86::CondCode CC = getCondFromNode(FlagUse.getUser());
3605
3606 switch (CC) {
3607 // Comparisons which don't examine the SF flag.
3608 case X86::COND_A: case X86::COND_AE:
3609 case X86::COND_B: case X86::COND_BE:
3610 case X86::COND_E: case X86::COND_NE:
3611 case X86::COND_O: case X86::COND_NO:
3612 case X86::COND_P: case X86::COND_NP:
3613 continue;
3614 // Anything else: assume conservatively.
3615 default:
3616 return false;
3617 }
3618 }
3619 }
3620 return true;
3621}
3622
3624 switch (CC) {
3625 // Comparisons which don't examine the CF flag.
3626 case X86::COND_O: case X86::COND_NO:
3627 case X86::COND_E: case X86::COND_NE:
3628 case X86::COND_S: case X86::COND_NS:
3629 case X86::COND_P: case X86::COND_NP:
3630 case X86::COND_L: case X86::COND_GE:
3631 case X86::COND_G: case X86::COND_LE:
3632 return false;
3633 // Anything else: assume conservatively.
3634 default:
3635 return true;
3636 }
3637}
3638
3639/// Test whether the given node which sets flags has any uses which require the
3640/// CF flag to be accurate.
3641 bool X86DAGToDAGISel::hasNoCarryFlagUses(SDValue Flags) const {
3642 // Examine each user of the node.
3643 for (SDUse &Use : Flags->uses()) {
3644 // Only check things that use the flags.
3645 if (Use.getResNo() != Flags.getResNo())
3646 continue;
3647
3648 SDNode *User = Use.getUser();
3649 unsigned UserOpc = User->getOpcode();
3650
3651 if (UserOpc == ISD::CopyToReg) {
3652 // Only examine CopyToReg uses that copy to EFLAGS.
3653 if (cast<RegisterSDNode>(User->getOperand(1))->getReg() != X86::EFLAGS)
3654 return false;
3655 // Examine each user of the CopyToReg use.
3656 for (SDUse &FlagUse : User->uses()) {
3657 // Only examine the Flag result.
3658 if (FlagUse.getResNo() != 1)
3659 continue;
3660 // Anything unusual: assume conservatively.
3661 if (!FlagUse.getUser()->isMachineOpcode())
3662 return false;
3663 // Examine the condition code of the user.
3664 X86::CondCode CC = getCondFromNode(FlagUse.getUser());
3665
3666 if (mayUseCarryFlag(CC))
3667 return false;
3668 }
3669
3670 // This CopyToReg is ok. Move on to the next user.
3671 continue;
3672 }
3673
3674 // This might be an unselected node. So look for the pre-isel opcodes that
3675 // use flags.
3676 unsigned CCOpNo;
3677 switch (UserOpc) {
3678 default:
3679 // Something unusual. Be conservative.
3680 return false;
3681 case X86ISD::SETCC: CCOpNo = 0; break;
3682 case X86ISD::SETCC_CARRY: CCOpNo = 0; break;
3683 case X86ISD::CMOV: CCOpNo = 2; break;
3684 case X86ISD::BRCOND: CCOpNo = 2; break;
3685 }
3686
3687 X86::CondCode CC = (X86::CondCode)User->getConstantOperandVal(CCOpNo);
3688 if (mayUseCarryFlag(CC))
3689 return false;
3690 }
3691 return true;
3692}
3693
3694/// Return true if \p Addr may be matched with a non-fixed frame index as base.
3696 const MachineFrameInfo &MFI,
3697 unsigned Depth = 0) {
3698 if (auto *FI = dyn_cast<FrameIndexSDNode>(Addr))
3699 return !MFI.isFixedObjectIndex(FI->getIndex());
3700 // Assume the worst if we can't see the whole address expression.
3702 return true;
3703 switch (Addr.getOpcode()) {
3704 case ISD::ADD:
3705 case ISD::OR:
3706 case ISD::XOR:
3707 return addrMayUseNonFixedFrameIndex(Addr.getOperand(0), MFI, Depth + 1) ||
3709 case ISD::SUB:
3710 return addrMayUseNonFixedFrameIndex(Addr.getOperand(0), MFI, Depth + 1);
3711 default:
3712 // Only add-like nodes and the LHS of a SUB can fold a frame index into the
3713 // base; anything else is matched as a register or symbol base.
3714 return false;
3715 }
3716}
3717
3718bool X86DAGToDAGISel::checkTCRetEnoughRegs(SDNode *N) const {
3719 assert(N->getOpcode() == X86ISD::TC_RETURN);
3720 // X86tcret args: (*chain, ptr, imm, regs..., glue)
3721 const SDValue &BasePtr = cast<LoadSDNode>(N->getOperand(1))->getBasePtr();
3722
3723 // The tail call executes after the epilogue, where only fixed stack objects
3724 // can still be addressed (the stack may end up realigned).
3725 if (addrMayUseNonFixedFrameIndex(BasePtr, MF->getFrameInfo()))
3726 return false;
3727
3728 // Check that there is enough volatile registers to load the callee address.
3729
3730 const X86RegisterInfo *RI = Subtarget->getRegisterInfo();
3731 unsigned AvailGPRs;
3732 // The register classes below must stay in sync with what's used for
3733 // TCRETURNri, TCRETURN_HIPE32ri, TCRETURN_WIN64ri, etc).
3734 if (Subtarget->is64Bit()) {
3735 const TargetRegisterClass *TCGPRs =
3736 Subtarget->isCallingConvWin64(MF->getFunction().getCallingConv())
3737 ? &X86::GR64_TCW64RegClass
3738 : &X86::GR64_TCRegClass;
3739 // Can't use RSP or RIP for the load in general.
3740 assert(TCGPRs->contains(X86::RSP));
3741 assert(TCGPRs->contains(X86::RIP));
3742 AvailGPRs = TCGPRs->getNumRegs() - 2;
3743 } else {
3744 const TargetRegisterClass *TCGPRs =
3745 MF->getFunction().getCallingConv() == CallingConv::HiPE
3746 ? &X86::GR32RegClass
3747 : &X86::GR32_TCRegClass;
3748 // Can't use ESP for the address in general.
3749 assert(TCGPRs->contains(X86::ESP));
3750 AvailGPRs = TCGPRs->getNumRegs() - 1;
3751 }
3752
3753 // The load's base and index need up to two registers.
3754 unsigned LoadGPRs = 2;
3755
3756 if (Subtarget->is32Bit()) {
3757 // FIXME: This was carried from X86tcret_1reg which was used for 32-bit,
3758 // but it could apply to 64-bit too.
3759 if (isa<FrameIndexSDNode>(BasePtr)) {
3760 LoadGPRs -= 2; // Base is fixed index off ESP; no regs needed.
3761 } else if (BasePtr.getOpcode() == X86ISD::Wrapper &&
3762 isa<GlobalAddressSDNode>(BasePtr->getOperand(0))) {
3763 if (getTargetMachine().isPositionIndependent())
3764 return false;
3765 LoadGPRs -= 1; // Base is a global (immediate since this is non-PIC), no
3766 // reg needed.
3767 }
3768 }
3769
3770 unsigned ArgGPRs = 0;
3771 for (unsigned I = 3, E = N->getNumOperands(); I != E; ++I) {
3772 if (const auto *RN = dyn_cast<RegisterSDNode>(N->getOperand(I))) {
3773 if (!RI->isGeneralPurposeRegister(*MF, RN->getReg()))
3774 continue;
3775 if (++ArgGPRs + LoadGPRs > AvailGPRs)
3776 return false;
3777 }
3778 }
3779
3780 return true;
3781}
3782
3783/// Check whether or not the chain ending in StoreNode is suitable for doing
3784/// the {load; op; store} to modify transformation.
3786 SDValue StoredVal, SelectionDAG *CurDAG,
3787 unsigned LoadOpNo,
3788 LoadSDNode *&LoadNode,
3789 SDValue &InputChain) {
3790 // Is the stored value result 0 of the operation?
3791 if (StoredVal.getResNo() != 0) return false;
3792
3793 // Are there other uses of the operation other than the store?
3794 if (!StoredVal.getNode()->hasNUsesOfValue(1, 0)) return false;
3795
3796 // Is the store non-extending and non-indexed?
3797 if (!ISD::isNormalStore(StoreNode) || StoreNode->isNonTemporal())
3798 return false;
3799
3800 SDValue Load = StoredVal->getOperand(LoadOpNo);
3801 // Is the stored value a non-extending and non-indexed load?
3802 if (!ISD::isNormalLoad(Load.getNode())) return false;
3803
3804 // Return LoadNode by reference.
3805 LoadNode = cast<LoadSDNode>(Load);
3806
3807 // Is store the only read of the loaded value?
3808 if (!Load.hasOneUse())
3809 return false;
3810
3811 // Is the address of the store the same as the load?
3812 if (LoadNode->getBasePtr() != StoreNode->getBasePtr() ||
3813 LoadNode->getOffset() != StoreNode->getOffset())
3814 return false;
3815
3816 bool FoundLoad = false;
3817 SmallVector<SDValue, 4> ChainOps;
3818 SmallVector<const SDNode *, 4> LoopWorklist;
3820 const unsigned int Max = 1024;
3821
3822 // Visualization of Load-Op-Store fusion:
3823 // -------------------------
3824 // Legend:
3825 // *-lines = Chain operand dependencies.
3826 // |-lines = Normal operand dependencies.
3827 // Dependencies flow down and right. n-suffix references multiple nodes.
3828 //
3829 // C Xn C
3830 // * * *
3831 // * * *
3832 // Xn A-LD Yn TF Yn
3833 // * * \ | * |
3834 // * * \ | * |
3835 // * * \ | => A--LD_OP_ST
3836 // * * \| \
3837 // TF OP \
3838 // * | \ Zn
3839 // * | \
3840 // A-ST Zn
3841 //
3842
3843 // This merge induced dependences from: #1: Xn -> LD, OP, Zn
3844 // #2: Yn -> LD
3845 // #3: ST -> Zn
3846
3847 // Ensure the transform is safe by checking for the dual
3848 // dependencies to make sure we do not induce a loop.
3849
3850 // As LD is a predecessor to both OP and ST we can do this by checking:
3851 // a). if LD is a predecessor to a member of Xn or Yn.
3852 // b). if a Zn is a predecessor to ST.
3853
3854 // However, (b) can only occur through being a chain predecessor to
3855 // ST, which is the same as Zn being a member or predecessor of Xn,
3856 // which is a subset of LD being a predecessor of Xn. So it's
3857 // subsumed by check (a).
3858
3859 SDValue Chain = StoreNode->getChain();
3860
3861 // Gather X elements in ChainOps.
3862 if (Chain == Load.getValue(1)) {
3863 FoundLoad = true;
3864 ChainOps.push_back(Load.getOperand(0));
3865 } else if (Chain.getOpcode() == ISD::TokenFactor) {
3866 for (unsigned i = 0, e = Chain.getNumOperands(); i != e; ++i) {
3867 SDValue Op = Chain.getOperand(i);
3868 if (Op == Load.getValue(1)) {
3869 FoundLoad = true;
3870 // Drop Load, but keep its chain. No cycle check necessary.
3871 ChainOps.push_back(Load.getOperand(0));
3872 continue;
3873 }
3874 LoopWorklist.push_back(Op.getNode());
3875 ChainOps.push_back(Op);
3876 }
3877 }
3878
3879 if (!FoundLoad)
3880 return false;
3881
3882 // Worklist is currently Xn. Add Yn to worklist.
3883 for (SDValue Op : StoredVal->ops())
3884 if (Op.getNode() != LoadNode)
3885 LoopWorklist.push_back(Op.getNode());
3886
3887 // Check (a) if Load is a predecessor to Xn + Yn
3888 if (SDNode::hasPredecessorHelper(Load.getNode(), Visited, LoopWorklist, Max,
3889 true))
3890 return false;
3891
3892 InputChain =
3893 CurDAG->getNode(ISD::TokenFactor, SDLoc(Chain), MVT::Other, ChainOps);
3894 return true;
3895}
3896
3897// Change a chain of {load; op; store} of the same value into a simple op
3898// through memory of that value, if the uses of the modified value and its
3899// address are suitable.
3900//
3901// The tablegen pattern memory operand pattern is currently not able to match
3902// the case where the EFLAGS on the original operation are used.
3903//
3904// To move this to tablegen, we'll need to improve tablegen to allow flags to
3905// be transferred from a node in the pattern to the result node, probably with
3906// a new keyword. For example, we have this
3907// def DEC64m : RI<0xFF, MRM1m, (outs), (ins i64mem:$dst), "dec{q}\t$dst",
3908// [(store (add (loadi64 addr:$dst), -1), addr:$dst)]>;
3909// but maybe need something like this
3910// def DEC64m : RI<0xFF, MRM1m, (outs), (ins i64mem:$dst), "dec{q}\t$dst",
3911// [(store (X86add_flag (loadi64 addr:$dst), -1), addr:$dst),
3912// (transferrable EFLAGS)]>;
3913//
3914// Until then, we manually fold these and instruction select the operation
3915// here.
3916bool X86DAGToDAGISel::foldLoadStoreIntoMemOperand(SDNode *Node) {
3917 auto *StoreNode = cast<StoreSDNode>(Node);
3918 SDValue StoredVal = StoreNode->getOperand(1);
3919 unsigned Opc = StoredVal->getOpcode();
3920
3921 // Before we try to select anything, make sure this is memory operand size
3922 // and opcode we can handle. Note that this must match the code below that
3923 // actually lowers the opcodes.
3924 EVT MemVT = StoreNode->getMemoryVT();
3925 if (MemVT != MVT::i64 && MemVT != MVT::i32 && MemVT != MVT::i16 &&
3926 MemVT != MVT::i8)
3927 return false;
3928
3929 bool IsCommutable = false;
3930 bool IsNegate = false;
3931 switch (Opc) {
3932 default:
3933 return false;
3934 case X86ISD::SUB:
3935 IsNegate = isNullConstant(StoredVal.getOperand(0));
3936 break;
3937 case X86ISD::SBB:
3938 break;
3939 case X86ISD::ADD:
3940 case X86ISD::ADC:
3941 case X86ISD::AND:
3942 case X86ISD::OR:
3943 case X86ISD::XOR:
3944 IsCommutable = true;
3945 break;
3946 }
3947
3948 unsigned LoadOpNo = IsNegate ? 1 : 0;
3949 LoadSDNode *LoadNode = nullptr;
3950 SDValue InputChain;
3951 if (!isFusableLoadOpStorePattern(StoreNode, StoredVal, CurDAG, LoadOpNo,
3952 LoadNode, InputChain)) {
3953 if (!IsCommutable)
3954 return false;
3955
3956 // This operation is commutable, try the other operand.
3957 LoadOpNo = 1;
3958 if (!isFusableLoadOpStorePattern(StoreNode, StoredVal, CurDAG, LoadOpNo,
3959 LoadNode, InputChain))
3960 return false;
3961 }
3962
3963 SDValue Base, Scale, Index, Disp, Segment;
3964 if (!selectAddr(LoadNode, LoadNode->getBasePtr(), Base, Scale, Index, Disp,
3965 Segment))
3966 return false;
3967
3968 auto SelectOpcode = [&](unsigned Opc64, unsigned Opc32, unsigned Opc16,
3969 unsigned Opc8) {
3970 switch (MemVT.getSimpleVT().SimpleTy) {
3971 case MVT::i64:
3972 return Opc64;
3973 case MVT::i32:
3974 return Opc32;
3975 case MVT::i16:
3976 return Opc16;
3977 case MVT::i8:
3978 return Opc8;
3979 default:
3980 llvm_unreachable("Invalid size!");
3981 }
3982 };
3983
3984 MachineSDNode *Result;
3985 switch (Opc) {
3986 case X86ISD::SUB:
3987 // Handle negate.
3988 if (IsNegate) {
3989 unsigned NewOpc = SelectOpcode(X86::NEG64m, X86::NEG32m, X86::NEG16m,
3990 X86::NEG8m);
3991 const SDValue Ops[] = {Base, Scale, Index, Disp, Segment, InputChain};
3992 Result = CurDAG->getMachineNode(NewOpc, SDLoc(Node), MVT::i32,
3993 MVT::Other, Ops);
3994 break;
3995 }
3996 [[fallthrough]];
3997 case X86ISD::ADD:
3998 // Try to match inc/dec.
3999 if (!Subtarget->slowIncDec() || CurDAG->shouldOptForSize()) {
4000 bool IsOne = isOneConstant(StoredVal.getOperand(1));
4001 bool IsNegOne = isAllOnesConstant(StoredVal.getOperand(1));
4002 // ADD/SUB with 1/-1 and carry flag isn't used can use inc/dec.
4003 if ((IsOne || IsNegOne) && hasNoCarryFlagUses(StoredVal.getValue(1))) {
4004 unsigned NewOpc =
4005 ((Opc == X86ISD::ADD) == IsOne)
4006 ? SelectOpcode(X86::INC64m, X86::INC32m, X86::INC16m, X86::INC8m)
4007 : SelectOpcode(X86::DEC64m, X86::DEC32m, X86::DEC16m, X86::DEC8m);
4008 const SDValue Ops[] = {Base, Scale, Index, Disp, Segment, InputChain};
4009 Result = CurDAG->getMachineNode(NewOpc, SDLoc(Node), MVT::i32,
4010 MVT::Other, Ops);
4011 break;
4012 }
4013 }
4014 [[fallthrough]];
4015 case X86ISD::ADC:
4016 case X86ISD::SBB:
4017 case X86ISD::AND:
4018 case X86ISD::OR:
4019 case X86ISD::XOR: {
4020 auto SelectRegOpcode = [SelectOpcode](unsigned Opc) {
4021 switch (Opc) {
4022 case X86ISD::ADD:
4023 return SelectOpcode(X86::ADD64mr, X86::ADD32mr, X86::ADD16mr,
4024 X86::ADD8mr);
4025 case X86ISD::ADC:
4026 return SelectOpcode(X86::ADC64mr, X86::ADC32mr, X86::ADC16mr,
4027 X86::ADC8mr);
4028 case X86ISD::SUB:
4029 return SelectOpcode(X86::SUB64mr, X86::SUB32mr, X86::SUB16mr,
4030 X86::SUB8mr);
4031 case X86ISD::SBB:
4032 return SelectOpcode(X86::SBB64mr, X86::SBB32mr, X86::SBB16mr,
4033 X86::SBB8mr);
4034 case X86ISD::AND:
4035 return SelectOpcode(X86::AND64mr, X86::AND32mr, X86::AND16mr,
4036 X86::AND8mr);
4037 case X86ISD::OR:
4038 return SelectOpcode(X86::OR64mr, X86::OR32mr, X86::OR16mr, X86::OR8mr);
4039 case X86ISD::XOR:
4040 return SelectOpcode(X86::XOR64mr, X86::XOR32mr, X86::XOR16mr,
4041 X86::XOR8mr);
4042 default:
4043 llvm_unreachable("Invalid opcode!");
4044 }
4045 };
4046 auto SelectImmOpcode = [SelectOpcode](unsigned Opc) {
4047 switch (Opc) {
4048 case X86ISD::ADD:
4049 return SelectOpcode(X86::ADD64mi32, X86::ADD32mi, X86::ADD16mi,
4050 X86::ADD8mi);
4051 case X86ISD::ADC:
4052 return SelectOpcode(X86::ADC64mi32, X86::ADC32mi, X86::ADC16mi,
4053 X86::ADC8mi);
4054 case X86ISD::SUB:
4055 return SelectOpcode(X86::SUB64mi32, X86::SUB32mi, X86::SUB16mi,
4056 X86::SUB8mi);
4057 case X86ISD::SBB:
4058 return SelectOpcode(X86::SBB64mi32, X86::SBB32mi, X86::SBB16mi,
4059 X86::SBB8mi);
4060 case X86ISD::AND:
4061 return SelectOpcode(X86::AND64mi32, X86::AND32mi, X86::AND16mi,
4062 X86::AND8mi);
4063 case X86ISD::OR:
4064 return SelectOpcode(X86::OR64mi32, X86::OR32mi, X86::OR16mi,
4065 X86::OR8mi);
4066 case X86ISD::XOR:
4067 return SelectOpcode(X86::XOR64mi32, X86::XOR32mi, X86::XOR16mi,
4068 X86::XOR8mi);
4069 default:
4070 llvm_unreachable("Invalid opcode!");
4071 }
4072 };
4073
4074 unsigned NewOpc = SelectRegOpcode(Opc);
4075 SDValue Operand = StoredVal->getOperand(1-LoadOpNo);
4076
4077 // See if the operand is a constant that we can fold into an immediate
4078 // operand.
4079 if (auto *OperandC = dyn_cast<ConstantSDNode>(Operand)) {
4080 int64_t OperandV = OperandC->getSExtValue();
4081
4082 // Check if we can shrink the operand enough to fit in an immediate (or
4083 // fit into a smaller immediate) by negating it and switching the
4084 // operation.
4085 if ((Opc == X86ISD::ADD || Opc == X86ISD::SUB) &&
4086 ((MemVT != MVT::i8 && !isInt<8>(OperandV) && isInt<8>(-OperandV)) ||
4087 (MemVT == MVT::i64 && !isInt<32>(OperandV) &&
4088 isInt<32>(-OperandV))) &&
4089 hasNoCarryFlagUses(StoredVal.getValue(1))) {
4090 OperandV = -OperandV;
4091 Opc = Opc == X86ISD::ADD ? X86ISD::SUB : X86ISD::ADD;
4092 }
4093
4094 if (MemVT != MVT::i64 || isInt<32>(OperandV)) {
4095 Operand = CurDAG->getSignedTargetConstant(OperandV, SDLoc(Node), MemVT);
4096 NewOpc = SelectImmOpcode(Opc);
4097 }
4098 }
4099
4100 if (Opc == X86ISD::ADC || Opc == X86ISD::SBB) {
4101 SDValue CopyTo =
4102 CurDAG->getCopyToReg(InputChain, SDLoc(Node), X86::EFLAGS,
4103 StoredVal.getOperand(2), SDValue());
4104
4105 const SDValue Ops[] = {Base, Scale, Index, Disp,
4106 Segment, Operand, CopyTo, CopyTo.getValue(1)};
4107 Result = CurDAG->getMachineNode(NewOpc, SDLoc(Node), MVT::i32, MVT::Other,
4108 Ops);
4109 } else {
4110 const SDValue Ops[] = {Base, Scale, Index, Disp,
4111 Segment, Operand, InputChain};
4112 Result = CurDAG->getMachineNode(NewOpc, SDLoc(Node), MVT::i32, MVT::Other,
4113 Ops);
4114 }
4115 break;
4116 }
4117 default:
4118 llvm_unreachable("Invalid opcode!");
4119 }
4120
4121 MachineMemOperand *MemOps[] = {StoreNode->getMemOperand(),
4122 LoadNode->getMemOperand()};
4123 CurDAG->setNodeMemRefs(Result, MemOps);
4124
4125 // Update Load Chain uses as well.
4126 ReplaceUses(SDValue(LoadNode, 1), SDValue(Result, 1));
4127 ReplaceUses(SDValue(StoreNode, 0), SDValue(Result, 1));
4128 ReplaceUses(SDValue(StoredVal.getNode(), 1), SDValue(Result, 0));
4129 CurDAG->RemoveDeadNode(Node);
4130 return true;
4131}
4132
4133// See if this is an X & Mask that we can match to BEXTR/BZHI.
4134// Where Mask is one of the following patterns:
4135// a) x & (1 << nbits) - 1
4136// b) x & ~(-1 << nbits)
4137// c) x & (-1 >> (32 - y))
4138// d) x << (32 - y) >> (32 - y)
4139// e) (1 << nbits) - 1
4140// f) ~(-1 << nbits)
4141bool X86DAGToDAGISel::matchBitExtract(SDNode *Node) {
4142 assert((Node->getOpcode() == ISD::ADD || Node->getOpcode() == ISD::AND ||
4143 Node->getOpcode() == ISD::XOR || Node->getOpcode() == ISD::SRL) &&
4144 "Should be either an and-mask, a standalone low-bits mask, or "
4145 "right-shift after clearing high bits.");
4146
4147 // BEXTR is BMI instruction, BZHI is BMI2 instruction. We need at least one.
4148 if (!Subtarget->hasBMI() && !Subtarget->hasBMI2())
4149 return false;
4150
4151 MVT NVT = Node->getSimpleValueType(0);
4152
4153 // Only supported for 32 and 64 bits.
4154 if (NVT != MVT::i32 && NVT != MVT::i64)
4155 return false;
4156
4157 SDValue NBits;
4158 bool NegateNBits;
4159
4160 // If we have BMI2's BZHI, we are ok with muti-use patterns.
4161 // Else, if we only have BMI1's BEXTR, we require one-use.
4162 const bool AllowExtraUsesByDefault = Subtarget->hasBMI2();
4163 auto checkUses = [AllowExtraUsesByDefault](
4164 SDValue Op, unsigned NUses,
4165 std::optional<bool> AllowExtraUses) {
4166 return AllowExtraUses.value_or(AllowExtraUsesByDefault) ||
4167 Op.getNode()->hasNUsesOfValue(NUses, Op.getResNo());
4168 };
4169 auto checkOneUse = [checkUses](SDValue Op,
4170 std::optional<bool> AllowExtraUses =
4171 std::nullopt) {
4172 return checkUses(Op, 1, AllowExtraUses);
4173 };
4174 auto checkTwoUse = [checkUses](SDValue Op,
4175 std::optional<bool> AllowExtraUses =
4176 std::nullopt) {
4177 return checkUses(Op, 2, AllowExtraUses);
4178 };
4179
4180 auto peekThroughOneUseTruncation = [checkOneUse](SDValue V) {
4181 if (V->getOpcode() == ISD::TRUNCATE && checkOneUse(V)) {
4182 assert(V.getSimpleValueType() == MVT::i32 &&
4183 V.getOperand(0).getSimpleValueType() == MVT::i64 &&
4184 "Expected i64 -> i32 truncation");
4185 V = V.getOperand(0);
4186 }
4187 return V;
4188 };
4189
4190 // a) x & ((1 << nbits) + (-1))
4191 auto matchPatternA = [checkOneUse, peekThroughOneUseTruncation, &NBits,
4192 &NegateNBits](SDValue Mask) -> bool {
4193 // Match `add`. Must only have one use!
4194 if (Mask->getOpcode() != ISD::ADD || !checkOneUse(Mask))
4195 return false;
4196 // We should be adding all-ones constant (i.e. subtracting one.)
4197 if (!isAllOnesConstant(Mask->getOperand(1)))
4198 return false;
4199 // Match `1 << nbits`. Might be truncated. Must only have one use!
4200 SDValue M0 = peekThroughOneUseTruncation(Mask->getOperand(0));
4201 if (M0->getOpcode() != ISD::SHL || !checkOneUse(M0))
4202 return false;
4203 if (!isOneConstant(M0->getOperand(0)))
4204 return false;
4205 NBits = M0->getOperand(1);
4206 NegateNBits = false;
4207 return true;
4208 };
4209
4210 auto isAllOnes = [this, peekThroughOneUseTruncation, NVT](SDValue V) {
4211 V = peekThroughOneUseTruncation(V);
4212 return CurDAG->MaskedValueIsAllOnes(
4213 V, APInt::getLowBitsSet(V.getSimpleValueType().getSizeInBits(),
4214 NVT.getSizeInBits()));
4215 };
4216
4217 // b) x & ~(-1 << nbits)
4218 auto matchPatternB = [checkOneUse, isAllOnes, peekThroughOneUseTruncation,
4219 &NBits, &NegateNBits](SDValue Mask) -> bool {
4220 // Match `~()`. Must only have one use!
4221 if (Mask.getOpcode() != ISD::XOR || !checkOneUse(Mask))
4222 return false;
4223 // The -1 only has to be all-ones for the final Node's NVT.
4224 if (!isAllOnes(Mask->getOperand(1)))
4225 return false;
4226 // Match `-1 << nbits`. Might be truncated. Must only have one use!
4227 SDValue M0 = peekThroughOneUseTruncation(Mask->getOperand(0));
4228 if (M0->getOpcode() != ISD::SHL || !checkOneUse(M0))
4229 return false;
4230 // The -1 only has to be all-ones for the final Node's NVT.
4231 if (!isAllOnes(M0->getOperand(0)))
4232 return false;
4233 NBits = M0->getOperand(1);
4234 NegateNBits = false;
4235 return true;
4236 };
4237
4238 // Try to match potentially-truncated shift amount as `(bitwidth - y)`,
4239 // or leave the shift amount as-is, but then we'll have to negate it.
4240 auto canonicalizeShiftAmt = [&NBits, &NegateNBits](SDValue ShiftAmt,
4241 unsigned Bitwidth) {
4242 NBits = ShiftAmt;
4243 NegateNBits = true;
4244 // Skip over a truncate of the shift amount, if any.
4245 if (NBits.getOpcode() == ISD::TRUNCATE)
4246 NBits = NBits.getOperand(0);
4247 // Try to match the shift amount as (bitwidth - y). It should go away, too.
4248 // If it doesn't match, that's fine, we'll just negate it ourselves.
4249 if (NBits.getOpcode() != ISD::SUB)
4250 return;
4251 auto *V0 = dyn_cast<ConstantSDNode>(NBits.getOperand(0));
4252 if (!V0 || V0->getZExtValue() != Bitwidth)
4253 return;
4254 NBits = NBits.getOperand(1);
4255 NegateNBits = false;
4256 };
4257
4258 // c) x & (-1 >> z) but then we'll have to subtract z from bitwidth
4259 // or
4260 // c) x & (-1 >> (32 - y))
4261 auto matchPatternC = [checkOneUse, peekThroughOneUseTruncation, &NegateNBits,
4262 canonicalizeShiftAmt](SDValue Mask) -> bool {
4263 // The mask itself may be truncated.
4264 Mask = peekThroughOneUseTruncation(Mask);
4265 unsigned Bitwidth = Mask.getSimpleValueType().getSizeInBits();
4266 // Match `l>>`. Must only have one use!
4267 if (Mask.getOpcode() != ISD::SRL || !checkOneUse(Mask))
4268 return false;
4269 // We should be shifting truly all-ones constant.
4270 if (!isAllOnesConstant(Mask.getOperand(0)))
4271 return false;
4272 SDValue M1 = Mask.getOperand(1);
4273 // The shift amount should not be used externally.
4274 if (!checkOneUse(M1))
4275 return false;
4276 canonicalizeShiftAmt(M1, Bitwidth);
4277 // Pattern c. is non-canonical, and is expanded into pattern d. iff there
4278 // is no extra use of the mask. Clearly, there was one since we are here.
4279 // But at the same time, if we need to negate the shift amount,
4280 // then we don't want the mask to stick around, else it's unprofitable.
4281 return !NegateNBits;
4282 };
4283
4284 SDValue X;
4285
4286 // d) x << z >> z but then we'll have to subtract z from bitwidth
4287 // or
4288 // d) x << (32 - y) >> (32 - y)
4289 auto matchPatternD = [checkOneUse, checkTwoUse, canonicalizeShiftAmt,
4290 AllowExtraUsesByDefault, &NegateNBits,
4291 &X](SDNode *Node) -> bool {
4292 if (Node->getOpcode() != ISD::SRL)
4293 return false;
4294 SDValue N0 = Node->getOperand(0);
4295 if (N0->getOpcode() != ISD::SHL)
4296 return false;
4297 unsigned Bitwidth = N0.getSimpleValueType().getSizeInBits();
4298 SDValue N1 = Node->getOperand(1);
4299 SDValue N01 = N0->getOperand(1);
4300 // Both of the shifts must be by the exact same value.
4301 if (N1 != N01)
4302 return false;
4303 canonicalizeShiftAmt(N1, Bitwidth);
4304 // There should not be any external uses of the inner shift / shift amount.
4305 // Note that while we are generally okay with external uses given BMI2,
4306 // iff we need to negate the shift amount, we are not okay with extra uses.
4307 const bool AllowExtraUses = AllowExtraUsesByDefault && !NegateNBits;
4308 if (!checkOneUse(N0, AllowExtraUses) || !checkTwoUse(N1, AllowExtraUses))
4309 return false;
4310 X = N0->getOperand(0);
4311 return true;
4312 };
4313
4314 auto matchLowBitMask = [matchPatternA, matchPatternB,
4315 matchPatternC](SDValue Mask) -> bool {
4316 return matchPatternA(Mask) || matchPatternB(Mask) || matchPatternC(Mask);
4317 };
4318
4319 if (Node->getOpcode() == ISD::AND) {
4320 X = Node->getOperand(0);
4321 SDValue Mask = Node->getOperand(1);
4322
4323 if (matchLowBitMask(Mask)) {
4324 // Great.
4325 } else {
4326 std::swap(X, Mask);
4327 if (!matchLowBitMask(Mask))
4328 return false;
4329 }
4330 } else if (matchLowBitMask(SDValue(Node, 0))) {
4331 X = CurDAG->getAllOnesConstant(SDLoc(Node), NVT);
4332 } else if (!matchPatternD(Node))
4333 return false;
4334
4335 // If we need to negate the shift amount, require BMI2 BZHI support.
4336 // It's just too unprofitable for BMI1 BEXTR.
4337 if (NegateNBits && !Subtarget->hasBMI2())
4338 return false;
4339
4340 SDLoc DL(Node);
4341
4342 if (NBits.getSimpleValueType() != MVT::i8) {
4343 // Truncate the shift amount.
4344 NBits = CurDAG->getNode(ISD::TRUNCATE, DL, MVT::i8, NBits);
4345 insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
4346 }
4347
4348 // Turn (i32)(x & imm8) into (i32)x & imm32.
4349 ConstantSDNode *Imm = nullptr;
4350 if (NBits->getOpcode() == ISD::AND)
4351 if ((Imm = dyn_cast<ConstantSDNode>(NBits->getOperand(1))))
4352 NBits = NBits->getOperand(0);
4353
4354 // Insert 8-bit NBits into lowest 8 bits of 32-bit register.
4355 // All the other bits are undefined, we do not care about them.
4356 SDValue ImplDef = SDValue(
4357 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::i32), 0);
4358 insertDAGNode(*CurDAG, SDValue(Node, 0), ImplDef);
4359
4360 SDValue SRIdxVal = CurDAG->getTargetConstant(X86::sub_8bit, DL, MVT::i32);
4361 insertDAGNode(*CurDAG, SDValue(Node, 0), SRIdxVal);
4362 NBits = SDValue(CurDAG->getMachineNode(TargetOpcode::INSERT_SUBREG, DL,
4363 MVT::i32, ImplDef, NBits, SRIdxVal),
4364 0);
4365 insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
4366
4367 if (Imm) {
4368 NBits =
4369 CurDAG->getNode(ISD::AND, DL, MVT::i32, NBits,
4370 CurDAG->getConstant(Imm->getZExtValue(), DL, MVT::i32));
4371 insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
4372 }
4373
4374 // We might have matched the amount of high bits to be cleared,
4375 // but we want the amount of low bits to be kept, so negate it then.
4376 if (NegateNBits) {
4377 SDValue BitWidthC = CurDAG->getConstant(NVT.getSizeInBits(), DL, MVT::i32);
4378 insertDAGNode(*CurDAG, SDValue(Node, 0), BitWidthC);
4379
4380 NBits = CurDAG->getNode(ISD::SUB, DL, MVT::i32, BitWidthC, NBits);
4381 insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
4382 }
4383
4384 if (Subtarget->hasBMI2()) {
4385 // Great, just emit the BZHI..
4386 if (NVT != MVT::i32) {
4387 // But have to place the bit count into the wide-enough register first.
4388 NBits = CurDAG->getNode(ISD::ANY_EXTEND, DL, NVT, NBits);
4389 insertDAGNode(*CurDAG, SDValue(Node, 0), NBits);
4390 }
4391
4392 SDValue Extract = CurDAG->getNode(X86ISD::BZHI, DL, NVT, X, NBits);
4393 ReplaceNode(Node, Extract.getNode());
4394 SelectCode(Extract.getNode());
4395 return true;
4396 }
4397
4398 // Else, if we do *NOT* have BMI2, let's find out if the if the 'X' is
4399 // *logically* shifted (potentially with one-use trunc inbetween),
4400 // and the truncation was the only use of the shift,
4401 // and if so look past one-use truncation.
4402 {
4403 SDValue RealX = peekThroughOneUseTruncation(X);
4404 // FIXME: only if the shift is one-use?
4405 if (RealX != X && RealX.getOpcode() == ISD::SRL)
4406 X = RealX;
4407 }
4408
4409 MVT XVT = X.getSimpleValueType();
4410
4411 // Else, emitting BEXTR requires one more step.
4412 // The 'control' of BEXTR has the pattern of:
4413 // [15...8 bit][ 7...0 bit] location
4414 // [ bit count][ shift] name
4415 // I.e. 0b000000011'00000001 means (x >> 0b1) & 0b11
4416
4417 // Shift NBits left by 8 bits, thus producing 'control'.
4418 // This makes the low 8 bits to be zero.
4419 SDValue C8 = CurDAG->getConstant(8, DL, MVT::i8);
4420 insertDAGNode(*CurDAG, SDValue(Node, 0), C8);
4421 SDValue Control = CurDAG->getNode(ISD::SHL, DL, MVT::i32, NBits, C8);
4422 insertDAGNode(*CurDAG, SDValue(Node, 0), Control);
4423
4424 // If the 'X' is *logically* shifted, we can fold that shift into 'control'.
4425 // FIXME: only if the shift is one-use?
4426 if (X.getOpcode() == ISD::SRL) {
4427 SDValue ShiftAmt = X.getOperand(1);
4428 X = X.getOperand(0);
4429
4430 assert(ShiftAmt.getValueType() == MVT::i8 &&
4431 "Expected shift amount to be i8");
4432
4433 // Now, *zero*-extend the shift amount. The bits 8...15 *must* be zero!
4434 // We could zext to i16 in some form, but we intentionally don't do that.
4435 SDValue OrigShiftAmt = ShiftAmt;
4436 ShiftAmt = CurDAG->getNode(ISD::ZERO_EXTEND, DL, MVT::i32, ShiftAmt);
4437 insertDAGNode(*CurDAG, OrigShiftAmt, ShiftAmt);
4438
4439 // And now 'or' these low 8 bits of shift amount into the 'control'.
4440 Control = CurDAG->getNode(ISD::OR, DL, MVT::i32, Control, ShiftAmt);
4441 insertDAGNode(*CurDAG, SDValue(Node, 0), Control);
4442 }
4443
4444 // But have to place the 'control' into the wide-enough register first.
4445 if (XVT != MVT::i32) {
4446 Control = CurDAG->getNode(ISD::ANY_EXTEND, DL, XVT, Control);
4447 insertDAGNode(*CurDAG, SDValue(Node, 0), Control);
4448 }
4449
4450 // And finally, form the BEXTR itself.
4451 SDValue Extract = CurDAG->getNode(X86ISD::BEXTR, DL, XVT, X, Control);
4452
4453 // The 'X' was originally truncated. Do that now.
4454 if (XVT != NVT) {
4455 insertDAGNode(*CurDAG, SDValue(Node, 0), Extract);
4456 Extract = CurDAG->getNode(ISD::TRUNCATE, DL, NVT, Extract);
4457 }
4458
4459 ReplaceNode(Node, Extract.getNode());
4460 SelectCode(Extract.getNode());
4461
4462 return true;
4463}
4464
4465// See if this is an (X >> C1) & C2 that we can match to BEXTR/BEXTRI.
4466MachineSDNode *X86DAGToDAGISel::matchBEXTRFromAndImm(SDNode *Node) {
4467 MVT NVT = Node->getSimpleValueType(0);
4468 SDLoc dl(Node);
4469
4470 SDValue N0 = Node->getOperand(0);
4471 SDValue N1 = Node->getOperand(1);
4472
4473 // If we have TBM we can use an immediate for the control. If we have BMI
4474 // we should only do this if the BEXTR instruction is implemented well.
4475 // Otherwise moving the control into a register makes this more costly.
4476 // TODO: Maybe load folding, greater than 32-bit masks, or a guarantee of LICM
4477 // hoisting the move immediate would make it worthwhile with a less optimal
4478 // BEXTR?
4479 bool PreferBEXTR =
4480 Subtarget->hasTBM() || (Subtarget->hasBMI() && Subtarget->hasFastBEXTR());
4481 if (!PreferBEXTR && !Subtarget->hasBMI2())
4482 return nullptr;
4483
4484 // Must have a shift right.
4485 if (N0->getOpcode() != ISD::SRL && N0->getOpcode() != ISD::SRA)
4486 return nullptr;
4487
4488 // Shift can't have additional users.
4489 if (!N0->hasOneUse())
4490 return nullptr;
4491
4492 // Only supported for 32 and 64 bits.
4493 if (NVT != MVT::i32 && NVT != MVT::i64)
4494 return nullptr;
4495
4496 // Shift amount and RHS of and must be constant.
4497 auto *MaskCst = dyn_cast<ConstantSDNode>(N1);
4498 auto *ShiftCst = dyn_cast<ConstantSDNode>(N0->getOperand(1));
4499 if (!MaskCst || !ShiftCst)
4500 return nullptr;
4501
4502 // And RHS must be a mask.
4503 uint64_t Mask = MaskCst->getZExtValue();
4504 if (!isMask_64(Mask))
4505 return nullptr;
4506
4507 uint64_t Shift = ShiftCst->getZExtValue();
4508 uint64_t MaskSize = llvm::popcount(Mask);
4509
4510 // Don't interfere with something that can be handled by extracting AH.
4511 // TODO: If we are able to fold a load, BEXTR might still be better than AH.
4512 if (Shift == 8 && MaskSize == 8)
4513 return nullptr;
4514
4515 // Make sure we are only using bits that were in the original value, not
4516 // shifted in.
4517 if (Shift + MaskSize > NVT.getSizeInBits())
4518 return nullptr;
4519
4520 // BZHI, if available, is always fast, unlike BEXTR. But even if we decide
4521 // that we can't use BEXTR, it is only worthwhile using BZHI if the mask
4522 // does not fit into 32 bits. Load folding is not a sufficient reason.
4523 if (!PreferBEXTR && MaskSize <= 32)
4524 return nullptr;
4525
4526 SDValue Control;
4527 unsigned ROpc, MOpc;
4528
4529#define GET_EGPR_IF_ENABLED(OPC) (Subtarget->hasEGPR() ? OPC##_EVEX : OPC)
4530 if (!PreferBEXTR) {
4531 assert(Subtarget->hasBMI2() && "We must have BMI2's BZHI then.");
4532 // If we can't make use of BEXTR then we can't fuse shift+mask stages.
4533 // Let's perform the mask first, and apply shift later. Note that we need to
4534 // widen the mask to account for the fact that we'll apply shift afterwards!
4535 Control = CurDAG->getTargetConstant(Shift + MaskSize, dl, NVT);
4536 ROpc = NVT == MVT::i64 ? GET_EGPR_IF_ENABLED(X86::BZHI64rr)
4537 : GET_EGPR_IF_ENABLED(X86::BZHI32rr);
4538 MOpc = NVT == MVT::i64 ? GET_EGPR_IF_ENABLED(X86::BZHI64rm)
4539 : GET_EGPR_IF_ENABLED(X86::BZHI32rm);
4540 unsigned NewOpc = NVT == MVT::i64 ? X86::MOV32ri64 : X86::MOV32ri;
4541 Control = SDValue(CurDAG->getMachineNode(NewOpc, dl, NVT, Control), 0);
4542 } else {
4543 // The 'control' of BEXTR has the pattern of:
4544 // [15...8 bit][ 7...0 bit] location
4545 // [ bit count][ shift] name
4546 // I.e. 0b000000011'00000001 means (x >> 0b1) & 0b11
4547 Control = CurDAG->getTargetConstant(Shift | (MaskSize << 8), dl, NVT);
4548 if (Subtarget->hasTBM()) {
4549 ROpc = NVT == MVT::i64 ? X86::BEXTRI64ri : X86::BEXTRI32ri;
4550 MOpc = NVT == MVT::i64 ? X86::BEXTRI64mi : X86::BEXTRI32mi;
4551 } else {
4552 assert(Subtarget->hasBMI() && "We must have BMI1's BEXTR then.");
4553 // BMI requires the immediate to placed in a register.
4554 ROpc = NVT == MVT::i64 ? GET_EGPR_IF_ENABLED(X86::BEXTR64rr)
4555 : GET_EGPR_IF_ENABLED(X86::BEXTR32rr);
4556 MOpc = NVT == MVT::i64 ? GET_EGPR_IF_ENABLED(X86::BEXTR64rm)
4557 : GET_EGPR_IF_ENABLED(X86::BEXTR32rm);
4558 unsigned NewOpc = NVT == MVT::i64 ? X86::MOV32ri64 : X86::MOV32ri;
4559 Control = SDValue(CurDAG->getMachineNode(NewOpc, dl, NVT, Control), 0);
4560 }
4561 }
4562
4563 MachineSDNode *NewNode;
4564 SDValue Input = N0->getOperand(0);
4565 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
4566 if (tryFoldLoad(Node, N0.getNode(), Input, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
4567 SDValue Ops[] = {
4568 Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, Control, Input.getOperand(0)};
4569 SDVTList VTs = CurDAG->getVTList(NVT, MVT::i32, MVT::Other);
4570 NewNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
4571 // Update the chain.
4572 ReplaceUses(Input.getValue(1), SDValue(NewNode, 2));
4573 // Record the mem-refs
4574 CurDAG->setNodeMemRefs(NewNode, {cast<LoadSDNode>(Input)->getMemOperand()});
4575 } else {
4576 NewNode = CurDAG->getMachineNode(ROpc, dl, NVT, MVT::i32, Input, Control);
4577 }
4578
4579 if (!PreferBEXTR) {
4580 // We still need to apply the shift.
4581 SDValue ShAmt = CurDAG->getTargetConstant(Shift, dl, NVT);
4582 unsigned NewOpc = NVT == MVT::i64 ? GET_ND_IF_ENABLED(X86::SHR64ri)
4583 : GET_ND_IF_ENABLED(X86::SHR32ri);
4584 NewNode =
4585 CurDAG->getMachineNode(NewOpc, dl, NVT, SDValue(NewNode, 0), ShAmt);
4586 }
4587
4588 return NewNode;
4589}
4590
4591// Emit a PCMISTR(I/M) instruction.
4592MachineSDNode *X86DAGToDAGISel::emitPCMPISTR(unsigned ROpc, unsigned MOpc,
4593 bool MayFoldLoad, const SDLoc &dl,
4594 MVT VT, SDNode *Node) {
4595 SDValue N0 = Node->getOperand(0);
4596 SDValue N1 = Node->getOperand(1);
4597 SDValue Imm = Node->getOperand(2);
4598 auto *Val = cast<ConstantSDNode>(Imm)->getConstantIntValue();
4599 Imm = CurDAG->getTargetConstant(*Val, SDLoc(Node), Imm.getValueType());
4600
4601 // Try to fold a load. No need to check alignment.
4602 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
4603 if (MayFoldLoad && tryFoldLoad(Node, N1, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
4604 SDValue Ops[] = { N0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, Imm,
4605 N1.getOperand(0) };
4606 SDVTList VTs = CurDAG->getVTList(VT, MVT::i32, MVT::Other);
4607 MachineSDNode *CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
4608 // Update the chain.
4609 ReplaceUses(N1.getValue(1), SDValue(CNode, 2));
4610 // Record the mem-refs
4611 CurDAG->setNodeMemRefs(CNode, {cast<LoadSDNode>(N1)->getMemOperand()});
4612 return CNode;
4613 }
4614
4615 SDValue Ops[] = { N0, N1, Imm };
4616 SDVTList VTs = CurDAG->getVTList(VT, MVT::i32);
4617 MachineSDNode *CNode = CurDAG->getMachineNode(ROpc, dl, VTs, Ops);
4618 return CNode;
4619}
4620
4621// Emit a PCMESTR(I/M) instruction. Also return the Glue result in case we need
4622// to emit a second instruction after this one. This is needed since we have two
4623// copyToReg nodes glued before this and we need to continue that glue through.
4624MachineSDNode *X86DAGToDAGISel::emitPCMPESTR(unsigned ROpc, unsigned MOpc,
4625 bool MayFoldLoad, const SDLoc &dl,
4626 MVT VT, SDNode *Node,
4627 SDValue &InGlue) {
4628 SDValue N0 = Node->getOperand(0);
4629 SDValue N2 = Node->getOperand(2);
4630 SDValue Imm = Node->getOperand(4);
4631 auto *Val = cast<ConstantSDNode>(Imm)->getConstantIntValue();
4632 Imm = CurDAG->getTargetConstant(*Val, SDLoc(Node), Imm.getValueType());
4633
4634 // Try to fold a load. No need to check alignment.
4635 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
4636 if (MayFoldLoad && tryFoldLoad(Node, N2, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
4637 SDValue Ops[] = { N0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, Imm,
4638 N2.getOperand(0), InGlue };
4639 SDVTList VTs = CurDAG->getVTList(VT, MVT::i32, MVT::Other, MVT::Glue);
4640 MachineSDNode *CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
4641 InGlue = SDValue(CNode, 3);
4642 // Update the chain.
4643 ReplaceUses(N2.getValue(1), SDValue(CNode, 2));
4644 // Record the mem-refs
4645 CurDAG->setNodeMemRefs(CNode, {cast<LoadSDNode>(N2)->getMemOperand()});
4646 return CNode;
4647 }
4648
4649 SDValue Ops[] = { N0, N2, Imm, InGlue };
4650 SDVTList VTs = CurDAG->getVTList(VT, MVT::i32, MVT::Glue);
4651 MachineSDNode *CNode = CurDAG->getMachineNode(ROpc, dl, VTs, Ops);
4652 InGlue = SDValue(CNode, 2);
4653 return CNode;
4654}
4655
4656bool X86DAGToDAGISel::tryShiftAmountMod(SDNode *N) {
4657 EVT VT = N->getValueType(0);
4658
4659 // Only handle scalar shifts.
4660 if (VT.isVector())
4661 return false;
4662
4663 // Narrower shifts only mask to 5 bits in hardware.
4664 unsigned Size = VT == MVT::i64 ? 64 : 32;
4665
4666 SDValue OrigShiftAmt = N->getOperand(1);
4667 SDValue ShiftAmt = OrigShiftAmt;
4668 SDLoc DL(N);
4669
4670 // Skip over a truncate of the shift amount.
4671 if (ShiftAmt->getOpcode() == ISD::TRUNCATE)
4672 ShiftAmt = ShiftAmt->getOperand(0);
4673
4674 // This function is called after X86DAGToDAGISel::matchBitExtract(),
4675 // so we are not afraid that we might mess up BZHI/BEXTR pattern.
4676
4677 SDValue NewShiftAmt;
4678 if (ShiftAmt->getOpcode() == ISD::ADD || ShiftAmt->getOpcode() == ISD::SUB ||
4679 ShiftAmt->getOpcode() == ISD::XOR) {
4680 SDValue Add0 = ShiftAmt->getOperand(0);
4681 SDValue Add1 = ShiftAmt->getOperand(1);
4682 auto *Add0C = dyn_cast<ConstantSDNode>(Add0);
4683 auto *Add1C = dyn_cast<ConstantSDNode>(Add1);
4684 // If we are shifting by X+/-/^N where N == 0 mod Size, then just shift by X
4685 // to avoid the ADD/SUB/XOR.
4686 if (Add1C && Add1C->getAPIntValue().urem(Size) == 0) {
4687 NewShiftAmt = Add0;
4688
4689 } else if (ShiftAmt->getOpcode() != ISD::ADD && ShiftAmt.hasOneUse() &&
4690 ((Add0C && Add0C->getAPIntValue().urem(Size) == Size - 1) ||
4691 (Add1C && Add1C->getAPIntValue().urem(Size) == Size - 1))) {
4692 // If we are doing a NOT on just the lower bits with (Size*N-1) -/^ X
4693 // we can replace it with a NOT. In the XOR case it may save some code
4694 // size, in the SUB case it also may save a move.
4695 assert(Add0C == nullptr || Add1C == nullptr);
4696
4697 // We can only do N-X, not X-N
4698 if (ShiftAmt->getOpcode() == ISD::SUB && Add0C == nullptr)
4699 return false;
4700
4701 EVT OpVT = ShiftAmt.getValueType();
4702
4703 SDValue AllOnes = CurDAG->getAllOnesConstant(DL, OpVT);
4704 NewShiftAmt = CurDAG->getNode(ISD::XOR, DL, OpVT,
4705 Add0C == nullptr ? Add0 : Add1, AllOnes);
4706 insertDAGNode(*CurDAG, OrigShiftAmt, AllOnes);
4707 insertDAGNode(*CurDAG, OrigShiftAmt, NewShiftAmt);
4708 // If we are shifting by N-X where N == 0 mod Size, then just shift by
4709 // -X to generate a NEG instead of a SUB of a constant.
4710 } else if (ShiftAmt->getOpcode() == ISD::SUB && Add0C &&
4711 Add0C->getZExtValue() != 0) {
4712 EVT SubVT = ShiftAmt.getValueType();
4713 SDValue X;
4714 if (Add0C->getZExtValue() % Size == 0)
4715 X = Add1;
4716 else if (ShiftAmt.hasOneUse() && Size == 64 &&
4717 Add0C->getZExtValue() % 32 == 0) {
4718 // We have a 64-bit shift by (n*32-x), turn it into -(x+n*32).
4719 // This is mainly beneficial if we already compute (x+n*32).
4720 if (Add1.getOpcode() == ISD::TRUNCATE) {
4721 Add1 = Add1.getOperand(0);
4722 SubVT = Add1.getValueType();
4723 }
4724 if (Add0.getValueType() != SubVT) {
4725 Add0 = CurDAG->getZExtOrTrunc(Add0, DL, SubVT);
4726 insertDAGNode(*CurDAG, OrigShiftAmt, Add0);
4727 }
4728
4729 X = CurDAG->getNode(ISD::ADD, DL, SubVT, Add1, Add0);
4730 insertDAGNode(*CurDAG, OrigShiftAmt, X);
4731 } else
4732 return false;
4733 // Insert a negate op.
4734 // TODO: This isn't guaranteed to replace the sub if there is a logic cone
4735 // that uses it that's not a shift.
4736 SDValue Zero = CurDAG->getConstant(0, DL, SubVT);
4737 SDValue Neg = CurDAG->getNode(ISD::SUB, DL, SubVT, Zero, X);
4738 NewShiftAmt = Neg;
4739
4740 // Insert these operands into a valid topological order so they can
4741 // get selected independently.
4742 insertDAGNode(*CurDAG, OrigShiftAmt, Zero);
4743 insertDAGNode(*CurDAG, OrigShiftAmt, Neg);
4744 } else
4745 return false;
4746 } else
4747 return false;
4748
4749 if (NewShiftAmt.getValueType() != MVT::i8) {
4750 // Need to truncate the shift amount.
4751 NewShiftAmt = CurDAG->getNode(ISD::TRUNCATE, DL, MVT::i8, NewShiftAmt);
4752 // Add to a correct topological ordering.
4753 insertDAGNode(*CurDAG, OrigShiftAmt, NewShiftAmt);
4754 }
4755
4756 // Insert a new mask to keep the shift amount legal. This should be removed
4757 // by isel patterns.
4758 NewShiftAmt = CurDAG->getNode(ISD::AND, DL, MVT::i8, NewShiftAmt,
4759 CurDAG->getConstant(Size - 1, DL, MVT::i8));
4760 // Place in a correct topological ordering.
4761 insertDAGNode(*CurDAG, OrigShiftAmt, NewShiftAmt);
4762
4763 SDNode *UpdatedNode = CurDAG->UpdateNodeOperands(N, N->getOperand(0),
4764 NewShiftAmt);
4765 if (UpdatedNode != N) {
4766 // If we found an existing node, we should replace ourselves with that node
4767 // and wait for it to be selected after its other users.
4768 ReplaceNode(N, UpdatedNode);
4769 return true;
4770 }
4771
4772 // If the original shift amount is now dead, delete it so that we don't run
4773 // it through isel.
4774 if (OrigShiftAmt.getNode()->use_empty())
4775 CurDAG->RemoveDeadNode(OrigShiftAmt.getNode());
4776
4777 // Now that we've optimized the shift amount, defer to normal isel to get
4778 // load folding and legacy vs BMI2 selection without repeating it here.
4779 SelectCode(N);
4780 return true;
4781}
4782
4783bool X86DAGToDAGISel::tryShrinkShlLogicImm(SDNode *N) {
4784 MVT NVT = N->getSimpleValueType(0);
4785 unsigned Opcode = N->getOpcode();
4786 SDLoc dl(N);
4787
4788 // For operations of the form (x << C1) op C2, check if we can use a smaller
4789 // encoding for C2 by transforming it into (x op (C2>>C1)) << C1.
4790 SDValue Shift = N->getOperand(0);
4791 SDValue N1 = N->getOperand(1);
4792
4793 auto *Cst = dyn_cast<ConstantSDNode>(N1);
4794 if (!Cst)
4795 return false;
4796
4797 int64_t Val = Cst->getSExtValue();
4798
4799 // If we have an any_extend feeding the AND, look through it to see if there
4800 // is a shift behind it. But only if the AND doesn't use the extended bits.
4801 // FIXME: Generalize this to other ANY_EXTEND than i32 to i64?
4802 bool FoundAnyExtend = false;
4803 if (Shift.getOpcode() == ISD::ANY_EXTEND && Shift.hasOneUse() &&
4804 Shift.getOperand(0).getSimpleValueType() == MVT::i32 &&
4805 isUInt<32>(Val)) {
4806 FoundAnyExtend = true;
4807 Shift = Shift.getOperand(0);
4808 }
4809
4810 if (Shift.getOpcode() != ISD::SHL || !Shift.hasOneUse())
4811 return false;
4812
4813 // i8 is unshrinkable, i16 should be promoted to i32.
4814 if (NVT != MVT::i32 && NVT != MVT::i64)
4815 return false;
4816
4817 auto *ShlCst = dyn_cast<ConstantSDNode>(Shift.getOperand(1));
4818 if (!ShlCst)
4819 return false;
4820
4821 uint64_t ShAmt = ShlCst->getZExtValue();
4822
4823 // Make sure that we don't change the operation by removing bits.
4824 // This only matters for OR and XOR, AND is unaffected.
4825 uint64_t RemovedBitsMask = (1ULL << ShAmt) - 1;
4826 if (Opcode != ISD::AND && (Val & RemovedBitsMask) != 0)
4827 return false;
4828
4829 // Check the minimum bitwidth for the new constant.
4830 // TODO: Using 16 and 8 bit operations is also possible for or32 & xor32.
4831 auto CanShrinkImmediate = [&](int64_t &ShiftedVal) {
4832 if (Opcode == ISD::AND) {
4833 // AND32ri is the same as AND64ri32 with zext imm.
4834 // Try this before sign extended immediates below.
4835 ShiftedVal = (uint64_t)Val >> ShAmt;
4836 if (NVT == MVT::i64 && !isUInt<32>(Val) && isUInt<32>(ShiftedVal))
4837 return true;
4838 // Also swap order when the AND can become MOVZX.
4839 if (ShiftedVal == UINT8_MAX || ShiftedVal == UINT16_MAX)
4840 return true;
4841 }
4842 ShiftedVal = Val >> ShAmt;
4843 if ((!isInt<8>(Val) && isInt<8>(ShiftedVal)) ||
4844 (!isInt<32>(Val) && isInt<32>(ShiftedVal)))
4845 return true;
4846 if (Opcode != ISD::AND) {
4847 // MOV32ri+OR64r/XOR64r is cheaper than MOV64ri64+OR64rr/XOR64rr
4848 ShiftedVal = (uint64_t)Val >> ShAmt;
4849 if (NVT == MVT::i64 && !isUInt<32>(Val) && isUInt<32>(ShiftedVal))
4850 return true;
4851 }
4852 return false;
4853 };
4854
4855 int64_t ShiftedVal;
4856 if (!CanShrinkImmediate(ShiftedVal))
4857 return false;
4858
4859 // Ok, we can reorder to get a smaller immediate.
4860
4861 // But, its possible the original immediate allowed an AND to become MOVZX.
4862 // Doing this late due to avoid the MakedValueIsZero call as late as
4863 // possible.
4864 if (Opcode == ISD::AND) {
4865 // Find the smallest zext this could possibly be.
4866 unsigned ZExtWidth = Cst->getAPIntValue().getActiveBits();
4867 ZExtWidth = llvm::bit_ceil(std::max(ZExtWidth, 8U));
4868
4869 // Figure out which bits need to be zero to achieve that mask.
4870 APInt NeededMask = APInt::getLowBitsSet(NVT.getSizeInBits(),
4871 ZExtWidth);
4872 NeededMask &= ~Cst->getAPIntValue();
4873
4874 if (CurDAG->MaskedValueIsZero(N->getOperand(0), NeededMask))
4875 return false;
4876 }
4877
4878 SDValue X = Shift.getOperand(0);
4879 if (FoundAnyExtend) {
4880 SDValue NewX = CurDAG->getNode(ISD::ANY_EXTEND, dl, NVT, X);
4881 insertDAGNode(*CurDAG, SDValue(N, 0), NewX);
4882 X = NewX;
4883 }
4884
4885 SDValue NewCst = CurDAG->getSignedConstant(ShiftedVal, dl, NVT);
4886 insertDAGNode(*CurDAG, SDValue(N, 0), NewCst);
4887 SDValue NewBinOp = CurDAG->getNode(Opcode, dl, NVT, X, NewCst);
4888 insertDAGNode(*CurDAG, SDValue(N, 0), NewBinOp);
4889 SDValue NewSHL = CurDAG->getNode(ISD::SHL, dl, NVT, NewBinOp,
4890 Shift.getOperand(1));
4891 ReplaceNode(N, NewSHL.getNode());
4892 SelectCode(NewSHL.getNode());
4893 return true;
4894}
4895
4896bool X86DAGToDAGISel::matchVPTERNLOG(SDNode *Root, SDNode *ParentA,
4897 SDNode *ParentB, SDNode *ParentC,
4898 SDValue A, SDValue B, SDValue C,
4899 uint8_t Imm) {
4900 assert(A.isOperandOf(ParentA) && B.isOperandOf(ParentB) &&
4901 C.isOperandOf(ParentC) && "Incorrect parent node");
4902
4903 auto tryFoldLoadOrBCast =
4904 [this](SDNode *Root, SDNode *P, SDValue &L, SDValue &Base, SDValue &Scale,
4905 SDValue &Index, SDValue &Disp, SDValue &Segment) {
4906 if (tryFoldLoad(Root, P, L, Base, Scale, Index, Disp, Segment))
4907 return true;
4908
4909 // Not a load, check for broadcast which may be behind a bitcast.
4910 if (L.getOpcode() == ISD::BITCAST && L.hasOneUse()) {
4911 P = L.getNode();
4912 L = L.getOperand(0);
4913 }
4914
4915 if (L.getOpcode() != X86ISD::VBROADCAST_LOAD)
4916 return false;
4917
4918 // Only 32 and 64 bit broadcasts are supported.
4919 auto *MemIntr = cast<MemIntrinsicSDNode>(L);
4920 unsigned Size = MemIntr->getMemoryVT().getSizeInBits();
4921 if (Size != 32 && Size != 64)
4922 return false;
4923
4924 return tryFoldBroadcast(Root, P, L, Base, Scale, Index, Disp, Segment);
4925 };
4926
4927 bool FoldedLoad = false;
4928 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
4929 if (tryFoldLoadOrBCast(Root, ParentC, C, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
4930 FoldedLoad = true;
4931 } else if (tryFoldLoadOrBCast(Root, ParentA, A, Tmp0, Tmp1, Tmp2, Tmp3,
4932 Tmp4)) {
4933 FoldedLoad = true;
4934 std::swap(A, C);
4935 // Swap bits 1/4 and 3/6.
4936 uint8_t OldImm = Imm;
4937 Imm = OldImm & 0xa5;
4938 if (OldImm & 0x02) Imm |= 0x10;
4939 if (OldImm & 0x10) Imm |= 0x02;
4940 if (OldImm & 0x08) Imm |= 0x40;
4941 if (OldImm & 0x40) Imm |= 0x08;
4942 } else if (tryFoldLoadOrBCast(Root, ParentB, B, Tmp0, Tmp1, Tmp2, Tmp3,
4943 Tmp4)) {
4944 FoldedLoad = true;
4945 std::swap(B, C);
4946 // Swap bits 1/2 and 5/6.
4947 uint8_t OldImm = Imm;
4948 Imm = OldImm & 0x99;
4949 if (OldImm & 0x02) Imm |= 0x04;
4950 if (OldImm & 0x04) Imm |= 0x02;
4951 if (OldImm & 0x20) Imm |= 0x40;
4952 if (OldImm & 0x40) Imm |= 0x20;
4953 }
4954
4955 SDLoc DL(Root);
4956
4957 SDValue TImm = CurDAG->getTargetConstant(Imm, DL, MVT::i8);
4958
4959 MVT NVT = Root->getSimpleValueType(0);
4960
4961 MachineSDNode *MNode;
4962 if (FoldedLoad) {
4963 SDVTList VTs = CurDAG->getVTList(NVT, MVT::Other);
4964
4965 unsigned Opc;
4966 if (C.getOpcode() == X86ISD::VBROADCAST_LOAD) {
4967 auto *MemIntr = cast<MemIntrinsicSDNode>(C);
4968 unsigned EltSize = MemIntr->getMemoryVT().getSizeInBits();
4969 assert((EltSize == 32 || EltSize == 64) && "Unexpected broadcast size!");
4970
4971 bool UseD = EltSize == 32;
4972 if (NVT.is128BitVector())
4973 Opc = UseD ? X86::VPTERNLOGDZ128rmbi : X86::VPTERNLOGQZ128rmbi;
4974 else if (NVT.is256BitVector())
4975 Opc = UseD ? X86::VPTERNLOGDZ256rmbi : X86::VPTERNLOGQZ256rmbi;
4976 else if (NVT.is512BitVector())
4977 Opc = UseD ? X86::VPTERNLOGDZrmbi : X86::VPTERNLOGQZrmbi;
4978 else
4979 llvm_unreachable("Unexpected vector size!");
4980 } else {
4981 bool UseD = NVT.getVectorElementType() == MVT::i32;
4982 if (NVT.is128BitVector())
4983 Opc = UseD ? X86::VPTERNLOGDZ128rmi : X86::VPTERNLOGQZ128rmi;
4984 else if (NVT.is256BitVector())
4985 Opc = UseD ? X86::VPTERNLOGDZ256rmi : X86::VPTERNLOGQZ256rmi;
4986 else if (NVT.is512BitVector())
4987 Opc = UseD ? X86::VPTERNLOGDZrmi : X86::VPTERNLOGQZrmi;
4988 else
4989 llvm_unreachable("Unexpected vector size!");
4990 }
4991
4992 SDValue Ops[] = {A, B, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, TImm, C.getOperand(0)};
4993 MNode = CurDAG->getMachineNode(Opc, DL, VTs, Ops);
4994
4995 // Update the chain.
4996 ReplaceUses(C.getValue(1), SDValue(MNode, 1));
4997 // Record the mem-refs
4998 CurDAG->setNodeMemRefs(MNode, {cast<MemSDNode>(C)->getMemOperand()});
4999 } else {
5000 bool UseD = NVT.getVectorElementType() == MVT::i32;
5001 unsigned Opc;
5002 if (NVT.is128BitVector())
5003 Opc = UseD ? X86::VPTERNLOGDZ128rri : X86::VPTERNLOGQZ128rri;
5004 else if (NVT.is256BitVector())
5005 Opc = UseD ? X86::VPTERNLOGDZ256rri : X86::VPTERNLOGQZ256rri;
5006 else if (NVT.is512BitVector())
5007 Opc = UseD ? X86::VPTERNLOGDZrri : X86::VPTERNLOGQZrri;
5008 else
5009 llvm_unreachable("Unexpected vector size!");
5010
5011 MNode = CurDAG->getMachineNode(Opc, DL, NVT, {A, B, C, TImm});
5012 }
5013
5014 ReplaceUses(SDValue(Root, 0), SDValue(MNode, 0));
5015 CurDAG->RemoveDeadNode(Root);
5016 return true;
5017}
5018
5019// Try to match two logic ops to a VPTERNLOG.
5020// FIXME: Handle more complex patterns that use an operand more than once?
5021bool X86DAGToDAGISel::tryVPTERNLOG(SDNode *N) {
5022 MVT NVT = N->getSimpleValueType(0);
5023
5024 // Make sure we support VPTERNLOG.
5025 if (!NVT.isVector() || !Subtarget->hasAVX512() ||
5026 NVT.getVectorElementType() == MVT::i1)
5027 return false;
5028
5029 // We need VLX for 128/256-bit.
5030 if (!(Subtarget->hasVLX() || NVT.is512BitVector()))
5031 return false;
5032
5033 auto getFoldableLogicOp = [](SDValue Op) {
5034 // Peek through single use bitcast.
5035 if (Op.getOpcode() == ISD::BITCAST && Op.hasOneUse())
5036 Op = Op.getOperand(0);
5037
5038 if (!Op.hasOneUse())
5039 return SDValue();
5040
5041 unsigned Opc = Op.getOpcode();
5042 if (Opc == ISD::AND || Opc == ISD::OR || Opc == ISD::XOR ||
5043 Opc == X86ISD::ANDNP)
5044 return Op;
5045
5046 return SDValue();
5047 };
5048
5049 SDValue N0, N1, A, FoldableOp;
5050
5051 // Identify and (optionally) peel an outer NOT that wraps a pure logic tree
5052 auto tryPeelOuterNotWrappingLogic = [&](SDNode *Op) {
5053 if (Op->getOpcode() == ISD::XOR && Op->hasOneUse() &&
5054 ISD::isBuildVectorAllOnes(Op->getOperand(1).getNode())) {
5055 SDValue InnerOp = getFoldableLogicOp(Op->getOperand(0));
5056
5057 if (!InnerOp)
5058 return SDValue();
5059
5060 N0 = InnerOp.getOperand(0);
5061 N1 = InnerOp.getOperand(1);
5062 if ((FoldableOp = getFoldableLogicOp(N1))) {
5063 A = N0;
5064 return InnerOp;
5065 }
5066 if ((FoldableOp = getFoldableLogicOp(N0))) {
5067 A = N1;
5068 return InnerOp;
5069 }
5070 }
5071 return SDValue();
5072 };
5073
5074 bool PeeledOuterNot = false;
5075 SDNode *OriN = N;
5076 if (SDValue InnerOp = tryPeelOuterNotWrappingLogic(N)) {
5077 PeeledOuterNot = true;
5078 N = InnerOp.getNode();
5079 } else {
5080 N0 = N->getOperand(0);
5081 N1 = N->getOperand(1);
5082
5083 if ((FoldableOp = getFoldableLogicOp(N1)))
5084 A = N0;
5085 else if ((FoldableOp = getFoldableLogicOp(N0)))
5086 A = N1;
5087 else
5088 return false;
5089 }
5090
5091 SDValue B = FoldableOp.getOperand(0);
5092 SDValue C = FoldableOp.getOperand(1);
5093 SDNode *ParentA = N;
5094 SDNode *ParentB = FoldableOp.getNode();
5095 SDNode *ParentC = FoldableOp.getNode();
5096
5097 // We can build the appropriate control immediate by performing the logic
5098 // operation we're matching using these constants for A, B, and C.
5099 uint8_t TernlogMagicA = 0xf0;
5100 uint8_t TernlogMagicB = 0xcc;
5101 uint8_t TernlogMagicC = 0xaa;
5102
5103 // Some of the inputs may be inverted, peek through them and invert the
5104 // magic values accordingly.
5105 // TODO: There may be a bitcast before the xor that we should peek through.
5106 auto PeekThroughNot = [](SDValue &Op, SDNode *&Parent, uint8_t &Magic) {
5107 if (Op.getOpcode() == ISD::XOR && Op.hasOneUse() &&
5108 ISD::isBuildVectorAllOnes(Op.getOperand(1).getNode())) {
5109 Magic = ~Magic;
5110 Parent = Op.getNode();
5111 Op = Op.getOperand(0);
5112 }
5113 };
5114
5115 PeekThroughNot(A, ParentA, TernlogMagicA);
5116 PeekThroughNot(B, ParentB, TernlogMagicB);
5117 PeekThroughNot(C, ParentC, TernlogMagicC);
5118
5119 uint8_t Imm;
5120 switch (FoldableOp.getOpcode()) {
5121 default: llvm_unreachable("Unexpected opcode!");
5122 case ISD::AND: Imm = TernlogMagicB & TernlogMagicC; break;
5123 case ISD::OR: Imm = TernlogMagicB | TernlogMagicC; break;
5124 case ISD::XOR: Imm = TernlogMagicB ^ TernlogMagicC; break;
5125 case X86ISD::ANDNP: Imm = ~(TernlogMagicB) & TernlogMagicC; break;
5126 }
5127
5128 switch (N->getOpcode()) {
5129 default: llvm_unreachable("Unexpected opcode!");
5130 case X86ISD::ANDNP:
5131 if (A == N0)
5132 Imm &= ~TernlogMagicA;
5133 else
5134 Imm = ~(Imm) & TernlogMagicA;
5135 break;
5136 case ISD::AND: Imm &= TernlogMagicA; break;
5137 case ISD::OR: Imm |= TernlogMagicA; break;
5138 case ISD::XOR: Imm ^= TernlogMagicA; break;
5139 }
5140
5141 if (PeeledOuterNot)
5142 Imm = ~Imm;
5143
5144 return matchVPTERNLOG(OriN, ParentA, ParentB, ParentC, A, B, C, Imm);
5145}
5146
5147/// If the high bits of an 'and' operand are known zero, try setting the
5148/// high bits of an 'and' constant operand to produce a smaller encoding by
5149/// creating a small, sign-extended negative immediate rather than a large
5150/// positive one. This reverses a transform in SimplifyDemandedBits that
5151/// shrinks mask constants by clearing bits. There is also a possibility that
5152/// the 'and' mask can be made -1, so the 'and' itself is unnecessary. In that
5153/// case, just replace the 'and'. Return 'true' if the node is replaced.
5154bool X86DAGToDAGISel::shrinkAndImmediate(SDNode *And) {
5155 // i8 is unshrinkable, i16 should be promoted to i32, and vector ops don't
5156 // have immediate operands.
5157 MVT VT = And->getSimpleValueType(0);
5158 if (VT != MVT::i32 && VT != MVT::i64)
5159 return false;
5160
5161 auto *And1C = dyn_cast<ConstantSDNode>(And->getOperand(1));
5162 if (!And1C)
5163 return false;
5164
5165 // Bail out if the mask constant is already negative. It's can't shrink more.
5166 // If the upper 32 bits of a 64 bit mask are all zeros, we have special isel
5167 // patterns to use a 32-bit and instead of a 64-bit and by relying on the
5168 // implicit zeroing of 32 bit ops. So we should check if the lower 32 bits
5169 // are negative too.
5170 APInt MaskVal = And1C->getAPIntValue();
5171 unsigned MaskLZ = MaskVal.countl_zero();
5172 if (!MaskLZ || (VT == MVT::i64 && MaskLZ == 32))
5173 return false;
5174
5175 // Don't extend into the upper 32 bits of a 64 bit mask.
5176 if (VT == MVT::i64 && MaskLZ >= 32) {
5177 MaskLZ -= 32;
5178 MaskVal = MaskVal.trunc(32);
5179 }
5180
5181 SDValue And0 = And->getOperand(0);
5182 APInt HighZeros = APInt::getHighBitsSet(MaskVal.getBitWidth(), MaskLZ);
5183 APInt NegMaskVal = MaskVal | HighZeros;
5184
5185 // If a negative constant would not allow a smaller encoding, there's no need
5186 // to continue. Only change the constant when we know it's a win.
5187 unsigned MinWidth = NegMaskVal.getSignificantBits();
5188 if (MinWidth > 32 || (MinWidth > 8 && MaskVal.getSignificantBits() <= 32))
5189 return false;
5190
5191 // Extend masks if we truncated above.
5192 if (VT == MVT::i64 && MaskVal.getBitWidth() < 64) {
5193 NegMaskVal = NegMaskVal.zext(64);
5194 HighZeros = HighZeros.zext(64);
5195 }
5196
5197 // The variable operand must be all zeros in the top bits to allow using the
5198 // new, negative constant as the mask.
5199 // TODO: Handle constant folding?
5200 KnownBits Known0 = CurDAG->computeKnownBits(And0);
5201 if (Known0.isConstant() || !HighZeros.isSubsetOf(Known0.Zero))
5202 return false;
5203
5204 // Check if the mask is -1. In that case, this is an unnecessary instruction
5205 // that escaped earlier analysis.
5206 if (NegMaskVal.isAllOnes()) {
5207 // The already-selected users of a 32-bit 'and' may rely on it zeroing the
5208 // upper 32 bits (def32), which a truncate operand doesn't guarantee.
5209 if (VT == MVT::i32 && !isDef32(And0.getNode()))
5210 return false;
5211 ReplaceNode(And, And0.getNode());
5212 return true;
5213 }
5214
5215 // A negative mask allows a smaller encoding. Create a new 'and' node.
5216 SDValue NewMask = CurDAG->getConstant(NegMaskVal, SDLoc(And), VT);
5217 insertDAGNode(*CurDAG, SDValue(And, 0), NewMask);
5218 SDValue NewAnd = CurDAG->getNode(ISD::AND, SDLoc(And), VT, And0, NewMask);
5219 ReplaceNode(And, NewAnd.getNode());
5220 SelectCode(NewAnd.getNode());
5221 return true;
5222}
5223
5224static unsigned getVPTESTMOpc(MVT TestVT, bool IsTestN, bool FoldedLoad,
5225 bool FoldedBCast, bool Masked) {
5226#define VPTESTM_CASE(VT, SUFFIX) \
5227case MVT::VT: \
5228 if (Masked) \
5229 return IsTestN ? X86::VPTESTNM##SUFFIX##k: X86::VPTESTM##SUFFIX##k; \
5230 return IsTestN ? X86::VPTESTNM##SUFFIX : X86::VPTESTM##SUFFIX;
5231
5232
5233#define VPTESTM_BROADCAST_CASES(SUFFIX) \
5234default: llvm_unreachable("Unexpected VT!"); \
5235VPTESTM_CASE(v4i32, DZ128##SUFFIX) \
5236VPTESTM_CASE(v2i64, QZ128##SUFFIX) \
5237VPTESTM_CASE(v8i32, DZ256##SUFFIX) \
5238VPTESTM_CASE(v4i64, QZ256##SUFFIX) \
5239VPTESTM_CASE(v16i32, DZ##SUFFIX) \
5240VPTESTM_CASE(v8i64, QZ##SUFFIX)
5241
5242#define VPTESTM_FULL_CASES(SUFFIX) \
5243VPTESTM_BROADCAST_CASES(SUFFIX) \
5244VPTESTM_CASE(v16i8, BZ128##SUFFIX) \
5245VPTESTM_CASE(v8i16, WZ128##SUFFIX) \
5246VPTESTM_CASE(v32i8, BZ256##SUFFIX) \
5247VPTESTM_CASE(v16i16, WZ256##SUFFIX) \
5248VPTESTM_CASE(v64i8, BZ##SUFFIX) \
5249VPTESTM_CASE(v32i16, WZ##SUFFIX)
5250
5251 if (FoldedBCast) {
5252 switch (TestVT.SimpleTy) {
5254 }
5255 }
5256
5257 if (FoldedLoad) {
5258 switch (TestVT.SimpleTy) {
5260 }
5261 }
5262
5263 switch (TestVT.SimpleTy) {
5265 }
5266
5267#undef VPTESTM_FULL_CASES
5268#undef VPTESTM_BROADCAST_CASES
5269#undef VPTESTM_CASE
5270}
5271
5272static void orderRegForMul(SDValue &N0, SDValue &N1, const unsigned LoReg,
5273 const MachineRegisterInfo &MRI) {
5274 auto GetPhysReg = [&](SDValue V) -> Register {
5275 if (V.getOpcode() != ISD::CopyFromReg)
5276 return Register();
5277 Register Reg = cast<RegisterSDNode>(V.getOperand(1))->getReg();
5278 if (Reg.isVirtual())
5279 return MRI.getLiveInPhysReg(Reg);
5280 return Reg;
5281 };
5282
5283 if (GetPhysReg(N1) == LoReg && GetPhysReg(N0) != LoReg)
5284 std::swap(N0, N1);
5285}
5286
5287// Try to create VPTESTM instruction. If InMask is not null, it will be used
5288// to form a masked operation.
5289bool X86DAGToDAGISel::tryVPTESTM(SDNode *Root, SDValue Setcc,
5290 SDValue InMask) {
5291 assert(Subtarget->hasAVX512() && "Expected AVX512!");
5292 assert(Setcc.getSimpleValueType().getVectorElementType() == MVT::i1 &&
5293 "Unexpected VT!");
5294
5295 // Look for equal and not equal compares.
5296 ISD::CondCode CC = cast<CondCodeSDNode>(Setcc.getOperand(2))->get();
5297 if (CC != ISD::SETEQ && CC != ISD::SETNE)
5298 return false;
5299
5300 SDValue SetccOp0 = Setcc.getOperand(0);
5301 SDValue SetccOp1 = Setcc.getOperand(1);
5302
5303 // Canonicalize the all zero vector to the RHS.
5304 if (ISD::isBuildVectorAllZeros(SetccOp0.getNode()))
5305 std::swap(SetccOp0, SetccOp1);
5306
5307 // See if we're comparing against zero.
5308 if (!ISD::isBuildVectorAllZeros(SetccOp1.getNode()))
5309 return false;
5310
5311 SDValue N0 = SetccOp0;
5312
5313 MVT CmpVT = N0.getSimpleValueType();
5314 MVT CmpSVT = CmpVT.getVectorElementType();
5315
5316 // Start with both operands the same. We'll try to refine this.
5317 SDValue Src0 = N0;
5318 SDValue Src1 = N0;
5319
5320 {
5321 // Look through single use bitcasts.
5322 SDValue N0Temp = N0;
5323 if (N0Temp.getOpcode() == ISD::BITCAST && N0Temp.hasOneUse())
5324 N0Temp = N0.getOperand(0);
5325
5326 // Look for single use AND.
5327 if (N0Temp.getOpcode() == ISD::AND && N0Temp.hasOneUse()) {
5328 Src0 = N0Temp.getOperand(0);
5329 Src1 = N0Temp.getOperand(1);
5330 }
5331 }
5332
5333 // Without VLX we need to widen the operation.
5334 bool Widen = !Subtarget->hasVLX() && !CmpVT.is512BitVector();
5335
5336 auto tryFoldLoadOrBCast = [&](SDNode *Root, SDNode *P, SDValue &L,
5337 SDValue &Base, SDValue &Scale, SDValue &Index,
5338 SDValue &Disp, SDValue &Segment) {
5339 // If we need to widen, we can't fold the load.
5340 if (!Widen)
5341 if (tryFoldLoad(Root, P, L, Base, Scale, Index, Disp, Segment))
5342 return true;
5343
5344 // If we didn't fold a load, try to match broadcast. No widening limitation
5345 // for this. But only 32 and 64 bit types are supported.
5346 if (CmpSVT != MVT::i32 && CmpSVT != MVT::i64)
5347 return false;
5348
5349 // Look through single use bitcasts.
5350 if (L.getOpcode() == ISD::BITCAST && L.hasOneUse()) {
5351 P = L.getNode();
5352 L = L.getOperand(0);
5353 }
5354
5355 if (L.getOpcode() != X86ISD::VBROADCAST_LOAD)
5356 return false;
5357
5358 auto *MemIntr = cast<MemIntrinsicSDNode>(L);
5359 if (MemIntr->getMemoryVT().getSizeInBits() != CmpSVT.getSizeInBits())
5360 return false;
5361
5362 return tryFoldBroadcast(Root, P, L, Base, Scale, Index, Disp, Segment);
5363 };
5364
5365 // We can only fold loads if the sources are unique.
5366 bool CanFoldLoads = Src0 != Src1;
5367
5368 bool FoldedLoad = false;
5369 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
5370 if (CanFoldLoads) {
5371 FoldedLoad = tryFoldLoadOrBCast(Root, N0.getNode(), Src1, Tmp0, Tmp1, Tmp2,
5372 Tmp3, Tmp4);
5373 if (!FoldedLoad) {
5374 // And is commutative.
5375 FoldedLoad = tryFoldLoadOrBCast(Root, N0.getNode(), Src0, Tmp0, Tmp1,
5376 Tmp2, Tmp3, Tmp4);
5377 if (FoldedLoad)
5378 std::swap(Src0, Src1);
5379 }
5380 }
5381
5382 bool FoldedBCast = FoldedLoad && Src1.getOpcode() == X86ISD::VBROADCAST_LOAD;
5383
5384 bool IsMasked = InMask.getNode() != nullptr;
5385
5386 SDLoc dl(Root);
5387
5388 MVT ResVT = Setcc.getSimpleValueType();
5389 MVT MaskVT = ResVT;
5390 if (Widen) {
5391 // Widen the inputs using insert_subreg or copy_to_regclass.
5392 unsigned Scale = CmpVT.is128BitVector() ? 4 : 2;
5393 unsigned SubReg = CmpVT.is128BitVector() ? X86::sub_xmm : X86::sub_ymm;
5394 unsigned NumElts = CmpVT.getVectorNumElements() * Scale;
5395 CmpVT = MVT::getVectorVT(CmpSVT, NumElts);
5396 MaskVT = MVT::getVectorVT(MVT::i1, NumElts);
5397 SDValue ImplDef = SDValue(CurDAG->getMachineNode(X86::IMPLICIT_DEF, dl,
5398 CmpVT), 0);
5399 Src0 = CurDAG->getTargetInsertSubreg(SubReg, dl, CmpVT, ImplDef, Src0);
5400
5401 if (!FoldedBCast)
5402 Src1 = CurDAG->getTargetInsertSubreg(SubReg, dl, CmpVT, ImplDef, Src1);
5403
5404 if (IsMasked) {
5405 // Widen the mask.
5406 unsigned RegClass = TLI->getRegClassFor(MaskVT)->getID();
5407 SDValue RC = CurDAG->getTargetConstant(RegClass, dl, MVT::i32);
5408 InMask = SDValue(CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS,
5409 dl, MaskVT, InMask, RC), 0);
5410 }
5411 }
5412
5413 bool IsTestN = CC == ISD::SETEQ;
5414 unsigned Opc = getVPTESTMOpc(CmpVT, IsTestN, FoldedLoad, FoldedBCast,
5415 IsMasked);
5416
5417 MachineSDNode *CNode;
5418 if (FoldedLoad) {
5419 SDVTList VTs = CurDAG->getVTList(MaskVT, MVT::Other);
5420
5421 if (IsMasked) {
5422 SDValue Ops[] = { InMask, Src0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4,
5423 Src1.getOperand(0) };
5424 CNode = CurDAG->getMachineNode(Opc, dl, VTs, Ops);
5425 } else {
5426 SDValue Ops[] = { Src0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4,
5427 Src1.getOperand(0) };
5428 CNode = CurDAG->getMachineNode(Opc, dl, VTs, Ops);
5429 }
5430
5431 // Update the chain.
5432 ReplaceUses(Src1.getValue(1), SDValue(CNode, 1));
5433 // Record the mem-refs
5434 CurDAG->setNodeMemRefs(CNode, {cast<MemSDNode>(Src1)->getMemOperand()});
5435 } else {
5436 if (IsMasked)
5437 CNode = CurDAG->getMachineNode(Opc, dl, MaskVT, InMask, Src0, Src1);
5438 else
5439 CNode = CurDAG->getMachineNode(Opc, dl, MaskVT, Src0, Src1);
5440 }
5441
5442 // If we widened, we need to shrink the mask VT.
5443 if (Widen) {
5444 unsigned RegClass = TLI->getRegClassFor(ResVT)->getID();
5445 SDValue RC = CurDAG->getTargetConstant(RegClass, dl, MVT::i32);
5446 CNode = CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS,
5447 dl, ResVT, SDValue(CNode, 0), RC);
5448 }
5449
5450 ReplaceUses(SDValue(Root, 0), SDValue(CNode, 0));
5451 CurDAG->RemoveDeadNode(Root);
5452 return true;
5453}
5454
5455// Try to match the bitselect pattern (or (and A, B), (andn A, C)). Turn it
5456// into vpternlog.
5457bool X86DAGToDAGISel::tryMatchBitSelect(SDNode *N) {
5458 assert(N->getOpcode() == ISD::OR && "Unexpected opcode!");
5459
5460 MVT NVT = N->getSimpleValueType(0);
5461
5462 // Make sure we support VPTERNLOG.
5463 if (!NVT.isVector() || !Subtarget->hasAVX512())
5464 return false;
5465
5466 // We need VLX for 128/256-bit.
5467 if (!(Subtarget->hasVLX() || NVT.is512BitVector()))
5468 return false;
5469
5470 SDValue N0 = N->getOperand(0);
5471 SDValue N1 = N->getOperand(1);
5472
5473 // Canonicalize AND to LHS.
5474 if (N1.getOpcode() == ISD::AND)
5475 std::swap(N0, N1);
5476
5477 if (N0.getOpcode() != ISD::AND ||
5478 N1.getOpcode() != X86ISD::ANDNP ||
5479 !N0.hasOneUse() || !N1.hasOneUse())
5480 return false;
5481
5482 // ANDN is not commutable, use it to pick down A and C.
5483 SDValue A = N1.getOperand(0);
5484 SDValue C = N1.getOperand(1);
5485
5486 // AND is commutable, if one operand matches A, the other operand is B.
5487 // Otherwise this isn't a match.
5488 SDValue B;
5489 if (N0.getOperand(0) == A)
5490 B = N0.getOperand(1);
5491 else if (N0.getOperand(1) == A)
5492 B = N0.getOperand(0);
5493 else
5494 return false;
5495
5496 SDLoc dl(N);
5497 SDValue Imm = CurDAG->getTargetConstant(0xCA, dl, MVT::i8);
5498 SDValue Ternlog = CurDAG->getNode(X86ISD::VPTERNLOG, dl, NVT, A, B, C, Imm);
5499 ReplaceNode(N, Ternlog.getNode());
5500
5501 return matchVPTERNLOG(Ternlog.getNode(), Ternlog.getNode(), Ternlog.getNode(),
5502 Ternlog.getNode(), A, B, C, 0xCA);
5503}
5504
5505void X86DAGToDAGISel::Select(SDNode *Node) {
5506 MVT NVT = Node->getSimpleValueType(0);
5507 unsigned Opcode = Node->getOpcode();
5508 SDLoc dl(Node);
5509
5510 if (Node->isMachineOpcode()) {
5511 LLVM_DEBUG(dbgs() << "== "; Node->dump(CurDAG); dbgs() << '\n');
5512 Node->setNodeId(-1);
5513 return; // Already selected.
5514 }
5515
5516 switch (Opcode) {
5517 default: break;
5519 unsigned IntNo = Node->getConstantOperandVal(1);
5520 switch (IntNo) {
5521 default: break;
5522 case Intrinsic::x86_encodekey128:
5523 case Intrinsic::x86_encodekey256: {
5524 if (!Subtarget->hasKL())
5525 break;
5526
5527 unsigned Opcode;
5528 switch (IntNo) {
5529 default: llvm_unreachable("Impossible intrinsic");
5530 case Intrinsic::x86_encodekey128:
5531 Opcode = X86::ENCODEKEY128;
5532 break;
5533 case Intrinsic::x86_encodekey256:
5534 Opcode = X86::ENCODEKEY256;
5535 break;
5536 }
5537
5538 SDValue Chain = Node->getOperand(0);
5539 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM0, Node->getOperand(3),
5540 SDValue());
5541 if (Opcode == X86::ENCODEKEY256)
5542 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM1, Node->getOperand(4),
5543 Chain.getValue(1));
5544
5545 MachineSDNode *Res = CurDAG->getMachineNode(
5546 Opcode, dl, Node->getVTList(),
5547 {Node->getOperand(2), Chain, Chain.getValue(1)});
5548 ReplaceNode(Node, Res);
5549 return;
5550 }
5551 case Intrinsic::x86_tileloaddrs64_internal:
5552 case Intrinsic::x86_tileloaddrst164_internal:
5553 if (!Subtarget->hasAMXMOVRS())
5554 break;
5555 [[fallthrough]];
5556 case Intrinsic::x86_tileloadd64_internal:
5557 case Intrinsic::x86_tileloaddt164_internal: {
5558 if (!Subtarget->hasAMXTILE())
5559 break;
5560 auto *MFI =
5561 CurDAG->getMachineFunction().getInfo<X86MachineFunctionInfo>();
5562 MFI->setAMXProgModel(AMXProgModelEnum::ManagedRA);
5563 unsigned Opc;
5564 switch (IntNo) {
5565 default:
5566 llvm_unreachable("Unexpected intrinsic!");
5567 case Intrinsic::x86_tileloaddrs64_internal:
5568 Opc = X86::PTILELOADDRSV;
5569 break;
5570 case Intrinsic::x86_tileloaddrst164_internal:
5571 Opc = X86::PTILELOADDRST1V;
5572 break;
5573 case Intrinsic::x86_tileloadd64_internal:
5574 Opc = X86::PTILELOADDV;
5575 break;
5576 case Intrinsic::x86_tileloaddt164_internal:
5577 Opc = X86::PTILELOADDT1V;
5578 break;
5579 }
5580 // _tile_loadd_internal(row, col, buf, STRIDE)
5581 SDValue Base = Node->getOperand(4);
5582 SDValue Scale = getI8Imm(1, dl);
5583 SDValue Index = Node->getOperand(5);
5584 SDValue Disp = CurDAG->getTargetConstant(0, dl, MVT::i32);
5585 SDValue Segment = CurDAG->getRegister(0, MVT::i16);
5586 SDValue Chain = Node->getOperand(0);
5587 MachineSDNode *CNode;
5588 SDValue Ops[] = {Node->getOperand(2),
5589 Node->getOperand(3),
5590 Base,
5591 Scale,
5592 Index,
5593 Disp,
5594 Segment,
5595 Chain};
5596 CNode = CurDAG->getMachineNode(Opc, dl, {MVT::x86amx, MVT::Other}, Ops);
5597 ReplaceNode(Node, CNode);
5598 return;
5599 }
5600 }
5601 break;
5602 }
5603 case ISD::INTRINSIC_VOID: {
5604 unsigned IntNo = Node->getConstantOperandVal(1);
5605 switch (IntNo) {
5606 default: break;
5607 case Intrinsic::x86_sse3_monitor:
5608 case Intrinsic::x86_monitorx:
5609 case Intrinsic::x86_clzero: {
5610 bool Use64BitPtr = Node->getOperand(2).getValueType() == MVT::i64;
5611
5612 unsigned Opc = 0;
5613 switch (IntNo) {
5614 default: llvm_unreachable("Unexpected intrinsic!");
5615 case Intrinsic::x86_sse3_monitor:
5616 if (!Subtarget->hasSSE3())
5617 break;
5618 Opc = Use64BitPtr ? X86::MONITOR64rrr : X86::MONITOR32rrr;
5619 break;
5620 case Intrinsic::x86_monitorx:
5621 if (!Subtarget->hasMWAITX())
5622 break;
5623 Opc = Use64BitPtr ? X86::MONITORX64rrr : X86::MONITORX32rrr;
5624 break;
5625 case Intrinsic::x86_clzero:
5626 if (!Subtarget->hasCLZERO())
5627 break;
5628 Opc = Use64BitPtr ? X86::CLZERO64r : X86::CLZERO32r;
5629 break;
5630 }
5631
5632 if (Opc) {
5633 unsigned PtrReg = Use64BitPtr ? X86::RAX : X86::EAX;
5634 SDValue Chain = CurDAG->getCopyToReg(Node->getOperand(0), dl, PtrReg,
5635 Node->getOperand(2), SDValue());
5636 SDValue InGlue = Chain.getValue(1);
5637
5638 if (IntNo == Intrinsic::x86_sse3_monitor ||
5639 IntNo == Intrinsic::x86_monitorx) {
5640 // Copy the other two operands to ECX and EDX.
5641 Chain = CurDAG->getCopyToReg(Chain, dl, X86::ECX, Node->getOperand(3),
5642 InGlue);
5643 InGlue = Chain.getValue(1);
5644 Chain = CurDAG->getCopyToReg(Chain, dl, X86::EDX, Node->getOperand(4),
5645 InGlue);
5646 InGlue = Chain.getValue(1);
5647 }
5648
5649 MachineSDNode *CNode = CurDAG->getMachineNode(Opc, dl, MVT::Other,
5650 { Chain, InGlue});
5651 ReplaceNode(Node, CNode);
5652 return;
5653 }
5654
5655 break;
5656 }
5657 case Intrinsic::x86_tilestored64_internal: {
5658 auto *MFI =
5659 CurDAG->getMachineFunction().getInfo<X86MachineFunctionInfo>();
5660 MFI->setAMXProgModel(AMXProgModelEnum::ManagedRA);
5661 unsigned Opc = X86::PTILESTOREDV;
5662 // _tile_stored_internal(row, col, buf, STRIDE, c)
5663 SDValue Base = Node->getOperand(4);
5664 SDValue Scale = getI8Imm(1, dl);
5665 SDValue Index = Node->getOperand(5);
5666 SDValue Disp = CurDAG->getTargetConstant(0, dl, MVT::i32);
5667 SDValue Segment = CurDAG->getRegister(0, MVT::i16);
5668 SDValue Chain = Node->getOperand(0);
5669 MachineSDNode *CNode;
5670 SDValue Ops[] = {Node->getOperand(2),
5671 Node->getOperand(3),
5672 Base,
5673 Scale,
5674 Index,
5675 Disp,
5676 Segment,
5677 Node->getOperand(6),
5678 Chain};
5679 CNode = CurDAG->getMachineNode(Opc, dl, MVT::Other, Ops);
5680 ReplaceNode(Node, CNode);
5681 return;
5682 }
5683 case Intrinsic::x86_tileloaddrs64:
5684 case Intrinsic::x86_tileloaddrst164:
5685 if (!Subtarget->hasAMXMOVRS())
5686 break;
5687 [[fallthrough]];
5688 case Intrinsic::x86_tileloadd64:
5689 case Intrinsic::x86_tileloaddt164:
5690 case Intrinsic::x86_tilestored64: {
5691 if (!Subtarget->hasAMXTILE())
5692 break;
5693 auto *MFI =
5694 CurDAG->getMachineFunction().getInfo<X86MachineFunctionInfo>();
5695 MFI->setAMXProgModel(AMXProgModelEnum::DirectReg);
5696 unsigned Opc;
5697 switch (IntNo) {
5698 default: llvm_unreachable("Unexpected intrinsic!");
5699 case Intrinsic::x86_tileloadd64: Opc = X86::PTILELOADD; break;
5700 case Intrinsic::x86_tileloaddrs64:
5701 Opc = X86::PTILELOADDRS;
5702 break;
5703 case Intrinsic::x86_tileloaddt164: Opc = X86::PTILELOADDT1; break;
5704 case Intrinsic::x86_tileloaddrst164:
5705 Opc = X86::PTILELOADDRST1;
5706 break;
5707 case Intrinsic::x86_tilestored64: Opc = X86::PTILESTORED; break;
5708 }
5709 // FIXME: Match displacement and scale.
5710 unsigned TIndex = Node->getConstantOperandVal(2);
5711 SDValue TReg = getI8Imm(TIndex, dl);
5712 SDValue Base = Node->getOperand(3);
5713 SDValue Scale = getI8Imm(1, dl);
5714 SDValue Index = Node->getOperand(4);
5715 SDValue Disp = CurDAG->getTargetConstant(0, dl, MVT::i32);
5716 SDValue Segment = CurDAG->getRegister(0, MVT::i16);
5717 SDValue Chain = Node->getOperand(0);
5718 MachineSDNode *CNode;
5719 if (Opc == X86::PTILESTORED) {
5720 SDValue Ops[] = { Base, Scale, Index, Disp, Segment, TReg, Chain };
5721 CNode = CurDAG->getMachineNode(Opc, dl, MVT::Other, Ops);
5722 } else {
5723 SDValue Ops[] = { TReg, Base, Scale, Index, Disp, Segment, Chain };
5724 CNode = CurDAG->getMachineNode(Opc, dl, MVT::Other, Ops);
5725 }
5726 ReplaceNode(Node, CNode);
5727 return;
5728 }
5729 }
5730 break;
5731 }
5732 case ISD::BRIND:
5733 case X86ISD::NT_BRIND: {
5734 if (Subtarget->isTarget64BitILP32()) {
5735 // Converts a 32-bit register to a 64-bit, zero-extended version of
5736 // it. This is needed because x86-64 can do many things, but jmp %r32
5737 // ain't one of them.
5738 SDValue Target = Node->getOperand(1);
5739 assert(Target.getValueType() == MVT::i32 && "Unexpected VT!");
5740 SDValue ZextTarget = CurDAG->getZExtOrTrunc(Target, dl, MVT::i64);
5741 insertDAGNode(*CurDAG, SDValue(Node, 0), ZextTarget);
5742
5743 unsigned Opc = Opcode == X86ISD::NT_BRIND ? X86::JMP64r_NT : X86::JMP64r;
5744 SDNode *Res = CurDAG->getMachineNode(Opc, dl, MVT::Other, ZextTarget,
5745 Node->getOperand(0));
5746 ReplaceNode(Node, Res);
5747 return;
5748 }
5749 break;
5750 }
5752 ReplaceNode(Node, getGlobalBaseReg());
5753 return;
5754
5755 case ISD::BITCAST:
5756 // Just drop all 128/256/512-bit bitcasts.
5757 if (NVT.is512BitVector() || NVT.is256BitVector() || NVT.is128BitVector() ||
5758 NVT == MVT::f128) {
5759 ReplaceUses(SDValue(Node, 0), Node->getOperand(0));
5760 CurDAG->RemoveDeadNode(Node);
5761 return;
5762 }
5763 break;
5764
5765 case ISD::SRL:
5766 if (matchBitExtract(Node))
5767 return;
5768 [[fallthrough]];
5769 case ISD::SRA:
5770 case ISD::SHL:
5771 if (tryShiftAmountMod(Node))
5772 return;
5773 break;
5774
5775 case X86ISD::VPTERNLOG: {
5776 uint8_t Imm = Node->getConstantOperandVal(3);
5777 if (matchVPTERNLOG(Node, Node, Node, Node, Node->getOperand(0),
5778 Node->getOperand(1), Node->getOperand(2), Imm))
5779 return;
5780 break;
5781 }
5782
5783 case X86ISD::ANDNP:
5784 if (tryVPTERNLOG(Node))
5785 return;
5786 break;
5787
5788 case ISD::AND:
5789 if (NVT.isVectorOf(MVT::i1)) {
5790 // Try to form a masked VPTESTM. Operands can be in either order.
5791 SDValue N0 = Node->getOperand(0);
5792 SDValue N1 = Node->getOperand(1);
5793 if (N0.getOpcode() == ISD::SETCC && N0.hasOneUse() &&
5794 tryVPTESTM(Node, N0, N1))
5795 return;
5796 if (N1.getOpcode() == ISD::SETCC && N1.hasOneUse() &&
5797 tryVPTESTM(Node, N1, N0))
5798 return;
5799 }
5800
5801 if (MachineSDNode *NewNode = matchBEXTRFromAndImm(Node)) {
5802 ReplaceUses(SDValue(Node, 0), SDValue(NewNode, 0));
5803 CurDAG->RemoveDeadNode(Node);
5804 return;
5805 }
5806 if (matchBitExtract(Node))
5807 return;
5808 if (Subtarget->getCLOpts().and_imm_shrink && shrinkAndImmediate(Node))
5809 return;
5810
5811 [[fallthrough]];
5812 case ISD::XOR:
5813 // A standalone ~(-1 << n) mask is (-1 & lowmask(n)): mov -1; bzhi beats
5814 // mov -1; shlx; not. AND falls through to here and has already tried.
5815 if (Opcode == ISD::XOR && Subtarget->hasBMI2() && matchBitExtract(Node))
5816 return;
5817 [[fallthrough]];
5818 case ISD::OR:
5819 if (tryShrinkShlLogicImm(Node))
5820 return;
5821 if (Opcode == ISD::OR && tryMatchBitSelect(Node))
5822 return;
5823 if (tryVPTERNLOG(Node))
5824 return;
5825
5826 [[fallthrough]];
5827 case ISD::ADD:
5828 if (Opcode == ISD::ADD && matchBitExtract(Node))
5829 return;
5830 [[fallthrough]];
5831 case ISD::SUB: {
5832 // Try to avoid folding immediates with multiple uses for optsize.
5833 // This code tries to select to register form directly to avoid going
5834 // through the isel table which might fold the immediate. We can't change
5835 // the patterns on the add/sub/and/or/xor with immediate paterns in the
5836 // tablegen files to check immediate use count without making the patterns
5837 // unavailable to the fast-isel table.
5838 if (!CurDAG->shouldOptForSize())
5839 break;
5840
5841 // Only handle i8/i16/i32/i64.
5842 if (NVT != MVT::i8 && NVT != MVT::i16 && NVT != MVT::i32 && NVT != MVT::i64)
5843 break;
5844
5845 SDValue N0 = Node->getOperand(0);
5846 SDValue N1 = Node->getOperand(1);
5847
5848 auto *Cst = dyn_cast<ConstantSDNode>(N1);
5849 if (!Cst)
5850 break;
5851
5852 int64_t Val = Cst->getSExtValue();
5853
5854 // Make sure its an immediate that is considered foldable.
5855 // FIXME: Handle unsigned 32 bit immediates for 64-bit AND.
5856 if (!isInt<8>(Val) && !isInt<32>(Val))
5857 break;
5858
5859 // If this can match to INC/DEC, let it go.
5860 if (Opcode == ISD::ADD && (Val == 1 || Val == -1))
5861 break;
5862
5863 // Check if we should avoid folding this immediate.
5864 if (!shouldAvoidImmediateInstFormsForSize(N1.getNode()))
5865 break;
5866
5867 // We should not fold the immediate. So we need a register form instead.
5868 unsigned ROpc, MOpc;
5869 switch (NVT.SimpleTy) {
5870 default: llvm_unreachable("Unexpected VT!");
5871 case MVT::i8:
5872 switch (Opcode) {
5873 default: llvm_unreachable("Unexpected opcode!");
5874 case ISD::ADD:
5875 ROpc = GET_ND_IF_ENABLED(X86::ADD8rr);
5876 MOpc = GET_NDM_IF_ENABLED(X86::ADD8rm);
5877 break;
5878 case ISD::SUB:
5879 ROpc = GET_ND_IF_ENABLED(X86::SUB8rr);
5880 MOpc = GET_NDM_IF_ENABLED(X86::SUB8rm);
5881 break;
5882 case ISD::AND:
5883 ROpc = GET_ND_IF_ENABLED(X86::AND8rr);
5884 MOpc = GET_NDM_IF_ENABLED(X86::AND8rm);
5885 break;
5886 case ISD::OR:
5887 ROpc = GET_ND_IF_ENABLED(X86::OR8rr);
5888 MOpc = GET_NDM_IF_ENABLED(X86::OR8rm);
5889 break;
5890 case ISD::XOR:
5891 ROpc = GET_ND_IF_ENABLED(X86::XOR8rr);
5892 MOpc = GET_NDM_IF_ENABLED(X86::XOR8rm);
5893 break;
5894 }
5895 break;
5896 case MVT::i16:
5897 switch (Opcode) {
5898 default: llvm_unreachable("Unexpected opcode!");
5899 case ISD::ADD:
5900 ROpc = GET_ND_IF_ENABLED(X86::ADD16rr);
5901 MOpc = GET_NDM_IF_ENABLED(X86::ADD16rm);
5902 break;
5903 case ISD::SUB:
5904 ROpc = GET_ND_IF_ENABLED(X86::SUB16rr);
5905 MOpc = GET_NDM_IF_ENABLED(X86::SUB16rm);
5906 break;
5907 case ISD::AND:
5908 ROpc = GET_ND_IF_ENABLED(X86::AND16rr);
5909 MOpc = GET_NDM_IF_ENABLED(X86::AND16rm);
5910 break;
5911 case ISD::OR:
5912 ROpc = GET_ND_IF_ENABLED(X86::OR16rr);
5913 MOpc = GET_NDM_IF_ENABLED(X86::OR16rm);
5914 break;
5915 case ISD::XOR:
5916 ROpc = GET_ND_IF_ENABLED(X86::XOR16rr);
5917 MOpc = GET_NDM_IF_ENABLED(X86::XOR16rm);
5918 break;
5919 }
5920 break;
5921 case MVT::i32:
5922 switch (Opcode) {
5923 default: llvm_unreachable("Unexpected opcode!");
5924 case ISD::ADD:
5925 ROpc = GET_ND_IF_ENABLED(X86::ADD32rr);
5926 MOpc = GET_NDM_IF_ENABLED(X86::ADD32rm);
5927 break;
5928 case ISD::SUB:
5929 ROpc = GET_ND_IF_ENABLED(X86::SUB32rr);
5930 MOpc = GET_NDM_IF_ENABLED(X86::SUB32rm);
5931 break;
5932 case ISD::AND:
5933 ROpc = GET_ND_IF_ENABLED(X86::AND32rr);
5934 MOpc = GET_NDM_IF_ENABLED(X86::AND32rm);
5935 break;
5936 case ISD::OR:
5937 ROpc = GET_ND_IF_ENABLED(X86::OR32rr);
5938 MOpc = GET_NDM_IF_ENABLED(X86::OR32rm);
5939 break;
5940 case ISD::XOR:
5941 ROpc = GET_ND_IF_ENABLED(X86::XOR32rr);
5942 MOpc = GET_NDM_IF_ENABLED(X86::XOR32rm);
5943 break;
5944 }
5945 break;
5946 case MVT::i64:
5947 switch (Opcode) {
5948 default: llvm_unreachable("Unexpected opcode!");
5949 case ISD::ADD:
5950 ROpc = GET_ND_IF_ENABLED(X86::ADD64rr);
5951 MOpc = GET_NDM_IF_ENABLED(X86::ADD64rm);
5952 break;
5953 case ISD::SUB:
5954 ROpc = GET_ND_IF_ENABLED(X86::SUB64rr);
5955 MOpc = GET_NDM_IF_ENABLED(X86::SUB64rm);
5956 break;
5957 case ISD::AND:
5958 ROpc = GET_ND_IF_ENABLED(X86::AND64rr);
5959 MOpc = GET_NDM_IF_ENABLED(X86::AND64rm);
5960 break;
5961 case ISD::OR:
5962 ROpc = GET_ND_IF_ENABLED(X86::OR64rr);
5963 MOpc = GET_NDM_IF_ENABLED(X86::OR64rm);
5964 break;
5965 case ISD::XOR:
5966 ROpc = GET_ND_IF_ENABLED(X86::XOR64rr);
5967 MOpc = GET_NDM_IF_ENABLED(X86::XOR64rm);
5968 break;
5969 }
5970 break;
5971 }
5972
5973 // Ok this is a AND/OR/XOR/ADD/SUB with constant.
5974
5975 // If this is a not a subtract, we can still try to fold a load.
5976 if (Opcode != ISD::SUB) {
5977 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
5978 if (tryFoldLoad(Node, N0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
5979 SDValue Ops[] = { N1, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, N0.getOperand(0) };
5980 SDVTList VTs = CurDAG->getVTList(NVT, MVT::i32, MVT::Other);
5981 MachineSDNode *CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
5982 // Update the chain.
5983 ReplaceUses(N0.getValue(1), SDValue(CNode, 2));
5984 // Record the mem-refs
5985 CurDAG->setNodeMemRefs(CNode, {cast<LoadSDNode>(N0)->getMemOperand()});
5986 ReplaceUses(SDValue(Node, 0), SDValue(CNode, 0));
5987 CurDAG->RemoveDeadNode(Node);
5988 return;
5989 }
5990 }
5991
5992 CurDAG->SelectNodeTo(Node, ROpc, NVT, MVT::i32, N0, N1);
5993 return;
5994 }
5995
5996 case X86ISD::SMUL:
5997 // i16/i32/i64 are handled with isel patterns.
5998 if (NVT != MVT::i8)
5999 break;
6000 [[fallthrough]];
6001 case X86ISD::UMUL: {
6002 SDValue N0 = Node->getOperand(0);
6003 SDValue N1 = Node->getOperand(1);
6004
6005 unsigned LoReg, ROpc, MOpc;
6006 switch (NVT.SimpleTy) {
6007 default: llvm_unreachable("Unsupported VT!");
6008 case MVT::i8:
6009 LoReg = X86::AL;
6010 ROpc = Opcode == X86ISD::SMUL ? X86::IMUL8r : X86::MUL8r;
6011 MOpc = Opcode == X86ISD::SMUL ? X86::IMUL8m : X86::MUL8m;
6012 break;
6013 case MVT::i16:
6014 LoReg = X86::AX;
6015 ROpc = X86::MUL16r;
6016 MOpc = X86::MUL16m;
6017 break;
6018 case MVT::i32:
6019 LoReg = X86::EAX;
6020 ROpc = X86::MUL32r;
6021 MOpc = X86::MUL32m;
6022 break;
6023 case MVT::i64:
6024 LoReg = X86::RAX;
6025 ROpc = X86::MUL64r;
6026 MOpc = X86::MUL64m;
6027 break;
6028 }
6029
6030 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
6031 bool FoldedLoad = tryFoldLoad(Node, N1, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4);
6032 // Multiply is commutative.
6033 if (!FoldedLoad) {
6034 FoldedLoad = tryFoldLoad(Node, N0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4);
6035 if (FoldedLoad)
6036 std::swap(N0, N1);
6037 }
6038
6039 // UMUL/SMUL have an implicit source in LoReg (AL/AX/EAX/RAX). Prefer the
6040 // operand that's already there to avoid an extra register-to-register move.
6041 if (!FoldedLoad)
6042 orderRegForMul(N0, N1, LoReg, CurDAG->getMachineFunction().getRegInfo());
6043
6044 SDValue InGlue = CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, LoReg,
6045 N0, SDValue()).getValue(1);
6046
6047 MachineSDNode *CNode;
6048 if (FoldedLoad) {
6049 // i16/i32/i64 use an instruction that produces a low and high result even
6050 // though only the low result is used.
6051 SDVTList VTs;
6052 if (NVT == MVT::i8)
6053 VTs = CurDAG->getVTList(NVT, MVT::i32, MVT::Other);
6054 else
6055 VTs = CurDAG->getVTList(NVT, NVT, MVT::i32, MVT::Other);
6056
6057 SDValue Ops[] = { Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, N1.getOperand(0),
6058 InGlue };
6059 CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
6060
6061 // Update the chain.
6062 ReplaceUses(N1.getValue(1), SDValue(CNode, NVT == MVT::i8 ? 2 : 3));
6063 // Record the mem-refs
6064 CurDAG->setNodeMemRefs(CNode, {cast<LoadSDNode>(N1)->getMemOperand()});
6065 } else {
6066 // i16/i32/i64 use an instruction that produces a low and high result even
6067 // though only the low result is used.
6068 SDVTList VTs;
6069 if (NVT == MVT::i8)
6070 VTs = CurDAG->getVTList(NVT, MVT::i32);
6071 else
6072 VTs = CurDAG->getVTList(NVT, NVT, MVT::i32);
6073
6074 CNode = CurDAG->getMachineNode(ROpc, dl, VTs, {N1, InGlue});
6075 }
6076
6077 ReplaceUses(SDValue(Node, 0), SDValue(CNode, 0));
6078 ReplaceUses(SDValue(Node, 1), SDValue(CNode, NVT == MVT::i8 ? 1 : 2));
6079 CurDAG->RemoveDeadNode(Node);
6080 return;
6081 }
6082
6083 case ISD::SMUL_LOHI:
6084 case ISD::UMUL_LOHI: {
6085 SDValue N0 = Node->getOperand(0);
6086 SDValue N1 = Node->getOperand(1);
6087
6088 unsigned Opc, MOpc;
6089 unsigned LoReg, HiReg;
6090 bool IsSigned = Opcode == ISD::SMUL_LOHI;
6091 bool UseMULX = !IsSigned && Subtarget->hasBMI2();
6092 bool UseMULXHi = UseMULX && SDValue(Node, 0).use_empty();
6093 switch (NVT.SimpleTy) {
6094 default: llvm_unreachable("Unsupported VT!");
6095 case MVT::i32:
6096 Opc = UseMULXHi ? X86::MULX32Hrr
6097 : UseMULX ? GET_EGPR_IF_ENABLED(X86::MULX32rr)
6098 : IsSigned ? X86::IMUL32r
6099 : X86::MUL32r;
6100 MOpc = UseMULXHi ? X86::MULX32Hrm
6101 : UseMULX ? GET_EGPR_IF_ENABLED(X86::MULX32rm)
6102 : IsSigned ? X86::IMUL32m
6103 : X86::MUL32m;
6104 LoReg = UseMULX ? X86::EDX : X86::EAX;
6105 HiReg = X86::EDX;
6106 break;
6107 case MVT::i64:
6108 Opc = UseMULXHi ? X86::MULX64Hrr
6109 : UseMULX ? GET_EGPR_IF_ENABLED(X86::MULX64rr)
6110 : IsSigned ? X86::IMUL64r
6111 : X86::MUL64r;
6112 MOpc = UseMULXHi ? X86::MULX64Hrm
6113 : UseMULX ? GET_EGPR_IF_ENABLED(X86::MULX64rm)
6114 : IsSigned ? X86::IMUL64m
6115 : X86::MUL64m;
6116 LoReg = UseMULX ? X86::RDX : X86::RAX;
6117 HiReg = X86::RDX;
6118 break;
6119 }
6120
6121 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
6122 bool foldedLoad = tryFoldLoad(Node, N1, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4);
6123 // Multiply is commutative.
6124 if (!foldedLoad) {
6125 foldedLoad = tryFoldLoad(Node, N0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4);
6126 if (foldedLoad)
6127 std::swap(N0, N1);
6128 }
6129
6130 // UMUL/SMUL_LOHI has an implicit source in LoReg (RDX for MULX, RAX for
6131 // MUL/IMUL). Prefer the operand that's already there.
6132 if (!foldedLoad)
6133 orderRegForMul(N0, N1, LoReg, CurDAG->getMachineFunction().getRegInfo());
6134
6135 SDValue InGlue = CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, LoReg,
6136 N0, SDValue()).getValue(1);
6137 SDValue ResHi, ResLo;
6138 if (foldedLoad) {
6139 SDValue Chain;
6140 MachineSDNode *CNode = nullptr;
6141 SDValue Ops[] = { Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, N1.getOperand(0),
6142 InGlue };
6143 if (UseMULXHi) {
6144 SDVTList VTs = CurDAG->getVTList(NVT, MVT::Other);
6145 CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
6146 ResHi = SDValue(CNode, 0);
6147 Chain = SDValue(CNode, 1);
6148 } else if (UseMULX) {
6149 SDVTList VTs = CurDAG->getVTList(NVT, NVT, MVT::Other);
6150 CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
6151 ResHi = SDValue(CNode, 0);
6152 ResLo = SDValue(CNode, 1);
6153 Chain = SDValue(CNode, 2);
6154 } else {
6155 SDVTList VTs = CurDAG->getVTList(MVT::Other, MVT::Glue);
6156 CNode = CurDAG->getMachineNode(MOpc, dl, VTs, Ops);
6157 Chain = SDValue(CNode, 0);
6158 InGlue = SDValue(CNode, 1);
6159 }
6160
6161 // Update the chain.
6162 ReplaceUses(N1.getValue(1), Chain);
6163 // Record the mem-refs
6164 CurDAG->setNodeMemRefs(CNode, {cast<LoadSDNode>(N1)->getMemOperand()});
6165 } else {
6166 SDValue Ops[] = { N1, InGlue };
6167 if (UseMULXHi) {
6168 SDVTList VTs = CurDAG->getVTList(NVT);
6169 SDNode *CNode = CurDAG->getMachineNode(Opc, dl, VTs, Ops);
6170 ResHi = SDValue(CNode, 0);
6171 } else if (UseMULX) {
6172 SDVTList VTs = CurDAG->getVTList(NVT, NVT);
6173 SDNode *CNode = CurDAG->getMachineNode(Opc, dl, VTs, Ops);
6174 ResHi = SDValue(CNode, 0);
6175 ResLo = SDValue(CNode, 1);
6176 } else {
6177 SDVTList VTs = CurDAG->getVTList(MVT::Glue);
6178 SDNode *CNode = CurDAG->getMachineNode(Opc, dl, VTs, Ops);
6179 InGlue = SDValue(CNode, 0);
6180 }
6181 }
6182
6183 // Copy the low half of the result, if it is needed.
6184 if (!SDValue(Node, 0).use_empty()) {
6185 if (!ResLo) {
6186 assert(LoReg && "Register for low half is not defined!");
6187 ResLo = CurDAG->getCopyFromReg(CurDAG->getEntryNode(), dl, LoReg,
6188 NVT, InGlue);
6189 InGlue = ResLo.getValue(2);
6190 }
6191 ReplaceUses(SDValue(Node, 0), ResLo);
6192 LLVM_DEBUG(dbgs() << "=> "; ResLo.getNode()->dump(CurDAG);
6193 dbgs() << '\n');
6194 }
6195 // Copy the high half of the result, if it is needed.
6196 if (!SDValue(Node, 1).use_empty()) {
6197 if (!ResHi) {
6198 assert(HiReg && "Register for high half is not defined!");
6199 ResHi = CurDAG->getCopyFromReg(CurDAG->getEntryNode(), dl, HiReg,
6200 NVT, InGlue);
6201 InGlue = ResHi.getValue(2);
6202 }
6203 ReplaceUses(SDValue(Node, 1), ResHi);
6204 LLVM_DEBUG(dbgs() << "=> "; ResHi.getNode()->dump(CurDAG);
6205 dbgs() << '\n');
6206 }
6207
6208 CurDAG->RemoveDeadNode(Node);
6209 return;
6210 }
6211
6212 case ISD::SDIVREM:
6213 case ISD::UDIVREM: {
6214 SDValue N0 = Node->getOperand(0);
6215 SDValue N1 = Node->getOperand(1);
6216
6217 unsigned ROpc, MOpc;
6218 bool isSigned = Opcode == ISD::SDIVREM;
6219 if (!isSigned) {
6220 switch (NVT.SimpleTy) {
6221 default: llvm_unreachable("Unsupported VT!");
6222 case MVT::i8: ROpc = X86::DIV8r; MOpc = X86::DIV8m; break;
6223 case MVT::i16: ROpc = X86::DIV16r; MOpc = X86::DIV16m; break;
6224 case MVT::i32: ROpc = X86::DIV32r; MOpc = X86::DIV32m; break;
6225 case MVT::i64: ROpc = X86::DIV64r; MOpc = X86::DIV64m; break;
6226 }
6227 } else {
6228 switch (NVT.SimpleTy) {
6229 default: llvm_unreachable("Unsupported VT!");
6230 case MVT::i8: ROpc = X86::IDIV8r; MOpc = X86::IDIV8m; break;
6231 case MVT::i16: ROpc = X86::IDIV16r; MOpc = X86::IDIV16m; break;
6232 case MVT::i32: ROpc = X86::IDIV32r; MOpc = X86::IDIV32m; break;
6233 case MVT::i64: ROpc = X86::IDIV64r; MOpc = X86::IDIV64m; break;
6234 }
6235 }
6236
6237 unsigned LoReg, HiReg, ClrReg;
6238 unsigned SExtOpcode;
6239 switch (NVT.SimpleTy) {
6240 default: llvm_unreachable("Unsupported VT!");
6241 case MVT::i8:
6242 LoReg = X86::AL; ClrReg = HiReg = X86::AH;
6243 SExtOpcode = 0; // Not used.
6244 break;
6245 case MVT::i16:
6246 LoReg = X86::AX; HiReg = X86::DX;
6247 ClrReg = X86::DX;
6248 SExtOpcode = X86::CWD;
6249 break;
6250 case MVT::i32:
6251 LoReg = X86::EAX; ClrReg = HiReg = X86::EDX;
6252 SExtOpcode = X86::CDQ;
6253 break;
6254 case MVT::i64:
6255 LoReg = X86::RAX; ClrReg = HiReg = X86::RDX;
6256 SExtOpcode = X86::CQO;
6257 break;
6258 }
6259
6260 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
6261 bool foldedLoad = tryFoldLoad(Node, N1, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4);
6262 bool signBitIsZero = CurDAG->SignBitIsZero(N0);
6263
6264 SDValue InGlue;
6265 if (NVT == MVT::i8) {
6266 // Special case for div8, just use a move with zero extension to AX to
6267 // clear the upper 8 bits (AH).
6268 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, Chain;
6269 MachineSDNode *Move;
6270 if (tryFoldLoad(Node, N0, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
6271 SDValue Ops[] = { Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, N0.getOperand(0) };
6272 unsigned Opc = (isSigned && !signBitIsZero) ? X86::MOVSX16rm8
6273 : X86::MOVZX16rm8;
6274 Move = CurDAG->getMachineNode(Opc, dl, MVT::i16, MVT::Other, Ops);
6275 Chain = SDValue(Move, 1);
6276 ReplaceUses(N0.getValue(1), Chain);
6277 // Record the mem-refs
6278 CurDAG->setNodeMemRefs(Move, {cast<LoadSDNode>(N0)->getMemOperand()});
6279 } else {
6280 unsigned Opc = (isSigned && !signBitIsZero) ? X86::MOVSX16rr8
6281 : X86::MOVZX16rr8;
6282 Move = CurDAG->getMachineNode(Opc, dl, MVT::i16, N0);
6283 Chain = CurDAG->getEntryNode();
6284 }
6285 Chain = CurDAG->getCopyToReg(Chain, dl, X86::AX, SDValue(Move, 0),
6286 SDValue());
6287 InGlue = Chain.getValue(1);
6288 } else {
6289 InGlue =
6290 CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl,
6291 LoReg, N0, SDValue()).getValue(1);
6292 if (isSigned && !signBitIsZero) {
6293 // Sign extend the low part into the high part.
6294 InGlue =
6295 SDValue(CurDAG->getMachineNode(SExtOpcode, dl, MVT::Glue, InGlue),0);
6296 } else {
6297 // Zero out the high part, effectively zero extending the input.
6298 SDVTList VTs = CurDAG->getVTList(MVT::i32, MVT::i32);
6299 SDValue ClrNode =
6300 SDValue(CurDAG->getMachineNode(X86::MOV32r0, dl, VTs, {}), 0);
6301 switch (NVT.SimpleTy) {
6302 case MVT::i16:
6303 ClrNode =
6304 SDValue(CurDAG->getMachineNode(
6305 TargetOpcode::EXTRACT_SUBREG, dl, MVT::i16, ClrNode,
6306 CurDAG->getTargetConstant(X86::sub_16bit, dl,
6307 MVT::i32)),
6308 0);
6309 break;
6310 case MVT::i32:
6311 break;
6312 case MVT::i64:
6313 ClrNode = SDValue(
6314 CurDAG->getMachineNode(
6315 TargetOpcode::SUBREG_TO_REG, dl, MVT::i64, ClrNode,
6316 CurDAG->getTargetConstant(X86::sub_32bit, dl, MVT::i32)),
6317 0);
6318 break;
6319 default:
6320 llvm_unreachable("Unexpected division source");
6321 }
6322
6323 InGlue = CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, ClrReg,
6324 ClrNode, InGlue).getValue(1);
6325 }
6326 }
6327
6328 if (foldedLoad) {
6329 SDValue Ops[] = { Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, N1.getOperand(0),
6330 InGlue };
6331 MachineSDNode *CNode =
6332 CurDAG->getMachineNode(MOpc, dl, MVT::Other, MVT::Glue, Ops);
6333 InGlue = SDValue(CNode, 1);
6334 // Update the chain.
6335 ReplaceUses(N1.getValue(1), SDValue(CNode, 0));
6336 // Record the mem-refs
6337 CurDAG->setNodeMemRefs(CNode, {cast<LoadSDNode>(N1)->getMemOperand()});
6338 } else {
6339 InGlue =
6340 SDValue(CurDAG->getMachineNode(ROpc, dl, MVT::Glue, N1, InGlue), 0);
6341 }
6342
6343 // Prevent use of AH in a REX instruction by explicitly copying it to
6344 // an ABCD_L register.
6345 //
6346 // The current assumption of the register allocator is that isel
6347 // won't generate explicit references to the GR8_ABCD_H registers. If
6348 // the allocator and/or the backend get enhanced to be more robust in
6349 // that regard, this can be, and should be, removed.
6350 if (HiReg == X86::AH && !SDValue(Node, 1).use_empty()) {
6351 SDValue AHCopy = CurDAG->getRegister(X86::AH, MVT::i8);
6352 unsigned AHExtOpcode =
6353 isSigned ? X86::MOVSX32rr8_NOREX : X86::MOVZX32rr8_NOREX;
6354
6355 SDNode *RNode = CurDAG->getMachineNode(AHExtOpcode, dl, MVT::i32,
6356 MVT::Glue, AHCopy, InGlue);
6357 SDValue Result(RNode, 0);
6358 InGlue = SDValue(RNode, 1);
6359
6360 Result =
6361 CurDAG->getTargetExtractSubreg(X86::sub_8bit, dl, MVT::i8, Result);
6362
6363 ReplaceUses(SDValue(Node, 1), Result);
6364 LLVM_DEBUG(dbgs() << "=> "; Result.getNode()->dump(CurDAG);
6365 dbgs() << '\n');
6366 }
6367 // Copy the division (low) result, if it is needed.
6368 if (!SDValue(Node, 0).use_empty()) {
6369 SDValue Result = CurDAG->getCopyFromReg(CurDAG->getEntryNode(), dl,
6370 LoReg, NVT, InGlue);
6371 InGlue = Result.getValue(2);
6372 ReplaceUses(SDValue(Node, 0), Result);
6373 LLVM_DEBUG(dbgs() << "=> "; Result.getNode()->dump(CurDAG);
6374 dbgs() << '\n');
6375 }
6376 // Copy the remainder (high) result, if it is needed.
6377 if (!SDValue(Node, 1).use_empty()) {
6378 SDValue Result = CurDAG->getCopyFromReg(CurDAG->getEntryNode(), dl,
6379 HiReg, NVT, InGlue);
6380 InGlue = Result.getValue(2);
6381 ReplaceUses(SDValue(Node, 1), Result);
6382 LLVM_DEBUG(dbgs() << "=> "; Result.getNode()->dump(CurDAG);
6383 dbgs() << '\n');
6384 }
6385 CurDAG->RemoveDeadNode(Node);
6386 return;
6387 }
6388
6389 case X86ISD::FCMP:
6390 case X86ISD::STRICT_FCMP:
6391 case X86ISD::STRICT_FCMPS: {
6392 bool IsStrictCmp = Node->getOpcode() == X86ISD::STRICT_FCMP ||
6393 Node->getOpcode() == X86ISD::STRICT_FCMPS;
6394 SDValue N0 = Node->getOperand(IsStrictCmp ? 1 : 0);
6395 SDValue N1 = Node->getOperand(IsStrictCmp ? 2 : 1);
6396
6397 // Save the original VT of the compare.
6398 MVT CmpVT = N0.getSimpleValueType();
6399
6400 // Floating point needs special handling if we don't have FCOMI.
6401 if (Subtarget->canUseCMOV())
6402 break;
6403
6404 bool IsSignaling = Node->getOpcode() == X86ISD::STRICT_FCMPS;
6405
6406 unsigned Opc;
6407 switch (CmpVT.SimpleTy) {
6408 default: llvm_unreachable("Unexpected type!");
6409 case MVT::f32:
6410 Opc = IsSignaling ? X86::COM_Fpr32 : X86::UCOM_Fpr32;
6411 break;
6412 case MVT::f64:
6413 Opc = IsSignaling ? X86::COM_Fpr64 : X86::UCOM_Fpr64;
6414 break;
6415 case MVT::f80:
6416 Opc = IsSignaling ? X86::COM_Fpr80 : X86::UCOM_Fpr80;
6417 break;
6418 }
6419
6420 SDValue Chain =
6421 IsStrictCmp ? Node->getOperand(0) : CurDAG->getEntryNode();
6422 SDValue Glue;
6423 if (IsStrictCmp) {
6424 SDVTList VTs = CurDAG->getVTList(MVT::Other, MVT::Glue);
6425 Chain = SDValue(CurDAG->getMachineNode(Opc, dl, VTs, {N0, N1, Chain}), 0);
6426 Glue = Chain.getValue(1);
6427 } else {
6428 Glue = SDValue(CurDAG->getMachineNode(Opc, dl, MVT::Glue, N0, N1), 0);
6429 }
6430
6431 // Move FPSW to AX.
6432 SDValue FNSTSW =
6433 SDValue(CurDAG->getMachineNode(X86::FNSTSW16r, dl, MVT::i16, Glue), 0);
6434
6435 // Extract upper 8-bits of AX.
6436 SDValue Extract =
6437 CurDAG->getTargetExtractSubreg(X86::sub_8bit_hi, dl, MVT::i8, FNSTSW);
6438
6439 // Move AH into flags.
6440 // Some 64-bit targets lack SAHF support, but they do support FCOMI.
6441 assert(Subtarget->canUseLAHFSAHF() &&
6442 "Target doesn't support SAHF or FCOMI?");
6443 SDValue AH = CurDAG->getCopyToReg(Chain, dl, X86::AH, Extract, SDValue());
6444 Chain = AH;
6445 SDValue SAHF = SDValue(
6446 CurDAG->getMachineNode(X86::SAHF, dl, MVT::i32, AH.getValue(1)), 0);
6447
6448 if (IsStrictCmp)
6449 ReplaceUses(SDValue(Node, 1), Chain);
6450
6451 ReplaceUses(SDValue(Node, 0), SAHF);
6452 CurDAG->RemoveDeadNode(Node);
6453 return;
6454 }
6455
6456 case X86ISD::CMP: {
6457 SDValue N0 = Node->getOperand(0);
6458 SDValue N1 = Node->getOperand(1);
6459
6460 // Optimizations for TEST compares.
6461 if (!isNullConstant(N1))
6462 break;
6463
6464 // Save the original VT of the compare.
6465 MVT CmpVT = N0.getSimpleValueType();
6466
6467 // If we are comparing (and (shr X, C, Mask) with 0, emit a BEXTR followed
6468 // by a test instruction. The test should be removed later by
6469 // analyzeCompare if we are using only the zero flag.
6470 // TODO: Should we check the users and use the BEXTR flags directly?
6471 if (N0.getOpcode() == ISD::AND && N0.hasOneUse()) {
6472 if (MachineSDNode *NewNode = matchBEXTRFromAndImm(N0.getNode())) {
6473 unsigned TestOpc = CmpVT == MVT::i64 ? X86::TEST64rr
6474 : X86::TEST32rr;
6475 SDValue BEXTR = SDValue(NewNode, 0);
6476 NewNode = CurDAG->getMachineNode(TestOpc, dl, MVT::i32, BEXTR, BEXTR);
6477 ReplaceUses(SDValue(Node, 0), SDValue(NewNode, 0));
6478 CurDAG->RemoveDeadNode(Node);
6479 return;
6480 }
6481 }
6482
6483 // We can peek through truncates, but we need to be careful below.
6484 if (N0.getOpcode() == ISD::TRUNCATE && N0.hasOneUse())
6485 N0 = N0.getOperand(0);
6486
6487 // Look for (X86cmp (and $op, $imm), 0) and see if we can convert it to
6488 // use a smaller encoding.
6489 // Look past the truncate if CMP is the only use of it.
6490 if (N0.getOpcode() == ISD::AND && N0.getNode()->hasOneUse() &&
6491 N0.getValueType() != MVT::i8) {
6492 auto *MaskC = dyn_cast<ConstantSDNode>(N0.getOperand(1));
6493 if (!MaskC)
6494 break;
6495
6496 // We may have looked through a truncate so mask off any bits that
6497 // shouldn't be part of the compare.
6498 uint64_t Mask = MaskC->getZExtValue();
6500
6501 // Check if we can replace AND+IMM{32,64} with a shift. This is possible
6502 // for masks like 0xFF000000 or 0x00FFFFFF and if we care only about the
6503 // zero flag.
6504 if (CmpVT == MVT::i64 && !isInt<8>(Mask) && isShiftedMask_64(Mask) &&
6505 onlyUsesZeroFlag(SDValue(Node, 0))) {
6506 unsigned ShiftOpcode = ISD::DELETED_NODE;
6507 unsigned ShiftAmt;
6508 unsigned SubRegIdx;
6509 MVT SubRegVT;
6510 unsigned TestOpcode;
6511 unsigned LeadingZeros = llvm::countl_zero(Mask);
6512 unsigned TrailingZeros = llvm::countr_zero(Mask);
6513
6514 // With leading/trailing zeros, the transform is profitable if we can
6515 // eliminate a movabsq or shrink a 32-bit immediate to 8-bit without
6516 // incurring any extra register moves.
6517 bool SavesBytes = !isInt<32>(Mask) || N0.getOperand(0).hasOneUse();
6518 if (LeadingZeros == 0 && SavesBytes) {
6519 // If the mask covers the most significant bit, then we can replace
6520 // TEST+AND with a SHR and check eflags.
6521 // This emits a redundant TEST which is subsequently eliminated.
6522 ShiftOpcode = GET_ND_IF_ENABLED(X86::SHR64ri);
6523 ShiftAmt = TrailingZeros;
6524 SubRegIdx = 0;
6525 TestOpcode = X86::TEST64rr;
6526 } else if (TrailingZeros == 0 && SavesBytes) {
6527 // If the mask covers the least significant bit, then we can replace
6528 // TEST+AND with a SHL and check eflags.
6529 // This emits a redundant TEST which is subsequently eliminated,
6530 // except for shift amounts 1 to 3: isDefConvertible() rejects those
6531 // SHLs to keep them convertible to LEA, so the TEST would survive.
6532 if (LeadingZeros == 1) {
6533 // Shift out the top bit by doubling with ADD reg,reg instead: it
6534 // is the same length and sets ZF identically, but the peephole
6535 // does fold the TEST into it, and it runs on more ports.
6536 MachineSDNode *Add = CurDAG->getMachineNode(
6537 GET_ND_IF_ENABLED(X86::ADD64rr), dl, MVT::i64, MVT::i32,
6538 N0.getOperand(0), N0.getOperand(0));
6539 MachineSDNode *Test = CurDAG->getMachineNode(
6540 X86::TEST64rr, dl, MVT::i32, SDValue(Add, 0), SDValue(Add, 0));
6541 ReplaceNode(Node, Test);
6542 return;
6543 }
6544 ShiftOpcode = GET_ND_IF_ENABLED(X86::SHL64ri);
6545 ShiftAmt = LeadingZeros;
6546 SubRegIdx = 0;
6547 TestOpcode = X86::TEST64rr;
6548 } else if (MaskC->hasOneUse() && !isInt<32>(Mask)) {
6549 // If the shifted mask extends into the high half and is 8/16/32 bits
6550 // wide, then replace it with a SHR and a TEST8rr/TEST16rr/TEST32rr.
6551 unsigned PopCount = 64 - LeadingZeros - TrailingZeros;
6552 if (PopCount == 8) {
6553 ShiftOpcode = GET_ND_IF_ENABLED(X86::SHR64ri);
6554 ShiftAmt = TrailingZeros;
6555 SubRegIdx = X86::sub_8bit;
6556 SubRegVT = MVT::i8;
6557 TestOpcode = X86::TEST8rr;
6558 } else if (PopCount == 16) {
6559 ShiftOpcode = GET_ND_IF_ENABLED(X86::SHR64ri);
6560 ShiftAmt = TrailingZeros;
6561 SubRegIdx = X86::sub_16bit;
6562 SubRegVT = MVT::i16;
6563 TestOpcode = X86::TEST16rr;
6564 } else if (PopCount == 32) {
6565 ShiftOpcode = GET_ND_IF_ENABLED(X86::SHR64ri);
6566 ShiftAmt = TrailingZeros;
6567 SubRegIdx = X86::sub_32bit;
6568 SubRegVT = MVT::i32;
6569 TestOpcode = X86::TEST32rr;
6570 }
6571 }
6572 if (ShiftOpcode != ISD::DELETED_NODE) {
6573 SDValue ShiftC = CurDAG->getTargetConstant(ShiftAmt, dl, MVT::i64);
6574 SDValue Shift = SDValue(
6575 CurDAG->getMachineNode(ShiftOpcode, dl, MVT::i64, MVT::i32,
6576 N0.getOperand(0), ShiftC),
6577 0);
6578 if (SubRegIdx != 0) {
6579 Shift =
6580 CurDAG->getTargetExtractSubreg(SubRegIdx, dl, SubRegVT, Shift);
6581 }
6582 MachineSDNode *Test =
6583 CurDAG->getMachineNode(TestOpcode, dl, MVT::i32, Shift, Shift);
6584 ReplaceNode(Node, Test);
6585 return;
6586 }
6587 }
6588
6589 MVT VT;
6590 int SubRegOp;
6591 unsigned ROpc, MOpc;
6592
6593 // For each of these checks we need to be careful if the sign flag is
6594 // being used. It is only safe to use the sign flag in two conditions,
6595 // either the sign bit in the shrunken mask is zero or the final test
6596 // size is equal to the original compare size.
6597
6598 if (isUInt<8>(Mask) &&
6599 (!(Mask & 0x80) || CmpVT == MVT::i8 ||
6600 hasNoSignFlagUses(SDValue(Node, 0)))) {
6601 // For example, convert "testl %eax, $8" to "testb %al, $8"
6602 VT = MVT::i8;
6603 SubRegOp = X86::sub_8bit;
6604 ROpc = X86::TEST8ri;
6605 MOpc = X86::TEST8mi;
6606 } else if (OptForMinSize && isUInt<16>(Mask) &&
6607 (!(Mask & 0x8000) || CmpVT == MVT::i16 ||
6608 hasNoSignFlagUses(SDValue(Node, 0)))) {
6609 // For example, "testl %eax, $32776" to "testw %ax, $32776".
6610 // NOTE: We only want to form TESTW instructions if optimizing for
6611 // min size. Otherwise we only save one byte and possibly get a length
6612 // changing prefix penalty in the decoders.
6613 VT = MVT::i16;
6614 SubRegOp = X86::sub_16bit;
6615 ROpc = X86::TEST16ri;
6616 MOpc = X86::TEST16mi;
6617 } else if (isUInt<32>(Mask) && N0.getValueType() != MVT::i16 &&
6618 ((!(Mask & 0x80000000) &&
6619 // Without minsize 16-bit Cmps can get here so we need to
6620 // be sure we calculate the correct sign flag if needed.
6621 (CmpVT != MVT::i16 || !(Mask & 0x8000))) ||
6622 CmpVT == MVT::i32 ||
6623 hasNoSignFlagUses(SDValue(Node, 0)))) {
6624 // For example, "testq %rax, $268468232" to "testl %eax, $268468232".
6625 // NOTE: We only want to run that transform if N0 is 32 or 64 bits.
6626 // Otherwize, we find ourselves in a position where we have to do
6627 // promotion. If previous passes did not promote the and, we assume
6628 // they had a good reason not to and do not promote here.
6629 VT = MVT::i32;
6630 SubRegOp = X86::sub_32bit;
6631 ROpc = X86::TEST32ri;
6632 MOpc = X86::TEST32mi;
6633 } else {
6634 // No eligible transformation was found.
6635 break;
6636 }
6637
6638 SDValue Imm = CurDAG->getTargetConstant(Mask, dl, VT);
6639 SDValue Reg = N0.getOperand(0);
6640
6641 // Emit a testl or testw.
6642 MachineSDNode *NewNode;
6643 SDValue Tmp0, Tmp1, Tmp2, Tmp3, Tmp4;
6644 if (tryFoldLoad(Node, N0.getNode(), Reg, Tmp0, Tmp1, Tmp2, Tmp3, Tmp4)) {
6645 if (auto *LoadN = dyn_cast<LoadSDNode>(N0.getOperand(0).getNode())) {
6646 if (!LoadN->isSimple()) {
6647 unsigned NumVolBits = LoadN->getValueType(0).getSizeInBits();
6648 if ((MOpc == X86::TEST8mi && NumVolBits != 8) ||
6649 (MOpc == X86::TEST16mi && NumVolBits != 16) ||
6650 (MOpc == X86::TEST32mi && NumVolBits != 32))
6651 break;
6652 }
6653 }
6654 SDValue Ops[] = { Tmp0, Tmp1, Tmp2, Tmp3, Tmp4, Imm,
6655 Reg.getOperand(0) };
6656 NewNode = CurDAG->getMachineNode(MOpc, dl, MVT::i32, MVT::Other, Ops);
6657 // Update the chain.
6658 ReplaceUses(Reg.getValue(1), SDValue(NewNode, 1));
6659 // Record the mem-refs
6660 CurDAG->setNodeMemRefs(NewNode,
6661 {cast<LoadSDNode>(Reg)->getMemOperand()});
6662 } else {
6663 // Extract the subregister if necessary.
6664 if (N0.getValueType() != VT)
6665 Reg = CurDAG->getTargetExtractSubreg(SubRegOp, dl, VT, Reg);
6666
6667 NewNode = CurDAG->getMachineNode(ROpc, dl, MVT::i32, Reg, Imm);
6668 }
6669 // Replace CMP with TEST.
6670 ReplaceNode(Node, NewNode);
6671 return;
6672 }
6673 break;
6674 }
6675 case X86ISD::PCMPISTR: {
6676 if (!Subtarget->hasSSE42())
6677 break;
6678
6679 bool NeedIndex = !SDValue(Node, 0).use_empty();
6680 bool NeedMask = !SDValue(Node, 1).use_empty();
6681 // We can't fold a load if we are going to make two instructions.
6682 bool MayFoldLoad = !NeedIndex || !NeedMask;
6683
6684 MachineSDNode *CNode;
6685 if (NeedMask) {
6686 unsigned ROpc =
6687 Subtarget->hasAVX() ? X86::VPCMPISTRMrri : X86::PCMPISTRMrri;
6688 unsigned MOpc =
6689 Subtarget->hasAVX() ? X86::VPCMPISTRMrmi : X86::PCMPISTRMrmi;
6690 CNode = emitPCMPISTR(ROpc, MOpc, MayFoldLoad, dl, MVT::v16i8, Node);
6691 ReplaceUses(SDValue(Node, 1), SDValue(CNode, 0));
6692 }
6693 if (NeedIndex || !NeedMask) {
6694 unsigned ROpc =
6695 Subtarget->hasAVX() ? X86::VPCMPISTRIrri : X86::PCMPISTRIrri;
6696 unsigned MOpc =
6697 Subtarget->hasAVX() ? X86::VPCMPISTRIrmi : X86::PCMPISTRIrmi;
6698 CNode = emitPCMPISTR(ROpc, MOpc, MayFoldLoad, dl, MVT::i32, Node);
6699 ReplaceUses(SDValue(Node, 0), SDValue(CNode, 0));
6700 }
6701
6702 // Connect the flag usage to the last instruction created.
6703 ReplaceUses(SDValue(Node, 2), SDValue(CNode, 1));
6704 CurDAG->RemoveDeadNode(Node);
6705 return;
6706 }
6707 case X86ISD::PCMPESTR: {
6708 if (!Subtarget->hasSSE42())
6709 break;
6710
6711 // Copy the two implicit register inputs.
6712 SDValue InGlue = CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, X86::EAX,
6713 Node->getOperand(1),
6714 SDValue()).getValue(1);
6715 InGlue = CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, X86::EDX,
6716 Node->getOperand(3), InGlue).getValue(1);
6717
6718 bool NeedIndex = !SDValue(Node, 0).use_empty();
6719 bool NeedMask = !SDValue(Node, 1).use_empty();
6720 // We can't fold a load if we are going to make two instructions.
6721 bool MayFoldLoad = !NeedIndex || !NeedMask;
6722
6723 MachineSDNode *CNode;
6724 if (NeedMask) {
6725 unsigned ROpc =
6726 Subtarget->hasAVX() ? X86::VPCMPESTRMrri : X86::PCMPESTRMrri;
6727 unsigned MOpc =
6728 Subtarget->hasAVX() ? X86::VPCMPESTRMrmi : X86::PCMPESTRMrmi;
6729 CNode =
6730 emitPCMPESTR(ROpc, MOpc, MayFoldLoad, dl, MVT::v16i8, Node, InGlue);
6731 ReplaceUses(SDValue(Node, 1), SDValue(CNode, 0));
6732 }
6733 if (NeedIndex || !NeedMask) {
6734 unsigned ROpc =
6735 Subtarget->hasAVX() ? X86::VPCMPESTRIrri : X86::PCMPESTRIrri;
6736 unsigned MOpc =
6737 Subtarget->hasAVX() ? X86::VPCMPESTRIrmi : X86::PCMPESTRIrmi;
6738 CNode = emitPCMPESTR(ROpc, MOpc, MayFoldLoad, dl, MVT::i32, Node, InGlue);
6739 ReplaceUses(SDValue(Node, 0), SDValue(CNode, 0));
6740 }
6741 // Connect the flag usage to the last instruction created.
6742 ReplaceUses(SDValue(Node, 2), SDValue(CNode, 1));
6743 CurDAG->RemoveDeadNode(Node);
6744 return;
6745 }
6746
6747 case ISD::SETCC: {
6748 if (NVT.isVector() && tryVPTESTM(Node, SDValue(Node, 0), SDValue()))
6749 return;
6750
6751 break;
6752 }
6753
6754 case ISD::STORE:
6755 if (foldLoadStoreIntoMemOperand(Node))
6756 return;
6757 break;
6758
6759 case X86ISD::SETCC_CARRY: {
6760 MVT VT = Node->getSimpleValueType(0);
6761 SDValue Result;
6762 if (Subtarget->hasSBBDepBreaking()) {
6763 // We have to do this manually because tblgen will put the eflags copy in
6764 // the wrong place if we use an extract_subreg in the pattern.
6765 // Copy flags to the EFLAGS register and glue it to next node.
6766 SDValue EFLAGS =
6767 CurDAG->getCopyToReg(CurDAG->getEntryNode(), dl, X86::EFLAGS,
6768 Node->getOperand(1), SDValue());
6769
6770 // Create a 64-bit instruction if the result is 64-bits otherwise use the
6771 // 32-bit version.
6772 unsigned Opc = VT == MVT::i64 ? X86::SETB_C64r : X86::SETB_C32r;
6773 MVT SetVT = VT == MVT::i64 ? MVT::i64 : MVT::i32;
6774 Result = SDValue(
6775 CurDAG->getMachineNode(Opc, dl, SetVT, EFLAGS, EFLAGS.getValue(1)),
6776 0);
6777 } else {
6778 // The target does not recognize sbb with the same reg operand as a
6779 // no-source idiom, so we explicitly zero the input values.
6780 Result = getSBBZero(Node);
6781 }
6782
6783 // For less than 32-bits we need to extract from the 32-bit node.
6784 if (VT == MVT::i8 || VT == MVT::i16) {
6785 int SubIndex = VT == MVT::i16 ? X86::sub_16bit : X86::sub_8bit;
6786 Result = CurDAG->getTargetExtractSubreg(SubIndex, dl, VT, Result);
6787 }
6788
6789 ReplaceUses(SDValue(Node, 0), Result);
6790 CurDAG->RemoveDeadNode(Node);
6791 return;
6792 }
6793 case X86ISD::SBB: {
6794 if (isNullConstant(Node->getOperand(0)) &&
6795 isNullConstant(Node->getOperand(1))) {
6796 SDValue Result = getSBBZero(Node);
6797
6798 // Replace the flag use.
6799 ReplaceUses(SDValue(Node, 1), Result.getValue(1));
6800
6801 // Replace the result use.
6802 if (!SDValue(Node, 0).use_empty()) {
6803 // For less than 32-bits we need to extract from the 32-bit node.
6804 MVT VT = Node->getSimpleValueType(0);
6805 if (VT == MVT::i8 || VT == MVT::i16) {
6806 int SubIndex = VT == MVT::i16 ? X86::sub_16bit : X86::sub_8bit;
6807 Result = CurDAG->getTargetExtractSubreg(SubIndex, dl, VT, Result);
6808 }
6809 ReplaceUses(SDValue(Node, 0), Result);
6810 }
6811
6812 CurDAG->RemoveDeadNode(Node);
6813 return;
6814 }
6815 break;
6816 }
6817 case X86ISD::MGATHER: {
6818 auto *Mgt = cast<X86MaskedGatherSDNode>(Node);
6819 SDValue IndexOp = Mgt->getIndex();
6820 SDValue Mask = Mgt->getMask();
6821 MVT IndexVT = IndexOp.getSimpleValueType();
6822 MVT ValueVT = Node->getSimpleValueType(0);
6823 MVT MaskVT = Mask.getSimpleValueType();
6824
6825 // This is just to prevent crashes if the nodes are malformed somehow. We're
6826 // otherwise only doing loose type checking in here based on type what
6827 // a type constraint would say just like table based isel.
6828 if (!ValueVT.isVector() || !MaskVT.isVector())
6829 break;
6830
6831 unsigned NumElts = ValueVT.getVectorNumElements();
6832 MVT ValueSVT = ValueVT.getVectorElementType();
6833
6834 bool IsFP = ValueSVT.isFloatingPoint();
6835 unsigned EltSize = ValueSVT.getSizeInBits();
6836
6837 unsigned Opc = 0;
6838 bool AVX512Gather = MaskVT.getVectorElementType() == MVT::i1;
6839 if (AVX512Gather) {
6840 if (IndexVT == MVT::v4i32 && NumElts == 4 && EltSize == 32)
6841 Opc = IsFP ? X86::VGATHERDPSZ128rm : X86::VPGATHERDDZ128rm;
6842 else if (IndexVT == MVT::v8i32 && NumElts == 8 && EltSize == 32)
6843 Opc = IsFP ? X86::VGATHERDPSZ256rm : X86::VPGATHERDDZ256rm;
6844 else if (IndexVT == MVT::v16i32 && NumElts == 16 && EltSize == 32)
6845 Opc = IsFP ? X86::VGATHERDPSZrm : X86::VPGATHERDDZrm;
6846 else if (IndexVT == MVT::v4i32 && NumElts == 2 && EltSize == 64)
6847 Opc = IsFP ? X86::VGATHERDPDZ128rm : X86::VPGATHERDQZ128rm;
6848 else if (IndexVT == MVT::v4i32 && NumElts == 4 && EltSize == 64)
6849 Opc = IsFP ? X86::VGATHERDPDZ256rm : X86::VPGATHERDQZ256rm;
6850 else if (IndexVT == MVT::v8i32 && NumElts == 8 && EltSize == 64)
6851 Opc = IsFP ? X86::VGATHERDPDZrm : X86::VPGATHERDQZrm;
6852 else if (IndexVT == MVT::v2i64 && NumElts == 4 && EltSize == 32)
6853 Opc = IsFP ? X86::VGATHERQPSZ128rm : X86::VPGATHERQDZ128rm;
6854 else if (IndexVT == MVT::v4i64 && NumElts == 4 && EltSize == 32)
6855 Opc = IsFP ? X86::VGATHERQPSZ256rm : X86::VPGATHERQDZ256rm;
6856 else if (IndexVT == MVT::v8i64 && NumElts == 8 && EltSize == 32)
6857 Opc = IsFP ? X86::VGATHERQPSZrm : X86::VPGATHERQDZrm;
6858 else if (IndexVT == MVT::v2i64 && NumElts == 2 && EltSize == 64)
6859 Opc = IsFP ? X86::VGATHERQPDZ128rm : X86::VPGATHERQQZ128rm;
6860 else if (IndexVT == MVT::v4i64 && NumElts == 4 && EltSize == 64)
6861 Opc = IsFP ? X86::VGATHERQPDZ256rm : X86::VPGATHERQQZ256rm;
6862 else if (IndexVT == MVT::v8i64 && NumElts == 8 && EltSize == 64)
6863 Opc = IsFP ? X86::VGATHERQPDZrm : X86::VPGATHERQQZrm;
6864 } else {
6865 assert(EVT(MaskVT) == EVT(ValueVT).changeVectorElementTypeToInteger() &&
6866 "Unexpected mask VT!");
6867 if (IndexVT == MVT::v4i32 && NumElts == 4 && EltSize == 32)
6868 Opc = IsFP ? X86::VGATHERDPSrm : X86::VPGATHERDDrm;
6869 else if (IndexVT == MVT::v8i32 && NumElts == 8 && EltSize == 32)
6870 Opc = IsFP ? X86::VGATHERDPSYrm : X86::VPGATHERDDYrm;
6871 else if (IndexVT == MVT::v4i32 && NumElts == 2 && EltSize == 64)
6872 Opc = IsFP ? X86::VGATHERDPDrm : X86::VPGATHERDQrm;
6873 else if (IndexVT == MVT::v4i32 && NumElts == 4 && EltSize == 64)
6874 Opc = IsFP ? X86::VGATHERDPDYrm : X86::VPGATHERDQYrm;
6875 else if (IndexVT == MVT::v2i64 && NumElts == 4 && EltSize == 32)
6876 Opc = IsFP ? X86::VGATHERQPSrm : X86::VPGATHERQDrm;
6877 else if (IndexVT == MVT::v4i64 && NumElts == 4 && EltSize == 32)
6878 Opc = IsFP ? X86::VGATHERQPSYrm : X86::VPGATHERQDYrm;
6879 else if (IndexVT == MVT::v2i64 && NumElts == 2 && EltSize == 64)
6880 Opc = IsFP ? X86::VGATHERQPDrm : X86::VPGATHERQQrm;
6881 else if (IndexVT == MVT::v4i64 && NumElts == 4 && EltSize == 64)
6882 Opc = IsFP ? X86::VGATHERQPDYrm : X86::VPGATHERQQYrm;
6883 }
6884
6885 if (!Opc)
6886 break;
6887
6888 SDValue Base, Scale, Index, Disp, Segment;
6889 if (!selectVectorAddr(Mgt, Mgt->getBasePtr(), IndexOp, Mgt->getScale(),
6890 Base, Scale, Index, Disp, Segment))
6891 break;
6892
6893 SDValue PassThru = Mgt->getPassThru();
6894 SDValue Chain = Mgt->getChain();
6895 // Gather instructions have a mask output not in the ISD node.
6896 SDVTList VTs = CurDAG->getVTList(ValueVT, MaskVT, MVT::Other);
6897
6898 MachineSDNode *NewNode;
6899 if (AVX512Gather) {
6900 SDValue Ops[] = {PassThru, Mask, Base, Scale,
6901 Index, Disp, Segment, Chain};
6902 NewNode = CurDAG->getMachineNode(Opc, SDLoc(dl), VTs, Ops);
6903 } else {
6904 SDValue Ops[] = {PassThru, Base, Scale, Index,
6905 Disp, Segment, Mask, Chain};
6906 NewNode = CurDAG->getMachineNode(Opc, SDLoc(dl), VTs, Ops);
6907 }
6908 CurDAG->setNodeMemRefs(NewNode, {Mgt->getMemOperand()});
6909 ReplaceUses(SDValue(Node, 0), SDValue(NewNode, 0));
6910 ReplaceUses(SDValue(Node, 1), SDValue(NewNode, 2));
6911 CurDAG->RemoveDeadNode(Node);
6912 return;
6913 }
6914 case X86ISD::MSCATTER: {
6915 auto *Sc = cast<X86MaskedScatterSDNode>(Node);
6916 SDValue Value = Sc->getValue();
6917 SDValue IndexOp = Sc->getIndex();
6918 MVT IndexVT = IndexOp.getSimpleValueType();
6919 MVT ValueVT = Value.getSimpleValueType();
6920
6921 // This is just to prevent crashes if the nodes are malformed somehow. We're
6922 // otherwise only doing loose type checking in here based on type what
6923 // a type constraint would say just like table based isel.
6924 if (!ValueVT.isVector())
6925 break;
6926
6927 unsigned NumElts = ValueVT.getVectorNumElements();
6928 MVT ValueSVT = ValueVT.getVectorElementType();
6929
6930 bool IsFP = ValueSVT.isFloatingPoint();
6931 unsigned EltSize = ValueSVT.getSizeInBits();
6932
6933 unsigned Opc;
6934 if (IndexVT == MVT::v4i32 && NumElts == 4 && EltSize == 32)
6935 Opc = IsFP ? X86::VSCATTERDPSZ128mr : X86::VPSCATTERDDZ128mr;
6936 else if (IndexVT == MVT::v8i32 && NumElts == 8 && EltSize == 32)
6937 Opc = IsFP ? X86::VSCATTERDPSZ256mr : X86::VPSCATTERDDZ256mr;
6938 else if (IndexVT == MVT::v16i32 && NumElts == 16 && EltSize == 32)
6939 Opc = IsFP ? X86::VSCATTERDPSZmr : X86::VPSCATTERDDZmr;
6940 else if (IndexVT == MVT::v4i32 && NumElts == 2 && EltSize == 64)
6941 Opc = IsFP ? X86::VSCATTERDPDZ128mr : X86::VPSCATTERDQZ128mr;
6942 else if (IndexVT == MVT::v4i32 && NumElts == 4 && EltSize == 64)
6943 Opc = IsFP ? X86::VSCATTERDPDZ256mr : X86::VPSCATTERDQZ256mr;
6944 else if (IndexVT == MVT::v8i32 && NumElts == 8 && EltSize == 64)
6945 Opc = IsFP ? X86::VSCATTERDPDZmr : X86::VPSCATTERDQZmr;
6946 else if (IndexVT == MVT::v2i64 && NumElts == 4 && EltSize == 32)
6947 Opc = IsFP ? X86::VSCATTERQPSZ128mr : X86::VPSCATTERQDZ128mr;
6948 else if (IndexVT == MVT::v4i64 && NumElts == 4 && EltSize == 32)
6949 Opc = IsFP ? X86::VSCATTERQPSZ256mr : X86::VPSCATTERQDZ256mr;
6950 else if (IndexVT == MVT::v8i64 && NumElts == 8 && EltSize == 32)
6951 Opc = IsFP ? X86::VSCATTERQPSZmr : X86::VPSCATTERQDZmr;
6952 else if (IndexVT == MVT::v2i64 && NumElts == 2 && EltSize == 64)
6953 Opc = IsFP ? X86::VSCATTERQPDZ128mr : X86::VPSCATTERQQZ128mr;
6954 else if (IndexVT == MVT::v4i64 && NumElts == 4 && EltSize == 64)
6955 Opc = IsFP ? X86::VSCATTERQPDZ256mr : X86::VPSCATTERQQZ256mr;
6956 else if (IndexVT == MVT::v8i64 && NumElts == 8 && EltSize == 64)
6957 Opc = IsFP ? X86::VSCATTERQPDZmr : X86::VPSCATTERQQZmr;
6958 else
6959 break;
6960
6961 SDValue Base, Scale, Index, Disp, Segment;
6962 if (!selectVectorAddr(Sc, Sc->getBasePtr(), IndexOp, Sc->getScale(),
6963 Base, Scale, Index, Disp, Segment))
6964 break;
6965
6966 SDValue Mask = Sc->getMask();
6967 SDValue Chain = Sc->getChain();
6968 // Scatter instructions have a mask output not in the ISD node.
6969 SDVTList VTs = CurDAG->getVTList(Mask.getValueType(), MVT::Other);
6970 SDValue Ops[] = {Base, Scale, Index, Disp, Segment, Mask, Value, Chain};
6971
6972 MachineSDNode *NewNode = CurDAG->getMachineNode(Opc, SDLoc(dl), VTs, Ops);
6973 CurDAG->setNodeMemRefs(NewNode, {Sc->getMemOperand()});
6974 ReplaceUses(SDValue(Node, 0), SDValue(NewNode, 1));
6975 CurDAG->RemoveDeadNode(Node);
6976 return;
6977 }
6979 auto *MFI = CurDAG->getMachineFunction().getInfo<X86MachineFunctionInfo>();
6980 auto CallId = MFI->getPreallocatedIdForCallSite(
6981 cast<SrcValueSDNode>(Node->getOperand(1))->getValue());
6982 SDValue Chain = Node->getOperand(0);
6983 SDValue CallIdValue = CurDAG->getTargetConstant(CallId, dl, MVT::i32);
6984 MachineSDNode *New = CurDAG->getMachineNode(
6985 TargetOpcode::PREALLOCATED_SETUP, dl, MVT::Other, CallIdValue, Chain);
6986 ReplaceUses(SDValue(Node, 0), SDValue(New, 0)); // Chain
6987 CurDAG->RemoveDeadNode(Node);
6988 return;
6989 }
6990 case ISD::PREALLOCATED_ARG: {
6991 auto *MFI = CurDAG->getMachineFunction().getInfo<X86MachineFunctionInfo>();
6992 auto CallId = MFI->getPreallocatedIdForCallSite(
6993 cast<SrcValueSDNode>(Node->getOperand(1))->getValue());
6994 SDValue Chain = Node->getOperand(0);
6995 SDValue CallIdValue = CurDAG->getTargetConstant(CallId, dl, MVT::i32);
6996 SDValue ArgIndex = Node->getOperand(2);
6997 SDValue Ops[3];
6998 Ops[0] = CallIdValue;
6999 Ops[1] = ArgIndex;
7000 Ops[2] = Chain;
7001 MachineSDNode *New = CurDAG->getMachineNode(
7002 TargetOpcode::PREALLOCATED_ARG, dl,
7003 CurDAG->getVTList(TLI->getPointerTy(CurDAG->getDataLayout()),
7004 MVT::Other),
7005 Ops);
7006 ReplaceUses(SDValue(Node, 0), SDValue(New, 0)); // Arg pointer
7007 ReplaceUses(SDValue(Node, 1), SDValue(New, 1)); // Chain
7008 CurDAG->RemoveDeadNode(Node);
7009 return;
7010 }
7015 if (!Subtarget->hasWIDEKL())
7016 break;
7017
7018 unsigned Opcode;
7019 switch (Node->getOpcode()) {
7020 default:
7021 llvm_unreachable("Unexpected opcode!");
7023 Opcode = X86::AESENCWIDE128KL;
7024 break;
7026 Opcode = X86::AESDECWIDE128KL;
7027 break;
7029 Opcode = X86::AESENCWIDE256KL;
7030 break;
7032 Opcode = X86::AESDECWIDE256KL;
7033 break;
7034 }
7035
7036 SDValue Chain = Node->getOperand(0);
7037 SDValue Addr = Node->getOperand(1);
7038
7039 SDValue Base, Scale, Index, Disp, Segment;
7040 if (!selectAddr(Node, Addr, Base, Scale, Index, Disp, Segment))
7041 break;
7042
7043 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM0, Node->getOperand(2),
7044 SDValue());
7045 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM1, Node->getOperand(3),
7046 Chain.getValue(1));
7047 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM2, Node->getOperand(4),
7048 Chain.getValue(1));
7049 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM3, Node->getOperand(5),
7050 Chain.getValue(1));
7051 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM4, Node->getOperand(6),
7052 Chain.getValue(1));
7053 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM5, Node->getOperand(7),
7054 Chain.getValue(1));
7055 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM6, Node->getOperand(8),
7056 Chain.getValue(1));
7057 Chain = CurDAG->getCopyToReg(Chain, dl, X86::XMM7, Node->getOperand(9),
7058 Chain.getValue(1));
7059
7060 MachineSDNode *Res = CurDAG->getMachineNode(
7061 Opcode, dl, Node->getVTList(),
7062 {Base, Scale, Index, Disp, Segment, Chain, Chain.getValue(1)});
7063 CurDAG->setNodeMemRefs(Res, cast<MemSDNode>(Node)->getMemOperand());
7064 ReplaceNode(Node, Res);
7065 return;
7066 }
7068 SDValue Chain = Node->getOperand(0);
7069 Register Reg = cast<RegisterSDNode>(Node->getOperand(1))->getReg();
7070 SDValue Glue;
7071 if (Node->getNumValues() == 3)
7072 Glue = Node->getOperand(2);
7073 SDValue Copy =
7074 CurDAG->getCopyFromReg(Chain, dl, Reg, Node->getValueType(0), Glue);
7075 ReplaceNode(Node, Copy.getNode());
7076 return;
7077 }
7078 }
7079
7080 SelectCode(Node);
7081}
7082
7083bool X86DAGToDAGISel::SelectInlineAsmMemoryOperand(
7084 const SDValue &Op, InlineAsm::ConstraintCode ConstraintID,
7085 std::vector<SDValue> &OutOps) {
7086 SDValue Op0, Op1, Op2, Op3, Op4;
7087 switch (ConstraintID) {
7088 default:
7089 llvm_unreachable("Unexpected asm memory constraint");
7090 case InlineAsm::ConstraintCode::o: // offsetable ??
7091 case InlineAsm::ConstraintCode::v: // not offsetable ??
7092 case InlineAsm::ConstraintCode::m: // memory
7093 case InlineAsm::ConstraintCode::X:
7094 case InlineAsm::ConstraintCode::p: // address
7095 if (!selectAddr(nullptr, Op, Op0, Op1, Op2, Op3, Op4))
7096 return true;
7097 break;
7098 }
7099
7100 OutOps.push_back(Op0);
7101 OutOps.push_back(Op1);
7102 OutOps.push_back(Op2);
7103 OutOps.push_back(Op3);
7104 OutOps.push_back(Op4);
7105 return false;
7106}
7107
7110 std::make_unique<X86DAGToDAGISel>(TM, TM.getOptLevel())) {}
7111
7112/// This pass converts a legalized DAG into a X86-specific DAG,
7113/// ready for instruction scheduling.
7115 CodeGenOptLevel OptLevel) {
7116 return new X86DAGToDAGISelLegacy(TM, OptLevel);
7117}
static SDValue Widen(SelectionDAG *CurDAG, SDValue N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
aarch64 promote const
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis false
#define CASE(ATTRNAME, AANAME,...)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
dxil translate DXIL Translate Metadata
static bool isSigned(unsigned Opcode)
#define DEBUG_TYPE
const HexagonInstrInfo * TII
Module.h This file contains the declarations for the Module class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
const MCPhysReg ArgGPRs[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
static bool isUndef(const MachineInstr &MI)
Register Reg
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
#define P(N)
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
BaseType
A given derived pointer can have multiple base pointers through phi/selects.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
static bool isFusableLoadOpStorePattern(StoreSDNode *StoreNode, SDValue StoredVal, SelectionDAG *CurDAG, LoadSDNode *&LoadNode, SDValue &InputChain)
static void insertDAGNode(SelectionDAG *DAG, SDNode *Pos, SDValue N)
#define PASS_NAME
static bool isRIPRelative(const MCInst &MI, const MCInstrInfo &MCII)
Check if the instruction uses RIP relative addressing.
#define FROM_TO(FROM, TO)
#define GET_EGPR_IF_ENABLED(OPC)
static bool isLegalMaskCompare(SDNode *N, const X86Subtarget *Subtarget)
static bool foldMaskAndShiftToScale(SelectionDAG &DAG, SDValue N, uint64_t Mask, SDValue Shift, SDValue X, X86ISelAddressMode &AM)
static bool foldMaskAndShiftToExtract(SelectionDAG &DAG, SDValue N, uint64_t Mask, SDValue Shift, SDValue X, X86ISelAddressMode &AM)
static bool addrMayUseNonFixedFrameIndex(SDValue Addr, const MachineFrameInfo &MFI, unsigned Depth=0)
Return true if Addr may be matched with a non-fixed frame index as base.
static bool needBWI(MVT VT)
static unsigned getVPTESTMOpc(MVT TestVT, bool IsTestN, bool FoldedLoad, bool FoldedBCast, bool Masked)
#define GET_NDM_IF_ENABLED(OPC)
static bool foldMaskedShiftToBEXTR(SelectionDAG &DAG, SDValue N, uint64_t Mask, SDValue Shift, SDValue X, X86ISelAddressMode &AM, const X86Subtarget &Subtarget)
static bool mayUseCarryFlag(X86::CondCode CC)
static bool isEndbrImm(uint64_t Imm, unsigned BitWidth)
static void moveBelowOrigChain(SelectionDAG *CurDAG, SDValue Load, SDValue Call, SDValue OrigChain)
Replace the original chain operand of the call with load's chain operand and move load below the call...
#define GET_ND_IF_ENABLED(OPC)
#define VPTESTM_BROADCAST_CASES(SUFFIX)
static bool foldMaskedShiftToScaledMask(SelectionDAG &DAG, SDValue N, X86ISelAddressMode &AM)
#define VPTESTM_FULL_CASES(SUFFIX)
static bool isCalleeLoad(SDValue Callee, SDValue &Chain, bool HasCallSeq)
Return true if call address is a load and it can be moved below CALLSEQ_START and the chains leading ...
static bool isDispSafeForFrameIndexOrRegBase(int64_t Val)
static void orderRegForMul(SDValue &N0, SDValue &N1, const unsigned LoReg, const MachineRegisterInfo &MRI)
#define GET_ND_IF_ENABLED(OPC)
#define CASE_ND(OP)
Value * RHS
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
Definition APInt.h:367
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1618
unsigned getSignificantBits() const
Get the minimum bit size for this signed APInt.
Definition APInt.h:1551
bool isSubsetOf(const APInt &RHS) const
This operation checks that all bits set in this APInt are also set in RHS.
Definition APInt.h:1261
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
bool isOne() const
Determine if this is a value of 1.
Definition APInt.h:385
unsigned countr_one() const
Count the number of trailing one bits.
Definition APInt.h:1676
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:696
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
Module * getParent()
Get the module that this global value is contained inside of...
LLVM_ABI std::optional< ConstantRange > getAbsoluteSymbolRange() const
If this is an absolute symbol reference, returns the range of the symbol, otherwise returns std::null...
Definition Globals.cpp:534
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
unsigned getID() const
getID() - Return the register class ID number.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
Machine Value Type.
bool isVectorOf(MVT EltVT) const
Return true if this is a vector with matching element type.
bool is128BitVector() const
Return true if this is a 128-bit vector type.
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool is512BitVector() const
Return true if this is a 512-bit vector type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
bool is256BitVector() const
Return true if this is a 256-bit vector type.
bool isScalarInteger() const
Return true if this is an integer, not including vectors.
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
MVT getHalfNumVectorElementsVT() const
Return a VT for a vector type with the same element type but half the number of elements.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI MCRegister getLiveInPhysReg(Register VReg) const
getLiveInPhysReg - If VReg is a live-in virtual register, return the corresponding live-in physical r...
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const MachinePointerInfo & getPointerInfo() const
const SDValue & getChain() const
bool isNonTemporal() const
Metadata * getModuleFlag(StringRef Key) const
Return the corresponding value if Key appears in module flags, otherwise return null.
Definition Module.cpp:358
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
int getNodeId() const
Return the unique node id.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
SDNodeFlags getFlags() const
MVT getSimpleValueType(unsigned ResNo) const
Return the type of a specified result as a simple type.
static bool hasPredecessorHelper(const SDNode *N, SmallPtrSetImpl< const SDNode * > &Visited, SmallVectorImpl< const SDNode * > &Worklist, unsigned int MaxSteps=0, bool TopologicalPrune=false)
Returns true if N is a predecessor of any node in Worklist.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
bool use_empty() const
Return true if there are no uses of this node.
const SDValue & getOperand(unsigned Num) const
bool hasNUsesOfValue(unsigned NUses, unsigned Value) const
Return true if there are exactly NUSES uses of the indicated value.
iterator_range< user_iterator > users()
op_iterator op_end() const
op_iterator op_begin() const
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
bool isMachineOpcode() const
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
uint64_t getScalarValueSizeInBits() const
unsigned getResNo() const
get the index which selects a specific result in the SDNode
uint64_t getConstantOperandVal(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getMachineOpcode() const
unsigned getOpcode() const
unsigned getNumOperands() const
SelectionDAGISelPass(std::unique_ptr< SelectionDAGISel > Selector)
SelectionDAGISel - This is the common base class used for SelectionDAG-based pattern-matching instruc...
static int getUninvalidatedNodeId(SDNode *N)
virtual bool runOnMachineFunction(MachineFunction &mf)
static void InvalidateNodeId(SDNode *N)
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
static constexpr unsigned MaxRecursionDepth
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
LLVM_ABI void RemoveDeadNode(SDNode *N)
Remove the specified node from the system.
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
void RepositionNode(allnodes_iterator Position, SDNode *N)
Move node N in the AllNodes list to be immediately before the given iterator Position.
ilist< SDNode >::iterator allnodes_iterator
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
This class is used to represent ISD::STORE nodes.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
std::pair< SDValue, SDValue > LowerCallTo(CallLoweringInfo &CLI) const
This function lowers an abstract call to a function into an actual call.
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
X86ISelDAGToDAGPass(X86TargetMachine &TM)
size_t getPreallocatedIdForCallSite(const Value *CS)
bool isScalarFPTypeInSSEReg(EVT VT) const
Return true if the specified scalar FP type is computed in an SSE register, not on the X87 floating p...
self_iterator getIterator()
Definition ilist_node.h:123
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
bool isNON_EXTLoad(const SDNode *N)
Returns true if the specified node is a non-extending load.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ DELETED_NODE
DELETED_NODE - This is an illegal value that is used to catch errors.
Definition ISDOpcodes.h:47
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ PREALLOCATED_SETUP
PREALLOCATED_SETUP - This has 2 operands: an input chain and a SRCVALUE with the preallocated call Va...
@ TargetExternalSymbol
Definition ISDOpcodes.h:192
@ PREALLOCATED_ARG
PREALLOCATED_ARG - This has 3 operands: an input chain, a SRCVALUE with the preallocated call Value,...
@ BRIND
BRIND - Indirect branch.
@ AssertAlign
AssertAlign - These nodes record if a register contains a value that has a known alignment and the tr...
Definition ISDOpcodes.h:71
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ TargetGlobalAddress
TargetGlobalAddress - Like GlobalAddress, but the DAG does no folding or anything else with this node...
Definition ISDOpcodes.h:187
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ LOCAL_RECOVER
LOCAL_RECOVER - Represents the llvm.localrecover intrinsic.
Definition ISDOpcodes.h:137
@ ANY_EXTEND_VECTOR_INREG
ANY_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register any-extension of the low la...
Definition ISDOpcodes.h:917
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:331
@ STRICT_FROUNDEVEN
Definition ISDOpcodes.h:469
@ STRICT_FP_TO_UINT
Definition ISDOpcodes.h:483
@ STRICT_FP_ROUND
X = STRICT_FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision ...
Definition ISDOpcodes.h:505
@ STRICT_FP_TO_SINT
STRICT_FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:482
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ STRICT_FP_EXTEND
X = STRICT_FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:510
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ FREEZE
FREEZE - FREEZE(VAL) returns an arbitrary value if VAL is UNDEF (or is evaluated to UNDEF),...
Definition ISDOpcodes.h:243
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ ZERO_EXTEND_VECTOR_INREG
ZERO_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register zero-extension of the low ...
Definition ISDOpcodes.h:939
@ STRICT_FNEARBYINT
Definition ISDOpcodes.h:461
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ CALLSEQ_START
CALLSEQ_START/CALLSEQ_END - These operators mark the beginning and end of a call sequence,...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ TargetGlobalTLSAddress
Definition ISDOpcodes.h:188
LLVM_ABI bool isBuildVectorOfConstantSDNodes(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR node of all ConstantSDNode or undef.
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LLVM_ABI bool isBuildVectorAllOnes(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are ~0 or undef.
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
@ GlobalBaseReg
The result of the mflr at function entry, used for PIC code.
@ X86
Windows x64, Windows Itanium (IA-64)
Definition MCAsmInfo.h:53
@ MO_NO_FLAG
MO_NO_FLAG - No flag for the operand.
@ EVEX
EVEX - Specifies that this instruction use EVEX form which provides syntax support up to 32 512-bit r...
@ VEX
VEX - encoding using 0xC4/0xC5.
@ XOP
XOP - Opcode prefix used by XOP instructions.
int getMemoryOperandNo(uint64_t TSFlags)
@ GlobalBaseReg
On Darwin, this node represents the result of the popl at function entry, used for PIC code.
@ POP_FROM_X87_REG
The same as ISD::CopyFromReg except that this node makes it explicit that it may lower to an x87 FPU ...
@ AddrNumOperands
Definition X86BaseInfo.h:37
int getCondSrcNoFromDesc(const MCInstrDesc &MCID)
Return the source operand # for condition code by MCID.
bool mayFoldLoad(SDValue Op, const X86Subtarget &Subtarget, bool AssumeSingleUse=false, bool IgnoreAlignment=false)
Check if Op is a load operation that could be folded into some other x86 instruction as a memory oper...
bool isOffsetSuitableForCodeModel(int64_t Offset, CodeModel::Model M, bool hasSymbolicDisplacement)
Returns true of the given offset can be fit into displacement field of the instruction.
bool isConstantSplat(SDValue Op, APInt &SplatVal, bool AllowPartialUndefs)
If Op is a constant whose elements are all the same constant or undefined, return true and return the...
@ User
could "use" a pointer
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
constexpr uint16_t Magic
Definition SFrame.h:32
This is an optimization pass for GlobalISel generic memory operations.
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
@ Offset
Definition DWP.cpp:577
InstructionCost Cost
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
bool isa_and_nonnull(const Y &Val)
Definition Casting.h:676
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Definition bit.h:362
Op::Description Desc
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
unsigned M1(unsigned Val)
Definition VE.h:377
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isMask_64(uint64_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
Definition MathExtras.h:262
FunctionPass * createX86ISelDag(X86TargetMachine &TM, CodeGenOptLevel OptLevel)
This pass converts a legalized DAG into a X86-specific DAG, ready for instruction scheduling.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:227
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ And
Bitwise or logical AND of integers.
@ Add
Sum of integers.
DWARFExpression::Operation Op
unsigned M0(unsigned Val)
Definition VE.h:376
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
bool is256BitVector() const
Return true if this is a 256-bit vector type.
Definition ValueTypes.h:235
bool isConstant() const
Returns true if we know the value of all bits.
Definition KnownBits.h:54
Matching combinators.
LLVM_ABI unsigned getAddrSpace() const
Return the LLVM IR address space number that this pointer points into.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
bool hasNoUnsignedWrap() const