LLVM 24.0.0git
WebAssemblyISelLowering.cpp
Go to the documentation of this file.
1//=- WebAssemblyISelLowering.cpp - WebAssembly DAG Lowering Implementation -==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// This file implements the WebAssemblyTargetLowering class.
11///
12//===----------------------------------------------------------------------===//
13
32#include "llvm/IR/Function.h"
33#include "llvm/IR/Intrinsics.h"
34#include "llvm/IR/IntrinsicsWebAssembly.h"
39using namespace llvm;
40
41#define DEBUG_TYPE "wasm-lower"
42
44 const TargetMachine &TM, const WebAssemblySubtarget &STI)
45 : TargetLowering(TM, STI), Subtarget(&STI) {
46 auto MVTPtr = Subtarget->hasAddr64() ? MVT::i64 : MVT::i32;
47
48 // Set the load count for memcmp expand optimization
51
52 // Booleans always contain 0 or 1.
54 // Except in SIMD vectors
56 // We don't know the microarchitecture here, so just reduce register pressure.
58 // Tell ISel that we have a stack pointer.
60 Subtarget->hasAddr64() ? WebAssembly::SP64 : WebAssembly::SP32);
61 // Set up the register classes.
62 addRegisterClass(MVT::i32, &WebAssembly::I32RegClass);
63 addRegisterClass(MVT::i64, &WebAssembly::I64RegClass);
64 addRegisterClass(MVT::f32, &WebAssembly::F32RegClass);
65 addRegisterClass(MVT::f64, &WebAssembly::F64RegClass);
66 if (Subtarget->hasSIMD128()) {
67 addRegisterClass(MVT::v16i8, &WebAssembly::V128RegClass);
68 addRegisterClass(MVT::v8i16, &WebAssembly::V128RegClass);
69 addRegisterClass(MVT::v4i32, &WebAssembly::V128RegClass);
70 addRegisterClass(MVT::v4f32, &WebAssembly::V128RegClass);
71 addRegisterClass(MVT::v2i64, &WebAssembly::V128RegClass);
72 addRegisterClass(MVT::v2f64, &WebAssembly::V128RegClass);
73 }
74 if (Subtarget->hasFP16()) {
75 addRegisterClass(MVT::v8f16, &WebAssembly::V128RegClass);
76 }
77 if (Subtarget->hasReferenceTypes()) {
78 addRegisterClass(MVT::externref, &WebAssembly::EXTERNREFRegClass);
79 addRegisterClass(MVT::funcref, &WebAssembly::FUNCREFRegClass);
80 if (Subtarget->hasExceptionHandling()) {
81 addRegisterClass(MVT::exnref, &WebAssembly::EXNREFRegClass);
82 }
83 }
84 // Compute derived properties from the register classes.
85 computeRegisterProperties(Subtarget->getRegisterInfo());
86
87 // Transform loads and stores to pointers in address space 1 to loads and
88 // stores to WebAssembly global variables, outside linear memory.
89 for (auto T : {MVT::i32, MVT::i64, MVT::f32, MVT::f64}) {
92 }
93 if (Subtarget->hasSIMD128()) {
94 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
95 MVT::v2f64}) {
98 }
99 }
100 if (Subtarget->hasFP16()) {
101 setOperationAction(ISD::LOAD, MVT::v8f16, Custom);
103 }
104 if (Subtarget->hasReferenceTypes()) {
105 // We need custom load and store lowering for both externref, funcref and
106 // Other. The MVT::Other here represents tables of reference types.
107 for (auto T : {MVT::externref, MVT::funcref, MVT::Other}) {
110 }
111 }
112
120
121 // Take the default expansion for va_arg, va_copy, and va_end. There is no
122 // default action for va_start, so we do that custom.
127
128 for (auto T : {MVT::f32, MVT::f64, MVT::v4f32, MVT::v2f64, MVT::v8f16}) {
129 if (!Subtarget->hasFP16() && T == MVT::v8f16) {
130 continue;
131 }
132 // Don't expand the floating-point types to constant pools.
134 // Expand floating-point comparisons.
135 for (auto CC : {ISD::SETO, ISD::SETUO, ISD::SETUEQ, ISD::SETONE,
138 // Expand floating-point library function operators.
141 // Expand vector FREM, but use a libcall rather than an expansion for scalar
142 if (MVT(T).isVector())
144 else
146 // Note supported floating-point library function operators that otherwise
147 // default to expand.
151 // Support minimum and maximum, which otherwise default to expand.
154 if (Subtarget->hasSIMD128() && MVT(T).isVector()) {
157 }
158 // When experimental v8f16 support is enabled these instructions don't need
159 // to be expanded.
160 if (T != MVT::v8f16) {
163 }
164 if (Subtarget->hasFP16() && T == MVT::f32) {
166 setTruncStoreAction(T, MVT::f16, Legal);
167 } else {
169 setTruncStoreAction(T, MVT::f16, Expand);
170 }
171 }
172
173 // Expand unavailable integer operations.
174 for (auto Op :
178 for (auto T : {MVT::i32, MVT::i64})
180 if (Subtarget->hasSIMD128())
181 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
183 }
184
185 if (Subtarget->hasWideArithmetic()) {
191 }
192
193 if (Subtarget->hasNontrappingFPToInt())
195 for (auto T : {MVT::i32, MVT::i64})
197
198 if (Subtarget->hasRelaxedSIMD()) {
201 {MVT::v4f32, MVT::v2f64}, Custom);
202 }
203
204 // Combine expands these operations, because wasi-libc and emscripten do not
205 // yet have the dedicated libcalls.
208
209 // SIMD-specific configuration
210 if (Subtarget->hasSIMD128()) {
211
213
214 // Combine wide-vector muls, with extend inputs, to extmul_half.
217
218 // Combine vector mask reductions into alltrue/anytrue
220
221 // Convert vector to integer bitcasts to bitmask
223
224 // Hoist bitcasts out of shuffles
226
227 // Combine extends of extract_subvectors into widening ops
229
230 // Combine int_to_fp or fp_extend of extract_vectors and vice versa into
231 // conversions ops
234
235 // Combine fp_to_{s,u}int_sat or fp_round of concat_vectors or vice versa
236 // into conversion ops
240
242
243 // Support saturating add/sub for i8x16 and i16x8
245 for (auto T : {MVT::v16i8, MVT::v8i16})
247
248 // Support integer abs
249 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
251
252 // Custom lower BUILD_VECTORs to minimize number of replace_lanes
253 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
254 MVT::v2f64})
256
257 if (Subtarget->hasFP16()) {
261 }
262
263 // We have custom shuffle lowering to expose the shuffle mask
264 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
265 MVT::v2f64})
267
268 if (Subtarget->hasFP16())
270
271 // Support splatting
272 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
273 MVT::v2f64})
275
276 setOperationAction(ISD::AVGCEILU, {MVT::v8i16, MVT::v16i8}, Legal);
277
278 // Custom lowering since wasm shifts must have a scalar shift amount
279 for (auto Op : {ISD::SHL, ISD::SRA, ISD::SRL})
280 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
282
283 // Custom lower lane accesses to expand out variable indices
285 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
286 MVT::v2f64})
288
289 // There is no i8x16.mul instruction
290 setOperationAction(ISD::MUL, MVT::v16i8, Expand);
291
292 // Expand integer operations supported for scalars but not SIMD
293 for (auto Op :
295 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64})
297
298 // But we do have integer min and max operations
299 for (auto Op : {ISD::SMIN, ISD::SMAX, ISD::UMIN, ISD::UMAX})
300 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32})
302
303 // And we have popcnt for i8x16. It can be used to expand ctlz/cttz.
304 setOperationAction(ISD::CTPOP, MVT::v16i8, Legal);
305 setOperationAction(ISD::CTLZ, MVT::v16i8, Expand);
306 setOperationAction(ISD::CTTZ, MVT::v16i8, Expand);
307
308 // Custom lower bit counting operations for other types to scalarize them.
309 for (auto Op : {ISD::CTLZ, ISD::CTTZ, ISD::CTPOP})
310 for (auto T : {MVT::v8i16, MVT::v4i32, MVT::v2i64})
312
313 // Expand float operations supported for scalars but not SIMD
316 for (auto T : {MVT::v4f32, MVT::v2f64})
318
319 // Unsigned comparison operations are unavailable for i64x2 vectors.
321 setCondCodeAction(CC, MVT::v2i64, Custom);
322
323 // 64x2 conversions are not in the spec
324 for (auto Op :
326 for (auto T : {MVT::v2i64, MVT::v2f64})
328
329 // But saturating fp_to_int conversions are
331 setOperationAction(Op, MVT::v4i32, Custom);
332 if (Subtarget->hasFP16()) {
333 setOperationAction(Op, MVT::v8i16, Custom);
334 }
335 }
336
337 // Support vector extending
342 }
343
344 if (Subtarget->hasFP16()) {
345 setOperationAction(ISD::FMA, MVT::v8f16, Legal);
346 }
347
348 if (Subtarget->hasRelaxedSIMD()) {
351 }
352
353 // Partial MLA reductions.
355 setPartialReduceMLAAction(Op, MVT::v4i32, MVT::v16i8, Legal);
356 setPartialReduceMLAAction(Op, MVT::v4i32, MVT::v8i16, Legal);
357 }
358 }
359
360 // As a special case, these operators use the type to mean the type to
361 // sign-extend from.
363 if (!Subtarget->hasSignExt()) {
364 // Sign extends are legal only when extending a vector extract
365 auto Action = Subtarget->hasSIMD128() ? Custom : Expand;
366 for (auto T : {MVT::i8, MVT::i16, MVT::i32})
368 }
371
372 // Dynamic stack allocation: use the default expansion.
376
380
381 // Expand these forms; we pattern-match the forms that we can handle in isel.
382 for (auto T : {MVT::i32, MVT::i64, MVT::f32, MVT::f64})
383 for (auto Op : {ISD::BR_CC, ISD::SELECT_CC})
385
386 if (Subtarget->hasReferenceTypes())
387 for (auto Op : {ISD::BR_CC, ISD::SELECT_CC})
388 for (auto T : {MVT::externref, MVT::funcref})
390
391 // There is no vector conditional select instruction
392 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v4f32, MVT::v2i64,
393 MVT::v2f64, MVT::v8f16})
395
396 // We have custom switch handling.
398
399 // WebAssembly doesn't have:
400 // - Floating-point extending loads.
401 // - Floating-point truncating stores.
402 // - i1 extending loads.
403 // - truncating SIMD stores and most extending loads
404 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
405 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
406 for (auto T : MVT::integer_valuetypes())
407 for (auto Ext : {ISD::EXTLOAD, ISD::ZEXTLOAD, ISD::SEXTLOAD})
408 setLoadExtAction(Ext, T, MVT::i1, Promote);
409 if (Subtarget->hasSIMD128()) {
410 for (auto T : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64, MVT::v4f32,
411 MVT::v2f64}) {
412 for (auto MemT : MVT::fixedlen_vector_valuetypes()) {
413 if (MVT(T) != MemT) {
415 for (auto Ext : {ISD::EXTLOAD, ISD::ZEXTLOAD, ISD::SEXTLOAD})
416 setLoadExtAction(Ext, T, MemT, Expand);
417 }
418 }
419 }
420 // But some vector extending loads are legal
421 for (auto Ext : {ISD::EXTLOAD, ISD::SEXTLOAD, ISD::ZEXTLOAD}) {
422 setLoadExtAction(Ext, MVT::v8i16, MVT::v8i8, Legal);
423 setLoadExtAction(Ext, MVT::v4i32, MVT::v4i16, Legal);
424 setLoadExtAction(Ext, MVT::v2i64, MVT::v2i32, Legal);
425 }
426 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f32, Legal);
427 }
428
429 // Don't do anything clever with build_pairs
431
432 // Trap lowers to wasm unreachable
433 setOperationAction(ISD::TRAP, MVT::Other, Legal);
435
436 // Exception handling intrinsics
440
442
443 // Always convert switches to br_tables unless there is only one case, which
444 // is equivalent to a simple branch. This reduces code size for wasm, and we
445 // defer possible jump table optimizations to the VM.
447}
448
450WebAssemblyTargetLowering::shouldExpandAtomicRMWInIR(
451 const AtomicRMWInst *AI) const {
452 // We have wasm instructions for these
453 switch (AI->getOperation()) {
461 default:
462 break;
463 }
465}
466
467bool WebAssemblyTargetLowering::shouldScalarizeBinop(SDValue VecOp) const {
468 // Implementation copied from X86TargetLowering.
469 unsigned Opc = VecOp.getOpcode();
470
471 // Assume target opcodes can't be scalarized.
472 // TODO - do we have any exceptions?
474 return false;
475
476 // If the vector op is not supported, try to convert to scalar.
477 EVT VecVT = VecOp.getValueType();
479 return true;
480
481 // If the vector op is supported, but the scalar op is not, the transform may
482 // not be worthwhile.
483 EVT ScalarVT = VecVT.getScalarType();
484 return isOperationLegalOrCustomOrPromote(Opc, ScalarVT);
485}
486
487FastISel *WebAssemblyTargetLowering::createFastISel(
488 FunctionLoweringInfo &FuncInfo, const TargetLibraryInfo *LibInfo,
489 const LibcallLoweringInfo *LibcallLowering) const {
490 return WebAssembly::createFastISel(FuncInfo, LibInfo, LibcallLowering);
491}
492
493MVT WebAssemblyTargetLowering::getScalarShiftAmountTy(const DataLayout & /*DL*/,
494 EVT VT) const {
495 unsigned BitWidth = NextPowerOf2(VT.getSizeInBits() - 1);
496 if (BitWidth > 1 && BitWidth < 8)
497 BitWidth = 8;
498
499 if (BitWidth > 64) {
500 // The shift will be lowered to a libcall, and compiler-rt libcalls expect
501 // the count to be an i32.
502 BitWidth = 32;
504 "32-bit shift counts ought to be enough for anyone");
505 }
506
509 "Unable to represent scalar shift amount type");
510 return Result;
511}
512
513// Lower an fp-to-int conversion operator from the LLVM opcode, which has an
514// undefined result on invalid/overflow, to the WebAssembly opcode, which
515// traps on invalid/overflow.
518 const TargetInstrInfo &TII,
519 bool IsUnsigned, bool Int64,
520 bool Float64, unsigned LoweredOpcode) {
522
523 Register OutReg = MI.getOperand(0).getReg();
524 Register InReg = MI.getOperand(1).getReg();
525
526 unsigned Abs = Float64 ? WebAssembly::ABS_F64 : WebAssembly::ABS_F32;
527 unsigned FConst = Float64 ? WebAssembly::CONST_F64 : WebAssembly::CONST_F32;
528 unsigned LT = Float64 ? WebAssembly::LT_F64 : WebAssembly::LT_F32;
529 unsigned GE = Float64 ? WebAssembly::GE_F64 : WebAssembly::GE_F32;
530 unsigned IConst = Int64 ? WebAssembly::CONST_I64 : WebAssembly::CONST_I32;
531 unsigned Eqz = WebAssembly::EQZ_I32;
532 unsigned And = WebAssembly::AND_I32;
533 int64_t Limit = Int64 ? INT64_MIN : INT32_MIN;
534 int64_t Substitute = IsUnsigned ? 0 : Limit;
535 double CmpVal = IsUnsigned ? -(double)Limit * 2.0 : -(double)Limit;
536 auto &Context = BB->getParent()->getFunction().getContext();
537 Type *Ty = Float64 ? Type::getDoubleTy(Context) : Type::getFloatTy(Context);
538
539 const BasicBlock *LLVMBB = BB->getBasicBlock();
540 MachineFunction *F = BB->getParent();
541 MachineBasicBlock *TrueMBB = F->CreateMachineBasicBlock(LLVMBB);
542 MachineBasicBlock *FalseMBB = F->CreateMachineBasicBlock(LLVMBB);
543 MachineBasicBlock *DoneMBB = F->CreateMachineBasicBlock(LLVMBB);
544
546 F->insert(It, FalseMBB);
547 F->insert(It, TrueMBB);
548 F->insert(It, DoneMBB);
549
550 // Transfer the remainder of BB and its successor edges to DoneMBB.
551 DoneMBB->splice(DoneMBB->begin(), BB, std::next(MI.getIterator()), BB->end());
553
554 BB->addSuccessor(TrueMBB);
555 BB->addSuccessor(FalseMBB);
556 TrueMBB->addSuccessor(DoneMBB);
557 FalseMBB->addSuccessor(DoneMBB);
558
559 unsigned Tmp0, Tmp1, CmpReg, EqzReg, FalseReg, TrueReg;
560 Tmp0 = MRI.createVirtualRegister(MRI.getRegClass(InReg));
561 Tmp1 = MRI.createVirtualRegister(MRI.getRegClass(InReg));
562 CmpReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
563 EqzReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
564 FalseReg = MRI.createVirtualRegister(MRI.getRegClass(OutReg));
565 TrueReg = MRI.createVirtualRegister(MRI.getRegClass(OutReg));
566
567 MI.eraseFromParent();
568 // For signed numbers, we can do a single comparison to determine whether
569 // fabs(x) is within range.
570 if (IsUnsigned) {
571 Tmp0 = InReg;
572 } else {
573 BuildMI(BB, DL, TII.get(Abs), Tmp0).addReg(InReg);
574 }
575 BuildMI(BB, DL, TII.get(FConst), Tmp1)
576 .addFPImm(cast<ConstantFP>(ConstantFP::get(Ty, CmpVal)));
577 BuildMI(BB, DL, TII.get(LT), CmpReg).addReg(Tmp0).addReg(Tmp1);
578
579 // For unsigned numbers, we have to do a separate comparison with zero.
580 if (IsUnsigned) {
581 Tmp1 = MRI.createVirtualRegister(MRI.getRegClass(InReg));
582 Register SecondCmpReg =
583 MRI.createVirtualRegister(&WebAssembly::I32RegClass);
584 Register AndReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
585 BuildMI(BB, DL, TII.get(FConst), Tmp1)
586 .addFPImm(cast<ConstantFP>(ConstantFP::get(Ty, 0.0)));
587 BuildMI(BB, DL, TII.get(GE), SecondCmpReg).addReg(Tmp0).addReg(Tmp1);
588 BuildMI(BB, DL, TII.get(And), AndReg).addReg(CmpReg).addReg(SecondCmpReg);
589 CmpReg = AndReg;
590 }
591
592 BuildMI(BB, DL, TII.get(Eqz), EqzReg).addReg(CmpReg);
593
594 // Create the CFG diamond to select between doing the conversion or using
595 // the substitute value.
596 BuildMI(BB, DL, TII.get(WebAssembly::BR_IF)).addMBB(TrueMBB).addReg(EqzReg);
597 BuildMI(FalseMBB, DL, TII.get(LoweredOpcode), FalseReg).addReg(InReg);
598 BuildMI(FalseMBB, DL, TII.get(WebAssembly::BR)).addMBB(DoneMBB);
599 BuildMI(TrueMBB, DL, TII.get(IConst), TrueReg).addImm(Substitute);
600 BuildMI(*DoneMBB, DoneMBB->begin(), DL, TII.get(TargetOpcode::PHI), OutReg)
601 .addReg(FalseReg)
602 .addMBB(FalseMBB)
603 .addReg(TrueReg)
604 .addMBB(TrueMBB);
605
606 return DoneMBB;
607}
608
609// Lower a `MEMCPY` instruction into a CFG triangle around a `MEMORY_COPY`
610// instruction to handle the zero-length case.
613 const TargetInstrInfo &TII, bool Int64) {
615
616 MachineOperand DstMem = MI.getOperand(0);
617 MachineOperand SrcMem = MI.getOperand(1);
618 MachineOperand Dst = MI.getOperand(2);
619 MachineOperand Src = MI.getOperand(3);
620 MachineOperand Len = MI.getOperand(4);
621
622 // If the length is a constant, we don't actually need the check.
623 if (MachineInstr *Def = MRI.getVRegDef(Len.getReg())) {
624 if (Def->getOpcode() == WebAssembly::CONST_I32 ||
625 Def->getOpcode() == WebAssembly::CONST_I64) {
626 if (Def->getOperand(1).getImm() == 0) {
627 // A zero-length memcpy is a no-op.
628 MI.eraseFromParent();
629 return BB;
630 }
631 // A non-zero-length memcpy doesn't need a zero check.
632 unsigned MemoryCopy =
633 Int64 ? WebAssembly::MEMORY_COPY_A64 : WebAssembly::MEMORY_COPY_A32;
634 BuildMI(*BB, MI, DL, TII.get(MemoryCopy))
635 .add(DstMem)
636 .add(SrcMem)
637 .add(Dst)
638 .add(Src)
639 .add(Len);
640 MI.eraseFromParent();
641 return BB;
642 }
643 }
644
645 // We're going to add an extra use to `Len` to test if it's zero; that
646 // use shouldn't be a kill, even if the original use is.
647 MachineOperand NoKillLen = Len;
648 NoKillLen.setIsKill(false);
649
650 // Decide on which `MachineInstr` opcode we're going to use.
651 unsigned Eqz = Int64 ? WebAssembly::EQZ_I64 : WebAssembly::EQZ_I32;
652 unsigned MemoryCopy =
653 Int64 ? WebAssembly::MEMORY_COPY_A64 : WebAssembly::MEMORY_COPY_A32;
654
655 // Create two new basic blocks; one for the new `memory.fill` that we can
656 // branch over, and one for the rest of the instructions after the original
657 // `memory.fill`.
658 const BasicBlock *LLVMBB = BB->getBasicBlock();
659 MachineFunction *F = BB->getParent();
660 MachineBasicBlock *TrueMBB = F->CreateMachineBasicBlock(LLVMBB);
661 MachineBasicBlock *DoneMBB = F->CreateMachineBasicBlock(LLVMBB);
662
664 F->insert(It, TrueMBB);
665 F->insert(It, DoneMBB);
666
667 // Transfer the remainder of BB and its successor edges to DoneMBB.
668 DoneMBB->splice(DoneMBB->begin(), BB, std::next(MI.getIterator()), BB->end());
670
671 // Connect the CFG edges.
672 BB->addSuccessor(TrueMBB);
673 BB->addSuccessor(DoneMBB);
674 TrueMBB->addSuccessor(DoneMBB);
675
676 // Create a virtual register for the `Eqz` result.
677 unsigned EqzReg;
678 EqzReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
679
680 // Erase the original `memory.copy`.
681 MI.eraseFromParent();
682
683 // Test if `Len` is zero.
684 BuildMI(BB, DL, TII.get(Eqz), EqzReg).add(NoKillLen);
685
686 // Insert a new `memory.copy`.
687 BuildMI(TrueMBB, DL, TII.get(MemoryCopy))
688 .add(DstMem)
689 .add(SrcMem)
690 .add(Dst)
691 .add(Src)
692 .add(Len);
693
694 // Create the CFG triangle.
695 BuildMI(BB, DL, TII.get(WebAssembly::BR_IF)).addMBB(DoneMBB).addReg(EqzReg);
696 BuildMI(TrueMBB, DL, TII.get(WebAssembly::BR)).addMBB(DoneMBB);
697
698 return DoneMBB;
699}
700
701// Lower a `MEMSET` instruction into a CFG triangle around a `MEMORY_FILL`
702// instruction to handle the zero-length case.
705 const TargetInstrInfo &TII, bool Int64) {
707
708 MachineOperand Mem = MI.getOperand(0);
709 MachineOperand Dst = MI.getOperand(1);
710 MachineOperand Val = MI.getOperand(2);
711 MachineOperand Len = MI.getOperand(3);
712
713 // If the length is a constant, we don't actually need the check.
714 if (MachineInstr *Def = MRI.getVRegDef(Len.getReg())) {
715 if (Def->getOpcode() == WebAssembly::CONST_I32 ||
716 Def->getOpcode() == WebAssembly::CONST_I64) {
717 if (Def->getOperand(1).getImm() == 0) {
718 // A zero-length memset is a no-op.
719 MI.eraseFromParent();
720 return BB;
721 }
722 // A non-zero-length memset doesn't need a zero check.
723 unsigned MemoryFill =
724 Int64 ? WebAssembly::MEMORY_FILL_A64 : WebAssembly::MEMORY_FILL_A32;
725 BuildMI(*BB, MI, DL, TII.get(MemoryFill))
726 .add(Mem)
727 .add(Dst)
728 .add(Val)
729 .add(Len);
730 MI.eraseFromParent();
731 return BB;
732 }
733 }
734
735 // We're going to add an extra use to `Len` to test if it's zero; that
736 // use shouldn't be a kill, even if the original use is.
737 MachineOperand NoKillLen = Len;
738 NoKillLen.setIsKill(false);
739
740 // Decide on which `MachineInstr` opcode we're going to use.
741 unsigned Eqz = Int64 ? WebAssembly::EQZ_I64 : WebAssembly::EQZ_I32;
742 unsigned MemoryFill =
743 Int64 ? WebAssembly::MEMORY_FILL_A64 : WebAssembly::MEMORY_FILL_A32;
744
745 // Create two new basic blocks; one for the new `memory.fill` that we can
746 // branch over, and one for the rest of the instructions after the original
747 // `memory.fill`.
748 const BasicBlock *LLVMBB = BB->getBasicBlock();
749 MachineFunction *F = BB->getParent();
750 MachineBasicBlock *TrueMBB = F->CreateMachineBasicBlock(LLVMBB);
751 MachineBasicBlock *DoneMBB = F->CreateMachineBasicBlock(LLVMBB);
752
754 F->insert(It, TrueMBB);
755 F->insert(It, DoneMBB);
756
757 // Transfer the remainder of BB and its successor edges to DoneMBB.
758 DoneMBB->splice(DoneMBB->begin(), BB, std::next(MI.getIterator()), BB->end());
760
761 // Connect the CFG edges.
762 BB->addSuccessor(TrueMBB);
763 BB->addSuccessor(DoneMBB);
764 TrueMBB->addSuccessor(DoneMBB);
765
766 // Create a virtual register for the `Eqz` result.
767 unsigned EqzReg;
768 EqzReg = MRI.createVirtualRegister(&WebAssembly::I32RegClass);
769
770 // Erase the original `memory.fill`.
771 MI.eraseFromParent();
772
773 // Test if `Len` is zero.
774 BuildMI(BB, DL, TII.get(Eqz), EqzReg).add(NoKillLen);
775
776 // Insert a new `memory.copy`.
777 BuildMI(TrueMBB, DL, TII.get(MemoryFill)).add(Mem).add(Dst).add(Val).add(Len);
778
779 // Create the CFG triangle.
780 BuildMI(BB, DL, TII.get(WebAssembly::BR_IF)).addMBB(DoneMBB).addReg(EqzReg);
781 BuildMI(TrueMBB, DL, TII.get(WebAssembly::BR)).addMBB(DoneMBB);
782
783 return DoneMBB;
784}
785
786static MachineBasicBlock *
788 const WebAssemblySubtarget *Subtarget,
789 const TargetInstrInfo &TII) {
790 MachineInstr &CallParams = *CallResults.getPrevNode();
791 assert(CallParams.getOpcode() == WebAssembly::CALL_PARAMS);
792 assert(CallResults.getOpcode() == WebAssembly::CALL_RESULTS ||
793 CallResults.getOpcode() == WebAssembly::RET_CALL_RESULTS);
794
795 bool IsIndirect =
796 CallParams.getOperand(0).isReg() || CallParams.getOperand(0).isFI();
797 bool IsRetCall = CallResults.getOpcode() == WebAssembly::RET_CALL_RESULTS;
798
799 bool IsFuncrefCall = false;
800 if (IsIndirect && CallParams.getOperand(0).isReg()) {
801 Register Reg = CallParams.getOperand(0).getReg();
802 const MachineFunction *MF = BB->getParent();
803 const MachineRegisterInfo &MRI = MF->getRegInfo();
804 const TargetRegisterClass *TRC = MRI.getRegClass(Reg);
805 IsFuncrefCall = (TRC == &WebAssembly::FUNCREFRegClass);
806 assert(!IsFuncrefCall || Subtarget->hasReferenceTypes());
807 }
808
809 unsigned CallOp;
810 if (IsIndirect && IsRetCall) {
811 CallOp = WebAssembly::RET_CALL_INDIRECT;
812 } else if (IsIndirect) {
813 CallOp = WebAssembly::CALL_INDIRECT;
814 } else if (IsRetCall) {
815 CallOp = WebAssembly::RET_CALL;
816 } else {
817 CallOp = WebAssembly::CALL;
818 }
819
820 MachineFunction &MF = *BB->getParent();
821 const MCInstrDesc &MCID = TII.get(CallOp);
822 MachineInstrBuilder MIB(MF, MF.CreateMachineInstr(MCID, DL));
823
824 // Move the function pointer to the end of the arguments for indirect calls
825 if (IsIndirect) {
826 auto FnPtr = CallParams.getOperand(0);
827 CallParams.removeOperand(0);
828
829 // For funcrefs, call_indirect is done through __funcref_call_table and the
830 // funcref is always installed in slot 0 of the table, therefore instead of
831 // having the function pointer added at the end of the params list, a zero
832 // (the index in
833 // __funcref_call_table is added).
834 if (IsFuncrefCall) {
835 Register RegZero =
836 MF.getRegInfo().createVirtualRegister(&WebAssembly::I32RegClass);
837 MachineInstrBuilder MIBC0 =
838 BuildMI(MF, DL, TII.get(WebAssembly::CONST_I32), RegZero).addImm(0);
839
840 BB->insert(CallResults.getIterator(), MIBC0);
841 MachineInstrBuilder(MF, CallParams).addReg(RegZero);
842 } else
843 CallParams.addOperand(FnPtr);
844 }
845
846 for (auto Def : CallResults.defs())
847 MIB.add(Def);
848
849 if (IsIndirect) {
850 // Placeholder for the type index.
851 // This gets replaced with the correct value in WebAssemblyMCInstLower.cpp
852 MIB.addImm(0);
853 // The table into which this call_indirect indexes.
854 MCSymbolWasm *Table = IsFuncrefCall
856 MF.getContext(), Subtarget)
858 MF.getContext(), Subtarget);
859 if (Subtarget->hasCallIndirectOverlong()) {
860 MIB.addSym(Table);
861 } else {
862 // For the MVP there is at most one table whose number is 0, but we can't
863 // write a table symbol or issue relocations. Instead we just ensure the
864 // table is live and write a zero.
865 Table->setNoStrip();
866 MIB.addImm(0);
867 }
868 }
869
870 // Avoid duplicating the implicit operands.
871 for (auto Use : CallParams.explicit_uses())
872 MIB.add(Use);
873
874 BB->insert(CallResults.getIterator(), MIB);
875 CallParams.eraseFromParent();
876 CallResults.eraseFromParent();
877
878 // If this is a funcref call, to avoid hidden GC roots, we need to clear the
879 // table slot with ref.null upon call_indirect return.
880 //
881 // This generates the following code, which comes right after a call_indirect
882 // of a funcref:
883 //
884 // i32.const 0
885 // ref.null func
886 // table.set __funcref_call_table
887 if (IsIndirect && IsFuncrefCall) {
889 MF.getContext(), Subtarget);
890 Register RegZero =
891 MF.getRegInfo().createVirtualRegister(&WebAssembly::I32RegClass);
892 MachineInstr *Const0 =
893 BuildMI(MF, DL, TII.get(WebAssembly::CONST_I32), RegZero).addImm(0);
894 BB->insertAfter(MIB.getInstr()->getIterator(), Const0);
895
896 Register RegFuncref =
897 MF.getRegInfo().createVirtualRegister(&WebAssembly::FUNCREFRegClass);
898 MachineInstr *RefNull =
899 BuildMI(MF, DL, TII.get(WebAssembly::REF_NULL_FUNCREF), RegFuncref);
900 BB->insertAfter(Const0->getIterator(), RefNull);
901
902 MachineInstr *TableSet =
903 BuildMI(MF, DL, TII.get(WebAssembly::TABLE_SET_FUNCREF))
904 .addSym(Table)
905 .addReg(RegZero)
906 .addReg(RegFuncref);
907 BB->insertAfter(RefNull->getIterator(), TableSet);
908 }
909
910 return BB;
911}
912
913MachineBasicBlock *WebAssemblyTargetLowering::EmitInstrWithCustomInserter(
914 MachineInstr &MI, MachineBasicBlock *BB) const {
915 const TargetInstrInfo &TII = *Subtarget->getInstrInfo();
916 DebugLoc DL = MI.getDebugLoc();
917
918 switch (MI.getOpcode()) {
919 default:
920 llvm_unreachable("Unexpected instr type to insert");
921 case WebAssembly::FP_TO_SINT_I32_F32:
922 return LowerFPToInt(MI, DL, BB, TII, false, false, false,
923 WebAssembly::I32_TRUNC_S_F32);
924 case WebAssembly::FP_TO_UINT_I32_F32:
925 return LowerFPToInt(MI, DL, BB, TII, true, false, false,
926 WebAssembly::I32_TRUNC_U_F32);
927 case WebAssembly::FP_TO_SINT_I64_F32:
928 return LowerFPToInt(MI, DL, BB, TII, false, true, false,
929 WebAssembly::I64_TRUNC_S_F32);
930 case WebAssembly::FP_TO_UINT_I64_F32:
931 return LowerFPToInt(MI, DL, BB, TII, true, true, false,
932 WebAssembly::I64_TRUNC_U_F32);
933 case WebAssembly::FP_TO_SINT_I32_F64:
934 return LowerFPToInt(MI, DL, BB, TII, false, false, true,
935 WebAssembly::I32_TRUNC_S_F64);
936 case WebAssembly::FP_TO_UINT_I32_F64:
937 return LowerFPToInt(MI, DL, BB, TII, true, false, true,
938 WebAssembly::I32_TRUNC_U_F64);
939 case WebAssembly::FP_TO_SINT_I64_F64:
940 return LowerFPToInt(MI, DL, BB, TII, false, true, true,
941 WebAssembly::I64_TRUNC_S_F64);
942 case WebAssembly::FP_TO_UINT_I64_F64:
943 return LowerFPToInt(MI, DL, BB, TII, true, true, true,
944 WebAssembly::I64_TRUNC_U_F64);
945 case WebAssembly::MEMCPY_A32:
946 return LowerMemcpy(MI, DL, BB, TII, false);
947 case WebAssembly::MEMCPY_A64:
948 return LowerMemcpy(MI, DL, BB, TII, true);
949 case WebAssembly::MEMSET_A32:
950 return LowerMemset(MI, DL, BB, TII, false);
951 case WebAssembly::MEMSET_A64:
952 return LowerMemset(MI, DL, BB, TII, true);
953 case WebAssembly::CALL_RESULTS:
954 case WebAssembly::RET_CALL_RESULTS:
955 return LowerCallResults(MI, DL, BB, Subtarget, TII);
956 }
957}
958
959std::pair<unsigned, const TargetRegisterClass *>
960WebAssemblyTargetLowering::getRegForInlineAsmConstraint(
961 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const {
962 // First, see if this is a constraint that directly corresponds to a
963 // WebAssembly register class.
964 if (Constraint.size() == 1) {
965 switch (Constraint[0]) {
966 case 'r':
967 assert(VT != MVT::iPTR && "Pointer MVT not expected here");
968 if (Subtarget->hasSIMD128() && VT.isVector()) {
969 if (VT.getSizeInBits() == 128)
970 return std::make_pair(0U, &WebAssembly::V128RegClass);
971 }
972 if (VT.isInteger() && !VT.isVector()) {
973 if (VT.getSizeInBits() <= 32)
974 return std::make_pair(0U, &WebAssembly::I32RegClass);
975 if (VT.getSizeInBits() <= 64)
976 return std::make_pair(0U, &WebAssembly::I64RegClass);
977 }
978 if (VT.isFloatingPoint() && !VT.isVector()) {
979 switch (VT.getSizeInBits()) {
980 case 32:
981 return std::make_pair(0U, &WebAssembly::F32RegClass);
982 case 64:
983 return std::make_pair(0U, &WebAssembly::F64RegClass);
984 default:
985 break;
986 }
987 }
988 break;
989 default:
990 break;
991 }
992 }
993
995}
996
997bool WebAssemblyTargetLowering::isCheapToSpeculateCttz(Type *Ty) const {
998 // Assume ctz is a relatively cheap operation.
999 return true;
1000}
1001
1002bool WebAssemblyTargetLowering::isCheapToSpeculateCtlz(Type *Ty) const {
1003 // Assume clz is a relatively cheap operation.
1004 return true;
1005}
1006
1007bool WebAssemblyTargetLowering::isLegalAddressingMode(const DataLayout &DL,
1008 const AddrMode &AM,
1009 Type *Ty, unsigned AS,
1010 Instruction *I) const {
1011 // WebAssembly offsets are added as unsigned without wrapping. The
1012 // isLegalAddressingMode gives us no way to determine if wrapping could be
1013 // happening, so we approximate this by accepting only non-negative offsets.
1014 if (AM.BaseOffs < 0)
1015 return false;
1016
1017 // WebAssembly has no scale register operands.
1018 if (AM.Scale != 0)
1019 return false;
1020
1021 // Everything else is legal.
1022 return true;
1023}
1024
1025bool WebAssemblyTargetLowering::allowsMisalignedMemoryAccesses(
1026 EVT /*VT*/, unsigned /*AddrSpace*/, Align /*Align*/,
1027 MachineMemOperand::Flags /*Flags*/, unsigned *Fast) const {
1028 // WebAssembly supports unaligned accesses, though it should be declared
1029 // with the p2align attribute on loads and stores which do so, and there
1030 // may be a performance impact. We tell LLVM they're "fast" because
1031 // for the kinds of things that LLVM uses this for (merging adjacent stores
1032 // of constants, etc.), WebAssembly implementations will either want the
1033 // unaligned access or they'll split anyway.
1034 if (Fast)
1035 *Fast = 1;
1036 return true;
1037}
1038
1039bool WebAssemblyTargetLowering::isIntDivCheap(EVT VT,
1040 AttributeList Attr) const {
1041 // The current thinking is that wasm engines will perform this optimization,
1042 // so we can save on code size.
1043 return true;
1044}
1045
1046bool WebAssemblyTargetLowering::isVectorLoadExtDesirable(SDValue ExtVal) const {
1047 EVT ExtT = ExtVal.getValueType();
1048 SDValue N0 = peekThroughFreeze(ExtVal->getOperand(0));
1049 auto *Load = dyn_cast<LoadSDNode>(N0);
1050 if (!Load)
1051 return false;
1052 EVT MemT = Load->getValueType(0);
1053 return (ExtT == MVT::v8i16 && MemT == MVT::v8i8) ||
1054 (ExtT == MVT::v4i32 && MemT == MVT::v4i16) ||
1055 (ExtT == MVT::v2i64 && MemT == MVT::v2i32);
1056}
1057
1058bool WebAssemblyTargetLowering::isOffsetFoldingLegal(
1059 const GlobalAddressSDNode *GA) const {
1060 // Wasm doesn't support function addresses with offsets
1061 const GlobalValue *GV = GA->getGlobal();
1063}
1064
1065EVT WebAssemblyTargetLowering::getSetCCResultType(const DataLayout &DL,
1066 LLVMContext &C,
1067 EVT VT) const {
1068 if (VT.isVector()) {
1069 if (VT.getVectorElementType() == MVT::f16 && !Subtarget->hasFP16())
1070 return VT.changeElementType(C, MVT::i1);
1071
1073 }
1074
1075 // So far, all branch instructions in Wasm take an I32 condition.
1076 // The default TargetLowering::getSetCCResultType returns the pointer size,
1077 // which would be useful to reduce instruction counts when testing
1078 // against 64-bit pointers/values if at some point Wasm supports that.
1079 return EVT::getIntegerVT(C, 32);
1080}
1081
1082void WebAssemblyTargetLowering::getTgtMemIntrinsic(
1084 MachineFunction &MF, unsigned Intrinsic) const {
1086 switch (Intrinsic) {
1087 case Intrinsic::wasm_memory_atomic_notify:
1089 Info.memVT = MVT::i32;
1090 Info.ptrVal = I.getArgOperand(0);
1091 Info.offset = 0;
1092 Info.align = Align(4);
1093 // atomic.notify instruction does not really load the memory specified with
1094 // this argument, but MachineMemOperand should either be load or store, so
1095 // we set this to a load.
1096 // FIXME Volatile isn't really correct, but currently all LLVM atomic
1097 // instructions are treated as volatiles in the backend, so we should be
1098 // consistent. The same applies for wasm_atomic_wait intrinsics too.
1100 Infos.push_back(Info);
1101 return;
1102 case Intrinsic::wasm_memory_atomic_wait32:
1104 Info.memVT = MVT::i32;
1105 Info.ptrVal = I.getArgOperand(0);
1106 Info.offset = 0;
1107 Info.align = Align(4);
1109 Infos.push_back(Info);
1110 return;
1111 case Intrinsic::wasm_memory_atomic_wait64:
1113 Info.memVT = MVT::i64;
1114 Info.ptrVal = I.getArgOperand(0);
1115 Info.offset = 0;
1116 Info.align = Align(8);
1118 Infos.push_back(Info);
1119 return;
1120 case Intrinsic::wasm_loadf16_f32:
1122 Info.memVT = MVT::f16;
1123 Info.ptrVal = I.getArgOperand(0);
1124 Info.offset = 0;
1125 Info.align = Align(2);
1127 Infos.push_back(Info);
1128 return;
1129 case Intrinsic::wasm_storef16_f32:
1131 Info.memVT = MVT::f16;
1132 Info.ptrVal = I.getArgOperand(1);
1133 Info.offset = 0;
1134 Info.align = Align(2);
1136 Infos.push_back(Info);
1137 return;
1138 default:
1139 return;
1140 }
1141}
1142
1143void WebAssemblyTargetLowering::computeKnownBitsForTargetNode(
1144 const SDValue Op, KnownBits &Known, const APInt &DemandedElts,
1145 const SelectionDAG &DAG, unsigned Depth) const {
1146 switch (Op.getOpcode()) {
1147 default:
1148 break;
1150 unsigned IntNo = Op.getConstantOperandVal(0);
1151 switch (IntNo) {
1152 default:
1153 break;
1154 case Intrinsic::wasm_bitmask: {
1155 unsigned BitWidth = Known.getBitWidth();
1156 EVT VT = Op.getOperand(1).getSimpleValueType();
1157 unsigned PossibleBits = VT.getVectorNumElements();
1158 APInt ZeroMask = APInt::getHighBitsSet(BitWidth, BitWidth - PossibleBits);
1159 Known.Zero |= ZeroMask;
1160 break;
1161 }
1162 }
1163 break;
1164 }
1165 case WebAssemblyISD::EXTEND_LOW_U:
1166 case WebAssemblyISD::EXTEND_HIGH_U: {
1167 // We know the high half, of each destination vector element, will be zero.
1168 SDValue SrcOp = Op.getOperand(0);
1169 EVT VT = SrcOp.getSimpleValueType();
1170 unsigned BitWidth = Known.getBitWidth();
1171 if (VT == MVT::v8i8 || VT == MVT::v16i8) {
1172 assert(BitWidth >= 8 && "Unexpected width!");
1174 Known.Zero |= Mask;
1175 } else if (VT == MVT::v4i16 || VT == MVT::v8i16) {
1176 assert(BitWidth >= 16 && "Unexpected width!");
1178 Known.Zero |= Mask;
1179 } else if (VT == MVT::v2i32 || VT == MVT::v4i32) {
1180 assert(BitWidth >= 32 && "Unexpected width!");
1182 Known.Zero |= Mask;
1183 }
1184 break;
1185 }
1186 // For 128-bit addition if the upper bits are all zero then it's known that
1187 // the upper bits of the result will have all bits guaranteed zero except the
1188 // first.
1189 case WebAssemblyISD::I64_ADD128:
1190 if (Op.getResNo() == 1) {
1191 SDValue LHS_HI = Op.getOperand(1);
1192 SDValue RHS_HI = Op.getOperand(3);
1193 if (isNullConstant(LHS_HI) && isNullConstant(RHS_HI))
1194 Known.Zero.setBitsFrom(1);
1195 }
1196 break;
1197 }
1198}
1199
1201WebAssemblyTargetLowering::getPreferredVectorAction(MVT VT) const {
1202 if (VT.isFixedLengthVector()) {
1203 MVT EltVT = VT.getVectorElementType();
1204 // We have legal vector types with these lane types, so widening the
1205 // vector would let us use some of the lanes directly without having to
1206 // extend or truncate values.
1207 if (EltVT == MVT::i8 || EltVT == MVT::i16 || EltVT == MVT::i32 ||
1208 EltVT == MVT::i64 || EltVT == MVT::f32 || EltVT == MVT::f64)
1209 return TypeWidenVector;
1210 }
1211
1213}
1214
1215bool WebAssemblyTargetLowering::isFMAFasterThanFMulAndFAdd(
1216 const MachineFunction &MF, EVT VT) const {
1217 if (!Subtarget->hasFP16() || !VT.isVector())
1218 return false;
1219
1220 EVT ScalarVT = VT.getScalarType();
1221 if (!ScalarVT.isSimple())
1222 return false;
1223
1224 return ScalarVT.getSimpleVT().SimpleTy == MVT::f16;
1225}
1226
1227bool WebAssemblyTargetLowering::shouldSimplifyDemandedVectorElts(
1228 SDValue Op, const TargetLoweringOpt &TLO) const {
1229 // ISel process runs DAGCombiner after legalization; this step is called
1230 // SelectionDAG optimization phase. This post-legalization combining process
1231 // runs DAGCombiner on each node, and if there was a change to be made,
1232 // re-runs legalization again on it and its user nodes to make sure
1233 // everythiing is in a legalized state.
1234 //
1235 // The legalization calls lowering routines, and we do our custom lowering for
1236 // build_vectors (LowerBUILD_VECTOR), which converts undef vector elements
1237 // into zeros. But there is a set of routines in DAGCombiner that turns unused
1238 // (= not demanded) nodes into undef, among which SimplifyDemandedVectorElts
1239 // turns unused vector elements into undefs. But this routine does not work
1240 // with our custom LowerBUILD_VECTOR, which turns undefs into zeros. This
1241 // combination can result in a infinite loop, in which undefs are converted to
1242 // zeros in legalization and back to undefs in combining.
1243 //
1244 // So after DAG is legalized, we prevent SimplifyDemandedVectorElts from
1245 // running for build_vectors.
1246 if (Op.getOpcode() == ISD::BUILD_VECTOR && TLO.LegalOps && TLO.LegalTys)
1247 return false;
1248 return true;
1249}
1250
1251//===----------------------------------------------------------------------===//
1252// WebAssembly Lowering private implementation.
1253//===----------------------------------------------------------------------===//
1254
1255//===----------------------------------------------------------------------===//
1256// Lowering Code
1257//===----------------------------------------------------------------------===//
1258
1259static void fail(const SDLoc &DL, SelectionDAG &DAG, const char *Msg) {
1261 DAG.getContext()->diagnose(
1262 DiagnosticInfoUnsupported(MF.getFunction(), Msg, DL.getDebugLoc()));
1263}
1264
1265// Test whether the given calling convention is supported.
1267 // We currently support the language-independent target-independent
1268 // conventions. We don't yet have a way to annotate calls with properties like
1269 // "cold", and we don't have any call-clobbered registers, so these are mostly
1270 // all handled the same.
1271 return CallConv == CallingConv::C || CallConv == CallingConv::Fast ||
1272 CallConv == CallingConv::Cold ||
1273 CallConv == CallingConv::PreserveMost ||
1274 CallConv == CallingConv::PreserveAll ||
1275 CallConv == CallingConv::CXX_FAST_TLS ||
1277 CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail;
1278}
1279
1280SDValue
1281WebAssemblyTargetLowering::LowerCall(CallLoweringInfo &CLI,
1282 SmallVectorImpl<SDValue> &InVals) const {
1283 SelectionDAG &DAG = CLI.DAG;
1284 SDLoc DL = CLI.DL;
1285 SDValue Chain = CLI.Chain;
1286 SDValue Callee = CLI.Callee;
1288 auto Layout = MF.getDataLayout();
1289
1290 // A call through a funcref is expressed in IR as a call through the pointer
1291 // produced by the llvm.wasm.funcref.to_ptr intrinsic. Detect this here and
1292 // recover the underlying funcref value so the call can be lowered to a
1293 // table.set + call_indirect through the dedicated __funcref_call_table.
1294 bool IsFuncrefCall = false;
1295 if (Callee.getOpcode() == ISD::INTRINSIC_WO_CHAIN &&
1296 Callee.getConstantOperandVal(0) == Intrinsic::wasm_funcref_to_ptr) {
1297 Callee = Callee.getOperand(1);
1298 IsFuncrefCall = true;
1299 }
1300
1301 CallingConv::ID CallConv = CLI.CallConv;
1302 if (!callingConvSupported(CallConv))
1303 fail(DL, DAG,
1304 "WebAssembly doesn't support language-specific or target-specific "
1305 "calling conventions yet");
1306 if (CLI.IsPatchPoint)
1307 fail(DL, DAG, "WebAssembly doesn't support patch point yet");
1308
1309 if (CLI.IsTailCall) {
1310 auto NoTail = [&](const char *Msg) {
1311 if (CLI.CB && CLI.CB->isMustTailCall())
1312 fail(DL, DAG, Msg);
1313 CLI.IsTailCall = false;
1314 };
1315
1316 if (!Subtarget->hasTailCall())
1317 NoTail("WebAssembly 'tail-call' feature not enabled");
1318
1319 // Varargs calls cannot be tail calls because the buffer is on the stack
1320 if (CLI.IsVarArg)
1321 NoTail("WebAssembly does not support varargs tail calls");
1322
1323 // Do not tail call unless caller and callee return types match
1324 const Function &F = MF.getFunction();
1325 const TargetMachine &TM = getTargetMachine();
1326 Type *RetTy = F.getReturnType();
1327 SmallVector<MVT, 4> CallerRetTys;
1328 SmallVector<MVT, 4> CalleeRetTys;
1329 computeLegalValueVTs(F, TM, RetTy, CallerRetTys);
1330 computeLegalValueVTs(F, TM, CLI.RetTy, CalleeRetTys);
1331 bool TypesMatch = CallerRetTys.size() == CalleeRetTys.size() &&
1332 std::equal(CallerRetTys.begin(), CallerRetTys.end(),
1333 CalleeRetTys.begin());
1334 if (!TypesMatch)
1335 NoTail("WebAssembly tail call requires caller and callee return types to "
1336 "match");
1337
1338 // If pointers to local stack values are passed, we cannot tail call
1339 if (CLI.CB) {
1340 for (auto &Arg : CLI.CB->args()) {
1341 Value *Val = Arg.get();
1342 // Trace the value back through pointer operations
1343 while (true) {
1344 Value *Src = Val->stripPointerCastsAndAliases();
1345 if (auto *GEP = dyn_cast<GetElementPtrInst>(Src))
1346 Src = GEP->getPointerOperand();
1347 if (Val == Src)
1348 break;
1349 Val = Src;
1350 }
1351 if (isa<AllocaInst>(Val)) {
1352 NoTail(
1353 "WebAssembly does not support tail calling with stack arguments");
1354 break;
1355 }
1356 }
1357 }
1358
1359 // A byval argument is copied into this function's stack frame below, and
1360 // a tail call releases that frame before the callee reads the copy.
1361 if (llvm::any_of(CLI.Outs, [](const ISD::OutputArg &Out) {
1362 return Out.Flags.isByVal() && Out.Flags.getByValSize() != 0;
1363 }))
1364 NoTail("WebAssembly does not support tail calling with byval arguments");
1365 }
1366
1367 SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins;
1368 SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs;
1369 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
1370
1371 // The generic code may have added an sret argument. If we're lowering an
1372 // invoke function, the ABI requires that the function pointer be the first
1373 // argument, so we may have to swap the arguments.
1374 if (CallConv == CallingConv::WASM_EmscriptenInvoke && Outs.size() >= 2 &&
1375 Outs[0].Flags.isSRet()) {
1376 std::swap(Outs[0], Outs[1]);
1377 std::swap(OutVals[0], OutVals[1]);
1378 }
1379
1380 bool HasSwiftSelfArg = false;
1381 bool HasSwiftErrorArg = false;
1382 bool HasSwiftAsyncArg = false;
1383 unsigned NumFixedArgs = 0;
1384 for (unsigned I = 0; I < Outs.size(); ++I) {
1385 const ISD::OutputArg &Out = Outs[I];
1386 SDValue &OutVal = OutVals[I];
1387 HasSwiftSelfArg |= Out.Flags.isSwiftSelf();
1388 HasSwiftErrorArg |= Out.Flags.isSwiftError();
1389 HasSwiftAsyncArg |= Out.Flags.isSwiftAsync();
1390 if (Out.Flags.isNest())
1391 fail(DL, DAG, "WebAssembly hasn't implemented nest arguments");
1392 if (Out.Flags.isInAlloca())
1393 fail(DL, DAG, "WebAssembly hasn't implemented inalloca arguments");
1394 if (Out.Flags.isInConsecutiveRegs())
1395 fail(DL, DAG, "WebAssembly hasn't implemented cons regs arguments");
1396 if (Out.Flags.isInConsecutiveRegsLast())
1397 fail(DL, DAG, "WebAssembly hasn't implemented cons regs last arguments");
1398 if (Out.Flags.isByVal() && Out.Flags.getByValSize() != 0) {
1399 auto &MFI = MF.getFrameInfo();
1400 int FI = MFI.CreateStackObject(Out.Flags.getByValSize(),
1401 Out.Flags.getNonZeroByValAlign(),
1402 /*isSS=*/false);
1403 SDValue SizeNode =
1404 DAG.getConstant(Out.Flags.getByValSize(), DL, MVT::i32);
1405 SDValue FINode = DAG.getFrameIndex(FI, getPointerTy(Layout));
1406 Align Alignment = Out.Flags.getNonZeroByValAlign();
1407 Chain = DAG.getMemcpy(Chain, DL, FINode, OutVal, SizeNode, Alignment,
1408 Alignment,
1409 /*isVolatile*/ false, /*AlwaysInline=*/false,
1410 /*CI=*/nullptr, std::nullopt, MachinePointerInfo(),
1411 MachinePointerInfo());
1412 OutVal = FINode;
1413 }
1414 // Count the number of fixed args *after* legalization.
1415 NumFixedArgs += !Out.Flags.isVarArg();
1416 }
1417
1418 bool IsVarArg = CLI.IsVarArg;
1419 auto PtrVT = getPointerTy(Layout);
1420
1421 // For swiftcc and swifttailcc, emit additional swiftself, swifterror, and
1422 // (for swifttailcc) swiftasync arguments if there aren't. These additional
1423 // arguments are also added for callee signature. They are necessary to match
1424 // callee and caller signature for indirect call.
1425 if (CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail) {
1426 Type *PtrTy = PointerType::getUnqual(*DAG.getContext());
1427 if (!HasSwiftSelfArg) {
1428 NumFixedArgs++;
1429 ISD::ArgFlagsTy Flags;
1430 Flags.setSwiftSelf();
1431 ISD::OutputArg Arg(Flags, PtrVT, EVT(PtrVT), PtrTy, 0, 0);
1432 CLI.Outs.push_back(Arg);
1433 SDValue ArgVal = DAG.getUNDEF(PtrVT);
1434 CLI.OutVals.push_back(ArgVal);
1435 }
1436 if (!HasSwiftErrorArg) {
1437 NumFixedArgs++;
1438 ISD::ArgFlagsTy Flags;
1439 Flags.setSwiftError();
1440 ISD::OutputArg Arg(Flags, PtrVT, EVT(PtrVT), PtrTy, 0, 0);
1441 CLI.Outs.push_back(Arg);
1442 SDValue ArgVal = DAG.getUNDEF(PtrVT);
1443 CLI.OutVals.push_back(ArgVal);
1444 }
1445 if (CallConv == CallingConv::SwiftTail && !HasSwiftAsyncArg) {
1446 NumFixedArgs++;
1447 ISD::ArgFlagsTy Flags;
1448 Flags.setSwiftAsync();
1449 ISD::OutputArg Arg(Flags, PtrVT, EVT(PtrVT), PtrTy, 0, 0);
1450 CLI.Outs.push_back(Arg);
1451 SDValue ArgVal = DAG.getUNDEF(PtrVT);
1452 CLI.OutVals.push_back(ArgVal);
1453 }
1454 }
1455
1456 // Analyze operands of the call, assigning locations to each operand.
1458 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
1459
1460 if (IsVarArg) {
1461 // Outgoing non-fixed arguments are placed in a buffer. First
1462 // compute their offsets and the total amount of buffer space needed.
1463 for (unsigned I = NumFixedArgs; I < Outs.size(); ++I) {
1464 const ISD::OutputArg &Out = Outs[I];
1465 SDValue &Arg = OutVals[I];
1466 EVT VT = Arg.getValueType();
1467 assert(VT != MVT::iPTR && "Legalized args should be concrete");
1468 Type *Ty = VT.getTypeForEVT(*DAG.getContext());
1470 std::max(Out.Flags.getNonZeroOrigAlign(), Layout.getABITypeAlign(Ty));
1471 unsigned Offset =
1472 CCInfo.AllocateStack(Layout.getTypeAllocSize(Ty), Alignment);
1473 CCInfo.addLoc(CCValAssign::getMem(ArgLocs.size(), VT.getSimpleVT(),
1474 Offset, VT.getSimpleVT(),
1476 }
1477 }
1478
1479 unsigned NumBytes = CCInfo.getAlignedCallFrameSize();
1480
1481 SDValue FINode;
1482 if (IsVarArg && NumBytes) {
1483 // For non-fixed arguments, next emit stores to store the argument values
1484 // to the stack buffer at the offsets computed above.
1485 MaybeAlign StackAlign = Layout.getStackAlignment();
1486 assert(StackAlign && "data layout string is missing stack alignment");
1487 int FI = MF.getFrameInfo().CreateStackObject(NumBytes, *StackAlign,
1488 /*isSS=*/false);
1489 unsigned ValNo = 0;
1491 for (SDValue Arg : drop_begin(OutVals, NumFixedArgs)) {
1492 assert(ArgLocs[ValNo].getValNo() == ValNo &&
1493 "ArgLocs should remain in order and only hold varargs args");
1494 unsigned Offset = ArgLocs[ValNo++].getLocMemOffset();
1495 FINode = DAG.getFrameIndex(FI, getPointerTy(Layout));
1496 SDValue Add = DAG.getNode(ISD::ADD, DL, PtrVT, FINode,
1497 DAG.getConstant(Offset, DL, PtrVT));
1498 Chains.push_back(
1499 DAG.getStore(Chain, DL, Arg, Add,
1501 }
1502 if (!Chains.empty())
1503 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
1504 } else if (IsVarArg) {
1505 FINode = DAG.getIntPtrConstant(0, DL);
1506 }
1507
1508 if (Callee->getOpcode() == ISD::GlobalAddress) {
1509 // If the callee is a GlobalAddress node (quite common, every direct call
1510 // is) turn it into a TargetGlobalAddress node so that LowerGlobalAddress
1511 // doesn't at MO_GOT which is not needed for direct calls.
1512 GlobalAddressSDNode *GA = cast<GlobalAddressSDNode>(Callee);
1515 GA->getOffset());
1516 Callee = DAG.getNode(WebAssemblyISD::Wrapper, DL,
1517 getPointerTy(DAG.getDataLayout()), Callee);
1518 }
1519
1520 // Compute the operands for the CALLn node.
1522 Ops.push_back(Chain);
1523 Ops.push_back(Callee);
1524
1525 // Add all fixed arguments. Note that for non-varargs calls, NumFixedArgs
1526 // isn't reliable.
1527 Ops.append(OutVals.begin(),
1528 IsVarArg ? OutVals.begin() + NumFixedArgs : OutVals.end());
1529 // Add a pointer to the vararg buffer.
1530 if (IsVarArg)
1531 Ops.push_back(FINode);
1532
1533 SmallVector<EVT, 8> InTys;
1534 for (const auto &In : Ins) {
1535 assert(!In.Flags.isByVal() && "byval is not valid for return values");
1536 assert(!In.Flags.isNest() && "nest is not valid for return values");
1537 if (In.Flags.isInAlloca())
1538 fail(DL, DAG, "WebAssembly hasn't implemented inalloca return values");
1539 if (In.Flags.isInConsecutiveRegs())
1540 fail(DL, DAG, "WebAssembly hasn't implemented cons regs return values");
1541 if (In.Flags.isInConsecutiveRegsLast())
1542 fail(DL, DAG,
1543 "WebAssembly hasn't implemented cons regs last return values");
1544 // Ignore In.getNonZeroOrigAlign() because all our arguments are passed in
1545 // registers.
1546 InTys.push_back(In.VT);
1547 }
1548
1549 // Lastly, if this is a call to a funcref we need to add an instruction
1550 // table.set to the chain and transform the call.
1551 if (IsFuncrefCall) {
1552 // In the absence of function references proposal where a funcref call is
1553 // lowered to call_ref, using reference types we generate a table.set to set
1554 // the funcref to a special table used solely for this purpose, followed by
1555 // a call_indirect. Here we just generate the table set, and return the
1556 // SDValue of the table.set so that LowerCall can finalize the lowering by
1557 // generating the call_indirect.
1558 SDValue Chain = Ops[0];
1559
1561 MF.getContext(), Subtarget);
1562 SDValue Sym = DAG.getMCSymbol(Table, PtrVT);
1563 SDValue TableSlot = DAG.getConstant(0, DL, MVT::i32);
1564 SDValue TableSetOps[] = {Chain, Sym, TableSlot, Callee};
1565 SDValue TableSet = DAG.getMemIntrinsicNode(
1566 WebAssemblyISD::TABLE_SET, DL, DAG.getVTList(MVT::Other), TableSetOps,
1567 MVT::funcref, MachinePointerInfo(), Align(1),
1569
1570 Ops[0] = TableSet; // The new chain is the TableSet itself
1571 }
1572
1573 if (CLI.IsTailCall) {
1574 // ret_calls do not return values to the current frame
1575 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
1576 return DAG.getNode(WebAssemblyISD::RET_CALL, DL, NodeTys, Ops);
1577 }
1578
1579 InTys.push_back(MVT::Other);
1580 SDVTList InTyList = DAG.getVTList(InTys);
1581 SDValue Res = DAG.getNode(WebAssemblyISD::CALL, DL, InTyList, Ops);
1582
1583 for (size_t I = 0; I < Ins.size(); ++I)
1584 InVals.push_back(Res.getValue(I));
1585
1586 // Return the chain
1587 return Res.getValue(Ins.size());
1588}
1589
1590bool WebAssemblyTargetLowering::CanLowerReturn(
1591 CallingConv::ID /*CallConv*/, MachineFunction & /*MF*/, bool /*IsVarArg*/,
1592 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext & /*Context*/,
1593 const Type *RetTy) const {
1594 // WebAssembly can only handle returning tuples with multivalue enabled
1595 return WebAssembly::canLowerReturn(Outs.size(), Subtarget);
1596}
1597
1598SDValue WebAssemblyTargetLowering::LowerReturn(
1599 SDValue Chain, CallingConv::ID CallConv, bool /*IsVarArg*/,
1601 const SmallVectorImpl<SDValue> &OutVals, const SDLoc &DL,
1602 SelectionDAG &DAG) const {
1603 assert(WebAssembly::canLowerReturn(Outs.size(), Subtarget) &&
1604 "MVP WebAssembly can only return up to one value");
1605 if (!callingConvSupported(CallConv))
1606 fail(DL, DAG, "WebAssembly doesn't support non-C calling conventions");
1607
1608 SmallVector<SDValue, 4> RetOps(1, Chain);
1609 RetOps.append(OutVals.begin(), OutVals.end());
1610 Chain = DAG.getNode(WebAssemblyISD::RETURN, DL, MVT::Other, RetOps);
1611
1612 // Record the number and types of the return values.
1613 for (const ISD::OutputArg &Out : Outs) {
1614 assert(!Out.Flags.isByVal() && "byval is not valid for return values");
1615 assert(!Out.Flags.isNest() && "nest is not valid for return values");
1616 assert(!Out.Flags.isVarArg() && "non-fixed return value is not valid");
1617 if (Out.Flags.isInAlloca())
1618 fail(DL, DAG, "WebAssembly hasn't implemented inalloca results");
1619 if (Out.Flags.isInConsecutiveRegs())
1620 fail(DL, DAG, "WebAssembly hasn't implemented cons regs results");
1621 if (Out.Flags.isInConsecutiveRegsLast())
1622 fail(DL, DAG, "WebAssembly hasn't implemented cons regs last results");
1623 }
1624
1625 return Chain;
1626}
1627
1628SDValue WebAssemblyTargetLowering::LowerFormalArguments(
1629 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
1630 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
1631 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
1632 if (!callingConvSupported(CallConv))
1633 fail(DL, DAG, "WebAssembly doesn't support non-C calling conventions");
1634
1636 auto *MFI = MF.getInfo<WebAssemblyFunctionInfo>();
1637
1638 // Set up the incoming ARGUMENTS value, which serves to represent the liveness
1639 // of the incoming values before they're represented by virtual registers.
1640 MF.getRegInfo().addLiveIn(WebAssembly::ARGUMENTS);
1641
1642 bool HasSwiftErrorArg = false;
1643 bool HasSwiftSelfArg = false;
1644 bool HasSwiftAsyncArg = false;
1645 for (const ISD::InputArg &In : Ins) {
1646 HasSwiftSelfArg |= In.Flags.isSwiftSelf();
1647 HasSwiftErrorArg |= In.Flags.isSwiftError();
1648 HasSwiftAsyncArg |= In.Flags.isSwiftAsync();
1649 if (In.Flags.isInAlloca())
1650 fail(DL, DAG, "WebAssembly hasn't implemented inalloca arguments");
1651 if (In.Flags.isNest())
1652 fail(DL, DAG, "WebAssembly hasn't implemented nest arguments");
1653 if (In.Flags.isInConsecutiveRegs())
1654 fail(DL, DAG, "WebAssembly hasn't implemented cons regs arguments");
1655 if (In.Flags.isInConsecutiveRegsLast())
1656 fail(DL, DAG, "WebAssembly hasn't implemented cons regs last arguments");
1657 // Ignore In.getNonZeroOrigAlign() because all our arguments are passed in
1658 // registers.
1659 InVals.push_back(In.Used ? DAG.getNode(WebAssemblyISD::ARGUMENT, DL, In.VT,
1660 DAG.getTargetConstant(InVals.size(),
1661 DL, MVT::i32))
1662 : DAG.getUNDEF(In.VT));
1663
1664 // Record the number and types of arguments.
1665 MFI->addParam(In.VT);
1666 }
1667
1668 // For swiftcc and swifttailcc, emit additional swiftself, swifterror, and
1669 // (for swifttailcc) swiftasync arguments if there aren't. These additional
1670 // arguments are also added for callee signature. They are necessary to match
1671 // callee and caller signature for indirect call.
1672 auto PtrVT = getPointerTy(MF.getDataLayout());
1673 if (CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail) {
1674 if (!HasSwiftSelfArg) {
1675 MFI->addParam(PtrVT);
1676 }
1677 if (!HasSwiftErrorArg) {
1678 MFI->addParam(PtrVT);
1679 }
1680 if (CallConv == CallingConv::SwiftTail && !HasSwiftAsyncArg) {
1681 MFI->addParam(PtrVT);
1682 }
1683 }
1684 // Varargs are copied into a buffer allocated by the caller, and a pointer to
1685 // the buffer is passed as an argument.
1686 if (IsVarArg) {
1687 MVT PtrVT = getPointerTy(MF.getDataLayout());
1688 Register VarargVreg =
1690 MFI->setVarargBufferVreg(VarargVreg);
1691 Chain = DAG.getCopyToReg(
1692 Chain, DL, VarargVreg,
1693 DAG.getNode(WebAssemblyISD::ARGUMENT, DL, PtrVT,
1694 DAG.getTargetConstant(Ins.size(), DL, MVT::i32)));
1695 MFI->addParam(PtrVT);
1696 }
1697
1698 // Record the number and types of arguments and results.
1699 SmallVector<MVT, 4> Params;
1702 MF.getFunction(), DAG.getTarget(), Params, Results);
1703 for (MVT VT : Results)
1704 MFI->addResult(VT);
1705 // TODO: Use signatures in WebAssemblyMachineFunctionInfo too and unify
1706 // the param logic here with ComputeSignatureVTs
1707 assert(MFI->getParams().size() == Params.size() &&
1708 std::equal(MFI->getParams().begin(), MFI->getParams().end(),
1709 Params.begin()));
1710
1711 return Chain;
1712}
1713
1714void WebAssemblyTargetLowering::ReplaceNodeResults(
1716 switch (N->getOpcode()) {
1718 // Do not add any results, signifying that N should not be custom lowered
1719 // after all. This happens because simd128 turns on custom lowering for
1720 // SIGN_EXTEND_INREG, but for non-vector sign extends the result might be an
1721 // illegal type.
1722 break;
1726 // Do not add any results, signifying that N should not be custom lowered.
1727 // EXTEND_VECTOR_INREG is implemented for some vectors, but not all.
1728 break;
1729 case ISD::FP_ROUND: {
1730 EVT VT = N->getValueType(0);
1731 SDValue Src = N->getOperand(0);
1732 if (VT == MVT::v4f16 && Src.getValueType() == MVT::v4f32) {
1733 Results.push_back(
1734 DAG.getNode(WebAssemblyISD::DEMOTE_ZERO, SDLoc(N), MVT::v8f16, Src));
1735 }
1736 break;
1737 }
1738 case ISD::ADD:
1739 case ISD::SUB:
1740 Results.push_back(Replace128Op(N, DAG));
1741 break;
1742 default:
1744 "ReplaceNodeResults not implemented for this op for WebAssembly!");
1745 }
1746}
1747
1748//===----------------------------------------------------------------------===//
1749// Custom lowering hooks.
1750//===----------------------------------------------------------------------===//
1751
1752SDValue WebAssemblyTargetLowering::LowerOperation(SDValue Op,
1753 SelectionDAG &DAG) const {
1754 SDLoc DL(Op);
1755 switch (Op.getOpcode()) {
1756 default:
1757 llvm_unreachable("unimplemented operation lowering");
1758 return SDValue();
1759 case ISD::FrameIndex:
1760 return LowerFrameIndex(Op, DAG);
1761 case ISD::GlobalAddress:
1762 return LowerGlobalAddress(Op, DAG);
1764 return LowerGlobalTLSAddress(Op, DAG);
1766 return LowerExternalSymbol(Op, DAG);
1767 case ISD::JumpTable:
1768 return LowerJumpTable(Op, DAG);
1769 case ISD::BR_JT:
1770 return LowerBR_JT(Op, DAG);
1771 case ISD::VASTART:
1772 return LowerVASTART(Op, DAG);
1773 case ISD::BlockAddress:
1774 case ISD::BRIND:
1775 fail(DL, DAG, "WebAssembly hasn't implemented computed gotos");
1776 return SDValue();
1777 case ISD::RETURNADDR:
1778 return LowerRETURNADDR(Op, DAG);
1779 case ISD::FRAMEADDR:
1780 return LowerFRAMEADDR(Op, DAG);
1781 case ISD::CopyToReg:
1782 return LowerCopyToReg(Op, DAG);
1785 return LowerAccessVectorElement(Op, DAG);
1789 return LowerIntrinsic(Op, DAG);
1791 return LowerSIGN_EXTEND_INREG(Op, DAG);
1795 return LowerEXTEND_VECTOR_INREG(Op, DAG);
1796 case ISD::BUILD_VECTOR:
1797 return LowerBUILD_VECTOR(Op, DAG);
1799 return LowerVECTOR_SHUFFLE(Op, DAG);
1800 case ISD::SETCC:
1801 return LowerSETCC(Op, DAG);
1802 case ISD::SHL:
1803 case ISD::SRA:
1804 case ISD::SRL:
1805 return LowerShift(Op, DAG);
1808 return LowerFP_TO_INT_SAT(Op, DAG);
1809 case ISD::FMINNUM:
1810 case ISD::FMINIMUMNUM:
1811 return LowerFMIN(Op, DAG);
1812 case ISD::FMAXNUM:
1813 case ISD::FMAXIMUMNUM:
1814 return LowerFMAX(Op, DAG);
1815 case ISD::LOAD:
1816 return LowerLoad(Op, DAG);
1817 case ISD::STORE:
1818 return LowerStore(Op, DAG);
1819 case ISD::CTPOP:
1820 case ISD::CTLZ:
1821 case ISD::CTTZ:
1822 return DAG.UnrollVectorOp(Op.getNode());
1823 case ISD::CLEAR_CACHE:
1824 // Report this as a diagnostic rather than aborting, like the other
1825 // unsupported features in this target. Pass the chain through so that
1826 // codegen can reach the point where the diagnostic is emitted.
1827 fail(SDLoc(Op), DAG, "llvm.clear_cache is not supported on wasm");
1828 return Op.getOperand(0);
1829 case ISD::SMUL_LOHI:
1830 case ISD::UMUL_LOHI:
1831 return LowerMUL_LOHI(Op, DAG);
1832 case ISD::UADDO:
1833 return LowerUADDO(Op, DAG);
1834 }
1835}
1836
1840
1841 return false;
1842}
1843
1844static std::optional<unsigned> IsWebAssemblyLocal(SDValue Op,
1845 SelectionDAG &DAG) {
1847 if (!FI)
1848 return std::nullopt;
1849
1850 auto &MF = DAG.getMachineFunction();
1852}
1853
1854SDValue WebAssemblyTargetLowering::LowerStore(SDValue Op,
1855 SelectionDAG &DAG) const {
1856 SDLoc DL(Op);
1857 StoreSDNode *SN = cast<StoreSDNode>(Op.getNode());
1858 const SDValue &Value = SN->getValue();
1859 const SDValue &Base = SN->getBasePtr();
1860 const SDValue &Offset = SN->getOffset();
1861
1863 if (!Offset->isUndef())
1864 report_fatal_error("unexpected offset when storing to webassembly global",
1865 false);
1866
1867 SDVTList Tys = DAG.getVTList(MVT::Other);
1868 SDValue Ops[] = {SN->getChain(), Value, Base};
1869 return DAG.getMemIntrinsicNode(WebAssemblyISD::GLOBAL_SET, DL, Tys, Ops,
1870 SN->getMemoryVT(), SN->getMemOperand());
1871 }
1872
1873 if (std::optional<unsigned> Local = IsWebAssemblyLocal(Base, DAG)) {
1874 if (!Offset->isUndef())
1875 report_fatal_error("unexpected offset when storing to webassembly local",
1876 false);
1877
1878 SDValue Idx = DAG.getTargetConstant(*Local, Base, MVT::i32);
1879 SDVTList Tys = DAG.getVTList(MVT::Other); // The chain.
1880 SDValue Ops[] = {SN->getChain(), Idx, Value};
1881 return DAG.getNode(WebAssemblyISD::LOCAL_SET, DL, Tys, Ops);
1882 }
1883
1886 "Encountered an unlowerable store to the wasm_var address space",
1887 false);
1888
1889 return Op;
1890}
1891
1892SDValue WebAssemblyTargetLowering::LowerLoad(SDValue Op,
1893 SelectionDAG &DAG) const {
1894 SDLoc DL(Op);
1895 LoadSDNode *LN = cast<LoadSDNode>(Op.getNode());
1896 const SDValue &Base = LN->getBasePtr();
1897 const SDValue &Offset = LN->getOffset();
1898
1900 if (!Offset->isUndef())
1902 "unexpected offset when loading from webassembly global", false);
1903
1904 SDVTList Tys = DAG.getVTList(LN->getValueType(0), MVT::Other);
1905 SDValue Ops[] = {LN->getChain(), Base};
1906 return DAG.getMemIntrinsicNode(WebAssemblyISD::GLOBAL_GET, DL, Tys, Ops,
1907 LN->getMemoryVT(), LN->getMemOperand());
1908 }
1909
1910 if (std::optional<unsigned> Local = IsWebAssemblyLocal(Base, DAG)) {
1911 if (!Offset->isUndef())
1913 "unexpected offset when loading from webassembly local", false);
1914
1915 SDValue Idx = DAG.getTargetConstant(*Local, Base, MVT::i32);
1916 EVT LocalVT = LN->getValueType(0);
1917 return DAG.getNode(WebAssemblyISD::LOCAL_GET, DL, {LocalVT, MVT::Other},
1918 {LN->getChain(), Idx});
1919 }
1920
1923 "Encountered an unlowerable load from the wasm_var address space",
1924 false);
1925
1926 return Op;
1927}
1928
1929SDValue WebAssemblyTargetLowering::LowerMUL_LOHI(SDValue Op,
1930 SelectionDAG &DAG) const {
1931 assert(Subtarget->hasWideArithmetic());
1932 assert(Op.getValueType() == MVT::i64);
1933 SDLoc DL(Op);
1934 unsigned Opcode;
1935 switch (Op.getOpcode()) {
1936 case ISD::UMUL_LOHI:
1937 Opcode = WebAssemblyISD::I64_MUL_WIDE_U;
1938 break;
1939 case ISD::SMUL_LOHI:
1940 Opcode = WebAssemblyISD::I64_MUL_WIDE_S;
1941 break;
1942 default:
1943 llvm_unreachable("unexpected opcode");
1944 }
1945 SDValue LHS = Op.getOperand(0);
1946 SDValue RHS = Op.getOperand(1);
1947 SDValue Lo =
1948 DAG.getNode(Opcode, DL, DAG.getVTList(MVT::i64, MVT::i64), LHS, RHS);
1949 SDValue Hi(Lo.getNode(), 1);
1950 SDValue Ops[] = {Lo, Hi};
1951 return DAG.getMergeValues(Ops, DL);
1952}
1953
1954// Lowers `UADDO` intrinsics to an `i64.add128` instruction when it's enabled.
1955//
1956// This enables generating a single wasm instruction for this operation where
1957// the upper half of both operands are constant zeros. The upper half of the
1958// result is then whether the overflow happened.
1959SDValue WebAssemblyTargetLowering::LowerUADDO(SDValue Op,
1960 SelectionDAG &DAG) const {
1961 assert(Subtarget->hasWideArithmetic());
1962 assert(Op.getValueType() == MVT::i64);
1963 assert(Op.getOpcode() == ISD::UADDO);
1964 SDLoc DL(Op);
1965 SDValue LHS = Op.getOperand(0);
1966 SDValue RHS = Op.getOperand(1);
1967 SDValue Zero = DAG.getConstant(0, DL, MVT::i64);
1968 SDValue Result =
1969 DAG.getNode(WebAssemblyISD::I64_ADD128, DL,
1970 DAG.getVTList(MVT::i64, MVT::i64), LHS, Zero, RHS, Zero);
1971 SDValue CarryI64(Result.getNode(), 1);
1972 SDValue CarryI32 = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, CarryI64);
1973 SDValue Ops[] = {Result, CarryI32};
1974 return DAG.getMergeValues(Ops, DL);
1975}
1976
1977SDValue WebAssemblyTargetLowering::Replace128Op(SDNode *N,
1978 SelectionDAG &DAG) const {
1979 assert(Subtarget->hasWideArithmetic());
1980 assert(N->getValueType(0) == MVT::i128);
1981 SDLoc DL(N);
1982 unsigned Opcode;
1983 switch (N->getOpcode()) {
1984 case ISD::ADD:
1985 Opcode = WebAssemblyISD::I64_ADD128;
1986 break;
1987 case ISD::SUB:
1988 Opcode = WebAssemblyISD::I64_SUB128;
1989 break;
1990 default:
1991 llvm_unreachable("unexpected opcode");
1992 }
1993 SDValue LHS = N->getOperand(0);
1994 SDValue RHS = N->getOperand(1);
1995
1996 SDValue C0 = DAG.getConstant(0, DL, MVT::i64);
1997 SDValue C1 = DAG.getConstant(1, DL, MVT::i64);
1998 SDValue LHS_0 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, LHS, C0);
1999 SDValue LHS_1 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, LHS, C1);
2000 SDValue RHS_0 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, RHS, C0);
2001 SDValue RHS_1 = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, RHS, C1);
2002 SDValue Result_LO = DAG.getNode(Opcode, DL, DAG.getVTList(MVT::i64, MVT::i64),
2003 LHS_0, LHS_1, RHS_0, RHS_1);
2004 SDValue Result_HI(Result_LO.getNode(), 1);
2005 return DAG.getNode(ISD::BUILD_PAIR, DL, N->getVTList(), Result_LO, Result_HI);
2006}
2007
2008SDValue WebAssemblyTargetLowering::LowerCopyToReg(SDValue Op,
2009 SelectionDAG &DAG) const {
2010 SDValue Src = Op.getOperand(2);
2011 if (isa<FrameIndexSDNode>(Src.getNode())) {
2012 // CopyToReg nodes don't support FrameIndex operands. Other targets select
2013 // the FI to some LEA-like instruction, but since we don't have that, we
2014 // need to insert some kind of instruction that can take an FI operand and
2015 // produces a value usable by CopyToReg (i.e. in a vreg). So insert a dummy
2016 // local.copy between Op and its FI operand.
2017 SDValue Chain = Op.getOperand(0);
2018 SDLoc DL(Op);
2019 Register Reg = cast<RegisterSDNode>(Op.getOperand(1))->getReg();
2020 EVT VT = Src.getValueType();
2021 SDValue Copy(DAG.getMachineNode(VT == MVT::i32 ? WebAssembly::COPY_I32
2022 : WebAssembly::COPY_I64,
2023 DL, VT, Src),
2024 0);
2025 return Op.getNode()->getNumValues() == 1
2026 ? DAG.getCopyToReg(Chain, DL, Reg, Copy)
2027 : DAG.getCopyToReg(Chain, DL, Reg, Copy,
2028 Op.getNumOperands() == 4 ? Op.getOperand(3)
2029 : SDValue());
2030 }
2031 return SDValue();
2032}
2033
2034SDValue WebAssemblyTargetLowering::LowerFrameIndex(SDValue Op,
2035 SelectionDAG &DAG) const {
2036 int FI = cast<FrameIndexSDNode>(Op)->getIndex();
2037 return DAG.getTargetFrameIndex(FI, Op.getValueType());
2038}
2039
2040SDValue WebAssemblyTargetLowering::LowerRETURNADDR(SDValue Op,
2041 SelectionDAG &DAG) const {
2042 SDLoc DL(Op);
2043
2044 if (!Subtarget->getTargetTriple().isOSEmscripten()) {
2045 fail(DL, DAG,
2046 "Non-Emscripten WebAssembly hasn't implemented "
2047 "__builtin_return_address");
2048 return SDValue();
2049 }
2050
2051 unsigned Depth = Op.getConstantOperandVal(0);
2052 MakeLibCallOptions CallOptions;
2053 return makeLibCall(DAG, RTLIB::RETURN_ADDRESS, Op.getValueType(),
2054 {DAG.getConstant(Depth, DL, MVT::i32)}, CallOptions, DL)
2055 .first;
2056}
2057
2058SDValue WebAssemblyTargetLowering::LowerFRAMEADDR(SDValue Op,
2059 SelectionDAG &DAG) const {
2060 // Non-zero depths are not supported by WebAssembly currently. Use the
2061 // legalizer's default expansion, which is to return 0 (what this function is
2062 // documented to do).
2063 if (Op.getConstantOperandVal(0) > 0)
2064 return SDValue();
2065
2067 EVT VT = Op.getValueType();
2068 Register FP =
2069 Subtarget->getRegisterInfo()->getFrameRegister(DAG.getMachineFunction());
2070 return DAG.getCopyFromReg(DAG.getEntryNode(), SDLoc(Op), FP, VT);
2071}
2072
2073SDValue
2074WebAssemblyTargetLowering::LowerGlobalTLSAddress(SDValue Op,
2075 SelectionDAG &DAG) const {
2076 SDLoc DL(Op);
2077 const auto *GA = cast<GlobalAddressSDNode>(Op);
2078
2080 if (!MF.getSubtarget<WebAssemblySubtarget>().hasBulkMemory())
2081 report_fatal_error("cannot use thread-local storage without bulk memory",
2082 false);
2083
2084 const GlobalValue *GV = GA->getGlobal();
2085
2086 // Currently only Emscripten supports dynamic linking with threads. Therefore,
2087 // on other targets, if we have thread-local storage, only the local-exec
2088 // model is possible.
2089 auto model = Subtarget->getTargetTriple().isOSEmscripten()
2090 ? GV->getThreadLocalMode()
2092
2093 // Unsupported TLS modes
2096
2097 if (model == GlobalValue::LocalExecTLSModel ||
2100 getTargetMachine().shouldAssumeDSOLocal(GV))) {
2101 // For DSO-local TLS variables we use offset from __tls_base, or
2102 // __wasm_get_tls_base() if using libcall thread context.
2103
2104 MVT PtrVT = getPointerTy(DAG.getDataLayout());
2105 SDValue BaseAddr(WebAssembly::getTLSBase(DAG, DL, Subtarget), 0);
2106
2107 SDValue TLSOffset = DAG.getTargetGlobalAddress(
2108 GV, DL, PtrVT, GA->getOffset(), WebAssemblyII::MO_TLS_BASE_REL);
2109 SDValue SymOffset =
2110 DAG.getNode(WebAssemblyISD::WrapperREL, DL, PtrVT, TLSOffset);
2111
2112 return DAG.getNode(ISD::ADD, DL, PtrVT, BaseAddr, SymOffset);
2113 }
2114
2116
2117 EVT VT = Op.getValueType();
2118 return DAG.getNode(WebAssemblyISD::Wrapper, DL, VT,
2119 DAG.getTargetGlobalAddress(GA->getGlobal(), DL, VT,
2120 GA->getOffset(),
2122}
2123
2124SDValue WebAssemblyTargetLowering::LowerGlobalAddress(SDValue Op,
2125 SelectionDAG &DAG) const {
2126 SDLoc DL(Op);
2127 const auto *GA = cast<GlobalAddressSDNode>(Op);
2128 EVT VT = Op.getValueType();
2129 assert(GA->getTargetFlags() == 0 &&
2130 "Unexpected target flags on generic GlobalAddressSDNode");
2132 fail(DL, DAG, "Invalid address space for WebAssembly target");
2133
2134 unsigned OperandFlags = 0;
2135 const GlobalValue *GV = GA->getGlobal();
2136 // Since WebAssembly tables cannot yet be shared across modules, we don't
2137 // need special treatment for tables in PIC mode.
2138 if (isPositionIndependent() &&
2140 if (getTargetMachine().shouldAssumeDSOLocal(GV)) {
2142 MVT PtrVT = getPointerTy(MF.getDataLayout());
2143 const char *BaseName;
2144 if (GV->getValueType()->isFunctionTy()) {
2145 BaseName = MF.createExternalSymbolName("__table_base");
2147 } else {
2148 BaseName = MF.createExternalSymbolName("__memory_base");
2150 }
2151 SDValue BaseAddr =
2152 DAG.getNode(WebAssemblyISD::Wrapper, DL, PtrVT,
2153 DAG.getTargetExternalSymbol(BaseName, PtrVT));
2154
2155 SDValue SymAddr = DAG.getNode(
2156 WebAssemblyISD::WrapperREL, DL, VT,
2157 DAG.getTargetGlobalAddress(GA->getGlobal(), DL, VT, GA->getOffset(),
2158 OperandFlags));
2159
2160 return DAG.getNode(ISD::ADD, DL, VT, BaseAddr, SymAddr);
2161 }
2163 }
2164
2165 return DAG.getNode(WebAssemblyISD::Wrapper, DL, VT,
2166 DAG.getTargetGlobalAddress(GA->getGlobal(), DL, VT,
2167 GA->getOffset(), OperandFlags));
2168}
2169
2170SDValue
2171WebAssemblyTargetLowering::LowerExternalSymbol(SDValue Op,
2172 SelectionDAG &DAG) const {
2173 SDLoc DL(Op);
2174 const auto *ES = cast<ExternalSymbolSDNode>(Op);
2175 EVT VT = Op.getValueType();
2176 assert(ES->getTargetFlags() == 0 &&
2177 "Unexpected target flags on generic ExternalSymbolSDNode");
2178 return DAG.getNode(WebAssemblyISD::Wrapper, DL, VT,
2179 DAG.getTargetExternalSymbol(ES->getSymbol(), VT));
2180}
2181
2182SDValue WebAssemblyTargetLowering::LowerJumpTable(SDValue Op,
2183 SelectionDAG &DAG) const {
2184 // There's no need for a Wrapper node because we always incorporate a jump
2185 // table operand into a BR_TABLE instruction, rather than ever
2186 // materializing it in a register.
2187 const JumpTableSDNode *JT = cast<JumpTableSDNode>(Op);
2188 return DAG.getTargetJumpTable(JT->getIndex(), Op.getValueType(),
2189 JT->getTargetFlags());
2190}
2191
2192SDValue WebAssemblyTargetLowering::LowerBR_JT(SDValue Op,
2193 SelectionDAG &DAG) const {
2194 SDLoc DL(Op);
2195 SDValue Chain = Op.getOperand(0);
2196 const auto *JT = cast<JumpTableSDNode>(Op.getOperand(1));
2197 SDValue Index = Op.getOperand(2);
2198 assert(JT->getTargetFlags() == 0 && "WebAssembly doesn't set target flags");
2199
2201 Ops.push_back(Chain);
2202 Ops.push_back(Index);
2203
2204 MachineJumpTableInfo *MJTI = DAG.getMachineFunction().getJumpTableInfo();
2205 const auto &MBBs = MJTI->getJumpTables()[JT->getIndex()].MBBs;
2206
2207 // Add an operand for each case.
2208 for (auto *MBB : MBBs)
2209 Ops.push_back(DAG.getBasicBlock(MBB));
2210
2211 // Add the first MBB as a dummy default target for now. This will be replaced
2212 // with the proper default target (and the preceding range check eliminated)
2213 // if possible by WebAssemblyFixBrTableDefaults.
2214 Ops.push_back(DAG.getBasicBlock(*MBBs.begin()));
2215 return DAG.getNode(WebAssemblyISD::BR_TABLE, DL, MVT::Other, Ops);
2216}
2217
2218SDValue WebAssemblyTargetLowering::LowerVASTART(SDValue Op,
2219 SelectionDAG &DAG) const {
2220 SDLoc DL(Op);
2221 EVT PtrVT = getPointerTy(DAG.getMachineFunction().getDataLayout());
2222
2223 auto *MFI = DAG.getMachineFunction().getInfo<WebAssemblyFunctionInfo>();
2224 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
2225
2226 SDValue ArgN = DAG.getCopyFromReg(DAG.getEntryNode(), DL,
2227 MFI->getVarargBufferVreg(), PtrVT);
2228 return DAG.getStore(Op.getOperand(0), DL, ArgN, Op.getOperand(1),
2229 MachinePointerInfo(SV));
2230}
2231
2232SDValue WebAssemblyTargetLowering::LowerIntrinsic(SDValue Op,
2233 SelectionDAG &DAG) const {
2235 unsigned IntNo;
2236 switch (Op.getOpcode()) {
2239 IntNo = Op.getConstantOperandVal(1);
2240 break;
2242 IntNo = Op.getConstantOperandVal(0);
2243 break;
2244 default:
2245 llvm_unreachable("Invalid intrinsic");
2246 }
2247 SDLoc DL(Op);
2248
2249 switch (IntNo) {
2250 default:
2251 return SDValue(); // Don't custom lower most intrinsics.
2252
2253 case Intrinsic::wasm_lsda: {
2254 auto PtrVT = getPointerTy(MF.getDataLayout());
2255 const char *SymName = MF.createExternalSymbolName(
2256 "GCC_except_table" + std::to_string(MF.getFunctionNumber()));
2257 if (isPositionIndependent()) {
2258 SDValue Node = DAG.getTargetExternalSymbol(
2259 SymName, PtrVT, WebAssemblyII::MO_MEMORY_BASE_REL);
2260 const char *BaseName = MF.createExternalSymbolName("__memory_base");
2261 SDValue BaseAddr =
2262 DAG.getNode(WebAssemblyISD::Wrapper, DL, PtrVT,
2263 DAG.getTargetExternalSymbol(BaseName, PtrVT));
2264 SDValue SymAddr =
2265 DAG.getNode(WebAssemblyISD::WrapperREL, DL, PtrVT, Node);
2266 return DAG.getNode(ISD::ADD, DL, PtrVT, BaseAddr, SymAddr);
2267 }
2268 SDValue Node = DAG.getTargetExternalSymbol(SymName, PtrVT);
2269 return DAG.getNode(WebAssemblyISD::Wrapper, DL, PtrVT, Node);
2270 }
2271
2272 case Intrinsic::wasm_shuffle: {
2273 // Drop in-chain and replace undefs, but otherwise pass through unchanged
2274 SDValue Ops[18];
2275 size_t OpIdx = 0;
2276 Ops[OpIdx++] = Op.getOperand(1);
2277 Ops[OpIdx++] = Op.getOperand(2);
2278 while (OpIdx < 18) {
2279 const SDValue &MaskIdx = Op.getOperand(OpIdx + 1);
2280 if (MaskIdx.isUndef() || MaskIdx.getNode()->getAsZExtVal() >= 32) {
2281 bool isTarget = MaskIdx.getNode()->getOpcode() == ISD::TargetConstant;
2282 Ops[OpIdx++] = DAG.getConstant(0, DL, MVT::i32, isTarget);
2283 } else {
2284 Ops[OpIdx++] = MaskIdx;
2285 }
2286 }
2287 return DAG.getNode(WebAssemblyISD::SHUFFLE, DL, Op.getValueType(), Ops);
2288 }
2289
2290 case Intrinsic::wasm_funcref_to_ptr: {
2291 // llvm.wasm.funcref.to_ptr only has a defined lowering when its result
2292 // feeds directly into an indirect call. Reaching here means the pointer
2293 // escapes a direct call. We haven't implemented conversion of a funcref
2294 // into a real function pointer so we crash if we get here.
2295 fail(DL, DAG,
2296 "a funcref can only be converted to a pointer to be directly called; "
2297 "the resulting pointer cannot otherwise be used");
2298 return DAG.getPOISON(Op.getValueType());
2299 }
2300
2301 case Intrinsic::thread_pointer: {
2302 return SDValue(WebAssembly::getTLSBase(DAG, DL, Subtarget), 0);
2303 }
2304 }
2305}
2306
2307SDValue
2308WebAssemblyTargetLowering::LowerSIGN_EXTEND_INREG(SDValue Op,
2309 SelectionDAG &DAG) const {
2310 SDLoc DL(Op);
2311 // If sign extension operations are disabled, allow sext_inreg only if operand
2312 // is a vector extract of an i8 or i16 lane. SIMD does not depend on sign
2313 // extension operations, but allowing sext_inreg in this context lets us have
2314 // simple patterns to select extract_lane_s instructions. Expanding sext_inreg
2315 // everywhere would be simpler in this file, but would necessitate large and
2316 // brittle patterns to undo the expansion and select extract_lane_s
2317 // instructions.
2318 assert(!Subtarget->hasSignExt() && Subtarget->hasSIMD128());
2319 if (Op.getOperand(0).getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2320 return SDValue();
2321
2322 const SDValue &Extract = Op.getOperand(0);
2323 MVT VecT = Extract.getOperand(0).getSimpleValueType();
2324 if (VecT.getVectorElementType().getSizeInBits() > 32)
2325 return SDValue();
2326 MVT ExtractedLaneT =
2327 cast<VTSDNode>(Op.getOperand(1).getNode())->getVT().getSimpleVT();
2328 MVT ExtractedVecT =
2329 MVT::getVectorVT(ExtractedLaneT, 128 / ExtractedLaneT.getSizeInBits());
2330 if (ExtractedVecT == VecT)
2331 return Op;
2332
2333 // Bitcast vector to appropriate type to ensure ISel pattern coverage
2334 const SDNode *Index = Extract.getOperand(1).getNode();
2335 if (!isa<ConstantSDNode>(Index))
2336 return SDValue();
2337 unsigned IndexVal = Index->getAsZExtVal();
2338 unsigned Scale =
2339 ExtractedVecT.getVectorNumElements() / VecT.getVectorNumElements();
2340 assert(Scale > 1);
2341 SDValue NewIndex =
2342 DAG.getConstant(IndexVal * Scale, DL, Index->getValueType(0));
2343 SDValue NewExtract = DAG.getNode(
2345 DAG.getBitcast(ExtractedVecT, Extract.getOperand(0)), NewIndex);
2346 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, Op.getValueType(), NewExtract,
2347 Op.getOperand(1));
2348}
2349
2350static SDValue GetExtendHigh(SDValue Op, unsigned UserOpc, EVT VT,
2351 SelectionDAG &DAG) {
2352 SDValue Source = peekThroughBitcasts(Op);
2353 if (Source.getOpcode() != ISD::VECTOR_SHUFFLE)
2354 return SDValue();
2355
2356 assert((UserOpc == WebAssemblyISD::EXTEND_LOW_U ||
2357 UserOpc == WebAssemblyISD::EXTEND_LOW_S) &&
2358 "expected extend_low");
2359 auto *Shuffle = cast<ShuffleVectorSDNode>(Source.getNode());
2360
2361 ArrayRef<int> Mask = Shuffle->getMask();
2362 // Look for a shuffle which moves from the high half to the low half.
2363 size_t FirstIdx = Mask.size() / 2;
2364 for (size_t i = 0; i < Mask.size() / 2; ++i) {
2365 if (Mask[i] != static_cast<int>(FirstIdx + i)) {
2366 return SDValue();
2367 }
2368 }
2369
2370 SDLoc DL(Op);
2371 unsigned Opc = UserOpc == WebAssemblyISD::EXTEND_LOW_S
2372 ? WebAssemblyISD::EXTEND_HIGH_S
2373 : WebAssemblyISD::EXTEND_HIGH_U;
2374 SDValue ShuffleSrc = Shuffle->getOperand(0);
2375 if (Op.getOpcode() == ISD::BITCAST)
2376 ShuffleSrc = DAG.getBitcast(Op.getValueType(), ShuffleSrc);
2377
2378 return DAG.getNode(Opc, DL, VT, ShuffleSrc);
2379}
2380
2381SDValue
2382WebAssemblyTargetLowering::LowerEXTEND_VECTOR_INREG(SDValue Op,
2383 SelectionDAG &DAG) const {
2384 SDLoc DL(Op);
2385 EVT VT = Op.getValueType();
2386 SDValue Src = Op.getOperand(0);
2387 EVT SrcVT = Src.getValueType();
2388
2389 if (SrcVT.getVectorElementType() == MVT::i1 ||
2390 SrcVT.getVectorElementType() == MVT::i64)
2391 return SDValue();
2392
2393 assert(VT.getScalarSizeInBits() % SrcVT.getScalarSizeInBits() == 0 &&
2394 "Unexpected extension factor.");
2395 unsigned Scale = VT.getScalarSizeInBits() / SrcVT.getScalarSizeInBits();
2396
2397 if (Scale != 2 && Scale != 4 && Scale != 8)
2398 return SDValue();
2399
2400 unsigned Ext;
2401 switch (Op.getOpcode()) {
2402 default:
2403 llvm_unreachable("unexpected opcode");
2406 Ext = WebAssemblyISD::EXTEND_LOW_U;
2407 break;
2409 Ext = WebAssemblyISD::EXTEND_LOW_S;
2410 break;
2411 }
2412
2413 if (Scale == 2) {
2414 // See if we can use EXTEND_HIGH.
2415 if (auto ExtendHigh = GetExtendHigh(Op.getOperand(0), Ext, VT, DAG))
2416 return ExtendHigh;
2417 }
2418
2419 SDValue Ret = Src;
2420 while (Scale != 1) {
2421 Ret = DAG.getNode(Ext, DL,
2422 Ret.getValueType()
2425 Ret);
2426 Scale /= 2;
2427 }
2428 assert(Ret.getValueType() == VT);
2429 return Ret;
2430}
2431
2433 SDLoc DL(Op);
2434 if (Op.getValueType() != MVT::v2f64 && Op.getValueType() != MVT::v4f32)
2435 return SDValue();
2436
2437 auto GetConvertedLane = [](SDValue Op, unsigned &Opcode, SDValue &SrcVec,
2438 unsigned &Index) -> bool {
2439 switch (Op.getOpcode()) {
2440 case ISD::SINT_TO_FP:
2441 Opcode = WebAssemblyISD::CONVERT_LOW_S;
2442 break;
2443 case ISD::UINT_TO_FP:
2444 Opcode = WebAssemblyISD::CONVERT_LOW_U;
2445 break;
2446 case ISD::FP_EXTEND:
2447 case ISD::FP16_TO_FP:
2448 Opcode = WebAssemblyISD::PROMOTE_LOW;
2449 break;
2450 default:
2451 return false;
2452 }
2453
2454 auto ExtractVector = Op.getOperand(0);
2455 if (ExtractVector.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2456 return false;
2457
2458 if (!isa<ConstantSDNode>(ExtractVector.getOperand(1).getNode()))
2459 return false;
2460
2461 SrcVec = ExtractVector.getOperand(0);
2462 Index = ExtractVector.getConstantOperandVal(1);
2463 return true;
2464 };
2465
2466 unsigned NumLanes = Op.getValueType() == MVT::v2f64 ? 2 : 4;
2467 unsigned FirstOpcode = 0, SecondOpcode = 0, ThirdOpcode = 0, FourthOpcode = 0;
2468 unsigned FirstIndex = 0, SecondIndex = 0, ThirdIndex = 0, FourthIndex = 0;
2469 SDValue FirstSrcVec, SecondSrcVec, ThirdSrcVec, FourthSrcVec;
2470
2471 if (!GetConvertedLane(Op.getOperand(0), FirstOpcode, FirstSrcVec,
2472 FirstIndex) ||
2473 !GetConvertedLane(Op.getOperand(1), SecondOpcode, SecondSrcVec,
2474 SecondIndex))
2475 return SDValue();
2476
2477 // If we're converting to v4f32, check the third and fourth lanes, too.
2478 if (NumLanes == 4 && (!GetConvertedLane(Op.getOperand(2), ThirdOpcode,
2479 ThirdSrcVec, ThirdIndex) ||
2480 !GetConvertedLane(Op.getOperand(3), FourthOpcode,
2481 FourthSrcVec, FourthIndex)))
2482 return SDValue();
2483
2484 if (FirstOpcode != SecondOpcode)
2485 return SDValue();
2486
2487 // TODO Add an optimization similar to the v2f64 below for shuffling the
2488 // vectors when the lanes are in the wrong order or come from different src
2489 // vectors.
2490 if (NumLanes == 4 &&
2491 (FirstOpcode != ThirdOpcode || FirstOpcode != FourthOpcode ||
2492 FirstSrcVec != SecondSrcVec || FirstSrcVec != ThirdSrcVec ||
2493 FirstSrcVec != FourthSrcVec || FirstIndex != 0 || SecondIndex != 1 ||
2494 ThirdIndex != 2 || FourthIndex != 3))
2495 return SDValue();
2496
2497 MVT ExpectedSrcVT;
2498 switch (FirstOpcode) {
2499 case WebAssemblyISD::CONVERT_LOW_S:
2500 case WebAssemblyISD::CONVERT_LOW_U:
2501 ExpectedSrcVT = MVT::v4i32;
2502 break;
2503 case WebAssemblyISD::PROMOTE_LOW:
2504 ExpectedSrcVT = NumLanes == 2 ? MVT::v4f32 : MVT::v8i16;
2505 break;
2506 }
2507 if (FirstSrcVec.getValueType() != ExpectedSrcVT)
2508 return SDValue();
2509
2510 auto Src = FirstSrcVec;
2511 if (NumLanes == 2 &&
2512 (FirstIndex != 0 || SecondIndex != 1 || FirstSrcVec != SecondSrcVec)) {
2513 // Shuffle the source vector so that the converted lanes are the low lanes.
2514 Src = DAG.getVectorShuffle(ExpectedSrcVT, DL, FirstSrcVec, SecondSrcVec,
2515 {static_cast<int>(FirstIndex),
2516 static_cast<int>(SecondIndex) + 4, -1, -1});
2517 }
2518 return DAG.getNode(FirstOpcode, DL, NumLanes == 2 ? MVT::v2f64 : MVT::v4f32,
2519 Src);
2520}
2521
2522SDValue WebAssemblyTargetLowering::LowerBUILD_VECTOR(SDValue Op,
2523 SelectionDAG &DAG) const {
2524 MVT VT = Op.getSimpleValueType();
2525 if (VT == MVT::v8f16) {
2526 // BUILD_VECTOR can't handle FP16 operands since Wasm doesn't have a scalar
2527 // FP16 type, so cast them to I16s.
2528 MVT IVT = VT.changeVectorElementType(MVT::i16);
2530 for (unsigned I = 0, E = Op.getNumOperands(); I < E; ++I)
2531 NewOps.push_back(DAG.getBitcast(MVT::i16, Op.getOperand(I)));
2532 SDValue Res = DAG.getNode(ISD::BUILD_VECTOR, SDLoc(), IVT, NewOps);
2533 return DAG.getBitcast(VT, Res);
2534 }
2535
2536 if (auto ConvertLow = LowerConvertLow(Op, DAG))
2537 return ConvertLow;
2538
2539 SDLoc DL(Op);
2540 const EVT VecT = Op.getValueType();
2541 const EVT LaneT = Op.getOperand(0).getValueType();
2542 const size_t Lanes = Op.getNumOperands();
2543 bool CanSwizzle = VecT == MVT::v16i8;
2544
2545 // BUILD_VECTORs are lowered to the instruction that initializes the highest
2546 // possible number of lanes at once followed by a sequence of replace_lane
2547 // instructions to individually initialize any remaining lanes.
2548
2549 // TODO: Tune this. For example, lanewise swizzling is very expensive, so
2550 // swizzled lanes should be given greater weight.
2551
2552 // TODO: Investigate looping rather than always extracting/replacing specific
2553 // lanes to fill gaps.
2554
2555 auto IsConstant = [](const SDValue &V) {
2556 return V.getOpcode() == ISD::Constant || V.getOpcode() == ISD::ConstantFP;
2557 };
2558
2559 // Returns the source vector and index vector pair if they exist. Checks for:
2560 // (extract_vector_elt
2561 // $src,
2562 // (sign_extend_inreg (extract_vector_elt $indices, $i))
2563 // )
2564 auto GetSwizzleSrcs = [](size_t I, const SDValue &Lane) {
2565 auto Bail = std::make_pair(SDValue(), SDValue());
2566 if (Lane->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2567 return Bail;
2568 const SDValue &SwizzleSrc = Lane->getOperand(0);
2569 const SDValue &IndexExt = Lane->getOperand(1);
2570 if (IndexExt->getOpcode() != ISD::SIGN_EXTEND_INREG)
2571 return Bail;
2572 const SDValue &Index = IndexExt->getOperand(0);
2573 if (Index->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2574 return Bail;
2575 const SDValue &SwizzleIndices = Index->getOperand(0);
2576 if (SwizzleSrc.getValueType() != MVT::v16i8 ||
2577 SwizzleIndices.getValueType() != MVT::v16i8 ||
2578 Index->getOperand(1)->getOpcode() != ISD::Constant ||
2579 Index->getConstantOperandVal(1) != I)
2580 return Bail;
2581 return std::make_pair(SwizzleSrc, SwizzleIndices);
2582 };
2583
2584 // If the lane is extracted from another vector at a constant index, return
2585 // that vector. The source vector must not have more lanes than the dest
2586 // because the shufflevector indices are in terms of the destination lanes and
2587 // would not be able to address the smaller individual source lanes.
2588 auto GetShuffleSrc = [&](const SDValue &Lane) {
2589 if (Lane->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
2590 return SDValue();
2591 if (!isa<ConstantSDNode>(Lane->getOperand(1).getNode()))
2592 return SDValue();
2593 if (Lane->getOperand(0).getValueType().getVectorNumElements() >
2594 VecT.getVectorNumElements())
2595 return SDValue();
2596 return Lane->getOperand(0);
2597 };
2598
2599 using ValueEntry = std::pair<SDValue, size_t>;
2600 SmallVector<ValueEntry, 16> SplatValueCounts;
2601
2602 using SwizzleEntry = std::pair<std::pair<SDValue, SDValue>, size_t>;
2603 SmallVector<SwizzleEntry, 16> SwizzleCounts;
2604
2605 using ShuffleEntry = std::pair<SDValue, size_t>;
2606 SmallVector<ShuffleEntry, 16> ShuffleCounts;
2607
2608 auto AddCount = [](auto &Counts, const auto &Val) {
2609 auto CountIt =
2610 llvm::find_if(Counts, [&Val](auto E) { return E.first == Val; });
2611 if (CountIt == Counts.end()) {
2612 Counts.emplace_back(Val, 1);
2613 } else {
2614 CountIt->second++;
2615 }
2616 };
2617
2618 auto GetMostCommon = [](auto &Counts) {
2619 auto CommonIt = llvm::max_element(Counts, llvm::less_second());
2620 assert(CommonIt != Counts.end() && "Unexpected all-undef build_vector");
2621 return *CommonIt;
2622 };
2623
2624 size_t NumConstantLanes = 0;
2625
2626 // Count eligible lanes for each type of vector creation op
2627 for (size_t I = 0; I < Lanes; ++I) {
2628 const SDValue &Lane = Op->getOperand(I);
2629 if (Lane.isUndef())
2630 continue;
2631
2632 AddCount(SplatValueCounts, Lane);
2633
2634 if (IsConstant(Lane))
2635 NumConstantLanes++;
2636 if (auto ShuffleSrc = GetShuffleSrc(Lane))
2637 AddCount(ShuffleCounts, ShuffleSrc);
2638 if (CanSwizzle) {
2639 auto SwizzleSrcs = GetSwizzleSrcs(I, Lane);
2640 if (SwizzleSrcs.first)
2641 AddCount(SwizzleCounts, SwizzleSrcs);
2642 }
2643 }
2644
2645 SDValue SplatValue;
2646 size_t NumSplatLanes;
2647 std::tie(SplatValue, NumSplatLanes) = GetMostCommon(SplatValueCounts);
2648
2649 SDValue SwizzleSrc;
2650 SDValue SwizzleIndices;
2651 size_t NumSwizzleLanes = 0;
2652 if (SwizzleCounts.size())
2653 std::forward_as_tuple(std::tie(SwizzleSrc, SwizzleIndices),
2654 NumSwizzleLanes) = GetMostCommon(SwizzleCounts);
2655
2656 // Shuffles can draw from up to two vectors, so find the two most common
2657 // sources.
2658 SDValue ShuffleSrc1, ShuffleSrc2;
2659 size_t NumShuffleLanes = 0;
2660 if (ShuffleCounts.size()) {
2661 std::tie(ShuffleSrc1, NumShuffleLanes) = GetMostCommon(ShuffleCounts);
2662 llvm::erase_if(ShuffleCounts,
2663 [&](const auto &Pair) { return Pair.first == ShuffleSrc1; });
2664 }
2665 if (ShuffleCounts.size()) {
2666 size_t AdditionalShuffleLanes;
2667 std::tie(ShuffleSrc2, AdditionalShuffleLanes) =
2668 GetMostCommon(ShuffleCounts);
2669 NumShuffleLanes += AdditionalShuffleLanes;
2670 }
2671
2672 // Predicate returning true if the lane is properly initialized by the
2673 // original instruction
2674 std::function<bool(size_t, const SDValue &)> IsLaneConstructed;
2675 SDValue Result;
2676 // Prefer swizzles over shuffles over vector consts over splats
2677 if (NumSwizzleLanes >= NumShuffleLanes &&
2678 NumSwizzleLanes >= NumConstantLanes && NumSwizzleLanes >= NumSplatLanes) {
2679 Result = DAG.getNode(WebAssemblyISD::SWIZZLE, DL, VecT, SwizzleSrc,
2680 SwizzleIndices);
2681 auto Swizzled = std::make_pair(SwizzleSrc, SwizzleIndices);
2682 IsLaneConstructed = [&, Swizzled](size_t I, const SDValue &Lane) {
2683 return Swizzled == GetSwizzleSrcs(I, Lane);
2684 };
2685 } else if (NumShuffleLanes >= NumConstantLanes &&
2686 NumShuffleLanes >= NumSplatLanes) {
2687 size_t DestLaneSize = VecT.getVectorElementType().getFixedSizeInBits() / 8;
2688 size_t DestLaneCount = VecT.getVectorNumElements();
2689 size_t Scale1 = 1;
2690 size_t Scale2 = 1;
2691 SDValue Src1 = ShuffleSrc1;
2692 SDValue Src2 = ShuffleSrc2 ? ShuffleSrc2 : DAG.getUNDEF(VecT);
2693 if (Src1.getValueType() != VecT) {
2694 size_t LaneSize =
2695 Src1.getValueType().getVectorElementType().getFixedSizeInBits() / 8;
2696 assert(LaneSize > DestLaneSize);
2697 Scale1 = LaneSize / DestLaneSize;
2698 Src1 = DAG.getBitcast(VecT, Src1);
2699 }
2700 if (Src2.getValueType() != VecT) {
2701 size_t LaneSize =
2702 Src2.getValueType().getVectorElementType().getFixedSizeInBits() / 8;
2703 assert(LaneSize > DestLaneSize);
2704 Scale2 = LaneSize / DestLaneSize;
2705 Src2 = DAG.getBitcast(VecT, Src2);
2706 }
2707
2708 int Mask[16];
2709 assert(DestLaneCount <= 16);
2710 for (size_t I = 0; I < DestLaneCount; ++I) {
2711 const SDValue &Lane = Op->getOperand(I);
2712 SDValue Src = GetShuffleSrc(Lane);
2713 if (Src == ShuffleSrc1) {
2714 Mask[I] = Lane->getConstantOperandVal(1) * Scale1;
2715 } else if (Src && Src == ShuffleSrc2) {
2716 Mask[I] = DestLaneCount + Lane->getConstantOperandVal(1) * Scale2;
2717 } else {
2718 Mask[I] = -1;
2719 }
2720 }
2721 ArrayRef<int> MaskRef(Mask, DestLaneCount);
2722 Result = DAG.getVectorShuffle(VecT, DL, Src1, Src2, MaskRef);
2723 IsLaneConstructed = [&](size_t, const SDValue &Lane) {
2724 auto Src = GetShuffleSrc(Lane);
2725 return Src == ShuffleSrc1 || (Src && Src == ShuffleSrc2);
2726 };
2727 } else if (NumConstantLanes >= NumSplatLanes) {
2728 SmallVector<SDValue, 16> ConstLanes;
2729 for (const SDValue &Lane : Op->op_values()) {
2730 if (IsConstant(Lane)) {
2731 // Values may need to be fixed so that they will sign extend to be
2732 // within the expected range during ISel. Check whether the value is in
2733 // bounds based on the lane bit width and if it is out of bounds, lop
2734 // off the extra bits.
2735 uint64_t LaneBits = 128 / Lanes;
2736 if (auto *Const = dyn_cast<ConstantSDNode>(Lane.getNode())) {
2737 ConstLanes.push_back(DAG.getConstant(
2738 Const->getAPIntValue().trunc(LaneBits).getZExtValue(),
2739 SDLoc(Lane), LaneT));
2740 } else {
2741 ConstLanes.push_back(Lane);
2742 }
2743 } else if (LaneT.isFloatingPoint()) {
2744 ConstLanes.push_back(DAG.getConstantFP(0, DL, LaneT));
2745 } else {
2746 ConstLanes.push_back(DAG.getConstant(0, DL, LaneT));
2747 }
2748 }
2749 Result = DAG.getBuildVector(VecT, DL, ConstLanes);
2750 IsLaneConstructed = [&IsConstant](size_t _, const SDValue &Lane) {
2751 return IsConstant(Lane);
2752 };
2753 } else {
2754 size_t DestLaneSize = VecT.getVectorElementType().getFixedSizeInBits();
2755 if (NumSplatLanes == 1 && Op->getOperand(0) == SplatValue &&
2756 (DestLaneSize == 32 || DestLaneSize == 64)) {
2757 // Could be selected to load_zero.
2758 Result = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VecT, SplatValue);
2759 } else {
2760 // Use a splat (which might be selected as a load splat)
2761 Result = DAG.getSplatBuildVector(VecT, DL, SplatValue);
2762 }
2763 IsLaneConstructed = [&SplatValue](size_t _, const SDValue &Lane) {
2764 return Lane == SplatValue;
2765 };
2766 }
2767
2768 assert(Result);
2769 assert(IsLaneConstructed);
2770
2771 // Add replace_lane instructions for any unhandled values
2772 for (size_t I = 0; I < Lanes; ++I) {
2773 const SDValue &Lane = Op->getOperand(I);
2774 if (!Lane.isUndef() && !IsLaneConstructed(I, Lane))
2775 Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VecT, Result, Lane,
2776 DAG.getConstant(I, DL, MVT::i32));
2777 }
2778
2779 return Result;
2780}
2781
2782SDValue
2783WebAssemblyTargetLowering::LowerVECTOR_SHUFFLE(SDValue Op,
2784 SelectionDAG &DAG) const {
2785 SDLoc DL(Op);
2786 ArrayRef<int> Mask = cast<ShuffleVectorSDNode>(Op.getNode())->getMask();
2787 MVT VecType = Op.getOperand(0).getSimpleValueType();
2788 assert(VecType.is128BitVector() && "Unexpected shuffle vector type");
2789 size_t LaneBytes = VecType.getVectorElementType().getSizeInBits() / 8;
2790
2791 // Space for two vector args and sixteen mask indices
2792 SDValue Ops[18];
2793 size_t OpIdx = 0;
2794 Ops[OpIdx++] = Op.getOperand(0);
2795 Ops[OpIdx++] = Op.getOperand(1);
2796
2797 // Expand mask indices to byte indices and materialize them as operands
2798 for (int M : Mask) {
2799 for (size_t J = 0; J < LaneBytes; ++J) {
2800 // Lower undefs (represented by -1 in mask) to {0..J}, which use a
2801 // whole lane of vector input, to allow further reduction at VM. E.g.
2802 // match an 8x16 byte shuffle to an equivalent cheaper 32x4 shuffle.
2803 uint64_t ByteIndex = M == -1 ? J : (uint64_t)M * LaneBytes + J;
2804 Ops[OpIdx++] = DAG.getConstant(ByteIndex, DL, MVT::i32);
2805 }
2806 }
2807
2808 return DAG.getNode(WebAssemblyISD::SHUFFLE, DL, Op.getValueType(), Ops);
2809}
2810
2811SDValue WebAssemblyTargetLowering::LowerSETCC(SDValue Op,
2812 SelectionDAG &DAG) const {
2813 SDLoc DL(Op);
2814 // The legalizer does not know how to expand the unsupported comparison modes
2815 // of i64x2 vectors, so we manually unroll them here.
2816 assert(Op->getOperand(0)->getSimpleValueType(0) == MVT::v2i64);
2818 DAG.ExtractVectorElements(Op->getOperand(0), LHS);
2819 DAG.ExtractVectorElements(Op->getOperand(1), RHS);
2820 const SDValue &CC = Op->getOperand(2);
2821 auto MakeLane = [&](unsigned I) {
2822 return DAG.getNode(ISD::SELECT_CC, DL, MVT::i64, LHS[I], RHS[I],
2823 DAG.getConstant(uint64_t(-1), DL, MVT::i64),
2824 DAG.getConstant(uint64_t(0), DL, MVT::i64), CC);
2825 };
2826 return DAG.getBuildVector(Op->getValueType(0), DL,
2827 {MakeLane(0), MakeLane(1)});
2828}
2829
2830SDValue
2831WebAssemblyTargetLowering::LowerAccessVectorElement(SDValue Op,
2832 SelectionDAG &DAG) const {
2833 if (Op.getOpcode() == ISD::INSERT_VECTOR_ELT &&
2834 Op.getValueType() == MVT::v8f16) {
2835 // INSERT_VECTOR_ELT can't handle FP16 operands since Wasm doesn't have a
2836 // scalar FP16 type, so cast them to I16s.
2837 SDLoc DL(Op);
2838 SDValue IntVector = DAG.getBitcast(MVT::v8i16, Op.getOperand(0));
2839 SDValue IntElement = DAG.getBitcast(MVT::i16, Op.getOperand(1));
2840 SDValue Inserted = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, MVT::v8i16,
2841 IntVector, IntElement, Op.getOperand(2));
2842 return DAG.getBitcast(MVT::v8f16, Inserted);
2843 }
2844
2845 // Allow constant lane indices, expand variable lane indices
2846 SDNode *IdxNode = Op.getOperand(Op.getNumOperands() - 1).getNode();
2847 if (isa<ConstantSDNode>(IdxNode)) {
2848 // Ensure the index type is i32 to match the tablegen patterns
2849 uint64_t Idx = IdxNode->getAsZExtVal();
2850 SmallVector<SDValue, 3> Ops(Op.getNode()->ops());
2851 Ops[Op.getNumOperands() - 1] =
2852 DAG.getConstant(Idx, SDLoc(IdxNode), MVT::i32);
2853 return DAG.getNode(Op.getOpcode(), SDLoc(Op), Op.getValueType(), Ops);
2854 }
2855 // Perform default expansion
2856 return SDValue();
2857}
2858
2860 EVT LaneT = Op.getSimpleValueType().getVectorElementType();
2861 // 32-bit and 64-bit unrolled shifts will have proper semantics
2862 if (LaneT.bitsGE(MVT::i32))
2863 return DAG.UnrollVectorOp(Op.getNode());
2864 // Otherwise mask the shift value to get proper semantics from 32-bit shift
2865 SDLoc DL(Op);
2866 size_t NumLanes = Op.getSimpleValueType().getVectorNumElements();
2867 SDValue Mask = DAG.getConstant(LaneT.getSizeInBits() - 1, DL, MVT::i32);
2868 unsigned ShiftOpcode = Op.getOpcode();
2869 SmallVector<SDValue, 16> ShiftedElements;
2870 DAG.ExtractVectorElements(Op.getOperand(0), ShiftedElements, 0, 0, MVT::i32);
2871 SmallVector<SDValue, 16> ShiftElements;
2872 DAG.ExtractVectorElements(Op.getOperand(1), ShiftElements, 0, 0, MVT::i32);
2873 SmallVector<SDValue, 16> UnrolledOps;
2874 for (size_t i = 0; i < NumLanes; ++i) {
2875 SDValue MaskedShiftValue =
2876 DAG.getNode(ISD::AND, DL, MVT::i32, ShiftElements[i], Mask);
2877 SDValue ShiftedValue = ShiftedElements[i];
2878 if (ShiftOpcode == ISD::SRA)
2879 ShiftedValue = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i32,
2880 ShiftedValue, DAG.getValueType(LaneT));
2881 UnrolledOps.push_back(
2882 DAG.getNode(ShiftOpcode, DL, MVT::i32, ShiftedValue, MaskedShiftValue));
2883 }
2884 return DAG.getBuildVector(Op.getValueType(), DL, UnrolledOps);
2885}
2886
2887SDValue WebAssemblyTargetLowering::LowerShift(SDValue Op,
2888 SelectionDAG &DAG) const {
2889 SDLoc DL(Op);
2890 // Only manually lower vector shifts
2891 assert(Op.getSimpleValueType().isVector());
2892
2893 uint64_t LaneBits = Op.getValueType().getScalarSizeInBits();
2894 auto ShiftVal = Op.getOperand(1);
2895
2896 // Try to skip bitmask operation since it is implied inside shift instruction
2897 auto SkipImpliedMask = [](SDValue MaskOp, uint64_t MaskBits) {
2898 if (MaskOp.getOpcode() != ISD::AND)
2899 return MaskOp;
2900 SDValue LHS = MaskOp.getOperand(0);
2901 SDValue RHS = MaskOp.getOperand(1);
2902 if (MaskOp.getValueType().isVector()) {
2903 APInt MaskVal;
2904 if (!ISD::isConstantSplatVector(RHS.getNode(), MaskVal))
2905 std::swap(LHS, RHS);
2906
2907 if (ISD::isConstantSplatVector(RHS.getNode(), MaskVal) &&
2908 MaskVal == MaskBits)
2909 MaskOp = LHS;
2910 } else {
2911 if (!isa<ConstantSDNode>(RHS.getNode()))
2912 std::swap(LHS, RHS);
2913
2914 auto ConstantRHS = dyn_cast<ConstantSDNode>(RHS.getNode());
2915 if (ConstantRHS && ConstantRHS->getAPIntValue() == MaskBits)
2916 MaskOp = LHS;
2917 }
2918
2919 return MaskOp;
2920 };
2921
2922 // Skip vector and operation
2923 ShiftVal = SkipImpliedMask(ShiftVal, LaneBits - 1);
2924 ShiftVal = DAG.getSplatValue(ShiftVal);
2925 if (!ShiftVal)
2926 return unrollVectorShift(Op, DAG);
2927
2928 // Skip scalar and operation
2929 ShiftVal = SkipImpliedMask(ShiftVal, LaneBits - 1);
2930 // Use anyext because none of the high bits can affect the shift
2931 ShiftVal = DAG.getAnyExtOrTrunc(ShiftVal, DL, MVT::i32);
2932
2933 unsigned Opcode;
2934 switch (Op.getOpcode()) {
2935 case ISD::SHL:
2936 Opcode = WebAssemblyISD::VEC_SHL;
2937 break;
2938 case ISD::SRA:
2939 Opcode = WebAssemblyISD::VEC_SHR_S;
2940 break;
2941 case ISD::SRL:
2942 Opcode = WebAssemblyISD::VEC_SHR_U;
2943 break;
2944 default:
2945 llvm_unreachable("unexpected opcode");
2946 }
2947
2948 return DAG.getNode(Opcode, DL, Op.getValueType(), Op.getOperand(0), ShiftVal);
2949}
2950
2951SDValue WebAssemblyTargetLowering::LowerFP_TO_INT_SAT(SDValue Op,
2952 SelectionDAG &DAG) const {
2953 EVT ResT = Op.getValueType();
2954 EVT SatVT = cast<VTSDNode>(Op.getOperand(1))->getVT();
2955
2956 if ((ResT == MVT::i32 || ResT == MVT::i64) &&
2957 (SatVT == MVT::i32 || SatVT == MVT::i64))
2958 return Op;
2959
2960 if (ResT == MVT::v4i32 && SatVT == MVT::i32)
2961 return Op;
2962
2963 if (ResT == MVT::v8i16 && SatVT == MVT::i16)
2964 return Op;
2965
2966 return SDValue();
2967}
2968
2970 return (Op->getFlags().hasNoNaNs() ||
2971 (DAG.isKnownNeverNaN(Op->getOperand(0)) &&
2972 DAG.isKnownNeverNaN(Op->getOperand(1)))) &&
2973 (Op->getFlags().hasNoSignedZeros() ||
2974 DAG.isKnownNeverLogicalZero(Op->getOperand(0)) ||
2975 DAG.isKnownNeverLogicalZero(Op->getOperand(1)));
2976}
2977
2978SDValue WebAssemblyTargetLowering::LowerFMIN(SDValue Op,
2979 SelectionDAG &DAG) const {
2980 if (Subtarget->hasRelaxedSIMD() && HasNoSignedZerosOrNaNs(Op, DAG)) {
2981 return DAG.getNode(WebAssemblyISD::RELAXED_FMIN, SDLoc(Op),
2982 Op.getValueType(), Op.getOperand(0), Op.getOperand(1));
2983 }
2984 return SDValue();
2985}
2986
2987SDValue WebAssemblyTargetLowering::LowerFMAX(SDValue Op,
2988 SelectionDAG &DAG) const {
2989 if (Subtarget->hasRelaxedSIMD() && HasNoSignedZerosOrNaNs(Op, DAG)) {
2990 return DAG.getNode(WebAssemblyISD::RELAXED_FMAX, SDLoc(Op),
2991 Op.getValueType(), Op.getOperand(0), Op.getOperand(1));
2992 }
2993 return SDValue();
2994}
2995
2996//===----------------------------------------------------------------------===//
2997// Custom DAG combine hooks
2998//===----------------------------------------------------------------------===//
2999static SDValue
3001 auto &DAG = DCI.DAG;
3002 auto Shuffle = cast<ShuffleVectorSDNode>(N);
3003
3004 // Hoist vector bitcasts that don't change the number of lanes out of unary
3005 // shuffles, where they are less likely to get in the way of other combines.
3006 // (shuffle (vNxT1 (bitcast (vNxT0 x))), undef, mask) ->
3007 // (vNxT1 (bitcast (vNxT0 (shuffle x, undef, mask))))
3008 SDValue Bitcast = N->getOperand(0);
3009 if (Bitcast.getOpcode() != ISD::BITCAST)
3010 return SDValue();
3011 if (!N->getOperand(1).isUndef())
3012 return SDValue();
3013 SDValue CastOp = Bitcast.getOperand(0);
3014 EVT SrcType = CastOp.getValueType();
3015 EVT DstType = Bitcast.getValueType();
3016 if (!SrcType.is128BitVector() ||
3017 SrcType.getVectorNumElements() != DstType.getVectorNumElements())
3018 return SDValue();
3019 SDValue NewShuffle = DAG.getVectorShuffle(
3020 SrcType, SDLoc(N), CastOp, DAG.getUNDEF(SrcType), Shuffle->getMask());
3021 return DAG.getBitcast(DstType, NewShuffle);
3022}
3023
3024/// Convert ({u,s}itofp vec) --> ({u,s}itofp ({s,z}ext vec)) so it doesn't get
3025/// split up into scalar instructions during legalization, and the vector
3026/// extending instructions are selected in performVectorExtendCombine below.
3027static SDValue
3029 const WebAssemblySubtarget *Subtarget) {
3030 auto &DAG = DCI.DAG;
3031 assert(N->getOpcode() == ISD::UINT_TO_FP ||
3032 N->getOpcode() == ISD::SINT_TO_FP);
3033
3034 EVT InVT = N->getOperand(0)->getValueType(0);
3035 EVT ResVT = N->getValueType(0);
3036 MVT ExtVT;
3037 if (ResVT == MVT::v4f32 && (InVT == MVT::v4i16 || InVT == MVT::v4i8))
3038 ExtVT = MVT::v4i32;
3039 else if (ResVT == MVT::v2f64 && (InVT == MVT::v2i16 || InVT == MVT::v2i8))
3040 ExtVT = MVT::v2i32;
3041 else if (Subtarget->hasFP16() && ResVT == MVT::v8f16 && InVT == MVT::v8i8)
3042 ExtVT = MVT::v8i16;
3043 else
3044 return SDValue();
3045
3046 unsigned Op =
3048 SDValue Conv = DAG.getNode(Op, SDLoc(N), ExtVT, N->getOperand(0));
3049 return DAG.getNode(N->getOpcode(), SDLoc(N), ResVT, Conv);
3050}
3051
3052static SDValue
3055 auto &DAG = DCI.DAG;
3056
3057 SDNodeFlags Flags = N->getFlags();
3058 SDValue Op0 = N->getOperand(0);
3059 EVT VT = N->getValueType(0);
3060
3061 // Optimize uitofp to sitofp when the sign bit is known to be zero.
3062 // Depending on the target (runtime) backend, this might be performance
3063 // neutral (e.g. AArch64) or a significant improvement (e.g. x86_64).
3064 if (VT.isVector() && (Flags.hasNonNeg() || DAG.SignBitIsZero(Op0))) {
3065 return DAG.getNode(ISD::SINT_TO_FP, SDLoc(N), VT, Op0);
3066 }
3067
3068 return SDValue();
3069}
3070
3071static SDValue
3073 auto &DAG = DCI.DAG;
3074 assert(N->getOpcode() == ISD::SIGN_EXTEND ||
3075 N->getOpcode() == ISD::ZERO_EXTEND);
3076
3077 EVT ResVT = N->getValueType(0);
3078 bool IsSext = N->getOpcode() == ISD::SIGN_EXTEND;
3079 SDLoc DL(N);
3080
3081 if (ResVT == MVT::v16i32 && N->getOperand(0)->getValueType(0) == MVT::v16i8) {
3082 // Use a tree of extend low/high to split and extend the input in two
3083 // layers to avoid doing several shuffles and even more extends.
3084 unsigned LowOp =
3085 IsSext ? WebAssemblyISD::EXTEND_LOW_S : WebAssemblyISD::EXTEND_LOW_U;
3086 unsigned HighOp =
3087 IsSext ? WebAssemblyISD::EXTEND_HIGH_S : WebAssemblyISD::EXTEND_HIGH_U;
3088 SDValue Input = N->getOperand(0);
3089 SDValue LowHalf = DAG.getNode(LowOp, DL, MVT::v8i16, Input);
3090 SDValue HighHalf = DAG.getNode(HighOp, DL, MVT::v8i16, Input);
3091 SDValue Subvectors[] = {
3092 DAG.getNode(LowOp, DL, MVT::v4i32, LowHalf),
3093 DAG.getNode(HighOp, DL, MVT::v4i32, LowHalf),
3094 DAG.getNode(LowOp, DL, MVT::v4i32, HighHalf),
3095 DAG.getNode(HighOp, DL, MVT::v4i32, HighHalf),
3096 };
3097 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ResVT, Subvectors);
3098 }
3099
3100 // Combine ({s,z}ext (extract_subvector src, i)) into a widening operation if
3101 // possible before the extract_subvector can be expanded.
3102 auto Extract = N->getOperand(0);
3103 if (Extract.getOpcode() != ISD::EXTRACT_SUBVECTOR)
3104 return SDValue();
3105 auto Source = Extract.getOperand(0);
3106 auto *IndexNode = dyn_cast<ConstantSDNode>(Extract.getOperand(1));
3107 if (IndexNode == nullptr)
3108 return SDValue();
3109 auto Index = IndexNode->getZExtValue();
3110
3111 // Only v8i8, v4i16, and v2i32 extracts can be widened, and only if the
3112 // extracted subvector is the low or high half of its source.
3113 if (ResVT == MVT::v8i16) {
3114 if (Extract.getValueType() != MVT::v8i8 ||
3115 Source.getValueType() != MVT::v16i8 || (Index != 0 && Index != 8))
3116 return SDValue();
3117 } else if (ResVT == MVT::v4i32) {
3118 if (Extract.getValueType() != MVT::v4i16 ||
3119 Source.getValueType() != MVT::v8i16 || (Index != 0 && Index != 4))
3120 return SDValue();
3121 } else if (ResVT == MVT::v2i64) {
3122 if (Extract.getValueType() != MVT::v2i32 ||
3123 Source.getValueType() != MVT::v4i32 || (Index != 0 && Index != 2))
3124 return SDValue();
3125 } else {
3126 return SDValue();
3127 }
3128
3129 bool IsLow = Index == 0;
3130
3131 unsigned Op = IsSext ? (IsLow ? WebAssemblyISD::EXTEND_LOW_S
3132 : WebAssemblyISD::EXTEND_HIGH_S)
3133 : (IsLow ? WebAssemblyISD::EXTEND_LOW_U
3134 : WebAssemblyISD::EXTEND_HIGH_U);
3135
3136 return DAG.getNode(Op, DL, ResVT, Source);
3137}
3138
3139static SDValue
3141 auto &DAG = DCI.DAG;
3142
3143 auto GetWasmConversionOp = [](unsigned Op) {
3144 switch (Op) {
3146 return WebAssemblyISD::TRUNC_SAT_ZERO_S;
3148 return WebAssemblyISD::TRUNC_SAT_ZERO_U;
3149 case ISD::FP_ROUND:
3150 return WebAssemblyISD::DEMOTE_ZERO;
3151 }
3152 llvm_unreachable("unexpected op");
3153 };
3154
3155 auto IsZeroSplat = [](SDValue SplatVal) {
3156 auto *Splat = dyn_cast<BuildVectorSDNode>(SplatVal.getNode());
3157 APInt SplatValue, SplatUndef;
3158 unsigned SplatBitSize;
3159 bool HasAnyUndefs;
3160 // Endianness doesn't matter in this context because we are looking for
3161 // an all-zero value.
3162 return Splat &&
3163 Splat->isConstantSplat(SplatValue, SplatUndef, SplatBitSize,
3164 HasAnyUndefs) &&
3165 SplatValue == 0;
3166 };
3167
3168 if (N->getOpcode() == ISD::CONCAT_VECTORS) {
3169 // Combine this:
3170 //
3171 // (concat_vectors (v2i32 (fp_to_{s,u}int_sat $x, 32)), (v2i32 (splat 0)))
3172 //
3173 // into (i32x4.trunc_sat_f64x2_zero_{s,u} $x).
3174 //
3175 // Or this:
3176 //
3177 // (concat_vectors ({v2f32, v4f16} (fp_round ({v2f64, v4f32} $x))),
3178 // ({v2f32, v4f16} (splat 0)))
3179 //
3180 // into ({f32x4, f16x8}.demote_zero_{f64x2, f32x4} $x).
3181 EVT ResVT;
3182 EVT ExpectedConversionType;
3183 auto Conversion = N->getOperand(0);
3184 auto ConversionOp = Conversion.getOpcode();
3185 switch (ConversionOp) {
3188 ResVT = MVT::v4i32;
3189 ExpectedConversionType = MVT::v2i32;
3190 break;
3191 case ISD::FP_ROUND:
3192 if (Conversion.getValueType() == MVT::v2f32) {
3193 ResVT = MVT::v4f32;
3194 ExpectedConversionType = MVT::v2f32;
3195 } else if (Conversion.getValueType() == MVT::v4f16) {
3196 ResVT = MVT::v8f16;
3197 ExpectedConversionType = MVT::v4f16;
3198 } else {
3199 return SDValue();
3200 }
3201 break;
3202 default:
3203 return SDValue();
3204 }
3205
3206 if (N->getValueType(0) != ResVT)
3207 return SDValue();
3208
3209 if (Conversion.getValueType() != ExpectedConversionType)
3210 return SDValue();
3211
3212 auto Source = Conversion.getOperand(0);
3213 if (!((Source.getValueType() == MVT::v2f64 && ResVT == MVT::v4f32) ||
3214 (Source.getValueType() == MVT::v2f64 && ResVT == MVT::v4i32) ||
3215 (Source.getValueType() == MVT::v4f32 && ResVT == MVT::v8f16)))
3216 return SDValue();
3217
3218 if (!IsZeroSplat(N->getOperand(1)) ||
3219 N->getOperand(1).getValueType() != ExpectedConversionType)
3220 return SDValue();
3221
3222 unsigned Op = GetWasmConversionOp(ConversionOp);
3223 return DAG.getNode(Op, SDLoc(N), ResVT, Source);
3224 }
3225
3226 // Combine this:
3227 //
3228 // (fp_to_{s,u}int_sat (concat_vectors $x, (v2f64 (splat 0))), 32)
3229 //
3230 // into (i32x4.trunc_sat_f64x2_zero_{s,u} $x).
3231 //
3232 // Or this:
3233 //
3234 // ({v4f32, v8f16} (fp_round (concat_vectors $x,
3235 // ({v2f64, v4f32} (splat 0)))))
3236 //
3237 // into ({f32x4, f16x8}.demote_zero_{f64x2, f32x4} $x).
3238 EVT ResVT;
3239 auto ConversionOp = N->getOpcode();
3240 switch (ConversionOp) {
3243 ResVT = MVT::v4i32;
3244 break;
3245 case ISD::FP_ROUND:
3246 ResVT = N->getValueType(0);
3247 break;
3248 default:
3249 llvm_unreachable("unexpected op");
3250 }
3251
3252 if (N->getValueType(0) != ResVT)
3253 return SDValue();
3254
3255 auto Concat = N->getOperand(0);
3256 if (Concat.getOpcode() != ISD::CONCAT_VECTORS)
3257 return SDValue();
3258 EVT ConcatVT = Concat.getValueType();
3259 EVT SourceVT = Concat.getOperand(0).getValueType();
3260
3261 if (!IsZeroSplat(Concat.getOperand(1)))
3262 return SDValue();
3263
3264 if (ConversionOp == ISD::FP_ROUND) {
3265 bool IsF64ToF32 =
3266 ConcatVT == MVT::v4f64 && SourceVT == MVT::v2f64 && ResVT == MVT::v4f32;
3267 bool IsF32ToF16 =
3268 ConcatVT == MVT::v8f32 && SourceVT == MVT::v4f32 && ResVT == MVT::v8f16;
3269 if (!(IsF64ToF32 || IsF32ToF16))
3270 return SDValue();
3271 } else {
3272 if (ConcatVT != MVT::v4f64 || SourceVT != MVT::v2f64 || ResVT != MVT::v4i32)
3273 return SDValue();
3274 }
3275
3276 unsigned Op = GetWasmConversionOp(ConversionOp);
3277 return DAG.getNode(Op, SDLoc(N), ResVT, Concat.getOperand(0));
3278}
3279
3280// Helper to extract VectorWidth bits from Vec, starting from IdxVal.
3281static SDValue extractSubVector(SDValue Vec, unsigned IdxVal, SelectionDAG &DAG,
3282 const SDLoc &DL, unsigned VectorWidth) {
3283 EVT VT = Vec.getValueType();
3284 EVT ElVT = VT.getVectorElementType();
3285 unsigned Factor = VT.getSizeInBits() / VectorWidth;
3286 EVT ResultVT = EVT::getVectorVT(*DAG.getContext(), ElVT,
3287 VT.getVectorNumElements() / Factor);
3288
3289 // Extract the relevant VectorWidth bits. Generate an EXTRACT_SUBVECTOR
3290 unsigned ElemsPerChunk = VectorWidth / ElVT.getSizeInBits();
3291 assert(isPowerOf2_32(ElemsPerChunk) && "Elements per chunk not power of 2");
3292
3293 // This is the index of the first element of the VectorWidth-bit chunk
3294 // we want. Since ElemsPerChunk is a power of 2 just need to clear bits.
3295 IdxVal &= ~(ElemsPerChunk - 1);
3296
3297 // If the input is a buildvector just emit a smaller one.
3298 if (Vec.getOpcode() == ISD::BUILD_VECTOR)
3299 return DAG.getBuildVector(ResultVT, DL,
3300 Vec->ops().slice(IdxVal, ElemsPerChunk));
3301
3302 SDValue VecIdx = DAG.getIntPtrConstant(IdxVal, DL);
3303 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, ResultVT, Vec, VecIdx);
3304}
3305
3306// Helper to recursively truncate vector elements in half with NARROW_U. DstVT
3307// is the expected destination value type after recursion. In is the initial
3308// input. Note that the input should have enough leading zero bits to prevent
3309// NARROW_U from saturating results.
3311 SelectionDAG &DAG) {
3312 EVT SrcVT = In.getValueType();
3313
3314 // No truncation required, we might get here due to recursive calls.
3315 if (SrcVT == DstVT)
3316 return In;
3317
3318 unsigned SrcSizeInBits = SrcVT.getSizeInBits();
3319 unsigned NumElems = SrcVT.getVectorNumElements();
3320 if (!isPowerOf2_32(NumElems))
3321 return SDValue();
3322 assert(DstVT.getVectorNumElements() == NumElems && "Illegal truncation");
3323 assert(SrcSizeInBits > DstVT.getSizeInBits() && "Illegal truncation");
3324
3325 LLVMContext &Ctx = *DAG.getContext();
3326 EVT PackedSVT = EVT::getIntegerVT(Ctx, SrcVT.getScalarSizeInBits() / 2);
3327
3328 // Narrow to the largest type possible:
3329 // vXi64/vXi32 -> i16x8.narrow_i32x4_u and vXi16 -> i8x16.narrow_i16x8_u.
3330 EVT InVT = MVT::i16, OutVT = MVT::i8;
3331 if (SrcVT.getScalarSizeInBits() > 16) {
3332 InVT = MVT::i32;
3333 OutVT = MVT::i16;
3334 }
3335 unsigned SubSizeInBits = SrcSizeInBits / 2;
3336 InVT = EVT::getVectorVT(Ctx, InVT, SubSizeInBits / InVT.getSizeInBits());
3337 OutVT = EVT::getVectorVT(Ctx, OutVT, SubSizeInBits / OutVT.getSizeInBits());
3338
3339 // Split lower/upper subvectors.
3340 SDValue Lo = extractSubVector(In, 0, DAG, DL, SubSizeInBits);
3341 SDValue Hi = extractSubVector(In, NumElems / 2, DAG, DL, SubSizeInBits);
3342
3343 // 256bit -> 128bit truncate - Narrow lower/upper 128-bit subvectors.
3344 if (SrcVT.is256BitVector() && DstVT.is128BitVector()) {
3345 Lo = DAG.getBitcast(InVT, Lo);
3346 Hi = DAG.getBitcast(InVT, Hi);
3347 SDValue Res = DAG.getNode(WebAssemblyISD::NARROW_U, DL, OutVT, Lo, Hi);
3348 return DAG.getBitcast(DstVT, Res);
3349 }
3350
3351 // Recursively narrow lower/upper subvectors, concat result and narrow again.
3352 EVT PackedVT = EVT::getVectorVT(Ctx, PackedSVT, NumElems / 2);
3353 Lo = truncateVectorWithNARROW(PackedVT, Lo, DL, DAG);
3354 Hi = truncateVectorWithNARROW(PackedVT, Hi, DL, DAG);
3355
3356 PackedVT = EVT::getVectorVT(Ctx, PackedSVT, NumElems);
3357 SDValue Res = DAG.getNode(ISD::CONCAT_VECTORS, DL, PackedVT, Lo, Hi);
3358 return truncateVectorWithNARROW(DstVT, Res, DL, DAG);
3359}
3360
3363 auto &DAG = DCI.DAG;
3364
3365 SDValue In = N->getOperand(0);
3366 EVT InVT = In.getValueType();
3367 if (!InVT.isSimple())
3368 return SDValue();
3369
3370 EVT OutVT = N->getValueType(0);
3371 if (!OutVT.isVector())
3372 return SDValue();
3373
3374 EVT OutSVT = OutVT.getVectorElementType();
3375 EVT InSVT = InVT.getVectorElementType();
3376 // Currently only cover truncate to v16i8 or v8i16.
3377 if (!((InSVT == MVT::i16 || InSVT == MVT::i32 || InSVT == MVT::i64) &&
3378 (OutSVT == MVT::i8 || OutSVT == MVT::i16) && OutVT.is128BitVector()))
3379 return SDValue();
3380
3381 SDLoc DL(N);
3383 OutVT.getScalarSizeInBits());
3384 In = DAG.getNode(ISD::AND, DL, InVT, In, DAG.getConstant(Mask, DL, InVT));
3385 return truncateVectorWithNARROW(OutVT, In, DL, DAG);
3386}
3387
3390 using namespace llvm::SDPatternMatch;
3391 auto &DAG = DCI.DAG;
3392 SDLoc DL(N);
3393 SDValue Src = N->getOperand(0);
3394 EVT VT = N->getValueType(0);
3395 EVT SrcVT = Src.getValueType();
3396
3397 if (!(DCI.isBeforeLegalize() && VT.isScalarInteger() &&
3398 SrcVT.isFixedLengthVectorOf(MVT::i1)))
3399 return SDValue();
3400
3401 unsigned NumElts = SrcVT.getVectorNumElements();
3402 EVT Width = MVT::getIntegerVT(128 / NumElts);
3403
3404 // bitcast <N x i1> to iN, where N = 2, 4, 8, 16 (legal)
3405 // ==> bitmask
3406 if (NumElts == 2 || NumElts == 4 || NumElts == 8 || NumElts == 16) {
3407 return DAG.getZExtOrTrunc(
3408 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3409 {DAG.getConstant(Intrinsic::wasm_bitmask, DL, MVT::i32),
3410 DAG.getSExtOrTrunc(N->getOperand(0), DL,
3411 SrcVT.changeVectorElementType(
3412 *DAG.getContext(), Width))}),
3413 DL, VT);
3414 }
3415
3416 // bitcast <N x i1>(setcc ...) to concat iN, where N = 32 and 64 (illegal)
3417 if (NumElts == 32 || NumElts == 64) {
3418 SDValue Concat, SetCCVector;
3419 ISD::CondCode SetCond;
3420
3421 if (!sd_match(N, m_BitCast(m_c_SetCC(SetCond, m_Value(Concat),
3422 m_Value(SetCCVector)))))
3423 return SDValue();
3424 if (Concat.getOpcode() != ISD::CONCAT_VECTORS)
3425 return SDValue();
3426
3427 // Reconstruct the wide bitmask from each CONCAT_VECTORS operand.
3428 // Derive the per-chunk mask/integer types from the actual operand type
3429 // instead of hardcoding v16i1 / i16 for every chunk.
3430 EVT ConcatOperandVT = Concat.getOperand(0).getValueType();
3431 unsigned ConcatOperandNumElts = ConcatOperandVT.getVectorNumElements();
3432
3433 EVT ConcatOperandMaskVT =
3434 EVT::getVectorVT(*DAG.getContext(), MVT::i1,
3435 ElementCount::getFixed(ConcatOperandNumElts));
3436 EVT ConcatOperandBitmaskVT =
3437 EVT::getIntegerVT(*DAG.getContext(), ConcatOperandNumElts);
3438 EVT ReturnVT = N->getValueType(0);
3439 SDValue ReconstructedBitmask = DAG.getConstant(0, DL, ReturnVT);
3440 // Example:
3441 // v32i16 = concat(v8i16, v8i16, v8i16, v8i16)
3442 // -> v8i1 + v8i1 + v8i1 + v8i1
3443 // -> i8 + i8 + i8 + i8
3444 // -> reconstructed i32 bitmask
3445 for (size_t I = 0; I < Concat->ops().size(); ++I) {
3446 SDValue ConcatOperand = Concat.getOperand(I);
3447 assert(ConcatOperand.getValueType() == ConcatOperandVT &&
3448 "concat_vectors operands must have the same type");
3449
3450 SDValue SetCCVectorOperand =
3451 extractSubVector(SetCCVector, I * ConcatOperandNumElts, DAG, DL, 128);
3452 if (!SetCCVectorOperand ||
3453 SetCCVectorOperand.getValueType() != ConcatOperandVT)
3454 return SDValue();
3455
3456 // Build the per-chunk mask using the correct chunk type:
3457 // v16i8 -> v16i1 -> i16
3458 // v8i16 -> v8i1 -> i8
3459 // v4i32 -> v4i1 -> i4
3460 // v2i64 -> v2i1 -> i2
3461 SDValue ConcatOperandMask = DAG.getSetCC(
3462 DL, ConcatOperandMaskVT, ConcatOperand, SetCCVectorOperand, SetCond);
3463 SDValue ConcatOperandBitmask =
3464 DAG.getBitcast(ConcatOperandBitmaskVT, ConcatOperandMask);
3465 SDValue ExtendedConcatOperandBitmask =
3466 DAG.getZExtOrTrunc(ConcatOperandBitmask, DL, ReturnVT);
3467
3468 // Shift each chunk's mask to its original lane position before merging it
3469 // into the result:
3470 // Result bit index = I * ConcatOperandNumElts + local lane index
3471 //
3472 // Example: four chunks, each containing 8 mask bits:
3473 // result = M0 | (M1 << 8) | (M2 << 16) | (M3 << 24)
3474 //
3475 SDValue PositionedChunkBitmask = ExtendedConcatOperandBitmask;
3476 if (I != 0) {
3477 PositionedChunkBitmask = DAG.getNode(
3478 ISD::SHL, DL, ReturnVT, ExtendedConcatOperandBitmask,
3479 DAG.getShiftAmountConstant(I * ConcatOperandNumElts, ReturnVT, DL));
3480 }
3481
3482 ReconstructedBitmask = DAG.getNode(
3483 ISD::OR, DL, ReturnVT, ReconstructedBitmask, PositionedChunkBitmask);
3484 }
3485
3486 return ReconstructedBitmask;
3487 }
3488
3489 return SDValue();
3490}
3491
3493 // bitmask (setcc <X>, 0, setlt) => bitmask X
3494 assert(N->getOpcode() == ISD::INTRINSIC_WO_CHAIN);
3495 using namespace llvm::SDPatternMatch;
3496
3497 if (N->getConstantOperandVal(0) != Intrinsic::wasm_bitmask)
3498 return SDValue();
3499
3500 SDValue LHS;
3501 if (!sd_match(N->getOperand(1),
3503 return SDValue();
3504
3505 SDLoc DL(N);
3506 return DAG.getNode(
3507 ISD::INTRINSIC_WO_CHAIN, DL, N->getValueType(0),
3508 {DAG.getConstant(Intrinsic::wasm_bitmask, DL, MVT::i32), LHS});
3509}
3510
3512 // any_true (setcc <X>, 0, eq) => (not (all_true X))
3513 // all_true (setcc <X>, 0, eq) => (not (any_true X))
3514 // any_true (setcc <X>, 0, ne) => (any_true X)
3515 // all_true (setcc <X>, 0, ne) => (all_true X)
3516 assert(N->getOpcode() == ISD::INTRINSIC_WO_CHAIN);
3517 using namespace llvm::SDPatternMatch;
3518
3519 SDValue LHS;
3520 if (N->getNumOperands() < 2 ||
3521 !sd_match(N->getOperand(1), m_c_SetCC(m_Value(LHS), m_Zero())))
3522 return SDValue();
3523 EVT LT = LHS.getValueType();
3524 if (LT.getScalarSizeInBits() > 128 / LT.getVectorNumElements())
3525 return SDValue();
3526
3527 auto CombineSetCC = [&N, &DAG](Intrinsic::WASMIntrinsics InPre,
3528 ISD::CondCode SetType,
3529 Intrinsic::WASMIntrinsics InPost) {
3530 if (N->getConstantOperandVal(0) != InPre)
3531 return SDValue();
3532
3533 SDValue LHS;
3534 if (!sd_match(N->getOperand(1),
3535 m_c_SpecificSetCC(SetType, m_Value(LHS), m_Zero())))
3536 return SDValue();
3537
3538 SDLoc DL(N);
3539 SDValue Ret = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3540 {DAG.getConstant(InPost, DL, MVT::i32), LHS});
3541 if (SetType == ISD::SETEQ)
3542 Ret = DAG.getNode(ISD::XOR, DL, MVT::i32, Ret,
3543 DAG.getConstant(1, DL, MVT::i32));
3544 return DAG.getZExtOrTrunc(Ret, DL, N->getValueType(0));
3545 };
3546
3547 if (SDValue AnyTrueEQ = CombineSetCC(Intrinsic::wasm_anytrue, ISD::SETEQ,
3548 Intrinsic::wasm_alltrue))
3549 return AnyTrueEQ;
3550 if (SDValue AllTrueEQ = CombineSetCC(Intrinsic::wasm_alltrue, ISD::SETEQ,
3551 Intrinsic::wasm_anytrue))
3552 return AllTrueEQ;
3553 if (SDValue AnyTrueNE = CombineSetCC(Intrinsic::wasm_anytrue, ISD::SETNE,
3554 Intrinsic::wasm_anytrue))
3555 return AnyTrueNE;
3556 if (SDValue AllTrueNE = CombineSetCC(Intrinsic::wasm_alltrue, ISD::SETNE,
3557 Intrinsic::wasm_alltrue))
3558 return AllTrueNE;
3559
3560 return SDValue();
3561}
3562
3568
3570 unsigned NumElts,
3571 const MaskReduceInfo &Info,
3572 SelectionDAG &DAG) {
3573 EVT VecVT = FromVT.changeVectorElementType(*DAG.getContext(),
3574 MVT::getIntegerVT(128 / NumElts));
3575 assert(VecVT.getSizeInBits() == 128 &&
3576 "mask reduction should be widened to a 128-bit vector");
3577
3578 SDLoc DL(N);
3579 SDValue Mask = N->getOperand(0)->getOperand(0);
3580 SDValue Ret = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3581 {DAG.getConstant(Info.IID, DL, MVT::i32),
3582 DAG.getSExtOrTrunc(Mask, DL, VecVT)});
3583 if (Info.Invert)
3584 Ret = DAG.getNode(ISD::XOR, DL, MVT::i32, Ret,
3585 DAG.getConstant(1, DL, MVT::i32));
3586 return DAG.getZExtOrTrunc(Ret, DL, N->getValueType(0));
3587}
3588
3590 unsigned NumElts,
3591 const MaskReduceInfo &Info,
3592 SelectionDAG &DAG) {
3593 assert((NumElts == 32 || NumElts == 64) &&
3594 "combineWideMaskReduction is only for wide masks");
3595 assert(MaskVT.isFixedLengthVector() &&
3596 MaskVT.getVectorElementType() == MVT::i1);
3597 SDLoc DL(N);
3598 unsigned ChunkElts = 16;
3599 EVT ChunkMaskVT = EVT::getVectorVT(*DAG.getContext(), MVT::i1,
3600 ElementCount::getFixed(ChunkElts));
3601 EVT LegalVecVT = ChunkMaskVT.changeVectorElementType(
3602 *DAG.getContext(), MVT::getIntegerVT(128 / ChunkElts));
3603
3604 SmallVector<SDValue, 4> ChunkResults;
3605 // Split the wide mask into v16i1 chunks and reduce each chunk separately.
3606 // For example:
3607 // v32i1: [0..15] [16..31]
3608 // | |
3609 // v v
3610 // chunk0 chunk1
3611 //
3612 // v64i1: [0..15] [16..31] [32..47] [48..63]
3613 // | | | |
3614 // v v v v
3615 // chunk0 chunk1 chunk2 chunk3
3616 //
3617 // each chunk:
3618 // v16i1 -> v16i8 -> wasm_anytrue/alltrue -> i32 0/1
3619 for (unsigned I = 0; I < NumElts; I += ChunkElts) {
3620 SDValue ChunkMask = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, ChunkMaskVT,
3621 Mask, DAG.getVectorIdxConstant(I, DL));
3622 SDValue LegalMask = DAG.getSExtOrTrunc(ChunkMask, DL, LegalVecVT);
3623 SDValue Reduced =
3624 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3625 DAG.getConstant(Info.IID, DL, MVT::i32), LegalMask);
3626 ChunkResults.push_back(Reduced);
3627 }
3628
3629 SDValue Acc = ChunkResults[0];
3630 for (unsigned I = 1; I < ChunkResults.size(); ++I)
3631 Acc =
3632 DAG.getNode(Info.WideCombineOpcode, DL, MVT::i32, Acc, ChunkResults[I]);
3633
3634 if (Info.Invert)
3635 Acc = DAG.getNode(ISD::XOR, DL, MVT::i32, Acc,
3636 DAG.getConstant(1, DL, MVT::i32));
3637
3638 return DAG.getZExtOrTrunc(Acc, DL, N->getValueType(0));
3639}
3640
3641static std::optional<MaskReduceInfo> classifyMaskReduction(SDNode *N) {
3642 auto *C = dyn_cast<ConstantSDNode>(N->getOperand(1));
3643 if (!C)
3644 return std::nullopt;
3645
3646 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
3647
3648 // setcc (bitcast mask), 0, ne -> any_true(mask)
3649 if (C->isZero() && CC == ISD::SETNE)
3650 return MaskReduceInfo{Intrinsic::wasm_anytrue, ISD::OR, false};
3651
3652 // setcc (bitcast mask), 0, eq -> !any_true(mask)
3653 if (C->isZero() && CC == ISD::SETEQ)
3654 return MaskReduceInfo{Intrinsic::wasm_anytrue, ISD::OR, true};
3655
3656 // setcc (bitcast mask), -1, eq -> all_true(mask)
3657 if (C->isAllOnes() && CC == ISD::SETEQ)
3658 return MaskReduceInfo{Intrinsic::wasm_alltrue, ISD::AND, false};
3659
3660 // setcc (bitcast mask), -1, ne -> !all_true(mask)
3661 if (C->isAllOnes() && CC == ISD::SETNE)
3662 return MaskReduceInfo{Intrinsic::wasm_alltrue, ISD::AND, true};
3663
3664 return std::nullopt;
3665}
3666
3667/// Try to convert a i128 comparison to a v16i8 comparison before type
3668/// legalization splits it up into chunks
3669static SDValue
3671 const WebAssemblySubtarget *Subtarget) {
3672
3673 SDLoc DL(N);
3674 SDValue X = N->getOperand(0);
3675 SDValue Y = N->getOperand(1);
3676 EVT VT = N->getValueType(0);
3677 EVT OpVT = X.getValueType();
3678
3679 SelectionDAG &DAG = DCI.DAG;
3681 Attribute::NoImplicitFloat))
3682 return SDValue();
3683
3684 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
3685 // We're looking for an oversized integer equality comparison with SIMD
3686 if (!OpVT.isScalarInteger() || !OpVT.isByteSized() || OpVT != MVT::i128 ||
3687 !Subtarget->hasSIMD128() || !isIntEqualitySetCC(CC))
3688 return SDValue();
3689
3690 // Don't perform this combine if constructing the vector will be expensive.
3691 auto IsVectorBitCastCheap = [](SDValue X) {
3693 return isa<ConstantSDNode>(X) || X.getOpcode() == ISD::LOAD;
3694 };
3695
3696 if (!IsVectorBitCastCheap(X) || !IsVectorBitCastCheap(Y))
3697 return SDValue();
3698
3699 SDValue VecX = DAG.getBitcast(MVT::v16i8, X);
3700 SDValue VecY = DAG.getBitcast(MVT::v16i8, Y);
3701 SDValue Cmp = DAG.getSetCC(DL, MVT::v16i8, VecX, VecY, CC);
3702
3703 SDValue Intr =
3704 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
3705 {DAG.getConstant(CC == ISD::SETEQ ? Intrinsic::wasm_alltrue
3706 : Intrinsic::wasm_anytrue,
3707 DL, MVT::i32),
3708 Cmp});
3709
3710 return DAG.getSetCC(DL, VT, Intr, DAG.getConstant(0, DL, MVT::i32),
3711 ISD::SETNE);
3712}
3713
3716 const WebAssemblySubtarget *Subtarget) {
3717 if (!DCI.isBeforeLegalize())
3718 return SDValue();
3719
3720 EVT VT = N->getValueType(0);
3721 if (!VT.isScalarInteger())
3722 return SDValue();
3723
3724 if (SDValue V = combineVectorSizedSetCCEquality(N, DCI, Subtarget))
3725 return V;
3726
3727 SDValue LHS = N->getOperand(0);
3728 if (LHS->getOpcode() != ISD::BITCAST)
3729 return SDValue();
3730
3731 EVT FromVT = LHS->getOperand(0).getValueType();
3732 if (!FromVT.isFixedLengthVectorOf(MVT::i1))
3733 return SDValue();
3734
3735 unsigned NumElts = FromVT.getVectorNumElements();
3736 auto Info = classifyMaskReduction(N);
3737 if (!Info)
3738 return SDValue();
3739
3740 auto &DAG = DCI.DAG;
3741 if (NumElts == 2 || NumElts == 4 || NumElts == 8 || NumElts == 16)
3742 return combineSmallMaskReduction(N, FromVT, NumElts, *Info, DAG);
3743
3744 if (NumElts == 32 || NumElts == 64)
3745 return combineWideMaskReduction(N, LHS.getOperand(0), FromVT, NumElts,
3746 *Info, DAG);
3747
3748 return SDValue();
3749}
3750
3752 EVT VT = N->getValueType(0);
3753 if (VT != MVT::v8i32 && VT != MVT::v16i32)
3754 return SDValue();
3755
3756 // Mul with extending inputs.
3757 SDValue LHS = N->getOperand(0);
3758 SDValue RHS = N->getOperand(1);
3759 if (LHS.getOpcode() != RHS.getOpcode())
3760 return SDValue();
3761
3762 if (LHS.getOpcode() != ISD::SIGN_EXTEND &&
3763 LHS.getOpcode() != ISD::ZERO_EXTEND)
3764 return SDValue();
3765
3766 if (LHS->getOperand(0).getValueType() != RHS->getOperand(0).getValueType())
3767 return SDValue();
3768
3769 EVT FromVT = LHS->getOperand(0).getValueType();
3770 EVT EltTy = FromVT.getVectorElementType();
3771 if (EltTy != MVT::i8)
3772 return SDValue();
3773
3774 // For an input DAG that looks like this
3775 // %a = input_type
3776 // %b = input_type
3777 // %lhs = extend %a to output_type
3778 // %rhs = extend %b to output_type
3779 // %mul = mul %lhs, %rhs
3780
3781 // input_type | output_type | instructions
3782 // v16i8 | v16i32 | %low = i16x8.extmul_low_i8x16_ %a, %b
3783 // | | %high = i16x8.extmul_high_i8x16_, %a, %b
3784 // | | %low_low = i32x4.ext_low_i16x8_ %low
3785 // | | %low_high = i32x4.ext_high_i16x8_ %low
3786 // | | %high_low = i32x4.ext_low_i16x8_ %high
3787 // | | %high_high = i32x4.ext_high_i16x8_ %high
3788 // | | %res = concat_vector(...)
3789 // v8i8 | v8i32 | %low = i16x8.extmul_low_i8x16_ %a, %b
3790 // | | %low_low = i32x4.ext_low_i16x8_ %low
3791 // | | %low_high = i32x4.ext_high_i16x8_ %low
3792 // | | %res = concat_vector(%low_low, %low_high)
3793
3794 SDLoc DL(N);
3795 unsigned NumElts = VT.getVectorNumElements();
3796 SDValue ExtendInLHS = LHS->getOperand(0);
3797 SDValue ExtendInRHS = RHS->getOperand(0);
3798 bool IsSigned = LHS->getOpcode() == ISD::SIGN_EXTEND;
3799 unsigned ExtendLowOpc =
3800 IsSigned ? WebAssemblyISD::EXTEND_LOW_S : WebAssemblyISD::EXTEND_LOW_U;
3801 unsigned ExtendHighOpc =
3802 IsSigned ? WebAssemblyISD::EXTEND_HIGH_S : WebAssemblyISD::EXTEND_HIGH_U;
3803
3804 auto GetExtendLow = [&DAG, &DL, &ExtendLowOpc](EVT VT, SDValue Op) {
3805 return DAG.getNode(ExtendLowOpc, DL, VT, Op);
3806 };
3807 auto GetExtendHigh = [&DAG, &DL, &ExtendHighOpc](EVT VT, SDValue Op) {
3808 return DAG.getNode(ExtendHighOpc, DL, VT, Op);
3809 };
3810
3811 if (NumElts == 16) {
3812 SDValue LowLHS = GetExtendLow(MVT::v8i16, ExtendInLHS);
3813 SDValue LowRHS = GetExtendLow(MVT::v8i16, ExtendInRHS);
3814 SDValue MulLow = DAG.getNode(ISD::MUL, DL, MVT::v8i16, LowLHS, LowRHS);
3815 SDValue HighLHS = GetExtendHigh(MVT::v8i16, ExtendInLHS);
3816 SDValue HighRHS = GetExtendHigh(MVT::v8i16, ExtendInRHS);
3817 SDValue MulHigh = DAG.getNode(ISD::MUL, DL, MVT::v8i16, HighLHS, HighRHS);
3818 SDValue SubVectors[] = {
3819 GetExtendLow(MVT::v4i32, MulLow),
3820 GetExtendHigh(MVT::v4i32, MulLow),
3821 GetExtendLow(MVT::v4i32, MulHigh),
3822 GetExtendHigh(MVT::v4i32, MulHigh),
3823 };
3824 return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, SubVectors);
3825 } else {
3826 assert(NumElts == 8);
3827 SDValue LowLHS = DAG.getNode(LHS->getOpcode(), DL, MVT::v8i16, ExtendInLHS);
3828 SDValue LowRHS = DAG.getNode(RHS->getOpcode(), DL, MVT::v8i16, ExtendInRHS);
3829 SDValue MulLow = DAG.getNode(ISD::MUL, DL, MVT::v8i16, LowLHS, LowRHS);
3830 SDValue Lo = GetExtendLow(MVT::v4i32, MulLow);
3831 SDValue Hi = GetExtendHigh(MVT::v4i32, MulLow);
3832 return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Lo, Hi);
3833 }
3834 return SDValue();
3835}
3836
3839 assert(N->getOpcode() == ISD::MUL);
3840 EVT VT = N->getValueType(0);
3841 if (!VT.isVector())
3842 return SDValue();
3843
3844 if (auto Res = TryWideExtMulCombine(N, DCI.DAG))
3845 return Res;
3846
3847 // We don't natively support v16i8 or v8i8 mul, but we do support v8i16. So,
3848 // extend them to v8i16.
3849 if (VT != MVT::v8i8 && VT != MVT::v16i8)
3850 return SDValue();
3851
3852 SDLoc DL(N);
3853 SelectionDAG &DAG = DCI.DAG;
3854 SDValue LHS = N->getOperand(0);
3855 SDValue RHS = N->getOperand(1);
3856 EVT MulVT = MVT::v8i16;
3857
3858 if (VT == MVT::v8i8) {
3859 SDValue PromotedLHS = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v16i8, LHS,
3860 DAG.getUNDEF(MVT::v8i8));
3861 SDValue PromotedRHS = DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v16i8, RHS,
3862 DAG.getUNDEF(MVT::v8i8));
3863 SDValue LowLHS =
3864 DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, PromotedLHS);
3865 SDValue LowRHS =
3866 DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, PromotedRHS);
3867 SDValue MulLow = DAG.getBitcast(
3868 MVT::v16i8, DAG.getNode(ISD::MUL, DL, MulVT, LowLHS, LowRHS));
3869 // Take the low byte of each lane.
3870 SDValue Shuffle = DAG.getVectorShuffle(
3871 MVT::v16i8, DL, MulLow, DAG.getUNDEF(MVT::v16i8),
3872 {0, 2, 4, 6, 8, 10, 12, 14, -1, -1, -1, -1, -1, -1, -1, -1});
3873 return extractSubVector(Shuffle, 0, DAG, DL, 64);
3874 } else {
3875 assert(VT == MVT::v16i8 && "Expected v16i8");
3876 SDValue LowLHS = DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, LHS);
3877 SDValue LowRHS = DAG.getNode(WebAssemblyISD::EXTEND_LOW_U, DL, MulVT, RHS);
3878 SDValue HighLHS =
3879 DAG.getNode(WebAssemblyISD::EXTEND_HIGH_U, DL, MulVT, LHS);
3880 SDValue HighRHS =
3881 DAG.getNode(WebAssemblyISD::EXTEND_HIGH_U, DL, MulVT, RHS);
3882
3883 SDValue MulLow =
3884 DAG.getBitcast(VT, DAG.getNode(ISD::MUL, DL, MulVT, LowLHS, LowRHS));
3885 SDValue MulHigh =
3886 DAG.getBitcast(VT, DAG.getNode(ISD::MUL, DL, MulVT, HighLHS, HighRHS));
3887
3888 // Take the low byte of each lane.
3889 return DAG.getVectorShuffle(
3890 VT, DL, MulLow, MulHigh,
3891 {0, 2, 4, 6, 8, 10, 12, 14, 16, 18, 20, 22, 24, 26, 28, 30});
3892 }
3893}
3894
3895SDValue DoubleVectorWidth(SDValue In, unsigned RequiredNumElems,
3896 SelectionDAG &DAG) {
3897 SDLoc DL(In);
3898 LLVMContext &Ctx = *DAG.getContext();
3899 EVT InVT = In.getValueType();
3900 unsigned NumElems = InVT.getVectorNumElements() * 2;
3901 EVT OutVT = EVT::getVectorVT(Ctx, InVT.getVectorElementType(), NumElems);
3902 SDValue Concat =
3903 DAG.getNode(ISD::CONCAT_VECTORS, DL, OutVT, In, DAG.getPOISON(InVT));
3904 if (NumElems < RequiredNumElems) {
3905 return DoubleVectorWidth(Concat, RequiredNumElems, DAG);
3906 }
3907 return Concat;
3908}
3909
3911 EVT OutVT = N->getValueType(0);
3912 if (!OutVT.isVector())
3913 return SDValue();
3914
3915 EVT OutElTy = OutVT.getVectorElementType();
3916 if (OutElTy != MVT::i8 && OutElTy != MVT::i16)
3917 return SDValue();
3918
3919 unsigned NumElems = OutVT.getVectorNumElements();
3920 if (!isPowerOf2_32(NumElems))
3921 return SDValue();
3922
3923 EVT FPVT = N->getOperand(0)->getValueType(0);
3924 if (FPVT.getVectorElementType() != MVT::f32)
3925 return SDValue();
3926
3927 SDLoc DL(N);
3928
3929 // First, convert to i32.
3930 LLVMContext &Ctx = *DAG.getContext();
3931 EVT IntVT = EVT::getVectorVT(Ctx, MVT::i32, NumElems);
3932 SDValue ToInt = DAG.getNode(N->getOpcode(), DL, IntVT, N->getOperand(0));
3934 OutVT.getScalarSizeInBits());
3935 // Mask out the top MSBs.
3936 SDValue Masked =
3937 DAG.getNode(ISD::AND, DL, IntVT, ToInt, DAG.getConstant(Mask, DL, IntVT));
3938
3939 if (OutVT.getSizeInBits() < 128) {
3940 // Create a wide enough vector that we can use narrow.
3941 EVT NarrowedVT = OutElTy == MVT::i8 ? MVT::v16i8 : MVT::v8i16;
3942 unsigned NumRequiredElems = NarrowedVT.getVectorNumElements();
3943 SDValue WideVector = DoubleVectorWidth(Masked, NumRequiredElems, DAG);
3944 SDValue Trunc = truncateVectorWithNARROW(NarrowedVT, WideVector, DL, DAG);
3945 return DAG.getBitcast(
3946 OutVT, extractSubVector(Trunc, 0, DAG, DL, OutVT.getSizeInBits()));
3947 } else {
3948 return truncateVectorWithNARROW(OutVT, Masked, DL, DAG);
3949 }
3950 return SDValue();
3951}
3952
3953// Wide vector shift operations such as v8i32 with sign-extended
3954// operands cause Type Legalizer crashes because the target-specific
3955// extension nodes cannot be directly mapped to the 256-bit size.
3956//
3957// To resolve the crash and optimize performance, we intercept the
3958// illegal v8i32 shift in DAGCombine. We convert the shift amounts
3959// into multipliers and manually split the vector into two v4i32 halves.
3960//
3961// Before: t1: v8i32 = shl (sign_extend v8i16), const_vec
3962// After : t2: v4i32 = mul (ext_low_s v8i16), (ext_low_s narrow_vec)
3963// t3: v4i32 = mul (ext_high_s v8i16), (ext_high_s narrow_vec)
3964// t4: v8i32 = concat_vectors t2, t3
3967 SelectionDAG &DAG = DCI.DAG;
3968 assert(N->getOpcode() == ISD::SHL);
3969 EVT VT = N->getValueType(0);
3970 if (VT != MVT::v8i32)
3971 return SDValue();
3972
3973 SDValue LHS = N->getOperand(0);
3974 SDValue RHS = N->getOperand(1);
3975 unsigned ExtOpc = LHS.getOpcode();
3976 if (ExtOpc != ISD::SIGN_EXTEND && ExtOpc != ISD::ZERO_EXTEND)
3977 return SDValue();
3978
3979 if (RHS.getOpcode() != ISD::BUILD_VECTOR)
3980 return SDValue();
3981
3982 SDLoc DL(N);
3983 SDValue ExtendIn = LHS.getOperand(0);
3984 EVT FromVT = ExtendIn.getValueType();
3985 if (FromVT != MVT::v8i16)
3986 return SDValue();
3987
3988 unsigned NumElts = VT.getVectorNumElements();
3989 unsigned BitWidth = FromVT.getScalarSizeInBits();
3990 bool IsSigned = (ExtOpc == ISD::SIGN_EXTEND);
3991 unsigned MaxValidShift = IsSigned ? (BitWidth - 1) : BitWidth;
3992 SmallVector<SDValue, 16> MulConsts;
3993 for (unsigned I = 0; I < NumElts; ++I) {
3994 auto *C = dyn_cast<ConstantSDNode>(RHS.getOperand(I));
3995 if (!C)
3996 return SDValue();
3997
3998 const APInt &ShiftAmt = C->getAPIntValue();
3999 if (ShiftAmt.uge(MaxValidShift))
4000 return SDValue();
4001
4002 APInt MulAmt = APInt::getOneBitSet(BitWidth, ShiftAmt.getZExtValue());
4003 MulConsts.push_back(DAG.getConstant(MulAmt, DL, FromVT.getScalarType(),
4004 /*isTarget=*/false, /*isOpaque=*/true));
4005 }
4006
4007 SDValue NarrowConst = DAG.getBuildVector(FromVT, DL, MulConsts);
4008 unsigned ExtLowOpc =
4009 IsSigned ? WebAssemblyISD::EXTEND_LOW_S : WebAssemblyISD::EXTEND_LOW_U;
4010 unsigned ExtHighOpc =
4011 IsSigned ? WebAssemblyISD::EXTEND_HIGH_S : WebAssemblyISD::EXTEND_HIGH_U;
4012
4013 EVT HalfVT = MVT::v4i32;
4014 SDValue LHSLo = DAG.getNode(ExtLowOpc, DL, HalfVT, ExtendIn);
4015 SDValue LHSHi = DAG.getNode(ExtHighOpc, DL, HalfVT, ExtendIn);
4016 SDValue RHSLo = DAG.getNode(ExtLowOpc, DL, HalfVT, NarrowConst);
4017 SDValue RHSHi = DAG.getNode(ExtHighOpc, DL, HalfVT, NarrowConst);
4018 SDValue MulLo = DAG.getNode(ISD::MUL, DL, HalfVT, LHSLo, RHSLo);
4019 SDValue MulHi = DAG.getNode(ISD::MUL, DL, HalfVT, LHSHi, RHSHi);
4020 return DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, MulLo, MulHi);
4021}
4022
4024 if (N->getValueType(0) != MVT::f128)
4025 return SDValue();
4026
4027 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4028 switch (N->getOpcode()) {
4029 // wasi-libc and emscripten do not currently define fminimuml and fmaximuml.
4030 case ISD::FMINIMUM:
4031 case ISD::FMAXIMUM:
4032 return TLI.expandFMINIMUM_FMAXIMUM(N, DAG);
4033
4034 // wasi-libc and emscripten do not currently define fminimum_numl and
4035 // fmaximum_numl.
4036 case ISD::FMINIMUMNUM:
4037 case ISD::FMAXIMUMNUM:
4038 return TLI.expandFMINIMUMNUM_FMAXIMUMNUM(N, DAG);
4039
4040 default:
4041 return SDValue();
4042 }
4043}
4044
4045SDValue
4046WebAssemblyTargetLowering::PerformDAGCombine(SDNode *N,
4047 DAGCombinerInfo &DCI) const {
4048 switch (N->getOpcode()) {
4049 default:
4050 return SDValue();
4051 case ISD::BITCAST:
4052 return performBitcastCombine(N, DCI);
4053 case ISD::SETCC:
4054 return performSETCCCombine(N, DCI, Subtarget);
4056 return performVECTOR_SHUFFLECombine(N, DCI);
4057 case ISD::SIGN_EXTEND:
4058 case ISD::ZERO_EXTEND:
4059 return performVectorExtendCombine(N, DCI);
4060 case ISD::UINT_TO_FP:
4061 if (auto ExtCombine = performVectorExtendToFPCombine(N, DCI, Subtarget))
4062 return ExtCombine;
4063 return performVectorNonNegToFPCombine(N, DCI);
4064 case ISD::SINT_TO_FP:
4065 return performVectorExtendToFPCombine(N, DCI, Subtarget);
4068 case ISD::FP_ROUND:
4070 return performVectorTruncZeroCombine(N, DCI);
4071 case ISD::FP_TO_SINT:
4072 case ISD::FP_TO_UINT:
4073 return performConvertFPCombine(N, DCI.DAG);
4074 case ISD::TRUNCATE:
4075 return performTruncateCombine(N, DCI);
4077 if (SDValue V = performBitmaskCombine(N, DCI.DAG))
4078 return V;
4079 return performAnyAllCombine(N, DCI.DAG);
4080 }
4081 case ISD::MUL:
4082 return performMulCombine(N, DCI);
4083 case ISD::SHL:
4084 return performShiftCombine(N, DCI);
4085 case ISD::FMINIMUM:
4086 case ISD::FMAXIMUM:
4087 case ISD::FMINIMUMNUM:
4088 case ISD::FMAXIMUMNUM:
4089 return performMinMaxF128Combine(N, DCI.DAG);
4090 }
4091}
static SDValue performMulCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const AArch64Subtarget *Subtarget)
static SDValue performTruncateCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI)
static SDValue performSETCCCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis false
Function Alias Analysis Results
static void fail(const SDLoc &DL, SelectionDAG &DAG, const Twine &Msg, SDValue Val={})
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
Hexagon Common GEP
const HexagonInstrInfo * TII
#define _
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
static SDValue performVECTOR_SHUFFLECombine(SDNode *N, SelectionDAG &DAG, const RISCVSubtarget &Subtarget, const RISCVTargetLowering &TLI)
static SDValue combineVectorSizedSetCCEquality(EVT VT, SDValue X, SDValue Y, ISD::CondCode CC, const SDLoc &DL, SelectionDAG &DAG, const RISCVSubtarget &Subtarget)
Try to map an integer comparison with size > XLEN to vector instructions before type legalization spl...
Contains matchers for matching SelectionDAG nodes and values.
const char * Msg
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static bool callingConvSupported(CallingConv::ID CallConv)
static MachineBasicBlock * LowerFPToInt(MachineInstr &MI, DebugLoc DL, MachineBasicBlock *BB, const TargetInstrInfo &TII, bool IsUnsigned, bool Int64, bool Float64, unsigned LoweredOpcode)
static SDValue TryWideExtMulCombine(SDNode *N, SelectionDAG &DAG)
static MachineBasicBlock * LowerMemcpy(MachineInstr &MI, DebugLoc DL, MachineBasicBlock *BB, const TargetInstrInfo &TII, bool Int64)
static SDValue performVectorExtendToFPCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const WebAssemblySubtarget *Subtarget)
Convert ({u,s}itofp vec) --> ({u,s}itofp ({s,z}ext vec)) so it doesn't get split up into scalar instr...
static std::optional< unsigned > IsWebAssemblyLocal(SDValue Op, SelectionDAG &DAG)
static SDValue performVectorExtendCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue performVectorNonNegToFPCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue unrollVectorShift(SDValue Op, SelectionDAG &DAG)
static SDValue performAnyAllCombine(SDNode *N, SelectionDAG &DAG)
static MachineBasicBlock * LowerCallResults(MachineInstr &CallResults, DebugLoc DL, MachineBasicBlock *BB, const WebAssemblySubtarget *Subtarget, const TargetInstrInfo &TII)
static std::optional< MaskReduceInfo > classifyMaskReduction(SDNode *N)
static SDValue GetExtendHigh(SDValue Op, unsigned UserOpc, EVT VT, SelectionDAG &DAG)
SDValue performConvertFPCombine(SDNode *N, SelectionDAG &DAG)
static SDValue performBitmaskCombine(SDNode *N, SelectionDAG &DAG)
static SDValue performVectorTruncZeroCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static bool IsWebAssemblyGlobal(SDValue Op)
static SDValue combineSmallMaskReduction(SDNode *N, EVT FromVT, unsigned NumElts, const MaskReduceInfo &Info, SelectionDAG &DAG)
static MachineBasicBlock * LowerMemset(MachineInstr &MI, DebugLoc DL, MachineBasicBlock *BB, const TargetInstrInfo &TII, bool Int64)
static bool HasNoSignedZerosOrNaNs(SDValue Op, SelectionDAG &DAG)
SDValue DoubleVectorWidth(SDValue In, unsigned RequiredNumElems, SelectionDAG &DAG)
static SDValue performShiftCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue LowerConvertLow(SDValue Op, SelectionDAG &DAG)
static SDValue extractSubVector(SDValue Vec, unsigned IdxVal, SelectionDAG &DAG, const SDLoc &DL, unsigned VectorWidth)
static SDValue performBitcastCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue truncateVectorWithNARROW(EVT DstVT, SDValue In, const SDLoc &DL, SelectionDAG &DAG)
static SDValue performMinMaxF128Combine(SDNode *N, SelectionDAG &DAG)
static SDValue combineWideMaskReduction(SDNode *N, SDValue Mask, EVT MaskVT, unsigned NumElts, const MaskReduceInfo &Info, SelectionDAG &DAG)
This file defines the interfaces that WebAssembly uses to lower LLVM code into a selection DAG.
This file provides WebAssembly-specific target descriptions.
This file declares WebAssembly-specific per-machine-function information.
This file declares the WebAssembly-specific subclass of TargetSubtarget.
This file declares the WebAssembly-specific subclass of TargetMachine.
This file contains the declaration of the WebAssembly-specific type parsing utility functions.
This file contains the declaration of the WebAssembly-specific utility functions.
X86 cmov Conversion
static constexpr int Concat[]
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
Definition APInt.h:235
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1225
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
an instruction that atomically reads a memory location, combines it with another value,...
@ Add
*p = old + v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
BinOp getOperation() const
LLVM Basic Block Representation.
Definition BasicBlock.h:62
static CCValAssign getMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP, bool IsCustom=false)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
A debug info location.
Definition DebugLoc.h:126
Diagnostic information for unsupported feature in backend.
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
This is a fast-path instruction selection class that generates poor code and doesn't support illegal ...
Definition FastISel.h:67
FunctionLoweringInfo - This contains information that is global to a function that is used when lower...
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
LLVM_ABI unsigned getAddressSpace() const
const GlobalValue * getGlobal() const
ThreadLocalMode getThreadLocalMode() const
Type * getValueType() const
unsigned getTargetFlags() const
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
Tracks which library functions to use for a particular subtarget or function.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
Describe properties that are true of each instruction in the target description file.
Machine Value Type.
bool is128BitVector() const
Return true if this is a 128-bit vector type.
@ INVALID_SIMPLE_VALUE_TYPE
static auto integer_fixedlen_vector_valuetypes()
SimpleValueType SimpleTy
MVT changeVectorElementType(MVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
static auto fixedlen_vector_valuetypes()
bool isFixedLengthVector() const
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
static MVT getIntegerVT(unsigned BitWidth)
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI instr_iterator insert(instr_iterator I, MachineInstr *M)
Insert MI into the instruction list before I, possibly inside a bundle.
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
iterator insertAfter(iterator I, MachineInstr *MI)
Insert MI into the instruction list after I.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
void setFrameAddressIsTaken(bool T)
unsigned getFunctionNumber() const
getFunctionNumber - Return a unique ID for the current function.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
const char * createExternalSymbolName(StringRef Name)
Allocate a string and populate it with the given external symbol name.
MCContext & getContext() const
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineJumpTableInfo * getJumpTableInfo() const
getJumpTableInfo - Return the jump table info object for the current function.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addSym(MCSymbol *Sym, unsigned char TargetFlags=0) const
const MachineInstrBuilder & addFPImm(const ConstantFP *Val) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
mop_range defs()
Returns all explicit operands that are register definitions.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
LLVM_ABI void addOperand(MachineFunction &MF, const MachineOperand &Op)
Add the specified operand to the instruction.
mop_range explicit_uses()
LLVM_ABI void removeOperand(unsigned OpNo)
Erase an operand from an instruction, leaving it with one fewer operand than it started with.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
const std::vector< MachineJumpTableEntry > & getJumpTables() const
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsKill(bool Val=true)
Register getReg() const
getReg - Returns the register number.
bool isFI() const
isFI - Tests if this is a MO_FrameIndex operand.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
static PointerType * getUnqual(LLVMContext &C)
This constructs an opaque pointer to an object in the default address space (address space zero).
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI bool isKnownNeverLogicalZero(SDValue Op, const APInt &DemandedElts, unsigned Depth=0) const
Test whether the given floating point SDValue (or all elements of it, if it is a vector) is known to ...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getSplatValue(SDValue V, bool LegalTypes=false)
If V is a splat vector, return its scalar source operand by extracting that element from the source v...
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
const DataLayout & getDataLayout() const
SDValue getTargetFrameIndex(int FI, EVT VT)
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
LLVM_ABI SDValue getBasicBlock(MachineBasicBlock *MBB)
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
const TargetMachine & getTarget() const
LLVM_ABI SDValue getAnyExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either any-extending or truncat...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
LLVM_ABI SDValue getMCSymbol(MCSymbol *Sym, EVT VT)
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
const SDValue & getBasePtr() const
const SDValue & getOffset() const
const SDValue & getValue() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
TargetInstrInfo - Interface to description of machine instruction set.
Provides information about what library functions are available for the current target.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
const TargetMachine & getTargetMachine() const
unsigned MaxLoadsPerMemcmp
Specify maximum number of load instructions per memcmp call.
LegalizeTypeAction
This enum indicates whether a types are legal for a target, and if not, what action should be used to...
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
virtual TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const
Return the preferred vector type legalization action.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setMinimumJumpTableEntries(unsigned Val)
Indicate the minimum number of blocks to generate jump tables.
void setPartialReduceMLAAction(unsigned Opc, MVT AccVT, MVT InputVT, LegalizeAction Action)
Indicate how a PARTIAL_REDUCE_U/SMLA node with Acc type AccVT and Input type InputVT should be treate...
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
unsigned MaxLoadsPerMemcmpOptSize
Likewise for functions with the OptSize attribute.
virtual bool isBinOp(unsigned Opcode) const
Return true if the node is a math/logic binary operator.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setCondCodeAction(ArrayRef< ISD::CondCode > CCs, MVT VT, LegalizeAction Action)
Indicate that the specified condition code is or isn't supported on the target and indicate what to d...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
bool isOperationLegalOrCustomOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
SDValue expandFMINIMUMNUM_FMAXIMUMNUM(SDNode *N, SelectionDAG &DAG) const
Expand fminimumnum/fmaximumnum into multiple comparison with selects.
SDValue expandFMINIMUM_FMAXIMUM(SDNode *N, SelectionDAG &DAG) const
Expand fminimum/fmaximum into multiple comparison with selects.
bool isPositionIndependent() const
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
TargetLowering(const TargetLowering &)=delete
virtual bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const
Return true if folding a constant offset with the given GlobalAddress is legal.
std::pair< SDValue, SDValue > makeLibCall(SelectionDAG &DAG, RTLIB::LibcallImpl LibcallImpl, EVT RetVT, ArrayRef< SDValue > Ops, MakeLibCallOptions CallOptions, const SDLoc &dl, SDValue Chain=SDValue()) const
Returns a pair of (return value, chain).
Primary interface to the complete machine description for the target machine.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isFunctionTy() const
True if this is an instance of FunctionType.
Definition Type.h:268
static LLVM_ABI Type * getDoubleTy(LLVMContext &C)
Definition Type.cpp:277
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI const Value * stripPointerCastsAndAliases() const
Strip off pointer casts, all-zero GEPs, address space casts, and aliases.
Definition Value.cpp:716
static std::optional< unsigned > getLocalForStackObject(MachineFunction &MF, int FrameIndex)
WebAssemblyTargetLowering(const TargetMachine &TM, const WebAssemblySubtarget &STI)
self_iterator getIterator()
Definition ilist_node.h:123
#define INT64_MIN
Definition DataTypes.h:74
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ Swift
Calling convention for Swift.
Definition CallingConv.h:69
@ PreserveMost
Used for runtime calls that preserves most registers.
Definition CallingConv.h:63
@ CXX_FAST_TLS
Used for access functions.
Definition CallingConv.h:72
@ WASM_EmscriptenInvoke
For emscripten __invoke_* functions.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ PreserveAll
Used for runtime calls that preserves (almost) all registers.
Definition CallingConv.h:66
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ SwiftTail
This follows the Swift calling convention in how arguments are passed but guarantees tail calls will ...
Definition CallingConv.h:87
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ PARTIAL_REDUCE_SMLA
PARTIAL_REDUCE_[U|S]MLA(Accumulator, Input1, Input2) The partial reduction nodes sign or zero extend ...
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ADDC
Carry-setting nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:296
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ PSEUDO_FMIN
PSEUDO_FMIN is strictly equivalent to op0 olt op1 ?
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ GlobalAddress
Definition ISDOpcodes.h:90
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:589
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ SIGN_EXTEND_VECTOR_INREG
SIGN_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register sign-extension of the low ...
Definition ISDOpcodes.h:928
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ FMULADD
FMULADD - Performs a * b + c, with, or without, intermediate rounding.
Definition ISDOpcodes.h:533
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ GlobalTLSAddress
Definition ISDOpcodes.h:91
@ PARTIAL_REDUCE_UMLA
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ BR_CC
BR_CC - Conditional branch.
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:377
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:249
@ SPLAT_VECTOR
SPLAT_VECTOR(VAL) - Returns a vector with the scalar value VAL duplicated in all lanes.
Definition ISDOpcodes.h:682
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ DEBUGTRAP
DEBUGTRAP - Trap intended to get the attention of a debugger.
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ ANY_EXTEND_VECTOR_INREG
ANY_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register any-extension of the low la...
Definition ISDOpcodes.h:917
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:112
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ TargetConstant
TargetConstant* - Like Constant*, but the DAG does not do any folding, simplification,...
Definition ISDOpcodes.h:181
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ ADDE
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:306
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ ExternalSymbol
Definition ISDOpcodes.h:95
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ CLEAR_CACHE
llvm.clear_cache intrinsic Operands: Input Chain, Start Addres, End Address Outputs: Output Chain
@ ZERO_EXTEND_VECTOR_INREG
ZERO_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register zero-extension of the low ...
Definition ISDOpcodes.h:939
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:963
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ VAARG
VAARG - VAARG has four operands: an input chain, a pointer, a SRCVALUE, and the alignment.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:851
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:368
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
OperandFlags
These are flags set on operands, but should be considered private, all access should go through the M...
Definition MCInstrDesc.h:51
auto m_Value()
Match an arbitrary value and ignore it.
CastOperator_match< OpTy, Instruction::BitCast > m_BitCast(const OpTy &Op)
Matches BitCast.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
TernaryOpc_match< T0_P, T1_P, CondCode_match, true, false > m_c_SetCC(const T0_P &LHS, const T1_P &RHS)
Match a SETCC with any condition code, allowing the operands to be commuted.
TernaryOpc_match< T0_P, T1_P, CondCode_match, true, false > m_c_SpecificSetCC(ISD::CondCode CC, const T0_P &LHS, const T1_P &RHS)
Match a SETCC with a specific condition code, allowing the operands to be commuted.
bool sd_match(SDValue N, Pattern &&P)
MCSymbolWasm * getOrCreateFunctionTableSymbol(MCContext &Ctx, const WebAssemblySubtarget *Subtarget)
Returns the __indirect_function_table, for use in call_indirect and in function bitcasts.
bool isWebAssemblyTableType(const Type *Ty)
Return true if the table represents a WebAssembly table type.
MCSymbolWasm * getOrCreateFuncrefCallTableSymbol(MCContext &Ctx, const WebAssemblySubtarget *Subtarget)
Returns the __funcref_call_table, for use in funcref calls when lowered to table.set + call_indirect.
bool isValidAddressSpace(unsigned AS)
FastISel * createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo, const LibcallLoweringInfo *libcallLowering)
bool canLowerReturn(size_t ResultSize, const WebAssemblySubtarget *Subtarget)
Returns true if the function's return value(s) can be lowered directly, i.e., not indirectly via a po...
MachineSDNode * getTLSBase(SelectionDAG &DAG, const SDLoc &DL, const WebAssemblySubtarget *Subtarget, const SDValue Chain=SDValue())
bool isWasmVarAddressSpace(unsigned AS)
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
void computeSignatureVTs(const FunctionType *Ty, const Function *TargetFunc, const Function &ContextFunc, const TargetMachine &TM, SmallVectorImpl< MVT > &Params, SmallVectorImpl< MVT > &Results)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
SDValue peekThroughFreeze(SDValue V)
Return the non-frozen source operand of V if it exists.
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ Add
Sum of integers.
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
auto max_element(R &&Range)
Provide wrappers to std::max_element which take ranges instead of having to pass begin/end explicitly...
Definition STLExtras.h:2104
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
void computeLegalValueVTs(const WebAssemblyTargetLowering &TLI, LLVMContext &Ctx, const DataLayout &DL, Type *Ty, SmallVectorImpl< MVT > &ValueVTs)
constexpr uint64_t NextPowerOf2(uint64_t A)
Returns the next power of two (in 64-bits) that is strictly greater than A.
Definition MathExtras.h:368
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
bool isByteSized() const
Return true if the bit size is a multiple of 8.
Definition ValueTypes.h:266
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
EVT widenIntegerVectorElementType(LLVMContext &Context) const
Return a VT for an integer vector type with the size of the elements doubled.
Definition ValueTypes.h:475
bool isFixedLengthVector() const
Definition ValueTypes.h:199
bool isFixedLengthVectorOf(EVT EltVT) const
Return true if this is a fixed length vector with matching element type.
Definition ValueTypes.h:205
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool bitsGE(EVT VT) const
Return true if this has no less bits than VT.
Definition ValueTypes.h:315
bool is256BitVector() const
Return true if this is a 256-bit vector type.
Definition ValueTypes.h:235
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
Definition ValueTypes.h:121
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
EVT getHalfNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:484
Matching combinators.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
These are IR-level optimization flags that may be propagated to SDNodes.
This structure is used to pass arguments to makeLibCall function.