LLVM 24.0.0git
AutoUpgrade.cpp
Go to the documentation of this file.
1//===-- AutoUpgrade.cpp - Implement auto-upgrade helper functions ---------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the auto-upgrade helper functions.
10// This is where deprecated IR intrinsics and other IR features are updated to
11// current specifications.
12//
13//===----------------------------------------------------------------------===//
14
15#include "llvm/IR/AutoUpgrade.h"
16#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/StringRef.h"
22#include "llvm/IR/Attributes.h"
23#include "llvm/IR/CallingConv.h"
24#include "llvm/IR/Constants.h"
25#include "llvm/IR/DebugInfo.h"
28#include "llvm/IR/Function.h"
29#include "llvm/IR/GlobalValue.h"
30#include "llvm/IR/IRBuilder.h"
31#include "llvm/IR/InstVisitor.h"
32#include "llvm/IR/Instruction.h"
34#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/IntrinsicsAArch64.h"
36#include "llvm/IR/IntrinsicsAMDGPU.h"
37#include "llvm/IR/IntrinsicsARM.h"
38#include "llvm/IR/IntrinsicsNVPTX.h"
39#include "llvm/IR/IntrinsicsRISCV.h"
40#include "llvm/IR/IntrinsicsWebAssembly.h"
41#include "llvm/IR/IntrinsicsX86.h"
42#include "llvm/IR/LLVMContext.h"
43#include "llvm/IR/MDBuilder.h"
44#include "llvm/IR/Metadata.h"
45#include "llvm/IR/Module.h"
47#include "llvm/IR/Value.h"
48#include "llvm/IR/Verifier.h"
52#include "llvm/Support/Debug.h"
56#include "llvm/Support/Regex.h"
59#include <cstdint>
60#include <cstring>
61#include <numeric>
62
63using namespace llvm;
64
65#define DEBUG_TYPE "auto-upgrade"
66
67static cl::opt<bool>
68 DisableAutoUpgradeDebugInfo("disable-auto-upgrade-debug-info",
69 cl::desc("Disable autoupgrade of debug info"));
70
71static void rename(GlobalValue *GV) { GV->setName(GV->getName() + ".old"); }
72
73// Report a fatal error along with the
74// Call Instruction which caused the error
75[[noreturn]] static void reportFatalUsageErrorWithCI(StringRef reason,
76 CallBase *CI) {
77 CI->print(llvm::errs());
78 llvm::errs() << "\n";
80}
81
82// Upgrade the declarations of the SSE4.1 ptest intrinsics whose arguments have
83// changed their type from v4f32 to v2i64.
85 Function *&NewFn) {
86 // Check whether this is an old version of the function, which received
87 // v4f32 arguments.
88 Type *Arg0Type = F->getFunctionType()->getParamType(0);
89 if (Arg0Type != FixedVectorType::get(Type::getFloatTy(F->getContext()), 4))
90 return false;
91
92 // Yes, it's old, replace it with new version.
93 rename(F);
94 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
95 return true;
96}
97
98// Upgrade the declarations of intrinsic functions whose 8-bit immediate mask
99// arguments have changed their type from i32 to i8.
101 Function *&NewFn) {
102 // Check that the last argument is an i32.
103 Type *LastArgType = F->getFunctionType()->getParamType(
104 F->getFunctionType()->getNumParams() - 1);
105 if (!LastArgType->isIntegerTy(32))
106 return false;
107
108 // Move this function aside and map down.
109 rename(F);
110 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
111 return true;
112}
113
114// Upgrade the declaration of fp compare intrinsics that change return type
115// from scalar to vXi1 mask.
117 Function *&NewFn) {
118 // Check if the return type is a vector.
119 if (F->getReturnType()->isVectorTy())
120 return false;
121
122 rename(F);
123 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
124 return true;
125}
126
127// Upgrade the declaration of multiply and add bytes intrinsics whose input
128// arguments' types have changed from vectors of i32 to vectors of i8
130 Function *&NewFn) {
131 // check if input argument type is a vector of i8
132 Type *Arg1Type = F->getFunctionType()->getParamType(1);
133 Type *Arg2Type = F->getFunctionType()->getParamType(2);
134 if (Arg1Type->isVectorTy() &&
135 cast<VectorType>(Arg1Type)->getElementType()->isIntegerTy(8) &&
136 Arg2Type->isVectorTy() &&
137 cast<VectorType>(Arg2Type)->getElementType()->isIntegerTy(8))
138 return false;
139
140 rename(F);
141 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
142 return true;
143}
144
145// Upgrade the declaration of multipy and add words intrinsics whose input
146// arguments' types have changed to vectors of i32 to vectors of i16
148 Function *&NewFn) {
149 // check if input argument type is a vector of i16
150 Type *Arg1Type = F->getFunctionType()->getParamType(1);
151 Type *Arg2Type = F->getFunctionType()->getParamType(2);
152 if (Arg1Type->isVectorTy() &&
153 cast<VectorType>(Arg1Type)->getElementType()->isIntegerTy(16) &&
154 Arg2Type->isVectorTy() &&
155 cast<VectorType>(Arg2Type)->getElementType()->isIntegerTy(16))
156 return false;
157
158 rename(F);
159 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
160 return true;
161}
162
164 Function *&NewFn) {
165 if (F->getReturnType()->getScalarType()->isBFloatTy())
166 return false;
167
168 rename(F);
169 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
170 return true;
171}
172
174 Function *&NewFn) {
175 if (F->getFunctionType()->getParamType(1)->getScalarType()->isBFloatTy())
176 return false;
177
178 rename(F);
179 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
180 return true;
181}
182
184 // All of the intrinsics matches below should be marked with which llvm
185 // version started autoupgrading them. At some point in the future we would
186 // like to use this information to remove upgrade code for some older
187 // intrinsics. It is currently undecided how we will determine that future
188 // point.
189 if (Name.consume_front("avx."))
190 return (Name.starts_with("blend.p") || // Added in 3.7
191 Name == "cvt.ps2.pd.256" || // Added in 3.9
192 Name == "cvtdq2.pd.256" || // Added in 3.9
193 Name == "cvtdq2.ps.256" || // Added in 7.0
194 Name.starts_with("movnt.") || // Added in 3.2
195 Name.starts_with("sqrt.p") || // Added in 7.0
196 Name.starts_with("storeu.") || // Added in 3.9
197 Name.starts_with("vbroadcast.s") || // Added in 3.5
198 Name.starts_with("vbroadcastf128") || // Added in 4.0
199 Name.starts_with("vextractf128.") || // Added in 3.7
200 Name.starts_with("vinsertf128.") || // Added in 3.7
201 Name.starts_with("vperm2f128.") || // Added in 6.0
202 Name.starts_with("vpermil.")); // Added in 3.1
203
204 if (Name.consume_front("avx2."))
205 return (Name == "movntdqa" || // Added in 5.0
206 Name.starts_with("pabs.") || // Added in 6.0
207 Name.starts_with("padds.") || // Added in 8.0
208 Name.starts_with("paddus.") || // Added in 8.0
209 Name.starts_with("pblendd.") || // Added in 3.7
210 Name == "pblendw" || // Added in 3.7
211 Name.starts_with("pbroadcast") || // Added in 3.8
212 Name.starts_with("pcmpeq.") || // Added in 3.1
213 Name.starts_with("pcmpgt.") || // Added in 3.1
214 Name.starts_with("pmax") || // Added in 3.9
215 Name.starts_with("pmin") || // Added in 3.9
216 Name.starts_with("pmovsx") || // Added in 3.9
217 Name.starts_with("pmovzx") || // Added in 3.9
218 Name.starts_with("pmulh.w") || // Added in 24.0
219 Name.starts_with("pmulhu.w") || // Added in 24.0
220 Name == "pmul.dq" || // Added in 7.0
221 Name == "pmulu.dq" || // Added in 7.0
222 Name.starts_with("psll.dq") || // Added in 3.7
223 Name.starts_with("psrl.dq") || // Added in 3.7
224 Name.starts_with("psubs.") || // Added in 8.0
225 Name.starts_with("psubus.") || // Added in 8.0
226 Name.starts_with("vbroadcast") || // Added in 3.8
227 Name == "vbroadcasti128" || // Added in 3.7
228 Name == "vextracti128" || // Added in 3.7
229 Name == "vinserti128" || // Added in 3.7
230 Name == "vperm2i128"); // Added in 6.0
231
232 if (Name.consume_front("avx512.")) {
233 if (Name.consume_front("mask."))
234 // 'avx512.mask.*'
235 return (Name.starts_with("add.p") || // Added in 7.0. 128/256 in 4.0
236 Name.starts_with("and.") || // Added in 3.9
237 Name.starts_with("andn.") || // Added in 3.9
238 Name.starts_with("broadcast.s") || // Added in 3.9
239 Name.starts_with("broadcastf32x4.") || // Added in 6.0
240 Name.starts_with("broadcastf32x8.") || // Added in 6.0
241 Name.starts_with("broadcastf64x2.") || // Added in 6.0
242 Name.starts_with("broadcastf64x4.") || // Added in 6.0
243 Name.starts_with("broadcasti32x4.") || // Added in 6.0
244 Name.starts_with("broadcasti32x8.") || // Added in 6.0
245 Name.starts_with("broadcasti64x2.") || // Added in 6.0
246 Name.starts_with("broadcasti64x4.") || // Added in 6.0
247 Name.starts_with("cmp.b") || // Added in 5.0
248 Name.starts_with("cmp.d") || // Added in 5.0
249 Name.starts_with("cmp.q") || // Added in 5.0
250 Name.starts_with("cmp.w") || // Added in 5.0
251 Name.starts_with("compress.b") || // Added in 9.0
252 Name.starts_with("compress.d") || // Added in 9.0
253 Name.starts_with("compress.p") || // Added in 9.0
254 Name.starts_with("compress.q") || // Added in 9.0
255 Name.starts_with("compress.store.") || // Added in 7.0
256 Name.starts_with("compress.w") || // Added in 9.0
257 Name.starts_with("conflict.") || // Added in 9.0
258 Name.starts_with("cvtdq2pd.") || // Added in 4.0
259 Name.starts_with("cvtdq2ps.") || // Added in 7.0 updated 9.0
260 Name == "cvtpd2dq.256" || // Added in 7.0
261 Name == "cvtpd2ps.256" || // Added in 7.0
262 Name == "cvtps2pd.128" || // Added in 7.0
263 Name == "cvtps2pd.256" || // Added in 7.0
264 Name.starts_with("cvtqq2pd.") || // Added in 7.0 updated 9.0
265 Name == "cvtqq2ps.256" || // Added in 9.0
266 Name == "cvtqq2ps.512" || // Added in 9.0
267 Name == "cvttpd2dq.256" || // Added in 7.0
268 Name == "cvttps2dq.128" || // Added in 7.0
269 Name == "cvttps2dq.256" || // Added in 7.0
270 Name.starts_with("cvtudq2pd.") || // Added in 4.0
271 Name.starts_with("cvtudq2ps.") || // Added in 7.0 updated 9.0
272 Name.starts_with("cvtuqq2pd.") || // Added in 7.0 updated 9.0
273 Name == "cvtuqq2ps.256" || // Added in 9.0
274 Name == "cvtuqq2ps.512" || // Added in 9.0
275 Name.starts_with("dbpsadbw.") || // Added in 7.0
276 Name.starts_with("div.p") || // Added in 7.0. 128/256 in 4.0
277 Name.starts_with("expand.b") || // Added in 9.0
278 Name.starts_with("expand.d") || // Added in 9.0
279 Name.starts_with("expand.load.") || // Added in 7.0
280 Name.starts_with("expand.p") || // Added in 9.0
281 Name.starts_with("expand.q") || // Added in 9.0
282 Name.starts_with("expand.w") || // Added in 9.0
283 Name.starts_with("fpclass.p") || // Added in 7.0
284 Name.starts_with("insert") || // Added in 4.0
285 Name.starts_with("load.") || // Added in 3.9
286 Name.starts_with("loadu.") || // Added in 3.9
287 Name.starts_with("lzcnt.") || // Added in 5.0
288 Name.starts_with("max.p") || // Added in 7.0. 128/256 in 5.0
289 Name.starts_with("min.p") || // Added in 7.0. 128/256 in 5.0
290 Name.starts_with("movddup") || // Added in 3.9
291 Name.starts_with("move.s") || // Added in 4.0
292 Name.starts_with("movshdup") || // Added in 3.9
293 Name.starts_with("movsldup") || // Added in 3.9
294 Name.starts_with("mul.p") || // Added in 7.0. 128/256 in 4.0
295 Name.starts_with("or.") || // Added in 3.9
296 Name.starts_with("pabs.") || // Added in 6.0
297 Name.starts_with("packssdw.") || // Added in 5.0
298 Name.starts_with("packsswb.") || // Added in 5.0
299 Name.starts_with("packusdw.") || // Added in 5.0
300 Name.starts_with("packuswb.") || // Added in 5.0
301 Name.starts_with("padd.") || // Added in 4.0
302 Name.starts_with("padds.") || // Added in 8.0
303 Name.starts_with("paddus.") || // Added in 8.0
304 Name.starts_with("palignr.") || // Added in 3.9
305 Name.starts_with("pand.") || // Added in 3.9
306 Name.starts_with("pandn.") || // Added in 3.9
307 Name.starts_with("pavg") || // Added in 6.0
308 Name.starts_with("pbroadcast") || // Added in 6.0
309 Name.starts_with("pcmpeq.") || // Added in 3.9
310 Name.starts_with("pcmpgt.") || // Added in 3.9
311 Name.starts_with("perm.df.") || // Added in 3.9
312 Name.starts_with("perm.di.") || // Added in 3.9
313 Name.starts_with("permvar.") || // Added in 7.0
314 Name.starts_with("pmaddubs.w.") || // Added in 7.0
315 Name.starts_with("pmaddw.d.") || // Added in 7.0
316 Name.starts_with("pmax") || // Added in 4.0
317 Name.starts_with("pmin") || // Added in 4.0
318 Name == "pmov.qd.256" || // Added in 9.0
319 Name == "pmov.qd.512" || // Added in 9.0
320 Name == "pmov.wb.256" || // Added in 9.0
321 Name == "pmov.wb.512" || // Added in 9.0
322 Name.starts_with("pmovsx") || // Added in 4.0
323 Name.starts_with("pmovzx") || // Added in 4.0
324 Name.starts_with("pmul.dq.") || // Added in 4.0
325 Name.starts_with("pmul.hr.sw.") || // Added in 7.0
326 Name.starts_with("pmulh.w.") || // Added in 7.0
327 Name.starts_with("pmulhu.w.") || // Added in 7.0
328 Name.starts_with("pmull.") || // Added in 4.0
329 Name.starts_with("pmultishift.qb.") || // Added in 8.0
330 Name.starts_with("pmulu.dq.") || // Added in 4.0
331 Name.starts_with("por.") || // Added in 3.9
332 Name.starts_with("prol.") || // Added in 8.0
333 Name.starts_with("prolv.") || // Added in 8.0
334 Name.starts_with("pror.") || // Added in 8.0
335 Name.starts_with("prorv.") || // Added in 8.0
336 Name.starts_with("pshuf.b.") || // Added in 4.0
337 Name.starts_with("pshuf.d.") || // Added in 3.9
338 Name.starts_with("pshufh.w.") || // Added in 3.9
339 Name.starts_with("pshufl.w.") || // Added in 3.9
340 Name.starts_with("psll.d") || // Added in 4.0
341 Name.starts_with("psll.q") || // Added in 4.0
342 Name.starts_with("psll.w") || // Added in 4.0
343 Name.starts_with("pslli") || // Added in 4.0
344 Name.starts_with("psllv") || // Added in 4.0
345 Name.starts_with("psra.d") || // Added in 4.0
346 Name.starts_with("psra.q") || // Added in 4.0
347 Name.starts_with("psra.w") || // Added in 4.0
348 Name.starts_with("psrai") || // Added in 4.0
349 Name.starts_with("psrav") || // Added in 4.0
350 Name.starts_with("psrl.d") || // Added in 4.0
351 Name.starts_with("psrl.q") || // Added in 4.0
352 Name.starts_with("psrl.w") || // Added in 4.0
353 Name.starts_with("psrli") || // Added in 4.0
354 Name.starts_with("psrlv") || // Added in 4.0
355 Name.starts_with("psub.") || // Added in 4.0
356 Name.starts_with("psubs.") || // Added in 8.0
357 Name.starts_with("psubus.") || // Added in 8.0
358 Name.starts_with("pternlog.") || // Added in 7.0
359 Name.starts_with("punpckh") || // Added in 3.9
360 Name.starts_with("punpckl") || // Added in 3.9
361 Name.starts_with("pxor.") || // Added in 3.9
362 Name.starts_with("shuf.f") || // Added in 6.0
363 Name.starts_with("shuf.i") || // Added in 6.0
364 Name.starts_with("shuf.p") || // Added in 4.0
365 Name.starts_with("sqrt.p") || // Added in 7.0
366 Name.starts_with("store.b.") || // Added in 3.9
367 Name.starts_with("store.d.") || // Added in 3.9
368 Name.starts_with("store.p") || // Added in 3.9
369 Name.starts_with("store.q.") || // Added in 3.9
370 Name.starts_with("store.w.") || // Added in 3.9
371 Name == "store.ss" || // Added in 7.0
372 Name.starts_with("storeu.") || // Added in 3.9
373 Name.starts_with("sub.p") || // Added in 7.0. 128/256 in 4.0
374 Name.starts_with("ucmp.") || // Added in 5.0
375 Name.starts_with("unpckh.") || // Added in 3.9
376 Name.starts_with("unpckl.") || // Added in 3.9
377 Name.starts_with("valign.") || // Added in 4.0
378 Name == "vcvtph2ps.128" || // Added in 11.0
379 Name == "vcvtph2ps.256" || // Added in 11.0
380 Name.starts_with("vextract") || // Added in 4.0
381 Name.starts_with("vfmadd.") || // Added in 7.0
382 Name.starts_with("vfmaddsub.") || // Added in 7.0
383 Name.starts_with("vfnmadd.") || // Added in 7.0
384 Name.starts_with("vfnmsub.") || // Added in 7.0
385 Name.starts_with("vpdpbusd.") || // Added in 7.0
386 Name.starts_with("vpdpbusds.") || // Added in 7.0
387 Name.starts_with("vpdpwssd.") || // Added in 7.0
388 Name.starts_with("vpdpwssds.") || // Added in 7.0
389 Name.starts_with("vpermi2var.") || // Added in 7.0
390 Name.starts_with("vpermil.p") || // Added in 3.9
391 Name.starts_with("vpermilvar.") || // Added in 4.0
392 Name.starts_with("vpermt2var.") || // Added in 7.0
393 Name.starts_with("vpmadd52") || // Added in 7.0
394 Name.starts_with("vpshld.") || // Added in 7.0
395 Name.starts_with("vpshldv.") || // Added in 8.0
396 Name.starts_with("vpshrd.") || // Added in 7.0
397 Name.starts_with("vpshrdv.") || // Added in 8.0
398 Name.starts_with("vpshufbitqmb.") || // Added in 8.0
399 Name.starts_with("xor.")); // Added in 3.9
400
401 if (Name.consume_front("mask3."))
402 // 'avx512.mask3.*'
403 return (Name.starts_with("vfmadd.") || // Added in 7.0
404 Name.starts_with("vfmaddsub.") || // Added in 7.0
405 Name.starts_with("vfmsub.") || // Added in 7.0
406 Name.starts_with("vfmsubadd.") || // Added in 7.0
407 Name.starts_with("vfnmsub.")); // Added in 7.0
408
409 if (Name.consume_front("maskz."))
410 // 'avx512.maskz.*'
411 return (Name.starts_with("pternlog.") || // Added in 7.0
412 Name.starts_with("vfmadd.") || // Added in 7.0
413 Name.starts_with("vfmaddsub.") || // Added in 7.0
414 Name.starts_with("vpdpbusd.") || // Added in 7.0
415 Name.starts_with("vpdpbusds.") || // Added in 7.0
416 Name.starts_with("vpdpwssd.") || // Added in 7.0
417 Name.starts_with("vpdpwssds.") || // Added in 7.0
418 Name.starts_with("vpermt2var.") || // Added in 7.0
419 Name.starts_with("vpmadd52") || // Added in 7.0
420 Name.starts_with("vpshldv.") || // Added in 8.0
421 Name.starts_with("vpshrdv.")); // Added in 8.0
422
423 // 'avx512.*'
424 return (Name == "movntdqa" || // Added in 5.0
425 Name == "pmul.dq.512" || // Added in 7.0
426 Name == "pmulu.dq.512" || // Added in 7.0
427 Name.starts_with("broadcastm") || // Added in 6.0
428 Name.starts_with("cmp.p") || // Added in 12.0
429 Name.starts_with("cvtb2mask.") || // Added in 7.0
430 Name.starts_with("cvtd2mask.") || // Added in 7.0
431 Name.starts_with("cvtmask2") || // Added in 5.0
432 Name.starts_with("cvtq2mask.") || // Added in 7.0
433 Name == "cvtusi2sd" || // Added in 7.0
434 Name.starts_with("cvtw2mask.") || // Added in 7.0
435 Name == "kand.w" || // Added in 7.0
436 Name == "kandn.w" || // Added in 7.0
437 Name == "knot.w" || // Added in 7.0
438 Name == "kor.w" || // Added in 7.0
439 Name == "kortestc.w" || // Added in 7.0
440 Name == "kortestz.w" || // Added in 7.0
441 Name.starts_with("kunpck") || // added in 6.0
442 Name == "kxnor.w" || // Added in 7.0
443 Name == "kxor.w" || // Added in 7.0
444 Name.starts_with("padds.") || // Added in 8.0
445 Name.starts_with("pbroadcast") || // Added in 3.9
446 Name.starts_with("pmulh.w") || // Added in 24.0
447 Name.starts_with("pmulhu.w") || // Added in 24.0
448 Name.starts_with("prol") || // Added in 8.0
449 Name.starts_with("pror") || // Added in 8.0
450 Name.starts_with("psll.dq") || // Added in 3.9
451 Name.starts_with("psrl.dq") || // Added in 3.9
452 Name.starts_with("psubs.") || // Added in 8.0
453 Name.starts_with("ptestm") || // Added in 6.0
454 Name.starts_with("ptestnm") || // Added in 6.0
455 Name.starts_with("storent.") || // Added in 3.9
456 Name.starts_with("vbroadcast.s") || // Added in 7.0
457 Name.starts_with("vpshld.") || // Added in 8.0
458 Name.starts_with("vpshrd.")); // Added in 8.0
459 }
460
461 if (Name.consume_front("fma."))
462 return (Name.starts_with("vfmadd.") || // Added in 7.0
463 Name.starts_with("vfmsub.") || // Added in 7.0
464 Name.starts_with("vfmsubadd.") || // Added in 7.0
465 Name.starts_with("vfnmadd.") || // Added in 7.0
466 Name.starts_with("vfnmsub.")); // Added in 7.0
467
468 if (Name.consume_front("fma4."))
469 return Name.starts_with("vfmadd.s"); // Added in 7.0
470
471 if (Name.consume_front("sse."))
472 return (Name == "add.ss" || // Added in 4.0
473 Name == "cvtsi2ss" || // Added in 7.0
474 Name == "cvtsi642ss" || // Added in 7.0
475 Name == "div.ss" || // Added in 4.0
476 Name == "mul.ss" || // Added in 4.0
477 Name.starts_with("sqrt.p") || // Added in 7.0
478 Name == "sqrt.ss" || // Added in 7.0
479 Name.starts_with("storeu.") || // Added in 3.9
480 Name == "sub.ss"); // Added in 4.0
481
482 if (Name.consume_front("sse2."))
483 return (Name == "add.sd" || // Added in 4.0
484 Name == "cvtdq2pd" || // Added in 3.9
485 Name == "cvtdq2ps" || // Added in 7.0
486 Name == "cvtps2pd" || // Added in 3.9
487 Name == "cvtsi2sd" || // Added in 7.0
488 Name == "cvtsi642sd" || // Added in 7.0
489 Name == "cvtss2sd" || // Added in 7.0
490 Name == "div.sd" || // Added in 4.0
491 Name == "mul.sd" || // Added in 4.0
492 Name.starts_with("padds.") || // Added in 8.0
493 Name.starts_with("paddus.") || // Added in 8.0
494 Name.starts_with("pcmpeq.") || // Added in 3.1
495 Name.starts_with("pcmpgt.") || // Added in 3.1
496 Name == "pmaxs.w" || // Added in 3.9
497 Name == "pmaxu.b" || // Added in 3.9
498 Name == "pmins.w" || // Added in 3.9
499 Name == "pminu.b" || // Added in 3.9
500 Name == "pmulh.w" || // Added in 24.0
501 Name == "pmulhu.w" || // Added in 24.0
502 Name == "pmulu.dq" || // Added in 7.0
503 Name.starts_with("pshuf") || // Added in 3.9
504 Name.starts_with("psll.dq") || // Added in 3.7
505 Name.starts_with("psrl.dq") || // Added in 3.7
506 Name.starts_with("psubs.") || // Added in 8.0
507 Name.starts_with("psubus.") || // Added in 8.0
508 Name.starts_with("sqrt.p") || // Added in 7.0
509 Name == "sqrt.sd" || // Added in 7.0
510 Name == "storel.dq" || // Added in 3.9
511 Name.starts_with("storeu.") || // Added in 3.9
512 Name == "sub.sd"); // Added in 4.0
513
514 if (Name.consume_front("sse41."))
515 return (Name.starts_with("blendp") || // Added in 3.7
516 Name == "movntdqa" || // Added in 5.0
517 Name == "pblendw" || // Added in 3.7
518 Name == "pmaxsb" || // Added in 3.9
519 Name == "pmaxsd" || // Added in 3.9
520 Name == "pmaxud" || // Added in 3.9
521 Name == "pmaxuw" || // Added in 3.9
522 Name == "pminsb" || // Added in 3.9
523 Name == "pminsd" || // Added in 3.9
524 Name == "pminud" || // Added in 3.9
525 Name == "pminuw" || // Added in 3.9
526 Name.starts_with("pmovsx") || // Added in 3.8
527 Name.starts_with("pmovzx") || // Added in 3.9
528 Name == "pmuldq"); // Added in 7.0
529
530 if (Name.consume_front("sse42."))
531 return Name == "crc32.64.8"; // Added in 3.4
532
533 if (Name.consume_front("sse4a."))
534 return Name.starts_with("movnt."); // Added in 3.9
535
536 if (Name.consume_front("ssse3."))
537 return (Name == "pabs.b.128" || // Added in 6.0
538 Name == "pabs.d.128" || // Added in 6.0
539 Name == "pabs.w.128"); // Added in 6.0
540
541 if (Name.consume_front("xop."))
542 return (Name == "vpcmov" || // Added in 3.8
543 Name == "vpcmov.256" || // Added in 5.0
544 Name.starts_with("vpcom") || // Added in 3.2, Updated in 9.0
545 Name.starts_with("vprot")); // Added in 8.0
546
547 if (Name.consume_front("bmi."))
548 return (Name.starts_with("pdep.") || // Added in 23.0
549 Name.starts_with("pext.")); // Added in 23.0
550
551 return (Name == "addcarry.u32" || // Added in 8.0
552 Name == "addcarry.u64" || // Added in 8.0
553 Name == "addcarryx.u32" || // Added in 8.0
554 Name == "addcarryx.u64" || // Added in 8.0
555 Name == "subborrow.u32" || // Added in 8.0
556 Name == "subborrow.u64" || // Added in 8.0
557 Name.starts_with("vcvtph2ps.")); // Added in 11.0
558}
559
561 Function *&NewFn) {
562 // Only handle intrinsics that start with "x86.".
563 if (!Name.consume_front("x86."))
564 return false;
565
566 if (shouldUpgradeX86Intrinsic(F, Name)) {
567 NewFn = nullptr;
568 return true;
569 }
570
571 if (Name == "rdtscp") { // Added in 8.0
572 // If this intrinsic has 0 operands, it's the new version.
573 if (F->getFunctionType()->getNumParams() == 0)
574 return false;
575
576 rename(F);
577 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
578 Intrinsic::x86_rdtscp);
579 return true;
580 }
581
582 Intrinsic::ID ID;
583
584 // SSE4.1 ptest functions may have an old signature.
585 if (Name.consume_front("sse41.ptest")) { // Added in 3.2
587 .Case("c", Intrinsic::x86_sse41_ptestc)
588 .Case("z", Intrinsic::x86_sse41_ptestz)
589 .Case("nzc", Intrinsic::x86_sse41_ptestnzc)
591 if (ID != Intrinsic::not_intrinsic)
592 return upgradePTESTIntrinsic(F, ID, NewFn);
593
594 return false;
595 }
596
597 // Several blend and other instructions with masks used the wrong number of
598 // bits.
599
600 // Added in 3.6
602 .Case("sse41.insertps", Intrinsic::x86_sse41_insertps)
603 .Case("sse41.dppd", Intrinsic::x86_sse41_dppd)
604 .Case("sse41.dpps", Intrinsic::x86_sse41_dpps)
605 .Case("sse41.mpsadbw", Intrinsic::x86_sse41_mpsadbw)
606 .Case("avx.dp.ps.256", Intrinsic::x86_avx_dp_ps_256)
607 .Case("avx2.mpsadbw", Intrinsic::x86_avx2_mpsadbw)
609 if (ID != Intrinsic::not_intrinsic)
610 return upgradeX86IntrinsicsWith8BitMask(F, ID, NewFn);
611
612 if (Name.consume_front("avx512.")) {
613 if (Name.consume_front("mask.cmp.")) {
614 // Added in 7.0
616 .Case("pd.128", Intrinsic::x86_avx512_mask_cmp_pd_128)
617 .Case("pd.256", Intrinsic::x86_avx512_mask_cmp_pd_256)
618 .Case("pd.512", Intrinsic::x86_avx512_mask_cmp_pd_512)
619 .Case("ps.128", Intrinsic::x86_avx512_mask_cmp_ps_128)
620 .Case("ps.256", Intrinsic::x86_avx512_mask_cmp_ps_256)
621 .Case("ps.512", Intrinsic::x86_avx512_mask_cmp_ps_512)
623 if (ID != Intrinsic::not_intrinsic)
624 return upgradeX86MaskedFPCompare(F, ID, NewFn);
625 } else if (Name.starts_with("vpdpbusd.") ||
626 Name.starts_with("vpdpbusds.")) {
627 // Added in 21.1
629 .Case("vpdpbusd.128", Intrinsic::x86_avx512_vpdpbusd_128)
630 .Case("vpdpbusd.256", Intrinsic::x86_avx512_vpdpbusd_256)
631 .Case("vpdpbusd.512", Intrinsic::x86_avx512_vpdpbusd_512)
632 .Case("vpdpbusds.128", Intrinsic::x86_avx512_vpdpbusds_128)
633 .Case("vpdpbusds.256", Intrinsic::x86_avx512_vpdpbusds_256)
634 .Case("vpdpbusds.512", Intrinsic::x86_avx512_vpdpbusds_512)
636 if (ID != Intrinsic::not_intrinsic)
637 return upgradeX86MultiplyAddBytes(F, ID, NewFn);
638 } else if (Name.starts_with("vpdpwssd.") ||
639 Name.starts_with("vpdpwssds.")) {
640 // Added in 21.1
642 .Case("vpdpwssd.128", Intrinsic::x86_avx512_vpdpwssd_128)
643 .Case("vpdpwssd.256", Intrinsic::x86_avx512_vpdpwssd_256)
644 .Case("vpdpwssd.512", Intrinsic::x86_avx512_vpdpwssd_512)
645 .Case("vpdpwssds.128", Intrinsic::x86_avx512_vpdpwssds_128)
646 .Case("vpdpwssds.256", Intrinsic::x86_avx512_vpdpwssds_256)
647 .Case("vpdpwssds.512", Intrinsic::x86_avx512_vpdpwssds_512)
649 if (ID != Intrinsic::not_intrinsic)
650 return upgradeX86MultiplyAddWords(F, ID, NewFn);
651 }
652 return false; // No other 'x86.avx512.*'.
653 }
654
655 if (Name.consume_front("avx2.")) {
656 if (Name.consume_front("vpdpb")) {
657 // Added in 21.1
659 .Case("ssd.128", Intrinsic::x86_avx2_vpdpbssd_128)
660 .Case("ssd.256", Intrinsic::x86_avx2_vpdpbssd_256)
661 .Case("ssds.128", Intrinsic::x86_avx2_vpdpbssds_128)
662 .Case("ssds.256", Intrinsic::x86_avx2_vpdpbssds_256)
663 .Case("sud.128", Intrinsic::x86_avx2_vpdpbsud_128)
664 .Case("sud.256", Intrinsic::x86_avx2_vpdpbsud_256)
665 .Case("suds.128", Intrinsic::x86_avx2_vpdpbsuds_128)
666 .Case("suds.256", Intrinsic::x86_avx2_vpdpbsuds_256)
667 .Case("uud.128", Intrinsic::x86_avx2_vpdpbuud_128)
668 .Case("uud.256", Intrinsic::x86_avx2_vpdpbuud_256)
669 .Case("uuds.128", Intrinsic::x86_avx2_vpdpbuuds_128)
670 .Case("uuds.256", Intrinsic::x86_avx2_vpdpbuuds_256)
672 if (ID != Intrinsic::not_intrinsic)
673 return upgradeX86MultiplyAddBytes(F, ID, NewFn);
674 } else if (Name.consume_front("vpdpw")) {
675 // Added in 21.1
677 .Case("sud.128", Intrinsic::x86_avx2_vpdpwsud_128)
678 .Case("sud.256", Intrinsic::x86_avx2_vpdpwsud_256)
679 .Case("suds.128", Intrinsic::x86_avx2_vpdpwsuds_128)
680 .Case("suds.256", Intrinsic::x86_avx2_vpdpwsuds_256)
681 .Case("usd.128", Intrinsic::x86_avx2_vpdpwusd_128)
682 .Case("usd.256", Intrinsic::x86_avx2_vpdpwusd_256)
683 .Case("usds.128", Intrinsic::x86_avx2_vpdpwusds_128)
684 .Case("usds.256", Intrinsic::x86_avx2_vpdpwusds_256)
685 .Case("uud.128", Intrinsic::x86_avx2_vpdpwuud_128)
686 .Case("uud.256", Intrinsic::x86_avx2_vpdpwuud_256)
687 .Case("uuds.128", Intrinsic::x86_avx2_vpdpwuuds_128)
688 .Case("uuds.256", Intrinsic::x86_avx2_vpdpwuuds_256)
690 if (ID != Intrinsic::not_intrinsic)
691 return upgradeX86MultiplyAddWords(F, ID, NewFn);
692 }
693 return false; // No other 'x86.avx2.*'
694 }
695
696 if (Name.consume_front("avx10.")) {
697 if (Name.consume_front("vpdpb")) {
698 // Added in 21.1
700 .Case("ssd.512", Intrinsic::x86_avx10_vpdpbssd_512)
701 .Case("ssds.512", Intrinsic::x86_avx10_vpdpbssds_512)
702 .Case("sud.512", Intrinsic::x86_avx10_vpdpbsud_512)
703 .Case("suds.512", Intrinsic::x86_avx10_vpdpbsuds_512)
704 .Case("uud.512", Intrinsic::x86_avx10_vpdpbuud_512)
705 .Case("uuds.512", Intrinsic::x86_avx10_vpdpbuuds_512)
707 if (ID != Intrinsic::not_intrinsic)
708 return upgradeX86MultiplyAddBytes(F, ID, NewFn);
709 } else if (Name.consume_front("vpdpw")) {
711 .Case("sud.512", Intrinsic::x86_avx10_vpdpwsud_512)
712 .Case("suds.512", Intrinsic::x86_avx10_vpdpwsuds_512)
713 .Case("usd.512", Intrinsic::x86_avx10_vpdpwusd_512)
714 .Case("usds.512", Intrinsic::x86_avx10_vpdpwusds_512)
715 .Case("uud.512", Intrinsic::x86_avx10_vpdpwuud_512)
716 .Case("uuds.512", Intrinsic::x86_avx10_vpdpwuuds_512)
718 if (ID != Intrinsic::not_intrinsic)
719 return upgradeX86MultiplyAddWords(F, ID, NewFn);
720 }
721 return false; // No other 'x86.avx10.*'
722 }
723
724 if (Name.consume_front("avx512bf16.")) {
725 // Added in 9.0
727 .Case("cvtne2ps2bf16.128",
728 Intrinsic::x86_avx512bf16_cvtne2ps2bf16_128)
729 .Case("cvtne2ps2bf16.256",
730 Intrinsic::x86_avx512bf16_cvtne2ps2bf16_256)
731 .Case("cvtne2ps2bf16.512",
732 Intrinsic::x86_avx512bf16_cvtne2ps2bf16_512)
733 .Case("mask.cvtneps2bf16.128",
734 Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128)
735 .Case("cvtneps2bf16.256",
736 Intrinsic::x86_avx512bf16_cvtneps2bf16_256)
737 .Case("cvtneps2bf16.512",
738 Intrinsic::x86_avx512bf16_cvtneps2bf16_512)
740 if (ID != Intrinsic::not_intrinsic)
741 return upgradeX86BF16Intrinsic(F, ID, NewFn);
742
743 // Added in 9.0
745 .Case("dpbf16ps.128", Intrinsic::x86_avx512bf16_dpbf16ps_128)
746 .Case("dpbf16ps.256", Intrinsic::x86_avx512bf16_dpbf16ps_256)
747 .Case("dpbf16ps.512", Intrinsic::x86_avx512bf16_dpbf16ps_512)
749 if (ID != Intrinsic::not_intrinsic)
750 return upgradeX86BF16DPIntrinsic(F, ID, NewFn);
751 return false; // No other 'x86.avx512bf16.*'.
752 }
753
754 if (Name.consume_front("xop.")) {
756 if (Name.starts_with("vpermil2")) { // Added in 3.9
757 // Upgrade any XOP PERMIL2 index operand still using a float/double
758 // vector.
759 auto Idx = F->getFunctionType()->getParamType(2);
760 if (Idx->isFPOrFPVectorTy()) {
761 unsigned IdxSize = Idx->getPrimitiveSizeInBits();
762 unsigned EltSize = Idx->getScalarSizeInBits();
763 if (EltSize == 64 && IdxSize == 128)
764 ID = Intrinsic::x86_xop_vpermil2pd;
765 else if (EltSize == 32 && IdxSize == 128)
766 ID = Intrinsic::x86_xop_vpermil2ps;
767 else if (EltSize == 64 && IdxSize == 256)
768 ID = Intrinsic::x86_xop_vpermil2pd_256;
769 else
770 ID = Intrinsic::x86_xop_vpermil2ps_256;
771 }
772 } else if (F->arg_size() == 2)
773 // frcz.ss/sd may need to have an argument dropped. Added in 3.2
775 .Case("vfrcz.ss", Intrinsic::x86_xop_vfrcz_ss)
776 .Case("vfrcz.sd", Intrinsic::x86_xop_vfrcz_sd)
778
779 if (ID != Intrinsic::not_intrinsic) {
780 rename(F);
781 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
782 return true;
783 }
784 return false; // No other 'x86.xop.*'
785 }
786
787 if (Name == "seh.recoverfp") {
788 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
789 Intrinsic::eh_recoverfp);
790 return true;
791 }
792
793 return false;
794}
795
796// Upgrade ARM (IsArm) or Aarch64 (!IsArm) intrinsic fns. Return true iff so.
797// IsArm: 'arm.*', !IsArm: 'aarch64.*'.
799 StringRef Name,
800 Function *&NewFn) {
801 if (Name.starts_with("rbit")) {
802 // '(arm|aarch64).rbit'.
804 F->getParent(), Intrinsic::bitreverse, F->arg_begin()->getType());
805 return true;
806 }
807
808 if (Name == "thread.pointer") {
809 // '(arm|aarch64).thread.pointer'.
811 F->getParent(), Intrinsic::thread_pointer, F->getReturnType());
812 return true;
813 }
814
815 bool Neon = Name.consume_front("neon.");
816 if (Neon) {
817 // '(arm|aarch64).neon.*'.
818 // Changed in 12.0: bfdot accept v4bf16 and v8bf16 instead of v8i8 and
819 // v16i8 respectively.
820 if (Name.consume_front("bfdot.")) {
821 // (arm|aarch64).neon.bfdot.*'.
822 Intrinsic::ID ID =
824 .Cases({"v2f32.v8i8", "v4f32.v16i8"},
825 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfdot
826 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfdot)
828 if (ID != Intrinsic::not_intrinsic) {
829 size_t OperandWidth = F->getReturnType()->getPrimitiveSizeInBits();
830 assert((OperandWidth == 64 || OperandWidth == 128) &&
831 "Unexpected operand width");
832 LLVMContext &Ctx = F->getParent()->getContext();
833 std::array<Type *, 2> Tys{
834 {F->getReturnType(),
835 FixedVectorType::get(Type::getBFloatTy(Ctx), OperandWidth / 16)}};
836 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID, Tys);
837 return true;
838 }
839 return false; // No other '(arm|aarch64).neon.bfdot.*'.
840 }
841
842 // Changed in 12.0: bfmmla, bfmlalb and bfmlalt are not polymorphic
843 // anymore and accept v8bf16 instead of v16i8.
844 if (Name.consume_front("bfm")) {
845 // (arm|aarch64).neon.bfm*'.
846 if (Name.consume_back(".v4f32.v16i8")) {
847 // (arm|aarch64).neon.bfm*.v4f32.v16i8'.
848 Intrinsic::ID ID =
850 .Case("mla",
851 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmmla
852 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmmla)
853 .Case("lalb",
854 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmlalb
855 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmlalb)
856 .Case("lalt",
857 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmlalt
858 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmlalt)
860 if (ID != Intrinsic::not_intrinsic) {
861 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
862 return true;
863 }
864 return false; // No other '(arm|aarch64).neon.bfm*.v16i8'.
865 }
866 return false; // No other '(arm|aarch64).neon.bfm*.
867 }
868 // Continue on to Aarch64 Neon or Arm Neon.
869 }
870 // Continue on to Arm or Aarch64.
871
872 if (IsArm) {
873 // 'arm.*'.
874 if (Neon) {
875 // 'arm.neon.*'.
877 .StartsWith("vclz.", Intrinsic::ctlz)
878 .StartsWith("vcnt.", Intrinsic::ctpop)
879 .StartsWith("vqadds.", Intrinsic::sadd_sat)
880 .StartsWith("vqaddu.", Intrinsic::uadd_sat)
881 .StartsWith("vqsubs.", Intrinsic::ssub_sat)
882 .StartsWith("vqsubu.", Intrinsic::usub_sat)
883 .StartsWith("vrinta.", Intrinsic::round)
884 .StartsWith("vrintn.", Intrinsic::roundeven)
885 .StartsWith("vrintm.", Intrinsic::floor)
886 .StartsWith("vrintp.", Intrinsic::ceil)
887 .StartsWith("vrintx.", Intrinsic::rint)
888 .StartsWith("vrintz.", Intrinsic::trunc)
890 if (ID != Intrinsic::not_intrinsic) {
891 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
892 F->arg_begin()->getType());
893 return true;
894 }
895
896 if (Name.consume_front("vst")) {
897 // 'arm.neon.vst*'.
898 static const Regex vstRegex("^([1234]|[234]lane)\\.v[a-z0-9]*$");
900 if (vstRegex.match(Name, &Groups)) {
901 static const Intrinsic::ID StoreInts[] = {
902 Intrinsic::arm_neon_vst1, Intrinsic::arm_neon_vst2,
903 Intrinsic::arm_neon_vst3, Intrinsic::arm_neon_vst4};
904
905 static const Intrinsic::ID StoreLaneInts[] = {
906 Intrinsic::arm_neon_vst2lane, Intrinsic::arm_neon_vst3lane,
907 Intrinsic::arm_neon_vst4lane};
908
909 auto fArgs = F->getFunctionType()->params();
910 Type *Tys[] = {fArgs[0], fArgs[1]};
911 if (Groups[1].size() == 1)
913 F->getParent(), StoreInts[fArgs.size() - 3], Tys);
914 else
916 F->getParent(), StoreLaneInts[fArgs.size() - 5], Tys);
917 return true;
918 }
919 return false; // No other 'arm.neon.vst*'.
920 }
921
922 return false; // No other 'arm.neon.*'.
923 }
924
925 if (Name.consume_front("mve.")) {
926 // 'arm.mve.*'.
927 if (Name == "vctp64") {
928 if (cast<FixedVectorType>(F->getReturnType())->getNumElements() == 4) {
929 // A vctp64 returning a v4i1 is converted to return a v2i1. Rename
930 // the function and deal with it below in UpgradeIntrinsicCall.
931 rename(F);
932 return true;
933 }
934 return false; // Not 'arm.mve.vctp64'.
935 }
936
937 if (Name.starts_with("vrintn.v")) {
939 F->getParent(), Intrinsic::roundeven, F->arg_begin()->getType());
940 return true;
941 }
942
943 // These too are changed to accept a v2i1 instead of the old v4i1.
944 if (Name.consume_back(".v4i1")) {
945 // 'arm.mve.*.v4i1'.
946 if (Name.consume_back(".predicated.v2i64.v4i32"))
947 // 'arm.mve.*.predicated.v2i64.v4i32.v4i1'
948 return Name == "mull.int" || Name == "vqdmull";
949
950 if (Name.consume_back(".v2i64")) {
951 // 'arm.mve.*.v2i64.v4i1'
952 bool IsGather = Name.consume_front("vldr.gather.");
953 if (IsGather || Name.consume_front("vstr.scatter.")) {
954 if (Name.consume_front("base.")) {
955 // Optional 'wb.' prefix.
956 Name.consume_front("wb.");
957 // 'arm.mve.(vldr.gather|vstr.scatter).base.(wb.)?
958 // predicated.v2i64.v2i64.v4i1'.
959 return Name == "predicated.v2i64";
960 }
961
962 if (Name.consume_front("offset.predicated."))
963 return Name == (IsGather ? "v2i64.p0i64" : "p0i64.v2i64") ||
964 Name == (IsGather ? "v2i64.p0" : "p0.v2i64");
965
966 // No other 'arm.mve.(vldr.gather|vstr.scatter).*.v2i64.v4i1'.
967 return false;
968 }
969
970 return false; // No other 'arm.mve.*.v2i64.v4i1'.
971 }
972 return false; // No other 'arm.mve.*.v4i1'.
973 }
974 return false; // No other 'arm.mve.*'.
975 }
976
977 if (Name.consume_front("cde.vcx")) {
978 // 'arm.cde.vcx*'.
979 if (Name.consume_back(".predicated.v2i64.v4i1"))
980 // 'arm.cde.vcx*.predicated.v2i64.v4i1'.
981 return Name == "1q" || Name == "1qa" || Name == "2q" || Name == "2qa" ||
982 Name == "3q" || Name == "3qa";
983
984 return false; // No other 'arm.cde.vcx*'.
985 }
986 } else {
987 // 'aarch64.*'.
988 if (Neon) {
989 // 'aarch64.neon.*'.
991 .StartsWith("frintn", Intrinsic::roundeven)
992 .StartsWith("rbit", Intrinsic::bitreverse)
994 if (ID != Intrinsic::not_intrinsic) {
995 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
996 F->arg_begin()->getType());
997 return true;
998 }
999
1000 Intrinsic::ID MinMaxID =
1001 StringSwitch<Intrinsic::ID>(Name.split('.').first)
1002 .Case("smax", Intrinsic::smax)
1003 .Case("smin", Intrinsic::smin)
1004 .Case("umax", Intrinsic::umax)
1005 .Case("umin", Intrinsic::umin)
1007 if (MinMaxID != Intrinsic::not_intrinsic) {
1008 if (F->arg_size() != 2 || !F->getReturnType()->isIntOrIntVectorTy())
1009 return false; // Invalid IR.
1010 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), MinMaxID,
1011 F->getReturnType());
1012 return true;
1013 }
1014
1015 if (Name.starts_with("addp")) {
1016 // 'aarch64.neon.addp*'.
1017 if (F->arg_size() != 2)
1018 return false; // Invalid IR.
1019 VectorType *Ty = dyn_cast<VectorType>(F->getReturnType());
1020 if (Ty && Ty->getElementType()->isFloatingPointTy()) {
1022 F->getParent(), Intrinsic::aarch64_neon_faddp, Ty);
1023 return true;
1024 }
1025 }
1026
1027 // Changed in 20.0: bfcvt/bfcvtn/bcvtn2 have been replaced with fptrunc.
1028 if (Name.starts_with("bfcvt")) {
1029 NewFn = nullptr;
1030 return true;
1031 }
1032
1033 // vcvtfp2hf and vcvthf2fp -> fpext and fptrunc
1034 if (Name == "vcvtfp2hf" || Name == "vcvthf2fp") {
1035 NewFn = nullptr;
1036 return true;
1037 }
1038
1039 return false; // No other 'aarch64.neon.*'.
1040 }
1041 if (Name.consume_front("sve.")) {
1042 // 'aarch64.sve.*'.
1043 if (Name.consume_front("bf")) {
1044 if (Name == "mmla") {
1045 Type *Tys[] = {F->getReturnType(),
1046 std::next(F->arg_begin())->getType()};
1048 F->getParent(), Intrinsic::aarch64_sve_fmmla, Tys);
1049 return true;
1050 }
1051 if (Name.consume_back(".lane")) {
1052 // 'aarch64.sve.bf*.lane'.
1053 Intrinsic::ID ID =
1055 .Case("dot", Intrinsic::aarch64_sve_bfdot_lane_v2)
1056 .Case("mlalb", Intrinsic::aarch64_sve_bfmlalb_lane_v2)
1057 .Case("mlalt", Intrinsic::aarch64_sve_bfmlalt_lane_v2)
1059 if (ID != Intrinsic::not_intrinsic) {
1060 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1061 return true;
1062 }
1063 return false; // No other 'aarch64.sve.bf*.lane'.
1064 }
1065 return false; // No other 'aarch64.sve.bf*'.
1066 }
1067
1068 // 'aarch64.sve.fcvt.bf16f32' || 'aarch64.sve.fcvtnt.bf16f32'
1069 if (Name == "fcvt.bf16f32" || Name == "fcvtnt.bf16f32") {
1070 NewFn = nullptr;
1071 return true;
1072 }
1073
1074 if (Name.consume_front("convert.from.svbool")) {
1075 // 'aarch64.sve.convert.from.svbool'
1076 auto *TTy = dyn_cast<TargetExtType>(F->getReturnType());
1077 if (!TTy || TTy->getName() != "aarch64.svcount")
1078 return false;
1079
1080 Intrinsic::ID ID = Intrinsic::aarch64_sve_convert_to_svcount;
1081 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1082 return true;
1083 }
1084
1085 if (Name.consume_front("convert.to.svbool")) {
1086 // 'aarch64.sve.convert.to.svbool'
1087 auto *TTy = dyn_cast<TargetExtType>(F->arg_begin()->getType());
1088 if (!TTy || TTy->getName() != "aarch64.svcount")
1089 return false;
1090
1091 Intrinsic::ID ID = Intrinsic::aarch64_sve_convert_from_svcount;
1092 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1093 return true;
1094 }
1095
1096 if (Name.consume_front("addqv")) {
1097 // 'aarch64.sve.addqv'.
1098 if (!F->getReturnType()->isFPOrFPVectorTy())
1099 return false;
1100
1101 auto Args = F->getFunctionType()->params();
1102 Type *Tys[] = {F->getReturnType(), Args[1]};
1104 F->getParent(), Intrinsic::aarch64_sve_faddqv, Tys);
1105 return true;
1106 }
1107
1108 if (Name.consume_front("ld")) {
1109 // 'aarch64.sve.ld*'.
1110 static const Regex LdRegex("^[234](.nxv[a-z0-9]+|$)");
1111 if (LdRegex.match(Name)) {
1112 Type *ScalarTy =
1113 cast<VectorType>(F->getReturnType())->getElementType();
1114 ElementCount EC =
1115 cast<VectorType>(F->arg_begin()->getType())->getElementCount();
1116 assert(F->arg_size() == 2 &&
1117 "Expected 2 arguments for ld* intrinsic.");
1118 Type *PtrTy = F->getArg(1)->getType();
1119 Type *Ty = VectorType::get(ScalarTy, EC);
1120 static const Intrinsic::ID LoadIDs[] = {
1121 Intrinsic::aarch64_sve_ld2_sret,
1122 Intrinsic::aarch64_sve_ld3_sret,
1123 Intrinsic::aarch64_sve_ld4_sret,
1124 };
1126 F->getParent(), LoadIDs[Name[0] - '2'], {Ty, PtrTy});
1127 return true;
1128 }
1129 return false; // No other 'aarch64.sve.ld*'.
1130 }
1131
1132 if (Name.consume_front("tuple.")) {
1133 // 'aarch64.sve.tuple.*'.
1134 if (Name.starts_with("get")) {
1135 // 'aarch64.sve.tuple.get*'.
1136 Type *Tys[] = {F->getReturnType(), F->arg_begin()->getType()};
1138 F->getParent(), Intrinsic::vector_extract, Tys);
1139 return true;
1140 }
1141
1142 if (Name.starts_with("set")) {
1143 // 'aarch64.sve.tuple.set*'.
1144 auto Args = F->getFunctionType()->params();
1145 Type *Tys[] = {Args[0], Args[2], Args[1]};
1147 F->getParent(), Intrinsic::vector_insert, Tys);
1148 return true;
1149 }
1150
1151 static const Regex CreateTupleRegex("^create[234](.nxv[a-z0-9]+|$)");
1152 if (CreateTupleRegex.match(Name)) {
1153 // 'aarch64.sve.tuple.create*'.
1154 auto Args = F->getFunctionType()->params();
1155 Type *Tys[] = {F->getReturnType(), Args[1]};
1157 F->getParent(), Intrinsic::vector_insert, Tys);
1158 return true;
1159 }
1160 return false; // No other 'aarch64.sve.tuple.*'.
1161 }
1162
1163 if (Name.starts_with("rev.nxv")) {
1164 // 'aarch64.sve.rev.<Ty>'
1166 F->getParent(), Intrinsic::vector_reverse, F->getReturnType());
1167 return true;
1168 }
1169
1170 return false; // No other 'aarch64.sve.*'.
1171 }
1172 if (Name.consume_front("sme.")) {
1173 // 'aarch64.sme.*'.
1174 if (Name.consume_front("ftmopa.")) {
1175 // The FP8 FTMOPA intrinsics were split out from the non-FP8 FTMOPA
1176 // intrinsics to model their FPMR dependency.
1177 Intrinsic::ID ID =
1179 .Case("za16.nxv16i8", Intrinsic::aarch64_sme_fp8_ftmopa_za16)
1180 .Case("za32.nxv16i8", Intrinsic::aarch64_sme_fp8_ftmopa_za32)
1182 if (ID != Intrinsic::not_intrinsic) {
1183 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1184 return true;
1185 }
1186 return false; // No other 'aarch64.sme.ftmopa.*'.
1187 }
1188
1189 return false; // No other 'aarch64.sme.*'.
1190 }
1191 }
1192 return false; // No other 'arm.*', 'aarch64.*'.
1193}
1194
1195// The TMA G2S (global-to-shared) tensor copy modes that have legacy
1196// declarations requiring an auto-upgrade. The same set applies to the
1197// cluster (g2s) and CTA (g2s_cta) variants.
1198#define NVVM_TMA_G2S_MODES(M) \
1199 M(tile_1d, "tile.1d") \
1200 M(tile_2d, "tile.2d") \
1201 M(tile_3d, "tile.3d") \
1202 M(tile_4d, "tile.4d") \
1203 M(tile_5d, "tile.5d") \
1204 M(tile_gather4_2d, "tile.gather4.2d") \
1205 M(im2col_3d, "im2col.3d") \
1206 M(im2col_4d, "im2col.4d") \
1207 M(im2col_5d, "im2col.5d") \
1208 M(im2col_w_3d, "im2col.w.3d") \
1209 M(im2col_w_4d, "im2col.w.4d") \
1210 M(im2col_w_5d, "im2col.w.5d") \
1211 M(im2col_w_128_3d, "im2col.w.128.3d") \
1212 M(im2col_w_128_4d, "im2col.w.128.4d") \
1213 M(im2col_w_128_5d, "im2col.w.128.5d")
1214
1215// Two legacy tails are:
1216//
1217// arg1, arg2, .. i64 %ch, i1 %flag_mc, i1 %flag_ch
1218// arg1, arg2, .. i64 %ch, i1 %flag_mc, i1 %flag_ch, i32 %cta_group
1219//
1220// The current tail appends a trailing i32 %flag_valid_pattern, so both
1221// legacy tails are recognized by an i1 at parameter N-2.
1222static Intrinsic::ID
1224 SmallVectorImpl<Type *> &OvlTys) {
1225 if (!Name.consume_front("cp.async.bulk.tensor.g2s."))
1227
1228#define G2S_ID(ID_SUFFIX, NAME) \
1229 .Case(NAME, Intrinsic::nvvm_cp_async_bulk_tensor_g2s_##ID_SUFFIX)
1230 // clang-format off
1234#undef G2S_ID
1235 // clang-format on
1236 if (ID == Intrinsic::not_intrinsic)
1237 return ID;
1238
1239 size_t NumParams = F->getFunctionType()->getNumParams();
1240
1241 // Parameter N-2 is i1 for both legacy tails; the current tail ends
1242 // with i32 %cta_group, i32 %flag_valid_pattern, for which N-2 is i32.
1243 if (!F->getFunctionType()->getParamType(NumParams - 2)->isIntegerTy(1))
1245
1246 // The multicast mask is the parameter immediately before the i64
1247 // cache-hint: N-4 for the 2-flag tail, N-5 for the 3-flag tail.
1248 ArrayRef<Type *> Params = F->getFunctionType()->params();
1249 size_t MaskIdx =
1250 Params[NumParams - 1]->isIntegerTy(1) ? NumParams - 4 : NumParams - 5;
1251 assert(Params[MaskIdx + 1]->isIntegerTy(64) &&
1252 "expected the i64 cache-hint after the multicast mask");
1253 Type *MaskTy = Params[MaskIdx];
1254 assert(MaskTy->isIntegerTy(16) && "unexpected multicast mask type");
1255 OvlTys.push_back(MaskTy);
1256
1257 return ID;
1258}
1259
1260// The legacy tail is:
1261//
1262// arg1, arg2, .. i64 %ch, i1 %flag_ch
1263//
1264// The current tail appends a trailing i32 %flag_valid_pattern, so the
1265// legacy tail is recognized by an i1 at parameter N-1.
1267 StringRef Name) {
1268 if (!Name.consume_front("cp.async.bulk.tensor.g2s.cta."))
1270
1271#define G2S_CTA_ID(ID_SUFFIX, NAME) \
1272 .Case(NAME, Intrinsic::nvvm_cp_async_bulk_tensor_g2s_cta_##ID_SUFFIX)
1273 // clang-format off
1277#undef G2S_CTA_ID
1278 // clang-format on
1279 if (ID == Intrinsic::not_intrinsic)
1280 return ID;
1281
1282 // Parameter N-1 is i1 for the legacy tail; the current tail ends
1283 // with i32 %flag_valid_pattern, for which N-1 is i32.
1284 if (!F->getFunctionType()
1285 ->getParamType(F->getFunctionType()->getNumParams() - 1)
1286 ->isIntegerTy(1))
1288
1289 return ID;
1290}
1291// The legacy tail of llvm.nvvm.cp.async.bulk.global.to.shared.cluster is:
1292//
1293// ..., i16 %mc, i64 %ch, i1 %flag_mc, i1 %flag_ch
1294//
1295// The current intrinsic is overloaded on the multicast-mask type and takes a
1296// trailing i32 %flag_valid_pattern; the legacy tail is recognized by an i1 at
1297// parameter N-1.
1298static Intrinsic::ID
1300 SmallVectorImpl<Type *> &OvlTys) {
1301 if (!Name.consume_front("cp.async.bulk.global.to.shared.cluster"))
1303
1304 // Parameter N-1 is i1 for the legacy tail; the current tail ends with
1305 // i32 %flag_valid_pattern, for which N-1 is i32.
1306 size_t NumParams = F->getFunctionType()->getNumParams();
1307 if (!F->getFunctionType()->getParamType(NumParams - 1)->isIntegerTy(1))
1309
1310 // The multicast mask is parameter 4; legacy IR only uses i16.
1311 Type *MaskTy = F->getFunctionType()->getParamType(NumParams - 4);
1312 if (!MaskTy->isIntegerTy(16))
1314 OvlTys.push_back(MaskTy);
1315
1316 return Intrinsic::nvvm_cp_async_bulk_global_to_shared_cluster;
1317}
1318
1319// The legacy tail of llvm.nvvm.cp.async.bulk.global.to.shared.cta is:
1320//
1321// ..., i64 %ch, i1 %flag_ch
1322//
1323// The current intrinsic adds %ignore_bytes_left/%ignore_bytes_right before
1324// %ch and trailing %flag_oob/%flag_valid_pattern; the legacy tail is
1325// recognized by an i1 at parameter N-1, whereas the current tail ends
1326// with an i32.
1328 StringRef Name) {
1329 if (!Name.consume_front("cp.async.bulk.global.to.shared.cta"))
1331
1332 // Parameter N-1 is i1 for the legacy tail; the current tail ends with
1333 // i32 %flag_valid_pattern, for which N-1 is i32.
1334 if (!F->getFunctionType()->getParamType(5)->isIntegerTy(1))
1336
1337 return Intrinsic::nvvm_cp_async_bulk_global_to_shared_cta;
1338}
1339
1340// The legacy TMA reduction intrinsics encode the reduction operator in their
1341// name, while the current ones take it as an immediate argument. Map the
1342// operator part of a legacy name to the corresponding immediate value.
1343static std::optional<unsigned> getNVPTXTMAReductionOp(StringRef Name) {
1345 .Case("add", static_cast<unsigned>(nvvm::TMAReductionOp::ADD))
1346 .Case("min", static_cast<unsigned>(nvvm::TMAReductionOp::MIN))
1347 .Case("max", static_cast<unsigned>(nvvm::TMAReductionOp::MAX))
1348 .Case("inc", static_cast<unsigned>(nvvm::TMAReductionOp::INC))
1349 .Case("dec", static_cast<unsigned>(nvvm::TMAReductionOp::DEC))
1350 .Case("and", static_cast<unsigned>(nvvm::TMAReductionOp::AND))
1351 .Case("or", static_cast<unsigned>(nvvm::TMAReductionOp::OR))
1352 .Case("xor", static_cast<unsigned>(nvvm::TMAReductionOp::XOR))
1353 .Default(std::nullopt);
1354}
1355
1357 if (!Name.consume_front("cp.async.bulk.tensor.reduce."))
1359
1360 auto [RedOpName, ShapeName] = Name.split('.');
1361 if (!getNVPTXTMAReductionOp(RedOpName))
1363
1364 return StringSwitch<Intrinsic::ID>(ShapeName)
1365 .Case("tile.1d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_1d)
1366 .Case("tile.2d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_2d)
1367 .Case("tile.3d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_3d)
1368 .Case("tile.4d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_4d)
1369 .Case("tile.5d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_5d)
1370 .Case("im2col.3d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_3d)
1371 .Case("im2col.4d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_4d)
1372 .Case("im2col.5d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_5d)
1374}
1375
1377 StringRef Name) {
1378 if (Name.consume_front("mapa.shared.cluster"))
1379 if (F->getReturnType()->getPointerAddressSpace() ==
1381 return Intrinsic::nvvm_mapa_shared_cluster;
1382
1383 if (Name.consume_front("cp.async.bulk.")) {
1384 Intrinsic::ID ID =
1386 .Case("shared.cta.to.cluster",
1387 Intrinsic::nvvm_cp_async_bulk_shared_cta_to_cluster)
1389
1390 if (ID != Intrinsic::not_intrinsic)
1391 if (F->getArg(0)->getType()->getPointerAddressSpace() ==
1393 return ID;
1394 }
1395
1397}
1398
1399static Intrinsic::ID
1401 if (!Name.consume_front("tcgen05.commit."))
1403
1404 if (Name.consume_front("shared."))
1405 return StringSwitch<Intrinsic::ID>(Name)
1406 .Case("cg1", Intrinsic::nvvm_tcgen05_commit_cg1)
1407 .Case("cg2", Intrinsic::nvvm_tcgen05_commit_cg2)
1409
1410 if (Name.consume_front("mc.shared.")) {
1411 // Only upgrade older i16 mc variants.
1412 if (!F->getArg(1)->getType()->isIntegerTy(16))
1414
1415 return StringSwitch<Intrinsic::ID>(Name)
1416 .Case("cg1", Intrinsic::nvvm_tcgen05_commit_mc_cg1)
1417 .Case("cg2", Intrinsic::nvvm_tcgen05_commit_mc_cg2)
1419 }
1420
1422}
1423
1424static Intrinsic::ID
1426 if (F->arg_size() != 2)
1428
1429 if (Name.consume_front("tcgen05.alloc.shared.") ||
1430 Name.consume_front("tcgen05.alloc."))
1431 return StringSwitch<Intrinsic::ID>(Name)
1432 .Case("cg1", Intrinsic::nvvm_tcgen05_alloc_cg1)
1433 .Case("cg2", Intrinsic::nvvm_tcgen05_alloc_cg2)
1435
1436 if (Name.consume_front("tcgen05.dealloc."))
1437 return StringSwitch<Intrinsic::ID>(Name)
1438 .Case("cg1", Intrinsic::nvvm_tcgen05_dealloc_cg1)
1439 .Case("cg2", Intrinsic::nvvm_tcgen05_dealloc_cg2)
1441
1443}
1444
1446 if (Name.consume_front("fma.rn."))
1447 return StringSwitch<Intrinsic::ID>(Name)
1448 .Case("bf16", Intrinsic::nvvm_fma_rn_bf16)
1449 .Case("bf16x2", Intrinsic::nvvm_fma_rn_bf16x2)
1450 .Case("relu.bf16", Intrinsic::nvvm_fma_rn_relu_bf16)
1451 .Case("relu.bf16x2", Intrinsic::nvvm_fma_rn_relu_bf16x2)
1453
1454 if (Name.consume_front("fmax."))
1455 return StringSwitch<Intrinsic::ID>(Name)
1456 .Case("bf16", Intrinsic::nvvm_fmax_bf16)
1457 .Case("bf16x2", Intrinsic::nvvm_fmax_bf16x2)
1458 .Case("ftz.bf16", Intrinsic::nvvm_fmax_ftz_bf16)
1459 .Case("ftz.bf16x2", Intrinsic::nvvm_fmax_ftz_bf16x2)
1460 .Case("ftz.nan.bf16", Intrinsic::nvvm_fmax_ftz_nan_bf16)
1461 .Case("ftz.nan.bf16x2", Intrinsic::nvvm_fmax_ftz_nan_bf16x2)
1462 .Case("ftz.nan.xorsign.abs.bf16",
1463 Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_bf16)
1464 .Case("ftz.nan.xorsign.abs.bf16x2",
1465 Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_bf16x2)
1466 .Case("ftz.xorsign.abs.bf16", Intrinsic::nvvm_fmax_ftz_xorsign_abs_bf16)
1467 .Case("ftz.xorsign.abs.bf16x2",
1468 Intrinsic::nvvm_fmax_ftz_xorsign_abs_bf16x2)
1469 .Case("nan.bf16", Intrinsic::nvvm_fmax_nan_bf16)
1470 .Case("nan.bf16x2", Intrinsic::nvvm_fmax_nan_bf16x2)
1471 .Case("nan.xorsign.abs.bf16", Intrinsic::nvvm_fmax_nan_xorsign_abs_bf16)
1472 .Case("nan.xorsign.abs.bf16x2",
1473 Intrinsic::nvvm_fmax_nan_xorsign_abs_bf16x2)
1474 .Case("xorsign.abs.bf16", Intrinsic::nvvm_fmax_xorsign_abs_bf16)
1475 .Case("xorsign.abs.bf16x2", Intrinsic::nvvm_fmax_xorsign_abs_bf16x2)
1477
1478 if (Name.consume_front("fmin."))
1479 return StringSwitch<Intrinsic::ID>(Name)
1480 .Case("bf16", Intrinsic::nvvm_fmin_bf16)
1481 .Case("bf16x2", Intrinsic::nvvm_fmin_bf16x2)
1482 .Case("ftz.bf16", Intrinsic::nvvm_fmin_ftz_bf16)
1483 .Case("ftz.bf16x2", Intrinsic::nvvm_fmin_ftz_bf16x2)
1484 .Case("ftz.nan.bf16", Intrinsic::nvvm_fmin_ftz_nan_bf16)
1485 .Case("ftz.nan.bf16x2", Intrinsic::nvvm_fmin_ftz_nan_bf16x2)
1486 .Case("ftz.nan.xorsign.abs.bf16",
1487 Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_bf16)
1488 .Case("ftz.nan.xorsign.abs.bf16x2",
1489 Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_bf16x2)
1490 .Case("ftz.xorsign.abs.bf16", Intrinsic::nvvm_fmin_ftz_xorsign_abs_bf16)
1491 .Case("ftz.xorsign.abs.bf16x2",
1492 Intrinsic::nvvm_fmin_ftz_xorsign_abs_bf16x2)
1493 .Case("nan.bf16", Intrinsic::nvvm_fmin_nan_bf16)
1494 .Case("nan.bf16x2", Intrinsic::nvvm_fmin_nan_bf16x2)
1495 .Case("nan.xorsign.abs.bf16", Intrinsic::nvvm_fmin_nan_xorsign_abs_bf16)
1496 .Case("nan.xorsign.abs.bf16x2",
1497 Intrinsic::nvvm_fmin_nan_xorsign_abs_bf16x2)
1498 .Case("xorsign.abs.bf16", Intrinsic::nvvm_fmin_xorsign_abs_bf16)
1499 .Case("xorsign.abs.bf16x2", Intrinsic::nvvm_fmin_xorsign_abs_bf16x2)
1501
1502 if (Name.consume_front("neg."))
1503 return StringSwitch<Intrinsic::ID>(Name)
1504 .Case("bf16", Intrinsic::nvvm_neg_bf16)
1505 .Case("bf16x2", Intrinsic::nvvm_neg_bf16x2)
1507
1509}
1510
1512 FunctionType *NewFnTy = Intrinsic::getType(F->getContext(), IID);
1513 FunctionType *OldFnTy = F->getFunctionType();
1514 auto IsOldBF16StorageTy = [](Type *OldTy, Type *NewTy) {
1515 return OldTy->getScalarType()->isIntegerTy() &&
1516 OldTy->getPrimitiveSizeInBits() == NewTy->getPrimitiveSizeInBits();
1517 };
1518
1519 if (!IsOldBF16StorageTy(OldFnTy->getReturnType(), NewFnTy->getReturnType()))
1520 return false;
1521
1522 if (OldFnTy->getNumParams() != NewFnTy->getNumParams())
1523 return false;
1524
1525 for (unsigned I = 0, E = OldFnTy->getNumParams(); I != E; ++I)
1526 if (!IsOldBF16StorageTy(OldFnTy->getParamType(I), NewFnTy->getParamType(I)))
1527 return false;
1528
1529 return true;
1530}
1531
1532// Overloaded fadd/fmul intrinsic IDs, indexed by [`.ftz`][`.sat`].
1533static constexpr Intrinsic::ID NVVMFAddIIDs[2][2] = {
1534 {Intrinsic::nvvm_fadd, Intrinsic::nvvm_fadd_sat},
1535 {Intrinsic::nvvm_fadd_ftz, Intrinsic::nvvm_fadd_ftz_sat}};
1536static constexpr Intrinsic::ID NVVMFMulIIDs[2][2] = {
1537 {Intrinsic::nvvm_fmul, Intrinsic::nvvm_fmul_sat},
1538 {Intrinsic::nvvm_fmul_ftz, Intrinsic::nvvm_fmul_ftz_sat}};
1539
1540static std::optional<std::pair<Intrinsic::ID, RoundingMode>>
1542 auto [Modifiers, Type] = Name.rsplit('.');
1543 if (!is_contained({"f", "d", "f16", "v2f16"}, Type))
1544 return std::nullopt;
1545
1546 std::optional<llvm::RoundingMode> RoundingMode =
1547 StringSwitch<std::optional<llvm::RoundingMode>>(Modifiers.take_front(2))
1552 .Default(std::nullopt);
1553 if (!RoundingMode)
1554 return std::nullopt;
1555
1556 StringRef Rest = Modifiers.drop_front(2);
1557 const bool IsFTZ = Rest.consume_front(".ftz");
1558 const bool IsSat = Rest.consume_front(".sat");
1559 if (!Rest.empty())
1560 return std::nullopt;
1561
1562 return std::make_pair(IIDs[IsFTZ][IsSat], *RoundingMode);
1563}
1564
1566 if (Name != "mbarrier.init" && Name != "mbarrier.init.shared")
1568
1569 return Intrinsic::nvvm_mbarrier_init;
1570}
1571
1573 return Name.consume_front("local") || Name.consume_front("shared") ||
1574 Name.consume_front("global") || Name.consume_front("constant") ||
1575 Name.consume_front("param");
1576}
1577
1579 if (!Name.consume_front("vp."))
1580 return 0;
1581 return StringSwitch<unsigned>(Name)
1582 .StartsWith("select", Instruction::Select)
1583 .StartsWith("add", Instruction::Add)
1584 .StartsWith("sub", Instruction::Sub)
1585 .StartsWith("mul", Instruction::Mul)
1586 .StartsWith("ashr", Instruction::AShr)
1587 .StartsWith("lshr", Instruction::LShr)
1588 .StartsWith("shl", Instruction::Shl)
1589 .StartsWith("or", Instruction::Or)
1590 .StartsWith("and", Instruction::And)
1591 .StartsWith("xor", Instruction::Xor)
1592 .StartsWith("fadd", Instruction::FAdd)
1593 .StartsWith("fsub", Instruction::FSub)
1594 .StartsWith("fmuladd", 0)
1595 .StartsWith("fmul", Instruction::FMul)
1596 .StartsWith("fdiv", Instruction::FDiv)
1597 .StartsWith("frem", Instruction::FRem)
1598 .StartsWith("fneg", Instruction::FNeg)
1599 .StartsWith("trunc", Instruction::Trunc)
1600 .StartsWith("zext", Instruction::ZExt)
1601 .StartsWith("sext", Instruction::SExt)
1602 .StartsWith("fptrunc", Instruction::FPTrunc)
1603 .StartsWith("fpext", Instruction::FPExt)
1604 .StartsWith("fptoui", Instruction::FPToUI)
1605 .StartsWith("fptosi", Instruction::FPToSI)
1606 .StartsWith("uitofp", Instruction::UIToFP)
1607 .StartsWith("sitofp", Instruction::SIToFP)
1608 .StartsWith("ptrtoint", Instruction::PtrToInt)
1609 .StartsWith("inttoptr", Instruction::IntToPtr)
1610 .StartsWith("icmp", Instruction::ICmp)
1611 .StartsWith("fcmp", Instruction::FCmp)
1612 .Default(0);
1613}
1614
1616 if (!Name.consume_front("vp."))
1617 return 0;
1618 return StringSwitch<Intrinsic::ID>(Name)
1619 .StartsWith("abs", Intrinsic::abs)
1620 .StartsWith("smax", Intrinsic::smax)
1621 .StartsWith("smin", Intrinsic::smin)
1622 .StartsWith("umax", Intrinsic::umax)
1623 .StartsWith("umin", Intrinsic::umin)
1624 .StartsWith("copysign", Intrinsic::copysign)
1625 .StartsWith("minnum", Intrinsic::minnum)
1626 .StartsWith("maxnum", Intrinsic::maxnum)
1627 .StartsWith("minimum", Intrinsic::minimum)
1628 .StartsWith("maximum", Intrinsic::maximum)
1629 .StartsWith("fabs", Intrinsic::fabs)
1630 .StartsWith("sqrt", Intrinsic::sqrt)
1631 .StartsWith("fma", Intrinsic::fma)
1632 .StartsWith("fmuladd", Intrinsic::fmuladd)
1633 .StartsWith("ceil", Intrinsic::ceil)
1634 .StartsWith("floor", Intrinsic::floor)
1635 .StartsWith("rint", Intrinsic::rint)
1636 .StartsWith("nearbyint", Intrinsic::nearbyint)
1637 .StartsWith("roundeven", Intrinsic::roundeven)
1638 .StartsWith("roundtozero", Intrinsic::trunc)
1639 .StartsWith("round", Intrinsic::round)
1640 .StartsWith("lrint", Intrinsic::lrint)
1641 .StartsWith("llrint", Intrinsic::llrint)
1642 .StartsWith("bitreverse", Intrinsic::bitreverse)
1643 .StartsWith("bswap", Intrinsic::bswap)
1644 .StartsWith("ctpop", Intrinsic::ctpop)
1645 .StartsWith("ctlz", Intrinsic::ctlz)
1646 .StartsWith("cttz.elts", 0)
1647 .StartsWith("cttz", Intrinsic::cttz)
1648 .StartsWith("sadd.sat", Intrinsic::sadd_sat)
1649 .StartsWith("uadd.sat", Intrinsic::uadd_sat)
1650 .StartsWith("ssub.sat", Intrinsic::ssub_sat)
1651 .StartsWith("usub.sat", Intrinsic::usub_sat)
1652 .StartsWith("fshl", Intrinsic::fshl)
1653 .StartsWith("fshr", Intrinsic::fshr)
1654 .StartsWith("is.fpclass", Intrinsic::is_fpclass)
1655 .Default(0);
1656}
1657
1661
1663 const FunctionType *FuncTy) {
1664 Type *HalfTy = Type::getHalfTy(FuncTy->getContext());
1665 if (Name.starts_with("to.fp16")) {
1666 return CastInst::castIsValid(Instruction::FPTrunc, FuncTy->getParamType(0),
1667 HalfTy) &&
1668 CastInst::castIsValid(Instruction::BitCast, HalfTy,
1669 FuncTy->getReturnType());
1670 }
1671
1672 if (Name.starts_with("from.fp16")) {
1673 return CastInst::castIsValid(Instruction::BitCast, FuncTy->getParamType(0),
1674 HalfTy) &&
1675 CastInst::castIsValid(Instruction::FPExt, HalfTy,
1676 FuncTy->getReturnType());
1677 }
1678
1679 return false;
1680}
1681
1682static unsigned
1684 SmallVectorImpl<Type *> &OverloadTys) {
1685 auto [FirstDefault, Defaults] = Intrinsic::getAllDefaultArgValues(IID);
1686 if (Defaults.empty())
1687 return 0;
1688
1689 unsigned FullArgCount = FirstDefault + Defaults.size();
1690
1691 // Only trailing default arguments can be missing.
1692 if (F->arg_size() < FirstDefault || F->arg_size() >= FullArgCount)
1693 return 0;
1694
1695 unsigned NumMissingTrailingParams = FullArgCount - F->arg_size();
1696 if (!Intrinsic::isSignatureValid(IID, F->getFunctionType(), OverloadTys,
1697 NumMissingTrailingParams))
1698 return 0;
1699
1700 return FullArgCount;
1701}
1702
1704 Intrinsic::ID IID = F->getIntrinsicID();
1705 SmallVector<Type *, 4> OverloadTys;
1706
1707 unsigned FullArgCount =
1708 getFullArgCountForDefaultArgUpgrade(F, IID, OverloadTys);
1709 if (FullArgCount == 0)
1710 return false;
1711
1712 rename(F);
1713 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID, OverloadTys);
1714 assert(NewFn->arg_size() == FullArgCount &&
1715 "total number of default args does not match intrinsic signature");
1716 return true;
1717}
1718
1720 bool CanUpgradeDebugIntrinsicsToRecords) {
1721 assert(F && "Illegal to upgrade a non-existent Function.");
1722
1723 StringRef Name = F->getName();
1724
1725 // Quickly eliminate it, if it's not a candidate.
1726 if (!Name.consume_front("llvm.") || Name.empty())
1727 return false;
1728
1729 switch (Name[0]) {
1730 default: break;
1731 case 'a': {
1732 bool IsArm = Name.consume_front("arm.");
1733 if (IsArm || Name.consume_front("aarch64.")) {
1734 if (upgradeArmOrAarch64IntrinsicFunction(IsArm, F, Name, NewFn))
1735 return true;
1736 break;
1737 }
1738
1739 if (Name.consume_front("amdgcn.")) {
1740 if (Name == "alignbit") {
1741 // Target specific intrinsic became redundant
1743 F->getParent(), Intrinsic::fshr, {F->getReturnType()});
1744 return true;
1745 }
1746
1747 if (Name.consume_front("atomic.")) {
1748 if (Name.starts_with("inc") || Name.starts_with("dec") ||
1749 Name.starts_with("cond.sub") || Name.starts_with("csub")) {
1750 // These were replaced with atomicrmw uinc_wrap, udec_wrap, usub_cond
1751 // and usub_sat so there's no new declaration.
1752 NewFn = nullptr;
1753 return true;
1754 }
1755 break; // No other 'amdgcn.atomic.*'
1756 }
1757
1758 if (Name.starts_with("addrspacecast.nonnull")) {
1759 // Replaced with an addrspacecast instruction carrying the nonnull flag,
1760 // so there's no new declaration.
1761 NewFn = nullptr;
1762 return true;
1763 }
1764
1765 switch (F->getIntrinsicID()) {
1766 default:
1767 break;
1768 // Legacy wmma iu intrinsics without the optional clamp operand.
1769 case Intrinsic::amdgcn_wmma_i32_16x16x64_iu8:
1770 if (F->arg_size() == 7) {
1771 NewFn = nullptr;
1772 return true;
1773 }
1774 break;
1775 case Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8:
1776 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
1777 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
1778 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
1779 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
1780 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
1781 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16:
1782 if (F->arg_size() == 8) {
1783 NewFn = nullptr;
1784 return true;
1785 }
1786 break;
1787 }
1788
1789 if (Name.consume_front("ds.") || Name.consume_front("global.atomic.") ||
1790 Name.consume_front("flat.atomic.")) {
1791 if (Name.starts_with("fadd") ||
1792 // FIXME: We should also remove fmin.num and fmax.num intrinsics.
1793 (Name.starts_with("fmin") && !Name.starts_with("fmin.num")) ||
1794 (Name.starts_with("fmax") && !Name.starts_with("fmax.num"))) {
1795 // Replaced with atomicrmw fadd/fmin/fmax, so there's no new
1796 // declaration.
1797 NewFn = nullptr;
1798 return true;
1799 }
1800 }
1801
1802 if (Name.starts_with("fcmp.") || Name.starts_with("icmp.")) {
1803 NewFn = nullptr;
1804 return true;
1805 }
1806
1807 if (Name.starts_with("ldexp.")) {
1808 // Target specific intrinsic became redundant
1810 F->getParent(), Intrinsic::ldexp,
1811 {F->getReturnType(), F->getArg(1)->getType()});
1812 return true;
1813 }
1814 break; // No other 'amdgcn.*'
1815 }
1816
1817 break;
1818 }
1819 case 'c': {
1820 if (F->arg_size() == 1) {
1821 if (Name.consume_front("convert.")) {
1822 if (convertIntrinsicValidType(Name, F->getFunctionType())) {
1823 NewFn = nullptr;
1824 return true;
1825 }
1826 }
1827
1829 .StartsWith("ctlz.", Intrinsic::ctlz)
1830 .StartsWith("cttz.", Intrinsic::cttz)
1832 if (ID != Intrinsic::not_intrinsic) {
1833 rename(F);
1834 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
1835 F->arg_begin()->getType());
1836 return true;
1837 }
1838 }
1839
1841 if (Name == "coro.end" &&
1842 (F->arg_size() == 2 || F->getReturnType()->isIntegerTy(1)))
1843 CoroEndID = Intrinsic::coro_end;
1844 else if (Name == "coro.end.async" && F->getReturnType()->isIntegerTy(1))
1845 CoroEndID = Intrinsic::coro_end_async;
1846
1847 if (CoroEndID != Intrinsic::not_intrinsic) {
1848 rename(F);
1849 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), CoroEndID);
1850 return true;
1851 }
1852
1853 break;
1854 }
1855 case 'd':
1856 if (Name.consume_front("dbg.")) {
1857 // Mark debug intrinsics for upgrade to new debug format.
1858 if (CanUpgradeDebugIntrinsicsToRecords) {
1859 if (Name == "addr" || Name == "value" || Name == "assign" ||
1860 Name == "declare" || Name == "label") {
1861 // There's no function to replace these with.
1862 NewFn = nullptr;
1863 // But we do want these to get upgraded.
1864 return true;
1865 }
1866 }
1867 // Update llvm.dbg.addr intrinsics even in "new debug mode"; they'll get
1868 // converted to DbgVariableRecords later.
1869 if (Name == "addr" || (Name == "value" && F->arg_size() == 4)) {
1870 rename(F);
1871 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
1872 Intrinsic::dbg_value);
1873 return true;
1874 }
1875 break; // No other 'dbg.*'.
1876 }
1877 break;
1878 case 'e':
1879 if (Name.consume_front("experimental.vector.")) {
1880 Intrinsic::ID ID =
1882 // Skip over extract.last.active, otherwise it will be 'upgraded'
1883 // to a regular vector extract which is a different operation.
1884 .StartsWith("extract.last.active.", Intrinsic::not_intrinsic)
1885 .StartsWith("extract.", Intrinsic::vector_extract)
1886 .StartsWith("insert.", Intrinsic::vector_insert)
1887 .StartsWith("reverse.", Intrinsic::vector_reverse)
1888 .StartsWith("interleave2.", Intrinsic::vector_interleave2)
1889 .StartsWith("deinterleave2.", Intrinsic::vector_deinterleave2)
1890 .StartsWith("partial.reduce.add",
1891 Intrinsic::vector_partial_reduce_add)
1893 if (ID != Intrinsic::not_intrinsic) {
1894 const auto *FT = F->getFunctionType();
1896 if (ID == Intrinsic::vector_extract ||
1897 ID == Intrinsic::vector_interleave2)
1898 // Extracting overloads the return type.
1899 Tys.push_back(FT->getReturnType());
1900 if (ID != Intrinsic::vector_interleave2)
1901 Tys.push_back(FT->getParamType(0));
1902 if (ID == Intrinsic::vector_insert ||
1903 ID == Intrinsic::vector_partial_reduce_add)
1904 // Inserting overloads the inserted type.
1905 Tys.push_back(FT->getParamType(1));
1906 rename(F);
1907 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID, Tys);
1908 return true;
1909 }
1910
1911 if (Name.consume_front("reduce.")) {
1913 static const Regex R("^([a-z]+)\\.[a-z][0-9]+");
1914 if (R.match(Name, &Groups))
1916 .Case("add", Intrinsic::vector_reduce_add)
1917 .Case("mul", Intrinsic::vector_reduce_mul)
1918 .Case("and", Intrinsic::vector_reduce_and)
1919 .Case("or", Intrinsic::vector_reduce_or)
1920 .Case("xor", Intrinsic::vector_reduce_xor)
1921 .Case("smax", Intrinsic::vector_reduce_smax)
1922 .Case("smin", Intrinsic::vector_reduce_smin)
1923 .Case("umax", Intrinsic::vector_reduce_umax)
1924 .Case("umin", Intrinsic::vector_reduce_umin)
1925 .Case("fmax", Intrinsic::vector_reduce_fmax)
1926 .Case("fmin", Intrinsic::vector_reduce_fmin)
1928
1929 bool V2 = false;
1930 if (ID == Intrinsic::not_intrinsic) {
1931 static const Regex R2("^v2\\.([a-z]+)\\.[fi][0-9]+");
1932 Groups.clear();
1933 V2 = true;
1934 if (R2.match(Name, &Groups))
1936 .Case("fadd", Intrinsic::vector_reduce_fadd)
1937 .Case("fmul", Intrinsic::vector_reduce_fmul)
1939 }
1940 if (ID != Intrinsic::not_intrinsic) {
1941 rename(F);
1942 auto Args = F->getFunctionType()->params();
1943 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
1944 {Args[V2 ? 1 : 0]});
1945 return true;
1946 }
1947 break; // No other 'expermental.vector.reduce.*'.
1948 }
1949
1950 if (Name.consume_front("splice"))
1951 return true;
1952 break; // No other 'experimental.vector.*'.
1953 }
1954 if (Name.consume_front("experimental.stepvector.")) {
1955 Intrinsic::ID ID = Intrinsic::stepvector;
1956 rename(F);
1958 F->getParent(), ID, F->getFunctionType()->getReturnType());
1959 return true;
1960 }
1961 break; // No other 'e*'.
1962 case 'f':
1963 if (Name.starts_with("flt.rounds")) {
1964 rename(F);
1965 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
1966 Intrinsic::get_rounding);
1967 return true;
1968 }
1969 break;
1970 case 'i':
1971 if (Name.starts_with("invariant.group.barrier")) {
1972 // Rename invariant.group.barrier to launder.invariant.group
1973 auto Args = F->getFunctionType()->params();
1974 Type* ObjectPtr[1] = {Args[0]};
1975 rename(F);
1977 F->getParent(), Intrinsic::launder_invariant_group, ObjectPtr);
1978 return true;
1979 }
1980 break;
1981 case 'l': {
1982 bool IsLifetimeStart = Name.consume_front("lifetime.start");
1983 bool IsLifetimeEnd = !IsLifetimeStart && Name.consume_front("lifetime.end");
1984 if (IsLifetimeStart || IsLifetimeEnd) {
1985 if (F->arg_size() == 2) {
1986 Intrinsic::ID IID = IsLifetimeStart ? Intrinsic::lifetime_start
1987 : Intrinsic::lifetime_end;
1988 rename(F);
1989 // Old 2 argument form of these intrinsics have [Size, Ptr] as
1990 // arguments. Use the Ptr argument to create new declaration.
1991 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
1992 F->getArg(1)->getType());
1993 return true;
1994 } else if (F->arg_size() == 1 && Name == ".i64") {
1995 // Matches @llvm.lifetime.{start/end}.i64 which used to be created by
1996 // Autoupgrade prior to
1997 // https://github.com/llvm/llvm-project/pull/204601. This is an invalid
1998 // intrinsic with no expected calls. To allow auto-upgrade process to
1999 // delete such invalid intrinsic declaration, set NewFn = nullptr
2000 // and return true here. If there are actual calls to this intrinsic
2001 // (which is not expected), they will be deleted in
2002 // UpgradeIntrinsicCall.
2003 NewFn = nullptr;
2004 return true;
2005 }
2006 }
2007 break;
2008 }
2009 case 'm': {
2010 // Updating the memory intrinsics (memcpy/memmove/memset) that have an
2011 // alignment parameter to embedding the alignment as an attribute of
2012 // the pointer args.
2013 if (unsigned ID = StringSwitch<unsigned>(Name)
2014 .StartsWith("memcpy.", Intrinsic::memcpy)
2015 .StartsWith("memmove.", Intrinsic::memmove)
2016 .Default(0)) {
2017 if (F->arg_size() == 5) {
2018 rename(F);
2019 // Get the types of dest, src, and len
2020 ArrayRef<Type *> ParamTypes =
2021 F->getFunctionType()->params().slice(0, 3);
2022 NewFn =
2023 Intrinsic::getOrInsertDeclaration(F->getParent(), ID, ParamTypes);
2024 return true;
2025 }
2026 }
2027 if (Name.starts_with("memset.") && F->arg_size() == 5) {
2028 rename(F);
2029 // Get the types of dest, and len
2030 const auto *FT = F->getFunctionType();
2031 Type *ParamTypes[2] = {
2032 FT->getParamType(0), // Dest
2033 FT->getParamType(2) // len
2034 };
2035 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
2036 Intrinsic::memset, ParamTypes);
2037 return true;
2038 }
2039
2040 unsigned MaskedID =
2042 .StartsWith("masked.load", Intrinsic::masked_load)
2043 .StartsWith("masked.gather", Intrinsic::masked_gather)
2044 .StartsWith("masked.store", Intrinsic::masked_store)
2045 .StartsWith("masked.scatter", Intrinsic::masked_scatter)
2046 .Default(0);
2047 if (MaskedID && F->arg_size() == 4) {
2048 rename(F);
2049 if (MaskedID == Intrinsic::masked_load ||
2050 MaskedID == Intrinsic::masked_gather) {
2052 F->getParent(), MaskedID,
2053 {F->getReturnType(), F->getArg(0)->getType()});
2054 return true;
2055 }
2057 F->getParent(), MaskedID,
2058 {F->getArg(0)->getType(), F->getArg(1)->getType()});
2059 return true;
2060 }
2061 break;
2062 }
2063 case 'n': {
2064 if (Name.consume_front("nvvm.")) {
2065 // Check for nvvm intrinsics corresponding exactly to an LLVM intrinsic.
2066 if (F->arg_size() == 1) {
2067 Intrinsic::ID IID =
2069 .Cases({"brev32", "brev64"}, Intrinsic::bitreverse)
2070 .Case("clz.i", Intrinsic::ctlz)
2071 .Case("popc.i", Intrinsic::ctpop)
2073 if (IID != Intrinsic::not_intrinsic) {
2074 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2075 {F->getReturnType()});
2076 return true;
2077 }
2078 } else if (F->arg_size() == 2) {
2079 Intrinsic::ID IID =
2081 .Cases({"max.s", "max.i", "max.ll"}, Intrinsic::smax)
2082 .Cases({"min.s", "min.i", "min.ll"}, Intrinsic::smin)
2083 .Cases({"max.us", "max.ui", "max.ull"}, Intrinsic::umax)
2084 .Cases({"min.us", "min.ui", "min.ull"}, Intrinsic::umin)
2085 .Cases({"mulhi.s", "mulhi.i", "mulhi.ll"}, Intrinsic::smulh)
2086 .Cases({"mulhi.us", "mulhi.ui", "mulhi.ull"}, Intrinsic::umulh)
2088 if (IID != Intrinsic::not_intrinsic) {
2089 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2090 {F->getReturnType()});
2091 return true;
2092 }
2093 }
2094
2095 // Check for nvvm intrinsics that need a return type adjustment.
2096 {
2098 if (IID != Intrinsic::not_intrinsic &&
2100 NewFn = nullptr;
2101 return true;
2102 }
2103 }
2104
2105 // Upgrade Distributed Shared Memory Intrinsics
2107 if (IID != Intrinsic::not_intrinsic) {
2108 rename(F);
2109 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2110 return true;
2111 }
2112
2113 // Upgrade TMA reduction intrinsics
2114 // llvm.nvvm.cp.async.bulk.tensor.reduce.<red_op>* =>
2115 // llvm.nvvm.cp.async.bulk.tensor.reduce.<shape>*
2117 if (IID != Intrinsic::not_intrinsic) {
2118 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2119 return true;
2120 }
2121
2122 // Upgrade tcgen05.commit shared variants to anyptr intrinsics.
2124 if (IID != Intrinsic::not_intrinsic) {
2125 rename(F);
2127 F->getParent(), IID, F->getReturnType(),
2128 F->getFunctionType()->params());
2129 return true;
2130 }
2131
2132 // Upgrade tcgen05.alloc/dealloc with the is_exclusive argument and
2133 // tcgen05.alloc shared variants to anyptr intrinsics.
2135 if (IID != Intrinsic::not_intrinsic) {
2136 rename(F);
2137 if (Intrinsic::isOverloaded(IID))
2138 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2139 {F->getArg(0)->getType()});
2140 else
2141 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2142 return true;
2143 }
2144
2145 // Upgrade TMA copy G2S CTA intrinsics.
2147 if (IID != Intrinsic::not_intrinsic) {
2148 rename(F);
2149 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2150 return true;
2151 }
2152
2153 // Upgrade TMA copy G2S (cluster) intrinsics.
2155 IID = shouldUpgradeNVPTXTMAG2SIntrinsics(F, Name, OvlTys);
2156 if (IID != Intrinsic::not_intrinsic) {
2157 rename(F);
2158 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID, OvlTys);
2159 return true;
2160 }
2161
2162 // Upgrade the legacy cp.async.bulk.global.to.shared.cluster signature
2163 // (multicast-mask overloading + trailing flag_valid_pattern).
2164 SmallVector<Type *, 1> BulkG2SOvlTys;
2165 IID = shouldUpgradeNVPTXBulkG2SClusterIntrinsic(F, Name, BulkG2SOvlTys);
2166 if (IID != Intrinsic::not_intrinsic) {
2167 rename(F);
2168 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2169 BulkG2SOvlTys);
2170 return true;
2171 }
2172
2173 // Upgrade the legacy cp.async.bulk.global.to.shared.cta signature
2174 // (no ignore_bytes_left/right + trailing flag_valid_pattern).
2176 if (IID != Intrinsic::not_intrinsic) {
2177 rename(F);
2178 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2179 return true;
2180 }
2181
2182 // Upgrade mbarrier.init intrinsics missing the layout operand.
2184 if (IID != Intrinsic::not_intrinsic) {
2185 rename(F);
2186 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2187 F->getArg(0)->getType());
2188 return true;
2189 }
2190
2191 // The following nvvm intrinsics correspond exactly to an LLVM idiom, but
2192 // not to an intrinsic alone. We expand them in UpgradeIntrinsicCall.
2193 //
2194 // TODO: We could add lohi.i2d.
2195 bool Expand = false;
2196 if (Name.consume_front("abs."))
2197 // nvvm.abs.{i,ii}
2198 Expand =
2199 Name == "i" || Name == "ll" || Name == "bf16" || Name == "bf16x2";
2200 else if (Name.consume_front("fabs."))
2201 // nvvm.fabs.{f,ftz.f,d}
2202 Expand = Name == "f" || Name == "ftz.f" || Name == "d";
2203 else if (Name.consume_front("add."))
2204 // nvvm.add.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
2205 Expand = getNVVMFPArithUpgrade(Name, NVVMFAddIIDs).has_value();
2206 else if (Name.consume_front("mul."))
2207 // nvvm.mul.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
2208 Expand = getNVVMFPArithUpgrade(Name, NVVMFMulIIDs).has_value();
2209 else if (Name.consume_front("ex2.approx."))
2210 // nvvm.ex2.approx.{f,ftz.f,d,f16x2}
2211 Expand =
2212 Name == "f" || Name == "ftz.f" || Name == "d" || Name == "f16x2";
2213 else if (Name.consume_front("atomic.load."))
2214 // nvvm.atomic.load.add.{f32,f64}.p
2215 // nvvm.atomic.load.{inc,dec}.32.p
2216 Expand = StringSwitch<bool>(Name)
2217 .StartsWith("add.f32.p", true)
2218 .StartsWith("add.f64.p", true)
2219 .StartsWith("inc.32.p", true)
2220 .StartsWith("dec.32.p", true)
2221 .Default(false);
2222 else if (Name.consume_front("atomic."))
2223 // nvvm.atomic.{add,exch,max,min,inc,dec,and,or,xor}.gen.{i,f}.{cta,sys}
2224 // nvvm.atomic.cas.gen.i.{cta,sys}
2225 Expand = StringSwitch<bool>(Name)
2226 .StartsWith("add.gen.", true)
2227 .StartsWith("exch.gen.", true)
2228 .StartsWith("max.gen.", true)
2229 .StartsWith("min.gen.", true)
2230 .StartsWith("inc.gen.", true)
2231 .StartsWith("dec.gen.", true)
2232 .StartsWith("and.gen.", true)
2233 .StartsWith("or.gen.", true)
2234 .StartsWith("xor.gen.", true)
2235 .StartsWith("cas.gen.", true)
2236 .Default(false);
2237 else if (Name.consume_front("bitcast."))
2238 // nvvm.bitcast.{f2i,i2f,ll2d,d2ll}
2239 Expand =
2240 Name == "f2i" || Name == "i2f" || Name == "ll2d" || Name == "d2ll";
2241 else if (Name.consume_front("rotate."))
2242 // nvvm.rotate.{b32,b64,right.b64}
2243 Expand = Name == "b32" || Name == "b64" || Name == "right.b64";
2244 else if (Name.consume_front("ptr.gen.to."))
2245 // nvvm.ptr.gen.to.{local,shared,global,constant,param}
2246 Expand = consumeNVVMPtrAddrSpace(Name);
2247 else if (Name.consume_front("ptr."))
2248 // nvvm.ptr.{local,shared,global,constant,param}.to.gen
2249 Expand = consumeNVVMPtrAddrSpace(Name) && Name.starts_with(".to.gen");
2250 else if (Name.consume_front("ldg.global."))
2251 // nvvm.ldg.global.{i,p,f}
2252 Expand = (Name.starts_with("i.") || Name.starts_with("f.") ||
2253 Name.starts_with("p."));
2254 else
2255 Expand = StringSwitch<bool>(Name)
2256 .Case("barrier0", true)
2257 .Case("barrier.n", true)
2258 .Case("barrier.sync.cnt", true)
2259 .Case("barrier.sync", true)
2260 .Case("barrier", true)
2261 .Case("bar.sync", true)
2262 .Case("barrier0.popc", true)
2263 .Case("barrier0.and", true)
2264 .Case("barrier0.or", true)
2265 .Case("clz.ll", true)
2266 .Case("popc.ll", true)
2267 .Case("h2f", true)
2268 .Case("swap.lo.hi.b64", true)
2269 .Case("tanh.approx.f32", true)
2270 .Default(false);
2271
2272 if (Expand) {
2273 NewFn = nullptr;
2274 return true;
2275 }
2276 break; // No other 'nvvm.*'.
2277 }
2278 break;
2279 }
2280 case 'o':
2281 if (Name.starts_with("objectsize.")) {
2282 Type *Tys[2] = { F->getReturnType(), F->arg_begin()->getType() };
2283 if (F->arg_size() == 2 || F->arg_size() == 3) {
2284 rename(F);
2285 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
2286 Intrinsic::objectsize, Tys);
2287 return true;
2288 }
2289 }
2290 break;
2291
2292 case 'p':
2293 if (Name.starts_with("ptr.annotation.") && F->arg_size() == 4) {
2294 rename(F);
2296 F->getParent(), Intrinsic::ptr_annotation,
2297 {F->arg_begin()->getType(), F->getArg(1)->getType()});
2298 return true;
2299 }
2300 break;
2301
2302 case 'r': {
2303 if (Name.consume_front("riscv.")) {
2304 Intrinsic::ID ID;
2306 .Case("aes32dsi", Intrinsic::riscv_aes32dsi)
2307 .Case("aes32dsmi", Intrinsic::riscv_aes32dsmi)
2308 .Case("aes32esi", Intrinsic::riscv_aes32esi)
2309 .Case("aes32esmi", Intrinsic::riscv_aes32esmi)
2311 if (ID != Intrinsic::not_intrinsic) {
2312 if (!F->getFunctionType()->getParamType(2)->isIntegerTy(32)) {
2313 rename(F);
2314 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2315 return true;
2316 }
2317 break; // No other applicable upgrades.
2318 }
2319
2321 .StartsWith("sm4ks", Intrinsic::riscv_sm4ks)
2322 .StartsWith("sm4ed", Intrinsic::riscv_sm4ed)
2324 if (ID != Intrinsic::not_intrinsic) {
2325 if (!F->getFunctionType()->getParamType(2)->isIntegerTy(32) ||
2326 F->getFunctionType()->getReturnType()->isIntegerTy(64)) {
2327 rename(F);
2328 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2329 return true;
2330 }
2331 break; // No other applicable upgrades.
2332 }
2333
2335 .StartsWith("sha256sig0", Intrinsic::riscv_sha256sig0)
2336 .StartsWith("sha256sig1", Intrinsic::riscv_sha256sig1)
2337 .StartsWith("sha256sum0", Intrinsic::riscv_sha256sum0)
2338 .StartsWith("sha256sum1", Intrinsic::riscv_sha256sum1)
2339 .StartsWith("sm3p0", Intrinsic::riscv_sm3p0)
2340 .StartsWith("sm3p1", Intrinsic::riscv_sm3p1)
2342 if (ID != Intrinsic::not_intrinsic) {
2343 if (F->getFunctionType()->getReturnType()->isIntegerTy(64)) {
2344 rename(F);
2345 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2346 return true;
2347 }
2348 break; // No other applicable upgrades.
2349 }
2350
2351 // Replace llvm.riscv.clmul with llvm.clmul.
2352 if (Name == "clmul.i32" || Name == "clmul.i64") {
2354 F->getParent(), Intrinsic::clmul, {F->getReturnType()});
2355 return true;
2356 }
2357
2358 break; // No other 'riscv.*' intrinsics
2359 }
2360 } break;
2361
2362 case 's':
2363 if (Name == "stackprotectorcheck") {
2364 NewFn = nullptr;
2365 return true;
2366 }
2367 if (Name.starts_with("strip.invariant.group")) {
2368 // For clang's usage it would be safe to just drop the
2369 // strip.invariant.group, but to be conservative replace with the
2370 // stronger launder.invariant.group instead.
2372 F->getParent(), Intrinsic::launder_invariant_group,
2373 F->getReturnType());
2374 return true;
2375 }
2376 break;
2377
2378 case 't':
2379 if (Name == "thread.pointer") {
2381 F->getParent(), Intrinsic::thread_pointer, F->getReturnType());
2382 return true;
2383 }
2384 break;
2385
2386 case 'v': {
2387 if (Name == "var.annotation" && F->arg_size() == 4) {
2388 rename(F);
2390 F->getParent(), Intrinsic::var_annotation,
2391 {{F->arg_begin()->getType(), F->getArg(1)->getType()}});
2392 return true;
2393 }
2394 if (Name.consume_front("vector.splice")) {
2395 if (Name.starts_with(".left") || Name.starts_with(".right"))
2396 break;
2397 return true;
2398 }
2399 if (shouldUpgradeVPIntrinsic(Name))
2400 return true;
2401 break;
2402 }
2403
2404 case 'w':
2405 if (Name.consume_front("wasm.")) {
2406 Intrinsic::ID ID =
2408 .StartsWith("fma.", Intrinsic::wasm_relaxed_madd)
2409 .StartsWith("fms.", Intrinsic::wasm_relaxed_nmadd)
2410 .StartsWith("laneselect.", Intrinsic::wasm_relaxed_laneselect)
2412 if (ID != Intrinsic::not_intrinsic) {
2413 rename(F);
2414 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
2415 F->getReturnType());
2416 return true;
2417 }
2418
2419 if (Name.consume_front("dot.i8x16.i7x16.")) {
2421 .Case("signed", Intrinsic::wasm_relaxed_dot_i8x16_i7x16_signed)
2422 .Case("add.signed",
2423 Intrinsic::wasm_relaxed_dot_i8x16_i7x16_add_signed)
2425 if (ID != Intrinsic::not_intrinsic) {
2426 rename(F);
2427 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2428 return true;
2429 }
2430 break; // No other 'wasm.dot.i8x16.i7x16.*'.
2431 }
2432 break; // No other 'wasm.*'.
2433 }
2434 break;
2435
2436 case 'x':
2437 if (upgradeX86IntrinsicFunction(F, Name, NewFn))
2438 return true;
2439 }
2440
2441 auto *ST = dyn_cast<StructType>(F->getReturnType());
2442 if (ST && (!ST->isLiteral() || ST->isPacked()) &&
2443 F->getIntrinsicID() != Intrinsic::not_intrinsic) {
2444 // Replace return type with literal non-packed struct. Only do this for
2445 // intrinsics declared to return a struct, not for intrinsics with
2446 // overloaded return type, in which case the exact struct type will be
2447 // mangled into the name.
2448 if (Intrinsic::hasStructReturnType(F->getIntrinsicID())) {
2449 FunctionType *FT = F->getFunctionType();
2450 auto *NewST = StructType::get(ST->getContext(), ST->elements());
2451 auto *NewFT = FunctionType::get(NewST, FT->params(), FT->isVarArg());
2452 std::string Name = F->getName().str();
2453 rename(F);
2454 NewFn = Function::Create(NewFT, F->getLinkage(), F->getAddressSpace(),
2455 Name, F->getParent());
2456
2457 // The new function may also need remangling.
2458 if (auto Result = llvm::Intrinsic::remangleIntrinsicFunction(NewFn))
2459 NewFn = *Result;
2460 return true;
2461 }
2462 }
2463
2464 // Remangle our intrinsic since we upgrade the mangling
2466 if (Result != std::nullopt) {
2467 NewFn = *Result;
2468 return true;
2469 }
2470
2472 return true;
2473
2474 // This may not belong here. This function is effectively being overloaded
2475 // to both detect an intrinsic which needs upgrading, and to provide the
2476 // upgraded form of the intrinsic. We should perhaps have two separate
2477 // functions for this.
2478
2479 return false;
2480}
2481
2483 bool CanUpgradeDebugIntrinsicsToRecords) {
2484 NewFn = nullptr;
2485 bool Upgraded =
2486 upgradeIntrinsicFunction1(F, NewFn, CanUpgradeDebugIntrinsicsToRecords);
2487
2488 // Upgrade intrinsic attributes. This does not change the function.
2489 if (NewFn)
2490 F = NewFn;
2491 if (Intrinsic::ID id = F->getIntrinsicID()) {
2492 // Only do this if the intrinsic signature is valid.
2493 SmallVector<Type *> OverloadTys;
2494 if (Intrinsic::isSignatureValid(id, F->getFunctionType(), OverloadTys))
2495 F->setAttributes(
2496 Intrinsic::getAttributes(F->getContext(), id, F->getFunctionType()));
2497 }
2498 return Upgraded;
2499}
2500
2502 if (!(GV->hasName() && (GV->getName() == "llvm.global_ctors" ||
2503 GV->getName() == "llvm.global_dtors")) ||
2504 !GV->hasInitializer())
2505 return nullptr;
2507 if (!ATy)
2508 return nullptr;
2510 if (!STy || STy->getNumElements() != 2)
2511 return nullptr;
2512
2513 IRBuilder<> IRB(*GV->getParent());
2514 auto EltTy = StructType::get(STy->getElementType(0), STy->getElementType(1),
2515 IRB.getPtrTy());
2516 Constant *Init = GV->getInitializer();
2517 unsigned N = Init->getNumOperands();
2518 std::vector<Constant *> NewCtors(N);
2519 for (unsigned i = 0; i != N; ++i) {
2520 auto Ctor = cast<Constant>(Init->getOperand(i));
2521 NewCtors[i] = ConstantStruct::get(EltTy, Ctor->getAggregateElement(0u),
2522 Ctor->getAggregateElement(1),
2524 }
2525 Constant *NewInit = ConstantArray::get(ArrayType::get(EltTy, N), NewCtors);
2526
2527 return new GlobalVariable(NewInit->getType(), false, GV->getLinkage(),
2528 NewInit, GV->getName());
2529}
2530
2531// Handles upgrading SSE2/AVX2/AVX512BW PSLLDQ intrinsics by converting them
2532// to byte shuffles.
2534 unsigned Shift) {
2535 auto *ResultTy = cast<FixedVectorType>(Op->getType());
2536 unsigned NumElts = ResultTy->getNumElements() * 8;
2537
2538 // Bitcast from a 64-bit element type to a byte element type.
2539 Type *VecTy = FixedVectorType::get(Builder.getInt8Ty(), NumElts);
2540 Op = Builder.CreateBitCast(Op, VecTy, "cast");
2541
2542 // We'll be shuffling in zeroes.
2543 Value *Res = Constant::getNullValue(VecTy);
2544
2545 // If shift is less than 16, emit a shuffle to move the bytes. Otherwise,
2546 // we'll just return the zero vector.
2547 if (Shift < 16) {
2548 int Idxs[64];
2549 // 256/512-bit version is split into 2/4 16-byte lanes.
2550 for (unsigned l = 0; l != NumElts; l += 16)
2551 for (unsigned i = 0; i != 16; ++i) {
2552 unsigned Idx = NumElts + i - Shift;
2553 if (Idx < NumElts)
2554 Idx -= NumElts - 16; // end of lane, switch operand.
2555 Idxs[l + i] = Idx + l;
2556 }
2557
2558 Res = Builder.CreateShuffleVector(Res, Op, ArrayRef(Idxs, NumElts));
2559 }
2560
2561 // Bitcast back to a 64-bit element type.
2562 return Builder.CreateBitCast(Res, ResultTy, "cast");
2563}
2564
2565// Handles upgrading SSE2/AVX2/AVX512BW PSRLDQ intrinsics by converting them
2566// to byte shuffles.
2568 unsigned Shift) {
2569 auto *ResultTy = cast<FixedVectorType>(Op->getType());
2570 unsigned NumElts = ResultTy->getNumElements() * 8;
2571
2572 // Bitcast from a 64-bit element type to a byte element type.
2573 Type *VecTy = FixedVectorType::get(Builder.getInt8Ty(), NumElts);
2574 Op = Builder.CreateBitCast(Op, VecTy, "cast");
2575
2576 // We'll be shuffling in zeroes.
2577 Value *Res = Constant::getNullValue(VecTy);
2578
2579 // If shift is less than 16, emit a shuffle to move the bytes. Otherwise,
2580 // we'll just return the zero vector.
2581 if (Shift < 16) {
2582 int Idxs[64];
2583 // 256/512-bit version is split into 2/4 16-byte lanes.
2584 for (unsigned l = 0; l != NumElts; l += 16)
2585 for (unsigned i = 0; i != 16; ++i) {
2586 unsigned Idx = i + Shift;
2587 if (Idx >= 16)
2588 Idx += NumElts - 16; // end of lane, switch operand.
2589 Idxs[l + i] = Idx + l;
2590 }
2591
2592 Res = Builder.CreateShuffleVector(Op, Res, ArrayRef(Idxs, NumElts));
2593 }
2594
2595 // Bitcast back to a 64-bit element type.
2596 return Builder.CreateBitCast(Res, ResultTy, "cast");
2597}
2598
2599static Value *getX86MaskVec(IRBuilder<> &Builder, Value *Mask,
2600 unsigned NumElts) {
2601 assert(isPowerOf2_32(NumElts) && "Expected power-of-2 mask elements");
2603 Builder.getInt1Ty(), cast<IntegerType>(Mask->getType())->getBitWidth());
2604 Mask = Builder.CreateBitCast(Mask, MaskTy);
2605
2606 // If we have less than 8 elements (1, 2 or 4), then the starting mask was an
2607 // i8 and we need to extract down to the right number of elements.
2608 if (NumElts <= 4) {
2609 int Indices[4];
2610 for (unsigned i = 0; i != NumElts; ++i)
2611 Indices[i] = i;
2612 Mask = Builder.CreateShuffleVector(Mask, Mask, ArrayRef(Indices, NumElts),
2613 "extract");
2614 }
2615
2616 return Mask;
2617}
2618
2619static Value *emitX86Select(IRBuilder<> &Builder, Value *Mask, Value *Op0,
2620 Value *Op1) {
2621 // If the mask is all ones just emit the first operation.
2622 if (const auto *C = dyn_cast<Constant>(Mask))
2623 if (C->isAllOnesValue())
2624 return Op0;
2625
2626 Mask = getX86MaskVec(Builder, Mask,
2627 cast<FixedVectorType>(Op0->getType())->getNumElements());
2628 return Builder.CreateSelect(Mask, Op0, Op1);
2629}
2630
2631static Value *emitX86ScalarSelect(IRBuilder<> &Builder, Value *Mask, Value *Op0,
2632 Value *Op1) {
2633 // If the mask is all ones just emit the first operation.
2634 if (const auto *C = dyn_cast<Constant>(Mask))
2635 if (C->isAllOnesValue())
2636 return Op0;
2637
2638 auto *MaskTy = FixedVectorType::get(Builder.getInt1Ty(),
2639 Mask->getType()->getIntegerBitWidth());
2640 Mask = Builder.CreateBitCast(Mask, MaskTy);
2641 Mask = Builder.CreateExtractElement(Mask, (uint64_t)0);
2642 return Builder.CreateSelect(Mask, Op0, Op1);
2643}
2644
2645// Handle autoupgrade for masked PALIGNR and VALIGND/Q intrinsics.
2646// PALIGNR handles large immediates by shifting while VALIGN masks the immediate
2647// so we need to handle both cases. VALIGN also doesn't have 128-bit lanes.
2649 Value *Op1, Value *Shift,
2650 Value *Passthru, Value *Mask,
2651 bool IsVALIGN) {
2652 unsigned ShiftVal = cast<llvm::ConstantInt>(Shift)->getZExtValue();
2653
2654 unsigned NumElts = cast<FixedVectorType>(Op0->getType())->getNumElements();
2655 assert((IsVALIGN || NumElts % 16 == 0) && "Illegal NumElts for PALIGNR!");
2656 assert((!IsVALIGN || NumElts <= 16) && "NumElts too large for VALIGN!");
2657 assert(isPowerOf2_32(NumElts) && "NumElts not a power of 2!");
2658
2659 // Mask the immediate for VALIGN.
2660 if (IsVALIGN)
2661 ShiftVal &= (NumElts - 1);
2662
2663 // If palignr is shifting the pair of vectors more than the size of two
2664 // lanes, emit zero.
2665 if (ShiftVal >= 32)
2667
2668 // If palignr is shifting the pair of input vectors more than one lane,
2669 // but less than two lanes, convert to shifting in zeroes.
2670 if (ShiftVal > 16) {
2671 ShiftVal -= 16;
2672 Op1 = Op0;
2674 }
2675
2676 int Indices[64];
2677 // 256-bit palignr operates on 128-bit lanes so we need to handle that
2678 for (unsigned l = 0; l < NumElts; l += 16) {
2679 for (unsigned i = 0; i != 16; ++i) {
2680 unsigned Idx = ShiftVal + i;
2681 if (!IsVALIGN && Idx >= 16) // Disable wrap for VALIGN.
2682 Idx += NumElts - 16; // End of lane, switch operand.
2683 Indices[l + i] = Idx + l;
2684 }
2685 }
2686
2687 Value *Align = Builder.CreateShuffleVector(
2688 Op1, Op0, ArrayRef(Indices, NumElts), "palignr");
2689
2690 return emitX86Select(Builder, Mask, Align, Passthru);
2691}
2692
2694 bool ZeroMask, bool IndexForm) {
2695 Type *Ty = CI.getType();
2696 unsigned VecWidth = Ty->getPrimitiveSizeInBits();
2697 unsigned EltWidth = Ty->getScalarSizeInBits();
2698 bool IsFloat = Ty->isFPOrFPVectorTy();
2699 Intrinsic::ID IID;
2700 if (VecWidth == 128 && EltWidth == 32 && IsFloat)
2701 IID = Intrinsic::x86_avx512_vpermi2var_ps_128;
2702 else if (VecWidth == 128 && EltWidth == 32 && !IsFloat)
2703 IID = Intrinsic::x86_avx512_vpermi2var_d_128;
2704 else if (VecWidth == 128 && EltWidth == 64 && IsFloat)
2705 IID = Intrinsic::x86_avx512_vpermi2var_pd_128;
2706 else if (VecWidth == 128 && EltWidth == 64 && !IsFloat)
2707 IID = Intrinsic::x86_avx512_vpermi2var_q_128;
2708 else if (VecWidth == 256 && EltWidth == 32 && IsFloat)
2709 IID = Intrinsic::x86_avx512_vpermi2var_ps_256;
2710 else if (VecWidth == 256 && EltWidth == 32 && !IsFloat)
2711 IID = Intrinsic::x86_avx512_vpermi2var_d_256;
2712 else if (VecWidth == 256 && EltWidth == 64 && IsFloat)
2713 IID = Intrinsic::x86_avx512_vpermi2var_pd_256;
2714 else if (VecWidth == 256 && EltWidth == 64 && !IsFloat)
2715 IID = Intrinsic::x86_avx512_vpermi2var_q_256;
2716 else if (VecWidth == 512 && EltWidth == 32 && IsFloat)
2717 IID = Intrinsic::x86_avx512_vpermi2var_ps_512;
2718 else if (VecWidth == 512 && EltWidth == 32 && !IsFloat)
2719 IID = Intrinsic::x86_avx512_vpermi2var_d_512;
2720 else if (VecWidth == 512 && EltWidth == 64 && IsFloat)
2721 IID = Intrinsic::x86_avx512_vpermi2var_pd_512;
2722 else if (VecWidth == 512 && EltWidth == 64 && !IsFloat)
2723 IID = Intrinsic::x86_avx512_vpermi2var_q_512;
2724 else if (VecWidth == 128 && EltWidth == 16)
2725 IID = Intrinsic::x86_avx512_vpermi2var_hi_128;
2726 else if (VecWidth == 256 && EltWidth == 16)
2727 IID = Intrinsic::x86_avx512_vpermi2var_hi_256;
2728 else if (VecWidth == 512 && EltWidth == 16)
2729 IID = Intrinsic::x86_avx512_vpermi2var_hi_512;
2730 else if (VecWidth == 128 && EltWidth == 8)
2731 IID = Intrinsic::x86_avx512_vpermi2var_qi_128;
2732 else if (VecWidth == 256 && EltWidth == 8)
2733 IID = Intrinsic::x86_avx512_vpermi2var_qi_256;
2734 else if (VecWidth == 512 && EltWidth == 8)
2735 IID = Intrinsic::x86_avx512_vpermi2var_qi_512;
2736 else
2737 llvm_unreachable("Unexpected intrinsic");
2738
2739 Value *Args[] = { CI.getArgOperand(0) , CI.getArgOperand(1),
2740 CI.getArgOperand(2) };
2741
2742 // If this isn't index form we need to swap operand 0 and 1.
2743 if (!IndexForm)
2744 std::swap(Args[0], Args[1]);
2745
2746 Value *V = Builder.CreateIntrinsic(IID, Args);
2747 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty)
2748 : Builder.CreateBitCast(CI.getArgOperand(1),
2749 Ty);
2750 return emitX86Select(Builder, CI.getArgOperand(3), V, PassThru);
2751}
2752
2754 Intrinsic::ID IID) {
2755 Type *Ty = CI.getType();
2756 Value *Op0 = CI.getOperand(0);
2757 Value *Op1 = CI.getOperand(1);
2758 Value *Res = Builder.CreateIntrinsic(IID, Ty, {Op0, Op1});
2759
2760 if (CI.arg_size() == 4) { // For masked intrinsics.
2761 Value *VecSrc = CI.getOperand(2);
2762 Value *Mask = CI.getOperand(3);
2763 Res = emitX86Select(Builder, Mask, Res, VecSrc);
2764 }
2765 return Res;
2766}
2767
2769 bool IsRotateRight) {
2770 Type *Ty = CI.getType();
2771 Value *Src = CI.getArgOperand(0);
2772 Value *Amt = CI.getArgOperand(1);
2773
2774 // Amount may be scalar immediate, in which case create a splat vector.
2775 // Funnel shifts amounts are treated as modulo and types are all power-of-2 so
2776 // we only care about the lowest log2 bits anyway.
2777 if (Amt->getType() != Ty) {
2778 unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements();
2779 Amt = Builder.CreateIntCast(Amt, Ty->getScalarType(), false);
2780 Amt = Builder.CreateVectorSplat(NumElts, Amt);
2781 }
2782
2783 Intrinsic::ID IID = IsRotateRight ? Intrinsic::fshr : Intrinsic::fshl;
2784 Value *Res = Builder.CreateIntrinsic(IID, Ty, {Src, Src, Amt});
2785
2786 if (CI.arg_size() == 4) { // For masked intrinsics.
2787 Value *VecSrc = CI.getOperand(2);
2788 Value *Mask = CI.getOperand(3);
2789 Res = emitX86Select(Builder, Mask, Res, VecSrc);
2790 }
2791 return Res;
2792}
2793
2794static Value *upgradeX86vpcom(IRBuilder<> &Builder, CallBase &CI, unsigned Imm,
2795 bool IsSigned) {
2796 Type *Ty = CI.getType();
2797 Value *LHS = CI.getArgOperand(0);
2798 Value *RHS = CI.getArgOperand(1);
2799
2800 CmpInst::Predicate Pred;
2801 switch (Imm) {
2802 case 0x0:
2803 Pred = IsSigned ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT;
2804 break;
2805 case 0x1:
2806 Pred = IsSigned ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE;
2807 break;
2808 case 0x2:
2809 Pred = IsSigned ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT;
2810 break;
2811 case 0x3:
2812 Pred = IsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE;
2813 break;
2814 case 0x4:
2815 Pred = ICmpInst::ICMP_EQ;
2816 break;
2817 case 0x5:
2818 Pred = ICmpInst::ICMP_NE;
2819 break;
2820 case 0x6:
2821 return Constant::getNullValue(Ty); // FALSE
2822 case 0x7:
2823 return Constant::getAllOnesValue(Ty); // TRUE
2824 default:
2825 llvm_unreachable("Unknown XOP vpcom/vpcomu predicate");
2826 }
2827
2828 Value *Cmp = Builder.CreateICmp(Pred, LHS, RHS);
2829 Value *Ext = Builder.CreateSExt(Cmp, Ty);
2830 return Ext;
2831}
2832
2834 bool IsShiftRight, bool ZeroMask) {
2835 Type *Ty = CI.getType();
2836 Value *Op0 = CI.getArgOperand(0);
2837 Value *Op1 = CI.getArgOperand(1);
2838 Value *Amt = CI.getArgOperand(2);
2839
2840 if (IsShiftRight)
2841 std::swap(Op0, Op1);
2842
2843 // Amount may be scalar immediate, in which case create a splat vector.
2844 // Funnel shifts amounts are treated as modulo and types are all power-of-2 so
2845 // we only care about the lowest log2 bits anyway.
2846 if (Amt->getType() != Ty) {
2847 unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements();
2848 Amt = Builder.CreateIntCast(Amt, Ty->getScalarType(), false);
2849 Amt = Builder.CreateVectorSplat(NumElts, Amt);
2850 }
2851
2852 Intrinsic::ID IID = IsShiftRight ? Intrinsic::fshr : Intrinsic::fshl;
2853 Value *Res = Builder.CreateIntrinsic(IID, Ty, {Op0, Op1, Amt});
2854
2855 unsigned NumArgs = CI.arg_size();
2856 if (NumArgs >= 4) { // For masked intrinsics.
2857 Value *VecSrc = NumArgs == 5 ? CI.getArgOperand(3) :
2858 ZeroMask ? ConstantAggregateZero::get(CI.getType()) :
2859 CI.getArgOperand(0);
2860 Value *Mask = CI.getOperand(NumArgs - 1);
2861 Res = emitX86Select(Builder, Mask, Res, VecSrc);
2862 }
2863 return Res;
2864}
2865
2867 Value *Mask, bool Aligned) {
2868 const Align Alignment =
2869 Aligned
2870 ? Align(Data->getType()->getPrimitiveSizeInBits().getFixedValue() / 8)
2871 : Align(1);
2872
2873 // If the mask is all ones just emit a regular store.
2874 if (const auto *C = dyn_cast<Constant>(Mask))
2875 if (C->isAllOnesValue())
2876 return Builder.CreateAlignedStore(Data, Ptr, Alignment);
2877
2878 // Convert the mask from an integer type to a vector of i1.
2879 unsigned NumElts = cast<FixedVectorType>(Data->getType())->getNumElements();
2880 Mask = getX86MaskVec(Builder, Mask, NumElts);
2881 return Builder.CreateMaskedStore(Data, Ptr, Alignment, Mask);
2882}
2883
2885 Value *Passthru, Value *Mask, bool Aligned) {
2886 Type *ValTy = Passthru->getType();
2887 const Align Alignment =
2888 Aligned
2889 ? Align(
2891 8)
2892 : Align(1);
2893
2894 // If the mask is all ones just emit a regular store.
2895 if (const auto *C = dyn_cast<Constant>(Mask))
2896 if (C->isAllOnesValue())
2897 return Builder.CreateAlignedLoad(ValTy, Ptr, Alignment);
2898
2899 // Convert the mask from an integer type to a vector of i1.
2900 unsigned NumElts = cast<FixedVectorType>(ValTy)->getNumElements();
2901 Mask = getX86MaskVec(Builder, Mask, NumElts);
2902 return Builder.CreateMaskedLoad(ValTy, Ptr, Alignment, Mask, Passthru);
2903}
2904
2905static Value *upgradeAbs(IRBuilder<> &Builder, CallBase &CI) {
2906 Type *Ty = CI.getType();
2907 Value *Op0 = CI.getArgOperand(0);
2908 Value *Res = Builder.CreateIntrinsic(Intrinsic::abs, Ty,
2909 {Op0, Builder.getInt1(false)});
2910 if (CI.arg_size() == 3)
2911 Res = emitX86Select(Builder, CI.getArgOperand(2), Res, CI.getArgOperand(1));
2912 return Res;
2913}
2914
2915static Value *upgradePMULDQ(IRBuilder<> &Builder, CallBase &CI, bool IsSigned) {
2916 Type *Ty = CI.getType();
2917
2918 // Arguments have a vXi32 type so cast to vXi64.
2919 Value *LHS = Builder.CreateBitCast(CI.getArgOperand(0), Ty);
2920 Value *RHS = Builder.CreateBitCast(CI.getArgOperand(1), Ty);
2921
2922 if (IsSigned) {
2923 // Shift left then arithmetic shift right.
2924 Constant *ShiftAmt = ConstantInt::get(Ty, 32);
2925 LHS = Builder.CreateShl(LHS, ShiftAmt);
2926 LHS = Builder.CreateAShr(LHS, ShiftAmt);
2927 RHS = Builder.CreateShl(RHS, ShiftAmt);
2928 RHS = Builder.CreateAShr(RHS, ShiftAmt);
2929 } else {
2930 // Clear the upper bits.
2931 Constant *Mask = ConstantInt::get(Ty, 0xffffffff);
2932 LHS = Builder.CreateAnd(LHS, Mask);
2933 RHS = Builder.CreateAnd(RHS, Mask);
2934 }
2935
2936 Value *Res = Builder.CreateMul(LHS, RHS);
2937
2938 if (CI.arg_size() == 4)
2939 Res = emitX86Select(Builder, CI.getArgOperand(3), Res, CI.getArgOperand(2));
2940
2941 return Res;
2942}
2943
2944// Applying mask on vector of i1's and make sure result is at least 8 bits wide.
2946 Value *Mask) {
2947 unsigned NumElts = cast<FixedVectorType>(Vec->getType())->getNumElements();
2948 if (Mask) {
2949 const auto *C = dyn_cast<Constant>(Mask);
2950 if (!C || !C->isAllOnesValue())
2951 Vec = Builder.CreateAnd(Vec, getX86MaskVec(Builder, Mask, NumElts));
2952 }
2953
2954 if (NumElts < 8) {
2955 int Indices[8];
2956 for (unsigned i = 0; i != NumElts; ++i)
2957 Indices[i] = i;
2958 for (unsigned i = NumElts; i != 8; ++i)
2959 Indices[i] = NumElts + i % NumElts;
2960 Vec = Builder.CreateShuffleVector(Vec,
2962 Indices);
2963 }
2964 return Builder.CreateBitCast(Vec, Builder.getIntNTy(std::max(NumElts, 8U)));
2965}
2966
2968 unsigned CC, bool Signed) {
2969 Value *Op0 = CI.getArgOperand(0);
2970 unsigned NumElts = cast<FixedVectorType>(Op0->getType())->getNumElements();
2971
2972 Value *Cmp;
2973 if (CC == 3) {
2975 FixedVectorType::get(Builder.getInt1Ty(), NumElts));
2976 } else if (CC == 7) {
2978 FixedVectorType::get(Builder.getInt1Ty(), NumElts));
2979 } else {
2981 switch (CC) {
2982 default: llvm_unreachable("Unknown condition code");
2983 case 0: Pred = ICmpInst::ICMP_EQ; break;
2984 case 1: Pred = Signed ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT; break;
2985 case 2: Pred = Signed ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE; break;
2986 case 4: Pred = ICmpInst::ICMP_NE; break;
2987 case 5: Pred = Signed ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE; break;
2988 case 6: Pred = Signed ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT; break;
2989 }
2990 Cmp = Builder.CreateICmp(Pred, Op0, CI.getArgOperand(1));
2991 }
2992
2993 Value *Mask = CI.getArgOperand(CI.arg_size() - 1);
2994
2995 return applyX86MaskOn1BitsVec(Builder, Cmp, Mask);
2996}
2997
2998// Replace a masked intrinsic with an older unmasked intrinsic.
3000 Intrinsic::ID IID) {
3001 Value *Rep =
3002 Builder.CreateIntrinsic(IID, {CI.getArgOperand(0), CI.getArgOperand(1)});
3003 return emitX86Select(Builder, CI.getArgOperand(3), Rep, CI.getArgOperand(2));
3004}
3005
3007 Value* A = CI.getArgOperand(0);
3008 Value* B = CI.getArgOperand(1);
3009 Value* Src = CI.getArgOperand(2);
3010 Value* Mask = CI.getArgOperand(3);
3011
3012 Value* AndNode = Builder.CreateAnd(Mask, APInt(8, 1));
3013 Value* Cmp = Builder.CreateIsNotNull(AndNode);
3014 Value* Extract1 = Builder.CreateExtractElement(B, (uint64_t)0);
3015 Value* Extract2 = Builder.CreateExtractElement(Src, (uint64_t)0);
3016 Value* Select = Builder.CreateSelect(Cmp, Extract1, Extract2);
3017 return Builder.CreateInsertElement(A, Select, (uint64_t)0);
3018}
3019
3021 Value* Op = CI.getArgOperand(0);
3022 Type* ReturnOp = CI.getType();
3023 unsigned NumElts = cast<FixedVectorType>(CI.getType())->getNumElements();
3024 Value *Mask = getX86MaskVec(Builder, Op, NumElts);
3025 return Builder.CreateSExt(Mask, ReturnOp, "vpmovm2");
3026}
3027
3028// Replace intrinsic with unmasked version and a select.
3030 CallBase &CI, Value *&Rep) {
3031 Name = Name.substr(12); // Remove avx512.mask.
3032
3033 unsigned VecWidth = CI.getType()->getPrimitiveSizeInBits();
3034 unsigned EltWidth = CI.getType()->getScalarSizeInBits();
3035 Intrinsic::ID IID;
3036 if (Name.starts_with("max.p")) {
3037 if (VecWidth == 128 && EltWidth == 32)
3038 IID = Intrinsic::x86_sse_max_ps;
3039 else if (VecWidth == 128 && EltWidth == 64)
3040 IID = Intrinsic::x86_sse2_max_pd;
3041 else if (VecWidth == 256 && EltWidth == 32)
3042 IID = Intrinsic::x86_avx_max_ps_256;
3043 else if (VecWidth == 256 && EltWidth == 64)
3044 IID = Intrinsic::x86_avx_max_pd_256;
3045 else
3046 llvm_unreachable("Unexpected intrinsic");
3047 } else if (Name.starts_with("min.p")) {
3048 if (VecWidth == 128 && EltWidth == 32)
3049 IID = Intrinsic::x86_sse_min_ps;
3050 else if (VecWidth == 128 && EltWidth == 64)
3051 IID = Intrinsic::x86_sse2_min_pd;
3052 else if (VecWidth == 256 && EltWidth == 32)
3053 IID = Intrinsic::x86_avx_min_ps_256;
3054 else if (VecWidth == 256 && EltWidth == 64)
3055 IID = Intrinsic::x86_avx_min_pd_256;
3056 else
3057 llvm_unreachable("Unexpected intrinsic");
3058 } else if (Name.starts_with("pshuf.b.")) {
3059 if (VecWidth == 128)
3060 IID = Intrinsic::x86_ssse3_pshuf_b_128;
3061 else if (VecWidth == 256)
3062 IID = Intrinsic::x86_avx2_pshuf_b;
3063 else if (VecWidth == 512)
3064 IID = Intrinsic::x86_avx512_pshuf_b_512;
3065 else
3066 llvm_unreachable("Unexpected intrinsic");
3067 } else if (Name.starts_with("pmul.hr.sw.")) {
3068 if (VecWidth == 128)
3069 IID = Intrinsic::x86_ssse3_pmul_hr_sw_128;
3070 else if (VecWidth == 256)
3071 IID = Intrinsic::x86_avx2_pmul_hr_sw;
3072 else if (VecWidth == 512)
3073 IID = Intrinsic::x86_avx512_pmul_hr_sw_512;
3074 else
3075 llvm_unreachable("Unexpected intrinsic");
3076 } else if (Name.starts_with("pmulh.w")) {
3077 assert((VecWidth == 128 || VecWidth == 256 || VecWidth == 512) &&
3078 "Unexpected intrinsic");
3079 Rep = upgradeX86BinaryIntrinsics(Builder, CI, Intrinsic::smulh);
3080 return true;
3081 } else if (Name.starts_with("pmulhu.w")) {
3082 assert((VecWidth == 128 || VecWidth == 256 || VecWidth == 512) &&
3083 "Unexpected intrinsic");
3084 Rep = upgradeX86BinaryIntrinsics(Builder, CI, Intrinsic::umulh);
3085 return true;
3086 } else if (Name.starts_with("pmaddw.d.")) {
3087 if (VecWidth == 128)
3088 IID = Intrinsic::x86_sse2_pmadd_wd;
3089 else if (VecWidth == 256)
3090 IID = Intrinsic::x86_avx2_pmadd_wd;
3091 else if (VecWidth == 512)
3092 IID = Intrinsic::x86_avx512_pmaddw_d_512;
3093 else
3094 llvm_unreachable("Unexpected intrinsic");
3095 } else if (Name.starts_with("pmaddubs.w.")) {
3096 if (VecWidth == 128)
3097 IID = Intrinsic::x86_ssse3_pmadd_ub_sw_128;
3098 else if (VecWidth == 256)
3099 IID = Intrinsic::x86_avx2_pmadd_ub_sw;
3100 else if (VecWidth == 512)
3101 IID = Intrinsic::x86_avx512_pmaddubs_w_512;
3102 else
3103 llvm_unreachable("Unexpected intrinsic");
3104 } else if (Name.starts_with("packsswb.")) {
3105 if (VecWidth == 128)
3106 IID = Intrinsic::x86_sse2_packsswb_128;
3107 else if (VecWidth == 256)
3108 IID = Intrinsic::x86_avx2_packsswb;
3109 else if (VecWidth == 512)
3110 IID = Intrinsic::x86_avx512_packsswb_512;
3111 else
3112 llvm_unreachable("Unexpected intrinsic");
3113 } else if (Name.starts_with("packssdw.")) {
3114 if (VecWidth == 128)
3115 IID = Intrinsic::x86_sse2_packssdw_128;
3116 else if (VecWidth == 256)
3117 IID = Intrinsic::x86_avx2_packssdw;
3118 else if (VecWidth == 512)
3119 IID = Intrinsic::x86_avx512_packssdw_512;
3120 else
3121 llvm_unreachable("Unexpected intrinsic");
3122 } else if (Name.starts_with("packuswb.")) {
3123 if (VecWidth == 128)
3124 IID = Intrinsic::x86_sse2_packuswb_128;
3125 else if (VecWidth == 256)
3126 IID = Intrinsic::x86_avx2_packuswb;
3127 else if (VecWidth == 512)
3128 IID = Intrinsic::x86_avx512_packuswb_512;
3129 else
3130 llvm_unreachable("Unexpected intrinsic");
3131 } else if (Name.starts_with("packusdw.")) {
3132 if (VecWidth == 128)
3133 IID = Intrinsic::x86_sse41_packusdw;
3134 else if (VecWidth == 256)
3135 IID = Intrinsic::x86_avx2_packusdw;
3136 else if (VecWidth == 512)
3137 IID = Intrinsic::x86_avx512_packusdw_512;
3138 else
3139 llvm_unreachable("Unexpected intrinsic");
3140 } else if (Name.starts_with("vpermilvar.")) {
3141 if (VecWidth == 128 && EltWidth == 32)
3142 IID = Intrinsic::x86_avx_vpermilvar_ps;
3143 else if (VecWidth == 128 && EltWidth == 64)
3144 IID = Intrinsic::x86_avx_vpermilvar_pd;
3145 else if (VecWidth == 256 && EltWidth == 32)
3146 IID = Intrinsic::x86_avx_vpermilvar_ps_256;
3147 else if (VecWidth == 256 && EltWidth == 64)
3148 IID = Intrinsic::x86_avx_vpermilvar_pd_256;
3149 else if (VecWidth == 512 && EltWidth == 32)
3150 IID = Intrinsic::x86_avx512_vpermilvar_ps_512;
3151 else if (VecWidth == 512 && EltWidth == 64)
3152 IID = Intrinsic::x86_avx512_vpermilvar_pd_512;
3153 else
3154 llvm_unreachable("Unexpected intrinsic");
3155 } else if (Name == "cvtpd2dq.256") {
3156 IID = Intrinsic::x86_avx_cvt_pd2dq_256;
3157 } else if (Name == "cvtpd2ps.256") {
3158 IID = Intrinsic::x86_avx_cvt_pd2_ps_256;
3159 } else if (Name == "cvttpd2dq.256") {
3160 IID = Intrinsic::x86_avx_cvtt_pd2dq_256;
3161 } else if (Name == "cvttps2dq.128") {
3162 IID = Intrinsic::x86_sse2_cvttps2dq;
3163 } else if (Name == "cvttps2dq.256") {
3164 IID = Intrinsic::x86_avx_cvtt_ps2dq_256;
3165 } else if (Name.starts_with("permvar.")) {
3166 bool IsFloat = CI.getType()->isFPOrFPVectorTy();
3167 if (VecWidth == 256 && EltWidth == 32 && IsFloat)
3168 IID = Intrinsic::x86_avx2_permps;
3169 else if (VecWidth == 256 && EltWidth == 32 && !IsFloat)
3170 IID = Intrinsic::x86_avx2_permd;
3171 else if (VecWidth == 256 && EltWidth == 64 && IsFloat)
3172 IID = Intrinsic::x86_avx512_permvar_df_256;
3173 else if (VecWidth == 256 && EltWidth == 64 && !IsFloat)
3174 IID = Intrinsic::x86_avx512_permvar_di_256;
3175 else if (VecWidth == 512 && EltWidth == 32 && IsFloat)
3176 IID = Intrinsic::x86_avx512_permvar_sf_512;
3177 else if (VecWidth == 512 && EltWidth == 32 && !IsFloat)
3178 IID = Intrinsic::x86_avx512_permvar_si_512;
3179 else if (VecWidth == 512 && EltWidth == 64 && IsFloat)
3180 IID = Intrinsic::x86_avx512_permvar_df_512;
3181 else if (VecWidth == 512 && EltWidth == 64 && !IsFloat)
3182 IID = Intrinsic::x86_avx512_permvar_di_512;
3183 else if (VecWidth == 128 && EltWidth == 16)
3184 IID = Intrinsic::x86_avx512_permvar_hi_128;
3185 else if (VecWidth == 256 && EltWidth == 16)
3186 IID = Intrinsic::x86_avx512_permvar_hi_256;
3187 else if (VecWidth == 512 && EltWidth == 16)
3188 IID = Intrinsic::x86_avx512_permvar_hi_512;
3189 else if (VecWidth == 128 && EltWidth == 8)
3190 IID = Intrinsic::x86_avx512_permvar_qi_128;
3191 else if (VecWidth == 256 && EltWidth == 8)
3192 IID = Intrinsic::x86_avx512_permvar_qi_256;
3193 else if (VecWidth == 512 && EltWidth == 8)
3194 IID = Intrinsic::x86_avx512_permvar_qi_512;
3195 else
3196 llvm_unreachable("Unexpected intrinsic");
3197 } else if (Name.starts_with("dbpsadbw.")) {
3198 if (VecWidth == 128)
3199 IID = Intrinsic::x86_avx512_dbpsadbw_128;
3200 else if (VecWidth == 256)
3201 IID = Intrinsic::x86_avx512_dbpsadbw_256;
3202 else if (VecWidth == 512)
3203 IID = Intrinsic::x86_avx512_dbpsadbw_512;
3204 else
3205 llvm_unreachable("Unexpected intrinsic");
3206 } else if (Name.starts_with("pmultishift.qb.")) {
3207 if (VecWidth == 128)
3208 IID = Intrinsic::x86_avx512_pmultishift_qb_128;
3209 else if (VecWidth == 256)
3210 IID = Intrinsic::x86_avx512_pmultishift_qb_256;
3211 else if (VecWidth == 512)
3212 IID = Intrinsic::x86_avx512_pmultishift_qb_512;
3213 else
3214 llvm_unreachable("Unexpected intrinsic");
3215 } else if (Name.starts_with("conflict.")) {
3216 if (Name[9] == 'd' && VecWidth == 128)
3217 IID = Intrinsic::x86_avx512_conflict_d_128;
3218 else if (Name[9] == 'd' && VecWidth == 256)
3219 IID = Intrinsic::x86_avx512_conflict_d_256;
3220 else if (Name[9] == 'd' && VecWidth == 512)
3221 IID = Intrinsic::x86_avx512_conflict_d_512;
3222 else if (Name[9] == 'q' && VecWidth == 128)
3223 IID = Intrinsic::x86_avx512_conflict_q_128;
3224 else if (Name[9] == 'q' && VecWidth == 256)
3225 IID = Intrinsic::x86_avx512_conflict_q_256;
3226 else if (Name[9] == 'q' && VecWidth == 512)
3227 IID = Intrinsic::x86_avx512_conflict_q_512;
3228 else
3229 llvm_unreachable("Unexpected intrinsic");
3230 } else if (Name.starts_with("pavg.")) {
3231 if (Name[5] == 'b' && VecWidth == 128)
3232 IID = Intrinsic::x86_sse2_pavg_b;
3233 else if (Name[5] == 'b' && VecWidth == 256)
3234 IID = Intrinsic::x86_avx2_pavg_b;
3235 else if (Name[5] == 'b' && VecWidth == 512)
3236 IID = Intrinsic::x86_avx512_pavg_b_512;
3237 else if (Name[5] == 'w' && VecWidth == 128)
3238 IID = Intrinsic::x86_sse2_pavg_w;
3239 else if (Name[5] == 'w' && VecWidth == 256)
3240 IID = Intrinsic::x86_avx2_pavg_w;
3241 else if (Name[5] == 'w' && VecWidth == 512)
3242 IID = Intrinsic::x86_avx512_pavg_w_512;
3243 else
3244 llvm_unreachable("Unexpected intrinsic");
3245 } else
3246 return false;
3247
3248 SmallVector<Value *, 4> Args(CI.args());
3249 Args.pop_back();
3250 Args.pop_back();
3251 Rep = Builder.CreateIntrinsic(IID, Args);
3252 unsigned NumArgs = CI.arg_size();
3253 Rep = emitX86Select(Builder, CI.getArgOperand(NumArgs - 1), Rep,
3254 CI.getArgOperand(NumArgs - 2));
3255 return true;
3256}
3257
3258/// Upgrade comment in call to inline asm that represents an objc retain release
3259/// marker.
3260void llvm::UpgradeInlineAsmString(std::string *AsmStr) {
3261 size_t Pos;
3262 if (AsmStr->find("mov\tfp") == 0 &&
3263 AsmStr->find("objc_retainAutoreleaseReturnValue") != std::string::npos &&
3264 (Pos = AsmStr->find("# marker")) != std::string::npos) {
3265 AsmStr->replace(Pos, 1, ";");
3266 }
3267}
3268
3270 StringRef Name,
3271 const Intrinsic::ID (&IIDs)[2][2]) {
3272 auto Result = getNVVMFPArithUpgrade(Name, IIDs);
3273 assert(Result && "unsupported nvvm.add.*/nvvm.mul.* intrinsic");
3274 auto [IID, RoundingMode] = *Result;
3275 Value *A = CI->getArgOperand(0);
3276 return Builder.CreateIntrinsic(
3277 A->getType(), IID,
3278 {A, CI->getArgOperand(1),
3279 Builder.getInt32(static_cast<int>(RoundingMode))});
3280}
3281
3283 Function *F, IRBuilder<> &Builder) {
3284 Value *Rep = nullptr;
3285
3286 if (Name == "abs.i" || Name == "abs.ll") {
3287 Value *Arg = CI->getArgOperand(0);
3288 Rep = Builder.CreateIntrinsic(Intrinsic::abs, {Arg->getType()},
3289 {Arg, Builder.getTrue()},
3290 /*FMFSource=*/nullptr, "abs");
3291 } else if (Name == "abs.bf16" || Name == "abs.bf16x2") {
3292 Type *Ty = (Name == "abs.bf16")
3293 ? Builder.getBFloatTy()
3294 : FixedVectorType::get(Builder.getBFloatTy(), 2);
3295 Value *Arg = Builder.CreateBitCast(CI->getArgOperand(0), Ty);
3296 Value *Abs = Builder.CreateUnaryIntrinsic(Intrinsic::nvvm_fabs, Arg);
3297 Rep = Builder.CreateBitCast(Abs, CI->getType());
3298 } else if (Name == "fabs.f" || Name == "fabs.ftz.f" || Name == "fabs.d") {
3299 Intrinsic::ID IID = (Name == "fabs.ftz.f") ? Intrinsic::nvvm_fabs_ftz
3300 : Intrinsic::nvvm_fabs;
3301 Rep = Builder.CreateUnaryIntrinsic(IID, CI->getArgOperand(0));
3302 } else if (Name.consume_front("add.")) {
3303 // nvvm.add.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
3304 Rep = upgradeNVVMFPArithCall(Builder, CI, Name, NVVMFAddIIDs);
3305 } else if (Name.consume_front("mul.")) {
3306 // nvvm.mul.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
3307 Rep = upgradeNVVMFPArithCall(Builder, CI, Name, NVVMFMulIIDs);
3308 } else if (Name.consume_front("ex2.approx.")) {
3309 // nvvm.ex2.approx.{f,ftz.f,d,f16x2}
3310 Intrinsic::ID IID = Name.starts_with("ftz") ? Intrinsic::nvvm_ex2_approx_ftz
3311 : Intrinsic::nvvm_ex2_approx;
3312 Rep = Builder.CreateUnaryIntrinsic(IID, CI->getArgOperand(0));
3313 } else if (Name.starts_with("atomic.load.add.f32.p") ||
3314 Name.starts_with("atomic.load.add.f64.p")) {
3315 Value *Ptr = CI->getArgOperand(0);
3316 Value *Val = CI->getArgOperand(1);
3317 Rep = Builder.CreateAtomicRMW(
3319 CI->getContext().getOrInsertSyncScopeID("device"));
3320 // The default scope for atomic.load.* intrinsics is device
3321 // (= gpu scope in ptx), but the default LLVM atomic scope is
3322 // "system"
3323 } else if (Name.starts_with("atomic.load.inc.32.p") ||
3324 Name.starts_with("atomic.load.dec.32.p")) {
3325 Value *Ptr = CI->getArgOperand(0);
3326 Value *Val = CI->getArgOperand(1);
3327 auto Op = Name.starts_with("atomic.load.inc") ? AtomicRMWInst::UIncWrap
3329 Rep = Builder.CreateAtomicRMW(
3331 CI->getContext().getOrInsertSyncScopeID("device"));
3332 // See comment above.
3333 } else if (Name.starts_with("atomic.") && Name.contains(".gen.")) {
3334 // nvvm.atomic.{op}.gen.{i,f}.{cta,sys} -> atomicrmw / cmpxchg.
3335 StringRef Op = Name.substr(StringRef("atomic.").size());
3336 Value *Ptr = CI->getArgOperand(0);
3337 Value *Val = CI->getArgOperand(1);
3339 Op.contains(".cta.") ? "block" : "");
3340 if (Op.starts_with("cas.")) {
3341 Value *New = CI->getArgOperand(2);
3342 Value *Pair = Builder.CreateAtomicCmpXchg(
3343 Ptr, Val, New, MaybeAlign(), AtomicOrdering::Monotonic,
3345 Rep = Builder.CreateExtractValue(Pair, 0);
3346 } else {
3347 // Note we don't upgrade anything to AtomicRMWInst::UMin/UMax. This is
3348 // because we were actually missing those intrinsics!
3349 AtomicRMWInst::BinOp BinOp =
3351 .StartsWith("add.gen.f", AtomicRMWInst::FAdd)
3352 .StartsWith("add.gen.i", AtomicRMWInst::Add)
3363 "unexpected nvvm scoped atomic intrinsic");
3364 Rep = Builder.CreateAtomicRMW(BinOp, Ptr, Val, MaybeAlign(),
3366 }
3367 } else if (Name == "clz.ll") {
3368 // llvm.nvvm.clz.ll returns an i32, but llvm.ctlz.i64 returns an i64.
3369 Value *Arg = CI->getArgOperand(0);
3370 Value *Ctlz = Builder.CreateIntrinsic(Intrinsic::ctlz, {Arg->getType()},
3371 {Arg, Builder.getFalse()},
3372 /*FMFSource=*/nullptr, "ctlz");
3373 Rep = Builder.CreateTrunc(Ctlz, Builder.getInt32Ty(), "ctlz.trunc");
3374 } else if (Name == "popc.ll") {
3375 // llvm.nvvm.popc.ll returns an i32, but llvm.ctpop.i64 returns an
3376 // i64.
3377 Value *Arg = CI->getArgOperand(0);
3378 Value *Popc = Builder.CreateIntrinsic(Intrinsic::ctpop, {Arg->getType()},
3379 Arg, /*FMFSource=*/nullptr, "ctpop");
3380 Rep = Builder.CreateTrunc(Popc, Builder.getInt32Ty(), "ctpop.trunc");
3381 } else if (Name == "h2f") {
3382 Value *Cast =
3383 Builder.CreateBitCast(CI->getArgOperand(0), Builder.getHalfTy());
3384 Rep = Builder.CreateFPExt(Cast, Builder.getFloatTy());
3385 } else if (Name.consume_front("bitcast.") &&
3386 (Name == "f2i" || Name == "i2f" || Name == "ll2d" ||
3387 Name == "d2ll")) {
3388 Rep = Builder.CreateBitCast(CI->getArgOperand(0), CI->getType());
3389 } else if (Name == "rotate.b32") {
3390 Value *Arg = CI->getOperand(0);
3391 Value *ShiftAmt = CI->getOperand(1);
3392 Rep = Builder.CreateIntrinsic(Builder.getInt32Ty(), Intrinsic::fshl,
3393 {Arg, Arg, ShiftAmt});
3394 } else if (Name == "rotate.b64") {
3395 Type *Int64Ty = Builder.getInt64Ty();
3396 Value *Arg = CI->getOperand(0);
3397 Value *ZExtShiftAmt = Builder.CreateZExt(CI->getOperand(1), Int64Ty);
3398 Rep = Builder.CreateIntrinsic(Int64Ty, Intrinsic::fshl,
3399 {Arg, Arg, ZExtShiftAmt});
3400 } else if (Name == "rotate.right.b64") {
3401 Type *Int64Ty = Builder.getInt64Ty();
3402 Value *Arg = CI->getOperand(0);
3403 Value *ZExtShiftAmt = Builder.CreateZExt(CI->getOperand(1), Int64Ty);
3404 Rep = Builder.CreateIntrinsic(Int64Ty, Intrinsic::fshr,
3405 {Arg, Arg, ZExtShiftAmt});
3406 } else if (Name == "swap.lo.hi.b64") {
3407 Type *Int64Ty = Builder.getInt64Ty();
3408 Value *Arg = CI->getOperand(0);
3409 Rep = Builder.CreateIntrinsic(Int64Ty, Intrinsic::fshl,
3410 {Arg, Arg, Builder.getInt64(32)});
3411 } else if ((Name.consume_front("ptr.gen.to.") &&
3412 consumeNVVMPtrAddrSpace(Name)) ||
3413 (Name.consume_front("ptr.") && consumeNVVMPtrAddrSpace(Name) &&
3414 Name.starts_with(".to.gen"))) {
3415 Rep = Builder.CreateAddrSpaceCast(CI->getArgOperand(0), CI->getType());
3416 } else if (Name.consume_front("ldg.global")) {
3417 Value *Ptr = CI->getArgOperand(0);
3418 Align PtrAlign = cast<ConstantInt>(CI->getArgOperand(1))->getAlignValue();
3419 // Use addrspace(1) for NVPTX ADDRESS_SPACE_GLOBAL
3420 Value *ASC = Builder.CreateAddrSpaceCast(Ptr, Builder.getPtrTy(1));
3421 Instruction *LD = Builder.CreateAlignedLoad(CI->getType(), ASC, PtrAlign);
3422 MDNode *MD = MDNode::get(Builder.getContext(), {});
3423 LD->setMetadata(LLVMContext::MD_invariant_load, MD);
3424 return LD;
3425 } else if (Name == "tanh.approx.f32") {
3426 // nvvm.tanh.approx.f32 -> afn llvm.tanh.f32
3427 FastMathFlags FMF;
3428 FMF.setApproxFunc();
3429 Rep = Builder.CreateUnaryIntrinsic(Intrinsic::tanh, CI->getArgOperand(0),
3430 FMF);
3431 } else if (Name == "barrier0" || Name == "barrier.n" || Name == "bar.sync") {
3432 Value *Arg =
3433 Name.ends_with('0') ? Builder.getInt32(0) : CI->getArgOperand(0);
3434 Rep = Builder.CreateIntrinsic(Intrinsic::nvvm_barrier_cta_sync_aligned_all,
3435 {}, {Arg});
3436 } else if (Name == "barrier") {
3437 Rep = Builder.CreateIntrinsic(
3438 Intrinsic::nvvm_barrier_cta_sync_aligned_count, {},
3439 {CI->getArgOperand(0), CI->getArgOperand(1)});
3440 } else if (Name == "barrier.sync") {
3441 Rep = Builder.CreateIntrinsic(Intrinsic::nvvm_barrier_cta_sync_all, {},
3442 {CI->getArgOperand(0)});
3443 } else if (Name == "barrier.sync.cnt") {
3444 Rep = Builder.CreateIntrinsic(Intrinsic::nvvm_barrier_cta_sync_count, {},
3445 {CI->getArgOperand(0), CI->getArgOperand(1)});
3446 } else if (Name == "barrier0.popc" || Name == "barrier0.and" ||
3447 Name == "barrier0.or") {
3448 Value *C = CI->getArgOperand(0);
3449 C = Builder.CreateICmpNE(C, Builder.getInt32(0));
3450
3451 Intrinsic::ID IID =
3453 .Case("barrier0.popc",
3454 Intrinsic::nvvm_barrier_cta_red_popc_aligned_all)
3455 .Case("barrier0.and",
3456 Intrinsic::nvvm_barrier_cta_red_and_aligned_all)
3457 .Case("barrier0.or",
3458 Intrinsic::nvvm_barrier_cta_red_or_aligned_all);
3459 Value *Bar = Builder.CreateIntrinsic(IID, {}, {Builder.getInt32(0), C});
3460 Rep = Builder.CreateZExt(Bar, CI->getType());
3461 } else {
3463 if (IID != Intrinsic::not_intrinsic &&
3465 rename(F);
3466 Function *NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
3468 for (size_t I = 0; I < NewFn->arg_size(); ++I) {
3469 Value *Arg = CI->getArgOperand(I);
3470 Type *OldType = Arg->getType();
3471 Type *NewType = NewFn->getArg(I)->getType();
3472 Args.push_back(
3473 (OldType->isIntegerTy() && NewType->getScalarType()->isBFloatTy())
3474 ? Builder.CreateBitCast(Arg, NewType)
3475 : Arg);
3476 }
3477 Rep = Builder.CreateCall(NewFn, Args);
3478 if (F->getReturnType()->isIntegerTy())
3479 Rep = Builder.CreateBitCast(Rep, F->getReturnType());
3480 }
3481 }
3482
3483 return Rep;
3484}
3485
3487 IRBuilder<> &Builder) {
3488 LLVMContext &C = F->getContext();
3489 Value *Rep = nullptr;
3490
3491 if (Name.starts_with("sse4a.movnt.")) {
3493 Elts.push_back(
3494 ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1)));
3495 MDNode *Node = MDNode::get(C, Elts);
3496
3497 Value *Arg0 = CI->getArgOperand(0);
3498 Value *Arg1 = CI->getArgOperand(1);
3499
3500 // Nontemporal (unaligned) store of the 0'th element of the float/double
3501 // vector.
3502 Value *Extract =
3503 Builder.CreateExtractElement(Arg1, (uint64_t)0, "extractelement");
3504
3505 StoreInst *SI = Builder.CreateAlignedStore(Extract, Arg0, Align(1));
3506 SI->setMetadata(LLVMContext::MD_nontemporal, Node);
3507 } else if (Name.starts_with("avx.movnt.") ||
3508 Name.starts_with("avx512.storent.")) {
3510 Elts.push_back(
3511 ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1)));
3512 MDNode *Node = MDNode::get(C, Elts);
3513
3514 Value *Arg0 = CI->getArgOperand(0);
3515 Value *Arg1 = CI->getArgOperand(1);
3516
3517 StoreInst *SI = Builder.CreateAlignedStore(
3518 Arg1, Arg0,
3520 SI->setMetadata(LLVMContext::MD_nontemporal, Node);
3521 } else if (Name == "sse2.storel.dq") {
3522 Value *Arg0 = CI->getArgOperand(0);
3523 Value *Arg1 = CI->getArgOperand(1);
3524
3525 auto *NewVecTy = FixedVectorType::get(Type::getInt64Ty(C), 2);
3526 Value *BC0 = Builder.CreateBitCast(Arg1, NewVecTy, "cast");
3527 Value *Elt = Builder.CreateExtractElement(BC0, (uint64_t)0);
3528 Builder.CreateAlignedStore(Elt, Arg0, Align(1));
3529 } else if (Name.starts_with("sse.storeu.") ||
3530 Name.starts_with("sse2.storeu.") ||
3531 Name.starts_with("avx.storeu.")) {
3532 Value *Arg0 = CI->getArgOperand(0);
3533 Value *Arg1 = CI->getArgOperand(1);
3534 Builder.CreateAlignedStore(Arg1, Arg0, Align(1));
3535 } else if (Name == "avx512.mask.store.ss") {
3536 Value *Mask = Builder.CreateAnd(CI->getArgOperand(2), Builder.getInt8(1));
3537 upgradeMaskedStore(Builder, CI->getArgOperand(0), CI->getArgOperand(1),
3538 Mask, false);
3539 } else if (Name.starts_with("avx512.mask.store")) {
3540 // "avx512.mask.storeu." or "avx512.mask.store."
3541 bool Aligned = Name[17] != 'u'; // "avx512.mask.storeu".
3542 upgradeMaskedStore(Builder, CI->getArgOperand(0), CI->getArgOperand(1),
3543 CI->getArgOperand(2), Aligned);
3544 } else if (Name.starts_with("sse2.pcmp") || Name.starts_with("avx2.pcmp")) {
3545 // Upgrade packed integer vector compare intrinsics to compare instructions.
3546 // "sse2.pcpmpeq." "sse2.pcmpgt." "avx2.pcmpeq." or "avx2.pcmpgt."
3547 bool CmpEq = Name[9] == 'e';
3548 Rep = Builder.CreateICmp(CmpEq ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_SGT,
3549 CI->getArgOperand(0), CI->getArgOperand(1));
3550 Rep = Builder.CreateSExt(Rep, CI->getType(), "");
3551 } else if (Name.starts_with("avx512.broadcastm")) {
3552 Type *ExtTy = Type::getInt32Ty(C);
3553 if (CI->getOperand(0)->getType()->isIntegerTy(8))
3554 ExtTy = Type::getInt64Ty(C);
3555 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() /
3556 ExtTy->getPrimitiveSizeInBits();
3557 Rep = Builder.CreateZExt(CI->getArgOperand(0), ExtTy);
3558 Rep = Builder.CreateVectorSplat(NumElts, Rep);
3559 } else if (Name == "sse.sqrt.ss" || Name == "sse2.sqrt.sd") {
3560 Value *Vec = CI->getArgOperand(0);
3561 Value *Elt0 = Builder.CreateExtractElement(Vec, (uint64_t)0);
3562 Elt0 = Builder.CreateIntrinsic(Intrinsic::sqrt, Elt0->getType(), Elt0);
3563 Rep = Builder.CreateInsertElement(Vec, Elt0, (uint64_t)0);
3564 } else if (Name.starts_with("avx.sqrt.p") ||
3565 Name.starts_with("sse2.sqrt.p") ||
3566 Name.starts_with("sse.sqrt.p")) {
3567 Rep = Builder.CreateIntrinsic(Intrinsic::sqrt, CI->getType(),
3568 {CI->getArgOperand(0)});
3569 } else if (Name.starts_with("avx512.mask.sqrt.p")) {
3570 if (CI->arg_size() == 4 &&
3571 (!isa<ConstantInt>(CI->getArgOperand(3)) ||
3572 cast<ConstantInt>(CI->getArgOperand(3))->getZExtValue() != 4)) {
3573 Intrinsic::ID IID = Name[18] == 's' ? Intrinsic::x86_avx512_sqrt_ps_512
3574 : Intrinsic::x86_avx512_sqrt_pd_512;
3575
3576 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(3)};
3577 Rep = Builder.CreateIntrinsic(IID, Args);
3578 } else {
3579 Rep = Builder.CreateIntrinsic(Intrinsic::sqrt, CI->getType(),
3580 {CI->getArgOperand(0)});
3581 }
3582 Rep =
3583 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
3584 } else if (Name.starts_with("avx512.ptestm") ||
3585 Name.starts_with("avx512.ptestnm")) {
3586 Value *Op0 = CI->getArgOperand(0);
3587 Value *Op1 = CI->getArgOperand(1);
3588 Value *Mask = CI->getArgOperand(2);
3589 Rep = Builder.CreateAnd(Op0, Op1);
3590 llvm::Type *Ty = Op0->getType();
3592 ICmpInst::Predicate Pred = Name.starts_with("avx512.ptestm")
3595 Rep = Builder.CreateICmp(Pred, Rep, Zero);
3596 Rep = applyX86MaskOn1BitsVec(Builder, Rep, Mask);
3597 } else if (Name.starts_with("avx512.mask.pbroadcast")) {
3598 unsigned NumElts = cast<FixedVectorType>(CI->getArgOperand(1)->getType())
3599 ->getNumElements();
3600 Rep = Builder.CreateVectorSplat(NumElts, CI->getArgOperand(0));
3601 Rep =
3602 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
3603 } else if (Name.starts_with("avx512.kunpck")) {
3604 unsigned NumElts = CI->getType()->getScalarSizeInBits();
3605 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), NumElts);
3606 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), NumElts);
3607 int Indices[64];
3608 for (unsigned i = 0; i != NumElts; ++i)
3609 Indices[i] = i;
3610
3611 // First extract half of each vector. This gives better codegen than
3612 // doing it in a single shuffle.
3613 LHS = Builder.CreateShuffleVector(LHS, LHS, ArrayRef(Indices, NumElts / 2));
3614 RHS = Builder.CreateShuffleVector(RHS, RHS, ArrayRef(Indices, NumElts / 2));
3615 // Concat the vectors.
3616 // NOTE: Operands have to be swapped to match intrinsic definition.
3617 Rep = Builder.CreateShuffleVector(RHS, LHS, ArrayRef(Indices, NumElts));
3618 Rep = Builder.CreateBitCast(Rep, CI->getType());
3619 } else if (Name == "avx512.kand.w") {
3620 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3621 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3622 Rep = Builder.CreateAnd(LHS, RHS);
3623 Rep = Builder.CreateBitCast(Rep, CI->getType());
3624 } else if (Name == "avx512.kandn.w") {
3625 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3626 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3627 LHS = Builder.CreateNot(LHS);
3628 Rep = Builder.CreateAnd(LHS, RHS);
3629 Rep = Builder.CreateBitCast(Rep, CI->getType());
3630 } else if (Name == "avx512.kor.w") {
3631 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3632 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3633 Rep = Builder.CreateOr(LHS, RHS);
3634 Rep = Builder.CreateBitCast(Rep, CI->getType());
3635 } else if (Name == "avx512.kxor.w") {
3636 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3637 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3638 Rep = Builder.CreateXor(LHS, RHS);
3639 Rep = Builder.CreateBitCast(Rep, CI->getType());
3640 } else if (Name == "avx512.kxnor.w") {
3641 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3642 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3643 LHS = Builder.CreateNot(LHS);
3644 Rep = Builder.CreateXor(LHS, RHS);
3645 Rep = Builder.CreateBitCast(Rep, CI->getType());
3646 } else if (Name == "avx512.knot.w") {
3647 Rep = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3648 Rep = Builder.CreateNot(Rep);
3649 Rep = Builder.CreateBitCast(Rep, CI->getType());
3650 } else if (Name == "avx512.kortestz.w" || Name == "avx512.kortestc.w") {
3651 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3652 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3653 Rep = Builder.CreateOr(LHS, RHS);
3654 Rep = Builder.CreateBitCast(Rep, Builder.getInt16Ty());
3655 Value *C;
3656 if (Name[14] == 'c')
3657 C = ConstantInt::getAllOnesValue(Builder.getInt16Ty());
3658 else
3659 C = ConstantInt::getNullValue(Builder.getInt16Ty());
3660 Rep = Builder.CreateICmpEQ(Rep, C);
3661 Rep = Builder.CreateZExt(Rep, Builder.getInt32Ty());
3662 } else if (Name == "sse.add.ss" || Name == "sse2.add.sd" ||
3663 Name == "sse.sub.ss" || Name == "sse2.sub.sd" ||
3664 Name == "sse.mul.ss" || Name == "sse2.mul.sd" ||
3665 Name == "sse.div.ss" || Name == "sse2.div.sd") {
3666 Type *I32Ty = Type::getInt32Ty(C);
3667 Value *Elt0 = Builder.CreateExtractElement(CI->getArgOperand(0),
3668 ConstantInt::get(I32Ty, 0));
3669 Value *Elt1 = Builder.CreateExtractElement(CI->getArgOperand(1),
3670 ConstantInt::get(I32Ty, 0));
3671 Value *EltOp;
3672 if (Name.contains(".add."))
3673 EltOp = Builder.CreateFAdd(Elt0, Elt1);
3674 else if (Name.contains(".sub."))
3675 EltOp = Builder.CreateFSub(Elt0, Elt1);
3676 else if (Name.contains(".mul."))
3677 EltOp = Builder.CreateFMul(Elt0, Elt1);
3678 else
3679 EltOp = Builder.CreateFDiv(Elt0, Elt1);
3680 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), EltOp,
3681 ConstantInt::get(I32Ty, 0));
3682 } else if (Name.starts_with("avx512.mask.pcmp")) {
3683 // "avx512.mask.pcmpeq." or "avx512.mask.pcmpgt."
3684 bool CmpEq = Name[16] == 'e';
3685 Rep = upgradeMaskedCompare(Builder, *CI, CmpEq ? 0 : 6, true);
3686 } else if (Name.starts_with("avx512.mask.vpshufbitqmb.")) {
3687 Type *OpTy = CI->getArgOperand(0)->getType();
3688 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3689 Intrinsic::ID IID;
3690 switch (VecWidth) {
3691 default:
3692 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
3693 break;
3694 case 128:
3695 IID = Intrinsic::x86_avx512_vpshufbitqmb_128;
3696 break;
3697 case 256:
3698 IID = Intrinsic::x86_avx512_vpshufbitqmb_256;
3699 break;
3700 case 512:
3701 IID = Intrinsic::x86_avx512_vpshufbitqmb_512;
3702 break;
3703 }
3704
3705 Rep =
3706 Builder.CreateIntrinsic(IID, {CI->getOperand(0), CI->getArgOperand(1)});
3707 Rep = applyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(2));
3708 } else if (Name.starts_with("avx512.mask.fpclass.p")) {
3709 Type *OpTy = CI->getArgOperand(0)->getType();
3710 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3711 unsigned EltWidth = OpTy->getScalarSizeInBits();
3712 Intrinsic::ID IID;
3713 if (VecWidth == 128 && EltWidth == 32)
3714 IID = Intrinsic::x86_avx512_fpclass_ps_128;
3715 else if (VecWidth == 256 && EltWidth == 32)
3716 IID = Intrinsic::x86_avx512_fpclass_ps_256;
3717 else if (VecWidth == 512 && EltWidth == 32)
3718 IID = Intrinsic::x86_avx512_fpclass_ps_512;
3719 else if (VecWidth == 128 && EltWidth == 64)
3720 IID = Intrinsic::x86_avx512_fpclass_pd_128;
3721 else if (VecWidth == 256 && EltWidth == 64)
3722 IID = Intrinsic::x86_avx512_fpclass_pd_256;
3723 else if (VecWidth == 512 && EltWidth == 64)
3724 IID = Intrinsic::x86_avx512_fpclass_pd_512;
3725 else
3726 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
3727
3728 Rep =
3729 Builder.CreateIntrinsic(IID, {CI->getOperand(0), CI->getArgOperand(1)});
3730 Rep = applyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(2));
3731 } else if (Name.starts_with("avx512.cmp.p")) {
3732 SmallVector<Value *, 4> Args(CI->args());
3733 Type *OpTy = Args[0]->getType();
3734 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3735 unsigned EltWidth = OpTy->getScalarSizeInBits();
3736 Intrinsic::ID IID;
3737 if (VecWidth == 128 && EltWidth == 32)
3738 IID = Intrinsic::x86_avx512_mask_cmp_ps_128;
3739 else if (VecWidth == 256 && EltWidth == 32)
3740 IID = Intrinsic::x86_avx512_mask_cmp_ps_256;
3741 else if (VecWidth == 512 && EltWidth == 32)
3742 IID = Intrinsic::x86_avx512_mask_cmp_ps_512;
3743 else if (VecWidth == 128 && EltWidth == 64)
3744 IID = Intrinsic::x86_avx512_mask_cmp_pd_128;
3745 else if (VecWidth == 256 && EltWidth == 64)
3746 IID = Intrinsic::x86_avx512_mask_cmp_pd_256;
3747 else if (VecWidth == 512 && EltWidth == 64)
3748 IID = Intrinsic::x86_avx512_mask_cmp_pd_512;
3749 else
3750 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
3751
3753 if (VecWidth == 512)
3754 std::swap(Mask, Args.back());
3755 Args.push_back(Mask);
3756
3757 Rep = Builder.CreateIntrinsic(IID, Args);
3758 } else if (Name.starts_with("avx512.mask.cmp.")) {
3759 // Integer compare intrinsics.
3760 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
3761 Rep = upgradeMaskedCompare(Builder, *CI, Imm, true);
3762 } else if (Name.starts_with("avx512.mask.ucmp.")) {
3763 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
3764 Rep = upgradeMaskedCompare(Builder, *CI, Imm, false);
3765 } else if (Name.starts_with("avx512.cvtb2mask.") ||
3766 Name.starts_with("avx512.cvtw2mask.") ||
3767 Name.starts_with("avx512.cvtd2mask.") ||
3768 Name.starts_with("avx512.cvtq2mask.")) {
3769 Value *Op = CI->getArgOperand(0);
3770 Value *Zero = llvm::Constant::getNullValue(Op->getType());
3771 Rep = Builder.CreateICmp(ICmpInst::ICMP_SLT, Op, Zero);
3772 Rep = applyX86MaskOn1BitsVec(Builder, Rep, nullptr);
3773 } else if (Name == "ssse3.pabs.b.128" || Name == "ssse3.pabs.w.128" ||
3774 Name == "ssse3.pabs.d.128" || Name.starts_with("avx2.pabs") ||
3775 Name.starts_with("avx512.mask.pabs")) {
3776 Rep = upgradeAbs(Builder, *CI);
3777 } else if (Name == "sse41.pmaxsb" || Name == "sse2.pmaxs.w" ||
3778 Name == "sse41.pmaxsd" || Name.starts_with("avx2.pmaxs") ||
3779 Name.starts_with("avx512.mask.pmaxs")) {
3780 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::smax);
3781 } else if (Name == "sse2.pmaxu.b" || Name == "sse41.pmaxuw" ||
3782 Name == "sse41.pmaxud" || Name.starts_with("avx2.pmaxu") ||
3783 Name.starts_with("avx512.mask.pmaxu")) {
3784 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::umax);
3785 } else if (Name == "sse41.pminsb" || Name == "sse2.pmins.w" ||
3786 Name == "sse41.pminsd" || Name.starts_with("avx2.pmins") ||
3787 Name.starts_with("avx512.mask.pmins")) {
3788 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::smin);
3789 } else if (Name == "sse2.pminu.b" || Name == "sse41.pminuw" ||
3790 Name == "sse41.pminud" || Name.starts_with("avx2.pminu") ||
3791 Name.starts_with("avx512.mask.pminu")) {
3792 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::umin);
3793 } else if (Name == "sse2.pmulh.w" || Name.starts_with("avx2.pmulh.w") ||
3794 Name.starts_with("avx512.pmulh.w")) {
3795 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::smulh);
3796 } else if (Name == "sse2.pmulhu.w" || Name.starts_with("avx2.pmulhu.w") ||
3797 Name.starts_with("avx512.pmulhu.w")) {
3798 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::umulh);
3799 } else if (Name == "sse2.pmulu.dq" || Name == "avx2.pmulu.dq" ||
3800 Name == "avx512.pmulu.dq.512" ||
3801 Name.starts_with("avx512.mask.pmulu.dq.")) {
3802 Rep = upgradePMULDQ(Builder, *CI, /*Signed*/ false);
3803 } else if (Name == "sse41.pmuldq" || Name == "avx2.pmul.dq" ||
3804 Name == "avx512.pmul.dq.512" ||
3805 Name.starts_with("avx512.mask.pmul.dq.")) {
3806 Rep = upgradePMULDQ(Builder, *CI, /*Signed*/ true);
3807 } else if (Name == "sse.cvtsi2ss" || Name == "sse2.cvtsi2sd" ||
3808 Name == "sse.cvtsi642ss" || Name == "sse2.cvtsi642sd") {
3809 Rep =
3810 Builder.CreateSIToFP(CI->getArgOperand(1),
3811 cast<VectorType>(CI->getType())->getElementType());
3812 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
3813 } else if (Name == "avx512.cvtusi2sd") {
3814 Rep =
3815 Builder.CreateUIToFP(CI->getArgOperand(1),
3816 cast<VectorType>(CI->getType())->getElementType());
3817 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
3818 } else if (Name == "sse2.cvtss2sd") {
3819 Rep = Builder.CreateExtractElement(CI->getArgOperand(1), (uint64_t)0);
3820 Rep = Builder.CreateFPExt(
3821 Rep, cast<VectorType>(CI->getType())->getElementType());
3822 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
3823 } else if (Name == "sse2.cvtdq2pd" || Name == "sse2.cvtdq2ps" ||
3824 Name == "avx.cvtdq2.pd.256" || Name == "avx.cvtdq2.ps.256" ||
3825 Name.starts_with("avx512.mask.cvtdq2pd.") ||
3826 Name.starts_with("avx512.mask.cvtudq2pd.") ||
3827 Name.starts_with("avx512.mask.cvtdq2ps.") ||
3828 Name.starts_with("avx512.mask.cvtudq2ps.") ||
3829 Name.starts_with("avx512.mask.cvtqq2pd.") ||
3830 Name.starts_with("avx512.mask.cvtuqq2pd.") ||
3831 Name == "avx512.mask.cvtqq2ps.256" ||
3832 Name == "avx512.mask.cvtqq2ps.512" ||
3833 Name == "avx512.mask.cvtuqq2ps.256" ||
3834 Name == "avx512.mask.cvtuqq2ps.512" || Name == "sse2.cvtps2pd" ||
3835 Name == "avx.cvt.ps2.pd.256" ||
3836 Name == "avx512.mask.cvtps2pd.128" ||
3837 Name == "avx512.mask.cvtps2pd.256") {
3838 auto *DstTy = cast<FixedVectorType>(CI->getType());
3839 Rep = CI->getArgOperand(0);
3840 auto *SrcTy = cast<FixedVectorType>(Rep->getType());
3841
3842 unsigned NumDstElts = DstTy->getNumElements();
3843 if (NumDstElts < SrcTy->getNumElements()) {
3844 assert(NumDstElts == 2 && "Unexpected vector size");
3845 Rep = Builder.CreateShuffleVector(Rep, Rep, ArrayRef<int>{0, 1});
3846 }
3847
3848 bool IsPS2PD = SrcTy->getElementType()->isFloatTy();
3849 bool IsUnsigned = Name.contains("cvtu");
3850 if (IsPS2PD)
3851 Rep = Builder.CreateFPExt(Rep, DstTy, "cvtps2pd");
3852 else if (CI->arg_size() == 4 &&
3853 (!isa<ConstantInt>(CI->getArgOperand(3)) ||
3854 cast<ConstantInt>(CI->getArgOperand(3))->getZExtValue() != 4)) {
3855 Intrinsic::ID IID = IsUnsigned ? Intrinsic::x86_avx512_uitofp_round
3856 : Intrinsic::x86_avx512_sitofp_round;
3857 Rep = Builder.CreateIntrinsic(IID, {DstTy, SrcTy},
3858 {Rep, CI->getArgOperand(3)});
3859 } else {
3860 Rep = IsUnsigned ? Builder.CreateUIToFP(Rep, DstTy, "cvt")
3861 : Builder.CreateSIToFP(Rep, DstTy, "cvt");
3862 }
3863
3864 if (CI->arg_size() >= 3)
3865 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
3866 CI->getArgOperand(1));
3867 } else if (Name.starts_with("avx512.mask.vcvtph2ps.") ||
3868 Name.starts_with("vcvtph2ps.")) {
3869 auto *DstTy = cast<FixedVectorType>(CI->getType());
3870 Rep = CI->getArgOperand(0);
3871 auto *SrcTy = cast<FixedVectorType>(Rep->getType());
3872 unsigned NumDstElts = DstTy->getNumElements();
3873 if (NumDstElts != SrcTy->getNumElements()) {
3874 assert(NumDstElts == 4 && "Unexpected vector size");
3875 Rep = Builder.CreateShuffleVector(Rep, Rep, ArrayRef<int>{0, 1, 2, 3});
3876 }
3877 Rep = Builder.CreateBitCast(
3878 Rep, FixedVectorType::get(Type::getHalfTy(C), NumDstElts));
3879 Rep = Builder.CreateFPExt(Rep, DstTy, "cvtph2ps");
3880 if (CI->arg_size() >= 3)
3881 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
3882 CI->getArgOperand(1));
3883 } else if (Name.starts_with("avx512.mask.load")) {
3884 // "avx512.mask.loadu." or "avx512.mask.load."
3885 bool Aligned = Name[16] != 'u'; // "avx512.mask.loadu".
3886 Rep = upgradeMaskedLoad(Builder, CI->getArgOperand(0), CI->getArgOperand(1),
3887 CI->getArgOperand(2), Aligned);
3888 } else if (Name.starts_with("avx512.mask.expand.load.")) {
3889 auto *ResultTy = cast<FixedVectorType>(CI->getType());
3890 auto *PtrTy = CI->getOperand(0)->getType();
3891 Value *MaskVec = getX86MaskVec(Builder, CI->getArgOperand(2),
3892 ResultTy->getNumElements());
3893 Rep = Builder.CreateIntrinsic(
3894 Intrinsic::masked_expandload, {ResultTy, PtrTy},
3895 {CI->getOperand(0), MaskVec, CI->getOperand(1)});
3896 } else if (Name.starts_with("avx512.mask.compress.store.")) {
3897 auto *ResultTy = cast<VectorType>(CI->getArgOperand(1)->getType());
3898 auto *PtrTy = CI->getArgOperand(0)->getType();
3899 Value *MaskVec =
3900 getX86MaskVec(Builder, CI->getArgOperand(2),
3901 cast<FixedVectorType>(ResultTy)->getNumElements());
3902 Rep = Builder.CreateIntrinsic(
3903 Intrinsic::masked_compressstore, {ResultTy, PtrTy},
3904 {CI->getArgOperand(1), CI->getArgOperand(0), MaskVec});
3905 } else if (Name.starts_with("avx512.mask.compress.") ||
3906 Name.starts_with("avx512.mask.expand.")) {
3907 auto *ResultTy = cast<FixedVectorType>(CI->getType());
3908
3909 Value *MaskVec = getX86MaskVec(Builder, CI->getArgOperand(2),
3910 ResultTy->getNumElements());
3911
3912 bool IsCompress = Name[12] == 'c';
3913 Intrinsic::ID IID = IsCompress ? Intrinsic::x86_avx512_mask_compress
3914 : Intrinsic::x86_avx512_mask_expand;
3915 Rep = Builder.CreateIntrinsic(
3916 IID, ResultTy, {CI->getOperand(0), CI->getOperand(1), MaskVec});
3917 } else if (Name.starts_with("xop.vpcom")) {
3918 bool IsSigned;
3919 if (Name.ends_with("ub") || Name.ends_with("uw") || Name.ends_with("ud") ||
3920 Name.ends_with("uq"))
3921 IsSigned = false;
3922 else if (Name.ends_with("b") || Name.ends_with("w") ||
3923 Name.ends_with("d") || Name.ends_with("q"))
3924 IsSigned = true;
3925 else
3926 reportFatalUsageErrorWithCI("Intrinsic has unknown suffix", CI);
3927
3928 unsigned Imm;
3929 if (CI->arg_size() == 3) {
3930 Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
3931 } else {
3932 Name = Name.substr(9); // strip off "xop.vpcom"
3933 if (Name.starts_with("lt"))
3934 Imm = 0;
3935 else if (Name.starts_with("le"))
3936 Imm = 1;
3937 else if (Name.starts_with("gt"))
3938 Imm = 2;
3939 else if (Name.starts_with("ge"))
3940 Imm = 3;
3941 else if (Name.starts_with("eq"))
3942 Imm = 4;
3943 else if (Name.starts_with("ne"))
3944 Imm = 5;
3945 else if (Name.starts_with("false"))
3946 Imm = 6;
3947 else if (Name.starts_with("true"))
3948 Imm = 7;
3949 else
3950 llvm_unreachable("Unknown condition");
3951 }
3952
3953 Rep = upgradeX86vpcom(Builder, *CI, Imm, IsSigned);
3954 } else if (Name.starts_with("xop.vpcmov")) {
3955 Value *Sel = CI->getArgOperand(2);
3956 Value *NotSel = Builder.CreateNot(Sel);
3957 Value *Sel0 = Builder.CreateAnd(CI->getArgOperand(0), Sel);
3958 Value *Sel1 = Builder.CreateAnd(CI->getArgOperand(1), NotSel);
3959 Rep = Builder.CreateOr(Sel0, Sel1);
3960 } else if (Name.starts_with("xop.vprot") || Name.starts_with("avx512.prol") ||
3961 Name.starts_with("avx512.mask.prol")) {
3962 Rep = upgradeX86Rotate(Builder, *CI, false);
3963 } else if (Name.starts_with("avx512.pror") ||
3964 Name.starts_with("avx512.mask.pror")) {
3965 Rep = upgradeX86Rotate(Builder, *CI, true);
3966 } else if (Name.starts_with("avx512.vpshld.") ||
3967 Name.starts_with("avx512.mask.vpshld") ||
3968 Name.starts_with("avx512.maskz.vpshld")) {
3969 bool ZeroMask = Name[11] == 'z';
3970 Rep = upgradeX86ConcatShift(Builder, *CI, false, ZeroMask);
3971 } else if (Name.starts_with("avx512.vpshrd.") ||
3972 Name.starts_with("avx512.mask.vpshrd") ||
3973 Name.starts_with("avx512.maskz.vpshrd")) {
3974 bool ZeroMask = Name[11] == 'z';
3975 Rep = upgradeX86ConcatShift(Builder, *CI, true, ZeroMask);
3976 } else if (Name == "sse42.crc32.64.8") {
3977 Value *Trunc0 =
3978 Builder.CreateTrunc(CI->getArgOperand(0), Type::getInt32Ty(C));
3979 Rep = Builder.CreateIntrinsic(Intrinsic::x86_sse42_crc32_32_8,
3980 {Trunc0, CI->getArgOperand(1)});
3981 Rep = Builder.CreateZExt(Rep, CI->getType(), "");
3982 } else if (Name.starts_with("avx.vbroadcast.s") ||
3983 Name.starts_with("avx512.vbroadcast.s")) {
3984 // Replace broadcasts with a series of insertelements.
3985 auto *VecTy = cast<FixedVectorType>(CI->getType());
3986 Type *EltTy = VecTy->getElementType();
3987 unsigned EltNum = VecTy->getNumElements();
3988 Value *Load = Builder.CreateLoad(EltTy, CI->getArgOperand(0));
3989 Type *I32Ty = Type::getInt32Ty(C);
3990 Rep = PoisonValue::get(VecTy);
3991 for (unsigned I = 0; I < EltNum; ++I)
3992 Rep = Builder.CreateInsertElement(Rep, Load, ConstantInt::get(I32Ty, I));
3993 } else if (Name.starts_with("sse41.pmovsx") ||
3994 Name.starts_with("sse41.pmovzx") ||
3995 Name.starts_with("avx2.pmovsx") ||
3996 Name.starts_with("avx2.pmovzx") ||
3997 Name.starts_with("avx512.mask.pmovsx") ||
3998 Name.starts_with("avx512.mask.pmovzx")) {
3999 auto *DstTy = cast<FixedVectorType>(CI->getType());
4000 unsigned NumDstElts = DstTy->getNumElements();
4001
4002 // Extract a subvector of the first NumDstElts lanes and sign/zero extend.
4003 SmallVector<int, 8> ShuffleMask(NumDstElts);
4004 for (unsigned i = 0; i != NumDstElts; ++i)
4005 ShuffleMask[i] = i;
4006
4007 Value *SV = Builder.CreateShuffleVector(CI->getArgOperand(0), ShuffleMask);
4008
4009 bool DoSext = Name.contains("pmovsx");
4010 Rep =
4011 DoSext ? Builder.CreateSExt(SV, DstTy) : Builder.CreateZExt(SV, DstTy);
4012 // If there are 3 arguments, it's a masked intrinsic so we need a select.
4013 if (CI->arg_size() == 3)
4014 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
4015 CI->getArgOperand(1));
4016 } else if (Name == "avx512.mask.pmov.qd.256" ||
4017 Name == "avx512.mask.pmov.qd.512" ||
4018 Name == "avx512.mask.pmov.wb.256" ||
4019 Name == "avx512.mask.pmov.wb.512") {
4020 Type *Ty = CI->getArgOperand(1)->getType();
4021 Rep = Builder.CreateTrunc(CI->getArgOperand(0), Ty);
4022 Rep =
4023 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4024 } else if (Name.starts_with("avx.vbroadcastf128") ||
4025 Name == "avx2.vbroadcasti128") {
4026 // Replace vbroadcastf128/vbroadcasti128 with a vector load+shuffle.
4027 Type *EltTy = cast<VectorType>(CI->getType())->getElementType();
4028 unsigned NumSrcElts = 128 / EltTy->getPrimitiveSizeInBits();
4029 auto *VT = FixedVectorType::get(EltTy, NumSrcElts);
4030 Value *Load = Builder.CreateAlignedLoad(VT, CI->getArgOperand(0), Align(1));
4031 if (NumSrcElts == 2)
4032 Rep = Builder.CreateShuffleVector(Load, ArrayRef<int>{0, 1, 0, 1});
4033 else
4034 Rep = Builder.CreateShuffleVector(Load,
4035 ArrayRef<int>{0, 1, 2, 3, 0, 1, 2, 3});
4036 } else if (Name.starts_with("avx512.mask.shuf.i") ||
4037 Name.starts_with("avx512.mask.shuf.f")) {
4038 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4039 Type *VT = CI->getType();
4040 unsigned NumLanes = VT->getPrimitiveSizeInBits() / 128;
4041 unsigned NumElementsInLane = 128 / VT->getScalarSizeInBits();
4042 unsigned ControlBitsMask = NumLanes - 1;
4043 unsigned NumControlBits = NumLanes / 2;
4044 SmallVector<int, 8> ShuffleMask(0);
4045
4046 for (unsigned l = 0; l != NumLanes; ++l) {
4047 unsigned LaneMask = (Imm >> (l * NumControlBits)) & ControlBitsMask;
4048 // We actually need the other source.
4049 if (l >= NumLanes / 2)
4050 LaneMask += NumLanes;
4051 for (unsigned i = 0; i != NumElementsInLane; ++i)
4052 ShuffleMask.push_back(LaneMask * NumElementsInLane + i);
4053 }
4054 Rep = Builder.CreateShuffleVector(CI->getArgOperand(0),
4055 CI->getArgOperand(1), ShuffleMask);
4056 Rep =
4057 emitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3));
4058 } else if (Name.starts_with("avx512.mask.broadcastf") ||
4059 Name.starts_with("avx512.mask.broadcasti")) {
4060 unsigned NumSrcElts = cast<FixedVectorType>(CI->getArgOperand(0)->getType())
4061 ->getNumElements();
4062 unsigned NumDstElts =
4063 cast<FixedVectorType>(CI->getType())->getNumElements();
4064
4065 SmallVector<int, 8> ShuffleMask(NumDstElts);
4066 for (unsigned i = 0; i != NumDstElts; ++i)
4067 ShuffleMask[i] = i % NumSrcElts;
4068
4069 Rep = Builder.CreateShuffleVector(CI->getArgOperand(0),
4070 CI->getArgOperand(0), ShuffleMask);
4071 Rep =
4072 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4073 } else if (Name.starts_with("avx2.pbroadcast") ||
4074 Name.starts_with("avx2.vbroadcast") ||
4075 Name.starts_with("avx512.pbroadcast") ||
4076 Name.starts_with("avx512.mask.broadcast.s")) {
4077 // Replace vp?broadcasts with a vector shuffle.
4078 Value *Op = CI->getArgOperand(0);
4079 ElementCount EC = cast<VectorType>(CI->getType())->getElementCount();
4080 Type *MaskTy = VectorType::get(Type::getInt32Ty(C), EC);
4083 Rep = Builder.CreateShuffleVector(Op, M);
4084
4085 if (CI->arg_size() == 3)
4086 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
4087 CI->getArgOperand(1));
4088 } else if (Name.starts_with("sse2.padds.") ||
4089 Name.starts_with("avx2.padds.") ||
4090 Name.starts_with("avx512.padds.") ||
4091 Name.starts_with("avx512.mask.padds.")) {
4092 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::sadd_sat);
4093 } else if (Name.starts_with("sse2.psubs.") ||
4094 Name.starts_with("avx2.psubs.") ||
4095 Name.starts_with("avx512.psubs.") ||
4096 Name.starts_with("avx512.mask.psubs.")) {
4097 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::ssub_sat);
4098 } else if (Name.starts_with("sse2.paddus.") ||
4099 Name.starts_with("avx2.paddus.") ||
4100 Name.starts_with("avx512.mask.paddus.")) {
4101 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::uadd_sat);
4102 } else if (Name.starts_with("sse2.psubus.") ||
4103 Name.starts_with("avx2.psubus.") ||
4104 Name.starts_with("avx512.mask.psubus.")) {
4105 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::usub_sat);
4106 } else if (Name.starts_with("avx512.mask.palignr.")) {
4107 Rep = upgradeX86ALIGNIntrinsics(Builder, CI->getArgOperand(0),
4108 CI->getArgOperand(1), CI->getArgOperand(2),
4109 CI->getArgOperand(3), CI->getArgOperand(4),
4110 false);
4111 } else if (Name.starts_with("avx512.mask.valign.")) {
4113 Builder, CI->getArgOperand(0), CI->getArgOperand(1),
4114 CI->getArgOperand(2), CI->getArgOperand(3), CI->getArgOperand(4), true);
4115 } else if (Name == "sse2.psll.dq" || Name == "avx2.psll.dq") {
4116 // 128/256-bit shift left specified in bits.
4117 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4118 Rep = upgradeX86PSLLDQIntrinsics(Builder, CI->getArgOperand(0),
4119 Shift / 8); // Shift is in bits.
4120 } else if (Name == "sse2.psrl.dq" || Name == "avx2.psrl.dq") {
4121 // 128/256-bit shift right specified in bits.
4122 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4123 Rep = upgradeX86PSRLDQIntrinsics(Builder, CI->getArgOperand(0),
4124 Shift / 8); // Shift is in bits.
4125 } else if (Name == "sse2.psll.dq.bs" || Name == "avx2.psll.dq.bs" ||
4126 Name == "avx512.psll.dq.512") {
4127 // 128/256/512-bit shift left specified in bytes.
4128 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4129 Rep = upgradeX86PSLLDQIntrinsics(Builder, CI->getArgOperand(0), Shift);
4130 } else if (Name == "sse2.psrl.dq.bs" || Name == "avx2.psrl.dq.bs" ||
4131 Name == "avx512.psrl.dq.512") {
4132 // 128/256/512-bit shift right specified in bytes.
4133 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4134 Rep = upgradeX86PSRLDQIntrinsics(Builder, CI->getArgOperand(0), Shift);
4135 } else if (Name == "sse41.pblendw" || Name.starts_with("sse41.blendp") ||
4136 Name.starts_with("avx.blend.p") || Name == "avx2.pblendw" ||
4137 Name.starts_with("avx2.pblendd.")) {
4138 Value *Op0 = CI->getArgOperand(0);
4139 Value *Op1 = CI->getArgOperand(1);
4140 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4141 auto *VecTy = cast<FixedVectorType>(CI->getType());
4142 unsigned NumElts = VecTy->getNumElements();
4143
4144 SmallVector<int, 16> Idxs(NumElts);
4145 for (unsigned i = 0; i != NumElts; ++i)
4146 Idxs[i] = ((Imm >> (i % 8)) & 1) ? i + NumElts : i;
4147
4148 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4149 } else if (Name.starts_with("avx.vinsertf128.") ||
4150 Name == "avx2.vinserti128" ||
4151 Name.starts_with("avx512.mask.insert")) {
4152 Value *Op0 = CI->getArgOperand(0);
4153 Value *Op1 = CI->getArgOperand(1);
4154 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4155 unsigned DstNumElts =
4156 cast<FixedVectorType>(CI->getType())->getNumElements();
4157 unsigned SrcNumElts =
4158 cast<FixedVectorType>(Op1->getType())->getNumElements();
4159 unsigned Scale = DstNumElts / SrcNumElts;
4160
4161 // Mask off the high bits of the immediate value; hardware ignores those.
4162 Imm = Imm % Scale;
4163
4164 // Extend the second operand into a vector the size of the destination.
4165 SmallVector<int, 8> Idxs(DstNumElts);
4166 for (unsigned i = 0; i != SrcNumElts; ++i)
4167 Idxs[i] = i;
4168 for (unsigned i = SrcNumElts; i != DstNumElts; ++i)
4169 Idxs[i] = SrcNumElts;
4170 Rep = Builder.CreateShuffleVector(Op1, Idxs);
4171
4172 // Insert the second operand into the first operand.
4173
4174 // Note that there is no guarantee that instruction lowering will actually
4175 // produce a vinsertf128 instruction for the created shuffles. In
4176 // particular, the 0 immediate case involves no lane changes, so it can
4177 // be handled as a blend.
4178
4179 // Example of shuffle mask for 32-bit elements:
4180 // Imm = 1 <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
4181 // Imm = 0 <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7 >
4182
4183 // First fill with identify mask.
4184 for (unsigned i = 0; i != DstNumElts; ++i)
4185 Idxs[i] = i;
4186 // Then replace the elements where we need to insert.
4187 for (unsigned i = 0; i != SrcNumElts; ++i)
4188 Idxs[i + Imm * SrcNumElts] = i + DstNumElts;
4189 Rep = Builder.CreateShuffleVector(Op0, Rep, Idxs);
4190
4191 // If the intrinsic has a mask operand, handle that.
4192 if (CI->arg_size() == 5)
4193 Rep = emitX86Select(Builder, CI->getArgOperand(4), Rep,
4194 CI->getArgOperand(3));
4195 } else if (Name.starts_with("avx.vextractf128.") ||
4196 Name == "avx2.vextracti128" ||
4197 Name.starts_with("avx512.mask.vextract")) {
4198 Value *Op0 = CI->getArgOperand(0);
4199 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4200 unsigned DstNumElts =
4201 cast<FixedVectorType>(CI->getType())->getNumElements();
4202 unsigned SrcNumElts =
4203 cast<FixedVectorType>(Op0->getType())->getNumElements();
4204 unsigned Scale = SrcNumElts / DstNumElts;
4205
4206 // Mask off the high bits of the immediate value; hardware ignores those.
4207 Imm = Imm % Scale;
4208
4209 // Get indexes for the subvector of the input vector.
4210 SmallVector<int, 8> Idxs(DstNumElts);
4211 for (unsigned i = 0; i != DstNumElts; ++i) {
4212 Idxs[i] = i + (Imm * DstNumElts);
4213 }
4214 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4215
4216 // If the intrinsic has a mask operand, handle that.
4217 if (CI->arg_size() == 4)
4218 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4219 CI->getArgOperand(2));
4220 } else if (Name.starts_with("avx512.mask.perm.df.") ||
4221 Name.starts_with("avx512.mask.perm.di.")) {
4222 Value *Op0 = CI->getArgOperand(0);
4223 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4224 auto *VecTy = cast<FixedVectorType>(CI->getType());
4225 unsigned NumElts = VecTy->getNumElements();
4226
4227 SmallVector<int, 8> Idxs(NumElts);
4228 for (unsigned i = 0; i != NumElts; ++i)
4229 Idxs[i] = (i & ~0x3) + ((Imm >> (2 * (i & 0x3))) & 3);
4230
4231 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4232
4233 if (CI->arg_size() == 4)
4234 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4235 CI->getArgOperand(2));
4236 } else if (Name.starts_with("avx.vperm2f128.") || Name == "avx2.vperm2i128") {
4237 // The immediate permute control byte looks like this:
4238 // [1:0] - select 128 bits from sources for low half of destination
4239 // [2] - ignore
4240 // [3] - zero low half of destination
4241 // [5:4] - select 128 bits from sources for high half of destination
4242 // [6] - ignore
4243 // [7] - zero high half of destination
4244
4245 uint8_t Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4246
4247 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4248 unsigned HalfSize = NumElts / 2;
4249 SmallVector<int, 8> ShuffleMask(NumElts);
4250
4251 // Determine which operand(s) are actually in use for this instruction.
4252 Value *V0 = (Imm & 0x02) ? CI->getArgOperand(1) : CI->getArgOperand(0);
4253 Value *V1 = (Imm & 0x20) ? CI->getArgOperand(1) : CI->getArgOperand(0);
4254
4255 // If needed, replace operands based on zero mask.
4256 V0 = (Imm & 0x08) ? ConstantAggregateZero::get(CI->getType()) : V0;
4257 V1 = (Imm & 0x80) ? ConstantAggregateZero::get(CI->getType()) : V1;
4258
4259 // Permute low half of result.
4260 unsigned StartIndex = (Imm & 0x01) ? HalfSize : 0;
4261 for (unsigned i = 0; i < HalfSize; ++i)
4262 ShuffleMask[i] = StartIndex + i;
4263
4264 // Permute high half of result.
4265 StartIndex = (Imm & 0x10) ? HalfSize : 0;
4266 for (unsigned i = 0; i < HalfSize; ++i)
4267 ShuffleMask[i + HalfSize] = NumElts + StartIndex + i;
4268
4269 Rep = Builder.CreateShuffleVector(V0, V1, ShuffleMask);
4270
4271 } else if (Name.starts_with("avx.vpermil.") || Name == "sse2.pshuf.d" ||
4272 Name.starts_with("avx512.mask.vpermil.p") ||
4273 Name.starts_with("avx512.mask.pshuf.d.")) {
4274 Value *Op0 = CI->getArgOperand(0);
4275 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4276 auto *VecTy = cast<FixedVectorType>(CI->getType());
4277 unsigned NumElts = VecTy->getNumElements();
4278 // Calculate the size of each index in the immediate.
4279 unsigned IdxSize = 64 / VecTy->getScalarSizeInBits();
4280 unsigned IdxMask = ((1 << IdxSize) - 1);
4281
4282 SmallVector<int, 8> Idxs(NumElts);
4283 // Lookup the bits for this element, wrapping around the immediate every
4284 // 8-bits. Elements are grouped into sets of 2 or 4 elements so we need
4285 // to offset by the first index of each group.
4286 for (unsigned i = 0; i != NumElts; ++i)
4287 Idxs[i] = ((Imm >> ((i * IdxSize) % 8)) & IdxMask) | (i & ~IdxMask);
4288
4289 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4290
4291 if (CI->arg_size() == 4)
4292 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4293 CI->getArgOperand(2));
4294 } else if (Name == "sse2.pshufl.w" ||
4295 Name.starts_with("avx512.mask.pshufl.w.")) {
4296 Value *Op0 = CI->getArgOperand(0);
4297 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4298 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4299
4300 if (Name == "sse2.pshufl.w" && NumElts % 8 != 0)
4301 reportFatalUsageErrorWithCI("Intrinsic has invalid signature", CI);
4302
4303 SmallVector<int, 16> Idxs(NumElts);
4304 for (unsigned l = 0; l != NumElts; l += 8) {
4305 for (unsigned i = 0; i != 4; ++i)
4306 Idxs[i + l] = ((Imm >> (2 * i)) & 0x3) + l;
4307 for (unsigned i = 4; i != 8; ++i)
4308 Idxs[i + l] = i + l;
4309 }
4310
4311 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4312
4313 if (CI->arg_size() == 4)
4314 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4315 CI->getArgOperand(2));
4316 } else if (Name == "sse2.pshufh.w" ||
4317 Name.starts_with("avx512.mask.pshufh.w.")) {
4318 Value *Op0 = CI->getArgOperand(0);
4319 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4320 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4321
4322 if (Name == "sse2.pshufh.w" && NumElts % 8 != 0)
4323 reportFatalUsageErrorWithCI("Intrinsic has invalid signature", CI);
4324
4325 SmallVector<int, 16> Idxs(NumElts);
4326 for (unsigned l = 0; l != NumElts; l += 8) {
4327 for (unsigned i = 0; i != 4; ++i)
4328 Idxs[i + l] = i + l;
4329 for (unsigned i = 0; i != 4; ++i)
4330 Idxs[i + l + 4] = ((Imm >> (2 * i)) & 0x3) + 4 + l;
4331 }
4332
4333 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4334
4335 if (CI->arg_size() == 4)
4336 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4337 CI->getArgOperand(2));
4338 } else if (Name.starts_with("avx512.mask.shuf.p")) {
4339 Value *Op0 = CI->getArgOperand(0);
4340 Value *Op1 = CI->getArgOperand(1);
4341 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4342 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4343
4344 unsigned NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4345 unsigned HalfLaneElts = NumLaneElts / 2;
4346
4347 SmallVector<int, 16> Idxs(NumElts);
4348 for (unsigned i = 0; i != NumElts; ++i) {
4349 // Base index is the starting element of the lane.
4350 Idxs[i] = i - (i % NumLaneElts);
4351 // If we are half way through the lane switch to the other source.
4352 if ((i % NumLaneElts) >= HalfLaneElts)
4353 Idxs[i] += NumElts;
4354 // Now select the specific element. By adding HalfLaneElts bits from
4355 // the immediate. Wrapping around the immediate every 8-bits.
4356 Idxs[i] += (Imm >> ((i * HalfLaneElts) % 8)) & ((1 << HalfLaneElts) - 1);
4357 }
4358
4359 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4360
4361 Rep =
4362 emitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3));
4363 } else if (Name.starts_with("avx512.mask.movddup") ||
4364 Name.starts_with("avx512.mask.movshdup") ||
4365 Name.starts_with("avx512.mask.movsldup")) {
4366 Value *Op0 = CI->getArgOperand(0);
4367 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4368 unsigned NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4369
4370 unsigned Offset = 0;
4371 if (Name.starts_with("avx512.mask.movshdup."))
4372 Offset = 1;
4373
4374 SmallVector<int, 16> Idxs(NumElts);
4375 for (unsigned l = 0; l != NumElts; l += NumLaneElts)
4376 for (unsigned i = 0; i != NumLaneElts; i += 2) {
4377 Idxs[i + l + 0] = i + l + Offset;
4378 Idxs[i + l + 1] = i + l + Offset;
4379 }
4380
4381 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4382
4383 Rep =
4384 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4385 } else if (Name.starts_with("avx512.mask.punpckl") ||
4386 Name.starts_with("avx512.mask.unpckl.")) {
4387 Value *Op0 = CI->getArgOperand(0);
4388 Value *Op1 = CI->getArgOperand(1);
4389 int NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4390 int NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4391
4392 SmallVector<int, 64> Idxs(NumElts);
4393 for (int l = 0; l != NumElts; l += NumLaneElts)
4394 for (int i = 0; i != NumLaneElts; ++i)
4395 Idxs[i + l] = l + (i / 2) + NumElts * (i % 2);
4396
4397 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4398
4399 Rep =
4400 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4401 } else if (Name.starts_with("avx512.mask.punpckh") ||
4402 Name.starts_with("avx512.mask.unpckh.")) {
4403 Value *Op0 = CI->getArgOperand(0);
4404 Value *Op1 = CI->getArgOperand(1);
4405 int NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4406 int NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4407
4408 SmallVector<int, 64> Idxs(NumElts);
4409 for (int l = 0; l != NumElts; l += NumLaneElts)
4410 for (int i = 0; i != NumLaneElts; ++i)
4411 Idxs[i + l] = (NumLaneElts / 2) + l + (i / 2) + NumElts * (i % 2);
4412
4413 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4414
4415 Rep =
4416 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4417 } else if (Name.starts_with("avx512.mask.and.") ||
4418 Name.starts_with("avx512.mask.pand.")) {
4419 VectorType *FTy = cast<VectorType>(CI->getType());
4421 Rep = Builder.CreateAnd(Builder.CreateBitCast(CI->getArgOperand(0), ITy),
4422 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4423 Rep = Builder.CreateBitCast(Rep, FTy);
4424 Rep =
4425 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4426 } else if (Name.starts_with("avx512.mask.andn.") ||
4427 Name.starts_with("avx512.mask.pandn.")) {
4428 VectorType *FTy = cast<VectorType>(CI->getType());
4430 Rep = Builder.CreateNot(Builder.CreateBitCast(CI->getArgOperand(0), ITy));
4431 Rep = Builder.CreateAnd(Rep,
4432 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4433 Rep = Builder.CreateBitCast(Rep, FTy);
4434 Rep =
4435 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4436 } else if (Name.starts_with("avx512.mask.or.") ||
4437 Name.starts_with("avx512.mask.por.")) {
4438 VectorType *FTy = cast<VectorType>(CI->getType());
4440 Rep = Builder.CreateOr(Builder.CreateBitCast(CI->getArgOperand(0), ITy),
4441 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4442 Rep = Builder.CreateBitCast(Rep, FTy);
4443 Rep =
4444 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4445 } else if (Name.starts_with("avx512.mask.xor.") ||
4446 Name.starts_with("avx512.mask.pxor.")) {
4447 VectorType *FTy = cast<VectorType>(CI->getType());
4449 Rep = Builder.CreateXor(Builder.CreateBitCast(CI->getArgOperand(0), ITy),
4450 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4451 Rep = Builder.CreateBitCast(Rep, FTy);
4452 Rep =
4453 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4454 } else if (Name.starts_with("avx512.mask.padd.")) {
4455 Rep = Builder.CreateAdd(CI->getArgOperand(0), CI->getArgOperand(1));
4456 Rep =
4457 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4458 } else if (Name.starts_with("avx512.mask.psub.")) {
4459 Rep = Builder.CreateSub(CI->getArgOperand(0), CI->getArgOperand(1));
4460 Rep =
4461 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4462 } else if (Name.starts_with("avx512.mask.pmull.")) {
4463 Rep = Builder.CreateMul(CI->getArgOperand(0), CI->getArgOperand(1));
4464 Rep =
4465 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4466 } else if (Name.starts_with("avx512.mask.add.p")) {
4467 if (Name.ends_with(".512")) {
4468 Intrinsic::ID IID;
4469 if (Name[17] == 's')
4470 IID = Intrinsic::x86_avx512_add_ps_512;
4471 else
4472 IID = Intrinsic::x86_avx512_add_pd_512;
4473
4474 Rep = Builder.CreateIntrinsic(
4475 IID,
4476 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4477 } else {
4478 Rep = Builder.CreateFAdd(CI->getArgOperand(0), CI->getArgOperand(1));
4479 }
4480 Rep =
4481 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4482 } else if (Name.starts_with("avx512.mask.div.p")) {
4483 if (Name.ends_with(".512")) {
4484 Intrinsic::ID IID;
4485 if (Name[17] == 's')
4486 IID = Intrinsic::x86_avx512_div_ps_512;
4487 else
4488 IID = Intrinsic::x86_avx512_div_pd_512;
4489
4490 Rep = Builder.CreateIntrinsic(
4491 IID,
4492 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4493 } else {
4494 Rep = Builder.CreateFDiv(CI->getArgOperand(0), CI->getArgOperand(1));
4495 }
4496 Rep =
4497 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4498 } else if (Name.starts_with("avx512.mask.mul.p")) {
4499 if (Name.ends_with(".512")) {
4500 Intrinsic::ID IID;
4501 if (Name[17] == 's')
4502 IID = Intrinsic::x86_avx512_mul_ps_512;
4503 else
4504 IID = Intrinsic::x86_avx512_mul_pd_512;
4505
4506 Rep = Builder.CreateIntrinsic(
4507 IID,
4508 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4509 } else {
4510 Rep = Builder.CreateFMul(CI->getArgOperand(0), CI->getArgOperand(1));
4511 }
4512 Rep =
4513 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4514 } else if (Name.starts_with("avx512.mask.sub.p")) {
4515 if (Name.ends_with(".512")) {
4516 Intrinsic::ID IID;
4517 if (Name[17] == 's')
4518 IID = Intrinsic::x86_avx512_sub_ps_512;
4519 else
4520 IID = Intrinsic::x86_avx512_sub_pd_512;
4521
4522 Rep = Builder.CreateIntrinsic(
4523 IID,
4524 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4525 } else {
4526 Rep = Builder.CreateFSub(CI->getArgOperand(0), CI->getArgOperand(1));
4527 }
4528 Rep =
4529 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4530 } else if ((Name.starts_with("avx512.mask.max.p") ||
4531 Name.starts_with("avx512.mask.min.p")) &&
4532 Name.drop_front(18) == ".512") {
4533 bool IsDouble = Name[17] == 'd';
4534 bool IsMin = Name[13] == 'i';
4535 static const Intrinsic::ID MinMaxTbl[2][2] = {
4536 {Intrinsic::x86_avx512_max_ps_512, Intrinsic::x86_avx512_max_pd_512},
4537 {Intrinsic::x86_avx512_min_ps_512, Intrinsic::x86_avx512_min_pd_512}};
4538 Intrinsic::ID IID = MinMaxTbl[IsMin][IsDouble];
4539
4540 Rep = Builder.CreateIntrinsic(
4541 IID,
4542 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4543 Rep =
4544 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4545 } else if (Name.starts_with("avx512.mask.lzcnt.")) {
4546 Rep =
4547 Builder.CreateIntrinsic(Intrinsic::ctlz, CI->getType(),
4548 {CI->getArgOperand(0), Builder.getInt1(false)});
4549 Rep =
4550 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4551 } else if (Name.starts_with("avx512.mask.psll")) {
4552 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4553 bool IsVariable = Name[16] == 'v';
4554 char Size = Name[16] == '.' ? Name[17]
4555 : Name[17] == '.' ? Name[18]
4556 : Name[18] == '.' ? Name[19]
4557 : Name[20];
4558
4559 Intrinsic::ID IID;
4560 if (IsVariable && Name[17] != '.') {
4561 if (Size == 'd' && Name[17] == '2') // avx512.mask.psllv2.di
4562 IID = Intrinsic::x86_avx2_psllv_q;
4563 else if (Size == 'd' && Name[17] == '4') // avx512.mask.psllv4.di
4564 IID = Intrinsic::x86_avx2_psllv_q_256;
4565 else if (Size == 's' && Name[17] == '4') // avx512.mask.psllv4.si
4566 IID = Intrinsic::x86_avx2_psllv_d;
4567 else if (Size == 's' && Name[17] == '8') // avx512.mask.psllv8.si
4568 IID = Intrinsic::x86_avx2_psllv_d_256;
4569 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psllv8.hi
4570 IID = Intrinsic::x86_avx512_psllv_w_128;
4571 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psllv16.hi
4572 IID = Intrinsic::x86_avx512_psllv_w_256;
4573 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psllv32hi
4574 IID = Intrinsic::x86_avx512_psllv_w_512;
4575 else
4576 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4577 } else if (Name.ends_with(".128")) {
4578 if (Size == 'd') // avx512.mask.psll.d.128, avx512.mask.psll.di.128
4579 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_d
4580 : Intrinsic::x86_sse2_psll_d;
4581 else if (Size == 'q') // avx512.mask.psll.q.128, avx512.mask.psll.qi.128
4582 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_q
4583 : Intrinsic::x86_sse2_psll_q;
4584 else if (Size == 'w') // avx512.mask.psll.w.128, avx512.mask.psll.wi.128
4585 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_w
4586 : Intrinsic::x86_sse2_psll_w;
4587 else
4588 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4589 } else if (Name.ends_with(".256")) {
4590 if (Size == 'd') // avx512.mask.psll.d.256, avx512.mask.psll.di.256
4591 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_d
4592 : Intrinsic::x86_avx2_psll_d;
4593 else if (Size == 'q') // avx512.mask.psll.q.256, avx512.mask.psll.qi.256
4594 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_q
4595 : Intrinsic::x86_avx2_psll_q;
4596 else if (Size == 'w') // avx512.mask.psll.w.256, avx512.mask.psll.wi.256
4597 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_w
4598 : Intrinsic::x86_avx2_psll_w;
4599 else
4600 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4601 } else {
4602 if (Size == 'd') // psll.di.512, pslli.d, psll.d, psllv.d.512
4603 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_d_512
4604 : IsVariable ? Intrinsic::x86_avx512_psllv_d_512
4605 : Intrinsic::x86_avx512_psll_d_512;
4606 else if (Size == 'q') // psll.qi.512, pslli.q, psll.q, psllv.q.512
4607 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_q_512
4608 : IsVariable ? Intrinsic::x86_avx512_psllv_q_512
4609 : Intrinsic::x86_avx512_psll_q_512;
4610 else if (Size == 'w') // psll.wi.512, pslli.w, psll.w
4611 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_w_512
4612 : Intrinsic::x86_avx512_psll_w_512;
4613 else
4614 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4615 }
4616
4617 Rep = upgradeX86MaskedShift(Builder, *CI, IID);
4618 } else if (Name.starts_with("avx512.mask.psrl")) {
4619 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4620 bool IsVariable = Name[16] == 'v';
4621 char Size = Name[16] == '.' ? Name[17]
4622 : Name[17] == '.' ? Name[18]
4623 : Name[18] == '.' ? Name[19]
4624 : Name[20];
4625
4626 Intrinsic::ID IID;
4627 if (IsVariable && Name[17] != '.') {
4628 if (Size == 'd' && Name[17] == '2') // avx512.mask.psrlv2.di
4629 IID = Intrinsic::x86_avx2_psrlv_q;
4630 else if (Size == 'd' && Name[17] == '4') // avx512.mask.psrlv4.di
4631 IID = Intrinsic::x86_avx2_psrlv_q_256;
4632 else if (Size == 's' && Name[17] == '4') // avx512.mask.psrlv4.si
4633 IID = Intrinsic::x86_avx2_psrlv_d;
4634 else if (Size == 's' && Name[17] == '8') // avx512.mask.psrlv8.si
4635 IID = Intrinsic::x86_avx2_psrlv_d_256;
4636 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrlv8.hi
4637 IID = Intrinsic::x86_avx512_psrlv_w_128;
4638 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrlv16.hi
4639 IID = Intrinsic::x86_avx512_psrlv_w_256;
4640 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrlv32hi
4641 IID = Intrinsic::x86_avx512_psrlv_w_512;
4642 else
4643 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4644 } else if (Name.ends_with(".128")) {
4645 if (Size == 'd') // avx512.mask.psrl.d.128, avx512.mask.psrl.di.128
4646 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_d
4647 : Intrinsic::x86_sse2_psrl_d;
4648 else if (Size == 'q') // avx512.mask.psrl.q.128, avx512.mask.psrl.qi.128
4649 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_q
4650 : Intrinsic::x86_sse2_psrl_q;
4651 else if (Size == 'w') // avx512.mask.psrl.w.128, avx512.mask.psrl.wi.128
4652 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_w
4653 : Intrinsic::x86_sse2_psrl_w;
4654 else
4655 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4656 } else if (Name.ends_with(".256")) {
4657 if (Size == 'd') // avx512.mask.psrl.d.256, avx512.mask.psrl.di.256
4658 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_d
4659 : Intrinsic::x86_avx2_psrl_d;
4660 else if (Size == 'q') // avx512.mask.psrl.q.256, avx512.mask.psrl.qi.256
4661 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_q
4662 : Intrinsic::x86_avx2_psrl_q;
4663 else if (Size == 'w') // avx512.mask.psrl.w.256, avx512.mask.psrl.wi.256
4664 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_w
4665 : Intrinsic::x86_avx2_psrl_w;
4666 else
4667 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4668 } else {
4669 if (Size == 'd') // psrl.di.512, psrli.d, psrl.d, psrl.d.512
4670 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_d_512
4671 : IsVariable ? Intrinsic::x86_avx512_psrlv_d_512
4672 : Intrinsic::x86_avx512_psrl_d_512;
4673 else if (Size == 'q') // psrl.qi.512, psrli.q, psrl.q, psrl.q.512
4674 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_q_512
4675 : IsVariable ? Intrinsic::x86_avx512_psrlv_q_512
4676 : Intrinsic::x86_avx512_psrl_q_512;
4677 else if (Size == 'w') // psrl.wi.512, psrli.w, psrl.w)
4678 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_w_512
4679 : Intrinsic::x86_avx512_psrl_w_512;
4680 else
4681 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4682 }
4683
4684 Rep = upgradeX86MaskedShift(Builder, *CI, IID);
4685 } else if (Name.starts_with("avx512.mask.psra")) {
4686 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4687 bool IsVariable = Name[16] == 'v';
4688 char Size = Name[16] == '.' ? Name[17]
4689 : Name[17] == '.' ? Name[18]
4690 : Name[18] == '.' ? Name[19]
4691 : Name[20];
4692
4693 Intrinsic::ID IID;
4694 if (IsVariable && Name[17] != '.') {
4695 if (Size == 's' && Name[17] == '4') // avx512.mask.psrav4.si
4696 IID = Intrinsic::x86_avx2_psrav_d;
4697 else if (Size == 's' && Name[17] == '8') // avx512.mask.psrav8.si
4698 IID = Intrinsic::x86_avx2_psrav_d_256;
4699 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrav8.hi
4700 IID = Intrinsic::x86_avx512_psrav_w_128;
4701 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrav16.hi
4702 IID = Intrinsic::x86_avx512_psrav_w_256;
4703 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrav32hi
4704 IID = Intrinsic::x86_avx512_psrav_w_512;
4705 else
4706 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4707 } else if (Name.ends_with(".128")) {
4708 if (Size == 'd') // avx512.mask.psra.d.128, avx512.mask.psra.di.128
4709 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_d
4710 : Intrinsic::x86_sse2_psra_d;
4711 else if (Size == 'q') // avx512.mask.psra.q.128, avx512.mask.psra.qi.128
4712 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_128
4713 : IsVariable ? Intrinsic::x86_avx512_psrav_q_128
4714 : Intrinsic::x86_avx512_psra_q_128;
4715 else if (Size == 'w') // avx512.mask.psra.w.128, avx512.mask.psra.wi.128
4716 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_w
4717 : Intrinsic::x86_sse2_psra_w;
4718 else
4719 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4720 } else if (Name.ends_with(".256")) {
4721 if (Size == 'd') // avx512.mask.psra.d.256, avx512.mask.psra.di.256
4722 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_d
4723 : Intrinsic::x86_avx2_psra_d;
4724 else if (Size == 'q') // avx512.mask.psra.q.256, avx512.mask.psra.qi.256
4725 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_256
4726 : IsVariable ? Intrinsic::x86_avx512_psrav_q_256
4727 : Intrinsic::x86_avx512_psra_q_256;
4728 else if (Size == 'w') // avx512.mask.psra.w.256, avx512.mask.psra.wi.256
4729 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_w
4730 : Intrinsic::x86_avx2_psra_w;
4731 else
4732 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4733 } else {
4734 if (Size == 'd') // psra.di.512, psrai.d, psra.d, psrav.d.512
4735 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_d_512
4736 : IsVariable ? Intrinsic::x86_avx512_psrav_d_512
4737 : Intrinsic::x86_avx512_psra_d_512;
4738 else if (Size == 'q') // psra.qi.512, psrai.q, psra.q
4739 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_512
4740 : IsVariable ? Intrinsic::x86_avx512_psrav_q_512
4741 : Intrinsic::x86_avx512_psra_q_512;
4742 else if (Size == 'w') // psra.wi.512, psrai.w, psra.w
4743 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_w_512
4744 : Intrinsic::x86_avx512_psra_w_512;
4745 else
4746 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4747 }
4748
4749 Rep = upgradeX86MaskedShift(Builder, *CI, IID);
4750 } else if (Name.starts_with("avx512.mask.move.s")) {
4751 Rep = upgradeMaskedMove(Builder, *CI);
4752 } else if (Name.starts_with("avx512.cvtmask2")) {
4753 Rep = upgradeMaskToInt(Builder, *CI);
4754 } else if (Name.ends_with(".movntdqa")) {
4756 C, ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1)));
4757
4758 LoadInst *LI = Builder.CreateAlignedLoad(
4759 CI->getType(), CI->getArgOperand(0),
4761 LI->setMetadata(LLVMContext::MD_nontemporal, Node);
4762 Rep = LI;
4763 } else if (Name.starts_with("fma.vfmadd.") ||
4764 Name.starts_with("fma.vfmsub.") ||
4765 Name.starts_with("fma.vfnmadd.") ||
4766 Name.starts_with("fma.vfnmsub.")) {
4767 bool NegMul = Name[6] == 'n';
4768 bool NegAcc = NegMul ? Name[8] == 's' : Name[7] == 's';
4769 bool IsScalar = NegMul ? Name[12] == 's' : Name[11] == 's';
4770
4771 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4772 CI->getArgOperand(2)};
4773
4774 if (IsScalar) {
4775 Ops[0] = Builder.CreateExtractElement(Ops[0], (uint64_t)0);
4776 Ops[1] = Builder.CreateExtractElement(Ops[1], (uint64_t)0);
4777 Ops[2] = Builder.CreateExtractElement(Ops[2], (uint64_t)0);
4778 }
4779
4780 if (NegMul && !IsScalar)
4781 Ops[0] = Builder.CreateFNeg(Ops[0]);
4782 if (NegMul && IsScalar)
4783 Ops[1] = Builder.CreateFNeg(Ops[1]);
4784 if (NegAcc)
4785 Ops[2] = Builder.CreateFNeg(Ops[2]);
4786
4787 Rep = Builder.CreateIntrinsic(Intrinsic::fma, Ops[0]->getType(), Ops);
4788
4789 if (IsScalar)
4790 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
4791 } else if (Name.starts_with("fma4.vfmadd.s")) {
4792 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4793 CI->getArgOperand(2)};
4794
4795 Ops[0] = Builder.CreateExtractElement(Ops[0], (uint64_t)0);
4796 Ops[1] = Builder.CreateExtractElement(Ops[1], (uint64_t)0);
4797 Ops[2] = Builder.CreateExtractElement(Ops[2], (uint64_t)0);
4798
4799 Rep = Builder.CreateIntrinsic(Intrinsic::fma, Ops[0]->getType(), Ops);
4800
4801 Rep = Builder.CreateInsertElement(Constant::getNullValue(CI->getType()),
4802 Rep, (uint64_t)0);
4803 } else if (Name.starts_with("avx512.mask.vfmadd.s") ||
4804 Name.starts_with("avx512.maskz.vfmadd.s") ||
4805 Name.starts_with("avx512.mask3.vfmadd.s") ||
4806 Name.starts_with("avx512.mask3.vfmsub.s") ||
4807 Name.starts_with("avx512.mask3.vfnmsub.s")) {
4808 bool IsMask3 = Name[11] == '3';
4809 bool IsMaskZ = Name[11] == 'z';
4810 // Drop the "avx512.mask." to make it easier.
4811 Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12);
4812 bool NegMul = Name[2] == 'n';
4813 bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's';
4814
4815 Value *A = CI->getArgOperand(0);
4816 Value *B = CI->getArgOperand(1);
4817 Value *C = CI->getArgOperand(2);
4818
4819 if (NegMul && (IsMask3 || IsMaskZ))
4820 A = Builder.CreateFNeg(A);
4821 if (NegMul && !(IsMask3 || IsMaskZ))
4822 B = Builder.CreateFNeg(B);
4823 if (NegAcc)
4824 C = Builder.CreateFNeg(C);
4825
4826 A = Builder.CreateExtractElement(A, (uint64_t)0);
4827 B = Builder.CreateExtractElement(B, (uint64_t)0);
4828 C = Builder.CreateExtractElement(C, (uint64_t)0);
4829
4830 if (!isa<ConstantInt>(CI->getArgOperand(4)) ||
4831 cast<ConstantInt>(CI->getArgOperand(4))->getZExtValue() != 4) {
4832 Value *Ops[] = {A, B, C, CI->getArgOperand(4)};
4833
4834 Intrinsic::ID IID;
4835 if (Name.back() == 'd')
4836 IID = Intrinsic::x86_avx512_vfmadd_f64;
4837 else
4838 IID = Intrinsic::x86_avx512_vfmadd_f32;
4839 Rep = Builder.CreateIntrinsic(IID, Ops);
4840 } else {
4841 Rep = Builder.CreateFMA(A, B, C);
4842 }
4843
4844 Value *PassThru = IsMaskZ ? Constant::getNullValue(Rep->getType())
4845 : IsMask3 ? C
4846 : A;
4847
4848 // For Mask3 with NegAcc, we need to create a new extractelement that
4849 // avoids the negation above.
4850 if (NegAcc && IsMask3)
4851 PassThru =
4852 Builder.CreateExtractElement(CI->getArgOperand(2), (uint64_t)0);
4853
4854 Rep = emitX86ScalarSelect(Builder, CI->getArgOperand(3), Rep, PassThru);
4855 Rep = Builder.CreateInsertElement(CI->getArgOperand(IsMask3 ? 2 : 0), Rep,
4856 (uint64_t)0);
4857 } else if (Name.starts_with("avx512.mask.vfmadd.p") ||
4858 Name.starts_with("avx512.mask.vfnmadd.p") ||
4859 Name.starts_with("avx512.mask.vfnmsub.p") ||
4860 Name.starts_with("avx512.mask3.vfmadd.p") ||
4861 Name.starts_with("avx512.mask3.vfmsub.p") ||
4862 Name.starts_with("avx512.mask3.vfnmsub.p") ||
4863 Name.starts_with("avx512.maskz.vfmadd.p")) {
4864 bool IsMask3 = Name[11] == '3';
4865 bool IsMaskZ = Name[11] == 'z';
4866 // Drop the "avx512.mask." to make it easier.
4867 Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12);
4868 bool NegMul = Name[2] == 'n';
4869 bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's';
4870
4871 Value *A = CI->getArgOperand(0);
4872 Value *B = CI->getArgOperand(1);
4873 Value *C = CI->getArgOperand(2);
4874
4875 if (NegMul && (IsMask3 || IsMaskZ))
4876 A = Builder.CreateFNeg(A);
4877 if (NegMul && !(IsMask3 || IsMaskZ))
4878 B = Builder.CreateFNeg(B);
4879 if (NegAcc)
4880 C = Builder.CreateFNeg(C);
4881
4882 if (CI->arg_size() == 5 &&
4883 (!isa<ConstantInt>(CI->getArgOperand(4)) ||
4884 cast<ConstantInt>(CI->getArgOperand(4))->getZExtValue() != 4)) {
4885 Intrinsic::ID IID;
4886 // Check the character before ".512" in string.
4887 if (Name[Name.size() - 5] == 's')
4888 IID = Intrinsic::x86_avx512_vfmadd_ps_512;
4889 else
4890 IID = Intrinsic::x86_avx512_vfmadd_pd_512;
4891
4892 Rep = Builder.CreateIntrinsic(IID, {A, B, C, CI->getArgOperand(4)});
4893 } else {
4894 Rep = Builder.CreateFMA(A, B, C);
4895 }
4896
4897 Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(CI->getType())
4898 : IsMask3 ? CI->getArgOperand(2)
4899 : CI->getArgOperand(0);
4900
4901 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
4902 } else if (Name.starts_with("fma.vfmsubadd.p")) {
4903 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4904 unsigned EltWidth = CI->getType()->getScalarSizeInBits();
4905 Intrinsic::ID IID;
4906 if (VecWidth == 128 && EltWidth == 32)
4907 IID = Intrinsic::x86_fma_vfmaddsub_ps;
4908 else if (VecWidth == 256 && EltWidth == 32)
4909 IID = Intrinsic::x86_fma_vfmaddsub_ps_256;
4910 else if (VecWidth == 128 && EltWidth == 64)
4911 IID = Intrinsic::x86_fma_vfmaddsub_pd;
4912 else if (VecWidth == 256 && EltWidth == 64)
4913 IID = Intrinsic::x86_fma_vfmaddsub_pd_256;
4914 else
4915 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
4916
4917 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4918 CI->getArgOperand(2)};
4919 Ops[2] = Builder.CreateFNeg(Ops[2]);
4920 Rep = Builder.CreateIntrinsic(IID, Ops);
4921 } else if (Name.starts_with("avx512.mask.vfmaddsub.p") ||
4922 Name.starts_with("avx512.mask3.vfmaddsub.p") ||
4923 Name.starts_with("avx512.maskz.vfmaddsub.p") ||
4924 Name.starts_with("avx512.mask3.vfmsubadd.p")) {
4925 bool IsMask3 = Name[11] == '3';
4926 bool IsMaskZ = Name[11] == 'z';
4927 // Drop the "avx512.mask." to make it easier.
4928 Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12);
4929 bool IsSubAdd = Name[3] == 's';
4930 if (CI->arg_size() == 5) {
4931 Intrinsic::ID IID;
4932 // Check the character before ".512" in string.
4933 if (Name[Name.size() - 5] == 's')
4934 IID = Intrinsic::x86_avx512_vfmaddsub_ps_512;
4935 else
4936 IID = Intrinsic::x86_avx512_vfmaddsub_pd_512;
4937
4938 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4939 CI->getArgOperand(2), CI->getArgOperand(4)};
4940 if (IsSubAdd)
4941 Ops[2] = Builder.CreateFNeg(Ops[2]);
4942
4943 Rep = Builder.CreateIntrinsic(IID, Ops);
4944 } else {
4945 int NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4946
4947 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4948 CI->getArgOperand(2)};
4949
4951 CI->getModule(), Intrinsic::fma, Ops[0]->getType());
4952 Value *Odd = Builder.CreateCall(FMA, Ops);
4953 Ops[2] = Builder.CreateFNeg(Ops[2]);
4954 Value *Even = Builder.CreateCall(FMA, Ops);
4955
4956 if (IsSubAdd)
4957 std::swap(Even, Odd);
4958
4959 SmallVector<int, 32> Idxs(NumElts);
4960 for (int i = 0; i != NumElts; ++i)
4961 Idxs[i] = i + (i % 2) * NumElts;
4962
4963 Rep = Builder.CreateShuffleVector(Even, Odd, Idxs);
4964 }
4965
4966 Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(CI->getType())
4967 : IsMask3 ? CI->getArgOperand(2)
4968 : CI->getArgOperand(0);
4969
4970 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
4971 } else if (Name.starts_with("avx512.mask.pternlog.") ||
4972 Name.starts_with("avx512.maskz.pternlog.")) {
4973 bool ZeroMask = Name[11] == 'z';
4974 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4975 unsigned EltWidth = CI->getType()->getScalarSizeInBits();
4976 Intrinsic::ID IID;
4977 if (VecWidth == 128 && EltWidth == 32)
4978 IID = Intrinsic::x86_avx512_pternlog_d_128;
4979 else if (VecWidth == 256 && EltWidth == 32)
4980 IID = Intrinsic::x86_avx512_pternlog_d_256;
4981 else if (VecWidth == 512 && EltWidth == 32)
4982 IID = Intrinsic::x86_avx512_pternlog_d_512;
4983 else if (VecWidth == 128 && EltWidth == 64)
4984 IID = Intrinsic::x86_avx512_pternlog_q_128;
4985 else if (VecWidth == 256 && EltWidth == 64)
4986 IID = Intrinsic::x86_avx512_pternlog_q_256;
4987 else if (VecWidth == 512 && EltWidth == 64)
4988 IID = Intrinsic::x86_avx512_pternlog_q_512;
4989 else
4990 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
4991
4992 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4993 CI->getArgOperand(2), CI->getArgOperand(3)};
4994 Rep = Builder.CreateIntrinsic(IID, Args);
4995 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
4996 : CI->getArgOperand(0);
4997 Rep = emitX86Select(Builder, CI->getArgOperand(4), Rep, PassThru);
4998 } else if (Name.starts_with("avx512.mask.vpmadd52") ||
4999 Name.starts_with("avx512.maskz.vpmadd52")) {
5000 bool ZeroMask = Name[11] == 'z';
5001 bool High = Name[20] == 'h' || Name[21] == 'h';
5002 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5003 Intrinsic::ID IID;
5004 if (VecWidth == 128 && !High)
5005 IID = Intrinsic::x86_avx512_vpmadd52l_uq_128;
5006 else if (VecWidth == 256 && !High)
5007 IID = Intrinsic::x86_avx512_vpmadd52l_uq_256;
5008 else if (VecWidth == 512 && !High)
5009 IID = Intrinsic::x86_avx512_vpmadd52l_uq_512;
5010 else if (VecWidth == 128 && High)
5011 IID = Intrinsic::x86_avx512_vpmadd52h_uq_128;
5012 else if (VecWidth == 256 && High)
5013 IID = Intrinsic::x86_avx512_vpmadd52h_uq_256;
5014 else if (VecWidth == 512 && High)
5015 IID = Intrinsic::x86_avx512_vpmadd52h_uq_512;
5016 else
5017 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5018
5019 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5020 CI->getArgOperand(2)};
5021 Rep = Builder.CreateIntrinsic(IID, Args);
5022 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
5023 : CI->getArgOperand(0);
5024 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
5025 } else if (Name.starts_with("avx512.mask.vpermi2var.") ||
5026 Name.starts_with("avx512.mask.vpermt2var.") ||
5027 Name.starts_with("avx512.maskz.vpermt2var.")) {
5028 bool ZeroMask = Name[11] == 'z';
5029 bool IndexForm = Name[17] == 'i';
5030 Rep = upgradeX86VPERMT2Intrinsics(Builder, *CI, ZeroMask, IndexForm);
5031 } else if (Name.starts_with("avx512.mask.vpdpbusd.") ||
5032 Name.starts_with("avx512.maskz.vpdpbusd.") ||
5033 Name.starts_with("avx512.mask.vpdpbusds.") ||
5034 Name.starts_with("avx512.maskz.vpdpbusds.")) {
5035 bool ZeroMask = Name[11] == 'z';
5036 bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's';
5037 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5038 Intrinsic::ID IID;
5039 if (VecWidth == 128 && !IsSaturating)
5040 IID = Intrinsic::x86_avx512_vpdpbusd_128;
5041 else if (VecWidth == 256 && !IsSaturating)
5042 IID = Intrinsic::x86_avx512_vpdpbusd_256;
5043 else if (VecWidth == 512 && !IsSaturating)
5044 IID = Intrinsic::x86_avx512_vpdpbusd_512;
5045 else if (VecWidth == 128 && IsSaturating)
5046 IID = Intrinsic::x86_avx512_vpdpbusds_128;
5047 else if (VecWidth == 256 && IsSaturating)
5048 IID = Intrinsic::x86_avx512_vpdpbusds_256;
5049 else if (VecWidth == 512 && IsSaturating)
5050 IID = Intrinsic::x86_avx512_vpdpbusds_512;
5051 else
5052 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5053
5054 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5055 CI->getArgOperand(2)};
5056
5057 // Input arguments types were incorrectly set to vectors of i32 before but
5058 // they should be vectors of i8. Insert bit cast when encountering the old
5059 // types
5060 if (Args[1]->getType()->isVectorTy() &&
5061 cast<VectorType>(Args[1]->getType())
5062 ->getElementType()
5063 ->isIntegerTy(32) &&
5064 Args[2]->getType()->isVectorTy() &&
5065 cast<VectorType>(Args[2]->getType())
5066 ->getElementType()
5067 ->isIntegerTy(32)) {
5068 Type *NewArgType = nullptr;
5069 if (VecWidth == 128)
5070 NewArgType = VectorType::get(Builder.getInt8Ty(), 16, false);
5071 else if (VecWidth == 256)
5072 NewArgType = VectorType::get(Builder.getInt8Ty(), 32, false);
5073 else if (VecWidth == 512)
5074 NewArgType = VectorType::get(Builder.getInt8Ty(), 64, false);
5075 else
5076 reportFatalUsageErrorWithCI("Intrinsic has unexpected vector bit width",
5077 CI);
5078
5079 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
5080 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
5081 }
5082
5083 Rep = Builder.CreateIntrinsic(IID, Args);
5084 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
5085 : CI->getArgOperand(0);
5086 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
5087 } else if (Name.starts_with("avx512.mask.vpdpwssd.") ||
5088 Name.starts_with("avx512.maskz.vpdpwssd.") ||
5089 Name.starts_with("avx512.mask.vpdpwssds.") ||
5090 Name.starts_with("avx512.maskz.vpdpwssds.")) {
5091 bool ZeroMask = Name[11] == 'z';
5092 bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's';
5093 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5094 Intrinsic::ID IID;
5095 if (VecWidth == 128 && !IsSaturating)
5096 IID = Intrinsic::x86_avx512_vpdpwssd_128;
5097 else if (VecWidth == 256 && !IsSaturating)
5098 IID = Intrinsic::x86_avx512_vpdpwssd_256;
5099 else if (VecWidth == 512 && !IsSaturating)
5100 IID = Intrinsic::x86_avx512_vpdpwssd_512;
5101 else if (VecWidth == 128 && IsSaturating)
5102 IID = Intrinsic::x86_avx512_vpdpwssds_128;
5103 else if (VecWidth == 256 && IsSaturating)
5104 IID = Intrinsic::x86_avx512_vpdpwssds_256;
5105 else if (VecWidth == 512 && IsSaturating)
5106 IID = Intrinsic::x86_avx512_vpdpwssds_512;
5107 else
5108 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5109
5110 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5111 CI->getArgOperand(2)};
5112
5113 // Input arguments types were incorrectly set to vectors of i32 before but
5114 // they should be vectors of i16. Insert bit cast when encountering the old
5115 // types
5116 if (Args[1]->getType()->isVectorTy() &&
5117 cast<VectorType>(Args[1]->getType())
5118 ->getElementType()
5119 ->isIntegerTy(32) &&
5120 Args[2]->getType()->isVectorTy() &&
5121 cast<VectorType>(Args[2]->getType())
5122 ->getElementType()
5123 ->isIntegerTy(32)) {
5124 Type *NewArgType = nullptr;
5125 if (VecWidth == 128)
5126 NewArgType = VectorType::get(Builder.getInt16Ty(), 8, false);
5127 else if (VecWidth == 256)
5128 NewArgType = VectorType::get(Builder.getInt16Ty(), 16, false);
5129 else if (VecWidth == 512)
5130 NewArgType = VectorType::get(Builder.getInt16Ty(), 32, false);
5131 else
5132 reportFatalUsageErrorWithCI("Intrinsic has unexpected vector bit width",
5133 CI);
5134
5135 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
5136 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
5137 }
5138
5139 Rep = Builder.CreateIntrinsic(IID, Args);
5140 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
5141 : CI->getArgOperand(0);
5142 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
5143 } else if (Name == "addcarryx.u32" || Name == "addcarryx.u64" ||
5144 Name == "addcarry.u32" || Name == "addcarry.u64" ||
5145 Name == "subborrow.u32" || Name == "subborrow.u64") {
5146 Intrinsic::ID IID;
5147 if (Name[0] == 'a' && Name.back() == '2')
5148 IID = Intrinsic::x86_addcarry_32;
5149 else if (Name[0] == 'a' && Name.back() == '4')
5150 IID = Intrinsic::x86_addcarry_64;
5151 else if (Name[0] == 's' && Name.back() == '2')
5152 IID = Intrinsic::x86_subborrow_32;
5153 else if (Name[0] == 's' && Name.back() == '4')
5154 IID = Intrinsic::x86_subborrow_64;
5155 else
5156 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5157
5158 // Make a call with 3 operands.
5159 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5160 CI->getArgOperand(2)};
5161 Value *NewCall = Builder.CreateIntrinsic(IID, Args);
5162
5163 // Extract the second result and store it.
5164 Value *Data = Builder.CreateExtractValue(NewCall, 1);
5165 Builder.CreateAlignedStore(Data, CI->getArgOperand(3), Align(1));
5166 // Replace the original call result with the first result of the new call.
5167 Value *CF = Builder.CreateExtractValue(NewCall, 0);
5168
5169 CI->replaceAllUsesWith(CF);
5170 Rep = nullptr;
5171 } else if (Name.starts_with("avx512.mask.") &&
5172 upgradeAVX512MaskToSelect(Name, Builder, *CI, Rep)) {
5173 // Rep will be updated by the call in the condition.
5174 } else if (Name.starts_with("bmi.pdep.")) {
5175 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::pdep);
5176 } else if (Name.starts_with("bmi.pext.")) {
5177 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::pext);
5178 } else
5179 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5180
5181 return Rep;
5182}
5183
5185 Function *F, IRBuilder<> &Builder) {
5186 if (Name.starts_with("neon.bfcvt")) {
5187 if (Name.starts_with("neon.bfcvtn2")) {
5188 SmallVector<int, 32> LoMask(4);
5189 std::iota(LoMask.begin(), LoMask.end(), 0);
5190 SmallVector<int, 32> ConcatMask(8);
5191 std::iota(ConcatMask.begin(), ConcatMask.end(), 0);
5192 Value *Inactive = Builder.CreateShuffleVector(CI->getOperand(0), LoMask);
5193 Value *Trunc =
5194 Builder.CreateFPTrunc(CI->getOperand(1), Inactive->getType());
5195 return Builder.CreateShuffleVector(Inactive, Trunc, ConcatMask);
5196 } else if (Name.starts_with("neon.bfcvtn")) {
5197 SmallVector<int, 32> ConcatMask(8);
5198 std::iota(ConcatMask.begin(), ConcatMask.end(), 0);
5199 Type *V4BF16 =
5200 FixedVectorType::get(Type::getBFloatTy(F->getContext()), 4);
5201 Value *Trunc = Builder.CreateFPTrunc(CI->getOperand(0), V4BF16);
5202 dbgs() << "Trunc: " << *Trunc << "\n";
5203 return Builder.CreateShuffleVector(
5204 Trunc, ConstantAggregateZero::get(V4BF16), ConcatMask);
5205 } else {
5206 return Builder.CreateFPTrunc(CI->getOperand(0),
5207 Type::getBFloatTy(F->getContext()));
5208 }
5209 } else if (Name.starts_with("sve.fcvt")) {
5210 Intrinsic::ID NewID =
5212 .Case("sve.fcvt.bf16f32", Intrinsic::aarch64_sve_fcvt_bf16f32_v2)
5213 .Case("sve.fcvtnt.bf16f32",
5214 Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2)
5216 if (NewID == Intrinsic::not_intrinsic)
5217 llvm_unreachable("Unhandled Intrinsic!");
5218
5219 SmallVector<Value *, 3> Args(CI->args());
5220
5221 // The original intrinsics incorrectly used a predicate based on the
5222 // smallest element type rather than the largest.
5223 Type *BadPredTy = ScalableVectorType::get(Builder.getInt1Ty(), 8);
5224 Type *GoodPredTy = ScalableVectorType::get(Builder.getInt1Ty(), 4);
5225
5226 if (Args[1]->getType() != BadPredTy)
5227 llvm_unreachable("Unexpected predicate type!");
5228
5229 Args[1] = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_convert_to_svbool,
5230 BadPredTy, Args[1]);
5231 Args[1] = Builder.CreateIntrinsic(
5232 Intrinsic::aarch64_sve_convert_from_svbool, GoodPredTy, Args[1]);
5233
5234 return Builder.CreateIntrinsic(NewID, Args, /*FMFSource=*/nullptr,
5235 CI->getName());
5236 }
5237
5238 if (Name == "neon.vcvtfp2hf")
5239 return Builder.CreateBitCast(
5240 Builder.CreateFPTrunc(
5241 CI->getOperand(0),
5242 FixedVectorType::get(Type::getHalfTy(F->getContext()), 4)),
5243 FixedVectorType::get(Type::getInt16Ty(F->getContext()), 4));
5244 if (Name == "neon.vcvthf2fp")
5245 return Builder.CreateFPExt(
5246 Builder.CreateBitCast(
5247 CI->getOperand(0),
5248 FixedVectorType::get(Type::getHalfTy(F->getContext()), 4)),
5249 FixedVectorType::get(Type::getFloatTy(F->getContext()), 4));
5250
5251 llvm_unreachable("Unhandled Intrinsic!");
5252}
5253
5255 IRBuilder<> &Builder) {
5256 if (Name == "mve.vctp64.old") {
5257 // Replace the old v4i1 vctp64 with a v2i1 vctp and predicate-casts to the
5258 // correct type.
5259 Value *VCTP = Builder.CreateIntrinsic(Intrinsic::arm_mve_vctp64, {},
5260 CI->getArgOperand(0),
5261 /*FMFSource=*/nullptr, CI->getName());
5262 Value *C1 = Builder.CreateIntrinsic(
5263 Intrinsic::arm_mve_pred_v2i,
5264 {VectorType::get(Builder.getInt1Ty(), 2, false)}, VCTP);
5265 return Builder.CreateIntrinsic(
5266 Intrinsic::arm_mve_pred_i2v,
5267 {VectorType::get(Builder.getInt1Ty(), 4, false)}, C1);
5268 } else if (Name == "mve.mull.int.predicated.v2i64.v4i32.v4i1" ||
5269 Name == "mve.vqdmull.predicated.v2i64.v4i32.v4i1" ||
5270 Name == "mve.vldr.gather.base.predicated.v2i64.v2i64.v4i1" ||
5271 Name == "mve.vldr.gather.base.wb.predicated.v2i64.v2i64.v4i1" ||
5272 Name ==
5273 "mve.vldr.gather.offset.predicated.v2i64.p0i64.v2i64.v4i1" ||
5274 Name == "mve.vldr.gather.offset.predicated.v2i64.p0.v2i64.v4i1" ||
5275 Name == "mve.vstr.scatter.base.predicated.v2i64.v2i64.v4i1" ||
5276 Name == "mve.vstr.scatter.base.wb.predicated.v2i64.v2i64.v4i1" ||
5277 Name ==
5278 "mve.vstr.scatter.offset.predicated.p0i64.v2i64.v2i64.v4i1" ||
5279 Name == "mve.vstr.scatter.offset.predicated.p0.v2i64.v2i64.v4i1" ||
5280 Name == "cde.vcx1q.predicated.v2i64.v4i1" ||
5281 Name == "cde.vcx1qa.predicated.v2i64.v4i1" ||
5282 Name == "cde.vcx2q.predicated.v2i64.v4i1" ||
5283 Name == "cde.vcx2qa.predicated.v2i64.v4i1" ||
5284 Name == "cde.vcx3q.predicated.v2i64.v4i1" ||
5285 Name == "cde.vcx3qa.predicated.v2i64.v4i1") {
5286 std::vector<Type *> Tys;
5287 unsigned ID = CI->getIntrinsicID();
5288 Type *V2I1Ty = FixedVectorType::get(Builder.getInt1Ty(), 2);
5289 switch (ID) {
5290 case Intrinsic::arm_mve_mull_int_predicated:
5291 case Intrinsic::arm_mve_vqdmull_predicated:
5292 case Intrinsic::arm_mve_vldr_gather_base_predicated:
5293 Tys = {CI->getType(), CI->getOperand(0)->getType(), V2I1Ty};
5294 break;
5295 case Intrinsic::arm_mve_vldr_gather_base_wb_predicated:
5296 case Intrinsic::arm_mve_vstr_scatter_base_predicated:
5297 case Intrinsic::arm_mve_vstr_scatter_base_wb_predicated:
5298 Tys = {CI->getOperand(0)->getType(), CI->getOperand(0)->getType(),
5299 V2I1Ty};
5300 break;
5301 case Intrinsic::arm_mve_vldr_gather_offset_predicated:
5302 Tys = {CI->getType(), CI->getOperand(0)->getType(),
5303 CI->getOperand(1)->getType(), V2I1Ty};
5304 break;
5305 case Intrinsic::arm_mve_vstr_scatter_offset_predicated:
5306 Tys = {CI->getOperand(0)->getType(), CI->getOperand(1)->getType(),
5307 CI->getOperand(2)->getType(), V2I1Ty};
5308 break;
5309 case Intrinsic::arm_cde_vcx1q_predicated:
5310 case Intrinsic::arm_cde_vcx1qa_predicated:
5311 case Intrinsic::arm_cde_vcx2q_predicated:
5312 case Intrinsic::arm_cde_vcx2qa_predicated:
5313 case Intrinsic::arm_cde_vcx3q_predicated:
5314 case Intrinsic::arm_cde_vcx3qa_predicated:
5315 Tys = {CI->getOperand(1)->getType(), V2I1Ty};
5316 break;
5317 default:
5318 llvm_unreachable("Unhandled Intrinsic!");
5319 }
5320
5321 std::vector<Value *> Ops;
5322 for (Value *Op : CI->args()) {
5323 Type *Ty = Op->getType();
5324 if (Ty->getScalarSizeInBits() == 1) {
5325 Value *C1 = Builder.CreateIntrinsic(
5326 Intrinsic::arm_mve_pred_v2i,
5327 {VectorType::get(Builder.getInt1Ty(), 4, false)}, Op);
5328 Op = Builder.CreateIntrinsic(Intrinsic::arm_mve_pred_i2v, {V2I1Ty}, C1);
5329 }
5330 Ops.push_back(Op);
5331 }
5332
5333 return Builder.CreateIntrinsic(ID, Tys, Ops, /*FMFSource=*/nullptr,
5334 CI->getName());
5335 }
5336 llvm_unreachable("Unknown function for ARM CallBase upgrade.");
5337}
5338
5339// These are expected to have the arguments:
5340// atomic.intrin (ptr, rmw_value, ordering, scope, isVolatile)
5341//
5342// Except for int_amdgcn_ds_fadd_v2bf16 which only has (ptr, rmw_value).
5343//
5345 Function *F, IRBuilder<> &Builder) {
5346 // Legacy WMMA iu intrinsics missed the optional clamp operand. Append clamp=0
5347 // for compatibility.
5348 auto UpgradeLegacyWMMAIUIntrinsicCall =
5349 [](Function *F, CallBase *CI, IRBuilder<> &Builder,
5350 ArrayRef<Type *> OverloadTys) -> Value * {
5351 // Prepare arguments, append clamp=0 for compatibility
5352 SmallVector<Value *, 10> Args(CI->args().begin(), CI->args().end());
5353 Args.push_back(Builder.getFalse());
5354
5355 // Insert the declaration for the right overload types
5357 F->getParent(), F->getIntrinsicID(), OverloadTys);
5358
5359 // Copy operand bundles if any
5361 CI->getOperandBundlesAsDefs(Bundles);
5362
5363 // Create the new call and copy calling properties
5364 auto *NewCall = cast<CallInst>(Builder.CreateCall(NewDecl, Args, Bundles));
5365 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
5366 NewCall->setCallingConv(CI->getCallingConv());
5367 NewCall->setAttributes(CI->getAttributes());
5368 NewCall->copyMetadata(*CI);
5369 return NewCall;
5370 };
5371
5372 if (F->getIntrinsicID() == Intrinsic::amdgcn_wmma_i32_16x16x64_iu8) {
5373 assert(CI->arg_size() == 7 && "Legacy int_amdgcn_wmma_i32_16x16x64_iu8 "
5374 "intrinsic should have 7 arguments");
5375 Type *T1 = CI->getArgOperand(4)->getType();
5376 Type *T2 = CI->getArgOperand(1)->getType();
5377 return UpgradeLegacyWMMAIUIntrinsicCall(F, CI, Builder, {T1, T2});
5378 }
5379 if (F->getIntrinsicID() == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8) {
5380 assert(CI->arg_size() == 8 && "Legacy int_amdgcn_swmmac_i32_16x16x128_iu8 "
5381 "intrinsic should have 8 arguments");
5382 Type *T1 = CI->getArgOperand(4)->getType();
5383 Type *T2 = CI->getArgOperand(1)->getType();
5384 Type *T3 = CI->getArgOperand(3)->getType();
5385 Type *T4 = CI->getArgOperand(5)->getType();
5386 return UpgradeLegacyWMMAIUIntrinsicCall(F, CI, Builder, {T1, T2, T3, T4});
5387 }
5388
5389 switch (F->getIntrinsicID()) {
5390 default:
5391 break;
5392 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
5393 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
5394 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
5395 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
5396 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
5397 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16: {
5398 // Drop src0 and src1 modifiers.
5399 const Value *Op0 = CI->getArgOperand(0);
5400 const Value *Op2 = CI->getArgOperand(2);
5401 assert(Op0->getType()->isIntegerTy() && Op2->getType()->isIntegerTy());
5402 const ConstantInt *ModA = dyn_cast<ConstantInt>(Op0);
5403 const ConstantInt *ModB = dyn_cast<ConstantInt>(Op2);
5404 if (!ModA->isZero() || !ModB->isZero())
5405 reportFatalUsageError(Name + " matrix A and B modifiers shall be zero");
5406
5408 for (int I = 4, E = CI->arg_size(); I < E; ++I)
5409 Args.push_back(CI->getArgOperand(I));
5410
5411 SmallVector<Type *, 3> Overloads{F->getReturnType(), Args[0]->getType()};
5412 if (F->getIntrinsicID() == Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16)
5413 Overloads.push_back(Args[3]->getType());
5415 F->getParent(), F->getIntrinsicID(), Overloads);
5416
5418 CI->getOperandBundlesAsDefs(Bundles);
5419
5420 auto *NewCall = cast<CallInst>(Builder.CreateCall(NewDecl, Args, Bundles));
5421 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
5422 NewCall->setCallingConv(CI->getCallingConv());
5423 NewCall->setAttributes(CI->getAttributes());
5424 NewCall->copyMetadata(*CI);
5425 NewCall->takeName(CI);
5426 return NewCall;
5427 }
5428 }
5429
5430 if (Name.starts_with("fcmp.") || Name.starts_with("icmp.")) {
5431 Value *LHS = CI->getArgOperand(0);
5432 Value *RHS = CI->getArgOperand(1);
5433 CmpInst::Predicate Pred = static_cast<CmpInst::Predicate>(
5434 cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue());
5435 Value *Cmp = Builder.CreateCmp(Pred, LHS, RHS);
5436 CallInst *NewCall = Builder.CreateIntrinsicWithoutFolding(
5437 CI->getType(), Intrinsic::amdgcn_ballot, Cmp);
5438 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
5439 NewCall->setCallingConv(CI->getCallingConv());
5440 NewCall->copyMetadata(*CI);
5441 NewCall->takeName(CI);
5442 return NewCall;
5443 }
5444
5445 if (Name.starts_with("addrspacecast.nonnull")) {
5446 if (CI->getNumOperands() < 2) // Malformed bitcode.
5447 return nullptr;
5448 Value *ASC = Builder.CreateAddrSpaceCast(
5449 CI->getArgOperand(0), CI->getType(), "", /*IsNonNull=*/true);
5450 ASC->takeName(CI);
5451 return ASC;
5452 }
5453
5454 AtomicRMWInst::BinOp RMWOp =
5456 .StartsWith("ds.fadd", AtomicRMWInst::FAdd)
5457 .StartsWith("ds.fmin", AtomicRMWInst::FMin)
5458 .StartsWith("ds.fmax", AtomicRMWInst::FMax)
5459 .StartsWith("atomic.inc.", AtomicRMWInst::UIncWrap)
5460 .StartsWith("atomic.dec.", AtomicRMWInst::UDecWrap)
5461 .StartsWith("global.atomic.fadd", AtomicRMWInst::FAdd)
5462 .StartsWith("flat.atomic.fadd", AtomicRMWInst::FAdd)
5463 .StartsWith("global.atomic.fmin", AtomicRMWInst::FMin)
5464 .StartsWith("flat.atomic.fmin", AtomicRMWInst::FMin)
5465 .StartsWith("global.atomic.fmax", AtomicRMWInst::FMax)
5466 .StartsWith("flat.atomic.fmax", AtomicRMWInst::FMax)
5467 .StartsWith("atomic.cond.sub", AtomicRMWInst::USubCond)
5468 .StartsWith("atomic.csub", AtomicRMWInst::USubSat);
5469
5470 unsigned NumOperands = CI->getNumOperands();
5471 if (NumOperands < 3) // Malformed bitcode.
5472 return nullptr;
5473
5474 Value *Ptr = CI->getArgOperand(0);
5475 PointerType *PtrTy = dyn_cast<PointerType>(Ptr->getType());
5476 if (!PtrTy) // Malformed.
5477 return nullptr;
5478
5479 Value *Val = CI->getArgOperand(1);
5480 if (Val->getType() != CI->getType()) // Malformed.
5481 return nullptr;
5482
5483 ConstantInt *OrderArg = nullptr;
5484 bool IsVolatile = false;
5485
5486 // These should have 5 arguments (plus the callee). A separate version of the
5487 // ds_fadd intrinsic was defined for bf16 which was missing arguments.
5488 if (NumOperands > 3)
5489 OrderArg = dyn_cast<ConstantInt>(CI->getArgOperand(2));
5490
5491 // Ignore scope argument at 3
5492
5493 if (NumOperands > 5) {
5494 ConstantInt *VolatileArg = dyn_cast<ConstantInt>(CI->getArgOperand(4));
5495 IsVolatile = !VolatileArg || !VolatileArg->isZero();
5496 }
5497
5499 if (OrderArg && isValidAtomicOrdering(OrderArg->getZExtValue()))
5500 Order = static_cast<AtomicOrdering>(OrderArg->getZExtValue());
5503
5504 LLVMContext &Ctx = F->getContext();
5505
5506 // Handle the v2bf16 intrinsic which used <2 x i16> instead of <2 x bfloat>
5507 Type *RetTy = CI->getType();
5508 if (VectorType *VT = dyn_cast<VectorType>(RetTy)) {
5509 if (VT->getElementType()->isIntegerTy(16)) {
5510 VectorType *AsBF16 =
5511 VectorType::get(Type::getBFloatTy(Ctx), VT->getElementCount());
5512 Val = Builder.CreateBitCast(Val, AsBF16);
5513 }
5514 }
5515
5516 // The scope argument never really worked correctly. Use agent as the most
5517 // conservative option which should still always produce the instruction.
5518 SyncScope::ID SSID = Ctx.getOrInsertSyncScopeID("agent");
5519 AtomicRMWInst *RMW =
5520 Builder.CreateAtomicRMW(RMWOp, Ptr, Val, std::nullopt, Order, SSID);
5521
5522 unsigned AddrSpace = PtrTy->getAddressSpace();
5523 if (AddrSpace != AMDGPUAS::LOCAL_ADDRESS) {
5524 MDNode *EmptyMD = MDNode::get(F->getContext(), {});
5525 RMW->setMetadata("amdgpu.no.fine.grained.memory", EmptyMD);
5526 if (RMWOp == AtomicRMWInst::FAdd && RetTy->isFloatTy())
5527 RMW->setMetadata(LLVMContext::MD_atomic_ignore_denormal_mode, EmptyMD);
5528 }
5529
5530 if (AddrSpace == AMDGPUAS::FLAT_ADDRESS) {
5531 MDBuilder MDB(F->getContext());
5532 MDNode *RangeNotPrivate =
5535 RMW->setMetadata(LLVMContext::MD_noalias_addrspace, RangeNotPrivate);
5536 }
5537
5538 if (IsVolatile)
5539 RMW->setVolatile(true);
5540
5541 return Builder.CreateBitCast(RMW, RetTy);
5542}
5543
5544/// Helper to unwrap intrinsic call MetadataAsValue operands. Return as a
5545/// plain MDNode, as it's the verifier's job to check these are the correct
5546/// types later.
5547static MDNode *unwrapMAVOp(CallBase *CI, unsigned Op) {
5548 if (Op < CI->arg_size()) {
5549 if (MetadataAsValue *MAV =
5551 Metadata *MD = MAV->getMetadata();
5552 return dyn_cast_if_present<MDNode>(MD);
5553 }
5554 }
5555 return nullptr;
5556}
5557
5558/// Helper to unwrap Metadata MetadataAsValue operands, such as the Value field.
5559static Metadata *unwrapMAVMetadataOp(CallBase *CI, unsigned Op) {
5560 if (Op < CI->arg_size())
5562 return MAV->getMetadata();
5563 return nullptr;
5564}
5565
5566/// Convert debug intrinsic calls to non-instruction debug records.
5567/// \p Name - Final part of the intrinsic name, e.g. 'value' in llvm.dbg.value.
5568/// \p CI - The debug intrinsic call.
5570 DbgRecord *DR = nullptr;
5571 if (Name == "label") {
5573 } else if (Name == "assign") {
5576 unwrapMAVOp(CI, 1), unwrapMAVOp(CI, 2), unwrapMAVOp(CI, 3),
5577 unwrapMAVMetadataOp(CI, 4),
5578 /*The address is a Value ref, it will be stored as a Metadata */
5579 unwrapMAVOp(CI, 5));
5580 } else if (Name == "declare") {
5583 unwrapMAVOp(CI, 1), unwrapMAVOp(CI, 2), nullptr, nullptr, nullptr);
5584 } else if (Name == "addr") {
5585 // Upgrade dbg.addr to dbg.value with DW_OP_deref.
5586 MDNode *ExprNode = unwrapMAVOp(CI, 2);
5587 // Don't try to add something to the expression if it's not an expression.
5588 // Instead, allow the verifier to fail later.
5589 if (DIExpression *Expr = dyn_cast<DIExpression>(ExprNode)) {
5590 ExprNode = DIExpression::append(Expr, dwarf::DW_OP_deref);
5591 }
5594 unwrapMAVOp(CI, 1), ExprNode, nullptr, nullptr, nullptr);
5595 } else if (Name == "value") {
5596 // An old version of dbg.value had an extra offset argument.
5597 unsigned VarOp = 1;
5598 unsigned ExprOp = 2;
5599 if (CI->arg_size() == 4) {
5601 // Nonzero offset dbg.values get dropped without a replacement.
5602 if (!Offset || !Offset->isNullValue())
5603 return;
5604 VarOp = 2;
5605 ExprOp = 3;
5606 }
5609 unwrapMAVOp(CI, VarOp), unwrapMAVOp(CI, ExprOp), nullptr, nullptr,
5610 nullptr);
5611 }
5612 DR->setDebugLoc(CI->getDebugLoc());
5613 assert(DR && "Unhandled intrinsic kind in upgrade to DbgRecord");
5614 CI->getParent()->insertDbgRecordBefore(DR, CI->getIterator());
5615}
5616
5619 if (!Offset)
5620 reportFatalUsageError("Invalid llvm.vector.splice offset argument");
5621 int64_t OffsetVal = Offset->getSExtValue();
5622 return Builder.CreateIntrinsic(OffsetVal >= 0
5623 ? Intrinsic::vector_splice_left
5624 : Intrinsic::vector_splice_right,
5625 CI->getType(),
5626 {CI->getArgOperand(0), CI->getArgOperand(1),
5627 Builder.getInt32(std::abs(OffsetVal))});
5628}
5629
5631 Function *F, IRBuilder<> &Builder) {
5632 if (Name.starts_with("to.fp16")) {
5633 Value *Cast =
5634 Builder.CreateFPTrunc(CI->getArgOperand(0), Builder.getHalfTy());
5635 return Builder.CreateBitCast(Cast, CI->getType());
5636 }
5637
5638 if (Name.starts_with("from.fp16")) {
5639 Value *Cast =
5640 Builder.CreateBitCast(CI->getArgOperand(0), Builder.getHalfTy());
5641 return Builder.CreateFPExt(Cast, CI->getType());
5642 }
5643
5644 return nullptr;
5645}
5646
5648 Metadata *MD = cast<MetadataAsValue>(Op)->getMetadata();
5649 if (!MD || !isa<MDString>(MD))
5651 return StringSwitch<ICmpInst::Predicate>(cast<MDString>(MD)->getString())
5652 .Case("eq", ICmpInst::ICMP_EQ)
5653 .Case("ne", ICmpInst::ICMP_NE)
5654 .Case("ugt", ICmpInst::ICMP_UGT)
5655 .Case("uge", ICmpInst::ICMP_UGE)
5656 .Case("ult", ICmpInst::ICMP_ULT)
5657 .Case("ule", ICmpInst::ICMP_ULE)
5658 .Case("sgt", ICmpInst::ICMP_SGT)
5659 .Case("sge", ICmpInst::ICMP_SGE)
5660 .Case("slt", ICmpInst::ICMP_SLT)
5661 .Case("sle", ICmpInst::ICMP_SLE)
5663}
5664
5666 Metadata *MD = cast<MetadataAsValue>(Op)->getMetadata();
5667 if (!MD || !isa<MDString>(MD))
5669 return StringSwitch<FCmpInst::Predicate>(cast<MDString>(MD)->getString())
5670 .Case("oeq", FCmpInst::FCMP_OEQ)
5671 .Case("ogt", FCmpInst::FCMP_OGT)
5672 .Case("oge", FCmpInst::FCMP_OGE)
5673 .Case("olt", FCmpInst::FCMP_OLT)
5674 .Case("ole", FCmpInst::FCMP_OLE)
5675 .Case("one", FCmpInst::FCMP_ONE)
5676 .Case("ord", FCmpInst::FCMP_ORD)
5677 .Case("uno", FCmpInst::FCMP_UNO)
5678 .Case("ueq", FCmpInst::FCMP_UEQ)
5679 .Case("ugt", FCmpInst::FCMP_UGT)
5680 .Case("uge", FCmpInst::FCMP_UGE)
5681 .Case("ult", FCmpInst::FCMP_ULT)
5682 .Case("ule", FCmpInst::FCMP_ULE)
5683 .Case("une", FCmpInst::FCMP_UNE)
5685}
5686
5688 IRBuilder<> &Builder) {
5689 Value *Rep;
5690 unsigned Opcode = getFunctionalOpcodeForVP(Name);
5691 if (Opcode && Instruction::isUnaryOp(Opcode))
5692 Rep =
5693 Builder.CreateUnOp((Instruction::UnaryOps)Opcode, CI->getArgOperand(0));
5694 else if (Opcode && Instruction::isBinaryOp(Opcode))
5695 Rep = Builder.CreateBinOp((Instruction::BinaryOps)Opcode,
5696 CI->getArgOperand(0), CI->getArgOperand(1));
5697 else if (Opcode && Instruction::isCast(Opcode))
5698 Rep = Builder.CreateCast((Instruction::CastOps)Opcode, CI->getArgOperand(0),
5699 CI->getType());
5700 else if (Opcode == Instruction::ICmp)
5701 Rep = Builder.CreateICmp(getVPIntPredicateFromMD(CI->getArgOperand(2)),
5702 CI->getArgOperand(0), CI->getArgOperand(1));
5703 else if (Opcode == Instruction::FCmp)
5704 Rep = Builder.CreateFCmp(getVPFPPredicateFromMD(CI->getArgOperand(2)),
5705 CI->getArgOperand(0), CI->getArgOperand(1));
5706 else if (Opcode == Instruction::Select)
5707 Rep = Builder.CreateSelect(CI->getArgOperand(0), CI->getArgOperand(1),
5708 CI->getArgOperand(2));
5709 else if (auto IntrinsicID = getFunctionalIntrinsicIDForVP(Name)) {
5710 SmallVector<Value *, 2> Args(drop_end(CI->args(), 2));
5711 Rep = Builder.CreateIntrinsic(CI->getType(), IntrinsicID, Args, {});
5712 } else
5713 llvm_unreachable("Unexpected vp intrinsic");
5714 Rep->takeName(CI);
5715 return Rep;
5716}
5717
5719 IRBuilder<> &Builder) {
5720 Intrinsic::ID IID = NewFn->getIntrinsicID();
5721
5722 auto [FirstDefault, Defaults] = Intrinsic::getAllDefaultArgValues(IID);
5723 if (Defaults.empty())
5724 return false;
5725
5726 unsigned OldArgCount = CI->arg_size();
5727 unsigned NewArgCount = NewFn->arg_size();
5728
5729 if (OldArgCount < FirstDefault)
5730 return false;
5731
5732 // More arguments than the new intrinsic accepts, cannot upgrade.
5733 if (OldArgCount > NewArgCount)
5734 return false;
5735
5736 // The call already passes the defaulted arguments explicitly; only the
5737 // callee is still the old, shorter declaration, so retarget it.
5738 if (OldArgCount == NewArgCount) {
5739 if (CI->getFunctionType() != NewFn->getFunctionType())
5740 return false;
5741 CI->setCalledFunction(NewFn);
5742 return true;
5743 }
5744
5745 // OldArgCount < NewArgCount: Fill in each missing trailing default
5746 // argument from the table.
5747 SmallVector<Value *, 8> NewArgs(CI->args());
5748
5749 FunctionType *NewFT = NewFn->getFunctionType();
5750 for (unsigned Idx = OldArgCount; Idx < NewArgCount; ++Idx) {
5751 assert(Idx >= FirstDefault && Idx - FirstDefault < Defaults.size() &&
5752 "missing argument outside the default range");
5753 Type *ParamTy = NewFT->getParamType(Idx);
5754
5755 // Only integer types are supported (i1, i8, i16, i32, i64).
5756 if (!ParamTy->isIntegerTy())
5757 return false;
5758 NewArgs.push_back(ConstantInt::get(ParamTy, Defaults[Idx - FirstDefault]));
5759 }
5760
5761 // Preserve operand bundles by creating the call with them.
5763 CI->getOperandBundlesAsDefs(OpBundles);
5764 CallInst *NewCall = Builder.CreateCall(NewFn, NewArgs, OpBundles);
5765
5766 NewCall->takeName(CI);
5767 NewCall->setCallingConv(CI->getCallingConv());
5768 NewCall->copyMetadata(*CI);
5769 if (auto *OldCI = dyn_cast<CallInst>(CI))
5770 NewCall->setTailCallKind(OldCI->getTailCallKind());
5771
5772 CI->replaceAllUsesWith(NewCall);
5773 CI->eraseFromParent();
5774 return true;
5775}
5776
5777/// Upgrade a call to an old intrinsic. All argument and return casting must be
5778/// provided to seamlessly integrate with existing context.
5780 // Note dyn_cast to Function is not quite the same as getCalledFunction, which
5781 // checks the callee's function type matches. It's likely we need to handle
5782 // type changes here.
5784 if (!F)
5785 return;
5786
5787 LLVMContext &C = CI->getContext();
5788 IRBuilder<> Builder(CI->getIterator());
5789 if (isa<FPMathOperator>(CI))
5790 Builder.setFastMathFlags(CI->getFastMathFlags());
5791
5792 if (!NewFn) {
5793 // Get the Function's name.
5794 StringRef Name = F->getName();
5795 if (!Name.consume_front("llvm."))
5796 llvm_unreachable("intrinsic doesn't start with 'llvm.'");
5797
5798 bool IsX86 = Name.consume_front("x86.");
5799 bool IsNVVM = Name.consume_front("nvvm.");
5800 bool IsAArch64 = Name.consume_front("aarch64.");
5801 bool IsARM = Name.consume_front("arm.");
5802 bool IsAMDGCN = Name.consume_front("amdgcn.");
5803 bool IsDbg = Name.consume_front("dbg.");
5804 bool IsOldSplice =
5805 (Name.consume_front("experimental.vector.splice") ||
5806 Name.consume_front("vector.splice")) &&
5807 !(Name.starts_with(".left") || Name.starts_with(".right"));
5808 Value *Rep = nullptr;
5809
5810 if (!IsX86 && Name == "stackprotectorcheck") {
5811 Rep = nullptr;
5812 } else if (IsNVVM) {
5813 Rep = upgradeNVVMIntrinsicCall(Name, CI, F, Builder);
5814 } else if (IsX86) {
5815 Rep = upgradeX86IntrinsicCall(Name, CI, F, Builder);
5816 } else if (IsAArch64) {
5817 Rep = upgradeAArch64IntrinsicCall(Name, CI, F, Builder);
5818 } else if (IsARM) {
5819 Rep = upgradeARMIntrinsicCall(Name, CI, F, Builder);
5820 } else if (IsAMDGCN) {
5821 Rep = upgradeAMDGCNIntrinsicCall(Name, CI, F, Builder);
5822 } else if (IsDbg) {
5824 } else if (IsOldSplice) {
5825 Rep = upgradeVectorSplice(CI, Builder);
5826 } else if (Name.consume_front("convert.")) {
5827 Rep = upgradeConvertIntrinsicCall(Name, CI, F, Builder);
5828 } else if (Name == "lifetime.start.i64" || Name == "lifetime.end.i64") {
5829 // Delete calls to invalid @llvm.lifetime.{start,end}.i64 intrinsics.
5830 Rep = nullptr;
5831 } else if (shouldUpgradeVPIntrinsic(Name)) {
5832 Rep = upgradeVPIntrinsicCall(Name, CI, Builder);
5833 } else {
5834 llvm_unreachable("Unknown function for CallBase upgrade.");
5835 }
5836
5837 if (Rep)
5838 CI->replaceAllUsesWith(Rep);
5839 CI->eraseFromParent();
5840 return;
5841 }
5842
5843 const auto &DefaultCase = [&]() -> void {
5844 if (F == NewFn)
5845 return;
5846
5847 if (CI->getFunctionType() == NewFn->getFunctionType()) {
5848 // Handle generic mangling change.
5849 assert(
5850 (CI->getCalledFunction()->getName() != NewFn->getName()) &&
5851 "Unknown function for CallBase upgrade and isn't just a name change");
5852 CI->setCalledFunction(NewFn);
5853 return;
5854 }
5855
5856 // This must be an upgrade from a named to a literal struct.
5857 if (auto *OldST = dyn_cast<StructType>(CI->getType())) {
5858 assert(OldST != NewFn->getReturnType() &&
5859 "Return type must have changed");
5860 assert(OldST->getNumElements() ==
5861 cast<StructType>(NewFn->getReturnType())->getNumElements() &&
5862 "Must have same number of elements");
5863
5864 SmallVector<Value *> Args(CI->args());
5865 CallInst *NewCI = Builder.CreateCall(NewFn, Args);
5866 NewCI->setAttributes(CI->getAttributes());
5867 Value *Res = PoisonValue::get(OldST);
5868 for (unsigned Idx = 0; Idx < OldST->getNumElements(); ++Idx) {
5869 Value *Elem = Builder.CreateExtractValue(NewCI, Idx);
5870 Res = Builder.CreateInsertValue(Res, Elem, Idx);
5871 }
5872 CI->replaceAllUsesWith(Res);
5873 CI->eraseFromParent();
5874 return;
5875 }
5876
5877 // We're probably about to produce something invalid. Let the verifier catch
5878 // it instead of dying here.
5879 CI->setCalledOperand(
5881 return;
5882 };
5883 CallInst *NewCall = nullptr;
5884 switch (NewFn->getIntrinsicID()) {
5885 default: {
5886 if (upgradeIntrinsicCallWithDefaultArgs(CI, NewFn, Builder))
5887 return;
5888 DefaultCase();
5889 return;
5890 }
5891 case Intrinsic::arm_neon_vst1:
5892 case Intrinsic::arm_neon_vst2:
5893 case Intrinsic::arm_neon_vst3:
5894 case Intrinsic::arm_neon_vst4:
5895 case Intrinsic::arm_neon_vst2lane:
5896 case Intrinsic::arm_neon_vst3lane:
5897 case Intrinsic::arm_neon_vst4lane: {
5898 SmallVector<Value *, 4> Args(CI->args());
5899 NewCall = Builder.CreateCall(NewFn, Args);
5900 break;
5901 }
5902 case Intrinsic::aarch64_sve_bfmlalb_lane_v2:
5903 case Intrinsic::aarch64_sve_bfmlalt_lane_v2:
5904 case Intrinsic::aarch64_sve_bfdot_lane_v2: {
5905 LLVMContext &Ctx = F->getParent()->getContext();
5906 SmallVector<Value *, 4> Args(CI->args());
5907 Args[3] = ConstantInt::get(Type::getInt32Ty(Ctx),
5908 cast<ConstantInt>(Args[3])->getZExtValue());
5909 NewCall = Builder.CreateCall(NewFn, Args);
5910 break;
5911 }
5912 case Intrinsic::aarch64_sve_ld3_sret:
5913 case Intrinsic::aarch64_sve_ld4_sret:
5914 case Intrinsic::aarch64_sve_ld2_sret: {
5915 // Is this a trivial remangle of the name to support ptr address spaces?
5916 if (isa<StructType>(F->getReturnType())) {
5917 DefaultCase();
5918 return;
5919 }
5920
5921 StringRef Name = F->getName();
5922 Name = Name.substr(5);
5923 unsigned N = StringSwitch<unsigned>(Name)
5924 .StartsWith("aarch64.sve.ld2", 2)
5925 .StartsWith("aarch64.sve.ld3", 3)
5926 .StartsWith("aarch64.sve.ld4", 4)
5927 .Default(0);
5928 auto *RetTy = cast<ScalableVectorType>(F->getReturnType());
5929 unsigned MinElts = RetTy->getMinNumElements() / N;
5930 SmallVector<Value *, 2> Args(CI->args());
5931 Value *NewLdCall = Builder.CreateCall(NewFn, Args);
5932 Value *Ret = llvm::PoisonValue::get(RetTy);
5933 for (unsigned I = 0; I < N; I++) {
5934 Value *SRet = Builder.CreateExtractValue(NewLdCall, I);
5935 Ret = Builder.CreateInsertVector(RetTy, Ret, SRet, I * MinElts);
5936 }
5937 NewCall = dyn_cast<CallInst>(Ret);
5938 break;
5939 }
5940
5941 case Intrinsic::coro_end_async:
5942 case Intrinsic::coro_end: {
5943 SmallVector<Value *, 3> Args(CI->args());
5944 if (NewFn->getIntrinsicID() == Intrinsic::coro_end && Args.size() == 2)
5945 Args.push_back(ConstantTokenNone::get(CI->getContext()));
5946 NewCall = Builder.CreateCall(NewFn, Args);
5947
5948 if (!CI->getType()->isVoidTy()) {
5949 if (!CI->use_empty()) {
5951 CI->getModule(), Intrinsic::coro_is_in_ramp);
5952 Value *InRamp = Builder.CreateCall(IsInRamp);
5953 CI->replaceAllUsesWith(Builder.CreateNot(InRamp));
5954 }
5955 CI->eraseFromParent();
5956 return;
5957 }
5958
5959 break;
5960 }
5961
5962 case Intrinsic::vector_extract: {
5963 StringRef Name = F->getName();
5964 Name = Name.substr(5); // Strip llvm
5965 if (!Name.starts_with("aarch64.sve.tuple.get")) {
5966 DefaultCase();
5967 return;
5968 }
5969 auto *RetTy = cast<ScalableVectorType>(F->getReturnType());
5970 unsigned MinElts = RetTy->getMinNumElements();
5971 uint64_t I = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
5972 Value *NewIdx = ConstantInt::get(Type::getInt64Ty(C), I * MinElts);
5973 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0), NewIdx});
5974 break;
5975 }
5976
5977 case Intrinsic::vector_insert: {
5978 StringRef Name = F->getName();
5979 Name = Name.substr(5);
5980 if (!Name.starts_with("aarch64.sve.tuple")) {
5981 DefaultCase();
5982 return;
5983 }
5984 if (Name.starts_with("aarch64.sve.tuple.set")) {
5985 uint64_t I = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
5986 auto *Ty = cast<ScalableVectorType>(CI->getArgOperand(2)->getType());
5987 Value *NewIdx =
5988 ConstantInt::get(Type::getInt64Ty(C), I * Ty->getMinNumElements());
5989 NewCall = Builder.CreateCall(
5990 NewFn, {CI->getArgOperand(0), CI->getArgOperand(2), NewIdx});
5991 break;
5992 }
5993 if (Name.starts_with("aarch64.sve.tuple.create")) {
5994 unsigned N = StringSwitch<unsigned>(Name)
5995 .StartsWith("aarch64.sve.tuple.create2", 2)
5996 .StartsWith("aarch64.sve.tuple.create3", 3)
5997 .StartsWith("aarch64.sve.tuple.create4", 4)
5998 .Default(0);
5999 assert(N > 1 && "Create is expected to be between 2-4");
6000 auto *RetTy = cast<ScalableVectorType>(F->getReturnType());
6001 Value *Ret = llvm::PoisonValue::get(RetTy);
6002 unsigned MinElts = RetTy->getMinNumElements() / N;
6003 for (unsigned I = 0; I < N; I++) {
6004 Value *V = CI->getArgOperand(I);
6005 Ret = Builder.CreateInsertVector(RetTy, Ret, V, I * MinElts);
6006 }
6007 NewCall = dyn_cast<CallInst>(Ret);
6008 }
6009 break;
6010 }
6011
6012 case Intrinsic::arm_neon_bfdot:
6013 case Intrinsic::arm_neon_bfmmla:
6014 case Intrinsic::arm_neon_bfmlalb:
6015 case Intrinsic::arm_neon_bfmlalt:
6016 case Intrinsic::aarch64_neon_bfdot:
6017 case Intrinsic::aarch64_neon_bfmmla:
6018 case Intrinsic::aarch64_neon_bfmlalb:
6019 case Intrinsic::aarch64_neon_bfmlalt: {
6021 assert(CI->arg_size() == 3 &&
6022 "Mismatch between function args and call args");
6023 size_t OperandWidth =
6025 assert((OperandWidth == 64 || OperandWidth == 128) &&
6026 "Unexpected operand width");
6027 Type *NewTy = FixedVectorType::get(Type::getBFloatTy(C), OperandWidth / 16);
6028 auto Iter = CI->args().begin();
6029 Args.push_back(*Iter++);
6030 Args.push_back(Builder.CreateBitCast(*Iter++, NewTy));
6031 Args.push_back(Builder.CreateBitCast(*Iter++, NewTy));
6032 NewCall = Builder.CreateCall(NewFn, Args);
6033 break;
6034 }
6035
6036 case Intrinsic::bitreverse:
6037 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)});
6038 break;
6039
6040 case Intrinsic::ctlz:
6041 case Intrinsic::cttz: {
6042 if (CI->arg_size() != 1) {
6043 DefaultCase();
6044 return;
6045 }
6046
6047 NewCall =
6048 Builder.CreateCall(NewFn, {CI->getArgOperand(0), Builder.getFalse()});
6049 break;
6050 }
6051
6052 case Intrinsic::objectsize: {
6053 Value *NullIsUnknownSize =
6054 CI->arg_size() == 2 ? Builder.getFalse() : CI->getArgOperand(2);
6055 Value *Dynamic =
6056 CI->arg_size() < 4 ? Builder.getFalse() : CI->getArgOperand(3);
6057 NewCall = Builder.CreateCall(
6058 NewFn, {CI->getArgOperand(0), CI->getArgOperand(1), NullIsUnknownSize, Dynamic});
6059 break;
6060 }
6061
6062 case Intrinsic::ctpop:
6063 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)});
6064 break;
6065 case Intrinsic::dbg_value: {
6066 StringRef Name = F->getName();
6067 Name = Name.substr(5); // Strip llvm.
6068 // Upgrade `dbg.addr` to `dbg.value` with `DW_OP_deref`.
6069 if (Name.starts_with("dbg.addr")) {
6071 cast<MetadataAsValue>(CI->getArgOperand(2))->getMetadata());
6072 Expr = DIExpression::append(Expr, dwarf::DW_OP_deref);
6073 NewCall =
6074 Builder.CreateCall(NewFn, {CI->getArgOperand(0), CI->getArgOperand(1),
6075 MetadataAsValue::get(C, Expr)});
6076 break;
6077 }
6078
6079 // Upgrade from the old version that had an extra offset argument.
6080 assert(CI->arg_size() == 4);
6081 // Drop nonzero offsets instead of attempting to upgrade them.
6083 if (Offset->isNullValue()) {
6084 NewCall = Builder.CreateCall(
6085 NewFn,
6086 {CI->getArgOperand(0), CI->getArgOperand(2), CI->getArgOperand(3)});
6087 break;
6088 }
6089 CI->eraseFromParent();
6090 return;
6091 }
6092
6093 case Intrinsic::ptr_annotation:
6094 // Upgrade from versions that lacked the annotation attribute argument.
6095 if (CI->arg_size() != 4) {
6096 DefaultCase();
6097 return;
6098 }
6099
6100 // Create a new call with an added null annotation attribute argument.
6101 NewCall = Builder.CreateCall(
6102 NewFn,
6103 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2),
6104 CI->getArgOperand(3), ConstantPointerNull::get(Builder.getPtrTy())});
6105 NewCall->takeName(CI);
6106 CI->replaceAllUsesWith(NewCall);
6107 CI->eraseFromParent();
6108 return;
6109
6110 case Intrinsic::var_annotation:
6111 // Upgrade from versions that lacked the annotation attribute argument.
6112 if (CI->arg_size() != 4) {
6113 DefaultCase();
6114 return;
6115 }
6116 // Create a new call with an added null annotation attribute argument.
6117 NewCall = Builder.CreateCall(
6118 NewFn,
6119 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2),
6120 CI->getArgOperand(3), ConstantPointerNull::get(Builder.getPtrTy())});
6121 NewCall->takeName(CI);
6122 CI->replaceAllUsesWith(NewCall);
6123 CI->eraseFromParent();
6124 return;
6125
6126 case Intrinsic::riscv_aes32dsi:
6127 case Intrinsic::riscv_aes32dsmi:
6128 case Intrinsic::riscv_aes32esi:
6129 case Intrinsic::riscv_aes32esmi:
6130 case Intrinsic::riscv_sm4ks:
6131 case Intrinsic::riscv_sm4ed: {
6132 // The last argument to these intrinsics used to be i8 and changed to i32.
6133 // The type overload for sm4ks and sm4ed was removed.
6134 Value *Arg2 = CI->getArgOperand(2);
6135 if (Arg2->getType()->isIntegerTy(32) && !CI->getType()->isIntegerTy(64))
6136 return;
6137
6138 Value *Arg0 = CI->getArgOperand(0);
6139 Value *Arg1 = CI->getArgOperand(1);
6140 if (CI->getType()->isIntegerTy(64)) {
6141 Arg0 = Builder.CreateTrunc(Arg0, Builder.getInt32Ty());
6142 Arg1 = Builder.CreateTrunc(Arg1, Builder.getInt32Ty());
6143 }
6144
6145 Arg2 = ConstantInt::get(Type::getInt32Ty(C),
6146 cast<ConstantInt>(Arg2)->getZExtValue());
6147
6148 NewCall = Builder.CreateCall(NewFn, {Arg0, Arg1, Arg2});
6149 Value *Res = NewCall;
6150 if (Res->getType() != CI->getType())
6151 Res = Builder.CreateIntCast(NewCall, CI->getType(), /*isSigned*/ true);
6152 NewCall->takeName(CI);
6153 CI->replaceAllUsesWith(Res);
6154 CI->eraseFromParent();
6155 return;
6156 }
6157 case Intrinsic::nvvm_mapa_shared_cluster: {
6158 // Create a new call with the correct address space.
6159 NewCall =
6160 Builder.CreateCall(NewFn, {CI->getArgOperand(0), CI->getArgOperand(1)});
6161 Value *Res = NewCall;
6162 Res = Builder.CreateAddrSpaceCast(
6163 Res, Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED));
6164 NewCall->takeName(CI);
6165 CI->replaceAllUsesWith(Res);
6166 CI->eraseFromParent();
6167 return;
6168 }
6169 case Intrinsic::nvvm_cp_async_bulk_global_to_shared_cluster: {
6170 SmallVector<Value *, 4> Args(CI->args());
6171 unsigned AS = Args[0]->getType()->getPointerAddressSpace();
6173 Args[0] = Builder.CreateAddrSpaceCast(
6174 Args[0], Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6175
6176 // Append the missing trailing flag_valid_pattern (0 = disabled).
6177 Args.push_back(Builder.getInt32(0));
6178
6179 NewCall = Builder.CreateCall(NewFn, Args);
6180 NewCall->takeName(CI);
6181 CI->replaceAllUsesWith(NewCall);
6182 CI->eraseFromParent();
6183 return;
6184 }
6185 case Intrinsic::nvvm_cp_async_bulk_global_to_shared_cta: {
6186 // (dst, mbar, src, size, ch, flag_ch)
6187 // -> (dst, mbar, src, size, i32 0, i32 0, ch, flag_ch, i1 false,
6188 // i32 0 /* flag_valid_pattern=disabled */)
6190 for (unsigned I = 0; I < 4; ++I)
6191 Args.push_back(CI->getArgOperand(I));
6192 Args.push_back(Builder.getInt32(0)); // ignore_bytes_left
6193 Args.push_back(Builder.getInt32(0)); // ignore_bytes_right
6194 Args.push_back(CI->getArgOperand(4)); // cache_hint
6195 Args.push_back(CI->getArgOperand(5)); // flag_ch
6196 Args.push_back(Builder.getInt1(false)); // flag_oob
6197 Args.push_back(Builder.getInt32(0)); // flag_valid_pattern
6198
6199 NewCall = Builder.CreateCall(NewFn, Args);
6200 NewCall->takeName(CI);
6201 CI->replaceAllUsesWith(NewCall);
6202 CI->eraseFromParent();
6203 return;
6204 }
6205 case Intrinsic::nvvm_cp_async_bulk_shared_cta_to_cluster: {
6206 // Create a new call with the correct address space.
6207 SmallVector<Value *, 4> Args(CI->args());
6208 Args[0] = Builder.CreateAddrSpaceCast(
6209 Args[0], Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6210
6211 NewCall = Builder.CreateCall(NewFn, Args);
6212 NewCall->takeName(CI);
6213 CI->replaceAllUsesWith(NewCall);
6214 CI->eraseFromParent();
6215 return;
6216 }
6217 // clang-format off
6218#define G2S_CLUSTER_CASE(ID_SUFFIX, NAME) \
6219 case Intrinsic::nvvm_cp_async_bulk_tensor_g2s_##ID_SUFFIX:
6221#undef G2S_CLUSTER_CASE
6222 {
6223 SmallVector<Value *, 16> Args(CI->args());
6224 unsigned AS = CI->getArgOperand(0)->getType()->getPointerAddressSpace();
6226 Args[0] = Builder.CreateAddrSpaceCast(
6227 Args[0], Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6228
6229 // Append the missing trailing arguments with default values (cta_group,
6230 // flag_valid_pattern).
6231 while (Args.size() < NewFn->getFunctionType()->getNumParams())
6232 Args.push_back(Builder.getInt32(0));
6233
6234 NewCall = Builder.CreateCall(NewFn, Args);
6235 NewCall->takeName(CI);
6236 CI->replaceAllUsesWith(NewCall);
6237 CI->eraseFromParent();
6238 return;
6239 }
6240
6241#define G2S_CTA_CASE(ID_SUFFIX, NAME) \
6242 case Intrinsic::nvvm_cp_async_bulk_tensor_g2s_cta_##ID_SUFFIX:
6244#undef G2S_CTA_CASE
6245 {
6246 SmallVector<Value *, 16> Args(CI->args());
6247 // Append the missing trailing flag_valid_pattern argument with default
6248 // value 0.
6249 assert(Args.size() + 1 == NewFn->getFunctionType()->getNumParams() &&
6250 "expected only the trailing flag_valid_pattern to be missing");
6251 Args.push_back(Builder.getInt32(0));
6252
6253 NewCall = Builder.CreateCall(NewFn, Args);
6254 NewCall->takeName(CI);
6255 CI->replaceAllUsesWith(NewCall);
6256 CI->eraseFromParent();
6257 return;
6258 }
6259#undef NVVM_TMA_G2S_MODES
6260 // clang-format on
6261
6262 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_1d:
6263 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_2d:
6264 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_3d:
6265 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_4d:
6266 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_5d:
6267 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_3d:
6268 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_4d:
6269 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_5d: {
6270 StringRef Name = F->getName();
6271 Name.consume_front("llvm.nvvm.cp.async.bulk.tensor.reduce.");
6272 auto RedOp = getNVPTXTMAReductionOp(Name.split('.').first);
6273
6274 SmallVector<Value *, 16> Args(CI->args());
6275 Args.insert(Args.end() - 1, Builder.getInt32(*RedOp));
6276 NewCall = Builder.CreateCall(NewFn, Args);
6277 break;
6278 }
6279 case Intrinsic::nvvm_tcgen05_alloc_cg1:
6280 case Intrinsic::nvvm_tcgen05_alloc_cg2:
6281 case Intrinsic::nvvm_tcgen05_dealloc_cg1:
6282 case Intrinsic::nvvm_tcgen05_dealloc_cg2:
6283 NewCall =
6284 Builder.CreateCall(NewFn, {CI->getArgOperand(0), CI->getArgOperand(1),
6285 Builder.getFalse()});
6286 break;
6287 case Intrinsic::nvvm_mbarrier_init: {
6288 SmallVector<Value *, 3> Args(CI->args());
6289 // The .shared variant folded into the overloaded form without gaining an
6290 // operand, so only the pre-layout two-argument form needs one appended.
6291 if (Args.size() == 2)
6292 Args.push_back(Builder.getInt32(0)); // layout = default(0)
6293 NewCall = Builder.CreateCall(NewFn, Args);
6294 break;
6295 }
6296 case Intrinsic::riscv_sha256sig0:
6297 case Intrinsic::riscv_sha256sig1:
6298 case Intrinsic::riscv_sha256sum0:
6299 case Intrinsic::riscv_sha256sum1:
6300 case Intrinsic::riscv_sm3p0:
6301 case Intrinsic::riscv_sm3p1: {
6302 // The last argument to these intrinsics used to be i8 and changed to i32.
6303 // The type overload for sm4ks and sm4ed was removed.
6304 if (!CI->getType()->isIntegerTy(64))
6305 return;
6306
6307 Value *Arg =
6308 Builder.CreateTrunc(CI->getArgOperand(0), Builder.getInt32Ty());
6309
6310 NewCall = Builder.CreateCall(NewFn, Arg);
6311 Value *Res =
6312 Builder.CreateIntCast(NewCall, CI->getType(), /*isSigned*/ true);
6313 NewCall->takeName(CI);
6314 CI->replaceAllUsesWith(Res);
6315 CI->eraseFromParent();
6316 return;
6317 }
6318
6319 case Intrinsic::x86_xop_vfrcz_ss:
6320 case Intrinsic::x86_xop_vfrcz_sd:
6321 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(1)});
6322 break;
6323
6324 case Intrinsic::x86_xop_vpermil2pd:
6325 case Intrinsic::x86_xop_vpermil2ps:
6326 case Intrinsic::x86_xop_vpermil2pd_256:
6327 case Intrinsic::x86_xop_vpermil2ps_256: {
6328 SmallVector<Value *, 4> Args(CI->args());
6329 VectorType *FltIdxTy = cast<VectorType>(Args[2]->getType());
6330 VectorType *IntIdxTy = VectorType::getInteger(FltIdxTy);
6331 Args[2] = Builder.CreateBitCast(Args[2], IntIdxTy);
6332 NewCall = Builder.CreateCall(NewFn, Args);
6333 break;
6334 }
6335
6336 case Intrinsic::x86_sse41_ptestc:
6337 case Intrinsic::x86_sse41_ptestz:
6338 case Intrinsic::x86_sse41_ptestnzc: {
6339 // The arguments for these intrinsics used to be v4f32, and changed
6340 // to v2i64. This is purely a nop, since those are bitwise intrinsics.
6341 // So, the only thing required is a bitcast for both arguments.
6342 // First, check the arguments have the old type.
6343 Value *Arg0 = CI->getArgOperand(0);
6344 if (Arg0->getType() != FixedVectorType::get(Type::getFloatTy(C), 4))
6345 return;
6346
6347 // Old intrinsic, add bitcasts
6348 Value *Arg1 = CI->getArgOperand(1);
6349
6350 auto *NewVecTy = FixedVectorType::get(Type::getInt64Ty(C), 2);
6351
6352 Value *BC0 = Builder.CreateBitCast(Arg0, NewVecTy, "cast");
6353 Value *BC1 = Builder.CreateBitCast(Arg1, NewVecTy, "cast");
6354
6355 NewCall = Builder.CreateCall(NewFn, {BC0, BC1});
6356 break;
6357 }
6358
6359 case Intrinsic::x86_rdtscp: {
6360 // This used to take 1 arguments. If we have no arguments, it is already
6361 // upgraded.
6362 if (CI->getNumOperands() == 0)
6363 return;
6364
6365 NewCall = Builder.CreateCall(NewFn);
6366 // Extract the second result and store it.
6367 Value *Data = Builder.CreateExtractValue(NewCall, 1);
6368 Builder.CreateAlignedStore(Data, CI->getArgOperand(0), Align(1));
6369 // Replace the original call result with the first result of the new call.
6370 Value *TSC = Builder.CreateExtractValue(NewCall, 0);
6371
6372 NewCall->takeName(CI);
6373 CI->replaceAllUsesWith(TSC);
6374 CI->eraseFromParent();
6375 return;
6376 }
6377
6378 case Intrinsic::x86_sse41_insertps:
6379 case Intrinsic::x86_sse41_dppd:
6380 case Intrinsic::x86_sse41_dpps:
6381 case Intrinsic::x86_sse41_mpsadbw:
6382 case Intrinsic::x86_avx_dp_ps_256:
6383 case Intrinsic::x86_avx2_mpsadbw: {
6384 // Need to truncate the last argument from i32 to i8 -- this argument models
6385 // an inherently 8-bit immediate operand to these x86 instructions.
6386 SmallVector<Value *, 4> Args(CI->args());
6387
6388 // Replace the last argument with a trunc.
6389 Args.back() = Builder.CreateTrunc(Args.back(), Type::getInt8Ty(C), "trunc");
6390 NewCall = Builder.CreateCall(NewFn, Args);
6391 break;
6392 }
6393
6394 case Intrinsic::x86_avx512_mask_cmp_pd_128:
6395 case Intrinsic::x86_avx512_mask_cmp_pd_256:
6396 case Intrinsic::x86_avx512_mask_cmp_pd_512:
6397 case Intrinsic::x86_avx512_mask_cmp_ps_128:
6398 case Intrinsic::x86_avx512_mask_cmp_ps_256:
6399 case Intrinsic::x86_avx512_mask_cmp_ps_512: {
6400 SmallVector<Value *, 4> Args(CI->args());
6401 unsigned NumElts =
6402 cast<FixedVectorType>(Args[0]->getType())->getNumElements();
6403 Args[3] = getX86MaskVec(Builder, Args[3], NumElts);
6404
6405 NewCall = Builder.CreateCall(NewFn, Args);
6406 Value *Res = applyX86MaskOn1BitsVec(Builder, NewCall, nullptr);
6407
6408 NewCall->takeName(CI);
6409 CI->replaceAllUsesWith(Res);
6410 CI->eraseFromParent();
6411 return;
6412 }
6413
6414 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_128:
6415 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_256:
6416 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_512:
6417 case Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128:
6418 case Intrinsic::x86_avx512bf16_cvtneps2bf16_256:
6419 case Intrinsic::x86_avx512bf16_cvtneps2bf16_512: {
6420 SmallVector<Value *, 4> Args(CI->args());
6421 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
6422 if (NewFn->getIntrinsicID() ==
6423 Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128)
6424 Args[1] = Builder.CreateBitCast(
6425 Args[1], FixedVectorType::get(Builder.getBFloatTy(), NumElts));
6426
6427 NewCall = Builder.CreateCall(NewFn, Args);
6428 Value *Res = Builder.CreateBitCast(
6429 NewCall, FixedVectorType::get(Builder.getInt16Ty(), NumElts));
6430
6431 NewCall->takeName(CI);
6432 CI->replaceAllUsesWith(Res);
6433 CI->eraseFromParent();
6434 return;
6435 }
6436 case Intrinsic::x86_avx512bf16_dpbf16ps_128:
6437 case Intrinsic::x86_avx512bf16_dpbf16ps_256:
6438 case Intrinsic::x86_avx512bf16_dpbf16ps_512:{
6439 SmallVector<Value *, 4> Args(CI->args());
6440 unsigned NumElts =
6441 cast<FixedVectorType>(CI->getType())->getNumElements() * 2;
6442 Args[1] = Builder.CreateBitCast(
6443 Args[1], FixedVectorType::get(Builder.getBFloatTy(), NumElts));
6444 Args[2] = Builder.CreateBitCast(
6445 Args[2], FixedVectorType::get(Builder.getBFloatTy(), NumElts));
6446
6447 NewCall = Builder.CreateCall(NewFn, Args);
6448 break;
6449 }
6450
6451 case Intrinsic::thread_pointer: {
6452 NewCall = Builder.CreateCall(NewFn, {});
6453 break;
6454 }
6455
6456 case Intrinsic::memcpy:
6457 case Intrinsic::memmove:
6458 case Intrinsic::memset: {
6459 // We have to make sure that the call signature is what we're expecting.
6460 // We only want to change the old signatures by removing the alignment arg:
6461 // @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i32, i1)
6462 // -> @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i1)
6463 // @llvm.memset...(i8*, i8, i[32|64], i32, i1)
6464 // -> @llvm.memset...(i8*, i8, i[32|64], i1)
6465 // Note: i8*'s in the above can be any pointer type
6466 if (CI->arg_size() != 5) {
6467 DefaultCase();
6468 return;
6469 }
6470 // Remove alignment argument (3), and add alignment attributes to the
6471 // dest/src pointers.
6472 Value *Args[4] = {CI->getArgOperand(0), CI->getArgOperand(1),
6473 CI->getArgOperand(2), CI->getArgOperand(4)};
6474 NewCall = Builder.CreateCall(NewFn, Args);
6475 AttributeList OldAttrs = CI->getAttributes();
6476 AttributeList NewAttrs = AttributeList::get(
6477 C, OldAttrs.getFnAttrs(), OldAttrs.getRetAttrs(),
6478 {OldAttrs.getParamAttrs(0), OldAttrs.getParamAttrs(1),
6479 OldAttrs.getParamAttrs(2), OldAttrs.getParamAttrs(4)});
6480 NewCall->setAttributes(NewAttrs);
6481 auto *MemCI = cast<MemIntrinsic>(NewCall);
6482 // All mem intrinsics support dest alignment.
6484 MemCI->setDestAlignment(Align->getMaybeAlignValue());
6485 // Memcpy/Memmove also support source alignment.
6486 if (auto *MTI = dyn_cast<MemTransferInst>(MemCI))
6487 MTI->setSourceAlignment(Align->getMaybeAlignValue());
6488 break;
6489 }
6490
6491 case Intrinsic::masked_load:
6492 case Intrinsic::masked_gather:
6493 case Intrinsic::masked_store:
6494 case Intrinsic::masked_scatter: {
6495 if (CI->arg_size() != 4) {
6496 DefaultCase();
6497 return;
6498 }
6499
6500 auto GetMaybeAlign = [](Value *Op) {
6501 if (auto *CI = dyn_cast<ConstantInt>(Op)) {
6502 uint64_t Val = CI->getZExtValue();
6503 if (Val == 0)
6504 return MaybeAlign();
6505 if (isPowerOf2_64(Val))
6506 return MaybeAlign(Val);
6507 }
6508 reportFatalUsageError("Invalid alignment argument");
6509 };
6510 auto GetAlign = [&](Value *Op) {
6511 MaybeAlign Align = GetMaybeAlign(Op);
6512 if (Align)
6513 return *Align;
6514 reportFatalUsageError("Invalid zero alignment argument");
6515 };
6516
6517 const DataLayout &DL = CI->getDataLayout();
6518 switch (NewFn->getIntrinsicID()) {
6519 case Intrinsic::masked_load:
6520 NewCall = Builder.CreateMaskedLoad(
6521 CI->getType(), CI->getArgOperand(0), GetAlign(CI->getArgOperand(1)),
6522 CI->getArgOperand(2), CI->getArgOperand(3));
6523 break;
6524 case Intrinsic::masked_gather:
6525 NewCall = Builder.CreateMaskedGather(
6526 CI->getType(), CI->getArgOperand(0),
6527 DL.getValueOrABITypeAlignment(GetMaybeAlign(CI->getArgOperand(1)),
6528 CI->getType()->getScalarType()),
6529 CI->getArgOperand(2), CI->getArgOperand(3));
6530 break;
6531 case Intrinsic::masked_store:
6532 NewCall = Builder.CreateMaskedStore(
6533 CI->getArgOperand(0), CI->getArgOperand(1),
6534 GetAlign(CI->getArgOperand(2)), CI->getArgOperand(3));
6535 break;
6536 case Intrinsic::masked_scatter:
6537 NewCall = Builder.CreateMaskedScatter(
6538 CI->getArgOperand(0), CI->getArgOperand(1),
6539 DL.getValueOrABITypeAlignment(
6540 GetMaybeAlign(CI->getArgOperand(2)),
6541 CI->getArgOperand(0)->getType()->getScalarType()),
6542 CI->getArgOperand(3));
6543 break;
6544 default:
6545 llvm_unreachable("Unexpected intrinsic ID");
6546 }
6547 // Previous metadata is still valid.
6548 NewCall->copyMetadata(*CI);
6549 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
6550 break;
6551 }
6552
6553 case Intrinsic::lifetime_start:
6554 case Intrinsic::lifetime_end: {
6555 if (CI->arg_size() != 2) {
6556 DefaultCase();
6557 return;
6558 }
6559
6560 Value *Ptr = CI->getArgOperand(1);
6561 // Try to strip pointer casts, such that the lifetime works on an alloca.
6562 Ptr = Ptr->stripPointerCasts();
6563 if (isa<AllocaInst>(Ptr)) {
6564 // Don't use NewFn, as we might have looked through an addrspacecast.
6565 if (NewFn->getIntrinsicID() == Intrinsic::lifetime_start)
6566 NewCall = Builder.CreateLifetimeStart(Ptr);
6567 else
6568 NewCall = Builder.CreateLifetimeEnd(Ptr);
6569 break;
6570 }
6571
6572 // Otherwise remove the lifetime marker.
6573 CI->eraseFromParent();
6574 return;
6575 }
6576
6577 case Intrinsic::x86_avx512_vpdpbusd_128:
6578 case Intrinsic::x86_avx512_vpdpbusd_256:
6579 case Intrinsic::x86_avx512_vpdpbusd_512:
6580 case Intrinsic::x86_avx512_vpdpbusds_128:
6581 case Intrinsic::x86_avx512_vpdpbusds_256:
6582 case Intrinsic::x86_avx512_vpdpbusds_512:
6583 case Intrinsic::x86_avx2_vpdpbssd_128:
6584 case Intrinsic::x86_avx2_vpdpbssd_256:
6585 case Intrinsic::x86_avx10_vpdpbssd_512:
6586 case Intrinsic::x86_avx2_vpdpbssds_128:
6587 case Intrinsic::x86_avx2_vpdpbssds_256:
6588 case Intrinsic::x86_avx10_vpdpbssds_512:
6589 case Intrinsic::x86_avx2_vpdpbsud_128:
6590 case Intrinsic::x86_avx2_vpdpbsud_256:
6591 case Intrinsic::x86_avx10_vpdpbsud_512:
6592 case Intrinsic::x86_avx2_vpdpbsuds_128:
6593 case Intrinsic::x86_avx2_vpdpbsuds_256:
6594 case Intrinsic::x86_avx10_vpdpbsuds_512:
6595 case Intrinsic::x86_avx2_vpdpbuud_128:
6596 case Intrinsic::x86_avx2_vpdpbuud_256:
6597 case Intrinsic::x86_avx10_vpdpbuud_512:
6598 case Intrinsic::x86_avx2_vpdpbuuds_128:
6599 case Intrinsic::x86_avx2_vpdpbuuds_256:
6600 case Intrinsic::x86_avx10_vpdpbuuds_512: {
6601 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / 8;
6602 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
6603 CI->getArgOperand(2)};
6604 Type *NewArgType = VectorType::get(Builder.getInt8Ty(), NumElts, false);
6605 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
6606 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
6607
6608 NewCall = Builder.CreateCall(NewFn, Args);
6609 break;
6610 }
6611 case Intrinsic::x86_avx512_vpdpwssd_128:
6612 case Intrinsic::x86_avx512_vpdpwssd_256:
6613 case Intrinsic::x86_avx512_vpdpwssd_512:
6614 case Intrinsic::x86_avx512_vpdpwssds_128:
6615 case Intrinsic::x86_avx512_vpdpwssds_256:
6616 case Intrinsic::x86_avx512_vpdpwssds_512:
6617 case Intrinsic::x86_avx2_vpdpwsud_128:
6618 case Intrinsic::x86_avx2_vpdpwsud_256:
6619 case Intrinsic::x86_avx10_vpdpwsud_512:
6620 case Intrinsic::x86_avx2_vpdpwsuds_128:
6621 case Intrinsic::x86_avx2_vpdpwsuds_256:
6622 case Intrinsic::x86_avx10_vpdpwsuds_512:
6623 case Intrinsic::x86_avx2_vpdpwusd_128:
6624 case Intrinsic::x86_avx2_vpdpwusd_256:
6625 case Intrinsic::x86_avx10_vpdpwusd_512:
6626 case Intrinsic::x86_avx2_vpdpwusds_128:
6627 case Intrinsic::x86_avx2_vpdpwusds_256:
6628 case Intrinsic::x86_avx10_vpdpwusds_512:
6629 case Intrinsic::x86_avx2_vpdpwuud_128:
6630 case Intrinsic::x86_avx2_vpdpwuud_256:
6631 case Intrinsic::x86_avx10_vpdpwuud_512:
6632 case Intrinsic::x86_avx2_vpdpwuuds_128:
6633 case Intrinsic::x86_avx2_vpdpwuuds_256:
6634 case Intrinsic::x86_avx10_vpdpwuuds_512:
6635 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / 16;
6636 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
6637 CI->getArgOperand(2)};
6638 Type *NewArgType = VectorType::get(Builder.getInt16Ty(), NumElts, false);
6639 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
6640 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
6641
6642 NewCall = Builder.CreateCall(NewFn, Args);
6643 break;
6644 }
6645 assert(NewCall && "Should have either set this variable or returned through "
6646 "the default case");
6647 NewCall->takeName(CI);
6648 CI->replaceAllUsesWith(NewCall);
6649 CI->eraseFromParent();
6650}
6651
6653 assert(F && "Illegal attempt to upgrade a non-existent intrinsic.");
6654
6655 // Check if this function should be upgraded and get the replacement function
6656 // if there is one.
6657 Function *NewFn;
6658 if (UpgradeIntrinsicFunction(F, NewFn)) {
6659 // Replace all users of the old function with the new function or new
6660 // instructions. This is not a range loop because the call is deleted.
6661 for (User *U : make_early_inc_range(F->users()))
6662 if (CallBase *CB = dyn_cast<CallBase>(U))
6663 UpgradeIntrinsicCall(CB, NewFn);
6664
6665 // Remove old function, no longer used, from the module.
6666 if (F != NewFn)
6667 F->eraseFromParent();
6668 }
6669}
6670
6672 const unsigned NumOperands = MD.getNumOperands();
6673 if (NumOperands == 0)
6674 return &MD; // Invalid, punt to a verifier error.
6675
6676 // Check if the tag uses struct-path aware TBAA format.
6677 if (isa<MDNode>(MD.getOperand(0)) && NumOperands >= 3)
6678 return &MD;
6679
6680 auto &Context = MD.getContext();
6681 if (NumOperands == 3) {
6682 Metadata *Elts[] = {MD.getOperand(0), MD.getOperand(1)};
6683 MDNode *ScalarType = MDNode::get(Context, Elts);
6684 // Create a MDNode <ScalarType, ScalarType, offset 0, const>
6685 Metadata *Elts2[] = {ScalarType, ScalarType,
6688 MD.getOperand(2)};
6689 return MDNode::get(Context, Elts2);
6690 }
6691 // Create a MDNode <MD, MD, offset 0>
6693 Type::getInt64Ty(Context)))};
6694 return MDNode::get(Context, Elts);
6695}
6696
6698 // !tbaa.struct is a list of (offset, size, tag) triples. Upgrade any
6699 // old-style scalar field tag to struct-path form via UpgradeTBAANode.
6700 unsigned NumOperands = MD.getNumOperands();
6701 if (NumOperands == 0 || NumOperands % 3 != 0)
6702 return &MD; // Malformed; leave it for the verifier to reject.
6703
6705 bool Changed = false;
6706 for (unsigned I = 2; I < NumOperands; I += 3) {
6707 auto *Tag = dyn_cast_or_null<MDNode>(Elts[I]);
6708 if (!Tag)
6709 continue;
6710 MDNode *Upgraded = UpgradeTBAANode(*Tag);
6711 if (Upgraded == Tag)
6712 continue;
6713 Elts[I] = Upgraded;
6714 Changed = true;
6715 }
6716 return Changed ? MDNode::get(MD.getContext(), Elts) : &MD;
6717}
6718
6720 Instruction *&Temp) {
6721 if (Opc != Instruction::BitCast)
6722 return nullptr;
6723
6724 Temp = nullptr;
6725 Type *SrcTy = V->getType();
6726 if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() &&
6727 SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) {
6728 LLVMContext &Context = V->getContext();
6729
6730 // We have no information about target data layout, so we assume that
6731 // the maximum pointer size is 64bit.
6732 Type *MidTy = Type::getInt64Ty(Context);
6733 Temp = CastInst::Create(Instruction::PtrToInt, V, MidTy);
6734
6735 return CastInst::Create(Instruction::IntToPtr, Temp, DestTy);
6736 }
6737
6738 return nullptr;
6739}
6740
6742 if (Opc != Instruction::BitCast)
6743 return nullptr;
6744
6745 Type *SrcTy = C->getType();
6746 if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() &&
6747 SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) {
6748 LLVMContext &Context = C->getContext();
6749
6750 // We have no information about target data layout, so we assume that
6751 // the maximum pointer size is 64bit.
6752 Type *MidTy = Type::getInt64Ty(Context);
6753
6755 DestTy);
6756 }
6757
6758 return nullptr;
6759}
6760
6761static std::optional<StringRef> getModuleFlagNameSafely(const MDNode &Flag) {
6762 if (Flag.getNumOperands() < 3)
6763 return std::nullopt;
6764 if (MDString *Name = dyn_cast_or_null<MDString>(Flag.getOperand(1)))
6765 return Name->getString();
6766 return std::nullopt;
6767}
6768
6769/// Check the debug info version number, if it is out-dated, drop the debug
6770/// info. Return true if module is modified.
6773 return false;
6774
6775 llvm::TimeTraceScope timeScope("Upgrade debug info");
6776 // We need to get metadata before the module is verified (i.e., getModuleFlag
6777 // makes assumptions that we haven't verified yet). Carefully extract the flag
6778 // from the metadata.
6779 unsigned Version = 0;
6780 if (NamedMDNode *ModFlags = M.getModuleFlagsMetadata()) {
6781 auto OpIt = find_if(ModFlags->operands(), [](const MDNode *Flag) {
6782 if (auto Name = getModuleFlagNameSafely(*Flag))
6783 return *Name == "Debug Info Version";
6784 return false;
6785 });
6786 if (OpIt != ModFlags->op_end()) {
6787 const MDOperand &ValOp = (*OpIt)->getOperand(2);
6788 if (auto *CI = mdconst::dyn_extract_or_null<ConstantInt>(ValOp))
6789 Version = CI->getZExtValue();
6790 }
6791 }
6792
6794 bool BrokenDebugInfo = false;
6795 if (verifyModule(M, &llvm::errs(), &BrokenDebugInfo))
6796 report_fatal_error("Broken module found, compilation aborted!");
6797 if (!BrokenDebugInfo)
6798 // Everything is ok.
6799 return false;
6800 else {
6801 // Diagnose malformed debug info.
6803 M.getContext().diagnose(Diag);
6804 }
6805 }
6806 bool Modified = StripDebugInfo(M);
6808 // Diagnose a version mismatch.
6810 M.getContext().diagnose(DiagVersion);
6811 }
6812 return Modified;
6813}
6814
6815static void upgradeNVVMFnVectorAttr(const StringRef Attr, const char DimC,
6816 GlobalValue *GV, const Metadata *V) {
6817 Function *F = cast<Function>(GV);
6818
6819 constexpr StringLiteral DefaultValue = "1";
6820 StringRef Vect3[3] = {DefaultValue, DefaultValue, DefaultValue};
6821 unsigned Length = 0;
6822
6823 if (F->hasFnAttribute(Attr)) {
6824 // We expect the existing attribute to have the form "x[,y[,z]]". Here we
6825 // parse these elements placing them into Vect3
6826 StringRef S = F->getFnAttribute(Attr).getValueAsString();
6827 for (; Length < 3 && !S.empty(); Length++) {
6828 auto [Part, Rest] = S.split(',');
6829 Vect3[Length] = Part.trim();
6830 S = Rest;
6831 }
6832 }
6833
6834 const unsigned Dim = DimC - 'x';
6835 assert(Dim < 3 && "Unexpected dim char");
6836
6837 const uint64_t VInt = mdconst::extract<ConstantInt>(V)->getZExtValue();
6838
6839 // local variable required for StringRef in Vect3 to point to.
6840 const std::string VStr = llvm::utostr(VInt);
6841 Vect3[Dim] = VStr;
6842 Length = std::max(Length, Dim + 1);
6843
6844 const std::string NewAttr = llvm::join(ArrayRef(Vect3, Length), ",");
6845 F->addFnAttr(Attr, NewAttr);
6846}
6847
6848static inline bool isXYZ(StringRef S) {
6849 return S == "x" || S == "y" || S == "z";
6850}
6851
6853 const Metadata *V) {
6854 if (K == "kernel") {
6856 cast<Function>(GV)->setCallingConv(CallingConv::PTX_Kernel);
6857 return true;
6858 }
6859 if (K == "align") {
6860 // V is a bitfeild specifying two 16-bit values. The alignment value is
6861 // specfied in low 16-bits, The index is specified in the high bits. For the
6862 // index, 0 indicates the return value while higher values correspond to
6863 // each parameter (idx = param + 1).
6864 const uint64_t AlignIdxValuePair =
6865 mdconst::extract<ConstantInt>(V)->getZExtValue();
6866 const unsigned Idx = (AlignIdxValuePair >> 16);
6867 const Align StackAlign = Align(AlignIdxValuePair & 0xFFFF);
6868 cast<Function>(GV)->addAttributeAtIndex(
6869 Idx, Attribute::getWithStackAlignment(GV->getContext(), StackAlign));
6870 return true;
6871 }
6872 if (K == "maxclusterrank" || K == "cluster_max_blocks") {
6873 const auto CV = mdconst::extract<ConstantInt>(V)->getZExtValue();
6875 return true;
6876 }
6877 if (K == "minctasm") {
6878 const auto CV = mdconst::extract<ConstantInt>(V)->getZExtValue();
6879 cast<Function>(GV)->addFnAttr(NVVMAttr::MinCTASm, llvm::utostr(CV));
6880 return true;
6881 }
6882 if (K == "maxnreg") {
6883 const auto CV = mdconst::extract<ConstantInt>(V)->getZExtValue();
6884 cast<Function>(GV)->addFnAttr(NVVMAttr::MaxNReg, llvm::utostr(CV));
6885 return true;
6886 }
6887 if (K.consume_front("maxntid") && isXYZ(K)) {
6889 return true;
6890 }
6891 if (K.consume_front("reqntid") && isXYZ(K)) {
6893 return true;
6894 }
6895 if (K.consume_front("cluster_dim_") && isXYZ(K)) {
6897 return true;
6898 }
6899 if (K == "grid_constant") {
6900 const auto Attr = Attribute::get(GV->getContext(), NVVMAttr::GridConstant);
6901 for (const auto &Op : cast<MDNode>(V)->operands()) {
6902 // For some reason, the index is 1-based in the metadata. Good thing we're
6903 // able to auto-upgrade it!
6904 const auto Index = mdconst::extract<ConstantInt>(Op)->getZExtValue() - 1;
6905 cast<Function>(GV)->addParamAttr(Index, Attr);
6906 }
6907 return true;
6908 }
6909
6910 return false;
6911}
6912
6914 NamedMDNode *NamedMD = M.getNamedMetadata("nvvm.annotations");
6915 if (!NamedMD)
6916 return;
6917
6918 SmallVector<MDNode *, 8> NewNodes;
6920 for (MDNode *MD : NamedMD->operands()) {
6921 if (!SeenNodes.insert(MD).second)
6922 continue;
6923
6924 auto *GV = mdconst::dyn_extract_or_null<GlobalValue>(MD->getOperand(0));
6925 if (!GV)
6926 continue;
6927
6928 assert((MD->getNumOperands() % 2) == 1 && "Invalid number of operands");
6929
6930 SmallVector<Metadata *, 8> NewOperands{MD->getOperand(0)};
6931 // Each nvvm.annotations metadata entry will be of the following form:
6932 // !{ ptr @gv, !"key1", value1, !"key2", value2, ... }
6933 // start index = 1, to skip the global variable key
6934 // increment = 2, to skip the value for each property-value pairs
6935 for (unsigned j = 1, je = MD->getNumOperands(); j < je; j += 2) {
6936 MDString *K = cast<MDString>(MD->getOperand(j));
6937 const MDOperand &V = MD->getOperand(j + 1);
6938 bool Upgraded = upgradeSingleNVVMAnnotation(GV, K->getString(), V);
6939 if (!Upgraded)
6940 NewOperands.append({K, V});
6941 }
6942
6943 if (NewOperands.size() > 1)
6944 NewNodes.push_back(MDNode::get(M.getContext(), NewOperands));
6945 }
6946
6947 NamedMD->clearOperands();
6948 for (MDNode *N : NewNodes)
6949 NamedMD->addOperand(N);
6950}
6951
6952/// This checks for objc retain release marker which should be upgraded. It
6953/// returns true if module is modified.
6955 bool Changed = false;
6956 const char *MarkerKey = "clang.arc.retainAutoreleasedReturnValueMarker";
6957 NamedMDNode *ModRetainReleaseMarker = M.getNamedMetadata(MarkerKey);
6958 if (ModRetainReleaseMarker) {
6959 MDNode *Op = ModRetainReleaseMarker->getOperand(0);
6960 if (Op) {
6961 MDString *ID = dyn_cast_or_null<MDString>(Op->getOperand(0));
6962 if (ID) {
6963 SmallVector<StringRef, 4> ValueComp;
6964 ID->getString().split(ValueComp, "#");
6965 if (ValueComp.size() == 2) {
6966 std::string NewValue = ValueComp[0].str() + ";" + ValueComp[1].str();
6967 ID = MDString::get(M.getContext(), NewValue);
6968 }
6969 M.addModuleFlag(Module::Error, MarkerKey, ID);
6970 M.eraseNamedMetadata(ModRetainReleaseMarker);
6971 Changed = true;
6972 }
6973 }
6974 }
6975 return Changed;
6976}
6977
6979 // This lambda converts normal function calls to ARC runtime functions to
6980 // intrinsic calls.
6981 auto UpgradeToIntrinsic = [&](const char *OldFunc,
6982 llvm::Intrinsic::ID IntrinsicFunc) {
6983 Function *Fn = M.getFunction(OldFunc);
6984
6985 if (!Fn)
6986 return;
6987
6988 Function *NewFn =
6989 llvm::Intrinsic::getOrInsertDeclaration(&M, IntrinsicFunc);
6990
6991 for (User *U : make_early_inc_range(Fn->users())) {
6993 if (!CI || CI->getCalledFunction() != Fn)
6994 continue;
6995
6996 IRBuilder<> Builder(CI->getIterator());
6997 FunctionType *NewFuncTy = NewFn->getFunctionType();
6999
7000 // Don't upgrade the intrinsic if it's not valid to bitcast the return
7001 // value to the return type of the old function.
7002 if (NewFuncTy->getReturnType() != CI->getType() &&
7003 !CastInst::castIsValid(Instruction::BitCast, CI,
7004 NewFuncTy->getReturnType()))
7005 continue;
7006
7007 bool InvalidCast = false;
7008
7009 for (unsigned I = 0, E = CI->arg_size(); I != E; ++I) {
7010 Value *Arg = CI->getArgOperand(I);
7011
7012 // Bitcast argument to the parameter type of the new function if it's
7013 // not a variadic argument.
7014 if (I < NewFuncTy->getNumParams()) {
7015 // Don't upgrade the intrinsic if it's not valid to bitcast the argument
7016 // to the parameter type of the new function.
7017 if (!CastInst::castIsValid(Instruction::BitCast, Arg,
7018 NewFuncTy->getParamType(I))) {
7019 InvalidCast = true;
7020 break;
7021 }
7022 Arg = Builder.CreateBitCast(Arg, NewFuncTy->getParamType(I));
7023 }
7024 Args.push_back(Arg);
7025 }
7026
7027 if (InvalidCast)
7028 continue;
7029
7030 // Create a call instruction that calls the new function.
7031 CallInst *NewCall = Builder.CreateCall(NewFuncTy, NewFn, Args);
7032 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
7033 NewCall->takeName(CI);
7034
7035 // Bitcast the return value back to the type of the old call.
7036 Value *NewRetVal = Builder.CreateBitCast(NewCall, CI->getType());
7037
7038 if (!CI->use_empty())
7039 CI->replaceAllUsesWith(NewRetVal);
7040 CI->eraseFromParent();
7041 }
7042
7043 if (Fn->use_empty())
7044 Fn->eraseFromParent();
7045 };
7046
7047 // Unconditionally convert a call to "clang.arc.use" to a call to
7048 // "llvm.objc.clang.arc.use".
7049 UpgradeToIntrinsic("clang.arc.use", llvm::Intrinsic::objc_clang_arc_use);
7050
7051 // Upgrade the retain release marker. If there is no need to upgrade
7052 // the marker, that means either the module is already new enough to contain
7053 // new intrinsics or it is not ARC. There is no need to upgrade runtime call.
7055 return;
7056
7057 std::pair<const char *, llvm::Intrinsic::ID> RuntimeFuncs[] = {
7058 {"objc_autorelease", llvm::Intrinsic::objc_autorelease},
7059 {"objc_autoreleasePoolPop", llvm::Intrinsic::objc_autoreleasePoolPop},
7060 {"objc_autoreleasePoolPush", llvm::Intrinsic::objc_autoreleasePoolPush},
7061 {"objc_autoreleaseReturnValue",
7062 llvm::Intrinsic::objc_autoreleaseReturnValue},
7063 {"objc_copyWeak", llvm::Intrinsic::objc_copyWeak},
7064 {"objc_destroyWeak", llvm::Intrinsic::objc_destroyWeak},
7065 {"objc_initWeak", llvm::Intrinsic::objc_initWeak},
7066 {"objc_loadWeak", llvm::Intrinsic::objc_loadWeak},
7067 {"objc_loadWeakRetained", llvm::Intrinsic::objc_loadWeakRetained},
7068 {"objc_moveWeak", llvm::Intrinsic::objc_moveWeak},
7069 {"objc_release", llvm::Intrinsic::objc_release},
7070 {"objc_retain", llvm::Intrinsic::objc_retain},
7071 {"objc_retainAutorelease", llvm::Intrinsic::objc_retainAutorelease},
7072 {"objc_retainAutoreleaseReturnValue",
7073 llvm::Intrinsic::objc_retainAutoreleaseReturnValue},
7074 {"objc_retainAutoreleasedReturnValue",
7075 llvm::Intrinsic::objc_retainAutoreleasedReturnValue},
7076 {"objc_retainBlock", llvm::Intrinsic::objc_retainBlock},
7077 {"objc_storeStrong", llvm::Intrinsic::objc_storeStrong},
7078 {"objc_storeWeak", llvm::Intrinsic::objc_storeWeak},
7079 {"objc_unsafeClaimAutoreleasedReturnValue",
7080 llvm::Intrinsic::objc_unsafeClaimAutoreleasedReturnValue},
7081 {"objc_retainedObject", llvm::Intrinsic::objc_retainedObject},
7082 {"objc_unretainedObject", llvm::Intrinsic::objc_unretainedObject},
7083 {"objc_unretainedPointer", llvm::Intrinsic::objc_unretainedPointer},
7084 {"objc_retain_autorelease", llvm::Intrinsic::objc_retain_autorelease},
7085 {"objc_sync_enter", llvm::Intrinsic::objc_sync_enter},
7086 {"objc_sync_exit", llvm::Intrinsic::objc_sync_exit},
7087 {"objc_arc_annotation_topdown_bbstart",
7088 llvm::Intrinsic::objc_arc_annotation_topdown_bbstart},
7089 {"objc_arc_annotation_topdown_bbend",
7090 llvm::Intrinsic::objc_arc_annotation_topdown_bbend},
7091 {"objc_arc_annotation_bottomup_bbstart",
7092 llvm::Intrinsic::objc_arc_annotation_bottomup_bbstart},
7093 {"objc_arc_annotation_bottomup_bbend",
7094 llvm::Intrinsic::objc_arc_annotation_bottomup_bbend}};
7095
7096 for (auto &I : RuntimeFuncs)
7097 UpgradeToIntrinsic(I.first, I.second);
7098}
7099
7100// Upgrade the way signing of pointers to init/fini functions is described.
7101//
7102// Originally, the `@llvm.global_(ctors|dtors)` arrays contained `ptrauth`
7103// constants, if signing was requested. After the upgrade, these arrays contain
7104// plain function pointers and the desired signing schema is described via a
7105// pair of module flags.
7106//
7107// Note that the upgrade is only performed if all elements of *both* arrays
7108// agree on a common signing schema.
7110 // As we cannot always decide whether the particular module should have
7111 // ptrauth-init-fini flags, we have to treat absent flags as having zero
7112 // values for compatibility reasons. Thus, upgradePtrauthInitFiniArrays
7113 // returns as soon as it spots any non-signed init/fini pointer: either we
7114 // should request non-signed pointers (safe to omit both flags) or there is
7115 // no common schema (and thus we do not modify anything).
7116 //
7117 // UseAddressDisc's value either represents "not decided yet" state (nullopt)
7118 // or whether we should request address diversity in addition to the basic
7119 // constant diversity. There is no value representing "decided not to sign"
7120 // for the reasons explained above.
7121 std::optional<bool> UseAddressDisc;
7122
7123 // Do not attempt upgrading if the new module flags already exist.
7124 if (const NamedMDNode *ModFlags = M.getModuleFlagsMetadata()) {
7125 for (const MDNode *Flag : ModFlags->operands()) {
7126 std::optional<StringRef> Name = getModuleFlagNameSafely(*Flag);
7127 if (Name && (*Name == "ptrauth-init-fini" ||
7128 *Name == "ptrauth-init-fini-address-discrimination"))
7129 return false;
7130 }
7131 }
7132
7133 auto UpgradeSinglePointer = [&UseAddressDisc](Constant *CV) -> Constant * {
7134 constexpr unsigned ExpectedConstDisc = 0xD9D4;
7135 constexpr unsigned ExpectedAddressMarker = 1;
7136
7137 auto *CPA = dyn_cast<ConstantPtrAuth>(CV);
7138 if (!CPA || !CPA->getDiscriminator()->equalsInt(ExpectedConstDisc))
7139 return nullptr; // Nothing to upgrade or unknown pattern found.
7140
7141 bool HasAddressDisc;
7142 if (!CPA->hasAddressDiscriminator())
7143 HasAddressDisc = false;
7144 else if (CPA->hasSpecialAddressDiscriminator(ExpectedAddressMarker))
7145 HasAddressDisc = true;
7146 else
7147 return nullptr; // Unknown pattern.
7148
7149 if (UseAddressDisc && *UseAddressDisc != HasAddressDisc)
7150 return nullptr; // Disagreement with the decided mode.
7151
7152 UseAddressDisc = HasAddressDisc;
7153 return CPA->getPointer();
7154 };
7155
7156 // Do not apply any changes until we know the upgrade is non-ambiguous.
7157 using PendingUpgrade = std::pair<GlobalVariable *, Constant *>;
7158 SmallVector<PendingUpgrade, 2> GlobalArraysToUpgrade;
7159
7160 for (const char *Name : {"llvm.global_ctors", "llvm.global_dtors"}) {
7161 auto *GV = dyn_cast_if_present<GlobalVariable>(M.getNamedValue(Name));
7162 if (!GV || !GV->hasInitializer())
7163 continue; // Skip, but it is okay to upgrade the other variable.
7164
7165 auto *OldStructorsArray = dyn_cast<ConstantArray>(GV->getInitializer());
7166 if (!OldStructorsArray || OldStructorsArray->getNumOperands() == 0)
7167 return false;
7168
7169 std::vector<Constant *> NewStructors;
7170 NewStructors.reserve(OldStructorsArray->getNumOperands());
7171
7172 for (Use &U : OldStructorsArray->operands()) {
7173 ConstantStruct *Structor = dyn_cast<ConstantStruct>(U.get());
7174 if (!Structor || Structor->getNumOperands() != 3)
7175 return false;
7176
7177 Constant *Prio = Structor->getOperand(0);
7178 Constant *Func = Structor->getOperand(1);
7179 Constant *Arg = Structor->getOperand(2);
7180
7181 Func = UpgradeSinglePointer(Func);
7182 if (!Func)
7183 return false;
7184
7185 NewStructors.push_back(
7186 ConstantStruct::get(Structor->getType(), {Prio, Func, Arg}));
7187 }
7188
7189 Constant *NewInit =
7190 ConstantArray::get(OldStructorsArray->getType(), NewStructors);
7191 GlobalArraysToUpgrade.emplace_back(GV, NewInit);
7192 }
7193
7194 if (GlobalArraysToUpgrade.empty())
7195 return false;
7196 assert(UseAddressDisc.has_value());
7197
7198 for (auto [GV, NewInit] : GlobalArraysToUpgrade)
7199 GV->setInitializer(NewInit);
7200
7201 M.addModuleFlag(Module::Error, "ptrauth-init-fini", 1);
7202 M.addModuleFlag(Module::Error, "ptrauth-init-fini-address-discrimination",
7203 *UseAddressDisc);
7204
7205 return true;
7206}
7207
7209 bool Changed = false;
7211
7212 NamedMDNode *ModFlags = M.getModuleFlagsMetadata();
7213 if (!ModFlags)
7214 return Changed;
7215
7216 bool HasObjCFlag = false, HasClassProperties = false;
7217 bool HasSwiftVersionFlag = false;
7218 uint8_t SwiftMajorVersion, SwiftMinorVersion;
7219 uint32_t SwiftABIVersion;
7220 auto Int8Ty = Type::getInt8Ty(M.getContext());
7221 auto Int32Ty = Type::getInt32Ty(M.getContext());
7222
7223 for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) {
7224 MDNode *Op = ModFlags->getOperand(I);
7225 if (Op->getNumOperands() != 3)
7226 continue;
7227 MDString *ID = dyn_cast_or_null<MDString>(Op->getOperand(1));
7228 if (!ID)
7229 continue;
7230 auto SetBehavior = [&](Module::ModFlagBehavior B) {
7231 Metadata *Ops[3] = {ConstantAsMetadata::get(ConstantInt::get(
7232 Type::getInt32Ty(M.getContext()), B)),
7233 MDString::get(M.getContext(), ID->getString()),
7234 Op->getOperand(2)};
7235 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7236 Changed = true;
7237 };
7238
7239 if (ID->getString() == "Objective-C Image Info Version")
7240 HasObjCFlag = true;
7241 if (ID->getString() == "Objective-C Class Properties")
7242 HasClassProperties = true;
7243 // Upgrade PIC from Error/Max to Min.
7244 if (ID->getString() == "PIC Level") {
7245 if (auto *Behavior =
7247 uint64_t V = Behavior->getLimitedValue();
7248 if (V == Module::Error || V == Module::Max)
7249 SetBehavior(Module::Min);
7250 }
7251 }
7252 // Upgrade "PIE Level" from Error to Max.
7253 if (ID->getString() == "PIE Level")
7254 if (auto *Behavior =
7256 if (Behavior->getLimitedValue() == Module::Error)
7257 SetBehavior(Module::Max);
7258
7259 // Upgrade branch protection and return address signing module flags. The
7260 // module flag behavior for these fields were Error and now they are Min.
7261 // The one exception is "sign-return-address-harden".
7262 if (ID->getString() == "branch-target-enforcement" ||
7263 (ID->getString().starts_with("sign-return-address") &&
7264 ID->getString() != "sign-return-address-harden")) {
7265 if (auto *Behavior =
7267 if (Behavior->getLimitedValue() == Module::Error) {
7268 Type *Int32Ty = Type::getInt32Ty(M.getContext());
7269 Metadata *Ops[3] = {
7270 ConstantAsMetadata::get(ConstantInt::get(Int32Ty, Module::Min)),
7271 Op->getOperand(1), Op->getOperand(2)};
7272 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7273 Changed = true;
7274 }
7275 }
7276 }
7277
7278 // Upgrade Objective-C Image Info Section. Removed the whitespce in the
7279 // section name so that llvm-lto will not complain about mismatching
7280 // module flags that is functionally the same.
7281 if (ID->getString() == "Objective-C Image Info Section") {
7282 if (auto *Value = dyn_cast_or_null<MDString>(Op->getOperand(2))) {
7283 SmallVector<StringRef, 4> ValueComp;
7284 Value->getString().split(ValueComp, " ");
7285 if (ValueComp.size() != 1) {
7286 std::string NewValue;
7287 for (auto &S : ValueComp)
7288 NewValue += S.str();
7289 Metadata *Ops[3] = {Op->getOperand(0), Op->getOperand(1),
7290 MDString::get(M.getContext(), NewValue)};
7291 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7292 Changed = true;
7293 }
7294 }
7295 }
7296
7297 // IRUpgrader turns a i32 type "Objective-C Garbage Collection" into i8 value.
7298 // If the higher bits are set, it adds new module flag for swift info.
7299 if (ID->getString() == "Objective-C Garbage Collection") {
7300 auto Md = dyn_cast<ConstantAsMetadata>(Op->getOperand(2));
7301 if (Md) {
7302 assert(Md->getValue() && "Expected non-empty metadata");
7303 auto Type = Md->getValue()->getType();
7304 if (Type == Int8Ty)
7305 continue;
7306 unsigned Val = Md->getValue()->getUniqueInteger().getZExtValue();
7307 if ((Val & 0xff) != Val) {
7308 HasSwiftVersionFlag = true;
7309 SwiftABIVersion = (Val & 0xff00) >> 8;
7310 SwiftMajorVersion = (Val & 0xff000000) >> 24;
7311 SwiftMinorVersion = (Val & 0xff0000) >> 16;
7312 }
7313 Metadata *Ops[3] = {
7314 ConstantAsMetadata::get(ConstantInt::get(Int32Ty,Module::Error)),
7315 Op->getOperand(1),
7316 ConstantAsMetadata::get(ConstantInt::get(Int8Ty,Val & 0xff))};
7317 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7318 Changed = true;
7319 }
7320 }
7321
7322 if (ID->getString() == "amdgpu_code_object_version") {
7323 Metadata *Ops[3] = {
7324 Op->getOperand(0),
7325 MDString::get(M.getContext(), "amdhsa_code_object_version"),
7326 Op->getOperand(2)};
7327 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7328 Changed = true;
7329 }
7330
7331 // clang/PowerPC used to use "float-abi" to describe the long double format;
7332 // it has been renamed to "long-double-type", with its values changed to the
7333 // corresponding IR floating-point type names.
7334 if (M.getTargetTriple().isPPC() && ID->getString() == "float-abi") {
7336 if (auto *S = dyn_cast_or_null<MDString>(Op->getOperand(2)))
7337 Format = S->getString();
7338
7339 // The "float-abi" key is now reserved for the target-independent
7340 // soft/hard ABI flag, so leave a valid value alone. Map any other value
7341 // (including unrecognized ones, which were never valid) to the default.
7343 LongDoubleFormat NewFormat =
7345 .Case("ieeequad", LongDoubleFormat::IEEEquad)
7346 .Case("ieeedouble", LongDoubleFormat::IEEEdouble)
7348 Metadata *Ops[3] = {
7349 Op->getOperand(0),
7350 MDString::get(M.getContext(), "long-double-type"),
7351 MDString::get(M.getContext(), getLongDoubleFormatName(NewFormat))};
7352 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7353 Changed = true;
7354 }
7355 }
7356 }
7357
7358 // "Objective-C Class Properties" is recently added for Objective-C. We
7359 // upgrade ObjC bitcodes to contain a "Objective-C Class Properties" module
7360 // flag of value 0, so we can correclty downgrade this flag when trying to
7361 // link an ObjC bitcode without this module flag with an ObjC bitcode with
7362 // this module flag.
7363 if (HasObjCFlag && !HasClassProperties) {
7364 M.addModuleFlag(llvm::Module::Override, "Objective-C Class Properties",
7365 (uint32_t)0);
7366 Changed = true;
7367 }
7368
7369 if (HasSwiftVersionFlag) {
7370 M.addModuleFlag(Module::Error, "Swift ABI Version",
7371 SwiftABIVersion);
7372 M.addModuleFlag(Module::Error, "Swift Major Version",
7373 ConstantInt::get(Int8Ty, SwiftMajorVersion));
7374 M.addModuleFlag(Module::Error, "Swift Minor Version",
7375 ConstantInt::get(Int8Ty, SwiftMinorVersion));
7376 Changed = true;
7377 }
7378
7379 return Changed;
7380}
7381
7383 NamedMDNode *CFIConsts = M.getNamedMetadata("cfi.functions");
7384 // If this metadata has operands, we expect all of them to be either from
7385 // before or from after the format change handled here, so we can bail out
7386 // fast if the first (if any) operands is of the new format.
7387 auto MatchesVersion = [](const MDNode *Op) {
7388 return Op->getNumOperands() >= 3 &&
7389 isa<ConstantAsMetadata>(Op->getOperand(2)) &&
7390 cast<ConstantAsMetadata>(Op->getOperand(2))
7391 ->getType()
7392 ->isIntegerTy(64);
7393 };
7394
7395 if (!CFIConsts || !CFIConsts->getNumOperands() ||
7396 MatchesVersion(CFIConsts->getOperand(0)))
7397 return false;
7398
7399 bool Changed = false;
7400 for (unsigned I = 0, E = CFIConsts->getNumOperands(); I != E; ++I) {
7401 MDNode *Op = CFIConsts->getOperand(I);
7402 assert(!MatchesVersion(Op) && "Unexpected mix of CFIConstant formats");
7403 assert(Op->getNumOperands() >= 2 &&
7404 "Expected at least 2 operands - name and linkage type");
7405 MDString *NameMD = dyn_cast<MDString>(Op->getOperand(0));
7406 StringRef Name = NameMD->getString();
7409
7411 Elts.push_back(Op->getOperand(0));
7412 Elts.push_back(Op->getOperand(1));
7414 ConstantInt::get(Type::getInt64Ty(M.getContext()), GUID)));
7415
7416 for (unsigned J = 2, EJ = Op->getNumOperands(); J != EJ; ++J)
7417 Elts.push_back(Op->getOperand(J));
7418
7419 CFIConsts->setOperand(I, MDNode::get(M.getContext(), Elts));
7420 Changed = true;
7421 }
7422
7423 return Changed;
7424}
7425
7427 auto TrimSpaces = [](StringRef Section) -> std::string {
7428 SmallVector<StringRef, 5> Components;
7429 Section.split(Components, ',');
7430
7431 SmallString<32> Buffer;
7432 raw_svector_ostream OS(Buffer);
7433
7434 for (auto Component : Components)
7435 OS << ',' << Component.trim();
7436
7437 return std::string(OS.str().substr(1));
7438 };
7439
7440 for (auto &GV : M.globals()) {
7441 if (!GV.hasSection())
7442 continue;
7443
7444 StringRef Section = GV.getSection();
7445
7446 if (!Section.starts_with("__DATA, __objc_catlist"))
7447 continue;
7448
7449 // __DATA, __objc_catlist, regular, no_dead_strip
7450 // __DATA,__objc_catlist,regular,no_dead_strip
7451 GV.setSection(TrimSpaces(Section));
7452 }
7453}
7454
7455namespace {
7456// Prior to LLVM 10.0, the strictfp attribute could be used on individual
7457// callsites within a function that did not also have the strictfp attribute.
7458// Since 10.0, if strict FP semantics are needed within a function, the
7459// function must have the strictfp attribute and all calls within the function
7460// must also have the strictfp attribute. This latter restriction is
7461// necessary to prevent unwanted libcall simplification when a function is
7462// being cloned (such as for inlining).
7463//
7464// The "dangling" strictfp attribute usage was only used to prevent constant
7465// folding and other libcall simplification. The nobuiltin attribute on the
7466// callsite has the same effect.
7467struct StrictFPUpgradeVisitor : public InstVisitor<StrictFPUpgradeVisitor> {
7468 StrictFPUpgradeVisitor() = default;
7469
7470 void visitCallBase(CallBase &Call) {
7471 if (!Call.isStrictFP())
7472 return;
7474 return;
7475 // If we get here, the caller doesn't have the strictfp attribute
7476 // but this callsite does. Replace the strictfp attribute with nobuiltin.
7477 Call.removeFnAttr(Attribute::StrictFP);
7478 Call.addFnAttr(Attribute::NoBuiltin);
7479 }
7480};
7481
7482/// Replace "amdgpu-unsafe-fp-atomics" metadata with atomicrmw metadata
7483struct AMDGPUUnsafeFPAtomicsUpgradeVisitor
7484 : public InstVisitor<AMDGPUUnsafeFPAtomicsUpgradeVisitor> {
7485 AMDGPUUnsafeFPAtomicsUpgradeVisitor() = default;
7486
7487 void visitAtomicRMWInst(AtomicRMWInst &RMW) {
7488 if (!RMW.isFloatingPointOperation())
7489 return;
7490
7491 MDNode *Empty = MDNode::get(RMW.getContext(), {});
7492 RMW.setMetadata("amdgpu.no.fine.grained.host.memory", Empty);
7493 RMW.setMetadata("amdgpu.no.remote.memory.access", Empty);
7494 RMW.setMetadata(LLVMContext::MD_atomic_ignore_denormal_mode, Empty);
7495 }
7496};
7497} // namespace
7498
7500 // If a function definition doesn't have the strictfp attribute,
7501 // convert any callsite strictfp attributes to nobuiltin.
7502 if (!F.isDeclaration() && !F.hasFnAttribute(Attribute::StrictFP)) {
7503 StrictFPUpgradeVisitor SFPV;
7504 SFPV.visit(F);
7505 }
7506
7507 // Remove all incompatibile attributes from function.
7508 F.removeRetAttrs(AttributeFuncs::typeIncompatible(
7509 F.getReturnType(), F.getAttributes().getRetAttrs()));
7510 for (auto &Arg : F.args())
7511 Arg.removeAttrs(
7512 AttributeFuncs::typeIncompatible(Arg.getType(), Arg.getAttributes()));
7513
7514 bool AddingAttrs = false, RemovingAttrs = false;
7515 AttrBuilder AttrsToAdd(F.getContext());
7516 AttributeMask AttrsToRemove;
7517
7518 // Older versions of LLVM treated an "implicit-section-name" attribute
7519 // similarly to directly setting the section on a Function.
7520 if (Attribute A = F.getFnAttribute("implicit-section-name");
7521 A.isValid() && A.isStringAttribute()) {
7522 F.setSection(A.getValueAsString());
7523 AttrsToRemove.addAttribute("implicit-section-name");
7524 RemovingAttrs = true;
7525 }
7526
7527 if (Attribute A = F.getFnAttribute("nooutline");
7528 A.isValid() && A.isStringAttribute()) {
7529 AttrsToRemove.addAttribute("nooutline");
7530 AttrsToAdd.addAttribute(Attribute::NoOutline);
7531 AddingAttrs = RemovingAttrs = true;
7532 }
7533
7534 if (Attribute A = F.getFnAttribute("uniform-work-group-size");
7535 A.isValid() && A.isStringAttribute() && !A.getValueAsString().empty()) {
7536 AttrsToRemove.addAttribute("uniform-work-group-size");
7537 RemovingAttrs = true;
7538 if (A.getValueAsString() == "true") {
7539 AttrsToAdd.addAttribute("uniform-work-group-size");
7540 AddingAttrs = true;
7541 }
7542 }
7543
7544 if (!F.empty()) {
7545 // For some reason this is called twice, and the first time is before any
7546 // instructions are loaded into the body.
7547
7548 if (Attribute A = F.getFnAttribute("amdgpu-unsafe-fp-atomics");
7549 A.isValid()) {
7550
7551 if (A.getValueAsBool()) {
7552 AMDGPUUnsafeFPAtomicsUpgradeVisitor Visitor;
7553 Visitor.visit(F);
7554 }
7555
7556 // We will leave behind dead attribute uses on external declarations, but
7557 // clang never added these to declarations anyway.
7558 AttrsToRemove.addAttribute("amdgpu-unsafe-fp-atomics");
7559 RemovingAttrs = true;
7560 }
7561 }
7562
7563 DenormalMode DenormalFPMath = DenormalMode::getIEEE();
7564 DenormalMode DenormalFPMathF32 = DenormalMode::getInvalid();
7565
7566 bool HandleDenormalMode = false;
7567
7568 if (Attribute Attr = F.getFnAttribute("denormal-fp-math"); Attr.isValid()) {
7569 DenormalMode ParsedMode = parseDenormalFPAttribute(Attr.getValueAsString());
7570 if (ParsedMode.isValid()) {
7571 DenormalFPMath = ParsedMode;
7572 AttrsToRemove.addAttribute("denormal-fp-math");
7573 AddingAttrs = RemovingAttrs = true;
7574 HandleDenormalMode = true;
7575 }
7576 }
7577
7578 if (Attribute Attr = F.getFnAttribute("denormal-fp-math-f32");
7579 Attr.isValid()) {
7580 DenormalMode ParsedMode = parseDenormalFPAttribute(Attr.getValueAsString());
7581 if (ParsedMode.isValid()) {
7582 DenormalFPMathF32 = ParsedMode;
7583 AttrsToRemove.addAttribute("denormal-fp-math-f32");
7584 AddingAttrs = RemovingAttrs = true;
7585 HandleDenormalMode = true;
7586 }
7587 }
7588
7589 if (HandleDenormalMode)
7590 AttrsToAdd.addDenormalFPEnvAttr(
7591 DenormalFPEnv(DenormalFPMath, DenormalFPMathF32));
7592
7593 if (RemovingAttrs)
7594 F.removeFnAttrs(AttrsToRemove);
7595
7596 if (AddingAttrs)
7597 F.addFnAttrs(AttrsToAdd);
7598}
7599
7600// Check if the function attribute is not present and set it.
7602 StringRef Value) {
7603 if (!F.hasFnAttribute(FnAttrName)) {
7604 F.addFnAttr(FnAttrName, Value);
7605 LLVM_DEBUG(dbgs() << "Set attribute: " << FnAttrName << "=\"" << Value
7606 << "\", function: " << F.getName() << "\n");
7607 }
7608}
7609
7610// Check if the function attribute is not present and set it if needed.
7611// If the attribute is "false" then removes it.
7612// If the attribute is "true" resets it to a valueless attribute.
7613static void ConvertFunctionAttr(Function &F, bool Set, StringRef FnAttrName) {
7614 if (!F.hasFnAttribute(FnAttrName)) {
7615 if (Set) {
7616 F.addFnAttr(FnAttrName);
7617 LLVM_DEBUG(dbgs() << "Added attribute: " << FnAttrName
7618 << ", function: " << F.getName() << "\n");
7619 }
7620 } else {
7621 auto A = F.getFnAttribute(FnAttrName);
7622 if ("false" == A.getValueAsString()) {
7623 F.removeFnAttr(FnAttrName);
7624 LLVM_DEBUG(dbgs() << "Removed attribute: " << FnAttrName
7625 << "=\"false\", function: " << F.getName() << "\n");
7626 } else if ("true" == A.getValueAsString()) {
7627 F.removeFnAttr(FnAttrName);
7628 F.addFnAttr(FnAttrName);
7629 LLVM_DEBUG(dbgs() << "Converted attribute: " << FnAttrName
7630 << "=\"true\", function: " << F.getName() << "\n");
7631 }
7632 }
7633}
7634
7636 StringRef Key, uint32_t Val) {
7637 M.setModuleFlag(Behavior, Key, Val);
7638 LLVM_DEBUG(dbgs() << "Converted module flag: " << "{" << Behavior << ", "
7639 << Key << ", " << Val << "}\n");
7640}
7641
7643 Triple T(M.getTargetTriple());
7644 if (!T.isThumb() && !T.isARM() && !T.isAArch64())
7645 return;
7646
7647 uint64_t BTEValue = 0;
7648 uint64_t BPPLRValue = 0;
7649 uint64_t GCSValue = 0;
7650 uint64_t SRAValue = 0;
7651 uint64_t SRAALLValue = 0;
7652 uint64_t SRABKeyValue = 0;
7653
7654 NamedMDNode *ModFlags = M.getModuleFlagsMetadata();
7655 if (ModFlags) {
7656 for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) {
7657 MDNode *Op = ModFlags->getOperand(I);
7658 if (Op->getNumOperands() != 3)
7659 continue;
7660
7661 MDString *ID = dyn_cast_or_null<MDString>(Op->getOperand(1));
7662 auto *CI = mdconst::dyn_extract<ConstantInt>(Op->getOperand(2));
7663 if (!ID || !CI)
7664 continue;
7665
7666 StringRef IDStr = ID->getString();
7667 uint64_t *ValPtr = IDStr == "branch-target-enforcement" ? &BTEValue
7668 : IDStr == "branch-protection-pauth-lr" ? &BPPLRValue
7669 : IDStr == "guarded-control-stack" ? &GCSValue
7670 : IDStr == "sign-return-address" ? &SRAValue
7671 : IDStr == "sign-return-address-all" ? &SRAALLValue
7672 : IDStr == "sign-return-address-with-bkey"
7673 ? &SRABKeyValue
7674 : nullptr;
7675 if (!ValPtr)
7676 continue;
7677
7678 *ValPtr = CI->getZExtValue();
7679 if (*ValPtr == 2)
7680 return;
7681
7682 LLVM_DEBUG(dbgs() << "Found module flag: " << IDStr << "(" << *ValPtr
7683 << ")\n");
7684 }
7685 }
7686
7687 bool BTE = BTEValue == 1;
7688 bool BPPLR = BPPLRValue == 1;
7689 bool GCS = GCSValue == 1;
7690 bool SRA = SRAValue == 1;
7691
7692 StringRef SignTypeValue = "non-leaf";
7693 if (SRA && SRAALLValue == 1)
7694 SignTypeValue = "all";
7695
7696 StringRef SignKeyValue = "a_key";
7697 if (SRA && SRABKeyValue == 1)
7698 SignKeyValue = "b_key";
7699
7700 for (Function &F : M.getFunctionList()) {
7701 if (F.isDeclaration())
7702 continue;
7703
7704 if (SRA) {
7705 setFunctionAttrIfNotSet(F, "sign-return-address", SignTypeValue);
7706 setFunctionAttrIfNotSet(F, "sign-return-address-key", SignKeyValue);
7707 } else {
7708 if (auto A = F.getFnAttribute("sign-return-address");
7709 A.isValid() && "none" == A.getValueAsString()) {
7710 F.removeFnAttr("sign-return-address");
7711 F.removeFnAttr("sign-return-address-key");
7712 }
7713 }
7714 ConvertFunctionAttr(F, BTE, "branch-target-enforcement");
7715 ConvertFunctionAttr(F, BPPLR, "branch-protection-pauth-lr");
7716 ConvertFunctionAttr(F, GCS, "guarded-control-stack");
7717 }
7718
7719 if (BTE)
7720 ConvertModuleFlag(M, llvm::Module::Min, "branch-target-enforcement", 2);
7721 if (BPPLR)
7722 ConvertModuleFlag(M, llvm::Module::Min, "branch-protection-pauth-lr", 2);
7723 if (GCS)
7724 ConvertModuleFlag(M, llvm::Module::Min, "guarded-control-stack", 2);
7725 if (SRA) {
7726 ConvertModuleFlag(M, llvm::Module::Min, "sign-return-address", 2);
7727 if (SRAALLValue == 1)
7728 ConvertModuleFlag(M, llvm::Module::Min, "sign-return-address-all", 2);
7729 if (SRABKeyValue == 1) {
7730 ConvertModuleFlag(M, llvm::Module::Min, "sign-return-address-with-bkey",
7731 2);
7732 }
7733 }
7734}
7735
7736/// Return the replacement tags if \p T still uses a removed two-operand form.
7738 if (T->getNumOperands() != 2 || !mdconst::hasa<ConstantInt>(T->getOperand(1)))
7739 return nullptr;
7740 auto *Tag = dyn_cast_or_null<MDString>(T->getOperand(0));
7741 return Tag ? findBooleanLoopTags(Tag->getString()) : nullptr;
7742}
7743
7744/// Build the single-operand node that replaces a boolean operand: nonzero
7745/// selects the enable tag, zero the disable tag.
7747 const BooleanLoopTags &Tags,
7748 const MDOperand &Op) {
7749 bool Enable = !mdconst::extract<ConstantInt>(Op)->isZero();
7750 return MDTuple::get(C,
7751 {MDString::get(C, Enable ? Tags.Enable : Tags.Disable)});
7752}
7753
7754static bool isOldLoopArgument(Metadata *MD) {
7755 auto *T = dyn_cast_or_null<MDTuple>(MD);
7756 if (!T)
7757 return false;
7758 if (T->getNumOperands() < 1)
7759 return false;
7760 auto *S = dyn_cast_or_null<MDString>(T->getOperand(0));
7761 if (!S)
7762 return false;
7763 if (S->getString().starts_with("llvm.vectorizer."))
7764 return true;
7765 return getOldBooleanLoopTags(T) != nullptr;
7766}
7767
7769 StringRef OldPrefix = "llvm.vectorizer.";
7770 assert(OldTag.starts_with(OldPrefix) && "Expected old prefix");
7771
7772 if (OldTag == "llvm.vectorizer.unroll")
7773 return MDString::get(C, "llvm.loop.interleave.count");
7774
7775 return MDString::get(
7776 C, (Twine("llvm.loop.vectorize.") + OldTag.drop_front(OldPrefix.size()))
7777 .str());
7778}
7779
7781 auto *T = dyn_cast_or_null<MDTuple>(MD);
7782 if (!T)
7783 return MD;
7784 if (T->getNumOperands() < 1)
7785 return MD;
7786 auto *OldTag = dyn_cast_or_null<MDString>(T->getOperand(0));
7787 if (!OldTag)
7788 return MD;
7789
7790 LLVMContext &C = T->getContext();
7791
7792 /// Rewrite a removed two-operand boolean form to the single-operand pair.
7793 if (const BooleanLoopTags *Tags = getOldBooleanLoopTags(T))
7794 return makeBooleanLoopNode(C, *Tags, T->getOperand(1));
7795
7796 if (!OldTag->getString().starts_with("llvm.vectorizer."))
7797 return MD;
7798
7799 // This has an old tag. Upgrade it.
7800 MDString *NewTag = upgradeLoopTag(C, OldTag->getString());
7801
7802 // The legacy !{!"llvm.vectorizer.enable", i1 X} maps onto the single-operand
7803 // vectorize.enable/disable pair, not a two-operand enable node.
7804 if (T->getNumOperands() == 2 && mdconst::hasa<ConstantInt>(T->getOperand(1)))
7805 if (const BooleanLoopTags *Tags = findBooleanLoopTags(NewTag->getString()))
7806 return makeBooleanLoopNode(C, *Tags, T->getOperand(1));
7807
7809 Ops.reserve(T->getNumOperands());
7810 Ops.push_back(NewTag);
7811 for (unsigned I = 1, E = T->getNumOperands(); I != E; ++I)
7812 Ops.push_back(T->getOperand(I));
7813
7814 return MDTuple::get(C, Ops);
7815}
7816
7818 auto *T = dyn_cast<MDTuple>(&N);
7819 if (!T)
7820 return &N;
7821
7822 if (none_of(T->operands(), isOldLoopArgument))
7823 return &N;
7824
7825 // Fix the removed two-operand boolean nodes in place: the Verifier rejects
7826 // any MDNode carrying those tags with more than one operand, so a leftover
7827 // reference (from the distinct loop-ID) would still trigger a diagnostic.
7828 // In-place mutation is safe on distinct MDNodes.
7829 if (T->isDistinct()) {
7830 for (unsigned I = 0, E = T->getNumOperands(); I < E; ++I) {
7831 auto *OpT = dyn_cast_or_null<MDTuple>(T->getOperand(I));
7832 if (OpT && getOldBooleanLoopTags(OpT))
7833 T->replaceOperandWith(I, upgradeLoopArgument(OpT));
7834 }
7835 if (none_of(T->operands(), isOldLoopArgument))
7836 return &N;
7837 }
7838
7839 // Remaining old arguments (e.g. llvm.vectorizer.*) are handled via a wrapper
7840 // attachment; the original distinct loop-ID is kept as the first operand.
7842 Ops.reserve(T->getNumOperands());
7843 for (Metadata *MD : T->operands())
7844 Ops.push_back(upgradeLoopArgument(MD));
7845
7846 return MDTuple::get(T->getContext(), Ops);
7847}
7848
7850 Triple T(TT);
7851 // The only data layout upgrades needed for pre-GCN, SPIR or SPIRV are setting
7852 // the address space of globals to 1. This does not apply to SPIRV Logical.
7853 if ((T.isSPIR() || (T.isSPIRV() && !T.isSPIRVLogical())) &&
7854 !DL.contains("-G") && !DL.starts_with("G")) {
7855 return DL.empty() ? std::string("G1") : (DL + "-G1").str();
7856 }
7857
7858 if (T.isLoongArch64() || T.isRISCV64()) {
7859 // Make i32 a native type for 64-bit LoongArch and RISC-V.
7860 auto I = DL.find("-n64-");
7861 if (I != StringRef::npos)
7862 return (DL.take_front(I) + "-n32:64-" + DL.drop_front(I + 5)).str();
7863 return DL.str();
7864 }
7865
7866 // AMDGPU data layout upgrades.
7867 std::string Res = DL.str();
7868 if (T.isAMDGPU()) {
7869 // Define address spaces for constants.
7870 if (!DL.contains("-G") && !DL.starts_with("G"))
7871 Res.append(Res.empty() ? "G1" : "-G1");
7872
7873 // AMDGCN data layout upgrades.
7874 if (T.isAMDGCN()) {
7875
7876 // Add missing non-integral declarations.
7877 // This goes before adding new address spaces to prevent incoherent string
7878 // values.
7879 if (!DL.contains("-ni") && !DL.starts_with("ni"))
7880 Res.append("-ni:7:8:9");
7881 // Update ni:7 to ni:7:8:9.
7882 if (DL.ends_with("ni:7"))
7883 Res.append(":8:9");
7884 if (DL.ends_with("ni:7:8"))
7885 Res.append(":9");
7886
7887 // Add sizing for address spaces 7 and 8 (fat raw buffers and buffer
7888 // resources) An empty data layout has already been upgraded to G1 by now.
7889 if (!DL.contains("-p7") && !DL.starts_with("p7"))
7890 Res.append("-p7:160:256:256:32");
7891 if (!DL.contains("-p8") && !DL.starts_with("p8"))
7892 Res.append("-p8:128:128:128:48");
7893 constexpr StringRef OldP8("-p8:128:128-");
7894 if (DL.contains(OldP8))
7895 Res.replace(Res.find(OldP8), OldP8.size(), "-p8:128:128:128:48-");
7896 if (!DL.contains("-p9") && !DL.starts_with("p9"))
7897 Res.append("-p9:192:256:256:32");
7898
7899 // Add sizing for address space 10 through 15.
7900 // AS 10-14 are reserved and defaulted to 32:32
7901 // AS 15 is in use and is 32:32.
7902 for (StringRef AS : {"p10", "p11", "p12", "p13", "p14", "p15"}) {
7903 if (!DL.contains(("-" + AS).str()) && !DL.starts_with(AS))
7904 Res.append(("-" + AS + ":32:32").str());
7905 }
7906 }
7907
7908 // Upgrade the ELF mangling mode.
7909 if (!DL.contains("m:e"))
7910 Res = Res.empty() ? "m:e" : "m:e-" + Res;
7911
7912 return Res;
7913 }
7914
7915 if (T.isSystemZ() && !DL.empty()) {
7916 // Make sure the stack alignment is present.
7917 if (!DL.contains("-S64"))
7918 return "E-S64" + DL.drop_front(1).str();
7919 return DL.str();
7920 }
7921
7922 auto AddPtr32Ptr64AddrSpaces = [&DL, &Res]() {
7923 // If the datalayout matches the expected format, add pointer size address
7924 // spaces to the datalayout.
7925 StringRef AddrSpaces{"-p270:32:32-p271:32:32-p272:64:64"};
7926 if (!DL.contains(AddrSpaces)) {
7928 Regex R("^([Ee]-m:[a-z](-p:32:32)?)(-.*)$");
7929 if (R.match(Res, &Groups))
7930 Res = (Groups[1] + AddrSpaces + Groups[3]).str();
7931 }
7932 };
7933
7934 // AArch64 data layout upgrades.
7935 if (T.isAArch64()) {
7936 // Add "-Fn32"
7937 if (!DL.empty() && !DL.contains("-Fn32"))
7938 Res.append("-Fn32");
7939 AddPtr32Ptr64AddrSpaces();
7940 return Res;
7941 }
7942
7943 if (T.isSPARC() || (T.isMIPS64() && !DL.contains("m:m")) || T.isPPC64() ||
7944 T.isWasm()) {
7945 // Mips64 with o32 ABI did not add "-i128:128".
7946 // Add "-i128:128"
7947 std::string I64 = "-i64:64";
7948 std::string I128 = "-i128:128";
7949 if (!StringRef(Res).contains(I128)) {
7950 size_t Pos = Res.find(I64);
7951 if (Pos != size_t(-1))
7952 Res.insert(Pos + I64.size(), I128);
7953 }
7954 }
7955
7956 if (T.isPPC() && T.isOSAIX() && !DL.contains("f64:32:64") && !DL.empty()) {
7957 size_t Pos = Res.find("-S128");
7958 if (Pos == StringRef::npos)
7959 Pos = Res.size();
7960 Res.insert(Pos, "-f64:32:64");
7961 }
7962
7963 // ARM data layout upgrades.
7964 // Add -Fi8 if a -F has not already been specified.
7965 if (T.isARM() && !DL.empty() && !DL.contains("Fi") && !DL.contains("Fn")) {
7966 const StringRef p3232 = "p:32:32";
7967 size_t Pos = Res.find(p3232);
7968 if (Pos != StringRef::npos)
7969 Res.insert(Pos + p3232.size(), "-Fi8");
7970 }
7971
7972 if (!T.isX86())
7973 return Res;
7974
7975 AddPtr32Ptr64AddrSpaces();
7976
7977 // i128 values need to be 16-byte-aligned. LLVM already called into libgcc
7978 // for i128 operations prior to this being reflected in the data layout, and
7979 // clang mostly produced LLVM IR that already aligned i128 to 16 byte
7980 // boundaries, so although this is a breaking change, the upgrade is expected
7981 // to fix more IR than it breaks.
7982 // Intel MCU is an exception and uses 4-byte-alignment.
7983 if (!T.isOSIAMCU()) {
7984 std::string I128 = "-i128:128";
7985 if (StringRef Ref = Res; !Ref.contains(I128)) {
7987 Regex R("^(e(-[mpi][^-]*)*)((-[^mpi][^-]*)*)$");
7988 if (R.match(Res, &Groups))
7989 Res = (Groups[1] + I128 + Groups[3]).str();
7990 }
7991 }
7992
7993 // For 32-bit MSVC targets, raise the alignment of f80 values to 16 bytes.
7994 // Raising the alignment is safe because Clang did not produce f80 values in
7995 // the MSVC environment before this upgrade was added.
7996 if (T.isWindowsMSVCEnvironment() && !T.isArch64Bit()) {
7997 StringRef Ref = Res;
7998 auto I = Ref.find("-f80:32-");
7999 if (I != StringRef::npos)
8000 Res = (Ref.take_front(I) + "-f80:128-" + Ref.drop_front(I + 8)).str();
8001 }
8002
8003 return Res;
8004}
8005
8006void llvm::UpgradeAttributes(AttrBuilder &B) {
8007 StringRef FramePointer;
8008 Attribute A = B.getAttribute("no-frame-pointer-elim");
8009 if (A.isValid()) {
8010 // The value can be "true" or "false".
8011 FramePointer = A.getValueAsString() == "true" ? "all" : "none";
8012 B.removeAttribute("no-frame-pointer-elim");
8013 }
8014 if (B.contains("no-frame-pointer-elim-non-leaf")) {
8015 // The value is ignored. "no-frame-pointer-elim"="true" takes priority.
8016 if (FramePointer != "all")
8017 FramePointer = "non-leaf";
8018 B.removeAttribute("no-frame-pointer-elim-non-leaf");
8019 }
8020 if (!FramePointer.empty())
8021 B.addAttribute("frame-pointer", FramePointer);
8022
8023 A = B.getAttribute("null-pointer-is-valid");
8024 if (A.isValid()) {
8025 // The value can be "true" or "false".
8026 bool NullPointerIsValid = A.getValueAsString() == "true";
8027 B.removeAttribute("null-pointer-is-valid");
8028 if (NullPointerIsValid)
8029 B.addAttribute(Attribute::NullPointerIsValid);
8030 }
8031
8032 A = B.getAttribute("uniform-work-group-size");
8033 if (A.isValid()) {
8034 StringRef Val = A.getValueAsString();
8035 if (!Val.empty()) {
8036 bool IsTrue = Val == "true";
8037 B.removeAttribute("uniform-work-group-size");
8038 if (IsTrue)
8039 B.addAttribute("uniform-work-group-size");
8040 }
8041 }
8042}
8043
8044void llvm::UpgradeOperandBundles(std::vector<OperandBundleDef> &Bundles) {
8045 // clang.arc.attachedcall bundles are now required to have an operand.
8046 // If they don't, it's okay to drop them entirely: when there is an operand,
8047 // the "attachedcall" is meaningful and required, but without an operand,
8048 // it's just a marker NOP. Dropping it merely prevents an optimization.
8049 erase_if(Bundles, [&](OperandBundleDef &OBD) {
8050 return OBD.getTag() == "clang.arc.attachedcall" &&
8051 OBD.inputs().empty();
8052 });
8053}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file contains the simple types necessary to represent the attributes associated with functions a...
static Value * upgradeX86VPERMT2Intrinsics(IRBuilder<> &Builder, CallBase &CI, bool ZeroMask, bool IndexForm)
static bool isLegacyNVPTXBF16IntSignature(Function *F, Intrinsic::ID IID)
static unsigned getFullArgCountForDefaultArgUpgrade(Function *F, Intrinsic::ID IID, SmallVectorImpl< Type * > &OverloadTys)
#define G2S_ID(ID_SUFFIX, NAME)
static Metadata * upgradeLoopArgument(Metadata *MD)
static Intrinsic::ID shouldUpgradeNVPTXMBarrierInitIntrinsic(StringRef Name)
static bool isXYZ(StringRef S)
static bool upgradeIntrinsicFunction1(Function *F, Function *&NewFn, bool CanUpgradeDebugIntrinsicsToRecords)
static Value * upgradeX86PSLLDQIntrinsics(IRBuilder<> &Builder, Value *Op, unsigned Shift)
static Intrinsic::ID shouldUpgradeNVPTXSharedClusterIntrinsic(Function *F, StringRef Name)
static Value * upgradeVPIntrinsicCall(StringRef Name, CallBase *CI, IRBuilder<> &Builder)
static std::optional< unsigned > getNVPTXTMAReductionOp(StringRef Name)
static Intrinsic::ID shouldUpgradeNVPTXTMAReductionIntrinsics(StringRef Name)
static bool upgradeRetainReleaseMarker(Module &M)
This checks for objc retain release marker which should be upgraded.
static Value * upgradeX86vpcom(IRBuilder<> &Builder, CallBase &CI, unsigned Imm, bool IsSigned)
static Value * upgradeMaskToInt(IRBuilder<> &Builder, CallBase &CI)
static bool convertIntrinsicValidType(StringRef Name, const FunctionType *FuncTy)
static Value * upgradeX86Rotate(IRBuilder<> &Builder, CallBase &CI, bool IsRotateRight)
static bool upgradeX86MultiplyAddBytes(Function *F, Intrinsic::ID IID, Function *&NewFn)
static Value * upgradeNVVMFPArithCall(IRBuilder<> &Builder, CallBase *CI, StringRef Name, const Intrinsic::ID(&IIDs)[2][2])
static Intrinsic::ID getFunctionalIntrinsicIDForVP(StringRef Name)
static void setFunctionAttrIfNotSet(Function &F, StringRef FnAttrName, StringRef Value)
static Intrinsic::ID shouldUpgradeNVPTXBF16Intrinsic(StringRef Name)
static bool upgradeSingleNVVMAnnotation(GlobalValue *GV, StringRef K, const Metadata *V)
static MDNode * unwrapMAVOp(CallBase *CI, unsigned Op)
Helper to unwrap intrinsic call MetadataAsValue operands.
static MDString * upgradeLoopTag(LLVMContext &C, StringRef OldTag)
static ICmpInst::Predicate getVPIntPredicateFromMD(const Value *Op)
static void upgradeNVVMFnVectorAttr(const StringRef Attr, const char DimC, GlobalValue *GV, const Metadata *V)
static bool upgradeX86MaskedFPCompare(Function *F, Intrinsic::ID IID, Function *&NewFn)
static Value * upgradeX86ALIGNIntrinsics(IRBuilder<> &Builder, Value *Op0, Value *Op1, Value *Shift, Value *Passthru, Value *Mask, bool IsVALIGN)
static Value * upgradeAbs(IRBuilder<> &Builder, CallBase &CI)
static bool shouldUpgradeVPIntrinsic(StringRef Name)
static Value * emitX86Select(IRBuilder<> &Builder, Value *Mask, Value *Op0, Value *Op1)
static Value * upgradeAArch64IntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static constexpr Intrinsic::ID NVVMFAddIIDs[2][2]
#define G2S_CTA_ID(ID_SUFFIX, NAME)
static Value * upgradeMaskedMove(IRBuilder<> &Builder, CallBase &CI)
static const BooleanLoopTags * getOldBooleanLoopTags(const MDTuple *T)
Return the replacement tags if T still uses a removed two-operand form.
static bool upgradeX86IntrinsicFunction(Function *F, StringRef Name, Function *&NewFn)
static Value * applyX86MaskOn1BitsVec(IRBuilder<> &Builder, Value *Vec, Value *Mask)
static Intrinsic::ID shouldUpgradeNVPTXTcgen05AllocDeallocIntrinsic(Function *F, StringRef Name)
static std::optional< StringRef > getModuleFlagNameSafely(const MDNode &Flag)
static bool consumeNVVMPtrAddrSpace(StringRef &Name)
static Metadata * makeBooleanLoopNode(LLVMContext &C, const BooleanLoopTags &Tags, const MDOperand &Op)
Build the single-operand node that replaces a boolean operand: nonzero selects the enable tag,...
#define G2S_CLUSTER_CASE(ID_SUFFIX, NAME)
static bool shouldUpgradeX86Intrinsic(Function *F, StringRef Name)
static std::optional< std::pair< Intrinsic::ID, RoundingMode > > getNVVMFPArithUpgrade(StringRef Name, const Intrinsic::ID(&IIDs)[2][2])
static Value * upgradeX86PSRLDQIntrinsics(IRBuilder<> &Builder, Value *Op, unsigned Shift)
static unsigned getFunctionalOpcodeForVP(StringRef Name)
static Intrinsic::ID shouldUpgradeNVPTXTMAG2SIntrinsics(Function *F, StringRef Name, SmallVectorImpl< Type * > &OvlTys)
static Intrinsic::ID shouldUpgradeNVPTXTcgen05CommitSharedIntrinsic(Function *F, StringRef Name)
static bool isOldLoopArgument(Metadata *MD)
static Value * upgradeARMIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static void ConvertModuleFlag(Module &M, Module::ModFlagBehavior Behavior, StringRef Key, uint32_t Val)
static bool upgradeX86IntrinsicsWith8BitMask(Function *F, Intrinsic::ID IID, Function *&NewFn)
static Value * upgradeVectorSplice(CallBase *CI, IRBuilder<> &Builder)
static Value * upgradeAMDGCNIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static Value * upgradeMaskedLoad(IRBuilder<> &Builder, Value *Ptr, Value *Passthru, Value *Mask, bool Aligned)
static Metadata * unwrapMAVMetadataOp(CallBase *CI, unsigned Op)
Helper to unwrap Metadata MetadataAsValue operands, such as the Value field.
static bool upgradeX86BF16Intrinsic(Function *F, Intrinsic::ID IID, Function *&NewFn)
static bool upgradeArmOrAarch64IntrinsicFunction(bool IsArm, Function *F, StringRef Name, Function *&NewFn)
static bool upgradeIntrinsicCallWithDefaultArgs(CallBase *CI, Function *NewFn, IRBuilder<> &Builder)
static Value * getX86MaskVec(IRBuilder<> &Builder, Value *Mask, unsigned NumElts)
static Value * emitX86ScalarSelect(IRBuilder<> &Builder, Value *Mask, Value *Op0, Value *Op1)
static bool upgradeIntrinsicWithDefaultArgs(Function *F, Function *&NewFn)
static Value * upgradeX86ConcatShift(IRBuilder<> &Builder, CallBase &CI, bool IsShiftRight, bool ZeroMask)
static void rename(GlobalValue *GV)
static bool upgradePTESTIntrinsic(Function *F, Intrinsic::ID IID, Function *&NewFn)
static bool upgradeX86BF16DPIntrinsic(Function *F, Intrinsic::ID IID, Function *&NewFn)
#define NVVM_TMA_G2S_MODES(M)
static cl::opt< bool > DisableAutoUpgradeDebugInfo("disable-auto-upgrade-debug-info", cl::desc("Disable autoupgrade of debug info"))
static Value * upgradeMaskedCompare(IRBuilder<> &Builder, CallBase &CI, unsigned CC, bool Signed)
static Value * upgradeX86BinaryIntrinsics(IRBuilder<> &Builder, CallBase &CI, Intrinsic::ID IID)
static Value * upgradeNVVMIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static Intrinsic::ID shouldUpgradeNVPTXBulkG2SClusterIntrinsic(Function *F, StringRef Name, SmallVectorImpl< Type * > &OvlTys)
static Value * upgradeX86MaskedShift(IRBuilder<> &Builder, CallBase &CI, Intrinsic::ID IID)
static bool upgradeAVX512MaskToSelect(StringRef Name, IRBuilder<> &Builder, CallBase &CI, Value *&Rep)
static constexpr Intrinsic::ID NVVMFMulIIDs[2][2]
static void upgradeDbgIntrinsicToDbgRecord(StringRef Name, CallBase *CI)
Convert debug intrinsic calls to non-instruction debug records.
static void ConvertFunctionAttr(Function &F, bool Set, StringRef FnAttrName)
static Value * upgradePMULDQ(IRBuilder<> &Builder, CallBase &CI, bool IsSigned)
static void reportFatalUsageErrorWithCI(StringRef reason, CallBase *CI)
static Value * upgradeMaskedStore(IRBuilder<> &Builder, Value *Ptr, Value *Data, Value *Mask, bool Aligned)
static Intrinsic::ID shouldUpgradeNVPTXBulkG2SCTAIntrinsic(Function *F, StringRef Name)
static Intrinsic::ID shouldUpgradeNVPTXTMAG2SCTAIntrinsics(Function *F, StringRef Name)
static Value * upgradeConvertIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
#define G2S_CTA_CASE(ID_SUFFIX, NAME)
static bool upgradeX86MultiplyAddWords(Function *F, Intrinsic::ID IID, Function *&NewFn)
static bool upgradePtrauthInitFiniArrays(Module &M)
static Value * upgradeX86IntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static FCmpInst::Predicate getVPFPPredicateFromMD(const Value *Op)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
@ Enable
This file contains constants used for implementing Dwarf debug support.
Module.h This file contains the declarations for the Module class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
Definition Lint.cpp:540
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define R2(n)
This file contains the declarations for metadata subclasses.
#define T
#define T1
NVPTX address space definition.
uint64_t High
This file contains the definitions of the enumerations and flags associated with NVVM Intrinsics,...
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file contains some functions that are useful when dealing with strings.
This file implements the StringSwitch template, which mimics a switch() statement whose cases are str...
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
LocallyHashedType DenseMapInfo< LocallyHashedType >::Empty
static const X86InstrFMA3Group Groups[]
Value * RHS
Value * LHS
Class for arbitrary precision integers.
Definition APInt.h:78
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Class to represent array types.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
Type * getElementType() const
an instruction that atomically reads a memory location, combines it with another value,...
void setVolatile(bool V)
Specify whether this is a volatile RMW or not.
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Add
*p = old + v
@ FAdd
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ Min
*p = old <signed v ? old : v
@ And
*p = old & v
@ Xor
*p = old ^ v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
bool isFloatingPointOperation() const
This class stores enough information to efficiently remove some attributes from an existing AttrBuild...
AttributeMask & addAttribute(Attribute::AttrKind Val)
Add an attribute to the mask.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
static LLVM_ABI Attribute getWithStackAlignment(LLVMContext &Context, Align Alignment)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
CallingConv::ID getCallingConv() const
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Value * getArgOperand(unsigned i) const
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
This class represents a function call, abstracting a target machine's calling convention.
void setTailCallKind(TailCallKind TCK)
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
static LLVM_ABI bool castIsValid(Instruction::CastOps op, Type *SrcTy, Type *DstTy)
This method can be used to determine if a cast from SrcTy to DstTy using Opcode op is valid or not.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
static ConstantAsMetadata * get(Constant *C)
Definition Metadata.h:548
static LLVM_ABI Constant * getIntToPtr(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getPtrToInt(Constant *C, Type *Ty, bool OnlyIfReduced=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
Definition Constants.h:219
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
StructType * getType() const
Specialization - reduce amount of casting.
Definition Constants.h:661
static LLVM_ABI ConstantTokenNone * get(LLVMContext &Context)
Return the ConstantTokenNone.
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
DWARF expression.
static LLVM_ABI DIExpression * append(const DIExpression *Expr, ArrayRef< uint64_t > Ops)
Append the opcodes Ops to DIExpr.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
static LLVM_ABI DbgLabelRecord * createUnresolvedDbgLabelRecord(MDNode *Label)
For use during parsing; creates a DbgLabelRecord from as-of-yet unresolved MDNodes.
Base class for non-instruction debug metadata records that have positions within IR.
void setDebugLoc(DebugLoc Loc)
static LLVM_ABI DbgVariableRecord * createUnresolvedDbgVariableRecord(LocationType Type, Metadata *Val, MDNode *Variable, MDNode *Expression, MDNode *AssignID, Metadata *Address, MDNode *AddressExpression)
Used to create DbgVariableRecords during parsing, where some metadata references may still be unresol...
Diagnostic information for debug metadata version reporting.
Diagnostic information for stripping invalid debug metadata.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
void setApproxFunc(bool B=true)
Definition FMF.h:93
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
Class to represent function types.
unsigned getNumParams() const
Return the number of fixed parameters this function type requires.
Type * getParamType(unsigned i) const
Parameter type accessors.
Type * getReturnType() const
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
Definition Function.h:169
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
Intrinsic::ID getIntrinsicID() const LLVM_READONLY
getIntrinsicID - This method returns the ID number of the specified function, or Intrinsic::not_intri...
Definition Function.h:247
const Function & getFunction() const
Definition Function.h:167
void eraseFromParent()
eraseFromParent - This method unlinks 'this' from the containing module and deletes it.
Definition Function.cpp:455
size_t arg_size() const
Definition Function.h:886
Type * getReturnType() const
Returns the type of the ret val.
Definition Function.h:217
Argument * getArg(unsigned i) const
Definition Function.h:871
static LLVM_ABI GUID getGUIDAssumingExternalLinkage(StringRef GlobalName)
Return a 64-bit global unique ID constructed from the name of a global symbol.
Definition Globals.cpp:80
LinkageTypes getLinkage() const
uint64_t GUID
Declare a type to represent a global unique identifier for a global value.
static StringRef dropLLVMManglingEscape(StringRef Name)
If the given string begins with the GlobalValue name mangling escape character '\1',...
Module * getParent()
Get the module that this global value is contained inside of...
Type * getValueType() const
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
bool hasInitializer() const
Definitions have initializers, declarations don't.
PointerType * getPtrTy(unsigned AddrSpace=0)
Fetch the type representing a pointer.
Definition IRBuilder.h:574
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2918
Base class for instruction visitors.
Definition InstVisitor.h:78
bool isCast() const
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
bool isBinaryOp() const
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
bool isUnaryOp() const
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI SyncScope::ID getOrInsertSyncScopeID(StringRef SSN)
getOrInsertSyncScopeID - Maps synchronization scope name to synchronization scope ID.
An instruction for reading from memory.
LLVM_ABI MDNode * createRange(const APInt &Lo, const APInt &Hi)
Return metadata describing the range [Lo, Hi).
Definition MDBuilder.cpp:96
Metadata node.
Definition Metadata.h:1081
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1437
op_iterator op_end() const
Definition Metadata.h:1431
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1443
op_iterator op_begin() const
Definition Metadata.h:1427
LLVMContext & getContext() const
Definition Metadata.h:1245
Tracking metadata reference owned by Metadata.
Definition Metadata.h:902
A single uniqued string.
Definition Metadata.h:733
LLVM_ABI StringRef getString() const
Definition Metadata.cpp:615
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:597
Tuple of metadata.
Definition Metadata.h:1496
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1525
Metadata wrapper in the Value hierarchy.
Definition Metadata.h:184
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:107
Root of the metadata hierarchy.
Definition Metadata.h:64
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
ModFlagBehavior
This enumeration defines the supported behaviors of module flags.
Definition Module.h:118
@ Override
Uses the specified value, regardless of the behavior or value of the other module.
Definition Module.h:139
@ Error
Emits an error if two values disagree, otherwise the resulting value is that of the operands.
Definition Module.h:121
@ Min
Takes the min of the two values, which are required to be integers.
Definition Module.h:153
@ Max
Takes the max of the two values, which are required to be integers.
Definition Module.h:150
A tuple of MDNodes.
Definition Metadata.h:1797
LLVM_ABI void setOperand(unsigned I, MDNode *New)
LLVM_ABI MDNode * getOperand(unsigned i) const
LLVM_ABI unsigned getNumOperands() const
LLVM_ABI void clearOperands()
Drop all references to this node's operands.
iterator_range< op_iterator > operands()
Definition Metadata.h:1893
LLVM_ABI void addOperand(MDNode *M)
ArrayRef< InputTy > inputs() const
StringRef getTag() const
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
LLVM_ABI bool match(StringRef String, SmallVectorImpl< StringRef > *Matches=nullptr, std::string *Error=nullptr) const
matches - Match the regex against a given String.
Definition Regex.cpp:84
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:865
ArrayRef< int > getShuffleMask() const
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
static constexpr size_t npos
Definition StringRef.h:58
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
Definition StringRef.h:597
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
Definition StringRef.h:635
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
size_t find(char C, size_t From=0) const
Search for the first character C in the string.
Definition StringRef.h:290
StringRef trim(char Char) const
Return string with consecutive Char characters starting from the left and right removed.
Definition StringRef.h:850
bool consume_front(char Prefix)
Returns true if this StringRef has the given prefix and removes that prefix.
Definition StringRef.h:661
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
StringSwitch & StartsWith(StringLiteral S, T Value)
StringSwitch & Cases(std::initializer_list< StringLiteral > CaseStrings, T Value)
Class to represent struct types.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
Definition Type.cpp:467
unsigned getNumElements() const
Random access to the elements.
Type * getElementType(unsigned N) const
The TimeTraceScope is a helper class to call the begin and end functions of the time trace profiler.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Definition Type.h:155
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:298
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:280
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
Definition Type.h:222
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
static LLVM_ABI Type * getBFloatTy(LLVMContext &C)
Definition Type.cpp:275
static LLVM_ABI Type * getHalfTy(LLVMContext &C)
Definition Type.cpp:274
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void print(raw_ostream &O, bool IsForDebug=false) const
Implement operator<< on Value.
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
Definition Value.cpp:394
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
iterator_range< user_iterator > users()
Definition Value.h:428
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:712
bool use_empty() const
Definition Value.h:348
bool hasName() const
Definition Value.h:263
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
A raw_ostream that writes to an SmallVector or SmallString.
StringRef str() const
Return a StringRef for the vector contents.
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ PRIVATE_ADDRESS
Address space for private memory.
@ PTX_Kernel
Call to a PTX kernel. Passes all arguments in parameter space.
std::optional< ABIType > parseABIType(StringRef S)
Parse the string spelling used by the "float-abi" IR module flag into an ABIType.
Definition CodeGen.h:167
LLVM_ABI std::optional< Function * > remangleIntrinsicFunction(Function *F)
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI AttributeList getAttributes(LLVMContext &C, ID id, FunctionType *FT)
Return the attributes for an intrinsic.
LLVM_ABI bool isOverloaded(ID id)
Returns true if the intrinsic can be overloaded.
LLVM_ABI FunctionType * getType(LLVMContext &Context, ID id, ArrayRef< Type * > OverloadTys={})
Return the function type for an intrinsic.
LLVM_ABI bool isSignatureValid(Intrinsic::ID ID, FunctionType *FT, SmallVectorImpl< Type * > &OverloadTys, raw_ostream &OS=nulls())
Returns true if FT is a valid function type for intrinsic ID.
LLVM_ABI bool hasStructReturnType(ID id)
Returns true if id has a struct return type.
LLVM_ABI std::pair< unsigned, ArrayRef< uint64_t > > getAllDefaultArgValues(ID IID)
Returns the first default argument index and an ArrayRef of all default values for the trailing param...
constexpr StringLiteral GridConstant("nvvm.grid_constant")
constexpr StringLiteral MaxNTID("nvvm.maxntid")
constexpr StringLiteral MaxNReg("nvvm.maxnreg")
constexpr StringLiteral MinCTASm("nvvm.minctasm")
constexpr StringLiteral ReqNTID("nvvm.reqntid")
constexpr StringLiteral MaxClusterRank("nvvm.maxclusterrank")
constexpr StringLiteral ClusterDim("nvvm.cluster_dim")
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > dyn_extract_or_null(Y &&MD)
Extract a Value from Metadata, if any, allowing null.
Definition Metadata.h:720
std::enable_if_t< detail::IsValidPointer< X, Y >::value, bool > hasa(Y &&MD)
Check whether Metadata has a Value.
Definition Metadata.h:662
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > dyn_extract(Y &&MD)
Extract a Value from Metadata, if any.
Definition Metadata.h:707
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
@ Length
Definition DWP.cpp:577
LLVM_ABI void UpgradeIntrinsicCall(CallBase *CB, Function *NewFn)
This is the complement to the above, replacing a specific call to an intrinsic function with a call t...
LLVM_ABI void UpgradeSectionAttributes(Module &M)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
LLVM_ABI void UpgradeInlineAsmString(std::string *AsmStr)
Upgrade comment in call to inline asm that represents an objc retain release marker.
bool isValidAtomicOrdering(Int I)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
StringRef getLongDoubleFormatName(LongDoubleFormat Format)
Returns the IR floating-point type name for a LongDoubleFormat.
Definition CodeGen.h:126
LongDoubleFormat
The floating-point format used for the target's "long double" type.
Definition CodeGen.h:117
LLVM_ABI bool UpgradeIntrinsicFunction(Function *F, Function *&NewFn, bool CanUpgradeDebugIntrinsicsToRecords=true)
This is a more granular function that simply checks an intrinsic function for upgrading,...
LLVM_ABI MDNode * upgradeInstructionLoopAttachment(MDNode &N)
Upgrade the loop attachment metadata node.
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_ABI void UpgradeAttributes(AttrBuilder &B)
Upgrade attributes that changed format or kind.
LLVM_ABI void UpgradeCallsToIntrinsic(Function *F)
This is an auto-upgrade hook for any old intrinsic function syntaxes which need to have both the func...
LLVM_ABI void UpgradeNVVMAnnotations(Module &M)
Convert legacy nvvm.annotations metadata to appropriate function attributes.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
LLVM_ABI bool UpgradeModuleFlags(Module &M)
This checks for module flags which should be upgraded.
std::string utostr(uint64_t X, bool isNeg=false)
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool UpgradeCFIFunctionsMetadata(Module &M)
Upgrade the cfi.functions metadata node by calculating and inserting the GUID for each function entry...
LLVM_ABI void copyModuleAttrToFunctions(Module &M)
Copies module attributes to the functions in the module.
LLVM_ABI void UpgradeOperandBundles(std::vector< OperandBundleDef > &OperandBundles)
Upgrade operand bundles (without knowing about their user instruction).
LLVM_ABI Constant * UpgradeBitCastExpr(unsigned Opc, Constant *C, Type *DestTy)
This is an auto-upgrade for bitcast constant expression between pointers with different address space...
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI std::string UpgradeDataLayoutString(StringRef DL, StringRef Triple)
Upgrade the datalayout string by adding a section for address space pointers.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
LLVM_ABI MDNode * UpgradeTBAAStructNode(MDNode &TBAAStructNode)
If the given !tbaa.struct node has old-style scalar field tags, return an equivalent node with each f...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI GlobalVariable * UpgradeGlobalVariable(GlobalVariable *GV)
This checks for global variables which should be upgraded.
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
LLVM_ABI bool StripDebugInfo(Module &M)
Strip debug info in the module if it exists.
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Ref
The access may reference the value stored in memory.
Definition ModRef.h:32
std::string join(IteratorT Begin, IteratorT End, StringRef Separator)
Joins the strings in the range [Begin, End), adding Separator between the elements.
const BooleanLoopTags * findBooleanLoopTags(StringRef Name)
Return the replacement tags for the enable tag Name, or nullptr.
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI Instruction * UpgradeBitCastInst(unsigned Opc, Value *V, Type *DestTy, Instruction *&Temp)
This is an auto-upgrade for bitcast between pointers with different address spaces: the instruction i...
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ TowardZero
roundTowardZero.
@ NearestTiesToEven
roundTiesToEven.
@ Dynamic
Denotes mode unknown at compile time.
@ TowardPositive
roundTowardPositive.
@ TowardNegative
roundTowardNegative.
ArrayRef(const T &OneElt) -> ArrayRef< T >
DenormalMode parseDenormalFPAttribute(StringRef Str)
Returns the denormal mode to use for inputs and outputs.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
LLVM_ABI bool UpgradeDebugInfo(Module &M)
Check the debug info version number, if it is out-dated, drop the debug info.
LLVM_ABI void UpgradeFunctionAttributes(Function &F)
Correct any IR that is relying on old function attribute behavior.
LLVM_ABI MDNode * UpgradeTBAANode(MDNode &TBAANode)
If the given TBAA tag uses the scalar TBAA format, create a new node corresponding to the upgrade to ...
LLVM_ABI void UpgradeARCRuntime(Module &M)
Convert calls to ARC runtime functions to intrinsic calls and upgrade the old retain release marker t...
@ DEBUG_METADATA_VERSION
Definition Metadata.h:54
LLVM_ABI bool verifyModule(const Module &M, raw_ostream *OS=nullptr, bool *BrokenDebugInfo=nullptr)
Check a module for errors.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Single-operand tags replacing a removed two-operand form !
StringLiteral Disable
StringLiteral Enable
Represents the full denormal controls for a function, including the default mode and the f32 specific...
Represent subnormal handling kind for floating point instruction inputs and outputs.
static constexpr DenormalMode getInvalid()
constexpr bool isValid() const
static constexpr DenormalMode getIEEE()
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106