LLVM 24.0.0git
AutoUpgrade.cpp
Go to the documentation of this file.
1//===-- AutoUpgrade.cpp - Implement auto-upgrade helper functions ---------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the auto-upgrade helper functions.
10// This is where deprecated IR intrinsics and other IR features are updated to
11// current specifications.
12//
13//===----------------------------------------------------------------------===//
14
15#include "llvm/IR/AutoUpgrade.h"
16#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/StringRef.h"
22#include "llvm/IR/Attributes.h"
23#include "llvm/IR/CallingConv.h"
24#include "llvm/IR/Constants.h"
25#include "llvm/IR/DebugInfo.h"
28#include "llvm/IR/Function.h"
29#include "llvm/IR/GlobalValue.h"
30#include "llvm/IR/IRBuilder.h"
31#include "llvm/IR/InstVisitor.h"
32#include "llvm/IR/Instruction.h"
34#include "llvm/IR/Intrinsics.h"
35#include "llvm/IR/IntrinsicsAArch64.h"
36#include "llvm/IR/IntrinsicsAMDGPU.h"
37#include "llvm/IR/IntrinsicsARM.h"
38#include "llvm/IR/IntrinsicsNVPTX.h"
39#include "llvm/IR/IntrinsicsRISCV.h"
40#include "llvm/IR/IntrinsicsWebAssembly.h"
41#include "llvm/IR/IntrinsicsX86.h"
42#include "llvm/IR/LLVMContext.h"
43#include "llvm/IR/MDBuilder.h"
44#include "llvm/IR/Metadata.h"
45#include "llvm/IR/Module.h"
47#include "llvm/IR/Value.h"
48#include "llvm/IR/Verifier.h"
55#include "llvm/Support/Regex.h"
58#include <cstdint>
59#include <cstring>
60#include <numeric>
61
62using namespace llvm;
63
64static cl::opt<bool>
65 DisableAutoUpgradeDebugInfo("disable-auto-upgrade-debug-info",
66 cl::desc("Disable autoupgrade of debug info"));
67
68static void rename(GlobalValue *GV) { GV->setName(GV->getName() + ".old"); }
69
70// Report a fatal error along with the
71// Call Instruction which caused the error
72[[noreturn]] static void reportFatalUsageErrorWithCI(StringRef reason,
73 CallBase *CI) {
74 CI->print(llvm::errs());
75 llvm::errs() << "\n";
77}
78
79// Upgrade the declarations of the SSE4.1 ptest intrinsics whose arguments have
80// changed their type from v4f32 to v2i64.
82 Function *&NewFn) {
83 // Check whether this is an old version of the function, which received
84 // v4f32 arguments.
85 Type *Arg0Type = F->getFunctionType()->getParamType(0);
86 if (Arg0Type != FixedVectorType::get(Type::getFloatTy(F->getContext()), 4))
87 return false;
88
89 // Yes, it's old, replace it with new version.
90 rename(F);
91 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
92 return true;
93}
94
95// Upgrade the declarations of intrinsic functions whose 8-bit immediate mask
96// arguments have changed their type from i32 to i8.
98 Function *&NewFn) {
99 // Check that the last argument is an i32.
100 Type *LastArgType = F->getFunctionType()->getParamType(
101 F->getFunctionType()->getNumParams() - 1);
102 if (!LastArgType->isIntegerTy(32))
103 return false;
104
105 // Move this function aside and map down.
106 rename(F);
107 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
108 return true;
109}
110
111// Upgrade the declaration of fp compare intrinsics that change return type
112// from scalar to vXi1 mask.
114 Function *&NewFn) {
115 // Check if the return type is a vector.
116 if (F->getReturnType()->isVectorTy())
117 return false;
118
119 rename(F);
120 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
121 return true;
122}
123
124// Upgrade the declaration of multiply and add bytes intrinsics whose input
125// arguments' types have changed from vectors of i32 to vectors of i8
127 Function *&NewFn) {
128 // check if input argument type is a vector of i8
129 Type *Arg1Type = F->getFunctionType()->getParamType(1);
130 Type *Arg2Type = F->getFunctionType()->getParamType(2);
131 if (Arg1Type->isVectorTy() &&
132 cast<VectorType>(Arg1Type)->getElementType()->isIntegerTy(8) &&
133 Arg2Type->isVectorTy() &&
134 cast<VectorType>(Arg2Type)->getElementType()->isIntegerTy(8))
135 return false;
136
137 rename(F);
138 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
139 return true;
140}
141
142// Upgrade the declaration of multipy and add words intrinsics whose input
143// arguments' types have changed to vectors of i32 to vectors of i16
145 Function *&NewFn) {
146 // check if input argument type is a vector of i16
147 Type *Arg1Type = F->getFunctionType()->getParamType(1);
148 Type *Arg2Type = F->getFunctionType()->getParamType(2);
149 if (Arg1Type->isVectorTy() &&
150 cast<VectorType>(Arg1Type)->getElementType()->isIntegerTy(16) &&
151 Arg2Type->isVectorTy() &&
152 cast<VectorType>(Arg2Type)->getElementType()->isIntegerTy(16))
153 return false;
154
155 rename(F);
156 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
157 return true;
158}
159
161 Function *&NewFn) {
162 if (F->getReturnType()->getScalarType()->isBFloatTy())
163 return false;
164
165 rename(F);
166 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
167 return true;
168}
169
171 Function *&NewFn) {
172 if (F->getFunctionType()->getParamType(1)->getScalarType()->isBFloatTy())
173 return false;
174
175 rename(F);
176 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
177 return true;
178}
179
181 // All of the intrinsics matches below should be marked with which llvm
182 // version started autoupgrading them. At some point in the future we would
183 // like to use this information to remove upgrade code for some older
184 // intrinsics. It is currently undecided how we will determine that future
185 // point.
186 if (Name.consume_front("avx."))
187 return (Name.starts_with("blend.p") || // Added in 3.7
188 Name == "cvt.ps2.pd.256" || // Added in 3.9
189 Name == "cvtdq2.pd.256" || // Added in 3.9
190 Name == "cvtdq2.ps.256" || // Added in 7.0
191 Name.starts_with("movnt.") || // Added in 3.2
192 Name.starts_with("sqrt.p") || // Added in 7.0
193 Name.starts_with("storeu.") || // Added in 3.9
194 Name.starts_with("vbroadcast.s") || // Added in 3.5
195 Name.starts_with("vbroadcastf128") || // Added in 4.0
196 Name.starts_with("vextractf128.") || // Added in 3.7
197 Name.starts_with("vinsertf128.") || // Added in 3.7
198 Name.starts_with("vperm2f128.") || // Added in 6.0
199 Name.starts_with("vpermil.")); // Added in 3.1
200
201 if (Name.consume_front("avx2."))
202 return (Name == "movntdqa" || // Added in 5.0
203 Name.starts_with("pabs.") || // Added in 6.0
204 Name.starts_with("padds.") || // Added in 8.0
205 Name.starts_with("paddus.") || // Added in 8.0
206 Name.starts_with("pblendd.") || // Added in 3.7
207 Name == "pblendw" || // Added in 3.7
208 Name.starts_with("pbroadcast") || // Added in 3.8
209 Name.starts_with("pcmpeq.") || // Added in 3.1
210 Name.starts_with("pcmpgt.") || // Added in 3.1
211 Name.starts_with("pmax") || // Added in 3.9
212 Name.starts_with("pmin") || // Added in 3.9
213 Name.starts_with("pmovsx") || // Added in 3.9
214 Name.starts_with("pmovzx") || // Added in 3.9
215 Name.starts_with("pmulh.w") || // Added in 24.0
216 Name.starts_with("pmulhu.w") || // Added in 24.0
217 Name == "pmul.dq" || // Added in 7.0
218 Name == "pmulu.dq" || // Added in 7.0
219 Name.starts_with("psll.dq") || // Added in 3.7
220 Name.starts_with("psrl.dq") || // Added in 3.7
221 Name.starts_with("psubs.") || // Added in 8.0
222 Name.starts_with("psubus.") || // Added in 8.0
223 Name.starts_with("vbroadcast") || // Added in 3.8
224 Name == "vbroadcasti128" || // Added in 3.7
225 Name == "vextracti128" || // Added in 3.7
226 Name == "vinserti128" || // Added in 3.7
227 Name == "vperm2i128"); // Added in 6.0
228
229 if (Name.consume_front("avx512.")) {
230 if (Name.consume_front("mask."))
231 // 'avx512.mask.*'
232 return (Name.starts_with("add.p") || // Added in 7.0. 128/256 in 4.0
233 Name.starts_with("and.") || // Added in 3.9
234 Name.starts_with("andn.") || // Added in 3.9
235 Name.starts_with("broadcast.s") || // Added in 3.9
236 Name.starts_with("broadcastf32x4.") || // Added in 6.0
237 Name.starts_with("broadcastf32x8.") || // Added in 6.0
238 Name.starts_with("broadcastf64x2.") || // Added in 6.0
239 Name.starts_with("broadcastf64x4.") || // Added in 6.0
240 Name.starts_with("broadcasti32x4.") || // Added in 6.0
241 Name.starts_with("broadcasti32x8.") || // Added in 6.0
242 Name.starts_with("broadcasti64x2.") || // Added in 6.0
243 Name.starts_with("broadcasti64x4.") || // Added in 6.0
244 Name.starts_with("cmp.b") || // Added in 5.0
245 Name.starts_with("cmp.d") || // Added in 5.0
246 Name.starts_with("cmp.q") || // Added in 5.0
247 Name.starts_with("cmp.w") || // Added in 5.0
248 Name.starts_with("compress.b") || // Added in 9.0
249 Name.starts_with("compress.d") || // Added in 9.0
250 Name.starts_with("compress.p") || // Added in 9.0
251 Name.starts_with("compress.q") || // Added in 9.0
252 Name.starts_with("compress.store.") || // Added in 7.0
253 Name.starts_with("compress.w") || // Added in 9.0
254 Name.starts_with("conflict.") || // Added in 9.0
255 Name.starts_with("cvtdq2pd.") || // Added in 4.0
256 Name.starts_with("cvtdq2ps.") || // Added in 7.0 updated 9.0
257 Name == "cvtpd2dq.256" || // Added in 7.0
258 Name == "cvtpd2ps.256" || // Added in 7.0
259 Name == "cvtps2pd.128" || // Added in 7.0
260 Name == "cvtps2pd.256" || // Added in 7.0
261 Name.starts_with("cvtqq2pd.") || // Added in 7.0 updated 9.0
262 Name == "cvtqq2ps.256" || // Added in 9.0
263 Name == "cvtqq2ps.512" || // Added in 9.0
264 Name == "cvttpd2dq.256" || // Added in 7.0
265 Name == "cvttps2dq.128" || // Added in 7.0
266 Name == "cvttps2dq.256" || // Added in 7.0
267 Name.starts_with("cvtudq2pd.") || // Added in 4.0
268 Name.starts_with("cvtudq2ps.") || // Added in 7.0 updated 9.0
269 Name.starts_with("cvtuqq2pd.") || // Added in 7.0 updated 9.0
270 Name == "cvtuqq2ps.256" || // Added in 9.0
271 Name == "cvtuqq2ps.512" || // Added in 9.0
272 Name.starts_with("dbpsadbw.") || // Added in 7.0
273 Name.starts_with("div.p") || // Added in 7.0. 128/256 in 4.0
274 Name.starts_with("expand.b") || // Added in 9.0
275 Name.starts_with("expand.d") || // Added in 9.0
276 Name.starts_with("expand.load.") || // Added in 7.0
277 Name.starts_with("expand.p") || // Added in 9.0
278 Name.starts_with("expand.q") || // Added in 9.0
279 Name.starts_with("expand.w") || // Added in 9.0
280 Name.starts_with("fpclass.p") || // Added in 7.0
281 Name.starts_with("insert") || // Added in 4.0
282 Name.starts_with("load.") || // Added in 3.9
283 Name.starts_with("loadu.") || // Added in 3.9
284 Name.starts_with("lzcnt.") || // Added in 5.0
285 Name.starts_with("max.p") || // Added in 7.0. 128/256 in 5.0
286 Name.starts_with("min.p") || // Added in 7.0. 128/256 in 5.0
287 Name.starts_with("movddup") || // Added in 3.9
288 Name.starts_with("move.s") || // Added in 4.0
289 Name.starts_with("movshdup") || // Added in 3.9
290 Name.starts_with("movsldup") || // Added in 3.9
291 Name.starts_with("mul.p") || // Added in 7.0. 128/256 in 4.0
292 Name.starts_with("or.") || // Added in 3.9
293 Name.starts_with("pabs.") || // Added in 6.0
294 Name.starts_with("packssdw.") || // Added in 5.0
295 Name.starts_with("packsswb.") || // Added in 5.0
296 Name.starts_with("packusdw.") || // Added in 5.0
297 Name.starts_with("packuswb.") || // Added in 5.0
298 Name.starts_with("padd.") || // Added in 4.0
299 Name.starts_with("padds.") || // Added in 8.0
300 Name.starts_with("paddus.") || // Added in 8.0
301 Name.starts_with("palignr.") || // Added in 3.9
302 Name.starts_with("pand.") || // Added in 3.9
303 Name.starts_with("pandn.") || // Added in 3.9
304 Name.starts_with("pavg") || // Added in 6.0
305 Name.starts_with("pbroadcast") || // Added in 6.0
306 Name.starts_with("pcmpeq.") || // Added in 3.9
307 Name.starts_with("pcmpgt.") || // Added in 3.9
308 Name.starts_with("perm.df.") || // Added in 3.9
309 Name.starts_with("perm.di.") || // Added in 3.9
310 Name.starts_with("permvar.") || // Added in 7.0
311 Name.starts_with("pmaddubs.w.") || // Added in 7.0
312 Name.starts_with("pmaddw.d.") || // Added in 7.0
313 Name.starts_with("pmax") || // Added in 4.0
314 Name.starts_with("pmin") || // Added in 4.0
315 Name == "pmov.qd.256" || // Added in 9.0
316 Name == "pmov.qd.512" || // Added in 9.0
317 Name == "pmov.wb.256" || // Added in 9.0
318 Name == "pmov.wb.512" || // Added in 9.0
319 Name.starts_with("pmovsx") || // Added in 4.0
320 Name.starts_with("pmovzx") || // Added in 4.0
321 Name.starts_with("pmul.dq.") || // Added in 4.0
322 Name.starts_with("pmul.hr.sw.") || // Added in 7.0
323 Name.starts_with("pmulh.w.") || // Added in 7.0
324 Name.starts_with("pmulhu.w.") || // Added in 7.0
325 Name.starts_with("pmull.") || // Added in 4.0
326 Name.starts_with("pmultishift.qb.") || // Added in 8.0
327 Name.starts_with("pmulu.dq.") || // Added in 4.0
328 Name.starts_with("por.") || // Added in 3.9
329 Name.starts_with("prol.") || // Added in 8.0
330 Name.starts_with("prolv.") || // Added in 8.0
331 Name.starts_with("pror.") || // Added in 8.0
332 Name.starts_with("prorv.") || // Added in 8.0
333 Name.starts_with("pshuf.b.") || // Added in 4.0
334 Name.starts_with("pshuf.d.") || // Added in 3.9
335 Name.starts_with("pshufh.w.") || // Added in 3.9
336 Name.starts_with("pshufl.w.") || // Added in 3.9
337 Name.starts_with("psll.d") || // Added in 4.0
338 Name.starts_with("psll.q") || // Added in 4.0
339 Name.starts_with("psll.w") || // Added in 4.0
340 Name.starts_with("pslli") || // Added in 4.0
341 Name.starts_with("psllv") || // Added in 4.0
342 Name.starts_with("psra.d") || // Added in 4.0
343 Name.starts_with("psra.q") || // Added in 4.0
344 Name.starts_with("psra.w") || // Added in 4.0
345 Name.starts_with("psrai") || // Added in 4.0
346 Name.starts_with("psrav") || // Added in 4.0
347 Name.starts_with("psrl.d") || // Added in 4.0
348 Name.starts_with("psrl.q") || // Added in 4.0
349 Name.starts_with("psrl.w") || // Added in 4.0
350 Name.starts_with("psrli") || // Added in 4.0
351 Name.starts_with("psrlv") || // Added in 4.0
352 Name.starts_with("psub.") || // Added in 4.0
353 Name.starts_with("psubs.") || // Added in 8.0
354 Name.starts_with("psubus.") || // Added in 8.0
355 Name.starts_with("pternlog.") || // Added in 7.0
356 Name.starts_with("punpckh") || // Added in 3.9
357 Name.starts_with("punpckl") || // Added in 3.9
358 Name.starts_with("pxor.") || // Added in 3.9
359 Name.starts_with("shuf.f") || // Added in 6.0
360 Name.starts_with("shuf.i") || // Added in 6.0
361 Name.starts_with("shuf.p") || // Added in 4.0
362 Name.starts_with("sqrt.p") || // Added in 7.0
363 Name.starts_with("store.b.") || // Added in 3.9
364 Name.starts_with("store.d.") || // Added in 3.9
365 Name.starts_with("store.p") || // Added in 3.9
366 Name.starts_with("store.q.") || // Added in 3.9
367 Name.starts_with("store.w.") || // Added in 3.9
368 Name == "store.ss" || // Added in 7.0
369 Name.starts_with("storeu.") || // Added in 3.9
370 Name.starts_with("sub.p") || // Added in 7.0. 128/256 in 4.0
371 Name.starts_with("ucmp.") || // Added in 5.0
372 Name.starts_with("unpckh.") || // Added in 3.9
373 Name.starts_with("unpckl.") || // Added in 3.9
374 Name.starts_with("valign.") || // Added in 4.0
375 Name == "vcvtph2ps.128" || // Added in 11.0
376 Name == "vcvtph2ps.256" || // Added in 11.0
377 Name.starts_with("vextract") || // Added in 4.0
378 Name.starts_with("vfmadd.") || // Added in 7.0
379 Name.starts_with("vfmaddsub.") || // Added in 7.0
380 Name.starts_with("vfnmadd.") || // Added in 7.0
381 Name.starts_with("vfnmsub.") || // Added in 7.0
382 Name.starts_with("vpdpbusd.") || // Added in 7.0
383 Name.starts_with("vpdpbusds.") || // Added in 7.0
384 Name.starts_with("vpdpwssd.") || // Added in 7.0
385 Name.starts_with("vpdpwssds.") || // Added in 7.0
386 Name.starts_with("vpermi2var.") || // Added in 7.0
387 Name.starts_with("vpermil.p") || // Added in 3.9
388 Name.starts_with("vpermilvar.") || // Added in 4.0
389 Name.starts_with("vpermt2var.") || // Added in 7.0
390 Name.starts_with("vpmadd52") || // Added in 7.0
391 Name.starts_with("vpshld.") || // Added in 7.0
392 Name.starts_with("vpshldv.") || // Added in 8.0
393 Name.starts_with("vpshrd.") || // Added in 7.0
394 Name.starts_with("vpshrdv.") || // Added in 8.0
395 Name.starts_with("vpshufbitqmb.") || // Added in 8.0
396 Name.starts_with("xor.")); // Added in 3.9
397
398 if (Name.consume_front("mask3."))
399 // 'avx512.mask3.*'
400 return (Name.starts_with("vfmadd.") || // Added in 7.0
401 Name.starts_with("vfmaddsub.") || // Added in 7.0
402 Name.starts_with("vfmsub.") || // Added in 7.0
403 Name.starts_with("vfmsubadd.") || // Added in 7.0
404 Name.starts_with("vfnmsub.")); // Added in 7.0
405
406 if (Name.consume_front("maskz."))
407 // 'avx512.maskz.*'
408 return (Name.starts_with("pternlog.") || // Added in 7.0
409 Name.starts_with("vfmadd.") || // Added in 7.0
410 Name.starts_with("vfmaddsub.") || // Added in 7.0
411 Name.starts_with("vpdpbusd.") || // Added in 7.0
412 Name.starts_with("vpdpbusds.") || // Added in 7.0
413 Name.starts_with("vpdpwssd.") || // Added in 7.0
414 Name.starts_with("vpdpwssds.") || // Added in 7.0
415 Name.starts_with("vpermt2var.") || // Added in 7.0
416 Name.starts_with("vpmadd52") || // Added in 7.0
417 Name.starts_with("vpshldv.") || // Added in 8.0
418 Name.starts_with("vpshrdv.")); // Added in 8.0
419
420 // 'avx512.*'
421 return (Name == "movntdqa" || // Added in 5.0
422 Name == "pmul.dq.512" || // Added in 7.0
423 Name == "pmulu.dq.512" || // Added in 7.0
424 Name.starts_with("broadcastm") || // Added in 6.0
425 Name.starts_with("cmp.p") || // Added in 12.0
426 Name.starts_with("cvtb2mask.") || // Added in 7.0
427 Name.starts_with("cvtd2mask.") || // Added in 7.0
428 Name.starts_with("cvtmask2") || // Added in 5.0
429 Name.starts_with("cvtq2mask.") || // Added in 7.0
430 Name == "cvtusi2sd" || // Added in 7.0
431 Name.starts_with("cvtw2mask.") || // Added in 7.0
432 Name == "kand.w" || // Added in 7.0
433 Name == "kandn.w" || // Added in 7.0
434 Name == "knot.w" || // Added in 7.0
435 Name == "kor.w" || // Added in 7.0
436 Name == "kortestc.w" || // Added in 7.0
437 Name == "kortestz.w" || // Added in 7.0
438 Name.starts_with("kunpck") || // added in 6.0
439 Name == "kxnor.w" || // Added in 7.0
440 Name == "kxor.w" || // Added in 7.0
441 Name.starts_with("padds.") || // Added in 8.0
442 Name.starts_with("pbroadcast") || // Added in 3.9
443 Name.starts_with("pmulh.w") || // Added in 24.0
444 Name.starts_with("pmulhu.w") || // Added in 24.0
445 Name.starts_with("prol") || // Added in 8.0
446 Name.starts_with("pror") || // Added in 8.0
447 Name.starts_with("psll.dq") || // Added in 3.9
448 Name.starts_with("psrl.dq") || // Added in 3.9
449 Name.starts_with("psubs.") || // Added in 8.0
450 Name.starts_with("ptestm") || // Added in 6.0
451 Name.starts_with("ptestnm") || // Added in 6.0
452 Name.starts_with("storent.") || // Added in 3.9
453 Name.starts_with("vbroadcast.s") || // Added in 7.0
454 Name.starts_with("vpshld.") || // Added in 8.0
455 Name.starts_with("vpshrd.")); // Added in 8.0
456 }
457
458 if (Name.consume_front("fma."))
459 return (Name.starts_with("vfmadd.") || // Added in 7.0
460 Name.starts_with("vfmsub.") || // Added in 7.0
461 Name.starts_with("vfmsubadd.") || // Added in 7.0
462 Name.starts_with("vfnmadd.") || // Added in 7.0
463 Name.starts_with("vfnmsub.")); // Added in 7.0
464
465 if (Name.consume_front("fma4."))
466 return Name.starts_with("vfmadd.s"); // Added in 7.0
467
468 if (Name.consume_front("sse."))
469 return (Name == "add.ss" || // Added in 4.0
470 Name == "cvtsi2ss" || // Added in 7.0
471 Name == "cvtsi642ss" || // Added in 7.0
472 Name == "div.ss" || // Added in 4.0
473 Name == "mul.ss" || // Added in 4.0
474 Name.starts_with("sqrt.p") || // Added in 7.0
475 Name == "sqrt.ss" || // Added in 7.0
476 Name.starts_with("storeu.") || // Added in 3.9
477 Name == "sub.ss"); // Added in 4.0
478
479 if (Name.consume_front("sse2."))
480 return (Name == "add.sd" || // Added in 4.0
481 Name == "cvtdq2pd" || // Added in 3.9
482 Name == "cvtdq2ps" || // Added in 7.0
483 Name == "cvtps2pd" || // Added in 3.9
484 Name == "cvtsi2sd" || // Added in 7.0
485 Name == "cvtsi642sd" || // Added in 7.0
486 Name == "cvtss2sd" || // Added in 7.0
487 Name == "div.sd" || // Added in 4.0
488 Name == "mul.sd" || // Added in 4.0
489 Name.starts_with("padds.") || // Added in 8.0
490 Name.starts_with("paddus.") || // Added in 8.0
491 Name.starts_with("pcmpeq.") || // Added in 3.1
492 Name.starts_with("pcmpgt.") || // Added in 3.1
493 Name == "pmaxs.w" || // Added in 3.9
494 Name == "pmaxu.b" || // Added in 3.9
495 Name == "pmins.w" || // Added in 3.9
496 Name == "pminu.b" || // Added in 3.9
497 Name == "pmulh.w" || // Added in 24.0
498 Name == "pmulhu.w" || // Added in 24.0
499 Name == "pmulu.dq" || // Added in 7.0
500 Name.starts_with("pshuf") || // Added in 3.9
501 Name.starts_with("psll.dq") || // Added in 3.7
502 Name.starts_with("psrl.dq") || // Added in 3.7
503 Name.starts_with("psubs.") || // Added in 8.0
504 Name.starts_with("psubus.") || // Added in 8.0
505 Name.starts_with("sqrt.p") || // Added in 7.0
506 Name == "sqrt.sd" || // Added in 7.0
507 Name == "storel.dq" || // Added in 3.9
508 Name.starts_with("storeu.") || // Added in 3.9
509 Name == "sub.sd"); // Added in 4.0
510
511 if (Name.consume_front("sse41."))
512 return (Name.starts_with("blendp") || // Added in 3.7
513 Name == "movntdqa" || // Added in 5.0
514 Name == "pblendw" || // Added in 3.7
515 Name == "pmaxsb" || // Added in 3.9
516 Name == "pmaxsd" || // Added in 3.9
517 Name == "pmaxud" || // Added in 3.9
518 Name == "pmaxuw" || // Added in 3.9
519 Name == "pminsb" || // Added in 3.9
520 Name == "pminsd" || // Added in 3.9
521 Name == "pminud" || // Added in 3.9
522 Name == "pminuw" || // Added in 3.9
523 Name.starts_with("pmovsx") || // Added in 3.8
524 Name.starts_with("pmovzx") || // Added in 3.9
525 Name == "pmuldq"); // Added in 7.0
526
527 if (Name.consume_front("sse42."))
528 return Name == "crc32.64.8"; // Added in 3.4
529
530 if (Name.consume_front("sse4a."))
531 return Name.starts_with("movnt."); // Added in 3.9
532
533 if (Name.consume_front("ssse3."))
534 return (Name == "pabs.b.128" || // Added in 6.0
535 Name == "pabs.d.128" || // Added in 6.0
536 Name == "pabs.w.128"); // Added in 6.0
537
538 if (Name.consume_front("xop."))
539 return (Name == "vpcmov" || // Added in 3.8
540 Name == "vpcmov.256" || // Added in 5.0
541 Name.starts_with("vpcom") || // Added in 3.2, Updated in 9.0
542 Name.starts_with("vprot")); // Added in 8.0
543
544 if (Name.consume_front("bmi."))
545 return (Name.starts_with("pdep.") || // Added in 23.0
546 Name.starts_with("pext.")); // Added in 23.0
547
548 return (Name == "addcarry.u32" || // Added in 8.0
549 Name == "addcarry.u64" || // Added in 8.0
550 Name == "addcarryx.u32" || // Added in 8.0
551 Name == "addcarryx.u64" || // Added in 8.0
552 Name == "subborrow.u32" || // Added in 8.0
553 Name == "subborrow.u64" || // Added in 8.0
554 Name.starts_with("vcvtph2ps.")); // Added in 11.0
555}
556
558 Function *&NewFn) {
559 // Only handle intrinsics that start with "x86.".
560 if (!Name.consume_front("x86."))
561 return false;
562
563 if (shouldUpgradeX86Intrinsic(F, Name)) {
564 NewFn = nullptr;
565 return true;
566 }
567
568 if (Name == "rdtscp") { // Added in 8.0
569 // If this intrinsic has 0 operands, it's the new version.
570 if (F->getFunctionType()->getNumParams() == 0)
571 return false;
572
573 rename(F);
574 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
575 Intrinsic::x86_rdtscp);
576 return true;
577 }
578
579 Intrinsic::ID ID;
580
581 // SSE4.1 ptest functions may have an old signature.
582 if (Name.consume_front("sse41.ptest")) { // Added in 3.2
584 .Case("c", Intrinsic::x86_sse41_ptestc)
585 .Case("z", Intrinsic::x86_sse41_ptestz)
586 .Case("nzc", Intrinsic::x86_sse41_ptestnzc)
588 if (ID != Intrinsic::not_intrinsic)
589 return upgradePTESTIntrinsic(F, ID, NewFn);
590
591 return false;
592 }
593
594 // Several blend and other instructions with masks used the wrong number of
595 // bits.
596
597 // Added in 3.6
599 .Case("sse41.insertps", Intrinsic::x86_sse41_insertps)
600 .Case("sse41.dppd", Intrinsic::x86_sse41_dppd)
601 .Case("sse41.dpps", Intrinsic::x86_sse41_dpps)
602 .Case("sse41.mpsadbw", Intrinsic::x86_sse41_mpsadbw)
603 .Case("avx.dp.ps.256", Intrinsic::x86_avx_dp_ps_256)
604 .Case("avx2.mpsadbw", Intrinsic::x86_avx2_mpsadbw)
606 if (ID != Intrinsic::not_intrinsic)
607 return upgradeX86IntrinsicsWith8BitMask(F, ID, NewFn);
608
609 if (Name.consume_front("avx512.")) {
610 if (Name.consume_front("mask.cmp.")) {
611 // Added in 7.0
613 .Case("pd.128", Intrinsic::x86_avx512_mask_cmp_pd_128)
614 .Case("pd.256", Intrinsic::x86_avx512_mask_cmp_pd_256)
615 .Case("pd.512", Intrinsic::x86_avx512_mask_cmp_pd_512)
616 .Case("ps.128", Intrinsic::x86_avx512_mask_cmp_ps_128)
617 .Case("ps.256", Intrinsic::x86_avx512_mask_cmp_ps_256)
618 .Case("ps.512", Intrinsic::x86_avx512_mask_cmp_ps_512)
620 if (ID != Intrinsic::not_intrinsic)
621 return upgradeX86MaskedFPCompare(F, ID, NewFn);
622 } else if (Name.starts_with("vpdpbusd.") ||
623 Name.starts_with("vpdpbusds.")) {
624 // Added in 21.1
626 .Case("vpdpbusd.128", Intrinsic::x86_avx512_vpdpbusd_128)
627 .Case("vpdpbusd.256", Intrinsic::x86_avx512_vpdpbusd_256)
628 .Case("vpdpbusd.512", Intrinsic::x86_avx512_vpdpbusd_512)
629 .Case("vpdpbusds.128", Intrinsic::x86_avx512_vpdpbusds_128)
630 .Case("vpdpbusds.256", Intrinsic::x86_avx512_vpdpbusds_256)
631 .Case("vpdpbusds.512", Intrinsic::x86_avx512_vpdpbusds_512)
633 if (ID != Intrinsic::not_intrinsic)
634 return upgradeX86MultiplyAddBytes(F, ID, NewFn);
635 } else if (Name.starts_with("vpdpwssd.") ||
636 Name.starts_with("vpdpwssds.")) {
637 // Added in 21.1
639 .Case("vpdpwssd.128", Intrinsic::x86_avx512_vpdpwssd_128)
640 .Case("vpdpwssd.256", Intrinsic::x86_avx512_vpdpwssd_256)
641 .Case("vpdpwssd.512", Intrinsic::x86_avx512_vpdpwssd_512)
642 .Case("vpdpwssds.128", Intrinsic::x86_avx512_vpdpwssds_128)
643 .Case("vpdpwssds.256", Intrinsic::x86_avx512_vpdpwssds_256)
644 .Case("vpdpwssds.512", Intrinsic::x86_avx512_vpdpwssds_512)
646 if (ID != Intrinsic::not_intrinsic)
647 return upgradeX86MultiplyAddWords(F, ID, NewFn);
648 }
649 return false; // No other 'x86.avx512.*'.
650 }
651
652 if (Name.consume_front("avx2.")) {
653 if (Name.consume_front("vpdpb")) {
654 // Added in 21.1
656 .Case("ssd.128", Intrinsic::x86_avx2_vpdpbssd_128)
657 .Case("ssd.256", Intrinsic::x86_avx2_vpdpbssd_256)
658 .Case("ssds.128", Intrinsic::x86_avx2_vpdpbssds_128)
659 .Case("ssds.256", Intrinsic::x86_avx2_vpdpbssds_256)
660 .Case("sud.128", Intrinsic::x86_avx2_vpdpbsud_128)
661 .Case("sud.256", Intrinsic::x86_avx2_vpdpbsud_256)
662 .Case("suds.128", Intrinsic::x86_avx2_vpdpbsuds_128)
663 .Case("suds.256", Intrinsic::x86_avx2_vpdpbsuds_256)
664 .Case("uud.128", Intrinsic::x86_avx2_vpdpbuud_128)
665 .Case("uud.256", Intrinsic::x86_avx2_vpdpbuud_256)
666 .Case("uuds.128", Intrinsic::x86_avx2_vpdpbuuds_128)
667 .Case("uuds.256", Intrinsic::x86_avx2_vpdpbuuds_256)
669 if (ID != Intrinsic::not_intrinsic)
670 return upgradeX86MultiplyAddBytes(F, ID, NewFn);
671 } else if (Name.consume_front("vpdpw")) {
672 // Added in 21.1
674 .Case("sud.128", Intrinsic::x86_avx2_vpdpwsud_128)
675 .Case("sud.256", Intrinsic::x86_avx2_vpdpwsud_256)
676 .Case("suds.128", Intrinsic::x86_avx2_vpdpwsuds_128)
677 .Case("suds.256", Intrinsic::x86_avx2_vpdpwsuds_256)
678 .Case("usd.128", Intrinsic::x86_avx2_vpdpwusd_128)
679 .Case("usd.256", Intrinsic::x86_avx2_vpdpwusd_256)
680 .Case("usds.128", Intrinsic::x86_avx2_vpdpwusds_128)
681 .Case("usds.256", Intrinsic::x86_avx2_vpdpwusds_256)
682 .Case("uud.128", Intrinsic::x86_avx2_vpdpwuud_128)
683 .Case("uud.256", Intrinsic::x86_avx2_vpdpwuud_256)
684 .Case("uuds.128", Intrinsic::x86_avx2_vpdpwuuds_128)
685 .Case("uuds.256", Intrinsic::x86_avx2_vpdpwuuds_256)
687 if (ID != Intrinsic::not_intrinsic)
688 return upgradeX86MultiplyAddWords(F, ID, NewFn);
689 }
690 return false; // No other 'x86.avx2.*'
691 }
692
693 if (Name.consume_front("avx10.")) {
694 if (Name.consume_front("vpdpb")) {
695 // Added in 21.1
697 .Case("ssd.512", Intrinsic::x86_avx10_vpdpbssd_512)
698 .Case("ssds.512", Intrinsic::x86_avx10_vpdpbssds_512)
699 .Case("sud.512", Intrinsic::x86_avx10_vpdpbsud_512)
700 .Case("suds.512", Intrinsic::x86_avx10_vpdpbsuds_512)
701 .Case("uud.512", Intrinsic::x86_avx10_vpdpbuud_512)
702 .Case("uuds.512", Intrinsic::x86_avx10_vpdpbuuds_512)
704 if (ID != Intrinsic::not_intrinsic)
705 return upgradeX86MultiplyAddBytes(F, ID, NewFn);
706 } else if (Name.consume_front("vpdpw")) {
708 .Case("sud.512", Intrinsic::x86_avx10_vpdpwsud_512)
709 .Case("suds.512", Intrinsic::x86_avx10_vpdpwsuds_512)
710 .Case("usd.512", Intrinsic::x86_avx10_vpdpwusd_512)
711 .Case("usds.512", Intrinsic::x86_avx10_vpdpwusds_512)
712 .Case("uud.512", Intrinsic::x86_avx10_vpdpwuud_512)
713 .Case("uuds.512", Intrinsic::x86_avx10_vpdpwuuds_512)
715 if (ID != Intrinsic::not_intrinsic)
716 return upgradeX86MultiplyAddWords(F, ID, NewFn);
717 }
718 return false; // No other 'x86.avx10.*'
719 }
720
721 if (Name.consume_front("avx512bf16.")) {
722 // Added in 9.0
724 .Case("cvtne2ps2bf16.128",
725 Intrinsic::x86_avx512bf16_cvtne2ps2bf16_128)
726 .Case("cvtne2ps2bf16.256",
727 Intrinsic::x86_avx512bf16_cvtne2ps2bf16_256)
728 .Case("cvtne2ps2bf16.512",
729 Intrinsic::x86_avx512bf16_cvtne2ps2bf16_512)
730 .Case("mask.cvtneps2bf16.128",
731 Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128)
732 .Case("cvtneps2bf16.256",
733 Intrinsic::x86_avx512bf16_cvtneps2bf16_256)
734 .Case("cvtneps2bf16.512",
735 Intrinsic::x86_avx512bf16_cvtneps2bf16_512)
737 if (ID != Intrinsic::not_intrinsic)
738 return upgradeX86BF16Intrinsic(F, ID, NewFn);
739
740 // Added in 9.0
742 .Case("dpbf16ps.128", Intrinsic::x86_avx512bf16_dpbf16ps_128)
743 .Case("dpbf16ps.256", Intrinsic::x86_avx512bf16_dpbf16ps_256)
744 .Case("dpbf16ps.512", Intrinsic::x86_avx512bf16_dpbf16ps_512)
746 if (ID != Intrinsic::not_intrinsic)
747 return upgradeX86BF16DPIntrinsic(F, ID, NewFn);
748 return false; // No other 'x86.avx512bf16.*'.
749 }
750
751 if (Name.consume_front("xop.")) {
753 if (Name.starts_with("vpermil2")) { // Added in 3.9
754 // Upgrade any XOP PERMIL2 index operand still using a float/double
755 // vector.
756 auto Idx = F->getFunctionType()->getParamType(2);
757 if (Idx->isFPOrFPVectorTy()) {
758 unsigned IdxSize = Idx->getPrimitiveSizeInBits();
759 unsigned EltSize = Idx->getScalarSizeInBits();
760 if (EltSize == 64 && IdxSize == 128)
761 ID = Intrinsic::x86_xop_vpermil2pd;
762 else if (EltSize == 32 && IdxSize == 128)
763 ID = Intrinsic::x86_xop_vpermil2ps;
764 else if (EltSize == 64 && IdxSize == 256)
765 ID = Intrinsic::x86_xop_vpermil2pd_256;
766 else
767 ID = Intrinsic::x86_xop_vpermil2ps_256;
768 }
769 } else if (F->arg_size() == 2)
770 // frcz.ss/sd may need to have an argument dropped. Added in 3.2
772 .Case("vfrcz.ss", Intrinsic::x86_xop_vfrcz_ss)
773 .Case("vfrcz.sd", Intrinsic::x86_xop_vfrcz_sd)
775
776 if (ID != Intrinsic::not_intrinsic) {
777 rename(F);
778 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
779 return true;
780 }
781 return false; // No other 'x86.xop.*'
782 }
783
784 if (Name == "seh.recoverfp") {
785 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
786 Intrinsic::eh_recoverfp);
787 return true;
788 }
789
790 return false;
791}
792
793// Upgrade ARM (IsArm) or Aarch64 (!IsArm) intrinsic fns. Return true iff so.
794// IsArm: 'arm.*', !IsArm: 'aarch64.*'.
796 StringRef Name,
797 Function *&NewFn) {
798 if (Name.starts_with("rbit")) {
799 // '(arm|aarch64).rbit'.
801 F->getParent(), Intrinsic::bitreverse, F->arg_begin()->getType());
802 return true;
803 }
804
805 if (Name == "thread.pointer") {
806 // '(arm|aarch64).thread.pointer'.
808 F->getParent(), Intrinsic::thread_pointer, F->getReturnType());
809 return true;
810 }
811
812 bool Neon = Name.consume_front("neon.");
813 if (Neon) {
814 // '(arm|aarch64).neon.*'.
815 // Changed in 12.0: bfdot accept v4bf16 and v8bf16 instead of v8i8 and
816 // v16i8 respectively.
817 if (Name.consume_front("bfdot.")) {
818 // (arm|aarch64).neon.bfdot.*'.
819 Intrinsic::ID ID =
821 .Cases({"v2f32.v8i8", "v4f32.v16i8"},
822 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfdot
823 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfdot)
825 if (ID != Intrinsic::not_intrinsic) {
826 size_t OperandWidth = F->getReturnType()->getPrimitiveSizeInBits();
827 assert((OperandWidth == 64 || OperandWidth == 128) &&
828 "Unexpected operand width");
829 LLVMContext &Ctx = F->getParent()->getContext();
830 std::array<Type *, 2> Tys{
831 {F->getReturnType(),
832 FixedVectorType::get(Type::getBFloatTy(Ctx), OperandWidth / 16)}};
833 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID, Tys);
834 return true;
835 }
836 return false; // No other '(arm|aarch64).neon.bfdot.*'.
837 }
838
839 // Changed in 12.0: bfmmla, bfmlalb and bfmlalt are not polymorphic
840 // anymore and accept v8bf16 instead of v16i8.
841 if (Name.consume_front("bfm")) {
842 // (arm|aarch64).neon.bfm*'.
843 if (Name.consume_back(".v4f32.v16i8")) {
844 // (arm|aarch64).neon.bfm*.v4f32.v16i8'.
845 Intrinsic::ID ID =
847 .Case("mla",
848 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmmla
849 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmmla)
850 .Case("lalb",
851 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmlalb
852 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmlalb)
853 .Case("lalt",
854 IsArm ? (Intrinsic::ID)Intrinsic::arm_neon_bfmlalt
855 : (Intrinsic::ID)Intrinsic::aarch64_neon_bfmlalt)
857 if (ID != Intrinsic::not_intrinsic) {
858 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
859 return true;
860 }
861 return false; // No other '(arm|aarch64).neon.bfm*.v16i8'.
862 }
863 return false; // No other '(arm|aarch64).neon.bfm*.
864 }
865 // Continue on to Aarch64 Neon or Arm Neon.
866 }
867 // Continue on to Arm or Aarch64.
868
869 if (IsArm) {
870 // 'arm.*'.
871 if (Neon) {
872 // 'arm.neon.*'.
874 .StartsWith("vclz.", Intrinsic::ctlz)
875 .StartsWith("vcnt.", Intrinsic::ctpop)
876 .StartsWith("vqadds.", Intrinsic::sadd_sat)
877 .StartsWith("vqaddu.", Intrinsic::uadd_sat)
878 .StartsWith("vqsubs.", Intrinsic::ssub_sat)
879 .StartsWith("vqsubu.", Intrinsic::usub_sat)
880 .StartsWith("vrinta.", Intrinsic::round)
881 .StartsWith("vrintn.", Intrinsic::roundeven)
882 .StartsWith("vrintm.", Intrinsic::floor)
883 .StartsWith("vrintp.", Intrinsic::ceil)
884 .StartsWith("vrintx.", Intrinsic::rint)
885 .StartsWith("vrintz.", Intrinsic::trunc)
887 if (ID != Intrinsic::not_intrinsic) {
888 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
889 F->arg_begin()->getType());
890 return true;
891 }
892
893 if (Name.consume_front("vst")) {
894 // 'arm.neon.vst*'.
895 static const Regex vstRegex("^([1234]|[234]lane)\\.v[a-z0-9]*$");
897 if (vstRegex.match(Name, &Groups)) {
898 static const Intrinsic::ID StoreInts[] = {
899 Intrinsic::arm_neon_vst1, Intrinsic::arm_neon_vst2,
900 Intrinsic::arm_neon_vst3, Intrinsic::arm_neon_vst4};
901
902 static const Intrinsic::ID StoreLaneInts[] = {
903 Intrinsic::arm_neon_vst2lane, Intrinsic::arm_neon_vst3lane,
904 Intrinsic::arm_neon_vst4lane};
905
906 auto fArgs = F->getFunctionType()->params();
907 Type *Tys[] = {fArgs[0], fArgs[1]};
908 if (Groups[1].size() == 1)
910 F->getParent(), StoreInts[fArgs.size() - 3], Tys);
911 else
913 F->getParent(), StoreLaneInts[fArgs.size() - 5], Tys);
914 return true;
915 }
916 return false; // No other 'arm.neon.vst*'.
917 }
918
919 return false; // No other 'arm.neon.*'.
920 }
921
922 if (Name.consume_front("mve.")) {
923 // 'arm.mve.*'.
924 if (Name == "vctp64") {
925 if (cast<FixedVectorType>(F->getReturnType())->getNumElements() == 4) {
926 // A vctp64 returning a v4i1 is converted to return a v2i1. Rename
927 // the function and deal with it below in UpgradeIntrinsicCall.
928 rename(F);
929 return true;
930 }
931 return false; // Not 'arm.mve.vctp64'.
932 }
933
934 if (Name.starts_with("vrintn.v")) {
936 F->getParent(), Intrinsic::roundeven, F->arg_begin()->getType());
937 return true;
938 }
939
940 // These too are changed to accept a v2i1 instead of the old v4i1.
941 if (Name.consume_back(".v4i1")) {
942 // 'arm.mve.*.v4i1'.
943 if (Name.consume_back(".predicated.v2i64.v4i32"))
944 // 'arm.mve.*.predicated.v2i64.v4i32.v4i1'
945 return Name == "mull.int" || Name == "vqdmull";
946
947 if (Name.consume_back(".v2i64")) {
948 // 'arm.mve.*.v2i64.v4i1'
949 bool IsGather = Name.consume_front("vldr.gather.");
950 if (IsGather || Name.consume_front("vstr.scatter.")) {
951 if (Name.consume_front("base.")) {
952 // Optional 'wb.' prefix.
953 Name.consume_front("wb.");
954 // 'arm.mve.(vldr.gather|vstr.scatter).base.(wb.)?
955 // predicated.v2i64.v2i64.v4i1'.
956 return Name == "predicated.v2i64";
957 }
958
959 if (Name.consume_front("offset.predicated."))
960 return Name == (IsGather ? "v2i64.p0i64" : "p0i64.v2i64") ||
961 Name == (IsGather ? "v2i64.p0" : "p0.v2i64");
962
963 // No other 'arm.mve.(vldr.gather|vstr.scatter).*.v2i64.v4i1'.
964 return false;
965 }
966
967 return false; // No other 'arm.mve.*.v2i64.v4i1'.
968 }
969 return false; // No other 'arm.mve.*.v4i1'.
970 }
971 return false; // No other 'arm.mve.*'.
972 }
973
974 if (Name.consume_front("cde.vcx")) {
975 // 'arm.cde.vcx*'.
976 if (Name.consume_back(".predicated.v2i64.v4i1"))
977 // 'arm.cde.vcx*.predicated.v2i64.v4i1'.
978 return Name == "1q" || Name == "1qa" || Name == "2q" || Name == "2qa" ||
979 Name == "3q" || Name == "3qa";
980
981 return false; // No other 'arm.cde.vcx*'.
982 }
983 } else {
984 // 'aarch64.*'.
985 if (Neon) {
986 // 'aarch64.neon.*'.
988 .StartsWith("frintn", Intrinsic::roundeven)
989 .StartsWith("rbit", Intrinsic::bitreverse)
991 if (ID != Intrinsic::not_intrinsic) {
992 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
993 F->arg_begin()->getType());
994 return true;
995 }
996
997 Intrinsic::ID MinMaxID =
998 StringSwitch<Intrinsic::ID>(Name.split('.').first)
999 .Case("smax", Intrinsic::smax)
1000 .Case("smin", Intrinsic::smin)
1001 .Case("umax", Intrinsic::umax)
1002 .Case("umin", Intrinsic::umin)
1004 if (MinMaxID != Intrinsic::not_intrinsic) {
1005 if (F->arg_size() != 2 || !F->getReturnType()->isIntOrIntVectorTy())
1006 return false; // Invalid IR.
1007 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), MinMaxID,
1008 F->getReturnType());
1009 return true;
1010 }
1011
1012 if (Name.starts_with("addp")) {
1013 // 'aarch64.neon.addp*'.
1014 if (F->arg_size() != 2)
1015 return false; // Invalid IR.
1016 VectorType *Ty = dyn_cast<VectorType>(F->getReturnType());
1017 if (Ty && Ty->getElementType()->isFloatingPointTy()) {
1019 F->getParent(), Intrinsic::aarch64_neon_faddp, Ty);
1020 return true;
1021 }
1022 }
1023
1024 // Changed in 20.0: bfcvt/bfcvtn/bcvtn2 have been replaced with fptrunc.
1025 if (Name.starts_with("bfcvt")) {
1026 NewFn = nullptr;
1027 return true;
1028 }
1029
1030 // vcvtfp2hf and vcvthf2fp -> fpext and fptrunc
1031 if (Name == "vcvtfp2hf" || Name == "vcvthf2fp") {
1032 NewFn = nullptr;
1033 return true;
1034 }
1035
1036 return false; // No other 'aarch64.neon.*'.
1037 }
1038 if (Name.consume_front("sve.")) {
1039 // 'aarch64.sve.*'.
1040 if (Name.consume_front("bf")) {
1041 if (Name == "mmla") {
1042 Type *Tys[] = {F->getReturnType(),
1043 std::next(F->arg_begin())->getType()};
1045 F->getParent(), Intrinsic::aarch64_sve_fmmla, Tys);
1046 return true;
1047 }
1048 if (Name.consume_back(".lane")) {
1049 // 'aarch64.sve.bf*.lane'.
1050 Intrinsic::ID ID =
1052 .Case("dot", Intrinsic::aarch64_sve_bfdot_lane_v2)
1053 .Case("mlalb", Intrinsic::aarch64_sve_bfmlalb_lane_v2)
1054 .Case("mlalt", Intrinsic::aarch64_sve_bfmlalt_lane_v2)
1056 if (ID != Intrinsic::not_intrinsic) {
1057 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1058 return true;
1059 }
1060 return false; // No other 'aarch64.sve.bf*.lane'.
1061 }
1062 return false; // No other 'aarch64.sve.bf*'.
1063 }
1064
1065 // 'aarch64.sve.fcvt.bf16f32' || 'aarch64.sve.fcvtnt.bf16f32'
1066 if (Name == "fcvt.bf16f32" || Name == "fcvtnt.bf16f32") {
1067 NewFn = nullptr;
1068 return true;
1069 }
1070
1071 if (Name.consume_front("convert.from.svbool")) {
1072 // 'aarch64.sve.convert.from.svbool'
1073 auto *TTy = dyn_cast<TargetExtType>(F->getReturnType());
1074 if (!TTy || TTy->getName() != "aarch64.svcount")
1075 return false;
1076
1077 Intrinsic::ID ID = Intrinsic::aarch64_sve_convert_to_svcount;
1078 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1079 return true;
1080 }
1081
1082 if (Name.consume_front("convert.to.svbool")) {
1083 // 'aarch64.sve.convert.to.svbool'
1084 auto *TTy = dyn_cast<TargetExtType>(F->arg_begin()->getType());
1085 if (!TTy || TTy->getName() != "aarch64.svcount")
1086 return false;
1087
1088 Intrinsic::ID ID = Intrinsic::aarch64_sve_convert_from_svcount;
1089 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1090 return true;
1091 }
1092
1093 if (Name.consume_front("addqv")) {
1094 // 'aarch64.sve.addqv'.
1095 if (!F->getReturnType()->isFPOrFPVectorTy())
1096 return false;
1097
1098 auto Args = F->getFunctionType()->params();
1099 Type *Tys[] = {F->getReturnType(), Args[1]};
1101 F->getParent(), Intrinsic::aarch64_sve_faddqv, Tys);
1102 return true;
1103 }
1104
1105 if (Name.consume_front("ld")) {
1106 // 'aarch64.sve.ld*'.
1107 static const Regex LdRegex("^[234](.nxv[a-z0-9]+|$)");
1108 if (LdRegex.match(Name)) {
1109 Type *ScalarTy =
1110 cast<VectorType>(F->getReturnType())->getElementType();
1111 ElementCount EC =
1112 cast<VectorType>(F->arg_begin()->getType())->getElementCount();
1113 assert(F->arg_size() == 2 &&
1114 "Expected 2 arguments for ld* intrinsic.");
1115 Type *PtrTy = F->getArg(1)->getType();
1116 Type *Ty = VectorType::get(ScalarTy, EC);
1117 static const Intrinsic::ID LoadIDs[] = {
1118 Intrinsic::aarch64_sve_ld2_sret,
1119 Intrinsic::aarch64_sve_ld3_sret,
1120 Intrinsic::aarch64_sve_ld4_sret,
1121 };
1123 F->getParent(), LoadIDs[Name[0] - '2'], {Ty, PtrTy});
1124 return true;
1125 }
1126 return false; // No other 'aarch64.sve.ld*'.
1127 }
1128
1129 if (Name.consume_front("tuple.")) {
1130 // 'aarch64.sve.tuple.*'.
1131 if (Name.starts_with("get")) {
1132 // 'aarch64.sve.tuple.get*'.
1133 Type *Tys[] = {F->getReturnType(), F->arg_begin()->getType()};
1135 F->getParent(), Intrinsic::vector_extract, Tys);
1136 return true;
1137 }
1138
1139 if (Name.starts_with("set")) {
1140 // 'aarch64.sve.tuple.set*'.
1141 auto Args = F->getFunctionType()->params();
1142 Type *Tys[] = {Args[0], Args[2], Args[1]};
1144 F->getParent(), Intrinsic::vector_insert, Tys);
1145 return true;
1146 }
1147
1148 static const Regex CreateTupleRegex("^create[234](.nxv[a-z0-9]+|$)");
1149 if (CreateTupleRegex.match(Name)) {
1150 // 'aarch64.sve.tuple.create*'.
1151 auto Args = F->getFunctionType()->params();
1152 Type *Tys[] = {F->getReturnType(), Args[1]};
1154 F->getParent(), Intrinsic::vector_insert, Tys);
1155 return true;
1156 }
1157 return false; // No other 'aarch64.sve.tuple.*'.
1158 }
1159
1160 if (Name.starts_with("rev.nxv")) {
1161 // 'aarch64.sve.rev.<Ty>'
1163 F->getParent(), Intrinsic::vector_reverse, F->getReturnType());
1164 return true;
1165 }
1166
1167 return false; // No other 'aarch64.sve.*'.
1168 }
1169 if (Name.consume_front("sme.")) {
1170 // 'aarch64.sme.*'.
1171 if (Name.consume_front("ftmopa.")) {
1172 // The FP8 FTMOPA intrinsics were split out from the non-FP8 FTMOPA
1173 // intrinsics to model their FPMR dependency.
1174 Intrinsic::ID ID =
1176 .Case("za16.nxv16i8", Intrinsic::aarch64_sme_fp8_ftmopa_za16)
1177 .Case("za32.nxv16i8", Intrinsic::aarch64_sme_fp8_ftmopa_za32)
1179 if (ID != Intrinsic::not_intrinsic) {
1180 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
1181 return true;
1182 }
1183 return false; // No other 'aarch64.sme.ftmopa.*'.
1184 }
1185
1186 return false; // No other 'aarch64.sme.*'.
1187 }
1188 }
1189 return false; // No other 'arm.*', 'aarch64.*'.
1190}
1191
1192// The TMA G2S (global-to-shared) tensor copy modes that have legacy
1193// declarations requiring an auto-upgrade. The same set applies to the
1194// cluster (g2s) and CTA (g2s_cta) variants.
1195#define NVVM_TMA_G2S_MODES(M) \
1196 M(tile_1d, "tile.1d") \
1197 M(tile_2d, "tile.2d") \
1198 M(tile_3d, "tile.3d") \
1199 M(tile_4d, "tile.4d") \
1200 M(tile_5d, "tile.5d") \
1201 M(tile_gather4_2d, "tile.gather4.2d") \
1202 M(im2col_3d, "im2col.3d") \
1203 M(im2col_4d, "im2col.4d") \
1204 M(im2col_5d, "im2col.5d") \
1205 M(im2col_w_3d, "im2col.w.3d") \
1206 M(im2col_w_4d, "im2col.w.4d") \
1207 M(im2col_w_5d, "im2col.w.5d") \
1208 M(im2col_w_128_3d, "im2col.w.128.3d") \
1209 M(im2col_w_128_4d, "im2col.w.128.4d") \
1210 M(im2col_w_128_5d, "im2col.w.128.5d")
1211
1212// Two legacy tails are:
1213//
1214// arg1, arg2, .. i64 %ch, i1 %flag_mc, i1 %flag_ch
1215// arg1, arg2, .. i64 %ch, i1 %flag_mc, i1 %flag_ch, i32 %cta_group
1216//
1217// The current tail appends a trailing i32 %flag_valid_pattern, so both
1218// legacy tails are recognized by an i1 at parameter N-2.
1219static Intrinsic::ID
1221 SmallVectorImpl<Type *> &OvlTys) {
1222 if (!Name.consume_front("cp.async.bulk.tensor.g2s."))
1224
1225#define G2S_ID(ID_SUFFIX, NAME) \
1226 .Case(NAME, Intrinsic::nvvm_cp_async_bulk_tensor_g2s_##ID_SUFFIX)
1227 // clang-format off
1231#undef G2S_ID
1232 // clang-format on
1233 if (ID == Intrinsic::not_intrinsic)
1234 return ID;
1235
1236 size_t NumParams = F->getFunctionType()->getNumParams();
1237
1238 // Parameter N-2 is i1 for both legacy tails; the current tail ends
1239 // with i32 %cta_group, i32 %flag_valid_pattern, for which N-2 is i32.
1240 if (!F->getFunctionType()->getParamType(NumParams - 2)->isIntegerTy(1))
1242
1243 // The multicast mask is the parameter immediately before the i64
1244 // cache-hint: N-4 for the 2-flag tail, N-5 for the 3-flag tail.
1245 ArrayRef<Type *> Params = F->getFunctionType()->params();
1246 size_t MaskIdx =
1247 Params[NumParams - 1]->isIntegerTy(1) ? NumParams - 4 : NumParams - 5;
1248 assert(Params[MaskIdx + 1]->isIntegerTy(64) &&
1249 "expected the i64 cache-hint after the multicast mask");
1250 Type *MaskTy = Params[MaskIdx];
1251 assert(MaskTy->isIntegerTy(16) && "unexpected multicast mask type");
1252 OvlTys.push_back(MaskTy);
1253
1254 return ID;
1255}
1256
1257// The legacy tail is:
1258//
1259// arg1, arg2, .. i64 %ch, i1 %flag_ch
1260//
1261// The current tail appends a trailing i32 %flag_valid_pattern, so the
1262// legacy tail is recognized by an i1 at parameter N-1.
1264 StringRef Name) {
1265 if (!Name.consume_front("cp.async.bulk.tensor.g2s.cta."))
1267
1268#define G2S_CTA_ID(ID_SUFFIX, NAME) \
1269 .Case(NAME, Intrinsic::nvvm_cp_async_bulk_tensor_g2s_cta_##ID_SUFFIX)
1270 // clang-format off
1274#undef G2S_CTA_ID
1275 // clang-format on
1276 if (ID == Intrinsic::not_intrinsic)
1277 return ID;
1278
1279 // Parameter N-1 is i1 for the legacy tail; the current tail ends
1280 // with i32 %flag_valid_pattern, for which N-1 is i32.
1281 if (!F->getFunctionType()
1282 ->getParamType(F->getFunctionType()->getNumParams() - 1)
1283 ->isIntegerTy(1))
1285
1286 return ID;
1287}
1288// The legacy tail of llvm.nvvm.cp.async.bulk.global.to.shared.cluster is:
1289//
1290// ..., i16 %mc, i64 %ch, i1 %flag_mc, i1 %flag_ch
1291//
1292// The current intrinsic is overloaded on the multicast-mask type and takes a
1293// trailing i32 %flag_valid_pattern; the legacy tail is recognized by an i1 at
1294// parameter N-1.
1295static Intrinsic::ID
1297 SmallVectorImpl<Type *> &OvlTys) {
1298 if (!Name.consume_front("cp.async.bulk.global.to.shared.cluster"))
1300
1301 // Parameter N-1 is i1 for the legacy tail; the current tail ends with
1302 // i32 %flag_valid_pattern, for which N-1 is i32.
1303 size_t NumParams = F->getFunctionType()->getNumParams();
1304 if (!F->getFunctionType()->getParamType(NumParams - 1)->isIntegerTy(1))
1306
1307 // The multicast mask is parameter 4; legacy IR only uses i16.
1308 Type *MaskTy = F->getFunctionType()->getParamType(NumParams - 4);
1309 if (!MaskTy->isIntegerTy(16))
1311 OvlTys.push_back(MaskTy);
1312
1313 return Intrinsic::nvvm_cp_async_bulk_global_to_shared_cluster;
1314}
1315
1316// The legacy tail of llvm.nvvm.cp.async.bulk.global.to.shared.cta is:
1317//
1318// ..., i64 %ch, i1 %flag_ch
1319//
1320// The current intrinsic adds %ignore_bytes_left/%ignore_bytes_right before
1321// %ch and trailing %flag_oob/%flag_valid_pattern; the legacy tail is
1322// recognized by an i1 at parameter N-1, whereas the current tail ends
1323// with an i32.
1325 StringRef Name) {
1326 if (!Name.consume_front("cp.async.bulk.global.to.shared.cta"))
1328
1329 // Parameter N-1 is i1 for the legacy tail; the current tail ends with
1330 // i32 %flag_valid_pattern, for which N-1 is i32.
1331 if (!F->getFunctionType()->getParamType(5)->isIntegerTy(1))
1333
1334 return Intrinsic::nvvm_cp_async_bulk_global_to_shared_cta;
1335}
1336
1337// The legacy TMA reduction intrinsics encode the reduction operator in their
1338// name, while the current ones take it as an immediate argument. Map the
1339// operator part of a legacy name to the corresponding immediate value.
1340static std::optional<unsigned> getNVPTXTMAReductionOp(StringRef Name) {
1342 .Case("add", static_cast<unsigned>(nvvm::TMAReductionOp::ADD))
1343 .Case("min", static_cast<unsigned>(nvvm::TMAReductionOp::MIN))
1344 .Case("max", static_cast<unsigned>(nvvm::TMAReductionOp::MAX))
1345 .Case("inc", static_cast<unsigned>(nvvm::TMAReductionOp::INC))
1346 .Case("dec", static_cast<unsigned>(nvvm::TMAReductionOp::DEC))
1347 .Case("and", static_cast<unsigned>(nvvm::TMAReductionOp::AND))
1348 .Case("or", static_cast<unsigned>(nvvm::TMAReductionOp::OR))
1349 .Case("xor", static_cast<unsigned>(nvvm::TMAReductionOp::XOR))
1350 .Default(std::nullopt);
1351}
1352
1354 if (!Name.consume_front("cp.async.bulk.tensor.reduce."))
1356
1357 auto [RedOpName, ShapeName] = Name.split('.');
1358 if (!getNVPTXTMAReductionOp(RedOpName))
1360
1361 return StringSwitch<Intrinsic::ID>(ShapeName)
1362 .Case("tile.1d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_1d)
1363 .Case("tile.2d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_2d)
1364 .Case("tile.3d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_3d)
1365 .Case("tile.4d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_4d)
1366 .Case("tile.5d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_5d)
1367 .Case("im2col.3d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_3d)
1368 .Case("im2col.4d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_4d)
1369 .Case("im2col.5d", Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_5d)
1371}
1372
1374 StringRef Name) {
1375 if (Name.consume_front("mapa.shared.cluster"))
1376 if (F->getReturnType()->getPointerAddressSpace() ==
1378 return Intrinsic::nvvm_mapa_shared_cluster;
1379
1380 if (Name.consume_front("cp.async.bulk.")) {
1381 Intrinsic::ID ID =
1383 .Case("shared.cta.to.cluster",
1384 Intrinsic::nvvm_cp_async_bulk_shared_cta_to_cluster)
1386
1387 if (ID != Intrinsic::not_intrinsic)
1388 if (F->getArg(0)->getType()->getPointerAddressSpace() ==
1390 return ID;
1391 }
1392
1394}
1395
1396static Intrinsic::ID
1398 if (!Name.consume_front("tcgen05.commit."))
1400
1401 if (Name.consume_front("shared."))
1402 return StringSwitch<Intrinsic::ID>(Name)
1403 .Case("cg1", Intrinsic::nvvm_tcgen05_commit_cg1)
1404 .Case("cg2", Intrinsic::nvvm_tcgen05_commit_cg2)
1406
1407 if (Name.consume_front("mc.shared.")) {
1408 // Only upgrade older i16 mc variants.
1409 if (!F->getArg(1)->getType()->isIntegerTy(16))
1411
1412 return StringSwitch<Intrinsic::ID>(Name)
1413 .Case("cg1", Intrinsic::nvvm_tcgen05_commit_mc_cg1)
1414 .Case("cg2", Intrinsic::nvvm_tcgen05_commit_mc_cg2)
1416 }
1417
1419}
1420
1421static Intrinsic::ID
1423 if (F->arg_size() != 2)
1425
1426 if (Name.consume_front("tcgen05.alloc.shared.") ||
1427 Name.consume_front("tcgen05.alloc."))
1428 return StringSwitch<Intrinsic::ID>(Name)
1429 .Case("cg1", Intrinsic::nvvm_tcgen05_alloc_cg1)
1430 .Case("cg2", Intrinsic::nvvm_tcgen05_alloc_cg2)
1432
1433 if (Name.consume_front("tcgen05.dealloc."))
1434 return StringSwitch<Intrinsic::ID>(Name)
1435 .Case("cg1", Intrinsic::nvvm_tcgen05_dealloc_cg1)
1436 .Case("cg2", Intrinsic::nvvm_tcgen05_dealloc_cg2)
1438
1440}
1441
1443 if (Name.consume_front("fma.rn."))
1444 return StringSwitch<Intrinsic::ID>(Name)
1445 .Case("bf16", Intrinsic::nvvm_fma_rn_bf16)
1446 .Case("bf16x2", Intrinsic::nvvm_fma_rn_bf16x2)
1447 .Case("relu.bf16", Intrinsic::nvvm_fma_rn_relu_bf16)
1448 .Case("relu.bf16x2", Intrinsic::nvvm_fma_rn_relu_bf16x2)
1450
1451 if (Name.consume_front("fmax."))
1452 return StringSwitch<Intrinsic::ID>(Name)
1453 .Case("bf16", Intrinsic::nvvm_fmax_bf16)
1454 .Case("bf16x2", Intrinsic::nvvm_fmax_bf16x2)
1455 .Case("ftz.bf16", Intrinsic::nvvm_fmax_ftz_bf16)
1456 .Case("ftz.bf16x2", Intrinsic::nvvm_fmax_ftz_bf16x2)
1457 .Case("ftz.nan.bf16", Intrinsic::nvvm_fmax_ftz_nan_bf16)
1458 .Case("ftz.nan.bf16x2", Intrinsic::nvvm_fmax_ftz_nan_bf16x2)
1459 .Case("ftz.nan.xorsign.abs.bf16",
1460 Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_bf16)
1461 .Case("ftz.nan.xorsign.abs.bf16x2",
1462 Intrinsic::nvvm_fmax_ftz_nan_xorsign_abs_bf16x2)
1463 .Case("ftz.xorsign.abs.bf16", Intrinsic::nvvm_fmax_ftz_xorsign_abs_bf16)
1464 .Case("ftz.xorsign.abs.bf16x2",
1465 Intrinsic::nvvm_fmax_ftz_xorsign_abs_bf16x2)
1466 .Case("nan.bf16", Intrinsic::nvvm_fmax_nan_bf16)
1467 .Case("nan.bf16x2", Intrinsic::nvvm_fmax_nan_bf16x2)
1468 .Case("nan.xorsign.abs.bf16", Intrinsic::nvvm_fmax_nan_xorsign_abs_bf16)
1469 .Case("nan.xorsign.abs.bf16x2",
1470 Intrinsic::nvvm_fmax_nan_xorsign_abs_bf16x2)
1471 .Case("xorsign.abs.bf16", Intrinsic::nvvm_fmax_xorsign_abs_bf16)
1472 .Case("xorsign.abs.bf16x2", Intrinsic::nvvm_fmax_xorsign_abs_bf16x2)
1474
1475 if (Name.consume_front("fmin."))
1476 return StringSwitch<Intrinsic::ID>(Name)
1477 .Case("bf16", Intrinsic::nvvm_fmin_bf16)
1478 .Case("bf16x2", Intrinsic::nvvm_fmin_bf16x2)
1479 .Case("ftz.bf16", Intrinsic::nvvm_fmin_ftz_bf16)
1480 .Case("ftz.bf16x2", Intrinsic::nvvm_fmin_ftz_bf16x2)
1481 .Case("ftz.nan.bf16", Intrinsic::nvvm_fmin_ftz_nan_bf16)
1482 .Case("ftz.nan.bf16x2", Intrinsic::nvvm_fmin_ftz_nan_bf16x2)
1483 .Case("ftz.nan.xorsign.abs.bf16",
1484 Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_bf16)
1485 .Case("ftz.nan.xorsign.abs.bf16x2",
1486 Intrinsic::nvvm_fmin_ftz_nan_xorsign_abs_bf16x2)
1487 .Case("ftz.xorsign.abs.bf16", Intrinsic::nvvm_fmin_ftz_xorsign_abs_bf16)
1488 .Case("ftz.xorsign.abs.bf16x2",
1489 Intrinsic::nvvm_fmin_ftz_xorsign_abs_bf16x2)
1490 .Case("nan.bf16", Intrinsic::nvvm_fmin_nan_bf16)
1491 .Case("nan.bf16x2", Intrinsic::nvvm_fmin_nan_bf16x2)
1492 .Case("nan.xorsign.abs.bf16", Intrinsic::nvvm_fmin_nan_xorsign_abs_bf16)
1493 .Case("nan.xorsign.abs.bf16x2",
1494 Intrinsic::nvvm_fmin_nan_xorsign_abs_bf16x2)
1495 .Case("xorsign.abs.bf16", Intrinsic::nvvm_fmin_xorsign_abs_bf16)
1496 .Case("xorsign.abs.bf16x2", Intrinsic::nvvm_fmin_xorsign_abs_bf16x2)
1498
1499 if (Name.consume_front("neg."))
1500 return StringSwitch<Intrinsic::ID>(Name)
1501 .Case("bf16", Intrinsic::nvvm_neg_bf16)
1502 .Case("bf16x2", Intrinsic::nvvm_neg_bf16x2)
1504
1506}
1507
1509 FunctionType *NewFnTy = Intrinsic::getType(F->getContext(), IID);
1510 FunctionType *OldFnTy = F->getFunctionType();
1511 auto IsOldBF16StorageTy = [](Type *OldTy, Type *NewTy) {
1512 return OldTy->getScalarType()->isIntegerTy() &&
1513 OldTy->getPrimitiveSizeInBits() == NewTy->getPrimitiveSizeInBits();
1514 };
1515
1516 if (!IsOldBF16StorageTy(OldFnTy->getReturnType(), NewFnTy->getReturnType()))
1517 return false;
1518
1519 if (OldFnTy->getNumParams() != NewFnTy->getNumParams())
1520 return false;
1521
1522 for (unsigned I = 0, E = OldFnTy->getNumParams(); I != E; ++I)
1523 if (!IsOldBF16StorageTy(OldFnTy->getParamType(I), NewFnTy->getParamType(I)))
1524 return false;
1525
1526 return true;
1527}
1528
1529static std::optional<std::pair<Intrinsic::ID, RoundingMode>>
1531 auto [Modifiers, Type] = Name.rsplit('.');
1532 if (!is_contained({"f", "d", "f16", "v2f16"}, Type))
1533 return std::nullopt;
1534
1535 std::optional<llvm::RoundingMode> RoundingMode =
1536 StringSwitch<std::optional<llvm::RoundingMode>>(Modifiers.take_front(2))
1541 .Default(std::nullopt);
1542 if (!RoundingMode)
1543 return std::nullopt;
1544
1545 Intrinsic::ID IID = StringSwitch<Intrinsic::ID>(Modifiers.drop_front(2))
1546 .Case("", Intrinsic::nvvm_fadd)
1547 .Case(".ftz", Intrinsic::nvvm_fadd_ftz)
1548 .Case(".sat", Intrinsic::nvvm_fadd_sat)
1549 .Case(".ftz.sat", Intrinsic::nvvm_fadd_ftz_sat)
1551 if (IID == Intrinsic::not_intrinsic)
1552 return std::nullopt;
1553
1554 return std::make_pair(IID, *RoundingMode);
1555}
1556
1558 if (Name != "mbarrier.init" && Name != "mbarrier.init.shared")
1560
1561 return Intrinsic::nvvm_mbarrier_init;
1562}
1563
1565 return Name.consume_front("local") || Name.consume_front("shared") ||
1566 Name.consume_front("global") || Name.consume_front("constant") ||
1567 Name.consume_front("param");
1568}
1569
1571 if (!Name.consume_front("vp."))
1572 return 0;
1573 return StringSwitch<unsigned>(Name)
1574 .StartsWith("select", Instruction::Select)
1575 .StartsWith("add", Instruction::Add)
1576 .StartsWith("sub", Instruction::Sub)
1577 .StartsWith("mul", Instruction::Mul)
1578 .StartsWith("ashr", Instruction::AShr)
1579 .StartsWith("lshr", Instruction::LShr)
1580 .StartsWith("shl", Instruction::Shl)
1581 .StartsWith("or", Instruction::Or)
1582 .StartsWith("and", Instruction::And)
1583 .StartsWith("xor", Instruction::Xor)
1584 .StartsWith("fadd", Instruction::FAdd)
1585 .StartsWith("fsub", Instruction::FSub)
1586 .StartsWith("fmuladd", 0)
1587 .StartsWith("fmul", Instruction::FMul)
1588 .StartsWith("fdiv", Instruction::FDiv)
1589 .StartsWith("frem", Instruction::FRem)
1590 .StartsWith("fneg", Instruction::FNeg)
1591 .StartsWith("trunc", Instruction::Trunc)
1592 .StartsWith("zext", Instruction::ZExt)
1593 .StartsWith("sext", Instruction::SExt)
1594 .StartsWith("fptrunc", Instruction::FPTrunc)
1595 .StartsWith("fpext", Instruction::FPExt)
1596 .StartsWith("fptoui", Instruction::FPToUI)
1597 .StartsWith("fptosi", Instruction::FPToSI)
1598 .StartsWith("uitofp", Instruction::UIToFP)
1599 .StartsWith("sitofp", Instruction::SIToFP)
1600 .StartsWith("ptrtoint", Instruction::PtrToInt)
1601 .StartsWith("inttoptr", Instruction::IntToPtr)
1602 .StartsWith("icmp", Instruction::ICmp)
1603 .StartsWith("fcmp", Instruction::FCmp)
1604 .Default(0);
1605}
1606
1608 if (!Name.consume_front("vp."))
1609 return 0;
1610 return StringSwitch<Intrinsic::ID>(Name)
1611 .StartsWith("abs", Intrinsic::abs)
1612 .StartsWith("smax", Intrinsic::smax)
1613 .StartsWith("smin", Intrinsic::smin)
1614 .StartsWith("umax", Intrinsic::umax)
1615 .StartsWith("umin", Intrinsic::umin)
1616 .StartsWith("copysign", Intrinsic::copysign)
1617 .StartsWith("minnum", Intrinsic::minnum)
1618 .StartsWith("maxnum", Intrinsic::maxnum)
1619 .StartsWith("minimum", Intrinsic::minimum)
1620 .StartsWith("maximum", Intrinsic::maximum)
1621 .StartsWith("fabs", Intrinsic::fabs)
1622 .StartsWith("sqrt", Intrinsic::sqrt)
1623 .StartsWith("fma", Intrinsic::fma)
1624 .StartsWith("fmuladd", Intrinsic::fmuladd)
1625 .StartsWith("ceil", Intrinsic::ceil)
1626 .StartsWith("floor", Intrinsic::floor)
1627 .StartsWith("rint", Intrinsic::rint)
1628 .StartsWith("nearbyint", Intrinsic::nearbyint)
1629 .StartsWith("roundeven", Intrinsic::roundeven)
1630 .StartsWith("roundtozero", Intrinsic::trunc)
1631 .StartsWith("round", Intrinsic::round)
1632 .StartsWith("lrint", Intrinsic::lrint)
1633 .StartsWith("llrint", Intrinsic::llrint)
1634 .StartsWith("bitreverse", Intrinsic::bitreverse)
1635 .StartsWith("bswap", Intrinsic::bswap)
1636 .StartsWith("ctpop", Intrinsic::ctpop)
1637 .StartsWith("ctlz", Intrinsic::ctlz)
1638 .StartsWith("cttz.elts", 0)
1639 .StartsWith("cttz", Intrinsic::cttz)
1640 .StartsWith("sadd.sat", Intrinsic::sadd_sat)
1641 .StartsWith("uadd.sat", Intrinsic::uadd_sat)
1642 .StartsWith("ssub.sat", Intrinsic::ssub_sat)
1643 .StartsWith("usub.sat", Intrinsic::usub_sat)
1644 .StartsWith("fshl", Intrinsic::fshl)
1645 .StartsWith("fshr", Intrinsic::fshr)
1646 .StartsWith("is.fpclass", Intrinsic::is_fpclass)
1647 .Default(0);
1648}
1649
1653
1655 const FunctionType *FuncTy) {
1656 Type *HalfTy = Type::getHalfTy(FuncTy->getContext());
1657 if (Name.starts_with("to.fp16")) {
1658 return CastInst::castIsValid(Instruction::FPTrunc, FuncTy->getParamType(0),
1659 HalfTy) &&
1660 CastInst::castIsValid(Instruction::BitCast, HalfTy,
1661 FuncTy->getReturnType());
1662 }
1663
1664 if (Name.starts_with("from.fp16")) {
1665 return CastInst::castIsValid(Instruction::BitCast, FuncTy->getParamType(0),
1666 HalfTy) &&
1667 CastInst::castIsValid(Instruction::FPExt, HalfTy,
1668 FuncTy->getReturnType());
1669 }
1670
1671 return false;
1672}
1673
1674static unsigned
1676 SmallVectorImpl<Type *> &OverloadTys) {
1677 auto [FirstDefault, Defaults] = Intrinsic::getAllDefaultArgValues(IID);
1678 if (Defaults.empty())
1679 return 0;
1680
1681 unsigned FullArgCount = FirstDefault + Defaults.size();
1682
1683 // Only trailing default arguments can be missing.
1684 if (F->arg_size() < FirstDefault || F->arg_size() >= FullArgCount)
1685 return 0;
1686
1687 unsigned NumMissingTrailingParams = FullArgCount - F->arg_size();
1688 if (!Intrinsic::isSignatureValid(IID, F->getFunctionType(), OverloadTys,
1689 NumMissingTrailingParams))
1690 return 0;
1691
1692 return FullArgCount;
1693}
1694
1696 Intrinsic::ID IID = F->getIntrinsicID();
1697 SmallVector<Type *, 4> OverloadTys;
1698
1699 unsigned FullArgCount =
1700 getFullArgCountForDefaultArgUpgrade(F, IID, OverloadTys);
1701 if (FullArgCount == 0)
1702 return false;
1703
1704 rename(F);
1705 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID, OverloadTys);
1706 assert(NewFn->arg_size() == FullArgCount &&
1707 "total number of default args does not match intrinsic signature");
1708 return true;
1709}
1710
1712 bool CanUpgradeDebugIntrinsicsToRecords) {
1713 assert(F && "Illegal to upgrade a non-existent Function.");
1714
1715 StringRef Name = F->getName();
1716
1717 // Quickly eliminate it, if it's not a candidate.
1718 if (!Name.consume_front("llvm.") || Name.empty())
1719 return false;
1720
1721 switch (Name[0]) {
1722 default: break;
1723 case 'a': {
1724 bool IsArm = Name.consume_front("arm.");
1725 if (IsArm || Name.consume_front("aarch64.")) {
1726 if (upgradeArmOrAarch64IntrinsicFunction(IsArm, F, Name, NewFn))
1727 return true;
1728 break;
1729 }
1730
1731 if (Name.consume_front("amdgcn.")) {
1732 if (Name == "alignbit") {
1733 // Target specific intrinsic became redundant
1735 F->getParent(), Intrinsic::fshr, {F->getReturnType()});
1736 return true;
1737 }
1738
1739 if (Name.consume_front("atomic.")) {
1740 if (Name.starts_with("inc") || Name.starts_with("dec") ||
1741 Name.starts_with("cond.sub") || Name.starts_with("csub")) {
1742 // These were replaced with atomicrmw uinc_wrap, udec_wrap, usub_cond
1743 // and usub_sat so there's no new declaration.
1744 NewFn = nullptr;
1745 return true;
1746 }
1747 break; // No other 'amdgcn.atomic.*'
1748 }
1749
1750 if (Name.starts_with("addrspacecast.nonnull")) {
1751 // Replaced with an addrspacecast instruction carrying the nonnull flag,
1752 // so there's no new declaration.
1753 NewFn = nullptr;
1754 return true;
1755 }
1756
1757 switch (F->getIntrinsicID()) {
1758 default:
1759 break;
1760 // Legacy wmma iu intrinsics without the optional clamp operand.
1761 case Intrinsic::amdgcn_wmma_i32_16x16x64_iu8:
1762 if (F->arg_size() == 7) {
1763 NewFn = nullptr;
1764 return true;
1765 }
1766 break;
1767 case Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8:
1768 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
1769 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
1770 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
1771 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
1772 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
1773 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16:
1774 if (F->arg_size() == 8) {
1775 NewFn = nullptr;
1776 return true;
1777 }
1778 break;
1779 }
1780
1781 if (Name.consume_front("ds.") || Name.consume_front("global.atomic.") ||
1782 Name.consume_front("flat.atomic.")) {
1783 if (Name.starts_with("fadd") ||
1784 // FIXME: We should also remove fmin.num and fmax.num intrinsics.
1785 (Name.starts_with("fmin") && !Name.starts_with("fmin.num")) ||
1786 (Name.starts_with("fmax") && !Name.starts_with("fmax.num"))) {
1787 // Replaced with atomicrmw fadd/fmin/fmax, so there's no new
1788 // declaration.
1789 NewFn = nullptr;
1790 return true;
1791 }
1792 }
1793
1794 if (Name.starts_with("fcmp.") || Name.starts_with("icmp.")) {
1795 NewFn = nullptr;
1796 return true;
1797 }
1798
1799 if (Name.starts_with("ldexp.")) {
1800 // Target specific intrinsic became redundant
1802 F->getParent(), Intrinsic::ldexp,
1803 {F->getReturnType(), F->getArg(1)->getType()});
1804 return true;
1805 }
1806 break; // No other 'amdgcn.*'
1807 }
1808
1809 break;
1810 }
1811 case 'c': {
1812 if (F->arg_size() == 1) {
1813 if (Name.consume_front("convert.")) {
1814 if (convertIntrinsicValidType(Name, F->getFunctionType())) {
1815 NewFn = nullptr;
1816 return true;
1817 }
1818 }
1819
1821 .StartsWith("ctlz.", Intrinsic::ctlz)
1822 .StartsWith("cttz.", Intrinsic::cttz)
1824 if (ID != Intrinsic::not_intrinsic) {
1825 rename(F);
1826 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
1827 F->arg_begin()->getType());
1828 return true;
1829 }
1830 }
1831
1833 if (Name == "coro.end" &&
1834 (F->arg_size() == 2 || F->getReturnType()->isIntegerTy(1)))
1835 CoroEndID = Intrinsic::coro_end;
1836 else if (Name == "coro.end.async" && F->getReturnType()->isIntegerTy(1))
1837 CoroEndID = Intrinsic::coro_end_async;
1838
1839 if (CoroEndID != Intrinsic::not_intrinsic) {
1840 rename(F);
1841 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), CoroEndID);
1842 return true;
1843 }
1844
1845 break;
1846 }
1847 case 'd':
1848 if (Name.consume_front("dbg.")) {
1849 // Mark debug intrinsics for upgrade to new debug format.
1850 if (CanUpgradeDebugIntrinsicsToRecords) {
1851 if (Name == "addr" || Name == "value" || Name == "assign" ||
1852 Name == "declare" || Name == "label") {
1853 // There's no function to replace these with.
1854 NewFn = nullptr;
1855 // But we do want these to get upgraded.
1856 return true;
1857 }
1858 }
1859 // Update llvm.dbg.addr intrinsics even in "new debug mode"; they'll get
1860 // converted to DbgVariableRecords later.
1861 if (Name == "addr" || (Name == "value" && F->arg_size() == 4)) {
1862 rename(F);
1863 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
1864 Intrinsic::dbg_value);
1865 return true;
1866 }
1867 break; // No other 'dbg.*'.
1868 }
1869 break;
1870 case 'e':
1871 if (Name.consume_front("experimental.vector.")) {
1872 Intrinsic::ID ID =
1874 // Skip over extract.last.active, otherwise it will be 'upgraded'
1875 // to a regular vector extract which is a different operation.
1876 .StartsWith("extract.last.active.", Intrinsic::not_intrinsic)
1877 .StartsWith("extract.", Intrinsic::vector_extract)
1878 .StartsWith("insert.", Intrinsic::vector_insert)
1879 .StartsWith("reverse.", Intrinsic::vector_reverse)
1880 .StartsWith("interleave2.", Intrinsic::vector_interleave2)
1881 .StartsWith("deinterleave2.", Intrinsic::vector_deinterleave2)
1882 .StartsWith("partial.reduce.add",
1883 Intrinsic::vector_partial_reduce_add)
1885 if (ID != Intrinsic::not_intrinsic) {
1886 const auto *FT = F->getFunctionType();
1888 if (ID == Intrinsic::vector_extract ||
1889 ID == Intrinsic::vector_interleave2)
1890 // Extracting overloads the return type.
1891 Tys.push_back(FT->getReturnType());
1892 if (ID != Intrinsic::vector_interleave2)
1893 Tys.push_back(FT->getParamType(0));
1894 if (ID == Intrinsic::vector_insert ||
1895 ID == Intrinsic::vector_partial_reduce_add)
1896 // Inserting overloads the inserted type.
1897 Tys.push_back(FT->getParamType(1));
1898 rename(F);
1899 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID, Tys);
1900 return true;
1901 }
1902
1903 if (Name.consume_front("reduce.")) {
1905 static const Regex R("^([a-z]+)\\.[a-z][0-9]+");
1906 if (R.match(Name, &Groups))
1908 .Case("add", Intrinsic::vector_reduce_add)
1909 .Case("mul", Intrinsic::vector_reduce_mul)
1910 .Case("and", Intrinsic::vector_reduce_and)
1911 .Case("or", Intrinsic::vector_reduce_or)
1912 .Case("xor", Intrinsic::vector_reduce_xor)
1913 .Case("smax", Intrinsic::vector_reduce_smax)
1914 .Case("smin", Intrinsic::vector_reduce_smin)
1915 .Case("umax", Intrinsic::vector_reduce_umax)
1916 .Case("umin", Intrinsic::vector_reduce_umin)
1917 .Case("fmax", Intrinsic::vector_reduce_fmax)
1918 .Case("fmin", Intrinsic::vector_reduce_fmin)
1920
1921 bool V2 = false;
1922 if (ID == Intrinsic::not_intrinsic) {
1923 static const Regex R2("^v2\\.([a-z]+)\\.[fi][0-9]+");
1924 Groups.clear();
1925 V2 = true;
1926 if (R2.match(Name, &Groups))
1928 .Case("fadd", Intrinsic::vector_reduce_fadd)
1929 .Case("fmul", Intrinsic::vector_reduce_fmul)
1931 }
1932 if (ID != Intrinsic::not_intrinsic) {
1933 rename(F);
1934 auto Args = F->getFunctionType()->params();
1935 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
1936 {Args[V2 ? 1 : 0]});
1937 return true;
1938 }
1939 break; // No other 'expermental.vector.reduce.*'.
1940 }
1941
1942 if (Name.consume_front("splice"))
1943 return true;
1944 break; // No other 'experimental.vector.*'.
1945 }
1946 if (Name.consume_front("experimental.stepvector.")) {
1947 Intrinsic::ID ID = Intrinsic::stepvector;
1948 rename(F);
1950 F->getParent(), ID, F->getFunctionType()->getReturnType());
1951 return true;
1952 }
1953 break; // No other 'e*'.
1954 case 'f':
1955 if (Name.starts_with("flt.rounds")) {
1956 rename(F);
1957 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
1958 Intrinsic::get_rounding);
1959 return true;
1960 }
1961 break;
1962 case 'i':
1963 if (Name.starts_with("invariant.group.barrier")) {
1964 // Rename invariant.group.barrier to launder.invariant.group
1965 auto Args = F->getFunctionType()->params();
1966 Type* ObjectPtr[1] = {Args[0]};
1967 rename(F);
1969 F->getParent(), Intrinsic::launder_invariant_group, ObjectPtr);
1970 return true;
1971 }
1972 break;
1973 case 'l': {
1974 bool IsLifetimeStart = Name.consume_front("lifetime.start");
1975 bool IsLifetimeEnd = !IsLifetimeStart && Name.consume_front("lifetime.end");
1976 if (IsLifetimeStart || IsLifetimeEnd) {
1977 if (F->arg_size() == 2) {
1978 Intrinsic::ID IID = IsLifetimeStart ? Intrinsic::lifetime_start
1979 : Intrinsic::lifetime_end;
1980 rename(F);
1981 // Old 2 argument form of these intrinsics have [Size, Ptr] as
1982 // arguments. Use the Ptr argument to create new declaration.
1983 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
1984 F->getArg(1)->getType());
1985 return true;
1986 } else if (F->arg_size() == 1 && Name == ".i64") {
1987 // Matches @llvm.lifetime.{start/end}.i64 which used to be created by
1988 // Autoupgrade prior to
1989 // https://github.com/llvm/llvm-project/pull/204601. This is an invalid
1990 // intrinsic with no expected calls. To allow auto-upgrade process to
1991 // delete such invalid intrinsic declaration, set NewFn = nullptr
1992 // and return true here. If there are actual calls to this intrinsic
1993 // (which is not expected), they will be deleted in
1994 // UpgradeIntrinsicCall.
1995 NewFn = nullptr;
1996 return true;
1997 }
1998 }
1999 break;
2000 }
2001 case 'm': {
2002 // Updating the memory intrinsics (memcpy/memmove/memset) that have an
2003 // alignment parameter to embedding the alignment as an attribute of
2004 // the pointer args.
2005 if (unsigned ID = StringSwitch<unsigned>(Name)
2006 .StartsWith("memcpy.", Intrinsic::memcpy)
2007 .StartsWith("memmove.", Intrinsic::memmove)
2008 .Default(0)) {
2009 if (F->arg_size() == 5) {
2010 rename(F);
2011 // Get the types of dest, src, and len
2012 ArrayRef<Type *> ParamTypes =
2013 F->getFunctionType()->params().slice(0, 3);
2014 NewFn =
2015 Intrinsic::getOrInsertDeclaration(F->getParent(), ID, ParamTypes);
2016 return true;
2017 }
2018 }
2019 if (Name.starts_with("memset.") && F->arg_size() == 5) {
2020 rename(F);
2021 // Get the types of dest, and len
2022 const auto *FT = F->getFunctionType();
2023 Type *ParamTypes[2] = {
2024 FT->getParamType(0), // Dest
2025 FT->getParamType(2) // len
2026 };
2027 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
2028 Intrinsic::memset, ParamTypes);
2029 return true;
2030 }
2031
2032 unsigned MaskedID =
2034 .StartsWith("masked.load", Intrinsic::masked_load)
2035 .StartsWith("masked.gather", Intrinsic::masked_gather)
2036 .StartsWith("masked.store", Intrinsic::masked_store)
2037 .StartsWith("masked.scatter", Intrinsic::masked_scatter)
2038 .Default(0);
2039 if (MaskedID && F->arg_size() == 4) {
2040 rename(F);
2041 if (MaskedID == Intrinsic::masked_load ||
2042 MaskedID == Intrinsic::masked_gather) {
2044 F->getParent(), MaskedID,
2045 {F->getReturnType(), F->getArg(0)->getType()});
2046 return true;
2047 }
2049 F->getParent(), MaskedID,
2050 {F->getArg(0)->getType(), F->getArg(1)->getType()});
2051 return true;
2052 }
2053 break;
2054 }
2055 case 'n': {
2056 if (Name.consume_front("nvvm.")) {
2057 // Check for nvvm intrinsics corresponding exactly to an LLVM intrinsic.
2058 if (F->arg_size() == 1) {
2059 Intrinsic::ID IID =
2061 .Cases({"brev32", "brev64"}, Intrinsic::bitreverse)
2062 .Case("clz.i", Intrinsic::ctlz)
2063 .Case("popc.i", Intrinsic::ctpop)
2065 if (IID != Intrinsic::not_intrinsic) {
2066 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2067 {F->getReturnType()});
2068 return true;
2069 }
2070 } else if (F->arg_size() == 2) {
2071 Intrinsic::ID IID =
2073 .Cases({"max.s", "max.i", "max.ll"}, Intrinsic::smax)
2074 .Cases({"min.s", "min.i", "min.ll"}, Intrinsic::smin)
2075 .Cases({"max.us", "max.ui", "max.ull"}, Intrinsic::umax)
2076 .Cases({"min.us", "min.ui", "min.ull"}, Intrinsic::umin)
2077 .Cases({"mulhi.s", "mulhi.i", "mulhi.ll"}, Intrinsic::smulh)
2078 .Cases({"mulhi.us", "mulhi.ui", "mulhi.ull"}, Intrinsic::umulh)
2080 if (IID != Intrinsic::not_intrinsic) {
2081 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2082 {F->getReturnType()});
2083 return true;
2084 }
2085 }
2086
2087 // Check for nvvm intrinsics that need a return type adjustment.
2088 {
2090 if (IID != Intrinsic::not_intrinsic &&
2092 NewFn = nullptr;
2093 return true;
2094 }
2095 }
2096
2097 // Upgrade Distributed Shared Memory Intrinsics
2099 if (IID != Intrinsic::not_intrinsic) {
2100 rename(F);
2101 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2102 return true;
2103 }
2104
2105 // Upgrade TMA reduction intrinsics
2106 // llvm.nvvm.cp.async.bulk.tensor.reduce.<red_op>* =>
2107 // llvm.nvvm.cp.async.bulk.tensor.reduce.<shape>*
2109 if (IID != Intrinsic::not_intrinsic) {
2110 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2111 return true;
2112 }
2113
2114 // Upgrade tcgen05.commit shared variants to anyptr intrinsics.
2116 if (IID != Intrinsic::not_intrinsic) {
2117 rename(F);
2119 F->getParent(), IID, F->getReturnType(),
2120 F->getFunctionType()->params());
2121 return true;
2122 }
2123
2124 // Upgrade tcgen05.alloc/dealloc with the is_exclusive argument and
2125 // tcgen05.alloc shared variants to anyptr intrinsics.
2127 if (IID != Intrinsic::not_intrinsic) {
2128 rename(F);
2129 if (Intrinsic::isOverloaded(IID))
2130 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2131 {F->getArg(0)->getType()});
2132 else
2133 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2134 return true;
2135 }
2136
2137 // Upgrade TMA copy G2S CTA intrinsics.
2139 if (IID != Intrinsic::not_intrinsic) {
2140 rename(F);
2141 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2142 return true;
2143 }
2144
2145 // Upgrade TMA copy G2S (cluster) intrinsics.
2147 IID = shouldUpgradeNVPTXTMAG2SIntrinsics(F, Name, OvlTys);
2148 if (IID != Intrinsic::not_intrinsic) {
2149 rename(F);
2150 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID, OvlTys);
2151 return true;
2152 }
2153
2154 // Upgrade the legacy cp.async.bulk.global.to.shared.cluster signature
2155 // (multicast-mask overloading + trailing flag_valid_pattern).
2156 SmallVector<Type *, 1> BulkG2SOvlTys;
2157 IID = shouldUpgradeNVPTXBulkG2SClusterIntrinsic(F, Name, BulkG2SOvlTys);
2158 if (IID != Intrinsic::not_intrinsic) {
2159 rename(F);
2160 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2161 BulkG2SOvlTys);
2162 return true;
2163 }
2164
2165 // Upgrade the legacy cp.async.bulk.global.to.shared.cta signature
2166 // (no ignore_bytes_left/right + trailing flag_valid_pattern).
2168 if (IID != Intrinsic::not_intrinsic) {
2169 rename(F);
2170 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
2171 return true;
2172 }
2173
2174 // Upgrade mbarrier.init intrinsics missing the layout operand.
2176 if (IID != Intrinsic::not_intrinsic) {
2177 rename(F);
2178 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID,
2179 F->getArg(0)->getType());
2180 return true;
2181 }
2182
2183 // The following nvvm intrinsics correspond exactly to an LLVM idiom, but
2184 // not to an intrinsic alone. We expand them in UpgradeIntrinsicCall.
2185 //
2186 // TODO: We could add lohi.i2d.
2187 bool Expand = false;
2188 if (Name.consume_front("abs."))
2189 // nvvm.abs.{i,ii}
2190 Expand =
2191 Name == "i" || Name == "ll" || Name == "bf16" || Name == "bf16x2";
2192 else if (Name.consume_front("fabs."))
2193 // nvvm.fabs.{f,ftz.f,d}
2194 Expand = Name == "f" || Name == "ftz.f" || Name == "d";
2195 else if (Name.consume_front("add."))
2196 // nvvm.add.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
2197 Expand = getNVVMFAddUpgrade(Name).has_value();
2198 else if (Name.consume_front("ex2.approx."))
2199 // nvvm.ex2.approx.{f,ftz.f,d,f16x2}
2200 Expand =
2201 Name == "f" || Name == "ftz.f" || Name == "d" || Name == "f16x2";
2202 else if (Name.consume_front("atomic.load."))
2203 // nvvm.atomic.load.add.{f32,f64}.p
2204 // nvvm.atomic.load.{inc,dec}.32.p
2205 Expand = StringSwitch<bool>(Name)
2206 .StartsWith("add.f32.p", true)
2207 .StartsWith("add.f64.p", true)
2208 .StartsWith("inc.32.p", true)
2209 .StartsWith("dec.32.p", true)
2210 .Default(false);
2211 else if (Name.consume_front("atomic."))
2212 // nvvm.atomic.{add,exch,max,min,inc,dec,and,or,xor}.gen.{i,f}.{cta,sys}
2213 // nvvm.atomic.cas.gen.i.{cta,sys}
2214 Expand = StringSwitch<bool>(Name)
2215 .StartsWith("add.gen.", true)
2216 .StartsWith("exch.gen.", true)
2217 .StartsWith("max.gen.", true)
2218 .StartsWith("min.gen.", true)
2219 .StartsWith("inc.gen.", true)
2220 .StartsWith("dec.gen.", true)
2221 .StartsWith("and.gen.", true)
2222 .StartsWith("or.gen.", true)
2223 .StartsWith("xor.gen.", true)
2224 .StartsWith("cas.gen.", true)
2225 .Default(false);
2226 else if (Name.consume_front("bitcast."))
2227 // nvvm.bitcast.{f2i,i2f,ll2d,d2ll}
2228 Expand =
2229 Name == "f2i" || Name == "i2f" || Name == "ll2d" || Name == "d2ll";
2230 else if (Name.consume_front("rotate."))
2231 // nvvm.rotate.{b32,b64,right.b64}
2232 Expand = Name == "b32" || Name == "b64" || Name == "right.b64";
2233 else if (Name.consume_front("ptr.gen.to."))
2234 // nvvm.ptr.gen.to.{local,shared,global,constant,param}
2235 Expand = consumeNVVMPtrAddrSpace(Name);
2236 else if (Name.consume_front("ptr."))
2237 // nvvm.ptr.{local,shared,global,constant,param}.to.gen
2238 Expand = consumeNVVMPtrAddrSpace(Name) && Name.starts_with(".to.gen");
2239 else if (Name.consume_front("ldg.global."))
2240 // nvvm.ldg.global.{i,p,f}
2241 Expand = (Name.starts_with("i.") || Name.starts_with("f.") ||
2242 Name.starts_with("p."));
2243 else
2244 Expand = StringSwitch<bool>(Name)
2245 .Case("barrier0", true)
2246 .Case("barrier.n", true)
2247 .Case("barrier.sync.cnt", true)
2248 .Case("barrier.sync", true)
2249 .Case("barrier", true)
2250 .Case("bar.sync", true)
2251 .Case("barrier0.popc", true)
2252 .Case("barrier0.and", true)
2253 .Case("barrier0.or", true)
2254 .Case("clz.ll", true)
2255 .Case("popc.ll", true)
2256 .Case("h2f", true)
2257 .Case("swap.lo.hi.b64", true)
2258 .Case("tanh.approx.f32", true)
2259 .Default(false);
2260
2261 if (Expand) {
2262 NewFn = nullptr;
2263 return true;
2264 }
2265 break; // No other 'nvvm.*'.
2266 }
2267 break;
2268 }
2269 case 'o':
2270 if (Name.starts_with("objectsize.")) {
2271 Type *Tys[2] = { F->getReturnType(), F->arg_begin()->getType() };
2272 if (F->arg_size() == 2 || F->arg_size() == 3) {
2273 rename(F);
2274 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(),
2275 Intrinsic::objectsize, Tys);
2276 return true;
2277 }
2278 }
2279 break;
2280
2281 case 'p':
2282 if (Name.starts_with("ptr.annotation.") && F->arg_size() == 4) {
2283 rename(F);
2285 F->getParent(), Intrinsic::ptr_annotation,
2286 {F->arg_begin()->getType(), F->getArg(1)->getType()});
2287 return true;
2288 }
2289 break;
2290
2291 case 'r': {
2292 if (Name.consume_front("riscv.")) {
2293 Intrinsic::ID ID;
2295 .Case("aes32dsi", Intrinsic::riscv_aes32dsi)
2296 .Case("aes32dsmi", Intrinsic::riscv_aes32dsmi)
2297 .Case("aes32esi", Intrinsic::riscv_aes32esi)
2298 .Case("aes32esmi", Intrinsic::riscv_aes32esmi)
2300 if (ID != Intrinsic::not_intrinsic) {
2301 if (!F->getFunctionType()->getParamType(2)->isIntegerTy(32)) {
2302 rename(F);
2303 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2304 return true;
2305 }
2306 break; // No other applicable upgrades.
2307 }
2308
2310 .StartsWith("sm4ks", Intrinsic::riscv_sm4ks)
2311 .StartsWith("sm4ed", Intrinsic::riscv_sm4ed)
2313 if (ID != Intrinsic::not_intrinsic) {
2314 if (!F->getFunctionType()->getParamType(2)->isIntegerTy(32) ||
2315 F->getFunctionType()->getReturnType()->isIntegerTy(64)) {
2316 rename(F);
2317 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2318 return true;
2319 }
2320 break; // No other applicable upgrades.
2321 }
2322
2324 .StartsWith("sha256sig0", Intrinsic::riscv_sha256sig0)
2325 .StartsWith("sha256sig1", Intrinsic::riscv_sha256sig1)
2326 .StartsWith("sha256sum0", Intrinsic::riscv_sha256sum0)
2327 .StartsWith("sha256sum1", Intrinsic::riscv_sha256sum1)
2328 .StartsWith("sm3p0", Intrinsic::riscv_sm3p0)
2329 .StartsWith("sm3p1", Intrinsic::riscv_sm3p1)
2331 if (ID != Intrinsic::not_intrinsic) {
2332 if (F->getFunctionType()->getReturnType()->isIntegerTy(64)) {
2333 rename(F);
2334 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2335 return true;
2336 }
2337 break; // No other applicable upgrades.
2338 }
2339
2340 // Replace llvm.riscv.clmul with llvm.clmul.
2341 if (Name == "clmul.i32" || Name == "clmul.i64") {
2343 F->getParent(), Intrinsic::clmul, {F->getReturnType()});
2344 return true;
2345 }
2346
2347 break; // No other 'riscv.*' intrinsics
2348 }
2349 } break;
2350
2351 case 's':
2352 if (Name == "stackprotectorcheck") {
2353 NewFn = nullptr;
2354 return true;
2355 }
2356 if (Name.starts_with("strip.invariant.group")) {
2357 // For clang's usage it would be safe to just drop the
2358 // strip.invariant.group, but to be conservative replace with the
2359 // stronger launder.invariant.group instead.
2361 F->getParent(), Intrinsic::launder_invariant_group,
2362 F->getReturnType());
2363 return true;
2364 }
2365 break;
2366
2367 case 't':
2368 if (Name == "thread.pointer") {
2370 F->getParent(), Intrinsic::thread_pointer, F->getReturnType());
2371 return true;
2372 }
2373 break;
2374
2375 case 'v': {
2376 if (Name == "var.annotation" && F->arg_size() == 4) {
2377 rename(F);
2379 F->getParent(), Intrinsic::var_annotation,
2380 {{F->arg_begin()->getType(), F->getArg(1)->getType()}});
2381 return true;
2382 }
2383 if (Name.consume_front("vector.splice")) {
2384 if (Name.starts_with(".left") || Name.starts_with(".right"))
2385 break;
2386 return true;
2387 }
2388 if (shouldUpgradeVPIntrinsic(Name))
2389 return true;
2390 break;
2391 }
2392
2393 case 'w':
2394 if (Name.consume_front("wasm.")) {
2395 Intrinsic::ID ID =
2397 .StartsWith("fma.", Intrinsic::wasm_relaxed_madd)
2398 .StartsWith("fms.", Intrinsic::wasm_relaxed_nmadd)
2399 .StartsWith("laneselect.", Intrinsic::wasm_relaxed_laneselect)
2401 if (ID != Intrinsic::not_intrinsic) {
2402 rename(F);
2403 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID,
2404 F->getReturnType());
2405 return true;
2406 }
2407
2408 if (Name.consume_front("dot.i8x16.i7x16.")) {
2410 .Case("signed", Intrinsic::wasm_relaxed_dot_i8x16_i7x16_signed)
2411 .Case("add.signed",
2412 Intrinsic::wasm_relaxed_dot_i8x16_i7x16_add_signed)
2414 if (ID != Intrinsic::not_intrinsic) {
2415 rename(F);
2416 NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), ID);
2417 return true;
2418 }
2419 break; // No other 'wasm.dot.i8x16.i7x16.*'.
2420 }
2421 break; // No other 'wasm.*'.
2422 }
2423 break;
2424
2425 case 'x':
2426 if (upgradeX86IntrinsicFunction(F, Name, NewFn))
2427 return true;
2428 }
2429
2430 auto *ST = dyn_cast<StructType>(F->getReturnType());
2431 if (ST && (!ST->isLiteral() || ST->isPacked()) &&
2432 F->getIntrinsicID() != Intrinsic::not_intrinsic) {
2433 // Replace return type with literal non-packed struct. Only do this for
2434 // intrinsics declared to return a struct, not for intrinsics with
2435 // overloaded return type, in which case the exact struct type will be
2436 // mangled into the name.
2437 if (Intrinsic::hasStructReturnType(F->getIntrinsicID())) {
2438 FunctionType *FT = F->getFunctionType();
2439 auto *NewST = StructType::get(ST->getContext(), ST->elements());
2440 auto *NewFT = FunctionType::get(NewST, FT->params(), FT->isVarArg());
2441 std::string Name = F->getName().str();
2442 rename(F);
2443 NewFn = Function::Create(NewFT, F->getLinkage(), F->getAddressSpace(),
2444 Name, F->getParent());
2445
2446 // The new function may also need remangling.
2447 if (auto Result = llvm::Intrinsic::remangleIntrinsicFunction(NewFn))
2448 NewFn = *Result;
2449 return true;
2450 }
2451 }
2452
2453 // Remangle our intrinsic since we upgrade the mangling
2455 if (Result != std::nullopt) {
2456 NewFn = *Result;
2457 return true;
2458 }
2459
2461 return true;
2462
2463 // This may not belong here. This function is effectively being overloaded
2464 // to both detect an intrinsic which needs upgrading, and to provide the
2465 // upgraded form of the intrinsic. We should perhaps have two separate
2466 // functions for this.
2467
2468 return false;
2469}
2470
2472 bool CanUpgradeDebugIntrinsicsToRecords) {
2473 NewFn = nullptr;
2474 bool Upgraded =
2475 upgradeIntrinsicFunction1(F, NewFn, CanUpgradeDebugIntrinsicsToRecords);
2476
2477 // Upgrade intrinsic attributes. This does not change the function.
2478 if (NewFn)
2479 F = NewFn;
2480 if (Intrinsic::ID id = F->getIntrinsicID()) {
2481 // Only do this if the intrinsic signature is valid.
2482 SmallVector<Type *> OverloadTys;
2483 if (Intrinsic::isSignatureValid(id, F->getFunctionType(), OverloadTys))
2484 F->setAttributes(
2485 Intrinsic::getAttributes(F->getContext(), id, F->getFunctionType()));
2486 }
2487 return Upgraded;
2488}
2489
2491 if (!(GV->hasName() && (GV->getName() == "llvm.global_ctors" ||
2492 GV->getName() == "llvm.global_dtors")) ||
2493 !GV->hasInitializer())
2494 return nullptr;
2496 if (!ATy)
2497 return nullptr;
2499 if (!STy || STy->getNumElements() != 2)
2500 return nullptr;
2501
2502 LLVMContext &C = GV->getContext();
2503 IRBuilder<> IRB(C);
2504 auto EltTy = StructType::get(STy->getElementType(0), STy->getElementType(1),
2505 IRB.getPtrTy());
2506 Constant *Init = GV->getInitializer();
2507 unsigned N = Init->getNumOperands();
2508 std::vector<Constant *> NewCtors(N);
2509 for (unsigned i = 0; i != N; ++i) {
2510 auto Ctor = cast<Constant>(Init->getOperand(i));
2511 NewCtors[i] = ConstantStruct::get(EltTy, Ctor->getAggregateElement(0u),
2512 Ctor->getAggregateElement(1),
2514 }
2515 Constant *NewInit = ConstantArray::get(ArrayType::get(EltTy, N), NewCtors);
2516
2517 return new GlobalVariable(NewInit->getType(), false, GV->getLinkage(),
2518 NewInit, GV->getName());
2519}
2520
2521// Handles upgrading SSE2/AVX2/AVX512BW PSLLDQ intrinsics by converting them
2522// to byte shuffles.
2524 unsigned Shift) {
2525 auto *ResultTy = cast<FixedVectorType>(Op->getType());
2526 unsigned NumElts = ResultTy->getNumElements() * 8;
2527
2528 // Bitcast from a 64-bit element type to a byte element type.
2529 Type *VecTy = FixedVectorType::get(Builder.getInt8Ty(), NumElts);
2530 Op = Builder.CreateBitCast(Op, VecTy, "cast");
2531
2532 // We'll be shuffling in zeroes.
2533 Value *Res = Constant::getNullValue(VecTy);
2534
2535 // If shift is less than 16, emit a shuffle to move the bytes. Otherwise,
2536 // we'll just return the zero vector.
2537 if (Shift < 16) {
2538 int Idxs[64];
2539 // 256/512-bit version is split into 2/4 16-byte lanes.
2540 for (unsigned l = 0; l != NumElts; l += 16)
2541 for (unsigned i = 0; i != 16; ++i) {
2542 unsigned Idx = NumElts + i - Shift;
2543 if (Idx < NumElts)
2544 Idx -= NumElts - 16; // end of lane, switch operand.
2545 Idxs[l + i] = Idx + l;
2546 }
2547
2548 Res = Builder.CreateShuffleVector(Res, Op, ArrayRef(Idxs, NumElts));
2549 }
2550
2551 // Bitcast back to a 64-bit element type.
2552 return Builder.CreateBitCast(Res, ResultTy, "cast");
2553}
2554
2555// Handles upgrading SSE2/AVX2/AVX512BW PSRLDQ intrinsics by converting them
2556// to byte shuffles.
2558 unsigned Shift) {
2559 auto *ResultTy = cast<FixedVectorType>(Op->getType());
2560 unsigned NumElts = ResultTy->getNumElements() * 8;
2561
2562 // Bitcast from a 64-bit element type to a byte element type.
2563 Type *VecTy = FixedVectorType::get(Builder.getInt8Ty(), NumElts);
2564 Op = Builder.CreateBitCast(Op, VecTy, "cast");
2565
2566 // We'll be shuffling in zeroes.
2567 Value *Res = Constant::getNullValue(VecTy);
2568
2569 // If shift is less than 16, emit a shuffle to move the bytes. Otherwise,
2570 // we'll just return the zero vector.
2571 if (Shift < 16) {
2572 int Idxs[64];
2573 // 256/512-bit version is split into 2/4 16-byte lanes.
2574 for (unsigned l = 0; l != NumElts; l += 16)
2575 for (unsigned i = 0; i != 16; ++i) {
2576 unsigned Idx = i + Shift;
2577 if (Idx >= 16)
2578 Idx += NumElts - 16; // end of lane, switch operand.
2579 Idxs[l + i] = Idx + l;
2580 }
2581
2582 Res = Builder.CreateShuffleVector(Op, Res, ArrayRef(Idxs, NumElts));
2583 }
2584
2585 // Bitcast back to a 64-bit element type.
2586 return Builder.CreateBitCast(Res, ResultTy, "cast");
2587}
2588
2589static Value *getX86MaskVec(IRBuilder<> &Builder, Value *Mask,
2590 unsigned NumElts) {
2591 assert(isPowerOf2_32(NumElts) && "Expected power-of-2 mask elements");
2593 Builder.getInt1Ty(), cast<IntegerType>(Mask->getType())->getBitWidth());
2594 Mask = Builder.CreateBitCast(Mask, MaskTy);
2595
2596 // If we have less than 8 elements (1, 2 or 4), then the starting mask was an
2597 // i8 and we need to extract down to the right number of elements.
2598 if (NumElts <= 4) {
2599 int Indices[4];
2600 for (unsigned i = 0; i != NumElts; ++i)
2601 Indices[i] = i;
2602 Mask = Builder.CreateShuffleVector(Mask, Mask, ArrayRef(Indices, NumElts),
2603 "extract");
2604 }
2605
2606 return Mask;
2607}
2608
2609static Value *emitX86Select(IRBuilder<> &Builder, Value *Mask, Value *Op0,
2610 Value *Op1) {
2611 // If the mask is all ones just emit the first operation.
2612 if (const auto *C = dyn_cast<Constant>(Mask))
2613 if (C->isAllOnesValue())
2614 return Op0;
2615
2616 Mask = getX86MaskVec(Builder, Mask,
2617 cast<FixedVectorType>(Op0->getType())->getNumElements());
2618 return Builder.CreateSelect(Mask, Op0, Op1);
2619}
2620
2621static Value *emitX86ScalarSelect(IRBuilder<> &Builder, Value *Mask, Value *Op0,
2622 Value *Op1) {
2623 // If the mask is all ones just emit the first operation.
2624 if (const auto *C = dyn_cast<Constant>(Mask))
2625 if (C->isAllOnesValue())
2626 return Op0;
2627
2628 auto *MaskTy = FixedVectorType::get(Builder.getInt1Ty(),
2629 Mask->getType()->getIntegerBitWidth());
2630 Mask = Builder.CreateBitCast(Mask, MaskTy);
2631 Mask = Builder.CreateExtractElement(Mask, (uint64_t)0);
2632 return Builder.CreateSelect(Mask, Op0, Op1);
2633}
2634
2635// Handle autoupgrade for masked PALIGNR and VALIGND/Q intrinsics.
2636// PALIGNR handles large immediates by shifting while VALIGN masks the immediate
2637// so we need to handle both cases. VALIGN also doesn't have 128-bit lanes.
2639 Value *Op1, Value *Shift,
2640 Value *Passthru, Value *Mask,
2641 bool IsVALIGN) {
2642 unsigned ShiftVal = cast<llvm::ConstantInt>(Shift)->getZExtValue();
2643
2644 unsigned NumElts = cast<FixedVectorType>(Op0->getType())->getNumElements();
2645 assert((IsVALIGN || NumElts % 16 == 0) && "Illegal NumElts for PALIGNR!");
2646 assert((!IsVALIGN || NumElts <= 16) && "NumElts too large for VALIGN!");
2647 assert(isPowerOf2_32(NumElts) && "NumElts not a power of 2!");
2648
2649 // Mask the immediate for VALIGN.
2650 if (IsVALIGN)
2651 ShiftVal &= (NumElts - 1);
2652
2653 // If palignr is shifting the pair of vectors more than the size of two
2654 // lanes, emit zero.
2655 if (ShiftVal >= 32)
2657
2658 // If palignr is shifting the pair of input vectors more than one lane,
2659 // but less than two lanes, convert to shifting in zeroes.
2660 if (ShiftVal > 16) {
2661 ShiftVal -= 16;
2662 Op1 = Op0;
2664 }
2665
2666 int Indices[64];
2667 // 256-bit palignr operates on 128-bit lanes so we need to handle that
2668 for (unsigned l = 0; l < NumElts; l += 16) {
2669 for (unsigned i = 0; i != 16; ++i) {
2670 unsigned Idx = ShiftVal + i;
2671 if (!IsVALIGN && Idx >= 16) // Disable wrap for VALIGN.
2672 Idx += NumElts - 16; // End of lane, switch operand.
2673 Indices[l + i] = Idx + l;
2674 }
2675 }
2676
2677 Value *Align = Builder.CreateShuffleVector(
2678 Op1, Op0, ArrayRef(Indices, NumElts), "palignr");
2679
2680 return emitX86Select(Builder, Mask, Align, Passthru);
2681}
2682
2684 bool ZeroMask, bool IndexForm) {
2685 Type *Ty = CI.getType();
2686 unsigned VecWidth = Ty->getPrimitiveSizeInBits();
2687 unsigned EltWidth = Ty->getScalarSizeInBits();
2688 bool IsFloat = Ty->isFPOrFPVectorTy();
2689 Intrinsic::ID IID;
2690 if (VecWidth == 128 && EltWidth == 32 && IsFloat)
2691 IID = Intrinsic::x86_avx512_vpermi2var_ps_128;
2692 else if (VecWidth == 128 && EltWidth == 32 && !IsFloat)
2693 IID = Intrinsic::x86_avx512_vpermi2var_d_128;
2694 else if (VecWidth == 128 && EltWidth == 64 && IsFloat)
2695 IID = Intrinsic::x86_avx512_vpermi2var_pd_128;
2696 else if (VecWidth == 128 && EltWidth == 64 && !IsFloat)
2697 IID = Intrinsic::x86_avx512_vpermi2var_q_128;
2698 else if (VecWidth == 256 && EltWidth == 32 && IsFloat)
2699 IID = Intrinsic::x86_avx512_vpermi2var_ps_256;
2700 else if (VecWidth == 256 && EltWidth == 32 && !IsFloat)
2701 IID = Intrinsic::x86_avx512_vpermi2var_d_256;
2702 else if (VecWidth == 256 && EltWidth == 64 && IsFloat)
2703 IID = Intrinsic::x86_avx512_vpermi2var_pd_256;
2704 else if (VecWidth == 256 && EltWidth == 64 && !IsFloat)
2705 IID = Intrinsic::x86_avx512_vpermi2var_q_256;
2706 else if (VecWidth == 512 && EltWidth == 32 && IsFloat)
2707 IID = Intrinsic::x86_avx512_vpermi2var_ps_512;
2708 else if (VecWidth == 512 && EltWidth == 32 && !IsFloat)
2709 IID = Intrinsic::x86_avx512_vpermi2var_d_512;
2710 else if (VecWidth == 512 && EltWidth == 64 && IsFloat)
2711 IID = Intrinsic::x86_avx512_vpermi2var_pd_512;
2712 else if (VecWidth == 512 && EltWidth == 64 && !IsFloat)
2713 IID = Intrinsic::x86_avx512_vpermi2var_q_512;
2714 else if (VecWidth == 128 && EltWidth == 16)
2715 IID = Intrinsic::x86_avx512_vpermi2var_hi_128;
2716 else if (VecWidth == 256 && EltWidth == 16)
2717 IID = Intrinsic::x86_avx512_vpermi2var_hi_256;
2718 else if (VecWidth == 512 && EltWidth == 16)
2719 IID = Intrinsic::x86_avx512_vpermi2var_hi_512;
2720 else if (VecWidth == 128 && EltWidth == 8)
2721 IID = Intrinsic::x86_avx512_vpermi2var_qi_128;
2722 else if (VecWidth == 256 && EltWidth == 8)
2723 IID = Intrinsic::x86_avx512_vpermi2var_qi_256;
2724 else if (VecWidth == 512 && EltWidth == 8)
2725 IID = Intrinsic::x86_avx512_vpermi2var_qi_512;
2726 else
2727 llvm_unreachable("Unexpected intrinsic");
2728
2729 Value *Args[] = { CI.getArgOperand(0) , CI.getArgOperand(1),
2730 CI.getArgOperand(2) };
2731
2732 // If this isn't index form we need to swap operand 0 and 1.
2733 if (!IndexForm)
2734 std::swap(Args[0], Args[1]);
2735
2736 Value *V = Builder.CreateIntrinsic(IID, Args);
2737 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(Ty)
2738 : Builder.CreateBitCast(CI.getArgOperand(1),
2739 Ty);
2740 return emitX86Select(Builder, CI.getArgOperand(3), V, PassThru);
2741}
2742
2744 Intrinsic::ID IID) {
2745 Type *Ty = CI.getType();
2746 Value *Op0 = CI.getOperand(0);
2747 Value *Op1 = CI.getOperand(1);
2748 Value *Res = Builder.CreateIntrinsic(IID, Ty, {Op0, Op1});
2749
2750 if (CI.arg_size() == 4) { // For masked intrinsics.
2751 Value *VecSrc = CI.getOperand(2);
2752 Value *Mask = CI.getOperand(3);
2753 Res = emitX86Select(Builder, Mask, Res, VecSrc);
2754 }
2755 return Res;
2756}
2757
2759 bool IsRotateRight) {
2760 Type *Ty = CI.getType();
2761 Value *Src = CI.getArgOperand(0);
2762 Value *Amt = CI.getArgOperand(1);
2763
2764 // Amount may be scalar immediate, in which case create a splat vector.
2765 // Funnel shifts amounts are treated as modulo and types are all power-of-2 so
2766 // we only care about the lowest log2 bits anyway.
2767 if (Amt->getType() != Ty) {
2768 unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements();
2769 Amt = Builder.CreateIntCast(Amt, Ty->getScalarType(), false);
2770 Amt = Builder.CreateVectorSplat(NumElts, Amt);
2771 }
2772
2773 Intrinsic::ID IID = IsRotateRight ? Intrinsic::fshr : Intrinsic::fshl;
2774 Value *Res = Builder.CreateIntrinsic(IID, Ty, {Src, Src, Amt});
2775
2776 if (CI.arg_size() == 4) { // For masked intrinsics.
2777 Value *VecSrc = CI.getOperand(2);
2778 Value *Mask = CI.getOperand(3);
2779 Res = emitX86Select(Builder, Mask, Res, VecSrc);
2780 }
2781 return Res;
2782}
2783
2784static Value *upgradeX86vpcom(IRBuilder<> &Builder, CallBase &CI, unsigned Imm,
2785 bool IsSigned) {
2786 Type *Ty = CI.getType();
2787 Value *LHS = CI.getArgOperand(0);
2788 Value *RHS = CI.getArgOperand(1);
2789
2790 CmpInst::Predicate Pred;
2791 switch (Imm) {
2792 case 0x0:
2793 Pred = IsSigned ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT;
2794 break;
2795 case 0x1:
2796 Pred = IsSigned ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE;
2797 break;
2798 case 0x2:
2799 Pred = IsSigned ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT;
2800 break;
2801 case 0x3:
2802 Pred = IsSigned ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE;
2803 break;
2804 case 0x4:
2805 Pred = ICmpInst::ICMP_EQ;
2806 break;
2807 case 0x5:
2808 Pred = ICmpInst::ICMP_NE;
2809 break;
2810 case 0x6:
2811 return Constant::getNullValue(Ty); // FALSE
2812 case 0x7:
2813 return Constant::getAllOnesValue(Ty); // TRUE
2814 default:
2815 llvm_unreachable("Unknown XOP vpcom/vpcomu predicate");
2816 }
2817
2818 Value *Cmp = Builder.CreateICmp(Pred, LHS, RHS);
2819 Value *Ext = Builder.CreateSExt(Cmp, Ty);
2820 return Ext;
2821}
2822
2824 bool IsShiftRight, bool ZeroMask) {
2825 Type *Ty = CI.getType();
2826 Value *Op0 = CI.getArgOperand(0);
2827 Value *Op1 = CI.getArgOperand(1);
2828 Value *Amt = CI.getArgOperand(2);
2829
2830 if (IsShiftRight)
2831 std::swap(Op0, Op1);
2832
2833 // Amount may be scalar immediate, in which case create a splat vector.
2834 // Funnel shifts amounts are treated as modulo and types are all power-of-2 so
2835 // we only care about the lowest log2 bits anyway.
2836 if (Amt->getType() != Ty) {
2837 unsigned NumElts = cast<FixedVectorType>(Ty)->getNumElements();
2838 Amt = Builder.CreateIntCast(Amt, Ty->getScalarType(), false);
2839 Amt = Builder.CreateVectorSplat(NumElts, Amt);
2840 }
2841
2842 Intrinsic::ID IID = IsShiftRight ? Intrinsic::fshr : Intrinsic::fshl;
2843 Value *Res = Builder.CreateIntrinsic(IID, Ty, {Op0, Op1, Amt});
2844
2845 unsigned NumArgs = CI.arg_size();
2846 if (NumArgs >= 4) { // For masked intrinsics.
2847 Value *VecSrc = NumArgs == 5 ? CI.getArgOperand(3) :
2848 ZeroMask ? ConstantAggregateZero::get(CI.getType()) :
2849 CI.getArgOperand(0);
2850 Value *Mask = CI.getOperand(NumArgs - 1);
2851 Res = emitX86Select(Builder, Mask, Res, VecSrc);
2852 }
2853 return Res;
2854}
2855
2857 Value *Mask, bool Aligned) {
2858 const Align Alignment =
2859 Aligned
2860 ? Align(Data->getType()->getPrimitiveSizeInBits().getFixedValue() / 8)
2861 : Align(1);
2862
2863 // If the mask is all ones just emit a regular store.
2864 if (const auto *C = dyn_cast<Constant>(Mask))
2865 if (C->isAllOnesValue())
2866 return Builder.CreateAlignedStore(Data, Ptr, Alignment);
2867
2868 // Convert the mask from an integer type to a vector of i1.
2869 unsigned NumElts = cast<FixedVectorType>(Data->getType())->getNumElements();
2870 Mask = getX86MaskVec(Builder, Mask, NumElts);
2871 return Builder.CreateMaskedStore(Data, Ptr, Alignment, Mask);
2872}
2873
2875 Value *Passthru, Value *Mask, bool Aligned) {
2876 Type *ValTy = Passthru->getType();
2877 const Align Alignment =
2878 Aligned
2879 ? Align(
2881 8)
2882 : Align(1);
2883
2884 // If the mask is all ones just emit a regular store.
2885 if (const auto *C = dyn_cast<Constant>(Mask))
2886 if (C->isAllOnesValue())
2887 return Builder.CreateAlignedLoad(ValTy, Ptr, Alignment);
2888
2889 // Convert the mask from an integer type to a vector of i1.
2890 unsigned NumElts = cast<FixedVectorType>(ValTy)->getNumElements();
2891 Mask = getX86MaskVec(Builder, Mask, NumElts);
2892 return Builder.CreateMaskedLoad(ValTy, Ptr, Alignment, Mask, Passthru);
2893}
2894
2895static Value *upgradeAbs(IRBuilder<> &Builder, CallBase &CI) {
2896 Type *Ty = CI.getType();
2897 Value *Op0 = CI.getArgOperand(0);
2898 Value *Res = Builder.CreateIntrinsic(Intrinsic::abs, Ty,
2899 {Op0, Builder.getInt1(false)});
2900 if (CI.arg_size() == 3)
2901 Res = emitX86Select(Builder, CI.getArgOperand(2), Res, CI.getArgOperand(1));
2902 return Res;
2903}
2904
2905static Value *upgradePMULDQ(IRBuilder<> &Builder, CallBase &CI, bool IsSigned) {
2906 Type *Ty = CI.getType();
2907
2908 // Arguments have a vXi32 type so cast to vXi64.
2909 Value *LHS = Builder.CreateBitCast(CI.getArgOperand(0), Ty);
2910 Value *RHS = Builder.CreateBitCast(CI.getArgOperand(1), Ty);
2911
2912 if (IsSigned) {
2913 // Shift left then arithmetic shift right.
2914 Constant *ShiftAmt = ConstantInt::get(Ty, 32);
2915 LHS = Builder.CreateShl(LHS, ShiftAmt);
2916 LHS = Builder.CreateAShr(LHS, ShiftAmt);
2917 RHS = Builder.CreateShl(RHS, ShiftAmt);
2918 RHS = Builder.CreateAShr(RHS, ShiftAmt);
2919 } else {
2920 // Clear the upper bits.
2921 Constant *Mask = ConstantInt::get(Ty, 0xffffffff);
2922 LHS = Builder.CreateAnd(LHS, Mask);
2923 RHS = Builder.CreateAnd(RHS, Mask);
2924 }
2925
2926 Value *Res = Builder.CreateMul(LHS, RHS);
2927
2928 if (CI.arg_size() == 4)
2929 Res = emitX86Select(Builder, CI.getArgOperand(3), Res, CI.getArgOperand(2));
2930
2931 return Res;
2932}
2933
2934// Applying mask on vector of i1's and make sure result is at least 8 bits wide.
2936 Value *Mask) {
2937 unsigned NumElts = cast<FixedVectorType>(Vec->getType())->getNumElements();
2938 if (Mask) {
2939 const auto *C = dyn_cast<Constant>(Mask);
2940 if (!C || !C->isAllOnesValue())
2941 Vec = Builder.CreateAnd(Vec, getX86MaskVec(Builder, Mask, NumElts));
2942 }
2943
2944 if (NumElts < 8) {
2945 int Indices[8];
2946 for (unsigned i = 0; i != NumElts; ++i)
2947 Indices[i] = i;
2948 for (unsigned i = NumElts; i != 8; ++i)
2949 Indices[i] = NumElts + i % NumElts;
2950 Vec = Builder.CreateShuffleVector(Vec,
2952 Indices);
2953 }
2954 return Builder.CreateBitCast(Vec, Builder.getIntNTy(std::max(NumElts, 8U)));
2955}
2956
2958 unsigned CC, bool Signed) {
2959 Value *Op0 = CI.getArgOperand(0);
2960 unsigned NumElts = cast<FixedVectorType>(Op0->getType())->getNumElements();
2961
2962 Value *Cmp;
2963 if (CC == 3) {
2965 FixedVectorType::get(Builder.getInt1Ty(), NumElts));
2966 } else if (CC == 7) {
2968 FixedVectorType::get(Builder.getInt1Ty(), NumElts));
2969 } else {
2971 switch (CC) {
2972 default: llvm_unreachable("Unknown condition code");
2973 case 0: Pred = ICmpInst::ICMP_EQ; break;
2974 case 1: Pred = Signed ? ICmpInst::ICMP_SLT : ICmpInst::ICMP_ULT; break;
2975 case 2: Pred = Signed ? ICmpInst::ICMP_SLE : ICmpInst::ICMP_ULE; break;
2976 case 4: Pred = ICmpInst::ICMP_NE; break;
2977 case 5: Pred = Signed ? ICmpInst::ICMP_SGE : ICmpInst::ICMP_UGE; break;
2978 case 6: Pred = Signed ? ICmpInst::ICMP_SGT : ICmpInst::ICMP_UGT; break;
2979 }
2980 Cmp = Builder.CreateICmp(Pred, Op0, CI.getArgOperand(1));
2981 }
2982
2983 Value *Mask = CI.getArgOperand(CI.arg_size() - 1);
2984
2985 return applyX86MaskOn1BitsVec(Builder, Cmp, Mask);
2986}
2987
2988// Replace a masked intrinsic with an older unmasked intrinsic.
2990 Intrinsic::ID IID) {
2991 Value *Rep =
2992 Builder.CreateIntrinsic(IID, {CI.getArgOperand(0), CI.getArgOperand(1)});
2993 return emitX86Select(Builder, CI.getArgOperand(3), Rep, CI.getArgOperand(2));
2994}
2995
2997 Value* A = CI.getArgOperand(0);
2998 Value* B = CI.getArgOperand(1);
2999 Value* Src = CI.getArgOperand(2);
3000 Value* Mask = CI.getArgOperand(3);
3001
3002 Value* AndNode = Builder.CreateAnd(Mask, APInt(8, 1));
3003 Value* Cmp = Builder.CreateIsNotNull(AndNode);
3004 Value* Extract1 = Builder.CreateExtractElement(B, (uint64_t)0);
3005 Value* Extract2 = Builder.CreateExtractElement(Src, (uint64_t)0);
3006 Value* Select = Builder.CreateSelect(Cmp, Extract1, Extract2);
3007 return Builder.CreateInsertElement(A, Select, (uint64_t)0);
3008}
3009
3011 Value* Op = CI.getArgOperand(0);
3012 Type* ReturnOp = CI.getType();
3013 unsigned NumElts = cast<FixedVectorType>(CI.getType())->getNumElements();
3014 Value *Mask = getX86MaskVec(Builder, Op, NumElts);
3015 return Builder.CreateSExt(Mask, ReturnOp, "vpmovm2");
3016}
3017
3018// Replace intrinsic with unmasked version and a select.
3020 CallBase &CI, Value *&Rep) {
3021 Name = Name.substr(12); // Remove avx512.mask.
3022
3023 unsigned VecWidth = CI.getType()->getPrimitiveSizeInBits();
3024 unsigned EltWidth = CI.getType()->getScalarSizeInBits();
3025 Intrinsic::ID IID;
3026 if (Name.starts_with("max.p")) {
3027 if (VecWidth == 128 && EltWidth == 32)
3028 IID = Intrinsic::x86_sse_max_ps;
3029 else if (VecWidth == 128 && EltWidth == 64)
3030 IID = Intrinsic::x86_sse2_max_pd;
3031 else if (VecWidth == 256 && EltWidth == 32)
3032 IID = Intrinsic::x86_avx_max_ps_256;
3033 else if (VecWidth == 256 && EltWidth == 64)
3034 IID = Intrinsic::x86_avx_max_pd_256;
3035 else
3036 llvm_unreachable("Unexpected intrinsic");
3037 } else if (Name.starts_with("min.p")) {
3038 if (VecWidth == 128 && EltWidth == 32)
3039 IID = Intrinsic::x86_sse_min_ps;
3040 else if (VecWidth == 128 && EltWidth == 64)
3041 IID = Intrinsic::x86_sse2_min_pd;
3042 else if (VecWidth == 256 && EltWidth == 32)
3043 IID = Intrinsic::x86_avx_min_ps_256;
3044 else if (VecWidth == 256 && EltWidth == 64)
3045 IID = Intrinsic::x86_avx_min_pd_256;
3046 else
3047 llvm_unreachable("Unexpected intrinsic");
3048 } else if (Name.starts_with("pshuf.b.")) {
3049 if (VecWidth == 128)
3050 IID = Intrinsic::x86_ssse3_pshuf_b_128;
3051 else if (VecWidth == 256)
3052 IID = Intrinsic::x86_avx2_pshuf_b;
3053 else if (VecWidth == 512)
3054 IID = Intrinsic::x86_avx512_pshuf_b_512;
3055 else
3056 llvm_unreachable("Unexpected intrinsic");
3057 } else if (Name.starts_with("pmul.hr.sw.")) {
3058 if (VecWidth == 128)
3059 IID = Intrinsic::x86_ssse3_pmul_hr_sw_128;
3060 else if (VecWidth == 256)
3061 IID = Intrinsic::x86_avx2_pmul_hr_sw;
3062 else if (VecWidth == 512)
3063 IID = Intrinsic::x86_avx512_pmul_hr_sw_512;
3064 else
3065 llvm_unreachable("Unexpected intrinsic");
3066 } else if (Name.starts_with("pmulh.w")) {
3067 assert((VecWidth == 128 || VecWidth == 256 || VecWidth == 512) &&
3068 "Unexpected intrinsic");
3069 Rep = upgradeX86BinaryIntrinsics(Builder, CI, Intrinsic::smulh);
3070 return true;
3071 } else if (Name.starts_with("pmulhu.w")) {
3072 assert((VecWidth == 128 || VecWidth == 256 || VecWidth == 512) &&
3073 "Unexpected intrinsic");
3074 Rep = upgradeX86BinaryIntrinsics(Builder, CI, Intrinsic::umulh);
3075 return true;
3076 } else if (Name.starts_with("pmaddw.d.")) {
3077 if (VecWidth == 128)
3078 IID = Intrinsic::x86_sse2_pmadd_wd;
3079 else if (VecWidth == 256)
3080 IID = Intrinsic::x86_avx2_pmadd_wd;
3081 else if (VecWidth == 512)
3082 IID = Intrinsic::x86_avx512_pmaddw_d_512;
3083 else
3084 llvm_unreachable("Unexpected intrinsic");
3085 } else if (Name.starts_with("pmaddubs.w.")) {
3086 if (VecWidth == 128)
3087 IID = Intrinsic::x86_ssse3_pmadd_ub_sw_128;
3088 else if (VecWidth == 256)
3089 IID = Intrinsic::x86_avx2_pmadd_ub_sw;
3090 else if (VecWidth == 512)
3091 IID = Intrinsic::x86_avx512_pmaddubs_w_512;
3092 else
3093 llvm_unreachable("Unexpected intrinsic");
3094 } else if (Name.starts_with("packsswb.")) {
3095 if (VecWidth == 128)
3096 IID = Intrinsic::x86_sse2_packsswb_128;
3097 else if (VecWidth == 256)
3098 IID = Intrinsic::x86_avx2_packsswb;
3099 else if (VecWidth == 512)
3100 IID = Intrinsic::x86_avx512_packsswb_512;
3101 else
3102 llvm_unreachable("Unexpected intrinsic");
3103 } else if (Name.starts_with("packssdw.")) {
3104 if (VecWidth == 128)
3105 IID = Intrinsic::x86_sse2_packssdw_128;
3106 else if (VecWidth == 256)
3107 IID = Intrinsic::x86_avx2_packssdw;
3108 else if (VecWidth == 512)
3109 IID = Intrinsic::x86_avx512_packssdw_512;
3110 else
3111 llvm_unreachable("Unexpected intrinsic");
3112 } else if (Name.starts_with("packuswb.")) {
3113 if (VecWidth == 128)
3114 IID = Intrinsic::x86_sse2_packuswb_128;
3115 else if (VecWidth == 256)
3116 IID = Intrinsic::x86_avx2_packuswb;
3117 else if (VecWidth == 512)
3118 IID = Intrinsic::x86_avx512_packuswb_512;
3119 else
3120 llvm_unreachable("Unexpected intrinsic");
3121 } else if (Name.starts_with("packusdw.")) {
3122 if (VecWidth == 128)
3123 IID = Intrinsic::x86_sse41_packusdw;
3124 else if (VecWidth == 256)
3125 IID = Intrinsic::x86_avx2_packusdw;
3126 else if (VecWidth == 512)
3127 IID = Intrinsic::x86_avx512_packusdw_512;
3128 else
3129 llvm_unreachable("Unexpected intrinsic");
3130 } else if (Name.starts_with("vpermilvar.")) {
3131 if (VecWidth == 128 && EltWidth == 32)
3132 IID = Intrinsic::x86_avx_vpermilvar_ps;
3133 else if (VecWidth == 128 && EltWidth == 64)
3134 IID = Intrinsic::x86_avx_vpermilvar_pd;
3135 else if (VecWidth == 256 && EltWidth == 32)
3136 IID = Intrinsic::x86_avx_vpermilvar_ps_256;
3137 else if (VecWidth == 256 && EltWidth == 64)
3138 IID = Intrinsic::x86_avx_vpermilvar_pd_256;
3139 else if (VecWidth == 512 && EltWidth == 32)
3140 IID = Intrinsic::x86_avx512_vpermilvar_ps_512;
3141 else if (VecWidth == 512 && EltWidth == 64)
3142 IID = Intrinsic::x86_avx512_vpermilvar_pd_512;
3143 else
3144 llvm_unreachable("Unexpected intrinsic");
3145 } else if (Name == "cvtpd2dq.256") {
3146 IID = Intrinsic::x86_avx_cvt_pd2dq_256;
3147 } else if (Name == "cvtpd2ps.256") {
3148 IID = Intrinsic::x86_avx_cvt_pd2_ps_256;
3149 } else if (Name == "cvttpd2dq.256") {
3150 IID = Intrinsic::x86_avx_cvtt_pd2dq_256;
3151 } else if (Name == "cvttps2dq.128") {
3152 IID = Intrinsic::x86_sse2_cvttps2dq;
3153 } else if (Name == "cvttps2dq.256") {
3154 IID = Intrinsic::x86_avx_cvtt_ps2dq_256;
3155 } else if (Name.starts_with("permvar.")) {
3156 bool IsFloat = CI.getType()->isFPOrFPVectorTy();
3157 if (VecWidth == 256 && EltWidth == 32 && IsFloat)
3158 IID = Intrinsic::x86_avx2_permps;
3159 else if (VecWidth == 256 && EltWidth == 32 && !IsFloat)
3160 IID = Intrinsic::x86_avx2_permd;
3161 else if (VecWidth == 256 && EltWidth == 64 && IsFloat)
3162 IID = Intrinsic::x86_avx512_permvar_df_256;
3163 else if (VecWidth == 256 && EltWidth == 64 && !IsFloat)
3164 IID = Intrinsic::x86_avx512_permvar_di_256;
3165 else if (VecWidth == 512 && EltWidth == 32 && IsFloat)
3166 IID = Intrinsic::x86_avx512_permvar_sf_512;
3167 else if (VecWidth == 512 && EltWidth == 32 && !IsFloat)
3168 IID = Intrinsic::x86_avx512_permvar_si_512;
3169 else if (VecWidth == 512 && EltWidth == 64 && IsFloat)
3170 IID = Intrinsic::x86_avx512_permvar_df_512;
3171 else if (VecWidth == 512 && EltWidth == 64 && !IsFloat)
3172 IID = Intrinsic::x86_avx512_permvar_di_512;
3173 else if (VecWidth == 128 && EltWidth == 16)
3174 IID = Intrinsic::x86_avx512_permvar_hi_128;
3175 else if (VecWidth == 256 && EltWidth == 16)
3176 IID = Intrinsic::x86_avx512_permvar_hi_256;
3177 else if (VecWidth == 512 && EltWidth == 16)
3178 IID = Intrinsic::x86_avx512_permvar_hi_512;
3179 else if (VecWidth == 128 && EltWidth == 8)
3180 IID = Intrinsic::x86_avx512_permvar_qi_128;
3181 else if (VecWidth == 256 && EltWidth == 8)
3182 IID = Intrinsic::x86_avx512_permvar_qi_256;
3183 else if (VecWidth == 512 && EltWidth == 8)
3184 IID = Intrinsic::x86_avx512_permvar_qi_512;
3185 else
3186 llvm_unreachable("Unexpected intrinsic");
3187 } else if (Name.starts_with("dbpsadbw.")) {
3188 if (VecWidth == 128)
3189 IID = Intrinsic::x86_avx512_dbpsadbw_128;
3190 else if (VecWidth == 256)
3191 IID = Intrinsic::x86_avx512_dbpsadbw_256;
3192 else if (VecWidth == 512)
3193 IID = Intrinsic::x86_avx512_dbpsadbw_512;
3194 else
3195 llvm_unreachable("Unexpected intrinsic");
3196 } else if (Name.starts_with("pmultishift.qb.")) {
3197 if (VecWidth == 128)
3198 IID = Intrinsic::x86_avx512_pmultishift_qb_128;
3199 else if (VecWidth == 256)
3200 IID = Intrinsic::x86_avx512_pmultishift_qb_256;
3201 else if (VecWidth == 512)
3202 IID = Intrinsic::x86_avx512_pmultishift_qb_512;
3203 else
3204 llvm_unreachable("Unexpected intrinsic");
3205 } else if (Name.starts_with("conflict.")) {
3206 if (Name[9] == 'd' && VecWidth == 128)
3207 IID = Intrinsic::x86_avx512_conflict_d_128;
3208 else if (Name[9] == 'd' && VecWidth == 256)
3209 IID = Intrinsic::x86_avx512_conflict_d_256;
3210 else if (Name[9] == 'd' && VecWidth == 512)
3211 IID = Intrinsic::x86_avx512_conflict_d_512;
3212 else if (Name[9] == 'q' && VecWidth == 128)
3213 IID = Intrinsic::x86_avx512_conflict_q_128;
3214 else if (Name[9] == 'q' && VecWidth == 256)
3215 IID = Intrinsic::x86_avx512_conflict_q_256;
3216 else if (Name[9] == 'q' && VecWidth == 512)
3217 IID = Intrinsic::x86_avx512_conflict_q_512;
3218 else
3219 llvm_unreachable("Unexpected intrinsic");
3220 } else if (Name.starts_with("pavg.")) {
3221 if (Name[5] == 'b' && VecWidth == 128)
3222 IID = Intrinsic::x86_sse2_pavg_b;
3223 else if (Name[5] == 'b' && VecWidth == 256)
3224 IID = Intrinsic::x86_avx2_pavg_b;
3225 else if (Name[5] == 'b' && VecWidth == 512)
3226 IID = Intrinsic::x86_avx512_pavg_b_512;
3227 else if (Name[5] == 'w' && VecWidth == 128)
3228 IID = Intrinsic::x86_sse2_pavg_w;
3229 else if (Name[5] == 'w' && VecWidth == 256)
3230 IID = Intrinsic::x86_avx2_pavg_w;
3231 else if (Name[5] == 'w' && VecWidth == 512)
3232 IID = Intrinsic::x86_avx512_pavg_w_512;
3233 else
3234 llvm_unreachable("Unexpected intrinsic");
3235 } else
3236 return false;
3237
3238 SmallVector<Value *, 4> Args(CI.args());
3239 Args.pop_back();
3240 Args.pop_back();
3241 Rep = Builder.CreateIntrinsic(IID, Args);
3242 unsigned NumArgs = CI.arg_size();
3243 Rep = emitX86Select(Builder, CI.getArgOperand(NumArgs - 1), Rep,
3244 CI.getArgOperand(NumArgs - 2));
3245 return true;
3246}
3247
3248/// Upgrade comment in call to inline asm that represents an objc retain release
3249/// marker.
3250void llvm::UpgradeInlineAsmString(std::string *AsmStr) {
3251 size_t Pos;
3252 if (AsmStr->find("mov\tfp") == 0 &&
3253 AsmStr->find("objc_retainAutoreleaseReturnValue") != std::string::npos &&
3254 (Pos = AsmStr->find("# marker")) != std::string::npos) {
3255 AsmStr->replace(Pos, 1, ";");
3256 }
3257}
3258
3260 Function *F, IRBuilder<> &Builder) {
3261 Value *Rep = nullptr;
3262
3263 if (Name == "abs.i" || Name == "abs.ll") {
3264 Value *Arg = CI->getArgOperand(0);
3265 Rep = Builder.CreateIntrinsic(Intrinsic::abs, {Arg->getType()},
3266 {Arg, Builder.getTrue()},
3267 /*FMFSource=*/nullptr, "abs");
3268 } else if (Name == "abs.bf16" || Name == "abs.bf16x2") {
3269 Type *Ty = (Name == "abs.bf16")
3270 ? Builder.getBFloatTy()
3271 : FixedVectorType::get(Builder.getBFloatTy(), 2);
3272 Value *Arg = Builder.CreateBitCast(CI->getArgOperand(0), Ty);
3273 Value *Abs = Builder.CreateUnaryIntrinsic(Intrinsic::nvvm_fabs, Arg);
3274 Rep = Builder.CreateBitCast(Abs, CI->getType());
3275 } else if (Name == "fabs.f" || Name == "fabs.ftz.f" || Name == "fabs.d") {
3276 Intrinsic::ID IID = (Name == "fabs.ftz.f") ? Intrinsic::nvvm_fabs_ftz
3277 : Intrinsic::nvvm_fabs;
3278 Rep = Builder.CreateUnaryIntrinsic(IID, CI->getArgOperand(0));
3279 } else if (Name.consume_front("add.")) {
3280 // nvvm.add.<rnd>{.ftz}{.sat}.{f,d,f16,v2f16}
3281 auto FAdd = getNVVMFAddUpgrade(Name);
3282 assert(FAdd && "unsupported nvvm.add.* intrinsic");
3283 auto [IID, RoundingMode] = *FAdd;
3284 Value *A = CI->getArgOperand(0);
3285 Rep = Builder.CreateIntrinsic(
3286 A->getType(), IID,
3287 {A, CI->getArgOperand(1),
3288 Builder.getInt32(static_cast<int>(RoundingMode))});
3289 } else if (Name.consume_front("ex2.approx.")) {
3290 // nvvm.ex2.approx.{f,ftz.f,d,f16x2}
3291 Intrinsic::ID IID = Name.starts_with("ftz") ? Intrinsic::nvvm_ex2_approx_ftz
3292 : Intrinsic::nvvm_ex2_approx;
3293 Rep = Builder.CreateUnaryIntrinsic(IID, CI->getArgOperand(0));
3294 } else if (Name.starts_with("atomic.load.add.f32.p") ||
3295 Name.starts_with("atomic.load.add.f64.p")) {
3296 Value *Ptr = CI->getArgOperand(0);
3297 Value *Val = CI->getArgOperand(1);
3298 Rep = Builder.CreateAtomicRMW(
3300 CI->getContext().getOrInsertSyncScopeID("device"));
3301 // The default scope for atomic.load.* intrinsics is device
3302 // (= gpu scope in ptx), but the default LLVM atomic scope is
3303 // "system"
3304 } else if (Name.starts_with("atomic.load.inc.32.p") ||
3305 Name.starts_with("atomic.load.dec.32.p")) {
3306 Value *Ptr = CI->getArgOperand(0);
3307 Value *Val = CI->getArgOperand(1);
3308 auto Op = Name.starts_with("atomic.load.inc") ? AtomicRMWInst::UIncWrap
3310 Rep = Builder.CreateAtomicRMW(
3312 CI->getContext().getOrInsertSyncScopeID("device"));
3313 // See comment above.
3314 } else if (Name.starts_with("atomic.") && Name.contains(".gen.")) {
3315 // nvvm.atomic.{op}.gen.{i,f}.{cta,sys} -> atomicrmw / cmpxchg.
3316 StringRef Op = Name.substr(StringRef("atomic.").size());
3317 Value *Ptr = CI->getArgOperand(0);
3318 Value *Val = CI->getArgOperand(1);
3320 Op.contains(".cta.") ? "block" : "");
3321 if (Op.starts_with("cas.")) {
3322 Value *New = CI->getArgOperand(2);
3323 Value *Pair = Builder.CreateAtomicCmpXchg(
3324 Ptr, Val, New, MaybeAlign(), AtomicOrdering::Monotonic,
3326 Rep = Builder.CreateExtractValue(Pair, 0);
3327 } else {
3328 // Note we don't upgrade anything to AtomicRMWInst::UMin/UMax. This is
3329 // because we were actually missing those intrinsics!
3330 AtomicRMWInst::BinOp BinOp =
3332 .StartsWith("add.gen.f", AtomicRMWInst::FAdd)
3333 .StartsWith("add.gen.i", AtomicRMWInst::Add)
3344 "unexpected nvvm scoped atomic intrinsic");
3345 Rep = Builder.CreateAtomicRMW(BinOp, Ptr, Val, MaybeAlign(),
3347 }
3348 } else if (Name == "clz.ll") {
3349 // llvm.nvvm.clz.ll returns an i32, but llvm.ctlz.i64 returns an i64.
3350 Value *Arg = CI->getArgOperand(0);
3351 Value *Ctlz = Builder.CreateIntrinsic(Intrinsic::ctlz, {Arg->getType()},
3352 {Arg, Builder.getFalse()},
3353 /*FMFSource=*/nullptr, "ctlz");
3354 Rep = Builder.CreateTrunc(Ctlz, Builder.getInt32Ty(), "ctlz.trunc");
3355 } else if (Name == "popc.ll") {
3356 // llvm.nvvm.popc.ll returns an i32, but llvm.ctpop.i64 returns an
3357 // i64.
3358 Value *Arg = CI->getArgOperand(0);
3359 Value *Popc = Builder.CreateIntrinsic(Intrinsic::ctpop, {Arg->getType()},
3360 Arg, /*FMFSource=*/nullptr, "ctpop");
3361 Rep = Builder.CreateTrunc(Popc, Builder.getInt32Ty(), "ctpop.trunc");
3362 } else if (Name == "h2f") {
3363 Value *Cast =
3364 Builder.CreateBitCast(CI->getArgOperand(0), Builder.getHalfTy());
3365 Rep = Builder.CreateFPExt(Cast, Builder.getFloatTy());
3366 } else if (Name.consume_front("bitcast.") &&
3367 (Name == "f2i" || Name == "i2f" || Name == "ll2d" ||
3368 Name == "d2ll")) {
3369 Rep = Builder.CreateBitCast(CI->getArgOperand(0), CI->getType());
3370 } else if (Name == "rotate.b32") {
3371 Value *Arg = CI->getOperand(0);
3372 Value *ShiftAmt = CI->getOperand(1);
3373 Rep = Builder.CreateIntrinsic(Builder.getInt32Ty(), Intrinsic::fshl,
3374 {Arg, Arg, ShiftAmt});
3375 } else if (Name == "rotate.b64") {
3376 Type *Int64Ty = Builder.getInt64Ty();
3377 Value *Arg = CI->getOperand(0);
3378 Value *ZExtShiftAmt = Builder.CreateZExt(CI->getOperand(1), Int64Ty);
3379 Rep = Builder.CreateIntrinsic(Int64Ty, Intrinsic::fshl,
3380 {Arg, Arg, ZExtShiftAmt});
3381 } else if (Name == "rotate.right.b64") {
3382 Type *Int64Ty = Builder.getInt64Ty();
3383 Value *Arg = CI->getOperand(0);
3384 Value *ZExtShiftAmt = Builder.CreateZExt(CI->getOperand(1), Int64Ty);
3385 Rep = Builder.CreateIntrinsic(Int64Ty, Intrinsic::fshr,
3386 {Arg, Arg, ZExtShiftAmt});
3387 } else if (Name == "swap.lo.hi.b64") {
3388 Type *Int64Ty = Builder.getInt64Ty();
3389 Value *Arg = CI->getOperand(0);
3390 Rep = Builder.CreateIntrinsic(Int64Ty, Intrinsic::fshl,
3391 {Arg, Arg, Builder.getInt64(32)});
3392 } else if ((Name.consume_front("ptr.gen.to.") &&
3393 consumeNVVMPtrAddrSpace(Name)) ||
3394 (Name.consume_front("ptr.") && consumeNVVMPtrAddrSpace(Name) &&
3395 Name.starts_with(".to.gen"))) {
3396 Rep = Builder.CreateAddrSpaceCast(CI->getArgOperand(0), CI->getType());
3397 } else if (Name.consume_front("ldg.global")) {
3398 Value *Ptr = CI->getArgOperand(0);
3399 Align PtrAlign = cast<ConstantInt>(CI->getArgOperand(1))->getAlignValue();
3400 // Use addrspace(1) for NVPTX ADDRESS_SPACE_GLOBAL
3401 Value *ASC = Builder.CreateAddrSpaceCast(Ptr, Builder.getPtrTy(1));
3402 Instruction *LD = Builder.CreateAlignedLoad(CI->getType(), ASC, PtrAlign);
3403 MDNode *MD = MDNode::get(Builder.getContext(), {});
3404 LD->setMetadata(LLVMContext::MD_invariant_load, MD);
3405 return LD;
3406 } else if (Name == "tanh.approx.f32") {
3407 // nvvm.tanh.approx.f32 -> afn llvm.tanh.f32
3408 FastMathFlags FMF;
3409 FMF.setApproxFunc();
3410 Rep = Builder.CreateUnaryIntrinsic(Intrinsic::tanh, CI->getArgOperand(0),
3411 FMF);
3412 } else if (Name == "barrier0" || Name == "barrier.n" || Name == "bar.sync") {
3413 Value *Arg =
3414 Name.ends_with('0') ? Builder.getInt32(0) : CI->getArgOperand(0);
3415 Rep = Builder.CreateIntrinsic(Intrinsic::nvvm_barrier_cta_sync_aligned_all,
3416 {}, {Arg});
3417 } else if (Name == "barrier") {
3418 Rep = Builder.CreateIntrinsic(
3419 Intrinsic::nvvm_barrier_cta_sync_aligned_count, {},
3420 {CI->getArgOperand(0), CI->getArgOperand(1)});
3421 } else if (Name == "barrier.sync") {
3422 Rep = Builder.CreateIntrinsic(Intrinsic::nvvm_barrier_cta_sync_all, {},
3423 {CI->getArgOperand(0)});
3424 } else if (Name == "barrier.sync.cnt") {
3425 Rep = Builder.CreateIntrinsic(Intrinsic::nvvm_barrier_cta_sync_count, {},
3426 {CI->getArgOperand(0), CI->getArgOperand(1)});
3427 } else if (Name == "barrier0.popc" || Name == "barrier0.and" ||
3428 Name == "barrier0.or") {
3429 Value *C = CI->getArgOperand(0);
3430 C = Builder.CreateICmpNE(C, Builder.getInt32(0));
3431
3432 Intrinsic::ID IID =
3434 .Case("barrier0.popc",
3435 Intrinsic::nvvm_barrier_cta_red_popc_aligned_all)
3436 .Case("barrier0.and",
3437 Intrinsic::nvvm_barrier_cta_red_and_aligned_all)
3438 .Case("barrier0.or",
3439 Intrinsic::nvvm_barrier_cta_red_or_aligned_all);
3440 Value *Bar = Builder.CreateIntrinsic(IID, {}, {Builder.getInt32(0), C});
3441 Rep = Builder.CreateZExt(Bar, CI->getType());
3442 } else {
3444 if (IID != Intrinsic::not_intrinsic &&
3446 rename(F);
3447 Function *NewFn = Intrinsic::getOrInsertDeclaration(F->getParent(), IID);
3449 for (size_t I = 0; I < NewFn->arg_size(); ++I) {
3450 Value *Arg = CI->getArgOperand(I);
3451 Type *OldType = Arg->getType();
3452 Type *NewType = NewFn->getArg(I)->getType();
3453 Args.push_back(
3454 (OldType->isIntegerTy() && NewType->getScalarType()->isBFloatTy())
3455 ? Builder.CreateBitCast(Arg, NewType)
3456 : Arg);
3457 }
3458 Rep = Builder.CreateCall(NewFn, Args);
3459 if (F->getReturnType()->isIntegerTy())
3460 Rep = Builder.CreateBitCast(Rep, F->getReturnType());
3461 }
3462 }
3463
3464 return Rep;
3465}
3466
3468 IRBuilder<> &Builder) {
3469 LLVMContext &C = F->getContext();
3470 Value *Rep = nullptr;
3471
3472 if (Name.starts_with("sse4a.movnt.")) {
3474 Elts.push_back(
3475 ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1)));
3476 MDNode *Node = MDNode::get(C, Elts);
3477
3478 Value *Arg0 = CI->getArgOperand(0);
3479 Value *Arg1 = CI->getArgOperand(1);
3480
3481 // Nontemporal (unaligned) store of the 0'th element of the float/double
3482 // vector.
3483 Value *Extract =
3484 Builder.CreateExtractElement(Arg1, (uint64_t)0, "extractelement");
3485
3486 StoreInst *SI = Builder.CreateAlignedStore(Extract, Arg0, Align(1));
3487 SI->setMetadata(LLVMContext::MD_nontemporal, Node);
3488 } else if (Name.starts_with("avx.movnt.") ||
3489 Name.starts_with("avx512.storent.")) {
3491 Elts.push_back(
3492 ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1)));
3493 MDNode *Node = MDNode::get(C, Elts);
3494
3495 Value *Arg0 = CI->getArgOperand(0);
3496 Value *Arg1 = CI->getArgOperand(1);
3497
3498 StoreInst *SI = Builder.CreateAlignedStore(
3499 Arg1, Arg0,
3501 SI->setMetadata(LLVMContext::MD_nontemporal, Node);
3502 } else if (Name == "sse2.storel.dq") {
3503 Value *Arg0 = CI->getArgOperand(0);
3504 Value *Arg1 = CI->getArgOperand(1);
3505
3506 auto *NewVecTy = FixedVectorType::get(Type::getInt64Ty(C), 2);
3507 Value *BC0 = Builder.CreateBitCast(Arg1, NewVecTy, "cast");
3508 Value *Elt = Builder.CreateExtractElement(BC0, (uint64_t)0);
3509 Builder.CreateAlignedStore(Elt, Arg0, Align(1));
3510 } else if (Name.starts_with("sse.storeu.") ||
3511 Name.starts_with("sse2.storeu.") ||
3512 Name.starts_with("avx.storeu.")) {
3513 Value *Arg0 = CI->getArgOperand(0);
3514 Value *Arg1 = CI->getArgOperand(1);
3515 Builder.CreateAlignedStore(Arg1, Arg0, Align(1));
3516 } else if (Name == "avx512.mask.store.ss") {
3517 Value *Mask = Builder.CreateAnd(CI->getArgOperand(2), Builder.getInt8(1));
3518 upgradeMaskedStore(Builder, CI->getArgOperand(0), CI->getArgOperand(1),
3519 Mask, false);
3520 } else if (Name.starts_with("avx512.mask.store")) {
3521 // "avx512.mask.storeu." or "avx512.mask.store."
3522 bool Aligned = Name[17] != 'u'; // "avx512.mask.storeu".
3523 upgradeMaskedStore(Builder, CI->getArgOperand(0), CI->getArgOperand(1),
3524 CI->getArgOperand(2), Aligned);
3525 } else if (Name.starts_with("sse2.pcmp") || Name.starts_with("avx2.pcmp")) {
3526 // Upgrade packed integer vector compare intrinsics to compare instructions.
3527 // "sse2.pcpmpeq." "sse2.pcmpgt." "avx2.pcmpeq." or "avx2.pcmpgt."
3528 bool CmpEq = Name[9] == 'e';
3529 Rep = Builder.CreateICmp(CmpEq ? ICmpInst::ICMP_EQ : ICmpInst::ICMP_SGT,
3530 CI->getArgOperand(0), CI->getArgOperand(1));
3531 Rep = Builder.CreateSExt(Rep, CI->getType(), "");
3532 } else if (Name.starts_with("avx512.broadcastm")) {
3533 Type *ExtTy = Type::getInt32Ty(C);
3534 if (CI->getOperand(0)->getType()->isIntegerTy(8))
3535 ExtTy = Type::getInt64Ty(C);
3536 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() /
3537 ExtTy->getPrimitiveSizeInBits();
3538 Rep = Builder.CreateZExt(CI->getArgOperand(0), ExtTy);
3539 Rep = Builder.CreateVectorSplat(NumElts, Rep);
3540 } else if (Name == "sse.sqrt.ss" || Name == "sse2.sqrt.sd") {
3541 Value *Vec = CI->getArgOperand(0);
3542 Value *Elt0 = Builder.CreateExtractElement(Vec, (uint64_t)0);
3543 Elt0 = Builder.CreateIntrinsic(Intrinsic::sqrt, Elt0->getType(), Elt0);
3544 Rep = Builder.CreateInsertElement(Vec, Elt0, (uint64_t)0);
3545 } else if (Name.starts_with("avx.sqrt.p") ||
3546 Name.starts_with("sse2.sqrt.p") ||
3547 Name.starts_with("sse.sqrt.p")) {
3548 Rep = Builder.CreateIntrinsic(Intrinsic::sqrt, CI->getType(),
3549 {CI->getArgOperand(0)});
3550 } else if (Name.starts_with("avx512.mask.sqrt.p")) {
3551 if (CI->arg_size() == 4 &&
3552 (!isa<ConstantInt>(CI->getArgOperand(3)) ||
3553 cast<ConstantInt>(CI->getArgOperand(3))->getZExtValue() != 4)) {
3554 Intrinsic::ID IID = Name[18] == 's' ? Intrinsic::x86_avx512_sqrt_ps_512
3555 : Intrinsic::x86_avx512_sqrt_pd_512;
3556
3557 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(3)};
3558 Rep = Builder.CreateIntrinsic(IID, Args);
3559 } else {
3560 Rep = Builder.CreateIntrinsic(Intrinsic::sqrt, CI->getType(),
3561 {CI->getArgOperand(0)});
3562 }
3563 Rep =
3564 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
3565 } else if (Name.starts_with("avx512.ptestm") ||
3566 Name.starts_with("avx512.ptestnm")) {
3567 Value *Op0 = CI->getArgOperand(0);
3568 Value *Op1 = CI->getArgOperand(1);
3569 Value *Mask = CI->getArgOperand(2);
3570 Rep = Builder.CreateAnd(Op0, Op1);
3571 llvm::Type *Ty = Op0->getType();
3573 ICmpInst::Predicate Pred = Name.starts_with("avx512.ptestm")
3576 Rep = Builder.CreateICmp(Pred, Rep, Zero);
3577 Rep = applyX86MaskOn1BitsVec(Builder, Rep, Mask);
3578 } else if (Name.starts_with("avx512.mask.pbroadcast")) {
3579 unsigned NumElts = cast<FixedVectorType>(CI->getArgOperand(1)->getType())
3580 ->getNumElements();
3581 Rep = Builder.CreateVectorSplat(NumElts, CI->getArgOperand(0));
3582 Rep =
3583 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
3584 } else if (Name.starts_with("avx512.kunpck")) {
3585 unsigned NumElts = CI->getType()->getScalarSizeInBits();
3586 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), NumElts);
3587 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), NumElts);
3588 int Indices[64];
3589 for (unsigned i = 0; i != NumElts; ++i)
3590 Indices[i] = i;
3591
3592 // First extract half of each vector. This gives better codegen than
3593 // doing it in a single shuffle.
3594 LHS = Builder.CreateShuffleVector(LHS, LHS, ArrayRef(Indices, NumElts / 2));
3595 RHS = Builder.CreateShuffleVector(RHS, RHS, ArrayRef(Indices, NumElts / 2));
3596 // Concat the vectors.
3597 // NOTE: Operands have to be swapped to match intrinsic definition.
3598 Rep = Builder.CreateShuffleVector(RHS, LHS, ArrayRef(Indices, NumElts));
3599 Rep = Builder.CreateBitCast(Rep, CI->getType());
3600 } else if (Name == "avx512.kand.w") {
3601 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3602 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3603 Rep = Builder.CreateAnd(LHS, RHS);
3604 Rep = Builder.CreateBitCast(Rep, CI->getType());
3605 } else if (Name == "avx512.kandn.w") {
3606 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3607 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3608 LHS = Builder.CreateNot(LHS);
3609 Rep = Builder.CreateAnd(LHS, RHS);
3610 Rep = Builder.CreateBitCast(Rep, CI->getType());
3611 } else if (Name == "avx512.kor.w") {
3612 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3613 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3614 Rep = Builder.CreateOr(LHS, RHS);
3615 Rep = Builder.CreateBitCast(Rep, CI->getType());
3616 } else if (Name == "avx512.kxor.w") {
3617 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3618 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3619 Rep = Builder.CreateXor(LHS, RHS);
3620 Rep = Builder.CreateBitCast(Rep, CI->getType());
3621 } else if (Name == "avx512.kxnor.w") {
3622 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3623 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3624 LHS = Builder.CreateNot(LHS);
3625 Rep = Builder.CreateXor(LHS, RHS);
3626 Rep = Builder.CreateBitCast(Rep, CI->getType());
3627 } else if (Name == "avx512.knot.w") {
3628 Rep = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3629 Rep = Builder.CreateNot(Rep);
3630 Rep = Builder.CreateBitCast(Rep, CI->getType());
3631 } else if (Name == "avx512.kortestz.w" || Name == "avx512.kortestc.w") {
3632 Value *LHS = getX86MaskVec(Builder, CI->getArgOperand(0), 16);
3633 Value *RHS = getX86MaskVec(Builder, CI->getArgOperand(1), 16);
3634 Rep = Builder.CreateOr(LHS, RHS);
3635 Rep = Builder.CreateBitCast(Rep, Builder.getInt16Ty());
3636 Value *C;
3637 if (Name[14] == 'c')
3638 C = ConstantInt::getAllOnesValue(Builder.getInt16Ty());
3639 else
3640 C = ConstantInt::getNullValue(Builder.getInt16Ty());
3641 Rep = Builder.CreateICmpEQ(Rep, C);
3642 Rep = Builder.CreateZExt(Rep, Builder.getInt32Ty());
3643 } else if (Name == "sse.add.ss" || Name == "sse2.add.sd" ||
3644 Name == "sse.sub.ss" || Name == "sse2.sub.sd" ||
3645 Name == "sse.mul.ss" || Name == "sse2.mul.sd" ||
3646 Name == "sse.div.ss" || Name == "sse2.div.sd") {
3647 Type *I32Ty = Type::getInt32Ty(C);
3648 Value *Elt0 = Builder.CreateExtractElement(CI->getArgOperand(0),
3649 ConstantInt::get(I32Ty, 0));
3650 Value *Elt1 = Builder.CreateExtractElement(CI->getArgOperand(1),
3651 ConstantInt::get(I32Ty, 0));
3652 Value *EltOp;
3653 if (Name.contains(".add."))
3654 EltOp = Builder.CreateFAdd(Elt0, Elt1);
3655 else if (Name.contains(".sub."))
3656 EltOp = Builder.CreateFSub(Elt0, Elt1);
3657 else if (Name.contains(".mul."))
3658 EltOp = Builder.CreateFMul(Elt0, Elt1);
3659 else
3660 EltOp = Builder.CreateFDiv(Elt0, Elt1);
3661 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), EltOp,
3662 ConstantInt::get(I32Ty, 0));
3663 } else if (Name.starts_with("avx512.mask.pcmp")) {
3664 // "avx512.mask.pcmpeq." or "avx512.mask.pcmpgt."
3665 bool CmpEq = Name[16] == 'e';
3666 Rep = upgradeMaskedCompare(Builder, *CI, CmpEq ? 0 : 6, true);
3667 } else if (Name.starts_with("avx512.mask.vpshufbitqmb.")) {
3668 Type *OpTy = CI->getArgOperand(0)->getType();
3669 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3670 Intrinsic::ID IID;
3671 switch (VecWidth) {
3672 default:
3673 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
3674 break;
3675 case 128:
3676 IID = Intrinsic::x86_avx512_vpshufbitqmb_128;
3677 break;
3678 case 256:
3679 IID = Intrinsic::x86_avx512_vpshufbitqmb_256;
3680 break;
3681 case 512:
3682 IID = Intrinsic::x86_avx512_vpshufbitqmb_512;
3683 break;
3684 }
3685
3686 Rep =
3687 Builder.CreateIntrinsic(IID, {CI->getOperand(0), CI->getArgOperand(1)});
3688 Rep = applyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(2));
3689 } else if (Name.starts_with("avx512.mask.fpclass.p")) {
3690 Type *OpTy = CI->getArgOperand(0)->getType();
3691 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3692 unsigned EltWidth = OpTy->getScalarSizeInBits();
3693 Intrinsic::ID IID;
3694 if (VecWidth == 128 && EltWidth == 32)
3695 IID = Intrinsic::x86_avx512_fpclass_ps_128;
3696 else if (VecWidth == 256 && EltWidth == 32)
3697 IID = Intrinsic::x86_avx512_fpclass_ps_256;
3698 else if (VecWidth == 512 && EltWidth == 32)
3699 IID = Intrinsic::x86_avx512_fpclass_ps_512;
3700 else if (VecWidth == 128 && EltWidth == 64)
3701 IID = Intrinsic::x86_avx512_fpclass_pd_128;
3702 else if (VecWidth == 256 && EltWidth == 64)
3703 IID = Intrinsic::x86_avx512_fpclass_pd_256;
3704 else if (VecWidth == 512 && EltWidth == 64)
3705 IID = Intrinsic::x86_avx512_fpclass_pd_512;
3706 else
3707 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
3708
3709 Rep =
3710 Builder.CreateIntrinsic(IID, {CI->getOperand(0), CI->getArgOperand(1)});
3711 Rep = applyX86MaskOn1BitsVec(Builder, Rep, CI->getArgOperand(2));
3712 } else if (Name.starts_with("avx512.cmp.p")) {
3713 SmallVector<Value *, 4> Args(CI->args());
3714 Type *OpTy = Args[0]->getType();
3715 unsigned VecWidth = OpTy->getPrimitiveSizeInBits();
3716 unsigned EltWidth = OpTy->getScalarSizeInBits();
3717 Intrinsic::ID IID;
3718 if (VecWidth == 128 && EltWidth == 32)
3719 IID = Intrinsic::x86_avx512_mask_cmp_ps_128;
3720 else if (VecWidth == 256 && EltWidth == 32)
3721 IID = Intrinsic::x86_avx512_mask_cmp_ps_256;
3722 else if (VecWidth == 512 && EltWidth == 32)
3723 IID = Intrinsic::x86_avx512_mask_cmp_ps_512;
3724 else if (VecWidth == 128 && EltWidth == 64)
3725 IID = Intrinsic::x86_avx512_mask_cmp_pd_128;
3726 else if (VecWidth == 256 && EltWidth == 64)
3727 IID = Intrinsic::x86_avx512_mask_cmp_pd_256;
3728 else if (VecWidth == 512 && EltWidth == 64)
3729 IID = Intrinsic::x86_avx512_mask_cmp_pd_512;
3730 else
3731 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
3732
3734 if (VecWidth == 512)
3735 std::swap(Mask, Args.back());
3736 Args.push_back(Mask);
3737
3738 Rep = Builder.CreateIntrinsic(IID, Args);
3739 } else if (Name.starts_with("avx512.mask.cmp.")) {
3740 // Integer compare intrinsics.
3741 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
3742 Rep = upgradeMaskedCompare(Builder, *CI, Imm, true);
3743 } else if (Name.starts_with("avx512.mask.ucmp.")) {
3744 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
3745 Rep = upgradeMaskedCompare(Builder, *CI, Imm, false);
3746 } else if (Name.starts_with("avx512.cvtb2mask.") ||
3747 Name.starts_with("avx512.cvtw2mask.") ||
3748 Name.starts_with("avx512.cvtd2mask.") ||
3749 Name.starts_with("avx512.cvtq2mask.")) {
3750 Value *Op = CI->getArgOperand(0);
3751 Value *Zero = llvm::Constant::getNullValue(Op->getType());
3752 Rep = Builder.CreateICmp(ICmpInst::ICMP_SLT, Op, Zero);
3753 Rep = applyX86MaskOn1BitsVec(Builder, Rep, nullptr);
3754 } else if (Name == "ssse3.pabs.b.128" || Name == "ssse3.pabs.w.128" ||
3755 Name == "ssse3.pabs.d.128" || Name.starts_with("avx2.pabs") ||
3756 Name.starts_with("avx512.mask.pabs")) {
3757 Rep = upgradeAbs(Builder, *CI);
3758 } else if (Name == "sse41.pmaxsb" || Name == "sse2.pmaxs.w" ||
3759 Name == "sse41.pmaxsd" || Name.starts_with("avx2.pmaxs") ||
3760 Name.starts_with("avx512.mask.pmaxs")) {
3761 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::smax);
3762 } else if (Name == "sse2.pmaxu.b" || Name == "sse41.pmaxuw" ||
3763 Name == "sse41.pmaxud" || Name.starts_with("avx2.pmaxu") ||
3764 Name.starts_with("avx512.mask.pmaxu")) {
3765 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::umax);
3766 } else if (Name == "sse41.pminsb" || Name == "sse2.pmins.w" ||
3767 Name == "sse41.pminsd" || Name.starts_with("avx2.pmins") ||
3768 Name.starts_with("avx512.mask.pmins")) {
3769 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::smin);
3770 } else if (Name == "sse2.pminu.b" || Name == "sse41.pminuw" ||
3771 Name == "sse41.pminud" || Name.starts_with("avx2.pminu") ||
3772 Name.starts_with("avx512.mask.pminu")) {
3773 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::umin);
3774 } else if (Name == "sse2.pmulh.w" || Name.starts_with("avx2.pmulh.w") ||
3775 Name.starts_with("avx512.pmulh.w")) {
3776 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::smulh);
3777 } else if (Name == "sse2.pmulhu.w" || Name.starts_with("avx2.pmulhu.w") ||
3778 Name.starts_with("avx512.pmulhu.w")) {
3779 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::umulh);
3780 } else if (Name == "sse2.pmulu.dq" || Name == "avx2.pmulu.dq" ||
3781 Name == "avx512.pmulu.dq.512" ||
3782 Name.starts_with("avx512.mask.pmulu.dq.")) {
3783 Rep = upgradePMULDQ(Builder, *CI, /*Signed*/ false);
3784 } else if (Name == "sse41.pmuldq" || Name == "avx2.pmul.dq" ||
3785 Name == "avx512.pmul.dq.512" ||
3786 Name.starts_with("avx512.mask.pmul.dq.")) {
3787 Rep = upgradePMULDQ(Builder, *CI, /*Signed*/ true);
3788 } else if (Name == "sse.cvtsi2ss" || Name == "sse2.cvtsi2sd" ||
3789 Name == "sse.cvtsi642ss" || Name == "sse2.cvtsi642sd") {
3790 Rep =
3791 Builder.CreateSIToFP(CI->getArgOperand(1),
3792 cast<VectorType>(CI->getType())->getElementType());
3793 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
3794 } else if (Name == "avx512.cvtusi2sd") {
3795 Rep =
3796 Builder.CreateUIToFP(CI->getArgOperand(1),
3797 cast<VectorType>(CI->getType())->getElementType());
3798 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
3799 } else if (Name == "sse2.cvtss2sd") {
3800 Rep = Builder.CreateExtractElement(CI->getArgOperand(1), (uint64_t)0);
3801 Rep = Builder.CreateFPExt(
3802 Rep, cast<VectorType>(CI->getType())->getElementType());
3803 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
3804 } else if (Name == "sse2.cvtdq2pd" || Name == "sse2.cvtdq2ps" ||
3805 Name == "avx.cvtdq2.pd.256" || Name == "avx.cvtdq2.ps.256" ||
3806 Name.starts_with("avx512.mask.cvtdq2pd.") ||
3807 Name.starts_with("avx512.mask.cvtudq2pd.") ||
3808 Name.starts_with("avx512.mask.cvtdq2ps.") ||
3809 Name.starts_with("avx512.mask.cvtudq2ps.") ||
3810 Name.starts_with("avx512.mask.cvtqq2pd.") ||
3811 Name.starts_with("avx512.mask.cvtuqq2pd.") ||
3812 Name == "avx512.mask.cvtqq2ps.256" ||
3813 Name == "avx512.mask.cvtqq2ps.512" ||
3814 Name == "avx512.mask.cvtuqq2ps.256" ||
3815 Name == "avx512.mask.cvtuqq2ps.512" || Name == "sse2.cvtps2pd" ||
3816 Name == "avx.cvt.ps2.pd.256" ||
3817 Name == "avx512.mask.cvtps2pd.128" ||
3818 Name == "avx512.mask.cvtps2pd.256") {
3819 auto *DstTy = cast<FixedVectorType>(CI->getType());
3820 Rep = CI->getArgOperand(0);
3821 auto *SrcTy = cast<FixedVectorType>(Rep->getType());
3822
3823 unsigned NumDstElts = DstTy->getNumElements();
3824 if (NumDstElts < SrcTy->getNumElements()) {
3825 assert(NumDstElts == 2 && "Unexpected vector size");
3826 Rep = Builder.CreateShuffleVector(Rep, Rep, ArrayRef<int>{0, 1});
3827 }
3828
3829 bool IsPS2PD = SrcTy->getElementType()->isFloatTy();
3830 bool IsUnsigned = Name.contains("cvtu");
3831 if (IsPS2PD)
3832 Rep = Builder.CreateFPExt(Rep, DstTy, "cvtps2pd");
3833 else if (CI->arg_size() == 4 &&
3834 (!isa<ConstantInt>(CI->getArgOperand(3)) ||
3835 cast<ConstantInt>(CI->getArgOperand(3))->getZExtValue() != 4)) {
3836 Intrinsic::ID IID = IsUnsigned ? Intrinsic::x86_avx512_uitofp_round
3837 : Intrinsic::x86_avx512_sitofp_round;
3838 Rep = Builder.CreateIntrinsic(IID, {DstTy, SrcTy},
3839 {Rep, CI->getArgOperand(3)});
3840 } else {
3841 Rep = IsUnsigned ? Builder.CreateUIToFP(Rep, DstTy, "cvt")
3842 : Builder.CreateSIToFP(Rep, DstTy, "cvt");
3843 }
3844
3845 if (CI->arg_size() >= 3)
3846 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
3847 CI->getArgOperand(1));
3848 } else if (Name.starts_with("avx512.mask.vcvtph2ps.") ||
3849 Name.starts_with("vcvtph2ps.")) {
3850 auto *DstTy = cast<FixedVectorType>(CI->getType());
3851 Rep = CI->getArgOperand(0);
3852 auto *SrcTy = cast<FixedVectorType>(Rep->getType());
3853 unsigned NumDstElts = DstTy->getNumElements();
3854 if (NumDstElts != SrcTy->getNumElements()) {
3855 assert(NumDstElts == 4 && "Unexpected vector size");
3856 Rep = Builder.CreateShuffleVector(Rep, Rep, ArrayRef<int>{0, 1, 2, 3});
3857 }
3858 Rep = Builder.CreateBitCast(
3859 Rep, FixedVectorType::get(Type::getHalfTy(C), NumDstElts));
3860 Rep = Builder.CreateFPExt(Rep, DstTy, "cvtph2ps");
3861 if (CI->arg_size() >= 3)
3862 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
3863 CI->getArgOperand(1));
3864 } else if (Name.starts_with("avx512.mask.load")) {
3865 // "avx512.mask.loadu." or "avx512.mask.load."
3866 bool Aligned = Name[16] != 'u'; // "avx512.mask.loadu".
3867 Rep = upgradeMaskedLoad(Builder, CI->getArgOperand(0), CI->getArgOperand(1),
3868 CI->getArgOperand(2), Aligned);
3869 } else if (Name.starts_with("avx512.mask.expand.load.")) {
3870 auto *ResultTy = cast<FixedVectorType>(CI->getType());
3871 auto *PtrTy = CI->getOperand(0)->getType();
3872 Value *MaskVec = getX86MaskVec(Builder, CI->getArgOperand(2),
3873 ResultTy->getNumElements());
3874 Rep = Builder.CreateIntrinsic(
3875 Intrinsic::masked_expandload, {ResultTy, PtrTy},
3876 {CI->getOperand(0), MaskVec, CI->getOperand(1)});
3877 } else if (Name.starts_with("avx512.mask.compress.store.")) {
3878 auto *ResultTy = cast<VectorType>(CI->getArgOperand(1)->getType());
3879 auto *PtrTy = CI->getArgOperand(0)->getType();
3880 Value *MaskVec =
3881 getX86MaskVec(Builder, CI->getArgOperand(2),
3882 cast<FixedVectorType>(ResultTy)->getNumElements());
3883 Rep = Builder.CreateIntrinsic(
3884 Intrinsic::masked_compressstore, {ResultTy, PtrTy},
3885 {CI->getArgOperand(1), CI->getArgOperand(0), MaskVec});
3886 } else if (Name.starts_with("avx512.mask.compress.") ||
3887 Name.starts_with("avx512.mask.expand.")) {
3888 auto *ResultTy = cast<FixedVectorType>(CI->getType());
3889
3890 Value *MaskVec = getX86MaskVec(Builder, CI->getArgOperand(2),
3891 ResultTy->getNumElements());
3892
3893 bool IsCompress = Name[12] == 'c';
3894 Intrinsic::ID IID = IsCompress ? Intrinsic::x86_avx512_mask_compress
3895 : Intrinsic::x86_avx512_mask_expand;
3896 Rep = Builder.CreateIntrinsic(
3897 IID, ResultTy, {CI->getOperand(0), CI->getOperand(1), MaskVec});
3898 } else if (Name.starts_with("xop.vpcom")) {
3899 bool IsSigned;
3900 if (Name.ends_with("ub") || Name.ends_with("uw") || Name.ends_with("ud") ||
3901 Name.ends_with("uq"))
3902 IsSigned = false;
3903 else if (Name.ends_with("b") || Name.ends_with("w") ||
3904 Name.ends_with("d") || Name.ends_with("q"))
3905 IsSigned = true;
3906 else
3907 reportFatalUsageErrorWithCI("Intrinsic has unknown suffix", CI);
3908
3909 unsigned Imm;
3910 if (CI->arg_size() == 3) {
3911 Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
3912 } else {
3913 Name = Name.substr(9); // strip off "xop.vpcom"
3914 if (Name.starts_with("lt"))
3915 Imm = 0;
3916 else if (Name.starts_with("le"))
3917 Imm = 1;
3918 else if (Name.starts_with("gt"))
3919 Imm = 2;
3920 else if (Name.starts_with("ge"))
3921 Imm = 3;
3922 else if (Name.starts_with("eq"))
3923 Imm = 4;
3924 else if (Name.starts_with("ne"))
3925 Imm = 5;
3926 else if (Name.starts_with("false"))
3927 Imm = 6;
3928 else if (Name.starts_with("true"))
3929 Imm = 7;
3930 else
3931 llvm_unreachable("Unknown condition");
3932 }
3933
3934 Rep = upgradeX86vpcom(Builder, *CI, Imm, IsSigned);
3935 } else if (Name.starts_with("xop.vpcmov")) {
3936 Value *Sel = CI->getArgOperand(2);
3937 Value *NotSel = Builder.CreateNot(Sel);
3938 Value *Sel0 = Builder.CreateAnd(CI->getArgOperand(0), Sel);
3939 Value *Sel1 = Builder.CreateAnd(CI->getArgOperand(1), NotSel);
3940 Rep = Builder.CreateOr(Sel0, Sel1);
3941 } else if (Name.starts_with("xop.vprot") || Name.starts_with("avx512.prol") ||
3942 Name.starts_with("avx512.mask.prol")) {
3943 Rep = upgradeX86Rotate(Builder, *CI, false);
3944 } else if (Name.starts_with("avx512.pror") ||
3945 Name.starts_with("avx512.mask.pror")) {
3946 Rep = upgradeX86Rotate(Builder, *CI, true);
3947 } else if (Name.starts_with("avx512.vpshld.") ||
3948 Name.starts_with("avx512.mask.vpshld") ||
3949 Name.starts_with("avx512.maskz.vpshld")) {
3950 bool ZeroMask = Name[11] == 'z';
3951 Rep = upgradeX86ConcatShift(Builder, *CI, false, ZeroMask);
3952 } else if (Name.starts_with("avx512.vpshrd.") ||
3953 Name.starts_with("avx512.mask.vpshrd") ||
3954 Name.starts_with("avx512.maskz.vpshrd")) {
3955 bool ZeroMask = Name[11] == 'z';
3956 Rep = upgradeX86ConcatShift(Builder, *CI, true, ZeroMask);
3957 } else if (Name == "sse42.crc32.64.8") {
3958 Value *Trunc0 =
3959 Builder.CreateTrunc(CI->getArgOperand(0), Type::getInt32Ty(C));
3960 Rep = Builder.CreateIntrinsic(Intrinsic::x86_sse42_crc32_32_8,
3961 {Trunc0, CI->getArgOperand(1)});
3962 Rep = Builder.CreateZExt(Rep, CI->getType(), "");
3963 } else if (Name.starts_with("avx.vbroadcast.s") ||
3964 Name.starts_with("avx512.vbroadcast.s")) {
3965 // Replace broadcasts with a series of insertelements.
3966 auto *VecTy = cast<FixedVectorType>(CI->getType());
3967 Type *EltTy = VecTy->getElementType();
3968 unsigned EltNum = VecTy->getNumElements();
3969 Value *Load = Builder.CreateLoad(EltTy, CI->getArgOperand(0));
3970 Type *I32Ty = Type::getInt32Ty(C);
3971 Rep = PoisonValue::get(VecTy);
3972 for (unsigned I = 0; I < EltNum; ++I)
3973 Rep = Builder.CreateInsertElement(Rep, Load, ConstantInt::get(I32Ty, I));
3974 } else if (Name.starts_with("sse41.pmovsx") ||
3975 Name.starts_with("sse41.pmovzx") ||
3976 Name.starts_with("avx2.pmovsx") ||
3977 Name.starts_with("avx2.pmovzx") ||
3978 Name.starts_with("avx512.mask.pmovsx") ||
3979 Name.starts_with("avx512.mask.pmovzx")) {
3980 auto *DstTy = cast<FixedVectorType>(CI->getType());
3981 unsigned NumDstElts = DstTy->getNumElements();
3982
3983 // Extract a subvector of the first NumDstElts lanes and sign/zero extend.
3984 SmallVector<int, 8> ShuffleMask(NumDstElts);
3985 for (unsigned i = 0; i != NumDstElts; ++i)
3986 ShuffleMask[i] = i;
3987
3988 Value *SV = Builder.CreateShuffleVector(CI->getArgOperand(0), ShuffleMask);
3989
3990 bool DoSext = Name.contains("pmovsx");
3991 Rep =
3992 DoSext ? Builder.CreateSExt(SV, DstTy) : Builder.CreateZExt(SV, DstTy);
3993 // If there are 3 arguments, it's a masked intrinsic so we need a select.
3994 if (CI->arg_size() == 3)
3995 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
3996 CI->getArgOperand(1));
3997 } else if (Name == "avx512.mask.pmov.qd.256" ||
3998 Name == "avx512.mask.pmov.qd.512" ||
3999 Name == "avx512.mask.pmov.wb.256" ||
4000 Name == "avx512.mask.pmov.wb.512") {
4001 Type *Ty = CI->getArgOperand(1)->getType();
4002 Rep = Builder.CreateTrunc(CI->getArgOperand(0), Ty);
4003 Rep =
4004 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4005 } else if (Name.starts_with("avx.vbroadcastf128") ||
4006 Name == "avx2.vbroadcasti128") {
4007 // Replace vbroadcastf128/vbroadcasti128 with a vector load+shuffle.
4008 Type *EltTy = cast<VectorType>(CI->getType())->getElementType();
4009 unsigned NumSrcElts = 128 / EltTy->getPrimitiveSizeInBits();
4010 auto *VT = FixedVectorType::get(EltTy, NumSrcElts);
4011 Value *Load = Builder.CreateAlignedLoad(VT, CI->getArgOperand(0), Align(1));
4012 if (NumSrcElts == 2)
4013 Rep = Builder.CreateShuffleVector(Load, ArrayRef<int>{0, 1, 0, 1});
4014 else
4015 Rep = Builder.CreateShuffleVector(Load,
4016 ArrayRef<int>{0, 1, 2, 3, 0, 1, 2, 3});
4017 } else if (Name.starts_with("avx512.mask.shuf.i") ||
4018 Name.starts_with("avx512.mask.shuf.f")) {
4019 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4020 Type *VT = CI->getType();
4021 unsigned NumLanes = VT->getPrimitiveSizeInBits() / 128;
4022 unsigned NumElementsInLane = 128 / VT->getScalarSizeInBits();
4023 unsigned ControlBitsMask = NumLanes - 1;
4024 unsigned NumControlBits = NumLanes / 2;
4025 SmallVector<int, 8> ShuffleMask(0);
4026
4027 for (unsigned l = 0; l != NumLanes; ++l) {
4028 unsigned LaneMask = (Imm >> (l * NumControlBits)) & ControlBitsMask;
4029 // We actually need the other source.
4030 if (l >= NumLanes / 2)
4031 LaneMask += NumLanes;
4032 for (unsigned i = 0; i != NumElementsInLane; ++i)
4033 ShuffleMask.push_back(LaneMask * NumElementsInLane + i);
4034 }
4035 Rep = Builder.CreateShuffleVector(CI->getArgOperand(0),
4036 CI->getArgOperand(1), ShuffleMask);
4037 Rep =
4038 emitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3));
4039 } else if (Name.starts_with("avx512.mask.broadcastf") ||
4040 Name.starts_with("avx512.mask.broadcasti")) {
4041 unsigned NumSrcElts = cast<FixedVectorType>(CI->getArgOperand(0)->getType())
4042 ->getNumElements();
4043 unsigned NumDstElts =
4044 cast<FixedVectorType>(CI->getType())->getNumElements();
4045
4046 SmallVector<int, 8> ShuffleMask(NumDstElts);
4047 for (unsigned i = 0; i != NumDstElts; ++i)
4048 ShuffleMask[i] = i % NumSrcElts;
4049
4050 Rep = Builder.CreateShuffleVector(CI->getArgOperand(0),
4051 CI->getArgOperand(0), ShuffleMask);
4052 Rep =
4053 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4054 } else if (Name.starts_with("avx2.pbroadcast") ||
4055 Name.starts_with("avx2.vbroadcast") ||
4056 Name.starts_with("avx512.pbroadcast") ||
4057 Name.starts_with("avx512.mask.broadcast.s")) {
4058 // Replace vp?broadcasts with a vector shuffle.
4059 Value *Op = CI->getArgOperand(0);
4060 ElementCount EC = cast<VectorType>(CI->getType())->getElementCount();
4061 Type *MaskTy = VectorType::get(Type::getInt32Ty(C), EC);
4064 Rep = Builder.CreateShuffleVector(Op, M);
4065
4066 if (CI->arg_size() == 3)
4067 Rep = emitX86Select(Builder, CI->getArgOperand(2), Rep,
4068 CI->getArgOperand(1));
4069 } else if (Name.starts_with("sse2.padds.") ||
4070 Name.starts_with("avx2.padds.") ||
4071 Name.starts_with("avx512.padds.") ||
4072 Name.starts_with("avx512.mask.padds.")) {
4073 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::sadd_sat);
4074 } else if (Name.starts_with("sse2.psubs.") ||
4075 Name.starts_with("avx2.psubs.") ||
4076 Name.starts_with("avx512.psubs.") ||
4077 Name.starts_with("avx512.mask.psubs.")) {
4078 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::ssub_sat);
4079 } else if (Name.starts_with("sse2.paddus.") ||
4080 Name.starts_with("avx2.paddus.") ||
4081 Name.starts_with("avx512.mask.paddus.")) {
4082 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::uadd_sat);
4083 } else if (Name.starts_with("sse2.psubus.") ||
4084 Name.starts_with("avx2.psubus.") ||
4085 Name.starts_with("avx512.mask.psubus.")) {
4086 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::usub_sat);
4087 } else if (Name.starts_with("avx512.mask.palignr.")) {
4088 Rep = upgradeX86ALIGNIntrinsics(Builder, CI->getArgOperand(0),
4089 CI->getArgOperand(1), CI->getArgOperand(2),
4090 CI->getArgOperand(3), CI->getArgOperand(4),
4091 false);
4092 } else if (Name.starts_with("avx512.mask.valign.")) {
4094 Builder, CI->getArgOperand(0), CI->getArgOperand(1),
4095 CI->getArgOperand(2), CI->getArgOperand(3), CI->getArgOperand(4), true);
4096 } else if (Name == "sse2.psll.dq" || Name == "avx2.psll.dq") {
4097 // 128/256-bit shift left specified in bits.
4098 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4099 Rep = upgradeX86PSLLDQIntrinsics(Builder, CI->getArgOperand(0),
4100 Shift / 8); // Shift is in bits.
4101 } else if (Name == "sse2.psrl.dq" || Name == "avx2.psrl.dq") {
4102 // 128/256-bit shift right specified in bits.
4103 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4104 Rep = upgradeX86PSRLDQIntrinsics(Builder, CI->getArgOperand(0),
4105 Shift / 8); // Shift is in bits.
4106 } else if (Name == "sse2.psll.dq.bs" || Name == "avx2.psll.dq.bs" ||
4107 Name == "avx512.psll.dq.512") {
4108 // 128/256/512-bit shift left specified in bytes.
4109 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4110 Rep = upgradeX86PSLLDQIntrinsics(Builder, CI->getArgOperand(0), Shift);
4111 } else if (Name == "sse2.psrl.dq.bs" || Name == "avx2.psrl.dq.bs" ||
4112 Name == "avx512.psrl.dq.512") {
4113 // 128/256/512-bit shift right specified in bytes.
4114 unsigned Shift = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4115 Rep = upgradeX86PSRLDQIntrinsics(Builder, CI->getArgOperand(0), Shift);
4116 } else if (Name == "sse41.pblendw" || Name.starts_with("sse41.blendp") ||
4117 Name.starts_with("avx.blend.p") || Name == "avx2.pblendw" ||
4118 Name.starts_with("avx2.pblendd.")) {
4119 Value *Op0 = CI->getArgOperand(0);
4120 Value *Op1 = CI->getArgOperand(1);
4121 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4122 auto *VecTy = cast<FixedVectorType>(CI->getType());
4123 unsigned NumElts = VecTy->getNumElements();
4124
4125 SmallVector<int, 16> Idxs(NumElts);
4126 for (unsigned i = 0; i != NumElts; ++i)
4127 Idxs[i] = ((Imm >> (i % 8)) & 1) ? i + NumElts : i;
4128
4129 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4130 } else if (Name.starts_with("avx.vinsertf128.") ||
4131 Name == "avx2.vinserti128" ||
4132 Name.starts_with("avx512.mask.insert")) {
4133 Value *Op0 = CI->getArgOperand(0);
4134 Value *Op1 = CI->getArgOperand(1);
4135 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4136 unsigned DstNumElts =
4137 cast<FixedVectorType>(CI->getType())->getNumElements();
4138 unsigned SrcNumElts =
4139 cast<FixedVectorType>(Op1->getType())->getNumElements();
4140 unsigned Scale = DstNumElts / SrcNumElts;
4141
4142 // Mask off the high bits of the immediate value; hardware ignores those.
4143 Imm = Imm % Scale;
4144
4145 // Extend the second operand into a vector the size of the destination.
4146 SmallVector<int, 8> Idxs(DstNumElts);
4147 for (unsigned i = 0; i != SrcNumElts; ++i)
4148 Idxs[i] = i;
4149 for (unsigned i = SrcNumElts; i != DstNumElts; ++i)
4150 Idxs[i] = SrcNumElts;
4151 Rep = Builder.CreateShuffleVector(Op1, Idxs);
4152
4153 // Insert the second operand into the first operand.
4154
4155 // Note that there is no guarantee that instruction lowering will actually
4156 // produce a vinsertf128 instruction for the created shuffles. In
4157 // particular, the 0 immediate case involves no lane changes, so it can
4158 // be handled as a blend.
4159
4160 // Example of shuffle mask for 32-bit elements:
4161 // Imm = 1 <i32 0, i32 1, i32 2, i32 3, i32 8, i32 9, i32 10, i32 11>
4162 // Imm = 0 <i32 8, i32 9, i32 10, i32 11, i32 4, i32 5, i32 6, i32 7 >
4163
4164 // First fill with identify mask.
4165 for (unsigned i = 0; i != DstNumElts; ++i)
4166 Idxs[i] = i;
4167 // Then replace the elements where we need to insert.
4168 for (unsigned i = 0; i != SrcNumElts; ++i)
4169 Idxs[i + Imm * SrcNumElts] = i + DstNumElts;
4170 Rep = Builder.CreateShuffleVector(Op0, Rep, Idxs);
4171
4172 // If the intrinsic has a mask operand, handle that.
4173 if (CI->arg_size() == 5)
4174 Rep = emitX86Select(Builder, CI->getArgOperand(4), Rep,
4175 CI->getArgOperand(3));
4176 } else if (Name.starts_with("avx.vextractf128.") ||
4177 Name == "avx2.vextracti128" ||
4178 Name.starts_with("avx512.mask.vextract")) {
4179 Value *Op0 = CI->getArgOperand(0);
4180 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4181 unsigned DstNumElts =
4182 cast<FixedVectorType>(CI->getType())->getNumElements();
4183 unsigned SrcNumElts =
4184 cast<FixedVectorType>(Op0->getType())->getNumElements();
4185 unsigned Scale = SrcNumElts / DstNumElts;
4186
4187 // Mask off the high bits of the immediate value; hardware ignores those.
4188 Imm = Imm % Scale;
4189
4190 // Get indexes for the subvector of the input vector.
4191 SmallVector<int, 8> Idxs(DstNumElts);
4192 for (unsigned i = 0; i != DstNumElts; ++i) {
4193 Idxs[i] = i + (Imm * DstNumElts);
4194 }
4195 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4196
4197 // If the intrinsic has a mask operand, handle that.
4198 if (CI->arg_size() == 4)
4199 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4200 CI->getArgOperand(2));
4201 } else if (Name.starts_with("avx512.mask.perm.df.") ||
4202 Name.starts_with("avx512.mask.perm.di.")) {
4203 Value *Op0 = CI->getArgOperand(0);
4204 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4205 auto *VecTy = cast<FixedVectorType>(CI->getType());
4206 unsigned NumElts = VecTy->getNumElements();
4207
4208 SmallVector<int, 8> Idxs(NumElts);
4209 for (unsigned i = 0; i != NumElts; ++i)
4210 Idxs[i] = (i & ~0x3) + ((Imm >> (2 * (i & 0x3))) & 3);
4211
4212 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4213
4214 if (CI->arg_size() == 4)
4215 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4216 CI->getArgOperand(2));
4217 } else if (Name.starts_with("avx.vperm2f128.") || Name == "avx2.vperm2i128") {
4218 // The immediate permute control byte looks like this:
4219 // [1:0] - select 128 bits from sources for low half of destination
4220 // [2] - ignore
4221 // [3] - zero low half of destination
4222 // [5:4] - select 128 bits from sources for high half of destination
4223 // [6] - ignore
4224 // [7] - zero high half of destination
4225
4226 uint8_t Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4227
4228 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4229 unsigned HalfSize = NumElts / 2;
4230 SmallVector<int, 8> ShuffleMask(NumElts);
4231
4232 // Determine which operand(s) are actually in use for this instruction.
4233 Value *V0 = (Imm & 0x02) ? CI->getArgOperand(1) : CI->getArgOperand(0);
4234 Value *V1 = (Imm & 0x20) ? CI->getArgOperand(1) : CI->getArgOperand(0);
4235
4236 // If needed, replace operands based on zero mask.
4237 V0 = (Imm & 0x08) ? ConstantAggregateZero::get(CI->getType()) : V0;
4238 V1 = (Imm & 0x80) ? ConstantAggregateZero::get(CI->getType()) : V1;
4239
4240 // Permute low half of result.
4241 unsigned StartIndex = (Imm & 0x01) ? HalfSize : 0;
4242 for (unsigned i = 0; i < HalfSize; ++i)
4243 ShuffleMask[i] = StartIndex + i;
4244
4245 // Permute high half of result.
4246 StartIndex = (Imm & 0x10) ? HalfSize : 0;
4247 for (unsigned i = 0; i < HalfSize; ++i)
4248 ShuffleMask[i + HalfSize] = NumElts + StartIndex + i;
4249
4250 Rep = Builder.CreateShuffleVector(V0, V1, ShuffleMask);
4251
4252 } else if (Name.starts_with("avx.vpermil.") || Name == "sse2.pshuf.d" ||
4253 Name.starts_with("avx512.mask.vpermil.p") ||
4254 Name.starts_with("avx512.mask.pshuf.d.")) {
4255 Value *Op0 = CI->getArgOperand(0);
4256 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4257 auto *VecTy = cast<FixedVectorType>(CI->getType());
4258 unsigned NumElts = VecTy->getNumElements();
4259 // Calculate the size of each index in the immediate.
4260 unsigned IdxSize = 64 / VecTy->getScalarSizeInBits();
4261 unsigned IdxMask = ((1 << IdxSize) - 1);
4262
4263 SmallVector<int, 8> Idxs(NumElts);
4264 // Lookup the bits for this element, wrapping around the immediate every
4265 // 8-bits. Elements are grouped into sets of 2 or 4 elements so we need
4266 // to offset by the first index of each group.
4267 for (unsigned i = 0; i != NumElts; ++i)
4268 Idxs[i] = ((Imm >> ((i * IdxSize) % 8)) & IdxMask) | (i & ~IdxMask);
4269
4270 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4271
4272 if (CI->arg_size() == 4)
4273 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4274 CI->getArgOperand(2));
4275 } else if (Name == "sse2.pshufl.w" ||
4276 Name.starts_with("avx512.mask.pshufl.w.")) {
4277 Value *Op0 = CI->getArgOperand(0);
4278 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4279 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4280
4281 if (Name == "sse2.pshufl.w" && NumElts % 8 != 0)
4282 reportFatalUsageErrorWithCI("Intrinsic has invalid signature", CI);
4283
4284 SmallVector<int, 16> Idxs(NumElts);
4285 for (unsigned l = 0; l != NumElts; l += 8) {
4286 for (unsigned i = 0; i != 4; ++i)
4287 Idxs[i + l] = ((Imm >> (2 * i)) & 0x3) + l;
4288 for (unsigned i = 4; i != 8; ++i)
4289 Idxs[i + l] = i + l;
4290 }
4291
4292 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4293
4294 if (CI->arg_size() == 4)
4295 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4296 CI->getArgOperand(2));
4297 } else if (Name == "sse2.pshufh.w" ||
4298 Name.starts_with("avx512.mask.pshufh.w.")) {
4299 Value *Op0 = CI->getArgOperand(0);
4300 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
4301 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4302
4303 if (Name == "sse2.pshufh.w" && NumElts % 8 != 0)
4304 reportFatalUsageErrorWithCI("Intrinsic has invalid signature", CI);
4305
4306 SmallVector<int, 16> Idxs(NumElts);
4307 for (unsigned l = 0; l != NumElts; l += 8) {
4308 for (unsigned i = 0; i != 4; ++i)
4309 Idxs[i + l] = i + l;
4310 for (unsigned i = 0; i != 4; ++i)
4311 Idxs[i + l + 4] = ((Imm >> (2 * i)) & 0x3) + 4 + l;
4312 }
4313
4314 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4315
4316 if (CI->arg_size() == 4)
4317 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep,
4318 CI->getArgOperand(2));
4319 } else if (Name.starts_with("avx512.mask.shuf.p")) {
4320 Value *Op0 = CI->getArgOperand(0);
4321 Value *Op1 = CI->getArgOperand(1);
4322 unsigned Imm = cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue();
4323 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4324
4325 unsigned NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4326 unsigned HalfLaneElts = NumLaneElts / 2;
4327
4328 SmallVector<int, 16> Idxs(NumElts);
4329 for (unsigned i = 0; i != NumElts; ++i) {
4330 // Base index is the starting element of the lane.
4331 Idxs[i] = i - (i % NumLaneElts);
4332 // If we are half way through the lane switch to the other source.
4333 if ((i % NumLaneElts) >= HalfLaneElts)
4334 Idxs[i] += NumElts;
4335 // Now select the specific element. By adding HalfLaneElts bits from
4336 // the immediate. Wrapping around the immediate every 8-bits.
4337 Idxs[i] += (Imm >> ((i * HalfLaneElts) % 8)) & ((1 << HalfLaneElts) - 1);
4338 }
4339
4340 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4341
4342 Rep =
4343 emitX86Select(Builder, CI->getArgOperand(4), Rep, CI->getArgOperand(3));
4344 } else if (Name.starts_with("avx512.mask.movddup") ||
4345 Name.starts_with("avx512.mask.movshdup") ||
4346 Name.starts_with("avx512.mask.movsldup")) {
4347 Value *Op0 = CI->getArgOperand(0);
4348 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4349 unsigned NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4350
4351 unsigned Offset = 0;
4352 if (Name.starts_with("avx512.mask.movshdup."))
4353 Offset = 1;
4354
4355 SmallVector<int, 16> Idxs(NumElts);
4356 for (unsigned l = 0; l != NumElts; l += NumLaneElts)
4357 for (unsigned i = 0; i != NumLaneElts; i += 2) {
4358 Idxs[i + l + 0] = i + l + Offset;
4359 Idxs[i + l + 1] = i + l + Offset;
4360 }
4361
4362 Rep = Builder.CreateShuffleVector(Op0, Op0, Idxs);
4363
4364 Rep =
4365 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4366 } else if (Name.starts_with("avx512.mask.punpckl") ||
4367 Name.starts_with("avx512.mask.unpckl.")) {
4368 Value *Op0 = CI->getArgOperand(0);
4369 Value *Op1 = CI->getArgOperand(1);
4370 int NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4371 int NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4372
4373 SmallVector<int, 64> Idxs(NumElts);
4374 for (int l = 0; l != NumElts; l += NumLaneElts)
4375 for (int i = 0; i != NumLaneElts; ++i)
4376 Idxs[i + l] = l + (i / 2) + NumElts * (i % 2);
4377
4378 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4379
4380 Rep =
4381 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4382 } else if (Name.starts_with("avx512.mask.punpckh") ||
4383 Name.starts_with("avx512.mask.unpckh.")) {
4384 Value *Op0 = CI->getArgOperand(0);
4385 Value *Op1 = CI->getArgOperand(1);
4386 int NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4387 int NumLaneElts = 128 / CI->getType()->getScalarSizeInBits();
4388
4389 SmallVector<int, 64> Idxs(NumElts);
4390 for (int l = 0; l != NumElts; l += NumLaneElts)
4391 for (int i = 0; i != NumLaneElts; ++i)
4392 Idxs[i + l] = (NumLaneElts / 2) + l + (i / 2) + NumElts * (i % 2);
4393
4394 Rep = Builder.CreateShuffleVector(Op0, Op1, Idxs);
4395
4396 Rep =
4397 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4398 } else if (Name.starts_with("avx512.mask.and.") ||
4399 Name.starts_with("avx512.mask.pand.")) {
4400 VectorType *FTy = cast<VectorType>(CI->getType());
4402 Rep = Builder.CreateAnd(Builder.CreateBitCast(CI->getArgOperand(0), ITy),
4403 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4404 Rep = Builder.CreateBitCast(Rep, FTy);
4405 Rep =
4406 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4407 } else if (Name.starts_with("avx512.mask.andn.") ||
4408 Name.starts_with("avx512.mask.pandn.")) {
4409 VectorType *FTy = cast<VectorType>(CI->getType());
4411 Rep = Builder.CreateNot(Builder.CreateBitCast(CI->getArgOperand(0), ITy));
4412 Rep = Builder.CreateAnd(Rep,
4413 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4414 Rep = Builder.CreateBitCast(Rep, FTy);
4415 Rep =
4416 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4417 } else if (Name.starts_with("avx512.mask.or.") ||
4418 Name.starts_with("avx512.mask.por.")) {
4419 VectorType *FTy = cast<VectorType>(CI->getType());
4421 Rep = Builder.CreateOr(Builder.CreateBitCast(CI->getArgOperand(0), ITy),
4422 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4423 Rep = Builder.CreateBitCast(Rep, FTy);
4424 Rep =
4425 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4426 } else if (Name.starts_with("avx512.mask.xor.") ||
4427 Name.starts_with("avx512.mask.pxor.")) {
4428 VectorType *FTy = cast<VectorType>(CI->getType());
4430 Rep = Builder.CreateXor(Builder.CreateBitCast(CI->getArgOperand(0), ITy),
4431 Builder.CreateBitCast(CI->getArgOperand(1), ITy));
4432 Rep = Builder.CreateBitCast(Rep, FTy);
4433 Rep =
4434 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4435 } else if (Name.starts_with("avx512.mask.padd.")) {
4436 Rep = Builder.CreateAdd(CI->getArgOperand(0), CI->getArgOperand(1));
4437 Rep =
4438 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4439 } else if (Name.starts_with("avx512.mask.psub.")) {
4440 Rep = Builder.CreateSub(CI->getArgOperand(0), CI->getArgOperand(1));
4441 Rep =
4442 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4443 } else if (Name.starts_with("avx512.mask.pmull.")) {
4444 Rep = Builder.CreateMul(CI->getArgOperand(0), CI->getArgOperand(1));
4445 Rep =
4446 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4447 } else if (Name.starts_with("avx512.mask.add.p")) {
4448 if (Name.ends_with(".512")) {
4449 Intrinsic::ID IID;
4450 if (Name[17] == 's')
4451 IID = Intrinsic::x86_avx512_add_ps_512;
4452 else
4453 IID = Intrinsic::x86_avx512_add_pd_512;
4454
4455 Rep = Builder.CreateIntrinsic(
4456 IID,
4457 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4458 } else {
4459 Rep = Builder.CreateFAdd(CI->getArgOperand(0), CI->getArgOperand(1));
4460 }
4461 Rep =
4462 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4463 } else if (Name.starts_with("avx512.mask.div.p")) {
4464 if (Name.ends_with(".512")) {
4465 Intrinsic::ID IID;
4466 if (Name[17] == 's')
4467 IID = Intrinsic::x86_avx512_div_ps_512;
4468 else
4469 IID = Intrinsic::x86_avx512_div_pd_512;
4470
4471 Rep = Builder.CreateIntrinsic(
4472 IID,
4473 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4474 } else {
4475 Rep = Builder.CreateFDiv(CI->getArgOperand(0), CI->getArgOperand(1));
4476 }
4477 Rep =
4478 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4479 } else if (Name.starts_with("avx512.mask.mul.p")) {
4480 if (Name.ends_with(".512")) {
4481 Intrinsic::ID IID;
4482 if (Name[17] == 's')
4483 IID = Intrinsic::x86_avx512_mul_ps_512;
4484 else
4485 IID = Intrinsic::x86_avx512_mul_pd_512;
4486
4487 Rep = Builder.CreateIntrinsic(
4488 IID,
4489 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4490 } else {
4491 Rep = Builder.CreateFMul(CI->getArgOperand(0), CI->getArgOperand(1));
4492 }
4493 Rep =
4494 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4495 } else if (Name.starts_with("avx512.mask.sub.p")) {
4496 if (Name.ends_with(".512")) {
4497 Intrinsic::ID IID;
4498 if (Name[17] == 's')
4499 IID = Intrinsic::x86_avx512_sub_ps_512;
4500 else
4501 IID = Intrinsic::x86_avx512_sub_pd_512;
4502
4503 Rep = Builder.CreateIntrinsic(
4504 IID,
4505 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4506 } else {
4507 Rep = Builder.CreateFSub(CI->getArgOperand(0), CI->getArgOperand(1));
4508 }
4509 Rep =
4510 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4511 } else if ((Name.starts_with("avx512.mask.max.p") ||
4512 Name.starts_with("avx512.mask.min.p")) &&
4513 Name.drop_front(18) == ".512") {
4514 bool IsDouble = Name[17] == 'd';
4515 bool IsMin = Name[13] == 'i';
4516 static const Intrinsic::ID MinMaxTbl[2][2] = {
4517 {Intrinsic::x86_avx512_max_ps_512, Intrinsic::x86_avx512_max_pd_512},
4518 {Intrinsic::x86_avx512_min_ps_512, Intrinsic::x86_avx512_min_pd_512}};
4519 Intrinsic::ID IID = MinMaxTbl[IsMin][IsDouble];
4520
4521 Rep = Builder.CreateIntrinsic(
4522 IID,
4523 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(4)});
4524 Rep =
4525 emitX86Select(Builder, CI->getArgOperand(3), Rep, CI->getArgOperand(2));
4526 } else if (Name.starts_with("avx512.mask.lzcnt.")) {
4527 Rep =
4528 Builder.CreateIntrinsic(Intrinsic::ctlz, CI->getType(),
4529 {CI->getArgOperand(0), Builder.getInt1(false)});
4530 Rep =
4531 emitX86Select(Builder, CI->getArgOperand(2), Rep, CI->getArgOperand(1));
4532 } else if (Name.starts_with("avx512.mask.psll")) {
4533 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4534 bool IsVariable = Name[16] == 'v';
4535 char Size = Name[16] == '.' ? Name[17]
4536 : Name[17] == '.' ? Name[18]
4537 : Name[18] == '.' ? Name[19]
4538 : Name[20];
4539
4540 Intrinsic::ID IID;
4541 if (IsVariable && Name[17] != '.') {
4542 if (Size == 'd' && Name[17] == '2') // avx512.mask.psllv2.di
4543 IID = Intrinsic::x86_avx2_psllv_q;
4544 else if (Size == 'd' && Name[17] == '4') // avx512.mask.psllv4.di
4545 IID = Intrinsic::x86_avx2_psllv_q_256;
4546 else if (Size == 's' && Name[17] == '4') // avx512.mask.psllv4.si
4547 IID = Intrinsic::x86_avx2_psllv_d;
4548 else if (Size == 's' && Name[17] == '8') // avx512.mask.psllv8.si
4549 IID = Intrinsic::x86_avx2_psllv_d_256;
4550 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psllv8.hi
4551 IID = Intrinsic::x86_avx512_psllv_w_128;
4552 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psllv16.hi
4553 IID = Intrinsic::x86_avx512_psllv_w_256;
4554 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psllv32hi
4555 IID = Intrinsic::x86_avx512_psllv_w_512;
4556 else
4557 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4558 } else if (Name.ends_with(".128")) {
4559 if (Size == 'd') // avx512.mask.psll.d.128, avx512.mask.psll.di.128
4560 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_d
4561 : Intrinsic::x86_sse2_psll_d;
4562 else if (Size == 'q') // avx512.mask.psll.q.128, avx512.mask.psll.qi.128
4563 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_q
4564 : Intrinsic::x86_sse2_psll_q;
4565 else if (Size == 'w') // avx512.mask.psll.w.128, avx512.mask.psll.wi.128
4566 IID = IsImmediate ? Intrinsic::x86_sse2_pslli_w
4567 : Intrinsic::x86_sse2_psll_w;
4568 else
4569 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4570 } else if (Name.ends_with(".256")) {
4571 if (Size == 'd') // avx512.mask.psll.d.256, avx512.mask.psll.di.256
4572 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_d
4573 : Intrinsic::x86_avx2_psll_d;
4574 else if (Size == 'q') // avx512.mask.psll.q.256, avx512.mask.psll.qi.256
4575 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_q
4576 : Intrinsic::x86_avx2_psll_q;
4577 else if (Size == 'w') // avx512.mask.psll.w.256, avx512.mask.psll.wi.256
4578 IID = IsImmediate ? Intrinsic::x86_avx2_pslli_w
4579 : Intrinsic::x86_avx2_psll_w;
4580 else
4581 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4582 } else {
4583 if (Size == 'd') // psll.di.512, pslli.d, psll.d, psllv.d.512
4584 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_d_512
4585 : IsVariable ? Intrinsic::x86_avx512_psllv_d_512
4586 : Intrinsic::x86_avx512_psll_d_512;
4587 else if (Size == 'q') // psll.qi.512, pslli.q, psll.q, psllv.q.512
4588 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_q_512
4589 : IsVariable ? Intrinsic::x86_avx512_psllv_q_512
4590 : Intrinsic::x86_avx512_psll_q_512;
4591 else if (Size == 'w') // psll.wi.512, pslli.w, psll.w
4592 IID = IsImmediate ? Intrinsic::x86_avx512_pslli_w_512
4593 : Intrinsic::x86_avx512_psll_w_512;
4594 else
4595 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4596 }
4597
4598 Rep = upgradeX86MaskedShift(Builder, *CI, IID);
4599 } else if (Name.starts_with("avx512.mask.psrl")) {
4600 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4601 bool IsVariable = Name[16] == 'v';
4602 char Size = Name[16] == '.' ? Name[17]
4603 : Name[17] == '.' ? Name[18]
4604 : Name[18] == '.' ? Name[19]
4605 : Name[20];
4606
4607 Intrinsic::ID IID;
4608 if (IsVariable && Name[17] != '.') {
4609 if (Size == 'd' && Name[17] == '2') // avx512.mask.psrlv2.di
4610 IID = Intrinsic::x86_avx2_psrlv_q;
4611 else if (Size == 'd' && Name[17] == '4') // avx512.mask.psrlv4.di
4612 IID = Intrinsic::x86_avx2_psrlv_q_256;
4613 else if (Size == 's' && Name[17] == '4') // avx512.mask.psrlv4.si
4614 IID = Intrinsic::x86_avx2_psrlv_d;
4615 else if (Size == 's' && Name[17] == '8') // avx512.mask.psrlv8.si
4616 IID = Intrinsic::x86_avx2_psrlv_d_256;
4617 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrlv8.hi
4618 IID = Intrinsic::x86_avx512_psrlv_w_128;
4619 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrlv16.hi
4620 IID = Intrinsic::x86_avx512_psrlv_w_256;
4621 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrlv32hi
4622 IID = Intrinsic::x86_avx512_psrlv_w_512;
4623 else
4624 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4625 } else if (Name.ends_with(".128")) {
4626 if (Size == 'd') // avx512.mask.psrl.d.128, avx512.mask.psrl.di.128
4627 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_d
4628 : Intrinsic::x86_sse2_psrl_d;
4629 else if (Size == 'q') // avx512.mask.psrl.q.128, avx512.mask.psrl.qi.128
4630 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_q
4631 : Intrinsic::x86_sse2_psrl_q;
4632 else if (Size == 'w') // avx512.mask.psrl.w.128, avx512.mask.psrl.wi.128
4633 IID = IsImmediate ? Intrinsic::x86_sse2_psrli_w
4634 : Intrinsic::x86_sse2_psrl_w;
4635 else
4636 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4637 } else if (Name.ends_with(".256")) {
4638 if (Size == 'd') // avx512.mask.psrl.d.256, avx512.mask.psrl.di.256
4639 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_d
4640 : Intrinsic::x86_avx2_psrl_d;
4641 else if (Size == 'q') // avx512.mask.psrl.q.256, avx512.mask.psrl.qi.256
4642 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_q
4643 : Intrinsic::x86_avx2_psrl_q;
4644 else if (Size == 'w') // avx512.mask.psrl.w.256, avx512.mask.psrl.wi.256
4645 IID = IsImmediate ? Intrinsic::x86_avx2_psrli_w
4646 : Intrinsic::x86_avx2_psrl_w;
4647 else
4648 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4649 } else {
4650 if (Size == 'd') // psrl.di.512, psrli.d, psrl.d, psrl.d.512
4651 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_d_512
4652 : IsVariable ? Intrinsic::x86_avx512_psrlv_d_512
4653 : Intrinsic::x86_avx512_psrl_d_512;
4654 else if (Size == 'q') // psrl.qi.512, psrli.q, psrl.q, psrl.q.512
4655 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_q_512
4656 : IsVariable ? Intrinsic::x86_avx512_psrlv_q_512
4657 : Intrinsic::x86_avx512_psrl_q_512;
4658 else if (Size == 'w') // psrl.wi.512, psrli.w, psrl.w)
4659 IID = IsImmediate ? Intrinsic::x86_avx512_psrli_w_512
4660 : Intrinsic::x86_avx512_psrl_w_512;
4661 else
4662 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4663 }
4664
4665 Rep = upgradeX86MaskedShift(Builder, *CI, IID);
4666 } else if (Name.starts_with("avx512.mask.psra")) {
4667 bool IsImmediate = Name[16] == 'i' || (Name.size() > 18 && Name[18] == 'i');
4668 bool IsVariable = Name[16] == 'v';
4669 char Size = Name[16] == '.' ? Name[17]
4670 : Name[17] == '.' ? Name[18]
4671 : Name[18] == '.' ? Name[19]
4672 : Name[20];
4673
4674 Intrinsic::ID IID;
4675 if (IsVariable && Name[17] != '.') {
4676 if (Size == 's' && Name[17] == '4') // avx512.mask.psrav4.si
4677 IID = Intrinsic::x86_avx2_psrav_d;
4678 else if (Size == 's' && Name[17] == '8') // avx512.mask.psrav8.si
4679 IID = Intrinsic::x86_avx2_psrav_d_256;
4680 else if (Size == 'h' && Name[17] == '8') // avx512.mask.psrav8.hi
4681 IID = Intrinsic::x86_avx512_psrav_w_128;
4682 else if (Size == 'h' && Name[17] == '1') // avx512.mask.psrav16.hi
4683 IID = Intrinsic::x86_avx512_psrav_w_256;
4684 else if (Name[17] == '3' && Name[18] == '2') // avx512.mask.psrav32hi
4685 IID = Intrinsic::x86_avx512_psrav_w_512;
4686 else
4687 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4688 } else if (Name.ends_with(".128")) {
4689 if (Size == 'd') // avx512.mask.psra.d.128, avx512.mask.psra.di.128
4690 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_d
4691 : Intrinsic::x86_sse2_psra_d;
4692 else if (Size == 'q') // avx512.mask.psra.q.128, avx512.mask.psra.qi.128
4693 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_128
4694 : IsVariable ? Intrinsic::x86_avx512_psrav_q_128
4695 : Intrinsic::x86_avx512_psra_q_128;
4696 else if (Size == 'w') // avx512.mask.psra.w.128, avx512.mask.psra.wi.128
4697 IID = IsImmediate ? Intrinsic::x86_sse2_psrai_w
4698 : Intrinsic::x86_sse2_psra_w;
4699 else
4700 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4701 } else if (Name.ends_with(".256")) {
4702 if (Size == 'd') // avx512.mask.psra.d.256, avx512.mask.psra.di.256
4703 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_d
4704 : Intrinsic::x86_avx2_psra_d;
4705 else if (Size == 'q') // avx512.mask.psra.q.256, avx512.mask.psra.qi.256
4706 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_256
4707 : IsVariable ? Intrinsic::x86_avx512_psrav_q_256
4708 : Intrinsic::x86_avx512_psra_q_256;
4709 else if (Size == 'w') // avx512.mask.psra.w.256, avx512.mask.psra.wi.256
4710 IID = IsImmediate ? Intrinsic::x86_avx2_psrai_w
4711 : Intrinsic::x86_avx2_psra_w;
4712 else
4713 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4714 } else {
4715 if (Size == 'd') // psra.di.512, psrai.d, psra.d, psrav.d.512
4716 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_d_512
4717 : IsVariable ? Intrinsic::x86_avx512_psrav_d_512
4718 : Intrinsic::x86_avx512_psra_d_512;
4719 else if (Size == 'q') // psra.qi.512, psrai.q, psra.q
4720 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_q_512
4721 : IsVariable ? Intrinsic::x86_avx512_psrav_q_512
4722 : Intrinsic::x86_avx512_psra_q_512;
4723 else if (Size == 'w') // psra.wi.512, psrai.w, psra.w
4724 IID = IsImmediate ? Intrinsic::x86_avx512_psrai_w_512
4725 : Intrinsic::x86_avx512_psra_w_512;
4726 else
4727 reportFatalUsageErrorWithCI("Intrinsic has unexpected size", CI);
4728 }
4729
4730 Rep = upgradeX86MaskedShift(Builder, *CI, IID);
4731 } else if (Name.starts_with("avx512.mask.move.s")) {
4732 Rep = upgradeMaskedMove(Builder, *CI);
4733 } else if (Name.starts_with("avx512.cvtmask2")) {
4734 Rep = upgradeMaskToInt(Builder, *CI);
4735 } else if (Name.ends_with(".movntdqa")) {
4737 C, ConstantAsMetadata::get(ConstantInt::get(Type::getInt32Ty(C), 1)));
4738
4739 LoadInst *LI = Builder.CreateAlignedLoad(
4740 CI->getType(), CI->getArgOperand(0),
4742 LI->setMetadata(LLVMContext::MD_nontemporal, Node);
4743 Rep = LI;
4744 } else if (Name.starts_with("fma.vfmadd.") ||
4745 Name.starts_with("fma.vfmsub.") ||
4746 Name.starts_with("fma.vfnmadd.") ||
4747 Name.starts_with("fma.vfnmsub.")) {
4748 bool NegMul = Name[6] == 'n';
4749 bool NegAcc = NegMul ? Name[8] == 's' : Name[7] == 's';
4750 bool IsScalar = NegMul ? Name[12] == 's' : Name[11] == 's';
4751
4752 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4753 CI->getArgOperand(2)};
4754
4755 if (IsScalar) {
4756 Ops[0] = Builder.CreateExtractElement(Ops[0], (uint64_t)0);
4757 Ops[1] = Builder.CreateExtractElement(Ops[1], (uint64_t)0);
4758 Ops[2] = Builder.CreateExtractElement(Ops[2], (uint64_t)0);
4759 }
4760
4761 if (NegMul && !IsScalar)
4762 Ops[0] = Builder.CreateFNeg(Ops[0]);
4763 if (NegMul && IsScalar)
4764 Ops[1] = Builder.CreateFNeg(Ops[1]);
4765 if (NegAcc)
4766 Ops[2] = Builder.CreateFNeg(Ops[2]);
4767
4768 Rep = Builder.CreateIntrinsic(Intrinsic::fma, Ops[0]->getType(), Ops);
4769
4770 if (IsScalar)
4771 Rep = Builder.CreateInsertElement(CI->getArgOperand(0), Rep, (uint64_t)0);
4772 } else if (Name.starts_with("fma4.vfmadd.s")) {
4773 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4774 CI->getArgOperand(2)};
4775
4776 Ops[0] = Builder.CreateExtractElement(Ops[0], (uint64_t)0);
4777 Ops[1] = Builder.CreateExtractElement(Ops[1], (uint64_t)0);
4778 Ops[2] = Builder.CreateExtractElement(Ops[2], (uint64_t)0);
4779
4780 Rep = Builder.CreateIntrinsic(Intrinsic::fma, Ops[0]->getType(), Ops);
4781
4782 Rep = Builder.CreateInsertElement(Constant::getNullValue(CI->getType()),
4783 Rep, (uint64_t)0);
4784 } else if (Name.starts_with("avx512.mask.vfmadd.s") ||
4785 Name.starts_with("avx512.maskz.vfmadd.s") ||
4786 Name.starts_with("avx512.mask3.vfmadd.s") ||
4787 Name.starts_with("avx512.mask3.vfmsub.s") ||
4788 Name.starts_with("avx512.mask3.vfnmsub.s")) {
4789 bool IsMask3 = Name[11] == '3';
4790 bool IsMaskZ = Name[11] == 'z';
4791 // Drop the "avx512.mask." to make it easier.
4792 Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12);
4793 bool NegMul = Name[2] == 'n';
4794 bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's';
4795
4796 Value *A = CI->getArgOperand(0);
4797 Value *B = CI->getArgOperand(1);
4798 Value *C = CI->getArgOperand(2);
4799
4800 if (NegMul && (IsMask3 || IsMaskZ))
4801 A = Builder.CreateFNeg(A);
4802 if (NegMul && !(IsMask3 || IsMaskZ))
4803 B = Builder.CreateFNeg(B);
4804 if (NegAcc)
4805 C = Builder.CreateFNeg(C);
4806
4807 A = Builder.CreateExtractElement(A, (uint64_t)0);
4808 B = Builder.CreateExtractElement(B, (uint64_t)0);
4809 C = Builder.CreateExtractElement(C, (uint64_t)0);
4810
4811 if (!isa<ConstantInt>(CI->getArgOperand(4)) ||
4812 cast<ConstantInt>(CI->getArgOperand(4))->getZExtValue() != 4) {
4813 Value *Ops[] = {A, B, C, CI->getArgOperand(4)};
4814
4815 Intrinsic::ID IID;
4816 if (Name.back() == 'd')
4817 IID = Intrinsic::x86_avx512_vfmadd_f64;
4818 else
4819 IID = Intrinsic::x86_avx512_vfmadd_f32;
4820 Rep = Builder.CreateIntrinsic(IID, Ops);
4821 } else {
4822 Rep = Builder.CreateFMA(A, B, C);
4823 }
4824
4825 Value *PassThru = IsMaskZ ? Constant::getNullValue(Rep->getType())
4826 : IsMask3 ? C
4827 : A;
4828
4829 // For Mask3 with NegAcc, we need to create a new extractelement that
4830 // avoids the negation above.
4831 if (NegAcc && IsMask3)
4832 PassThru =
4833 Builder.CreateExtractElement(CI->getArgOperand(2), (uint64_t)0);
4834
4835 Rep = emitX86ScalarSelect(Builder, CI->getArgOperand(3), Rep, PassThru);
4836 Rep = Builder.CreateInsertElement(CI->getArgOperand(IsMask3 ? 2 : 0), Rep,
4837 (uint64_t)0);
4838 } else if (Name.starts_with("avx512.mask.vfmadd.p") ||
4839 Name.starts_with("avx512.mask.vfnmadd.p") ||
4840 Name.starts_with("avx512.mask.vfnmsub.p") ||
4841 Name.starts_with("avx512.mask3.vfmadd.p") ||
4842 Name.starts_with("avx512.mask3.vfmsub.p") ||
4843 Name.starts_with("avx512.mask3.vfnmsub.p") ||
4844 Name.starts_with("avx512.maskz.vfmadd.p")) {
4845 bool IsMask3 = Name[11] == '3';
4846 bool IsMaskZ = Name[11] == 'z';
4847 // Drop the "avx512.mask." to make it easier.
4848 Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12);
4849 bool NegMul = Name[2] == 'n';
4850 bool NegAcc = NegMul ? Name[4] == 's' : Name[3] == 's';
4851
4852 Value *A = CI->getArgOperand(0);
4853 Value *B = CI->getArgOperand(1);
4854 Value *C = CI->getArgOperand(2);
4855
4856 if (NegMul && (IsMask3 || IsMaskZ))
4857 A = Builder.CreateFNeg(A);
4858 if (NegMul && !(IsMask3 || IsMaskZ))
4859 B = Builder.CreateFNeg(B);
4860 if (NegAcc)
4861 C = Builder.CreateFNeg(C);
4862
4863 if (CI->arg_size() == 5 &&
4864 (!isa<ConstantInt>(CI->getArgOperand(4)) ||
4865 cast<ConstantInt>(CI->getArgOperand(4))->getZExtValue() != 4)) {
4866 Intrinsic::ID IID;
4867 // Check the character before ".512" in string.
4868 if (Name[Name.size() - 5] == 's')
4869 IID = Intrinsic::x86_avx512_vfmadd_ps_512;
4870 else
4871 IID = Intrinsic::x86_avx512_vfmadd_pd_512;
4872
4873 Rep = Builder.CreateIntrinsic(IID, {A, B, C, CI->getArgOperand(4)});
4874 } else {
4875 Rep = Builder.CreateFMA(A, B, C);
4876 }
4877
4878 Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(CI->getType())
4879 : IsMask3 ? CI->getArgOperand(2)
4880 : CI->getArgOperand(0);
4881
4882 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
4883 } else if (Name.starts_with("fma.vfmsubadd.p")) {
4884 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4885 unsigned EltWidth = CI->getType()->getScalarSizeInBits();
4886 Intrinsic::ID IID;
4887 if (VecWidth == 128 && EltWidth == 32)
4888 IID = Intrinsic::x86_fma_vfmaddsub_ps;
4889 else if (VecWidth == 256 && EltWidth == 32)
4890 IID = Intrinsic::x86_fma_vfmaddsub_ps_256;
4891 else if (VecWidth == 128 && EltWidth == 64)
4892 IID = Intrinsic::x86_fma_vfmaddsub_pd;
4893 else if (VecWidth == 256 && EltWidth == 64)
4894 IID = Intrinsic::x86_fma_vfmaddsub_pd_256;
4895 else
4896 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
4897
4898 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4899 CI->getArgOperand(2)};
4900 Ops[2] = Builder.CreateFNeg(Ops[2]);
4901 Rep = Builder.CreateIntrinsic(IID, Ops);
4902 } else if (Name.starts_with("avx512.mask.vfmaddsub.p") ||
4903 Name.starts_with("avx512.mask3.vfmaddsub.p") ||
4904 Name.starts_with("avx512.maskz.vfmaddsub.p") ||
4905 Name.starts_with("avx512.mask3.vfmsubadd.p")) {
4906 bool IsMask3 = Name[11] == '3';
4907 bool IsMaskZ = Name[11] == 'z';
4908 // Drop the "avx512.mask." to make it easier.
4909 Name = Name.drop_front(IsMask3 || IsMaskZ ? 13 : 12);
4910 bool IsSubAdd = Name[3] == 's';
4911 if (CI->arg_size() == 5) {
4912 Intrinsic::ID IID;
4913 // Check the character before ".512" in string.
4914 if (Name[Name.size() - 5] == 's')
4915 IID = Intrinsic::x86_avx512_vfmaddsub_ps_512;
4916 else
4917 IID = Intrinsic::x86_avx512_vfmaddsub_pd_512;
4918
4919 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4920 CI->getArgOperand(2), CI->getArgOperand(4)};
4921 if (IsSubAdd)
4922 Ops[2] = Builder.CreateFNeg(Ops[2]);
4923
4924 Rep = Builder.CreateIntrinsic(IID, Ops);
4925 } else {
4926 int NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
4927
4928 Value *Ops[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4929 CI->getArgOperand(2)};
4930
4932 CI->getModule(), Intrinsic::fma, Ops[0]->getType());
4933 Value *Odd = Builder.CreateCall(FMA, Ops);
4934 Ops[2] = Builder.CreateFNeg(Ops[2]);
4935 Value *Even = Builder.CreateCall(FMA, Ops);
4936
4937 if (IsSubAdd)
4938 std::swap(Even, Odd);
4939
4940 SmallVector<int, 32> Idxs(NumElts);
4941 for (int i = 0; i != NumElts; ++i)
4942 Idxs[i] = i + (i % 2) * NumElts;
4943
4944 Rep = Builder.CreateShuffleVector(Even, Odd, Idxs);
4945 }
4946
4947 Value *PassThru = IsMaskZ ? llvm::Constant::getNullValue(CI->getType())
4948 : IsMask3 ? CI->getArgOperand(2)
4949 : CI->getArgOperand(0);
4950
4951 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
4952 } else if (Name.starts_with("avx512.mask.pternlog.") ||
4953 Name.starts_with("avx512.maskz.pternlog.")) {
4954 bool ZeroMask = Name[11] == 'z';
4955 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4956 unsigned EltWidth = CI->getType()->getScalarSizeInBits();
4957 Intrinsic::ID IID;
4958 if (VecWidth == 128 && EltWidth == 32)
4959 IID = Intrinsic::x86_avx512_pternlog_d_128;
4960 else if (VecWidth == 256 && EltWidth == 32)
4961 IID = Intrinsic::x86_avx512_pternlog_d_256;
4962 else if (VecWidth == 512 && EltWidth == 32)
4963 IID = Intrinsic::x86_avx512_pternlog_d_512;
4964 else if (VecWidth == 128 && EltWidth == 64)
4965 IID = Intrinsic::x86_avx512_pternlog_q_128;
4966 else if (VecWidth == 256 && EltWidth == 64)
4967 IID = Intrinsic::x86_avx512_pternlog_q_256;
4968 else if (VecWidth == 512 && EltWidth == 64)
4969 IID = Intrinsic::x86_avx512_pternlog_q_512;
4970 else
4971 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
4972
4973 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
4974 CI->getArgOperand(2), CI->getArgOperand(3)};
4975 Rep = Builder.CreateIntrinsic(IID, Args);
4976 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
4977 : CI->getArgOperand(0);
4978 Rep = emitX86Select(Builder, CI->getArgOperand(4), Rep, PassThru);
4979 } else if (Name.starts_with("avx512.mask.vpmadd52") ||
4980 Name.starts_with("avx512.maskz.vpmadd52")) {
4981 bool ZeroMask = Name[11] == 'z';
4982 bool High = Name[20] == 'h' || Name[21] == 'h';
4983 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
4984 Intrinsic::ID IID;
4985 if (VecWidth == 128 && !High)
4986 IID = Intrinsic::x86_avx512_vpmadd52l_uq_128;
4987 else if (VecWidth == 256 && !High)
4988 IID = Intrinsic::x86_avx512_vpmadd52l_uq_256;
4989 else if (VecWidth == 512 && !High)
4990 IID = Intrinsic::x86_avx512_vpmadd52l_uq_512;
4991 else if (VecWidth == 128 && High)
4992 IID = Intrinsic::x86_avx512_vpmadd52h_uq_128;
4993 else if (VecWidth == 256 && High)
4994 IID = Intrinsic::x86_avx512_vpmadd52h_uq_256;
4995 else if (VecWidth == 512 && High)
4996 IID = Intrinsic::x86_avx512_vpmadd52h_uq_512;
4997 else
4998 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
4999
5000 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5001 CI->getArgOperand(2)};
5002 Rep = Builder.CreateIntrinsic(IID, Args);
5003 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
5004 : CI->getArgOperand(0);
5005 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
5006 } else if (Name.starts_with("avx512.mask.vpermi2var.") ||
5007 Name.starts_with("avx512.mask.vpermt2var.") ||
5008 Name.starts_with("avx512.maskz.vpermt2var.")) {
5009 bool ZeroMask = Name[11] == 'z';
5010 bool IndexForm = Name[17] == 'i';
5011 Rep = upgradeX86VPERMT2Intrinsics(Builder, *CI, ZeroMask, IndexForm);
5012 } else if (Name.starts_with("avx512.mask.vpdpbusd.") ||
5013 Name.starts_with("avx512.maskz.vpdpbusd.") ||
5014 Name.starts_with("avx512.mask.vpdpbusds.") ||
5015 Name.starts_with("avx512.maskz.vpdpbusds.")) {
5016 bool ZeroMask = Name[11] == 'z';
5017 bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's';
5018 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5019 Intrinsic::ID IID;
5020 if (VecWidth == 128 && !IsSaturating)
5021 IID = Intrinsic::x86_avx512_vpdpbusd_128;
5022 else if (VecWidth == 256 && !IsSaturating)
5023 IID = Intrinsic::x86_avx512_vpdpbusd_256;
5024 else if (VecWidth == 512 && !IsSaturating)
5025 IID = Intrinsic::x86_avx512_vpdpbusd_512;
5026 else if (VecWidth == 128 && IsSaturating)
5027 IID = Intrinsic::x86_avx512_vpdpbusds_128;
5028 else if (VecWidth == 256 && IsSaturating)
5029 IID = Intrinsic::x86_avx512_vpdpbusds_256;
5030 else if (VecWidth == 512 && IsSaturating)
5031 IID = Intrinsic::x86_avx512_vpdpbusds_512;
5032 else
5033 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5034
5035 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5036 CI->getArgOperand(2)};
5037
5038 // Input arguments types were incorrectly set to vectors of i32 before but
5039 // they should be vectors of i8. Insert bit cast when encountering the old
5040 // types
5041 if (Args[1]->getType()->isVectorTy() &&
5042 cast<VectorType>(Args[1]->getType())
5043 ->getElementType()
5044 ->isIntegerTy(32) &&
5045 Args[2]->getType()->isVectorTy() &&
5046 cast<VectorType>(Args[2]->getType())
5047 ->getElementType()
5048 ->isIntegerTy(32)) {
5049 Type *NewArgType = nullptr;
5050 if (VecWidth == 128)
5051 NewArgType = VectorType::get(Builder.getInt8Ty(), 16, false);
5052 else if (VecWidth == 256)
5053 NewArgType = VectorType::get(Builder.getInt8Ty(), 32, false);
5054 else if (VecWidth == 512)
5055 NewArgType = VectorType::get(Builder.getInt8Ty(), 64, false);
5056 else
5057 reportFatalUsageErrorWithCI("Intrinsic has unexpected vector bit width",
5058 CI);
5059
5060 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
5061 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
5062 }
5063
5064 Rep = Builder.CreateIntrinsic(IID, Args);
5065 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
5066 : CI->getArgOperand(0);
5067 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
5068 } else if (Name.starts_with("avx512.mask.vpdpwssd.") ||
5069 Name.starts_with("avx512.maskz.vpdpwssd.") ||
5070 Name.starts_with("avx512.mask.vpdpwssds.") ||
5071 Name.starts_with("avx512.maskz.vpdpwssds.")) {
5072 bool ZeroMask = Name[11] == 'z';
5073 bool IsSaturating = Name[ZeroMask ? 21 : 20] == 's';
5074 unsigned VecWidth = CI->getType()->getPrimitiveSizeInBits();
5075 Intrinsic::ID IID;
5076 if (VecWidth == 128 && !IsSaturating)
5077 IID = Intrinsic::x86_avx512_vpdpwssd_128;
5078 else if (VecWidth == 256 && !IsSaturating)
5079 IID = Intrinsic::x86_avx512_vpdpwssd_256;
5080 else if (VecWidth == 512 && !IsSaturating)
5081 IID = Intrinsic::x86_avx512_vpdpwssd_512;
5082 else if (VecWidth == 128 && IsSaturating)
5083 IID = Intrinsic::x86_avx512_vpdpwssds_128;
5084 else if (VecWidth == 256 && IsSaturating)
5085 IID = Intrinsic::x86_avx512_vpdpwssds_256;
5086 else if (VecWidth == 512 && IsSaturating)
5087 IID = Intrinsic::x86_avx512_vpdpwssds_512;
5088 else
5089 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5090
5091 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5092 CI->getArgOperand(2)};
5093
5094 // Input arguments types were incorrectly set to vectors of i32 before but
5095 // they should be vectors of i16. Insert bit cast when encountering the old
5096 // types
5097 if (Args[1]->getType()->isVectorTy() &&
5098 cast<VectorType>(Args[1]->getType())
5099 ->getElementType()
5100 ->isIntegerTy(32) &&
5101 Args[2]->getType()->isVectorTy() &&
5102 cast<VectorType>(Args[2]->getType())
5103 ->getElementType()
5104 ->isIntegerTy(32)) {
5105 Type *NewArgType = nullptr;
5106 if (VecWidth == 128)
5107 NewArgType = VectorType::get(Builder.getInt16Ty(), 8, false);
5108 else if (VecWidth == 256)
5109 NewArgType = VectorType::get(Builder.getInt16Ty(), 16, false);
5110 else if (VecWidth == 512)
5111 NewArgType = VectorType::get(Builder.getInt16Ty(), 32, false);
5112 else
5113 reportFatalUsageErrorWithCI("Intrinsic has unexpected vector bit width",
5114 CI);
5115
5116 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
5117 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
5118 }
5119
5120 Rep = Builder.CreateIntrinsic(IID, Args);
5121 Value *PassThru = ZeroMask ? ConstantAggregateZero::get(CI->getType())
5122 : CI->getArgOperand(0);
5123 Rep = emitX86Select(Builder, CI->getArgOperand(3), Rep, PassThru);
5124 } else if (Name == "addcarryx.u32" || Name == "addcarryx.u64" ||
5125 Name == "addcarry.u32" || Name == "addcarry.u64" ||
5126 Name == "subborrow.u32" || Name == "subborrow.u64") {
5127 Intrinsic::ID IID;
5128 if (Name[0] == 'a' && Name.back() == '2')
5129 IID = Intrinsic::x86_addcarry_32;
5130 else if (Name[0] == 'a' && Name.back() == '4')
5131 IID = Intrinsic::x86_addcarry_64;
5132 else if (Name[0] == 's' && Name.back() == '2')
5133 IID = Intrinsic::x86_subborrow_32;
5134 else if (Name[0] == 's' && Name.back() == '4')
5135 IID = Intrinsic::x86_subborrow_64;
5136 else
5137 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5138
5139 // Make a call with 3 operands.
5140 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
5141 CI->getArgOperand(2)};
5142 Value *NewCall = Builder.CreateIntrinsic(IID, Args);
5143
5144 // Extract the second result and store it.
5145 Value *Data = Builder.CreateExtractValue(NewCall, 1);
5146 Builder.CreateAlignedStore(Data, CI->getArgOperand(3), Align(1));
5147 // Replace the original call result with the first result of the new call.
5148 Value *CF = Builder.CreateExtractValue(NewCall, 0);
5149
5150 CI->replaceAllUsesWith(CF);
5151 Rep = nullptr;
5152 } else if (Name.starts_with("avx512.mask.") &&
5153 upgradeAVX512MaskToSelect(Name, Builder, *CI, Rep)) {
5154 // Rep will be updated by the call in the condition.
5155 } else if (Name.starts_with("bmi.pdep.")) {
5156 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::pdep);
5157 } else if (Name.starts_with("bmi.pext.")) {
5158 Rep = upgradeX86BinaryIntrinsics(Builder, *CI, Intrinsic::pext);
5159 } else
5160 reportFatalUsageErrorWithCI("Unexpected intrinsic", CI);
5161
5162 return Rep;
5163}
5164
5166 Function *F, IRBuilder<> &Builder) {
5167 if (Name.starts_with("neon.bfcvt")) {
5168 if (Name.starts_with("neon.bfcvtn2")) {
5169 SmallVector<int, 32> LoMask(4);
5170 std::iota(LoMask.begin(), LoMask.end(), 0);
5171 SmallVector<int, 32> ConcatMask(8);
5172 std::iota(ConcatMask.begin(), ConcatMask.end(), 0);
5173 Value *Inactive = Builder.CreateShuffleVector(CI->getOperand(0), LoMask);
5174 Value *Trunc =
5175 Builder.CreateFPTrunc(CI->getOperand(1), Inactive->getType());
5176 return Builder.CreateShuffleVector(Inactive, Trunc, ConcatMask);
5177 } else if (Name.starts_with("neon.bfcvtn")) {
5178 SmallVector<int, 32> ConcatMask(8);
5179 std::iota(ConcatMask.begin(), ConcatMask.end(), 0);
5180 Type *V4BF16 =
5181 FixedVectorType::get(Type::getBFloatTy(F->getContext()), 4);
5182 Value *Trunc = Builder.CreateFPTrunc(CI->getOperand(0), V4BF16);
5183 dbgs() << "Trunc: " << *Trunc << "\n";
5184 return Builder.CreateShuffleVector(
5185 Trunc, ConstantAggregateZero::get(V4BF16), ConcatMask);
5186 } else {
5187 return Builder.CreateFPTrunc(CI->getOperand(0),
5188 Type::getBFloatTy(F->getContext()));
5189 }
5190 } else if (Name.starts_with("sve.fcvt")) {
5191 Intrinsic::ID NewID =
5193 .Case("sve.fcvt.bf16f32", Intrinsic::aarch64_sve_fcvt_bf16f32_v2)
5194 .Case("sve.fcvtnt.bf16f32",
5195 Intrinsic::aarch64_sve_fcvtnt_bf16f32_v2)
5197 if (NewID == Intrinsic::not_intrinsic)
5198 llvm_unreachable("Unhandled Intrinsic!");
5199
5200 SmallVector<Value *, 3> Args(CI->args());
5201
5202 // The original intrinsics incorrectly used a predicate based on the
5203 // smallest element type rather than the largest.
5204 Type *BadPredTy = ScalableVectorType::get(Builder.getInt1Ty(), 8);
5205 Type *GoodPredTy = ScalableVectorType::get(Builder.getInt1Ty(), 4);
5206
5207 if (Args[1]->getType() != BadPredTy)
5208 llvm_unreachable("Unexpected predicate type!");
5209
5210 Args[1] = Builder.CreateIntrinsic(Intrinsic::aarch64_sve_convert_to_svbool,
5211 BadPredTy, Args[1]);
5212 Args[1] = Builder.CreateIntrinsic(
5213 Intrinsic::aarch64_sve_convert_from_svbool, GoodPredTy, Args[1]);
5214
5215 return Builder.CreateIntrinsic(NewID, Args, /*FMFSource=*/nullptr,
5216 CI->getName());
5217 }
5218
5219 if (Name == "neon.vcvtfp2hf")
5220 return Builder.CreateBitCast(
5221 Builder.CreateFPTrunc(
5222 CI->getOperand(0),
5223 FixedVectorType::get(Type::getHalfTy(F->getContext()), 4)),
5224 FixedVectorType::get(Type::getInt16Ty(F->getContext()), 4));
5225 if (Name == "neon.vcvthf2fp")
5226 return Builder.CreateFPExt(
5227 Builder.CreateBitCast(
5228 CI->getOperand(0),
5229 FixedVectorType::get(Type::getHalfTy(F->getContext()), 4)),
5230 FixedVectorType::get(Type::getFloatTy(F->getContext()), 4));
5231
5232 llvm_unreachable("Unhandled Intrinsic!");
5233}
5234
5236 IRBuilder<> &Builder) {
5237 if (Name == "mve.vctp64.old") {
5238 // Replace the old v4i1 vctp64 with a v2i1 vctp and predicate-casts to the
5239 // correct type.
5240 Value *VCTP = Builder.CreateIntrinsic(Intrinsic::arm_mve_vctp64, {},
5241 CI->getArgOperand(0),
5242 /*FMFSource=*/nullptr, CI->getName());
5243 Value *C1 = Builder.CreateIntrinsic(
5244 Intrinsic::arm_mve_pred_v2i,
5245 {VectorType::get(Builder.getInt1Ty(), 2, false)}, VCTP);
5246 return Builder.CreateIntrinsic(
5247 Intrinsic::arm_mve_pred_i2v,
5248 {VectorType::get(Builder.getInt1Ty(), 4, false)}, C1);
5249 } else if (Name == "mve.mull.int.predicated.v2i64.v4i32.v4i1" ||
5250 Name == "mve.vqdmull.predicated.v2i64.v4i32.v4i1" ||
5251 Name == "mve.vldr.gather.base.predicated.v2i64.v2i64.v4i1" ||
5252 Name == "mve.vldr.gather.base.wb.predicated.v2i64.v2i64.v4i1" ||
5253 Name ==
5254 "mve.vldr.gather.offset.predicated.v2i64.p0i64.v2i64.v4i1" ||
5255 Name == "mve.vldr.gather.offset.predicated.v2i64.p0.v2i64.v4i1" ||
5256 Name == "mve.vstr.scatter.base.predicated.v2i64.v2i64.v4i1" ||
5257 Name == "mve.vstr.scatter.base.wb.predicated.v2i64.v2i64.v4i1" ||
5258 Name ==
5259 "mve.vstr.scatter.offset.predicated.p0i64.v2i64.v2i64.v4i1" ||
5260 Name == "mve.vstr.scatter.offset.predicated.p0.v2i64.v2i64.v4i1" ||
5261 Name == "cde.vcx1q.predicated.v2i64.v4i1" ||
5262 Name == "cde.vcx1qa.predicated.v2i64.v4i1" ||
5263 Name == "cde.vcx2q.predicated.v2i64.v4i1" ||
5264 Name == "cde.vcx2qa.predicated.v2i64.v4i1" ||
5265 Name == "cde.vcx3q.predicated.v2i64.v4i1" ||
5266 Name == "cde.vcx3qa.predicated.v2i64.v4i1") {
5267 std::vector<Type *> Tys;
5268 unsigned ID = CI->getIntrinsicID();
5269 Type *V2I1Ty = FixedVectorType::get(Builder.getInt1Ty(), 2);
5270 switch (ID) {
5271 case Intrinsic::arm_mve_mull_int_predicated:
5272 case Intrinsic::arm_mve_vqdmull_predicated:
5273 case Intrinsic::arm_mve_vldr_gather_base_predicated:
5274 Tys = {CI->getType(), CI->getOperand(0)->getType(), V2I1Ty};
5275 break;
5276 case Intrinsic::arm_mve_vldr_gather_base_wb_predicated:
5277 case Intrinsic::arm_mve_vstr_scatter_base_predicated:
5278 case Intrinsic::arm_mve_vstr_scatter_base_wb_predicated:
5279 Tys = {CI->getOperand(0)->getType(), CI->getOperand(0)->getType(),
5280 V2I1Ty};
5281 break;
5282 case Intrinsic::arm_mve_vldr_gather_offset_predicated:
5283 Tys = {CI->getType(), CI->getOperand(0)->getType(),
5284 CI->getOperand(1)->getType(), V2I1Ty};
5285 break;
5286 case Intrinsic::arm_mve_vstr_scatter_offset_predicated:
5287 Tys = {CI->getOperand(0)->getType(), CI->getOperand(1)->getType(),
5288 CI->getOperand(2)->getType(), V2I1Ty};
5289 break;
5290 case Intrinsic::arm_cde_vcx1q_predicated:
5291 case Intrinsic::arm_cde_vcx1qa_predicated:
5292 case Intrinsic::arm_cde_vcx2q_predicated:
5293 case Intrinsic::arm_cde_vcx2qa_predicated:
5294 case Intrinsic::arm_cde_vcx3q_predicated:
5295 case Intrinsic::arm_cde_vcx3qa_predicated:
5296 Tys = {CI->getOperand(1)->getType(), V2I1Ty};
5297 break;
5298 default:
5299 llvm_unreachable("Unhandled Intrinsic!");
5300 }
5301
5302 std::vector<Value *> Ops;
5303 for (Value *Op : CI->args()) {
5304 Type *Ty = Op->getType();
5305 if (Ty->getScalarSizeInBits() == 1) {
5306 Value *C1 = Builder.CreateIntrinsic(
5307 Intrinsic::arm_mve_pred_v2i,
5308 {VectorType::get(Builder.getInt1Ty(), 4, false)}, Op);
5309 Op = Builder.CreateIntrinsic(Intrinsic::arm_mve_pred_i2v, {V2I1Ty}, C1);
5310 }
5311 Ops.push_back(Op);
5312 }
5313
5314 return Builder.CreateIntrinsic(ID, Tys, Ops, /*FMFSource=*/nullptr,
5315 CI->getName());
5316 }
5317 llvm_unreachable("Unknown function for ARM CallBase upgrade.");
5318}
5319
5320// These are expected to have the arguments:
5321// atomic.intrin (ptr, rmw_value, ordering, scope, isVolatile)
5322//
5323// Except for int_amdgcn_ds_fadd_v2bf16 which only has (ptr, rmw_value).
5324//
5326 Function *F, IRBuilder<> &Builder) {
5327 // Legacy WMMA iu intrinsics missed the optional clamp operand. Append clamp=0
5328 // for compatibility.
5329 auto UpgradeLegacyWMMAIUIntrinsicCall =
5330 [](Function *F, CallBase *CI, IRBuilder<> &Builder,
5331 ArrayRef<Type *> OverloadTys) -> Value * {
5332 // Prepare arguments, append clamp=0 for compatibility
5333 SmallVector<Value *, 10> Args(CI->args().begin(), CI->args().end());
5334 Args.push_back(Builder.getFalse());
5335
5336 // Insert the declaration for the right overload types
5338 F->getParent(), F->getIntrinsicID(), OverloadTys);
5339
5340 // Copy operand bundles if any
5342 CI->getOperandBundlesAsDefs(Bundles);
5343
5344 // Create the new call and copy calling properties
5345 auto *NewCall = cast<CallInst>(Builder.CreateCall(NewDecl, Args, Bundles));
5346 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
5347 NewCall->setCallingConv(CI->getCallingConv());
5348 NewCall->setAttributes(CI->getAttributes());
5349 NewCall->copyMetadata(*CI);
5350 return NewCall;
5351 };
5352
5353 if (F->getIntrinsicID() == Intrinsic::amdgcn_wmma_i32_16x16x64_iu8) {
5354 assert(CI->arg_size() == 7 && "Legacy int_amdgcn_wmma_i32_16x16x64_iu8 "
5355 "intrinsic should have 7 arguments");
5356 Type *T1 = CI->getArgOperand(4)->getType();
5357 Type *T2 = CI->getArgOperand(1)->getType();
5358 return UpgradeLegacyWMMAIUIntrinsicCall(F, CI, Builder, {T1, T2});
5359 }
5360 if (F->getIntrinsicID() == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8) {
5361 assert(CI->arg_size() == 8 && "Legacy int_amdgcn_swmmac_i32_16x16x128_iu8 "
5362 "intrinsic should have 8 arguments");
5363 Type *T1 = CI->getArgOperand(4)->getType();
5364 Type *T2 = CI->getArgOperand(1)->getType();
5365 Type *T3 = CI->getArgOperand(3)->getType();
5366 Type *T4 = CI->getArgOperand(5)->getType();
5367 return UpgradeLegacyWMMAIUIntrinsicCall(F, CI, Builder, {T1, T2, T3, T4});
5368 }
5369
5370 switch (F->getIntrinsicID()) {
5371 default:
5372 break;
5373 case Intrinsic::amdgcn_wmma_f32_16x16x4_f32:
5374 case Intrinsic::amdgcn_wmma_f32_16x16x32_bf16:
5375 case Intrinsic::amdgcn_wmma_f32_16x16x32_f16:
5376 case Intrinsic::amdgcn_wmma_f16_16x16x32_f16:
5377 case Intrinsic::amdgcn_wmma_bf16_16x16x32_bf16:
5378 case Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16: {
5379 // Drop src0 and src1 modifiers.
5380 const Value *Op0 = CI->getArgOperand(0);
5381 const Value *Op2 = CI->getArgOperand(2);
5382 assert(Op0->getType()->isIntegerTy() && Op2->getType()->isIntegerTy());
5383 const ConstantInt *ModA = dyn_cast<ConstantInt>(Op0);
5384 const ConstantInt *ModB = dyn_cast<ConstantInt>(Op2);
5385 if (!ModA->isZero() || !ModB->isZero())
5386 reportFatalUsageError(Name + " matrix A and B modifiers shall be zero");
5387
5389 for (int I = 4, E = CI->arg_size(); I < E; ++I)
5390 Args.push_back(CI->getArgOperand(I));
5391
5392 SmallVector<Type *, 3> Overloads{F->getReturnType(), Args[0]->getType()};
5393 if (F->getIntrinsicID() == Intrinsic::amdgcn_wmma_bf16f32_16x16x32_bf16)
5394 Overloads.push_back(Args[3]->getType());
5396 F->getParent(), F->getIntrinsicID(), Overloads);
5397
5399 CI->getOperandBundlesAsDefs(Bundles);
5400
5401 auto *NewCall = cast<CallInst>(Builder.CreateCall(NewDecl, Args, Bundles));
5402 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
5403 NewCall->setCallingConv(CI->getCallingConv());
5404 NewCall->setAttributes(CI->getAttributes());
5405 NewCall->copyMetadata(*CI);
5406 NewCall->takeName(CI);
5407 return NewCall;
5408 }
5409 }
5410
5411 if (Name.starts_with("fcmp.") || Name.starts_with("icmp.")) {
5412 Value *LHS = CI->getArgOperand(0);
5413 Value *RHS = CI->getArgOperand(1);
5414 CmpInst::Predicate Pred = static_cast<CmpInst::Predicate>(
5415 cast<ConstantInt>(CI->getArgOperand(2))->getZExtValue());
5416 Value *Cmp = Builder.CreateCmp(Pred, LHS, RHS);
5417 CallInst *NewCall = Builder.CreateIntrinsicWithoutFolding(
5418 CI->getType(), Intrinsic::amdgcn_ballot, Cmp);
5419 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
5420 NewCall->setCallingConv(CI->getCallingConv());
5421 NewCall->copyMetadata(*CI);
5422 NewCall->takeName(CI);
5423 return NewCall;
5424 }
5425
5426 if (Name.starts_with("addrspacecast.nonnull")) {
5427 if (CI->getNumOperands() < 2) // Malformed bitcode.
5428 return nullptr;
5429 Value *ASC = Builder.CreateAddrSpaceCast(
5430 CI->getArgOperand(0), CI->getType(), "", /*IsNonNull=*/true);
5431 ASC->takeName(CI);
5432 return ASC;
5433 }
5434
5435 AtomicRMWInst::BinOp RMWOp =
5437 .StartsWith("ds.fadd", AtomicRMWInst::FAdd)
5438 .StartsWith("ds.fmin", AtomicRMWInst::FMin)
5439 .StartsWith("ds.fmax", AtomicRMWInst::FMax)
5440 .StartsWith("atomic.inc.", AtomicRMWInst::UIncWrap)
5441 .StartsWith("atomic.dec.", AtomicRMWInst::UDecWrap)
5442 .StartsWith("global.atomic.fadd", AtomicRMWInst::FAdd)
5443 .StartsWith("flat.atomic.fadd", AtomicRMWInst::FAdd)
5444 .StartsWith("global.atomic.fmin", AtomicRMWInst::FMin)
5445 .StartsWith("flat.atomic.fmin", AtomicRMWInst::FMin)
5446 .StartsWith("global.atomic.fmax", AtomicRMWInst::FMax)
5447 .StartsWith("flat.atomic.fmax", AtomicRMWInst::FMax)
5448 .StartsWith("atomic.cond.sub", AtomicRMWInst::USubCond)
5449 .StartsWith("atomic.csub", AtomicRMWInst::USubSat);
5450
5451 unsigned NumOperands = CI->getNumOperands();
5452 if (NumOperands < 3) // Malformed bitcode.
5453 return nullptr;
5454
5455 Value *Ptr = CI->getArgOperand(0);
5456 PointerType *PtrTy = dyn_cast<PointerType>(Ptr->getType());
5457 if (!PtrTy) // Malformed.
5458 return nullptr;
5459
5460 Value *Val = CI->getArgOperand(1);
5461 if (Val->getType() != CI->getType()) // Malformed.
5462 return nullptr;
5463
5464 ConstantInt *OrderArg = nullptr;
5465 bool IsVolatile = false;
5466
5467 // These should have 5 arguments (plus the callee). A separate version of the
5468 // ds_fadd intrinsic was defined for bf16 which was missing arguments.
5469 if (NumOperands > 3)
5470 OrderArg = dyn_cast<ConstantInt>(CI->getArgOperand(2));
5471
5472 // Ignore scope argument at 3
5473
5474 if (NumOperands > 5) {
5475 ConstantInt *VolatileArg = dyn_cast<ConstantInt>(CI->getArgOperand(4));
5476 IsVolatile = !VolatileArg || !VolatileArg->isZero();
5477 }
5478
5480 if (OrderArg && isValidAtomicOrdering(OrderArg->getZExtValue()))
5481 Order = static_cast<AtomicOrdering>(OrderArg->getZExtValue());
5484
5485 LLVMContext &Ctx = F->getContext();
5486
5487 // Handle the v2bf16 intrinsic which used <2 x i16> instead of <2 x bfloat>
5488 Type *RetTy = CI->getType();
5489 if (VectorType *VT = dyn_cast<VectorType>(RetTy)) {
5490 if (VT->getElementType()->isIntegerTy(16)) {
5491 VectorType *AsBF16 =
5492 VectorType::get(Type::getBFloatTy(Ctx), VT->getElementCount());
5493 Val = Builder.CreateBitCast(Val, AsBF16);
5494 }
5495 }
5496
5497 // The scope argument never really worked correctly. Use agent as the most
5498 // conservative option which should still always produce the instruction.
5499 SyncScope::ID SSID = Ctx.getOrInsertSyncScopeID("agent");
5500 AtomicRMWInst *RMW =
5501 Builder.CreateAtomicRMW(RMWOp, Ptr, Val, std::nullopt, Order, SSID);
5502
5503 unsigned AddrSpace = PtrTy->getAddressSpace();
5504 if (AddrSpace != AMDGPUAS::LOCAL_ADDRESS) {
5505 MDNode *EmptyMD = MDNode::get(F->getContext(), {});
5506 RMW->setMetadata("amdgpu.no.fine.grained.memory", EmptyMD);
5507 if (RMWOp == AtomicRMWInst::FAdd && RetTy->isFloatTy())
5508 RMW->setMetadata(LLVMContext::MD_atomic_ignore_denormal_mode, EmptyMD);
5509 }
5510
5511 if (AddrSpace == AMDGPUAS::FLAT_ADDRESS) {
5512 MDBuilder MDB(F->getContext());
5513 MDNode *RangeNotPrivate =
5516 RMW->setMetadata(LLVMContext::MD_noalias_addrspace, RangeNotPrivate);
5517 }
5518
5519 if (IsVolatile)
5520 RMW->setVolatile(true);
5521
5522 return Builder.CreateBitCast(RMW, RetTy);
5523}
5524
5525/// Helper to unwrap intrinsic call MetadataAsValue operands. Return as a
5526/// plain MDNode, as it's the verifier's job to check these are the correct
5527/// types later.
5528static MDNode *unwrapMAVOp(CallBase *CI, unsigned Op) {
5529 if (Op < CI->arg_size()) {
5530 if (MetadataAsValue *MAV =
5532 Metadata *MD = MAV->getMetadata();
5533 return dyn_cast_if_present<MDNode>(MD);
5534 }
5535 }
5536 return nullptr;
5537}
5538
5539/// Helper to unwrap Metadata MetadataAsValue operands, such as the Value field.
5540static Metadata *unwrapMAVMetadataOp(CallBase *CI, unsigned Op) {
5541 if (Op < CI->arg_size())
5543 return MAV->getMetadata();
5544 return nullptr;
5545}
5546
5547/// Convert debug intrinsic calls to non-instruction debug records.
5548/// \p Name - Final part of the intrinsic name, e.g. 'value' in llvm.dbg.value.
5549/// \p CI - The debug intrinsic call.
5551 DbgRecord *DR = nullptr;
5552 if (Name == "label") {
5554 } else if (Name == "assign") {
5557 unwrapMAVOp(CI, 1), unwrapMAVOp(CI, 2), unwrapMAVOp(CI, 3),
5558 unwrapMAVMetadataOp(CI, 4),
5559 /*The address is a Value ref, it will be stored as a Metadata */
5560 unwrapMAVOp(CI, 5));
5561 } else if (Name == "declare") {
5564 unwrapMAVOp(CI, 1), unwrapMAVOp(CI, 2), nullptr, nullptr, nullptr);
5565 } else if (Name == "addr") {
5566 // Upgrade dbg.addr to dbg.value with DW_OP_deref.
5567 MDNode *ExprNode = unwrapMAVOp(CI, 2);
5568 // Don't try to add something to the expression if it's not an expression.
5569 // Instead, allow the verifier to fail later.
5570 if (DIExpression *Expr = dyn_cast<DIExpression>(ExprNode)) {
5571 ExprNode = DIExpression::append(Expr, dwarf::DW_OP_deref);
5572 }
5575 unwrapMAVOp(CI, 1), ExprNode, nullptr, nullptr, nullptr);
5576 } else if (Name == "value") {
5577 // An old version of dbg.value had an extra offset argument.
5578 unsigned VarOp = 1;
5579 unsigned ExprOp = 2;
5580 if (CI->arg_size() == 4) {
5582 // Nonzero offset dbg.values get dropped without a replacement.
5583 if (!Offset || !Offset->isNullValue())
5584 return;
5585 VarOp = 2;
5586 ExprOp = 3;
5587 }
5590 unwrapMAVOp(CI, VarOp), unwrapMAVOp(CI, ExprOp), nullptr, nullptr,
5591 nullptr);
5592 }
5593 DR->setDebugLoc(CI->getDebugLoc());
5594 assert(DR && "Unhandled intrinsic kind in upgrade to DbgRecord");
5595 CI->getParent()->insertDbgRecordBefore(DR, CI->getIterator());
5596}
5597
5600 if (!Offset)
5601 reportFatalUsageError("Invalid llvm.vector.splice offset argument");
5602 int64_t OffsetVal = Offset->getSExtValue();
5603 return Builder.CreateIntrinsic(OffsetVal >= 0
5604 ? Intrinsic::vector_splice_left
5605 : Intrinsic::vector_splice_right,
5606 CI->getType(),
5607 {CI->getArgOperand(0), CI->getArgOperand(1),
5608 Builder.getInt32(std::abs(OffsetVal))});
5609}
5610
5612 Function *F, IRBuilder<> &Builder) {
5613 if (Name.starts_with("to.fp16")) {
5614 Value *Cast =
5615 Builder.CreateFPTrunc(CI->getArgOperand(0), Builder.getHalfTy());
5616 return Builder.CreateBitCast(Cast, CI->getType());
5617 }
5618
5619 if (Name.starts_with("from.fp16")) {
5620 Value *Cast =
5621 Builder.CreateBitCast(CI->getArgOperand(0), Builder.getHalfTy());
5622 return Builder.CreateFPExt(Cast, CI->getType());
5623 }
5624
5625 return nullptr;
5626}
5627
5629 Metadata *MD = cast<MetadataAsValue>(Op)->getMetadata();
5630 if (!MD || !isa<MDString>(MD))
5632 return StringSwitch<ICmpInst::Predicate>(cast<MDString>(MD)->getString())
5633 .Case("eq", ICmpInst::ICMP_EQ)
5634 .Case("ne", ICmpInst::ICMP_NE)
5635 .Case("ugt", ICmpInst::ICMP_UGT)
5636 .Case("uge", ICmpInst::ICMP_UGE)
5637 .Case("ult", ICmpInst::ICMP_ULT)
5638 .Case("ule", ICmpInst::ICMP_ULE)
5639 .Case("sgt", ICmpInst::ICMP_SGT)
5640 .Case("sge", ICmpInst::ICMP_SGE)
5641 .Case("slt", ICmpInst::ICMP_SLT)
5642 .Case("sle", ICmpInst::ICMP_SLE)
5644}
5645
5647 Metadata *MD = cast<MetadataAsValue>(Op)->getMetadata();
5648 if (!MD || !isa<MDString>(MD))
5650 return StringSwitch<FCmpInst::Predicate>(cast<MDString>(MD)->getString())
5651 .Case("oeq", FCmpInst::FCMP_OEQ)
5652 .Case("ogt", FCmpInst::FCMP_OGT)
5653 .Case("oge", FCmpInst::FCMP_OGE)
5654 .Case("olt", FCmpInst::FCMP_OLT)
5655 .Case("ole", FCmpInst::FCMP_OLE)
5656 .Case("one", FCmpInst::FCMP_ONE)
5657 .Case("ord", FCmpInst::FCMP_ORD)
5658 .Case("uno", FCmpInst::FCMP_UNO)
5659 .Case("ueq", FCmpInst::FCMP_UEQ)
5660 .Case("ugt", FCmpInst::FCMP_UGT)
5661 .Case("uge", FCmpInst::FCMP_UGE)
5662 .Case("ult", FCmpInst::FCMP_ULT)
5663 .Case("ule", FCmpInst::FCMP_ULE)
5664 .Case("une", FCmpInst::FCMP_UNE)
5666}
5667
5669 IRBuilder<> &Builder) {
5670 Value *Rep;
5671 unsigned Opcode = getFunctionalOpcodeForVP(Name);
5672 if (Opcode && Instruction::isUnaryOp(Opcode))
5673 Rep =
5674 Builder.CreateUnOp((Instruction::UnaryOps)Opcode, CI->getArgOperand(0));
5675 else if (Opcode && Instruction::isBinaryOp(Opcode))
5676 Rep = Builder.CreateBinOp((Instruction::BinaryOps)Opcode,
5677 CI->getArgOperand(0), CI->getArgOperand(1));
5678 else if (Opcode && Instruction::isCast(Opcode))
5679 Rep = Builder.CreateCast((Instruction::CastOps)Opcode, CI->getArgOperand(0),
5680 CI->getType());
5681 else if (Opcode == Instruction::ICmp)
5682 Rep = Builder.CreateICmp(getVPIntPredicateFromMD(CI->getArgOperand(2)),
5683 CI->getArgOperand(0), CI->getArgOperand(1));
5684 else if (Opcode == Instruction::FCmp)
5685 Rep = Builder.CreateFCmp(getVPFPPredicateFromMD(CI->getArgOperand(2)),
5686 CI->getArgOperand(0), CI->getArgOperand(1));
5687 else if (Opcode == Instruction::Select)
5688 Rep = Builder.CreateSelect(CI->getArgOperand(0), CI->getArgOperand(1),
5689 CI->getArgOperand(2));
5690 else if (auto IntrinsicID = getFunctionalIntrinsicIDForVP(Name)) {
5691 SmallVector<Value *, 2> Args(drop_end(CI->args(), 2));
5692 Rep = Builder.CreateIntrinsic(CI->getType(), IntrinsicID, Args, {});
5693 } else
5694 llvm_unreachable("Unexpected vp intrinsic");
5695 Rep->takeName(CI);
5696 return Rep;
5697}
5698
5700 IRBuilder<> &Builder) {
5701 Intrinsic::ID IID = NewFn->getIntrinsicID();
5702
5703 auto [FirstDefault, Defaults] = Intrinsic::getAllDefaultArgValues(IID);
5704 if (Defaults.empty())
5705 return false;
5706
5707 unsigned OldArgCount = CI->arg_size();
5708 unsigned NewArgCount = NewFn->arg_size();
5709
5710 if (OldArgCount < FirstDefault)
5711 return false;
5712
5713 // More arguments than the new intrinsic accepts, cannot upgrade.
5714 if (OldArgCount > NewArgCount)
5715 return false;
5716
5717 // The call already passes the defaulted arguments explicitly; only the
5718 // callee is still the old, shorter declaration, so retarget it.
5719 if (OldArgCount == NewArgCount) {
5720 if (CI->getFunctionType() != NewFn->getFunctionType())
5721 return false;
5722 CI->setCalledFunction(NewFn);
5723 return true;
5724 }
5725
5726 // OldArgCount < NewArgCount: Fill in each missing trailing default
5727 // argument from the table.
5728 SmallVector<Value *, 8> NewArgs(CI->args());
5729
5730 FunctionType *NewFT = NewFn->getFunctionType();
5731 for (unsigned Idx = OldArgCount; Idx < NewArgCount; ++Idx) {
5732 assert(Idx >= FirstDefault && Idx - FirstDefault < Defaults.size() &&
5733 "missing argument outside the default range");
5734 Type *ParamTy = NewFT->getParamType(Idx);
5735
5736 // Only integer types are supported (i1, i8, i16, i32, i64).
5737 if (!ParamTy->isIntegerTy())
5738 return false;
5739 NewArgs.push_back(ConstantInt::get(ParamTy, Defaults[Idx - FirstDefault]));
5740 }
5741
5742 // Preserve operand bundles by creating the call with them.
5744 CI->getOperandBundlesAsDefs(OpBundles);
5745 CallInst *NewCall = Builder.CreateCall(NewFn, NewArgs, OpBundles);
5746
5747 NewCall->takeName(CI);
5748 NewCall->setCallingConv(CI->getCallingConv());
5749 NewCall->copyMetadata(*CI);
5750 if (auto *OldCI = dyn_cast<CallInst>(CI))
5751 NewCall->setTailCallKind(OldCI->getTailCallKind());
5752
5753 CI->replaceAllUsesWith(NewCall);
5754 CI->eraseFromParent();
5755 return true;
5756}
5757
5758/// Upgrade a call to an old intrinsic. All argument and return casting must be
5759/// provided to seamlessly integrate with existing context.
5761 // Note dyn_cast to Function is not quite the same as getCalledFunction, which
5762 // checks the callee's function type matches. It's likely we need to handle
5763 // type changes here.
5765 if (!F)
5766 return;
5767
5768 LLVMContext &C = CI->getContext();
5769 IRBuilder<> Builder(CI->getParent(), CI->getIterator());
5770 if (isa<FPMathOperator>(CI))
5771 Builder.setFastMathFlags(CI->getFastMathFlags());
5772
5773 if (!NewFn) {
5774 // Get the Function's name.
5775 StringRef Name = F->getName();
5776 if (!Name.consume_front("llvm."))
5777 llvm_unreachable("intrinsic doesn't start with 'llvm.'");
5778
5779 bool IsX86 = Name.consume_front("x86.");
5780 bool IsNVVM = Name.consume_front("nvvm.");
5781 bool IsAArch64 = Name.consume_front("aarch64.");
5782 bool IsARM = Name.consume_front("arm.");
5783 bool IsAMDGCN = Name.consume_front("amdgcn.");
5784 bool IsDbg = Name.consume_front("dbg.");
5785 bool IsOldSplice =
5786 (Name.consume_front("experimental.vector.splice") ||
5787 Name.consume_front("vector.splice")) &&
5788 !(Name.starts_with(".left") || Name.starts_with(".right"));
5789 Value *Rep = nullptr;
5790
5791 if (!IsX86 && Name == "stackprotectorcheck") {
5792 Rep = nullptr;
5793 } else if (IsNVVM) {
5794 Rep = upgradeNVVMIntrinsicCall(Name, CI, F, Builder);
5795 } else if (IsX86) {
5796 Rep = upgradeX86IntrinsicCall(Name, CI, F, Builder);
5797 } else if (IsAArch64) {
5798 Rep = upgradeAArch64IntrinsicCall(Name, CI, F, Builder);
5799 } else if (IsARM) {
5800 Rep = upgradeARMIntrinsicCall(Name, CI, F, Builder);
5801 } else if (IsAMDGCN) {
5802 Rep = upgradeAMDGCNIntrinsicCall(Name, CI, F, Builder);
5803 } else if (IsDbg) {
5805 } else if (IsOldSplice) {
5806 Rep = upgradeVectorSplice(CI, Builder);
5807 } else if (Name.consume_front("convert.")) {
5808 Rep = upgradeConvertIntrinsicCall(Name, CI, F, Builder);
5809 } else if (Name == "lifetime.start.i64" || Name == "lifetime.end.i64") {
5810 // Delete calls to invalid @llvm.lifetime.{start,end}.i64 intrinsics.
5811 Rep = nullptr;
5812 } else if (shouldUpgradeVPIntrinsic(Name)) {
5813 Rep = upgradeVPIntrinsicCall(Name, CI, Builder);
5814 } else {
5815 llvm_unreachable("Unknown function for CallBase upgrade.");
5816 }
5817
5818 if (Rep)
5819 CI->replaceAllUsesWith(Rep);
5820 CI->eraseFromParent();
5821 return;
5822 }
5823
5824 const auto &DefaultCase = [&]() -> void {
5825 if (F == NewFn)
5826 return;
5827
5828 if (CI->getFunctionType() == NewFn->getFunctionType()) {
5829 // Handle generic mangling change.
5830 assert(
5831 (CI->getCalledFunction()->getName() != NewFn->getName()) &&
5832 "Unknown function for CallBase upgrade and isn't just a name change");
5833 CI->setCalledFunction(NewFn);
5834 return;
5835 }
5836
5837 // This must be an upgrade from a named to a literal struct.
5838 if (auto *OldST = dyn_cast<StructType>(CI->getType())) {
5839 assert(OldST != NewFn->getReturnType() &&
5840 "Return type must have changed");
5841 assert(OldST->getNumElements() ==
5842 cast<StructType>(NewFn->getReturnType())->getNumElements() &&
5843 "Must have same number of elements");
5844
5845 SmallVector<Value *> Args(CI->args());
5846 CallInst *NewCI = Builder.CreateCall(NewFn, Args);
5847 NewCI->setAttributes(CI->getAttributes());
5848 Value *Res = PoisonValue::get(OldST);
5849 for (unsigned Idx = 0; Idx < OldST->getNumElements(); ++Idx) {
5850 Value *Elem = Builder.CreateExtractValue(NewCI, Idx);
5851 Res = Builder.CreateInsertValue(Res, Elem, Idx);
5852 }
5853 CI->replaceAllUsesWith(Res);
5854 CI->eraseFromParent();
5855 return;
5856 }
5857
5858 // We're probably about to produce something invalid. Let the verifier catch
5859 // it instead of dying here.
5860 CI->setCalledOperand(
5862 return;
5863 };
5864 CallInst *NewCall = nullptr;
5865 switch (NewFn->getIntrinsicID()) {
5866 default: {
5867 if (upgradeIntrinsicCallWithDefaultArgs(CI, NewFn, Builder))
5868 return;
5869 DefaultCase();
5870 return;
5871 }
5872 case Intrinsic::arm_neon_vst1:
5873 case Intrinsic::arm_neon_vst2:
5874 case Intrinsic::arm_neon_vst3:
5875 case Intrinsic::arm_neon_vst4:
5876 case Intrinsic::arm_neon_vst2lane:
5877 case Intrinsic::arm_neon_vst3lane:
5878 case Intrinsic::arm_neon_vst4lane: {
5879 SmallVector<Value *, 4> Args(CI->args());
5880 NewCall = Builder.CreateCall(NewFn, Args);
5881 break;
5882 }
5883 case Intrinsic::aarch64_sve_bfmlalb_lane_v2:
5884 case Intrinsic::aarch64_sve_bfmlalt_lane_v2:
5885 case Intrinsic::aarch64_sve_bfdot_lane_v2: {
5886 LLVMContext &Ctx = F->getParent()->getContext();
5887 SmallVector<Value *, 4> Args(CI->args());
5888 Args[3] = ConstantInt::get(Type::getInt32Ty(Ctx),
5889 cast<ConstantInt>(Args[3])->getZExtValue());
5890 NewCall = Builder.CreateCall(NewFn, Args);
5891 break;
5892 }
5893 case Intrinsic::aarch64_sve_ld3_sret:
5894 case Intrinsic::aarch64_sve_ld4_sret:
5895 case Intrinsic::aarch64_sve_ld2_sret: {
5896 // Is this a trivial remangle of the name to support ptr address spaces?
5897 if (isa<StructType>(F->getReturnType())) {
5898 DefaultCase();
5899 return;
5900 }
5901
5902 StringRef Name = F->getName();
5903 Name = Name.substr(5);
5904 unsigned N = StringSwitch<unsigned>(Name)
5905 .StartsWith("aarch64.sve.ld2", 2)
5906 .StartsWith("aarch64.sve.ld3", 3)
5907 .StartsWith("aarch64.sve.ld4", 4)
5908 .Default(0);
5909 auto *RetTy = cast<ScalableVectorType>(F->getReturnType());
5910 unsigned MinElts = RetTy->getMinNumElements() / N;
5911 SmallVector<Value *, 2> Args(CI->args());
5912 Value *NewLdCall = Builder.CreateCall(NewFn, Args);
5913 Value *Ret = llvm::PoisonValue::get(RetTy);
5914 for (unsigned I = 0; I < N; I++) {
5915 Value *SRet = Builder.CreateExtractValue(NewLdCall, I);
5916 Ret = Builder.CreateInsertVector(RetTy, Ret, SRet, I * MinElts);
5917 }
5918 NewCall = dyn_cast<CallInst>(Ret);
5919 break;
5920 }
5921
5922 case Intrinsic::coro_end_async:
5923 case Intrinsic::coro_end: {
5924 SmallVector<Value *, 3> Args(CI->args());
5925 if (NewFn->getIntrinsicID() == Intrinsic::coro_end && Args.size() == 2)
5926 Args.push_back(ConstantTokenNone::get(CI->getContext()));
5927 NewCall = Builder.CreateCall(NewFn, Args);
5928
5929 if (!CI->getType()->isVoidTy()) {
5930 if (!CI->use_empty()) {
5932 CI->getModule(), Intrinsic::coro_is_in_ramp);
5933 Value *InRamp = Builder.CreateCall(IsInRamp);
5934 CI->replaceAllUsesWith(Builder.CreateNot(InRamp));
5935 }
5936 CI->eraseFromParent();
5937 return;
5938 }
5939
5940 break;
5941 }
5942
5943 case Intrinsic::vector_extract: {
5944 StringRef Name = F->getName();
5945 Name = Name.substr(5); // Strip llvm
5946 if (!Name.starts_with("aarch64.sve.tuple.get")) {
5947 DefaultCase();
5948 return;
5949 }
5950 auto *RetTy = cast<ScalableVectorType>(F->getReturnType());
5951 unsigned MinElts = RetTy->getMinNumElements();
5952 uint64_t I = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
5953 Value *NewIdx = ConstantInt::get(Type::getInt64Ty(C), I * MinElts);
5954 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0), NewIdx});
5955 break;
5956 }
5957
5958 case Intrinsic::vector_insert: {
5959 StringRef Name = F->getName();
5960 Name = Name.substr(5);
5961 if (!Name.starts_with("aarch64.sve.tuple")) {
5962 DefaultCase();
5963 return;
5964 }
5965 if (Name.starts_with("aarch64.sve.tuple.set")) {
5966 uint64_t I = cast<ConstantInt>(CI->getArgOperand(1))->getZExtValue();
5967 auto *Ty = cast<ScalableVectorType>(CI->getArgOperand(2)->getType());
5968 Value *NewIdx =
5969 ConstantInt::get(Type::getInt64Ty(C), I * Ty->getMinNumElements());
5970 NewCall = Builder.CreateCall(
5971 NewFn, {CI->getArgOperand(0), CI->getArgOperand(2), NewIdx});
5972 break;
5973 }
5974 if (Name.starts_with("aarch64.sve.tuple.create")) {
5975 unsigned N = StringSwitch<unsigned>(Name)
5976 .StartsWith("aarch64.sve.tuple.create2", 2)
5977 .StartsWith("aarch64.sve.tuple.create3", 3)
5978 .StartsWith("aarch64.sve.tuple.create4", 4)
5979 .Default(0);
5980 assert(N > 1 && "Create is expected to be between 2-4");
5981 auto *RetTy = cast<ScalableVectorType>(F->getReturnType());
5982 Value *Ret = llvm::PoisonValue::get(RetTy);
5983 unsigned MinElts = RetTy->getMinNumElements() / N;
5984 for (unsigned I = 0; I < N; I++) {
5985 Value *V = CI->getArgOperand(I);
5986 Ret = Builder.CreateInsertVector(RetTy, Ret, V, I * MinElts);
5987 }
5988 NewCall = dyn_cast<CallInst>(Ret);
5989 }
5990 break;
5991 }
5992
5993 case Intrinsic::arm_neon_bfdot:
5994 case Intrinsic::arm_neon_bfmmla:
5995 case Intrinsic::arm_neon_bfmlalb:
5996 case Intrinsic::arm_neon_bfmlalt:
5997 case Intrinsic::aarch64_neon_bfdot:
5998 case Intrinsic::aarch64_neon_bfmmla:
5999 case Intrinsic::aarch64_neon_bfmlalb:
6000 case Intrinsic::aarch64_neon_bfmlalt: {
6002 assert(CI->arg_size() == 3 &&
6003 "Mismatch between function args and call args");
6004 size_t OperandWidth =
6006 assert((OperandWidth == 64 || OperandWidth == 128) &&
6007 "Unexpected operand width");
6008 Type *NewTy = FixedVectorType::get(Type::getBFloatTy(C), OperandWidth / 16);
6009 auto Iter = CI->args().begin();
6010 Args.push_back(*Iter++);
6011 Args.push_back(Builder.CreateBitCast(*Iter++, NewTy));
6012 Args.push_back(Builder.CreateBitCast(*Iter++, NewTy));
6013 NewCall = Builder.CreateCall(NewFn, Args);
6014 break;
6015 }
6016
6017 case Intrinsic::bitreverse:
6018 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)});
6019 break;
6020
6021 case Intrinsic::ctlz:
6022 case Intrinsic::cttz: {
6023 if (CI->arg_size() != 1) {
6024 DefaultCase();
6025 return;
6026 }
6027
6028 NewCall =
6029 Builder.CreateCall(NewFn, {CI->getArgOperand(0), Builder.getFalse()});
6030 break;
6031 }
6032
6033 case Intrinsic::objectsize: {
6034 Value *NullIsUnknownSize =
6035 CI->arg_size() == 2 ? Builder.getFalse() : CI->getArgOperand(2);
6036 Value *Dynamic =
6037 CI->arg_size() < 4 ? Builder.getFalse() : CI->getArgOperand(3);
6038 NewCall = Builder.CreateCall(
6039 NewFn, {CI->getArgOperand(0), CI->getArgOperand(1), NullIsUnknownSize, Dynamic});
6040 break;
6041 }
6042
6043 case Intrinsic::ctpop:
6044 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(0)});
6045 break;
6046 case Intrinsic::dbg_value: {
6047 StringRef Name = F->getName();
6048 Name = Name.substr(5); // Strip llvm.
6049 // Upgrade `dbg.addr` to `dbg.value` with `DW_OP_deref`.
6050 if (Name.starts_with("dbg.addr")) {
6052 cast<MetadataAsValue>(CI->getArgOperand(2))->getMetadata());
6053 Expr = DIExpression::append(Expr, dwarf::DW_OP_deref);
6054 NewCall =
6055 Builder.CreateCall(NewFn, {CI->getArgOperand(0), CI->getArgOperand(1),
6056 MetadataAsValue::get(C, Expr)});
6057 break;
6058 }
6059
6060 // Upgrade from the old version that had an extra offset argument.
6061 assert(CI->arg_size() == 4);
6062 // Drop nonzero offsets instead of attempting to upgrade them.
6064 if (Offset->isNullValue()) {
6065 NewCall = Builder.CreateCall(
6066 NewFn,
6067 {CI->getArgOperand(0), CI->getArgOperand(2), CI->getArgOperand(3)});
6068 break;
6069 }
6070 CI->eraseFromParent();
6071 return;
6072 }
6073
6074 case Intrinsic::ptr_annotation:
6075 // Upgrade from versions that lacked the annotation attribute argument.
6076 if (CI->arg_size() != 4) {
6077 DefaultCase();
6078 return;
6079 }
6080
6081 // Create a new call with an added null annotation attribute argument.
6082 NewCall = Builder.CreateCall(
6083 NewFn,
6084 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2),
6085 CI->getArgOperand(3), ConstantPointerNull::get(Builder.getPtrTy())});
6086 NewCall->takeName(CI);
6087 CI->replaceAllUsesWith(NewCall);
6088 CI->eraseFromParent();
6089 return;
6090
6091 case Intrinsic::var_annotation:
6092 // Upgrade from versions that lacked the annotation attribute argument.
6093 if (CI->arg_size() != 4) {
6094 DefaultCase();
6095 return;
6096 }
6097 // Create a new call with an added null annotation attribute argument.
6098 NewCall = Builder.CreateCall(
6099 NewFn,
6100 {CI->getArgOperand(0), CI->getArgOperand(1), CI->getArgOperand(2),
6101 CI->getArgOperand(3), ConstantPointerNull::get(Builder.getPtrTy())});
6102 NewCall->takeName(CI);
6103 CI->replaceAllUsesWith(NewCall);
6104 CI->eraseFromParent();
6105 return;
6106
6107 case Intrinsic::riscv_aes32dsi:
6108 case Intrinsic::riscv_aes32dsmi:
6109 case Intrinsic::riscv_aes32esi:
6110 case Intrinsic::riscv_aes32esmi:
6111 case Intrinsic::riscv_sm4ks:
6112 case Intrinsic::riscv_sm4ed: {
6113 // The last argument to these intrinsics used to be i8 and changed to i32.
6114 // The type overload for sm4ks and sm4ed was removed.
6115 Value *Arg2 = CI->getArgOperand(2);
6116 if (Arg2->getType()->isIntegerTy(32) && !CI->getType()->isIntegerTy(64))
6117 return;
6118
6119 Value *Arg0 = CI->getArgOperand(0);
6120 Value *Arg1 = CI->getArgOperand(1);
6121 if (CI->getType()->isIntegerTy(64)) {
6122 Arg0 = Builder.CreateTrunc(Arg0, Builder.getInt32Ty());
6123 Arg1 = Builder.CreateTrunc(Arg1, Builder.getInt32Ty());
6124 }
6125
6126 Arg2 = ConstantInt::get(Type::getInt32Ty(C),
6127 cast<ConstantInt>(Arg2)->getZExtValue());
6128
6129 NewCall = Builder.CreateCall(NewFn, {Arg0, Arg1, Arg2});
6130 Value *Res = NewCall;
6131 if (Res->getType() != CI->getType())
6132 Res = Builder.CreateIntCast(NewCall, CI->getType(), /*isSigned*/ true);
6133 NewCall->takeName(CI);
6134 CI->replaceAllUsesWith(Res);
6135 CI->eraseFromParent();
6136 return;
6137 }
6138 case Intrinsic::nvvm_mapa_shared_cluster: {
6139 // Create a new call with the correct address space.
6140 NewCall =
6141 Builder.CreateCall(NewFn, {CI->getArgOperand(0), CI->getArgOperand(1)});
6142 Value *Res = NewCall;
6143 Res = Builder.CreateAddrSpaceCast(
6144 Res, Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED));
6145 NewCall->takeName(CI);
6146 CI->replaceAllUsesWith(Res);
6147 CI->eraseFromParent();
6148 return;
6149 }
6150 case Intrinsic::nvvm_cp_async_bulk_global_to_shared_cluster: {
6151 SmallVector<Value *, 4> Args(CI->args());
6152 unsigned AS = Args[0]->getType()->getPointerAddressSpace();
6154 Args[0] = Builder.CreateAddrSpaceCast(
6155 Args[0], Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6156
6157 // Append the missing trailing flag_valid_pattern (0 = disabled).
6158 Args.push_back(Builder.getInt32(0));
6159
6160 NewCall = Builder.CreateCall(NewFn, Args);
6161 NewCall->takeName(CI);
6162 CI->replaceAllUsesWith(NewCall);
6163 CI->eraseFromParent();
6164 return;
6165 }
6166 case Intrinsic::nvvm_cp_async_bulk_global_to_shared_cta: {
6167 // (dst, mbar, src, size, ch, flag_ch)
6168 // -> (dst, mbar, src, size, i32 0, i32 0, ch, flag_ch, i1 false,
6169 // i32 0 /* flag_valid_pattern=disabled */)
6171 for (unsigned I = 0; I < 4; ++I)
6172 Args.push_back(CI->getArgOperand(I));
6173 Args.push_back(Builder.getInt32(0)); // ignore_bytes_left
6174 Args.push_back(Builder.getInt32(0)); // ignore_bytes_right
6175 Args.push_back(CI->getArgOperand(4)); // cache_hint
6176 Args.push_back(CI->getArgOperand(5)); // flag_ch
6177 Args.push_back(Builder.getInt1(false)); // flag_oob
6178 Args.push_back(Builder.getInt32(0)); // flag_valid_pattern
6179
6180 NewCall = Builder.CreateCall(NewFn, Args);
6181 NewCall->takeName(CI);
6182 CI->replaceAllUsesWith(NewCall);
6183 CI->eraseFromParent();
6184 return;
6185 }
6186 case Intrinsic::nvvm_cp_async_bulk_shared_cta_to_cluster: {
6187 // Create a new call with the correct address space.
6188 SmallVector<Value *, 4> Args(CI->args());
6189 Args[0] = Builder.CreateAddrSpaceCast(
6190 Args[0], Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6191
6192 NewCall = Builder.CreateCall(NewFn, Args);
6193 NewCall->takeName(CI);
6194 CI->replaceAllUsesWith(NewCall);
6195 CI->eraseFromParent();
6196 return;
6197 }
6198 // clang-format off
6199#define G2S_CLUSTER_CASE(ID_SUFFIX, NAME) \
6200 case Intrinsic::nvvm_cp_async_bulk_tensor_g2s_##ID_SUFFIX:
6202#undef G2S_CLUSTER_CASE
6203 {
6204 SmallVector<Value *, 16> Args(CI->args());
6205 unsigned AS = CI->getArgOperand(0)->getType()->getPointerAddressSpace();
6207 Args[0] = Builder.CreateAddrSpaceCast(
6208 Args[0], Builder.getPtrTy(NVPTXAS::ADDRESS_SPACE_SHARED_CLUSTER));
6209
6210 // Append the missing trailing arguments with default values (cta_group,
6211 // flag_valid_pattern).
6212 while (Args.size() < NewFn->getFunctionType()->getNumParams())
6213 Args.push_back(Builder.getInt32(0));
6214
6215 NewCall = Builder.CreateCall(NewFn, Args);
6216 NewCall->takeName(CI);
6217 CI->replaceAllUsesWith(NewCall);
6218 CI->eraseFromParent();
6219 return;
6220 }
6221
6222#define G2S_CTA_CASE(ID_SUFFIX, NAME) \
6223 case Intrinsic::nvvm_cp_async_bulk_tensor_g2s_cta_##ID_SUFFIX:
6225#undef G2S_CTA_CASE
6226 {
6227 SmallVector<Value *, 16> Args(CI->args());
6228 // Append the missing trailing flag_valid_pattern argument with default
6229 // value 0.
6230 assert(Args.size() + 1 == NewFn->getFunctionType()->getNumParams() &&
6231 "expected only the trailing flag_valid_pattern to be missing");
6232 Args.push_back(Builder.getInt32(0));
6233
6234 NewCall = Builder.CreateCall(NewFn, Args);
6235 NewCall->takeName(CI);
6236 CI->replaceAllUsesWith(NewCall);
6237 CI->eraseFromParent();
6238 return;
6239 }
6240#undef NVVM_TMA_G2S_MODES
6241 // clang-format on
6242
6243 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_1d:
6244 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_2d:
6245 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_3d:
6246 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_4d:
6247 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_tile_5d:
6248 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_3d:
6249 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_4d:
6250 case Intrinsic::nvvm_cp_async_bulk_tensor_reduce_im2col_5d: {
6251 StringRef Name = F->getName();
6252 Name.consume_front("llvm.nvvm.cp.async.bulk.tensor.reduce.");
6253 auto RedOp = getNVPTXTMAReductionOp(Name.split('.').first);
6254
6255 SmallVector<Value *, 16> Args(CI->args());
6256 Args.insert(Args.end() - 1, Builder.getInt32(*RedOp));
6257 NewCall = Builder.CreateCall(NewFn, Args);
6258 break;
6259 }
6260 case Intrinsic::nvvm_tcgen05_alloc_cg1:
6261 case Intrinsic::nvvm_tcgen05_alloc_cg2:
6262 case Intrinsic::nvvm_tcgen05_dealloc_cg1:
6263 case Intrinsic::nvvm_tcgen05_dealloc_cg2:
6264 NewCall =
6265 Builder.CreateCall(NewFn, {CI->getArgOperand(0), CI->getArgOperand(1),
6266 Builder.getFalse()});
6267 break;
6268 case Intrinsic::nvvm_mbarrier_init: {
6269 SmallVector<Value *, 3> Args(CI->args());
6270 // The .shared variant folded into the overloaded form without gaining an
6271 // operand, so only the pre-layout two-argument form needs one appended.
6272 if (Args.size() == 2)
6273 Args.push_back(Builder.getInt32(0)); // layout = default(0)
6274 NewCall = Builder.CreateCall(NewFn, Args);
6275 break;
6276 }
6277 case Intrinsic::riscv_sha256sig0:
6278 case Intrinsic::riscv_sha256sig1:
6279 case Intrinsic::riscv_sha256sum0:
6280 case Intrinsic::riscv_sha256sum1:
6281 case Intrinsic::riscv_sm3p0:
6282 case Intrinsic::riscv_sm3p1: {
6283 // The last argument to these intrinsics used to be i8 and changed to i32.
6284 // The type overload for sm4ks and sm4ed was removed.
6285 if (!CI->getType()->isIntegerTy(64))
6286 return;
6287
6288 Value *Arg =
6289 Builder.CreateTrunc(CI->getArgOperand(0), Builder.getInt32Ty());
6290
6291 NewCall = Builder.CreateCall(NewFn, Arg);
6292 Value *Res =
6293 Builder.CreateIntCast(NewCall, CI->getType(), /*isSigned*/ true);
6294 NewCall->takeName(CI);
6295 CI->replaceAllUsesWith(Res);
6296 CI->eraseFromParent();
6297 return;
6298 }
6299
6300 case Intrinsic::x86_xop_vfrcz_ss:
6301 case Intrinsic::x86_xop_vfrcz_sd:
6302 NewCall = Builder.CreateCall(NewFn, {CI->getArgOperand(1)});
6303 break;
6304
6305 case Intrinsic::x86_xop_vpermil2pd:
6306 case Intrinsic::x86_xop_vpermil2ps:
6307 case Intrinsic::x86_xop_vpermil2pd_256:
6308 case Intrinsic::x86_xop_vpermil2ps_256: {
6309 SmallVector<Value *, 4> Args(CI->args());
6310 VectorType *FltIdxTy = cast<VectorType>(Args[2]->getType());
6311 VectorType *IntIdxTy = VectorType::getInteger(FltIdxTy);
6312 Args[2] = Builder.CreateBitCast(Args[2], IntIdxTy);
6313 NewCall = Builder.CreateCall(NewFn, Args);
6314 break;
6315 }
6316
6317 case Intrinsic::x86_sse41_ptestc:
6318 case Intrinsic::x86_sse41_ptestz:
6319 case Intrinsic::x86_sse41_ptestnzc: {
6320 // The arguments for these intrinsics used to be v4f32, and changed
6321 // to v2i64. This is purely a nop, since those are bitwise intrinsics.
6322 // So, the only thing required is a bitcast for both arguments.
6323 // First, check the arguments have the old type.
6324 Value *Arg0 = CI->getArgOperand(0);
6325 if (Arg0->getType() != FixedVectorType::get(Type::getFloatTy(C), 4))
6326 return;
6327
6328 // Old intrinsic, add bitcasts
6329 Value *Arg1 = CI->getArgOperand(1);
6330
6331 auto *NewVecTy = FixedVectorType::get(Type::getInt64Ty(C), 2);
6332
6333 Value *BC0 = Builder.CreateBitCast(Arg0, NewVecTy, "cast");
6334 Value *BC1 = Builder.CreateBitCast(Arg1, NewVecTy, "cast");
6335
6336 NewCall = Builder.CreateCall(NewFn, {BC0, BC1});
6337 break;
6338 }
6339
6340 case Intrinsic::x86_rdtscp: {
6341 // This used to take 1 arguments. If we have no arguments, it is already
6342 // upgraded.
6343 if (CI->getNumOperands() == 0)
6344 return;
6345
6346 NewCall = Builder.CreateCall(NewFn);
6347 // Extract the second result and store it.
6348 Value *Data = Builder.CreateExtractValue(NewCall, 1);
6349 Builder.CreateAlignedStore(Data, CI->getArgOperand(0), Align(1));
6350 // Replace the original call result with the first result of the new call.
6351 Value *TSC = Builder.CreateExtractValue(NewCall, 0);
6352
6353 NewCall->takeName(CI);
6354 CI->replaceAllUsesWith(TSC);
6355 CI->eraseFromParent();
6356 return;
6357 }
6358
6359 case Intrinsic::x86_sse41_insertps:
6360 case Intrinsic::x86_sse41_dppd:
6361 case Intrinsic::x86_sse41_dpps:
6362 case Intrinsic::x86_sse41_mpsadbw:
6363 case Intrinsic::x86_avx_dp_ps_256:
6364 case Intrinsic::x86_avx2_mpsadbw: {
6365 // Need to truncate the last argument from i32 to i8 -- this argument models
6366 // an inherently 8-bit immediate operand to these x86 instructions.
6367 SmallVector<Value *, 4> Args(CI->args());
6368
6369 // Replace the last argument with a trunc.
6370 Args.back() = Builder.CreateTrunc(Args.back(), Type::getInt8Ty(C), "trunc");
6371 NewCall = Builder.CreateCall(NewFn, Args);
6372 break;
6373 }
6374
6375 case Intrinsic::x86_avx512_mask_cmp_pd_128:
6376 case Intrinsic::x86_avx512_mask_cmp_pd_256:
6377 case Intrinsic::x86_avx512_mask_cmp_pd_512:
6378 case Intrinsic::x86_avx512_mask_cmp_ps_128:
6379 case Intrinsic::x86_avx512_mask_cmp_ps_256:
6380 case Intrinsic::x86_avx512_mask_cmp_ps_512: {
6381 SmallVector<Value *, 4> Args(CI->args());
6382 unsigned NumElts =
6383 cast<FixedVectorType>(Args[0]->getType())->getNumElements();
6384 Args[3] = getX86MaskVec(Builder, Args[3], NumElts);
6385
6386 NewCall = Builder.CreateCall(NewFn, Args);
6387 Value *Res = applyX86MaskOn1BitsVec(Builder, NewCall, nullptr);
6388
6389 NewCall->takeName(CI);
6390 CI->replaceAllUsesWith(Res);
6391 CI->eraseFromParent();
6392 return;
6393 }
6394
6395 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_128:
6396 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_256:
6397 case Intrinsic::x86_avx512bf16_cvtne2ps2bf16_512:
6398 case Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128:
6399 case Intrinsic::x86_avx512bf16_cvtneps2bf16_256:
6400 case Intrinsic::x86_avx512bf16_cvtneps2bf16_512: {
6401 SmallVector<Value *, 4> Args(CI->args());
6402 unsigned NumElts = cast<FixedVectorType>(CI->getType())->getNumElements();
6403 if (NewFn->getIntrinsicID() ==
6404 Intrinsic::x86_avx512bf16_mask_cvtneps2bf16_128)
6405 Args[1] = Builder.CreateBitCast(
6406 Args[1], FixedVectorType::get(Builder.getBFloatTy(), NumElts));
6407
6408 NewCall = Builder.CreateCall(NewFn, Args);
6409 Value *Res = Builder.CreateBitCast(
6410 NewCall, FixedVectorType::get(Builder.getInt16Ty(), NumElts));
6411
6412 NewCall->takeName(CI);
6413 CI->replaceAllUsesWith(Res);
6414 CI->eraseFromParent();
6415 return;
6416 }
6417 case Intrinsic::x86_avx512bf16_dpbf16ps_128:
6418 case Intrinsic::x86_avx512bf16_dpbf16ps_256:
6419 case Intrinsic::x86_avx512bf16_dpbf16ps_512:{
6420 SmallVector<Value *, 4> Args(CI->args());
6421 unsigned NumElts =
6422 cast<FixedVectorType>(CI->getType())->getNumElements() * 2;
6423 Args[1] = Builder.CreateBitCast(
6424 Args[1], FixedVectorType::get(Builder.getBFloatTy(), NumElts));
6425 Args[2] = Builder.CreateBitCast(
6426 Args[2], FixedVectorType::get(Builder.getBFloatTy(), NumElts));
6427
6428 NewCall = Builder.CreateCall(NewFn, Args);
6429 break;
6430 }
6431
6432 case Intrinsic::thread_pointer: {
6433 NewCall = Builder.CreateCall(NewFn, {});
6434 break;
6435 }
6436
6437 case Intrinsic::memcpy:
6438 case Intrinsic::memmove:
6439 case Intrinsic::memset: {
6440 // We have to make sure that the call signature is what we're expecting.
6441 // We only want to change the old signatures by removing the alignment arg:
6442 // @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i32, i1)
6443 // -> @llvm.mem[cpy|move]...(i8*, i8*, i[32|i64], i1)
6444 // @llvm.memset...(i8*, i8, i[32|64], i32, i1)
6445 // -> @llvm.memset...(i8*, i8, i[32|64], i1)
6446 // Note: i8*'s in the above can be any pointer type
6447 if (CI->arg_size() != 5) {
6448 DefaultCase();
6449 return;
6450 }
6451 // Remove alignment argument (3), and add alignment attributes to the
6452 // dest/src pointers.
6453 Value *Args[4] = {CI->getArgOperand(0), CI->getArgOperand(1),
6454 CI->getArgOperand(2), CI->getArgOperand(4)};
6455 NewCall = Builder.CreateCall(NewFn, Args);
6456 AttributeList OldAttrs = CI->getAttributes();
6457 AttributeList NewAttrs = AttributeList::get(
6458 C, OldAttrs.getFnAttrs(), OldAttrs.getRetAttrs(),
6459 {OldAttrs.getParamAttrs(0), OldAttrs.getParamAttrs(1),
6460 OldAttrs.getParamAttrs(2), OldAttrs.getParamAttrs(4)});
6461 NewCall->setAttributes(NewAttrs);
6462 auto *MemCI = cast<MemIntrinsic>(NewCall);
6463 // All mem intrinsics support dest alignment.
6465 MemCI->setDestAlignment(Align->getMaybeAlignValue());
6466 // Memcpy/Memmove also support source alignment.
6467 if (auto *MTI = dyn_cast<MemTransferInst>(MemCI))
6468 MTI->setSourceAlignment(Align->getMaybeAlignValue());
6469 break;
6470 }
6471
6472 case Intrinsic::masked_load:
6473 case Intrinsic::masked_gather:
6474 case Intrinsic::masked_store:
6475 case Intrinsic::masked_scatter: {
6476 if (CI->arg_size() != 4) {
6477 DefaultCase();
6478 return;
6479 }
6480
6481 auto GetMaybeAlign = [](Value *Op) {
6482 if (auto *CI = dyn_cast<ConstantInt>(Op)) {
6483 uint64_t Val = CI->getZExtValue();
6484 if (Val == 0)
6485 return MaybeAlign();
6486 if (isPowerOf2_64(Val))
6487 return MaybeAlign(Val);
6488 }
6489 reportFatalUsageError("Invalid alignment argument");
6490 };
6491 auto GetAlign = [&](Value *Op) {
6492 MaybeAlign Align = GetMaybeAlign(Op);
6493 if (Align)
6494 return *Align;
6495 reportFatalUsageError("Invalid zero alignment argument");
6496 };
6497
6498 const DataLayout &DL = CI->getDataLayout();
6499 switch (NewFn->getIntrinsicID()) {
6500 case Intrinsic::masked_load:
6501 NewCall = Builder.CreateMaskedLoad(
6502 CI->getType(), CI->getArgOperand(0), GetAlign(CI->getArgOperand(1)),
6503 CI->getArgOperand(2), CI->getArgOperand(3));
6504 break;
6505 case Intrinsic::masked_gather:
6506 NewCall = Builder.CreateMaskedGather(
6507 CI->getType(), CI->getArgOperand(0),
6508 DL.getValueOrABITypeAlignment(GetMaybeAlign(CI->getArgOperand(1)),
6509 CI->getType()->getScalarType()),
6510 CI->getArgOperand(2), CI->getArgOperand(3));
6511 break;
6512 case Intrinsic::masked_store:
6513 NewCall = Builder.CreateMaskedStore(
6514 CI->getArgOperand(0), CI->getArgOperand(1),
6515 GetAlign(CI->getArgOperand(2)), CI->getArgOperand(3));
6516 break;
6517 case Intrinsic::masked_scatter:
6518 NewCall = Builder.CreateMaskedScatter(
6519 CI->getArgOperand(0), CI->getArgOperand(1),
6520 DL.getValueOrABITypeAlignment(
6521 GetMaybeAlign(CI->getArgOperand(2)),
6522 CI->getArgOperand(0)->getType()->getScalarType()),
6523 CI->getArgOperand(3));
6524 break;
6525 default:
6526 llvm_unreachable("Unexpected intrinsic ID");
6527 }
6528 // Previous metadata is still valid.
6529 NewCall->copyMetadata(*CI);
6530 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
6531 break;
6532 }
6533
6534 case Intrinsic::lifetime_start:
6535 case Intrinsic::lifetime_end: {
6536 if (CI->arg_size() != 2) {
6537 DefaultCase();
6538 return;
6539 }
6540
6541 Value *Ptr = CI->getArgOperand(1);
6542 // Try to strip pointer casts, such that the lifetime works on an alloca.
6543 Ptr = Ptr->stripPointerCasts();
6544 if (isa<AllocaInst>(Ptr)) {
6545 // Don't use NewFn, as we might have looked through an addrspacecast.
6546 if (NewFn->getIntrinsicID() == Intrinsic::lifetime_start)
6547 NewCall = Builder.CreateLifetimeStart(Ptr);
6548 else
6549 NewCall = Builder.CreateLifetimeEnd(Ptr);
6550 break;
6551 }
6552
6553 // Otherwise remove the lifetime marker.
6554 CI->eraseFromParent();
6555 return;
6556 }
6557
6558 case Intrinsic::x86_avx512_vpdpbusd_128:
6559 case Intrinsic::x86_avx512_vpdpbusd_256:
6560 case Intrinsic::x86_avx512_vpdpbusd_512:
6561 case Intrinsic::x86_avx512_vpdpbusds_128:
6562 case Intrinsic::x86_avx512_vpdpbusds_256:
6563 case Intrinsic::x86_avx512_vpdpbusds_512:
6564 case Intrinsic::x86_avx2_vpdpbssd_128:
6565 case Intrinsic::x86_avx2_vpdpbssd_256:
6566 case Intrinsic::x86_avx10_vpdpbssd_512:
6567 case Intrinsic::x86_avx2_vpdpbssds_128:
6568 case Intrinsic::x86_avx2_vpdpbssds_256:
6569 case Intrinsic::x86_avx10_vpdpbssds_512:
6570 case Intrinsic::x86_avx2_vpdpbsud_128:
6571 case Intrinsic::x86_avx2_vpdpbsud_256:
6572 case Intrinsic::x86_avx10_vpdpbsud_512:
6573 case Intrinsic::x86_avx2_vpdpbsuds_128:
6574 case Intrinsic::x86_avx2_vpdpbsuds_256:
6575 case Intrinsic::x86_avx10_vpdpbsuds_512:
6576 case Intrinsic::x86_avx2_vpdpbuud_128:
6577 case Intrinsic::x86_avx2_vpdpbuud_256:
6578 case Intrinsic::x86_avx10_vpdpbuud_512:
6579 case Intrinsic::x86_avx2_vpdpbuuds_128:
6580 case Intrinsic::x86_avx2_vpdpbuuds_256:
6581 case Intrinsic::x86_avx10_vpdpbuuds_512: {
6582 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / 8;
6583 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
6584 CI->getArgOperand(2)};
6585 Type *NewArgType = VectorType::get(Builder.getInt8Ty(), NumElts, false);
6586 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
6587 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
6588
6589 NewCall = Builder.CreateCall(NewFn, Args);
6590 break;
6591 }
6592 case Intrinsic::x86_avx512_vpdpwssd_128:
6593 case Intrinsic::x86_avx512_vpdpwssd_256:
6594 case Intrinsic::x86_avx512_vpdpwssd_512:
6595 case Intrinsic::x86_avx512_vpdpwssds_128:
6596 case Intrinsic::x86_avx512_vpdpwssds_256:
6597 case Intrinsic::x86_avx512_vpdpwssds_512:
6598 case Intrinsic::x86_avx2_vpdpwsud_128:
6599 case Intrinsic::x86_avx2_vpdpwsud_256:
6600 case Intrinsic::x86_avx10_vpdpwsud_512:
6601 case Intrinsic::x86_avx2_vpdpwsuds_128:
6602 case Intrinsic::x86_avx2_vpdpwsuds_256:
6603 case Intrinsic::x86_avx10_vpdpwsuds_512:
6604 case Intrinsic::x86_avx2_vpdpwusd_128:
6605 case Intrinsic::x86_avx2_vpdpwusd_256:
6606 case Intrinsic::x86_avx10_vpdpwusd_512:
6607 case Intrinsic::x86_avx2_vpdpwusds_128:
6608 case Intrinsic::x86_avx2_vpdpwusds_256:
6609 case Intrinsic::x86_avx10_vpdpwusds_512:
6610 case Intrinsic::x86_avx2_vpdpwuud_128:
6611 case Intrinsic::x86_avx2_vpdpwuud_256:
6612 case Intrinsic::x86_avx10_vpdpwuud_512:
6613 case Intrinsic::x86_avx2_vpdpwuuds_128:
6614 case Intrinsic::x86_avx2_vpdpwuuds_256:
6615 case Intrinsic::x86_avx10_vpdpwuuds_512:
6616 unsigned NumElts = CI->getType()->getPrimitiveSizeInBits() / 16;
6617 Value *Args[] = {CI->getArgOperand(0), CI->getArgOperand(1),
6618 CI->getArgOperand(2)};
6619 Type *NewArgType = VectorType::get(Builder.getInt16Ty(), NumElts, false);
6620 Args[1] = Builder.CreateBitCast(Args[1], NewArgType);
6621 Args[2] = Builder.CreateBitCast(Args[2], NewArgType);
6622
6623 NewCall = Builder.CreateCall(NewFn, Args);
6624 break;
6625 }
6626 assert(NewCall && "Should have either set this variable or returned through "
6627 "the default case");
6628 NewCall->takeName(CI);
6629 CI->replaceAllUsesWith(NewCall);
6630 CI->eraseFromParent();
6631}
6632
6634 assert(F && "Illegal attempt to upgrade a non-existent intrinsic.");
6635
6636 // Check if this function should be upgraded and get the replacement function
6637 // if there is one.
6638 Function *NewFn;
6639 if (UpgradeIntrinsicFunction(F, NewFn)) {
6640 // Replace all users of the old function with the new function or new
6641 // instructions. This is not a range loop because the call is deleted.
6642 for (User *U : make_early_inc_range(F->users()))
6643 if (CallBase *CB = dyn_cast<CallBase>(U))
6644 UpgradeIntrinsicCall(CB, NewFn);
6645
6646 // Remove old function, no longer used, from the module.
6647 if (F != NewFn)
6648 F->eraseFromParent();
6649 }
6650}
6651
6653 const unsigned NumOperands = MD.getNumOperands();
6654 if (NumOperands == 0)
6655 return &MD; // Invalid, punt to a verifier error.
6656
6657 // Check if the tag uses struct-path aware TBAA format.
6658 if (isa<MDNode>(MD.getOperand(0)) && NumOperands >= 3)
6659 return &MD;
6660
6661 auto &Context = MD.getContext();
6662 if (NumOperands == 3) {
6663 Metadata *Elts[] = {MD.getOperand(0), MD.getOperand(1)};
6664 MDNode *ScalarType = MDNode::get(Context, Elts);
6665 // Create a MDNode <ScalarType, ScalarType, offset 0, const>
6666 Metadata *Elts2[] = {ScalarType, ScalarType,
6669 MD.getOperand(2)};
6670 return MDNode::get(Context, Elts2);
6671 }
6672 // Create a MDNode <MD, MD, offset 0>
6674 Type::getInt64Ty(Context)))};
6675 return MDNode::get(Context, Elts);
6676}
6677
6679 // !tbaa.struct is a list of (offset, size, tag) triples. Upgrade any
6680 // old-style scalar field tag to struct-path form via UpgradeTBAANode.
6681 unsigned NumOperands = MD.getNumOperands();
6682 if (NumOperands == 0 || NumOperands % 3 != 0)
6683 return &MD; // Malformed; leave it for the verifier to reject.
6684
6686 bool Changed = false;
6687 for (unsigned I = 2; I < NumOperands; I += 3) {
6688 auto *Tag = dyn_cast_or_null<MDNode>(Elts[I]);
6689 if (!Tag)
6690 continue;
6691 MDNode *Upgraded = UpgradeTBAANode(*Tag);
6692 if (Upgraded == Tag)
6693 continue;
6694 Elts[I] = Upgraded;
6695 Changed = true;
6696 }
6697 return Changed ? MDNode::get(MD.getContext(), Elts) : &MD;
6698}
6699
6701 Instruction *&Temp) {
6702 if (Opc != Instruction::BitCast)
6703 return nullptr;
6704
6705 Temp = nullptr;
6706 Type *SrcTy = V->getType();
6707 if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() &&
6708 SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) {
6709 LLVMContext &Context = V->getContext();
6710
6711 // We have no information about target data layout, so we assume that
6712 // the maximum pointer size is 64bit.
6713 Type *MidTy = Type::getInt64Ty(Context);
6714 Temp = CastInst::Create(Instruction::PtrToInt, V, MidTy);
6715
6716 return CastInst::Create(Instruction::IntToPtr, Temp, DestTy);
6717 }
6718
6719 return nullptr;
6720}
6721
6723 if (Opc != Instruction::BitCast)
6724 return nullptr;
6725
6726 Type *SrcTy = C->getType();
6727 if (SrcTy->isPtrOrPtrVectorTy() && DestTy->isPtrOrPtrVectorTy() &&
6728 SrcTy->getPointerAddressSpace() != DestTy->getPointerAddressSpace()) {
6729 LLVMContext &Context = C->getContext();
6730
6731 // We have no information about target data layout, so we assume that
6732 // the maximum pointer size is 64bit.
6733 Type *MidTy = Type::getInt64Ty(Context);
6734
6736 DestTy);
6737 }
6738
6739 return nullptr;
6740}
6741
6742static std::optional<StringRef> getModuleFlagNameSafely(const MDNode &Flag) {
6743 if (Flag.getNumOperands() < 3)
6744 return std::nullopt;
6745 if (MDString *Name = dyn_cast_or_null<MDString>(Flag.getOperand(1)))
6746 return Name->getString();
6747 return std::nullopt;
6748}
6749
6750/// Check the debug info version number, if it is out-dated, drop the debug
6751/// info. Return true if module is modified.
6754 return false;
6755
6756 llvm::TimeTraceScope timeScope("Upgrade debug info");
6757 // We need to get metadata before the module is verified (i.e., getModuleFlag
6758 // makes assumptions that we haven't verified yet). Carefully extract the flag
6759 // from the metadata.
6760 unsigned Version = 0;
6761 if (NamedMDNode *ModFlags = M.getModuleFlagsMetadata()) {
6762 auto OpIt = find_if(ModFlags->operands(), [](const MDNode *Flag) {
6763 if (auto Name = getModuleFlagNameSafely(*Flag))
6764 return *Name == "Debug Info Version";
6765 return false;
6766 });
6767 if (OpIt != ModFlags->op_end()) {
6768 const MDOperand &ValOp = (*OpIt)->getOperand(2);
6769 if (auto *CI = mdconst::dyn_extract_or_null<ConstantInt>(ValOp))
6770 Version = CI->getZExtValue();
6771 }
6772 }
6773
6775 bool BrokenDebugInfo = false;
6776 if (verifyModule(M, &llvm::errs(), &BrokenDebugInfo))
6777 report_fatal_error("Broken module found, compilation aborted!");
6778 if (!BrokenDebugInfo)
6779 // Everything is ok.
6780 return false;
6781 else {
6782 // Diagnose malformed debug info.
6784 M.getContext().diagnose(Diag);
6785 }
6786 }
6787 bool Modified = StripDebugInfo(M);
6789 // Diagnose a version mismatch.
6791 M.getContext().diagnose(DiagVersion);
6792 }
6793 return Modified;
6794}
6795
6796static void upgradeNVVMFnVectorAttr(const StringRef Attr, const char DimC,
6797 GlobalValue *GV, const Metadata *V) {
6798 Function *F = cast<Function>(GV);
6799
6800 constexpr StringLiteral DefaultValue = "1";
6801 StringRef Vect3[3] = {DefaultValue, DefaultValue, DefaultValue};
6802 unsigned Length = 0;
6803
6804 if (F->hasFnAttribute(Attr)) {
6805 // We expect the existing attribute to have the form "x[,y[,z]]". Here we
6806 // parse these elements placing them into Vect3
6807 StringRef S = F->getFnAttribute(Attr).getValueAsString();
6808 for (; Length < 3 && !S.empty(); Length++) {
6809 auto [Part, Rest] = S.split(',');
6810 Vect3[Length] = Part.trim();
6811 S = Rest;
6812 }
6813 }
6814
6815 const unsigned Dim = DimC - 'x';
6816 assert(Dim < 3 && "Unexpected dim char");
6817
6818 const uint64_t VInt = mdconst::extract<ConstantInt>(V)->getZExtValue();
6819
6820 // local variable required for StringRef in Vect3 to point to.
6821 const std::string VStr = llvm::utostr(VInt);
6822 Vect3[Dim] = VStr;
6823 Length = std::max(Length, Dim + 1);
6824
6825 const std::string NewAttr = llvm::join(ArrayRef(Vect3, Length), ",");
6826 F->addFnAttr(Attr, NewAttr);
6827}
6828
6829static inline bool isXYZ(StringRef S) {
6830 return S == "x" || S == "y" || S == "z";
6831}
6832
6834 const Metadata *V) {
6835 if (K == "kernel") {
6837 cast<Function>(GV)->setCallingConv(CallingConv::PTX_Kernel);
6838 return true;
6839 }
6840 if (K == "align") {
6841 // V is a bitfeild specifying two 16-bit values. The alignment value is
6842 // specfied in low 16-bits, The index is specified in the high bits. For the
6843 // index, 0 indicates the return value while higher values correspond to
6844 // each parameter (idx = param + 1).
6845 const uint64_t AlignIdxValuePair =
6846 mdconst::extract<ConstantInt>(V)->getZExtValue();
6847 const unsigned Idx = (AlignIdxValuePair >> 16);
6848 const Align StackAlign = Align(AlignIdxValuePair & 0xFFFF);
6849 cast<Function>(GV)->addAttributeAtIndex(
6850 Idx, Attribute::getWithStackAlignment(GV->getContext(), StackAlign));
6851 return true;
6852 }
6853 if (K == "maxclusterrank" || K == "cluster_max_blocks") {
6854 const auto CV = mdconst::extract<ConstantInt>(V)->getZExtValue();
6856 return true;
6857 }
6858 if (K == "minctasm") {
6859 const auto CV = mdconst::extract<ConstantInt>(V)->getZExtValue();
6860 cast<Function>(GV)->addFnAttr(NVVMAttr::MinCTASm, llvm::utostr(CV));
6861 return true;
6862 }
6863 if (K == "maxnreg") {
6864 const auto CV = mdconst::extract<ConstantInt>(V)->getZExtValue();
6865 cast<Function>(GV)->addFnAttr(NVVMAttr::MaxNReg, llvm::utostr(CV));
6866 return true;
6867 }
6868 if (K.consume_front("maxntid") && isXYZ(K)) {
6870 return true;
6871 }
6872 if (K.consume_front("reqntid") && isXYZ(K)) {
6874 return true;
6875 }
6876 if (K.consume_front("cluster_dim_") && isXYZ(K)) {
6878 return true;
6879 }
6880 if (K == "grid_constant") {
6881 const auto Attr = Attribute::get(GV->getContext(), NVVMAttr::GridConstant);
6882 for (const auto &Op : cast<MDNode>(V)->operands()) {
6883 // For some reason, the index is 1-based in the metadata. Good thing we're
6884 // able to auto-upgrade it!
6885 const auto Index = mdconst::extract<ConstantInt>(Op)->getZExtValue() - 1;
6886 cast<Function>(GV)->addParamAttr(Index, Attr);
6887 }
6888 return true;
6889 }
6890
6891 return false;
6892}
6893
6895 NamedMDNode *NamedMD = M.getNamedMetadata("nvvm.annotations");
6896 if (!NamedMD)
6897 return;
6898
6899 SmallVector<MDNode *, 8> NewNodes;
6901 for (MDNode *MD : NamedMD->operands()) {
6902 if (!SeenNodes.insert(MD).second)
6903 continue;
6904
6905 auto *GV = mdconst::dyn_extract_or_null<GlobalValue>(MD->getOperand(0));
6906 if (!GV)
6907 continue;
6908
6909 assert((MD->getNumOperands() % 2) == 1 && "Invalid number of operands");
6910
6911 SmallVector<Metadata *, 8> NewOperands{MD->getOperand(0)};
6912 // Each nvvm.annotations metadata entry will be of the following form:
6913 // !{ ptr @gv, !"key1", value1, !"key2", value2, ... }
6914 // start index = 1, to skip the global variable key
6915 // increment = 2, to skip the value for each property-value pairs
6916 for (unsigned j = 1, je = MD->getNumOperands(); j < je; j += 2) {
6917 MDString *K = cast<MDString>(MD->getOperand(j));
6918 const MDOperand &V = MD->getOperand(j + 1);
6919 bool Upgraded = upgradeSingleNVVMAnnotation(GV, K->getString(), V);
6920 if (!Upgraded)
6921 NewOperands.append({K, V});
6922 }
6923
6924 if (NewOperands.size() > 1)
6925 NewNodes.push_back(MDNode::get(M.getContext(), NewOperands));
6926 }
6927
6928 NamedMD->clearOperands();
6929 for (MDNode *N : NewNodes)
6930 NamedMD->addOperand(N);
6931}
6932
6933/// This checks for objc retain release marker which should be upgraded. It
6934/// returns true if module is modified.
6936 bool Changed = false;
6937 const char *MarkerKey = "clang.arc.retainAutoreleasedReturnValueMarker";
6938 NamedMDNode *ModRetainReleaseMarker = M.getNamedMetadata(MarkerKey);
6939 if (ModRetainReleaseMarker) {
6940 MDNode *Op = ModRetainReleaseMarker->getOperand(0);
6941 if (Op) {
6942 MDString *ID = dyn_cast_or_null<MDString>(Op->getOperand(0));
6943 if (ID) {
6944 SmallVector<StringRef, 4> ValueComp;
6945 ID->getString().split(ValueComp, "#");
6946 if (ValueComp.size() == 2) {
6947 std::string NewValue = ValueComp[0].str() + ";" + ValueComp[1].str();
6948 ID = MDString::get(M.getContext(), NewValue);
6949 }
6950 M.addModuleFlag(Module::Error, MarkerKey, ID);
6951 M.eraseNamedMetadata(ModRetainReleaseMarker);
6952 Changed = true;
6953 }
6954 }
6955 }
6956 return Changed;
6957}
6958
6960 // This lambda converts normal function calls to ARC runtime functions to
6961 // intrinsic calls.
6962 auto UpgradeToIntrinsic = [&](const char *OldFunc,
6963 llvm::Intrinsic::ID IntrinsicFunc) {
6964 Function *Fn = M.getFunction(OldFunc);
6965
6966 if (!Fn)
6967 return;
6968
6969 Function *NewFn =
6970 llvm::Intrinsic::getOrInsertDeclaration(&M, IntrinsicFunc);
6971
6972 for (User *U : make_early_inc_range(Fn->users())) {
6974 if (!CI || CI->getCalledFunction() != Fn)
6975 continue;
6976
6977 IRBuilder<> Builder(CI->getParent(), CI->getIterator());
6978 FunctionType *NewFuncTy = NewFn->getFunctionType();
6980
6981 // Don't upgrade the intrinsic if it's not valid to bitcast the return
6982 // value to the return type of the old function.
6983 if (NewFuncTy->getReturnType() != CI->getType() &&
6984 !CastInst::castIsValid(Instruction::BitCast, CI,
6985 NewFuncTy->getReturnType()))
6986 continue;
6987
6988 bool InvalidCast = false;
6989
6990 for (unsigned I = 0, E = CI->arg_size(); I != E; ++I) {
6991 Value *Arg = CI->getArgOperand(I);
6992
6993 // Bitcast argument to the parameter type of the new function if it's
6994 // not a variadic argument.
6995 if (I < NewFuncTy->getNumParams()) {
6996 // Don't upgrade the intrinsic if it's not valid to bitcast the argument
6997 // to the parameter type of the new function.
6998 if (!CastInst::castIsValid(Instruction::BitCast, Arg,
6999 NewFuncTy->getParamType(I))) {
7000 InvalidCast = true;
7001 break;
7002 }
7003 Arg = Builder.CreateBitCast(Arg, NewFuncTy->getParamType(I));
7004 }
7005 Args.push_back(Arg);
7006 }
7007
7008 if (InvalidCast)
7009 continue;
7010
7011 // Create a call instruction that calls the new function.
7012 CallInst *NewCall = Builder.CreateCall(NewFuncTy, NewFn, Args);
7013 NewCall->setTailCallKind(cast<CallInst>(CI)->getTailCallKind());
7014 NewCall->takeName(CI);
7015
7016 // Bitcast the return value back to the type of the old call.
7017 Value *NewRetVal = Builder.CreateBitCast(NewCall, CI->getType());
7018
7019 if (!CI->use_empty())
7020 CI->replaceAllUsesWith(NewRetVal);
7021 CI->eraseFromParent();
7022 }
7023
7024 if (Fn->use_empty())
7025 Fn->eraseFromParent();
7026 };
7027
7028 // Unconditionally convert a call to "clang.arc.use" to a call to
7029 // "llvm.objc.clang.arc.use".
7030 UpgradeToIntrinsic("clang.arc.use", llvm::Intrinsic::objc_clang_arc_use);
7031
7032 // Upgrade the retain release marker. If there is no need to upgrade
7033 // the marker, that means either the module is already new enough to contain
7034 // new intrinsics or it is not ARC. There is no need to upgrade runtime call.
7036 return;
7037
7038 std::pair<const char *, llvm::Intrinsic::ID> RuntimeFuncs[] = {
7039 {"objc_autorelease", llvm::Intrinsic::objc_autorelease},
7040 {"objc_autoreleasePoolPop", llvm::Intrinsic::objc_autoreleasePoolPop},
7041 {"objc_autoreleasePoolPush", llvm::Intrinsic::objc_autoreleasePoolPush},
7042 {"objc_autoreleaseReturnValue",
7043 llvm::Intrinsic::objc_autoreleaseReturnValue},
7044 {"objc_copyWeak", llvm::Intrinsic::objc_copyWeak},
7045 {"objc_destroyWeak", llvm::Intrinsic::objc_destroyWeak},
7046 {"objc_initWeak", llvm::Intrinsic::objc_initWeak},
7047 {"objc_loadWeak", llvm::Intrinsic::objc_loadWeak},
7048 {"objc_loadWeakRetained", llvm::Intrinsic::objc_loadWeakRetained},
7049 {"objc_moveWeak", llvm::Intrinsic::objc_moveWeak},
7050 {"objc_release", llvm::Intrinsic::objc_release},
7051 {"objc_retain", llvm::Intrinsic::objc_retain},
7052 {"objc_retainAutorelease", llvm::Intrinsic::objc_retainAutorelease},
7053 {"objc_retainAutoreleaseReturnValue",
7054 llvm::Intrinsic::objc_retainAutoreleaseReturnValue},
7055 {"objc_retainAutoreleasedReturnValue",
7056 llvm::Intrinsic::objc_retainAutoreleasedReturnValue},
7057 {"objc_retainBlock", llvm::Intrinsic::objc_retainBlock},
7058 {"objc_storeStrong", llvm::Intrinsic::objc_storeStrong},
7059 {"objc_storeWeak", llvm::Intrinsic::objc_storeWeak},
7060 {"objc_unsafeClaimAutoreleasedReturnValue",
7061 llvm::Intrinsic::objc_unsafeClaimAutoreleasedReturnValue},
7062 {"objc_retainedObject", llvm::Intrinsic::objc_retainedObject},
7063 {"objc_unretainedObject", llvm::Intrinsic::objc_unretainedObject},
7064 {"objc_unretainedPointer", llvm::Intrinsic::objc_unretainedPointer},
7065 {"objc_retain_autorelease", llvm::Intrinsic::objc_retain_autorelease},
7066 {"objc_sync_enter", llvm::Intrinsic::objc_sync_enter},
7067 {"objc_sync_exit", llvm::Intrinsic::objc_sync_exit},
7068 {"objc_arc_annotation_topdown_bbstart",
7069 llvm::Intrinsic::objc_arc_annotation_topdown_bbstart},
7070 {"objc_arc_annotation_topdown_bbend",
7071 llvm::Intrinsic::objc_arc_annotation_topdown_bbend},
7072 {"objc_arc_annotation_bottomup_bbstart",
7073 llvm::Intrinsic::objc_arc_annotation_bottomup_bbstart},
7074 {"objc_arc_annotation_bottomup_bbend",
7075 llvm::Intrinsic::objc_arc_annotation_bottomup_bbend}};
7076
7077 for (auto &I : RuntimeFuncs)
7078 UpgradeToIntrinsic(I.first, I.second);
7079}
7080
7081// Upgrade the way signing of pointers to init/fini functions is described.
7082//
7083// Originally, the `@llvm.global_(ctors|dtors)` arrays contained `ptrauth`
7084// constants, if signing was requested. After the upgrade, these arrays contain
7085// plain function pointers and the desired signing schema is described via a
7086// pair of module flags.
7087//
7088// Note that the upgrade is only performed if all elements of *both* arrays
7089// agree on a common signing schema.
7091 // As we cannot always decide whether the particular module should have
7092 // ptrauth-init-fini flags, we have to treat absent flags as having zero
7093 // values for compatibility reasons. Thus, upgradePtrauthInitFiniArrays
7094 // returns as soon as it spots any non-signed init/fini pointer: either we
7095 // should request non-signed pointers (safe to omit both flags) or there is
7096 // no common schema (and thus we do not modify anything).
7097 //
7098 // UseAddressDisc's value either represents "not decided yet" state (nullopt)
7099 // or whether we should request address diversity in addition to the basic
7100 // constant diversity. There is no value representing "decided not to sign"
7101 // for the reasons explained above.
7102 std::optional<bool> UseAddressDisc;
7103
7104 // Do not attempt upgrading if the new module flags already exist.
7105 if (const NamedMDNode *ModFlags = M.getModuleFlagsMetadata()) {
7106 for (const MDNode *Flag : ModFlags->operands()) {
7107 std::optional<StringRef> Name = getModuleFlagNameSafely(*Flag);
7108 if (Name && (*Name == "ptrauth-init-fini" ||
7109 *Name == "ptrauth-init-fini-address-discrimination"))
7110 return false;
7111 }
7112 }
7113
7114 auto UpgradeSinglePointer = [&UseAddressDisc](Constant *CV) -> Constant * {
7115 constexpr unsigned ExpectedConstDisc = 0xD9D4;
7116 constexpr unsigned ExpectedAddressMarker = 1;
7117
7118 auto *CPA = dyn_cast<ConstantPtrAuth>(CV);
7119 if (!CPA || !CPA->getDiscriminator()->equalsInt(ExpectedConstDisc))
7120 return nullptr; // Nothing to upgrade or unknown pattern found.
7121
7122 bool HasAddressDisc;
7123 if (!CPA->hasAddressDiscriminator())
7124 HasAddressDisc = false;
7125 else if (CPA->hasSpecialAddressDiscriminator(ExpectedAddressMarker))
7126 HasAddressDisc = true;
7127 else
7128 return nullptr; // Unknown pattern.
7129
7130 if (UseAddressDisc && *UseAddressDisc != HasAddressDisc)
7131 return nullptr; // Disagreement with the decided mode.
7132
7133 UseAddressDisc = HasAddressDisc;
7134 return CPA->getPointer();
7135 };
7136
7137 // Do not apply any changes until we know the upgrade is non-ambiguous.
7138 using PendingUpgrade = std::pair<GlobalVariable *, Constant *>;
7139 SmallVector<PendingUpgrade, 2> GlobalArraysToUpgrade;
7140
7141 for (const char *Name : {"llvm.global_ctors", "llvm.global_dtors"}) {
7142 auto *GV = dyn_cast_if_present<GlobalVariable>(M.getNamedValue(Name));
7143 if (!GV || !GV->hasInitializer())
7144 continue; // Skip, but it is okay to upgrade the other variable.
7145
7146 auto *OldStructorsArray = dyn_cast<ConstantArray>(GV->getInitializer());
7147 if (!OldStructorsArray || OldStructorsArray->getNumOperands() == 0)
7148 return false;
7149
7150 std::vector<Constant *> NewStructors;
7151 NewStructors.reserve(OldStructorsArray->getNumOperands());
7152
7153 for (Use &U : OldStructorsArray->operands()) {
7154 ConstantStruct *Structor = dyn_cast<ConstantStruct>(U.get());
7155 if (!Structor || Structor->getNumOperands() != 3)
7156 return false;
7157
7158 Constant *Prio = Structor->getOperand(0);
7159 Constant *Func = Structor->getOperand(1);
7160 Constant *Arg = Structor->getOperand(2);
7161
7162 Func = UpgradeSinglePointer(Func);
7163 if (!Func)
7164 return false;
7165
7166 NewStructors.push_back(
7167 ConstantStruct::get(Structor->getType(), {Prio, Func, Arg}));
7168 }
7169
7170 Constant *NewInit =
7171 ConstantArray::get(OldStructorsArray->getType(), NewStructors);
7172 GlobalArraysToUpgrade.emplace_back(GV, NewInit);
7173 }
7174
7175 if (GlobalArraysToUpgrade.empty())
7176 return false;
7177 assert(UseAddressDisc.has_value());
7178
7179 for (auto [GV, NewInit] : GlobalArraysToUpgrade)
7180 GV->setInitializer(NewInit);
7181
7182 M.addModuleFlag(Module::Error, "ptrauth-init-fini", 1);
7183 M.addModuleFlag(Module::Error, "ptrauth-init-fini-address-discrimination",
7184 *UseAddressDisc);
7185
7186 return true;
7187}
7188
7190 bool Changed = false;
7192
7193 NamedMDNode *ModFlags = M.getModuleFlagsMetadata();
7194 if (!ModFlags)
7195 return Changed;
7196
7197 bool HasObjCFlag = false, HasClassProperties = false;
7198 bool HasSwiftVersionFlag = false;
7199 uint8_t SwiftMajorVersion, SwiftMinorVersion;
7200 uint32_t SwiftABIVersion;
7201 auto Int8Ty = Type::getInt8Ty(M.getContext());
7202 auto Int32Ty = Type::getInt32Ty(M.getContext());
7203
7204 for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) {
7205 MDNode *Op = ModFlags->getOperand(I);
7206 if (Op->getNumOperands() != 3)
7207 continue;
7208 MDString *ID = dyn_cast_or_null<MDString>(Op->getOperand(1));
7209 if (!ID)
7210 continue;
7211 auto SetBehavior = [&](Module::ModFlagBehavior B) {
7212 Metadata *Ops[3] = {ConstantAsMetadata::get(ConstantInt::get(
7213 Type::getInt32Ty(M.getContext()), B)),
7214 MDString::get(M.getContext(), ID->getString()),
7215 Op->getOperand(2)};
7216 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7217 Changed = true;
7218 };
7219
7220 if (ID->getString() == "Objective-C Image Info Version")
7221 HasObjCFlag = true;
7222 if (ID->getString() == "Objective-C Class Properties")
7223 HasClassProperties = true;
7224 // Upgrade PIC from Error/Max to Min.
7225 if (ID->getString() == "PIC Level") {
7226 if (auto *Behavior =
7228 uint64_t V = Behavior->getLimitedValue();
7229 if (V == Module::Error || V == Module::Max)
7230 SetBehavior(Module::Min);
7231 }
7232 }
7233 // Upgrade "PIE Level" from Error to Max.
7234 if (ID->getString() == "PIE Level")
7235 if (auto *Behavior =
7237 if (Behavior->getLimitedValue() == Module::Error)
7238 SetBehavior(Module::Max);
7239
7240 // Upgrade branch protection and return address signing module flags. The
7241 // module flag behavior for these fields were Error and now they are Min.
7242 if (ID->getString() == "branch-target-enforcement" ||
7243 ID->getString().starts_with("sign-return-address")) {
7244 if (auto *Behavior =
7246 if (Behavior->getLimitedValue() == Module::Error) {
7247 Type *Int32Ty = Type::getInt32Ty(M.getContext());
7248 Metadata *Ops[3] = {
7249 ConstantAsMetadata::get(ConstantInt::get(Int32Ty, Module::Min)),
7250 Op->getOperand(1), Op->getOperand(2)};
7251 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7252 Changed = true;
7253 }
7254 }
7255 }
7256
7257 // Upgrade Objective-C Image Info Section. Removed the whitespce in the
7258 // section name so that llvm-lto will not complain about mismatching
7259 // module flags that is functionally the same.
7260 if (ID->getString() == "Objective-C Image Info Section") {
7261 if (auto *Value = dyn_cast_or_null<MDString>(Op->getOperand(2))) {
7262 SmallVector<StringRef, 4> ValueComp;
7263 Value->getString().split(ValueComp, " ");
7264 if (ValueComp.size() != 1) {
7265 std::string NewValue;
7266 for (auto &S : ValueComp)
7267 NewValue += S.str();
7268 Metadata *Ops[3] = {Op->getOperand(0), Op->getOperand(1),
7269 MDString::get(M.getContext(), NewValue)};
7270 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7271 Changed = true;
7272 }
7273 }
7274 }
7275
7276 // IRUpgrader turns a i32 type "Objective-C Garbage Collection" into i8 value.
7277 // If the higher bits are set, it adds new module flag for swift info.
7278 if (ID->getString() == "Objective-C Garbage Collection") {
7279 auto Md = dyn_cast<ConstantAsMetadata>(Op->getOperand(2));
7280 if (Md) {
7281 assert(Md->getValue() && "Expected non-empty metadata");
7282 auto Type = Md->getValue()->getType();
7283 if (Type == Int8Ty)
7284 continue;
7285 unsigned Val = Md->getValue()->getUniqueInteger().getZExtValue();
7286 if ((Val & 0xff) != Val) {
7287 HasSwiftVersionFlag = true;
7288 SwiftABIVersion = (Val & 0xff00) >> 8;
7289 SwiftMajorVersion = (Val & 0xff000000) >> 24;
7290 SwiftMinorVersion = (Val & 0xff0000) >> 16;
7291 }
7292 Metadata *Ops[3] = {
7293 ConstantAsMetadata::get(ConstantInt::get(Int32Ty,Module::Error)),
7294 Op->getOperand(1),
7295 ConstantAsMetadata::get(ConstantInt::get(Int8Ty,Val & 0xff))};
7296 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7297 Changed = true;
7298 }
7299 }
7300
7301 if (ID->getString() == "amdgpu_code_object_version") {
7302 Metadata *Ops[3] = {
7303 Op->getOperand(0),
7304 MDString::get(M.getContext(), "amdhsa_code_object_version"),
7305 Op->getOperand(2)};
7306 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7307 Changed = true;
7308 }
7309
7310 // clang/PowerPC used to use "float-abi" to describe the long double format;
7311 // it has been renamed to "long-double-type", with its values changed to the
7312 // corresponding IR floating-point type names.
7313 if (M.getTargetTriple().isPPC() && ID->getString() == "float-abi") {
7315 if (auto *S = dyn_cast_or_null<MDString>(Op->getOperand(2)))
7316 Format = S->getString();
7317
7318 // The "float-abi" key is now reserved for the target-independent
7319 // soft/hard ABI flag, so leave a valid value alone. Map any other value
7320 // (including unrecognized ones, which were never valid) to the default.
7322 LongDoubleFormat NewFormat =
7324 .Case("ieeequad", LongDoubleFormat::IEEEquad)
7325 .Case("ieeedouble", LongDoubleFormat::IEEEdouble)
7327 Metadata *Ops[3] = {
7328 Op->getOperand(0),
7329 MDString::get(M.getContext(), "long-double-type"),
7330 MDString::get(M.getContext(), getLongDoubleFormatName(NewFormat))};
7331 ModFlags->setOperand(I, MDNode::get(M.getContext(), Ops));
7332 Changed = true;
7333 }
7334 }
7335 }
7336
7337 // "Objective-C Class Properties" is recently added for Objective-C. We
7338 // upgrade ObjC bitcodes to contain a "Objective-C Class Properties" module
7339 // flag of value 0, so we can correclty downgrade this flag when trying to
7340 // link an ObjC bitcode without this module flag with an ObjC bitcode with
7341 // this module flag.
7342 if (HasObjCFlag && !HasClassProperties) {
7343 M.addModuleFlag(llvm::Module::Override, "Objective-C Class Properties",
7344 (uint32_t)0);
7345 Changed = true;
7346 }
7347
7348 if (HasSwiftVersionFlag) {
7349 M.addModuleFlag(Module::Error, "Swift ABI Version",
7350 SwiftABIVersion);
7351 M.addModuleFlag(Module::Error, "Swift Major Version",
7352 ConstantInt::get(Int8Ty, SwiftMajorVersion));
7353 M.addModuleFlag(Module::Error, "Swift Minor Version",
7354 ConstantInt::get(Int8Ty, SwiftMinorVersion));
7355 Changed = true;
7356 }
7357
7358 return Changed;
7359}
7360
7362 NamedMDNode *CFIConsts = M.getNamedMetadata("cfi.functions");
7363 // If this metadata has operands, we expect all of them to be either from
7364 // before or from after the format change handled here, so we can bail out
7365 // fast if the first (if any) operands is of the new format.
7366 auto MatchesVersion = [](const MDNode *Op) {
7367 return Op->getNumOperands() >= 3 &&
7368 isa<ConstantAsMetadata>(Op->getOperand(2)) &&
7369 cast<ConstantAsMetadata>(Op->getOperand(2))
7370 ->getType()
7371 ->isIntegerTy(64);
7372 };
7373
7374 if (!CFIConsts || !CFIConsts->getNumOperands() ||
7375 MatchesVersion(CFIConsts->getOperand(0)))
7376 return false;
7377
7378 bool Changed = false;
7379 for (unsigned I = 0, E = CFIConsts->getNumOperands(); I != E; ++I) {
7380 MDNode *Op = CFIConsts->getOperand(I);
7381 assert(!MatchesVersion(Op) && "Unexpected mix of CFIConstant formats");
7382 assert(Op->getNumOperands() >= 2 &&
7383 "Expected at least 2 operands - name and linkage type");
7384 MDString *NameMD = dyn_cast<MDString>(Op->getOperand(0));
7385 StringRef Name = NameMD->getString();
7388
7390 Elts.push_back(Op->getOperand(0));
7391 Elts.push_back(Op->getOperand(1));
7393 ConstantInt::get(Type::getInt64Ty(M.getContext()), GUID)));
7394
7395 for (unsigned J = 2, EJ = Op->getNumOperands(); J != EJ; ++J)
7396 Elts.push_back(Op->getOperand(J));
7397
7398 CFIConsts->setOperand(I, MDNode::get(M.getContext(), Elts));
7399 Changed = true;
7400 }
7401
7402 return Changed;
7403}
7404
7406 auto TrimSpaces = [](StringRef Section) -> std::string {
7407 SmallVector<StringRef, 5> Components;
7408 Section.split(Components, ',');
7409
7410 SmallString<32> Buffer;
7411 raw_svector_ostream OS(Buffer);
7412
7413 for (auto Component : Components)
7414 OS << ',' << Component.trim();
7415
7416 return std::string(OS.str().substr(1));
7417 };
7418
7419 for (auto &GV : M.globals()) {
7420 if (!GV.hasSection())
7421 continue;
7422
7423 StringRef Section = GV.getSection();
7424
7425 if (!Section.starts_with("__DATA, __objc_catlist"))
7426 continue;
7427
7428 // __DATA, __objc_catlist, regular, no_dead_strip
7429 // __DATA,__objc_catlist,regular,no_dead_strip
7430 GV.setSection(TrimSpaces(Section));
7431 }
7432}
7433
7434namespace {
7435// Prior to LLVM 10.0, the strictfp attribute could be used on individual
7436// callsites within a function that did not also have the strictfp attribute.
7437// Since 10.0, if strict FP semantics are needed within a function, the
7438// function must have the strictfp attribute and all calls within the function
7439// must also have the strictfp attribute. This latter restriction is
7440// necessary to prevent unwanted libcall simplification when a function is
7441// being cloned (such as for inlining).
7442//
7443// The "dangling" strictfp attribute usage was only used to prevent constant
7444// folding and other libcall simplification. The nobuiltin attribute on the
7445// callsite has the same effect.
7446struct StrictFPUpgradeVisitor : public InstVisitor<StrictFPUpgradeVisitor> {
7447 StrictFPUpgradeVisitor() = default;
7448
7449 void visitCallBase(CallBase &Call) {
7450 if (!Call.isStrictFP())
7451 return;
7453 return;
7454 // If we get here, the caller doesn't have the strictfp attribute
7455 // but this callsite does. Replace the strictfp attribute with nobuiltin.
7456 Call.removeFnAttr(Attribute::StrictFP);
7457 Call.addFnAttr(Attribute::NoBuiltin);
7458 }
7459};
7460
7461/// Replace "amdgpu-unsafe-fp-atomics" metadata with atomicrmw metadata
7462struct AMDGPUUnsafeFPAtomicsUpgradeVisitor
7463 : public InstVisitor<AMDGPUUnsafeFPAtomicsUpgradeVisitor> {
7464 AMDGPUUnsafeFPAtomicsUpgradeVisitor() = default;
7465
7466 void visitAtomicRMWInst(AtomicRMWInst &RMW) {
7467 if (!RMW.isFloatingPointOperation())
7468 return;
7469
7470 MDNode *Empty = MDNode::get(RMW.getContext(), {});
7471 RMW.setMetadata("amdgpu.no.fine.grained.host.memory", Empty);
7472 RMW.setMetadata("amdgpu.no.remote.memory.access", Empty);
7473 RMW.setMetadata(LLVMContext::MD_atomic_ignore_denormal_mode, Empty);
7474 }
7475};
7476} // namespace
7477
7479 // If a function definition doesn't have the strictfp attribute,
7480 // convert any callsite strictfp attributes to nobuiltin.
7481 if (!F.isDeclaration() && !F.hasFnAttribute(Attribute::StrictFP)) {
7482 StrictFPUpgradeVisitor SFPV;
7483 SFPV.visit(F);
7484 }
7485
7486 // Remove all incompatibile attributes from function.
7487 F.removeRetAttrs(AttributeFuncs::typeIncompatible(
7488 F.getReturnType(), F.getAttributes().getRetAttrs()));
7489 for (auto &Arg : F.args())
7490 Arg.removeAttrs(
7491 AttributeFuncs::typeIncompatible(Arg.getType(), Arg.getAttributes()));
7492
7493 bool AddingAttrs = false, RemovingAttrs = false;
7494 AttrBuilder AttrsToAdd(F.getContext());
7495 AttributeMask AttrsToRemove;
7496
7497 // Older versions of LLVM treated an "implicit-section-name" attribute
7498 // similarly to directly setting the section on a Function.
7499 if (Attribute A = F.getFnAttribute("implicit-section-name");
7500 A.isValid() && A.isStringAttribute()) {
7501 F.setSection(A.getValueAsString());
7502 AttrsToRemove.addAttribute("implicit-section-name");
7503 RemovingAttrs = true;
7504 }
7505
7506 if (Attribute A = F.getFnAttribute("nooutline");
7507 A.isValid() && A.isStringAttribute()) {
7508 AttrsToRemove.addAttribute("nooutline");
7509 AttrsToAdd.addAttribute(Attribute::NoOutline);
7510 AddingAttrs = RemovingAttrs = true;
7511 }
7512
7513 if (Attribute A = F.getFnAttribute("uniform-work-group-size");
7514 A.isValid() && A.isStringAttribute() && !A.getValueAsString().empty()) {
7515 AttrsToRemove.addAttribute("uniform-work-group-size");
7516 RemovingAttrs = true;
7517 if (A.getValueAsString() == "true") {
7518 AttrsToAdd.addAttribute("uniform-work-group-size");
7519 AddingAttrs = true;
7520 }
7521 }
7522
7523 if (!F.empty()) {
7524 // For some reason this is called twice, and the first time is before any
7525 // instructions are loaded into the body.
7526
7527 if (Attribute A = F.getFnAttribute("amdgpu-unsafe-fp-atomics");
7528 A.isValid()) {
7529
7530 if (A.getValueAsBool()) {
7531 AMDGPUUnsafeFPAtomicsUpgradeVisitor Visitor;
7532 Visitor.visit(F);
7533 }
7534
7535 // We will leave behind dead attribute uses on external declarations, but
7536 // clang never added these to declarations anyway.
7537 AttrsToRemove.addAttribute("amdgpu-unsafe-fp-atomics");
7538 RemovingAttrs = true;
7539 }
7540 }
7541
7542 DenormalMode DenormalFPMath = DenormalMode::getIEEE();
7543 DenormalMode DenormalFPMathF32 = DenormalMode::getInvalid();
7544
7545 bool HandleDenormalMode = false;
7546
7547 if (Attribute Attr = F.getFnAttribute("denormal-fp-math"); Attr.isValid()) {
7548 DenormalMode ParsedMode = parseDenormalFPAttribute(Attr.getValueAsString());
7549 if (ParsedMode.isValid()) {
7550 DenormalFPMath = ParsedMode;
7551 AttrsToRemove.addAttribute("denormal-fp-math");
7552 AddingAttrs = RemovingAttrs = true;
7553 HandleDenormalMode = true;
7554 }
7555 }
7556
7557 if (Attribute Attr = F.getFnAttribute("denormal-fp-math-f32");
7558 Attr.isValid()) {
7559 DenormalMode ParsedMode = parseDenormalFPAttribute(Attr.getValueAsString());
7560 if (ParsedMode.isValid()) {
7561 DenormalFPMathF32 = ParsedMode;
7562 AttrsToRemove.addAttribute("denormal-fp-math-f32");
7563 AddingAttrs = RemovingAttrs = true;
7564 HandleDenormalMode = true;
7565 }
7566 }
7567
7568 if (HandleDenormalMode)
7569 AttrsToAdd.addDenormalFPEnvAttr(
7570 DenormalFPEnv(DenormalFPMath, DenormalFPMathF32));
7571
7572 if (RemovingAttrs)
7573 F.removeFnAttrs(AttrsToRemove);
7574
7575 if (AddingAttrs)
7576 F.addFnAttrs(AttrsToAdd);
7577}
7578
7579// Check if the function attribute is not present and set it.
7581 StringRef Value) {
7582 if (!F.hasFnAttribute(FnAttrName))
7583 F.addFnAttr(FnAttrName, Value);
7584}
7585
7586// Check if the function attribute is not present and set it if needed.
7587// If the attribute is "false" then removes it.
7588// If the attribute is "true" resets it to a valueless attribute.
7589static void ConvertFunctionAttr(Function &F, bool Set, StringRef FnAttrName) {
7590 if (!F.hasFnAttribute(FnAttrName)) {
7591 if (Set)
7592 F.addFnAttr(FnAttrName);
7593 } else {
7594 auto A = F.getFnAttribute(FnAttrName);
7595 if ("false" == A.getValueAsString())
7596 F.removeFnAttr(FnAttrName);
7597 else if ("true" == A.getValueAsString()) {
7598 F.removeFnAttr(FnAttrName);
7599 F.addFnAttr(FnAttrName);
7600 }
7601 }
7602}
7603
7605 Triple T(M.getTargetTriple());
7606 if (!T.isThumb() && !T.isARM() && !T.isAArch64())
7607 return;
7608
7609 uint64_t BTEValue = 0;
7610 uint64_t BPPLRValue = 0;
7611 uint64_t GCSValue = 0;
7612 uint64_t SRAValue = 0;
7613 uint64_t SRAALLValue = 0;
7614 uint64_t SRABKeyValue = 0;
7615
7616 NamedMDNode *ModFlags = M.getModuleFlagsMetadata();
7617 if (ModFlags) {
7618 for (unsigned I = 0, E = ModFlags->getNumOperands(); I != E; ++I) {
7619 MDNode *Op = ModFlags->getOperand(I);
7620 if (Op->getNumOperands() != 3)
7621 continue;
7622
7623 MDString *ID = dyn_cast_or_null<MDString>(Op->getOperand(1));
7624 auto *CI = mdconst::dyn_extract<ConstantInt>(Op->getOperand(2));
7625 if (!ID || !CI)
7626 continue;
7627
7628 StringRef IDStr = ID->getString();
7629 uint64_t *ValPtr = IDStr == "branch-target-enforcement" ? &BTEValue
7630 : IDStr == "branch-protection-pauth-lr" ? &BPPLRValue
7631 : IDStr == "guarded-control-stack" ? &GCSValue
7632 : IDStr == "sign-return-address" ? &SRAValue
7633 : IDStr == "sign-return-address-all" ? &SRAALLValue
7634 : IDStr == "sign-return-address-with-bkey"
7635 ? &SRABKeyValue
7636 : nullptr;
7637 if (!ValPtr)
7638 continue;
7639
7640 *ValPtr = CI->getZExtValue();
7641 if (*ValPtr == 2)
7642 return;
7643 }
7644 }
7645
7646 bool BTE = BTEValue == 1;
7647 bool BPPLR = BPPLRValue == 1;
7648 bool GCS = GCSValue == 1;
7649 bool SRA = SRAValue == 1;
7650
7651 StringRef SignTypeValue = "non-leaf";
7652 if (SRA && SRAALLValue == 1)
7653 SignTypeValue = "all";
7654
7655 StringRef SignKeyValue = "a_key";
7656 if (SRA && SRABKeyValue == 1)
7657 SignKeyValue = "b_key";
7658
7659 for (Function &F : M.getFunctionList()) {
7660 if (F.isDeclaration())
7661 continue;
7662
7663 if (SRA) {
7664 setFunctionAttrIfNotSet(F, "sign-return-address", SignTypeValue);
7665 setFunctionAttrIfNotSet(F, "sign-return-address-key", SignKeyValue);
7666 } else {
7667 if (auto A = F.getFnAttribute("sign-return-address");
7668 A.isValid() && "none" == A.getValueAsString()) {
7669 F.removeFnAttr("sign-return-address");
7670 F.removeFnAttr("sign-return-address-key");
7671 }
7672 }
7673 ConvertFunctionAttr(F, BTE, "branch-target-enforcement");
7674 ConvertFunctionAttr(F, BPPLR, "branch-protection-pauth-lr");
7675 ConvertFunctionAttr(F, GCS, "guarded-control-stack");
7676 }
7677
7678 if (BTE)
7679 M.setModuleFlag(llvm::Module::Min, "branch-target-enforcement", 2);
7680 if (BPPLR)
7681 M.setModuleFlag(llvm::Module::Min, "branch-protection-pauth-lr", 2);
7682 if (GCS)
7683 M.setModuleFlag(llvm::Module::Min, "guarded-control-stack", 2);
7684 if (SRA) {
7685 M.setModuleFlag(llvm::Module::Min, "sign-return-address", 2);
7686 if (SRAALLValue == 1)
7687 M.setModuleFlag(llvm::Module::Min, "sign-return-address-all", 2);
7688 if (SRABKeyValue == 1)
7689 M.setModuleFlag(llvm::Module::Min, "sign-return-address-with-bkey", 2);
7690 }
7691}
7692
7693/// Return the replacement tags if \p T still uses a removed two-operand form.
7695 if (T->getNumOperands() != 2 || !mdconst::hasa<ConstantInt>(T->getOperand(1)))
7696 return nullptr;
7697 auto *Tag = dyn_cast_or_null<MDString>(T->getOperand(0));
7698 return Tag ? findBooleanLoopTags(Tag->getString()) : nullptr;
7699}
7700
7701/// Build the single-operand node that replaces a boolean operand: nonzero
7702/// selects the enable tag, zero the disable tag.
7704 const BooleanLoopTags &Tags,
7705 const MDOperand &Op) {
7706 bool Enable = !mdconst::extract<ConstantInt>(Op)->isZero();
7707 return MDTuple::get(C,
7708 {MDString::get(C, Enable ? Tags.Enable : Tags.Disable)});
7709}
7710
7711static bool isOldLoopArgument(Metadata *MD) {
7712 auto *T = dyn_cast_or_null<MDTuple>(MD);
7713 if (!T)
7714 return false;
7715 if (T->getNumOperands() < 1)
7716 return false;
7717 auto *S = dyn_cast_or_null<MDString>(T->getOperand(0));
7718 if (!S)
7719 return false;
7720 if (S->getString().starts_with("llvm.vectorizer."))
7721 return true;
7722 return getOldBooleanLoopTags(T) != nullptr;
7723}
7724
7726 StringRef OldPrefix = "llvm.vectorizer.";
7727 assert(OldTag.starts_with(OldPrefix) && "Expected old prefix");
7728
7729 if (OldTag == "llvm.vectorizer.unroll")
7730 return MDString::get(C, "llvm.loop.interleave.count");
7731
7732 return MDString::get(
7733 C, (Twine("llvm.loop.vectorize.") + OldTag.drop_front(OldPrefix.size()))
7734 .str());
7735}
7736
7738 auto *T = dyn_cast_or_null<MDTuple>(MD);
7739 if (!T)
7740 return MD;
7741 if (T->getNumOperands() < 1)
7742 return MD;
7743 auto *OldTag = dyn_cast_or_null<MDString>(T->getOperand(0));
7744 if (!OldTag)
7745 return MD;
7746
7747 LLVMContext &C = T->getContext();
7748
7749 /// Rewrite a removed two-operand boolean form to the single-operand pair.
7750 if (const BooleanLoopTags *Tags = getOldBooleanLoopTags(T))
7751 return makeBooleanLoopNode(C, *Tags, T->getOperand(1));
7752
7753 if (!OldTag->getString().starts_with("llvm.vectorizer."))
7754 return MD;
7755
7756 // This has an old tag. Upgrade it.
7757 MDString *NewTag = upgradeLoopTag(C, OldTag->getString());
7758
7759 // The legacy !{!"llvm.vectorizer.enable", i1 X} maps onto the single-operand
7760 // vectorize.enable/disable pair, not a two-operand enable node.
7761 if (T->getNumOperands() == 2 && mdconst::hasa<ConstantInt>(T->getOperand(1)))
7762 if (const BooleanLoopTags *Tags = findBooleanLoopTags(NewTag->getString()))
7763 return makeBooleanLoopNode(C, *Tags, T->getOperand(1));
7764
7766 Ops.reserve(T->getNumOperands());
7767 Ops.push_back(NewTag);
7768 for (unsigned I = 1, E = T->getNumOperands(); I != E; ++I)
7769 Ops.push_back(T->getOperand(I));
7770
7771 return MDTuple::get(C, Ops);
7772}
7773
7775 auto *T = dyn_cast<MDTuple>(&N);
7776 if (!T)
7777 return &N;
7778
7779 if (none_of(T->operands(), isOldLoopArgument))
7780 return &N;
7781
7782 // Fix the removed two-operand boolean nodes in place: the Verifier rejects
7783 // any MDNode carrying those tags with more than one operand, so a leftover
7784 // reference (from the distinct loop-ID) would still trigger a diagnostic.
7785 // In-place mutation is safe on distinct MDNodes.
7786 if (T->isDistinct()) {
7787 for (unsigned I = 0, E = T->getNumOperands(); I < E; ++I) {
7788 auto *OpT = dyn_cast_or_null<MDTuple>(T->getOperand(I));
7789 if (OpT && getOldBooleanLoopTags(OpT))
7790 T->replaceOperandWith(I, upgradeLoopArgument(OpT));
7791 }
7792 if (none_of(T->operands(), isOldLoopArgument))
7793 return &N;
7794 }
7795
7796 // Remaining old arguments (e.g. llvm.vectorizer.*) are handled via a wrapper
7797 // attachment; the original distinct loop-ID is kept as the first operand.
7799 Ops.reserve(T->getNumOperands());
7800 for (Metadata *MD : T->operands())
7801 Ops.push_back(upgradeLoopArgument(MD));
7802
7803 return MDTuple::get(T->getContext(), Ops);
7804}
7805
7807 Triple T(TT);
7808 // The only data layout upgrades needed for pre-GCN, SPIR or SPIRV are setting
7809 // the address space of globals to 1. This does not apply to SPIRV Logical.
7810 if ((T.isSPIR() || (T.isSPIRV() && !T.isSPIRVLogical())) &&
7811 !DL.contains("-G") && !DL.starts_with("G")) {
7812 return DL.empty() ? std::string("G1") : (DL + "-G1").str();
7813 }
7814
7815 if (T.isLoongArch64() || T.isRISCV64()) {
7816 // Make i32 a native type for 64-bit LoongArch and RISC-V.
7817 auto I = DL.find("-n64-");
7818 if (I != StringRef::npos)
7819 return (DL.take_front(I) + "-n32:64-" + DL.drop_front(I + 5)).str();
7820 return DL.str();
7821 }
7822
7823 // AMDGPU data layout upgrades.
7824 std::string Res = DL.str();
7825 if (T.isAMDGPU()) {
7826 // Define address spaces for constants.
7827 if (!DL.contains("-G") && !DL.starts_with("G"))
7828 Res.append(Res.empty() ? "G1" : "-G1");
7829
7830 // AMDGCN data layout upgrades.
7831 if (T.isAMDGCN()) {
7832
7833 // Add missing non-integral declarations.
7834 // This goes before adding new address spaces to prevent incoherent string
7835 // values.
7836 if (!DL.contains("-ni") && !DL.starts_with("ni"))
7837 Res.append("-ni:7:8:9");
7838 // Update ni:7 to ni:7:8:9.
7839 if (DL.ends_with("ni:7"))
7840 Res.append(":8:9");
7841 if (DL.ends_with("ni:7:8"))
7842 Res.append(":9");
7843
7844 // Add sizing for address spaces 7 and 8 (fat raw buffers and buffer
7845 // resources) An empty data layout has already been upgraded to G1 by now.
7846 if (!DL.contains("-p7") && !DL.starts_with("p7"))
7847 Res.append("-p7:160:256:256:32");
7848 if (!DL.contains("-p8") && !DL.starts_with("p8"))
7849 Res.append("-p8:128:128:128:48");
7850 constexpr StringRef OldP8("-p8:128:128-");
7851 if (DL.contains(OldP8))
7852 Res.replace(Res.find(OldP8), OldP8.size(), "-p8:128:128:128:48-");
7853 if (!DL.contains("-p9") && !DL.starts_with("p9"))
7854 Res.append("-p9:192:256:256:32");
7855
7856 // Add sizing for address space 10 through 15.
7857 // AS 10-14 are reserved and defaulted to 32:32
7858 // AS 15 is in use and is 32:32.
7859 for (StringRef AS : {"p10", "p11", "p12", "p13", "p14", "p15"}) {
7860 if (!DL.contains(("-" + AS).str()) && !DL.starts_with(AS))
7861 Res.append(("-" + AS + ":32:32").str());
7862 }
7863 }
7864
7865 // Upgrade the ELF mangling mode.
7866 if (!DL.contains("m:e"))
7867 Res = Res.empty() ? "m:e" : "m:e-" + Res;
7868
7869 return Res;
7870 }
7871
7872 if (T.isSystemZ() && !DL.empty()) {
7873 // Make sure the stack alignment is present.
7874 if (!DL.contains("-S64"))
7875 return "E-S64" + DL.drop_front(1).str();
7876 return DL.str();
7877 }
7878
7879 auto AddPtr32Ptr64AddrSpaces = [&DL, &Res]() {
7880 // If the datalayout matches the expected format, add pointer size address
7881 // spaces to the datalayout.
7882 StringRef AddrSpaces{"-p270:32:32-p271:32:32-p272:64:64"};
7883 if (!DL.contains(AddrSpaces)) {
7885 Regex R("^([Ee]-m:[a-z](-p:32:32)?)(-.*)$");
7886 if (R.match(Res, &Groups))
7887 Res = (Groups[1] + AddrSpaces + Groups[3]).str();
7888 }
7889 };
7890
7891 // AArch64 data layout upgrades.
7892 if (T.isAArch64()) {
7893 // Add "-Fn32"
7894 if (!DL.empty() && !DL.contains("-Fn32"))
7895 Res.append("-Fn32");
7896 AddPtr32Ptr64AddrSpaces();
7897 return Res;
7898 }
7899
7900 if (T.isSPARC() || (T.isMIPS64() && !DL.contains("m:m")) || T.isPPC64() ||
7901 T.isWasm()) {
7902 // Mips64 with o32 ABI did not add "-i128:128".
7903 // Add "-i128:128"
7904 std::string I64 = "-i64:64";
7905 std::string I128 = "-i128:128";
7906 if (!StringRef(Res).contains(I128)) {
7907 size_t Pos = Res.find(I64);
7908 if (Pos != size_t(-1))
7909 Res.insert(Pos + I64.size(), I128);
7910 }
7911 }
7912
7913 if (T.isPPC() && T.isOSAIX() && !DL.contains("f64:32:64") && !DL.empty()) {
7914 size_t Pos = Res.find("-S128");
7915 if (Pos == StringRef::npos)
7916 Pos = Res.size();
7917 Res.insert(Pos, "-f64:32:64");
7918 }
7919
7920 // ARM data layout upgrades.
7921 // Add -Fi8 if a -F has not already been specified.
7922 if (T.isARM() && !DL.empty() && !DL.contains("Fi") && !DL.contains("Fn")) {
7923 const StringRef p3232 = "p:32:32";
7924 size_t Pos = Res.find(p3232);
7925 if (Pos != StringRef::npos)
7926 Res.insert(Pos + p3232.size(), "-Fi8");
7927 }
7928
7929 if (!T.isX86())
7930 return Res;
7931
7932 AddPtr32Ptr64AddrSpaces();
7933
7934 // i128 values need to be 16-byte-aligned. LLVM already called into libgcc
7935 // for i128 operations prior to this being reflected in the data layout, and
7936 // clang mostly produced LLVM IR that already aligned i128 to 16 byte
7937 // boundaries, so although this is a breaking change, the upgrade is expected
7938 // to fix more IR than it breaks.
7939 // Intel MCU is an exception and uses 4-byte-alignment.
7940 if (!T.isOSIAMCU()) {
7941 std::string I128 = "-i128:128";
7942 if (StringRef Ref = Res; !Ref.contains(I128)) {
7944 Regex R("^(e(-[mpi][^-]*)*)((-[^mpi][^-]*)*)$");
7945 if (R.match(Res, &Groups))
7946 Res = (Groups[1] + I128 + Groups[3]).str();
7947 }
7948 }
7949
7950 // For 32-bit MSVC targets, raise the alignment of f80 values to 16 bytes.
7951 // Raising the alignment is safe because Clang did not produce f80 values in
7952 // the MSVC environment before this upgrade was added.
7953 if (T.isWindowsMSVCEnvironment() && !T.isArch64Bit()) {
7954 StringRef Ref = Res;
7955 auto I = Ref.find("-f80:32-");
7956 if (I != StringRef::npos)
7957 Res = (Ref.take_front(I) + "-f80:128-" + Ref.drop_front(I + 8)).str();
7958 }
7959
7960 return Res;
7961}
7962
7963void llvm::UpgradeAttributes(AttrBuilder &B) {
7964 StringRef FramePointer;
7965 Attribute A = B.getAttribute("no-frame-pointer-elim");
7966 if (A.isValid()) {
7967 // The value can be "true" or "false".
7968 FramePointer = A.getValueAsString() == "true" ? "all" : "none";
7969 B.removeAttribute("no-frame-pointer-elim");
7970 }
7971 if (B.contains("no-frame-pointer-elim-non-leaf")) {
7972 // The value is ignored. "no-frame-pointer-elim"="true" takes priority.
7973 if (FramePointer != "all")
7974 FramePointer = "non-leaf";
7975 B.removeAttribute("no-frame-pointer-elim-non-leaf");
7976 }
7977 if (!FramePointer.empty())
7978 B.addAttribute("frame-pointer", FramePointer);
7979
7980 A = B.getAttribute("null-pointer-is-valid");
7981 if (A.isValid()) {
7982 // The value can be "true" or "false".
7983 bool NullPointerIsValid = A.getValueAsString() == "true";
7984 B.removeAttribute("null-pointer-is-valid");
7985 if (NullPointerIsValid)
7986 B.addAttribute(Attribute::NullPointerIsValid);
7987 }
7988
7989 A = B.getAttribute("uniform-work-group-size");
7990 if (A.isValid()) {
7991 StringRef Val = A.getValueAsString();
7992 if (!Val.empty()) {
7993 bool IsTrue = Val == "true";
7994 B.removeAttribute("uniform-work-group-size");
7995 if (IsTrue)
7996 B.addAttribute("uniform-work-group-size");
7997 }
7998 }
7999}
8000
8001void llvm::UpgradeOperandBundles(std::vector<OperandBundleDef> &Bundles) {
8002 // clang.arc.attachedcall bundles are now required to have an operand.
8003 // If they don't, it's okay to drop them entirely: when there is an operand,
8004 // the "attachedcall" is meaningful and required, but without an operand,
8005 // it's just a marker NOP. Dropping it merely prevents an optimization.
8006 erase_if(Bundles, [&](OperandBundleDef &OBD) {
8007 return OBD.getTag() == "clang.arc.attachedcall" &&
8008 OBD.inputs().empty();
8009 });
8010}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
This file contains the simple types necessary to represent the attributes associated with functions a...
static Value * upgradeX86VPERMT2Intrinsics(IRBuilder<> &Builder, CallBase &CI, bool ZeroMask, bool IndexForm)
static bool isLegacyNVPTXBF16IntSignature(Function *F, Intrinsic::ID IID)
static unsigned getFullArgCountForDefaultArgUpgrade(Function *F, Intrinsic::ID IID, SmallVectorImpl< Type * > &OverloadTys)
#define G2S_ID(ID_SUFFIX, NAME)
static Metadata * upgradeLoopArgument(Metadata *MD)
static Intrinsic::ID shouldUpgradeNVPTXMBarrierInitIntrinsic(StringRef Name)
static bool isXYZ(StringRef S)
static bool upgradeIntrinsicFunction1(Function *F, Function *&NewFn, bool CanUpgradeDebugIntrinsicsToRecords)
static Value * upgradeX86PSLLDQIntrinsics(IRBuilder<> &Builder, Value *Op, unsigned Shift)
static Intrinsic::ID shouldUpgradeNVPTXSharedClusterIntrinsic(Function *F, StringRef Name)
static Value * upgradeVPIntrinsicCall(StringRef Name, CallBase *CI, IRBuilder<> &Builder)
static std::optional< unsigned > getNVPTXTMAReductionOp(StringRef Name)
static Intrinsic::ID shouldUpgradeNVPTXTMAReductionIntrinsics(StringRef Name)
static bool upgradeRetainReleaseMarker(Module &M)
This checks for objc retain release marker which should be upgraded.
static Value * upgradeX86vpcom(IRBuilder<> &Builder, CallBase &CI, unsigned Imm, bool IsSigned)
static Value * upgradeMaskToInt(IRBuilder<> &Builder, CallBase &CI)
static bool convertIntrinsicValidType(StringRef Name, const FunctionType *FuncTy)
static Value * upgradeX86Rotate(IRBuilder<> &Builder, CallBase &CI, bool IsRotateRight)
static bool upgradeX86MultiplyAddBytes(Function *F, Intrinsic::ID IID, Function *&NewFn)
static Intrinsic::ID getFunctionalIntrinsicIDForVP(StringRef Name)
static void setFunctionAttrIfNotSet(Function &F, StringRef FnAttrName, StringRef Value)
static Intrinsic::ID shouldUpgradeNVPTXBF16Intrinsic(StringRef Name)
static bool upgradeSingleNVVMAnnotation(GlobalValue *GV, StringRef K, const Metadata *V)
static MDNode * unwrapMAVOp(CallBase *CI, unsigned Op)
Helper to unwrap intrinsic call MetadataAsValue operands.
static MDString * upgradeLoopTag(LLVMContext &C, StringRef OldTag)
static ICmpInst::Predicate getVPIntPredicateFromMD(const Value *Op)
static void upgradeNVVMFnVectorAttr(const StringRef Attr, const char DimC, GlobalValue *GV, const Metadata *V)
static bool upgradeX86MaskedFPCompare(Function *F, Intrinsic::ID IID, Function *&NewFn)
static Value * upgradeX86ALIGNIntrinsics(IRBuilder<> &Builder, Value *Op0, Value *Op1, Value *Shift, Value *Passthru, Value *Mask, bool IsVALIGN)
static Value * upgradeAbs(IRBuilder<> &Builder, CallBase &CI)
static bool shouldUpgradeVPIntrinsic(StringRef Name)
static Value * emitX86Select(IRBuilder<> &Builder, Value *Mask, Value *Op0, Value *Op1)
static Value * upgradeAArch64IntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
#define G2S_CTA_ID(ID_SUFFIX, NAME)
static Value * upgradeMaskedMove(IRBuilder<> &Builder, CallBase &CI)
static const BooleanLoopTags * getOldBooleanLoopTags(const MDTuple *T)
Return the replacement tags if T still uses a removed two-operand form.
static bool upgradeX86IntrinsicFunction(Function *F, StringRef Name, Function *&NewFn)
static Value * applyX86MaskOn1BitsVec(IRBuilder<> &Builder, Value *Vec, Value *Mask)
static Intrinsic::ID shouldUpgradeNVPTXTcgen05AllocDeallocIntrinsic(Function *F, StringRef Name)
static std::optional< StringRef > getModuleFlagNameSafely(const MDNode &Flag)
static bool consumeNVVMPtrAddrSpace(StringRef &Name)
static Metadata * makeBooleanLoopNode(LLVMContext &C, const BooleanLoopTags &Tags, const MDOperand &Op)
Build the single-operand node that replaces a boolean operand: nonzero selects the enable tag,...
#define G2S_CLUSTER_CASE(ID_SUFFIX, NAME)
static bool shouldUpgradeX86Intrinsic(Function *F, StringRef Name)
static Value * upgradeX86PSRLDQIntrinsics(IRBuilder<> &Builder, Value *Op, unsigned Shift)
static unsigned getFunctionalOpcodeForVP(StringRef Name)
static Intrinsic::ID shouldUpgradeNVPTXTMAG2SIntrinsics(Function *F, StringRef Name, SmallVectorImpl< Type * > &OvlTys)
static Intrinsic::ID shouldUpgradeNVPTXTcgen05CommitSharedIntrinsic(Function *F, StringRef Name)
static std::optional< std::pair< Intrinsic::ID, RoundingMode > > getNVVMFAddUpgrade(StringRef Name)
static bool isOldLoopArgument(Metadata *MD)
static Value * upgradeARMIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static bool upgradeX86IntrinsicsWith8BitMask(Function *F, Intrinsic::ID IID, Function *&NewFn)
static Value * upgradeVectorSplice(CallBase *CI, IRBuilder<> &Builder)
static Value * upgradeAMDGCNIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static Value * upgradeMaskedLoad(IRBuilder<> &Builder, Value *Ptr, Value *Passthru, Value *Mask, bool Aligned)
static Metadata * unwrapMAVMetadataOp(CallBase *CI, unsigned Op)
Helper to unwrap Metadata MetadataAsValue operands, such as the Value field.
static bool upgradeX86BF16Intrinsic(Function *F, Intrinsic::ID IID, Function *&NewFn)
static bool upgradeArmOrAarch64IntrinsicFunction(bool IsArm, Function *F, StringRef Name, Function *&NewFn)
static bool upgradeIntrinsicCallWithDefaultArgs(CallBase *CI, Function *NewFn, IRBuilder<> &Builder)
static Value * getX86MaskVec(IRBuilder<> &Builder, Value *Mask, unsigned NumElts)
static Value * emitX86ScalarSelect(IRBuilder<> &Builder, Value *Mask, Value *Op0, Value *Op1)
static bool upgradeIntrinsicWithDefaultArgs(Function *F, Function *&NewFn)
static Value * upgradeX86ConcatShift(IRBuilder<> &Builder, CallBase &CI, bool IsShiftRight, bool ZeroMask)
static void rename(GlobalValue *GV)
static bool upgradePTESTIntrinsic(Function *F, Intrinsic::ID IID, Function *&NewFn)
static bool upgradeX86BF16DPIntrinsic(Function *F, Intrinsic::ID IID, Function *&NewFn)
#define NVVM_TMA_G2S_MODES(M)
static cl::opt< bool > DisableAutoUpgradeDebugInfo("disable-auto-upgrade-debug-info", cl::desc("Disable autoupgrade of debug info"))
static Value * upgradeMaskedCompare(IRBuilder<> &Builder, CallBase &CI, unsigned CC, bool Signed)
static Value * upgradeX86BinaryIntrinsics(IRBuilder<> &Builder, CallBase &CI, Intrinsic::ID IID)
static Value * upgradeNVVMIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static Intrinsic::ID shouldUpgradeNVPTXBulkG2SClusterIntrinsic(Function *F, StringRef Name, SmallVectorImpl< Type * > &OvlTys)
static Value * upgradeX86MaskedShift(IRBuilder<> &Builder, CallBase &CI, Intrinsic::ID IID)
static bool upgradeAVX512MaskToSelect(StringRef Name, IRBuilder<> &Builder, CallBase &CI, Value *&Rep)
static void upgradeDbgIntrinsicToDbgRecord(StringRef Name, CallBase *CI)
Convert debug intrinsic calls to non-instruction debug records.
static void ConvertFunctionAttr(Function &F, bool Set, StringRef FnAttrName)
static Value * upgradePMULDQ(IRBuilder<> &Builder, CallBase &CI, bool IsSigned)
static void reportFatalUsageErrorWithCI(StringRef reason, CallBase *CI)
static Value * upgradeMaskedStore(IRBuilder<> &Builder, Value *Ptr, Value *Data, Value *Mask, bool Aligned)
static Intrinsic::ID shouldUpgradeNVPTXBulkG2SCTAIntrinsic(Function *F, StringRef Name)
static Intrinsic::ID shouldUpgradeNVPTXTMAG2SCTAIntrinsics(Function *F, StringRef Name)
static Value * upgradeConvertIntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
#define G2S_CTA_CASE(ID_SUFFIX, NAME)
static bool upgradeX86MultiplyAddWords(Function *F, Intrinsic::ID IID, Function *&NewFn)
static bool upgradePtrauthInitFiniArrays(Module &M)
static Value * upgradeX86IntrinsicCall(StringRef Name, CallBase *CI, Function *F, IRBuilder<> &Builder)
static FCmpInst::Predicate getVPFPPredicateFromMD(const Value *Op)
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
@ Enable
This file contains constants used for implementing Dwarf debug support.
Module.h This file contains the declarations for the Module class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static bool isZero(Value *V, const DataLayout &DL, DominatorTree *DT, AssumptionCache *AC)
Definition Lint.cpp:540
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define R2(n)
This file contains the declarations for metadata subclasses.
#define T
#define T1
NVPTX address space definition.
uint64_t High
This file contains the definitions of the enumerations and flags associated with NVVM Intrinsics,...
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file contains some functions that are useful when dealing with strings.
This file implements the StringSwitch template, which mimics a switch() statement whose cases are str...
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
LocallyHashedType DenseMapInfo< LocallyHashedType >::Empty
static const X86InstrFMA3Group Groups[]
Value * RHS
Value * LHS
Class for arbitrary precision integers.
Definition APInt.h:78
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Class to represent array types.
static LLVM_ABI ArrayType * get(Type *ElementType, uint64_t NumElements)
This static method is the primary way to construct an ArrayType.
Type * getElementType() const
an instruction that atomically reads a memory location, combines it with another value,...
void setVolatile(bool V)
Specify whether this is a volatile RMW or not.
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Add
*p = old + v
@ FAdd
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ Min
*p = old <signed v ? old : v
@ And
*p = old & v
@ Xor
*p = old ^ v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
bool isFloatingPointOperation() const
This class stores enough information to efficiently remove some attributes from an existing AttrBuild...
AttributeMask & addAttribute(Attribute::AttrKind Val)
Add an attribute to the mask.
Functions, function parameters, and return types can have attributes to indicate how they should be t...
Definition Attributes.h:106
static LLVM_ABI Attribute getWithStackAlignment(LLVMContext &Context, Align Alignment)
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
void setCallingConv(CallingConv::ID CC)
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
CallingConv::ID getCallingConv() const
Value * getCalledOperand() const
void setAttributes(AttributeList A)
Set the attributes for this call.
Value * getArgOperand(unsigned i) const
FunctionType * getFunctionType() const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
void setCalledOperand(Value *V)
unsigned arg_size() const
AttributeList getAttributes() const
Return the attributes for this call.
void setCalledFunction(Function *Fn)
Sets the function called, including updating the function type.
This class represents a function call, abstracting a target machine's calling convention.
void setTailCallKind(TailCallKind TCK)
static LLVM_ABI CastInst * Create(Instruction::CastOps, Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Provides a way to construct any of the CastInst subclasses using an opcode instead of the subclass's ...
static LLVM_ABI bool castIsValid(Instruction::CastOps op, Type *SrcTy, Type *DstTy)
This method can be used to determine if a cast from SrcTy to DstTy using Opcode op is valid or not.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ FCMP_OEQ
0 0 0 1 True if ordered and equal
Definition InstrTypes.h:743
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ FCMP_OLT
0 1 0 0 True if ordered and less than
Definition InstrTypes.h:746
@ FCMP_ULE
1 1 0 1 True if unordered, less than, or equal
Definition InstrTypes.h:755
@ FCMP_OGT
0 0 1 0 True if ordered and greater than
Definition InstrTypes.h:744
@ FCMP_OGE
0 0 1 1 True if ordered and greater than or equal
Definition InstrTypes.h:745
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ FCMP_ULT
1 1 0 0 True if unordered or less than
Definition InstrTypes.h:754
@ FCMP_ONE
0 1 1 0 True if ordered and operands are unequal
Definition InstrTypes.h:748
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_UGT
1 0 1 0 True if unordered or greater than
Definition InstrTypes.h:752
@ FCMP_OLE
0 1 0 1 True if ordered and less than or equal
Definition InstrTypes.h:747
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_NE
not equal
Definition InstrTypes.h:762
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ FCMP_UNE
1 1 1 0 True if unordered or not equal
Definition InstrTypes.h:756
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UGE
1 0 1 1 True if unordered, greater than, or equal
Definition InstrTypes.h:753
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
static LLVM_ABI ConstantAggregateZero * get(Type *Ty)
static LLVM_ABI Constant * get(ArrayType *T, ArrayRef< Constant * > V)
static ConstantAsMetadata * get(Constant *C)
Definition Metadata.h:548
static LLVM_ABI Constant * getIntToPtr(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static LLVM_ABI Constant * getPointerCast(Constant *C, Type *Ty)
Create a BitCast, AddrSpaceCast, or a PtrToInt cast constant expression.
static LLVM_ABI Constant * getPtrToInt(Constant *C, Type *Ty, bool OnlyIfReduced=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
Definition Constants.h:219
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
static LLVM_ABI ConstantPointerNull * get(PointerType *T)
Static factory methods - Return objects of the specified value.
static LLVM_ABI Constant * get(StructType *T, ArrayRef< Constant * > V)
StructType * getType() const
Specialization - reduce amount of casting.
Definition Constants.h:661
static LLVM_ABI ConstantTokenNone * get(LLVMContext &Context)
Return the ConstantTokenNone.
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
DWARF expression.
static LLVM_ABI DIExpression * append(const DIExpression *Expr, ArrayRef< uint64_t > Ops)
Append the opcodes Ops to DIExpr.
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
static LLVM_ABI DbgLabelRecord * createUnresolvedDbgLabelRecord(MDNode *Label)
For use during parsing; creates a DbgLabelRecord from as-of-yet unresolved MDNodes.
Base class for non-instruction debug metadata records that have positions within IR.
void setDebugLoc(DebugLoc Loc)
static LLVM_ABI DbgVariableRecord * createUnresolvedDbgVariableRecord(LocationType Type, Metadata *Val, MDNode *Variable, MDNode *Expression, MDNode *AssignID, Metadata *Address, MDNode *AddressExpression)
Used to create DbgVariableRecords during parsing, where some metadata references may still be unresol...
Diagnostic information for debug metadata version reporting.
Diagnostic information for stripping invalid debug metadata.
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
void setApproxFunc(bool B=true)
Definition FMF.h:93
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
Class to represent function types.
unsigned getNumParams() const
Return the number of fixed parameters this function type requires.
Type * getParamType(unsigned i) const
Parameter type accessors.
Type * getReturnType() const
static LLVM_ABI FunctionType * get(Type *Result, ArrayRef< Type * > Params, bool isVarArg)
This static method is the primary way of constructing a FunctionType.
static Function * Create(FunctionType *Ty, LinkageTypes Linkage, unsigned AddrSpace, const Twine &N="", Module *M=nullptr)
Definition Function.h:169
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
Intrinsic::ID getIntrinsicID() const LLVM_READONLY
getIntrinsicID - This method returns the ID number of the specified function, or Intrinsic::not_intri...
Definition Function.h:247
const Function & getFunction() const
Definition Function.h:167
void eraseFromParent()
eraseFromParent - This method unlinks 'this' from the containing module and deletes it.
Definition Function.cpp:451
size_t arg_size() const
Definition Function.h:886
Type * getReturnType() const
Returns the type of the ret val.
Definition Function.h:217
Argument * getArg(unsigned i) const
Definition Function.h:871
static LLVM_ABI GUID getGUIDAssumingExternalLinkage(StringRef GlobalName)
Return a 64-bit global unique ID constructed from the name of a global symbol.
Definition Globals.cpp:80
LinkageTypes getLinkage() const
uint64_t GUID
Declare a type to represent a global unique identifier for a global value.
static StringRef dropLLVMManglingEscape(StringRef Name)
If the given string begins with the GlobalValue name mangling escape character '\1',...
Type * getValueType() const
const Constant * getInitializer() const
getInitializer - Return the initializer for this global variable.
bool hasInitializer() const
Definitions have initializers, declarations don't.
PointerType * getPtrTy(unsigned AddrSpace=0)
Fetch the type representing a pointer.
Definition IRBuilder.h:556
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2901
Base class for instruction visitors.
Definition InstVisitor.h:78
bool isCast() const
const DebugLoc & getDebugLoc() const
Return the debug location for this node as a DebugLoc.
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
bool isBinaryOp() const
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
bool isUnaryOp() const
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI SyncScope::ID getOrInsertSyncScopeID(StringRef SSN)
getOrInsertSyncScopeID - Maps synchronization scope name to synchronization scope ID.
An instruction for reading from memory.
LLVM_ABI MDNode * createRange(const APInt &Lo, const APInt &Hi)
Return metadata describing the range [Lo, Hi).
Definition MDBuilder.cpp:96
Metadata node.
Definition Metadata.h:1081
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1437
op_iterator op_end() const
Definition Metadata.h:1431
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1579
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1443
op_iterator op_begin() const
Definition Metadata.h:1427
LLVMContext & getContext() const
Definition Metadata.h:1245
Tracking metadata reference owned by Metadata.
Definition Metadata.h:902
A single uniqued string.
Definition Metadata.h:733
LLVM_ABI StringRef getString() const
Definition Metadata.cpp:615
static LLVM_ABI MDString * get(LLVMContext &Context, StringRef Str)
Definition Metadata.cpp:597
Tuple of metadata.
Definition Metadata.h:1496
static MDTuple * get(LLVMContext &Context, ArrayRef< Metadata * > MDs)
Definition Metadata.h:1525
Metadata wrapper in the Value hierarchy.
Definition Metadata.h:184
static LLVM_ABI MetadataAsValue * get(LLVMContext &Context, Metadata *MD)
Definition Metadata.cpp:107
Root of the metadata hierarchy.
Definition Metadata.h:64
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
ModFlagBehavior
This enumeration defines the supported behaviors of module flags.
Definition Module.h:118
@ Override
Uses the specified value, regardless of the behavior or value of the other module.
Definition Module.h:139
@ Error
Emits an error if two values disagree, otherwise the resulting value is that of the operands.
Definition Module.h:121
@ Min
Takes the min of the two values, which are required to be integers.
Definition Module.h:153
@ Max
Takes the max of the two values, which are required to be integers.
Definition Module.h:150
A tuple of MDNodes.
Definition Metadata.h:1797
LLVM_ABI void setOperand(unsigned I, MDNode *New)
LLVM_ABI MDNode * getOperand(unsigned i) const
LLVM_ABI unsigned getNumOperands() const
LLVM_ABI void clearOperands()
Drop all references to this node's operands.
iterator_range< op_iterator > operands()
Definition Metadata.h:1893
LLVM_ABI void addOperand(MDNode *M)
ArrayRef< InputTy > inputs() const
StringRef getTag() const
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
LLVM_ABI bool match(StringRef String, SmallVectorImpl< StringRef > *Matches=nullptr, std::string *Error=nullptr) const
matches - Match the regex against a given String.
Definition Regex.cpp:84
static LLVM_ABI ScalableVectorType * get(Type *ElementType, unsigned MinNumElts)
Definition Type.cpp:865
ArrayRef< int > getShuffleMask() const
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
static constexpr size_t npos
Definition StringRef.h:58
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
Definition StringRef.h:597
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
Definition StringRef.h:635
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
size_t find(char C, size_t From=0) const
Search for the first character C in the string.
Definition StringRef.h:290
StringRef trim(char Char) const
Return string with consecutive Char characters starting from the left and right removed.
Definition StringRef.h:850
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
StringSwitch & StartsWith(StringLiteral S, T Value)
StringSwitch & Cases(std::initializer_list< StringLiteral > CaseStrings, T Value)
Class to represent struct types.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
Definition Type.cpp:467
unsigned getNumElements() const
Random access to the elements.
Type * getElementType(unsigned N) const
The TimeTraceScope is a helper class to call the begin and end functions of the time trace profiler.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Definition Type.h:155
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
static LLVM_ABI IntegerType * getInt8Ty(LLVMContext &C)
Definition Type.cpp:297
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:298
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isPtrOrPtrVectorTy() const
Return true if this is a pointer type or a vector of pointer types.
Definition Type.h:280
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
Definition Type.h:222
static LLVM_ABI Type * getFloatTy(LLVMContext &C)
Definition Type.cpp:276
static LLVM_ABI Type * getBFloatTy(LLVMContext &C)
Definition Type.cpp:275
static LLVM_ABI Type * getHalfTy(LLVMContext &C)
Definition Type.cpp:274
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void print(raw_ostream &O, bool IsForDebug=false) const
Implement operator<< on Value.
LLVM_ABI void setName(const Twine &Name)
Change the name of the value.
Definition Value.cpp:394
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
iterator_range< user_iterator > users()
Definition Value.h:428
LLVM_ABI const Value * stripPointerCasts() const
Strip off pointer casts, all-zero GEPs and address space casts.
Definition Value.cpp:712
bool use_empty() const
Definition Value.h:348
bool hasName() const
Definition Value.h:263
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
static VectorType * getInteger(VectorType *VTy)
This static method gets a VectorType with the same number of elements as the input type,...
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
A raw_ostream that writes to an SmallVector or SmallString.
StringRef str() const
Return a StringRef for the vector contents.
CallInst * Call
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ LOCAL_ADDRESS
Address space for local memory.
@ FLAT_ADDRESS
Address space for flat memory.
@ PRIVATE_ADDRESS
Address space for private memory.
@ PTX_Kernel
Call to a PTX kernel. Passes all arguments in parameter space.
std::optional< ABIType > parseABIType(StringRef S)
Parse the string spelling used by the "float-abi" IR module flag into an ABIType.
Definition CodeGen.h:167
LLVM_ABI std::optional< Function * > remangleIntrinsicFunction(Function *F)
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI AttributeList getAttributes(LLVMContext &C, ID id, FunctionType *FT)
Return the attributes for an intrinsic.
LLVM_ABI bool isOverloaded(ID id)
Returns true if the intrinsic can be overloaded.
LLVM_ABI FunctionType * getType(LLVMContext &Context, ID id, ArrayRef< Type * > OverloadTys={})
Return the function type for an intrinsic.
LLVM_ABI bool isSignatureValid(Intrinsic::ID ID, FunctionType *FT, SmallVectorImpl< Type * > &OverloadTys, raw_ostream &OS=nulls())
Returns true if FT is a valid function type for intrinsic ID.
LLVM_ABI bool hasStructReturnType(ID id)
Returns true if id has a struct return type.
LLVM_ABI std::pair< unsigned, ArrayRef< uint64_t > > getAllDefaultArgValues(ID IID)
Returns the first default argument index and an ArrayRef of all default values for the trailing param...
constexpr StringLiteral GridConstant("nvvm.grid_constant")
constexpr StringLiteral MaxNTID("nvvm.maxntid")
constexpr StringLiteral MaxNReg("nvvm.maxnreg")
constexpr StringLiteral MinCTASm("nvvm.minctasm")
constexpr StringLiteral ReqNTID("nvvm.reqntid")
constexpr StringLiteral MaxClusterRank("nvvm.maxclusterrank")
constexpr StringLiteral ClusterDim("nvvm.cluster_dim")
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > dyn_extract_or_null(Y &&MD)
Extract a Value from Metadata, if any, allowing null.
Definition Metadata.h:720
std::enable_if_t< detail::IsValidPointer< X, Y >::value, bool > hasa(Y &&MD)
Check whether Metadata has a Value.
Definition Metadata.h:662
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > dyn_extract(Y &&MD)
Extract a Value from Metadata, if any.
Definition Metadata.h:707
std::enable_if_t< detail::IsValidPointer< X, Y >::value, X * > extract(Y &&MD)
Extract a Value from Metadata.
Definition Metadata.h:679
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
@ Length
Definition DWP.cpp:577
LLVM_ABI void UpgradeIntrinsicCall(CallBase *CB, Function *NewFn)
This is the complement to the above, replacing a specific call to an intrinsic function with a call t...
LLVM_ABI void UpgradeSectionAttributes(Module &M)
auto size(R &&Range, std::enable_if_t< std::is_base_of< std::random_access_iterator_tag, typename std::iterator_traits< decltype(Range.begin())>::iterator_category >::value, void > *=nullptr)
Get the size of a range.
Definition STLExtras.h:1685
LLVM_ABI void UpgradeInlineAsmString(std::string *AsmStr)
Upgrade comment in call to inline asm that represents an objc retain release marker.
bool isValidAtomicOrdering(Int I)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
StringRef getLongDoubleFormatName(LongDoubleFormat Format)
Returns the IR floating-point type name for a LongDoubleFormat.
Definition CodeGen.h:126
LongDoubleFormat
The floating-point format used for the target's "long double" type.
Definition CodeGen.h:117
LLVM_ABI bool UpgradeIntrinsicFunction(Function *F, Function *&NewFn, bool CanUpgradeDebugIntrinsicsToRecords=true)
This is a more granular function that simply checks an intrinsic function for upgrading,...
LLVM_ABI MDNode * upgradeInstructionLoopAttachment(MDNode &N)
Upgrade the loop attachment metadata node.
auto dyn_cast_if_present(const Y &Val)
dyn_cast_if_present<X> - Functionally identical to dyn_cast, except that a null (or none in the case ...
Definition Casting.h:732
LLVM_ABI void UpgradeAttributes(AttrBuilder &B)
Upgrade attributes that changed format or kind.
LLVM_ABI void UpgradeCallsToIntrinsic(Function *F)
This is an auto-upgrade hook for any old intrinsic function syntaxes which need to have both the func...
LLVM_ABI void UpgradeNVVMAnnotations(Module &M)
Convert legacy nvvm.annotations metadata to appropriate function attributes.
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
LLVM_ABI bool UpgradeModuleFlags(Module &M)
This checks for module flags which should be upgraded.
std::string utostr(uint64_t X, bool isNeg=false)
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool UpgradeCFIFunctionsMetadata(Module &M)
Upgrade the cfi.functions metadata node by calculating and inserting the GUID for each function entry...
LLVM_ABI void copyModuleAttrToFunctions(Module &M)
Copies module attributes to the functions in the module.
LLVM_ABI void UpgradeOperandBundles(std::vector< OperandBundleDef > &OperandBundles)
Upgrade operand bundles (without knowing about their user instruction).
LLVM_ABI Constant * UpgradeBitCastExpr(unsigned Opc, Constant *C, Type *DestTy)
This is an auto-upgrade for bitcast constant expression between pointers with different address space...
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI std::string UpgradeDataLayoutString(StringRef DL, StringRef Triple)
Upgrade the datalayout string by adding a section for address space pointers.
bool none_of(R &&Range, UnaryPredicate P)
Provide wrappers to std::none_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1769
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
LLVM_ABI MDNode * UpgradeTBAAStructNode(MDNode &TBAAStructNode)
If the given !tbaa.struct node has old-style scalar field tags, return an equivalent node with each f...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI GlobalVariable * UpgradeGlobalVariable(GlobalVariable *GV)
This checks for global variables which should be upgraded.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
LLVM_ABI bool StripDebugInfo(Module &M)
Strip debug info in the module if it exists.
auto drop_end(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the last N elements excluded.
Definition STLExtras.h:323
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Ref
The access may reference the value stored in memory.
Definition ModRef.h:32
std::string join(IteratorT Begin, IteratorT End, StringRef Separator)
Joins the strings in the range [Begin, End), adding Separator between the elements.
const BooleanLoopTags * findBooleanLoopTags(StringRef Name)
Return the replacement tags for the enable tag Name, or nullptr.
OperandBundleDefT< Value * > OperandBundleDef
Definition AutoUpgrade.h:34
LLVM_ABI Instruction * UpgradeBitCastInst(unsigned Opc, Value *V, Type *DestTy, Instruction *&Temp)
This is an auto-upgrade for bitcast between pointers with different address spaces: the instruction i...
@ FAdd
Sum of floats.
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ TowardZero
roundTowardZero.
@ NearestTiesToEven
roundTiesToEven.
@ Dynamic
Denotes mode unknown at compile time.
@ TowardPositive
roundTowardPositive.
@ TowardNegative
roundTowardNegative.
ArrayRef(const T &OneElt) -> ArrayRef< T >
DenormalMode parseDenormalFPAttribute(StringRef Str)
Returns the denormal mode to use for inputs and outputs.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
void erase_if(Container &C, UnaryPredicate P)
Provide a container algorithm similar to C++ Library Fundamentals v2's erase_if which is equivalent t...
Definition STLExtras.h:2208
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
LLVM_ABI bool UpgradeDebugInfo(Module &M)
Check the debug info version number, if it is out-dated, drop the debug info.
LLVM_ABI void UpgradeFunctionAttributes(Function &F)
Correct any IR that is relying on old function attribute behavior.
LLVM_ABI MDNode * UpgradeTBAANode(MDNode &TBAANode)
If the given TBAA tag uses the scalar TBAA format, create a new node corresponding to the upgrade to ...
LLVM_ABI void UpgradeARCRuntime(Module &M)
Convert calls to ARC runtime functions to intrinsic calls and upgrade the old retain release marker t...
@ DEBUG_METADATA_VERSION
Definition Metadata.h:54
@ Default
The result value is uniform if and only if all operands are uniform.
Definition Uniformity.h:20
LLVM_ABI bool verifyModule(const Module &M, raw_ostream *OS=nullptr, bool *BrokenDebugInfo=nullptr)
Check a module for errors.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Single-operand tags replacing a removed two-operand form !
StringLiteral Disable
StringLiteral Enable
Represents the full denormal controls for a function, including the default mode and the f32 specific...
Represent subnormal handling kind for floating point instruction inputs and outputs.
static constexpr DenormalMode getInvalid()
constexpr bool isValid() const
static constexpr DenormalMode getIEEE()
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106