LLVM 24.0.0git
RISCVLegalizerInfo.cpp
Go to the documentation of this file.
1//===-- RISCVLegalizerInfo.cpp ----------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the Machinelegalizer class for RISC-V.
10/// \todo This should be generated by TableGen.
11//===----------------------------------------------------------------------===//
12
13#include "RISCVLegalizerInfo.h"
16#include "RISCVSubtarget.h"
30#include "llvm/IR/Intrinsics.h"
31#include "llvm/IR/IntrinsicsRISCV.h"
32#include "llvm/IR/Type.h"
33
34using namespace llvm;
35using namespace LegalityPredicates;
36using namespace LegalizeMutations;
37using namespace MIPatternMatch;
38
40typeIsLegalIntOrFPVec(unsigned TypeIdx,
41 std::initializer_list<LLT> IntOrFPVecTys,
42 const RISCVSubtarget &ST) {
43 LegalityPredicate P = [=, &ST](const LegalityQuery &Query) {
44 return ST.hasVInstructions() &&
45 (Query.Types[TypeIdx].getScalarSizeInBits() != 64 ||
46 ST.hasVInstructionsI64()) &&
47 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 1 ||
48 ST.getELen() == 64);
49 };
50
51 return all(typeInSet(TypeIdx, IntOrFPVecTys), P);
52}
53
55typeIsLegalBoolVec(unsigned TypeIdx, std::initializer_list<LLT> BoolVecTys,
56 const RISCVSubtarget &ST) {
57 LegalityPredicate P = [=, &ST](const LegalityQuery &Query) {
58 return ST.hasVInstructions() &&
59 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 1 ||
60 ST.getELen() == 64);
61 };
62 return all(typeInSet(TypeIdx, BoolVecTys), P);
63}
64
65static LegalityPredicate typeIsLegalPtrVec(unsigned TypeIdx,
66 std::initializer_list<LLT> PtrVecTys,
67 const RISCVSubtarget &ST) {
68 LegalityPredicate P = [=, &ST](const LegalityQuery &Query) {
69 return ST.hasVInstructions() &&
70 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 1 ||
71 ST.getELen() == 64) &&
72 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 16 ||
73 Query.Types[TypeIdx].getScalarSizeInBits() == 32);
74 };
75 return all(typeInSet(TypeIdx, PtrVecTys), P);
76}
77
79 : STI(ST), XLen(STI.getXLen()), sXLen(LLT::scalar(XLen)) {
80 const LLT sDoubleXLen = LLT::scalar(2 * XLen);
81 const LLT p0 = LLT::pointer(0, XLen);
82 const LLT s1 = LLT::scalar(1);
83 const LLT s8 = LLT::scalar(8);
84 const LLT s16 = LLT::scalar(16);
85 const LLT f16 = LLT::float16();
86 const LLT s32 = LLT::scalar(32);
87 const LLT s64 = LLT::scalar(64);
88 const LLT s128 = LLT::scalar(128);
89
90 const LLT nxv1s1 = LLT::scalable_vector(1, s1);
91 const LLT nxv2s1 = LLT::scalable_vector(2, s1);
92 const LLT nxv4s1 = LLT::scalable_vector(4, s1);
93 const LLT nxv8s1 = LLT::scalable_vector(8, s1);
94 const LLT nxv16s1 = LLT::scalable_vector(16, s1);
95 const LLT nxv32s1 = LLT::scalable_vector(32, s1);
96 const LLT nxv64s1 = LLT::scalable_vector(64, s1);
97
98 const LLT nxv1s8 = LLT::scalable_vector(1, s8);
99 const LLT nxv2s8 = LLT::scalable_vector(2, s8);
100 const LLT nxv4s8 = LLT::scalable_vector(4, s8);
101 const LLT nxv8s8 = LLT::scalable_vector(8, s8);
102 const LLT nxv16s8 = LLT::scalable_vector(16, s8);
103 const LLT nxv32s8 = LLT::scalable_vector(32, s8);
104 const LLT nxv64s8 = LLT::scalable_vector(64, s8);
105
106 const LLT nxv1s16 = LLT::scalable_vector(1, s16);
107 const LLT nxv2s16 = LLT::scalable_vector(2, s16);
108 const LLT nxv4s16 = LLT::scalable_vector(4, s16);
109 const LLT nxv8s16 = LLT::scalable_vector(8, s16);
110 const LLT nxv16s16 = LLT::scalable_vector(16, s16);
111 const LLT nxv32s16 = LLT::scalable_vector(32, s16);
112
113 const LLT nxv1s32 = LLT::scalable_vector(1, s32);
114 const LLT nxv2s32 = LLT::scalable_vector(2, s32);
115 const LLT nxv4s32 = LLT::scalable_vector(4, s32);
116 const LLT nxv8s32 = LLT::scalable_vector(8, s32);
117 const LLT nxv16s32 = LLT::scalable_vector(16, s32);
118
119 const LLT nxv1s64 = LLT::scalable_vector(1, s64);
120 const LLT nxv2s64 = LLT::scalable_vector(2, s64);
121 const LLT nxv4s64 = LLT::scalable_vector(4, s64);
122 const LLT nxv8s64 = LLT::scalable_vector(8, s64);
123
124 const LLT nxv1p0 = LLT::scalable_vector(1, p0);
125 const LLT nxv2p0 = LLT::scalable_vector(2, p0);
126 const LLT nxv4p0 = LLT::scalable_vector(4, p0);
127 const LLT nxv8p0 = LLT::scalable_vector(8, p0);
128 const LLT nxv16p0 = LLT::scalable_vector(16, p0);
129
130 using namespace TargetOpcode;
131
132 auto BoolVecTys = {nxv1s1, nxv2s1, nxv4s1, nxv8s1, nxv16s1, nxv32s1, nxv64s1};
133
134 auto IntOrFPVecTys = {nxv1s8, nxv2s8, nxv4s8, nxv8s8, nxv16s8, nxv32s8,
135 nxv64s8, nxv1s16, nxv2s16, nxv4s16, nxv8s16, nxv16s16,
136 nxv32s16, nxv1s32, nxv2s32, nxv4s32, nxv8s32, nxv16s32,
137 nxv1s64, nxv2s64, nxv4s64, nxv8s64};
138
139 auto PtrVecTys = {nxv1p0, nxv2p0, nxv4p0, nxv8p0, nxv16p0};
140
141 getActionDefinitionsBuilder({G_ADD, G_SUB})
142 .legalFor({sXLen})
143 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
144 .customFor(ST.is64Bit(), {s32})
146 .clampScalar(0, sXLen, sXLen);
147
148 getActionDefinitionsBuilder({G_AND, G_OR, G_XOR})
149 .legalFor({sXLen})
150 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
152 .clampScalar(0, sXLen, sXLen);
153
155 {G_UADDE, G_UADDO, G_USUBE, G_USUBO, G_READ_REGISTER, G_WRITE_REGISTER})
156 .lower();
157
158 getActionDefinitionsBuilder({G_SADDE, G_SADDO, G_SSUBE, G_SSUBO})
159 .minScalar(0, sXLen)
160 .lower();
161
162 // TODO: Use Vector Single-Width Saturating Instructions for vector types.
164 {G_UADDSAT, G_SADDSAT, G_USUBSAT, G_SSUBSAT, G_SSHLSAT, G_USHLSAT})
165 .lower();
166
167 getActionDefinitionsBuilder({G_SHL, G_ASHR, G_LSHR})
168 .legalFor({{sXLen, sXLen}})
169 .customFor(ST.is64Bit(), {{s32, s32}})
170 .widenScalarToNextPow2(0)
171 .clampScalar(1, sXLen, sXLen)
172 .clampScalar(0, sXLen, sXLen);
173
174 getActionDefinitionsBuilder({G_ZEXT, G_SEXT, G_ANYEXT})
175 .legalFor({{s32, s16}})
176 .legalFor(ST.is64Bit(), {{s64, s16}, {s64, s32}})
177 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
178 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST)))
179 .customIf(typeIsLegalBoolVec(1, BoolVecTys, ST))
180 .maxScalar(0, sXLen);
181
182 getActionDefinitionsBuilder(G_TRUNC).alwaysLegal();
183
184 {
185 LegalityPredicate ValidSextInRegWidth = all(sizeIs(0, 64), immIs(0, 32));
186
187 if (STI.hasStdExtZbb())
188 ValidSextInRegWidth =
189 LegalityPredicates::any(ValidSextInRegWidth, immInSet(0, {8, 16}));
190
191 getActionDefinitionsBuilder(G_SEXT_INREG)
192 .legalIf(all(typeIs(0, sXLen), ValidSextInRegWidth))
193 .clampScalar(0, sXLen, sXLen)
194 .lower();
195 }
196
197 // Merge/Unmerge
198 for (unsigned Op : {G_MERGE_VALUES, G_UNMERGE_VALUES}) {
199 auto &MergeUnmergeActions = getActionDefinitionsBuilder(Op);
200 unsigned BigTyIdx = Op == G_MERGE_VALUES ? 0 : 1;
201 unsigned LitTyIdx = Op == G_MERGE_VALUES ? 1 : 0;
202 if (XLen == 32 && ST.hasStdExtD()) {
203 MergeUnmergeActions.legalIf(
204 all(typeIs(BigTyIdx, s64), typeIs(LitTyIdx, s32)));
205 }
206 MergeUnmergeActions.widenScalarToNextPow2(LitTyIdx, XLen)
207 .widenScalarToNextPow2(BigTyIdx, XLen)
208 .clampScalar(LitTyIdx, sXLen, sXLen)
209 .clampScalar(BigTyIdx, sXLen, sXLen);
210 }
211
212 getActionDefinitionsBuilder({G_FSHL, G_FSHR}).lower();
213
214 getActionDefinitionsBuilder({G_ROTR, G_ROTL})
215 .legalFor(ST.hasStdExtZbb() || ST.hasStdExtZbkb(), {{sXLen, sXLen}})
216 .customFor(ST.is64Bit() && (ST.hasStdExtZbb() || ST.hasStdExtZbkb()),
217 {{s32, s32}})
218 .lower();
219
220 getActionDefinitionsBuilder(G_BITREVERSE)
221 .customFor(ST.hasStdExtZbkb(), {s8})
222 .maxScalar(0, sXLen)
223 .lower();
224
225 getActionDefinitionsBuilder(G_BITCAST).legalIf(
227 typeIsLegalBoolVec(0, BoolVecTys, ST)),
229 typeIsLegalBoolVec(1, BoolVecTys, ST))));
230
231 auto &BSWAPActions = getActionDefinitionsBuilder(G_BSWAP);
232 if (ST.hasStdExtZbb() || ST.hasStdExtZbkb())
233 BSWAPActions.legalFor({sXLen}).clampScalar(0, sXLen, sXLen);
234 else
235 BSWAPActions.maxScalar(0, sXLen).lower();
236
237 getActionDefinitionsBuilder(G_CLMUL)
238 .legalFor(ST.hasStdExtZbkc(), {sXLen})
239 .unsupported();
240
241 getActionDefinitionsBuilder(G_CLMULH)
242 .legalFor(ST.hasStdExtZbkc(), {sXLen})
243 .customFor(ST.is64Bit() && ST.hasStdExtZbkc(), {s32})
244 .unsupported();
245
246 // CLMULR is Zbc-only; Zbkc is a subset that has CLMUL/CLMULH but not CLMULR.
247 getActionDefinitionsBuilder(G_CLMULR)
248 .legalFor(ST.hasStdExtZbc(), {sXLen})
249 .customFor(ST.is64Bit() && ST.hasStdExtZbc(), {s32})
250 .unsupported();
251
252 auto &CountZerosActions = getActionDefinitionsBuilder({G_CTLZ, G_CTTZ});
253 auto &CountZerosPoisonActions =
254 getActionDefinitionsBuilder({G_CTLZ_ZERO_POISON, G_CTTZ_ZERO_POISON});
255 if (ST.hasStdExtZbb()) {
256 CountZerosActions.legalFor({{sXLen, sXLen}})
257 .customFor({{s32, s32}})
258 .clampScalar(0, s32, sXLen)
259 .widenScalarToNextPow2(0)
260 .scalarSameSizeAs(1, 0);
261 } else {
262 CountZerosActions.maxScalar(0, sXLen).scalarSameSizeAs(1, 0).lower();
263 CountZerosPoisonActions.maxScalar(0, sXLen).scalarSameSizeAs(1, 0);
264 }
265 CountZerosPoisonActions.lower();
266
267 auto &CountSignActions = getActionDefinitionsBuilder(G_CTLS);
268 if (ST.hasStdExtP()) {
269 CountSignActions.legalFor({{sXLen, sXLen}})
270 .customFor({{s32, s32}})
271 .clampScalar(0, s32, sXLen)
272 .widenScalarToNextPow2(0)
273 .scalarSameSizeAs(1, 0);
274 } else {
275 CountSignActions.maxScalar(0, sXLen).scalarSameSizeAs(1, 0).lower();
276 }
277
278 auto &CTPOPActions = getActionDefinitionsBuilder(G_CTPOP);
279 if (ST.hasStdExtZbb()) {
280 CTPOPActions.legalFor({{sXLen, sXLen}})
281 .clampScalar(0, sXLen, sXLen)
282 .scalarSameSizeAs(1, 0);
283 } else {
284 CTPOPActions.widenScalarToNextPow2(0, /*Min*/ 8)
285 .clampScalar(0, s8, sXLen)
286 .scalarSameSizeAs(1, 0)
287 .lower();
288 }
289
290 getActionDefinitionsBuilder(G_CONSTANT)
291 .legalFor({p0})
292 .legalFor(!ST.is64Bit(), {s32})
293 .customFor(ST.is64Bit(), {s64})
294 .widenScalarToNextPow2(0)
295 .clampScalar(0, sXLen, sXLen);
296
297 // TODO: transform illegal vector types into legal vector type
298 getActionDefinitionsBuilder(G_FREEZE)
299 .legalFor({s16, s32, p0})
300 .legalFor(ST.is64Bit(), {s64})
301 .legalIf(typeIsLegalBoolVec(0, BoolVecTys, ST))
302 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
303 .widenScalarToNextPow2(0)
304 .clampScalar(0, s16, sXLen);
305
306 // TODO: transform illegal vector types into legal vector type
307 // TODO: Merge with G_FREEZE?
308 getActionDefinitionsBuilder(
309 {G_IMPLICIT_DEF, G_CONSTANT_FOLD_BARRIER})
310 .legalFor({s32, sXLen, p0})
311 .legalIf(typeIsLegalBoolVec(0, BoolVecTys, ST))
312 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
313 .widenScalarToNextPow2(0)
314 .clampScalar(0, s32, sXLen);
315
316 getActionDefinitionsBuilder(G_ICMP)
317 .legalFor({{sXLen, sXLen}, {sXLen, p0}})
318 .legalIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST),
319 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST)))
320 .widenScalarOrEltToNextPow2OrMinSize(1, 8)
321 .clampScalar(1, sXLen, sXLen)
322 .clampScalar(0, sXLen, sXLen);
323
324 getActionDefinitionsBuilder(G_SELECT)
325 .legalFor({{s32, sXLen}, {p0, sXLen}})
326 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
327 typeIsLegalBoolVec(1, BoolVecTys, ST)))
328 .legalFor(XLen == 64 || ST.hasStdExtD(), {{s64, sXLen}})
329 .widenScalarToNextPow2(0)
330 .clampScalar(0, s32, (XLen == 64 || ST.hasStdExtD()) ? s64 : s32)
331 .clampScalar(1, sXLen, sXLen);
332
333 auto &LoadActions = getActionDefinitionsBuilder(G_LOAD);
334 auto &StoreActions = getActionDefinitionsBuilder(G_STORE);
335 auto &ExtLoadActions = getActionDefinitionsBuilder({G_SEXTLOAD, G_ZEXTLOAD});
336
337 // Return the alignment needed for scalar memory ops. If unaligned scalar mem
338 // is supported, we only require byte alignment. Otherwise, we need the memory
339 // op to be natively aligned.
340 auto getScalarMemAlign = [&ST](unsigned Size) {
341 return ST.enableUnalignedScalarMem() ? 8 : Size;
342 };
343
344 LoadActions.legalForTypesWithMemDesc(
345 {{s16, p0, s8, getScalarMemAlign(8)},
346 {s32, p0, s8, getScalarMemAlign(8)},
347 {s16, p0, s16, getScalarMemAlign(16)},
348 {s32, p0, s16, getScalarMemAlign(16)},
349 {s32, p0, s32, getScalarMemAlign(32)},
350 {p0, p0, sXLen, getScalarMemAlign(XLen)}});
351 StoreActions.legalForTypesWithMemDesc(
352 {{s16, p0, s8, getScalarMemAlign(8)},
353 {s32, p0, s8, getScalarMemAlign(8)},
354 {s16, p0, s16, getScalarMemAlign(16)},
355 {s32, p0, s16, getScalarMemAlign(16)},
356 {s32, p0, s32, getScalarMemAlign(32)},
357 {p0, p0, sXLen, getScalarMemAlign(XLen)}});
358 ExtLoadActions.legalForTypesWithMemDesc(
359 {{sXLen, p0, s8, getScalarMemAlign(8)},
360 {sXLen, p0, s16, getScalarMemAlign(16)}});
361 if (XLen == 64) {
362 LoadActions.legalForTypesWithMemDesc(
363 {{s64, p0, s8, getScalarMemAlign(8)},
364 {s64, p0, s16, getScalarMemAlign(16)},
365 {s64, p0, s32, getScalarMemAlign(32)},
366 {s64, p0, s64, getScalarMemAlign(64)}});
367 StoreActions.legalForTypesWithMemDesc(
368 {{s64, p0, s8, getScalarMemAlign(8)},
369 {s64, p0, s16, getScalarMemAlign(16)},
370 {s64, p0, s32, getScalarMemAlign(32)},
371 {s64, p0, s64, getScalarMemAlign(64)}});
372 ExtLoadActions.legalForTypesWithMemDesc(
373 {{s64, p0, s32, getScalarMemAlign(32)}});
374 } else if (ST.hasStdExtD()) {
375 LoadActions.legalForTypesWithMemDesc(
376 {{s64, p0, s64, getScalarMemAlign(64)}});
377 StoreActions.legalForTypesWithMemDesc(
378 {{s64, p0, s64, getScalarMemAlign(64)}});
379 }
380
381 // Vector loads/stores.
382 if (ST.hasVInstructions()) {
383 LoadActions.legalForTypesWithMemDesc({{nxv2s8, p0, nxv2s8, 8},
384 {nxv4s8, p0, nxv4s8, 8},
385 {nxv8s8, p0, nxv8s8, 8},
386 {nxv16s8, p0, nxv16s8, 8},
387 {nxv32s8, p0, nxv32s8, 8},
388 {nxv64s8, p0, nxv64s8, 8},
389 {nxv2s16, p0, nxv2s16, 16},
390 {nxv4s16, p0, nxv4s16, 16},
391 {nxv8s16, p0, nxv8s16, 16},
392 {nxv16s16, p0, nxv16s16, 16},
393 {nxv32s16, p0, nxv32s16, 16},
394 {nxv2s32, p0, nxv2s32, 32},
395 {nxv4s32, p0, nxv4s32, 32},
396 {nxv8s32, p0, nxv8s32, 32},
397 {nxv16s32, p0, nxv16s32, 32}});
398 StoreActions.legalForTypesWithMemDesc({{nxv2s8, p0, nxv2s8, 8},
399 {nxv4s8, p0, nxv4s8, 8},
400 {nxv8s8, p0, nxv8s8, 8},
401 {nxv16s8, p0, nxv16s8, 8},
402 {nxv32s8, p0, nxv32s8, 8},
403 {nxv64s8, p0, nxv64s8, 8},
404 {nxv2s16, p0, nxv2s16, 16},
405 {nxv4s16, p0, nxv4s16, 16},
406 {nxv8s16, p0, nxv8s16, 16},
407 {nxv16s16, p0, nxv16s16, 16},
408 {nxv32s16, p0, nxv32s16, 16},
409 {nxv2s32, p0, nxv2s32, 32},
410 {nxv4s32, p0, nxv4s32, 32},
411 {nxv8s32, p0, nxv8s32, 32},
412 {nxv16s32, p0, nxv16s32, 32}});
413
414 if (ST.getELen() == 64) {
415 LoadActions.legalForTypesWithMemDesc({{nxv1s8, p0, nxv1s8, 8},
416 {nxv1s16, p0, nxv1s16, 16},
417 {nxv1s32, p0, nxv1s32, 32}});
418 StoreActions.legalForTypesWithMemDesc({{nxv1s8, p0, nxv1s8, 8},
419 {nxv1s16, p0, nxv1s16, 16},
420 {nxv1s32, p0, nxv1s32, 32}});
421 }
422
423 if (ST.hasVInstructionsI64()) {
424 LoadActions.legalForTypesWithMemDesc({{nxv1s64, p0, nxv1s64, 64},
425 {nxv2s64, p0, nxv2s64, 64},
426 {nxv4s64, p0, nxv4s64, 64},
427 {nxv8s64, p0, nxv8s64, 64}});
428 StoreActions.legalForTypesWithMemDesc({{nxv1s64, p0, nxv1s64, 64},
429 {nxv2s64, p0, nxv2s64, 64},
430 {nxv4s64, p0, nxv4s64, 64},
431 {nxv8s64, p0, nxv8s64, 64}});
432 }
433
434 // we will take the custom lowering logic if we have scalable vector types
435 // with non-standard alignments
436 LoadActions.customIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST));
437 StoreActions.customIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST));
438
439 // Pointers require that XLen sized elements are legal.
440 if (XLen <= ST.getELen()) {
441 LoadActions.customIf(typeIsLegalPtrVec(0, PtrVecTys, ST));
442 StoreActions.customIf(typeIsLegalPtrVec(0, PtrVecTys, ST));
443 }
444 }
445
446 LoadActions.widenScalarToNextPow2(0, /* MinSize = */ 8)
447 .lowerIfMemSizeNotByteSizePow2()
448 .clampScalar(0, s16, sXLen)
449 .lower();
450 StoreActions
451 .clampScalar(0, s16, sXLen)
452 .lowerIfMemSizeNotByteSizePow2()
453 .lower();
454
455 ExtLoadActions.widenScalarToNextPow2(0).clampScalar(0, sXLen, sXLen).lower();
456
457 getActionDefinitionsBuilder({G_PTR_ADD, G_PTRMASK}).legalFor({{p0, sXLen}});
458
459 getActionDefinitionsBuilder(G_PTRTOINT)
460 .legalFor({{sXLen, p0}})
461 .clampScalar(0, sXLen, sXLen);
462
463 getActionDefinitionsBuilder(G_INTTOPTR)
464 .legalFor({{p0, sXLen}})
465 .clampScalar(1, sXLen, sXLen);
466
467 getActionDefinitionsBuilder(G_BR).alwaysLegal();
468
469 getActionDefinitionsBuilder(G_BRCOND).legalFor({sXLen}).minScalar(0, sXLen);
470
471 getActionDefinitionsBuilder(G_BRJT).customFor({{p0, sXLen}});
472
473 getActionDefinitionsBuilder(G_BRINDIRECT).legalFor({p0});
474
475 getActionDefinitionsBuilder(G_PHI)
476 .legalFor({p0, s32, sXLen})
477 .widenScalarToNextPow2(0)
478 .clampScalar(0, s32, sXLen);
479
480 getActionDefinitionsBuilder({G_GLOBAL_VALUE, G_JUMP_TABLE, G_CONSTANT_POOL})
481 .legalFor({p0});
482
483 if (ST.hasStdExtZmmul()) {
484 getActionDefinitionsBuilder(G_MUL)
485 .legalFor({sXLen})
486 .widenScalarToNextPow2(0)
487 .clampScalar(0, sXLen, sXLen);
488
489 // clang-format off
490 getActionDefinitionsBuilder({G_SMULH, G_UMULH})
491 .legalFor({sXLen})
492 .lower();
493 // clang-format on
494
495 getActionDefinitionsBuilder({G_SMULO, G_UMULO}).minScalar(0, sXLen).lower();
496 } else {
497 getActionDefinitionsBuilder(G_MUL)
498 .libcallFor({sXLen, sDoubleXLen})
499 .widenScalarToNextPow2(0)
500 .clampScalar(0, sXLen, sDoubleXLen);
501
502 getActionDefinitionsBuilder({G_SMULH, G_UMULH}).lowerFor({sXLen});
503
504 getActionDefinitionsBuilder({G_SMULO, G_UMULO})
505 .minScalar(0, sXLen)
506 // Widen sXLen to sDoubleXLen so we can use a single libcall to get
507 // the low bits for the mul result and high bits to do the overflow
508 // check.
509 .widenScalarIf(typeIs(0, sXLen),
510 LegalizeMutations::changeTo(0, sDoubleXLen))
511 .lower();
512 }
513
514 if (ST.hasStdExtM()) {
515 getActionDefinitionsBuilder({G_SDIV, G_UDIV, G_UREM})
516 .legalFor({sXLen})
517 .customFor({s32})
518 .libcallFor({sDoubleXLen})
519 .clampScalar(0, s32, sDoubleXLen)
520 .widenScalarToNextPow2(0);
521 getActionDefinitionsBuilder(G_SREM)
522 .legalFor({sXLen})
523 .libcallFor({sDoubleXLen})
524 .clampScalar(0, sXLen, sDoubleXLen)
525 .widenScalarToNextPow2(0);
526 } else {
527 getActionDefinitionsBuilder({G_UDIV, G_SDIV, G_UREM, G_SREM})
528 .libcallFor({sXLen, sDoubleXLen})
529 .clampScalar(0, sXLen, sDoubleXLen)
530 .widenScalarToNextPow2(0);
531 }
532
533 // TODO: Use libcall for sDoubleXLen.
534 getActionDefinitionsBuilder({G_SDIVREM, G_UDIVREM}).lower();
535
536 getActionDefinitionsBuilder(G_ABS)
537 .customFor(ST.hasStdExtZbb(), {sXLen})
538 .minScalar(ST.hasStdExtZbb(), 0, sXLen)
539 .lower();
540
541 getActionDefinitionsBuilder({G_ABDS, G_ABDU})
542 .minScalar(ST.hasStdExtZbb(), 0, sXLen)
543 .lower();
544
545 getActionDefinitionsBuilder({G_UMAX, G_UMIN, G_SMAX, G_SMIN})
546 .legalFor(ST.hasStdExtZbb(), {sXLen})
547 .minScalar(ST.hasStdExtZbb(), 0, sXLen)
548 .lower();
549
550 getActionDefinitionsBuilder({G_SCMP, G_UCMP}).lower();
551
552 getActionDefinitionsBuilder(G_FRAME_INDEX).legalFor({p0});
553
554 getActionDefinitionsBuilder({G_MEMCPY, G_MEMMOVE, G_MEMSET}).libcall();
555
556 getActionDefinitionsBuilder({G_MEMCPY_INLINE, G_MEMSET_INLINE}).lower();
557
558 getActionDefinitionsBuilder({G_DYN_STACKALLOC, G_STACKSAVE, G_STACKRESTORE})
559 .lower();
560
561 // On RV64 the 64-bit counter CSRs (cycle/time) are read directly. On RV32
562 // they are custom-legally lowered to a re-read-the-high-half loop (see
563 // legalizeReadCounter).
564 getActionDefinitionsBuilder({G_READCYCLECOUNTER, G_READSTEADYCOUNTER})
565 .legalFor(ST.is64Bit(), {s64})
566 .customFor(!ST.is64Bit(), {s64});
567
568 // FP Operations
569
570 // FIXME: Support s128 for rv32 when libcall handling is able to use sret.
571 getActionDefinitionsBuilder({G_FADD, G_FSUB, G_FMUL, G_FDIV, G_FMA, G_FSQRT,
572 G_FMAXNUM, G_FMINNUM, G_FMAXIMUMNUM,
573 G_FMINIMUMNUM})
574 .legalFor(ST.hasStdExtF(), {s32})
575 .legalFor(ST.hasStdExtD(), {s64})
576 .legalFor(ST.hasStdExtZfh(), {s16})
577 .libcallFor({s32, s64})
578 .libcallFor(ST.is64Bit(), {s128});
579
580 getActionDefinitionsBuilder({G_FNEG, G_FABS})
581 .legalFor(ST.hasStdExtF(), {s32})
582 .legalFor(ST.hasStdExtD(), {s64})
583 .legalFor(ST.hasStdExtZfh(), {s16})
584 .lowerFor({s32, s64, s128});
585
586 getActionDefinitionsBuilder(G_FREM)
587 .libcallFor({s32, s64})
588 .libcallFor(ST.is64Bit(), {s128})
589 .minScalar(0, s32)
590 .scalarize(0);
591
592 getActionDefinitionsBuilder(G_FCOPYSIGN)
593 .legalFor(ST.hasStdExtF(), {{s32, s32}})
594 .legalFor(ST.hasStdExtD(), {{s64, s64}, {s32, s64}, {s64, s32}})
595 .legalFor(ST.hasStdExtZfh(), {{s16, s16}, {s16, s32}, {s32, s16}})
596 .legalFor(ST.hasStdExtZfh() && ST.hasStdExtD(), {{s16, s64}, {s64, s16}})
597 .lower();
598
599 // FIXME: Use Zfhmin.
600 getActionDefinitionsBuilder(G_FPTRUNC)
601 .legalFor(ST.hasStdExtD(), {{s32, s64}})
602 .legalFor(ST.hasStdExtZfh(), {{s16, s32}})
603 .legalFor(ST.hasStdExtZfh() && ST.hasStdExtD(), {{s16, s64}})
604 .libcallFor({{s32, s64}})
605 .libcallFor(ST.is64Bit(), {{s32, s128}, {s64, s128}});
606 getActionDefinitionsBuilder(G_FPEXT)
607 .legalFor(ST.hasStdExtD(), {{s64, s32}})
608 .legalFor(ST.hasStdExtZfhmin(), {{s32, s16}})
609 .legalFor(ST.hasStdExtZfh() && ST.hasStdExtD(), {{s64, s16}})
610 .libcallFor(!ST.hasStdExtZfhmin(), {{s32, s16}})
611 .libcallFor({{s64, s32}})
612 .libcallFor(ST.is64Bit(), {{s128, s32}, {s128, s64}});
613
614 getActionDefinitionsBuilder(G_FCMP)
615 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
616 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
617 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
618 .clampScalar(0, sXLen, sXLen)
619 .libcallFor({{sXLen, s32}, {sXLen, s64}})
620 .libcallFor(ST.is64Bit(), {{sXLen, s128}});
621
622 // TODO: Support vector version of G_IS_FPCLASS.
623 getActionDefinitionsBuilder(G_IS_FPCLASS)
624 .customFor(ST.hasStdExtF(), {{s1, s32}})
625 .customFor(ST.hasStdExtD(), {{s1, s64}})
626 .customFor(ST.hasStdExtZfh(), {{s1, s16}})
627 .lower();
628
629 getActionDefinitionsBuilder(G_FCONSTANT)
630 .legalFor(ST.hasStdExtF(), {s32})
631 .legalFor(ST.hasStdExtD(), {s64})
632 .legalFor(ST.hasStdExtZfh(), {s16})
633 .customFor(!ST.is64Bit(), {s32})
634 .customFor(ST.is64Bit(), {s32, s64})
635 .lowerFor({s64, s128});
636
637 getActionDefinitionsBuilder({G_FPTOSI, G_FPTOUI})
638 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
639 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
640 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
641 .customFor(ST.is64Bit() && ST.hasStdExtF(), {{s32, s32}})
642 .customFor(ST.is64Bit() && ST.hasStdExtD(), {{s32, s64}})
643 .customFor(ST.is64Bit() && ST.hasStdExtZfh(), {{s32, s16}})
644 .widenScalarToNextPow2(0)
645 .minScalar(0, s32)
646 // The magnitude of a half is at most 65504, so with Zfh use fcvt.w[u].h
647 // and extend the i32 result. Without Zfh, use the libcalls.
648 .libcallFor(!ST.hasStdExtZfh(), {{s64, f16}})
649 .narrowScalarFor({{s64, f16}}, changeTo(0, s32))
650 .libcallFor({{s32, s32}, {s64, s32}, {s32, s64}, {s64, s64}})
651 .libcallFor(ST.is64Bit(), {{s32, s128}, {s64, s128}}) // FIXME RV32.
652 .libcallFor(ST.is64Bit(), {{s128, s32}, {s128, s64}, {s128, s128}});
653
654 getActionDefinitionsBuilder({G_LROUND, G_LLROUND})
655 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
656 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
657 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
658 .customFor(ST.is64Bit() && ST.hasStdExtF(), {{s32, s32}})
659 .customFor(ST.is64Bit() && ST.hasStdExtD(), {{s32, s64}})
660 .customFor(ST.is64Bit() && ST.hasStdExtZfh(), {{s32, s16}})
661 .widenScalarIf(typeIs(1, s16), LegalizeMutations::changeTo(1, s32))
662 .libcallFor({{s32, s32},
663 {s64, s32},
664 {s32, s64},
665 {s64, s64},
666 {s32, s128},
667 {s64, s128}});
668
669 getActionDefinitionsBuilder({G_INTRINSIC_LRINT, G_INTRINSIC_LLRINT})
670 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
671 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
672 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
673 .minScalar(0, sXLen)
674 .widenScalarIf(typeIs(1, s16), LegalizeMutations::changeTo(1, s32))
675 .libcallFor({{s32, s32},
676 {s64, s32},
677 {s32, s64},
678 {s64, s64},
679 {s32, s128},
680 {s64, s128}});
681
682 getActionDefinitionsBuilder({G_SITOFP, G_UITOFP})
683 .legalFor(ST.hasStdExtF(), {{s32, sXLen}})
684 .legalFor(ST.hasStdExtD(), {{s64, sXLen}})
685 .legalFor(ST.hasStdExtZfh(), {{s16, sXLen}})
686 .widenScalarToNextPow2(1)
687 // Promote to XLen if the operation is legal.
688 .widenScalarIf(
689 [=, &ST](const LegalityQuery &Query) {
690 return Query.Types[0].isScalar() && Query.Types[1].isScalar() &&
691 (Query.Types[1].getSizeInBits() < ST.getXLen()) &&
692 ((ST.hasStdExtF() && Query.Types[0].getSizeInBits() == 32) ||
693 (ST.hasStdExtD() && Query.Types[0].getSizeInBits() == 64) ||
694 (ST.hasStdExtZfh() &&
695 Query.Types[0].getSizeInBits() == 16));
696 },
698 // Otherwise only promote to s32 since we have si libcalls.
699 .minScalar(1, s32)
700 .libcallFor({{s32, s32}, {s64, s32}, {s32, s64}, {s64, s64}})
701 .libcallFor(ST.is64Bit(), {{s128, s32}, {s128, s64}}) // FIXME RV32.
702 .libcallFor(ST.is64Bit(), {{s32, s128}, {s64, s128}, {s128, s128}});
703
704 // FIXME: We can do custom inline expansion like SelectionDAG.
705 getActionDefinitionsBuilder({G_FCEIL, G_FFLOOR, G_FRINT, G_FNEARBYINT,
706 G_INTRINSIC_TRUNC, G_INTRINSIC_ROUND,
707 G_INTRINSIC_ROUNDEVEN})
708 .legalFor(ST.hasStdExtZfa(), {s32})
709 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtD(), {s64})
710 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtZfh(), {s16})
711 .libcallFor({s32, s64})
712 .libcallFor(ST.is64Bit(), {s128});
713
714 getActionDefinitionsBuilder({G_FMAXIMUM, G_FMINIMUM})
715 .legalFor(ST.hasStdExtZfa(), {s32})
716 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtD(), {s64})
717 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtZfh(), {s16});
718
719 getActionDefinitionsBuilder({G_FCOS, G_FSIN, G_FTAN, G_FPOW, G_FLOG, G_FLOG2,
720 G_FLOG10, G_FEXP, G_FEXP2, G_FEXP10, G_FACOS,
721 G_FASIN, G_FATAN, G_FATAN2, G_FCOSH, G_FSINH,
722 G_FTANH, G_FMODF})
723 .libcallFor({s32, s64})
724 .libcallFor(ST.is64Bit(), {s128});
725 getActionDefinitionsBuilder({G_FPOWI, G_FLDEXP})
726 .libcallFor({{s32, s32}, {s64, s32}})
727 .libcallFor(ST.is64Bit(), {s128, s32});
728
729 getActionDefinitionsBuilder(G_FCANONICALIZE)
730 .legalFor(ST.hasStdExtF(), {s32})
731 .legalFor(ST.hasStdExtD(), {s64})
732 .legalFor(ST.hasStdExtZfh(), {s16});
733
734 getActionDefinitionsBuilder(G_VASTART).customFor({p0});
735
736 // va_list must be a pointer, but most sized types are pretty easy to handle
737 // as the destination.
738 getActionDefinitionsBuilder(G_VAARG)
739 // TODO: Implement narrowScalar and widenScalar for G_VAARG for types
740 // other than sXLen.
741 .clampScalar(0, sXLen, sXLen)
742 .lowerForCartesianProduct({sXLen, p0}, {p0});
743
744 getActionDefinitionsBuilder(G_VSCALE)
745 .clampScalar(0, sXLen, sXLen)
746 .customFor({sXLen});
747
748 auto &SplatActions =
749 getActionDefinitionsBuilder(G_SPLAT_VECTOR)
750 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
751 typeIs(1, sXLen)))
752 .customIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST), typeIs(1, s1)));
753 // Handle case of s64 element vectors on RV32. If the subtarget does not have
754 // f64, then try to lower it to G_SPLAT_VECTOR_SPLIT_64_VL. If the subtarget
755 // does have f64, then we don't know whether the type is an f64 or an i64,
756 // so mark the G_SPLAT_VECTOR as legal and decide later what to do with it,
757 // depending on how the instructions it consumes are legalized. They are not
758 // legalized yet since legalization is in reverse postorder, so we cannot
759 // make the decision at this moment.
760 if (XLen == 32) {
761 if (ST.hasVInstructionsF64() && ST.hasStdExtD())
762 SplatActions.legalIf(all(
763 typeInSet(0, {nxv1s64, nxv2s64, nxv4s64, nxv8s64}), typeIs(1, s64)));
764 else if (ST.hasVInstructionsI64())
765 SplatActions.customIf(all(
766 typeInSet(0, {nxv1s64, nxv2s64, nxv4s64, nxv8s64}), typeIs(1, s64)));
767 }
768
769 SplatActions.clampScalar(1, sXLen, sXLen);
770
771 LegalityPredicate ExtractSubvecBitcastPred = [=](const LegalityQuery &Query) {
772 LLT DstTy = Query.Types[0];
773 LLT SrcTy = Query.Types[1];
774 return DstTy.getElementType() == LLT::scalar(1) &&
775 DstTy.getElementCount().getKnownMinValue() >= 8 &&
776 SrcTy.getElementCount().getKnownMinValue() >= 8;
777 };
778 getActionDefinitionsBuilder(G_EXTRACT_SUBVECTOR)
779 // We don't have the ability to slide mask vectors down indexed by their
780 // i1 elements; the smallest we can do is i8. Often we are able to bitcast
781 // to equivalent i8 vectors.
782 .bitcastIf(
783 all(typeIsLegalBoolVec(0, BoolVecTys, ST),
784 typeIsLegalBoolVec(1, BoolVecTys, ST), ExtractSubvecBitcastPred),
785 [=](const LegalityQuery &Query) {
786 LLT CastTy = LLT::vector(
787 Query.Types[0].getElementCount().divideCoefficientBy(8), 8);
788 return std::pair(0, CastTy);
789 })
790 .customIf(LegalityPredicates::any(
791 all(typeIsLegalBoolVec(0, BoolVecTys, ST),
792 typeIsLegalBoolVec(1, BoolVecTys, ST)),
793 all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
794 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST))));
795
796 getActionDefinitionsBuilder(G_INSERT_SUBVECTOR)
797 .customIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST),
798 typeIsLegalBoolVec(1, BoolVecTys, ST)))
799 .customIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
800 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST)));
801
802 getActionDefinitionsBuilder(G_ATOMIC_CMPXCHG_WITH_SUCCESS)
803 .lowerIf(all(typeInSet(0, {s8, s16, s32, s64}), typeIs(2, p0)));
804
805 getActionDefinitionsBuilder({G_ATOMIC_CMPXCHG, G_ATOMICRMW_ADD,
806 G_ATOMICRMW_XCHG, G_ATOMICRMW_AND,
807 G_ATOMICRMW_OR, G_ATOMICRMW_XOR})
808 .legalFor(ST.hasStdExtA(), {{sXLen, p0}})
809 .libcallFor(!ST.hasStdExtA(), {{s8, p0}, {s16, p0}, {s32, p0}, {s64, p0}})
810 .clampScalar(0, sXLen, sXLen);
811
812 getActionDefinitionsBuilder(G_ATOMICRMW_SUB)
813 .libcallFor(!ST.hasStdExtA(), {{s8, p0}, {s16, p0}, {s32, p0}, {s64, p0}})
814 .clampScalar(0, sXLen, sXLen)
815 .lower();
816
817 getActionDefinitionsBuilder(
818 {G_ATOMICRMW_MAX, G_ATOMICRMW_MIN, G_ATOMICRMW_UMAX, G_ATOMICRMW_UMIN})
819 .legalFor(ST.hasStdExtA(), {{sXLen, p0}})
820 .clampScalar(0, sXLen, sXLen)
821 .unsupported();
822
823 getActionDefinitionsBuilder(G_PREFETCH).legalIf(typeIs(0, p0));
824
825 LegalityPredicate InsertVectorEltPred = [=](const LegalityQuery &Query) {
826 LLT VecTy = Query.Types[0];
827 LLT EltTy = Query.Types[1];
828 return VecTy.getElementType() == EltTy;
829 };
830
831 getActionDefinitionsBuilder(G_INSERT_VECTOR_ELT)
832 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
833 InsertVectorEltPred, typeIs(2, sXLen)))
834 .legalIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST), InsertVectorEltPred,
835 typeIs(2, sXLen)));
836
837 getActionDefinitionsBuilder({G_INTRINSIC, G_INTRINSIC_W_SIDE_EFFECTS})
838 .alwaysLegal();
839
840 getActionDefinitionsBuilder(G_FENCE).alwaysLegal();
841
842 getActionDefinitionsBuilder({G_TRAP, G_DEBUGTRAP, G_UBSANTRAP}).alwaysLegal();
843
844 verify(*ST.getInstrInfo());
845}
846
848 MachineInstr &MI) const {
849 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
850
852 RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IntrinsicID)) {
853 if (II->hasScalarOperand() && !II->IsFPIntrinsic) {
854 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
855 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
856
857 auto OldScalar = MI.getOperand(II->ScalarOperand + 2).getReg();
858 // Legalize integer vx form intrinsic.
859 if (MRI.getType(OldScalar).isScalar()) {
860 if (MRI.getType(OldScalar).getSizeInBits() < sXLen.getSizeInBits()) {
861 Helper.Observer.changingInstr(MI);
862 Helper.widenScalarSrc(MI, sXLen, II->ScalarOperand + 2,
863 TargetOpcode::G_ANYEXT);
864 Helper.Observer.changedInstr(MI);
865 } else if (MRI.getType(OldScalar).getSizeInBits() >
866 sXLen.getSizeInBits()) {
867 // TODO: i64 in riscv32.
868 return false;
869 }
870 }
871 }
872 return true;
873 }
874
875 switch (IntrinsicID) {
876 default:
877 return false;
878 case Intrinsic::riscv_clmulh:
879 Helper.MIRBuilder.buildInstr(TargetOpcode::G_CLMULH, {MI.getOperand(0)},
880 {MI.getOperand(2), MI.getOperand(3)});
881 MI.eraseFromParent();
882 return true;
883 case Intrinsic::riscv_clmulr:
884 Helper.MIRBuilder.buildInstr(TargetOpcode::G_CLMULR, {MI.getOperand(0)},
885 {MI.getOperand(2), MI.getOperand(3)});
886 MI.eraseFromParent();
887 return true;
888 case Intrinsic::vacopy: {
889 // vacopy arguments must be legal because of the intrinsic signature.
890 // No need to check here.
891
892 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
893 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
894 MachineFunction &MF = *MI.getMF();
895 const DataLayout &DL = MIRBuilder.getDataLayout();
896 LLVMContext &Ctx = MF.getFunction().getContext();
897
898 Register DstLst = MI.getOperand(1).getReg();
899 LLT PtrTy = MRI.getType(DstLst);
900
901 // Load the source va_list
902 Align Alignment = DL.getABITypeAlign(getTypeForLLT(PtrTy, Ctx));
904 MachinePointerInfo(), MachineMemOperand::MOLoad, PtrTy, Alignment);
905 auto Tmp = MIRBuilder.buildLoad(PtrTy, MI.getOperand(2), *LoadMMO);
906
907 // Store the result in the destination va_list
910 MIRBuilder.buildStore(Tmp, DstLst, *StoreMMO);
911
912 MI.eraseFromParent();
913 return true;
914 }
915 case Intrinsic::riscv_vsetvli:
916 case Intrinsic::riscv_vsetvlimax:
917 case Intrinsic::riscv_masked_atomicrmw_add:
918 case Intrinsic::riscv_masked_atomicrmw_sub:
919 case Intrinsic::riscv_masked_atomicrmw_xchg:
920 case Intrinsic::riscv_masked_atomicrmw_max:
921 case Intrinsic::riscv_masked_atomicrmw_min:
922 case Intrinsic::riscv_masked_atomicrmw_umax:
923 case Intrinsic::riscv_masked_atomicrmw_umin:
924 case Intrinsic::riscv_masked_cmpxchg:
925 return true;
926 }
927}
928
929bool RISCVLegalizerInfo::legalizeVAStart(MachineInstr &MI,
930 MachineIRBuilder &MIRBuilder) const {
931 // Stores the address of the VarArgsFrameIndex slot into the memory location
932 assert(MI.getOpcode() == TargetOpcode::G_VASTART);
933 MachineFunction *MF = MI.getParent()->getParent();
935 int FI = FuncInfo->getVarArgsFrameIndex();
936 LLT AddrTy = MIRBuilder.getMRI()->getType(MI.getOperand(0).getReg());
937 auto FINAddr = MIRBuilder.buildFrameIndex(AddrTy, FI);
938 assert(MI.hasOneMemOperand());
939 MIRBuilder.buildStore(FINAddr, MI.getOperand(0).getReg(),
940 *MI.memoperands()[0]);
941 MI.eraseFromParent();
942 return true;
943}
944
945bool RISCVLegalizerInfo::legalizeReadCounter(
946 MachineInstr &MI, MachineIRBuilder &MIRBuilder,
947 GISelChangeObserver &Observer) const {
948 assert((MI.getOpcode() == TargetOpcode::G_READCYCLECOUNTER ||
949 MI.getOpcode() == TargetOpcode::G_READSTEADYCOUNTER) &&
950 "Unexpected opcode");
951 assert(!STI.is64Bit() && "READCYCLECOUNTER/READSTEADYCOUNTER only "
952 "has custom type legalization on riscv32");
953
954 // On RV32 a 64-bit counter CSR must be read as two 32-bit halves. Because
955 // the count may wrap between the two reads, re-read the high half and loop
956 // until the two high reads agree.
957 int64_t LoCounter, HiCounter;
958 if (MI.getOpcode() == TargetOpcode::G_READCYCLECOUNTER) {
959 LoCounter = RISCVSysReg::cycle;
960 HiCounter = RISCVSysReg::cycleh;
961 } else {
962 LoCounter = RISCVSysReg::time;
963 HiCounter = RISCVSysReg::timeh;
964 }
965
966 MachineBasicBlock *BB = MI.getParent();
967 MachineFunction &MF = *BB->getParent();
968 const BasicBlock *LLVMBB = BB->getBasicBlock();
969 DebugLoc DL = MI.getDebugLoc();
970 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
971
972 // Split BB into an entry that falls through into a loop block, and a done
973 // block that receives the remainder of BB and its original successors.
974 MachineFunction::iterator It = std::next(BB->getIterator());
975 MachineBasicBlock *LoopMBB = MF.CreateMachineBasicBlock(LLVMBB);
976 MachineBasicBlock *DoneMBB = MF.CreateMachineBasicBlock(LLVMBB);
977 MF.insert(It, LoopMBB);
978 MF.insert(It, DoneMBB);
979
980 // Splice the instructions after the readcyclecounter into DoneMBB, notifying
981 // the observer about each moved instruction so CSEInfo stays consistent.
982 for (MachineBasicBlock::iterator I = std::next(MI.getIterator()),
983 E = BB->end();
984 I != E; ++I)
985 Observer.changingInstr(*I);
986 DoneMBB->splice(DoneMBB->begin(), BB,
987 std::next(MachineBasicBlock::iterator(MI)), BB->end());
988 for (MachineInstr &MovedMI : DoneMBB->instrs())
989 Observer.changedInstr(MovedMI);
991 BB->addSuccessor(LoopMBB);
992
993 LLT S32 = LLT::scalar(32);
994 // Generic vregs carry the s32 type for G_MERGE_VALUES below, but are also
995 // constrained to GPR so the target CSRRS/BNE instructions satisfy the
996 // verifier's register-class constraints.
997 auto CreateGPR = [&]() {
999 MRI.setRegClass(R, &RISCV::GPRRegClass);
1000 return R;
1001 };
1002 Register LoReg = CreateGPR();
1003 Register HiReg = CreateGPR();
1004 Register ReadAgainReg = CreateGPR();
1005
1006 // read:
1007 // csrrs HiReg, counterh # high word
1008 // csrrs LoReg, counter # low word
1009 // csrrs ReadAgainReg, counterh
1010 // bne HiReg, ReadAgainReg, read
1011 // Emit the target instructions directly with BuildMI.
1012 const RISCVInstrInfo *TII = STI.getInstrInfo();
1013 BuildMI(LoopMBB, DL, TII->get(RISCV::CSRRS), HiReg)
1014 .addImm(HiCounter)
1015 .addReg(RISCV::X0);
1016 BuildMI(LoopMBB, DL, TII->get(RISCV::CSRRS), LoReg)
1017 .addImm(LoCounter)
1018 .addReg(RISCV::X0);
1019 BuildMI(LoopMBB, DL, TII->get(RISCV::CSRRS), ReadAgainReg)
1020 .addImm(HiCounter)
1021 .addReg(RISCV::X0);
1022
1023 BuildMI(LoopMBB, DL, TII->get(RISCV::BNE))
1024 .addReg(HiReg)
1025 .addReg(ReadAgainReg)
1026 .addMBB(LoopMBB);
1027
1028 LoopMBB->addSuccessor(LoopMBB);
1029 LoopMBB->addSuccessor(DoneMBB);
1030
1031 // Re-pair the two halves into the 64-bit result.
1032 Register DstReg = MI.getOperand(0).getReg();
1033 Observer.erasingInstr(MI);
1034 MI.eraseFromParent();
1035
1036 MIRBuilder.setInsertPt(*DoneMBB, DoneMBB->begin());
1037 MIRBuilder.setDebugLoc(DL);
1038 MIRBuilder.buildMergeValues(DstReg, {LoReg, HiReg});
1039 return true;
1040}
1041
1042bool RISCVLegalizerInfo::legalizeBRJT(MachineInstr &MI,
1043 MachineIRBuilder &MIRBuilder) const {
1044 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
1045 auto &MF = *MI.getParent()->getParent();
1046 const MachineJumpTableInfo *MJTI = MF.getJumpTableInfo();
1047 unsigned EntrySize = MJTI->getEntrySize(MF.getDataLayout());
1048
1049 Register PtrReg = MI.getOperand(0).getReg();
1050 LLT PtrTy = MRI.getType(PtrReg);
1051 Register IndexReg = MI.getOperand(2).getReg();
1052 LLT IndexTy = MRI.getType(IndexReg);
1053
1054 if (!isPowerOf2_32(EntrySize))
1055 return false;
1056
1057 auto ShiftAmt = MIRBuilder.buildConstant(IndexTy, Log2_32(EntrySize));
1058 IndexReg = MIRBuilder.buildShl(IndexTy, IndexReg, ShiftAmt).getReg(0);
1059
1060 auto Addr = MIRBuilder.buildPtrAdd(PtrTy, PtrReg, IndexReg);
1061
1062 MachineMemOperand *MMO = MF.getMachineMemOperand(
1064 EntrySize, Align(MJTI->getEntryAlignment(MF.getDataLayout())));
1065
1066 Register TargetReg;
1067 switch (MJTI->getEntryKind()) {
1068 default:
1069 return false;
1071 // For PIC, the sequence is:
1072 // BRIND(load(Jumptable + index) + RelocBase)
1073 // RelocBase can be JumpTable, GOT or some sort of global base.
1074 unsigned LoadOpc =
1075 STI.is64Bit() ? TargetOpcode::G_SEXTLOAD : TargetOpcode::G_LOAD;
1076 auto Load = MIRBuilder.buildLoadInstr(LoadOpc, IndexTy, Addr, *MMO);
1077 TargetReg = MIRBuilder.buildPtrAdd(PtrTy, PtrReg, Load).getReg(0);
1078 break;
1079 }
1081 auto Load = MIRBuilder.buildLoadInstr(TargetOpcode::G_SEXTLOAD, IndexTy,
1082 Addr, *MMO);
1083 TargetReg = MIRBuilder.buildIntToPtr(PtrTy, Load).getReg(0);
1084 break;
1085 }
1087 TargetReg = MIRBuilder.buildLoad(PtrTy, Addr, *MMO).getReg(0);
1088 break;
1089 }
1090
1091 MIRBuilder.buildBrIndirect(TargetReg);
1092
1093 MI.eraseFromParent();
1094 return true;
1095}
1096
1097bool RISCVLegalizerInfo::shouldBeInConstantPool(const APInt &APImm,
1098 bool ShouldOptForSize) const {
1099 assert(APImm.getBitWidth() == 32 || APImm.getBitWidth() == 64);
1100 int64_t Imm = APImm.getSExtValue();
1101 // All simm32 constants should be handled by isel.
1102 // NOTE: The getMaxBuildIntsCost call below should return a value >= 2 making
1103 // this check redundant, but small immediates are common so this check
1104 // should have better compile time.
1105 if (isInt<32>(Imm))
1106 return false;
1107
1108 // We only need to cost the immediate, if constant pool lowering is enabled.
1109 if (!STI.useConstantPoolForLargeInts())
1110 return false;
1111
1113 if (Seq.size() <= STI.getMaxBuildIntsCost())
1114 return false;
1115
1116 // Optimizations below are disabled for opt size. If we're optimizing for
1117 // size, use a constant pool.
1118 if (ShouldOptForSize)
1119 return true;
1120 //
1121 // Special case. See if we can build the constant as (ADD (SLLI X, C), X) do
1122 // that if it will avoid a constant pool.
1123 // It will require an extra temporary register though.
1124 // If we have Zba we can use (ADD_UW X, (SLLI X, 32)) to handle cases where
1125 // low and high 32 bits are the same and bit 31 and 63 are set.
1126 unsigned ShiftAmt, AddOpc;
1127 RISCVMatInt::InstSeq SeqLo =
1128 RISCVMatInt::generateTwoRegInstSeq(Imm, STI, ShiftAmt, AddOpc);
1129 return !(!SeqLo.empty() && (SeqLo.size() + 2) <= STI.getMaxBuildIntsCost());
1130}
1131
1132bool RISCVLegalizerInfo::legalizeVScale(MachineInstr &MI,
1133 MachineIRBuilder &MIB) const {
1134 Register Dst = MI.getOperand(0).getReg();
1135
1136 // We define our scalable vector types for lmul=1 to use a 64 bit known
1137 // minimum size. e.g. <vscale x 2 x i32>. VLENB is in bytes so we calculate
1138 // vscale as VLENB / 8.
1139 static_assert(RISCV::RVVBitsPerBlock == 64, "Unexpected bits per block!");
1140 if (STI.getRealMinVLen() < RISCV::RVVBitsPerBlock)
1141 // Support for VLEN==32 is incomplete.
1142 return false;
1143
1144 // We assume VLENB is a multiple of 8. We manually choose the best shift
1145 // here because SimplifyDemandedBits isn't always able to simplify it.
1146 uint64_t Val = MI.getOperand(1).getCImm()->getZExtValue();
1147 if (isPowerOf2_64(Val)) {
1148 uint64_t Log2 = Log2_64(Val);
1149 if (Log2 < 3) {
1150 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1151 MIB.buildLShr(Dst, VLENB, MIB.buildConstant(sXLen, 3 - Log2),
1153 } else if (Log2 > 3) {
1154 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1155 MIB.buildShl(Dst, VLENB, MIB.buildConstant(sXLen, Log2 - 3));
1156 } else {
1157 MIB.buildInstr(RISCV::G_READ_VLENB, {Dst}, {});
1158 }
1159 } else if ((Val % 8) == 0) {
1160 // If the multiplier is a multiple of 8, scale it down to avoid needing
1161 // to shift the VLENB value.
1162 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1163 MIB.buildMul(Dst, VLENB, MIB.buildConstant(sXLen, Val / 8));
1164 } else {
1165 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1166 auto VScale = MIB.buildLShr(sXLen, VLENB, MIB.buildConstant(sXLen, 3),
1168 MIB.buildMul(Dst, VScale, MIB.buildConstant(sXLen, Val));
1169 }
1170 MI.eraseFromParent();
1171 return true;
1172}
1173
1174// Custom-lower extensions from mask vectors by using a vselect either with 1
1175// for zero/any-extension or -1 for sign-extension:
1176// (vXiN = (s|z)ext vXi1:vmask) -> (vXiN = vselect vmask, (-1 or 1), 0)
1177// Note that any-extension is lowered identically to zero-extension.
1178bool RISCVLegalizerInfo::legalizeExt(MachineInstr &MI,
1179 MachineIRBuilder &MIB) const {
1180
1181 unsigned Opc = MI.getOpcode();
1182 assert(Opc == TargetOpcode::G_ZEXT || Opc == TargetOpcode::G_SEXT ||
1183 Opc == TargetOpcode::G_ANYEXT);
1184
1185 MachineRegisterInfo &MRI = *MIB.getMRI();
1186 Register Dst = MI.getOperand(0).getReg();
1187 Register Src = MI.getOperand(1).getReg();
1188
1189 LLT DstTy = MRI.getType(Dst);
1190 int64_t ExtTrueVal = Opc == TargetOpcode::G_SEXT ? -1 : 1;
1191 LLT DstEltTy = DstTy.getElementType();
1192 auto SplatZero = MIB.buildSplatVector(DstTy, MIB.buildConstant(DstEltTy, 0));
1193 auto SplatTrue =
1194 MIB.buildSplatVector(DstTy, MIB.buildConstant(DstEltTy, ExtTrueVal));
1195 MIB.buildSelect(Dst, Src, SplatTrue, SplatZero);
1196
1197 MI.eraseFromParent();
1198 return true;
1199}
1200
1201bool RISCVLegalizerInfo::legalizeLoadStore(MachineInstr &MI,
1202 LegalizerHelper &Helper,
1203 MachineIRBuilder &MIB) const {
1205 "Machine instructions must be Load/Store.");
1206 MachineRegisterInfo &MRI = *MIB.getMRI();
1207 MachineFunction *MF = MI.getMF();
1208 const DataLayout &DL = MIB.getDataLayout();
1209 LLVMContext &Ctx = MF->getFunction().getContext();
1210
1211 Register DstReg = MI.getOperand(0).getReg();
1212 LLT DataTy = MRI.getType(DstReg);
1213 if (!DataTy.isVector())
1214 return false;
1215
1216 if (!MI.hasOneMemOperand())
1217 return false;
1218
1219 MachineMemOperand *MMO = *MI.memoperands_begin();
1220
1221 const auto *TLI = STI.getTargetLowering();
1222 EVT VT = EVT::getEVT(getTypeForLLT(DataTy, Ctx));
1223
1224 if (TLI->allowsMemoryAccessForAlignment(Ctx, DL, VT, *MMO))
1225 return true;
1226
1227 unsigned EltSizeBits = DataTy.getScalarSizeInBits();
1228 assert((EltSizeBits == 16 || EltSizeBits == 32 || EltSizeBits == 64) &&
1229 "Unexpected unaligned RVV load type");
1230
1231 // Calculate the new vector type with i8 elements
1232 unsigned NumElements =
1233 DataTy.getElementCount().getKnownMinValue() * (EltSizeBits / 8);
1234 LLT NewDataTy = LLT::scalable_vector(NumElements, 8);
1235
1236 Helper.bitcast(MI, 0, NewDataTy);
1237
1238 return true;
1239}
1240
1241/// Return the type of the mask type suitable for masking the provided
1242/// vector type. This is simply an i1 element type vector of the same
1243/// (possibly scalable) length.
1244static LLT getMaskTypeFor(LLT VecTy) {
1245 assert(VecTy.isVector());
1246 ElementCount EC = VecTy.getElementCount();
1247 return LLT::vector(EC, LLT::scalar(1));
1248}
1249
1250/// Creates an all ones mask suitable for masking a vector of type VecTy with
1251/// vector length VL.
1253 MachineIRBuilder &MIB,
1254 MachineRegisterInfo &MRI) {
1255 LLT MaskTy = getMaskTypeFor(VecTy);
1256 return MIB.buildInstr(RISCV::G_VMSET_VL, {MaskTy}, {VL});
1257}
1258
1259/// Gets the two common "VL" operands: an all-ones mask and the vector length.
1260/// VecTy is a scalable vector type.
1261static std::pair<MachineInstrBuilder, MachineInstrBuilder>
1263 assert(VecTy.isScalableVector() && "Expecting scalable container type");
1264 const RISCVSubtarget &STI = MIB.getMF().getSubtarget<RISCVSubtarget>();
1265 LLT XLenTy(STI.getXLenVT());
1266 auto VL = MIB.buildConstant(XLenTy, -1);
1267 auto Mask = buildAllOnesMask(VecTy, VL, MIB, MRI);
1268 return {Mask, VL};
1269}
1270
1272buildSplatPartsS64WithVL(const DstOp &Dst, const SrcOp &Passthru, Register Lo,
1273 Register Hi, const SrcOp &VL, MachineIRBuilder &MIB,
1274 MachineRegisterInfo &MRI) {
1275 // TODO: If the Hi bits of the splat are undefined, then it's fine to just
1276 // splat Lo even if it might be sign extended. I don't think we have
1277 // introduced a case where we're build a s64 where the upper bits are undef
1278 // yet.
1279
1280 // Fall back to a stack store and stride x0 vector load.
1281 // TODO: need to lower G_SPLAT_VECTOR_SPLIT_I64. This is done in
1282 // preprocessDAG in SDAG.
1283 return MIB.buildInstr(RISCV::G_SPLAT_VECTOR_SPLIT_I64_VL, {Dst},
1284 {Passthru, Lo, Hi, VL});
1285}
1286
1288buildSplatSplitS64WithVL(const DstOp &Dst, const SrcOp &Passthru,
1289 const SrcOp &Scalar, const SrcOp &VL,
1291 assert(Scalar.getLLTTy(MRI) == LLT::scalar(64) && "Unexpected VecTy!");
1292 auto Unmerge = MIB.buildUnmerge(LLT::scalar(32), Scalar);
1293 return buildSplatPartsS64WithVL(Dst, Passthru, Unmerge.getReg(0),
1294 Unmerge.getReg(1), VL, MIB, MRI);
1295}
1296
1297// Lower splats of s1 types to G_ICMP. For each mask vector type, we have a
1298// legal equivalently-sized i8 type, so we can use that as a go-between.
1299// Splats of s1 types that have constant value can be legalized as VMSET_VL or
1300// VMCLR_VL.
1301bool RISCVLegalizerInfo::legalizeSplatVector(MachineInstr &MI,
1302 MachineIRBuilder &MIB) const {
1303 assert(MI.getOpcode() == TargetOpcode::G_SPLAT_VECTOR);
1304
1305 MachineRegisterInfo &MRI = *MIB.getMRI();
1306
1307 Register Dst = MI.getOperand(0).getReg();
1308 Register SplatVal = MI.getOperand(1).getReg();
1309
1310 LLT VecTy = MRI.getType(Dst);
1311 LLT XLenTy(STI.getXLenVT());
1312
1313 // Handle case of s64 element vectors on rv32
1314 if (XLenTy.getSizeInBits() == 32 &&
1315 VecTy.getElementType().getSizeInBits() == 64) {
1316 auto [_, VL] = buildDefaultVLOps(MRI.getType(Dst), MIB, MRI);
1317 buildSplatSplitS64WithVL(Dst, MIB.buildUndef(VecTy), SplatVal, VL, MIB,
1318 MRI);
1319 MI.eraseFromParent();
1320 return true;
1321 }
1322
1323 // All-zeros or all-ones splats are handled specially.
1324 MachineInstr &SplatValMI = *MRI.getVRegDef(SplatVal);
1325 if (isAllOnesOrAllOnesSplat(SplatValMI, MRI)) {
1326 auto VL = buildDefaultVLOps(VecTy, MIB, MRI).second;
1327 MIB.buildInstr(RISCV::G_VMSET_VL, {Dst}, {VL});
1328 MI.eraseFromParent();
1329 return true;
1330 }
1331 if (isNullOrNullSplat(SplatValMI, MRI)) {
1332 auto VL = buildDefaultVLOps(VecTy, MIB, MRI).second;
1333 MIB.buildInstr(RISCV::G_VMCLR_VL, {Dst}, {VL});
1334 MI.eraseFromParent();
1335 return true;
1336 }
1337
1338 // Handle non-constant mask splat (i.e. not sure if it's all zeros or all
1339 // ones) by promoting it to an s8 splat.
1340 LLT InterEltTy = LLT::scalar(8);
1341 LLT InterTy = VecTy.changeElementType(InterEltTy);
1342 auto ZExtSplatVal = MIB.buildZExt(InterEltTy, SplatVal);
1343 auto And =
1344 MIB.buildAnd(InterEltTy, ZExtSplatVal, MIB.buildConstant(InterEltTy, 1));
1345 auto LHS = MIB.buildSplatVector(InterTy, And);
1346 auto ZeroSplat =
1347 MIB.buildSplatVector(InterTy, MIB.buildConstant(InterEltTy, 0));
1348 MIB.buildICmp(CmpInst::Predicate::ICMP_NE, Dst, LHS, ZeroSplat);
1349 MI.eraseFromParent();
1350 return true;
1351}
1352
1353static LLT getLMUL1Ty(LLT VecTy) {
1354 assert(VecTy.getElementType().getSizeInBits() <= 64 &&
1355 "Unexpected vector LLT");
1357 VecTy.getElementType().getSizeInBits(),
1358 VecTy.getElementType());
1359}
1360
1361bool RISCVLegalizerInfo::legalizeExtractSubvector(MachineInstr &MI,
1362 MachineIRBuilder &MIB) const {
1363 GExtractSubvector &ES = cast<GExtractSubvector>(MI);
1364
1365 MachineRegisterInfo &MRI = *MIB.getMRI();
1366
1367 Register Dst = ES.getReg(0);
1368 Register Src = ES.getSrcVec();
1369 uint64_t Idx = ES.getIndexImm();
1370
1371 // With an index of 0 this is a cast-like subvector, which can be performed
1372 // with subregister operations.
1373 if (Idx == 0)
1374 return true;
1375
1376 LLT LitTy = MRI.getType(Dst);
1377 LLT BigTy = MRI.getType(Src);
1378
1379 if (LitTy.getElementType() == LLT::scalar(1)) {
1380 // We can't slide this mask vector up indexed by its i1 elements.
1381 // This poses a problem when we wish to insert a scalable vector which
1382 // can't be re-expressed as a larger type. Just choose the slow path and
1383 // extend to a larger type, then truncate back down.
1384 LLT ExtBigTy = BigTy.changeElementType(LLT::scalar(8));
1385 LLT ExtLitTy = LitTy.changeElementType(LLT::scalar(8));
1386 auto BigZExt = MIB.buildZExt(ExtBigTy, Src);
1387 auto ExtractZExt = MIB.buildExtractSubvector(ExtLitTy, BigZExt, Idx);
1388 auto SplatZero = MIB.buildSplatVector(
1389 ExtLitTy, MIB.buildConstant(ExtLitTy.getElementType(), 0));
1390 MIB.buildICmp(CmpInst::Predicate::ICMP_NE, Dst, ExtractZExt, SplatZero);
1391 MI.eraseFromParent();
1392 return true;
1393 }
1394
1395 // extract_subvector scales the index by vscale if the subvector is scalable,
1396 // and decomposeSubvectorInsertExtractToSubRegs takes this into account.
1397 const RISCVRegisterInfo *TRI = STI.getRegisterInfo();
1398 MVT LitTyMVT = getMVTForLLT(LitTy);
1399 auto Decompose =
1401 getMVTForLLT(BigTy), LitTyMVT, Idx, TRI);
1402 unsigned RemIdx = Decompose.second;
1403
1404 // If the Idx has been completely eliminated then this is a subvector extract
1405 // which naturally aligns to a vector register. These can easily be handled
1406 // using subregister manipulation.
1407 if (RemIdx == 0)
1408 return true;
1409
1410 // Else LitTy is M1 or smaller and may need to be slid down: if LitTy
1411 // was > M1 then the index would need to be a multiple of VLMAX, and so would
1412 // divide exactly.
1413 assert(
1416
1417 // If the vector type is an LMUL-group type, extract a subvector equal to the
1418 // nearest full vector register type.
1419 LLT InterLitTy = BigTy;
1420 Register Vec = Src;
1422 getLMUL1Ty(BigTy).getSizeInBits())) {
1423 // If BigTy has an LMUL > 1, then LitTy should have a smaller LMUL, and
1424 // we should have successfully decomposed the extract into a subregister.
1425 assert(Decompose.first != RISCV::NoSubRegister);
1426 InterLitTy = getLMUL1Ty(BigTy);
1427 // SDAG builds a TargetExtractSubreg. We cannot create a a Copy with SubReg
1428 // specified on the source Register (the equivalent) since generic virtual
1429 // register does not allow subregister index.
1430 Vec = MIB.buildExtractSubvector(InterLitTy, Src, Idx - RemIdx).getReg(0);
1431 }
1432
1433 // Slide this vector register down by the desired number of elements in order
1434 // to place the desired subvector starting at element 0.
1435 const LLT XLenTy(STI.getXLenVT());
1436 auto SlidedownAmt = MIB.buildVScale(XLenTy, RemIdx);
1437 auto [Mask, VL] = buildDefaultVLOps(InterLitTy, MIB, MRI);
1439 auto Slidedown = MIB.buildInstr(
1440 RISCV::G_VSLIDEDOWN_VL, {InterLitTy},
1441 {MIB.buildUndef(InterLitTy), Vec, SlidedownAmt, Mask, VL, Policy});
1442
1443 // Now the vector is in the right position, extract our final subvector. This
1444 // should resolve to a COPY.
1445 MIB.buildExtractSubvector(Dst, Slidedown, 0);
1446
1447 MI.eraseFromParent();
1448 return true;
1449}
1450
1451bool RISCVLegalizerInfo::legalizeInsertSubvector(MachineInstr &MI,
1452 LegalizerHelper &Helper,
1453 MachineIRBuilder &MIB) const {
1454 GInsertSubvector &IS = cast<GInsertSubvector>(MI);
1455
1456 MachineRegisterInfo &MRI = *MIB.getMRI();
1457
1458 Register Dst = IS.getReg(0);
1459 Register BigVec = IS.getBigVec();
1460 Register LitVec = IS.getSubVec();
1461 uint64_t Idx = IS.getIndexImm();
1462
1463 LLT BigTy = MRI.getType(BigVec);
1464 LLT LitTy = MRI.getType(LitVec);
1465
1466 if (Idx == 0 && mi_match(BigVec, MRI, m_GImplicitDef()))
1467 return true;
1468
1469 // We don't have the ability to slide mask vectors up indexed by their i1
1470 // elements; the smallest we can do is i8. Often we are able to bitcast to
1471 // equivalent i8 vectors. Otherwise, we can must zeroextend to equivalent i8
1472 // vectors and truncate down after the insert.
1473 if (LitTy.getElementType() == LLT::scalar(1)) {
1474 auto BigTyMinElts = BigTy.getElementCount().getKnownMinValue();
1475 auto LitTyMinElts = LitTy.getElementCount().getKnownMinValue();
1476 if (BigTyMinElts >= 8 && LitTyMinElts >= 8)
1477 return Helper.bitcast(
1478 IS, 0,
1480
1481 // We can't slide this mask vector up indexed by its i1 elements.
1482 // This poses a problem when we wish to insert a scalable vector which
1483 // can't be re-expressed as a larger type. Just choose the slow path and
1484 // extend to a larger type, then truncate back down.
1485 LLT ExtBigTy = BigTy.changeElementType(LLT::scalar(8));
1486 return Helper.widenScalar(IS, 0, ExtBigTy);
1487 }
1488
1489 const RISCVRegisterInfo *TRI = STI.getRegisterInfo();
1490 unsigned SubRegIdx, RemIdx;
1491 std::tie(SubRegIdx, RemIdx) =
1493 getMVTForLLT(BigTy), getMVTForLLT(LitTy), Idx, TRI);
1494
1495 TypeSize VecRegSize = TypeSize::getScalable(RISCV::RVVBitsPerBlock);
1497 STI.expandVScale(LitTy.getSizeInBits()).getKnownMinValue()));
1498 bool ExactlyVecRegSized =
1499 STI.expandVScale(LitTy.getSizeInBits())
1500 .isKnownMultipleOf(STI.expandVScale(VecRegSize));
1501
1502 // If the Idx has been completely eliminated and this subvector's size is a
1503 // vector register or a multiple thereof, or the surrounding elements are
1504 // undef, then this is a subvector insert which naturally aligns to a vector
1505 // register. These can easily be handled using subregister manipulation.
1506 if (RemIdx == 0 && ExactlyVecRegSized)
1507 return true;
1508
1509 // If the subvector is smaller than a vector register, then the insertion
1510 // must preserve the undisturbed elements of the register. We do this by
1511 // lowering to an EXTRACT_SUBVECTOR grabbing the nearest LMUL=1 vector type
1512 // (which resolves to a subregister copy), performing a VSLIDEUP to place the
1513 // subvector within the vector register, and an INSERT_SUBVECTOR of that
1514 // LMUL=1 type back into the larger vector (resolving to another subregister
1515 // operation). See below for how our VSLIDEUP works. We go via a LMUL=1 type
1516 // to avoid allocating a large register group to hold our subvector.
1517
1518 // VSLIDEUP works by leaving elements 0<i<OFFSET undisturbed, elements
1519 // OFFSET<=i<VL set to the "subvector" and vl<=i<VLMAX set to the tail policy
1520 // (in our case undisturbed). This means we can set up a subvector insertion
1521 // where OFFSET is the insertion offset, and the VL is the OFFSET plus the
1522 // size of the subvector.
1523 const LLT XLenTy(STI.getXLenVT());
1524 LLT InterLitTy = BigTy;
1525 Register AlignedExtract = BigVec;
1526 unsigned AlignedIdx = Idx - RemIdx;
1528 getLMUL1Ty(BigTy).getSizeInBits())) {
1529 InterLitTy = getLMUL1Ty(BigTy);
1530 // Extract a subvector equal to the nearest full vector register type. This
1531 // should resolve to a G_EXTRACT on a subreg.
1532 AlignedExtract =
1533 MIB.buildExtractSubvector(InterLitTy, BigVec, AlignedIdx).getReg(0);
1534 }
1535
1536 auto Insert = MIB.buildInsertSubvector(InterLitTy, MIB.buildUndef(InterLitTy),
1537 LitVec, 0);
1538
1539 auto [Mask, _] = buildDefaultVLOps(InterLitTy, MIB, MRI);
1540 auto VL = MIB.buildVScale(XLenTy, LitTy.getElementCount().getKnownMinValue());
1541
1542 // If we're inserting into the lowest elements, use a tail undisturbed
1543 // vmv.v.v.
1544 MachineInstrBuilder Inserted;
1545 bool NeedInsertSubvec =
1546 TypeSize::isKnownGT(BigTy.getSizeInBits(), InterLitTy.getSizeInBits());
1547 Register InsertedDst =
1548 NeedInsertSubvec ? MRI.createGenericVirtualRegister(InterLitTy) : Dst;
1549 if (RemIdx == 0) {
1550 Inserted = MIB.buildInstr(RISCV::G_VMV_V_V_VL, {InsertedDst},
1551 {AlignedExtract, Insert, VL});
1552 } else {
1553 auto SlideupAmt = MIB.buildVScale(XLenTy, RemIdx);
1554 // Construct the vector length corresponding to RemIdx + length(LitTy).
1555 VL = MIB.buildAdd(XLenTy, SlideupAmt, VL);
1556 // Use tail agnostic policy if we're inserting over InterLitTy's tail.
1557 ElementCount EndIndex =
1560 if (STI.expandVScale(EndIndex) ==
1561 STI.expandVScale(InterLitTy.getElementCount()))
1563
1564 Inserted =
1565 MIB.buildInstr(RISCV::G_VSLIDEUP_VL, {InsertedDst},
1566 {AlignedExtract, Insert, SlideupAmt, Mask, VL, Policy});
1567 }
1568
1569 // If required, insert this subvector back into the correct vector register.
1570 // This should resolve to an INSERT_SUBREG instruction.
1571 if (NeedInsertSubvec)
1572 MIB.buildInsertSubvector(Dst, BigVec, Inserted, AlignedIdx);
1573
1574 MI.eraseFromParent();
1575 return true;
1576}
1577
1578bool RISCVLegalizerInfo::legalizeBitreverse(MachineInstr &MI,
1579 MachineIRBuilder &MIB) const {
1580 assert(MI.getOpcode() == TargetOpcode::G_BITREVERSE && "Unexpected opcode");
1581
1582 if (!STI.hasStdExtZbkb())
1583 return false;
1584
1585 MachineRegisterInfo &MRI = *MIB.getMRI();
1586
1587 Register Dst = MI.getOperand(0).getReg();
1588 Register Src = MI.getOperand(1).getReg();
1589
1590 if (!MRI.getType(Dst).isScalar(8))
1591 return false;
1592
1593 auto WideSrc = MIB.buildAnyExt(sXLen, Src);
1594 auto Brev = MIB.buildInstr(RISCV::G_BREV8, {sXLen}, {WideSrc.getReg(0)});
1595 MIB.buildTrunc(Dst, Brev.getReg(0));
1596
1597 MI.eraseFromParent();
1598 return true;
1599}
1600
1601static unsigned getRISCVWOpcode(unsigned Opcode) {
1602 switch (Opcode) {
1603 default:
1604 llvm_unreachable("Unexpected opcode");
1605 case TargetOpcode::G_ASHR:
1606 return RISCV::G_SRAW;
1607 case TargetOpcode::G_LSHR:
1608 return RISCV::G_SRLW;
1609 case TargetOpcode::G_SHL:
1610 return RISCV::G_SLLW;
1611 case TargetOpcode::G_SDIV:
1612 return RISCV::G_DIVW;
1613 case TargetOpcode::G_UDIV:
1614 return RISCV::G_DIVUW;
1615 case TargetOpcode::G_UREM:
1616 return RISCV::G_REMUW;
1617 case TargetOpcode::G_ROTL:
1618 return RISCV::G_ROLW;
1619 case TargetOpcode::G_ROTR:
1620 return RISCV::G_RORW;
1621 case TargetOpcode::G_CTLZ:
1622 return RISCV::G_CLZW;
1623 case TargetOpcode::G_CTTZ:
1624 return RISCV::G_CTZW;
1625 case TargetOpcode::G_CTLS:
1626 return RISCV::G_CLSW;
1627 case TargetOpcode::G_FPTOSI:
1628 return RISCV::G_FCVT_W_RV64;
1629 case TargetOpcode::G_FPTOUI:
1630 return RISCV::G_FCVT_WU_RV64;
1631 }
1632}
1633
1636 LostDebugLocObserver &LocObserver) const {
1637 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
1638 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
1639 MachineFunction &MF = *MI.getParent()->getParent();
1640 switch (MI.getOpcode()) {
1641 default:
1642 // No idea what to do.
1643 return false;
1644 case TargetOpcode::G_ABS:
1645 return Helper.lowerAbsToMaxNeg(MI);
1646 case TargetOpcode::G_CLMULH:
1647 case TargetOpcode::G_CLMULR: {
1648 assert(STI.is64Bit() &&
1649 MRI.getType(MI.getOperand(0).getReg()) == LLT::scalar(32) &&
1650 "Unexpected custom legalization");
1651 // Shift both inputs by 32 so the full product has 64 trailing zeros.
1652 // Perform CLMULH or CLMULR on the shifted inputs, then extract the upper
1653 // 32 bits of the result.
1654 auto Shift = MIRBuilder.buildConstant(sXLen, 32);
1655 auto LHS = MIRBuilder.buildAnyExt(sXLen, MI.getOperand(1));
1656 auto RHS = MIRBuilder.buildAnyExt(sXLen, MI.getOperand(2));
1657 auto ShiftedLHS = MIRBuilder.buildShl(sXLen, LHS, Shift);
1658 auto ShiftedRHS = MIRBuilder.buildShl(sXLen, RHS, Shift);
1659 auto Product = MIRBuilder.buildInstr(MI.getOpcode(), {sXLen},
1660 {ShiftedLHS, ShiftedRHS});
1661 auto High = MIRBuilder.buildLShr(sXLen, Product, Shift);
1662 MIRBuilder.buildTrunc(MI.getOperand(0), High);
1663 MI.eraseFromParent();
1664 return true;
1665 }
1666 case TargetOpcode::G_FCONSTANT: {
1667 const APFloat &FVal = MI.getOperand(1).getFPImm()->getValueAPF();
1668
1669 // Convert G_FCONSTANT to G_CONSTANT.
1670 Register DstReg = MI.getOperand(0).getReg();
1671 MIRBuilder.buildConstant(DstReg, FVal.bitcastToAPInt());
1672
1673 MI.eraseFromParent();
1674 return true;
1675 }
1676 case TargetOpcode::G_CONSTANT: {
1677 const Function &F = MF.getFunction();
1678 // TODO: if PSI and BFI are present, add " ||
1679 // llvm::shouldOptForSize(*CurMBB, PSI, BFI)".
1680 bool ShouldOptForSize = F.hasOptSize();
1681 const ConstantInt *ConstVal = MI.getOperand(1).getCImm();
1682 if (!shouldBeInConstantPool(ConstVal->getValue(), ShouldOptForSize))
1683 return true;
1684 return Helper.lowerConstant(MI);
1685 }
1686 case TargetOpcode::G_SUB:
1687 case TargetOpcode::G_ADD: {
1688 Helper.Observer.changingInstr(MI);
1689 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1690 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ANYEXT);
1691
1692 Register DstALU = MRI.createGenericVirtualRegister(sXLen);
1693
1694 MachineOperand &MO = MI.getOperand(0);
1695 MIRBuilder.setInsertPt(MIRBuilder.getMBB(), ++MIRBuilder.getInsertPt());
1696 auto DstSext = MIRBuilder.buildSExtInReg(sXLen, DstALU, 32);
1697
1698 MIRBuilder.buildInstr(TargetOpcode::G_TRUNC, {MO}, {DstSext});
1699 MO.setReg(DstALU);
1700
1701 Helper.Observer.changedInstr(MI);
1702 return true;
1703 }
1704 case TargetOpcode::G_ASHR:
1705 case TargetOpcode::G_LSHR:
1706 case TargetOpcode::G_SHL: {
1707 if (getIConstantVRegValWithLookThrough(MI.getOperand(2).getReg(), MRI)) {
1708 // We don't need a custom node for shift by constant. Just widen the
1709 // source and the shift amount.
1710 unsigned ExtOpc = TargetOpcode::G_ANYEXT;
1711 if (MI.getOpcode() == TargetOpcode::G_ASHR)
1712 ExtOpc = TargetOpcode::G_SEXT;
1713 else if (MI.getOpcode() == TargetOpcode::G_LSHR)
1714 ExtOpc = TargetOpcode::G_ZEXT;
1715
1716 Helper.Observer.changingInstr(MI);
1717 Helper.widenScalarSrc(MI, sXLen, 1, ExtOpc);
1718 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ZEXT);
1719 Helper.widenScalarDst(MI, sXLen);
1720 Helper.Observer.changedInstr(MI);
1721 return true;
1722 }
1723
1724 Helper.Observer.changingInstr(MI);
1725 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1726 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ANYEXT);
1727 Helper.widenScalarDst(MI, sXLen);
1728 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1729 Helper.Observer.changedInstr(MI);
1730 return true;
1731 }
1732 case TargetOpcode::G_SDIV:
1733 case TargetOpcode::G_UDIV:
1734 case TargetOpcode::G_UREM:
1735 case TargetOpcode::G_ROTL:
1736 case TargetOpcode::G_ROTR: {
1737 Helper.Observer.changingInstr(MI);
1738 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1739 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ANYEXT);
1740 Helper.widenScalarDst(MI, sXLen);
1741 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1742 Helper.Observer.changedInstr(MI);
1743 return true;
1744 }
1745 case TargetOpcode::G_CTLZ:
1746 case TargetOpcode::G_CTTZ:
1747 case TargetOpcode::G_CTLS: {
1748 Helper.Observer.changingInstr(MI);
1749 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1750 Helper.widenScalarDst(MI, sXLen);
1751 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1752 Helper.Observer.changedInstr(MI);
1753 return true;
1754 }
1755 case TargetOpcode::G_FPTOSI:
1756 case TargetOpcode::G_FPTOUI: {
1757 Helper.Observer.changingInstr(MI);
1758 Helper.widenScalarDst(MI, sXLen);
1759 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1761 Helper.Observer.changedInstr(MI);
1762 return true;
1763 }
1764 case TargetOpcode::G_LROUND: {
1765 // The (i32 any_lround) Pat is IsRV32-only; on RV64 lower to
1766 // riscv_fcvt_w_rv64 with FRM_RMM.
1767 Helper.Observer.changingInstr(MI);
1768 Helper.widenScalarDst(MI, sXLen);
1769 MI.setDesc(MIRBuilder.getTII().get(RISCV::G_FCVT_W_RV64));
1771 Helper.Observer.changedInstr(MI);
1772 return true;
1773 }
1774 case TargetOpcode::G_READCYCLECOUNTER:
1775 case TargetOpcode::G_READSTEADYCOUNTER:
1776 return legalizeReadCounter(MI, MIRBuilder, Helper.Observer);
1777 case TargetOpcode::G_IS_FPCLASS: {
1778 Register GISFPCLASS = MI.getOperand(0).getReg();
1779 Register Src = MI.getOperand(1).getReg();
1780 const MachineOperand &ImmOp = MI.getOperand(2);
1781 MachineIRBuilder MIB(MI);
1782
1783 // Turn LLVM IR's floating point classes to that in RISC-V,
1784 // by simply rotating the 10-bit immediate right by two bits.
1785 APInt GFpClassImm(10, static_cast<uint64_t>(ImmOp.getImm()));
1786 auto FClassMask = MIB.buildConstant(sXLen, GFpClassImm.rotr(2).zext(XLen));
1787 auto ConstZero = MIB.buildConstant(sXLen, 0);
1788
1789 auto GFClass = MIB.buildInstr(RISCV::G_FCLASS, {sXLen}, {Src});
1790 auto And = MIB.buildAnd(sXLen, GFClass, FClassMask);
1791 MIB.buildICmp(CmpInst::ICMP_NE, GISFPCLASS, And, ConstZero);
1792
1793 MI.eraseFromParent();
1794 return true;
1795 }
1796 case TargetOpcode::G_BRJT:
1797 return legalizeBRJT(MI, MIRBuilder);
1798 case TargetOpcode::G_VASTART:
1799 return legalizeVAStart(MI, MIRBuilder);
1800 case TargetOpcode::G_VSCALE:
1801 return legalizeVScale(MI, MIRBuilder);
1802 case TargetOpcode::G_ZEXT:
1803 case TargetOpcode::G_SEXT:
1804 case TargetOpcode::G_ANYEXT:
1805 return legalizeExt(MI, MIRBuilder);
1806 case TargetOpcode::G_SPLAT_VECTOR:
1807 return legalizeSplatVector(MI, MIRBuilder);
1808 case TargetOpcode::G_EXTRACT_SUBVECTOR:
1809 return legalizeExtractSubvector(MI, MIRBuilder);
1810 case TargetOpcode::G_INSERT_SUBVECTOR:
1811 return legalizeInsertSubvector(MI, Helper, MIRBuilder);
1812 case TargetOpcode::G_BITREVERSE:
1813 return legalizeBitreverse(MI, MIRBuilder);
1814 case TargetOpcode::G_LOAD:
1815 case TargetOpcode::G_STORE:
1816 return legalizeLoadStore(MI, Helper, MIRBuilder);
1817 }
1818
1819 llvm_unreachable("expected switch to return");
1820}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
constexpr LLT S32
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
#define _
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
This file declares the MachineIRBuilder class.
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t High
uint64_t IntrinsicInst * II
#define P(N)
ppc ctr loops verify
static LLT getLMUL1Ty(LLT VecTy)
static MachineInstrBuilder buildAllOnesMask(LLT VecTy, const SrcOp &VL, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
Creates an all ones mask suitable for masking a vector of type VecTy with vector length VL.
static std::pair< MachineInstrBuilder, MachineInstrBuilder > buildDefaultVLOps(LLT VecTy, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
Gets the two common "VL" operands: an all-ones mask and the vector length.
static LegalityPredicate typeIsLegalBoolVec(unsigned TypeIdx, std::initializer_list< LLT > BoolVecTys, const RISCVSubtarget &ST)
static MachineInstrBuilder buildSplatSplitS64WithVL(const DstOp &Dst, const SrcOp &Passthru, const SrcOp &Scalar, const SrcOp &VL, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
static LegalityPredicate typeIsLegalIntOrFPVec(unsigned TypeIdx, std::initializer_list< LLT > IntOrFPVecTys, const RISCVSubtarget &ST)
static MachineInstrBuilder buildSplatPartsS64WithVL(const DstOp &Dst, const SrcOp &Passthru, Register Lo, Register Hi, const SrcOp &VL, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
static LLT getMaskTypeFor(LLT VecTy)
Return the type of the mask type suitable for masking the provided vector type.
static LegalityPredicate typeIsLegalPtrVec(unsigned TypeIdx, std::initializer_list< LLT > PtrVecTys, const RISCVSubtarget &ST)
static unsigned getRISCVWOpcode(unsigned Opcode)
This file declares the targeting of the Machinelegalizer class for RISC-V.
Value * LHS
APInt bitcastToAPInt() const
Definition APFloat.h:1475
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
LLVM_ABI APInt rotr(unsigned rotateAmt) const
Rotate right by rotateAmt.
Definition APInt.cpp:1199
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
@ ICMP_NE
not equal
Definition InstrTypes.h:762
This is the shared class of boolean and integer constants.
Definition Constants.h:87
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:308
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
Abstract class that contains various methods for clients to notify about changes.
virtual void changingInstr(MachineInstr &MI)=0
This instruction is about to be mutated in some way.
virtual void changedInstr(MachineInstr &MI)=0
This instruction was mutated in some way.
virtual void erasingInstr(MachineInstr &MI)=0
An instruction is about to be erased.
Register getReg(unsigned Idx) const
Access the Idx'th operand as a register and return it.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr bool isScalar() const
static constexpr LLT scalable_vector(unsigned MinNumElements, unsigned ScalarSizeInBits)
Get a low-level scalable vector of some number of elements and element width.
constexpr LLT changeElementType(LLT NewEltTy) const
If this type is a vector, return a vector with the same number of elements but the new element type.
static constexpr LLT vector(ElementCount EC, unsigned ScalarSizeInBits)
Get a low-level vector of some number of elements and element width.
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr bool isVector() const
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr ElementCount getElementCount() const
static constexpr LLT float16()
Get a 16-bit IEEE half value.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LegalizeRuleSet & maxScalar(unsigned TypeIdx, const LLT Ty)
Ensure the scalar is at most as wide as Ty.
LegalizeRuleSet & lower()
The instruction is lowered.
LegalizeRuleSet & clampScalar(unsigned TypeIdx, const LLT MinTy, const LLT MaxTy)
Limit the range of scalar sizes to MinTy and MaxTy.
LegalizeRuleSet & alwaysLegal()
LegalizeRuleSet & customIf(LegalityPredicate Predicate)
LegalizeRuleSet & widenScalarToNextPow2(unsigned TypeIdx, unsigned MinSize=0)
Widen the scalar to the next power of two that is at least MinSize.
LegalizeRuleSet & customFor(std::initializer_list< LLT > Types)
LLVM_ABI void widenScalarSrc(MachineInstr &MI, LLT WideTy, unsigned OpIdx, unsigned ExtOpcode)
Legalize a single operand OpIdx of the machine instruction MI as a Use by extending the operand's typ...
LLVM_ABI LegalizeResult lowerAbsToMaxNeg(MachineInstr &MI)
LLVM_ABI LegalizeResult bitcast(MachineInstr &MI, unsigned TypeIdx, LLT Ty)
Legalize an instruction by replacing the value type.
GISelChangeObserver & Observer
To keep track of changes made by the LegalizerHelper.
LLVM_ABI LegalizeResult widenScalar(MachineInstr &MI, unsigned TypeIdx, LLT WideTy)
Legalize an instruction by performing the operation on a wider scalar type (for example a 16-bit addi...
MachineIRBuilder & MIRBuilder
Expose MIRBuilder so clients can set their own RecordInsertInstruction functions.
LLVM_ABI LegalizeResult lowerConstant(MachineInstr &MI)
LLVM_ABI void widenScalarDst(MachineInstr &MI, LLT WideTy, unsigned OpIdx=0, unsigned TruncOpcode=TargetOpcode::G_TRUNC)
Legalize a single operand OpIdx of the machine instruction MI as a Def by extending the operand's typ...
LegalizeRuleSet & getActionDefinitionsBuilder(unsigned Opcode)
Get the action definition builder for the given opcode.
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineJumpTableInfo * getJumpTableInfo() const
getJumpTableInfo - Return the jump table info object for the current function.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
Helper class to build MachineInstr.
void setInsertPt(MachineBasicBlock &MBB, MachineBasicBlock::iterator II)
Set the insertion point before the specified position.
MachineInstrBuilder buildAdd(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_ADD Op0, Op1.
MachineInstrBuilder buildUndef(const DstOp &Res)
Build and insert Res = IMPLICIT_DEF.
MachineInstrBuilder buildUnmerge(ArrayRef< LLT > Res, const SrcOp &Op)
Build and insert Res0, ... = G_UNMERGE_VALUES Op.
MachineInstrBuilder buildSelect(const DstOp &Res, const SrcOp &Tst, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert a Res = G_SELECT Tst, Op0, Op1.
MachineInstrBuilder buildMul(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_MUL Op0, Op1.
MachineInstrBuilder buildInsertSubvector(const DstOp &Res, const SrcOp &Src0, const SrcOp &Src1, unsigned Index)
Build and insert Res = G_INSERT_SUBVECTOR Src0, Src1, Idx.
MachineInstrBuilder buildAnd(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1)
Build and insert Res = G_AND Op0, Op1.
const TargetInstrInfo & getTII()
MachineInstrBuilder buildICmp(CmpInst::Predicate Pred, const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert a Res = G_ICMP Pred, Op0, Op1.
MachineInstrBuilder buildLShr(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
MachineBasicBlock::iterator getInsertPt()
Current insertion point for new instructions.
MachineInstrBuilder buildZExt(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_ZEXT Op.
MachineInstrBuilder buildVScale(const DstOp &Res, unsigned MinElts)
Build and insert Res = G_VSCALE MinElts.
MachineInstrBuilder buildIntToPtr(const DstOp &Dst, const SrcOp &Src)
Build and insert a G_INTTOPTR instruction.
MachineInstrBuilder buildLoad(const DstOp &Res, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert Res = G_LOAD Addr, MMO.
MachineInstrBuilder buildPtrAdd(const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_PTR_ADD Op0, Op1.
MachineInstrBuilder buildShl(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
MachineInstrBuilder buildStore(const SrcOp &Val, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert G_STORE Val, Addr, MMO.
MachineInstrBuilder buildInstr(unsigned Opcode)
Build and insert <empty> = Opcode <empty>.
MachineInstrBuilder buildFrameIndex(const DstOp &Res, int Idx)
Build and insert Res = G_FRAME_INDEX Idx.
MachineFunction & getMF()
Getter for the function we currently build.
MachineInstrBuilder buildMergeValues(const DstOp &Res, ArrayRef< Register > Ops)
Build and insert Res = G_MERGE_VALUES Op0, ...
MachineInstrBuilder buildTrunc(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_TRUNC Op.
void setDebugLoc(const DebugLoc &DL)
Set the debug location to DL for all the next build instructions.
const MachineBasicBlock & getMBB() const
Getter for the basic block we currently build.
MachineInstrBuilder buildAnyExt(const DstOp &Res, const SrcOp &Op)
Build and insert Res = G_ANYEXT Op0.
MachineRegisterInfo * getMRI()
Getter for MRI.
MachineInstrBuilder buildExtractSubvector(const DstOp &Res, const SrcOp &Src, unsigned Index)
Build and insert Res = G_EXTRACT_SUBVECTOR Src, Idx0.
const DataLayout & getDataLayout() const
MachineInstrBuilder buildBrIndirect(Register Tgt)
Build and insert G_BRINDIRECT Tgt.
MachineInstrBuilder buildSplatVector(const DstOp &Res, const SrcOp &Val)
Build and insert Res = G_SPLAT_VECTOR Val.
MachineInstrBuilder buildLoadInstr(unsigned Opcode, const DstOp &Res, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert Res = <opcode> Addr, MMO.
virtual MachineInstrBuilder buildConstant(const DstOp &Res, const ConstantInt &Val)
Build and insert Res = G_CONSTANT Val.
MachineInstrBuilder buildSExtInReg(const DstOp &Res, const SrcOp &Op, int64_t ImmOp)
Build and insert Res = G_SEXT_INREG Op, ImmOp.
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
LLVM_ABI unsigned getEntrySize(const DataLayout &TD) const
getEntrySize - Return the size of each entry in the jump table.
@ EK_LabelDifference32
EK_LabelDifference32 - Each entry is the address of the block minus the address of the jump table.
@ EK_Custom32
EK_Custom32 - Each entry is a 32-bit value that is custom lowered by the TargetLowering::LowerCustomJ...
@ EK_BlockAddress
EK_BlockAddress - Each entry is a plain address of block, e.g.: .word LBB123.
LLVM_ABI unsigned getEntryAlignment(const DataLayout &TD) const
getEntryAlignment - Return the alignment of each entry in the jump table.
A description of a memory reference used in the backend.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
static MachineOperand CreateImm(int64_t Val)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register createGenericVirtualRegister(LLT Ty, StringRef Name="")
Create and return a new generic virtual register with low-level type Ty.
bool legalizeCustom(LegalizerHelper &Helper, MachineInstr &MI, LostDebugLocObserver &LocObserver) const override
Called for instructions with the Custom LegalizationAction.
bool legalizeIntrinsic(LegalizerHelper &Helper, MachineInstr &MI) const override
RISCVLegalizerInfo(const RISCVSubtarget &ST)
RISCVMachineFunctionInfo - This class is derived from MachineFunctionInfo and contains private RISCV-...
static std::pair< unsigned, unsigned > decomposeSubvectorInsertExtractToSubRegs(MVT VecVT, MVT SubVecVT, unsigned InsertExtractIdx, const RISCVRegisterInfo *TRI)
static RISCVVType::VLMUL getLMUL(MVT VT)
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Register getReg() const
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
self_iterator getIterator()
Definition ilist_node.h:123
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
LLVM_ABI LegalityPredicate typeInSet(unsigned TypeIdx, std::initializer_list< LLT > TypesInit)
True iff the given type index is one of the specified types.
LLVM_ABI LegalityPredicate immIs(unsigned ImmIdx, int64_t Imm)
True iff the immediate at the given index has the specified value.
Predicate any(Predicate P0, Predicate P1)
True iff P0 or P1 are true.
LLVM_ABI LegalityPredicate sizeIs(unsigned TypeIdx, unsigned Size)
True if the total bitwidth of the specified type index is Size bits.
Predicate all(Predicate P0, Predicate P1)
True iff P0 and P1 are true.
LLVM_ABI LegalityPredicate typeIs(unsigned TypeIdx, LLT TypesInit)
True iff the given type index is the specified type.
LLVM_ABI LegalityPredicate immInSet(unsigned ImmIdx, std::initializer_list< int64_t > ImmsInit)
True iff the immediate at the given index has one of the specified values.
LLVM_ABI LegalizeMutation changeTo(unsigned TypeIdx, LLT Ty)
Select this specific type for the given type index.
ImplicitDefMatch m_GImplicitDef()
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
InstSeq generateInstSeq(int64_t Val, const MCSubtargetInfo &STI)
InstSeq generateTwoRegInstSeq(int64_t Val, const MCSubtargetInfo &STI, unsigned &ShiftAmt, unsigned &AddOpc)
SmallVector< Inst, 8 > InstSeq
Definition RISCVMatInt.h:43
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
Invariant opcodes: All instruction sets have these as their low opcodes.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Type * getTypeForLLT(LLT Ty, LLVMContext &C)
Get the type back from LLT.
Definition Utils.cpp:1973
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isAllOnesOrAllOnesSplat(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndefs=false)
Return true if the value is a constant -1 integer or a splatted vector of a constant -1 integer (with...
Definition Utils.cpp:1557
@ Load
The value being inserted comes from a load (InsertElement only).
LLVM_ABI MVT getMVTForLLT(LLT Ty)
Get a rough equivalent of an MVT for a given LLT.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isNullOrNullSplat(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndefs=false)
Return true if the value is a constant 0 integer or a splatted vector of a constant 0 integer (with n...
Definition Utils.cpp:1539
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
std::function< bool(const LegalityQuery &)> LegalityPredicate
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ And
Bitwise or logical AND of integers.
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
The LegalityQuery object bundles together all the information that's needed to decide whether a given...
ArrayRef< LLT > Types
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getJumpTable(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a jump table entry.