LLVM 24.0.0git
InstCombineSimplifyDemanded.cpp
Go to the documentation of this file.
1//===- InstCombineSimplifyDemanded.cpp ------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains logic for simplifying instructions based on information
10// about how they are used.
11//
12//===----------------------------------------------------------------------===//
13
14#include "InstCombineInternal.h"
22
23using namespace llvm;
24using namespace llvm::PatternMatch;
25
26#define DEBUG_TYPE "instcombine"
27
28/// Check to see if the specified operand of the specified instruction is a
29/// constant integer. If so, check to see if there are any bits set in the
30/// constant that are not demanded. If so, shrink the constant and return true.
31static bool ShrinkDemandedConstant(Instruction *I, unsigned OpNo,
32 const APInt &Demanded) {
33 assert(I && "No instruction?");
34 assert(OpNo < I->getNumOperands() && "Operand index too large");
35
36 // The operand must be a constant integer or splat integer.
37 Value *Op = I->getOperand(OpNo);
38 const APInt *C;
39 if (!match(Op, m_APInt(C)))
40 return false;
41
42 // If there are no bits set that aren't demanded, nothing to do.
43 if (C->isSubsetOf(Demanded))
44 return false;
45
46 // This instruction is producing bits that are not demanded. Shrink the RHS.
47 I->setOperand(OpNo, ConstantInt::get(Op->getType(), *C & Demanded));
48
49 return true;
50}
51
52/// Let N = 2 * M.
53/// Given an N-bit integer representing a pack of two M-bit integers,
54/// we can select one of the packed integers by right-shifting by either
55/// zero or M (which is the most straightforward to check if M is a power
56/// of 2), and then isolating the lower M bits. In this case, we can
57/// represent the shift as a select on whether the shr amount is nonzero.
59 const APInt &DemandedMask,
61 unsigned Depth) {
62 assert(I->getOpcode() == Instruction::LShr &&
63 "Only lshr instruction supported");
64
65 uint64_t ShlAmt;
66 Value *Upper, *Lower;
67 if (!match(I->getOperand(0),
70 m_Value(Lower)))))
71 return nullptr;
72
73 if (!isPowerOf2_64(ShlAmt))
74 return nullptr;
75
76 const uint64_t DemandedBitWidth = DemandedMask.getActiveBits();
77 if (DemandedBitWidth > ShlAmt)
78 return nullptr;
79
80 // Check that upper demanded bits are not lost from lshift.
81 if (Upper->getType()->getScalarSizeInBits() < ShlAmt + DemandedBitWidth)
82 return nullptr;
83
84 KnownBits KnownLowerBits = IC.computeKnownBits(Lower, I, Depth);
85 if (!KnownLowerBits.getMaxValue().isIntN(ShlAmt))
86 return nullptr;
87
88 Value *ShrAmt = I->getOperand(1);
89 KnownBits KnownShrBits = IC.computeKnownBits(ShrAmt, I, Depth);
90
91 // Verify that ShrAmt is either exactly ShlAmt (which is a power of 2) or
92 // zero.
93 if (~KnownShrBits.Zero != ShlAmt)
94 return nullptr;
95
98 Value *ShrAmtZ =
100 ShrAmt->getName() + ".z");
101 // There is no existing !prof metadata we can derive the !prof metadata for
102 // this select.
105 Select->takeName(I);
106 return Select;
107}
108
109/// Returns the bitwidth of the given scalar or pointer type. For vector types,
110/// returns the element type's bitwidth.
111static unsigned getBitWidth(Type *Ty, const DataLayout &DL) {
112 if (unsigned BitWidth = Ty->getScalarSizeInBits())
113 return BitWidth;
114
115 return DL.getPointerTypeSizeInBits(Ty);
116}
117
118/// Inst is an integer instruction that SimplifyDemandedBits knows about. See if
119/// the instruction has any properties that allow us to simplify its operands.
121 KnownBits &Known) {
122 APInt DemandedMask(APInt::getAllOnes(Known.getBitWidth()));
123 Value *V = SimplifyDemandedUseBits(&Inst, DemandedMask, Known,
124 SQ.getWithInstruction(&Inst));
125 if (!V) return false;
126 if (V == &Inst) return true;
127 replaceInstUsesWith(Inst, V);
128 return true;
129}
130
131/// Inst is an integer instruction that SimplifyDemandedBits knows about. See if
132/// the instruction has any properties that allow us to simplify its operands.
137
140
142 SQ.getWithInstruction(&Inst));
143 if (!V)
144 return false;
145 if (V == &Inst)
146 return true;
147 replaceInstUsesWith(Inst, V);
148 return true;
149}
150
151/// This form of SimplifyDemandedBits simplifies the specified instruction
152/// operand if possible, updating it in place. It returns true if it made any
153/// change and false otherwise.
155 const APInt &DemandedMask,
157 const SimplifyQuery &Q,
158 unsigned Depth) {
159 Use &U = I->getOperandUse(OpNo);
160 Value *V = U.get();
161 if (isa<Constant>(V)) {
163 return false;
164 }
165
166 Known.resetAll();
167 if (DemandedMask.isZero()) {
168 // Not demanding any bits from V.
169 replaceUse(U, UndefValue::get(V->getType()));
170 return true;
171 }
172
174 if (!VInst) {
176 return false;
177 }
178
180 return false;
181
182 Value *NewVal;
183 if (VInst->hasOneUse()) {
184 // If the instruction has one use, we can directly simplify it.
185 NewVal = SimplifyDemandedUseBits(VInst, DemandedMask, Known, Q, Depth);
186 } else {
187 // If there are multiple uses of this instruction, then we can simplify
188 // VInst to some other value, but not modify the instruction.
189 NewVal =
190 SimplifyMultipleUseDemandedBits(VInst, DemandedMask, Known, Q, Depth);
191 }
192 if (!NewVal) return false;
193 if (Instruction* OpInst = dyn_cast<Instruction>(U))
194 salvageDebugInfo(*OpInst);
195
196 replaceUse(U, NewVal);
197 return true;
198}
199
200/// This function attempts to replace V with a simpler value based on the
201/// demanded bits. When this function is called, it is known that only the bits
202/// set in DemandedMask of the result of V are ever used downstream.
203/// Consequently, depending on the mask and V, it may be possible to replace V
204/// with a constant or one of its operands. In such cases, this function does
205/// the replacement and returns true. In all other cases, it returns false after
206/// analyzing the expression and setting KnownOne and known to be one in the
207/// expression. Known.Zero contains all the bits that are known to be zero in
208/// the expression. These are provided to potentially allow the caller (which
209/// might recursively be SimplifyDemandedBits itself) to simplify the
210/// expression.
211/// Known.One and Known.Zero always follow the invariant that:
212/// Known.One & Known.Zero == 0.
213/// That is, a bit can't be both 1 and 0. The bits in Known.One and Known.Zero
214/// are accurate even for bits not in DemandedMask. Note
215/// also that the bitwidth of V, DemandedMask, Known.Zero and Known.One must all
216/// be the same.
217///
218/// This returns null if it did not change anything and it permits no
219/// simplification. This returns V itself if it did some simplification of V's
220/// operands based on the information about what bits are demanded. This returns
221/// some other non-null value if it found out that V is equal to another value
222/// in the context where the specified bits are demanded, but not for all users.
224 const APInt &DemandedMask,
226 const SimplifyQuery &Q,
227 unsigned Depth) {
228 assert(I != nullptr && "Null pointer of Value???");
229 assert(Depth <= MaxAnalysisRecursionDepth && "Limit Search Depth");
230 uint32_t BitWidth = DemandedMask.getBitWidth();
231 Type *VTy = I->getType();
232 assert(
233 (!VTy->isIntOrIntVectorTy() || VTy->getScalarSizeInBits() == BitWidth) &&
234 Known.getBitWidth() == BitWidth &&
235 "Value *V, DemandedMask and Known must have same BitWidth");
236
237 KnownBits LHSKnown(BitWidth), RHSKnown(BitWidth);
238
239 // Update flags after simplifying an operand based on the fact that some high
240 // order bits are not demanded.
241 auto disableWrapFlagsBasedOnUnusedHighBits = [](Instruction *I,
242 unsigned NLZ) {
243 if (NLZ > 0) {
244 // Disable the nsw and nuw flags here: We can no longer guarantee that
245 // we won't wrap after simplification. Removing the nsw/nuw flags is
246 // legal here because the top bit is not demanded.
247 I->setHasNoSignedWrap(false);
248 I->setHasNoUnsignedWrap(false);
249 }
250 return I;
251 };
252
253 // If the high-bits of an ADD/SUB/MUL are not demanded, then we do not care
254 // about the high bits of the operands.
255 auto simplifyOperandsBasedOnUnusedHighBits = [&](APInt &DemandedFromOps) {
256 unsigned NLZ = DemandedMask.countl_zero();
257 // Right fill the mask of bits for the operands to demand the most
258 // significant bit and all those below it.
259 DemandedFromOps = APInt::getLowBitsSet(BitWidth, BitWidth - NLZ);
260 if (ShrinkDemandedConstant(I, 0, DemandedFromOps) ||
261 SimplifyDemandedBits(I, 0, DemandedFromOps, LHSKnown, Q, Depth + 1) ||
262 ShrinkDemandedConstant(I, 1, DemandedFromOps) ||
263 SimplifyDemandedBits(I, 1, DemandedFromOps, RHSKnown, Q, Depth + 1)) {
264 disableWrapFlagsBasedOnUnusedHighBits(I, NLZ);
265 return true;
266 }
267 return false;
268 };
269
270 switch (I->getOpcode()) {
271 default:
273 break;
274 case Instruction::And: {
275 // If either the LHS or the RHS are Zero, the result is zero.
276 if (SimplifyDemandedBits(I, 1, DemandedMask, RHSKnown, Q, Depth + 1) ||
277 SimplifyDemandedBits(I, 0, DemandedMask & ~RHSKnown.Zero, LHSKnown, Q,
278 Depth + 1))
279 return I;
280
281 Known = analyzeKnownBitsFromAndXorOr(cast<Operator>(I), LHSKnown, RHSKnown,
282 Q, Depth);
283
284 // If the client is only demanding bits that we know, return the known
285 // constant.
286 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
287 return Constant::getIntegerValue(VTy, Known.One);
288
289 // If all of the demanded bits are known 1 on one side, return the other.
290 // These bits cannot contribute to the result of the 'and'.
291 if (DemandedMask.isSubsetOf(LHSKnown.Zero | RHSKnown.One))
292 return I->getOperand(0);
293 if (DemandedMask.isSubsetOf(RHSKnown.Zero | LHSKnown.One))
294 return I->getOperand(1);
295
296 // If the RHS is a constant, see if we can simplify it.
297 if (ShrinkDemandedConstant(I, 1, DemandedMask & ~LHSKnown.Zero))
298 return I;
299
300 break;
301 }
302 case Instruction::Or: {
303 // If either the LHS or the RHS are One, the result is One.
304 if (SimplifyDemandedBits(I, 1, DemandedMask, RHSKnown, Q, Depth + 1) ||
305 SimplifyDemandedBits(I, 0, DemandedMask & ~RHSKnown.One, LHSKnown, Q,
306 Depth + 1)) {
307 // Disjoint flag may not longer hold.
308 I->dropPoisonGeneratingFlags();
309 return I;
310 }
311
312 Known = analyzeKnownBitsFromAndXorOr(cast<Operator>(I), LHSKnown, RHSKnown,
313 Q, Depth);
314
315 // If the client is only demanding bits that we know, return the known
316 // constant.
317 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
318 return Constant::getIntegerValue(VTy, Known.One);
319
320 // If all of the demanded bits are known zero on one side, return the other.
321 // These bits cannot contribute to the result of the 'or'.
322 if (DemandedMask.isSubsetOf(LHSKnown.One | RHSKnown.Zero))
323 return I->getOperand(0);
324 if (DemandedMask.isSubsetOf(RHSKnown.One | LHSKnown.Zero))
325 return I->getOperand(1);
326
327 // If the RHS is a constant, see if we can simplify it.
328 if (ShrinkDemandedConstant(I, 1, DemandedMask))
329 return I;
330
331 // Infer disjoint flag if no common bits are set.
332 if (!cast<PossiblyDisjointInst>(I)->isDisjoint()) {
333 WithCache<const Value *> LHSCache(I->getOperand(0), LHSKnown),
334 RHSCache(I->getOperand(1), RHSKnown);
335 if (haveNoCommonBitsSet(LHSCache, RHSCache, Q)) {
336 cast<PossiblyDisjointInst>(I)->setIsDisjoint(true);
337 return I;
338 }
339 }
340
341 break;
342 }
343 case Instruction::Xor: {
344 if (SimplifyDemandedBits(I, 1, DemandedMask, RHSKnown, Q, Depth + 1) ||
345 SimplifyDemandedBits(I, 0, DemandedMask, LHSKnown, Q, Depth + 1))
346 return I;
347 Value *LHS, *RHS;
348 if (DemandedMask == 1 && match(I->getOperand(0), m_Ctpop(m_Value(LHS))) &&
349 match(I->getOperand(1), m_Ctpop(m_Value(RHS)))) {
350 // (ctpop(X) ^ ctpop(Y)) & 1 --> ctpop(X^Y) & 1
352 Builder.SetInsertPoint(I);
353 auto *Xor = Builder.CreateXor(LHS, RHS);
354 return Builder.CreateUnaryIntrinsic(Intrinsic::ctpop, Xor);
355 }
356
357 Known = analyzeKnownBitsFromAndXorOr(cast<Operator>(I), LHSKnown, RHSKnown,
358 Q, Depth);
359
360 // If the client is only demanding bits that we know, return the known
361 // constant.
362 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
363 return Constant::getIntegerValue(VTy, Known.One);
364
365 // If all of the demanded bits are known zero on one side, return the other.
366 // These bits cannot contribute to the result of the 'xor'.
367 if (DemandedMask.isSubsetOf(RHSKnown.Zero))
368 return I->getOperand(0);
369 if (DemandedMask.isSubsetOf(LHSKnown.Zero))
370 return I->getOperand(1);
371
372 // If all of the demanded bits are known to be zero on one side or the
373 // other, turn this into an *inclusive* or.
374 // e.g. (A & C1)^(B & C2) -> (A & C1)|(B & C2) iff C1&C2 == 0
375 if (DemandedMask.isSubsetOf(RHSKnown.Zero | LHSKnown.Zero)) {
376 Instruction *Or =
377 BinaryOperator::CreateOr(I->getOperand(0), I->getOperand(1));
378 if (DemandedMask.isAllOnes())
379 cast<PossiblyDisjointInst>(Or)->setIsDisjoint(true);
380 Or->takeName(I);
381 return InsertNewInstWith(Or, I->getIterator());
382 }
383
384 // If all of the demanded bits on one side are known, and all of the set
385 // bits on that side are also known to be set on the other side, turn this
386 // into an AND, as we know the bits will be cleared.
387 // e.g. (X | C1) ^ C2 --> (X | C1) & ~C2 iff (C1&C2) == C2
388 if (DemandedMask.isSubsetOf(RHSKnown.Zero|RHSKnown.One) &&
389 RHSKnown.One.isSubsetOf(LHSKnown.One)) {
391 ~RHSKnown.One & DemandedMask);
392 Instruction *And = BinaryOperator::CreateAnd(I->getOperand(0), AndC);
393 return InsertNewInstWith(And, I->getIterator());
394 }
395
396 // If the RHS is a constant, see if we can change it. Don't alter a -1
397 // constant because that's a canonical 'not' op, and that is better for
398 // combining, SCEV, and codegen.
399 const APInt *C;
400 if (match(I->getOperand(1), m_APInt(C)) && !C->isAllOnes()) {
401 if ((*C | ~DemandedMask).isAllOnes()) {
402 // Force bits to 1 to create a 'not' op.
403 I->setOperand(1, ConstantInt::getAllOnesValue(VTy));
404 return I;
405 }
406 // If we can't turn this into a 'not', try to shrink the constant.
407 if (ShrinkDemandedConstant(I, 1, DemandedMask))
408 return I;
409 }
410
411 // If our LHS is an 'and' and if it has one use, and if any of the bits we
412 // are flipping are known to be set, then the xor is just resetting those
413 // bits to zero. We can just knock out bits from the 'and' and the 'xor',
414 // simplifying both of them.
415 if (Instruction *LHSInst = dyn_cast<Instruction>(I->getOperand(0))) {
416 ConstantInt *AndRHS, *XorRHS;
417 if (LHSInst->getOpcode() == Instruction::And && LHSInst->hasOneUse() &&
418 match(I->getOperand(1), m_ConstantInt(XorRHS)) &&
419 match(LHSInst->getOperand(1), m_ConstantInt(AndRHS)) &&
420 (LHSKnown.One & RHSKnown.One & DemandedMask) != 0) {
421 APInt NewMask = ~(LHSKnown.One & RHSKnown.One & DemandedMask);
422
423 Constant *AndC = ConstantInt::get(VTy, NewMask & AndRHS->getValue());
424 Instruction *NewAnd = BinaryOperator::CreateAnd(I->getOperand(0), AndC);
425 InsertNewInstWith(NewAnd, I->getIterator());
426
427 Constant *XorC = ConstantInt::get(VTy, NewMask & XorRHS->getValue());
428 Instruction *NewXor = BinaryOperator::CreateXor(NewAnd, XorC);
429 return InsertNewInstWith(NewXor, I->getIterator());
430 }
431 }
432 break;
433 }
434 case Instruction::Select: {
435 if (SimplifyDemandedBits(I, 2, DemandedMask, RHSKnown, Q, Depth + 1) ||
436 SimplifyDemandedBits(I, 1, DemandedMask, LHSKnown, Q, Depth + 1))
437 return I;
438
439 // If the operands are constants, see if we can simplify them.
440 // This is similar to ShrinkDemandedConstant, but for a select we want to
441 // try to keep the selected constants the same as icmp value constants, if
442 // we can. This helps not break apart (or helps put back together)
443 // canonical patterns like min and max.
444 auto CanonicalizeSelectConstant = [](Instruction *I, unsigned OpNo,
445 const APInt &DemandedMask) {
446 const APInt *SelC;
447 if (!match(I->getOperand(OpNo), m_APInt(SelC)))
448 return false;
449
450 // Get the constant out of the ICmp, if there is one.
451 // Only try this when exactly 1 operand is a constant (if both operands
452 // are constant, the icmp should eventually simplify). Otherwise, we may
453 // invert the transform that reduces set bits and infinite-loop.
454 Value *X;
455 const APInt *CmpC;
456 if (!match(I->getOperand(0), m_ICmp(m_Value(X), m_APInt(CmpC))) ||
457 isa<Constant>(X) || CmpC->getBitWidth() != SelC->getBitWidth())
458 return ShrinkDemandedConstant(I, OpNo, DemandedMask);
459
460 // If the constant is already the same as the ICmp, leave it as-is.
461 if (*CmpC == *SelC)
462 return false;
463 // If the constants are not already the same, but can be with the demand
464 // mask, use the constant value from the ICmp.
465 if ((*CmpC & DemandedMask) == (*SelC & DemandedMask)) {
466 I->setOperand(OpNo, ConstantInt::get(I->getType(), *CmpC));
467 return true;
468 }
469 return ShrinkDemandedConstant(I, OpNo, DemandedMask);
470 };
471 if (CanonicalizeSelectConstant(I, 1, DemandedMask) ||
472 CanonicalizeSelectConstant(I, 2, DemandedMask))
473 return I;
474
475 // Only known if known in both the LHS and RHS.
476 adjustKnownBitsForSelectArm(LHSKnown, I->getOperand(0), I->getOperand(1),
477 /*Invert=*/false, Q, Depth);
478 adjustKnownBitsForSelectArm(RHSKnown, I->getOperand(0), I->getOperand(2),
479 /*Invert=*/true, Q, Depth);
480 Known = LHSKnown.intersectWith(RHSKnown);
481 break;
482 }
483 case Instruction::Trunc: {
484 // If we do not demand the high bits of a right-shifted and truncated value,
485 // then we may be able to truncate it before the shift.
486 Value *X;
487 const APInt *C;
488 if (match(I->getOperand(0), m_OneUse(m_LShr(m_Value(X), m_APInt(C))))) {
489 // The shift amount must be valid (not poison) in the narrow type, and
490 // it must not be greater than the high bits demanded of the result.
491 if (C->ult(VTy->getScalarSizeInBits()) &&
492 C->ule(DemandedMask.countl_zero())) {
493 // trunc (lshr X, C) --> lshr (trunc X), C
495 Builder.SetInsertPoint(I);
496 Value *Trunc = Builder.CreateTrunc(X, VTy);
497 return Builder.CreateLShr(Trunc, C->getZExtValue());
498 }
499 }
500 }
501 [[fallthrough]];
502 case Instruction::ZExt: {
503 unsigned SrcBitWidth = I->getOperand(0)->getType()->getScalarSizeInBits();
504
505 APInt InputDemandedMask = DemandedMask.zextOrTrunc(SrcBitWidth);
506 KnownBits InputKnown(SrcBitWidth);
507 if (SimplifyDemandedBits(I, 0, InputDemandedMask, InputKnown, Q,
508 Depth + 1)) {
509 // For zext nneg, we may have dropped the instruction which made the
510 // input non-negative.
511 I->dropPoisonGeneratingFlags();
512 return I;
513 }
514 assert(InputKnown.getBitWidth() == SrcBitWidth && "Src width changed?");
515 if (I->getOpcode() == Instruction::ZExt && I->hasNonNeg() &&
516 !InputKnown.isNegative())
517 InputKnown.makeNonNegative();
518 Known = InputKnown.zextOrTrunc(BitWidth);
519
520 break;
521 }
522 case Instruction::SExt: {
523 // Compute the bits in the result that are not present in the input.
524 unsigned SrcBitWidth = I->getOperand(0)->getType()->getScalarSizeInBits();
525
526 APInt InputDemandedBits = DemandedMask.trunc(SrcBitWidth);
527
528 // If any of the sign extended bits are demanded, we know that the sign
529 // bit is demanded.
530 if (DemandedMask.getActiveBits() > SrcBitWidth)
531 InputDemandedBits.setBit(SrcBitWidth-1);
532
533 KnownBits InputKnown(SrcBitWidth);
534 if (SimplifyDemandedBits(I, 0, InputDemandedBits, InputKnown, Q, Depth + 1))
535 return I;
536
537 // If the input sign bit is known zero, or if the NewBits are not demanded
538 // convert this into a zero extension.
539 if (InputKnown.isNonNegative() ||
540 DemandedMask.getActiveBits() <= SrcBitWidth) {
541 // Convert to ZExt cast.
542 CastInst *NewCast = new ZExtInst(I->getOperand(0), VTy);
543 NewCast->takeName(I);
544 return InsertNewInstWith(NewCast, I->getIterator());
545 }
546
547 // If the sign bit of the input is known set or clear, then we know the
548 // top bits of the result.
549 Known = InputKnown.sext(BitWidth);
550 break;
551 }
552 case Instruction::Add: {
553 if ((DemandedMask & 1) == 0) {
554 // If we do not need the low bit, try to convert bool math to logic:
555 // add iN (zext i1 X), (sext i1 Y) --> sext (~X & Y) to iN
556 Value *X, *Y;
558 m_OneUse(m_SExt(m_Value(Y))))) &&
559 X->getType()->isIntOrIntVectorTy(1) && X->getType() == Y->getType()) {
560 // Truth table for inputs and output signbits:
561 // X:0 | X:1
562 // ----------
563 // Y:0 | 0 | 0 |
564 // Y:1 | -1 | 0 |
565 // ----------
567 Builder.SetInsertPoint(I);
568 Value *AndNot = Builder.CreateAnd(Builder.CreateNot(X), Y);
569 return Builder.CreateSExt(AndNot, VTy);
570 }
571
572 // add iN (sext i1 X), (sext i1 Y) --> sext (X | Y) to iN
573 if (match(I, m_Add(m_SExt(m_Value(X)), m_SExt(m_Value(Y)))) &&
574 X->getType()->isIntOrIntVectorTy(1) && X->getType() == Y->getType() &&
575 (I->getOperand(0)->hasOneUse() || I->getOperand(1)->hasOneUse())) {
576
577 // Truth table for inputs and output signbits:
578 // X:0 | X:1
579 // -----------
580 // Y:0 | 0 | -1 |
581 // Y:1 | -1 | -1 |
582 // -----------
584 Builder.SetInsertPoint(I);
585 Value *Or = Builder.CreateOr(X, Y);
586 return Builder.CreateSExt(Or, VTy);
587 }
588 }
589
590 // Right fill the mask of bits for the operands to demand the most
591 // significant bit and all those below it.
592 unsigned NLZ = DemandedMask.countl_zero();
593 APInt DemandedFromOps = APInt::getLowBitsSet(BitWidth, BitWidth - NLZ);
594 if (ShrinkDemandedConstant(I, 1, DemandedFromOps) ||
595 SimplifyDemandedBits(I, 1, DemandedFromOps, RHSKnown, Q, Depth + 1))
596 return disableWrapFlagsBasedOnUnusedHighBits(I, NLZ);
597
598 // If low order bits are not demanded and known to be zero in one operand,
599 // then we don't need to demand them from the other operand, since they
600 // can't cause overflow into any bits that are demanded in the result.
601 unsigned NTZ = (~DemandedMask & RHSKnown.Zero).countr_one();
602 APInt DemandedFromLHS = DemandedFromOps;
603 DemandedFromLHS.clearLowBits(NTZ);
604 if (ShrinkDemandedConstant(I, 0, DemandedFromLHS) ||
605 SimplifyDemandedBits(I, 0, DemandedFromLHS, LHSKnown, Q, Depth + 1))
606 return disableWrapFlagsBasedOnUnusedHighBits(I, NLZ);
607
608 unsigned NtzLHS = (~DemandedMask & LHSKnown.Zero).countr_one();
609 APInt DemandedFromRHS = DemandedFromOps;
610 DemandedFromRHS.clearLowBits(NtzLHS);
611 if (ShrinkDemandedConstant(I, 1, DemandedFromRHS))
612 return disableWrapFlagsBasedOnUnusedHighBits(I, NLZ);
613
614 // If we are known to be adding zeros to every bit below
615 // the highest demanded bit, we just return the other side.
616 if (DemandedFromOps.isSubsetOf(RHSKnown.Zero))
617 return I->getOperand(0);
618 if (DemandedFromOps.isSubsetOf(LHSKnown.Zero))
619 return I->getOperand(1);
620
621 // (add X, C) --> (xor X, C) IFF C is equal to the top bit of the DemandMask
622 {
623 const APInt *C;
624 if (match(I->getOperand(1), m_APInt(C)) &&
625 C->isOneBitSet(DemandedMask.getActiveBits() - 1)) {
627 Builder.SetInsertPoint(I);
628 return Builder.CreateXor(I->getOperand(0), ConstantInt::get(VTy, *C));
629 }
630 }
631
632 // Otherwise just compute the known bits of the result.
633 bool NSW = cast<OverflowingBinaryOperator>(I)->hasNoSignedWrap();
634 bool NUW = cast<OverflowingBinaryOperator>(I)->hasNoUnsignedWrap();
635 Known = KnownBits::add(LHSKnown, RHSKnown, NSW, NUW);
636 break;
637 }
638 case Instruction::Sub: {
639 // Right fill the mask of bits for the operands to demand the most
640 // significant bit and all those below it.
641 unsigned NLZ = DemandedMask.countl_zero();
642 APInt DemandedFromOps = APInt::getLowBitsSet(BitWidth, BitWidth - NLZ);
643 if (ShrinkDemandedConstant(I, 1, DemandedFromOps) ||
644 SimplifyDemandedBits(I, 1, DemandedFromOps, RHSKnown, Q, Depth + 1))
645 return disableWrapFlagsBasedOnUnusedHighBits(I, NLZ);
646
647 // If low order bits are not demanded and are known to be zero in RHS,
648 // then we don't need to demand them from LHS, since they can't cause a
649 // borrow from any bits that are demanded in the result.
650 unsigned NTZ = (~DemandedMask & RHSKnown.Zero).countr_one();
651 APInt DemandedFromLHS = DemandedFromOps;
652 DemandedFromLHS.clearLowBits(NTZ);
653 if (ShrinkDemandedConstant(I, 0, DemandedFromLHS) ||
654 SimplifyDemandedBits(I, 0, DemandedFromLHS, LHSKnown, Q, Depth + 1))
655 return disableWrapFlagsBasedOnUnusedHighBits(I, NLZ);
656
657 // If we are known to be subtracting zeros from every bit below
658 // the highest demanded bit, we just return the other side.
659 if (DemandedFromOps.isSubsetOf(RHSKnown.Zero))
660 return I->getOperand(0);
661 // We can't do this with the LHS for subtraction, unless we are only
662 // demanding the LSB.
663 if (DemandedFromOps.isOne() && DemandedFromOps.isSubsetOf(LHSKnown.Zero))
664 return I->getOperand(1);
665
666 // Canonicalize sub mask, X -> ~X
667 const APInt *LHSC;
668 if (match(I->getOperand(0), m_LowBitMask(LHSC)) &&
669 DemandedFromOps.isSubsetOf(*LHSC)) {
671 Builder.SetInsertPoint(I);
672 return Builder.CreateNot(I->getOperand(1));
673 }
674
675 // Otherwise just compute the known bits of the result.
676 bool NSW = cast<OverflowingBinaryOperator>(I)->hasNoSignedWrap();
677 bool NUW = cast<OverflowingBinaryOperator>(I)->hasNoUnsignedWrap();
678 Known = KnownBits::sub(LHSKnown, RHSKnown, NSW, NUW);
679 break;
680 }
681 case Instruction::Mul: {
682 APInt DemandedFromOps;
683 if (simplifyOperandsBasedOnUnusedHighBits(DemandedFromOps))
684 return I;
685
686 if (DemandedMask.isPowerOf2()) {
687 // The LSB of X*Y is set only if (X & 1) == 1 and (Y & 1) == 1.
688 // If we demand exactly one bit N and we have "X * (C' << N)" where C' is
689 // odd (has LSB set), then the left-shifted low bit of X is the answer.
690 unsigned CTZ = DemandedMask.countr_zero();
691 const APInt *C;
692 if (match(I->getOperand(1), m_APInt(C)) && C->countr_zero() == CTZ) {
693 Constant *ShiftC = ConstantInt::get(VTy, CTZ);
694 Instruction *Shl = BinaryOperator::CreateShl(I->getOperand(0), ShiftC);
695 return InsertNewInstWith(Shl, I->getIterator());
696 }
697 }
698 // For a squared value "X * X", the bottom 2 bits are 0 and X[0] because:
699 // X * X is odd iff X is odd.
700 // 'Quadratic Reciprocity': X * X -> 0 for bit[1]
701 if (I->getOperand(0) == I->getOperand(1) && DemandedMask.ult(4)) {
702 Constant *One = ConstantInt::get(VTy, 1);
703 Instruction *And1 = BinaryOperator::CreateAnd(I->getOperand(0), One);
704 return InsertNewInstWith(And1, I->getIterator());
705 }
706
708 break;
709 }
710 case Instruction::Shl: {
711 const APInt *SA;
712 if (match(I->getOperand(1), m_APInt(SA))) {
713 const APInt *ShrAmt;
714 if (match(I->getOperand(0), m_Shr(m_Value(), m_APInt(ShrAmt))))
715 if (Instruction *Shr = dyn_cast<Instruction>(I->getOperand(0)))
716 if (Value *R = simplifyShrShlDemandedBits(Shr, *ShrAmt, I, *SA,
717 DemandedMask, Known))
718 return R;
719
720 // Do not simplify if shl is part of funnel-shift pattern
721 if (I->hasOneUse()) {
722 Instruction *Inst = I->user_back();
723 if (Inst->getOpcode() == BinaryOperator::Or) {
724 if (auto Opt = convertOrOfShiftsToFunnelShift(*Inst)) {
725 auto [IID, FShiftArgs] = *Opt;
726 if ((IID == Intrinsic::fshl || IID == Intrinsic::fshr) &&
727 FShiftArgs[0] == FShiftArgs[1]) {
729 break;
730 }
731 }
732 }
733 }
734
735 // We only want bits that already match the signbit then we don't
736 // need to shift.
737 uint64_t ShiftAmt = SA->getLimitedValue(BitWidth - 1);
738 if (DemandedMask.countr_zero() >= ShiftAmt) {
739 if (I->hasNoSignedWrap()) {
740 unsigned NumHiDemandedBits = BitWidth - DemandedMask.countr_zero();
741 unsigned SignBits =
742 ComputeNumSignBits(I->getOperand(0), Q.CtxI, Depth + 1);
743 if (SignBits > ShiftAmt && SignBits - ShiftAmt >= NumHiDemandedBits)
744 return I->getOperand(0);
745 }
746
747 // If we can pre-shift a right-shifted constant to the left without
748 // losing any high bits and we don't demand the low bits, then eliminate
749 // the left-shift:
750 // (C >> X) << LeftShiftAmtC --> (C << LeftShiftAmtC) >> X
751 Value *X;
752 Constant *C;
753 if (match(I->getOperand(0), m_LShr(m_ImmConstant(C), m_Value(X)))) {
754 Constant *LeftShiftAmtC = ConstantInt::get(VTy, ShiftAmt);
755 Constant *NewC = ConstantFoldBinaryOpOperands(Instruction::Shl, C,
756 LeftShiftAmtC, DL);
757 if (ConstantFoldBinaryOpOperands(Instruction::LShr, NewC,
758 LeftShiftAmtC, DL) == C) {
759 Instruction *Lshr = BinaryOperator::CreateLShr(NewC, X);
760 return InsertNewInstWith(Lshr, I->getIterator());
761 }
762 }
763 }
764
765 APInt DemandedMaskIn(DemandedMask.lshr(ShiftAmt));
766
767 // If the shift is NUW/NSW, then it does demand the high bits.
769 if (IOp->hasNoSignedWrap())
770 DemandedMaskIn.setHighBits(ShiftAmt+1);
771 else if (IOp->hasNoUnsignedWrap())
772 DemandedMaskIn.setHighBits(ShiftAmt);
773
774 if (SimplifyDemandedBits(I, 0, DemandedMaskIn, Known, Q, Depth + 1))
775 return I;
776
779 /* NUW */ IOp->hasNoUnsignedWrap(),
780 /* NSW */ IOp->hasNoSignedWrap());
781 } else {
782 // This is a variable shift, so we can't shift the demand mask by a known
783 // amount. But if we are not demanding high bits, then we are not
784 // demanding those bits from the pre-shifted operand either.
785 if (unsigned CTLZ = DemandedMask.countl_zero()) {
786 APInt DemandedFromOp(APInt::getLowBitsSet(BitWidth, BitWidth - CTLZ));
787 if (SimplifyDemandedBits(I, 0, DemandedFromOp, Known, Q, Depth + 1)) {
788 // We can't guarantee that nsw/nuw hold after simplifying the operand.
789 I->dropPoisonGeneratingFlags();
790 return I;
791 }
792 }
794 }
795 break;
796 }
797 case Instruction::LShr: {
798 const APInt *SA;
799 if (match(I->getOperand(1), m_APInt(SA))) {
800 uint64_t ShiftAmt = SA->getLimitedValue(BitWidth-1);
801
802 // Do not simplify if lshr is part of funnel-shift pattern
803 if (I->hasOneUse()) {
804 Instruction *Inst = I->user_back();
805 if (Inst->getOpcode() == BinaryOperator::Or) {
806 if (auto Opt = convertOrOfShiftsToFunnelShift(*Inst)) {
807 auto [IID, FShiftArgs] = *Opt;
808 if ((IID == Intrinsic::fshl || IID == Intrinsic::fshr) &&
809 FShiftArgs[0] == FShiftArgs[1]) {
811 break;
812 }
813 }
814 }
815 }
816
817 // If we are just demanding the shifted sign bit and below, then this can
818 // be treated as an ASHR in disguise.
819 if (DemandedMask.countl_zero() >= ShiftAmt) {
820 // If we only want bits that already match the signbit then we don't
821 // need to shift.
822 unsigned NumHiDemandedBits = BitWidth - DemandedMask.countr_zero();
823 unsigned SignBits =
824 ComputeNumSignBits(I->getOperand(0), Q.CtxI, Depth + 1);
825 if (SignBits >= NumHiDemandedBits)
826 return I->getOperand(0);
827
828 // If we can pre-shift a left-shifted constant to the right without
829 // losing any low bits (we already know we don't demand the high bits),
830 // then eliminate the right-shift:
831 // (C << X) >> RightShiftAmtC --> (C >> RightShiftAmtC) << X
832 Value *X;
833 Constant *C;
834 if (match(I->getOperand(0), m_Shl(m_ImmConstant(C), m_Value(X)))) {
835 Constant *RightShiftAmtC = ConstantInt::get(VTy, ShiftAmt);
836 Constant *NewC = ConstantFoldBinaryOpOperands(Instruction::LShr, C,
837 RightShiftAmtC, DL);
838 if (ConstantFoldBinaryOpOperands(Instruction::Shl, NewC,
839 RightShiftAmtC, DL) == C) {
840 Instruction *Shl = BinaryOperator::CreateShl(NewC, X);
841 return InsertNewInstWith(Shl, I->getIterator());
842 }
843 }
844
845 const APInt *Factor;
846 if (match(I->getOperand(0),
847 m_OneUse(m_Mul(m_Value(X), m_APInt(Factor)))) &&
848 Factor->countr_zero() >= ShiftAmt) {
849 BinaryOperator *Mul = BinaryOperator::CreateMul(
850 X, ConstantInt::get(X->getType(), Factor->lshr(ShiftAmt)));
851 return InsertNewInstWith(Mul, I->getIterator());
852 }
853 }
854
855 // Unsigned shift right.
856 APInt DemandedMaskIn(DemandedMask.shl(ShiftAmt));
857 if (SimplifyDemandedBits(I, 0, DemandedMaskIn, Known, Q, Depth + 1)) {
858 // exact flag may not longer hold.
859 I->dropPoisonGeneratingFlags();
860 return I;
861 }
862 Known >>= ShiftAmt;
863 if (ShiftAmt)
864 Known.Zero.setHighBits(ShiftAmt); // high bits known zero.
865 break;
866 }
867 if (Value *V =
868 simplifyShiftSelectingPackedElement(I, DemandedMask, *this, Depth))
869 return V;
870
872 break;
873 }
874 case Instruction::AShr: {
875 unsigned SignBits = ComputeNumSignBits(I->getOperand(0), Q.CtxI, Depth + 1);
876
877 // If we only want bits that already match the signbit then we don't need
878 // to shift.
879 unsigned NumHiDemandedBits = BitWidth - DemandedMask.countr_zero();
880 if (SignBits >= NumHiDemandedBits)
881 return I->getOperand(0);
882
883 // If this is an arithmetic shift right and only the low-bit is set, we can
884 // always convert this into a logical shr, even if the shift amount is
885 // variable. The low bit of the shift cannot be an input sign bit unless
886 // the shift amount is >= the size of the datatype, which is undefined.
887 if (DemandedMask.isOne()) {
888 // Perform the logical shift right.
889 Instruction *NewVal = BinaryOperator::CreateLShr(
890 I->getOperand(0), I->getOperand(1), I->getName());
891 return InsertNewInstWith(NewVal, I->getIterator());
892 }
893
894 const APInt *SA;
895 if (match(I->getOperand(1), m_APInt(SA))) {
896 uint32_t ShiftAmt = SA->getLimitedValue(BitWidth-1);
897
898 // Signed shift right.
899 APInt DemandedMaskIn(DemandedMask.shl(ShiftAmt));
900 // If any of the bits being shifted in are demanded, then we should set
901 // the sign bit as demanded.
902 bool ShiftedInBitsDemanded = DemandedMask.countl_zero() < ShiftAmt;
903 if (ShiftedInBitsDemanded)
904 DemandedMaskIn.setSignBit();
905 if (SimplifyDemandedBits(I, 0, DemandedMaskIn, Known, Q, Depth + 1)) {
906 // exact flag may not longer hold.
907 I->dropPoisonGeneratingFlags();
908 return I;
909 }
910
911 // If the input sign bit is known to be zero, or if none of the shifted in
912 // bits are demanded, turn this into an unsigned shift right.
913 if (Known.Zero[BitWidth - 1] || !ShiftedInBitsDemanded) {
914 BinaryOperator *LShr = BinaryOperator::CreateLShr(I->getOperand(0),
915 I->getOperand(1));
916 LShr->setIsExact(cast<BinaryOperator>(I)->isExact());
917 LShr->takeName(I);
918 return InsertNewInstWith(LShr, I->getIterator());
919 }
920
923 ShiftAmt != 0, I->isExact());
924 } else {
926 }
927 break;
928 }
929 case Instruction::UDiv: {
930 // UDiv doesn't demand low bits that are zero in the divisor.
931 const APInt *SA;
932 if (match(I->getOperand(1), m_APInt(SA))) {
933 // TODO: Take the demanded mask of the result into account.
934 unsigned RHSTrailingZeros = SA->countr_zero();
935 APInt DemandedMaskIn =
936 APInt::getHighBitsSet(BitWidth, BitWidth - RHSTrailingZeros);
937 if (SimplifyDemandedBits(I, 0, DemandedMaskIn, LHSKnown, Q, Depth + 1)) {
938 // We can't guarantee that "exact" is still true after changing the
939 // the dividend.
940 I->dropPoisonGeneratingFlags();
941 return I;
942 }
943
945 cast<BinaryOperator>(I)->isExact());
946 } else {
948 }
949 break;
950 }
951 case Instruction::SRem: {
952 const APInt *Rem;
953 if (match(I->getOperand(1), m_APInt(Rem)) && Rem->isPowerOf2()) {
954 if (DemandedMask.ult(*Rem)) // srem won't affect demanded bits
955 return I->getOperand(0);
956
957 APInt LowBits = *Rem - 1;
958 APInt Mask2 = LowBits | APInt::getSignMask(BitWidth);
959 if (SimplifyDemandedBits(I, 0, Mask2, LHSKnown, Q, Depth + 1))
960 return I;
962 break;
963 }
964
966 break;
967 }
968 case Instruction::Call: {
969 bool KnownBitsComputed = false;
971 switch (II->getIntrinsicID()) {
972 case Intrinsic::abs: {
973 if (DemandedMask == 1)
974 return II->getArgOperand(0);
975 break;
976 }
977 case Intrinsic::ctpop: {
978 // Checking if the number of clear bits is odd (parity)? If the type has
979 // an even number of bits, that's the same as checking if the number of
980 // set bits is odd, so we can eliminate the 'not' op.
981 Value *X;
982 if (DemandedMask == 1 && VTy->getScalarSizeInBits() % 2 == 0 &&
983 match(II->getArgOperand(0), m_Not(m_Value(X)))) {
985 II->getModule(), Intrinsic::ctpop, VTy);
986 return InsertNewInstWith(CallInst::Create(Ctpop, {X}), I->getIterator());
987 }
988 break;
989 }
990 case Intrinsic::bswap: {
991 // If the only bits demanded come from one byte of the bswap result,
992 // just shift the input byte into position to eliminate the bswap.
993 unsigned NLZ = DemandedMask.countl_zero();
994 unsigned NTZ = DemandedMask.countr_zero();
995
996 // Round NTZ down to the next byte. If we have 11 trailing zeros, then
997 // we need all the bits down to bit 8. Likewise, round NLZ. If we
998 // have 14 leading zeros, round to 8.
999 NLZ = alignDown(NLZ, 8);
1000 NTZ = alignDown(NTZ, 8);
1001 // If we need exactly one byte, we can do this transformation.
1002 if (BitWidth - NLZ - NTZ == 8) {
1003 // Replace this with either a left or right shift to get the byte into
1004 // the right place.
1005 Instruction *NewVal;
1006 if (NLZ > NTZ)
1007 NewVal = BinaryOperator::CreateLShr(
1008 II->getArgOperand(0), ConstantInt::get(VTy, NLZ - NTZ));
1009 else
1010 NewVal = BinaryOperator::CreateShl(
1011 II->getArgOperand(0), ConstantInt::get(VTy, NTZ - NLZ));
1012 NewVal->takeName(I);
1013 return InsertNewInstWith(NewVal, I->getIterator());
1014 }
1015 break;
1016 }
1017 case Intrinsic::ptrmask: {
1018 unsigned MaskWidth = I->getOperand(1)->getType()->getScalarSizeInBits();
1019 RHSKnown = KnownBits(MaskWidth);
1020 // If either the LHS or the RHS are Zero, the result is zero.
1021 if (SimplifyDemandedBits(I, 0, DemandedMask, LHSKnown, Q, Depth + 1) ||
1023 I, 1, (DemandedMask & ~LHSKnown.Zero).zextOrTrunc(MaskWidth),
1024 RHSKnown, Q, Depth + 1))
1025 return I;
1026
1027 // TODO: Should be 1-extend
1028 RHSKnown = RHSKnown.anyextOrTrunc(BitWidth);
1029
1030 Known = LHSKnown & RHSKnown;
1031 KnownBitsComputed = true;
1032
1033 // If the client is only demanding bits we know to be zero, return
1034 // `llvm.ptrmask(p, 0)`. We can't return `null` here due to pointer
1035 // provenance, but making the mask zero will be easily optimizable in
1036 // the backend.
1037 if (DemandedMask.isSubsetOf(Known.Zero) &&
1038 !match(I->getOperand(1), m_Zero()))
1039 return replaceOperand(
1040 *I, 1, Constant::getNullValue(I->getOperand(1)->getType()));
1041
1042 // Mask in demanded space does nothing.
1043 // NOTE: We may have attributes associated with the return value of the
1044 // llvm.ptrmask intrinsic that will be lost when we just return the
1045 // operand. We should try to preserve them.
1046 if (DemandedMask.isSubsetOf(RHSKnown.One | LHSKnown.Zero))
1047 return I->getOperand(0);
1048
1049 // If the RHS is a constant, see if we can simplify it.
1051 I, 1, (DemandedMask & ~LHSKnown.Zero).zextOrTrunc(MaskWidth)))
1052 return I;
1053
1054 // Combine:
1055 // (ptrmask (getelementptr i8, ptr p, imm i), imm mask)
1056 // -> (ptrmask (getelementptr i8, ptr p, imm (i & mask)), imm mask)
1057 // where only the low bits known to be zero in the pointer are changed
1058 Value *InnerPtr;
1059 uint64_t GEPIndex;
1060 uint64_t PtrMaskImmediate;
1062 m_PtrAdd(m_Value(InnerPtr), m_ConstantInt(GEPIndex)),
1063 m_ConstantInt(PtrMaskImmediate)))) {
1064
1065 LHSKnown = computeKnownBits(InnerPtr, I, Depth + 1);
1066 if (!LHSKnown.isZero()) {
1067 const unsigned trailingZeros = LHSKnown.countMinTrailingZeros();
1068 uint64_t PointerAlignBits = (uint64_t(1) << trailingZeros) - 1;
1069
1070 uint64_t HighBitsGEPIndex = GEPIndex & ~PointerAlignBits;
1071 uint64_t MaskedLowBitsGEPIndex =
1072 GEPIndex & PointerAlignBits & PtrMaskImmediate;
1073
1074 uint64_t MaskedGEPIndex = HighBitsGEPIndex | MaskedLowBitsGEPIndex;
1075
1076 if (MaskedGEPIndex != GEPIndex) {
1077 auto *GEP = cast<GEPOperator>(II->getArgOperand(0));
1078 Builder.SetInsertPoint(I);
1079 Type *GEPIndexType =
1080 DL.getIndexType(GEP->getPointerOperand()->getType());
1081 Value *MaskedGEP = Builder.CreateGEP(
1082 GEP->getSourceElementType(), InnerPtr,
1083 ConstantInt::get(GEPIndexType, MaskedGEPIndex),
1084 GEP->getName(), GEP->isInBounds());
1085
1086 replaceOperand(*I, 0, MaskedGEP);
1087 return I;
1088 }
1089 }
1090 }
1091
1092 break;
1093 }
1094
1095 case Intrinsic::fshr:
1096 case Intrinsic::fshl: {
1097 const APInt *SA;
1098 if (!match(I->getOperand(2), m_APInt(SA)))
1099 break;
1100
1101 // Normalize to funnel shift left. APInt shifts of BitWidth are well-
1102 // defined, so no need to special-case zero shifts here.
1103 uint64_t ShiftAmt = SA->urem(BitWidth);
1104 if (II->getIntrinsicID() == Intrinsic::fshr)
1105 ShiftAmt = BitWidth - ShiftAmt;
1106
1107 APInt DemandedMaskLHS(DemandedMask.lshr(ShiftAmt));
1108 APInt DemandedMaskRHS(DemandedMask.shl(BitWidth - ShiftAmt));
1109 if (I->getOperand(0) != I->getOperand(1)) {
1110 if (SimplifyDemandedBits(I, 0, DemandedMaskLHS, LHSKnown, Q,
1111 Depth + 1) ||
1112 SimplifyDemandedBits(I, 1, DemandedMaskRHS, RHSKnown, Q,
1113 Depth + 1)) {
1114 // Range attribute or metadata may no longer hold.
1115 I->dropPoisonGeneratingAnnotations();
1116 return I;
1117 }
1118 } else { // fshl is a rotate
1119 // Avoid converting rotate into funnel shift.
1120 // Only simplify if one operand is constant.
1121 LHSKnown = computeKnownBits(I->getOperand(0), I, Depth + 1);
1122 if (DemandedMaskLHS.isSubsetOf(LHSKnown.Zero | LHSKnown.One) &&
1123 !match(I->getOperand(0), m_SpecificInt(LHSKnown.One))) {
1124 replaceOperand(*I, 0, Constant::getIntegerValue(VTy, LHSKnown.One));
1125 // Range attribute or metadata may no longer hold.
1126 I->dropPoisonGeneratingAnnotations();
1127 return I;
1128 }
1129
1130 RHSKnown = computeKnownBits(I->getOperand(1), I, Depth + 1);
1131 if (DemandedMaskRHS.isSubsetOf(RHSKnown.Zero | RHSKnown.One) &&
1132 !match(I->getOperand(1), m_SpecificInt(RHSKnown.One))) {
1133 replaceOperand(*I, 1, Constant::getIntegerValue(VTy, RHSKnown.One));
1134 // Range attribute or metadata may no longer hold.
1135 I->dropPoisonGeneratingAnnotations();
1136 return I;
1137 }
1138 }
1139
1140 LHSKnown <<= ShiftAmt;
1141 RHSKnown >>= BitWidth - ShiftAmt;
1142 Known = LHSKnown.unionWith(RHSKnown);
1143 KnownBitsComputed = true;
1144 break;
1145 }
1146 case Intrinsic::umax: {
1147 // UMax(A, C) == A if ...
1148 // The lowest non-zero bit of DemandMask is higher than the highest
1149 // non-zero bit of C.
1150 const APInt *C;
1151 unsigned CTZ = DemandedMask.countr_zero();
1152 if (match(II->getArgOperand(1), m_APInt(C)) &&
1153 CTZ >= C->getActiveBits())
1154 return II->getArgOperand(0);
1155 break;
1156 }
1157 case Intrinsic::umin: {
1158 // UMin(A, C) == A if ...
1159 // The lowest non-zero bit of DemandMask is higher than the highest
1160 // non-one bit of C.
1161 // This comes from using DeMorgans on the above umax example.
1162 const APInt *C;
1163 unsigned CTZ = DemandedMask.countr_zero();
1164 if (match(II->getArgOperand(1), m_APInt(C)) &&
1165 CTZ >= C->getBitWidth() - C->countl_one())
1166 return II->getArgOperand(0);
1167 break;
1168 }
1169 default: {
1170 // Handle target specific intrinsics
1171 std::optional<Value *> V = targetSimplifyDemandedUseBitsIntrinsic(
1172 *II, DemandedMask, Known, KnownBitsComputed);
1173 if (V)
1174 return *V;
1175 break;
1176 }
1177 }
1178 }
1179
1180 if (!KnownBitsComputed)
1182 break;
1183 }
1184 }
1185
1186 if (I->getType()->isPointerTy()) {
1187 Align Alignment = I->getPointerAlignment(DL);
1188 Known.Zero.setLowBits(Log2(Alignment));
1189 }
1190
1191 // If the client is only demanding bits that we know, return the known
1192 // constant. We can't directly simplify pointers as a constant because of
1193 // pointer provenance.
1194 // TODO: We could return `(inttoptr const)` for pointers.
1195 if (!I->getType()->isPointerTy() &&
1196 DemandedMask.isSubsetOf(Known.Zero | Known.One))
1197 return Constant::getIntegerValue(VTy, Known.One);
1198
1199 if (CLOpts.verify_known_bits) {
1200 KnownBits ReferenceKnown = llvm::computeKnownBits(I, Q, Depth);
1201 if (Known != ReferenceKnown) {
1202 errs() << "Mismatched known bits for " << *I << " in "
1203 << I->getFunction()->getName() << "\n";
1204 errs() << "computeKnownBits(): " << ReferenceKnown << "\n";
1205 errs() << "SimplifyDemandedBits(): " << Known << "\n";
1206 std::abort();
1207 }
1208 }
1209
1210 return nullptr;
1211}
1212
1213/// Helper routine of SimplifyDemandedUseBits. It computes Known
1214/// bits. It also tries to handle simplifications that can be done based on
1215/// DemandedMask, but without modifying the Instruction.
1217 Instruction *I, const APInt &DemandedMask, KnownBits &Known,
1218 const SimplifyQuery &Q, unsigned Depth) {
1219 unsigned BitWidth = DemandedMask.getBitWidth();
1220 Type *ITy = I->getType();
1221
1222 KnownBits LHSKnown(BitWidth);
1223 KnownBits RHSKnown(BitWidth);
1224
1225 // Despite the fact that we can't simplify this instruction in all User's
1226 // context, we can at least compute the known bits, and we can
1227 // do simplifications that apply to *just* the one user if we know that
1228 // this instruction has a simpler value in that context.
1229 switch (I->getOpcode()) {
1230 case Instruction::And: {
1231 llvm::computeKnownBits(I->getOperand(1), RHSKnown, Q, Depth + 1);
1232 llvm::computeKnownBits(I->getOperand(0), LHSKnown, Q, Depth + 1);
1233 Known = analyzeKnownBitsFromAndXorOr(cast<Operator>(I), LHSKnown, RHSKnown,
1234 Q, Depth);
1236
1237 // If the client is only demanding bits that we know, return the known
1238 // constant.
1239 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
1240 return Constant::getIntegerValue(ITy, Known.One);
1241
1242 // If all of the demanded bits are known 1 on one side, return the other.
1243 // These bits cannot contribute to the result of the 'and' in this context.
1244 if (DemandedMask.isSubsetOf(LHSKnown.Zero | RHSKnown.One))
1245 return I->getOperand(0);
1246 if (DemandedMask.isSubsetOf(RHSKnown.Zero | LHSKnown.One))
1247 return I->getOperand(1);
1248
1249 break;
1250 }
1251 case Instruction::Or: {
1252 llvm::computeKnownBits(I->getOperand(1), RHSKnown, Q, Depth + 1);
1253 llvm::computeKnownBits(I->getOperand(0), LHSKnown, Q, Depth + 1);
1254 Known = analyzeKnownBitsFromAndXorOr(cast<Operator>(I), LHSKnown, RHSKnown,
1255 Q, Depth);
1257
1258 // If the client is only demanding bits that we know, return the known
1259 // constant.
1260 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
1261 return Constant::getIntegerValue(ITy, Known.One);
1262
1263 // We can simplify (X|Y) -> X or Y in the user's context if we know that
1264 // only bits from X or Y are demanded.
1265 // If all of the demanded bits are known zero on one side, return the other.
1266 // These bits cannot contribute to the result of the 'or' in this context.
1267 if (DemandedMask.isSubsetOf(LHSKnown.One | RHSKnown.Zero))
1268 return I->getOperand(0);
1269 if (DemandedMask.isSubsetOf(RHSKnown.One | LHSKnown.Zero))
1270 return I->getOperand(1);
1271
1272 break;
1273 }
1274 case Instruction::Xor: {
1275 llvm::computeKnownBits(I->getOperand(1), RHSKnown, Q, Depth + 1);
1276 llvm::computeKnownBits(I->getOperand(0), LHSKnown, Q, Depth + 1);
1277 Known = analyzeKnownBitsFromAndXorOr(cast<Operator>(I), LHSKnown, RHSKnown,
1278 Q, Depth);
1280
1281 // If the client is only demanding bits that we know, return the known
1282 // constant.
1283 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
1284 return Constant::getIntegerValue(ITy, Known.One);
1285
1286 // We can simplify (X^Y) -> X or Y in the user's context if we know that
1287 // only bits from X or Y are demanded.
1288 // If all of the demanded bits are known zero on one side, return the other.
1289 if (DemandedMask.isSubsetOf(RHSKnown.Zero))
1290 return I->getOperand(0);
1291 if (DemandedMask.isSubsetOf(LHSKnown.Zero))
1292 return I->getOperand(1);
1293
1294 break;
1295 }
1296 case Instruction::Add: {
1297 unsigned NLZ = DemandedMask.countl_zero();
1298 APInt DemandedFromOps = APInt::getLowBitsSet(BitWidth, BitWidth - NLZ);
1299
1300 // If an operand adds zeros to every bit below the highest demanded bit,
1301 // that operand doesn't change the result. Return the other side.
1302 llvm::computeKnownBits(I->getOperand(1), RHSKnown, Q, Depth + 1);
1303 if (DemandedFromOps.isSubsetOf(RHSKnown.Zero))
1304 return I->getOperand(0);
1305
1306 llvm::computeKnownBits(I->getOperand(0), LHSKnown, Q, Depth + 1);
1307 if (DemandedFromOps.isSubsetOf(LHSKnown.Zero))
1308 return I->getOperand(1);
1309
1310 bool NSW = cast<OverflowingBinaryOperator>(I)->hasNoSignedWrap();
1311 bool NUW = cast<OverflowingBinaryOperator>(I)->hasNoUnsignedWrap();
1312 Known = KnownBits::add(LHSKnown, RHSKnown, NSW, NUW);
1314 break;
1315 }
1316 case Instruction::Sub: {
1317 unsigned NLZ = DemandedMask.countl_zero();
1318 APInt DemandedFromOps = APInt::getLowBitsSet(BitWidth, BitWidth - NLZ);
1319
1320 // If an operand subtracts zeros from every bit below the highest demanded
1321 // bit, that operand doesn't change the result. Return the other side.
1322 llvm::computeKnownBits(I->getOperand(1), RHSKnown, Q, Depth + 1);
1323 if (DemandedFromOps.isSubsetOf(RHSKnown.Zero))
1324 return I->getOperand(0);
1325
1326 bool NSW = cast<OverflowingBinaryOperator>(I)->hasNoSignedWrap();
1327 bool NUW = cast<OverflowingBinaryOperator>(I)->hasNoUnsignedWrap();
1328 llvm::computeKnownBits(I->getOperand(0), LHSKnown, Q, Depth + 1);
1329 Known = KnownBits::sub(LHSKnown, RHSKnown, NSW, NUW);
1331 break;
1332 }
1333 case Instruction::AShr: {
1334 // Compute the Known bits to simplify things downstream.
1336
1337 // If this user is only demanding bits that we know, return the known
1338 // constant.
1339 if (DemandedMask.isSubsetOf(Known.Zero | Known.One))
1340 return Constant::getIntegerValue(ITy, Known.One);
1341
1342 // If the right shift operand 0 is a result of a left shift by the same
1343 // amount, this is probably a zero/sign extension, which may be unnecessary,
1344 // if we do not demand any of the new sign bits. So, return the original
1345 // operand instead.
1346 const APInt *ShiftRC;
1347 const APInt *ShiftLC;
1348 Value *X;
1349 unsigned BitWidth = DemandedMask.getBitWidth();
1350 if (match(I,
1351 m_AShr(m_Shl(m_Value(X), m_APInt(ShiftLC)), m_APInt(ShiftRC))) &&
1352 ShiftLC == ShiftRC && ShiftLC->ult(BitWidth) &&
1353 DemandedMask.isSubsetOf(APInt::getLowBitsSet(
1354 BitWidth, BitWidth - ShiftRC->getZExtValue()))) {
1355 return X;
1356 }
1357
1358 break;
1359 }
1360 default:
1361 // Compute the Known bits to simplify things downstream.
1363
1364 // If this user is only demanding bits that we know, return the known
1365 // constant.
1366 if (DemandedMask.isSubsetOf(Known.Zero|Known.One))
1367 return Constant::getIntegerValue(ITy, Known.One);
1368
1369 break;
1370 }
1371
1372 return nullptr;
1373}
1374
1375/// Helper routine of SimplifyDemandedUseBits. It tries to simplify
1376/// "E1 = (X lsr C1) << C2", where the C1 and C2 are constant, into
1377/// "E2 = X << (C2 - C1)" or "E2 = X >> (C1 - C2)", depending on the sign
1378/// of "C2-C1".
1379///
1380/// Suppose E1 and E2 are generally different in bits S={bm, bm+1,
1381/// ..., bn}, without considering the specific value X is holding.
1382/// This transformation is legal iff one of following conditions is hold:
1383/// 1) All the bit in S are 0, in this case E1 == E2.
1384/// 2) We don't care those bits in S, per the input DemandedMask.
1385/// 3) Combination of 1) and 2). Some bits in S are 0, and we don't care the
1386/// rest bits.
1387///
1388/// Currently we only test condition 2).
1389///
1390/// As with SimplifyDemandedUseBits, it returns NULL if the simplification was
1391/// not successful.
1393 Instruction *Shr, const APInt &ShrOp1, Instruction *Shl,
1394 const APInt &ShlOp1, const APInt &DemandedMask, KnownBits &Known) {
1395 if (!ShlOp1 || !ShrOp1)
1396 return nullptr; // No-op.
1397
1398 Value *VarX = Shr->getOperand(0);
1399 Type *Ty = VarX->getType();
1400 unsigned BitWidth = Ty->getScalarSizeInBits();
1401 if (ShlOp1.uge(BitWidth) || ShrOp1.uge(BitWidth))
1402 return nullptr; // Undef.
1403
1404 unsigned ShlAmt = ShlOp1.getZExtValue();
1405 unsigned ShrAmt = ShrOp1.getZExtValue();
1406
1407 Known.One.clearAllBits();
1408 Known.Zero.setLowBits(ShlAmt - 1);
1409 Known.Zero &= DemandedMask;
1410
1411 APInt BitMask1(APInt::getAllOnes(BitWidth));
1412 APInt BitMask2(APInt::getAllOnes(BitWidth));
1413
1414 bool isLshr = (Shr->getOpcode() == Instruction::LShr);
1415 BitMask1 = isLshr ? (BitMask1.lshr(ShrAmt) << ShlAmt) :
1416 (BitMask1.ashr(ShrAmt) << ShlAmt);
1417
1418 if (ShrAmt <= ShlAmt) {
1419 BitMask2 <<= (ShlAmt - ShrAmt);
1420 } else {
1421 BitMask2 = isLshr ? BitMask2.lshr(ShrAmt - ShlAmt):
1422 BitMask2.ashr(ShrAmt - ShlAmt);
1423 }
1424
1425 // Check if condition-2 (see the comment to this function) is satified.
1426 if ((BitMask1 & DemandedMask) == (BitMask2 & DemandedMask)) {
1427 if (ShrAmt == ShlAmt)
1428 return VarX;
1429
1430 if (!Shr->hasOneUse())
1431 return nullptr;
1432
1433 BinaryOperator *New;
1434 if (ShrAmt < ShlAmt) {
1435 Constant *Amt = ConstantInt::get(VarX->getType(), ShlAmt - ShrAmt);
1436 New = BinaryOperator::CreateShl(VarX, Amt);
1438 New->setHasNoSignedWrap(Orig->hasNoSignedWrap());
1439 New->setHasNoUnsignedWrap(Orig->hasNoUnsignedWrap());
1440 } else {
1441 Constant *Amt = ConstantInt::get(VarX->getType(), ShrAmt - ShlAmt);
1442 New = isLshr ? BinaryOperator::CreateLShr(VarX, Amt) :
1443 BinaryOperator::CreateAShr(VarX, Amt);
1444 if (cast<BinaryOperator>(Shr)->isExact())
1445 New->setIsExact(true);
1446 }
1447
1448 return InsertNewInstWith(New, Shl->getIterator());
1449 }
1450
1451 return nullptr;
1452}
1453
1454/// Return true if the top-level all-lanes demanded-elements query can be
1455/// skipped for an intermediate insertelement chain node. This is limited to a
1456/// bounded one-use chain with distinct in-range constant indices, where SDVE
1457/// cannot remove a dead insert before hitting its depth limit.
1459 unsigned VWidth,
1460 unsigned DepthLimit) {
1461 // Only skip chain nodes that feed another insertelement; the final chain root
1462 // still runs the full query.
1463 if (!IE.hasOneUse())
1464 return false;
1465 auto *UserIE = dyn_cast<InsertElementInst>(IE.user_back());
1466 if (!UserIE || UserIE->getOperand(0) != &IE)
1467 return false;
1468
1469 SmallBitVector SeenIndices(VWidth);
1470 auto HasNewIndexInRange = [&](InsertElementInst &Insert) {
1471 auto *Idx = dyn_cast<ConstantInt>(Insert.getOperand(2));
1472 // Let the normal SDVE path handle variable or out-of-range indices. The
1473 // latter may simplify the chain and must not be passed to getZExtValue().
1474 if (!Idx || Idx->getValue().uge(VWidth))
1475 return false;
1476
1477 unsigned Index = Idx->getZExtValue();
1478 if (SeenIndices.test(Index))
1479 return false;
1480
1481 SeenIndices.set(Index);
1482 return true;
1483 };
1484
1485 auto *Cur = &IE;
1486 for (unsigned I = 0; I != DepthLimit; ++I) {
1487 // This loop scans the same base-chain window that the SDVE query would
1488 // inspect before hitting its depth limit. With distinct insert indices in
1489 // that window, the all-lanes query cannot remove a dead insert; with
1490 // VWidth > DepthLimit, it also cannot narrow demand to a single lane.
1491 if (!HasNewIndexInRange(*Cur))
1492 return false;
1493
1494 Value *Base = Cur->getOperand(0);
1495 if (match(Base, m_Poison()))
1496 return true;
1497
1499 if (!Cur || !Cur->hasOneUse())
1500 return false;
1501 }
1502
1503 return true;
1504}
1505
1506/// The specified value produces a vector with any number of elements.
1507/// This method analyzes which elements of the operand are poison and
1508/// returns that information in PoisonElts.
1509///
1510/// DemandedElts contains the set of elements that are actually used by the
1511/// caller, and by default (AllowMultipleUsers equals false) the value is
1512/// simplified only if it has a single caller. If AllowMultipleUsers is set
1513/// to true, DemandedElts refers to the union of sets of elements that are
1514/// used by all callers.
1515///
1516/// If the information about demanded elements can be used to simplify the
1517/// operation, the operation is simplified, then the resultant value is
1518/// returned. This returns null if no change was made.
1520 APInt DemandedElts,
1521 APInt &PoisonElts,
1522 unsigned Depth,
1523 bool AllowMultipleUsers) {
1524 // Cannot analyze scalable type. The number of vector elements is not a
1525 // compile-time constant.
1526 if (isa<ScalableVectorType>(V->getType()))
1527 return nullptr;
1528
1529 unsigned VWidth = cast<FixedVectorType>(V->getType())->getNumElements();
1530 APInt EltMask(APInt::getAllOnes(VWidth));
1531 assert((DemandedElts & ~EltMask) == 0 && "Invalid DemandedElts!");
1532
1533 if (match(V, m_Poison())) {
1534 // If the entire vector is poison, just return this info.
1535 PoisonElts = EltMask;
1536 return nullptr;
1537 }
1538
1539 if (DemandedElts.isZero()) { // If nothing is demanded, provide poison.
1540 PoisonElts = EltMask;
1541 return PoisonValue::get(V->getType());
1542 }
1543
1544 PoisonElts = 0;
1545
1546 if (auto *C = dyn_cast<Constant>(V)) {
1547 // Check if this is identity. If so, return 0 since we are not simplifying
1548 // anything.
1549 if (DemandedElts.isAllOnes())
1550 return nullptr;
1551
1552 Type *EltTy = cast<VectorType>(V->getType())->getElementType();
1555 for (unsigned i = 0; i != VWidth; ++i) {
1556 if (!DemandedElts[i]) { // If not demanded, set to poison.
1557 Elts.push_back(Poison);
1558 PoisonElts.setBit(i);
1559 continue;
1560 }
1561
1562 Constant *Elt = C->getAggregateElement(i);
1563 if (!Elt) return nullptr;
1564
1565 Elts.push_back(Elt);
1566 if (isa<PoisonValue>(Elt)) // Already poison.
1567 PoisonElts.setBit(i);
1568 }
1569
1570 // If we changed the constant, return it.
1571 Constant *NewCV = ConstantVector::get(Elts);
1572 return NewCV != C ? NewCV : nullptr;
1573 }
1574
1575 // Limit search depth.
1576 if (Depth == CLOpts.simplify_vector_elts_depth)
1577 return nullptr;
1578
1579 if (!AllowMultipleUsers) {
1580 // If multiple users are using the root value, proceed with
1581 // simplification conservatively assuming that all elements
1582 // are needed.
1583 if (!V->hasOneUse()) {
1584 // Quit if we find multiple users of a non-root value though.
1585 // They'll be handled when it's their turn to be visited by
1586 // the main instcombine process.
1587 if (Depth != 0)
1588 // TODO: Just compute the PoisonElts information recursively.
1589 return nullptr;
1590
1591 // Conservatively assume that all elements are needed.
1592 DemandedElts = EltMask;
1593 }
1594 }
1595
1597 if (!I) return nullptr; // Only analyze instructions.
1598
1599 bool MadeChange = false;
1600 auto simplifyAndSetOp = [&](Instruction *Inst, unsigned OpNum,
1601 APInt Demanded, APInt &Undef) {
1602 auto *II = dyn_cast<IntrinsicInst>(Inst);
1603 Value *Op = II ? II->getArgOperand(OpNum) : Inst->getOperand(OpNum);
1604 if (Value *V = SimplifyDemandedVectorElts(Op, Demanded, Undef, Depth + 1)) {
1605 replaceOperand(*Inst, OpNum, V);
1606 MadeChange = true;
1607 }
1608 };
1609
1610 APInt PoisonElts2(VWidth, 0);
1611 APInt PoisonElts3(VWidth, 0);
1612 switch (I->getOpcode()) {
1613 default: break;
1614
1615 case Instruction::GetElementPtr: {
1616 // The LangRef requires that struct geps have all constant indices. As
1617 // such, we can't convert any operand to partial undef.
1618 auto mayIndexStructType = [](GetElementPtrInst &GEP) {
1619 for (auto I = gep_type_begin(GEP), E = gep_type_end(GEP);
1620 I != E; I++)
1621 if (I.isStruct())
1622 return true;
1623 return false;
1624 };
1625 if (mayIndexStructType(cast<GetElementPtrInst>(*I)))
1626 break;
1627
1628 // Conservatively track the demanded elements back through any vector
1629 // operands we may have. We know there must be at least one, or we
1630 // wouldn't have a vector result to get here. Note that we intentionally
1631 // merge the undef bits here since gepping with either an poison base or
1632 // index results in poison.
1633 for (unsigned i = 0; i < I->getNumOperands(); i++) {
1634 if (i == 0 ? match(I->getOperand(i), m_Undef())
1635 : match(I->getOperand(i), m_Poison())) {
1636 // If the entire vector is undefined, just return this info.
1637 PoisonElts = EltMask;
1638 return nullptr;
1639 }
1640 if (I->getOperand(i)->getType()->isVectorTy()) {
1641 APInt PoisonEltsOp(VWidth, 0);
1642 simplifyAndSetOp(I, i, DemandedElts, PoisonEltsOp);
1643 // gep(x, undef) is not undef, so skip considering idx ops here
1644 // Note that we could propagate poison, but we can't distinguish between
1645 // undef & poison bits ATM
1646 if (i == 0)
1647 PoisonElts |= PoisonEltsOp;
1648 }
1649 }
1650
1651 break;
1652 }
1653 case Instruction::InsertElement: {
1654 unsigned DepthLimit = CLOpts.simplify_vector_elts_depth;
1655 auto *IE = cast<InsertElementInst>(I);
1656 // Skip only when SDVE cannot simplify this insert chain before the limit.
1657 if (Depth == 0 && DemandedElts.isAllOnes() && VWidth > DepthLimit &&
1658 canSkipDemandedEltsInInsertChain(*IE, VWidth, DepthLimit))
1659 return nullptr;
1660
1661 // If this is a variable index, we don't know which element it overwrites.
1662 // demand exactly the same input as we produce.
1663 ConstantInt *Idx = dyn_cast<ConstantInt>(I->getOperand(2));
1664 if (!Idx) {
1665 // Note that we can't propagate undef elt info, because we don't know
1666 // which elt is getting updated.
1667 simplifyAndSetOp(I, 0, DemandedElts, PoisonElts2);
1668 break;
1669 }
1670
1671 // The element inserted overwrites whatever was there, so the input demanded
1672 // set is simpler than the output set.
1673 unsigned IdxNo = Idx->getZExtValue();
1674 APInt PreInsertDemandedElts = DemandedElts;
1675 if (IdxNo < VWidth)
1676 PreInsertDemandedElts.clearBit(IdxNo);
1677
1678 // If we only demand the element that is being inserted and that element
1679 // was extracted from the same index in another vector with the same type,
1680 // replace this insert with that other vector.
1681 // Note: This is attempted before the call to simplifyAndSetOp because that
1682 // may change PoisonElts to a value that does not match with Vec.
1683 Value *Vec;
1684 if (PreInsertDemandedElts == 0 &&
1685 match(I->getOperand(1),
1686 m_ExtractElt(m_Value(Vec), m_SpecificInt(IdxNo))) &&
1687 Vec->getType() == I->getType()) {
1688 return Vec;
1689 }
1690
1691 simplifyAndSetOp(I, 0, PreInsertDemandedElts, PoisonElts);
1692
1693 // If this is inserting an element that isn't demanded, remove this
1694 // insertelement.
1695 if (IdxNo >= VWidth || !DemandedElts[IdxNo]) {
1696 Worklist.push(I);
1697 return I->getOperand(0);
1698 }
1699
1700 // The inserted element is defined.
1701 PoisonElts.clearBit(IdxNo);
1702 break;
1703 }
1704 case Instruction::ShuffleVector: {
1705 auto *Shuffle = cast<ShuffleVectorInst>(I);
1706 assert(Shuffle->getOperand(0)->getType() ==
1707 Shuffle->getOperand(1)->getType() &&
1708 "Expected shuffle operands to have same type");
1709 unsigned OpWidth = cast<FixedVectorType>(Shuffle->getOperand(0)->getType())
1710 ->getNumElements();
1711 // Handle trivial case of a splat. Only check the first element of LHS
1712 // operand.
1713 if (all_of(Shuffle->getShuffleMask(), equal_to(0)) &&
1714 DemandedElts.isAllOnes()) {
1715 if (!isa<PoisonValue>(I->getOperand(1))) {
1716 I->setOperand(1, PoisonValue::get(I->getOperand(1)->getType()));
1717 MadeChange = true;
1718 }
1719 APInt LeftDemanded(OpWidth, 1);
1720 APInt LHSPoisonElts(OpWidth, 0);
1721 simplifyAndSetOp(I, 0, LeftDemanded, LHSPoisonElts);
1722 if (LHSPoisonElts[0])
1723 PoisonElts = EltMask;
1724 else
1725 PoisonElts.clearAllBits();
1726 break;
1727 }
1728
1729 APInt LeftDemanded(OpWidth, 0), RightDemanded(OpWidth, 0);
1730 for (unsigned i = 0; i < VWidth; i++) {
1731 if (DemandedElts[i]) {
1732 unsigned MaskVal = Shuffle->getMaskValue(i);
1733 if (MaskVal != -1u) {
1734 assert(MaskVal < OpWidth * 2 &&
1735 "shufflevector mask index out of range!");
1736 if (MaskVal < OpWidth)
1737 LeftDemanded.setBit(MaskVal);
1738 else
1739 RightDemanded.setBit(MaskVal - OpWidth);
1740 }
1741 }
1742 }
1743
1744 APInt LHSPoisonElts(OpWidth, 0);
1745 simplifyAndSetOp(I, 0, LeftDemanded, LHSPoisonElts);
1746
1747 APInt RHSPoisonElts(OpWidth, 0);
1748 simplifyAndSetOp(I, 1, RightDemanded, RHSPoisonElts);
1749
1750 // If this shuffle does not change the vector length and the elements
1751 // demanded by this shuffle are an identity mask, then this shuffle is
1752 // unnecessary.
1753 //
1754 // We are assuming canonical form for the mask, so the source vector is
1755 // operand 0 and operand 1 is not used.
1756 //
1757 // Note that if an element is demanded and this shuffle mask is undefined
1758 // for that element, then the shuffle is not considered an identity
1759 // operation. The shuffle prevents poison from the operand vector from
1760 // leaking to the result by replacing poison with an undefined value.
1761 if (VWidth == OpWidth) {
1762 bool IsIdentityShuffle = true;
1763 for (unsigned i = 0; i < VWidth; i++) {
1764 unsigned MaskVal = Shuffle->getMaskValue(i);
1765 if (DemandedElts[i] && i != MaskVal) {
1766 IsIdentityShuffle = false;
1767 break;
1768 }
1769 }
1770 if (IsIdentityShuffle)
1771 return Shuffle->getOperand(0);
1772 }
1773
1774 bool NewPoisonElts = false;
1775 unsigned LHSIdx = -1u, LHSValIdx = -1u;
1776 unsigned RHSIdx = -1u, RHSValIdx = -1u;
1777 bool LHSUniform = true;
1778 bool RHSUniform = true;
1779 for (unsigned i = 0; i < VWidth; i++) {
1780 unsigned MaskVal = Shuffle->getMaskValue(i);
1781 if (MaskVal == -1u) {
1782 PoisonElts.setBit(i);
1783 } else if (!DemandedElts[i]) {
1784 NewPoisonElts = true;
1785 PoisonElts.setBit(i);
1786 } else if (MaskVal < OpWidth) {
1787 if (LHSPoisonElts[MaskVal]) {
1788 NewPoisonElts = true;
1789 PoisonElts.setBit(i);
1790 } else {
1791 LHSIdx = LHSIdx == -1u ? i : OpWidth;
1792 LHSValIdx = LHSValIdx == -1u ? MaskVal : OpWidth;
1793 LHSUniform = LHSUniform && (MaskVal == i);
1794 }
1795 } else {
1796 if (RHSPoisonElts[MaskVal - OpWidth]) {
1797 NewPoisonElts = true;
1798 PoisonElts.setBit(i);
1799 } else {
1800 RHSIdx = RHSIdx == -1u ? i : OpWidth;
1801 RHSValIdx = RHSValIdx == -1u ? MaskVal - OpWidth : OpWidth;
1802 RHSUniform = RHSUniform && (MaskVal - OpWidth == i);
1803 }
1804 }
1805 }
1806
1807 // Try to transform shuffle with constant vector and single element from
1808 // this constant vector to single insertelement instruction.
1809 // shufflevector V, C, <v1, v2, .., ci, .., vm> ->
1810 // insertelement V, C[ci], ci-n
1811 if (OpWidth ==
1812 cast<FixedVectorType>(Shuffle->getType())->getNumElements()) {
1813 Value *Op = nullptr;
1814 Constant *Value = nullptr;
1815 unsigned Idx = -1u;
1816
1817 // Find constant vector with the single element in shuffle (LHS or RHS).
1818 if (LHSIdx < OpWidth && RHSUniform) {
1819 if (auto *CV = dyn_cast<ConstantVector>(Shuffle->getOperand(0))) {
1820 Op = Shuffle->getOperand(1);
1821 Value = CV->getOperand(LHSValIdx);
1822 Idx = LHSIdx;
1823 }
1824 }
1825 if (RHSIdx < OpWidth && LHSUniform) {
1826 if (auto *CV = dyn_cast<ConstantVector>(Shuffle->getOperand(1))) {
1827 Op = Shuffle->getOperand(0);
1828 Value = CV->getOperand(RHSValIdx);
1829 Idx = RHSIdx;
1830 }
1831 }
1832 // Found constant vector with single element - convert to insertelement.
1833 if (Op && Value) {
1835 Op, Value, ConstantInt::get(Type::getInt64Ty(I->getContext()), Idx),
1836 Shuffle->getName());
1837 InsertNewInstWith(New, Shuffle->getIterator());
1838 return New;
1839 }
1840 }
1841 if (NewPoisonElts) {
1842 // Add additional discovered undefs.
1844 for (unsigned i = 0; i < VWidth; ++i) {
1845 if (PoisonElts[i])
1847 else
1848 Elts.push_back(Shuffle->getMaskValue(i));
1849 }
1850 Shuffle->setShuffleMask(Elts);
1851 MadeChange = true;
1852 }
1853 break;
1854 }
1855 case Instruction::Select: {
1856 // If this is a vector select, try to transform the select condition based
1857 // on the current demanded elements.
1859 if (Sel->getCondition()->getType()->isVectorTy()) {
1860 // TODO: We are not doing anything with PoisonElts based on this call.
1861 // It is overwritten below based on the other select operands. If an
1862 // element of the select condition is known undef, then we are free to
1863 // choose the output value from either arm of the select. If we know that
1864 // one of those values is undef, then the output can be undef.
1865 simplifyAndSetOp(I, 0, DemandedElts, PoisonElts);
1866 }
1867
1868 // Next, see if we can transform the arms of the select.
1869 APInt DemandedLHS(DemandedElts), DemandedRHS(DemandedElts);
1870 if (auto *CV = dyn_cast<ConstantVector>(Sel->getCondition())) {
1871 for (unsigned i = 0; i < VWidth; i++) {
1872 Constant *CElt = CV->getAggregateElement(i);
1873
1874 // isNullValue() always returns false when called on a ConstantExpr.
1875 if (CElt->isNullValue())
1876 DemandedLHS.clearBit(i);
1877 else if (CElt->isOneValue())
1878 DemandedRHS.clearBit(i);
1879 }
1880 }
1881
1882 simplifyAndSetOp(I, 1, DemandedLHS, PoisonElts2);
1883 simplifyAndSetOp(I, 2, DemandedRHS, PoisonElts3);
1884
1885 // Output elements are undefined if the element from each arm is undefined.
1886 // TODO: This can be improved. See comment in select condition handling.
1887 PoisonElts = PoisonElts2 & PoisonElts3;
1888 break;
1889 }
1890 case Instruction::BitCast: {
1891 // Vector->vector casts only.
1892 VectorType *VTy = dyn_cast<VectorType>(I->getOperand(0)->getType());
1893 if (!VTy) break;
1894 unsigned InVWidth = cast<FixedVectorType>(VTy)->getNumElements();
1895 APInt InputDemandedElts(InVWidth, 0);
1896 PoisonElts2 = APInt(InVWidth, 0);
1897 unsigned Ratio;
1898
1899 if (VWidth == InVWidth) {
1900 // If we are converting from <4 x i32> -> <4 x f32>, we demand the same
1901 // elements as are demanded of us.
1902 Ratio = 1;
1903 InputDemandedElts = DemandedElts;
1904 } else if ((VWidth % InVWidth) == 0) {
1905 // If the number of elements in the output is a multiple of the number of
1906 // elements in the input then an input element is live if any of the
1907 // corresponding output elements are live.
1908 Ratio = VWidth / InVWidth;
1909 for (unsigned OutIdx = 0; OutIdx != VWidth; ++OutIdx)
1910 if (DemandedElts[OutIdx])
1911 InputDemandedElts.setBit(OutIdx / Ratio);
1912 } else if ((InVWidth % VWidth) == 0) {
1913 // If the number of elements in the input is a multiple of the number of
1914 // elements in the output then an input element is live if the
1915 // corresponding output element is live.
1916 Ratio = InVWidth / VWidth;
1917 for (unsigned InIdx = 0; InIdx != InVWidth; ++InIdx)
1918 if (DemandedElts[InIdx / Ratio])
1919 InputDemandedElts.setBit(InIdx);
1920 } else {
1921 // Unsupported so far.
1922 break;
1923 }
1924
1925 simplifyAndSetOp(I, 0, InputDemandedElts, PoisonElts2);
1926
1927 if (VWidth == InVWidth) {
1928 PoisonElts = PoisonElts2;
1929 } else if ((VWidth % InVWidth) == 0) {
1930 // If the number of elements in the output is a multiple of the number of
1931 // elements in the input then an output element is undef if the
1932 // corresponding input element is undef.
1933 for (unsigned OutIdx = 0; OutIdx != VWidth; ++OutIdx)
1934 if (PoisonElts2[OutIdx / Ratio])
1935 PoisonElts.setBit(OutIdx);
1936 } else if ((InVWidth % VWidth) == 0) {
1937 // If the number of elements in the input is a multiple of the number of
1938 // elements in the output then an output element is undef if all of the
1939 // corresponding input elements are undef.
1940 for (unsigned OutIdx = 0; OutIdx != VWidth; ++OutIdx) {
1941 APInt SubUndef = PoisonElts2.lshr(OutIdx * Ratio).zextOrTrunc(Ratio);
1942 if (SubUndef.popcount() == Ratio)
1943 PoisonElts.setBit(OutIdx);
1944 }
1945 } else {
1946 llvm_unreachable("Unimp");
1947 }
1948 break;
1949 }
1950 case Instruction::FPTrunc:
1951 case Instruction::FPExt:
1952 simplifyAndSetOp(I, 0, DemandedElts, PoisonElts);
1953 break;
1954
1955 case Instruction::Call: {
1957 if (!II) break;
1958 switch (II->getIntrinsicID()) {
1959 case Intrinsic::masked_gather: // fallthrough
1960 case Intrinsic::masked_load: {
1961 // Subtlety: If we load from a pointer, the pointer must be valid
1962 // regardless of whether the element is demanded. Doing otherwise risks
1963 // segfaults which didn't exist in the original program.
1964 APInt DemandedPtrs(APInt::getAllOnes(VWidth)),
1965 DemandedPassThrough(DemandedElts);
1966 if (auto *CMask = dyn_cast<Constant>(II->getOperand(1))) {
1967 for (unsigned i = 0; i < VWidth; i++) {
1968 if (Constant *CElt = CMask->getAggregateElement(i)) {
1969 if (CElt->isNullValue())
1970 DemandedPtrs.clearBit(i);
1971 else if (CElt->isAllOnesValue())
1972 DemandedPassThrough.clearBit(i);
1973 }
1974 }
1975 }
1976
1977 if (II->getIntrinsicID() == Intrinsic::masked_gather)
1978 simplifyAndSetOp(II, 0, DemandedPtrs, PoisonElts2);
1979 simplifyAndSetOp(II, 2, DemandedPassThrough, PoisonElts3);
1980
1981 // Output elements are undefined if the element from both sources are.
1982 // TODO: can strengthen via mask as well.
1983 PoisonElts = PoisonElts2 & PoisonElts3;
1984 break;
1985 }
1986 case Intrinsic::smulh:
1987 case Intrinsic::umulh:
1988 simplifyAndSetOp(II, 0, DemandedElts, PoisonElts);
1989 simplifyAndSetOp(II, 1, DemandedElts, PoisonElts);
1990 PoisonElts = PoisonElts2 | PoisonElts3;
1991 break;
1992 default: {
1993 // Handle target specific intrinsics
1994 std::optional<Value *> V = targetSimplifyDemandedVectorEltsIntrinsic(
1995 *II, DemandedElts, PoisonElts, PoisonElts2, PoisonElts3,
1996 simplifyAndSetOp);
1997 if (V)
1998 return *V;
1999 break;
2000 }
2001 } // switch on IntrinsicID
2002 break;
2003 } // case Call
2004 } // switch on Opcode
2005
2006 // TODO: We bail completely on integer div/rem and shifts because they have
2007 // UB/poison potential, but that should be refined.
2008 BinaryOperator *BO;
2009 if (match(I, m_BinOp(BO)) && !BO->isIntDivRem() && !BO->isShift()) {
2010 Value *X = BO->getOperand(0);
2011 Value *Y = BO->getOperand(1);
2012
2013 // Look for an equivalent binop except that one operand has been shuffled.
2014 // If the demand for this binop only includes elements that are the same as
2015 // the other binop, then we may be able to replace this binop with a use of
2016 // the earlier one.
2017 //
2018 // Example:
2019 // %other_bo = bo (shuf X, {0}), Y
2020 // %this_extracted_bo = extelt (bo X, Y), 0
2021 // -->
2022 // %other_bo = bo (shuf X, {0}), Y
2023 // %this_extracted_bo = extelt %other_bo, 0
2024 //
2025 // TODO: Handle demand of an arbitrary single element or more than one
2026 // element instead of just element 0.
2027 // TODO: Unlike general demanded elements transforms, this should be safe
2028 // for any (div/rem/shift) opcode too.
2029 if (DemandedElts == 1 && !X->hasOneUse() && !Y->hasOneUse() &&
2030 BO->hasOneUse() ) {
2031
2032 auto findShufBO = [&](bool MatchShufAsOp0) -> User * {
2033 // Try to use shuffle-of-operand in place of an operand:
2034 // bo X, Y --> bo (shuf X), Y
2035 // bo X, Y --> bo X, (shuf Y)
2036
2037 Value *OtherOp = MatchShufAsOp0 ? Y : X;
2038 if (!OtherOp->hasUseList())
2039 return nullptr;
2040
2041 BinaryOperator::BinaryOps Opcode = BO->getOpcode();
2042 Value *ShufOp = MatchShufAsOp0 ? X : Y;
2043
2044 for (User *U : OtherOp->users()) {
2045 ArrayRef<int> Mask;
2046 auto Shuf = m_Shuffle(m_Specific(ShufOp), m_Value(), m_Mask(Mask));
2047 if (BO->isCommutative()
2048 ? match(U, m_c_BinOp(Opcode, Shuf, m_Specific(OtherOp)))
2049 : MatchShufAsOp0
2050 ? match(U, m_BinOp(Opcode, Shuf, m_Specific(OtherOp)))
2051 : match(U, m_BinOp(Opcode, m_Specific(OtherOp), Shuf)))
2052 if (match(Mask, m_ZeroMask()) && Mask[0] != PoisonMaskElem)
2053 if (DT.dominates(U, I))
2054 return U;
2055 }
2056 return nullptr;
2057 };
2058
2059 User *ShufBO = findShufBO(/* MatchShufAsOp0 */ true);
2060 if (!ShufBO)
2061 ShufBO = findShufBO(/* MatchShufAsOp0 */ false);
2062 if (ShufBO) {
2063 auto *ShufBOI = cast<Instruction>(ShufBO);
2064 ShufBOI->andIRFlags(BO);
2065 Worklist.add(ShufBOI);
2066 return ShufBO;
2067 }
2068 }
2069
2070 simplifyAndSetOp(I, 0, DemandedElts, PoisonElts);
2071 simplifyAndSetOp(I, 1, DemandedElts, PoisonElts2);
2072
2073 // Output elements are undefined if both are undefined. Consider things
2074 // like undef & 0. The result is known zero, not undef.
2075 PoisonElts &= PoisonElts2;
2076 }
2077
2078 // If we've proven all of the lanes poison, return a poison value.
2079 // TODO: Intersect w/demanded lanes
2080 if (PoisonElts.isAllOnes())
2081 return PoisonValue::get(I->getType());
2082
2083 return MadeChange ? I : nullptr;
2084}
2085
2086/// For floating-point classes that resolve to a single bit pattern, return that
2087/// value.
2089 bool IsCanonicalizing = false) {
2090 if (Mask == fcNone)
2091 return PoisonValue::get(Ty);
2092
2093 if (Mask == fcPosZero)
2094 return Constant::getNullValue(Ty);
2095
2096 // TODO: Support aggregate types that are allowed by FPMathOperator.
2097 if (Ty->isAggregateType())
2098 return nullptr;
2099
2100 // Turn any possible snans into quiet if we can.
2101 if (Mask == fcNan && IsCanonicalizing)
2102 return ConstantFP::getQNaN(Ty);
2103
2104 switch (Mask) {
2105 case fcNegZero:
2106 return ConstantFP::getZero(Ty, true);
2107 case fcPosInf:
2108 return ConstantFP::getInfinity(Ty);
2109 case fcNegInf:
2110 return ConstantFP::getInfinity(Ty, true);
2111 case fcQNan:
2112 // Payload bits cannot be dropped for pure signbit operations.
2113 return IsCanonicalizing ? ConstantFP::getQNaN(Ty) : nullptr;
2114 default:
2115 return nullptr;
2116 }
2117}
2118
2119/// Perform multiple-use aware simplfications for fabs(\p Src). Returns a
2120/// replacement value if it's simplified, otherwise nullptr. Updates \p Known
2121/// with the known fpclass if not simplified.
2123 FPClassTest DemandedMask,
2124 KnownFPClass KnownSrc, bool NSZ) {
2125 if ((DemandedMask & fcNan) == fcNone)
2126 KnownSrc.knownNot(fcNan);
2127 if ((DemandedMask & fcInf) == fcNone)
2128 KnownSrc.knownNot(fcInf);
2129
2130 if (KnownSrc.getSignBit() == false ||
2131 ((DemandedMask & fcNan) == fcNone && KnownSrc.isKnownNever(fcNegative)))
2132 return Src;
2133
2134 // If the only sign bit difference is due to -0, ignore it with nsz
2135 if (NSZ &&
2137 return Src;
2138
2139 Known = KnownFPClass::fabs(KnownSrc);
2140 Known.knownNot(~DemandedMask);
2141 return nullptr;
2142}
2143
2144/// Try to set an inferred no-nans or no-infs in \p FMF. \p ValidResults is a
2145/// mask of known valid results for the operator (already computed from the
2146/// result, and the known operand inputs in \p Known)
2148 FPClassTest ValidResults,
2150 if (!FMF.noNaNs() && (ValidResults & fcNan) == fcNone) {
2151 if (all_of(Known, [](const KnownFPClass KnownSrc) {
2152 return KnownSrc.isKnownNeverNaN();
2153 }))
2154 FMF.setNoNaNs();
2155 }
2156
2157 if (!FMF.noInfs() && (ValidResults & fcInf) == fcNone) {
2158 if (all_of(Known, [](const KnownFPClass KnownSrc) {
2159 return KnownSrc.isKnownNeverInfinity();
2160 }))
2161 FMF.setNoInfs();
2162 }
2163
2164 return FMF;
2165}
2166
2168 FastMathFlags FMF) {
2169 if (FMF.noNaNs())
2170 DemandedMask &= ~fcNan;
2171
2172 if (FMF.noInfs())
2173 DemandedMask &= ~fcInf;
2174 return DemandedMask;
2175}
2176
2177/// Apply epilog fixups to a floating-point intrinsic. See if the result can
2178/// fold to a constant, or apply fast math flags.
2180 FastMathFlags FMF,
2181 FPClassTest DemandedMask,
2183 ArrayRef<KnownFPClass> KnownSrcs) {
2184 FPClassTest ValidResults = DemandedMask & Known.getKnownFPClasses();
2185 Constant *SingleVal = getFPClassConstant(FPOp->getType(), ValidResults,
2186 /*IsCanonicalizing=*/true);
2187 if (SingleVal)
2188 return SingleVal;
2189
2190 FastMathFlags InferredFMF =
2191 inferFastMathValueFlags(FMF, ValidResults, KnownSrcs);
2192 if (InferredFMF != FMF) {
2194 FPOp->setFastMathFlags(InferredFMF);
2195 return FPOp;
2196 }
2197
2198 return nullptr;
2199}
2200
2201/// Perform multiple-use aware simplfications for fneg(fabs(\p Src)). Returns a
2202/// replacement value if it's simplified, otherwise nullptr. Updates \p Known
2203/// with the known fpclass if not simplified.
2205 FPClassTest DemandedMask,
2206 KnownFPClass KnownSrc, bool NSZ) {
2207 if ((DemandedMask & fcNan) == fcNone)
2208 KnownSrc.knownNot(fcNan);
2209 if ((DemandedMask & fcInf) == fcNone)
2210 KnownSrc.knownNot(fcInf);
2211
2212 // If the source value is known negative, we can directly fold to it.
2213 if (KnownSrc.getSignBit() == true)
2214 return Src;
2215
2216 // If the only sign bit difference is for 0, ignore it with nsz.
2217 if (NSZ &&
2219 return Src;
2220
2222 Known.knownNot(~DemandedMask);
2223 return nullptr;
2224}
2225
2227 FPClassTest DemandedMask,
2228 KnownFPClass KnownSrc,
2229 bool NSZ) {
2230 if (NSZ) {
2231 constexpr FPClassTest NegOrZero = fcNegative | fcPosZero;
2232 constexpr FPClassTest PosOrZero = fcPositive | fcNegZero;
2233
2234 if ((DemandedMask & ~NegOrZero) == fcNone &&
2235 KnownSrc.isKnownAlways(NegOrZero))
2236 return MagSrc;
2237
2238 if ((DemandedMask & ~PosOrZero) == fcNone &&
2239 KnownSrc.isKnownAlways(PosOrZero))
2240 return MagSrc;
2241 } else {
2242 if ((DemandedMask & ~fcNegative) == fcNone && KnownSrc.getSignBit() == true)
2243 return MagSrc;
2244
2245 if ((DemandedMask & ~fcPositive) == fcNone &&
2246 KnownSrc.getSignBit() == false)
2247 return MagSrc;
2248 }
2249
2250 return nullptr;
2251}
2252
2253static Value *
2255 const CallInst *CI, FPClassTest DemandedMask,
2256 KnownFPClass KnownLHS, KnownFPClass KnownRHS,
2257 const Function &F, bool NSZ) {
2258 bool OrderedZeroSign = !NSZ;
2259
2261 switch (IID) {
2262 case Intrinsic::maximum: {
2264
2265 // If one operand is known greater than the other, it must be that
2266 // operand unless the other is a nan.
2268 KnownRHS.getKnownFPClasses(),
2269 OrderedZeroSign) &&
2270 KnownRHS.isKnownNever(fcNan))
2271 return CI->getArgOperand(0);
2272
2274 KnownRHS.getKnownFPClasses(),
2275 OrderedZeroSign) &&
2276 KnownLHS.isKnownNever(fcNan))
2277 return CI->getArgOperand(1);
2278
2279 break;
2280 }
2281 case Intrinsic::minimum: {
2283
2284 // If one operand is known less than the other, it must be that operand
2285 // unless the other is a nan.
2287 KnownRHS.getKnownFPClasses(),
2288 OrderedZeroSign) &&
2289 KnownRHS.isKnownNever(fcNan))
2290 return CI->getArgOperand(0);
2291
2293 KnownRHS.getKnownFPClasses(),
2294 OrderedZeroSign) &&
2295 KnownLHS.isKnownNever(fcNan))
2296 return CI->getArgOperand(1);
2297
2298 break;
2299 }
2300 case Intrinsic::maxnum:
2301 case Intrinsic::maximumnum: {
2302 OpKind = IID == Intrinsic::maxnum ? KnownFPClass::MinMaxKind::maxnum
2304
2306 KnownRHS.getKnownFPClasses(),
2307 OrderedZeroSign) &&
2308 KnownLHS.isKnownNever(fcNan))
2309 return CI->getArgOperand(0);
2310
2312 KnownRHS.getKnownFPClasses(),
2313 OrderedZeroSign) &&
2314 KnownRHS.isKnownNever(fcNan))
2315 return CI->getArgOperand(1);
2316
2317 break;
2318 }
2319 case Intrinsic::minnum:
2320 case Intrinsic::minimumnum: {
2321 OpKind = IID == Intrinsic::minnum ? KnownFPClass::MinMaxKind::minnum
2323
2325 KnownRHS.getKnownFPClasses(),
2326 OrderedZeroSign) &&
2327 KnownLHS.isKnownNever(fcNan))
2328 return CI->getArgOperand(0);
2329
2331 KnownRHS.getKnownFPClasses(),
2332 OrderedZeroSign) &&
2333 KnownRHS.isKnownNever(fcNan))
2334 return CI->getArgOperand(1);
2335
2336 break;
2337 }
2338 default:
2339 llvm_unreachable("not a min/max intrinsic");
2340 }
2341
2342 Type *EltTy = CI->getType()->getScalarType();
2343 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2344 Known = KnownFPClass::minMaxLike(KnownLHS, KnownRHS, OpKind, Mode);
2345 Known.knownNot(~DemandedMask);
2346
2347 return getFPClassConstant(CI->getType(), Known.getKnownFPClasses(),
2348 /*IsCanonicalizing=*/true);
2349}
2350
2351static Value *
2353 FastMathFlags FMF, FPClassTest DemandedMask,
2354 KnownFPClass &Known, const SimplifyQuery &SQ,
2355 unsigned Depth) {
2356
2357 FPClassTest SrcDemandedMask = DemandedMask;
2358 if (DemandedMask & fcNan)
2359 SrcDemandedMask |= fcNan;
2360
2361 // Zero results may have been rounded from subnormal or normal sources.
2362 if (DemandedMask & fcNegZero)
2363 SrcDemandedMask |= fcNegSubnormal | fcNegNormal;
2364 if (DemandedMask & fcPosZero)
2365 SrcDemandedMask |= fcPosSubnormal | fcPosNormal;
2366
2367 // Subnormal results may have been normal in the source type
2368 if (DemandedMask & fcNegSubnormal)
2369 SrcDemandedMask |= fcNegNormal;
2370 if (DemandedMask & fcPosSubnormal)
2371 SrcDemandedMask |= fcPosNormal;
2372
2373 if (DemandedMask & fcPosInf)
2374 SrcDemandedMask |= fcPosNormal;
2375 if (DemandedMask & fcNegInf)
2376 SrcDemandedMask |= fcNegNormal;
2377
2378 KnownFPClass KnownSrc;
2379 if (IC.SimplifyDemandedFPClass(&I, 0, SrcDemandedMask, KnownSrc, SQ,
2380 Depth + 1))
2381 return &I;
2382
2383 Known = KnownFPClass::fptrunc(KnownSrc);
2384 Known.knownNot(~DemandedMask);
2385
2386 return simplifyDemandedFPClassResult(&I, FMF, DemandedMask, Known,
2387 {KnownSrc});
2388}
2389
2391 FPClassTest DemandedMask,
2393 const SimplifyQuery &SQ,
2394 unsigned Depth) {
2395 assert(Depth <= MaxAnalysisRecursionDepth && "Limit Search Depth");
2396 assert(Known == KnownFPClass() && "expected uninitialized state");
2397
2398 Type *VTy = I->getType();
2399
2400 FastMathFlags FMF;
2401 if (auto *FPOp = dyn_cast<FPMathOperator>(I)) {
2402 FMF = FPOp->getFastMathFlags();
2403 DemandedMask = adjustDemandedMaskFromFlags(DemandedMask, FMF);
2404 }
2405
2406 switch (I->getOpcode()) {
2407 case Instruction::FNeg: {
2408 // Special case fneg(fabs(x))
2409
2410 Value *FNegSrc = I->getOperand(0);
2411 Value *FNegFAbsSrc;
2412 if (match(FNegSrc, m_OneUse(m_FAbs(m_Value(FNegFAbsSrc))))) {
2413 KnownFPClass KnownSrc;
2415 llvm::unknown_sign(DemandedMask), KnownSrc,
2416 SQ, Depth + 1))
2417 return I;
2418
2419 FastMathFlags FabsFMF = cast<FPMathOperator>(FNegSrc)->getFastMathFlags();
2420 FPClassTest ThisDemandedMask =
2421 adjustDemandedMaskFromFlags(DemandedMask, FabsFMF);
2422
2423 bool IsNSZ = FMF.noSignedZeros() || FabsFMF.noSignedZeros();
2424 if (Value *Simplified = simplifyDemandedFPClassFnegFabs(
2425 Known, FNegFAbsSrc, ThisDemandedMask, KnownSrc, IsNSZ))
2426 return Simplified;
2427
2428 if ((ThisDemandedMask & fcNan) == fcNone)
2429 KnownSrc.knownNot(fcNan);
2430 if ((ThisDemandedMask & fcInf) == fcNone)
2431 KnownSrc.knownNot(fcInf);
2432
2433 // fneg(fabs(x)) => fneg(x)
2434 if (KnownSrc.getSignBit() == false)
2435 return replaceOperand(*I, 0, FNegFAbsSrc);
2436
2437 // fneg(fabs(x)) => fneg(x), ignoring -0 if nsz.
2438 if (IsNSZ &&
2440 return replaceOperand(*I, 0, FNegFAbsSrc);
2441
2442 break;
2443 }
2444
2445 if (SimplifyDemandedFPClass(I, 0, llvm::fneg(DemandedMask), Known, SQ,
2446 Depth + 1))
2447 return I;
2448 Known.fneg();
2449 Known.knownNot(~DemandedMask);
2450 break;
2451 }
2452 case Instruction::FAdd:
2453 case Instruction::FSub: {
2454 KnownFPClass KnownLHS, KnownRHS;
2455
2456 // fadd x, x can be handled more aggressively.
2457 if (I->getOperand(0) == I->getOperand(1) &&
2458 I->getOpcode() == Instruction::FAdd &&
2459 isGuaranteedNotToBeUndef(I->getOperand(0), SQ.AC, SQ.CtxI, SQ.DT,
2460 Depth + 1)) {
2461 Type *EltTy = VTy->getScalarType();
2462 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2463
2464 FPClassTest SrcDemandedMask = DemandedMask;
2465 if (DemandedMask & fcNan)
2466 SrcDemandedMask |= fcNan;
2467
2468 // Doubling a subnormal could have resulted in a normal value.
2469 if (DemandedMask & fcPosNormal)
2470 SrcDemandedMask |= fcPosSubnormal;
2471 if (DemandedMask & fcNegNormal)
2472 SrcDemandedMask |= fcNegSubnormal;
2473
2474 // Doubling a subnormal may produce 0 if FTZ/DAZ.
2475 if (Mode != DenormalMode::getIEEE()) {
2476 if (DemandedMask & fcPosZero) {
2477 SrcDemandedMask |= fcPosSubnormal;
2478
2479 if (Mode.inputsMayBePositiveZero() || Mode.outputsMayBePositiveZero())
2480 SrcDemandedMask |= fcNegSubnormal;
2481 }
2482
2483 if (DemandedMask & fcNegZero)
2484 SrcDemandedMask |= fcNegSubnormal;
2485 }
2486
2487 // Doubling a normal could have resulted in an infinity.
2488 if (DemandedMask & fcPosInf)
2489 SrcDemandedMask |= fcPosNormal;
2490 if (DemandedMask & fcNegInf)
2491 SrcDemandedMask |= fcNegNormal;
2492
2493 if (SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownLHS, SQ,
2494 Depth + 1))
2495 return I;
2496
2497 Known = KnownFPClass::fadd_self(KnownLHS, Mode);
2498 KnownRHS = KnownLHS;
2499 } else {
2500 FPClassTest SrcDemandedMask = fcFinite;
2501
2502 // inf + (-inf) = nan
2503 if (DemandedMask & fcNan)
2504 SrcDemandedMask |= fcNan | fcInf;
2505
2506 if (DemandedMask & fcInf)
2507 SrcDemandedMask |= fcInf;
2508
2509 if (SimplifyDemandedFPClass(I, 1, SrcDemandedMask, KnownRHS, SQ,
2510 Depth + 1) ||
2511 SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownLHS, SQ,
2512 Depth + 1))
2513 return I;
2514
2515 Type *EltTy = VTy->getScalarType();
2516 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2517
2518 Known = I->getOpcode() == Instruction::FAdd
2519 ? KnownFPClass::fadd(KnownLHS, KnownRHS, Mode)
2520 : KnownFPClass::fsub(KnownLHS, KnownRHS, Mode);
2521 }
2522
2523 Known.knownNot(~DemandedMask);
2524
2525 if (Constant *SingleVal = getFPClassConstant(VTy, Known.getKnownFPClasses(),
2526 /*IsCanonicalizing=*/true))
2527 return SingleVal;
2528
2529 // Propagate known result to simplify edge case checks.
2530 bool ResultNotNan = (DemandedMask & fcNan) == fcNone;
2531
2532 // With nnan: X + {+/-}Inf --> {+/-}Inf
2533 if (ResultNotNan && I->getOpcode() == Instruction::FAdd &&
2534 KnownRHS.isKnownAlways(fcInf | fcNan) && KnownLHS.isKnownNever(fcNan))
2535 return I->getOperand(1);
2536
2537 // With nnan: {+/-}Inf + X --> {+/-}Inf
2538 // With nnan: {+/-}Inf - X --> {+/-}Inf
2539 if (ResultNotNan && KnownLHS.isKnownAlways(fcInf | fcNan) &&
2540 KnownRHS.isKnownNever(fcNan))
2541 return I->getOperand(0);
2542
2544 FMF, Known.getKnownFPClasses(), {KnownLHS, KnownRHS});
2545 if (InferredFMF != FMF) {
2546 I->setFastMathFlags(InferredFMF);
2547 return I;
2548 }
2549
2550 return nullptr;
2551 }
2552 case Instruction::FMul: {
2553 KnownFPClass KnownLHS, KnownRHS;
2554
2555 Value *X = I->getOperand(0);
2556 Value *Y = I->getOperand(1);
2557
2558 FPClassTest SrcDemandedMask =
2559 DemandedMask & (fcNan | fcZero | fcSubnormal | fcNormal);
2560
2561 if (DemandedMask & fcInf) {
2562 // mul x, inf = inf
2563 // mul large_x, large_y = inf
2564 SrcDemandedMask |= fcSubnormal | fcNormal | fcInf;
2565 }
2566
2567 if (DemandedMask & fcNan) {
2568 // mul +/-inf, 0 => nan
2569 SrcDemandedMask |= fcZero | fcInf | fcNan;
2570
2571 // TODO: Mode check
2572 // mul +/-inf, sub => nan if daz
2573 SrcDemandedMask |= fcSubnormal;
2574 }
2575
2576 // mul normal, subnormal = normal
2577 // Normal inputs may result in underflow.
2578 if (DemandedMask & (fcNormal | fcSubnormal))
2579 SrcDemandedMask |= fcNormal | fcSubnormal;
2580
2581 if (DemandedMask & fcZero)
2582 SrcDemandedMask |= fcNormal | fcSubnormal;
2583
2584 if (X == Y &&
2585 isGuaranteedNotToBeUndef(X, SQ.AC, SQ.CtxI, SQ.DT, Depth + 1)) {
2586 if (SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownLHS, SQ,
2587 Depth + 1))
2588 return I;
2589 Type *EltTy = VTy->getScalarType();
2590
2591 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2592 Known = KnownFPClass::square(KnownLHS, Mode);
2593 Known.knownNot(~DemandedMask);
2594
2595 if (Constant *Folded = getFPClassConstant(VTy, Known.getKnownFPClasses(),
2596 /*IsCanonicalizing=*/true))
2597 return Folded;
2598
2599 if (Known.isKnownAlways(fcPosZero | fcPosInf | fcNan) &&
2600 KnownLHS.isKnownNever(fcSubnormal | fcNormal)) {
2601 // We can skip the fabs if the source was already known positive.
2602 if (KnownLHS.isKnownAlways(fcPositive))
2603 return X;
2604
2605 // => fabs(x), in case this was a -inf or -0.
2606 // Note: Dropping canonicalize.
2608 Builder.SetInsertPoint(I);
2609 Value *Fabs = Builder.CreateFAbs(X, FMF);
2610 Fabs->takeName(I);
2611 return Fabs;
2612 }
2613
2614 return nullptr;
2615 }
2616
2617 if (SimplifyDemandedFPClass(I, 1, SrcDemandedMask, KnownRHS, SQ,
2618 Depth + 1) ||
2619 SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownLHS, SQ, Depth + 1))
2620 return I;
2621
2622 if (FMF.noInfs()) {
2623 // Flag implies inputs cannot be infinity.
2624 KnownLHS.knownNot(fcInf);
2625 KnownRHS.knownNot(fcInf);
2626 }
2627
2628 bool NonNanResult = (DemandedMask & fcNan) == fcNone;
2629
2630 // With no-nans/no-infs:
2631 // X * 0.0 --> copysign(0.0, X)
2632 // X * -0.0 --> copysign(0.0, -X)
2633 if ((NonNanResult || KnownLHS.isKnownNeverInfOrNaN()) &&
2634 KnownRHS.isKnownAlways(fcPosZero | fcNan)) {
2636 Builder.SetInsertPoint(I);
2637
2638 // => copysign(+0, lhs)
2639 // Note: Dropping canonicalize
2640 Value *Copysign = Builder.CreateCopySign(Y, X, FMF);
2641 Copysign->takeName(I);
2642 return Copysign;
2643 }
2644
2645 if (KnownLHS.isKnownAlways(fcPosZero | fcNan) &&
2646 (NonNanResult || KnownRHS.isKnownNeverInfOrNaN())) {
2648 Builder.SetInsertPoint(I);
2649
2650 // => copysign(+0, rhs)
2651 // Note: Dropping canonicalize
2652 Value *Copysign = Builder.CreateCopySign(X, Y, FMF);
2653 Copysign->takeName(I);
2654 return Copysign;
2655 }
2656
2657 if ((NonNanResult || KnownLHS.isKnownNeverInfOrNaN()) &&
2658 KnownRHS.isKnownAlways(fcNegZero | fcNan)) {
2660 Builder.SetInsertPoint(I);
2661
2662 // => copysign(0, fneg(lhs))
2663 // Note: Dropping canonicalize
2664 Value *Copysign =
2665 Builder.CreateCopySign(Y, Builder.CreateFNegFMF(X, FMF), FMF);
2666 Copysign->takeName(I);
2667 return Copysign;
2668 }
2669
2670 if (KnownLHS.isKnownAlways(fcNegZero | fcNan) &&
2671 (NonNanResult || KnownRHS.isKnownNeverInfOrNaN())) {
2673 Builder.SetInsertPoint(I);
2674
2675 // => copysign(+0, fneg(rhs))
2676 // Note: Dropping canonicalize
2677 Value *Copysign =
2678 Builder.CreateCopySign(X, Builder.CreateFNegFMF(Y, FMF), FMF);
2679 Copysign->takeName(I);
2680 return Copysign;
2681 }
2682
2683 Type *EltTy = VTy->getScalarType();
2684 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2685
2686 if (KnownLHS.isKnownAlways(fcInf | fcNan) &&
2687 (KnownRHS.isKnownNeverNaN() &&
2688 KnownRHS.cannotBeOrderedGreaterEqZero(Mode))) {
2690 Builder.SetInsertPoint(I);
2691
2692 // Note: Dropping canonicalize
2693 Value *Neg = Builder.CreateFNegFMF(X, FMF);
2694 Neg->takeName(I);
2695 return Neg;
2696 }
2697
2698 if (KnownRHS.isKnownAlways(fcInf | fcNan) &&
2699 (KnownLHS.isKnownNeverNaN() &&
2700 KnownLHS.cannotBeOrderedGreaterEqZero(Mode))) {
2702 Builder.SetInsertPoint(I);
2703
2704 // Note: Dropping canonicalize
2705 Value *Neg = Builder.CreateFNegFMF(Y, FMF);
2706 Neg->takeName(I);
2707 return Neg;
2708 }
2709
2710 Known = KnownFPClass::fmul(KnownLHS, KnownRHS, Mode);
2711 Known.knownNot(~DemandedMask);
2712
2713 if (Constant *SingleVal = getFPClassConstant(VTy, Known.getKnownFPClasses(),
2714 /*IsCanonicalizing=*/true))
2715 return SingleVal;
2716
2718 FMF, Known.getKnownFPClasses(), {KnownLHS, KnownRHS});
2719 if (InferredFMF != FMF) {
2720 I->setFastMathFlags(InferredFMF);
2721 return I;
2722 }
2723
2724 return nullptr;
2725 }
2726 case Instruction::FDiv: {
2727 Value *X = I->getOperand(0);
2728 Value *Y = I->getOperand(1);
2729 if (X == Y &&
2730 isGuaranteedNotToBeUndef(X, SQ.AC, SQ.CtxI, SQ.DT, Depth + 1)) {
2731 // If the source is 0, inf or nan, the result is a nan
2733 Builder.SetInsertPoint(I);
2734
2735 Value *IsZeroOrNan = Builder.CreateFCmpFMF(
2736 FCmpInst::FCMP_UEQ, I->getOperand(0), ConstantFP::getZero(VTy), FMF);
2737
2738 Value *Fabs = Builder.CreateFAbs(I->getOperand(0), FMF);
2739 Value *IsInfOrNan = Builder.CreateFCmpFMF(
2741
2742 Value *IsInfOrZeroOrNan = Builder.CreateOr(IsInfOrNan, IsZeroOrNan);
2743
2744 return Builder.CreateSelectFMFWithUnknownProfile(
2745 IsInfOrZeroOrNan, ConstantFP::getQNaN(VTy),
2746 ConstantFP::get(
2748 FMF, DEBUG_TYPE);
2749 }
2750
2751 Type *EltTy = VTy->getScalarType();
2752 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2753
2754 // Every output class could require denormal inputs (except for the
2755 // degenerate case of only-nan results, without DAZ).
2756 FPClassTest SrcDemandedMask = (DemandedMask & fcNan) | fcSubnormal;
2757
2758 // Normal inputs may result in underflow.
2759 // x / x = 1.0 for non0/inf/nan
2760 // -x = +y / -z
2761 // -x = -y / +z
2762 if (DemandedMask & (fcSubnormal | fcNormal))
2763 SrcDemandedMask |= fcNormal;
2764
2765 if (DemandedMask & fcNan) {
2766 // 0 / 0 = nan
2767 // inf / inf = nan
2768
2769 // Subnormal is added in case of DAZ, but this isn't strictly
2770 // necessary. Every other input class implies a possible subnormal source,
2771 // so this only could matter in the degenerate case of only-nan results.
2772 SrcDemandedMask |= fcZero | fcInf | fcNan;
2773 }
2774
2775 // Zero outputs may be the result of underflow.
2776 if (DemandedMask & fcZero)
2777 SrcDemandedMask |= fcNormal | fcSubnormal;
2778
2779 FPClassTest LHSDemandedMask = SrcDemandedMask;
2780 FPClassTest RHSDemandedMask = SrcDemandedMask;
2781
2782 // 0 / inf = 0
2783 if (DemandedMask & fcZero) {
2784 assert((LHSDemandedMask & fcSubnormal) &&
2785 "should not have to worry about daz here");
2786 LHSDemandedMask |= fcZero;
2787 RHSDemandedMask |= fcInf;
2788 }
2789
2790 // x / 0 = inf
2791 // large_normal / small_normal = inf
2792 // inf / 1 = inf
2793 // large_normal / subnormal = inf
2794 if (DemandedMask & fcInf) {
2795 LHSDemandedMask |= fcInf | fcNormal | fcSubnormal;
2796 RHSDemandedMask |= fcZero | fcSubnormal | fcNormal;
2797 }
2798
2799 KnownFPClass KnownLHS, KnownRHS;
2800 if (SimplifyDemandedFPClass(I, 0, LHSDemandedMask, KnownLHS, SQ,
2801 Depth + 1) ||
2802 SimplifyDemandedFPClass(I, 1, RHSDemandedMask, KnownRHS, SQ, Depth + 1))
2803 return I;
2804
2805 bool ResultNotNan = (DemandedMask & fcNan) == fcNone;
2806 bool ResultNotInf = (DemandedMask & fcInf) == fcNone;
2807
2808 // Replacing 0/x with a zero is only valid when the divisor can't be
2809 // (logical) zero, since 0/0 is NaN -- unless NaN results aren't demanded. A
2810 // subnormal divisor can flush to zero under a flushing denormal mode.
2811 bool CanIgnoreZeroByZeroNan =
2812 ResultNotNan || KnownRHS.isKnownNeverLogicalZero(Mode);
2813
2814 // nsz [+-]0 / x -> 0
2815 if (FMF.noSignedZeros() && KnownLHS.isKnownAlways(fcZero) &&
2816 KnownRHS.isKnownNeverNaN() && CanIgnoreZeroByZeroNan)
2817 return ConstantFP::getZero(VTy);
2818
2819 if (KnownLHS.isKnownAlways(fcPosZero) && KnownRHS.isKnownNeverNaN() &&
2820 CanIgnoreZeroByZeroNan) {
2822 Builder.SetInsertPoint(I);
2823
2824 // nnan +0 / x -> copysign(0, rhs)
2825 // TODO: -0 / x => copysign(0, fneg(rhs))
2826 Value *Copysign = Builder.CreateCopySign(X, Y, FMF);
2827 Copysign->takeName(I);
2828 return Copysign;
2829 }
2830
2831 if (!ResultNotInf &&
2832 ((ResultNotNan || (KnownLHS.isKnownNeverNaN() &&
2833 KnownLHS.isKnownNeverLogicalZero(Mode))) &&
2834 (KnownRHS.isKnownAlways(fcPosZero) ||
2835 (FMF.noSignedZeros() && KnownRHS.isKnownAlways(fcZero))))) {
2837 Builder.SetInsertPoint(I);
2838
2839 // nnan x / 0 => copysign(inf, x);
2840 // nnan nsz x / -0 => copysign(inf, x);
2841 Value *Copysign =
2842 Builder.CreateCopySign(ConstantFP::getInfinity(VTy), X, FMF);
2843 Copysign->takeName(I);
2844 return Copysign;
2845 }
2846
2847 // nnan ninf X / [-]0.0 -> poison
2848 if (ResultNotNan && ResultNotInf && KnownRHS.isKnownAlways(fcZero))
2849 return PoisonValue::get(VTy);
2850
2851 Known = KnownFPClass::fdiv(KnownLHS, KnownRHS, Mode);
2852 Known.knownNot(~DemandedMask);
2853
2854 if (Constant *SingleVal = getFPClassConstant(VTy, Known.getKnownFPClasses(),
2855 /*IsCanonicalizing=*/true))
2856 return SingleVal;
2857
2859 FMF, Known.getKnownFPClasses(), {KnownLHS, KnownRHS});
2860 if (InferredFMF != FMF) {
2861 I->setFastMathFlags(InferredFMF);
2862 return I;
2863 }
2864
2865 return nullptr;
2866 }
2867 case Instruction::FPTrunc:
2868 return simplifyDemandedUseFPClassFPTrunc(*this, *I, FMF, DemandedMask,
2869 Known, SQ, Depth);
2870 case Instruction::FPExt: {
2871 FPClassTest SrcDemandedMask = DemandedMask;
2872 if (DemandedMask & fcNan)
2873 SrcDemandedMask |= fcNan;
2874
2875 // No subnormal result does not imply not-subnormal in the source type.
2876 if ((DemandedMask & fcNegNormal) != fcNone)
2877 SrcDemandedMask |= fcNegSubnormal;
2878 if ((DemandedMask & fcPosNormal) != fcNone)
2879 SrcDemandedMask |= fcPosSubnormal;
2880
2881 KnownFPClass KnownSrc;
2882 if (SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownSrc, SQ, Depth + 1))
2883 return I;
2884
2885 const fltSemantics &DstTy = VTy->getScalarType()->getFltSemantics();
2886 const fltSemantics &SrcTy =
2887 I->getOperand(0)->getType()->getScalarType()->getFltSemantics();
2888
2889 Known = KnownFPClass::fpext(KnownSrc, DstTy, SrcTy);
2890 Known.knownNot(~DemandedMask);
2891
2892 return simplifyDemandedFPClassResult(I, FMF, DemandedMask, Known,
2893 {KnownSrc});
2894 }
2895 case Instruction::Call: {
2896 CallInst *CI = cast<CallInst>(I);
2897 const Intrinsic::ID IID = CI->getIntrinsicID();
2898 switch (IID) {
2899 case Intrinsic::fabs: {
2900 KnownFPClass KnownSrc;
2901 if (SimplifyDemandedFPClass(I, 0, llvm::inverse_fabs(DemandedMask),
2902 KnownSrc, SQ, Depth + 1))
2903 return I;
2904
2905 if (Value *Simplified = simplifyDemandedFPClassFabs(
2906 Known, CI->getArgOperand(0), DemandedMask, KnownSrc,
2907 FMF.noSignedZeros()))
2908 return Simplified;
2909 break;
2910 }
2911 case Intrinsic::arithmetic_fence:
2912 if (SimplifyDemandedFPClass(I, 0, DemandedMask, Known, SQ, Depth + 1))
2913 return I;
2914 break;
2915 case Intrinsic::copysign: {
2916 // Flip on more potentially demanded classes
2917 const FPClassTest DemandedMaskAnySign = llvm::unknown_sign(DemandedMask);
2918 KnownFPClass KnownMag;
2919 if (SimplifyDemandedFPClass(CI, 0, DemandedMaskAnySign, KnownMag, SQ,
2920 Depth + 1))
2921 return I;
2922
2923 if ((DemandedMask & fcNegative) == DemandedMask) {
2924 // Roundabout way of replacing with fneg(fabs)
2925 CI->setOperand(1, ConstantFP::get(VTy, -1.0));
2926 return I;
2927 }
2928
2929 if ((DemandedMask & fcPositive) == DemandedMask) {
2930 // Roundabout way of replacing with fabs
2931 CI->setOperand(1, ConstantFP::getZero(VTy));
2932 return I;
2933 }
2934
2935 if (Value *Simplified = simplifyDemandedFPClassCopysignMag(
2936 CI->getArgOperand(0), DemandedMask, KnownMag,
2937 FMF.noSignedZeros()))
2938 return Simplified;
2939
2940 KnownFPClass KnownSign =
2942 if (KnownMag.getSignBit() && KnownSign.getSignBit() &&
2943 *KnownMag.getSignBit() == *KnownSign.getSignBit())
2944 return CI->getOperand(0);
2945
2946 // TODO: Call argument attribute not considered
2947 // Input implied not-nan from flag.
2948 if (FMF.noNaNs())
2949 KnownSign.knownNot(fcNan);
2950
2951 if (KnownSign.getSignBit() == false) {
2953 CI->setOperand(1, ConstantFP::getZero(VTy));
2954 return I;
2955 }
2956
2957 if (KnownSign.getSignBit() == true) {
2959 CI->setOperand(1, ConstantFP::get(VTy, -1.0));
2960 return I;
2961 }
2962
2963 Known = KnownFPClass::copysign(KnownMag, KnownSign);
2964 Known.knownNot(~DemandedMask);
2965 break;
2966 }
2967 case Intrinsic::fma:
2968 case Intrinsic::fmuladd: {
2969 // We can't do any simplification on the source besides stripping out
2970 // unneeded nans.
2971 FPClassTest SrcDemandedMask = DemandedMask | ~fcNan;
2972 if (DemandedMask & fcNan)
2973 SrcDemandedMask |= fcNan;
2974
2975 KnownFPClass KnownSrc[3];
2976
2977 Type *EltTy = VTy->getScalarType();
2978 if (CI->getArgOperand(0) == CI->getArgOperand(1) &&
2979 isGuaranteedNotToBeUndef(CI->getArgOperand(0), SQ.AC, SQ.CtxI, SQ.DT,
2980 Depth + 1)) {
2981 if (SimplifyDemandedFPClass(CI, 0, SrcDemandedMask, KnownSrc[0], SQ,
2982 Depth + 1) ||
2983 SimplifyDemandedFPClass(CI, 2, SrcDemandedMask, KnownSrc[2], SQ,
2984 Depth + 1))
2985 return I;
2986
2987 KnownSrc[1] = KnownSrc[0];
2988 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2989 Known = KnownFPClass::fma_square(KnownSrc[0], KnownSrc[2], Mode);
2990 } else {
2991 for (int OpIdx = 0; OpIdx != 3; ++OpIdx) {
2992 if (SimplifyDemandedFPClass(CI, OpIdx, SrcDemandedMask,
2993 KnownSrc[OpIdx], SQ, Depth + 1))
2994 return CI;
2995 }
2996
2997 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
2998 Known = KnownFPClass::fma(KnownSrc[0], KnownSrc[1], KnownSrc[2], Mode);
2999 }
3000
3001 return simplifyDemandedFPClassResult(CI, FMF, DemandedMask, Known,
3002 {KnownSrc});
3003 }
3004 case Intrinsic::maximum:
3005 case Intrinsic::minimum:
3006 case Intrinsic::maximumnum:
3007 case Intrinsic::minimumnum:
3008 case Intrinsic::maxnum:
3009 case Intrinsic::minnum: {
3010 const bool PropagateNaN =
3011 IID == Intrinsic::maximum || IID == Intrinsic::minimum;
3012
3013 // We can't tell much based on the demanded result without inspecting the
3014 // operands (e.g., a known-positive result could have been clamped), but
3015 // we can still prune known-nan inputs.
3016 FPClassTest SrcDemandedMask =
3017 PropagateNaN && ((DemandedMask & fcNan) == fcNone)
3018 ? DemandedMask | ~fcNan
3019 : fcAllFlags;
3020
3021 KnownFPClass KnownLHS, KnownRHS;
3022 if (SimplifyDemandedFPClass(CI, 1, SrcDemandedMask, KnownRHS, SQ,
3023 Depth + 1) ||
3024 SimplifyDemandedFPClass(CI, 0, SrcDemandedMask, KnownLHS, SQ,
3025 Depth + 1))
3026 return I;
3027
3028 Value *Simplified =
3029 simplifyDemandedFPClassMinMax(Known, IID, CI, DemandedMask, KnownLHS,
3030 KnownRHS, F, FMF.noSignedZeros());
3031 if (Simplified)
3032 return Simplified;
3033
3034 auto *FPOp = cast<FPMathOperator>(CI);
3035
3036 FPClassTest ValidResults = DemandedMask & Known.getKnownFPClasses();
3037 FastMathFlags InferredFMF = FMF;
3038
3039 if (!FMF.noSignedZeros()) {
3040 // Add NSZ flag if we know the result will not be sensitive to the sign
3041 // of 0.
3042 FPClassTest ZeroMask = fcZero;
3043
3044 Type *EltTy = VTy->getScalarType();
3045 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
3046 if (Mode != DenormalMode::getIEEE())
3047 ZeroMask |= fcSubnormal;
3048
3049 bool ResultNotLogical0 = (ValidResults & ZeroMask) == fcNone;
3050 if (ResultNotLogical0 || ((KnownLHS.isKnownNeverLogicalNegZero(Mode) ||
3051 KnownRHS.isKnownNeverLogicalPosZero(Mode)) &&
3052 (KnownLHS.isKnownNeverLogicalPosZero(Mode) ||
3053 KnownRHS.isKnownNeverLogicalNegZero(Mode))))
3054 InferredFMF.setNoSignedZeros(true);
3055 }
3056
3057 if (!FMF.noNaNs() &&
3058 ((PropagateNaN && (ValidResults & fcNan) == fcNone) ||
3059 (KnownLHS.isKnownNeverNaN() && KnownRHS.isKnownNeverNaN()))) {
3061 InferredFMF.setNoNaNs(true);
3062 }
3063
3064 if (InferredFMF != FMF) {
3065 CI->setFastMathFlags(InferredFMF);
3066 return FPOp;
3067 }
3068
3069 return nullptr;
3070 }
3071 case Intrinsic::exp:
3072 case Intrinsic::exp2:
3073 case Intrinsic::exp10: {
3074 if ((DemandedMask & fcPositive) == fcNone) {
3075 // Only returns positive values or nans.
3076 if ((DemandedMask & fcNan) == fcNone)
3077 return PoisonValue::get(VTy);
3078
3079 // Only need nan propagation.
3080 if ((DemandedMask & ~fcNan) == fcNone)
3081 return ConstantFP::getQNaN(VTy);
3082
3083 return CI->getArgOperand(0);
3084 }
3085
3086 FPClassTest SrcDemandedMask = DemandedMask & fcNan;
3087 if (DemandedMask & fcNan)
3088 SrcDemandedMask |= fcNan;
3089
3090 if (DemandedMask & fcZero) {
3091 // exp(-infinity) = 0
3092 SrcDemandedMask |= fcNegInf;
3093
3094 // exp(-largest_normal) = 0
3095 //
3096 // Negative numbers of sufficiently large magnitude underflow to 0. No
3097 // subnormal input has a 0 result.
3098 SrcDemandedMask |= fcNegNormal;
3099 }
3100
3101 if (DemandedMask & fcPosSubnormal) {
3102 // Negative numbers of sufficiently large magnitude underflow to 0. No
3103 // subnormal input has a 0 result.
3104 SrcDemandedMask |= fcNegNormal;
3105 }
3106
3107 if (DemandedMask & fcPosNormal) {
3108 // exp(0) = 1
3109 // exp(+/- smallest_normal) = 1
3110 // exp(+/- largest_denormal) = 1
3111 // exp(+/- smallest_denormal) = 1
3112 // exp(-1) = pos normal
3113 SrcDemandedMask |= fcNormal | fcSubnormal | fcZero;
3114 }
3115
3116 // exp(inf), exp(largest_normal) = inf
3117 if (DemandedMask & fcPosInf)
3118 SrcDemandedMask |= fcPosInf | fcPosNormal;
3119
3120 KnownFPClass KnownSrc;
3121
3122 // TODO: This could really make use of KnownFPClass of specific value
3123 // range, (i.e., close enough to 1)
3124 if (SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownSrc, SQ,
3125 Depth + 1))
3126 return I;
3127
3128 // exp(+/-0) = 1
3129 if (KnownSrc.isKnownAlways(fcZero))
3130 return ConstantFP::get(VTy, 1.0);
3131
3132 // Only perform nan propagation.
3133 // Note: Dropping canonicalize / quiet of signaling nan.
3134 if (KnownSrc.isKnownAlways(fcNan))
3135 return CI->getArgOperand(0);
3136
3137 // exp(0 | nan) => x == 0.0 ? 1.0 : x
3138 if (KnownSrc.isKnownAlways(fcZero | fcNan)) {
3140 Builder.SetInsertPoint(CI);
3141
3142 // fadd +/-0, 1.0 => 1.0
3143 // fadd nan, 1.0 => nan
3144 return Builder.CreateFAddFMF(CI->getArgOperand(0),
3145 ConstantFP::get(VTy, 1.0), FMF);
3146 }
3147
3148 if (KnownSrc.isKnownAlways(fcInf | fcNan)) {
3149 // exp(-inf) = 0
3150 // exp(+inf) = +inf
3152 Builder.SetInsertPoint(CI);
3153
3154 // Note: Dropping canonicalize / quiet of signaling nan.
3155 Value *X = CI->getArgOperand(0);
3156 Value *IsPosInfOrNan = Builder.CreateFCmpFMF(
3158 // We do not know whether an infinity or a NaN is more likely here,
3159 // so mark the branch weights as unkown.
3160 Value *ZeroOrInf = Builder.CreateSelectFMFWithUnknownProfile(
3161 IsPosInfOrNan, X, ConstantFP::getZero(VTy), FMF, DEBUG_TYPE);
3162 return ZeroOrInf;
3163 }
3164
3165 Known = KnownFPClass::exp(KnownSrc);
3166 Known.knownNot(~DemandedMask);
3167
3168 return simplifyDemandedFPClassResult(CI, FMF, DemandedMask, Known,
3169 KnownSrc);
3170 }
3171 case Intrinsic::log:
3172 case Intrinsic::log2:
3173 case Intrinsic::log10: {
3174 FPClassTest DemandedSrcMask = DemandedMask & (fcNan | fcPosInf);
3175 if (DemandedMask & fcNan)
3176 DemandedSrcMask |= fcNan;
3177
3178 Type *EltTy = VTy->getScalarType();
3179 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
3180
3181 // log(x < 0) = nan
3182 if (DemandedMask & fcNan)
3183 DemandedSrcMask |= (fcNegative & ~fcNegZero);
3184
3185 // log(0) = -inf
3186 if (DemandedMask & fcNegInf) {
3187 DemandedSrcMask |= fcZero;
3188
3189 // No value produces subnormal result.
3190 if (Mode.inputsMayBeZero())
3191 DemandedSrcMask |= fcSubnormal;
3192 }
3193
3194 if (DemandedMask & fcNormal)
3195 DemandedSrcMask |= fcNormal | fcSubnormal;
3196
3197 // log(1) = 0
3198 if (DemandedMask & fcZero)
3199 DemandedSrcMask |= fcPosNormal;
3200
3201 KnownFPClass KnownSrc;
3202 if (SimplifyDemandedFPClass(I, 0, DemandedSrcMask, KnownSrc, SQ,
3203 Depth + 1))
3204 return I;
3205
3206 Known = KnownFPClass::log(KnownSrc, Mode);
3207 Known.knownNot(~DemandedMask);
3208
3209 return simplifyDemandedFPClassResult(CI, FMF, DemandedMask, Known,
3210 KnownSrc);
3211 }
3212 case Intrinsic::sqrt: {
3213 FPClassTest DemandedSrcMask =
3214 DemandedMask & (fcNegZero | fcPositive | fcNan);
3215
3216 if (DemandedMask & fcNan)
3217 DemandedSrcMask |= fcNan | (fcNegative & ~fcNegZero);
3218
3219 // sqrt(max_subnormal) is a normal value
3220 if (DemandedMask & fcPosNormal)
3221 DemandedSrcMask |= fcPosSubnormal;
3222
3223 KnownFPClass KnownSrc;
3224 if (SimplifyDemandedFPClass(I, 0, DemandedSrcMask, KnownSrc, SQ,
3225 Depth + 1))
3226 return I;
3227
3228 // Infer the source cannot be negative if the result cannot be nan.
3229 if ((DemandedMask & fcNan) == fcNone)
3230 KnownSrc.knownNot((fcNegative & ~fcNegZero) | fcNan);
3231
3232 // Infer the source cannot be +inf if the result is not +nf
3233 if ((DemandedMask & fcPosInf) == fcNone)
3234 KnownSrc.knownNot(fcPosInf);
3235
3236 Type *EltTy = VTy->getScalarType();
3237 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
3238
3239 // sqrt(-x) = nan, but be careful of negative subnormals flushed to 0.
3240 if (KnownSrc.isKnownNever(fcPositive) &&
3241 KnownSrc.isKnownNeverLogicalZero(Mode))
3242 return ConstantFP::getQNaN(VTy);
3243
3244 Known = KnownFPClass::sqrt(KnownSrc, Mode);
3245 Known.knownNot(~DemandedMask);
3246
3247 if (Known.getKnownFPClasses() == fcZero) {
3248 if (FMF.noSignedZeros())
3249 return ConstantFP::getZero(VTy);
3251 Builder.SetInsertPoint(CI);
3252
3253 Value *Copysign = Builder.CreateCopySign(ConstantFP::getZero(VTy),
3254 CI->getArgOperand(0), FMF);
3255 Copysign->takeName(CI);
3256 return Copysign;
3257 }
3258
3259 return simplifyDemandedFPClassResult(CI, FMF, DemandedMask, Known,
3260 {KnownSrc});
3261 }
3262 case Intrinsic::ldexp: {
3263 FPClassTest SrcDemandedMask = DemandedMask & fcInf;
3264 if (DemandedMask & fcNan)
3265 SrcDemandedMask |= fcNan;
3266
3267 if (DemandedMask & fcPosInf)
3268 SrcDemandedMask |= fcPosNormal | fcPosSubnormal;
3269 if (DemandedMask & fcNegInf)
3270 SrcDemandedMask |= fcNegNormal | fcNegSubnormal;
3271
3272 if (DemandedMask & (fcPosNormal | fcPosSubnormal))
3273 SrcDemandedMask |= fcPosNormal | fcPosSubnormal;
3274 if (DemandedMask & (fcNegNormal | fcNegSubnormal))
3275 SrcDemandedMask |= fcNegNormal | fcNegSubnormal;
3276
3277 if (DemandedMask & fcPosZero)
3278 SrcDemandedMask |= fcPosFinite;
3279 if (DemandedMask & fcNegZero)
3280 SrcDemandedMask |= fcNegFinite;
3281
3282 KnownFPClass KnownSrc;
3283 if (SimplifyDemandedFPClass(CI, 0, SrcDemandedMask, KnownSrc, SQ,
3284 Depth + 1))
3285 return CI;
3286
3287 Type *EltTy = VTy->getScalarType();
3288 const fltSemantics &FltSem = EltTy->getFltSemantics();
3289 DenormalMode Mode = F.getDenormalMode(FltSem);
3290
3291 KnownBits KnownExpBits =
3293
3294 Known = KnownFPClass::ldexp(KnownSrc, KnownExpBits, FltSem, Mode);
3295 Known.knownNot(~DemandedMask);
3296
3297 return simplifyDemandedFPClassResult(CI, FMF, DemandedMask, Known,
3298 {KnownSrc});
3299 }
3300 case Intrinsic::trunc:
3301 case Intrinsic::floor:
3302 case Intrinsic::ceil:
3303 case Intrinsic::rint:
3304 case Intrinsic::nearbyint:
3305 case Intrinsic::round:
3306 case Intrinsic::roundeven: {
3307 FPClassTest DemandedSrcMask = DemandedMask;
3308 if (DemandedMask & fcNan)
3309 DemandedSrcMask |= fcNan;
3310
3311 // Zero results imply valid subnormal sources.
3312 if (DemandedMask & fcNegZero)
3313 DemandedSrcMask |= fcNegSubnormal | fcNegNormal;
3314
3315 if (DemandedMask & fcPosZero)
3316 DemandedSrcMask |= fcPosSubnormal | fcPosNormal;
3317
3318 KnownFPClass KnownSrc;
3319 if (SimplifyDemandedFPClass(CI, 0, DemandedSrcMask, KnownSrc, SQ,
3320 Depth + 1))
3321 return I;
3322
3323 // Note: Possibly dropping snan quiet.
3324 if (KnownSrc.isKnownAlways(fcInf | fcNan | fcZero))
3325 return CI->getArgOperand(0);
3326
3327 bool IsRoundNearestOrTrunc =
3328 IID == Intrinsic::round || IID == Intrinsic::roundeven ||
3329 IID == Intrinsic::nearbyint || IID == Intrinsic::rint ||
3330 IID == Intrinsic::trunc;
3331
3332 // Ignore denormals-as-zero, as canonicalization is not mandated.
3333 if ((IID == Intrinsic::floor || IsRoundNearestOrTrunc) &&
3335 return ConstantFP::getZero(VTy);
3336
3337 if ((IID == Intrinsic::ceil || IsRoundNearestOrTrunc) &&
3339 return ConstantFP::getZero(VTy, true);
3340
3341 if (IID == Intrinsic::floor && KnownSrc.isKnownAlways(fcNegSubnormal))
3342 return ConstantFP::get(VTy, -1.0);
3343
3344 if (IID == Intrinsic::ceil && KnownSrc.isKnownAlways(fcPosSubnormal))
3345 return ConstantFP::get(VTy, 1.0);
3346
3348 KnownSrc, IID == Intrinsic::trunc,
3350
3351 Known.knownNot(~DemandedMask);
3352
3353 if (Constant *SingleVal =
3354 getFPClassConstant(VTy, Known.getKnownFPClasses(),
3355 /*IsCanonicalizing=*/true))
3356 return SingleVal;
3357
3358 if ((IID == Intrinsic::trunc || IsRoundNearestOrTrunc) &&
3359 KnownSrc.isKnownAlways(fcZero | fcSubnormal)) {
3361 Builder.SetInsertPoint(CI);
3362
3363 Value *Copysign = Builder.CreateCopySign(ConstantFP::getZero(VTy),
3364 CI->getArgOperand(0));
3365 Copysign->takeName(CI);
3366 return Copysign;
3367 }
3368
3369 FastMathFlags InferredFMF =
3370 inferFastMathValueFlags(FMF, Known.getKnownFPClasses(), KnownSrc);
3371 if (InferredFMF != FMF) {
3373 CI->setFastMathFlags(InferredFMF);
3374 return CI;
3375 }
3376
3377 return nullptr;
3378 }
3379 case Intrinsic::fptrunc_round:
3380 return simplifyDemandedUseFPClassFPTrunc(*this, *CI, FMF, DemandedMask,
3381 Known, SQ, Depth);
3382 case Intrinsic::canonicalize: {
3383 Type *EltTy = VTy->getScalarType();
3384
3385 // TODO: This could have more refined support for PositiveZero denormal
3386 // mode.
3387 if (EltTy->isIEEELikeFPTy()) {
3388 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
3389
3390 FPClassTest SrcDemandedMask = DemandedMask;
3391
3392 // A demanded quiet nan result may have come from a signaling nan, so we
3393 // need to expand the demanded mask.
3394 if ((DemandedMask & fcQNan) != fcNone)
3395 SrcDemandedMask |= fcSNan;
3396
3397 if (Mode != DenormalMode::getIEEE()) {
3398 // Any zero results may have come from flushed denormals.
3399 if (DemandedMask & fcPosZero)
3400 SrcDemandedMask |= fcPosSubnormal;
3401 if (DemandedMask & fcNegZero)
3402 SrcDemandedMask |= fcNegSubnormal;
3403 }
3404
3405 if (Mode == DenormalMode::getPreserveSign()) {
3406 // If a denormal input will be flushed, and we don't need zeros, we
3407 // don't need denormals either.
3408 if ((DemandedMask & fcPosZero) == fcNone)
3409 SrcDemandedMask &= ~fcPosSubnormal;
3410
3411 if ((DemandedMask & fcNegZero) == fcNone)
3412 SrcDemandedMask &= ~fcNegSubnormal;
3413 }
3414
3415 KnownFPClass KnownSrc;
3416
3417 // Simplify upstream operations before trying to simplify this call.
3418 if (SimplifyDemandedFPClass(I, 0, SrcDemandedMask, KnownSrc, SQ,
3419 Depth + 1))
3420 return I;
3421
3422 // Perform the canonicalization to see if this folded to a constant.
3423 Known = KnownFPClass::canonicalize(KnownSrc, Mode);
3424 Known.knownNot(~DemandedMask);
3425
3426 if (Constant *SingleVal =
3427 getFPClassConstant(VTy, Known.getKnownFPClasses()))
3428 return SingleVal;
3429
3430 // For IEEE handling, there is only a bit change for nan inputs, so we
3431 // can drop it if we do not demand nan results or we know the input
3432 // isn't a nan.
3433 // Otherwise, we also need to avoid denormal inputs to drop the
3434 // canonicalize.
3435 if (KnownSrc.isKnownNeverNaN() && (Mode == DenormalMode::getIEEE() ||
3436 KnownSrc.isKnownNeverSubnormal()))
3437 return CI->getArgOperand(0);
3438
3439 FastMathFlags InferredFMF =
3440 inferFastMathValueFlags(FMF, Known.getKnownFPClasses(), KnownSrc);
3441 if (InferredFMF != FMF) {
3443 CI->setFastMathFlags(InferredFMF);
3444 return CI;
3445 }
3446
3447 return nullptr;
3448 }
3449
3450 [[fallthrough]];
3451 }
3452 default:
3453 Known = computeKnownFPClass(I, DemandedMask, SQ, Depth + 1);
3454 Known.knownNot(~DemandedMask);
3455 break;
3456 }
3457
3458 break;
3459 }
3460 case Instruction::Select: {
3461 KnownFPClass KnownLHS, KnownRHS;
3462 if (SimplifyDemandedFPClass(I, 2, DemandedMask, KnownRHS, SQ, Depth + 1) ||
3463 SimplifyDemandedFPClass(I, 1, DemandedMask, KnownLHS, SQ, Depth + 1))
3464 return I;
3465
3466 if (KnownLHS.isKnownNever(DemandedMask))
3467 return I->getOperand(2);
3468 if (KnownRHS.isKnownNever(DemandedMask))
3469 return I->getOperand(1);
3470
3471 adjustKnownFPClassForSelectArm(KnownLHS, I->getOperand(0), I->getOperand(1),
3472 /*Invert=*/false, SQ, Depth);
3473 adjustKnownFPClassForSelectArm(KnownRHS, I->getOperand(0), I->getOperand(2),
3474 /*Invert=*/true, SQ, Depth);
3475 Known = KnownLHS.intersectWith(KnownRHS);
3476 Known.knownNot(~DemandedMask);
3477 break;
3478 }
3479 case Instruction::ExtractElement: {
3480 // TODO: Handle demanded element mask
3481 if (SimplifyDemandedFPClass(I, 0, DemandedMask, Known, SQ, Depth + 1))
3482 return I;
3483 Known.knownNot(~DemandedMask);
3484 break;
3485 }
3486 case Instruction::InsertElement: {
3487 KnownFPClass KnownInserted, KnownVec;
3488 if (SimplifyDemandedFPClass(I, 1, DemandedMask, KnownInserted, SQ,
3489 Depth + 1) ||
3490 SimplifyDemandedFPClass(I, 0, DemandedMask, KnownVec, SQ, Depth + 1))
3491 return I;
3492
3493 // TODO: Use demanded elements logic from computeKnownFPClass
3494 Known = KnownVec | KnownInserted;
3495 Known.knownNot(~DemandedMask);
3496 break;
3497 }
3498 case Instruction::ShuffleVector: {
3499 KnownFPClass KnownLHS, KnownRHS;
3500 if (SimplifyDemandedFPClass(I, 1, DemandedMask, KnownRHS, SQ, Depth + 1) ||
3501 SimplifyDemandedFPClass(I, 0, DemandedMask, KnownLHS, SQ, Depth + 1))
3502 return I;
3503
3504 // TODO: This is overly conservative and should consider demanded elements,
3505 // and splats.
3506 Known = KnownLHS | KnownRHS;
3507 Known.knownNot(~DemandedMask);
3508 break;
3509 }
3510 case Instruction::InsertValue: {
3511 KnownFPClass KnownAgg, KnownElt;
3512 if (SimplifyDemandedFPClass(I, 0, DemandedMask, KnownAgg, SQ, Depth + 1) ||
3513 SimplifyDemandedFPClass(I, 1, DemandedMask, KnownElt, SQ, Depth + 1))
3514 return I;
3515
3516 Known = KnownAgg | KnownElt;
3517 break;
3518 }
3519 case Instruction::ExtractValue: {
3520 Value *ExtractSrc;
3521 if (match(I, m_ExtractValue<0>(m_OneUse(m_Value(ExtractSrc))))) {
3522 if (auto *II = dyn_cast<IntrinsicInst>(ExtractSrc)) {
3523 const Intrinsic::ID IID = II->getIntrinsicID();
3524 switch (IID) {
3525 case Intrinsic::frexp: {
3526 FPClassTest SrcDemandedMask = fcNone;
3527 if (DemandedMask & fcNan)
3528 SrcDemandedMask |= fcNan;
3529 if (DemandedMask & fcNegFinite)
3530 SrcDemandedMask |= fcNegFinite;
3531 if (DemandedMask & fcPosFinite)
3532 SrcDemandedMask |= fcPosFinite;
3533 if (DemandedMask & fcPosInf)
3534 SrcDemandedMask |= fcPosInf;
3535 if (DemandedMask & fcNegInf)
3536 SrcDemandedMask |= fcNegInf;
3537
3538 KnownFPClass KnownSrc;
3539 if (SimplifyDemandedFPClass(II, 0, SrcDemandedMask, KnownSrc, SQ,
3540 Depth + 1))
3541 return I;
3542
3543 Type *EltTy = VTy->getScalarType();
3544 DenormalMode Mode = F.getDenormalMode(EltTy->getFltSemantics());
3545
3546 Known = KnownFPClass::frexp_mant(KnownSrc, Mode);
3547 Known.setKnownFPClasses(Known.getKnownFPClasses() & DemandedMask);
3548
3549 if (Constant *SingleVal =
3550 getFPClassConstant(VTy, Known.getKnownFPClasses(),
3551 /*IsCanonicalizing=*/true))
3552 return SingleVal;
3553
3554 if (Known.isKnownAlways(fcInf | fcNan))
3555 return II->getArgOperand(0);
3556
3557 return nullptr;
3558 }
3559 default:
3560 break;
3561 }
3562 }
3563 }
3564
3565 KnownFPClass KnownSrc;
3566 if (SimplifyDemandedFPClass(I, 0, DemandedMask, KnownSrc, SQ, Depth + 1))
3567 return I;
3568 Known = KnownSrc;
3569 break;
3570 }
3571 case Instruction::PHI: {
3572 const unsigned PhiRecursionLimit = MaxAnalysisRecursionDepth - 2;
3573 if (Depth >= PhiRecursionLimit)
3574 break;
3575
3577 SimplifyQuery ContextSQ = SQ.getWithoutCondContext();
3578
3579 bool First = true;
3580 bool Changed = false;
3581 for (unsigned I = 0, E = P->getNumIncomingValues(); I != E; ++I) {
3582 // TODO: Better support for self recursive phi
3583 BasicBlock *PredBB = P->getIncomingBlock(I);
3584 const Instruction *CtxI = PredBB->getTerminator();
3585
3586 // Attempt to simplify all incoming edges at a time. If we simplify one
3587 // incoming edge, the phi may fold away, losing information on a later
3588 // visit.
3589 KnownFPClass KnownSrc;
3591 P, P->getOperandNumForIncomingValue(I), DemandedMask, KnownSrc,
3592 ContextSQ.getWithInstruction(CtxI), Depth + 1)) {
3593 // Fixup the other block references to the simplified value.
3594 P->setIncomingValueForBlock(PredBB, P->getIncomingValue(I));
3595 Changed = true;
3596 }
3597
3598 if (First) {
3599 Known = KnownSrc;
3600 First = false;
3601 } else {
3602 Known |= KnownSrc;
3603 }
3604 }
3605
3606 if (Changed)
3607 return P;
3608
3609 Known.knownNot(~DemandedMask);
3610 break;
3611 }
3612 default:
3613 Known = computeKnownFPClass(I, DemandedMask, SQ, Depth + 1);
3614 Known.knownNot(~DemandedMask);
3615 break;
3616 }
3617
3618 return getFPClassConstant(VTy, Known.getKnownFPClasses());
3619}
3620
3621/// Helper routine of SimplifyDemandedUseFPClass. It computes Known
3622/// floating-point classes. It also tries to handle simplifications that can be
3623/// done based on DemandedMask, but without modifying the Instruction.
3625 Instruction *I, FPClassTest DemandedMask, KnownFPClass &Known,
3626 const SimplifyQuery &SQ, unsigned Depth) {
3627 FastMathFlags FMF;
3628 if (auto *FPOp = dyn_cast<FPMathOperator>(I)) {
3629 FMF = FPOp->getFastMathFlags();
3630 DemandedMask = adjustDemandedMaskFromFlags(DemandedMask, FMF);
3631 }
3632
3633 switch (I->getOpcode()) {
3634 case Instruction::Select: {
3635 // TODO: Can we infer which side it came from based on adjusted result
3636 // class?
3637 KnownFPClass KnownRHS =
3638 computeKnownFPClass(I->getOperand(2), DemandedMask, SQ, Depth + 1);
3639 if (KnownRHS.isKnownNever(DemandedMask))
3640 return I->getOperand(1);
3641
3642 KnownFPClass KnownLHS =
3643 computeKnownFPClass(I->getOperand(1), DemandedMask, SQ, Depth + 1);
3644 if (KnownLHS.isKnownNever(DemandedMask))
3645 return I->getOperand(2);
3646
3647 adjustKnownFPClassForSelectArm(KnownLHS, I->getOperand(0), I->getOperand(1),
3648 /*Invert=*/false, SQ, Depth);
3649 adjustKnownFPClassForSelectArm(KnownRHS, I->getOperand(0), I->getOperand(2),
3650 /*Invert=*/true, SQ, Depth);
3651 Known = KnownLHS.intersectWith(KnownRHS);
3652 Known.knownNot(~DemandedMask);
3653 break;
3654 }
3655 case Instruction::FNeg: {
3656 // Special case fneg(fabs(x))
3657 Value *Src;
3658
3659 Value *FNegSrc = I->getOperand(0);
3660 if (!match(FNegSrc, m_FAbs(m_Value(Src)))) {
3661 Known = computeKnownFPClass(I, DemandedMask, SQ, Depth + 1);
3662 break;
3663 }
3664
3665 KnownFPClass KnownSrc = computeKnownFPClass(Src, fcAllFlags, SQ, Depth + 1);
3666
3667 FastMathFlags FabsFMF = cast<FPMathOperator>(FNegSrc)->getFastMathFlags();
3668 FPClassTest ThisDemandedMask =
3669 adjustDemandedMaskFromFlags(DemandedMask, FabsFMF);
3670
3671 // We cannot apply the NSZ logic with multiple uses. We can apply it if the
3672 // inner fabs has it and this is the only use.
3673 if (Value *Simplified = simplifyDemandedFPClassFnegFabs(
3674 Known, Src, ThisDemandedMask, KnownSrc, /*NSZ=*/false))
3675 return Simplified;
3676 break;
3677 }
3678 case Instruction::Call: {
3679 const CallInst *CI = cast<CallInst>(I);
3680 const Intrinsic::ID IID = CI->getIntrinsicID();
3681 switch (IID) {
3682 case Intrinsic::fabs: {
3683 Value *Src = CI->getArgOperand(0);
3684 KnownFPClass KnownSrc =
3686
3687 // NSZ cannot be applied in multiple use case (maybe it could if all uses
3688 // were known nsz)
3689 if (Value *Simplified = simplifyDemandedFPClassFabs(
3690 Known, CI->getArgOperand(0), DemandedMask, KnownSrc,
3691 /*NSZ=*/false))
3692 return Simplified;
3693 break;
3694 }
3695 case Intrinsic::copysign: {
3696 Value *Mag = CI->getArgOperand(0);
3697 Value *Sign = CI->getArgOperand(1);
3698 KnownFPClass KnownMag =
3700
3701 // Rule out some cases by magnitude, which may help prove the sign bit is
3702 // one direction or the other.
3703 KnownMag.knownNot(~llvm::unknown_sign(DemandedMask));
3704
3705 // Cannot use nsz in the multiple use case.
3706 if (Value *Simplified = simplifyDemandedFPClassCopysignMag(
3707 Mag, DemandedMask, KnownMag, /*NSZ=*/false))
3708 return Simplified;
3709
3710 KnownFPClass KnownSign =
3712
3713 if (FMF.noInfs())
3714 KnownSign.knownNot(fcInf);
3715 if (FMF.noNaNs())
3716 KnownSign.knownNot(fcNan);
3717
3718 if (KnownSign.getSignBit() && KnownMag.getSignBit() &&
3719 *KnownSign.getSignBit() == *KnownMag.getSignBit())
3720 return Mag;
3721
3722 Known = KnownFPClass::copysign(KnownMag, KnownSign);
3723 break;
3724 }
3725 case Intrinsic::maxnum:
3726 case Intrinsic::minnum:
3727 case Intrinsic::maximum:
3728 case Intrinsic::minimum:
3729 case Intrinsic::maximumnum:
3730 case Intrinsic::minimumnum: {
3732 DemandedMask, SQ, Depth + 1);
3733 if (KnownRHS.isUnknown())
3734 return nullptr;
3735
3737 DemandedMask, SQ, Depth + 1);
3738
3739 // Cannot use NSZ in the multiple use case.
3740 return simplifyDemandedFPClassMinMax(Known, IID, CI, DemandedMask,
3741 KnownLHS, KnownRHS, F,
3742 /*NSZ=*/false);
3743 }
3744 default:
3745 break;
3746 }
3747
3748 [[fallthrough]];
3749 }
3750 default:
3751 Known = computeKnownFPClass(I, DemandedMask, SQ, Depth + 1);
3752 Known.knownNot(~DemandedMask);
3753 break;
3754 }
3755
3756 return getFPClassConstant(I->getType(), Known.getKnownFPClasses());
3757}
3758
3760 FPClassTest DemandedMask,
3762 const SimplifyQuery &SQ,
3763 unsigned Depth) {
3764 Use &U = I->getOperandUse(OpNo);
3765 Value *V = U.get();
3766 Type *VTy = V->getType();
3767
3768 if (DemandedMask == fcNone) {
3769 if (isa<PoisonValue>(V))
3770 return false;
3772 return true;
3773 }
3774
3775 // Handle constant
3777 if (!VInst) {
3778 // Handle constants and arguments
3780 Known.knownNot(~DemandedMask);
3781
3782 if (Known.getKnownFPClasses() == fcNone) {
3783 if (isa<PoisonValue>(V))
3784 return false;
3786 return true;
3787 }
3788
3789 // Do not try to replace values which are already constants (unless we are
3790 // folding to poison). Doing so could promote poison elements to non-poison
3791 // constants.
3792 if (isa<Constant>(V))
3793 return false;
3794
3795 Value *FoldedToConst = getFPClassConstant(VTy, Known.getKnownFPClasses());
3796 if (!FoldedToConst || FoldedToConst == V)
3797 return false;
3798
3799 replaceUse(U, FoldedToConst);
3800 return true;
3801 }
3802
3804 Known.knownNot(~DemandedMask);
3805 return false;
3806 }
3807
3808 Value *NewVal;
3809
3810 if (VInst->hasOneUse()) {
3811 // If the instruction has one use, we can directly simplify it.
3812 NewVal = SimplifyDemandedUseFPClass(VInst, DemandedMask, Known, SQ, Depth);
3813 } else {
3814 // If there are multiple uses of this instruction, then we can simplify
3815 // VInst to some other value, but not modify the instruction.
3816 NewVal = SimplifyMultipleUseDemandedFPClass(VInst, DemandedMask, Known, SQ,
3817 Depth);
3818 }
3819
3820 if (!NewVal)
3821 return false;
3822 if (Instruction *OpInst = dyn_cast<Instruction>(U))
3823 salvageDebugInfo(*OpInst);
3824
3825 replaceUse(U, NewVal);
3826 return true;
3827}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
AMDGPU Register Bank Select
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
#define DEBUG_TYPE
Hexagon Common GEP
This file provides internal interfaces used to implement the InstCombine.
static Constant * getFPClassConstant(Type *Ty, FPClassTest Mask, bool IsCanonicalizing=false)
For floating-point classes that resolve to a single bit pattern, return that value.
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
static Value * simplifyDemandedFPClassFabs(KnownFPClass &Known, Value *Src, FPClassTest DemandedMask, KnownFPClass KnownSrc, bool NSZ)
Perform multiple-use aware simplfications for fabs(Src).
static Value * simplifyDemandedUseFPClassFPTrunc(InstCombinerImpl &IC, Instruction &I, FastMathFlags FMF, FPClassTest DemandedMask, KnownFPClass &Known, const SimplifyQuery &SQ, unsigned Depth)
static Value * simplifyDemandedFPClassFnegFabs(KnownFPClass &Known, Value *Src, FPClassTest DemandedMask, KnownFPClass KnownSrc, bool NSZ)
Perform multiple-use aware simplfications for fneg(fabs(Src)).
static bool ShrinkDemandedConstant(Instruction *I, unsigned OpNo, const APInt &Demanded)
Check to see if the specified operand of the specified instruction is a constant integer.
static Value * simplifyShiftSelectingPackedElement(Instruction *I, const APInt &DemandedMask, InstCombinerImpl &IC, unsigned Depth)
Let N = 2 * M.
static Value * simplifyDemandedFPClassMinMax(KnownFPClass &Known, Intrinsic::ID IID, const CallInst *CI, FPClassTest DemandedMask, KnownFPClass KnownLHS, KnownFPClass KnownRHS, const Function &F, bool NSZ)
static bool canSkipDemandedEltsInInsertChain(InsertElementInst &IE, unsigned VWidth, unsigned DepthLimit)
Return true if the top-level all-lanes demanded-elements query can be skipped for an intermediate ins...
static Value * simplifyDemandedFPClassCopysignMag(Value *MagSrc, FPClassTest DemandedMask, KnownFPClass KnownSrc, bool NSZ)
static FPClassTest adjustDemandedMaskFromFlags(FPClassTest DemandedMask, FastMathFlags FMF)
static FastMathFlags inferFastMathValueFlags(FastMathFlags FMF, FPClassTest ValidResults, ArrayRef< KnownFPClass > Known)
Try to set an inferred no-nans or no-infs in FMF.
static Value * simplifyDemandedFPClassResult(Instruction *FPOp, FastMathFlags FMF, FPClassTest DemandedMask, KnownFPClass &Known, ArrayRef< KnownFPClass > KnownSrcs)
Apply epilog fixups to a floating-point intrinsic.
This file provides the interface for the instcombine pass implementation.
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
uint64_t IntrinsicInst * II
#define P(N)
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
This file implements the SmallBitVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static unsigned getBitWidth(Type *Ty, const DataLayout &DL)
Returns the bitwidth of the given scalar or pointer type.
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
void clearBit(unsigned BitPosition)
Set a given bit to 0.
Definition APInt.h:1426
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:225
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
void setHighBits(unsigned hiBits)
Set the top hiBits bits.
Definition APInt.h:1411
unsigned popcount() const
Count the number of bits set.
Definition APInt.h:1690
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1532
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
Definition APInt.h:367
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1695
void setSignBit()
Set the sign bit to 1.
Definition APInt.h:1360
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1115
void clearAllBits()
Set every bit to 0.
Definition APInt.h:1416
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1618
void clearLowBits(unsigned loBits)
Set bottom loBits bits to 0.
Definition APInt.h:1455
uint64_t getLimitedValue(uint64_t Limit=UINT64_MAX) const
If this value is smaller than the specified limit, return it, otherwise return the limit value.
Definition APInt.h:471
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:829
APInt shl(unsigned shiftAmt) const
Left-shift function.
Definition APInt.h:875
bool isSubsetOf(const APInt &RHS) const
This operation checks that all bits set in this APInt are also set in RHS.
Definition APInt.h:1261
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
Definition APInt.h:436
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
bool isIntN(unsigned N) const
Check if this APInt has an N-bits unsigned integer value.
Definition APInt.h:428
bool isOne() const
Determine if this is a value of 1.
Definition APInt.h:385
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1225
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
LLVM Basic Block Representation.
Definition BasicBlock.h:62
const Instruction * getTerminator() const LLVM_READONLY
Returns the terminator instruction; assumes that the block is well-formed.
Definition BasicBlock.h:237
BinaryOps getOpcode() const
Definition InstrTypes.h:409
Value * getArgOperand(unsigned i) const
LLVM_ABI Intrinsic::ID getIntrinsicID() const
Returns the intrinsic ID of the intrinsic called or Intrinsic::not_intrinsic if the called function i...
This class represents a function call, abstracting a target machine's calling convention.
static CallInst * Create(FunctionType *Ty, Value *F, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
This is the base class for all instructions that perform data casts.
Definition InstrTypes.h:512
@ FCMP_UEQ
1 0 0 1 True if unordered or equal
Definition InstrTypes.h:751
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getQNaN(Type *Ty, bool Negative=false, APInt *Payload=nullptr)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
static LLVM_ABI Constant * get(ArrayRef< Constant * > V)
This is an important base class in LLVM.
Definition Constant.h:43
static LLVM_ABI Constant * getIntegerValue(Type *Ty, const APInt &V)
Return the value for an integer or pointer constant, or a vector thereof, with the given scalar value...
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
Definition Constant.h:64
static LLVM_ABI Constant * getAllOnesValue(Type *Ty)
LLVM_ABI bool isOneValue() const
Returns true if the value is one.
Definition Constants.cpp:89
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
LLVM_ABI Constant * getAggregateElement(unsigned Elt) const
For aggregates (struct/array/vector) return the constant that corresponds to the specified element if...
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Convenience struct for specifying and reasoning about fast-math flags.
Definition FMF.h:23
bool noSignedZeros() const
Definition FMF.h:67
bool noInfs() const
Definition FMF.h:66
void setNoSignedZeros(bool B=true)
Definition FMF.h:84
void setNoNaNs(bool B=true)
Definition FMF.h:78
bool noNaNs() const
Definition FMF.h:65
void setNoInfs(bool B=true)
Definition FMF.h:81
an instruction for type-safe pointer arithmetic to access elements of arrays and structs
Value * CreateICmpEQ(Value *LHS, Value *RHS, const Twine &Name="")
Definition IRBuilder.h:2374
LLVM_ABI Value * CreateSelectWithUnknownProfile(Value *C, Value *True, Value *False, StringRef PassName, const Twine &Name="")
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Definition IRBuilder.h:179
This instruction inserts a single (scalar) element into a VectorType value.
static InsertElementInst * Create(Value *Vec, Value *NewElt, Value *Idx, const Twine &NameStr="", InsertPosition InsertBefore=nullptr)
bool SimplifyDemandedInstructionFPClass(Instruction &Inst)
Value * SimplifyDemandedVectorElts(Value *V, APInt DemandedElts, APInt &PoisonElts, unsigned Depth=0, bool AllowMultipleUsers=false) override
The specified value produces a vector with any number of elements.
Value * SimplifyDemandedUseFPClass(Instruction *I, FPClassTest DemandedMask, KnownFPClass &Known, const SimplifyQuery &Q, unsigned Depth=0)
Attempts to replace V with a simpler value based on the demanded floating-point classes.
bool SimplifyDemandedBits(Instruction *I, unsigned Op, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0) override
This form of SimplifyDemandedBits simplifies the specified instruction operand if possible,...
std::optional< std::pair< Intrinsic::ID, SmallVector< Value *, 3 > > > convertOrOfShiftsToFunnelShift(Instruction &Or)
Value * SimplifyMultipleUseDemandedFPClass(Instruction *I, FPClassTest DemandedMask, KnownFPClass &Known, const SimplifyQuery &Q, unsigned Depth)
Helper routine of SimplifyDemandedUseFPClass.
const InstCombineCLOptions & CLOpts
Value * simplifyShrShlDemandedBits(Instruction *Shr, const APInt &ShrOp1, Instruction *Shl, const APInt &ShlOp1, const APInt &DemandedMask, KnownBits &Known)
Helper routine of SimplifyDemandedUseBits.
bool SimplifyDemandedFPClass(Instruction *I, unsigned Op, FPClassTest DemandedMask, KnownFPClass &Known, const SimplifyQuery &Q, unsigned Depth=0)
Value * SimplifyDemandedUseBits(Instruction *I, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0)
Attempts to replace I with a simpler value based on the demanded bits.
bool SimplifyDemandedInstructionBits(Instruction &Inst)
Tries to simplify operands to an integer instruction based on its demanded bits.
Value * SimplifyMultipleUseDemandedBits(Instruction *I, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0)
Helper routine of SimplifyDemandedUseBits.
SimplifyQuery SQ
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
void replaceUse(Use &U, Value *NewValue)
Replace use and add the previously used value to the worklist.
InstructionWorklist & Worklist
A worklist of the instructions that need to be simplified.
Instruction * InsertNewInstWith(Instruction *New, BasicBlock::iterator Old)
Same as InsertNewInstBefore, but also sets the debug loc.
const DataLayout & DL
unsigned ComputeNumSignBits(const Value *Op, const Instruction *CtxI=nullptr, unsigned Depth=0) const
LLVM_ABI std::optional< Value * > targetSimplifyDemandedVectorEltsIntrinsic(IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp)
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
DominatorTree & DT
LLVM_ABI std::optional< Value * > targetSimplifyDemandedUseBitsIntrinsic(IntrinsicInst &II, APInt DemandedMask, KnownBits &Known, bool &KnownBitsComputed)
void computeKnownBits(const Value *V, KnownBits &Known, const Instruction *CtxI, unsigned Depth=0) const
LLVM_ABI void dropUBImplyingAttrsAndMetadata(ArrayRef< unsigned > Keep={})
Drop any attributes or metadata that can cause immediate undefined behavior.
LLVM_ABI bool hasNoUnsignedWrap() const LLVM_READONLY
Determine whether the no unsigned wrap flag is set.
LLVM_ABI bool hasNoSignedWrap() const LLVM_READONLY
Determine whether the no signed wrap flag is set.
LLVM_ABI bool isCommutative() const LLVM_READONLY
Return true if the instruction is commutative:
LLVM_ABI void setFastMathFlags(FastMathFlags FMF)
Convenience function for setting multiple fast-math flags on this instruction, which must be an opera...
unsigned getOpcode() const
Returns a member of one of the enums like Instruction::Add.
LLVM_ABI void setIsExact(bool b=true)
Set or clear the exact flag on this instruction, which must be an operator which supports this flag.
bool isShift() const
bool isIntDivRem() const
A wrapper class for inspecting calls to intrinsic functions.
bool hasNoSignedWrap() const
Test whether this operation is known to never undergo signed overflow, aka the nsw property.
Definition Operator.h:113
bool hasNoUnsignedWrap() const
Test whether this operation is known to never undergo unsigned overflow, aka the nuw property.
Definition Operator.h:107
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
This class represents the LLVM 'select' instruction.
const Value * getCondition() const
This is a 'bitvector' (really, a variable-sized bit array), optimized for the case when the array is ...
SmallBitVector & set()
bool test(unsigned Idx) const
Returns true if bit Idx is set.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:300
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
bool isIntOrIntVectorTy() const
Return true if this is an integer type or a vector of integer types.
Definition Type.h:258
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isMultiUnitFPType() const
Returns true if this is a floating-point type that is an unevaluated sum of multiple floating-point u...
Definition Type.h:195
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isIEEELikeFPTy() const
Return true if this is a well-behaved IEEE-like type, which has a IEEE compatible layout,...
Definition Type.h:172
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:96
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
void setOperand(unsigned i, Value *Val)
Definition User.h:212
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
iterator_range< user_iterator > users()
Definition Value.h:428
bool hasUseList() const
Check if this Value has a use-list.
Definition Value.h:346
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
Definition Value.cpp:400
Base class of all SIMD vector types.
This class represents zero extension of integer types.
self_iterator getIterator()
Definition ilist_node.h:123
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
BinaryOp_match< SrcTy, SpecificConstantMatch, TargetOpcode::G_XOR, true > m_Not(const SrcTy &&Src)
Matches a register not-ed by a G_XOR.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
cst_pred_ty< is_lowbit_mask > m_LowBitMask()
Match an integer or vector with only the low bit(s) set.
PtrAdd_match< PointerOpTy, OffsetOpTy > m_PtrAdd(const PointerOpTy &PointerOp, const OffsetOpTy &OffsetOp)
Matches GEP with i8 source element type.
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::AShr > m_AShr(const LHS &L, const RHS &R)
auto m_Poison()
Match an arbitrary poison constant.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
specific_intval< false > m_SpecificInt(const APInt &V)
Match a specific integer value or vector with all elements equal to the value.
bool match(Val *V, const Pattern &P)
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
BinOpPred_match< LHS, RHS, is_right_shift_op > m_Shr(const LHS &L, const RHS &R)
Matches logical shift operations.
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
auto m_BinOp()
Match an arbitrary binary operation and ignore it.
ExtractValue_match< Ind, Val_t > m_ExtractValue(const Val_t &V)
Match a single index ExtractValue instruction.
auto m_Value()
Match an arbitrary value and ignore it.
auto m_Ctpop(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::Mul > m_Mul(const LHS &L, const RHS &R)
TwoOps_match< V1_t, V2_t, Instruction::ShuffleVector > m_Shuffle(const V1_t &v1, const V2_t &v2)
Matches ShuffleVectorInst independently of mask value.
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
match_immconstant_ty m_ImmConstant()
Match an arbitrary immediate Constant and ignore it.
DisjointOr_match< LHS, RHS, true > m_c_DisjointOr(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Add, true > m_c_Add(const LHS &L, const RHS &R)
Matches a Add with LHS and RHS in either order.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
auto m_FAbs(const Opnd0 &Op0)
AnyBinaryOp_match< LHS, RHS, true > m_c_BinOp(const LHS &L, const RHS &R)
Matches a BinaryOperator with LHS and RHS in either order.
BinaryOp_match< LHS, RHS, Instruction::LShr > m_LShr(const LHS &L, const RHS &R)
CmpClass_match< LHS, RHS, ICmpInst > m_ICmp(CmpPredicate &Pred, const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
auto m_Undef()
Match an arbitrary undef constant.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI bool haveNoCommonBitsSet(const WithCache< const Value * > &LHSCache, const WithCache< const Value * > &RHSCache, const SimplifyQuery &SQ)
Return true if LHS and RHS have no common bits set.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
LLVM_ABI void computeKnownBitsFromContext(const Value *V, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0)
Merge bits known from context-dependent facts into Known.
@ Known
Known to have no common set bits.
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
int countr_one(T Value)
Count the number of ones from the least significant bit to the first zero bit.
Definition bit.h:315
LLVM_ABI void salvageDebugInfo(const MachineRegisterInfo &MRI, MachineInstr &MI)
Assuming the instruction MI is going to be deleted, attempt to salvage debug users of MI by writing t...
Definition Utils.cpp:1676
LLVM_ABI void computeKnownBits(const Value *V, KnownBits &Known, const DataLayout &DL, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, bool UseInstrInfo=true, unsigned Depth=0)
Determine which bits of V are known to be either zero or one and return them in the KnownZero/KnownOn...
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
gep_type_iterator gep_type_end(const User *GEP)
constexpr auto equal_to(T &&Arg)
Functor variant of std::equal_to that can be used as a UnaryPredicate in functional algorithms like a...
Definition STLExtras.h:2189
LLVM_ABI bool isGuaranteedNotToBeUndef(const Value *V, AssumptionCache *AC=nullptr, const Instruction *CtxI=nullptr, const DominatorTree *DT=nullptr, unsigned Depth=0)
Returns true if V cannot be undef, but may be poison.
LLVM_ABI bool cannotOrderStrictlyLess(FPClassTest LHS, FPClassTest RHS, bool OrderedZeroSign=false)
Returns true if all values in LHS must be greater than or equal to those in RHS.
LLVM_ABI bool cannotOrderStrictlyGreater(FPClassTest LHS, FPClassTest RHS, bool OrderedZeroSign=false)
Returns true if all values in LHS must be less than or equal to those in RHS.
constexpr unsigned MaxAnalysisRecursionDepth
LLVM_ABI void adjustKnownBitsForSelectArm(KnownBits &Known, Value *Cond, Value *Arm, bool Invert, const SimplifyQuery &Q, unsigned Depth=0)
Adjust Known for the given select Arm to include information from the select Cond.
LLVM_ABI FPClassTest fneg(FPClassTest Mask)
Return the test mask which returns true if the value's sign bit is flipped.
FPClassTest
Floating-point class tests, supported by 'is_fpclass' intrinsic.
LLVM_ABI void adjustKnownFPClassForSelectArm(KnownFPClass &Known, Value *Cond, Value *Arm, bool Invert, const SimplifyQuery &Q, unsigned Depth=0)
Adjust Known for the given select Arm to include information from the select Cond.
LLVM_ABI FPClassTest inverse_fabs(FPClassTest Mask)
Return the test mask which returns true after fabs is applied to the value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI Constant * ConstantFoldBinaryOpOperands(unsigned Opcode, Constant *LHS, Constant *RHS, const DataLayout &DL)
Attempt to constant fold a binary operation with the specified operands.
constexpr int PoisonMaskElem
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ Mul
Product of integers.
@ Xor
Bitwise or logical XOR of integers.
LLVM_ABI FPClassTest unknown_sign(FPClassTest Mask)
Return the test mask which returns true if the value could have the same set of classes,...
DWARFExpression::Operation Op
constexpr unsigned BitWidth
LLVM_ABI KnownBits analyzeKnownBitsFromAndXorOr(const Operator *I, const KnownBits &KnownLHS, const KnownBits &KnownRHS, const SimplifyQuery &SQ, unsigned Depth=0)
Using KnownBits LHS/RHS produce the known bits for logic op (and/xor/or).
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
gep_type_iterator gep_type_begin(const User *GEP)
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Represent subnormal handling kind for floating point instruction inputs and outputs.
static constexpr DenormalMode getPreserveSign()
static constexpr DenormalMode getIEEE()
static KnownBits makeConstant(const APInt &C)
Create known bits from a known constant.
Definition KnownBits.h:315
KnownBits anyextOrTrunc(unsigned BitWidth) const
Return known bits for an "any" extension or truncation of the value we're tracking.
Definition KnownBits.h:190
bool isNonNegative() const
Returns true if this value is known to be non-negative.
Definition KnownBits.h:106
void makeNonNegative()
Make this value non-negative.
Definition KnownBits.h:125
static LLVM_ABI KnownBits ashr(const KnownBits &LHS, const KnownBits &RHS, bool ShAmtNonZero=false, bool Exact=false)
Compute known bits for ashr(LHS, RHS).
unsigned getBitWidth() const
Get the bit width of this value.
Definition KnownBits.h:44
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
Definition KnownBits.h:361
KnownBits sext(unsigned BitWidth) const
Return known bits for a sign extension of the value we're tracking.
Definition KnownBits.h:184
KnownBits zextOrTrunc(unsigned BitWidth) const
Return known bits for a zero extension or truncation of the value we're tracking.
Definition KnownBits.h:200
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
Definition KnownBits.h:146
static LLVM_ABI KnownBits srem(const KnownBits &LHS, const KnownBits &RHS)
Compute known bits for srem(LHS, RHS).
static LLVM_ABI KnownBits udiv(const KnownBits &LHS, const KnownBits &RHS, bool Exact=false)
Compute known bits for udiv(LHS, RHS).
bool isNegative() const
Returns true if this value is known to be negative.
Definition KnownBits.h:103
static KnownBits sub(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false)
Compute knownbits resulting from subtraction of LHS and RHS.
Definition KnownBits.h:376
static LLVM_ABI KnownBits shl(const KnownBits &LHS, const KnownBits &RHS, bool NUW=false, bool NSW=false, bool ShAmtNonZero=false)
Compute known bits for shl(LHS, RHS).
bool isKnownNeverInfOrNaN() const
Return true if it's known this can never be an infinity or nan.
bool isKnownNeverInfinity() const
Return true if it's known this can never be an infinity.
static constexpr FPClassTest OrderedGreaterThanZeroMask
static constexpr FPClassTest OrderedLessThanZeroMask
void knownNot(FPClassTest RuleOut)
static LLVM_ABI KnownFPClass fmul(const KnownFPClass &LHS, const KnownFPClass &RHS, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fmul.
static LLVM_ABI KnownFPClass fadd_self(const KnownFPClass &Src, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fadd x, x.
void copysign(const KnownFPClass &Sign)
static KnownFPClass square(const KnownFPClass &Src, DenormalMode Mode=DenormalMode::getDynamic())
static LLVM_ABI KnownFPClass fsub(const KnownFPClass &LHS, const KnownFPClass &RHS, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fsub.
bool isKnownNeverSubnormal() const
Return true if it's known this can never be a subnormal.
bool isKnownAlways(FPClassTest Mask) const
static LLVM_ABI KnownFPClass canonicalize(const KnownFPClass &Src, DenormalMode DenormMode=DenormalMode::getDynamic())
Apply the canonicalize intrinsic to this value.
LLVM_ABI bool isKnownNeverLogicalZero(DenormalMode Mode) const
Return true if it's known this can never be interpreted as a zero.
static LLVM_ABI KnownFPClass log(const KnownFPClass &Src, DenormalMode Mode=DenormalMode::getDynamic())
Propagate known class for log/log2/log10.
static LLVM_ABI KnownFPClass fdiv(const KnownFPClass &LHS, const KnownFPClass &RHS, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fdiv.
static LLVM_ABI KnownFPClass roundToIntegral(const KnownFPClass &Src, bool IsTrunc, bool IsMultiUnitFPType)
Propagate known class for rounding intrinsics (trunc, floor, ceil, rint, nearbyint,...
static LLVM_ABI KnownFPClass minMaxLike(const KnownFPClass &LHS, const KnownFPClass &RHS, MinMaxKind Kind, DenormalMode DenormMode=DenormalMode::getDynamic())
bool isUnknown() const
KnownFPClass intersectWith(const KnownFPClass &RHS) const
static LLVM_ABI KnownFPClass exp(const KnownFPClass &Src)
Report known values for exp, exp2 and exp10.
static LLVM_ABI KnownFPClass frexp_mant(const KnownFPClass &Src, DenormalMode Mode=DenormalMode::getDynamic())
Propagate known class for mantissa component of frexp.
bool isKnownNeverNaN() const
Return true if it's known this can never be a nan.
bool isKnownNever(FPClassTest Mask) const
Return true if it's known this can never be one of the mask entries.
std::optional< bool > getSignBit() const
std::nullopt if the sign bit is unknown, true if the sign bit is definitely set or false if the sign ...
static LLVM_ABI KnownFPClass fpext(const KnownFPClass &KnownSrc, const fltSemantics &DstTy, const fltSemantics &SrcTy)
Propagate known class for fpext.
FPClassTest getKnownFPClasses() const
Floating-point classes the value could be one of.
static LLVM_ABI KnownFPClass fma(const KnownFPClass &LHS, const KnownFPClass &RHS, const KnownFPClass &Addend, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fma.
static LLVM_ABI KnownFPClass fptrunc(const KnownFPClass &KnownSrc)
Propagate known class for fptrunc.
static LLVM_ABI KnownFPClass sqrt(const KnownFPClass &Src, DenormalMode Mode=DenormalMode::getDynamic())
Propagate known class for sqrt.
LLVM_ABI bool isKnownNeverLogicalPosZero(DenormalMode Mode) const
Return true if it's known this can never be interpreted as a positive zero.
bool cannotBeOrderedGreaterEqZero(DenormalMode Mode) const
Return true if it's know this can never be a negative value or a logical 0.
static LLVM_ABI KnownFPClass fadd(const KnownFPClass &LHS, const KnownFPClass &RHS, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fadd.
LLVM_ABI bool isKnownNeverLogicalNegZero(DenormalMode Mode) const
Return true if it's known this can never be interpreted as a negative zero.
static LLVM_ABI KnownFPClass fma_square(const KnownFPClass &Squared, const KnownFPClass &Addend, DenormalMode Mode=DenormalMode::getDynamic())
Report known values for fma squared, squared, addend.
static LLVM_ABI KnownFPClass ldexp(const KnownFPClass &Src, const APInt &ConstantRangeMin, const APInt &ConstantRangeMax, const fltSemantics &Flt, DenormalMode Mode=DenormalMode::getDynamic())
Propagate known class for ldexp, assuming the exponent is known to be within [ConstantRangeMin,...
Matching combinators.
SimplifyQuery getWithInstruction(const Instruction *I) const
const Instruction * CtxI