2010-01-04 15:53:58 +08:00
|
|
|
//===- InstCombineCasts.cpp -----------------------------------------------===//
|
|
|
|
//
|
2019-01-19 16:50:56 +08:00
|
|
|
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
|
|
|
// See https://llvm.org/LICENSE.txt for license information.
|
|
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
2010-01-04 15:53:58 +08:00
|
|
|
//
|
|
|
|
//===----------------------------------------------------------------------===//
|
|
|
|
//
|
|
|
|
// This file implements the visit functions for cast operations.
|
|
|
|
//
|
|
|
|
//===----------------------------------------------------------------------===//
|
|
|
|
|
2015-01-22 13:25:13 +08:00
|
|
|
#include "InstCombineInternal.h"
|
2016-10-26 04:43:42 +08:00
|
|
|
#include "llvm/ADT/SetVector.h"
|
2011-07-21 05:57:23 +08:00
|
|
|
#include "llvm/Analysis/ConstantFolding.h"
|
2017-04-27 00:39:58 +08:00
|
|
|
#include "llvm/Analysis/TargetLibraryInfo.h"
|
2013-01-02 19:36:10 +08:00
|
|
|
#include "llvm/IR/DataLayout.h"
|
2018-01-27 06:02:52 +08:00
|
|
|
#include "llvm/IR/DIBuilder.h"
|
2014-03-04 19:08:18 +08:00
|
|
|
#include "llvm/IR/PatternMatch.h"
|
2017-04-27 00:39:58 +08:00
|
|
|
#include "llvm/Support/KnownBits.h"
|
2019-11-29 06:18:28 +08:00
|
|
|
#include <numeric>
|
2010-01-04 15:53:58 +08:00
|
|
|
using namespace llvm;
|
|
|
|
using namespace PatternMatch;
|
|
|
|
|
2014-04-22 10:55:47 +08:00
|
|
|
#define DEBUG_TYPE "instcombine"
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Analyze 'Val', seeing if it is a simple linear expression.
|
|
|
|
/// If so, decompose it, returning some value X, such that Val is
|
2010-01-04 15:59:07 +08:00
|
|
|
/// X*Scale+Offset.
|
|
|
|
///
|
2015-09-09 22:54:29 +08:00
|
|
|
static Value *decomposeSimpleLinearExpr(Value *Val, unsigned &Scale,
|
2010-05-28 12:33:04 +08:00
|
|
|
uint64_t &Offset) {
|
2010-01-04 15:59:07 +08:00
|
|
|
if (ConstantInt *CI = dyn_cast<ConstantInt>(Val)) {
|
|
|
|
Offset = CI->getZExtValue();
|
|
|
|
Scale = 0;
|
2010-05-28 12:33:04 +08:00
|
|
|
return ConstantInt::get(Val->getType(), 0);
|
2010-01-06 04:57:30 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-06 04:57:30 +08:00
|
|
|
if (BinaryOperator *I = dyn_cast<BinaryOperator>(Val)) {
|
2011-07-09 06:09:33 +08:00
|
|
|
// Cannot look past anything that might overflow.
|
|
|
|
OverflowingBinaryOperator *OBI = dyn_cast<OverflowingBinaryOperator>(Val);
|
2012-05-05 15:09:40 +08:00
|
|
|
if (OBI && !OBI->hasNoUnsignedWrap() && !OBI->hasNoSignedWrap()) {
|
2011-07-09 06:09:33 +08:00
|
|
|
Scale = 1;
|
|
|
|
Offset = 0;
|
|
|
|
return Val;
|
|
|
|
}
|
|
|
|
|
2010-01-04 15:59:07 +08:00
|
|
|
if (ConstantInt *RHS = dyn_cast<ConstantInt>(I->getOperand(1))) {
|
|
|
|
if (I->getOpcode() == Instruction::Shl) {
|
|
|
|
// This is a value scaled by '1 << the shift amt'.
|
2010-05-28 12:33:04 +08:00
|
|
|
Scale = UINT64_C(1) << RHS->getZExtValue();
|
2010-01-04 15:59:07 +08:00
|
|
|
Offset = 0;
|
|
|
|
return I->getOperand(0);
|
2010-01-06 04:57:30 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-06 04:57:30 +08:00
|
|
|
if (I->getOpcode() == Instruction::Mul) {
|
2010-01-04 15:59:07 +08:00
|
|
|
// This value is scaled by 'RHS'.
|
|
|
|
Scale = RHS->getZExtValue();
|
|
|
|
Offset = 0;
|
|
|
|
return I->getOperand(0);
|
2010-01-06 04:57:30 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-06 04:57:30 +08:00
|
|
|
if (I->getOpcode() == Instruction::Add) {
|
2013-01-24 13:22:40 +08:00
|
|
|
// We have X+C. Check to see if we really have (X*C2)+C1,
|
2010-01-04 15:59:07 +08:00
|
|
|
// where C1 is divisible by C2.
|
|
|
|
unsigned SubScale;
|
2013-01-24 13:22:40 +08:00
|
|
|
Value *SubVal =
|
2015-09-09 22:54:29 +08:00
|
|
|
decomposeSimpleLinearExpr(I->getOperand(0), SubScale, Offset);
|
2010-01-04 15:59:07 +08:00
|
|
|
Offset += RHS->getZExtValue();
|
|
|
|
Scale = SubScale;
|
|
|
|
return SubVal;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
// Otherwise, we can't look past this.
|
|
|
|
Scale = 1;
|
|
|
|
Offset = 0;
|
|
|
|
return Val;
|
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// If we find a cast of an allocation instruction, try to eliminate the cast by
|
|
|
|
/// moving the type information into the alloc.
|
2010-01-04 15:59:07 +08:00
|
|
|
Instruction *InstCombiner::PromoteCastOfAllocation(BitCastInst &CI,
|
|
|
|
AllocaInst &AI) {
|
2011-07-18 12:54:35 +08:00
|
|
|
PointerType *PTy = cast<PointerType>(CI.getType());
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-02-17 01:02:29 +08:00
|
|
|
IRBuilderBase::InsertPointGuard Guard(Builder);
|
|
|
|
Builder.SetInsertPoint(&AI);
|
2010-01-04 15:59:07 +08:00
|
|
|
|
|
|
|
// Get the type really allocated and the type casted to.
|
2011-07-18 12:54:35 +08:00
|
|
|
Type *AllocElTy = AI.getAllocatedType();
|
|
|
|
Type *CastElTy = PTy->getElementType();
|
2014-04-25 13:29:35 +08:00
|
|
|
if (!AllocElTy->isSized() || !CastElTy->isSized()) return nullptr;
|
2010-01-04 15:59:07 +08:00
|
|
|
|
2015-03-10 10:37:25 +08:00
|
|
|
unsigned AllocElTyAlign = DL.getABITypeAlignment(AllocElTy);
|
|
|
|
unsigned CastElTyAlign = DL.getABITypeAlignment(CastElTy);
|
2014-04-25 13:29:35 +08:00
|
|
|
if (CastElTyAlign < AllocElTyAlign) return nullptr;
|
2010-01-04 15:59:07 +08:00
|
|
|
|
|
|
|
// If the allocation has multiple uses, only promote it if we are strictly
|
|
|
|
// increasing the alignment of the resultant allocation. If we keep it the
|
2011-03-09 06:12:11 +08:00
|
|
|
// same, we open the door to infinite loops of various kinds.
|
2014-04-25 13:29:35 +08:00
|
|
|
if (!AI.hasOneUse() && CastElTyAlign == AllocElTyAlign) return nullptr;
|
2010-01-04 15:59:07 +08:00
|
|
|
|
2015-03-10 10:37:25 +08:00
|
|
|
uint64_t AllocElTySize = DL.getTypeAllocSize(AllocElTy);
|
|
|
|
uint64_t CastElTySize = DL.getTypeAllocSize(CastElTy);
|
2014-04-25 13:29:35 +08:00
|
|
|
if (CastElTySize == 0 || AllocElTySize == 0) return nullptr;
|
2010-01-04 15:59:07 +08:00
|
|
|
|
2013-03-06 13:44:53 +08:00
|
|
|
// If the allocation has multiple uses, only promote it if we're not
|
|
|
|
// shrinking the amount of memory being allocated.
|
2015-03-10 10:37:25 +08:00
|
|
|
uint64_t AllocElTyStoreSize = DL.getTypeStoreSize(AllocElTy);
|
|
|
|
uint64_t CastElTyStoreSize = DL.getTypeStoreSize(CastElTy);
|
2014-04-25 13:29:35 +08:00
|
|
|
if (!AI.hasOneUse() && CastElTyStoreSize < AllocElTyStoreSize) return nullptr;
|
2013-03-06 13:44:53 +08:00
|
|
|
|
2010-01-04 15:59:07 +08:00
|
|
|
// See if we can satisfy the modulus by pulling a scale out of the array
|
|
|
|
// size argument.
|
|
|
|
unsigned ArraySizeScale;
|
2010-05-28 12:33:04 +08:00
|
|
|
uint64_t ArrayOffset;
|
2010-01-04 15:59:07 +08:00
|
|
|
Value *NumElements = // See if the array size is a decomposable linear expr.
|
2015-09-09 22:54:29 +08:00
|
|
|
decomposeSimpleLinearExpr(AI.getOperand(0), ArraySizeScale, ArrayOffset);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:59:07 +08:00
|
|
|
// If we can now satisfy the modulus, by using a non-1 scale, we really can
|
|
|
|
// do the xform.
|
|
|
|
if ((AllocElTySize*ArraySizeScale) % CastElTySize != 0 ||
|
2014-04-25 13:29:35 +08:00
|
|
|
(AllocElTySize*ArrayOffset ) % CastElTySize != 0) return nullptr;
|
2010-01-04 15:59:07 +08:00
|
|
|
|
|
|
|
unsigned Scale = (AllocElTySize*ArraySizeScale)/CastElTySize;
|
2014-04-25 13:29:35 +08:00
|
|
|
Value *Amt = nullptr;
|
2010-01-04 15:59:07 +08:00
|
|
|
if (Scale == 1) {
|
|
|
|
Amt = NumElements;
|
|
|
|
} else {
|
2010-05-28 12:33:04 +08:00
|
|
|
Amt = ConstantInt::get(AI.getArraySize()->getType(), Scale);
|
2010-01-04 15:59:07 +08:00
|
|
|
// Insert before the alloca, not before the cast.
|
2020-02-17 01:02:29 +08:00
|
|
|
Amt = Builder.CreateMul(Amt, NumElements);
|
2010-01-04 15:59:07 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-05-28 12:33:04 +08:00
|
|
|
if (uint64_t Offset = (AllocElTySize*ArrayOffset)/CastElTySize) {
|
|
|
|
Value *Off = ConstantInt::get(AI.getArraySize()->getType(),
|
2010-01-04 15:59:07 +08:00
|
|
|
Offset, true);
|
2020-02-17 01:02:29 +08:00
|
|
|
Amt = Builder.CreateAdd(Amt, Off);
|
2010-01-04 15:59:07 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-02-17 01:02:29 +08:00
|
|
|
AllocaInst *New = Builder.CreateAlloca(CastElTy, Amt);
|
2020-05-16 04:23:14 +08:00
|
|
|
New->setAlignment(AI.getAlign());
|
2010-01-04 15:59:07 +08:00
|
|
|
New->takeName(&AI);
|
2014-04-29 01:40:03 +08:00
|
|
|
New->setUsedWithInAlloca(AI.isUsedWithInAlloca());
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:59:07 +08:00
|
|
|
// If the allocation has multiple real uses, insert a cast and change all
|
|
|
|
// things that used it to use the new cast. This will also hack on CI, but it
|
|
|
|
// will die soon.
|
2011-03-09 06:12:11 +08:00
|
|
|
if (!AI.hasOneUse()) {
|
2010-01-04 15:59:07 +08:00
|
|
|
// New is the allocation instruction, pointer typed. AI is the original
|
|
|
|
// allocation instruction, also pointer typed. Thus, cast to use is BitCast.
|
2020-02-17 01:02:29 +08:00
|
|
|
Value *NewCast = Builder.CreateBitCast(New, AI.getType(), "tmpcast");
|
2016-02-02 06:23:39 +08:00
|
|
|
replaceInstUsesWith(AI, NewCast);
|
2020-03-31 02:58:58 +08:00
|
|
|
eraseInstFromFunction(AI);
|
2010-01-04 15:59:07 +08:00
|
|
|
}
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, New);
|
2010-01-04 15:59:07 +08:00
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Given an expression that CanEvaluateTruncated or CanEvaluateSExtd returns
|
|
|
|
/// true for, actually insert the code to evaluate the expression.
|
2013-01-24 13:22:40 +08:00
|
|
|
Value *InstCombiner::EvaluateInDifferentType(Value *V, Type *Ty,
|
2010-01-04 15:54:59 +08:00
|
|
|
bool isSigned) {
|
2010-01-09 03:28:47 +08:00
|
|
|
if (Constant *C = dyn_cast<Constant>(V)) {
|
|
|
|
C = ConstantExpr::getIntegerCast(C, Ty, isSigned /*Sext or ZExt*/);
|
2014-02-21 08:06:31 +08:00
|
|
|
// If we got a constantexpr back, try to simplify it with DL info.
|
2020-03-04 02:16:11 +08:00
|
|
|
return ConstantFoldConstant(C, DL, &TLI);
|
2010-01-09 03:28:47 +08:00
|
|
|
}
|
2010-01-04 15:54:59 +08:00
|
|
|
|
|
|
|
// Otherwise, it must be an instruction.
|
|
|
|
Instruction *I = cast<Instruction>(V);
|
2014-04-25 13:29:35 +08:00
|
|
|
Instruction *Res = nullptr;
|
2010-01-04 15:54:59 +08:00
|
|
|
unsigned Opc = I->getOpcode();
|
|
|
|
switch (Opc) {
|
|
|
|
case Instruction::Add:
|
|
|
|
case Instruction::Sub:
|
|
|
|
case Instruction::Mul:
|
|
|
|
case Instruction::And:
|
|
|
|
case Instruction::Or:
|
|
|
|
case Instruction::Xor:
|
|
|
|
case Instruction::AShr:
|
|
|
|
case Instruction::LShr:
|
|
|
|
case Instruction::Shl:
|
|
|
|
case Instruction::UDiv:
|
|
|
|
case Instruction::URem: {
|
[InstCombine] don't try to evaluate instructions with >1 use (revert r324014)
This example causes a compile-time explosion:
define i16 @foo(i16 %in) {
%x = zext i16 %in to i32
%a1 = mul i32 %x, %x
%a2 = mul i32 %a1, %a1
%a3 = mul i32 %a2, %a2
%a4 = mul i32 %a3, %a3
%a5 = mul i32 %a4, %a4
%a6 = mul i32 %a5, %a5
%a7 = mul i32 %a6, %a6
%a8 = mul i32 %a7, %a7
%a9 = mul i32 %a8, %a8
%a10 = mul i32 %a9, %a9
%a11 = mul i32 %a10, %a10
%a12 = mul i32 %a11, %a11
%a13 = mul i32 %a12, %a12
%a14 = mul i32 %a13, %a13
%a15 = mul i32 %a14, %a14
%a16 = mul i32 %a15, %a15
%a17 = mul i32 %a16, %a16
%a18 = mul i32 %a17, %a17
%a19 = mul i32 %a18, %a18
%a20 = mul i32 %a19, %a19
%a21 = mul i32 %a20, %a20
%a22 = mul i32 %a21, %a21
%a23 = mul i32 %a22, %a22
%a24 = mul i32 %a23, %a23
%T = trunc i32 %a24 to i16
ret i16 %T
}
llvm-svn: 324276
2018-02-06 05:50:32 +08:00
|
|
|
Value *LHS = EvaluateInDifferentType(I->getOperand(0), Ty, isSigned);
|
|
|
|
Value *RHS = EvaluateInDifferentType(I->getOperand(1), Ty, isSigned);
|
2010-01-04 15:54:59 +08:00
|
|
|
Res = BinaryOperator::Create((Instruction::BinaryOps)Opc, LHS, RHS);
|
|
|
|
break;
|
2013-01-24 13:22:40 +08:00
|
|
|
}
|
2010-01-04 15:54:59 +08:00
|
|
|
case Instruction::Trunc:
|
|
|
|
case Instruction::ZExt:
|
|
|
|
case Instruction::SExt:
|
|
|
|
// If the source type of the cast is the type we're trying for then we can
|
|
|
|
// just return the source. There's no need to insert it because it is not
|
|
|
|
// new.
|
|
|
|
if (I->getOperand(0)->getType() == Ty)
|
|
|
|
return I->getOperand(0);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:54:59 +08:00
|
|
|
// Otherwise, must be the same type of cast, so just reinsert a new one.
|
2010-01-11 04:25:54 +08:00
|
|
|
// This also handles the case of zext(trunc(x)) -> zext(x).
|
|
|
|
Res = CastInst::CreateIntegerCast(I->getOperand(0), Ty,
|
|
|
|
Opc == Instruction::SExt);
|
2010-01-04 15:54:59 +08:00
|
|
|
break;
|
|
|
|
case Instruction::Select: {
|
|
|
|
Value *True = EvaluateInDifferentType(I->getOperand(1), Ty, isSigned);
|
|
|
|
Value *False = EvaluateInDifferentType(I->getOperand(2), Ty, isSigned);
|
|
|
|
Res = SelectInst::Create(I->getOperand(0), True, False);
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
case Instruction::PHI: {
|
|
|
|
PHINode *OPN = cast<PHINode>(I);
|
2011-03-30 19:28:46 +08:00
|
|
|
PHINode *NPN = PHINode::Create(Ty, OPN->getNumIncomingValues());
|
2010-01-04 15:54:59 +08:00
|
|
|
for (unsigned i = 0, e = OPN->getNumIncomingValues(); i != e; ++i) {
|
2015-03-10 10:37:25 +08:00
|
|
|
Value *V =
|
|
|
|
EvaluateInDifferentType(OPN->getIncomingValue(i), Ty, isSigned);
|
2010-01-04 15:54:59 +08:00
|
|
|
NPN->addIncoming(V, OPN->getIncomingBlock(i));
|
|
|
|
}
|
|
|
|
Res = NPN;
|
|
|
|
break;
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
default:
|
2010-01-04 15:54:59 +08:00
|
|
|
// TODO: Can handle more cases here.
|
|
|
|
llvm_unreachable("Unreachable!");
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:54:59 +08:00
|
|
|
Res->takeName(I);
|
2011-05-27 08:19:40 +08:00
|
|
|
return InsertNewInstWith(Res, *I);
|
2010-01-04 15:54:59 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2016-07-19 17:06:08 +08:00
|
|
|
Instruction::CastOps InstCombiner::isEliminableCastPair(const CastInst *CI1,
|
|
|
|
const CastInst *CI2) {
|
|
|
|
Type *SrcTy = CI1->getSrcTy();
|
|
|
|
Type *MidTy = CI1->getDestTy();
|
|
|
|
Type *DstTy = CI2->getDestTy();
|
|
|
|
|
2017-08-04 13:12:35 +08:00
|
|
|
Instruction::CastOps firstOp = CI1->getOpcode();
|
|
|
|
Instruction::CastOps secondOp = CI2->getOpcode();
|
2015-03-10 10:37:25 +08:00
|
|
|
Type *SrcIntPtrTy =
|
|
|
|
SrcTy->isPtrOrPtrVectorTy() ? DL.getIntPtrType(SrcTy) : nullptr;
|
|
|
|
Type *MidIntPtrTy =
|
|
|
|
MidTy->isPtrOrPtrVectorTy() ? DL.getIntPtrType(MidTy) : nullptr;
|
|
|
|
Type *DstIntPtrTy =
|
|
|
|
DstTy->isPtrOrPtrVectorTy() ? DL.getIntPtrType(DstTy) : nullptr;
|
2010-01-04 15:53:58 +08:00
|
|
|
unsigned Res = CastInst::isEliminableCastPair(firstOp, secondOp, SrcTy, MidTy,
|
2012-10-31 00:03:32 +08:00
|
|
|
DstTy, SrcIntPtrTy, MidIntPtrTy,
|
|
|
|
DstIntPtrTy);
|
2012-10-24 23:52:52 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// We don't want to form an inttoptr or ptrtoint that converts to an integer
|
|
|
|
// type that differs from the pointer size.
|
2012-10-31 00:03:32 +08:00
|
|
|
if ((Res == Instruction::IntToPtr && SrcTy != DstIntPtrTy) ||
|
|
|
|
(Res == Instruction::PtrToInt && DstTy != SrcIntPtrTy))
|
2010-01-04 15:53:58 +08:00
|
|
|
Res = 0;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
return Instruction::CastOps(Res);
|
|
|
|
}
|
|
|
|
|
2018-05-02 00:10:38 +08:00
|
|
|
/// Implement the transforms common to all CastInst visitors.
|
2010-01-04 15:53:58 +08:00
|
|
|
Instruction *InstCombiner::commonCastTransforms(CastInst &CI) {
|
|
|
|
Value *Src = CI.getOperand(0);
|
|
|
|
|
2016-10-26 22:52:35 +08:00
|
|
|
// Try to eliminate a cast of a cast.
|
|
|
|
if (auto *CSrc = dyn_cast<CastInst>(Src)) { // A->B->C cast
|
|
|
|
if (Instruction::CastOps NewOpc = isEliminableCastPair(CSrc, &CI)) {
|
2010-01-04 15:53:58 +08:00
|
|
|
// The first cast (CSrc) is eliminable so we need to fix up or replace
|
|
|
|
// the second cast (CI). CSrc will then have a good chance of being dead.
|
2018-06-27 08:47:53 +08:00
|
|
|
auto *Ty = CI.getType();
|
|
|
|
auto *Res = CastInst::Create(NewOpc, CSrc->getOperand(0), Ty);
|
2018-07-07 01:32:39 +08:00
|
|
|
// Point debug users of the dying cast to the new one.
|
|
|
|
if (CSrc->hasOneUse())
|
|
|
|
replaceAllDbgUsesWith(*CSrc, *Res, CI, DT);
|
2018-01-27 06:02:52 +08:00
|
|
|
return Res;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2018-05-31 08:16:58 +08:00
|
|
|
if (auto *Sel = dyn_cast<SelectInst>(Src)) {
|
2020-01-28 05:33:13 +08:00
|
|
|
// We are casting a select. Try to fold the cast into the select if the
|
|
|
|
// select does not have a compare instruction with matching operand types
|
|
|
|
// or the select is likely better done in a narrow type.
|
|
|
|
// Creating a select with operands that are different sizes than its
|
2018-05-31 08:16:58 +08:00
|
|
|
// condition may inhibit other folds and lead to worse codegen.
|
|
|
|
auto *Cmp = dyn_cast<CmpInst>(Sel->getCondition());
|
2020-01-28 05:33:13 +08:00
|
|
|
if (!Cmp || Cmp->getOperand(0)->getType() != Sel->getType() ||
|
|
|
|
(CI.getOpcode() == Instruction::Trunc &&
|
|
|
|
shouldChangeType(CI.getSrcTy(), CI.getType()))) {
|
2018-07-18 02:08:36 +08:00
|
|
|
if (Instruction *NV = FoldOpIntoSelect(CI, Sel)) {
|
|
|
|
replaceAllDbgUsesWith(*Sel, *NV, CI, DT);
|
2018-05-31 08:16:58 +08:00
|
|
|
return NV;
|
2018-07-18 02:08:36 +08:00
|
|
|
}
|
2020-01-28 05:33:13 +08:00
|
|
|
}
|
2018-05-31 08:16:58 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2016-10-26 22:52:35 +08:00
|
|
|
// If we are casting a PHI, then fold the cast into the PHI.
|
2017-04-15 03:20:12 +08:00
|
|
|
if (auto *PN = dyn_cast<PHINode>(Src)) {
|
2016-10-26 22:52:35 +08:00
|
|
|
// Don't do this if it would create a PHI node with an illegal type from a
|
|
|
|
// legal type.
|
2015-03-10 10:37:25 +08:00
|
|
|
if (!Src->getType()->isIntegerTy() || !CI.getType()->isIntegerTy() ||
|
2020-02-04 20:02:01 +08:00
|
|
|
shouldChangeType(CI.getSrcTy(), CI.getType()))
|
2017-04-15 03:20:12 +08:00
|
|
|
if (Instruction *NV = foldOpIntoPhi(CI, PN))
|
2010-01-04 15:53:58 +08:00
|
|
|
return NV;
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2018-01-31 22:55:53 +08:00
|
|
|
/// Constants and extensions/truncates from the destination type are always
|
|
|
|
/// free to be evaluated in that type. This is a helper for canEvaluate*.
|
|
|
|
static bool canAlwaysEvaluateInType(Value *V, Type *Ty) {
|
|
|
|
if (isa<Constant>(V))
|
|
|
|
return true;
|
|
|
|
Value *X;
|
|
|
|
if ((match(V, m_ZExtOrSExt(m_Value(X))) || match(V, m_Trunc(m_Value(X)))) &&
|
|
|
|
X->getType() == Ty)
|
|
|
|
return true;
|
|
|
|
|
|
|
|
return false;
|
|
|
|
}
|
|
|
|
|
|
|
|
/// Filter out values that we can not evaluate in the destination type for free.
|
|
|
|
/// This is a helper for canEvaluate*.
|
|
|
|
static bool canNotEvaluateInType(Value *V, Type *Ty) {
|
|
|
|
assert(!isa<Constant>(V) && "Constant should already be handled.");
|
|
|
|
if (!isa<Instruction>(V))
|
|
|
|
return true;
|
[InstCombine] don't try to evaluate instructions with >1 use (revert r324014)
This example causes a compile-time explosion:
define i16 @foo(i16 %in) {
%x = zext i16 %in to i32
%a1 = mul i32 %x, %x
%a2 = mul i32 %a1, %a1
%a3 = mul i32 %a2, %a2
%a4 = mul i32 %a3, %a3
%a5 = mul i32 %a4, %a4
%a6 = mul i32 %a5, %a5
%a7 = mul i32 %a6, %a6
%a8 = mul i32 %a7, %a7
%a9 = mul i32 %a8, %a8
%a10 = mul i32 %a9, %a9
%a11 = mul i32 %a10, %a10
%a12 = mul i32 %a11, %a11
%a13 = mul i32 %a12, %a12
%a14 = mul i32 %a13, %a13
%a15 = mul i32 %a14, %a14
%a16 = mul i32 %a15, %a15
%a17 = mul i32 %a16, %a16
%a18 = mul i32 %a17, %a17
%a19 = mul i32 %a18, %a18
%a20 = mul i32 %a19, %a19
%a21 = mul i32 %a20, %a20
%a22 = mul i32 %a21, %a21
%a23 = mul i32 %a22, %a22
%a24 = mul i32 %a23, %a23
%T = trunc i32 %a24 to i16
ret i16 %T
}
llvm-svn: 324276
2018-02-06 05:50:32 +08:00
|
|
|
// We don't extend or shrink something that has multiple uses -- doing so
|
|
|
|
// would require duplicating the instruction which isn't profitable.
|
|
|
|
if (!V->hasOneUse())
|
|
|
|
return true;
|
|
|
|
|
2018-01-31 22:55:53 +08:00
|
|
|
return false;
|
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Return true if we can evaluate the specified expression tree as type Ty
|
|
|
|
/// instead of its larger type, and arrive with the same value.
|
|
|
|
/// This is used by code that tries to eliminate truncates.
|
2010-01-10 08:58:42 +08:00
|
|
|
///
|
|
|
|
/// Ty will always be a type smaller than V. We should return true if trunc(V)
|
|
|
|
/// can be computed by computing V in the smaller type. If V is an instruction,
|
|
|
|
/// then trunc(inst(x,y)) can be computed as inst(trunc(x),trunc(y)), which only
|
|
|
|
/// makes sense if x and y can be efficiently truncated.
|
|
|
|
///
|
2010-01-11 10:43:35 +08:00
|
|
|
/// This function works on both vectors and scalars.
|
|
|
|
///
|
2015-09-09 22:54:29 +08:00
|
|
|
static bool canEvaluateTruncated(Value *V, Type *Ty, InstCombiner &IC,
|
Make use of @llvm.assume in ValueTracking (computeKnownBits, etc.)
This change, which allows @llvm.assume to be used from within computeKnownBits
(and other associated functions in ValueTracking), adds some (optional)
parameters to computeKnownBits and friends. These functions now (optionally)
take a "context" instruction pointer, an AssumptionTracker pointer, and also a
DomTree pointer, and most of the changes are just to pass this new information
when it is easily available from InstSimplify, InstCombine, etc.
As explained below, the significant conceptual change is that known properties
of a value might depend on the control-flow location of the use (because we
care that the @llvm.assume dominates the use because assumptions have
control-flow dependencies). This means that, when we ask if bits are known in a
value, we might get different answers for different uses.
The significant changes are all in ValueTracking. Two main changes: First, as
with the rest of the code, new parameters need to be passed around. To make
this easier, I grouped them into a structure, and I made internal static
versions of the relevant functions that take this structure as a parameter. The
new code does as you might expect, it looks for @llvm.assume calls that make
use of the value we're trying to learn something about (often indirectly),
attempts to pattern match that expression, and uses the result if successful.
By making use of the AssumptionTracker, the process of finding @llvm.assume
calls is not expensive.
Part of the structure being passed around inside ValueTracking is a set of
already-considered @llvm.assume calls. This is to prevent a query using, for
example, the assume(a == b), to recurse on itself. The context and DT params
are used to find applicable assumptions. An assumption needs to dominate the
context instruction, or come after it deterministically. In this latter case we
only handle the specific case where both the assumption and the context
instruction are in the same block, and we need to exclude assumptions from
being used to simplify their own ephemeral values (those which contribute only
to the assumption) because otherwise the assumption would prove its feeding
comparison trivial and would be removed.
This commit adds the plumbing and the logic for a simple masked-bit propagation
(just enough to write a regression test). Future commits add more patterns
(and, correspondingly, more regression tests).
llvm-svn: 217342
2014-09-08 02:57:58 +08:00
|
|
|
Instruction *CxtI) {
|
2018-01-31 22:55:53 +08:00
|
|
|
if (canAlwaysEvaluateInType(V, Ty))
|
2010-01-10 08:58:42 +08:00
|
|
|
return true;
|
2018-01-31 22:55:53 +08:00
|
|
|
if (canNotEvaluateInType(V, Ty))
|
|
|
|
return false;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-01-31 22:55:53 +08:00
|
|
|
auto *I = cast<Instruction>(V);
|
2011-07-18 12:54:35 +08:00
|
|
|
Type *OrigTy = V->getType();
|
2018-01-31 22:55:53 +08:00
|
|
|
switch (I->getOpcode()) {
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::Add:
|
|
|
|
case Instruction::Sub:
|
|
|
|
case Instruction::Mul:
|
|
|
|
case Instruction::And:
|
|
|
|
case Instruction::Or:
|
|
|
|
case Instruction::Xor:
|
|
|
|
// These operators can all arbitrarily be extended or truncated.
|
2015-09-09 22:54:29 +08:00
|
|
|
return canEvaluateTruncated(I->getOperand(0), Ty, IC, CxtI) &&
|
|
|
|
canEvaluateTruncated(I->getOperand(1), Ty, IC, CxtI);
|
2010-01-08 07:41:00 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::UDiv:
|
|
|
|
case Instruction::URem: {
|
|
|
|
// UDiv and URem can be truncated if all the truncated bits are zero.
|
|
|
|
uint32_t OrigBitWidth = OrigTy->getScalarSizeInBits();
|
|
|
|
uint32_t BitWidth = Ty->getScalarSizeInBits();
|
2018-05-11 06:45:28 +08:00
|
|
|
assert(BitWidth < OrigBitWidth && "Unexpected bitwidths!");
|
|
|
|
APInt Mask = APInt::getBitsSetFrom(OrigBitWidth, BitWidth);
|
|
|
|
if (IC.MaskedValueIsZero(I->getOperand(0), Mask, 0, CxtI) &&
|
|
|
|
IC.MaskedValueIsZero(I->getOperand(1), Mask, 0, CxtI)) {
|
|
|
|
return canEvaluateTruncated(I->getOperand(0), Ty, IC, CxtI) &&
|
|
|
|
canEvaluateTruncated(I->getOperand(1), Ty, IC, CxtI);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
break;
|
2010-01-08 07:41:00 +08:00
|
|
|
}
|
2017-08-16 06:48:41 +08:00
|
|
|
case Instruction::Shl: {
|
2010-01-10 08:58:42 +08:00
|
|
|
// If we are truncating the result of this SHL, and if it's a shift of a
|
|
|
|
// constant amount, we can always perform a SHL in a smaller type.
|
2017-08-16 06:48:41 +08:00
|
|
|
const APInt *Amt;
|
|
|
|
if (match(I->getOperand(1), m_APInt(Amt))) {
|
2010-01-10 08:58:42 +08:00
|
|
|
uint32_t BitWidth = Ty->getScalarSizeInBits();
|
2017-08-16 06:48:41 +08:00
|
|
|
if (Amt->getLimitedValue(BitWidth) < BitWidth)
|
2015-09-09 22:54:29 +08:00
|
|
|
return canEvaluateTruncated(I->getOperand(0), Ty, IC, CxtI);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
break;
|
2017-08-16 06:48:41 +08:00
|
|
|
}
|
|
|
|
case Instruction::LShr: {
|
2010-01-10 08:58:42 +08:00
|
|
|
// If this is a truncate of a logical shr, we can truncate it to a smaller
|
2012-09-27 18:14:43 +08:00
|
|
|
// lshr iff we know that the bits we would otherwise be shifting in are
|
2010-01-10 08:58:42 +08:00
|
|
|
// already zeros.
|
2017-08-16 06:48:41 +08:00
|
|
|
const APInt *Amt;
|
|
|
|
if (match(I->getOperand(1), m_APInt(Amt))) {
|
2010-01-10 08:58:42 +08:00
|
|
|
uint32_t OrigBitWidth = OrigTy->getScalarSizeInBits();
|
|
|
|
uint32_t BitWidth = Ty->getScalarSizeInBits();
|
2018-05-10 08:53:25 +08:00
|
|
|
if (Amt->getLimitedValue(BitWidth) < BitWidth &&
|
|
|
|
IC.MaskedValueIsZero(I->getOperand(0),
|
|
|
|
APInt::getBitsSetFrom(OrigBitWidth, BitWidth), 0, CxtI)) {
|
2015-09-09 22:54:29 +08:00
|
|
|
return canEvaluateTruncated(I->getOperand(0), Ty, IC, CxtI);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
break;
|
2017-08-16 06:48:41 +08:00
|
|
|
}
|
2017-08-17 06:42:38 +08:00
|
|
|
case Instruction::AShr: {
|
|
|
|
// If this is a truncate of an arithmetic shr, we can truncate it to a
|
|
|
|
// smaller ashr iff we know that all the bits from the sign bit of the
|
|
|
|
// original type and the sign bit of the truncate type are similar.
|
|
|
|
// TODO: It is enough to check that the bits we would be shifting in are
|
|
|
|
// similar to sign bit of the truncate type.
|
|
|
|
const APInt *Amt;
|
|
|
|
if (match(I->getOperand(1), m_APInt(Amt))) {
|
|
|
|
uint32_t OrigBitWidth = OrigTy->getScalarSizeInBits();
|
|
|
|
uint32_t BitWidth = Ty->getScalarSizeInBits();
|
|
|
|
if (Amt->getLimitedValue(BitWidth) < BitWidth &&
|
|
|
|
OrigBitWidth - BitWidth <
|
|
|
|
IC.ComputeNumSignBits(I->getOperand(0), 0, CxtI))
|
|
|
|
return canEvaluateTruncated(I->getOperand(0), Ty, IC, CxtI);
|
|
|
|
}
|
|
|
|
break;
|
|
|
|
}
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::Trunc:
|
|
|
|
// trunc(trunc(x)) -> trunc(x)
|
|
|
|
return true;
|
2010-08-28 04:32:06 +08:00
|
|
|
case Instruction::ZExt:
|
|
|
|
case Instruction::SExt:
|
|
|
|
// trunc(ext(x)) -> ext(x) if the source type is smaller than the new dest
|
|
|
|
// trunc(ext(x)) -> trunc(x) if the source type is larger than the new dest
|
|
|
|
return true;
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::Select: {
|
|
|
|
SelectInst *SI = cast<SelectInst>(I);
|
2015-09-09 22:54:29 +08:00
|
|
|
return canEvaluateTruncated(SI->getTrueValue(), Ty, IC, CxtI) &&
|
|
|
|
canEvaluateTruncated(SI->getFalseValue(), Ty, IC, CxtI);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
case Instruction::PHI: {
|
|
|
|
// We can change a phi if we can change all operands. Note that we never
|
|
|
|
// get into trouble with cyclic PHIs here because we only consider
|
|
|
|
// instructions with a single use.
|
|
|
|
PHINode *PN = cast<PHINode>(I);
|
2015-05-13 04:05:31 +08:00
|
|
|
for (Value *IncValue : PN->incoming_values())
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateTruncated(IncValue, Ty, IC, CxtI))
|
2010-01-10 08:58:42 +08:00
|
|
|
return false;
|
|
|
|
return true;
|
2010-01-06 09:56:21 +08:00
|
|
|
}
|
2010-01-10 08:58:42 +08:00
|
|
|
default:
|
|
|
|
// TODO: Can handle more cases here.
|
|
|
|
break;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
return false;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2015-12-15 00:16:54 +08:00
|
|
|
/// Given a vector that is bitcast to an integer, optionally logically
|
|
|
|
/// right-shifted, and truncated, convert it to an extractelement.
|
|
|
|
/// Example (big endian):
|
|
|
|
/// trunc (lshr (bitcast <4 x i32> %X to i128), 32) to i32
|
|
|
|
/// --->
|
|
|
|
/// extractelement <4 x i32> %X, 1
|
2017-07-07 07:18:43 +08:00
|
|
|
static Instruction *foldVecTruncToExtElt(TruncInst &Trunc, InstCombiner &IC) {
|
2015-12-15 00:16:54 +08:00
|
|
|
Value *TruncOp = Trunc.getOperand(0);
|
|
|
|
Type *DestType = Trunc.getType();
|
|
|
|
if (!TruncOp->hasOneUse() || !isa<IntegerType>(DestType))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
Value *VecInput = nullptr;
|
|
|
|
ConstantInt *ShiftVal = nullptr;
|
|
|
|
if (!match(TruncOp, m_CombineOr(m_BitCast(m_Value(VecInput)),
|
|
|
|
m_LShr(m_BitCast(m_Value(VecInput)),
|
|
|
|
m_ConstantInt(ShiftVal)))) ||
|
|
|
|
!isa<VectorType>(VecInput->getType()))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
VectorType *VecType = cast<VectorType>(VecInput->getType());
|
|
|
|
unsigned VecWidth = VecType->getPrimitiveSizeInBits();
|
|
|
|
unsigned DestWidth = DestType->getPrimitiveSizeInBits();
|
|
|
|
unsigned ShiftAmount = ShiftVal ? ShiftVal->getZExtValue() : 0;
|
|
|
|
|
|
|
|
if ((VecWidth % DestWidth != 0) || (ShiftAmount % DestWidth != 0))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
// If the element type of the vector doesn't match the result type,
|
|
|
|
// bitcast it to a vector type that we can extract from.
|
|
|
|
unsigned NumVecElts = VecWidth / DestWidth;
|
|
|
|
if (VecType->getElementType() != DestType) {
|
2020-05-30 06:24:15 +08:00
|
|
|
VecType = FixedVectorType::get(DestType, NumVecElts);
|
2017-07-08 07:16:26 +08:00
|
|
|
VecInput = IC.Builder.CreateBitCast(VecInput, VecType, "bc");
|
2015-12-15 00:16:54 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
unsigned Elt = ShiftAmount / DestWidth;
|
2017-07-07 07:18:43 +08:00
|
|
|
if (IC.getDataLayout().isBigEndian())
|
2015-12-15 00:16:54 +08:00
|
|
|
Elt = NumVecElts - 1 - Elt;
|
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
return ExtractElementInst::Create(VecInput, IC.Builder.getInt32(Elt));
|
2015-12-15 00:16:54 +08:00
|
|
|
}
|
|
|
|
|
2017-08-10 02:37:41 +08:00
|
|
|
/// Rotate left/right may occur in a wider type than necessary because of type
|
2019-01-05 01:38:12 +08:00
|
|
|
/// promotion rules. Try to narrow the inputs and convert to funnel shift.
|
2017-08-10 02:37:41 +08:00
|
|
|
Instruction *InstCombiner::narrowRotate(TruncInst &Trunc) {
|
|
|
|
assert((isa<VectorType>(Trunc.getSrcTy()) ||
|
|
|
|
shouldChangeType(Trunc.getSrcTy(), Trunc.getType())) &&
|
|
|
|
"Don't narrow to an illegal scalar type");
|
|
|
|
|
2018-11-16 01:19:14 +08:00
|
|
|
// Bail out on strange types. It is possible to handle some of these patterns
|
|
|
|
// even with non-power-of-2 sizes, but it is not a likely scenario.
|
|
|
|
Type *DestTy = Trunc.getType();
|
|
|
|
unsigned NarrowWidth = DestTy->getScalarSizeInBits();
|
|
|
|
if (!isPowerOf2_32(NarrowWidth))
|
|
|
|
return nullptr;
|
|
|
|
|
2017-08-10 02:37:41 +08:00
|
|
|
// First, find an or'd pair of opposite shifts with the same shifted operand:
|
|
|
|
// trunc (or (lshr ShVal, ShAmt0), (shl ShVal, ShAmt1))
|
|
|
|
Value *Or0, *Or1;
|
|
|
|
if (!match(Trunc.getOperand(0), m_OneUse(m_Or(m_Value(Or0), m_Value(Or1)))))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
Value *ShVal, *ShAmt0, *ShAmt1;
|
|
|
|
if (!match(Or0, m_OneUse(m_LogicalShift(m_Value(ShVal), m_Value(ShAmt0)))) ||
|
|
|
|
!match(Or1, m_OneUse(m_LogicalShift(m_Specific(ShVal), m_Value(ShAmt1)))))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
auto ShiftOpcode0 = cast<BinaryOperator>(Or0)->getOpcode();
|
|
|
|
auto ShiftOpcode1 = cast<BinaryOperator>(Or1)->getOpcode();
|
|
|
|
if (ShiftOpcode0 == ShiftOpcode1)
|
|
|
|
return nullptr;
|
|
|
|
|
2018-11-13 06:00:00 +08:00
|
|
|
// Match the shift amount operands for a rotate pattern. This always matches
|
|
|
|
// a subtraction on the R operand.
|
|
|
|
auto matchShiftAmount = [](Value *L, Value *R, unsigned Width) -> Value * {
|
|
|
|
// The shift amounts may add up to the narrow bit width:
|
|
|
|
// (shl ShVal, L) | (lshr ShVal, Width - L)
|
|
|
|
if (match(R, m_OneUse(m_Sub(m_SpecificInt(Width), m_Specific(L)))))
|
|
|
|
return L;
|
|
|
|
|
[InstCombine] narrow width of rotate patterns, part 2 (PR39624)
The sub-pattern for the shift amount in a rotate can take on
several different forms, and there's apparently no way to
canonicalize those without seeing the entire rotate sequence.
This is the form noted in:
https://bugs.llvm.org/show_bug.cgi?id=39624
https://rise4fun.com/Alive/qnT
%zx = zext i8 %x to i32
%maskedShAmt = and i32 %shAmt, 7
%shl = shl i32 %zx, %maskedShAmt
%negShAmt = sub i32 0, %shAmt
%maskedNegShAmt = and i32 %negShAmt, 7
%shr = lshr i32 %zx, %maskedNegShAmt
%rot = or i32 %shl, %shr
%r = trunc i32 %rot to i8
=>
%truncShAmt = trunc i32 %shAmt to i8
%maskedShAmt2 = and i8 %truncShAmt, 7
%shl2 = shl i8 %x, %maskedShAmt2
%negShAmt2 = sub i8 0, %truncShAmt
%maskedNegShAmt2 = and i8 %negShAmt2, 7
%shr2 = lshr i8 %x, %maskedNegShAmt2
%r = or i8 %shl2, %shr2
llvm-svn: 346713
2018-11-13 06:11:09 +08:00
|
|
|
// The shift amount may be masked with negation:
|
|
|
|
// (shl ShVal, (X & (Width - 1))) | (lshr ShVal, ((-X) & (Width - 1)))
|
|
|
|
Value *X;
|
|
|
|
unsigned Mask = Width - 1;
|
|
|
|
if (match(L, m_And(m_Value(X), m_SpecificInt(Mask))) &&
|
|
|
|
match(R, m_And(m_Neg(m_Specific(X)), m_SpecificInt(Mask))))
|
|
|
|
return X;
|
|
|
|
|
[InstCombine] narrow width of rotate patterns, part 3
This is a longer variant for the pattern handled in
rL346713
This one includes zexts.
Eventually, we should canonicalize all rotate patterns
to the funnel shift intrinsics, but we need a bit more
infrastructure to make sure the vectorizers handle those
intrinsics as well as the shift+logic ops.
https://rise4fun.com/Alive/FMn
Name: narrow rotateright
%neg = sub i8 0, %shamt
%rshamt = and i8 %shamt, 7
%rshamtconv = zext i8 %rshamt to i32
%lshamt = and i8 %neg, 7
%lshamtconv = zext i8 %lshamt to i32
%conv = zext i8 %x to i32
%shr = lshr i32 %conv, %rshamtconv
%shl = shl i32 %conv, %lshamtconv
%or = or i32 %shl, %shr
%r = trunc i32 %or to i8
=>
%maskedShAmt2 = and i8 %shamt, 7
%negShAmt2 = sub i8 0, %shamt
%maskedNegShAmt2 = and i8 %negShAmt2, 7
%shl2 = lshr i8 %x, %maskedShAmt2
%shr2 = shl i8 %x, %maskedNegShAmt2
%r = or i8 %shl2, %shr2
llvm-svn: 346716
2018-11-13 06:52:25 +08:00
|
|
|
// Same as above, but the shift amount may be extended after masking:
|
|
|
|
if (match(L, m_ZExt(m_And(m_Value(X), m_SpecificInt(Mask)))) &&
|
|
|
|
match(R, m_ZExt(m_And(m_Neg(m_Specific(X)), m_SpecificInt(Mask)))))
|
|
|
|
return X;
|
|
|
|
|
2018-11-13 06:00:00 +08:00
|
|
|
return nullptr;
|
|
|
|
};
|
|
|
|
|
|
|
|
Value *ShAmt = matchShiftAmount(ShAmt0, ShAmt1, NarrowWidth);
|
|
|
|
bool SubIsOnLHS = false;
|
|
|
|
if (!ShAmt) {
|
|
|
|
ShAmt = matchShiftAmount(ShAmt1, ShAmt0, NarrowWidth);
|
2017-08-10 02:37:41 +08:00
|
|
|
SubIsOnLHS = true;
|
|
|
|
}
|
2018-11-13 06:00:00 +08:00
|
|
|
if (!ShAmt)
|
|
|
|
return nullptr;
|
2017-08-10 02:37:41 +08:00
|
|
|
|
|
|
|
// The shifted value must have high zeros in the wide type. Typically, this
|
|
|
|
// will be a zext, but it could also be the result of an 'and' or 'shift'.
|
|
|
|
unsigned WideWidth = Trunc.getSrcTy()->getScalarSizeInBits();
|
|
|
|
APInt HiBitMask = APInt::getHighBitsSet(WideWidth, WideWidth - NarrowWidth);
|
|
|
|
if (!MaskedValueIsZero(ShVal, HiBitMask, 0, &Trunc))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
// We have an unnecessarily wide rotate!
|
|
|
|
// trunc (or (lshr ShVal, ShAmt), (shl ShVal, BitWidth - ShAmt))
|
2019-01-05 01:38:12 +08:00
|
|
|
// Narrow the inputs and convert to funnel shift intrinsic:
|
|
|
|
// llvm.fshl.i8(trunc(ShVal), trunc(ShVal), trunc(ShAmt))
|
2017-08-10 02:37:41 +08:00
|
|
|
Value *NarrowShAmt = Builder.CreateTrunc(ShAmt, DestTy);
|
|
|
|
Value *X = Builder.CreateTrunc(ShVal, DestTy);
|
2019-01-05 01:38:12 +08:00
|
|
|
bool IsFshl = (!SubIsOnLHS && ShiftOpcode0 == BinaryOperator::Shl) ||
|
|
|
|
(SubIsOnLHS && ShiftOpcode1 == BinaryOperator::Shl);
|
|
|
|
Intrinsic::ID IID = IsFshl ? Intrinsic::fshl : Intrinsic::fshr;
|
|
|
|
Function *F = Intrinsic::getDeclaration(Trunc.getModule(), IID, DestTy);
|
|
|
|
return IntrinsicInst::Create(F, { X, X, NarrowShAmt });
|
2017-08-10 02:37:41 +08:00
|
|
|
}
|
|
|
|
|
2017-08-05 23:19:18 +08:00
|
|
|
/// Try to narrow the width of math or bitwise logic instructions by pulling a
|
|
|
|
/// truncate ahead of binary operators.
|
|
|
|
/// TODO: Transforms for truncated shifts should be moved into here.
|
|
|
|
Instruction *InstCombiner::narrowBinOp(TruncInst &Trunc) {
|
2016-12-01 04:48:54 +08:00
|
|
|
Type *SrcTy = Trunc.getSrcTy();
|
|
|
|
Type *DestTy = Trunc.getType();
|
2017-08-10 02:37:41 +08:00
|
|
|
if (!isa<VectorType>(SrcTy) && !shouldChangeType(SrcTy, DestTy))
|
2016-12-01 04:48:54 +08:00
|
|
|
return nullptr;
|
|
|
|
|
2017-08-05 23:19:18 +08:00
|
|
|
BinaryOperator *BinOp;
|
|
|
|
if (!match(Trunc.getOperand(0), m_OneUse(m_BinOp(BinOp))))
|
2016-12-01 04:48:54 +08:00
|
|
|
return nullptr;
|
|
|
|
|
2017-11-16 03:12:01 +08:00
|
|
|
Value *BinOp0 = BinOp->getOperand(0);
|
|
|
|
Value *BinOp1 = BinOp->getOperand(1);
|
2017-08-05 23:19:18 +08:00
|
|
|
switch (BinOp->getOpcode()) {
|
|
|
|
case Instruction::And:
|
|
|
|
case Instruction::Or:
|
|
|
|
case Instruction::Xor:
|
|
|
|
case Instruction::Add:
|
2017-11-16 22:40:51 +08:00
|
|
|
case Instruction::Sub:
|
2017-08-05 23:19:18 +08:00
|
|
|
case Instruction::Mul: {
|
|
|
|
Constant *C;
|
2017-11-16 22:40:51 +08:00
|
|
|
if (match(BinOp0, m_Constant(C))) {
|
|
|
|
// trunc (binop C, X) --> binop (trunc C', X)
|
|
|
|
Constant *NarrowC = ConstantExpr::getTrunc(C, DestTy);
|
|
|
|
Value *TruncX = Builder.CreateTrunc(BinOp1, DestTy);
|
|
|
|
return BinaryOperator::Create(BinOp->getOpcode(), NarrowC, TruncX);
|
|
|
|
}
|
2017-11-16 03:12:01 +08:00
|
|
|
if (match(BinOp1, m_Constant(C))) {
|
2017-08-05 23:19:18 +08:00
|
|
|
// trunc (binop X, C) --> binop (trunc X, C')
|
|
|
|
Constant *NarrowC = ConstantExpr::getTrunc(C, DestTy);
|
2017-11-16 03:12:01 +08:00
|
|
|
Value *TruncX = Builder.CreateTrunc(BinOp0, DestTy);
|
2017-08-05 23:19:18 +08:00
|
|
|
return BinaryOperator::Create(BinOp->getOpcode(), TruncX, NarrowC);
|
|
|
|
}
|
2017-11-16 03:12:01 +08:00
|
|
|
Value *X;
|
|
|
|
if (match(BinOp0, m_ZExtOrSExt(m_Value(X))) && X->getType() == DestTy) {
|
|
|
|
// trunc (binop (ext X), Y) --> binop X, (trunc Y)
|
|
|
|
Value *NarrowOp1 = Builder.CreateTrunc(BinOp1, DestTy);
|
|
|
|
return BinaryOperator::Create(BinOp->getOpcode(), X, NarrowOp1);
|
|
|
|
}
|
|
|
|
if (match(BinOp1, m_ZExtOrSExt(m_Value(X))) && X->getType() == DestTy) {
|
|
|
|
// trunc (binop Y, (ext X)) --> binop (trunc Y), X
|
|
|
|
Value *NarrowOp0 = Builder.CreateTrunc(BinOp0, DestTy);
|
|
|
|
return BinaryOperator::Create(BinOp->getOpcode(), NarrowOp0, X);
|
|
|
|
}
|
2017-08-05 23:19:18 +08:00
|
|
|
break;
|
|
|
|
}
|
|
|
|
|
|
|
|
default: break;
|
|
|
|
}
|
|
|
|
|
2017-08-10 02:37:41 +08:00
|
|
|
if (Instruction *NarrowOr = narrowRotate(Trunc))
|
|
|
|
return NarrowOr;
|
|
|
|
|
2017-08-05 23:19:18 +08:00
|
|
|
return nullptr;
|
2016-12-01 04:48:54 +08:00
|
|
|
}
|
|
|
|
|
2017-03-08 05:45:16 +08:00
|
|
|
/// Try to narrow the width of a splat shuffle. This could be generalized to any
|
|
|
|
/// shuffle with a constant operand, but we limit the transform to avoid
|
|
|
|
/// creating a shuffle type that targets may not be able to lower effectively.
|
|
|
|
static Instruction *shrinkSplatShuffle(TruncInst &Trunc,
|
|
|
|
InstCombiner::BuilderTy &Builder) {
|
|
|
|
auto *Shuf = dyn_cast<ShuffleVectorInst>(Trunc.getOperand(0));
|
|
|
|
if (Shuf && Shuf->hasOneUse() && isa<UndefValue>(Shuf->getOperand(1)) &&
|
2020-04-01 04:08:59 +08:00
|
|
|
is_splat(Shuf->getShuffleMask()) &&
|
2017-03-08 23:02:23 +08:00
|
|
|
Shuf->getType() == Shuf->getOperand(0)->getType()) {
|
2017-03-08 05:45:16 +08:00
|
|
|
// trunc (shuf X, Undef, SplatMask) --> shuf (trunc X), Undef, SplatMask
|
|
|
|
Constant *NarrowUndef = UndefValue::get(Trunc.getType());
|
|
|
|
Value *NarrowOp = Builder.CreateTrunc(Shuf->getOperand(0), Trunc.getType());
|
2020-04-01 04:08:59 +08:00
|
|
|
return new ShuffleVectorInst(NarrowOp, NarrowUndef, Shuf->getShuffleMask());
|
2017-03-08 05:45:16 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
2017-03-08 07:27:14 +08:00
|
|
|
/// Try to narrow the width of an insert element. This could be generalized for
|
|
|
|
/// any vector constant, but we limit the transform to insertion into undef to
|
|
|
|
/// avoid potential backend problems from unsupported insertion widths. This
|
|
|
|
/// could also be extended to handle the case of inserting a scalar constant
|
|
|
|
/// into a vector variable.
|
|
|
|
static Instruction *shrinkInsertElt(CastInst &Trunc,
|
|
|
|
InstCombiner::BuilderTy &Builder) {
|
|
|
|
Instruction::CastOps Opcode = Trunc.getOpcode();
|
|
|
|
assert((Opcode == Instruction::Trunc || Opcode == Instruction::FPTrunc) &&
|
|
|
|
"Unexpected instruction for shrinking");
|
|
|
|
|
|
|
|
auto *InsElt = dyn_cast<InsertElementInst>(Trunc.getOperand(0));
|
|
|
|
if (!InsElt || !InsElt->hasOneUse())
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
Type *DestTy = Trunc.getType();
|
|
|
|
Type *DestScalarTy = DestTy->getScalarType();
|
|
|
|
Value *VecOp = InsElt->getOperand(0);
|
|
|
|
Value *ScalarOp = InsElt->getOperand(1);
|
|
|
|
Value *Index = InsElt->getOperand(2);
|
|
|
|
|
|
|
|
if (isa<UndefValue>(VecOp)) {
|
|
|
|
// trunc (inselt undef, X, Index) --> inselt undef, (trunc X), Index
|
|
|
|
// fptrunc (inselt undef, X, Index) --> inselt undef, (fptrunc X), Index
|
|
|
|
UndefValue *NarrowUndef = UndefValue::get(DestTy);
|
|
|
|
Value *NarrowOp = Builder.CreateCast(Opcode, ScalarOp, DestScalarTy);
|
|
|
|
return InsertElementInst::Create(NarrowUndef, NarrowOp, Index);
|
|
|
|
}
|
|
|
|
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
Instruction *InstCombiner::visitTrunc(TruncInst &Trunc) {
|
|
|
|
if (Instruction *Result = commonCastTransforms(Trunc))
|
2010-01-04 15:53:58 +08:00
|
|
|
return Result;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
Value *Src = Trunc.getOperand(0);
|
|
|
|
Type *DestTy = Trunc.getType(), *SrcTy = Src->getType();
|
|
|
|
unsigned DestWidth = DestTy->getScalarSizeInBits();
|
|
|
|
unsigned SrcWidth = SrcTy->getScalarSizeInBits();
|
2020-03-30 03:48:54 +08:00
|
|
|
ConstantInt *Cst;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
// Attempt to truncate the entire input expression tree to the destination
|
|
|
|
// type. Only do this if the dest type is a simple type, don't convert the
|
|
|
|
// expression tree to something weird like i93 unless the source is also
|
|
|
|
// strange.
|
2017-02-01 01:25:42 +08:00
|
|
|
if ((DestTy->isVectorTy() || shouldChangeType(SrcTy, DestTy)) &&
|
2020-06-08 23:15:13 +08:00
|
|
|
canEvaluateTruncated(Src, DestTy, *this, &Trunc)) {
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
// If this cast is a truncate, evaluting in a different type always
|
|
|
|
// eliminates the cast, so it is always a win.
|
2018-05-14 20:53:11 +08:00
|
|
|
LLVM_DEBUG(
|
|
|
|
dbgs() << "ICE: EvaluateInDifferentType converting expression type"
|
|
|
|
" to avoid cast: "
|
2020-06-08 23:15:13 +08:00
|
|
|
<< Trunc << '\n');
|
2010-01-10 08:58:42 +08:00
|
|
|
Value *Res = EvaluateInDifferentType(Src, DestTy, false);
|
|
|
|
assert(Res->getType() == DestTy);
|
2020-06-08 23:15:13 +08:00
|
|
|
return replaceInstUsesWith(Trunc, Res);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2018-07-05 01:44:04 +08:00
|
|
|
// Test if the trunc is the user of a select which is part of a
|
|
|
|
// minimum or maximum operation. If so, don't do any more simplification.
|
|
|
|
// Even simplifying demanded bits can break the canonical form of a
|
|
|
|
// min/max.
|
|
|
|
Value *LHS, *RHS;
|
2020-06-08 23:15:13 +08:00
|
|
|
if (SelectInst *Sel = dyn_cast<SelectInst>(Src))
|
|
|
|
if (matchSelectPattern(Sel, LHS, RHS).Flavor != SPF_UNKNOWN)
|
2018-07-05 01:44:04 +08:00
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
// See if we can simplify any instructions used by the input whose sole
|
|
|
|
// purpose is to compute bits we don't care about.
|
2020-06-08 23:15:13 +08:00
|
|
|
if (SimplifyDemandedInstructionBits(Trunc))
|
|
|
|
return &Trunc;
|
2018-07-05 01:44:04 +08:00
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
if (DestWidth == 1) {
|
|
|
|
Value *Zero = Constant::getNullValue(SrcTy);
|
[InstCombine] reverse 'trunc X to <N x i1>' canonicalization; 2nd try
Re-trying r344082 because it unintentionally included extra diffs.
Original commit message:
icmp ne (and X, 1), 0 --> trunc X to N x i1
Ideally, we'd do the same for scalars, but there will likely be
regressions unless we add more trunc folds as we're doing here
for vectors.
The motivating vector case is from PR37549:
https://bugs.llvm.org/show_bug.cgi?id=37549
define <4 x float> @bitwise_select(<4 x float> %x, <4 x float> %y, <4 x float> %z, <4 x float> %w) {
%c = fcmp ole <4 x float> %x, %y
%s = sext <4 x i1> %c to <4 x i32>
%s1 = shufflevector <4 x i32> %s, <4 x i32> undef, <4 x i32> <i32 0, i32 0, i32 1, i32 1>
%s2 = shufflevector <4 x i32> %s, <4 x i32> undef, <4 x i32> <i32 2, i32 2, i32 3, i32 3>
%cond = or <4 x i32> %s1, %s2
%condtr = trunc <4 x i32> %cond to <4 x i1>
%r = select <4 x i1> %condtr, <4 x float> %z, <4 x float> %w
ret <4 x float> %r
}
Here's a sampling of the vector codegen for that case using
mask+icmp (current behavior) vs. trunc (with this patch):
AVX before:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vandps LCPI0_0(%rip), %xmm0, %xmm0
vxorps %xmm1, %xmm1, %xmm1
vpcmpeqd %xmm1, %xmm0, %xmm0
vblendvps %xmm0, %xmm3, %xmm2, %xmm0
AVX after:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vblendvps %xmm0, %xmm2, %xmm3, %xmm0
AVX512f before:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vpbroadcastd LCPI0_0(%rip), %xmm1 ## xmm1 = [1,1,1,1]
vptestnmd %zmm1, %zmm0, %k1
vblendmps %zmm3, %zmm2, %zmm0 {%k1}
AVX512f after:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vpslld $31, %xmm0, %xmm0
vptestmd %zmm0, %zmm0, %k1
vblendmps %zmm2, %zmm3, %zmm0 {%k1}
AArch64 before:
fcmge v0.4s, v1.4s, v0.4s
zip1 v1.4s, v0.4s, v0.4s
zip2 v0.4s, v0.4s, v0.4s
orr v0.16b, v1.16b, v0.16b
movi v1.4s, #1
and v0.16b, v0.16b, v1.16b
cmeq v0.4s, v0.4s, #0
bsl v0.16b, v3.16b, v2.16b
AArch64 after:
fcmge v0.4s, v1.4s, v0.4s
zip1 v1.4s, v0.4s, v0.4s
zip2 v0.4s, v0.4s, v0.4s
orr v0.16b, v1.16b, v0.16b
bsl v0.16b, v2.16b, v3.16b
PowerPC-le before:
xvcmpgesp 34, 35, 34
vspltisw 0, 1
vmrglw 3, 2, 2
vmrghw 2, 2, 2
xxlor 0, 35, 34
xxlxor 35, 35, 35
xxland 34, 0, 32
vcmpequw 2, 2, 3
xxsel 34, 36, 37, 34
PowerPC-le after:
xvcmpgesp 34, 35, 34
vmrglw 3, 2, 2
vmrghw 2, 2, 2
xxlor 0, 35, 34
xxsel 34, 37, 36, 0
Differential Revision: https://reviews.llvm.org/D52747
llvm-svn: 344181
2018-10-11 04:47:46 +08:00
|
|
|
if (DestTy->isIntegerTy()) {
|
|
|
|
// Canonicalize trunc x to i1 -> icmp ne (and x, 1), 0 (scalar only).
|
|
|
|
// TODO: We canonicalize to more instructions here because we are probably
|
|
|
|
// lacking equivalent analysis for trunc relative to icmp. There may also
|
|
|
|
// be codegen concerns. If those trunc limitations were removed, we could
|
|
|
|
// remove this transform.
|
|
|
|
Value *And = Builder.CreateAnd(Src, ConstantInt::get(SrcTy, 1));
|
|
|
|
return new ICmpInst(ICmpInst::ICMP_NE, And, Zero);
|
|
|
|
}
|
|
|
|
|
|
|
|
// For vectors, we do not canonicalize all truncs to icmp, so optimize
|
|
|
|
// patterns that would be covered within visitICmpInst.
|
|
|
|
Value *X;
|
|
|
|
const APInt *C;
|
|
|
|
if (match(Src, m_OneUse(m_LShr(m_Value(X), m_APInt(C))))) {
|
|
|
|
// trunc (lshr X, C) to i1 --> icmp ne (and X, C'), 0
|
2020-06-08 23:15:13 +08:00
|
|
|
APInt MaskC = APInt(SrcWidth, 1).shl(*C);
|
[InstCombine] reverse 'trunc X to <N x i1>' canonicalization; 2nd try
Re-trying r344082 because it unintentionally included extra diffs.
Original commit message:
icmp ne (and X, 1), 0 --> trunc X to N x i1
Ideally, we'd do the same for scalars, but there will likely be
regressions unless we add more trunc folds as we're doing here
for vectors.
The motivating vector case is from PR37549:
https://bugs.llvm.org/show_bug.cgi?id=37549
define <4 x float> @bitwise_select(<4 x float> %x, <4 x float> %y, <4 x float> %z, <4 x float> %w) {
%c = fcmp ole <4 x float> %x, %y
%s = sext <4 x i1> %c to <4 x i32>
%s1 = shufflevector <4 x i32> %s, <4 x i32> undef, <4 x i32> <i32 0, i32 0, i32 1, i32 1>
%s2 = shufflevector <4 x i32> %s, <4 x i32> undef, <4 x i32> <i32 2, i32 2, i32 3, i32 3>
%cond = or <4 x i32> %s1, %s2
%condtr = trunc <4 x i32> %cond to <4 x i1>
%r = select <4 x i1> %condtr, <4 x float> %z, <4 x float> %w
ret <4 x float> %r
}
Here's a sampling of the vector codegen for that case using
mask+icmp (current behavior) vs. trunc (with this patch):
AVX before:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vandps LCPI0_0(%rip), %xmm0, %xmm0
vxorps %xmm1, %xmm1, %xmm1
vpcmpeqd %xmm1, %xmm0, %xmm0
vblendvps %xmm0, %xmm3, %xmm2, %xmm0
AVX after:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vblendvps %xmm0, %xmm2, %xmm3, %xmm0
AVX512f before:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vpbroadcastd LCPI0_0(%rip), %xmm1 ## xmm1 = [1,1,1,1]
vptestnmd %zmm1, %zmm0, %k1
vblendmps %zmm3, %zmm2, %zmm0 {%k1}
AVX512f after:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vpslld $31, %xmm0, %xmm0
vptestmd %zmm0, %zmm0, %k1
vblendmps %zmm2, %zmm3, %zmm0 {%k1}
AArch64 before:
fcmge v0.4s, v1.4s, v0.4s
zip1 v1.4s, v0.4s, v0.4s
zip2 v0.4s, v0.4s, v0.4s
orr v0.16b, v1.16b, v0.16b
movi v1.4s, #1
and v0.16b, v0.16b, v1.16b
cmeq v0.4s, v0.4s, #0
bsl v0.16b, v3.16b, v2.16b
AArch64 after:
fcmge v0.4s, v1.4s, v0.4s
zip1 v1.4s, v0.4s, v0.4s
zip2 v0.4s, v0.4s, v0.4s
orr v0.16b, v1.16b, v0.16b
bsl v0.16b, v2.16b, v3.16b
PowerPC-le before:
xvcmpgesp 34, 35, 34
vspltisw 0, 1
vmrglw 3, 2, 2
vmrghw 2, 2, 2
xxlor 0, 35, 34
xxlxor 35, 35, 35
xxland 34, 0, 32
vcmpequw 2, 2, 3
xxsel 34, 36, 37, 34
PowerPC-le after:
xvcmpgesp 34, 35, 34
vmrglw 3, 2, 2
vmrghw 2, 2, 2
xxlor 0, 35, 34
xxsel 34, 37, 36, 0
Differential Revision: https://reviews.llvm.org/D52747
llvm-svn: 344181
2018-10-11 04:47:46 +08:00
|
|
|
Value *And = Builder.CreateAnd(X, ConstantInt::get(SrcTy, MaskC));
|
|
|
|
return new ICmpInst(ICmpInst::ICMP_NE, And, Zero);
|
|
|
|
}
|
|
|
|
if (match(Src, m_OneUse(m_c_Or(m_LShr(m_Value(X), m_APInt(C)),
|
|
|
|
m_Deferred(X))))) {
|
|
|
|
// trunc (or (lshr X, C), X) to i1 --> icmp ne (and X, C'), 0
|
2020-06-08 23:15:13 +08:00
|
|
|
APInt MaskC = APInt(SrcWidth, 1).shl(*C) | 1;
|
[InstCombine] reverse 'trunc X to <N x i1>' canonicalization; 2nd try
Re-trying r344082 because it unintentionally included extra diffs.
Original commit message:
icmp ne (and X, 1), 0 --> trunc X to N x i1
Ideally, we'd do the same for scalars, but there will likely be
regressions unless we add more trunc folds as we're doing here
for vectors.
The motivating vector case is from PR37549:
https://bugs.llvm.org/show_bug.cgi?id=37549
define <4 x float> @bitwise_select(<4 x float> %x, <4 x float> %y, <4 x float> %z, <4 x float> %w) {
%c = fcmp ole <4 x float> %x, %y
%s = sext <4 x i1> %c to <4 x i32>
%s1 = shufflevector <4 x i32> %s, <4 x i32> undef, <4 x i32> <i32 0, i32 0, i32 1, i32 1>
%s2 = shufflevector <4 x i32> %s, <4 x i32> undef, <4 x i32> <i32 2, i32 2, i32 3, i32 3>
%cond = or <4 x i32> %s1, %s2
%condtr = trunc <4 x i32> %cond to <4 x i1>
%r = select <4 x i1> %condtr, <4 x float> %z, <4 x float> %w
ret <4 x float> %r
}
Here's a sampling of the vector codegen for that case using
mask+icmp (current behavior) vs. trunc (with this patch):
AVX before:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vandps LCPI0_0(%rip), %xmm0, %xmm0
vxorps %xmm1, %xmm1, %xmm1
vpcmpeqd %xmm1, %xmm0, %xmm0
vblendvps %xmm0, %xmm3, %xmm2, %xmm0
AVX after:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vblendvps %xmm0, %xmm2, %xmm3, %xmm0
AVX512f before:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vpbroadcastd LCPI0_0(%rip), %xmm1 ## xmm1 = [1,1,1,1]
vptestnmd %zmm1, %zmm0, %k1
vblendmps %zmm3, %zmm2, %zmm0 {%k1}
AVX512f after:
vcmpleps %xmm1, %xmm0, %xmm0
vpermilps $80, %xmm0, %xmm1 ## xmm1 = xmm0[0,0,1,1]
vpermilps $250, %xmm0, %xmm0 ## xmm0 = xmm0[2,2,3,3]
vorps %xmm0, %xmm1, %xmm0
vpslld $31, %xmm0, %xmm0
vptestmd %zmm0, %zmm0, %k1
vblendmps %zmm2, %zmm3, %zmm0 {%k1}
AArch64 before:
fcmge v0.4s, v1.4s, v0.4s
zip1 v1.4s, v0.4s, v0.4s
zip2 v0.4s, v0.4s, v0.4s
orr v0.16b, v1.16b, v0.16b
movi v1.4s, #1
and v0.16b, v0.16b, v1.16b
cmeq v0.4s, v0.4s, #0
bsl v0.16b, v3.16b, v2.16b
AArch64 after:
fcmge v0.4s, v1.4s, v0.4s
zip1 v1.4s, v0.4s, v0.4s
zip2 v0.4s, v0.4s, v0.4s
orr v0.16b, v1.16b, v0.16b
bsl v0.16b, v2.16b, v3.16b
PowerPC-le before:
xvcmpgesp 34, 35, 34
vspltisw 0, 1
vmrglw 3, 2, 2
vmrghw 2, 2, 2
xxlor 0, 35, 34
xxlxor 35, 35, 35
xxland 34, 0, 32
vcmpequw 2, 2, 3
xxsel 34, 36, 37, 34
PowerPC-le after:
xvcmpgesp 34, 35, 34
vmrglw 3, 2, 2
vmrghw 2, 2, 2
xxlor 0, 35, 34
xxsel 34, 37, 36, 0
Differential Revision: https://reviews.llvm.org/D52747
llvm-svn: 344181
2018-10-11 04:47:46 +08:00
|
|
|
Value *And = Builder.CreateAnd(X, ConstantInt::get(SrcTy, MaskC));
|
|
|
|
return new ICmpInst(ICmpInst::ICMP_NE, And, Zero);
|
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2017-05-10 00:24:59 +08:00
|
|
|
// FIXME: Maybe combine the next two transforms to handle the no cast case
|
|
|
|
// more efficiently. Support vector types. Cleanup code by using m_OneUse.
|
|
|
|
|
Add an instcombine to clean up a common pattern produced
by the SRoA "promote to large integer" code, eliminating
some type conversions like this:
%94 = zext i16 %93 to i32 ; <i32> [#uses=2]
%96 = lshr i32 %94, 8 ; <i32> [#uses=1]
%101 = trunc i32 %96 to i8 ; <i8> [#uses=1]
This also unblocks other xforms from happening, now clang is able to compile:
struct S { float A, B, C, D; };
float foo(struct S A) { return A.A + A.B+A.C+A.D; }
into:
_foo: ## @foo
## BB#0: ## %entry
pshufd $1, %xmm0, %xmm2
addss %xmm0, %xmm2
movdqa %xmm1, %xmm3
addss %xmm2, %xmm3
pshufd $1, %xmm1, %xmm0
addss %xmm3, %xmm0
ret
on x86-64, instead of:
_foo: ## @foo
## BB#0: ## %entry
movd %xmm0, %rax
shrq $32, %rax
movd %eax, %xmm2
addss %xmm0, %xmm2
movapd %xmm1, %xmm3
addss %xmm2, %xmm3
movd %xmm1, %rax
shrq $32, %rax
movd %eax, %xmm0
addss %xmm3, %xmm0
ret
This seems pretty close to optimal to me, at least without
using horizontal adds. This also triggers in lots of other
code, including SPEC.
llvm-svn: 112278
2010-08-28 02:31:05 +08:00
|
|
|
// Transform trunc(lshr (zext A), Cst) to eliminate one type conversion.
|
2020-03-30 03:48:54 +08:00
|
|
|
Value *A = nullptr;
|
implement an instcombine xform that canonicalizes casts outside of and-with-constant operations.
This fixes rdar://8808586 which observed that we used to compile:
union xy {
struct x { _Bool b[15]; } x;
__attribute__((packed))
struct y {
__attribute__((packed)) unsigned long b0to7;
__attribute__((packed)) unsigned int b8to11;
__attribute__((packed)) unsigned short b12to13;
__attribute__((packed)) unsigned char b14;
} y;
};
struct x
foo(union xy *xy)
{
return xy->x;
}
into:
_foo: ## @foo
movq (%rdi), %rax
movabsq $1095216660480, %rcx ## imm = 0xFF00000000
andq %rax, %rcx
movabsq $-72057594037927936, %rdx ## imm = 0xFF00000000000000
andq %rax, %rdx
movzbl %al, %esi
orq %rdx, %rsi
movq %rax, %rdx
andq $65280, %rdx ## imm = 0xFF00
orq %rsi, %rdx
movq %rax, %rsi
andq $16711680, %rsi ## imm = 0xFF0000
orq %rdx, %rsi
movl %eax, %edx
andl $-16777216, %edx ## imm = 0xFFFFFFFFFF000000
orq %rsi, %rdx
orq %rcx, %rdx
movabsq $280375465082880, %rcx ## imm = 0xFF0000000000
movq %rax, %rsi
andq %rcx, %rsi
orq %rdx, %rsi
movabsq $71776119061217280, %r8 ## imm = 0xFF000000000000
andq %r8, %rax
orq %rsi, %rax
movzwl 12(%rdi), %edx
movzbl 14(%rdi), %esi
shlq $16, %rsi
orl %edx, %esi
movq %rsi, %r9
shlq $32, %r9
movl 8(%rdi), %edx
orq %r9, %rdx
andq %rdx, %rcx
movzbl %sil, %esi
shlq $32, %rsi
orq %rcx, %rsi
movl %edx, %ecx
andl $-16777216, %ecx ## imm = 0xFFFFFFFFFF000000
orq %rsi, %rcx
movq %rdx, %rsi
andq $16711680, %rsi ## imm = 0xFF0000
orq %rcx, %rsi
movq %rdx, %rcx
andq $65280, %rcx ## imm = 0xFF00
orq %rsi, %rcx
movzbl %dl, %esi
orq %rcx, %rsi
andq %r8, %rdx
orq %rsi, %rdx
ret
We now compile this into:
_foo: ## @foo
## BB#0: ## %entry
movzwl 12(%rdi), %eax
movzbl 14(%rdi), %ecx
shlq $16, %rcx
orl %eax, %ecx
shlq $32, %rcx
movl 8(%rdi), %edx
orq %rcx, %rdx
movq (%rdi), %rax
ret
A small improvement :-)
llvm-svn: 123520
2011-01-15 14:32:33 +08:00
|
|
|
if (Src->hasOneUse() &&
|
|
|
|
match(Src, m_LShr(m_ZExt(m_Value(A)), m_ConstantInt(Cst)))) {
|
Add an instcombine to clean up a common pattern produced
by the SRoA "promote to large integer" code, eliminating
some type conversions like this:
%94 = zext i16 %93 to i32 ; <i32> [#uses=2]
%96 = lshr i32 %94, 8 ; <i32> [#uses=1]
%101 = trunc i32 %96 to i8 ; <i8> [#uses=1]
This also unblocks other xforms from happening, now clang is able to compile:
struct S { float A, B, C, D; };
float foo(struct S A) { return A.A + A.B+A.C+A.D; }
into:
_foo: ## @foo
## BB#0: ## %entry
pshufd $1, %xmm0, %xmm2
addss %xmm0, %xmm2
movdqa %xmm1, %xmm3
addss %xmm2, %xmm3
pshufd $1, %xmm1, %xmm0
addss %xmm3, %xmm0
ret
on x86-64, instead of:
_foo: ## @foo
## BB#0: ## %entry
movd %xmm0, %rax
shrq $32, %rax
movd %eax, %xmm2
addss %xmm0, %xmm2
movapd %xmm1, %xmm3
addss %xmm2, %xmm3
movd %xmm1, %rax
shrq $32, %rax
movd %eax, %xmm0
addss %xmm3, %xmm0
ret
This seems pretty close to optimal to me, at least without
using horizontal adds. This also triggers in lots of other
code, including SPEC.
llvm-svn: 112278
2010-08-28 02:31:05 +08:00
|
|
|
// We have three types to worry about here, the type of A, the source of
|
|
|
|
// the truncate (MidSize), and the destination of the truncate. We know that
|
|
|
|
// ASize < MidSize and MidSize > ResultSize, but don't know the relation
|
|
|
|
// between ASize and ResultSize.
|
|
|
|
unsigned ASize = A->getType()->getPrimitiveSizeInBits();
|
2013-01-24 13:22:40 +08:00
|
|
|
|
Add an instcombine to clean up a common pattern produced
by the SRoA "promote to large integer" code, eliminating
some type conversions like this:
%94 = zext i16 %93 to i32 ; <i32> [#uses=2]
%96 = lshr i32 %94, 8 ; <i32> [#uses=1]
%101 = trunc i32 %96 to i8 ; <i8> [#uses=1]
This also unblocks other xforms from happening, now clang is able to compile:
struct S { float A, B, C, D; };
float foo(struct S A) { return A.A + A.B+A.C+A.D; }
into:
_foo: ## @foo
## BB#0: ## %entry
pshufd $1, %xmm0, %xmm2
addss %xmm0, %xmm2
movdqa %xmm1, %xmm3
addss %xmm2, %xmm3
pshufd $1, %xmm1, %xmm0
addss %xmm3, %xmm0
ret
on x86-64, instead of:
_foo: ## @foo
## BB#0: ## %entry
movd %xmm0, %rax
shrq $32, %rax
movd %eax, %xmm2
addss %xmm0, %xmm2
movapd %xmm1, %xmm3
addss %xmm2, %xmm3
movd %xmm1, %rax
shrq $32, %rax
movd %eax, %xmm0
addss %xmm3, %xmm0
ret
This seems pretty close to optimal to me, at least without
using horizontal adds. This also triggers in lots of other
code, including SPEC.
llvm-svn: 112278
2010-08-28 02:31:05 +08:00
|
|
|
// If the shift amount is larger than the size of A, then the result is
|
|
|
|
// known to be zero because all the input bits got shifted out.
|
|
|
|
if (Cst->getZExtValue() >= ASize)
|
2020-06-08 23:15:13 +08:00
|
|
|
return replaceInstUsesWith(Trunc, Constant::getNullValue(DestTy));
|
Add an instcombine to clean up a common pattern produced
by the SRoA "promote to large integer" code, eliminating
some type conversions like this:
%94 = zext i16 %93 to i32 ; <i32> [#uses=2]
%96 = lshr i32 %94, 8 ; <i32> [#uses=1]
%101 = trunc i32 %96 to i8 ; <i8> [#uses=1]
This also unblocks other xforms from happening, now clang is able to compile:
struct S { float A, B, C, D; };
float foo(struct S A) { return A.A + A.B+A.C+A.D; }
into:
_foo: ## @foo
## BB#0: ## %entry
pshufd $1, %xmm0, %xmm2
addss %xmm0, %xmm2
movdqa %xmm1, %xmm3
addss %xmm2, %xmm3
pshufd $1, %xmm1, %xmm0
addss %xmm3, %xmm0
ret
on x86-64, instead of:
_foo: ## @foo
## BB#0: ## %entry
movd %xmm0, %rax
shrq $32, %rax
movd %eax, %xmm2
addss %xmm0, %xmm2
movapd %xmm1, %xmm3
addss %xmm2, %xmm3
movd %xmm1, %rax
shrq $32, %rax
movd %eax, %xmm0
addss %xmm3, %xmm0
ret
This seems pretty close to optimal to me, at least without
using horizontal adds. This also triggers in lots of other
code, including SPEC.
llvm-svn: 112278
2010-08-28 02:31:05 +08:00
|
|
|
|
|
|
|
// Since we're doing an lshr and a zero extend, and know that the shift
|
|
|
|
// amount is smaller than ASize, it is always safe to do the shift in A's
|
|
|
|
// type, then zero extend or truncate to the result.
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *Shift = Builder.CreateLShr(A, Cst->getZExtValue());
|
Add an instcombine to clean up a common pattern produced
by the SRoA "promote to large integer" code, eliminating
some type conversions like this:
%94 = zext i16 %93 to i32 ; <i32> [#uses=2]
%96 = lshr i32 %94, 8 ; <i32> [#uses=1]
%101 = trunc i32 %96 to i8 ; <i8> [#uses=1]
This also unblocks other xforms from happening, now clang is able to compile:
struct S { float A, B, C, D; };
float foo(struct S A) { return A.A + A.B+A.C+A.D; }
into:
_foo: ## @foo
## BB#0: ## %entry
pshufd $1, %xmm0, %xmm2
addss %xmm0, %xmm2
movdqa %xmm1, %xmm3
addss %xmm2, %xmm3
pshufd $1, %xmm1, %xmm0
addss %xmm3, %xmm0
ret
on x86-64, instead of:
_foo: ## @foo
## BB#0: ## %entry
movd %xmm0, %rax
shrq $32, %rax
movd %eax, %xmm2
addss %xmm0, %xmm2
movapd %xmm1, %xmm3
addss %xmm2, %xmm3
movd %xmm1, %rax
shrq $32, %rax
movd %eax, %xmm0
addss %xmm3, %xmm0
ret
This seems pretty close to optimal to me, at least without
using horizontal adds. This also triggers in lots of other
code, including SPEC.
llvm-svn: 112278
2010-08-28 02:31:05 +08:00
|
|
|
Shift->takeName(Src);
|
2015-11-18 02:37:23 +08:00
|
|
|
return CastInst::CreateIntegerCast(Shift, DestTy, false);
|
Add an instcombine to clean up a common pattern produced
by the SRoA "promote to large integer" code, eliminating
some type conversions like this:
%94 = zext i16 %93 to i32 ; <i32> [#uses=2]
%96 = lshr i32 %94, 8 ; <i32> [#uses=1]
%101 = trunc i32 %96 to i8 ; <i8> [#uses=1]
This also unblocks other xforms from happening, now clang is able to compile:
struct S { float A, B, C, D; };
float foo(struct S A) { return A.A + A.B+A.C+A.D; }
into:
_foo: ## @foo
## BB#0: ## %entry
pshufd $1, %xmm0, %xmm2
addss %xmm0, %xmm2
movdqa %xmm1, %xmm3
addss %xmm2, %xmm3
pshufd $1, %xmm1, %xmm0
addss %xmm3, %xmm0
ret
on x86-64, instead of:
_foo: ## @foo
## BB#0: ## %entry
movd %xmm0, %rax
shrq $32, %rax
movd %eax, %xmm2
addss %xmm0, %xmm2
movapd %xmm1, %xmm3
addss %xmm2, %xmm3
movd %xmm1, %rax
shrq $32, %rax
movd %eax, %xmm0
addss %xmm3, %xmm0
ret
This seems pretty close to optimal to me, at least without
using horizontal adds. This also triggers in lots of other
code, including SPEC.
llvm-svn: 112278
2010-08-28 02:31:05 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-06-08 23:37:43 +08:00
|
|
|
const APInt *C;
|
2020-06-09 02:09:04 +08:00
|
|
|
if (match(Src, m_LShr(m_SExt(m_Value(A)), m_APInt(C)))) {
|
|
|
|
unsigned AWidth = A->getType()->getScalarSizeInBits();
|
|
|
|
unsigned MaxShiftAmt = SrcWidth - std::max(DestWidth, AWidth);
|
2020-06-08 23:37:43 +08:00
|
|
|
|
2020-06-09 02:09:04 +08:00
|
|
|
// If the shift is small enough, all zero bits created by the shift are
|
|
|
|
// removed by the trunc.
|
|
|
|
if (C->getZExtValue() <= MaxShiftAmt) {
|
|
|
|
// trunc (lshr (sext A), C) --> ashr A, C
|
|
|
|
if (A->getType() == DestTy) {
|
|
|
|
unsigned ShAmt = std::min((unsigned)C->getZExtValue(), DestWidth - 1);
|
|
|
|
return BinaryOperator::CreateAShr(A, ConstantInt::get(DestTy, ShAmt));
|
|
|
|
}
|
|
|
|
// The types are mismatched, so create a cast after shifting:
|
|
|
|
// trunc (lshr (sext A), C) --> sext/trunc (ashr A, C)
|
|
|
|
if (Src->hasOneUse()) {
|
|
|
|
unsigned ShAmt = std::min((unsigned)C->getZExtValue(), AWidth - 1);
|
|
|
|
Value *Shift = Builder.CreateAShr(A, ShAmt);
|
2020-06-08 23:15:13 +08:00
|
|
|
return CastInst::CreateIntegerCast(Shift, DestTy, true);
|
2017-05-22 04:30:27 +08:00
|
|
|
}
|
2015-09-10 19:31:20 +08:00
|
|
|
}
|
2020-06-09 02:09:04 +08:00
|
|
|
// TODO: Mask high bits with 'and'.
|
2015-09-10 19:31:20 +08:00
|
|
|
}
|
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
if (Instruction *I = narrowBinOp(Trunc))
|
2016-12-01 04:48:54 +08:00
|
|
|
return I;
|
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
if (Instruction *I = shrinkSplatShuffle(Trunc, Builder))
|
2017-03-08 05:45:16 +08:00
|
|
|
return I;
|
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
if (Instruction *I = shrinkInsertElt(Trunc, Builder))
|
2017-03-08 07:27:14 +08:00
|
|
|
return I;
|
|
|
|
|
2015-11-18 02:37:23 +08:00
|
|
|
if (Src->hasOneUse() && isa<IntegerType>(SrcTy) &&
|
2017-02-01 01:25:42 +08:00
|
|
|
shouldChangeType(SrcTy, DestTy)) {
|
2016-09-14 03:43:57 +08:00
|
|
|
// Transform "trunc (shl X, cst)" -> "shl (trunc X), cst" so long as the
|
|
|
|
// dest type is native and cst < dest size.
|
|
|
|
if (match(Src, m_Shl(m_Value(A), m_ConstantInt(Cst))) &&
|
|
|
|
!match(A, m_Shr(m_Value(), m_Constant()))) {
|
|
|
|
// Skip shifts of shift by constants. It undoes a combine in
|
|
|
|
// FoldShiftByConstant and is the extend in reg pattern.
|
2020-06-08 23:15:13 +08:00
|
|
|
if (Cst->getValue().ult(DestWidth)) {
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *NewTrunc = Builder.CreateTrunc(A, DestTy, A->getName() + ".tr");
|
2016-09-14 03:43:57 +08:00
|
|
|
|
|
|
|
return BinaryOperator::Create(
|
|
|
|
Instruction::Shl, NewTrunc,
|
2020-06-08 23:15:13 +08:00
|
|
|
ConstantInt::get(DestTy, Cst->getValue().trunc(DestWidth)));
|
2016-09-14 03:43:57 +08:00
|
|
|
}
|
|
|
|
}
|
implement an instcombine xform that canonicalizes casts outside of and-with-constant operations.
This fixes rdar://8808586 which observed that we used to compile:
union xy {
struct x { _Bool b[15]; } x;
__attribute__((packed))
struct y {
__attribute__((packed)) unsigned long b0to7;
__attribute__((packed)) unsigned int b8to11;
__attribute__((packed)) unsigned short b12to13;
__attribute__((packed)) unsigned char b14;
} y;
};
struct x
foo(union xy *xy)
{
return xy->x;
}
into:
_foo: ## @foo
movq (%rdi), %rax
movabsq $1095216660480, %rcx ## imm = 0xFF00000000
andq %rax, %rcx
movabsq $-72057594037927936, %rdx ## imm = 0xFF00000000000000
andq %rax, %rdx
movzbl %al, %esi
orq %rdx, %rsi
movq %rax, %rdx
andq $65280, %rdx ## imm = 0xFF00
orq %rsi, %rdx
movq %rax, %rsi
andq $16711680, %rsi ## imm = 0xFF0000
orq %rdx, %rsi
movl %eax, %edx
andl $-16777216, %edx ## imm = 0xFFFFFFFFFF000000
orq %rsi, %rdx
orq %rcx, %rdx
movabsq $280375465082880, %rcx ## imm = 0xFF0000000000
movq %rax, %rsi
andq %rcx, %rsi
orq %rdx, %rsi
movabsq $71776119061217280, %r8 ## imm = 0xFF000000000000
andq %r8, %rax
orq %rsi, %rax
movzwl 12(%rdi), %edx
movzbl 14(%rdi), %esi
shlq $16, %rsi
orl %edx, %esi
movq %rsi, %r9
shlq $32, %r9
movl 8(%rdi), %edx
orq %r9, %rdx
andq %rdx, %rcx
movzbl %sil, %esi
shlq $32, %rsi
orq %rcx, %rsi
movl %edx, %ecx
andl $-16777216, %ecx ## imm = 0xFFFFFFFFFF000000
orq %rsi, %rcx
movq %rdx, %rsi
andq $16711680, %rsi ## imm = 0xFF0000
orq %rcx, %rsi
movq %rdx, %rcx
andq $65280, %rcx ## imm = 0xFF00
orq %rsi, %rcx
movzbl %dl, %esi
orq %rcx, %rsi
andq %r8, %rdx
orq %rsi, %rdx
ret
We now compile this into:
_foo: ## @foo
## BB#0: ## %entry
movzwl 12(%rdi), %eax
movzbl 14(%rdi), %ecx
shlq $16, %rcx
orl %eax, %ecx
shlq $32, %rcx
movl 8(%rdi), %edx
orq %rcx, %rdx
movq (%rdi), %rax
ret
A small improvement :-)
llvm-svn: 123520
2011-01-15 14:32:33 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2020-06-08 23:15:13 +08:00
|
|
|
if (Instruction *I = foldVecTruncToExtElt(Trunc, *this))
|
2015-12-15 00:16:54 +08:00
|
|
|
return I;
|
|
|
|
|
2020-03-30 03:48:54 +08:00
|
|
|
// Whenever an element is extracted from a vector, and then truncated,
|
|
|
|
// canonicalize by converting it to a bitcast followed by an
|
|
|
|
// extractelement.
|
|
|
|
//
|
|
|
|
// Example (little endian):
|
|
|
|
// trunc (extractelement <4 x i64> %X, 0) to i32
|
|
|
|
// --->
|
|
|
|
// extractelement <8 x i32> (bitcast <4 x i64> %X to <8 x i32>), i32 0
|
|
|
|
Value *VecOp;
|
2020-05-23 22:13:50 +08:00
|
|
|
if (match(Src, m_OneUse(m_ExtractElt(m_Value(VecOp), m_ConstantInt(Cst))))) {
|
2020-04-09 01:42:22 +08:00
|
|
|
auto *VecOpTy = cast<VectorType>(VecOp->getType());
|
|
|
|
unsigned VecNumElts = VecOpTy->getNumElements();
|
2020-03-30 03:48:54 +08:00
|
|
|
|
|
|
|
// A badly fit destination size would result in an invalid cast.
|
2020-06-08 23:15:13 +08:00
|
|
|
if (SrcWidth % DestWidth == 0) {
|
|
|
|
uint64_t TruncRatio = SrcWidth / DestWidth;
|
2020-03-30 03:48:54 +08:00
|
|
|
uint64_t BitCastNumElts = VecNumElts * TruncRatio;
|
|
|
|
uint64_t VecOpIdx = Cst->getZExtValue();
|
|
|
|
uint64_t NewIdx = DL.isBigEndian() ? (VecOpIdx + 1) * TruncRatio - 1
|
|
|
|
: VecOpIdx * TruncRatio;
|
|
|
|
assert(BitCastNumElts <= std::numeric_limits<uint32_t>::max() &&
|
|
|
|
"overflow 32-bits");
|
|
|
|
|
2020-05-30 06:24:15 +08:00
|
|
|
auto *BitCastTo = FixedVectorType::get(DestTy, BitCastNumElts);
|
2020-03-30 03:48:54 +08:00
|
|
|
Value *BitCast = Builder.CreateBitCast(VecOp, BitCastTo);
|
|
|
|
return ExtractElementInst::Create(BitCast, Builder.getInt32(NewIdx));
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2019-12-06 22:34:41 +08:00
|
|
|
Instruction *InstCombiner::transformZExtICmp(ICmpInst *Cmp, ZExtInst &Zext,
|
2016-07-19 17:06:08 +08:00
|
|
|
bool DoTransform) {
|
2010-01-04 15:53:58 +08:00
|
|
|
// If we are just checking for a icmp eq of a single bit and zext'ing it
|
|
|
|
// to an integer, then shift the bit to the appropriate place and then
|
|
|
|
// cast to integer to avoid the comparison.
|
2019-12-07 03:24:14 +08:00
|
|
|
const APInt *Op1CV;
|
|
|
|
if (match(Cmp->getOperand(1), m_APInt(Op1CV))) {
|
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// zext (x <s 0) to i32 --> x>>u31 true if signbit set.
|
|
|
|
// zext (x >s -1) to i32 --> (x>>u31)^1 true if signbit clear.
|
2019-12-07 03:24:14 +08:00
|
|
|
if ((Cmp->getPredicate() == ICmpInst::ICMP_SLT && Op1CV->isNullValue()) ||
|
|
|
|
(Cmp->getPredicate() == ICmpInst::ICMP_SGT && Op1CV->isAllOnesValue())) {
|
2019-12-06 22:34:41 +08:00
|
|
|
if (!DoTransform) return Cmp;
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
Value *In = Cmp->getOperand(0);
|
|
|
|
Value *Sh = ConstantInt::get(In->getType(),
|
|
|
|
In->getType()->getScalarSizeInBits() - 1);
|
|
|
|
In = Builder.CreateLShr(In, Sh, In->getName() + ".lobit");
|
|
|
|
if (In->getType() != Zext.getType())
|
|
|
|
In = Builder.CreateIntCast(In, Zext.getType(), false /*ZExt*/);
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
if (Cmp->getPredicate() == ICmpInst::ICMP_SGT) {
|
|
|
|
Constant *One = ConstantInt::get(In->getType(), 1);
|
2019-12-07 03:20:44 +08:00
|
|
|
In = Builder.CreateXor(In, One, In->getName() + ".not");
|
|
|
|
}
|
|
|
|
|
|
|
|
return replaceInstUsesWith(Zext, In);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2011-11-30 09:59:59 +08:00
|
|
|
|
2012-09-27 18:14:43 +08:00
|
|
|
// zext (X == 0) to i32 --> X^1 iff X has only the low bit set.
|
|
|
|
// zext (X == 0) to i32 --> (X>>1)^1 iff X has only the 2nd bit set.
|
|
|
|
// zext (X == 1) to i32 --> X iff X has only the low bit set.
|
|
|
|
// zext (X == 2) to i32 --> X>>1 iff X has only the 2nd bit set.
|
|
|
|
// zext (X != 0) to i32 --> X iff X has only the low bit set.
|
|
|
|
// zext (X != 0) to i32 --> X>>1 iff X has only the 2nd bit set.
|
|
|
|
// zext (X != 1) to i32 --> X^1 iff X has only the low bit set.
|
|
|
|
// zext (X != 2) to i32 --> (X>>1)^1 iff X has only the 2nd bit set.
|
2019-12-07 03:24:14 +08:00
|
|
|
if ((Op1CV->isNullValue() || Op1CV->isPowerOf2()) &&
|
2010-01-04 15:53:58 +08:00
|
|
|
// This only works for EQ and NE
|
2019-12-06 22:34:41 +08:00
|
|
|
Cmp->isEquality()) {
|
2010-01-04 15:53:58 +08:00
|
|
|
// If Op1C some other power of two, convert:
|
2019-12-07 03:24:14 +08:00
|
|
|
KnownBits Known = computeKnownBits(Cmp->getOperand(0), 0, &Zext);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2017-04-27 00:39:58 +08:00
|
|
|
APInt KnownZeroMask(~Known.Zero);
|
2010-01-04 15:53:58 +08:00
|
|
|
if (KnownZeroMask.isPowerOf2()) { // Exactly 1 possible 1?
|
2019-12-06 22:34:41 +08:00
|
|
|
if (!DoTransform) return Cmp;
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
bool isNE = Cmp->getPredicate() == ICmpInst::ICMP_NE;
|
|
|
|
if (!Op1CV->isNullValue() && (*Op1CV != KnownZeroMask)) {
|
2010-01-04 15:53:58 +08:00
|
|
|
// (X&4) == 2 --> false
|
|
|
|
// (X&4) != 2 --> true
|
2019-12-07 03:24:14 +08:00
|
|
|
Constant *Res = ConstantInt::get(Zext.getType(), isNE);
|
2019-12-06 22:34:41 +08:00
|
|
|
return replaceInstUsesWith(Zext, Res);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2015-12-31 02:31:30 +08:00
|
|
|
uint32_t ShAmt = KnownZeroMask.logBase2();
|
2019-12-07 03:24:14 +08:00
|
|
|
Value *In = Cmp->getOperand(0);
|
2015-12-31 02:31:30 +08:00
|
|
|
if (ShAmt) {
|
2010-01-04 15:53:58 +08:00
|
|
|
// Perform a logical shr by shiftamt.
|
|
|
|
// Insert the shift to put the result in the low bit.
|
2019-12-07 03:24:14 +08:00
|
|
|
In = Builder.CreateLShr(In, ConstantInt::get(In->getType(), ShAmt),
|
2017-07-08 07:16:26 +08:00
|
|
|
In->getName() + ".lobit");
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
if (!Op1CV->isNullValue() == isNE) { // Toggle the low bit.
|
|
|
|
Constant *One = ConstantInt::get(In->getType(), 1);
|
2017-07-08 07:16:26 +08:00
|
|
|
In = Builder.CreateXor(In, One);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
if (Zext.getType() == In->getType())
|
2019-12-06 22:34:41 +08:00
|
|
|
return replaceInstUsesWith(Zext, In);
|
[InstCombine] Refactor optimization of zext(or(icmp, icmp)) to enable more aggressive cast-folding
Summary:
InstCombine unfolds expressions of the form `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` such that in a later iteration of InstCombine the exposed `zext(icmp)` instructions can be optimized. We now combine this unfolding and the subsequent `zext(icmp)` optimization to be performed together. Since the unfolding doesn't happen separately anymore, we also again enable the folding of `logic(cast(icmp), cast(icmp))` expressions to `cast(logic(icmp, icmp))` which had been disabled due to its interference with the unfolding transformation.
Tested via `make check` and `lnt`.
Background
==========
For a better understanding on how it came to this change we subsequently summarize its history. In commit r275989 we've already tried to enable the folding of `logic(cast(icmp), cast(icmp))` to `cast(logic(icmp, icmp))` which had to be reverted in r276106 because it could lead to an endless loop in InstCombine (also see http://lists.llvm.org/pipermail/llvm-commits/Week-of-Mon-20160718/374347.html). The root of this problem is that in `visitZExt()` in InstCombineCasts.cpp there also exists a reverse of the above folding transformation, that unfolds `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` in order to expose `zext(icmp)` operations which would then possibly be eliminated by subsequent iterations of InstCombine. However, before these `zext(icmp)` would be eliminated the folding from r275989 could kick in and cause InstCombine to endlessly switch back and forth between the folding and the unfolding transformation. This is the reason why we now combine the `zext`-unfolding and the elimination of the exposed `zext(icmp)` to happen at one go because this enables us to still allow the cast-folding in `logic(cast(icmp), cast(icmp))` without entering an endless loop again.
Details on the submitted changes
================================
- In `visitZExt()` we combine the unfolding and optimization of `zext` instructions.
- In `transformZExtICmp()` we have to use `Builder->CreateIntCast()` instead of `CastInst::CreateIntegerCast()` to make sure that the new `CastInst` is inserted in a `BasicBlock`. The new calls to `transformZExtICmp()` that we introduce in `visitZExt()` would otherwise cause according assertions to be triggered (in our case this happend, for example, with lnt for the MultiSource/Applications/sqlite3 and SingleSource/Regression/C++/EH/recursive-throw tests). The subsequent usage of `replaceInstUsesWith()` is necessary to ensure that the new `CastInst` replaces the `ZExtInst` accordingly.
- In InstCombineAndOrXor.cpp we again allow the folding of casts on `icmp` instructions.
- The instruction order in the optimized IR for the zext-or-icmp.ll test case is different with the introduced changes.
- The test cases in zext.ll have been adopted from the reverted commits r275989 and r276105.
Reviewers: grosser, majnemer, spatel
Subscribers: eli.friedman, majnemer, llvm-commits
Differential Revision: https://reviews.llvm.org/D22864
Contributed-by: Matthias Reisinger <d412vv1n@gmail.com>
llvm-svn: 277635
2016-08-04 03:30:35 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
Value *IntCast = Builder.CreateIntCast(In, Zext.getType(), false);
|
2019-12-06 22:34:41 +08:00
|
|
|
return replaceInstUsesWith(Zext, IntCast);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2020-01-08 18:21:21 +08:00
|
|
|
// icmp ne A, B is equal to xor A, B when A and B only really have one bit.
|
|
|
|
// It is also profitable to transform icmp eq into not(xor(A, B)) because that
|
|
|
|
// may lead to additional simplifications.
|
2019-12-07 03:24:14 +08:00
|
|
|
if (Cmp->isEquality() && Zext.getType() == Cmp->getOperand(0)->getType()) {
|
|
|
|
if (IntegerType *ITy = dyn_cast<IntegerType>(Zext.getType())) {
|
|
|
|
Value *LHS = Cmp->getOperand(0);
|
|
|
|
Value *RHS = Cmp->getOperand(1);
|
|
|
|
|
|
|
|
KnownBits KnownLHS = computeKnownBits(LHS, 0, &Zext);
|
|
|
|
KnownBits KnownRHS = computeKnownBits(RHS, 0, &Zext);
|
2019-12-07 03:19:02 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
if (KnownLHS.Zero == KnownRHS.Zero && KnownLHS.One == KnownRHS.One) {
|
|
|
|
APInt KnownBits = KnownLHS.Zero | KnownLHS.One;
|
2019-12-07 03:19:02 +08:00
|
|
|
APInt UnknownBit = ~KnownBits;
|
|
|
|
if (UnknownBit.countPopulation() == 1) {
|
|
|
|
if (!DoTransform) return Cmp;
|
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
Value *Result = Builder.CreateXor(LHS, RHS);
|
2019-12-07 03:19:02 +08:00
|
|
|
|
|
|
|
// Mask off any bits that are set and won't be shifted away.
|
2019-12-07 03:24:14 +08:00
|
|
|
if (KnownLHS.One.uge(UnknownBit))
|
2019-12-07 03:19:02 +08:00
|
|
|
Result = Builder.CreateAnd(Result,
|
2019-12-07 03:24:14 +08:00
|
|
|
ConstantInt::get(ITy, UnknownBit));
|
2019-12-07 03:19:02 +08:00
|
|
|
|
|
|
|
// Shift the bit we're testing down to the lsb.
|
|
|
|
Result = Builder.CreateLShr(
|
2019-12-07 03:24:14 +08:00
|
|
|
Result, ConstantInt::get(ITy, UnknownBit.countTrailingZeros()));
|
2019-12-07 03:19:02 +08:00
|
|
|
|
2019-12-07 03:24:14 +08:00
|
|
|
if (Cmp->getPredicate() == ICmpInst::ICMP_EQ)
|
2019-12-07 03:19:02 +08:00
|
|
|
Result = Builder.CreateXor(Result, ConstantInt::get(ITy, 1));
|
|
|
|
Result->takeName(Cmp);
|
|
|
|
return replaceInstUsesWith(Zext, Result);
|
|
|
|
}
|
|
|
|
}
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Determine if the specified value can be computed in the specified wider type
|
|
|
|
/// and produce the same low bits. If not, return false.
|
2010-01-11 10:43:35 +08:00
|
|
|
///
|
2010-01-11 11:32:00 +08:00
|
|
|
/// If this function returns true, it can also return a non-zero number of bits
|
|
|
|
/// (in BitsToClear) which indicates that the value it computes is correct for
|
|
|
|
/// the zero extend, but that the additional BitsToClear bits need to be zero'd
|
|
|
|
/// out. For example, to promote something like:
|
|
|
|
///
|
|
|
|
/// %B = trunc i64 %A to i32
|
|
|
|
/// %C = lshr i32 %B, 8
|
|
|
|
/// %E = zext i32 %C to i64
|
|
|
|
///
|
|
|
|
/// CanEvaluateZExtd for the 'lshr' will return true, and BitsToClear will be
|
|
|
|
/// set to 8 to indicate that the promoted value needs to have bits 24-31
|
|
|
|
/// cleared in addition to bits 32-63. Since an 'and' will be generated to
|
|
|
|
/// clear the top bits anyway, doing this has no extra cost.
|
|
|
|
///
|
2010-01-11 10:43:35 +08:00
|
|
|
/// This function works on both vectors and scalars.
|
2015-09-09 22:54:29 +08:00
|
|
|
static bool canEvaluateZExtd(Value *V, Type *Ty, unsigned &BitsToClear,
|
Make use of @llvm.assume in ValueTracking (computeKnownBits, etc.)
This change, which allows @llvm.assume to be used from within computeKnownBits
(and other associated functions in ValueTracking), adds some (optional)
parameters to computeKnownBits and friends. These functions now (optionally)
take a "context" instruction pointer, an AssumptionTracker pointer, and also a
DomTree pointer, and most of the changes are just to pass this new information
when it is easily available from InstSimplify, InstCombine, etc.
As explained below, the significant conceptual change is that known properties
of a value might depend on the control-flow location of the use (because we
care that the @llvm.assume dominates the use because assumptions have
control-flow dependencies). This means that, when we ask if bits are known in a
value, we might get different answers for different uses.
The significant changes are all in ValueTracking. Two main changes: First, as
with the rest of the code, new parameters need to be passed around. To make
this easier, I grouped them into a structure, and I made internal static
versions of the relevant functions that take this structure as a parameter. The
new code does as you might expect, it looks for @llvm.assume calls that make
use of the value we're trying to learn something about (often indirectly),
attempts to pattern match that expression, and uses the result if successful.
By making use of the AssumptionTracker, the process of finding @llvm.assume
calls is not expensive.
Part of the structure being passed around inside ValueTracking is a set of
already-considered @llvm.assume calls. This is to prevent a query using, for
example, the assume(a == b), to recurse on itself. The context and DT params
are used to find applicable assumptions. An assumption needs to dominate the
context instruction, or come after it deterministically. In this latter case we
only handle the specific case where both the assumption and the context
instruction are in the same block, and we need to exclude assumptions from
being used to simplify their own ephemeral values (those which contribute only
to the assumption) because otherwise the assumption would prove its feeding
comparison trivial and would be removed.
This commit adds the plumbing and the logic for a simple masked-bit propagation
(just enough to write a regression test). Future commits add more patterns
(and, correspondingly, more regression tests).
llvm-svn: 217342
2014-09-08 02:57:58 +08:00
|
|
|
InstCombiner &IC, Instruction *CxtI) {
|
2010-01-11 11:32:00 +08:00
|
|
|
BitsToClear = 0;
|
2018-01-31 22:55:53 +08:00
|
|
|
if (canAlwaysEvaluateInType(V, Ty))
|
2010-01-10 10:50:04 +08:00
|
|
|
return true;
|
2018-01-31 22:55:53 +08:00
|
|
|
if (canNotEvaluateInType(V, Ty))
|
|
|
|
return false;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-01-31 22:55:53 +08:00
|
|
|
auto *I = cast<Instruction>(V);
|
|
|
|
unsigned Tmp;
|
|
|
|
switch (I->getOpcode()) {
|
2010-01-11 04:25:54 +08:00
|
|
|
case Instruction::ZExt: // zext(zext(x)) -> zext(x).
|
|
|
|
case Instruction::SExt: // zext(sext(x)) -> sext(x).
|
|
|
|
case Instruction::Trunc: // zext(trunc(x)) -> trunc(x) or zext(x)
|
|
|
|
return true;
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::And:
|
|
|
|
case Instruction::Or:
|
|
|
|
case Instruction::Xor:
|
|
|
|
case Instruction::Add:
|
|
|
|
case Instruction::Sub:
|
|
|
|
case Instruction::Mul:
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateZExtd(I->getOperand(0), Ty, BitsToClear, IC, CxtI) ||
|
|
|
|
!canEvaluateZExtd(I->getOperand(1), Ty, Tmp, IC, CxtI))
|
2010-01-11 11:32:00 +08:00
|
|
|
return false;
|
|
|
|
// These can all be promoted if neither operand has 'bits to clear'.
|
|
|
|
if (BitsToClear == 0 && Tmp == 0)
|
|
|
|
return true;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
// If the operation is an AND/OR/XOR and the bits to clear are zero in the
|
|
|
|
// other side, BitsToClear is ok.
|
2016-11-23 06:54:36 +08:00
|
|
|
if (Tmp == 0 && I->isBitwiseLogicOp()) {
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
// We use MaskedValueIsZero here for generality, but the case we care
|
|
|
|
// about the most is constant RHS.
|
|
|
|
unsigned VSize = V->getType()->getScalarSizeInBits();
|
Make use of @llvm.assume in ValueTracking (computeKnownBits, etc.)
This change, which allows @llvm.assume to be used from within computeKnownBits
(and other associated functions in ValueTracking), adds some (optional)
parameters to computeKnownBits and friends. These functions now (optionally)
take a "context" instruction pointer, an AssumptionTracker pointer, and also a
DomTree pointer, and most of the changes are just to pass this new information
when it is easily available from InstSimplify, InstCombine, etc.
As explained below, the significant conceptual change is that known properties
of a value might depend on the control-flow location of the use (because we
care that the @llvm.assume dominates the use because assumptions have
control-flow dependencies). This means that, when we ask if bits are known in a
value, we might get different answers for different uses.
The significant changes are all in ValueTracking. Two main changes: First, as
with the rest of the code, new parameters need to be passed around. To make
this easier, I grouped them into a structure, and I made internal static
versions of the relevant functions that take this structure as a parameter. The
new code does as you might expect, it looks for @llvm.assume calls that make
use of the value we're trying to learn something about (often indirectly),
attempts to pattern match that expression, and uses the result if successful.
By making use of the AssumptionTracker, the process of finding @llvm.assume
calls is not expensive.
Part of the structure being passed around inside ValueTracking is a set of
already-considered @llvm.assume calls. This is to prevent a query using, for
example, the assume(a == b), to recurse on itself. The context and DT params
are used to find applicable assumptions. An assumption needs to dominate the
context instruction, or come after it deterministically. In this latter case we
only handle the specific case where both the assumption and the context
instruction are in the same block, and we need to exclude assumptions from
being used to simplify their own ephemeral values (those which contribute only
to the assumption) because otherwise the assumption would prove its feeding
comparison trivial and would be removed.
This commit adds the plumbing and the logic for a simple masked-bit propagation
(just enough to write a regression test). Future commits add more patterns
(and, correspondingly, more regression tests).
llvm-svn: 217342
2014-09-08 02:57:58 +08:00
|
|
|
if (IC.MaskedValueIsZero(I->getOperand(1),
|
|
|
|
APInt::getHighBitsSet(VSize, BitsToClear),
|
2017-08-22 00:04:11 +08:00
|
|
|
0, CxtI)) {
|
|
|
|
// If this is an And instruction and all of the BitsToClear are
|
|
|
|
// known to be zero we can reset BitsToClear.
|
2018-01-31 22:55:53 +08:00
|
|
|
if (I->getOpcode() == Instruction::And)
|
2017-08-22 00:04:11 +08:00
|
|
|
BitsToClear = 0;
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
return true;
|
2017-08-22 00:04:11 +08:00
|
|
|
}
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
// Otherwise, we don't know how to analyze this BitsToClear case yet.
|
2010-01-11 11:32:00 +08:00
|
|
|
return false;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2017-08-16 06:48:41 +08:00
|
|
|
case Instruction::Shl: {
|
2013-05-11 00:26:37 +08:00
|
|
|
// We can promote shl(x, cst) if we can promote x. Since shl overwrites the
|
|
|
|
// upper bits we can reduce BitsToClear by the shift amount.
|
2017-08-16 06:48:41 +08:00
|
|
|
const APInt *Amt;
|
|
|
|
if (match(I->getOperand(1), m_APInt(Amt))) {
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateZExtd(I->getOperand(0), Ty, BitsToClear, IC, CxtI))
|
2013-05-11 00:26:37 +08:00
|
|
|
return false;
|
|
|
|
uint64_t ShiftAmt = Amt->getZExtValue();
|
|
|
|
BitsToClear = ShiftAmt < BitsToClear ? BitsToClear - ShiftAmt : 0;
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
return false;
|
2017-08-16 06:48:41 +08:00
|
|
|
}
|
|
|
|
case Instruction::LShr: {
|
2010-01-11 11:32:00 +08:00
|
|
|
// We can promote lshr(x, cst) if we can promote x. This requires the
|
|
|
|
// ultimate 'and' to clear out the high zero bits we're clearing out though.
|
2017-08-16 06:48:41 +08:00
|
|
|
const APInt *Amt;
|
|
|
|
if (match(I->getOperand(1), m_APInt(Amt))) {
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateZExtd(I->getOperand(0), Ty, BitsToClear, IC, CxtI))
|
2010-01-11 11:32:00 +08:00
|
|
|
return false;
|
|
|
|
BitsToClear += Amt->getZExtValue();
|
|
|
|
if (BitsToClear > V->getType()->getScalarSizeInBits())
|
|
|
|
BitsToClear = V->getType()->getScalarSizeInBits();
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
// Cannot promote variable LSHR.
|
|
|
|
return false;
|
2017-08-16 06:48:41 +08:00
|
|
|
}
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::Select:
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateZExtd(I->getOperand(1), Ty, Tmp, IC, CxtI) ||
|
|
|
|
!canEvaluateZExtd(I->getOperand(2), Ty, BitsToClear, IC, CxtI) ||
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
// TODO: If important, we could handle the case when the BitsToClear are
|
|
|
|
// known zero in the disagreeing side.
|
2010-01-11 11:32:00 +08:00
|
|
|
Tmp != BitsToClear)
|
|
|
|
return false;
|
|
|
|
return true;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::PHI: {
|
|
|
|
// We can change a phi if we can change all operands. Note that we never
|
|
|
|
// get into trouble with cyclic PHIs here because we only consider
|
|
|
|
// instructions with a single use.
|
|
|
|
PHINode *PN = cast<PHINode>(I);
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateZExtd(PN->getIncomingValue(0), Ty, BitsToClear, IC, CxtI))
|
2010-01-11 11:32:00 +08:00
|
|
|
return false;
|
2010-01-10 10:50:04 +08:00
|
|
|
for (unsigned i = 1, e = PN->getNumIncomingValues(); i != e; ++i)
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateZExtd(PN->getIncomingValue(i), Ty, Tmp, IC, CxtI) ||
|
Extend CanEvaluateZExtd to handle and/or/xor more aggressively in the
BitsToClear case. This allows it to promote expressions which have an
and/or/xor after the lshr, promoting cases like test2 (from PR4216)
and test3 (random extample extracted from a spec benchmark).
clang now compiles the code in PR4216 into:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
orl $32768, %edi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
instead of:
_test_bitfield: ## @test_bitfield
movl %edi, %eax
orl $194, %eax
movl $4294902010, %ecx
andq %rax, %rcx
shrl $8, %edi
orl $128, %edi
shlq $8, %rdi
andq $39936, %rdi
movq %rdi, %rax
orq %rcx, %rax
ret
which is still not great, but is progress.
llvm-svn: 93145
2010-01-11 12:05:13 +08:00
|
|
|
// TODO: If important, we could handle the case when the BitsToClear
|
|
|
|
// are known zero in the disagreeing input.
|
2010-01-11 11:32:00 +08:00
|
|
|
Tmp != BitsToClear)
|
|
|
|
return false;
|
2010-01-10 10:50:04 +08:00
|
|
|
return true;
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
default:
|
|
|
|
// TODO: Can handle more cases here.
|
2010-01-10 10:50:04 +08:00
|
|
|
return false;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
Instruction *InstCombiner::visitZExt(ZExtInst &CI) {
|
2013-01-15 04:56:10 +08:00
|
|
|
// If this zero extend is only used by a truncate, let the truncate be
|
2010-01-10 10:39:31 +08:00
|
|
|
// eliminated before we try to optimize this zext.
|
2014-03-09 11:16:01 +08:00
|
|
|
if (CI.hasOneUse() && isa<TruncInst>(CI.user_back()))
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// If one of the common conversion will work, do it.
|
2010-01-10 09:00:46 +08:00
|
|
|
if (Instruction *Result = commonCastTransforms(CI))
|
2010-01-04 15:53:58 +08:00
|
|
|
return Result;
|
|
|
|
|
2010-01-10 09:00:46 +08:00
|
|
|
Value *Src = CI.getOperand(0);
|
2011-07-18 12:54:35 +08:00
|
|
|
Type *SrcTy = Src->getType(), *DestTy = CI.getType();
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-12-18 04:27:43 +08:00
|
|
|
// Try to extend the entire expression tree to the wide destination type.
|
2010-01-11 11:32:00 +08:00
|
|
|
unsigned BitsToClear;
|
2018-12-18 04:27:43 +08:00
|
|
|
if (shouldChangeType(SrcTy, DestTy) &&
|
2015-09-09 22:54:29 +08:00
|
|
|
canEvaluateZExtd(Src, DestTy, BitsToClear, *this, &CI)) {
|
[InstCombine] Liberate assert in InstCombiner::visitZExt
Summary:
The call to canEvaluateZExtd in InstCombiner::visitZExt may
return with BitsToClear == SrcTy->getScalarSizeInBits(), but
there is an assert that BitsToClear should be smaller than
SrcTy->getScalarSizeInBits().
I have a test case that triggers the assert, but it only happens
for my downstream target. I've not been able to trigger it for
any upstream target.
The assert triggered for a piece of code such as this
%shr1 = lshr i16 undef, 15
...
%shr2 = lshr i16 %shr1, 1
%conv = zext i16 %shr2 to i32
Normally the lshr instructions are constant folded before we
visit the zext (that is why it is so hard to reproduce).
The original pattern, before instcombine, is of course a lot more
complicated in my test case. The shift count in the second lshr
is for example determined by the outcome of a PHI instruction.
It seems like other rewrites by instcombine leads up to
the pattern above. And then the zext is pulled from the
worklist, and visited (hitting the assert), before we detect
that the lshr instrucions can be constant folded.
Anyway, since the canEvaluateZExtd may return with BitsToClear
equal to SrcTy->getScalarSizeInBits(), and since the rewrite
that converts the expression type to avoid a zero extend works
also for the case where SrcBitsKept ends up being zero, then
it should be OK to liberate the assert to
assert(BitsToClear <= SrcTy->getScalarSizeInBits() &&
"Unreasonable BitsToClear");
Reviewers: hfinkel
Reviewed By: hfinkel
Subscribers: hfinkel, llvm-commits
Differential Revision: https://reviews.llvm.org/D30993
llvm-svn: 297952
2017-03-16 21:22:01 +08:00
|
|
|
assert(BitsToClear <= SrcTy->getScalarSizeInBits() &&
|
|
|
|
"Can't clear more bits than in SrcTy");
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 10:39:31 +08:00
|
|
|
// Okay, we can transform this! Insert the new expression now.
|
2018-05-14 20:53:11 +08:00
|
|
|
LLVM_DEBUG(
|
|
|
|
dbgs() << "ICE: EvaluateInDifferentType converting expression type"
|
|
|
|
" to avoid zero extend: "
|
|
|
|
<< CI << '\n');
|
2010-01-10 10:39:31 +08:00
|
|
|
Value *Res = EvaluateInDifferentType(Src, DestTy, false);
|
|
|
|
assert(Res->getType() == DestTy);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-07-07 01:32:39 +08:00
|
|
|
// Preserve debug values referring to Src if the zext is its last use.
|
|
|
|
if (auto *SrcOp = dyn_cast<Instruction>(Src))
|
|
|
|
if (SrcOp->hasOneUse())
|
|
|
|
replaceAllDbgUsesWith(*SrcOp, *Res, CI, DT);
|
2018-07-04 17:55:46 +08:00
|
|
|
|
2010-01-11 11:32:00 +08:00
|
|
|
uint32_t SrcBitsKept = SrcTy->getScalarSizeInBits()-BitsToClear;
|
|
|
|
uint32_t DestBitSize = DestTy->getScalarSizeInBits();
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 10:39:31 +08:00
|
|
|
// If the high bits are already filled with zeros, just replace this
|
|
|
|
// cast with the result.
|
Make use of @llvm.assume in ValueTracking (computeKnownBits, etc.)
This change, which allows @llvm.assume to be used from within computeKnownBits
(and other associated functions in ValueTracking), adds some (optional)
parameters to computeKnownBits and friends. These functions now (optionally)
take a "context" instruction pointer, an AssumptionTracker pointer, and also a
DomTree pointer, and most of the changes are just to pass this new information
when it is easily available from InstSimplify, InstCombine, etc.
As explained below, the significant conceptual change is that known properties
of a value might depend on the control-flow location of the use (because we
care that the @llvm.assume dominates the use because assumptions have
control-flow dependencies). This means that, when we ask if bits are known in a
value, we might get different answers for different uses.
The significant changes are all in ValueTracking. Two main changes: First, as
with the rest of the code, new parameters need to be passed around. To make
this easier, I grouped them into a structure, and I made internal static
versions of the relevant functions that take this structure as a parameter. The
new code does as you might expect, it looks for @llvm.assume calls that make
use of the value we're trying to learn something about (often indirectly),
attempts to pattern match that expression, and uses the result if successful.
By making use of the AssumptionTracker, the process of finding @llvm.assume
calls is not expensive.
Part of the structure being passed around inside ValueTracking is a set of
already-considered @llvm.assume calls. This is to prevent a query using, for
example, the assume(a == b), to recurse on itself. The context and DT params
are used to find applicable assumptions. An assumption needs to dominate the
context instruction, or come after it deterministically. In this latter case we
only handle the specific case where both the assumption and the context
instruction are in the same block, and we need to exclude assumptions from
being used to simplify their own ephemeral values (those which contribute only
to the assumption) because otherwise the assumption would prove its feeding
comparison trivial and would be removed.
This commit adds the plumbing and the logic for a simple masked-bit propagation
(just enough to write a regression test). Future commits add more patterns
(and, correspondingly, more regression tests).
llvm-svn: 217342
2014-09-08 02:57:58 +08:00
|
|
|
if (MaskedValueIsZero(Res,
|
|
|
|
APInt::getHighBitsSet(DestBitSize,
|
|
|
|
DestBitSize-SrcBitsKept),
|
|
|
|
0, &CI))
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, Res);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 10:39:31 +08:00
|
|
|
// We need to emit an AND to clear the high bits.
|
2010-01-11 04:25:54 +08:00
|
|
|
Constant *C = ConstantInt::get(Res->getType(),
|
2010-01-11 11:32:00 +08:00
|
|
|
APInt::getLowBitsSet(DestBitSize, SrcBitsKept));
|
2010-01-10 10:39:31 +08:00
|
|
|
return BinaryOperator::CreateAnd(Res, C);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
|
|
|
// If this is a TRUNC followed by a ZEXT then we are dealing with integral
|
|
|
|
// types and if the sizes are just right we can convert this into a logical
|
|
|
|
// 'and' which will be much cheaper than the pair of casts.
|
|
|
|
if (TruncInst *CSrc = dyn_cast<TruncInst>(Src)) { // A->B->C cast
|
2010-01-10 15:08:30 +08:00
|
|
|
// TODO: Subsume this into EvaluateInDifferentType.
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// Get the sizes of the types involved. We know that the intermediate type
|
|
|
|
// will be smaller than A or C, but don't know the relation between A and C.
|
|
|
|
Value *A = CSrc->getOperand(0);
|
|
|
|
unsigned SrcSize = A->getType()->getScalarSizeInBits();
|
|
|
|
unsigned MidSize = CSrc->getType()->getScalarSizeInBits();
|
|
|
|
unsigned DstSize = CI.getType()->getScalarSizeInBits();
|
|
|
|
// If we're actually extending zero bits, then if
|
|
|
|
// SrcSize < DstSize: zext(a & mask)
|
|
|
|
// SrcSize == DstSize: a & mask
|
|
|
|
// SrcSize > DstSize: trunc(a) & mask
|
|
|
|
if (SrcSize < DstSize) {
|
|
|
|
APInt AndValue(APInt::getLowBitsSet(SrcSize, MidSize));
|
|
|
|
Constant *AndConst = ConstantInt::get(A->getType(), AndValue);
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *And = Builder.CreateAnd(A, AndConst, CSrc->getName() + ".mask");
|
2010-01-04 15:53:58 +08:00
|
|
|
return new ZExtInst(And, CI.getType());
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
if (SrcSize == DstSize) {
|
|
|
|
APInt AndValue(APInt::getLowBitsSet(SrcSize, MidSize));
|
|
|
|
return BinaryOperator::CreateAnd(A, ConstantInt::get(A->getType(),
|
|
|
|
AndValue));
|
|
|
|
}
|
|
|
|
if (SrcSize > DstSize) {
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *Trunc = Builder.CreateTrunc(A, CI.getType());
|
2010-01-04 15:53:58 +08:00
|
|
|
APInt AndValue(APInt::getLowBitsSet(DstSize, MidSize));
|
2013-01-24 13:22:40 +08:00
|
|
|
return BinaryOperator::CreateAnd(Trunc,
|
2010-01-04 15:53:58 +08:00
|
|
|
ConstantInt::get(Trunc->getType(),
|
2010-01-10 15:08:30 +08:00
|
|
|
AndValue));
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-12-06 22:34:41 +08:00
|
|
|
if (ICmpInst *Cmp = dyn_cast<ICmpInst>(Src))
|
|
|
|
return transformZExtICmp(Cmp, CI);
|
2010-01-04 15:53:58 +08:00
|
|
|
|
|
|
|
BinaryOperator *SrcI = dyn_cast<BinaryOperator>(Src);
|
|
|
|
if (SrcI && SrcI->getOpcode() == Instruction::Or) {
|
[InstCombine] Refactor optimization of zext(or(icmp, icmp)) to enable more aggressive cast-folding
Summary:
InstCombine unfolds expressions of the form `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` such that in a later iteration of InstCombine the exposed `zext(icmp)` instructions can be optimized. We now combine this unfolding and the subsequent `zext(icmp)` optimization to be performed together. Since the unfolding doesn't happen separately anymore, we also again enable the folding of `logic(cast(icmp), cast(icmp))` expressions to `cast(logic(icmp, icmp))` which had been disabled due to its interference with the unfolding transformation.
Tested via `make check` and `lnt`.
Background
==========
For a better understanding on how it came to this change we subsequently summarize its history. In commit r275989 we've already tried to enable the folding of `logic(cast(icmp), cast(icmp))` to `cast(logic(icmp, icmp))` which had to be reverted in r276106 because it could lead to an endless loop in InstCombine (also see http://lists.llvm.org/pipermail/llvm-commits/Week-of-Mon-20160718/374347.html). The root of this problem is that in `visitZExt()` in InstCombineCasts.cpp there also exists a reverse of the above folding transformation, that unfolds `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` in order to expose `zext(icmp)` operations which would then possibly be eliminated by subsequent iterations of InstCombine. However, before these `zext(icmp)` would be eliminated the folding from r275989 could kick in and cause InstCombine to endlessly switch back and forth between the folding and the unfolding transformation. This is the reason why we now combine the `zext`-unfolding and the elimination of the exposed `zext(icmp)` to happen at one go because this enables us to still allow the cast-folding in `logic(cast(icmp), cast(icmp))` without entering an endless loop again.
Details on the submitted changes
================================
- In `visitZExt()` we combine the unfolding and optimization of `zext` instructions.
- In `transformZExtICmp()` we have to use `Builder->CreateIntCast()` instead of `CastInst::CreateIntegerCast()` to make sure that the new `CastInst` is inserted in a `BasicBlock`. The new calls to `transformZExtICmp()` that we introduce in `visitZExt()` would otherwise cause according assertions to be triggered (in our case this happend, for example, with lnt for the MultiSource/Applications/sqlite3 and SingleSource/Regression/C++/EH/recursive-throw tests). The subsequent usage of `replaceInstUsesWith()` is necessary to ensure that the new `CastInst` replaces the `ZExtInst` accordingly.
- In InstCombineAndOrXor.cpp we again allow the folding of casts on `icmp` instructions.
- The instruction order in the optimized IR for the zext-or-icmp.ll test case is different with the introduced changes.
- The test cases in zext.ll have been adopted from the reverted commits r275989 and r276105.
Reviewers: grosser, majnemer, spatel
Subscribers: eli.friedman, majnemer, llvm-commits
Differential Revision: https://reviews.llvm.org/D22864
Contributed-by: Matthias Reisinger <d412vv1n@gmail.com>
llvm-svn: 277635
2016-08-04 03:30:35 +08:00
|
|
|
// zext (or icmp, icmp) -> or (zext icmp), (zext icmp) if at least one
|
|
|
|
// of the (zext icmp) can be eliminated. If so, immediately perform the
|
|
|
|
// according elimination.
|
2010-01-04 15:53:58 +08:00
|
|
|
ICmpInst *LHS = dyn_cast<ICmpInst>(SrcI->getOperand(0));
|
|
|
|
ICmpInst *RHS = dyn_cast<ICmpInst>(SrcI->getOperand(1));
|
|
|
|
if (LHS && RHS && LHS->hasOneUse() && RHS->hasOneUse() &&
|
|
|
|
(transformZExtICmp(LHS, CI, false) ||
|
|
|
|
transformZExtICmp(RHS, CI, false))) {
|
[InstCombine] Refactor optimization of zext(or(icmp, icmp)) to enable more aggressive cast-folding
Summary:
InstCombine unfolds expressions of the form `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` such that in a later iteration of InstCombine the exposed `zext(icmp)` instructions can be optimized. We now combine this unfolding and the subsequent `zext(icmp)` optimization to be performed together. Since the unfolding doesn't happen separately anymore, we also again enable the folding of `logic(cast(icmp), cast(icmp))` expressions to `cast(logic(icmp, icmp))` which had been disabled due to its interference with the unfolding transformation.
Tested via `make check` and `lnt`.
Background
==========
For a better understanding on how it came to this change we subsequently summarize its history. In commit r275989 we've already tried to enable the folding of `logic(cast(icmp), cast(icmp))` to `cast(logic(icmp, icmp))` which had to be reverted in r276106 because it could lead to an endless loop in InstCombine (also see http://lists.llvm.org/pipermail/llvm-commits/Week-of-Mon-20160718/374347.html). The root of this problem is that in `visitZExt()` in InstCombineCasts.cpp there also exists a reverse of the above folding transformation, that unfolds `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` in order to expose `zext(icmp)` operations which would then possibly be eliminated by subsequent iterations of InstCombine. However, before these `zext(icmp)` would be eliminated the folding from r275989 could kick in and cause InstCombine to endlessly switch back and forth between the folding and the unfolding transformation. This is the reason why we now combine the `zext`-unfolding and the elimination of the exposed `zext(icmp)` to happen at one go because this enables us to still allow the cast-folding in `logic(cast(icmp), cast(icmp))` without entering an endless loop again.
Details on the submitted changes
================================
- In `visitZExt()` we combine the unfolding and optimization of `zext` instructions.
- In `transformZExtICmp()` we have to use `Builder->CreateIntCast()` instead of `CastInst::CreateIntegerCast()` to make sure that the new `CastInst` is inserted in a `BasicBlock`. The new calls to `transformZExtICmp()` that we introduce in `visitZExt()` would otherwise cause according assertions to be triggered (in our case this happend, for example, with lnt for the MultiSource/Applications/sqlite3 and SingleSource/Regression/C++/EH/recursive-throw tests). The subsequent usage of `replaceInstUsesWith()` is necessary to ensure that the new `CastInst` replaces the `ZExtInst` accordingly.
- In InstCombineAndOrXor.cpp we again allow the folding of casts on `icmp` instructions.
- The instruction order in the optimized IR for the zext-or-icmp.ll test case is different with the introduced changes.
- The test cases in zext.ll have been adopted from the reverted commits r275989 and r276105.
Reviewers: grosser, majnemer, spatel
Subscribers: eli.friedman, majnemer, llvm-commits
Differential Revision: https://reviews.llvm.org/D22864
Contributed-by: Matthias Reisinger <d412vv1n@gmail.com>
llvm-svn: 277635
2016-08-04 03:30:35 +08:00
|
|
|
// zext (or icmp, icmp) -> or (zext icmp), (zext icmp)
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *LCast = Builder.CreateZExt(LHS, CI.getType(), LHS->getName());
|
|
|
|
Value *RCast = Builder.CreateZExt(RHS, CI.getType(), RHS->getName());
|
[InstCombine] Insert instructions before adding them to worklist
Summary:
This patch adds instructions to the InstCombine worklist after they are properly inserted. This way we don't get `<badref>`s printed when logging added instructions.
It also adds a check in `Worklist::Add` that ensures that all added instructions have parents.
Simple test case that illustrates the difference when run with `--debug-only=instcombine`:
```
define i32 @test35(i32 %a, i32 %b) {
%1 = or i32 %a, 1135
%2 = or i32 %1, %b
ret i32 %2
}
```
Before this patch:
```
INSTCOMBINE ITERATION #1 on test35
IC: ADDING: 3 instrs to worklist
IC: Visiting: %1 = or i32 %a, 1135
IC: Visiting: %2 = or i32 %1, %b
IC: ADD: %2 = or i32 %a, %b
IC: Old = %3 = or i32 %1, %b
New = <badref> = or i32 %2, 1135
IC: ADD: <badref> = or i32 %2, 1135
...
```
With this patch:
```
INSTCOMBINE ITERATION #1 on test35
IC: ADDING: 3 instrs to worklist
IC: Visiting: %1 = or i32 %a, 1135
IC: Visiting: %2 = or i32 %1, %b
IC: ADD: %2 = or i32 %a, %b
IC: Old = %3 = or i32 %1, %b
New = <badref> = or i32 %2, 1135
IC: ADD: %3 = or i32 %2, 1135
...
```
Reviewers: fhahn, davide, spatel, foad, grosser, nikic
Reviewed By: nikic
Subscribers: nikic, lebedev.ri, hiraditya, llvm-commits
Tags: #llvm
Differential Revision: https://reviews.llvm.org/D71093
2019-12-19 03:55:41 +08:00
|
|
|
Value *Or = Builder.CreateOr(LCast, RCast, CI.getName());
|
|
|
|
if (auto *OrInst = dyn_cast<Instruction>(Or))
|
|
|
|
Builder.SetInsertPoint(OrInst);
|
[InstCombine] Refactor optimization of zext(or(icmp, icmp)) to enable more aggressive cast-folding
Summary:
InstCombine unfolds expressions of the form `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` such that in a later iteration of InstCombine the exposed `zext(icmp)` instructions can be optimized. We now combine this unfolding and the subsequent `zext(icmp)` optimization to be performed together. Since the unfolding doesn't happen separately anymore, we also again enable the folding of `logic(cast(icmp), cast(icmp))` expressions to `cast(logic(icmp, icmp))` which had been disabled due to its interference with the unfolding transformation.
Tested via `make check` and `lnt`.
Background
==========
For a better understanding on how it came to this change we subsequently summarize its history. In commit r275989 we've already tried to enable the folding of `logic(cast(icmp), cast(icmp))` to `cast(logic(icmp, icmp))` which had to be reverted in r276106 because it could lead to an endless loop in InstCombine (also see http://lists.llvm.org/pipermail/llvm-commits/Week-of-Mon-20160718/374347.html). The root of this problem is that in `visitZExt()` in InstCombineCasts.cpp there also exists a reverse of the above folding transformation, that unfolds `zext(or(icmp, icmp))` to `or(zext(icmp), zext(icmp))` in order to expose `zext(icmp)` operations which would then possibly be eliminated by subsequent iterations of InstCombine. However, before these `zext(icmp)` would be eliminated the folding from r275989 could kick in and cause InstCombine to endlessly switch back and forth between the folding and the unfolding transformation. This is the reason why we now combine the `zext`-unfolding and the elimination of the exposed `zext(icmp)` to happen at one go because this enables us to still allow the cast-folding in `logic(cast(icmp), cast(icmp))` without entering an endless loop again.
Details on the submitted changes
================================
- In `visitZExt()` we combine the unfolding and optimization of `zext` instructions.
- In `transformZExtICmp()` we have to use `Builder->CreateIntCast()` instead of `CastInst::CreateIntegerCast()` to make sure that the new `CastInst` is inserted in a `BasicBlock`. The new calls to `transformZExtICmp()` that we introduce in `visitZExt()` would otherwise cause according assertions to be triggered (in our case this happend, for example, with lnt for the MultiSource/Applications/sqlite3 and SingleSource/Regression/C++/EH/recursive-throw tests). The subsequent usage of `replaceInstUsesWith()` is necessary to ensure that the new `CastInst` replaces the `ZExtInst` accordingly.
- In InstCombineAndOrXor.cpp we again allow the folding of casts on `icmp` instructions.
- The instruction order in the optimized IR for the zext-or-icmp.ll test case is different with the introduced changes.
- The test cases in zext.ll have been adopted from the reverted commits r275989 and r276105.
Reviewers: grosser, majnemer, spatel
Subscribers: eli.friedman, majnemer, llvm-commits
Differential Revision: https://reviews.llvm.org/D22864
Contributed-by: Matthias Reisinger <d412vv1n@gmail.com>
llvm-svn: 277635
2016-08-04 03:30:35 +08:00
|
|
|
|
|
|
|
// Perform the elimination.
|
|
|
|
if (auto *LZExt = dyn_cast<ZExtInst>(LCast))
|
|
|
|
transformZExtICmp(LHS, *LZExt);
|
|
|
|
if (auto *RZExt = dyn_cast<ZExtInst>(RCast))
|
|
|
|
transformZExtICmp(RHS, *RZExt);
|
|
|
|
|
[InstCombine] Insert instructions before adding them to worklist
Summary:
This patch adds instructions to the InstCombine worklist after they are properly inserted. This way we don't get `<badref>`s printed when logging added instructions.
It also adds a check in `Worklist::Add` that ensures that all added instructions have parents.
Simple test case that illustrates the difference when run with `--debug-only=instcombine`:
```
define i32 @test35(i32 %a, i32 %b) {
%1 = or i32 %a, 1135
%2 = or i32 %1, %b
ret i32 %2
}
```
Before this patch:
```
INSTCOMBINE ITERATION #1 on test35
IC: ADDING: 3 instrs to worklist
IC: Visiting: %1 = or i32 %a, 1135
IC: Visiting: %2 = or i32 %1, %b
IC: ADD: %2 = or i32 %a, %b
IC: Old = %3 = or i32 %1, %b
New = <badref> = or i32 %2, 1135
IC: ADD: <badref> = or i32 %2, 1135
...
```
With this patch:
```
INSTCOMBINE ITERATION #1 on test35
IC: ADDING: 3 instrs to worklist
IC: Visiting: %1 = or i32 %a, 1135
IC: Visiting: %2 = or i32 %1, %b
IC: ADD: %2 = or i32 %a, %b
IC: Old = %3 = or i32 %1, %b
New = <badref> = or i32 %2, 1135
IC: ADD: %3 = or i32 %2, 1135
...
```
Reviewers: fhahn, davide, spatel, foad, grosser, nikic
Reviewed By: nikic
Subscribers: nikic, lebedev.ri, hiraditya, llvm-commits
Tags: #llvm
Differential Revision: https://reviews.llvm.org/D71093
2019-12-19 03:55:41 +08:00
|
|
|
return replaceInstUsesWith(CI, Or);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2014-01-20 04:05:13 +08:00
|
|
|
// zext(trunc(X) & C) -> (X & zext(C)).
|
|
|
|
Constant *C;
|
|
|
|
Value *X;
|
|
|
|
if (SrcI &&
|
|
|
|
match(SrcI, m_OneUse(m_And(m_Trunc(m_Value(X)), m_Constant(C)))) &&
|
|
|
|
X->getType() == CI.getType())
|
|
|
|
return BinaryOperator::CreateAnd(X, ConstantExpr::getZExt(C, CI.getType()));
|
|
|
|
|
|
|
|
// zext((trunc(X) & C) ^ C) -> ((X & zext(C)) ^ zext(C)).
|
|
|
|
Value *And;
|
|
|
|
if (SrcI && match(SrcI, m_OneUse(m_Xor(m_Value(And), m_Constant(C)))) &&
|
|
|
|
match(And, m_OneUse(m_And(m_Trunc(m_Value(X)), m_Specific(C)))) &&
|
|
|
|
X->getType() == CI.getType()) {
|
|
|
|
Constant *ZC = ConstantExpr::getZExt(C, CI.getType());
|
2017-07-08 07:16:26 +08:00
|
|
|
return BinaryOperator::CreateXor(Builder.CreateAnd(X, ZC), ZC);
|
2014-01-20 04:05:13 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Transform (sext icmp) to bitwise / integer operations to eliminate the icmp.
|
2011-04-02 04:09:03 +08:00
|
|
|
Instruction *InstCombiner::transformSExtICmp(ICmpInst *ICI, Instruction &CI) {
|
|
|
|
Value *Op0 = ICI->getOperand(0), *Op1 = ICI->getOperand(1);
|
|
|
|
ICmpInst::Predicate Pred = ICI->getPredicate();
|
|
|
|
|
2014-10-27 13:47:49 +08:00
|
|
|
// Don't bother if Op1 isn't of vector or integer type.
|
|
|
|
if (!Op1->getType()->isIntOrIntVectorTy())
|
|
|
|
return nullptr;
|
|
|
|
|
2018-06-22 01:51:44 +08:00
|
|
|
if ((Pred == ICmpInst::ICMP_SLT && match(Op1, m_ZeroInt())) ||
|
|
|
|
(Pred == ICmpInst::ICMP_SGT && match(Op1, m_AllOnes()))) {
|
2011-04-02 06:29:18 +08:00
|
|
|
// (x <s 0) ? -1 : 0 -> ashr x, 31 -> all ones if negative
|
|
|
|
// (x >s -1) ? -1 : 0 -> not (ashr x, 31) -> all ones if positive
|
2018-06-22 01:51:44 +08:00
|
|
|
Value *Sh = ConstantInt::get(Op0->getType(),
|
|
|
|
Op0->getType()->getScalarSizeInBits() - 1);
|
|
|
|
Value *In = Builder.CreateAShr(Op0, Sh, Op0->getName() + ".lobit");
|
|
|
|
if (In->getType() != CI.getType())
|
|
|
|
In = Builder.CreateIntCast(In, CI.getType(), true /*SExt*/);
|
|
|
|
|
|
|
|
if (Pred == ICmpInst::ICMP_SGT)
|
|
|
|
In = Builder.CreateNot(In, In->getName() + ".not");
|
|
|
|
return replaceInstUsesWith(CI, In);
|
2014-01-20 04:05:13 +08:00
|
|
|
}
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
|
2014-01-20 04:05:13 +08:00
|
|
|
if (ConstantInt *Op1C = dyn_cast<ConstantInt>(Op1)) {
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
// If we know that only one bit of the LHS of the icmp can be set and we
|
|
|
|
// have an equality comparison with zero or a power of 2, we can transform
|
|
|
|
// the icmp and sext into bitwise/integer operations.
|
2011-04-02 06:22:11 +08:00
|
|
|
if (ICI->hasOneUse() &&
|
|
|
|
ICI->isEquality() && (Op1C->isZero() || Op1C->getValue().isPowerOf2())){
|
2017-05-25 00:53:07 +08:00
|
|
|
KnownBits Known = computeKnownBits(Op0, 0, &CI);
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
|
2017-04-27 00:39:58 +08:00
|
|
|
APInt KnownZeroMask(~Known.Zero);
|
2011-04-02 04:15:16 +08:00
|
|
|
if (KnownZeroMask.isPowerOf2()) {
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
Value *In = ICI->getOperand(0);
|
|
|
|
|
2011-04-03 02:50:58 +08:00
|
|
|
// If the icmp tests for a known zero bit we can constant fold it.
|
|
|
|
if (!Op1C->isZero() && Op1C->getValue() != KnownZeroMask) {
|
|
|
|
Value *V = Pred == ICmpInst::ICMP_NE ?
|
|
|
|
ConstantInt::getAllOnesValue(CI.getType()) :
|
|
|
|
ConstantInt::getNullValue(CI.getType());
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, V);
|
2011-04-03 02:50:58 +08:00
|
|
|
}
|
2011-04-02 06:22:11 +08:00
|
|
|
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
if (!Op1C->isZero() == (Pred == ICmpInst::ICMP_NE)) {
|
|
|
|
// sext ((x & 2^n) == 0) -> (x >> n) - 1
|
|
|
|
// sext ((x & 2^n) != 2^n) -> (x >> n) - 1
|
|
|
|
unsigned ShiftAmt = KnownZeroMask.countTrailingZeros();
|
|
|
|
// Perform a right shift to place the desired bit in the LSB.
|
|
|
|
if (ShiftAmt)
|
2017-07-08 07:16:26 +08:00
|
|
|
In = Builder.CreateLShr(In,
|
|
|
|
ConstantInt::get(In->getType(), ShiftAmt));
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
|
|
|
|
// At this point "In" is either 1 or 0. Subtract 1 to turn
|
|
|
|
// {1, 0} -> {0, -1}.
|
2017-07-08 07:16:26 +08:00
|
|
|
In = Builder.CreateAdd(In,
|
|
|
|
ConstantInt::getAllOnesValue(In->getType()),
|
|
|
|
"sext");
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
} else {
|
|
|
|
// sext ((x & 2^n) != 0) -> (x << bitwidth-n) a>> bitwidth-1
|
2011-04-02 06:22:11 +08:00
|
|
|
// sext ((x & 2^n) == 2^n) -> (x << bitwidth-n) a>> bitwidth-1
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
unsigned ShiftAmt = KnownZeroMask.countLeadingZeros();
|
|
|
|
// Perform a left shift to place the desired bit in the MSB.
|
|
|
|
if (ShiftAmt)
|
2017-07-08 07:16:26 +08:00
|
|
|
In = Builder.CreateShl(In,
|
|
|
|
ConstantInt::get(In->getType(), ShiftAmt));
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
|
|
|
|
// Distribute the bit over the whole bit width.
|
2017-07-08 07:16:26 +08:00
|
|
|
In = Builder.CreateAShr(In, ConstantInt::get(In->getType(),
|
|
|
|
KnownZeroMask.getBitWidth() - 1), "sext");
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
if (CI.getType() == In->getType())
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, In);
|
InstCombine: Turn icmp + sext into bitwise/integer ops when the input has only one unknown bit.
int test1(unsigned x) { return (x&8) ? 0 : -1; }
int test3(unsigned x) { return (x&8) ? -1 : 0; }
before (x86_64):
_test1:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
ret
_test3:
andl $8, %edi
cmpl $1, %edi
sbbl %eax, %eax
notl %eax
ret
after:
_test1:
shrl $3, %edi
andl $1, %edi
leal -1(%rdi), %eax
ret
_test3:
shll $28, %edi
movl %edi, %eax
sarl $31, %eax
ret
llvm-svn: 128732
2011-04-02 04:09:10 +08:00
|
|
|
return CastInst::CreateIntegerCast(In, CI.getType(), true/*SExt*/);
|
|
|
|
}
|
|
|
|
}
|
2011-04-02 04:09:03 +08:00
|
|
|
}
|
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2011-04-02 04:09:03 +08:00
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Return true if we can take the specified value and return it as type Ty
|
|
|
|
/// without inserting any new casts and without changing the value of the common
|
|
|
|
/// low bits. This is used by code that tries to promote integer operations to
|
|
|
|
/// a wider types will allow us to eliminate the extension.
|
2010-01-10 08:58:42 +08:00
|
|
|
///
|
2010-01-10 15:57:20 +08:00
|
|
|
/// This function works on both vectors and scalars.
|
2010-01-10 08:58:42 +08:00
|
|
|
///
|
2015-09-09 22:54:29 +08:00
|
|
|
static bool canEvaluateSExtd(Value *V, Type *Ty) {
|
2010-01-10 08:58:42 +08:00
|
|
|
assert(V->getType()->getScalarSizeInBits() < Ty->getScalarSizeInBits() &&
|
|
|
|
"Can't sign extend type to a smaller type");
|
2018-01-31 22:55:53 +08:00
|
|
|
if (canAlwaysEvaluateInType(V, Ty))
|
2010-01-10 15:57:20 +08:00
|
|
|
return true;
|
2018-01-31 22:55:53 +08:00
|
|
|
if (canNotEvaluateInType(V, Ty))
|
|
|
|
return false;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-01-31 22:55:53 +08:00
|
|
|
auto *I = cast<Instruction>(V);
|
2010-01-10 15:57:20 +08:00
|
|
|
switch (I->getOpcode()) {
|
2010-01-11 04:30:41 +08:00
|
|
|
case Instruction::SExt: // sext(sext(x)) -> sext(x)
|
|
|
|
case Instruction::ZExt: // sext(zext(x)) -> zext(x)
|
|
|
|
case Instruction::Trunc: // sext(trunc(x)) -> trunc(x) or sext(x)
|
|
|
|
return true;
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::And:
|
|
|
|
case Instruction::Or:
|
|
|
|
case Instruction::Xor:
|
|
|
|
case Instruction::Add:
|
|
|
|
case Instruction::Sub:
|
|
|
|
case Instruction::Mul:
|
2010-01-10 15:57:20 +08:00
|
|
|
// These operators can all arbitrarily be extended if their inputs can.
|
2015-09-09 22:54:29 +08:00
|
|
|
return canEvaluateSExtd(I->getOperand(0), Ty) &&
|
|
|
|
canEvaluateSExtd(I->getOperand(1), Ty);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
//case Instruction::Shl: TODO
|
|
|
|
//case Instruction::LShr: TODO
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 15:57:20 +08:00
|
|
|
case Instruction::Select:
|
2015-09-09 22:54:29 +08:00
|
|
|
return canEvaluateSExtd(I->getOperand(1), Ty) &&
|
|
|
|
canEvaluateSExtd(I->getOperand(2), Ty);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
case Instruction::PHI: {
|
|
|
|
// We can change a phi if we can change all operands. Note that we never
|
|
|
|
// get into trouble with cyclic PHIs here because we only consider
|
|
|
|
// instructions with a single use.
|
|
|
|
PHINode *PN = cast<PHINode>(I);
|
2015-05-13 04:05:31 +08:00
|
|
|
for (Value *IncValue : PN->incoming_values())
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!canEvaluateSExtd(IncValue, Ty)) return false;
|
2010-01-10 15:57:20 +08:00
|
|
|
return true;
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
default:
|
|
|
|
// TODO: Can handle more cases here.
|
|
|
|
break;
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 15:57:20 +08:00
|
|
|
return false;
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
Instruction *InstCombiner::visitSExt(SExtInst &CI) {
|
2013-02-13 08:19:19 +08:00
|
|
|
// If this sign extend is only used by a truncate, let the truncate be
|
|
|
|
// eliminated before we try to optimize this sext.
|
2014-03-09 11:16:01 +08:00
|
|
|
if (CI.hasOneUse() && isa<TruncInst>(CI.user_back()))
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 09:00:46 +08:00
|
|
|
if (Instruction *I = commonCastTransforms(CI))
|
2010-01-04 15:53:58 +08:00
|
|
|
return I;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
Value *Src = CI.getOperand(0);
|
2011-07-18 12:54:35 +08:00
|
|
|
Type *SrcTy = Src->getType(), *DestTy = CI.getType();
|
2010-01-10 08:58:42 +08:00
|
|
|
|
2015-02-14 08:05:36 +08:00
|
|
|
// If we know that the value being extended is positive, we can use a zext
|
2016-08-05 09:09:48 +08:00
|
|
|
// instead.
|
2017-05-15 14:39:41 +08:00
|
|
|
KnownBits Known = computeKnownBits(Src, 0, &CI);
|
2019-05-09 04:59:21 +08:00
|
|
|
if (Known.isNonNegative())
|
|
|
|
return CastInst::Create(Instruction::ZExt, Src, DestTy);
|
2015-02-14 08:05:36 +08:00
|
|
|
|
2018-12-18 04:27:43 +08:00
|
|
|
// Try to extend the entire expression tree to the wide destination type.
|
|
|
|
if (shouldChangeType(SrcTy, DestTy) && canEvaluateSExtd(Src, DestTy)) {
|
2010-01-10 15:40:50 +08:00
|
|
|
// Okay, we can transform this! Insert the new expression now.
|
2018-05-14 20:53:11 +08:00
|
|
|
LLVM_DEBUG(
|
|
|
|
dbgs() << "ICE: EvaluateInDifferentType converting expression type"
|
|
|
|
" to avoid sign extend: "
|
|
|
|
<< CI << '\n');
|
2010-01-10 15:40:50 +08:00
|
|
|
Value *Res = EvaluateInDifferentType(Src, DestTy, true);
|
|
|
|
assert(Res->getType() == DestTy);
|
|
|
|
|
2010-01-10 08:58:42 +08:00
|
|
|
uint32_t SrcBitSize = SrcTy->getScalarSizeInBits();
|
|
|
|
uint32_t DestBitSize = DestTy->getScalarSizeInBits();
|
2010-01-10 15:40:50 +08:00
|
|
|
|
|
|
|
// If the high bits are already filled with sign bit, just replace this
|
|
|
|
// cast with the result.
|
Make use of @llvm.assume in ValueTracking (computeKnownBits, etc.)
This change, which allows @llvm.assume to be used from within computeKnownBits
(and other associated functions in ValueTracking), adds some (optional)
parameters to computeKnownBits and friends. These functions now (optionally)
take a "context" instruction pointer, an AssumptionTracker pointer, and also a
DomTree pointer, and most of the changes are just to pass this new information
when it is easily available from InstSimplify, InstCombine, etc.
As explained below, the significant conceptual change is that known properties
of a value might depend on the control-flow location of the use (because we
care that the @llvm.assume dominates the use because assumptions have
control-flow dependencies). This means that, when we ask if bits are known in a
value, we might get different answers for different uses.
The significant changes are all in ValueTracking. Two main changes: First, as
with the rest of the code, new parameters need to be passed around. To make
this easier, I grouped them into a structure, and I made internal static
versions of the relevant functions that take this structure as a parameter. The
new code does as you might expect, it looks for @llvm.assume calls that make
use of the value we're trying to learn something about (often indirectly),
attempts to pattern match that expression, and uses the result if successful.
By making use of the AssumptionTracker, the process of finding @llvm.assume
calls is not expensive.
Part of the structure being passed around inside ValueTracking is a set of
already-considered @llvm.assume calls. This is to prevent a query using, for
example, the assume(a == b), to recurse on itself. The context and DT params
are used to find applicable assumptions. An assumption needs to dominate the
context instruction, or come after it deterministically. In this latter case we
only handle the specific case where both the assumption and the context
instruction are in the same block, and we need to exclude assumptions from
being used to simplify their own ephemeral values (those which contribute only
to the assumption) because otherwise the assumption would prove its feeding
comparison trivial and would be removed.
This commit adds the plumbing and the logic for a simple masked-bit propagation
(just enough to write a regression test). Future commits add more patterns
(and, correspondingly, more regression tests).
llvm-svn: 217342
2014-09-08 02:57:58 +08:00
|
|
|
if (ComputeNumSignBits(Res, 0, &CI) > DestBitSize - SrcBitSize)
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, Res);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-10 15:40:50 +08:00
|
|
|
// We need to emit a shl + ashr to do the sign extend.
|
|
|
|
Value *ShAmt = ConstantInt::get(DestTy, DestBitSize-SrcBitSize);
|
2017-07-08 07:16:26 +08:00
|
|
|
return BinaryOperator::CreateAShr(Builder.CreateShl(Res, ShAmt, "sext"),
|
2010-01-10 15:40:50 +08:00
|
|
|
ShAmt);
|
2010-01-10 08:58:42 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2017-02-24 00:26:03 +08:00
|
|
|
// If the input is a trunc from the destination type, then turn sext(trunc(x))
|
2010-01-19 06:19:16 +08:00
|
|
|
// into shifts.
|
2017-02-24 00:26:03 +08:00
|
|
|
Value *X;
|
|
|
|
if (match(Src, m_OneUse(m_Trunc(m_Value(X)))) && X->getType() == DestTy) {
|
|
|
|
// sext(trunc(X)) --> ashr(shl(X, C), C)
|
|
|
|
unsigned SrcBitSize = SrcTy->getScalarSizeInBits();
|
|
|
|
unsigned DestBitSize = DestTy->getScalarSizeInBits();
|
|
|
|
Constant *ShAmt = ConstantInt::get(DestTy, DestBitSize - SrcBitSize);
|
2017-07-08 07:16:26 +08:00
|
|
|
return BinaryOperator::CreateAShr(Builder.CreateShl(X, ShAmt), ShAmt);
|
2017-02-24 00:26:03 +08:00
|
|
|
}
|
2010-12-18 07:27:41 +08:00
|
|
|
|
2011-04-02 04:09:03 +08:00
|
|
|
if (ICmpInst *ICI = dyn_cast<ICmpInst>(Src))
|
|
|
|
return transformSExtICmp(ICI, CI);
|
2010-12-18 07:27:41 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// If the input is a shl/ashr pair of a same constant, then this is a sign
|
|
|
|
// extension from a smaller value. If we could trust arbitrary bitwidth
|
|
|
|
// integers, we could turn this into a truncate to the smaller bit and then
|
|
|
|
// use a sext for the whole extension. Since we don't, look deeper and check
|
|
|
|
// for a truncate. If the source and dest are the same type, eliminate the
|
|
|
|
// trunc and extend and just do shifts. For example, turn:
|
|
|
|
// %a = trunc i32 %i to i8
|
|
|
|
// %b = shl i8 %a, 6
|
|
|
|
// %c = ashr i8 %b, 6
|
|
|
|
// %d = sext i8 %c to i32
|
|
|
|
// into:
|
|
|
|
// %a = shl i32 %i, 30
|
|
|
|
// %d = ashr i32 %a, 30
|
2014-04-25 13:29:35 +08:00
|
|
|
Value *A = nullptr;
|
2010-01-10 09:04:31 +08:00
|
|
|
// TODO: Eventually this could be subsumed by EvaluateInDifferentType.
|
2014-04-25 13:29:35 +08:00
|
|
|
ConstantInt *BA = nullptr, *CA = nullptr;
|
2010-01-10 09:04:31 +08:00
|
|
|
if (match(Src, m_AShr(m_Shl(m_Trunc(m_Value(A)), m_ConstantInt(BA)),
|
2010-01-04 15:53:58 +08:00
|
|
|
m_ConstantInt(CA))) &&
|
2010-01-10 09:04:31 +08:00
|
|
|
BA == CA && A->getType() == CI.getType()) {
|
|
|
|
unsigned MidSize = Src->getType()->getScalarSizeInBits();
|
|
|
|
unsigned SrcDstSize = CI.getType()->getScalarSizeInBits();
|
|
|
|
unsigned ShAmt = CA->getZExtValue()+SrcDstSize-MidSize;
|
|
|
|
Constant *ShAmtV = ConstantInt::get(CI.getType(), ShAmt);
|
2017-07-08 07:16:26 +08:00
|
|
|
A = Builder.CreateShl(A, ShAmtV, CI.getName());
|
2010-01-10 09:04:31 +08:00
|
|
|
return BinaryOperator::CreateAShr(A, ShAmtV);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Return a Constant* for the specified floating-point constant if it fits
|
2010-01-04 15:53:58 +08:00
|
|
|
/// in the specified FP type without changing its value.
|
2018-03-03 05:25:18 +08:00
|
|
|
static bool fitsInFPType(ConstantFP *CFP, const fltSemantics &Sem) {
|
2010-01-04 15:53:58 +08:00
|
|
|
bool losesInfo;
|
|
|
|
APFloat F = CFP->getValueAPF();
|
|
|
|
(void)F.convert(Sem, APFloat::rmNearestTiesToEven, &losesInfo);
|
2018-03-03 05:25:18 +08:00
|
|
|
return !losesInfo;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2018-03-03 05:25:18 +08:00
|
|
|
static Type *shrinkFPConstant(ConstantFP *CFP) {
|
2018-03-01 04:14:34 +08:00
|
|
|
if (CFP->getType() == Type::getPPC_FP128Ty(CFP->getContext()))
|
|
|
|
return nullptr; // No constant folding of this.
|
|
|
|
// See if the value can be truncated to half and then reextended.
|
2018-03-03 05:25:18 +08:00
|
|
|
if (fitsInFPType(CFP, APFloat::IEEEhalf()))
|
|
|
|
return Type::getHalfTy(CFP->getContext());
|
2018-03-01 04:14:34 +08:00
|
|
|
// See if the value can be truncated to float and then reextended.
|
2018-03-03 05:25:18 +08:00
|
|
|
if (fitsInFPType(CFP, APFloat::IEEEsingle()))
|
|
|
|
return Type::getFloatTy(CFP->getContext());
|
2018-03-01 04:14:34 +08:00
|
|
|
if (CFP->getType()->isDoubleTy())
|
|
|
|
return nullptr; // Won't shrink.
|
2018-03-03 05:25:18 +08:00
|
|
|
if (fitsInFPType(CFP, APFloat::IEEEdouble()))
|
|
|
|
return Type::getDoubleTy(CFP->getContext());
|
2018-03-01 04:14:34 +08:00
|
|
|
// Don't try to shrink to various long double types.
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
2018-03-06 02:04:12 +08:00
|
|
|
// Determine if this is a vector of ConstantFPs and if so, return the minimal
|
|
|
|
// type we can safely truncate all elements to.
|
|
|
|
// TODO: Make these support undef elements.
|
|
|
|
static Type *shrinkFPConstantVector(Value *V) {
|
|
|
|
auto *CV = dyn_cast<Constant>(V);
|
2020-04-09 01:42:22 +08:00
|
|
|
auto *CVVTy = dyn_cast<VectorType>(V->getType());
|
|
|
|
if (!CV || !CVVTy)
|
2018-03-06 02:04:12 +08:00
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
Type *MinType = nullptr;
|
|
|
|
|
2020-04-09 01:42:22 +08:00
|
|
|
unsigned NumElts = CVVTy->getNumElements();
|
2018-03-06 02:04:12 +08:00
|
|
|
for (unsigned i = 0; i != NumElts; ++i) {
|
|
|
|
auto *CFP = dyn_cast_or_null<ConstantFP>(CV->getAggregateElement(i));
|
|
|
|
if (!CFP)
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
Type *T = shrinkFPConstant(CFP);
|
|
|
|
if (!T)
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
// If we haven't found a type yet or this type has a larger mantissa than
|
|
|
|
// our previous type, this is our new minimal type.
|
|
|
|
if (!MinType || T->getFPMantissaWidth() > MinType->getFPMantissaWidth())
|
|
|
|
MinType = T;
|
|
|
|
}
|
|
|
|
|
|
|
|
// Make a vector type from the minimal type.
|
2020-05-30 06:24:15 +08:00
|
|
|
return FixedVectorType::get(MinType, NumElts);
|
2018-03-06 02:04:12 +08:00
|
|
|
}
|
|
|
|
|
2018-03-03 05:25:18 +08:00
|
|
|
/// Find the minimum FP type we can safely truncate to.
|
|
|
|
static Type *getMinimumFPType(Value *V) {
|
|
|
|
if (auto *FPExt = dyn_cast<FPExtInst>(V))
|
|
|
|
return FPExt->getOperand(0)->getType();
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// If this value is a constant, return the constant in the smallest FP type
|
|
|
|
// that can accurately represent it. This allows us to turn
|
|
|
|
// (float)((double)X+2.0) into x+2.0f.
|
2018-03-01 04:14:34 +08:00
|
|
|
if (auto *CFP = dyn_cast<ConstantFP>(V))
|
2018-03-03 05:25:18 +08:00
|
|
|
if (Type *T = shrinkFPConstant(CFP))
|
|
|
|
return T;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-03-06 02:04:12 +08:00
|
|
|
// Try to shrink a vector of FP constants.
|
|
|
|
if (Type *T = shrinkFPConstantVector(V))
|
|
|
|
return T;
|
|
|
|
|
2018-03-03 05:25:18 +08:00
|
|
|
return V->getType();
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2020-05-24 21:30:19 +08:00
|
|
|
/// Return true if the cast from integer to FP can be proven to be exact for all
|
|
|
|
/// possible inputs (the conversion does not lose any precision).
|
|
|
|
static bool isKnownExactCastIntToFP(CastInst &I) {
|
|
|
|
CastInst::CastOps Opcode = I.getOpcode();
|
|
|
|
assert((Opcode == CastInst::SIToFP || Opcode == CastInst::UIToFP) &&
|
|
|
|
"Unexpected cast");
|
|
|
|
Value *Src = I.getOperand(0);
|
|
|
|
Type *SrcTy = Src->getType();
|
|
|
|
Type *FPTy = I.getType();
|
|
|
|
bool IsSigned = Opcode == Instruction::SIToFP;
|
|
|
|
int SrcSize = (int)SrcTy->getScalarSizeInBits() - IsSigned;
|
|
|
|
|
|
|
|
// Easy case - if the source integer type has less bits than the FP mantissa,
|
|
|
|
// then the cast must be exact.
|
|
|
|
int DestNumSigBits = FPTy->getFPMantissaWidth();
|
|
|
|
if (SrcSize <= DestNumSigBits)
|
|
|
|
return true;
|
|
|
|
|
|
|
|
// Cast from FP to integer and back to FP is independent of the intermediate
|
|
|
|
// integer width because of poison on overflow.
|
|
|
|
Value *F;
|
|
|
|
if (match(Src, m_FPToSI(m_Value(F))) || match(Src, m_FPToUI(m_Value(F)))) {
|
|
|
|
// If this is uitofp (fptosi F), the source needs an extra bit to avoid
|
|
|
|
// potential rounding of negative FP input values.
|
|
|
|
int SrcNumSigBits = F->getType()->getFPMantissaWidth();
|
|
|
|
if (!IsSigned && match(Src, m_FPToSI(m_Value())))
|
|
|
|
SrcNumSigBits++;
|
|
|
|
|
|
|
|
// [su]itofp (fpto[su]i F) --> exact if the source type has less or equal
|
|
|
|
// significant bits than the destination (and make sure neither type is
|
|
|
|
// weird -- ppc_fp128).
|
|
|
|
if (SrcNumSigBits > 0 && DestNumSigBits > 0 &&
|
|
|
|
SrcNumSigBits <= DestNumSigBits)
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
|
|
|
|
// TODO:
|
|
|
|
// Try harder to find if the source integer type has less significant bits.
|
|
|
|
// For example, compute number of sign bits or compute low bit mask.
|
|
|
|
return false;
|
|
|
|
}
|
|
|
|
|
2018-03-24 23:41:59 +08:00
|
|
|
Instruction *InstCombiner::visitFPTrunc(FPTruncInst &FPT) {
|
|
|
|
if (Instruction *I = commonCastTransforms(FPT))
|
2010-01-04 15:53:58 +08:00
|
|
|
return I;
|
2018-03-24 23:41:59 +08:00
|
|
|
|
2013-11-29 05:38:05 +08:00
|
|
|
// If we have fptrunc(OpI (fpextend x), (fpextend y)), we would like to
|
2015-11-22 00:16:29 +08:00
|
|
|
// simplify this expression to avoid one or more of the trunc/extend
|
2013-11-29 05:38:05 +08:00
|
|
|
// operations if we can do so without changing the numerical results.
|
|
|
|
//
|
|
|
|
// The exact manner in which the widths of the operands interact to limit
|
|
|
|
// what we can and cannot do safely varies from operation to operation, and
|
|
|
|
// is explained below in the various case statements.
|
2018-03-24 23:41:59 +08:00
|
|
|
Type *Ty = FPT.getType();
|
2019-09-12 06:31:34 +08:00
|
|
|
auto *BO = dyn_cast<BinaryOperator>(FPT.getOperand(0));
|
|
|
|
if (BO && BO->hasOneUse()) {
|
|
|
|
Type *LHSMinType = getMinimumFPType(BO->getOperand(0));
|
|
|
|
Type *RHSMinType = getMinimumFPType(BO->getOperand(1));
|
|
|
|
unsigned OpWidth = BO->getType()->getFPMantissaWidth();
|
2018-03-03 05:25:18 +08:00
|
|
|
unsigned LHSWidth = LHSMinType->getFPMantissaWidth();
|
|
|
|
unsigned RHSWidth = RHSMinType->getFPMantissaWidth();
|
2013-11-29 05:38:05 +08:00
|
|
|
unsigned SrcWidth = std::max(LHSWidth, RHSWidth);
|
2018-03-24 23:41:59 +08:00
|
|
|
unsigned DstWidth = Ty->getFPMantissaWidth();
|
2019-09-12 06:31:34 +08:00
|
|
|
switch (BO->getOpcode()) {
|
2013-11-29 05:38:05 +08:00
|
|
|
default: break;
|
|
|
|
case Instruction::FAdd:
|
|
|
|
case Instruction::FSub:
|
|
|
|
// For addition and subtraction, the infinitely precise result can
|
|
|
|
// essentially be arbitrarily wide; proving that double rounding
|
|
|
|
// will not occur because the result of OpI is exact (as we will for
|
|
|
|
// FMul, for example) is hopeless. However, we *can* nonetheless
|
|
|
|
// frequently know that double rounding cannot occur (or that it is
|
2014-01-25 01:20:08 +08:00
|
|
|
// innocuous) by taking advantage of the specific structure of
|
2013-11-29 05:38:05 +08:00
|
|
|
// infinitely-precise results that admit double rounding.
|
|
|
|
//
|
2014-01-25 01:20:08 +08:00
|
|
|
// Specifically, if OpWidth >= 2*DstWdith+1 and DstWidth is sufficient
|
2013-11-29 05:38:05 +08:00
|
|
|
// to represent both sources, we can guarantee that the double
|
|
|
|
// rounding is innocuous (See p50 of Figueroa's 2000 PhD thesis,
|
|
|
|
// "A Rigorous Framework for Fully Supporting the IEEE Standard ..."
|
|
|
|
// for proof of this fact).
|
|
|
|
//
|
|
|
|
// Note: Figueroa does not consider the case where DstFormat !=
|
|
|
|
// SrcFormat. It's possible (likely even!) that this analysis
|
|
|
|
// could be tightened for those cases, but they are rare (the main
|
|
|
|
// case of interest here is (float)((double)float + float)).
|
|
|
|
if (OpWidth >= 2*DstWidth+1 && DstWidth >= SrcWidth) {
|
2019-09-12 06:31:34 +08:00
|
|
|
Value *LHS = Builder.CreateFPTrunc(BO->getOperand(0), Ty);
|
|
|
|
Value *RHS = Builder.CreateFPTrunc(BO->getOperand(1), Ty);
|
|
|
|
Instruction *RI = BinaryOperator::Create(BO->getOpcode(), LHS, RHS);
|
|
|
|
RI->copyFastMathFlags(BO);
|
2014-01-18 08:48:14 +08:00
|
|
|
return RI;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-11-29 05:38:05 +08:00
|
|
|
break;
|
|
|
|
case Instruction::FMul:
|
|
|
|
// For multiplication, the infinitely precise result has at most
|
|
|
|
// LHSWidth + RHSWidth significant bits; if OpWidth is sufficient
|
|
|
|
// that such a value can be exactly represented, then no double
|
|
|
|
// rounding can possibly occur; we can safely perform the operation
|
|
|
|
// in the destination format if it can represent both sources.
|
|
|
|
if (OpWidth >= LHSWidth + RHSWidth && DstWidth >= SrcWidth) {
|
2019-09-12 06:31:34 +08:00
|
|
|
Value *LHS = Builder.CreateFPTrunc(BO->getOperand(0), Ty);
|
|
|
|
Value *RHS = Builder.CreateFPTrunc(BO->getOperand(1), Ty);
|
|
|
|
return BinaryOperator::CreateFMulFMF(LHS, RHS, BO);
|
2013-11-29 05:38:05 +08:00
|
|
|
}
|
|
|
|
break;
|
|
|
|
case Instruction::FDiv:
|
|
|
|
// For division, we use again use the bound from Figueroa's
|
|
|
|
// dissertation. I am entirely certain that this bound can be
|
|
|
|
// tightened in the unbalanced operand case by an analysis based on
|
|
|
|
// the diophantine rational approximation bound, but the well-known
|
|
|
|
// condition used here is a good conservative first pass.
|
|
|
|
// TODO: Tighten bound via rigorous analysis of the unbalanced case.
|
|
|
|
if (OpWidth >= 2*DstWidth && DstWidth >= SrcWidth) {
|
2019-09-12 06:31:34 +08:00
|
|
|
Value *LHS = Builder.CreateFPTrunc(BO->getOperand(0), Ty);
|
|
|
|
Value *RHS = Builder.CreateFPTrunc(BO->getOperand(1), Ty);
|
|
|
|
return BinaryOperator::CreateFDivFMF(LHS, RHS, BO);
|
2013-11-29 05:38:05 +08:00
|
|
|
}
|
|
|
|
break;
|
2018-03-03 05:25:18 +08:00
|
|
|
case Instruction::FRem: {
|
2013-11-29 05:38:05 +08:00
|
|
|
// Remainder is straightforward. Remainder is always exact, so the
|
|
|
|
// type of OpI doesn't enter into things at all. We simply evaluate
|
|
|
|
// in whichever source type is larger, then convert to the
|
|
|
|
// destination type.
|
2014-12-13 02:48:37 +08:00
|
|
|
if (SrcWidth == OpWidth)
|
2014-12-13 01:21:54 +08:00
|
|
|
break;
|
2018-03-03 05:25:18 +08:00
|
|
|
Value *LHS, *RHS;
|
|
|
|
if (LHSWidth == SrcWidth) {
|
2019-09-12 06:31:34 +08:00
|
|
|
LHS = Builder.CreateFPTrunc(BO->getOperand(0), LHSMinType);
|
|
|
|
RHS = Builder.CreateFPTrunc(BO->getOperand(1), LHSMinType);
|
2018-03-03 05:25:18 +08:00
|
|
|
} else {
|
2019-09-12 06:31:34 +08:00
|
|
|
LHS = Builder.CreateFPTrunc(BO->getOperand(0), RHSMinType);
|
|
|
|
RHS = Builder.CreateFPTrunc(BO->getOperand(1), RHSMinType);
|
2014-11-19 05:30:02 +08:00
|
|
|
}
|
2018-03-03 05:25:18 +08:00
|
|
|
|
2019-09-12 06:31:34 +08:00
|
|
|
Value *ExactResult = Builder.CreateFRemFMF(LHS, RHS, BO);
|
2018-03-24 23:41:59 +08:00
|
|
|
return CastInst::CreateFPCast(ExactResult, Ty);
|
2018-03-03 05:25:18 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2019-06-11 23:45:41 +08:00
|
|
|
}
|
2013-01-11 06:06:52 +08:00
|
|
|
|
2019-06-11 23:45:41 +08:00
|
|
|
// (fptrunc (fneg x)) -> (fneg (fptrunc x))
|
|
|
|
Value *X;
|
|
|
|
Instruction *Op = dyn_cast<Instruction>(FPT.getOperand(0));
|
|
|
|
if (Op && Op->hasOneUse()) {
|
2019-12-05 23:50:43 +08:00
|
|
|
// FIXME: The FMF should propagate from the fptrunc, not the source op.
|
|
|
|
IRBuilder<>::FastMathFlagGuard FMFG(Builder);
|
|
|
|
if (isa<FPMathOperator>(Op))
|
|
|
|
Builder.setFastMathFlags(Op->getFastMathFlags());
|
|
|
|
|
2019-06-11 23:45:41 +08:00
|
|
|
if (match(Op, m_FNeg(m_Value(X)))) {
|
2018-10-26 02:09:33 +08:00
|
|
|
Value *InnerTrunc = Builder.CreateFPTrunc(X, Ty);
|
2019-06-11 23:45:41 +08:00
|
|
|
|
|
|
|
return UnaryOperator::CreateFNegFMF(InnerTrunc, Op);
|
2013-01-11 06:06:52 +08:00
|
|
|
}
|
2019-12-06 00:12:44 +08:00
|
|
|
|
|
|
|
// If we are truncating a select that has an extended operand, we can
|
|
|
|
// narrow the other operand and do the select as a narrow op.
|
|
|
|
Value *Cond, *X, *Y;
|
|
|
|
if (match(Op, m_Select(m_Value(Cond), m_FPExt(m_Value(X)), m_Value(Y))) &&
|
|
|
|
X->getType() == Ty) {
|
|
|
|
// fptrunc (select Cond, (fpext X), Y --> select Cond, X, (fptrunc Y)
|
|
|
|
Value *NarrowY = Builder.CreateFPTrunc(Y, Ty);
|
|
|
|
Value *Sel = Builder.CreateSelect(Cond, X, NarrowY, "narrow.sel", Op);
|
|
|
|
return replaceInstUsesWith(FPT, Sel);
|
|
|
|
}
|
|
|
|
if (match(Op, m_Select(m_Value(Cond), m_Value(Y), m_FPExt(m_Value(X)))) &&
|
|
|
|
X->getType() == Ty) {
|
|
|
|
// fptrunc (select Cond, Y, (fpext X) --> select Cond, (fptrunc Y), X
|
|
|
|
Value *NarrowY = Builder.CreateFPTrunc(Y, Ty);
|
|
|
|
Value *Sel = Builder.CreateSelect(Cond, NarrowY, X, "narrow.sel", Op);
|
|
|
|
return replaceInstUsesWith(FPT, Sel);
|
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-11 06:06:52 +08:00
|
|
|
|
2018-03-24 23:41:59 +08:00
|
|
|
if (auto *II = dyn_cast<IntrinsicInst>(FPT.getOperand(0))) {
|
2013-01-11 06:06:52 +08:00
|
|
|
switch (II->getIntrinsicID()) {
|
2017-01-17 08:10:40 +08:00
|
|
|
default: break;
|
2017-01-24 07:55:08 +08:00
|
|
|
case Intrinsic::ceil:
|
2018-03-24 23:41:59 +08:00
|
|
|
case Intrinsic::fabs:
|
2017-01-24 07:55:08 +08:00
|
|
|
case Intrinsic::floor:
|
2018-03-24 23:41:59 +08:00
|
|
|
case Intrinsic::nearbyint:
|
2017-01-24 07:55:08 +08:00
|
|
|
case Intrinsic::rint:
|
|
|
|
case Intrinsic::round:
|
2020-05-26 20:24:05 +08:00
|
|
|
case Intrinsic::roundeven:
|
2017-01-24 07:55:08 +08:00
|
|
|
case Intrinsic::trunc: {
|
2017-03-21 05:59:24 +08:00
|
|
|
Value *Src = II->getArgOperand(0);
|
|
|
|
if (!Src->hasOneUse())
|
|
|
|
break;
|
|
|
|
|
|
|
|
// Except for fabs, this transformation requires the input of the unary FP
|
|
|
|
// operation to be itself an fpext from the type to which we're
|
|
|
|
// truncating.
|
|
|
|
if (II->getIntrinsicID() != Intrinsic::fabs) {
|
|
|
|
FPExtInst *FPExtSrc = dyn_cast<FPExtInst>(Src);
|
2018-03-24 23:41:59 +08:00
|
|
|
if (!FPExtSrc || FPExtSrc->getSrcTy() != Ty)
|
2017-03-21 05:59:24 +08:00
|
|
|
break;
|
|
|
|
}
|
|
|
|
|
2017-01-24 07:55:08 +08:00
|
|
|
// Do unary FP operation on smaller type.
|
2017-01-17 08:10:40 +08:00
|
|
|
// (fptrunc (fabs x)) -> (fabs (fptrunc x))
|
2018-03-24 23:41:59 +08:00
|
|
|
Value *InnerTrunc = Builder.CreateFPTrunc(Src, Ty);
|
|
|
|
Function *Overload = Intrinsic::getDeclaration(FPT.getModule(),
|
|
|
|
II->getIntrinsicID(), Ty);
|
2017-01-17 08:10:40 +08:00
|
|
|
SmallVector<OperandBundleDef, 1> OpBundles;
|
|
|
|
II->getOperandBundlesAsDefs(OpBundles);
|
2019-02-02 04:43:25 +08:00
|
|
|
CallInst *NewCI =
|
|
|
|
CallInst::Create(Overload, {InnerTrunc}, OpBundles, II->getName());
|
2017-01-17 08:10:40 +08:00
|
|
|
NewCI->copyFastMathFlags(II);
|
|
|
|
return NewCI;
|
|
|
|
}
|
2013-01-11 06:06:52 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2018-03-24 23:41:59 +08:00
|
|
|
if (Instruction *I = shrinkInsertElt(FPT, Builder))
|
2017-03-08 07:27:14 +08:00
|
|
|
return I;
|
|
|
|
|
2020-05-24 21:30:19 +08:00
|
|
|
Value *Src = FPT.getOperand(0);
|
|
|
|
if (isa<SIToFPInst>(Src) || isa<UIToFPInst>(Src)) {
|
|
|
|
auto *FPCast = cast<CastInst>(Src);
|
|
|
|
if (isKnownExactCastIntToFP(*FPCast))
|
|
|
|
return CastInst::Create(FPCast->getOpcode(), FPCast->getOperand(0), Ty);
|
2020-05-17 20:43:40 +08:00
|
|
|
}
|
|
|
|
|
2020-05-24 21:30:19 +08:00
|
|
|
return nullptr;
|
2020-05-09 02:40:04 +08:00
|
|
|
}
|
|
|
|
|
2020-05-10 18:59:30 +08:00
|
|
|
Instruction *InstCombiner::visitFPExt(CastInst &FPExt) {
|
|
|
|
// If the source operand is a cast from integer to FP and known exact, then
|
|
|
|
// cast the integer operand directly to the destination type.
|
|
|
|
Type *Ty = FPExt.getType();
|
|
|
|
Value *Src = FPExt.getOperand(0);
|
|
|
|
if (isa<SIToFPInst>(Src) || isa<UIToFPInst>(Src)) {
|
|
|
|
auto *FPCast = cast<CastInst>(Src);
|
|
|
|
if (isKnownExactCastIntToFP(*FPCast))
|
|
|
|
return CastInst::Create(FPCast->getOpcode(), FPCast->getOperand(0), Ty);
|
|
|
|
}
|
|
|
|
|
|
|
|
return commonCastTransforms(FPExt);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2020-05-09 00:06:38 +08:00
|
|
|
/// fpto{s/u}i({u/s}itofp(X)) --> X or zext(X) or sext(X) or trunc(X)
|
|
|
|
/// This is safe if the intermediate type has enough bits in its mantissa to
|
|
|
|
/// accurately represent all values of X. For example, this won't work with
|
|
|
|
/// i64 -> float -> i64.
|
|
|
|
Instruction *InstCombiner::foldItoFPtoI(CastInst &FI) {
|
2015-02-17 05:47:54 +08:00
|
|
|
if (!isa<UIToFPInst>(FI.getOperand(0)) && !isa<SIToFPInst>(FI.getOperand(0)))
|
|
|
|
return nullptr;
|
|
|
|
|
2020-05-09 02:40:04 +08:00
|
|
|
auto *OpI = cast<CastInst>(FI.getOperand(0));
|
2020-05-09 00:06:38 +08:00
|
|
|
Value *X = OpI->getOperand(0);
|
|
|
|
Type *XType = X->getType();
|
2020-05-09 02:40:04 +08:00
|
|
|
Type *DestType = FI.getType();
|
2015-02-17 05:47:54 +08:00
|
|
|
bool IsOutputSigned = isa<FPToSIInst>(FI);
|
|
|
|
|
|
|
|
// Since we can assume the conversion won't overflow, our decision as to
|
|
|
|
// whether the input will fit in the float should depend on the minimum
|
|
|
|
// of the input range and output range.
|
|
|
|
|
|
|
|
// This means this is also safe for a signed input and unsigned output, since
|
|
|
|
// a negative input would lead to undefined behavior.
|
2020-05-09 02:40:04 +08:00
|
|
|
if (!isKnownExactCastIntToFP(*OpI)) {
|
|
|
|
// The first cast may not round exactly based on the source integer width
|
|
|
|
// and FP width, but the overflow UB rules can still allow this to fold.
|
|
|
|
// If the destination type is narrow, that means the intermediate FP value
|
|
|
|
// must be large enough to hold the source value exactly.
|
|
|
|
// For example, (uint8_t)((float)(uint32_t 16777217) is undefined behavior.
|
|
|
|
int OutputSize = (int)DestType->getScalarSizeInBits() - IsOutputSigned;
|
|
|
|
if (OutputSize > OpI->getType()->getFPMantissaWidth())
|
|
|
|
return nullptr;
|
|
|
|
}
|
2020-05-09 00:06:38 +08:00
|
|
|
|
|
|
|
if (DestType->getScalarSizeInBits() > XType->getScalarSizeInBits()) {
|
2020-05-09 02:40:04 +08:00
|
|
|
bool IsInputSigned = isa<SIToFPInst>(OpI);
|
2020-05-09 00:06:38 +08:00
|
|
|
if (IsInputSigned && IsOutputSigned)
|
|
|
|
return new SExtInst(X, DestType);
|
|
|
|
return new ZExtInst(X, DestType);
|
2015-02-17 05:47:54 +08:00
|
|
|
}
|
2020-05-09 00:06:38 +08:00
|
|
|
if (DestType->getScalarSizeInBits() < XType->getScalarSizeInBits())
|
|
|
|
return new TruncInst(X, DestType);
|
|
|
|
|
|
|
|
assert(XType == DestType && "Unexpected types for int to FP to int casts");
|
|
|
|
return replaceInstUsesWith(FI, X);
|
2015-02-17 05:47:54 +08:00
|
|
|
}
|
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
Instruction *InstCombiner::visitFPToUI(FPToUIInst &FI) {
|
2020-05-09 00:06:38 +08:00
|
|
|
if (Instruction *I = foldItoFPtoI(FI))
|
2015-02-17 05:47:54 +08:00
|
|
|
return I;
|
2010-01-04 15:53:58 +08:00
|
|
|
|
|
|
|
return commonCastTransforms(FI);
|
|
|
|
}
|
|
|
|
|
|
|
|
Instruction *InstCombiner::visitFPToSI(FPToSIInst &FI) {
|
2020-05-09 00:06:38 +08:00
|
|
|
if (Instruction *I = foldItoFPtoI(FI))
|
2015-02-17 05:47:54 +08:00
|
|
|
return I;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
return commonCastTransforms(FI);
|
|
|
|
}
|
|
|
|
|
|
|
|
Instruction *InstCombiner::visitUIToFP(CastInst &CI) {
|
|
|
|
return commonCastTransforms(CI);
|
|
|
|
}
|
|
|
|
|
|
|
|
Instruction *InstCombiner::visitSIToFP(CastInst &CI) {
|
|
|
|
return commonCastTransforms(CI);
|
|
|
|
}
|
|
|
|
|
|
|
|
Instruction *InstCombiner::visitIntToPtr(IntToPtrInst &CI) {
|
2010-02-02 09:44:02 +08:00
|
|
|
// If the source integer type is not the intptr_t type for this target, do a
|
|
|
|
// trunc or zext to the intptr_t type, then inttoptr of it. This allows the
|
|
|
|
// cast to be exposed to other transforms.
|
2015-03-10 10:37:25 +08:00
|
|
|
unsigned AS = CI.getAddressSpace();
|
|
|
|
if (CI.getOperand(0)->getType()->getScalarSizeInBits() !=
|
|
|
|
DL.getPointerSizeInBits(AS)) {
|
|
|
|
Type *Ty = DL.getIntPtrType(CI.getContext(), AS);
|
2020-04-09 01:42:22 +08:00
|
|
|
// Handle vectors of pointers.
|
|
|
|
if (auto *CIVTy = dyn_cast<VectorType>(CI.getType()))
|
|
|
|
Ty = VectorType::get(Ty, CIVTy->getElementCount());
|
2015-03-10 10:37:25 +08:00
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *P = Builder.CreateZExtOrTrunc(CI.getOperand(0), Ty);
|
2015-03-10 10:37:25 +08:00
|
|
|
return new IntToPtrInst(P, CI.getType());
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
if (Instruction *I = commonCastTransforms(CI))
|
|
|
|
return I;
|
|
|
|
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2018-05-02 00:10:38 +08:00
|
|
|
/// Implement the transforms for cast of pointer (bitcast/ptrtoint)
|
2010-01-06 06:21:18 +08:00
|
|
|
Instruction *InstCombiner::commonPointerCastTransforms(CastInst &CI) {
|
|
|
|
Value *Src = CI.getOperand(0);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-06 06:21:18 +08:00
|
|
|
if (GetElementPtrInst *GEP = dyn_cast<GetElementPtrInst>(Src)) {
|
|
|
|
// If casting the result of a getelementptr instruction with no offset, turn
|
|
|
|
// this into a cast of the original pointer!
|
2014-06-07 05:52:55 +08:00
|
|
|
if (GEP->hasAllZeroIndices() &&
|
|
|
|
// If CI is an addrspacecast and GEP changes the poiner type, merging
|
|
|
|
// GEP into CI would undo canonicalizing addrspacecast with different
|
|
|
|
// pointer types, causing infinite loops.
|
|
|
|
(!isa<AddrSpaceCastInst>(CI) ||
|
2017-04-19 06:00:54 +08:00
|
|
|
GEP->getType() == GEP->getPointerOperandType())) {
|
2010-01-06 06:21:18 +08:00
|
|
|
// Changing the cast operand is usually not a good idea but it is safe
|
2013-01-24 13:22:40 +08:00
|
|
|
// here because the pointer operand is being replaced with another
|
2010-01-06 06:21:18 +08:00
|
|
|
// pointer operand so the opcode doesn't need to change.
|
2020-02-01 05:23:33 +08:00
|
|
|
return replaceOperand(CI, 0, GEP->getOperand(0));
|
2010-01-06 06:21:18 +08:00
|
|
|
}
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-01-06 06:21:18 +08:00
|
|
|
return commonCastTransforms(CI);
|
|
|
|
}
|
|
|
|
|
|
|
|
Instruction *InstCombiner::visitPtrToInt(PtrToIntInst &CI) {
|
2010-02-02 09:44:02 +08:00
|
|
|
// If the destination integer type is not the intptr_t type for this target,
|
|
|
|
// do a ptrtoint to intptr_t then do a trunc or zext. This allows the cast
|
|
|
|
// to be exposed to other transforms.
|
2013-02-06 03:21:56 +08:00
|
|
|
|
2013-08-22 03:53:10 +08:00
|
|
|
Type *Ty = CI.getType();
|
|
|
|
unsigned AS = CI.getPointerAddressSpace();
|
|
|
|
|
2019-12-13 17:55:45 +08:00
|
|
|
if (Ty->getScalarSizeInBits() == DL.getPointerSizeInBits(AS))
|
2013-08-22 03:53:10 +08:00
|
|
|
return commonPointerCastTransforms(CI);
|
|
|
|
|
2015-03-10 10:37:25 +08:00
|
|
|
Type *PtrTy = DL.getIntPtrType(CI.getContext(), AS);
|
2020-05-30 06:24:15 +08:00
|
|
|
if (auto *VTy = dyn_cast<VectorType>(Ty)) {
|
|
|
|
// Handle vectors of pointers.
|
|
|
|
// FIXME: what should happen for scalable vectors?
|
|
|
|
PtrTy = FixedVectorType::get(PtrTy, VTy->getNumElements());
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *P = Builder.CreatePtrToInt(CI.getOperand(0), PtrTy);
|
2013-08-22 03:53:10 +08:00
|
|
|
return CastInst::CreateIntegerCast(P, Ty, /*isSigned=*/false);
|
2010-01-06 06:21:18 +08:00
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// This input value (which is known to have vector type) is being zero extended
|
2019-11-29 06:18:28 +08:00
|
|
|
/// or truncated to the specified vector type. Since the zext/trunc is done
|
|
|
|
/// using an integer type, we have a (bitcast(cast(bitcast))) pattern,
|
|
|
|
/// endianness will impact which end of the vector that is extended or
|
|
|
|
/// truncated.
|
|
|
|
///
|
|
|
|
/// A vector is always stored with index 0 at the lowest address, which
|
|
|
|
/// corresponds to the most significant bits for a big endian stored integer and
|
|
|
|
/// the least significant bits for little endian. A trunc/zext of an integer
|
|
|
|
/// impacts the big end of the integer. Thus, we need to add/remove elements at
|
|
|
|
/// the front of the vector for big endian targets, and the back of the vector
|
|
|
|
/// for little endian targets.
|
|
|
|
///
|
2015-09-09 22:34:26 +08:00
|
|
|
/// Try to replace it with a shuffle (and vector/vector bitcast) if possible.
|
2010-05-09 05:50:26 +08:00
|
|
|
///
|
|
|
|
/// The source and destination vector types may have different element types.
|
2019-11-29 06:18:28 +08:00
|
|
|
static Instruction *optimizeVectorResizeWithIntegerBitCasts(Value *InVal,
|
|
|
|
VectorType *DestTy,
|
|
|
|
InstCombiner &IC) {
|
2010-05-09 05:50:26 +08:00
|
|
|
// We can only do this optimization if the output is a multiple of the input
|
|
|
|
// element size, or the input is a multiple of the output element size.
|
|
|
|
// Convert the input type to have the same element type as the output.
|
2011-07-18 12:54:35 +08:00
|
|
|
VectorType *SrcTy = cast<VectorType>(InVal->getType());
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2010-05-09 05:50:26 +08:00
|
|
|
if (SrcTy->getElementType() != DestTy->getElementType()) {
|
|
|
|
// The input types don't need to be identical, but for now they must be the
|
|
|
|
// same size. There is no specific reason we couldn't handle things like
|
|
|
|
// <4 x i16> -> <4 x i32> by bitcasting to <2 x i32> but haven't gotten
|
2013-01-24 13:22:40 +08:00
|
|
|
// there yet.
|
2010-05-09 05:50:26 +08:00
|
|
|
if (SrcTy->getElementType()->getPrimitiveSizeInBits() !=
|
|
|
|
DestTy->getElementType()->getPrimitiveSizeInBits())
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-05-30 06:24:15 +08:00
|
|
|
SrcTy =
|
|
|
|
FixedVectorType::get(DestTy->getElementType(), SrcTy->getNumElements());
|
2017-07-08 07:16:26 +08:00
|
|
|
InVal = IC.Builder.CreateBitCast(InVal, SrcTy);
|
2010-05-09 05:50:26 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2019-11-29 06:18:28 +08:00
|
|
|
bool IsBigEndian = IC.getDataLayout().isBigEndian();
|
|
|
|
unsigned SrcElts = SrcTy->getNumElements();
|
|
|
|
unsigned DestElts = DestTy->getNumElements();
|
|
|
|
|
|
|
|
assert(SrcElts != DestElts && "Element counts should be different.");
|
|
|
|
|
2010-05-09 05:50:26 +08:00
|
|
|
// Now that the element types match, get the shuffle mask and RHS of the
|
|
|
|
// shuffle to use, which depends on whether we're increasing or decreasing the
|
|
|
|
// size of the input.
|
2020-04-15 20:29:09 +08:00
|
|
|
SmallVector<int, 16> ShuffleMaskStorage;
|
|
|
|
ArrayRef<int> ShuffleMask;
|
2010-05-09 05:50:26 +08:00
|
|
|
Value *V2;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2019-11-29 06:18:28 +08:00
|
|
|
// Produce an identify shuffle mask for the src vector.
|
|
|
|
ShuffleMaskStorage.resize(SrcElts);
|
|
|
|
std::iota(ShuffleMaskStorage.begin(), ShuffleMaskStorage.end(), 0);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2019-11-29 06:18:28 +08:00
|
|
|
if (SrcElts > DestElts) {
|
|
|
|
// If we're shrinking the number of elements (rewriting an integer
|
|
|
|
// truncate), just shuffle in the elements corresponding to the least
|
|
|
|
// significant bits from the input and use undef as the second shuffle
|
|
|
|
// input.
|
|
|
|
V2 = UndefValue::get(SrcTy);
|
|
|
|
// Make sure the shuffle mask selects the "least significant bits" by
|
|
|
|
// keeping elements from back of the src vector for big endian, and from the
|
|
|
|
// front for little endian.
|
|
|
|
ShuffleMask = ShuffleMaskStorage;
|
|
|
|
if (IsBigEndian)
|
|
|
|
ShuffleMask = ShuffleMask.take_back(DestElts);
|
|
|
|
else
|
|
|
|
ShuffleMask = ShuffleMask.take_front(DestElts);
|
2010-05-09 05:50:26 +08:00
|
|
|
} else {
|
2019-11-29 06:18:28 +08:00
|
|
|
// If we're increasing the number of elements (rewriting an integer zext),
|
|
|
|
// shuffle in all of the elements from InVal. Fill the rest of the result
|
|
|
|
// elements with zeros from a constant zero.
|
2010-05-09 05:50:26 +08:00
|
|
|
V2 = Constant::getNullValue(SrcTy);
|
2019-11-29 06:18:28 +08:00
|
|
|
// Use first elt from V2 when indicating zero in the shuffle mask.
|
|
|
|
uint32_t NullElt = SrcElts;
|
|
|
|
// Extend with null values in the "most significant bits" by adding elements
|
|
|
|
// in front of the src vector for big endian, and at the back for little
|
|
|
|
// endian.
|
|
|
|
unsigned DeltaElts = DestElts - SrcElts;
|
|
|
|
if (IsBigEndian)
|
|
|
|
ShuffleMaskStorage.insert(ShuffleMaskStorage.begin(), DeltaElts, NullElt);
|
|
|
|
else
|
|
|
|
ShuffleMaskStorage.append(DeltaElts, NullElt);
|
|
|
|
ShuffleMask = ShuffleMaskStorage;
|
2010-05-09 05:50:26 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-04-15 20:29:09 +08:00
|
|
|
return new ShuffleVectorInst(InVal, V2, ShuffleMask);
|
2010-05-09 05:50:26 +08:00
|
|
|
}
|
|
|
|
|
2011-07-18 12:54:35 +08:00
|
|
|
static bool isMultipleOfTypeSize(unsigned Value, Type *Ty) {
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
return Value % Ty->getPrimitiveSizeInBits() == 0;
|
|
|
|
}
|
|
|
|
|
2011-07-18 12:54:35 +08:00
|
|
|
static unsigned getTypeSizeIndex(unsigned Value, Type *Ty) {
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
return Value / Ty->getPrimitiveSizeInBits();
|
|
|
|
}
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// V is a value which is inserted into a vector of VecEltTy.
|
|
|
|
/// Look through the value to see if we can decompose it into
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
/// insertions into the vector. See the example in the comment for
|
|
|
|
/// OptimizeIntegerToVectorInsertions for the pattern this handles.
|
|
|
|
/// The type of V is always a non-zero multiple of VecEltTy's size.
|
2013-08-12 15:26:09 +08:00
|
|
|
/// Shift is the number of bits between the lsb of V and the lsb of
|
|
|
|
/// the vector.
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
///
|
|
|
|
/// This returns false if the pattern can't be matched or true if it can,
|
|
|
|
/// filling in Elements with the elements found here.
|
2015-09-09 22:54:29 +08:00
|
|
|
static bool collectInsertionElements(Value *V, unsigned Shift,
|
2015-03-10 10:37:25 +08:00
|
|
|
SmallVectorImpl<Value *> &Elements,
|
|
|
|
Type *VecEltTy, bool isBigEndian) {
|
2013-08-12 15:26:09 +08:00
|
|
|
assert(isMultipleOfTypeSize(Shift, VecEltTy) &&
|
|
|
|
"Shift should be a multiple of the element type size");
|
|
|
|
|
2010-08-28 11:36:51 +08:00
|
|
|
// Undef values never contribute useful bits to the result.
|
|
|
|
if (isa<UndefValue>(V)) return true;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
// If we got down to a value of the right type, we win, try inserting into the
|
|
|
|
// right element.
|
|
|
|
if (V->getType() == VecEltTy) {
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
// Inserting null doesn't actually insert any elements.
|
|
|
|
if (Constant *C = dyn_cast<Constant>(V))
|
|
|
|
if (C->isNullValue())
|
|
|
|
return true;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2013-08-12 15:26:09 +08:00
|
|
|
unsigned ElementIndex = getTypeSizeIndex(Shift, VecEltTy);
|
2015-03-10 10:37:25 +08:00
|
|
|
if (isBigEndian)
|
2013-08-12 15:26:09 +08:00
|
|
|
ElementIndex = Elements.size() - ElementIndex - 1;
|
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
// Fail if multiple elements are inserted into this slot.
|
2014-04-25 13:29:35 +08:00
|
|
|
if (Elements[ElementIndex])
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
return false;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
Elements[ElementIndex] = V;
|
|
|
|
return true;
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
if (Constant *C = dyn_cast<Constant>(V)) {
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
// Figure out the # elements this provides, and bitcast it or slice it up
|
|
|
|
// as required.
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
unsigned NumElts = getTypeSizeIndex(C->getType()->getPrimitiveSizeInBits(),
|
|
|
|
VecEltTy);
|
|
|
|
// If the constant is the size of a vector element, we just need to bitcast
|
|
|
|
// it to the right type so it gets properly inserted.
|
|
|
|
if (NumElts == 1)
|
2015-09-09 22:54:29 +08:00
|
|
|
return collectInsertionElements(ConstantExpr::getBitCast(C, VecEltTy),
|
2015-03-10 10:37:25 +08:00
|
|
|
Shift, Elements, VecEltTy, isBigEndian);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
// Okay, this is a constant that covers multiple elements. Slice it up into
|
|
|
|
// pieces and insert each element-sized piece into the vector.
|
|
|
|
if (!isa<IntegerType>(C->getType()))
|
|
|
|
C = ConstantExpr::getBitCast(C, IntegerType::get(V->getContext(),
|
|
|
|
C->getType()->getPrimitiveSizeInBits()));
|
|
|
|
unsigned ElementSize = VecEltTy->getPrimitiveSizeInBits();
|
2011-07-18 12:54:35 +08:00
|
|
|
Type *ElementIntTy = IntegerType::get(C->getContext(), ElementSize);
|
2013-01-24 13:22:40 +08:00
|
|
|
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
for (unsigned i = 0; i != NumElts; ++i) {
|
2013-08-12 15:26:09 +08:00
|
|
|
unsigned ShiftI = Shift+i*ElementSize;
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
Constant *Piece = ConstantExpr::getLShr(C, ConstantInt::get(C->getType(),
|
2013-08-12 15:26:09 +08:00
|
|
|
ShiftI));
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
Piece = ConstantExpr::getTrunc(Piece, ElementIntTy);
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!collectInsertionElements(Piece, ShiftI, Elements, VecEltTy,
|
2015-03-10 10:37:25 +08:00
|
|
|
isBigEndian))
|
handle the constant case of vector insertion. For something
like this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.B;
A.A = 42;
return A;
}
we now generate:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
pshufd $16, %xmm2, %xmm2
movss LCPI1_1(%rip), %xmm0
pshufd $16, %xmm0, %xmm0
unpcklps %xmm2, %xmm0
ret
instead of:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss 12(%rax), %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
unpcklps %xmm0, %xmm1
addss LCPI1_0(%rip), %xmm2
movd %xmm2, %eax
shlq $32, %rax
addq $1109917696, %rax ## imm = 0x42280000
movd %rax, %xmm0
ret
llvm-svn: 112345
2010-08-28 09:50:57 +08:00
|
|
|
return false;
|
|
|
|
}
|
|
|
|
return true;
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
if (!V->hasOneUse()) return false;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
Instruction *I = dyn_cast<Instruction>(V);
|
2014-04-25 13:29:35 +08:00
|
|
|
if (!I) return false;
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
switch (I->getOpcode()) {
|
|
|
|
default: return false; // Unhandled case.
|
|
|
|
case Instruction::BitCast:
|
2015-09-09 22:54:29 +08:00
|
|
|
return collectInsertionElements(I->getOperand(0), Shift, Elements, VecEltTy,
|
2015-03-10 10:37:25 +08:00
|
|
|
isBigEndian);
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
case Instruction::ZExt:
|
|
|
|
if (!isMultipleOfTypeSize(
|
|
|
|
I->getOperand(0)->getType()->getPrimitiveSizeInBits(),
|
|
|
|
VecEltTy))
|
|
|
|
return false;
|
2015-09-09 22:54:29 +08:00
|
|
|
return collectInsertionElements(I->getOperand(0), Shift, Elements, VecEltTy,
|
2015-03-10 10:37:25 +08:00
|
|
|
isBigEndian);
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
case Instruction::Or:
|
2015-09-09 22:54:29 +08:00
|
|
|
return collectInsertionElements(I->getOperand(0), Shift, Elements, VecEltTy,
|
2015-03-10 10:37:25 +08:00
|
|
|
isBigEndian) &&
|
2015-09-09 22:54:29 +08:00
|
|
|
collectInsertionElements(I->getOperand(1), Shift, Elements, VecEltTy,
|
2015-03-10 10:37:25 +08:00
|
|
|
isBigEndian);
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
case Instruction::Shl: {
|
|
|
|
// Must be shifting by a constant that is a multiple of the element size.
|
|
|
|
ConstantInt *CI = dyn_cast<ConstantInt>(I->getOperand(1));
|
2014-04-25 13:29:35 +08:00
|
|
|
if (!CI) return false;
|
2013-08-12 15:26:09 +08:00
|
|
|
Shift += CI->getZExtValue();
|
|
|
|
if (!isMultipleOfTypeSize(Shift, VecEltTy)) return false;
|
2015-09-09 22:54:29 +08:00
|
|
|
return collectInsertionElements(I->getOperand(0), Shift, Elements, VecEltTy,
|
2015-03-10 10:37:25 +08:00
|
|
|
isBigEndian);
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
|
|
|
|
2015-09-09 22:34:26 +08:00
|
|
|
/// If the input is an 'or' instruction, we may be doing shifts and ors to
|
|
|
|
/// assemble the elements of the vector manually.
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
/// Try to rip the code out and replace it with insertelements. This is to
|
|
|
|
/// optimize code like this:
|
|
|
|
///
|
|
|
|
/// %tmp37 = bitcast float %inc to i32
|
|
|
|
/// %tmp38 = zext i32 %tmp37 to i64
|
|
|
|
/// %tmp31 = bitcast float %inc5 to i32
|
|
|
|
/// %tmp32 = zext i32 %tmp31 to i64
|
|
|
|
/// %tmp33 = shl i64 %tmp32, 32
|
|
|
|
/// %ins35 = or i64 %tmp33, %tmp38
|
|
|
|
/// %tmp43 = bitcast i64 %ins35 to <2 x float>
|
|
|
|
///
|
|
|
|
/// Into two insertelements that do "buildvector{%inc, %inc5}".
|
2015-09-09 22:54:29 +08:00
|
|
|
static Value *optimizeIntegerToVectorInsertions(BitCastInst &CI,
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
InstCombiner &IC) {
|
2011-07-18 12:54:35 +08:00
|
|
|
VectorType *DestVecTy = cast<VectorType>(CI.getType());
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
Value *IntInput = CI.getOperand(0);
|
|
|
|
|
|
|
|
SmallVector<Value*, 8> Elements(DestVecTy->getNumElements());
|
2015-09-09 22:54:29 +08:00
|
|
|
if (!collectInsertionElements(IntInput, 0, Elements,
|
2015-03-10 10:37:25 +08:00
|
|
|
DestVecTy->getElementType(),
|
|
|
|
IC.getDataLayout().isBigEndian()))
|
2014-04-25 13:29:35 +08:00
|
|
|
return nullptr;
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
|
|
|
|
// If we succeeded, we know that all of the element are specified by Elements
|
|
|
|
// or are zero if Elements has a null entry. Recast this as a set of
|
|
|
|
// insertions.
|
|
|
|
Value *Result = Constant::getNullValue(CI.getType());
|
|
|
|
for (unsigned i = 0, e = Elements.size(); i != e; ++i) {
|
2014-04-25 13:29:35 +08:00
|
|
|
if (!Elements[i]) continue; // Unset element.
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
Result = IC.Builder.CreateInsertElement(Result, Elements[i],
|
|
|
|
IC.Builder.getInt32(i));
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
return Result;
|
|
|
|
}
|
|
|
|
|
2015-12-13 00:44:48 +08:00
|
|
|
/// Canonicalize scalar bitcasts of extracted elements into a bitcast of the
|
|
|
|
/// vector followed by extract element. The backend tends to handle bitcasts of
|
|
|
|
/// vectors better than bitcasts of scalars because vector registers are
|
|
|
|
/// usually not type-specific like scalar integer or scalar floating-point.
|
|
|
|
static Instruction *canonicalizeBitCastExtElt(BitCastInst &BitCast,
|
2017-07-07 07:18:43 +08:00
|
|
|
InstCombiner &IC) {
|
2015-12-11 01:09:28 +08:00
|
|
|
// TODO: Create and use a pattern matcher for ExtractElementInst.
|
|
|
|
auto *ExtElt = dyn_cast<ExtractElementInst>(BitCast.getOperand(0));
|
|
|
|
if (!ExtElt || !ExtElt->hasOneUse())
|
|
|
|
return nullptr;
|
|
|
|
|
2015-12-13 00:44:48 +08:00
|
|
|
// The bitcast must be to a vectorizable type, otherwise we can't make a new
|
|
|
|
// type to extract from.
|
|
|
|
Type *DestType = BitCast.getType();
|
|
|
|
if (!VectorType::isValidElementType(DestType))
|
2015-12-11 01:09:28 +08:00
|
|
|
return nullptr;
|
|
|
|
|
2015-12-13 00:44:48 +08:00
|
|
|
unsigned NumElts = ExtElt->getVectorOperandType()->getNumElements();
|
2020-05-30 06:24:15 +08:00
|
|
|
auto *NewVecType = FixedVectorType::get(DestType, NumElts);
|
2017-07-08 07:16:26 +08:00
|
|
|
auto *NewBC = IC.Builder.CreateBitCast(ExtElt->getVectorOperand(),
|
|
|
|
NewVecType, "bc");
|
2015-12-13 00:44:48 +08:00
|
|
|
return ExtractElementInst::Create(NewBC, ExtElt->getIndexOperand());
|
2015-12-11 01:09:28 +08:00
|
|
|
}
|
|
|
|
|
2016-11-23 06:05:48 +08:00
|
|
|
/// Change the type of a bitwise logic operation if we can eliminate a bitcast.
|
|
|
|
static Instruction *foldBitCastBitwiseLogic(BitCastInst &BitCast,
|
|
|
|
InstCombiner::BuilderTy &Builder) {
|
|
|
|
Type *DestTy = BitCast.getType();
|
2016-11-23 06:54:36 +08:00
|
|
|
BinaryOperator *BO;
|
2017-07-09 15:04:00 +08:00
|
|
|
if (!DestTy->isIntOrIntVectorTy() ||
|
2016-11-23 06:54:36 +08:00
|
|
|
!match(BitCast.getOperand(0), m_OneUse(m_BinOp(BO))) ||
|
|
|
|
!BO->isBitwiseLogicOp())
|
2016-11-23 06:05:48 +08:00
|
|
|
return nullptr;
|
2018-02-14 14:58:08 +08:00
|
|
|
|
2016-11-23 06:05:48 +08:00
|
|
|
// FIXME: This transform is restricted to vector types to avoid backend
|
|
|
|
// problems caused by creating potentially illegal operations. If a fix-up is
|
|
|
|
// added to handle that situation, we can remove this check.
|
|
|
|
if (!DestTy->isVectorTy() || !BO->getType()->isVectorTy())
|
|
|
|
return nullptr;
|
2018-02-14 14:58:08 +08:00
|
|
|
|
2016-11-23 06:05:48 +08:00
|
|
|
Value *X;
|
|
|
|
if (match(BO->getOperand(0), m_OneUse(m_BitCast(m_Value(X)))) &&
|
|
|
|
X->getType() == DestTy && !isa<Constant>(X)) {
|
|
|
|
// bitcast(logic(bitcast(X), Y)) --> logic'(X, bitcast(Y))
|
|
|
|
Value *CastedOp1 = Builder.CreateBitCast(BO->getOperand(1), DestTy);
|
2016-11-23 06:54:36 +08:00
|
|
|
return BinaryOperator::Create(BO->getOpcode(), X, CastedOp1);
|
2016-11-23 06:05:48 +08:00
|
|
|
}
|
|
|
|
|
|
|
|
if (match(BO->getOperand(1), m_OneUse(m_BitCast(m_Value(X)))) &&
|
|
|
|
X->getType() == DestTy && !isa<Constant>(X)) {
|
|
|
|
// bitcast(logic(Y, bitcast(X))) --> logic'(bitcast(Y), X)
|
|
|
|
Value *CastedOp0 = Builder.CreateBitCast(BO->getOperand(0), DestTy);
|
2016-11-23 06:54:36 +08:00
|
|
|
return BinaryOperator::Create(BO->getOpcode(), CastedOp0, X);
|
2016-11-23 06:05:48 +08:00
|
|
|
}
|
|
|
|
|
2017-06-22 23:46:54 +08:00
|
|
|
// Canonicalize vector bitcasts to come before vector bitwise logic with a
|
|
|
|
// constant. This eases recognition of special constants for later ops.
|
|
|
|
// Example:
|
|
|
|
// icmp u/s (a ^ signmask), (b ^ signmask) --> icmp s/u a, b
|
|
|
|
Constant *C;
|
|
|
|
if (match(BO->getOperand(1), m_Constant(C))) {
|
|
|
|
// bitcast (logic X, C) --> logic (bitcast X, C')
|
|
|
|
Value *CastedOp0 = Builder.CreateBitCast(BO->getOperand(0), DestTy);
|
2020-02-29 05:29:54 +08:00
|
|
|
Value *CastedC = Builder.CreateBitCast(C, DestTy);
|
2017-06-22 23:46:54 +08:00
|
|
|
return BinaryOperator::Create(BO->getOpcode(), CastedOp0, CastedC);
|
|
|
|
}
|
|
|
|
|
2016-11-23 06:05:48 +08:00
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
2016-12-03 23:25:16 +08:00
|
|
|
/// Change the type of a select if we can eliminate a bitcast.
|
|
|
|
static Instruction *foldBitCastSelect(BitCastInst &BitCast,
|
|
|
|
InstCombiner::BuilderTy &Builder) {
|
|
|
|
Value *Cond, *TVal, *FVal;
|
|
|
|
if (!match(BitCast.getOperand(0),
|
|
|
|
m_OneUse(m_Select(m_Value(Cond), m_Value(TVal), m_Value(FVal)))))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
// A vector select must maintain the same number of elements in its operands.
|
|
|
|
Type *CondTy = Cond->getType();
|
|
|
|
Type *DestTy = BitCast.getType();
|
2020-04-09 01:42:22 +08:00
|
|
|
if (auto *CondVTy = dyn_cast<VectorType>(CondTy)) {
|
2016-12-03 23:25:16 +08:00
|
|
|
if (!DestTy->isVectorTy())
|
|
|
|
return nullptr;
|
2020-04-09 01:42:22 +08:00
|
|
|
if (cast<VectorType>(DestTy)->getNumElements() != CondVTy->getNumElements())
|
2016-12-03 23:25:16 +08:00
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
|
|
|
// FIXME: This transform is restricted from changing the select between
|
|
|
|
// scalars and vectors to avoid backend problems caused by creating
|
|
|
|
// potentially illegal operations. If a fix-up is added to handle that
|
|
|
|
// situation, we can remove this check.
|
|
|
|
if (DestTy->isVectorTy() != TVal->getType()->isVectorTy())
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
auto *Sel = cast<Instruction>(BitCast.getOperand(0));
|
|
|
|
Value *X;
|
|
|
|
if (match(TVal, m_OneUse(m_BitCast(m_Value(X)))) && X->getType() == DestTy &&
|
|
|
|
!isa<Constant>(X)) {
|
|
|
|
// bitcast(select(Cond, bitcast(X), Y)) --> select'(Cond, X, bitcast(Y))
|
|
|
|
Value *CastedVal = Builder.CreateBitCast(FVal, DestTy);
|
|
|
|
return SelectInst::Create(Cond, X, CastedVal, "", nullptr, Sel);
|
|
|
|
}
|
|
|
|
|
|
|
|
if (match(FVal, m_OneUse(m_BitCast(m_Value(X)))) && X->getType() == DestTy &&
|
|
|
|
!isa<Constant>(X)) {
|
|
|
|
// bitcast(select(Cond, Y, bitcast(X))) --> select'(Cond, bitcast(Y), X)
|
|
|
|
Value *CastedVal = Builder.CreateBitCast(TVal, DestTy);
|
|
|
|
return SelectInst::Create(Cond, CastedVal, X, "", nullptr, Sel);
|
|
|
|
}
|
|
|
|
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
2016-10-26 04:43:42 +08:00
|
|
|
/// Check if all users of CI are StoreInsts.
|
|
|
|
static bool hasStoreUsersOnly(CastInst &CI) {
|
|
|
|
for (User *U : CI.users()) {
|
|
|
|
if (!isa<StoreInst>(U))
|
|
|
|
return false;
|
|
|
|
}
|
|
|
|
return true;
|
|
|
|
}
|
|
|
|
|
|
|
|
/// This function handles following case
|
|
|
|
///
|
|
|
|
/// A -> B cast
|
|
|
|
/// PHI
|
|
|
|
/// B -> A cast
|
|
|
|
///
|
|
|
|
/// All the related PHI nodes can be replaced by new PHI nodes with type A.
|
|
|
|
/// The uses of \p CI can be changed to the new PHI node corresponding to \p PN.
|
|
|
|
Instruction *InstCombiner::optimizeBitCastFromPhi(CastInst &CI, PHINode *PN) {
|
|
|
|
// BitCast used by Store can be handled in InstCombineLoadStoreAlloca.cpp.
|
|
|
|
if (hasStoreUsersOnly(CI))
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
Value *Src = CI.getOperand(0);
|
|
|
|
Type *SrcTy = Src->getType(); // Type B
|
|
|
|
Type *DestTy = CI.getType(); // Type A
|
|
|
|
|
|
|
|
SmallVector<PHINode *, 4> PhiWorklist;
|
|
|
|
SmallSetVector<PHINode *, 4> OldPhiNodes;
|
|
|
|
|
|
|
|
// Find all of the A->B casts and PHI nodes.
|
2019-02-09 09:44:28 +08:00
|
|
|
// We need to inspect all related PHI nodes, but PHIs can be cyclic, so
|
2016-10-26 04:43:42 +08:00
|
|
|
// OldPhiNodes is used to track all known PHI nodes, before adding a new
|
|
|
|
// PHI to PhiWorklist, it is checked against and added to OldPhiNodes first.
|
|
|
|
PhiWorklist.push_back(PN);
|
|
|
|
OldPhiNodes.insert(PN);
|
|
|
|
while (!PhiWorklist.empty()) {
|
|
|
|
auto *OldPN = PhiWorklist.pop_back_val();
|
|
|
|
for (Value *IncValue : OldPN->incoming_values()) {
|
|
|
|
if (isa<Constant>(IncValue))
|
|
|
|
continue;
|
|
|
|
|
|
|
|
if (auto *LI = dyn_cast<LoadInst>(IncValue)) {
|
|
|
|
// If there is a sequence of one or more load instructions, each loaded
|
|
|
|
// value is used as address of later load instruction, bitcast is
|
|
|
|
// necessary to change the value type, don't optimize it. For
|
|
|
|
// simplicity we give up if the load address comes from another load.
|
|
|
|
Value *Addr = LI->getOperand(0);
|
|
|
|
if (Addr == &CI || isa<LoadInst>(Addr))
|
|
|
|
return nullptr;
|
|
|
|
if (LI->hasOneUse() && LI->isSimple())
|
|
|
|
continue;
|
|
|
|
// If a LoadInst has more than one use, changing the type of loaded
|
|
|
|
// value may create another bitcast.
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
|
|
|
|
if (auto *PNode = dyn_cast<PHINode>(IncValue)) {
|
|
|
|
if (OldPhiNodes.insert(PNode))
|
|
|
|
PhiWorklist.push_back(PNode);
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
|
|
|
|
auto *BCI = dyn_cast<BitCastInst>(IncValue);
|
|
|
|
// We can't handle other instructions.
|
|
|
|
if (!BCI)
|
|
|
|
return nullptr;
|
|
|
|
|
|
|
|
// Verify it's a A->B cast.
|
|
|
|
Type *TyA = BCI->getOperand(0)->getType();
|
|
|
|
Type *TyB = BCI->getType();
|
|
|
|
if (TyA != DestTy || TyB != SrcTy)
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-12-31 19:11:35 +08:00
|
|
|
// Check that each user of each old PHI node is something that we can
|
|
|
|
// rewrite, so that all of the old PHI nodes can be cleaned up afterwards.
|
|
|
|
for (auto *OldPN : OldPhiNodes) {
|
|
|
|
for (User *V : OldPN->users()) {
|
|
|
|
if (auto *SI = dyn_cast<StoreInst>(V)) {
|
|
|
|
if (!SI->isSimple() || SI->getOperand(0) != OldPN)
|
|
|
|
return nullptr;
|
|
|
|
} else if (auto *BCI = dyn_cast<BitCastInst>(V)) {
|
|
|
|
// Verify it's a B->A cast.
|
|
|
|
Type *TyB = BCI->getOperand(0)->getType();
|
|
|
|
Type *TyA = BCI->getType();
|
|
|
|
if (TyA != DestTy || TyB != SrcTy)
|
|
|
|
return nullptr;
|
|
|
|
} else if (auto *PHI = dyn_cast<PHINode>(V)) {
|
|
|
|
// As long as the user is another old PHI node, then even if we don't
|
|
|
|
// rewrite it, the PHI web we're considering won't have any users
|
|
|
|
// outside itself, so it'll be dead.
|
|
|
|
if (OldPhiNodes.count(PHI) == 0)
|
|
|
|
return nullptr;
|
|
|
|
} else {
|
|
|
|
return nullptr;
|
|
|
|
}
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2016-10-26 04:43:42 +08:00
|
|
|
// For each old PHI node, create a corresponding new PHI node with a type A.
|
|
|
|
SmallDenseMap<PHINode *, PHINode *> NewPNodes;
|
|
|
|
for (auto *OldPN : OldPhiNodes) {
|
2017-07-08 07:16:26 +08:00
|
|
|
Builder.SetInsertPoint(OldPN);
|
|
|
|
PHINode *NewPN = Builder.CreatePHI(DestTy, OldPN->getNumOperands());
|
2016-10-26 04:43:42 +08:00
|
|
|
NewPNodes[OldPN] = NewPN;
|
|
|
|
}
|
|
|
|
|
|
|
|
// Fill in the operands of new PHI nodes.
|
|
|
|
for (auto *OldPN : OldPhiNodes) {
|
|
|
|
PHINode *NewPN = NewPNodes[OldPN];
|
|
|
|
for (unsigned j = 0, e = OldPN->getNumOperands(); j != e; ++j) {
|
|
|
|
Value *V = OldPN->getOperand(j);
|
|
|
|
Value *NewV = nullptr;
|
|
|
|
if (auto *C = dyn_cast<Constant>(V)) {
|
|
|
|
NewV = ConstantExpr::getBitCast(C, DestTy);
|
|
|
|
} else if (auto *LI = dyn_cast<LoadInst>(V)) {
|
2020-01-14 01:57:14 +08:00
|
|
|
// Explicitly perform load combine to make sure no opposing transform
|
|
|
|
// can remove the bitcast in the meantime and trigger an infinite loop.
|
|
|
|
Builder.SetInsertPoint(LI);
|
|
|
|
NewV = combineLoadToNewType(*LI, DestTy);
|
|
|
|
// Remove the old load and its use in the old phi, which itself becomes
|
|
|
|
// dead once the whole transform finishes.
|
|
|
|
replaceInstUsesWith(*LI, UndefValue::get(LI->getType()));
|
|
|
|
eraseInstFromFunction(*LI);
|
2016-10-26 04:43:42 +08:00
|
|
|
} else if (auto *BCI = dyn_cast<BitCastInst>(V)) {
|
|
|
|
NewV = BCI->getOperand(0);
|
|
|
|
} else if (auto *PrevPN = dyn_cast<PHINode>(V)) {
|
|
|
|
NewV = NewPNodes[PrevPN];
|
|
|
|
}
|
|
|
|
assert(NewV);
|
|
|
|
NewPN->addIncoming(NewV, OldPN->getIncomingBlock(j));
|
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-02-09 09:44:28 +08:00
|
|
|
// Traverse all accumulated PHI nodes and process its users,
|
|
|
|
// which are Stores and BitcCasts. Without this processing
|
|
|
|
// NewPHI nodes could be replicated and could lead to extra
|
|
|
|
// moves generated after DeSSA.
|
2016-10-26 04:43:42 +08:00
|
|
|
// If there is a store with type B, change it to type A.
|
2019-02-09 09:44:28 +08:00
|
|
|
|
|
|
|
|
|
|
|
// Replace users of BitCast B->A with NewPHI. These will help
|
|
|
|
// later to get rid off a closure formed by OldPHI nodes.
|
|
|
|
Instruction *RetVal = nullptr;
|
|
|
|
for (auto *OldPN : OldPhiNodes) {
|
|
|
|
PHINode *NewPN = NewPNodes[OldPN];
|
2020-01-14 06:54:42 +08:00
|
|
|
for (auto It = OldPN->user_begin(), End = OldPN->user_end(); It != End; ) {
|
|
|
|
User *V = *It;
|
|
|
|
// We may remove this user, advance to avoid iterator invalidation.
|
|
|
|
++It;
|
2019-02-09 09:44:28 +08:00
|
|
|
if (auto *SI = dyn_cast<StoreInst>(V)) {
|
2019-12-31 19:11:35 +08:00
|
|
|
assert(SI->isSimple() && SI->getOperand(0) == OldPN);
|
|
|
|
Builder.SetInsertPoint(SI);
|
|
|
|
auto *NewBC =
|
|
|
|
cast<BitCastInst>(Builder.CreateBitCast(NewPN, SrcTy));
|
|
|
|
SI->setOperand(0, NewBC);
|
2020-01-31 05:32:46 +08:00
|
|
|
Worklist.push(SI);
|
2019-12-31 19:11:35 +08:00
|
|
|
assert(hasStoreUsersOnly(*NewBC));
|
2019-02-09 09:44:28 +08:00
|
|
|
}
|
|
|
|
else if (auto *BCI = dyn_cast<BitCastInst>(V)) {
|
|
|
|
Type *TyB = BCI->getOperand(0)->getType();
|
|
|
|
Type *TyA = BCI->getType();
|
2019-12-31 19:11:35 +08:00
|
|
|
assert(TyA == DestTy && TyB == SrcTy);
|
|
|
|
(void) TyA;
|
|
|
|
(void) TyB;
|
|
|
|
Instruction *I = replaceInstUsesWith(*BCI, NewPN);
|
|
|
|
if (BCI == &CI)
|
|
|
|
RetVal = I;
|
|
|
|
} else if (auto *PHI = dyn_cast<PHINode>(V)) {
|
|
|
|
assert(OldPhiNodes.count(PHI) > 0);
|
|
|
|
(void) PHI;
|
|
|
|
} else {
|
|
|
|
llvm_unreachable("all uses should be handled");
|
2019-02-09 09:44:28 +08:00
|
|
|
}
|
2016-10-26 04:43:42 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-02-09 09:44:28 +08:00
|
|
|
return RetVal;
|
2016-10-26 04:43:42 +08:00
|
|
|
}
|
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
Instruction *InstCombiner::visitBitCast(BitCastInst &CI) {
|
|
|
|
// If the operands are integer typed then apply the integer transforms,
|
|
|
|
// otherwise just apply the common ones.
|
|
|
|
Value *Src = CI.getOperand(0);
|
2011-07-18 12:54:35 +08:00
|
|
|
Type *SrcTy = Src->getType();
|
|
|
|
Type *DestTy = CI.getType();
|
2010-01-04 15:53:58 +08:00
|
|
|
|
|
|
|
// Get rid of casts from one type to the same type. These are useless and can
|
|
|
|
// be replaced by the operand.
|
|
|
|
if (DestTy == Src->getType())
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, Src);
|
2010-01-04 15:53:58 +08:00
|
|
|
|
2011-07-18 12:54:35 +08:00
|
|
|
if (PointerType *DstPTy = dyn_cast<PointerType>(DestTy)) {
|
|
|
|
PointerType *SrcPTy = cast<PointerType>(SrcTy);
|
|
|
|
Type *DstElTy = DstPTy->getElementType();
|
|
|
|
Type *SrcElTy = SrcPTy->getElementType();
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2018-07-31 23:53:03 +08:00
|
|
|
// Casting pointers between the same type, but with different address spaces
|
|
|
|
// is an addrspace cast rather than a bitcast.
|
|
|
|
if ((DstElTy == SrcElTy) &&
|
|
|
|
(DstPTy->getAddressSpace() != SrcPTy->getAddressSpace()))
|
|
|
|
return new AddrSpaceCastInst(Src, DestTy);
|
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// If we are casting a alloca to a pointer to a type of the same
|
|
|
|
// size, rewrite the allocation instruction to allocate the "right" type.
|
|
|
|
// There is no need to modify malloc calls because it is their bitcast that
|
|
|
|
// needs to be cleaned up.
|
|
|
|
if (AllocaInst *AI = dyn_cast<AllocaInst>(Src))
|
|
|
|
if (Instruction *V = PromoteCastOfAllocation(CI, *AI))
|
|
|
|
return V;
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2016-05-24 03:23:17 +08:00
|
|
|
// When the type pointed to is not sized the cast cannot be
|
|
|
|
// turned into a gep.
|
|
|
|
Type *PointeeType =
|
|
|
|
cast<PointerType>(Src->getType()->getScalarType())->getElementType();
|
|
|
|
if (!PointeeType->isSized())
|
|
|
|
return nullptr;
|
|
|
|
|
2010-01-04 15:53:58 +08:00
|
|
|
// If the source and destination are pointers, and this cast is equivalent
|
|
|
|
// to a getelementptr X, 0, 0, 0... turn it into the appropriate gep.
|
|
|
|
// This can enhance SROA and other transforms that want type-safe pointers.
|
|
|
|
unsigned NumZeros = 0;
|
2020-03-04 07:42:16 +08:00
|
|
|
while (SrcElTy && SrcElTy != DstElTy) {
|
|
|
|
SrcElTy = GetElementPtrInst::getTypeAtIndex(SrcElTy, (uint64_t)0);
|
2010-01-04 15:53:58 +08:00
|
|
|
++NumZeros;
|
|
|
|
}
|
|
|
|
|
|
|
|
// If we found a path from the src to dest, create the getelementptr now.
|
|
|
|
if (SrcElTy == DstElTy) {
|
2017-07-08 07:16:26 +08:00
|
|
|
SmallVector<Value *, 8> Idxs(NumZeros + 1, Builder.getInt32(0));
|
2019-10-06 21:08:08 +08:00
|
|
|
GetElementPtrInst *GEP =
|
|
|
|
GetElementPtrInst::Create(SrcPTy->getElementType(), Src, Idxs);
|
|
|
|
|
|
|
|
// If the source pointer is dereferenceable, then assume it points to an
|
|
|
|
// allocated object and apply "inbounds" to the GEP.
|
|
|
|
bool CanBeNull;
|
2019-10-14 01:19:08 +08:00
|
|
|
if (Src->getPointerDereferenceableBytes(DL, CanBeNull)) {
|
|
|
|
// In a non-default address space (not 0), a null pointer can not be
|
|
|
|
// assumed inbounds, so ignore that case (dereferenceable_or_null).
|
|
|
|
// The reason is that 'null' is not treated differently in these address
|
|
|
|
// spaces, and we consequently ignore the 'gep inbounds' special case
|
|
|
|
// for 'null' which allows 'inbounds' on 'null' if the indices are
|
|
|
|
// zeros.
|
|
|
|
if (SrcPTy->getAddressSpace() == 0 || !CanBeNull)
|
|
|
|
GEP->setIsInBounds();
|
|
|
|
}
|
2019-10-06 21:08:08 +08:00
|
|
|
return GEP;
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2020-05-26 23:07:46 +08:00
|
|
|
if (FixedVectorType *DestVTy = dyn_cast<FixedVectorType>(DestTy)) {
|
2020-05-06 23:24:59 +08:00
|
|
|
// Beware: messing with this target-specific oddity may cause trouble.
|
|
|
|
if (DestVTy->getNumElements() == 1 && SrcTy->isX86_MMXTy()) {
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *Elem = Builder.CreateBitCast(Src, DestVTy->getElementType());
|
2010-01-06 06:21:18 +08:00
|
|
|
return InsertElementInst::Create(UndefValue::get(DestTy), Elem,
|
2010-01-04 15:53:58 +08:00
|
|
|
Constant::getNullValue(Type::getInt32Ty(CI.getContext())));
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
if (isa<IntegerType>(SrcTy)) {
|
|
|
|
// If this is a cast from an integer to vector, check to see if the input
|
|
|
|
// is a trunc or zext of a bitcast from vector. If so, we can replace all
|
|
|
|
// the casts with a shuffle and (potentially) a bitcast.
|
|
|
|
if (isa<TruncInst>(Src) || isa<ZExtInst>(Src)) {
|
|
|
|
CastInst *SrcCast = cast<CastInst>(Src);
|
|
|
|
if (BitCastInst *BCIn = dyn_cast<BitCastInst>(SrcCast->getOperand(0)))
|
|
|
|
if (isa<VectorType>(BCIn->getOperand(0)->getType()))
|
2019-11-29 06:18:28 +08:00
|
|
|
if (Instruction *I = optimizeVectorResizeWithIntegerBitCasts(
|
|
|
|
BCIn->getOperand(0), cast<VectorType>(DestTy), *this))
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
return I;
|
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
optimize bitcasts from large integers to vector into vector
element insertion from the pieces that feed into the vector.
This handles a pattern that occurs frequently due to code
generated for the x86-64 abi. We now compile something like
this:
struct S { float A, B, C, D; };
struct S g;
struct S bar() {
struct S A = g;
++A.A;
++A.C;
return A;
}
into all nice vector operations:
_bar: ## @bar
## BB#0: ## %entry
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
pshufd $16, %xmm0, %xmm0
movss 4(%rax), %xmm2
movss 12(%rax), %xmm3
pshufd $16, %xmm2, %xmm2
unpcklps %xmm2, %xmm0
addss 8(%rax), %xmm1
pshufd $16, %xmm1, %xmm1
pshufd $16, %xmm3, %xmm2
unpcklps %xmm2, %xmm1
ret
instead of icky integer operations:
_bar: ## @bar
movq _g@GOTPCREL(%rip), %rax
movss LCPI1_0(%rip), %xmm1
movss (%rax), %xmm0
addss %xmm1, %xmm0
movd %xmm0, %ecx
movl 4(%rax), %edx
movl 12(%rax), %esi
shlq $32, %rdx
addq %rcx, %rdx
movd %rdx, %xmm0
addss 8(%rax), %xmm1
movd %xmm1, %eax
shlq $32, %rsi
addq %rax, %rsi
movd %rsi, %xmm1
ret
This resolves rdar://8360454
llvm-svn: 112343
2010-08-28 09:20:38 +08:00
|
|
|
// If the input is an 'or' instruction, we may be doing shifts and ors to
|
|
|
|
// assemble the elements of the vector manually. Try to rip the code out
|
|
|
|
// and replace it with insertelements.
|
2015-09-09 22:54:29 +08:00
|
|
|
if (Value *V = optimizeIntegerToVectorInsertions(CI, *this))
|
2016-02-02 06:23:39 +08:00
|
|
|
return replaceInstUsesWith(CI, V);
|
2010-05-09 05:50:26 +08:00
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
|
2020-05-26 23:07:46 +08:00
|
|
|
if (FixedVectorType *SrcVTy = dyn_cast<FixedVectorType>(SrcTy)) {
|
2013-02-12 05:41:44 +08:00
|
|
|
if (SrcVTy->getNumElements() == 1) {
|
|
|
|
// If our destination is not a vector, then make this a straight
|
|
|
|
// scalar-scalar cast.
|
2019-12-04 05:48:39 +08:00
|
|
|
if (!DestTy->isVectorTy()) {
|
2013-02-12 05:41:44 +08:00
|
|
|
Value *Elem =
|
2017-07-08 07:16:26 +08:00
|
|
|
Builder.CreateExtractElement(Src,
|
2013-02-12 05:41:44 +08:00
|
|
|
Constant::getNullValue(Type::getInt32Ty(CI.getContext())));
|
|
|
|
return CastInst::Create(Instruction::BitCast, Elem, DestTy);
|
|
|
|
}
|
|
|
|
|
|
|
|
// Otherwise, see if our source is an insert. If so, then use the scalar
|
2019-06-06 05:26:52 +08:00
|
|
|
// component directly:
|
|
|
|
// bitcast (inselt <1 x elt> V, X, 0) to <n x m> --> bitcast X to <n x m>
|
|
|
|
if (auto *InsElt = dyn_cast<InsertElementInst>(Src))
|
|
|
|
return new BitCastInst(InsElt->getOperand(1), DestTy);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
|
|
|
|
2019-08-30 03:36:18 +08:00
|
|
|
if (auto *Shuf = dyn_cast<ShuffleVectorInst>(Src)) {
|
2010-01-06 06:21:18 +08:00
|
|
|
// Okay, we have (bitcast (shuffle ..)). Check to see if this is
|
2010-04-08 07:22:42 +08:00
|
|
|
// a bitcast to a vector with the same # elts.
|
2019-08-30 03:36:18 +08:00
|
|
|
Value *ShufOp0 = Shuf->getOperand(0);
|
|
|
|
Value *ShufOp1 = Shuf->getOperand(1);
|
2020-04-09 01:42:22 +08:00
|
|
|
unsigned NumShufElts = Shuf->getType()->getNumElements();
|
|
|
|
unsigned NumSrcVecElts =
|
|
|
|
cast<VectorType>(ShufOp0->getType())->getNumElements();
|
2019-08-30 03:36:18 +08:00
|
|
|
if (Shuf->hasOneUse() && DestTy->isVectorTy() &&
|
2020-04-09 01:42:22 +08:00
|
|
|
cast<VectorType>(DestTy)->getNumElements() == NumShufElts &&
|
2019-08-30 03:36:18 +08:00
|
|
|
NumShufElts == NumSrcVecElts) {
|
2010-01-06 06:21:18 +08:00
|
|
|
BitCastInst *Tmp;
|
|
|
|
// If either of the operands is a cast from CI.getType(), then
|
|
|
|
// evaluating the shuffle in the casted destination's type will allow
|
|
|
|
// us to eliminate at least one cast.
|
2019-08-30 03:36:18 +08:00
|
|
|
if (((Tmp = dyn_cast<BitCastInst>(ShufOp0)) &&
|
2010-01-06 06:21:18 +08:00
|
|
|
Tmp->getOperand(0)->getType() == DestTy) ||
|
2019-08-30 03:36:18 +08:00
|
|
|
((Tmp = dyn_cast<BitCastInst>(ShufOp1)) &&
|
2010-01-06 06:21:18 +08:00
|
|
|
Tmp->getOperand(0)->getType() == DestTy)) {
|
2019-08-30 03:36:18 +08:00
|
|
|
Value *LHS = Builder.CreateBitCast(ShufOp0, DestTy);
|
|
|
|
Value *RHS = Builder.CreateBitCast(ShufOp1, DestTy);
|
2010-01-06 06:21:18 +08:00
|
|
|
// Return a new shuffle vector. Use the same element ID's, as we
|
|
|
|
// know the vector types match #elts.
|
2020-04-01 04:08:59 +08:00
|
|
|
return new ShuffleVectorInst(LHS, RHS, Shuf->getShuffleMask());
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
|
|
|
}
|
[InstCombine] recognize bswap disguised as shufflevector
bitcast <N x i8> (shuf X, undef, <N, N-1,...0>) to i{N*8} --> bswap (bitcast X to i{N*8})
In PR43146:
https://bugs.llvm.org/show_bug.cgi?id=43146
...we have a more complicated case where SLP is making a mess of bswap. This patch won't
do anything for that currently, but we need to improve bswap recognition in instcombine,
SLP, and/or a standalone pass to avoid that problem.
This is limited using the data-layout so we don't try to do this transform with actual
vector types. The backend does not appear to have folds to convert in either direction,
so we don't want to mess up something that is actually better lowered as a shuffle.
On x86, we're trading something like this:
vmovd %edi, %xmm0
vpshufb LCPI0_0(%rip), %xmm0, %xmm0 ## xmm0 = xmm0[3,2,1,0,u,u,u,u,u,u,u,u,u,u,u,u]
vmovd %xmm0, %eax
For:
movl %edi, %eax
bswapl %eax
Differential Revision: https://reviews.llvm.org/D66965
llvm-svn: 370659
2019-09-02 21:33:20 +08:00
|
|
|
|
|
|
|
// A bitcasted-to-scalar and byte-reversing shuffle is better recognized as
|
|
|
|
// a byte-swap:
|
|
|
|
// bitcast <N x i8> (shuf X, undef, <N, N-1,...0>) --> bswap (bitcast X)
|
|
|
|
// TODO: We should match the related pattern for bitreverse.
|
|
|
|
if (DestTy->isIntegerTy() &&
|
|
|
|
DL.isLegalInteger(DestTy->getScalarSizeInBits()) &&
|
|
|
|
SrcTy->getScalarSizeInBits() == 8 && NumShufElts % 2 == 0 &&
|
|
|
|
Shuf->hasOneUse() && Shuf->isReverse()) {
|
|
|
|
assert(ShufOp0->getType() == SrcTy && "Unexpected shuffle mask");
|
|
|
|
assert(isa<UndefValue>(ShufOp1) && "Unexpected shuffle op");
|
|
|
|
Function *Bswap =
|
|
|
|
Intrinsic::getDeclaration(CI.getModule(), Intrinsic::bswap, DestTy);
|
|
|
|
Value *ScalarX = Builder.CreateBitCast(ShufOp0, DestTy);
|
|
|
|
return IntrinsicInst::Create(Bswap, { ScalarX });
|
|
|
|
}
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-01-24 13:22:40 +08:00
|
|
|
|
2016-10-26 04:43:42 +08:00
|
|
|
// Handle the A->B->A cast, and there is an intervening PHI node.
|
|
|
|
if (PHINode *PN = dyn_cast<PHINode>(Src))
|
|
|
|
if (Instruction *I = optimizeBitCastFromPhi(CI, PN))
|
|
|
|
return I;
|
|
|
|
|
2017-07-07 07:18:43 +08:00
|
|
|
if (Instruction *I = canonicalizeBitCastExtElt(CI, *this))
|
2015-12-11 01:09:28 +08:00
|
|
|
return I;
|
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
if (Instruction *I = foldBitCastBitwiseLogic(CI, Builder))
|
2016-11-23 06:05:48 +08:00
|
|
|
return I;
|
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
if (Instruction *I = foldBitCastSelect(CI, Builder))
|
2016-12-03 23:25:16 +08:00
|
|
|
return I;
|
|
|
|
|
2010-02-16 19:11:14 +08:00
|
|
|
if (SrcTy->isPointerTy())
|
2010-01-06 06:21:18 +08:00
|
|
|
return commonPointerCastTransforms(CI);
|
|
|
|
return commonCastTransforms(CI);
|
2010-01-04 15:53:58 +08:00
|
|
|
}
|
2013-11-15 13:45:08 +08:00
|
|
|
|
|
|
|
Instruction *InstCombiner::visitAddrSpaceCast(AddrSpaceCastInst &CI) {
|
2014-07-16 09:34:21 +08:00
|
|
|
// If the destination pointer element type is not the same as the source's
|
|
|
|
// first do a bitcast to the destination type, and then the addrspacecast.
|
|
|
|
// This allows the cast to be exposed to other transforms.
|
2014-06-07 05:52:55 +08:00
|
|
|
Value *Src = CI.getOperand(0);
|
|
|
|
PointerType *SrcTy = cast<PointerType>(Src->getType()->getScalarType());
|
|
|
|
PointerType *DestTy = cast<PointerType>(CI.getType()->getScalarType());
|
|
|
|
|
|
|
|
Type *DestElemTy = DestTy->getElementType();
|
|
|
|
if (SrcTy->getElementType() != DestElemTy) {
|
|
|
|
Type *MidTy = PointerType::get(DestElemTy, SrcTy->getAddressSpace());
|
2014-06-16 05:40:57 +08:00
|
|
|
if (VectorType *VT = dyn_cast<VectorType>(CI.getType())) {
|
|
|
|
// Handle vectors of pointers.
|
2020-05-30 06:24:15 +08:00
|
|
|
// FIXME: what should happen for scalable vectors?
|
|
|
|
MidTy = FixedVectorType::get(MidTy, VT->getNumElements());
|
2014-06-16 05:40:57 +08:00
|
|
|
}
|
2014-06-07 05:52:55 +08:00
|
|
|
|
2017-07-08 07:16:26 +08:00
|
|
|
Value *NewBitCast = Builder.CreateBitCast(Src, MidTy);
|
2014-06-07 05:52:55 +08:00
|
|
|
return new AddrSpaceCastInst(NewBitCast, CI.getType());
|
|
|
|
}
|
|
|
|
|
2014-01-15 04:00:45 +08:00
|
|
|
return commonPointerCastTransforms(CI);
|
2013-11-15 13:45:08 +08:00
|
|
|
}
|