mirror of
https://github.com/RPCS3/llvm.git
synced 2026-08-26 18:26:51 -04:00
[DAGCombiner] rearrange extract_element+bitcast fold; NFC
I want to add another pattern here that includes scalar_to_vector, so this makes that patch smaller. I was hoping to remove the hasOneUse() check because it shouldn't be necessary for common codegen, but an AMDGPU test has a comment suggesting that the extra check makes things better on one of those targets. git-svn-id: https://llvm.org/svn/llvm-project/llvm/trunk@344320 91177308-0d34-0410-b5e6-96231b3b80d8
This commit is contained in:
@@ -15499,13 +15499,15 @@ SDValue DAGCombiner::visitEXTRACT_VECTOR_ELT(SDNode *N) {
|
||||
// converts.
|
||||
}
|
||||
|
||||
// extract_vector_elt (v2i32 (bitcast i64:x)), EltTrunc -> i32 (trunc i64:x)
|
||||
bool isLE = DAG.getDataLayout().isLittleEndian();
|
||||
unsigned EltTrunc = isLE ? 0 : VT.getVectorNumElements() - 1;
|
||||
if (ConstEltNo && InVec.getOpcode() == ISD::BITCAST && InVec.hasOneUse() &&
|
||||
ConstEltNo->getZExtValue() == EltTrunc && VT.isInteger()) {
|
||||
if (ConstEltNo && InVec.getOpcode() == ISD::BITCAST) {
|
||||
// The vector index of the LSBs of the source depend on the endian-ness.
|
||||
bool IsLE = DAG.getDataLayout().isLittleEndian();
|
||||
|
||||
// extract_elt (v2i32 (bitcast i64:x)), BCTruncElt -> i32 (trunc i64:x)
|
||||
unsigned BCTruncElt = IsLE ? 0 : VT.getVectorNumElements() - 1;
|
||||
SDValue BCSrc = InVec.getOperand(0);
|
||||
if (BCSrc.getValueType().isScalarInteger())
|
||||
if (InVec.hasOneUse() && ConstEltNo->getZExtValue() == BCTruncElt &&
|
||||
VT.isInteger() && BCSrc.getValueType().isScalarInteger())
|
||||
return DAG.getNode(ISD::TRUNCATE, SDLoc(N), NVT, BCSrc);
|
||||
}
|
||||
|
||||
|
||||
@@ -28,6 +28,10 @@ define i8 @extractelt_bitcast(i32 %x) nounwind {
|
||||
ret i8 %ext
|
||||
}
|
||||
|
||||
; TODO: This should have folded to avoid vector ops, but the transform
|
||||
; is guarded by 'hasOneUse'. That limitation apparently makes some AMDGPU
|
||||
; codegen better.
|
||||
|
||||
define i8 @extractelt_bitcast_extra_use(i32 %x, <4 x i8>* %p) nounwind {
|
||||
; X86-LABEL: extractelt_bitcast_extra_use:
|
||||
; X86: # %bb.0:
|
||||
|
||||
Reference in New Issue
Block a user