[X86] Add a DAG combines to turn vXi64 muls into VPMULDQ/VPMULUDQ if the upper bits are all sign bits or zeros.

Normally we catch this during lowering, but vXi64 mul is considered legal when we have AVX512DQ. This DAG combine allows us to avoid PMULLQ with AVX512DQ if we can prove its unnecessary. PMULLQ is 3 uops that take 4 cycles each. While pmuldq/pmuludq is only one 4 cycle uop. llvm-svn: 321437
author: Craig Topper <craig.topper@intel.com> 2017-12-25 06:47:10 +0000
committer: Craig Topper <craig.topper@intel.com> 2017-12-25 06:47:10 +0000
commit: 705fef3ef3d26718f34de45e1d4c2ef0f37c9bb2 (patch)
tree: 582343f5fcf0a1f01720cff28bd6e7d08c93096a /llvm/lib
parent: b28460a0d6c02e7938d69daf54cf9b8e17bf431b (diff)
download: bcm5719-llvm-705fef3ef3d26718f34de45e1d4c2ef0f37c9bb2.tar.gz
bcm5719-llvm-705fef3ef3d26718f34de45e1d4c2ef0f37c9bb2.zip
1 files changed, 34 insertions, 0 deletions
diff --git a/llvm/lib/Target/X86/X86ISelLowering.cpp b/llvm/lib/Target/X86/X86ISelLowering.cpp
index 363cdee48f3..10dcf2168ba 100644
--- a/llvm/lib/Target/X86/X86ISelLowering.cpp
+++ b/llvm/lib/Target/X86/X86ISelLowering.cpp
@@ -32423,6 +32423,37 @@ static SDValue combineMulSpecial(uint64_t MulAmt, SDNode *N, SelectionDAG &DAG,
   return SDValue();
 }
 
+static SDValue combineVMUL(SDNode *N, SelectionDAG &DAG,
+                           const X86Subtarget &Subtarget) {
+  EVT VT = N->getValueType(0);
+  SDLoc dl(N);
+
+  if (VT.getScalarType() != MVT::i64)
+    return SDValue();
+
+  MVT MulVT = MVT::getVectorVT(MVT::i32, VT.getVectorNumElements() * 2);
+
+  SDValue LHS = N->getOperand(0);
+  SDValue RHS = N->getOperand(1);
+
+  // MULDQ returns the 64-bit result of the signed multiplication of the lower
+  // 32-bits. We can lower with this if the sign bits stretch that far.
+  if (Subtarget.hasSSE41() && DAG.ComputeNumSignBits(LHS) > 32 &&
+      DAG.ComputeNumSignBits(RHS) > 32) {
+    return DAG.getNode(X86ISD::PMULDQ, dl, VT, DAG.getBitcast(MulVT, LHS),
+                       DAG.getBitcast(MulVT, RHS));
+  }
+
+  // If the upper bits are zero we can use a single pmuludq.
+  APInt Mask = APInt::getHighBitsSet(64, 32);
+  if (DAG.MaskedValueIsZero(LHS, Mask) && DAG.MaskedValueIsZero(RHS, Mask)) {
+    return DAG.getNode(X86ISD::PMULUDQ, dl, VT, DAG.getBitcast(MulVT, LHS),
+                       DAG.getBitcast(MulVT, RHS));
+  }
+
+  return SDValue();
+}
+
 /// Optimize a single multiply with constant into two operations in order to
 /// implement it with two cheaper instructions, e.g. LEA + SHL, LEA + LEA.
 static SDValue combineMul(SDNode *N, SelectionDAG &DAG,
@@ -32432,6 +32463,9 @@ static SDValue combineMul(SDNode *N, SelectionDAG &DAG,
   if (DCI.isBeforeLegalize() && VT.isVector())
     return reduceVMULWidth(N, DAG, Subtarget);
 
+  if (!DCI.isBeforeLegalize() && VT.isVector())
+    return combineVMUL(N, DAG, Subtarget);
+
   if (!MulConstantOptimization)
     return SDValue();
   // An imul is usually smaller than the alternative sequence.
author	Craig Topper <craig.topper@intel.com>	2017-12-25 06:47:10 +0000
committer	Craig Topper <craig.topper@intel.com>	2017-12-25 06:47:10 +0000
commit	705fef3ef3d26718f34de45e1d4c2ef0f37c9bb2 (patch)
tree	582343f5fcf0a1f01720cff28bd6e7d08c93096a /llvm/lib
parent	b28460a0d6c02e7938d69daf54cf9b8e17bf431b (diff)
download	bcm5719-llvm-705fef3ef3d26718f34de45e1d4c2ef0f37c9bb2.tar.gz bcm5719-llvm-705fef3ef3d26718f34de45e1d4c2ef0f37c9bb2.zip