[InstCombine, ARM, AArch64] Convert table lookup to shuffle vector

Turning a table lookup intrinsic into a shuffle vector instruction can be beneficial. If the mask used for the lookup is the constant vector {7,6,5,4,3,2,1,0}, then the back-end generates byte reverse instructions instead. Differential Revision: https://reviews.llvm.org/D46133 llvm-svn: 333550
author: Alexandros Lamprineas <alexandros.lamprineas@arm.com> 2018-05-30 14:38:50 +0000
committer: Alexandros Lamprineas <alexandros.lamprineas@arm.com> 2018-05-30 14:38:50 +0000
commit: 52457d33b23c1657e0825f99bab3064e463d62e8 (patch)
tree: 0f4cea0291349e8f7dc8540433e629843a461f1e /llvm/lib/Transforms/InstCombine/InstCombineCalls.cpp
parent: 8df8b129ce53abc88dca53b7aaad46c49d408bee (diff)
download: bcm5719-llvm-52457d33b23c1657e0825f99bab3064e463d62e8.tar.gz
bcm5719-llvm-52457d33b23c1657e0825f99bab3064e463d62e8.zip
1 files changed, 46 insertions, 0 deletions
diff --git a/llvm/lib/Transforms/InstCombine/InstCombineCalls.cpp b/llvm/lib/Transforms/InstCombine/InstCombineCalls.cpp
index a5a0edcd96f..7fbf068f506 100644
--- a/llvm/lib/Transforms/InstCombine/InstCombineCalls.cpp
+++ b/llvm/lib/Transforms/InstCombine/InstCombineCalls.cpp
@@ -1387,6 +1387,46 @@ static APFloat fmed3AMDGCN(const APFloat &Src0, const APFloat &Src1,
   return maxnum(Src0, Src1);
 }
 
+/// Convert a table lookup to shufflevector if the mask is constant.
+/// This could benefit tbl1 if the mask is { 7,6,5,4,3,2,1,0 }, in
+/// which case we could lower the shufflevector with rev64 instructions
+/// as it's actually a byte reverse.
+static Value *simplifyNeonTbl1(const IntrinsicInst &II,
+                               InstCombiner::BuilderTy &Builder) {
+  // Bail out if the mask is not a constant.
+  auto *C = dyn_cast<Constant>(II.getArgOperand(1));
+  if (!C)
+    return nullptr;
+
+  auto *VecTy = cast<VectorType>(II.getType());
+  unsigned NumElts = VecTy->getNumElements();
+
+  // Only perform this transformation for <8 x i8> vector types.
+  if (!VecTy->getElementType()->isIntegerTy(8) || NumElts != 8)
+    return nullptr;
+
+  uint32_t Indexes[8];
+
+  for (unsigned I = 0; I < NumElts; ++I) {
+    Constant *COp = C->getAggregateElement(I);
+
+    if (!COp || !isa<ConstantInt>(COp))
+      return nullptr;
+
+    Indexes[I] = cast<ConstantInt>(COp)->getLimitedValue();
+
+    // Make sure the mask indices are in range.
+    if (Indexes[I] >= NumElts)
+      return nullptr;
+  }
+
+  auto *ShuffleMask = ConstantDataVector::get(II.getContext(),
+                                              makeArrayRef(Indexes));
+  auto *V1 = II.getArgOperand(0);
+  auto *V2 = Constant::getNullValue(V1->getType());
+  return Builder.CreateShuffleVector(V1, V2, ShuffleMask);
+}
+
 // Returns true iff the 2 intrinsics have the same operands, limiting the
 // comparison to the first NumOperands.
 static bool haveSameOperands(const IntrinsicInst &I, const IntrinsicInst &E,
@@ -2928,6 +2968,12 @@ Instruction *InstCombiner::visitCallInst(CallInst &CI) {
     break;
   }
 
+  case Intrinsic::arm_neon_vtbl1:
+  case Intrinsic::aarch64_neon_tbl1:
+    if (Value *V = simplifyNeonTbl1(*II, Builder))
+      return replaceInstUsesWith(*II, V);
+    break;
+
   case Intrinsic::arm_neon_vmulls:
   case Intrinsic::arm_neon_vmullu:
   case Intrinsic::aarch64_neon_smull:
author	Alexandros Lamprineas <alexandros.lamprineas@arm.com>	2018-05-30 14:38:50 +0000
committer	Alexandros Lamprineas <alexandros.lamprineas@arm.com>	2018-05-30 14:38:50 +0000
commit	52457d33b23c1657e0825f99bab3064e463d62e8 (patch)
tree	0f4cea0291349e8f7dc8540433e629843a461f1e /llvm/lib/Transforms/InstCombine/InstCombineCalls.cpp
parent	8df8b129ce53abc88dca53b7aaad46c49d408bee (diff)
download	bcm5719-llvm-52457d33b23c1657e0825f99bab3064e463d62e8.tar.gz bcm5719-llvm-52457d33b23c1657e0825f99bab3064e463d62e8.zip