LoopVectorizer: Preserve fast-math flags

author Arnold Schwaighofer <aschwaighofer@apple.com>

Wed, 5 Mar 2014 21:10:47 +0000 (21:10 +0000)

committer Arnold Schwaighofer <aschwaighofer@apple.com>

Wed, 5 Mar 2014 21:10:47 +0000 (21:10 +0000)
author Arnold Schwaighofer <aschwaighofer@apple.com>
Wed, 5 Mar 2014 21:10:47 +0000 (21:10 +0000)
committer Arnold Schwaighofer <aschwaighofer@apple.com>
Wed, 5 Mar 2014 21:10:47 +0000 (21:10 +0000)
diff --git a/lib/Transforms/Vectorize/LoopVectorize.cpp b/lib/Transforms/Vectorize/LoopVectorize.cpp

index 573df567de9d09a56c9ce99ee48f1c98bed508e6..77633a562a6759549ea9c6fb77b40aa5ba0f7926 100644 (file)
--- a/lib/Transforms/Vectorize/LoopVectorize.cpp
+++ b/lib/Transforms/Vectorize/LoopVectorize.cpp
@@ -2489,6 +2489,16 @@ static void cse(SmallVector<BasicBlock *, 4> &BBs) {
    }
  }
  
+/// \brief Adds a 'fast' flag to floating point operations.
+static Value *addFastMathFlag(Value *V) {
+  if (isa<FPMathOperator>(V)){
+    FastMathFlags Flags;
+    Flags.setUnsafeAlgebra();
+    cast<Instruction>(V)->setFastMathFlags(Flags);
+  }
+  return V;
+}
+
  void InnerLoopVectorizer::vectorizeLoop() {
    //===------------------------------------------------===//
    //
@@ -2632,9 +2642,10 @@ void InnerLoopVectorizer::vectorizeLoop() {
      setDebugLocFromInst(Builder, ReducedPartRdx);
      for (unsigned part = 1; part < UF; ++part) {
        if (Op != Instruction::ICmp && Op != Instruction::FCmp)
-        ReducedPartRdx = Builder.CreateBinOp((Instruction::BinaryOps)Op,
-                                             RdxParts[part], ReducedPartRdx,
-                                             "bin.rdx");
+        // Floating point operations had to be 'fast' to enable the reduction.
+        ReducedPartRdx = addFastMathFlag(
+            Builder.CreateBinOp((Instruction::BinaryOps)Op, RdxParts[part],
+                                ReducedPartRdx, "bin.rdx"));
        else
          ReducedPartRdx = createMinMaxOp(Builder, RdxDesc.MinMaxKind,
                                          ReducedPartRdx, RdxParts[part]);
@@ -2664,8 +2675,9 @@ void InnerLoopVectorizer::vectorizeLoop() {
                                      "rdx.shuf");
  
          if (Op != Instruction::ICmp && Op != Instruction::FCmp)
-          TmpVec = Builder.CreateBinOp((Instruction::BinaryOps)Op, TmpVec, Shuf,
-                                       "bin.rdx");
+          // Floating point operations had to be 'fast' to enable the reduction.
+          TmpVec = addFastMathFlag(Builder.CreateBinOp(
+              (Instruction::BinaryOps)Op, TmpVec, Shuf, "bin.rdx"));
          else
            TmpVec = createMinMaxOp(Builder, RdxDesc.MinMaxKind, TmpVec, Shuf);
        }
@@ -2999,6 +3011,10 @@ void InnerLoopVectorizer::vectorizeBlockInLoop(BasicBlock *BB, PhiVector *PV) {
          if (VecOp && isa<PossiblyExactOperator>(VecOp))
            VecOp->setIsExact(BinOp->isExact());
  
+        // Copy the fast-math flags.
+        if (VecOp && isa<FPMathOperator>(V))
+          VecOp->setFastMathFlags(it->getFastMathFlags());
+
          Entry[Part] = V;
        }
        break;
diff --git a/test/Transforms/LoopVectorize/flags.ll b/test/Transforms/LoopVectorize/flags.ll

index a4ebb4284881221560e8458593c5dc73dcb6b4eb..21d09372d546aefc40dc88ddff9ec72b563963ce 100644 (file)
--- a/test/Transforms/LoopVectorize/flags.ll
+++ b/test/Transforms/LoopVectorize/flags.ll
@@ -51,3 +51,29 @@ define i32 @flags2(i32 %n, i32* nocapture %A) nounwind uwtable ssp {
  ._crit_edge:                                      ; preds = %.lr.ph, %0
    ret i32 undef
  }
+
+; Make sure we copy fast math flags and use them for the final reduction.
+; CHECK-LABEL: fast_math
+; CHECK: load <4 x float>
+; CHECK: fadd fast <4 x float>
+; CHECK: br
+; CHECK: fadd fast <4 x float>
+; CHECK: fadd fast <4 x float>
+define float @fast_math(float* noalias %s) {
+entry:
+  br label %for.body
+
+for.body:
+  %indvars.iv = phi i64 [ 0, %entry ], [ %indvars.iv.next, %for.body ]
+  %q.04 = phi float [ 0.000000e+00, %entry ], [ %add, %for.body ]
+  %arrayidx = getelementptr inbounds float* %s, i64 %indvars.iv
+  %0 = load float* %arrayidx, align 4
+  %add = fadd fast float %q.04, %0
+  %indvars.iv.next = add nuw nsw i64 %indvars.iv, 1
+  %exitcond = icmp eq i64 %indvars.iv.next, 256
+  br i1 %exitcond, label %for.end, label %for.body
+
+for.end:
+  %add.lcssa = phi float [ %add, %for.body ]
+  ret float %add.lcssa
+}
diff --git a/test/Transforms/LoopVectorize/float-reduction.ll b/test/Transforms/LoopVectorize/float-reduction.ll

index c45098dd2c3b92869f639046d7b0390e066452a1..0dfbab07279ac0bb1277e8f0a78a60aef4e9ffaa 100644 (file)
--- a/test/Transforms/LoopVectorize/float-reduction.ll
+++ b/test/Transforms/LoopVectorize/float-reduction.ll
@@ -3,7 +3,7 @@
  target datalayout = "e-p:64:64:64-i1:8:8-i8:8:8-i16:16:16-i32:32:32-i64:64:64-f32:32:32-f64:64:64-v64:64:64-v128:128:128-a0:0:64-s0:64:64-f80:128:128-n8:16:32:64-S128"
  target triple = "x86_64-apple-macosx10.8.0"
  ;CHECK-LABEL: @foo(
-;CHECK: fadd <4 x float>
+;CHECK: fadd fast <4 x float>
  ;CHECK: ret
  define float @foo(float* nocapture %A, i32* nocapture %n) nounwind uwtable readonly ssp {
  entry:
author	Arnold Schwaighofer <aschwaighofer@apple.com>
	Wed, 5 Mar 2014 21:10:47 +0000 (21:10 +0000)
committer	Arnold Schwaighofer <aschwaighofer@apple.com>
	Wed, 5 Mar 2014 21:10:47 +0000 (21:10 +0000)
lib/Transforms/Vectorize/LoopVectorize.cpp		patch \| blob \| history
test/Transforms/LoopVectorize/flags.ll		patch \| blob \| history
test/Transforms/LoopVectorize/float-reduction.ll		patch \| blob \| history