[IR] redefine 'UnsafeAlgebra' / 'reassoc' fast-math-flags and add 'trans' fast-math...

[android-x86/external-llvm.git] / lib / Transforms / Scalar / Reassociate.cpp
diff --git a/lib/Transforms/Scalar/Reassociate.cpp b/lib/Transforms/Scalar/Reassociate.cpp

index b11567c..1f32f9f 100644 (file)
--- a/lib/Transforms/Scalar/Reassociate.cpp
+++ b/lib/Transforms/Scalar/Reassociate.cpp
@@ -20,26 +20,47 @@
  //
  //===----------------------------------------------------------------------===//
  
-#include "llvm/Transforms/Scalar.h"
+#include "llvm/Transforms/Scalar/Reassociate.h"
+#include "llvm/ADT/APFloat.h"
+#include "llvm/ADT/APInt.h"
  #include "llvm/ADT/DenseMap.h"
  #include "llvm/ADT/PostOrderIterator.h"
-#include "llvm/ADT/STLExtras.h"
  #include "llvm/ADT/SetVector.h"
+#include "llvm/ADT/SmallPtrSet.h"
+#include "llvm/ADT/SmallVector.h"
  #include "llvm/ADT/Statistic.h"
+#include "llvm/Analysis/GlobalsModRef.h"
+#include "llvm/Analysis/ValueTracking.h"
+#include "llvm/IR/Argument.h"
+#include "llvm/IR/BasicBlock.h"
  #include "llvm/IR/CFG.h"
+#include "llvm/IR/Constant.h"
  #include "llvm/IR/Constants.h"
-#include "llvm/IR/DerivedTypes.h"
  #include "llvm/IR/Function.h"
  #include "llvm/IR/IRBuilder.h"
+#include "llvm/IR/InstrTypes.h"
+#include "llvm/IR/Instruction.h"
  #include "llvm/IR/Instructions.h"
-#include "llvm/IR/IntrinsicInst.h"
+#include "llvm/IR/Operator.h"
+#include "llvm/IR/PassManager.h"
+#include "llvm/IR/PatternMatch.h"
+#include "llvm/IR/Type.h"
+#include "llvm/IR/User.h"
+#include "llvm/IR/Value.h"
  #include "llvm/IR/ValueHandle.h"
  #include "llvm/Pass.h"
+#include "llvm/Support/Casting.h"
  #include "llvm/Support/Debug.h"
+#include "llvm/Support/ErrorHandling.h"
  #include "llvm/Support/raw_ostream.h"
+#include "llvm/Transforms/Scalar.h"
  #include "llvm/Transforms/Utils/Local.h"
  #include <algorithm>
+#include <cassert>
+#include <utility>
+
  using namespace llvm;
+using namespace reassociate;
  
  #define DEBUG_TYPE "reassociate"
  
@@ -47,22 +68,10 @@ STATISTIC(NumChanged, "Number of insts reassociated");
  STATISTIC(NumAnnihil, "Number of expr tree annihilated");
  STATISTIC(NumFactor , "Number of multiplies factored");
  
-namespace {
-  struct ValueEntry {
-    unsigned Rank;
-    Value *Op;
-    ValueEntry(unsigned R, Value *O) : Rank(R), Op(O) {}
-  };
-  inline bool operator<(const ValueEntry &LHS, const ValueEntry &RHS) {
-    return LHS.Rank > RHS.Rank;   // Sort so that highest rank goes to start.
-  }
-}
-
  #ifndef NDEBUG
-/// PrintOps - Print out the expression identified in the Ops list.
-///
+/// Print out the expression identified in the Ops list.
  static void PrintOps(Instruction *I, const SmallVectorImpl<ValueEntry> &Ops) {
-  Module *M = I->getParent()->getParent()->getParent();
+  Module *M = I->getModule();
    dbgs() << Instruction::getOpcodeName(I->getOpcode()) << " "
         << *Ops[0].Op->getType() << '\t';
    for (unsigned i = 0, e = Ops.size(); i != e; ++i) {
@@ -73,129 +82,35 @@ static void PrintOps(Instruction *I, const SmallVectorImpl<ValueEntry> &Ops) {
  }
  #endif
  
-namespace {
-  /// \brief Utility class representing a base and exponent pair which form one
-  /// factor of some product.
-  struct Factor {
-    Value *Base;
-    unsigned Power;
-
-    Factor(Value *Base, unsigned Power) : Base(Base), Power(Power) {}
-
-    /// \brief Sort factors by their Base.
-    struct BaseSorter {
-      bool operator()(const Factor &LHS, const Factor &RHS) {
-        return LHS.Base < RHS.Base;
-      }
-    };
-
-    /// \brief Compare factors for equal bases.
-    struct BaseEqual {
-      bool operator()(const Factor &LHS, const Factor &RHS) {
-        return LHS.Base == RHS.Base;
-      }
-    };
-
-    /// \brief Sort factors in descending order by their power.
-    struct PowerDescendingSorter {
-      bool operator()(const Factor &LHS, const Factor &RHS) {
-        return LHS.Power > RHS.Power;
-      }
-    };
-
-    /// \brief Compare factors for equal powers.
-    struct PowerEqual {
-      bool operator()(const Factor &LHS, const Factor &RHS) {
-        return LHS.Power == RHS.Power;
-      }
-    };
-  };
-  
-  /// Utility class representing a non-constant Xor-operand. We classify
-  /// non-constant Xor-Operands into two categories:
-  ///  C1) The operand is in the form "X & C", where C is a constant and C != ~0
-  ///  C2)
-  ///    C2.1) The operand is in the form of "X | C", where C is a non-zero
-  ///          constant.
-  ///    C2.2) Any operand E which doesn't fall into C1 and C2.1, we view this
-  ///          operand as "E | 0"
-  class XorOpnd {
-  public:
-    XorOpnd(Value *V);
-
-    bool isInvalid() const { return SymbolicPart == nullptr; }
-    bool isOrExpr() const { return isOr; }
-    Value *getValue() const { return OrigVal; }
-    Value *getSymbolicPart() const { return SymbolicPart; }
-    unsigned getSymbolicRank() const { return SymbolicRank; }
-    const APInt &getConstPart() const { return ConstPart; }
-
-    void Invalidate() { SymbolicPart = OrigVal = nullptr; }
-    void setSymbolicRank(unsigned R) { SymbolicRank = R; }
-
-    // Sort the XorOpnd-Pointer in ascending order of symbolic-value-rank.
-    // The purpose is twofold:
-    // 1) Cluster together the operands sharing the same symbolic-value.
-    // 2) Operand having smaller symbolic-value-rank is permuted earlier, which 
-    //   could potentially shorten crital path, and expose more loop-invariants.
-    //   Note that values' rank are basically defined in RPO order (FIXME). 
-    //   So, if Rank(X) < Rank(Y) < Rank(Z), it means X is defined earlier 
-    //   than Y which is defined earlier than Z. Permute "x | 1", "Y & 2",
-    //   "z" in the order of X-Y-Z is better than any other orders.
-    struct PtrSortFunctor {
-      bool operator()(XorOpnd * const &LHS, XorOpnd * const &RHS) {
-        return LHS->getSymbolicRank() < RHS->getSymbolicRank();
-      }
-    };
-  private:
-    Value *OrigVal;
-    Value *SymbolicPart;
-    APInt ConstPart;
-    unsigned SymbolicRank;
-    bool isOr;
-  };
-}
-
-namespace {
-  class Reassociate : public FunctionPass {
-    DenseMap<BasicBlock*, unsigned> RankMap;
-    DenseMap<AssertingVH<Value>, unsigned> ValueRankMap;
-    SetVector<AssertingVH<Instruction> > RedoInsts;
-    bool MadeChange;
-  public:
-    static char ID; // Pass identification, replacement for typeid
-    Reassociate() : FunctionPass(ID) {
-      initializeReassociatePass(*PassRegistry::getPassRegistry());
-    }
-
-    bool runOnFunction(Function &F) override;
-
-    void getAnalysisUsage(AnalysisUsage &AU) const override {
-      AU.setPreservesCFG();
-    }
-  private:
-    void BuildRankMap(Function &F);
-    unsigned getRank(Value *V);
-    void ReassociateExpression(BinaryOperator *I);
-    void RewriteExprTree(BinaryOperator *I, SmallVectorImpl<ValueEntry> &Ops);
-    Value *OptimizeExpression(BinaryOperator *I,
-                              SmallVectorImpl<ValueEntry> &Ops);
-    Value *OptimizeAdd(Instruction *I, SmallVectorImpl<ValueEntry> &Ops);
-    Value *OptimizeXor(Instruction *I, SmallVectorImpl<ValueEntry> &Ops);
-    bool CombineXorOpnd(Instruction *I, XorOpnd *Opnd1, APInt &ConstOpnd,
-                        Value *&Res);
-    bool CombineXorOpnd(Instruction *I, XorOpnd *Opnd1, XorOpnd *Opnd2,
-                        APInt &ConstOpnd, Value *&Res);
-    bool collectMultiplyFactors(SmallVectorImpl<ValueEntry> &Ops,
-                                SmallVectorImpl<Factor> &Factors);
-    Value *buildMinimalMultiplyDAG(IRBuilder<> &Builder,
-                                   SmallVectorImpl<Factor> &Factors);
-    Value *OptimizeMul(BinaryOperator *I, SmallVectorImpl<ValueEntry> &Ops);
-    Value *RemoveFactorFromExpression(Value *V, Value *Factor);
-    void EraseInst(Instruction *I);
-    void OptimizeInst(Instruction *I);
-  };
-}
+/// Utility class representing a non-constant Xor-operand. We classify
+/// non-constant Xor-Operands into two categories:
+///  C1) The operand is in the form "X & C", where C is a constant and C != ~0
+///  C2)
+///    C2.1) The operand is in the form of "X | C", where C is a non-zero
+///          constant.
+///    C2.2) Any operand E which doesn't fall into C1 and C2.1, we view this
+///          operand as "E | 0"
+class llvm::reassociate::XorOpnd {
+public:
+  XorOpnd(Value *V);
+
+  bool isInvalid() const { return SymbolicPart == nullptr; }
+  bool isOrExpr() const { return isOr; }
+  Value *getValue() const { return OrigVal; }
+  Value *getSymbolicPart() const { return SymbolicPart; }
+  unsigned getSymbolicRank() const { return SymbolicRank; }
+  const APInt &getConstPart() const { return ConstPart; }
+
+  void Invalidate() { SymbolicPart = OrigVal = nullptr; }
+  void setSymbolicRank(unsigned R) { SymbolicRank = R; }
+
+private:
+  Value *OrigVal;
+  Value *SymbolicPart;
+  APInt ConstPart;
+  unsigned SymbolicRank;
+  bool isOr;
+};
  
  XorOpnd::XorOpnd(Value *V) {
    assert(!isa<ConstantInt>(V) && "No ConstantInt");
@@ -207,11 +122,12 @@ XorOpnd::XorOpnd(Value *V) {
              I->getOpcode() == Instruction::And)) {
      Value *V0 = I->getOperand(0);
      Value *V1 = I->getOperand(1);
-    if (isa<ConstantInt>(V0))
+    const APInt *C;
+    if (match(V0, PatternMatch::m_APInt(C)))
        std::swap(V0, V1);
  
-    if (ConstantInt *C = dyn_cast<ConstantInt>(V1)) {
-      ConstPart = C->getValue();
+    if (match(V1, PatternMatch::m_APInt(C))) {
+      ConstPart = *C;
        SymbolicPart = V0;
        isOr = (I->getOpcode() == Instruction::Or);
        return;
@@ -220,22 +136,16 @@ XorOpnd::XorOpnd(Value *V) {
  
    // view the operand as "V | 0"
    SymbolicPart = V;
-  ConstPart = APInt::getNullValue(V->getType()->getIntegerBitWidth());
+  ConstPart = APInt::getNullValue(V->getType()->getScalarSizeInBits());
    isOr = true;
  }
  
-char Reassociate::ID = 0;
-INITIALIZE_PASS(Reassociate, "reassociate",
-                "Reassociate expressions", false, false)
-
-// Public interface to the Reassociate pass
-FunctionPass *llvm::createReassociatePass() { return new Reassociate(); }
-
-/// isReassociableOp - Return true if V is an instruction of the specified
-/// opcode and if it only has one use.
+/// Return true if V is an instruction of the specified opcode and if it
+/// only has one use.
  static BinaryOperator *isReassociableOp(Value *V, unsigned Opcode) {
    if (V->hasOneUse() && isa<Instruction>(V) &&
-      cast<Instruction>(V)->getOpcode() == Opcode)
+      cast<Instruction>(V)->getOpcode() == Opcode &&
+      (!isa<FPMathOperator>(V) || cast<Instruction>(V)->isFast()))
      return cast<BinaryOperator>(V);
    return nullptr;
  }
@@ -244,55 +154,37 @@ static BinaryOperator *isReassociableOp(Value *V, unsigned Opcode1,
                                          unsigned Opcode2) {
    if (V->hasOneUse() && isa<Instruction>(V) &&
        (cast<Instruction>(V)->getOpcode() == Opcode1 ||
-       cast<Instruction>(V)->getOpcode() == Opcode2))
+       cast<Instruction>(V)->getOpcode() == Opcode2) &&
+      (!isa<FPMathOperator>(V) || cast<Instruction>(V)->isFast()))
      return cast<BinaryOperator>(V);
    return nullptr;
  }
  
-static bool isUnmovableInstruction(Instruction *I) {
-  switch (I->getOpcode()) {
-  case Instruction::PHI:
-  case Instruction::LandingPad:
-  case Instruction::Alloca:
-  case Instruction::Load:
-  case Instruction::Invoke:
-  case Instruction::UDiv:
-  case Instruction::SDiv:
-  case Instruction::FDiv:
-  case Instruction::URem:
-  case Instruction::SRem:
-  case Instruction::FRem:
-    return true;
-  case Instruction::Call:
-    return !isa<DbgInfoIntrinsic>(I);
-  default:
-    return false;
-  }
-}
-
-void Reassociate::BuildRankMap(Function &F) {
-  unsigned i = 2;
+void ReassociatePass::BuildRankMap(Function &F,
+                                   ReversePostOrderTraversal<Function*> &RPOT) {
+  unsigned Rank = 2;
  
-  // Assign distinct ranks to function arguments
-  for (Function::arg_iterator I = F.arg_begin(), E = F.arg_end(); I != E; ++I)
-    ValueRankMap[&*I] = ++i;
+  // Assign distinct ranks to function arguments.
+  for (auto &Arg : F.args()) {
+    ValueRankMap[&Arg] = ++Rank;
+    DEBUG(dbgs() << "Calculated Rank[" << Arg.getName() << "] = " << Rank
+                 << "\n");
+  }
  
-  ReversePostOrderTraversal<Function*> RPOT(&F);
-  for (ReversePostOrderTraversal<Function*>::rpo_iterator I = RPOT.begin(),
-         E = RPOT.end(); I != E; ++I) {
-    BasicBlock *BB = *I;
-    unsigned BBRank = RankMap[BB] = ++i << 16;
+  // Traverse basic blocks in ReversePostOrder
+  for (BasicBlock *BB : RPOT) {
+    unsigned BBRank = RankMap[BB] = ++Rank << 16;
  
      // Walk the basic block, adding precomputed ranks for any instructions that
      // we cannot move.  This ensures that the ranks for these instructions are
      // all different in the block.
-    for (BasicBlock::iterator I = BB->begin(), E = BB->end(); I != E; ++I)
-      if (isUnmovableInstruction(I))
-        ValueRankMap[&*I] = ++BBRank;
+    for (Instruction &I : *BB)
+      if (mayBeMemoryDependent(I))
+        ValueRankMap[&I] = ++BBRank;
    }
  }
  
-unsigned Reassociate::getRank(Value *V) {
+unsigned ReassociatePass::getRank(Value *V) {
    Instruction *I = dyn_cast<Instruction>(V);
    if (!I) {
      if (isa<Argument>(V)) return ValueRankMap[V];   // Function argument.
@@ -313,21 +205,31 @@ unsigned Reassociate::getRank(Value *V) {
  
    // If this is a not or neg instruction, do not count it for rank.  This
    // assures us that X and ~X will have the same rank.
-  Type *Ty = V->getType();
-  if ((!Ty->isIntegerTy() && !Ty->isFloatingPointTy()) ||
-      (!BinaryOperator::isNot(I) && !BinaryOperator::isNeg(I) &&
-       !BinaryOperator::isFNeg(I)))
+  if  (!BinaryOperator::isNot(I) && !BinaryOperator::isNeg(I) &&
+       !BinaryOperator::isFNeg(I))
      ++Rank;
  
-  //DEBUG(dbgs() << "Calculated Rank[" << V->getName() << "] = "
-  //     << Rank << "\n");
+  DEBUG(dbgs() << "Calculated Rank[" << V->getName() << "] = " << Rank << "\n");
  
    return ValueRankMap[I] = Rank;
  }
  
+// Canonicalize constants to RHS.  Otherwise, sort the operands by rank.
+void ReassociatePass::canonicalizeOperands(Instruction *I) {
+  assert(isa<BinaryOperator>(I) && "Expected binary operator.");
+  assert(I->isCommutative() && "Expected commutative operator.");
+
+  Value *LHS = I->getOperand(0);
+  Value *RHS = I->getOperand(1);
+  if (LHS == RHS || isa<Constant>(RHS))
+    return;
+  if (isa<Constant>(LHS) || getRank(RHS) < getRank(LHS))
+    cast<BinaryOperator>(I)->swapOperands();
+}
+
  static BinaryOperator *CreateAdd(Value *S1, Value *S2, const Twine &Name,
                                   Instruction *InsertBefore, Value *FlagsOp) {
-  if (S1->getType()->isIntegerTy())
+  if (S1->getType()->isIntOrIntVectorTy())
      return BinaryOperator::CreateAdd(S1, S2, Name, InsertBefore);
    else {
      BinaryOperator *Res =
@@ -339,7 +241,7 @@ static BinaryOperator *CreateAdd(Value *S1, Value *S2, const Twine &Name,
  
  static BinaryOperator *CreateMul(Value *S1, Value *S2, const Twine &Name,
                                   Instruction *InsertBefore, Value *FlagsOp) {
-  if (S1->getType()->isIntegerTy())
+  if (S1->getType()->isIntOrIntVectorTy())
      return BinaryOperator::CreateMul(S1, S2, Name, InsertBefore);
    else {
      BinaryOperator *Res =
@@ -351,7 +253,7 @@ static BinaryOperator *CreateMul(Value *S1, Value *S2, const Twine &Name,
  
  static BinaryOperator *CreateNeg(Value *S1, const Twine &Name,
                                   Instruction *InsertBefore, Value *FlagsOp) {
-  if (S1->getType()->isIntegerTy())
+  if (S1->getType()->isIntOrIntVectorTy())
      return BinaryOperator::CreateNeg(S1, Name, InsertBefore);
    else {
      BinaryOperator *Res = BinaryOperator::CreateFNeg(S1, Name, InsertBefore);
@@ -360,12 +262,11 @@ static BinaryOperator *CreateNeg(Value *S1, const Twine &Name,
    }
  }
  
-/// LowerNegateToMultiply - Replace 0-X with X*-1.
-///
+/// Replace 0-X with X*-1.
  static BinaryOperator *LowerNegateToMultiply(Instruction *Neg) {
    Type *Ty = Neg->getType();
-  Constant *NegOne = Ty->isIntegerTy() ? ConstantInt::getAllOnesValue(Ty)
-                                       : ConstantFP::get(Ty, -1.0);
+  Constant *NegOne = Ty->isIntOrIntVectorTy() ?
+    ConstantInt::getAllOnesValue(Ty) : ConstantFP::get(Ty, -1.0);
  
    BinaryOperator *Res = CreateMul(Neg->getOperand(1), NegOne, "", Neg, Neg);
    Neg->setOperand(1, Constant::getNullValue(Ty)); // Drop use of op.
@@ -375,8 +276,8 @@ static BinaryOperator *LowerNegateToMultiply(Instruction *Neg) {
    return Res;
  }
  
-/// CarmichaelShift - Returns k such that lambda(2^Bitwidth) = 2^k, where lambda
-/// is the Carmichael function. This means that x^(2^k) === 1 mod 2^Bitwidth for
+/// Returns k such that lambda(2^Bitwidth) = 2^k, where lambda is the Carmichael
+/// function. This means that x^(2^k) === 1 mod 2^Bitwidth for
  /// every odd x, i.e. x^(2^k) = 1 for every odd x in Bitwidth-bit arithmetic.
  /// Note that 0 <= k < Bitwidth, and if Bitwidth > 3 then x^(2^k) = 0 for every
  /// even x in Bitwidth-bit arithmetic.
@@ -386,7 +287,7 @@ static unsigned CarmichaelShift(unsigned Bitwidth) {
    return Bitwidth - 2;
  }
  
-/// IncorporateWeight - Add the extra weight 'RHS' to the existing weight 'LHS',
+/// Add the extra weight 'RHS' to the existing weight 'LHS',
  /// reducing the combined weight using any special properties of the operation.
  /// The existing weight LHS represents the computation X op X op ... op X where
  /// X occurs LHS times.  The combined weight represents  X op X op ... op X with
@@ -466,9 +367,9 @@ static void IncorporateWeight(APInt &LHS, const APInt &RHS, unsigned Opcode) {
    }
  }
  
-typedef std::pair<Value*, APInt> RepeatedValue;
+using RepeatedValue = std::pair<Value*, APInt>;
  
-/// LinearizeExprTree - Given an associative binary expression, return the leaf
+/// Given an associative binary expression, return the leaf
  /// nodes in Ops along with their weights (how many times the leaf occurs).  The
  /// original expression is the same as
  ///   (Ops[0].first op Ops[0].first op ... Ops[0].first)  <- Ops[0].second times
@@ -541,7 +442,6 @@ typedef std::pair<Value*, APInt> RepeatedValue;
  /// that have all uses inside the expression (i.e. only used by non-leaf nodes
  /// of the expression) if it can turn them into binary operators of the right
  /// type and thus make the expression bigger.
-
  static bool LinearizeExprTree(BinaryOperator *I,
                                SmallVectorImpl<RepeatedValue> &Ops) {
    DEBUG(dbgs() << "LINEARIZE: " << *I << '\n');
@@ -562,7 +462,7 @@ static bool LinearizeExprTree(BinaryOperator *I,
    // ways to get to it.
    SmallVector<std::pair<BinaryOperator*, APInt>, 8> Worklist; // (Op, Weight)
    Worklist.push_back(std::make_pair(I, APInt(Bitwidth, 1)));
-  bool MadeChange = false;
+  bool Changed = false;
  
    // Leaves of the expression are values that either aren't the right kind of
    // operation (eg: a constant, or a multiply in an add tree), or are, but have
@@ -579,12 +479,12 @@ static bool LinearizeExprTree(BinaryOperator *I,
  
    // Leaves - Keeps track of the set of putative leaves as well as the number of
    // paths to each leaf seen so far.
-  typedef DenseMap<Value*, APInt> LeafMap;
+  using LeafMap = DenseMap<Value *, APInt>;
    LeafMap Leaves; // Leaf -> Total weight so far.
-  SmallVector<Value*, 8> LeafOrder; // Ensure deterministic leaf output order.
+  SmallVector<Value *, 8> LeafOrder; // Ensure deterministic leaf output order.
  
  #ifndef NDEBUG
-  SmallPtrSet<Value*, 8> Visited; // For sanity checking the iteration scheme.
+  SmallPtrSet<Value *, 8> Visited; // For sanity checking the iteration scheme.
  #endif
    while (!Worklist.empty()) {
      std::pair<BinaryOperator*, APInt> P = Worklist.pop_back_val();
@@ -599,7 +499,7 @@ static bool LinearizeExprTree(BinaryOperator *I,
        // If this is a binary operation of the right kind with only one use then
        // add its operands to the expression.
        if (BinaryOperator *BO = isReassociableOp(Op, Opcode)) {
-        assert(Visited.insert(Op) && "Not first visit!");
+        assert(Visited.insert(Op).second && "Not first visit!");
          DEBUG(dbgs() << "DIRECT ADD: " << *Op << " (" << Weight << ")\n");
          Worklist.push_back(std::make_pair(BO, Weight));
          continue;
@@ -609,7 +509,7 @@ static bool LinearizeExprTree(BinaryOperator *I,
        LeafMap::iterator It = Leaves.find(Op);
        if (It == Leaves.end()) {
          // Not in the leaf map.  Must be the first time we saw this operand.
-        assert(Visited.insert(Op) && "Not first visit!");
+        assert(Visited.insert(Op).second && "Not first visit!");
          if (!Op->hasOneUse()) {
            // This value has uses not accounted for by the expression, so it is
            // not safe to modify.  Mark it as being a leaf.
@@ -619,9 +519,10 @@ static bool LinearizeExprTree(BinaryOperator *I,
            continue;
          }
          // No uses outside the expression, try morphing it.
-      } else if (It != Leaves.end()) {
+      } else {
          // Already in the leaf map.
-        assert(Visited.count(Op) && "In leaf map but not visited!");
+        assert(It != Leaves.end() && Visited.count(Op) &&
+               "In leaf map but not visited!");
  
          // Update the number of paths to the leaf.
          IncorporateWeight(It->second, Weight, Opcode);
@@ -631,7 +532,7 @@ static bool LinearizeExprTree(BinaryOperator *I,
          // exactly one such use, drop this new use of the leaf.
          assert(!Op->hasOneUse() && "Only one use, but we got here twice!");
          I->setOperand(OpIdx, UndefValue::get(I->getType()));
-        MadeChange = true;
+        Changed = true;
  
          // If the leaf is a binary operation of the right kind and we now see
          // that its multiple original uses were in fact all by nodes belonging
@@ -660,7 +561,9 @@ static bool LinearizeExprTree(BinaryOperator *I,
        // expression.  This means that it can safely be modified.  See if we
        // can usefully morph it into an expression of the right kind.
        assert((!isa<Instruction>(Op) ||
-              cast<Instruction>(Op)->getOpcode() != Opcode) &&
+              cast<Instruction>(Op)->getOpcode() != Opcode
+              || (isa<FPMathOperator>(Op) &&
+                  !cast<Instruction>(Op)->isFast())) &&
               "Should have been handled above!");
        assert(Op->hasOneUse() && "Has uses outside the expression tree!");
  
@@ -673,7 +576,7 @@ static bool LinearizeExprTree(BinaryOperator *I,
            BO = LowerNegateToMultiply(BO);
            DEBUG(dbgs() << *BO << '\n');
            Worklist.push_back(std::make_pair(BO, Weight));
-          MadeChange = true;
+          Changed = true;
            continue;
          }
  
@@ -710,16 +613,16 @@ static bool LinearizeExprTree(BinaryOperator *I,
    if (Ops.empty()) {
      Constant *Identity = ConstantExpr::getBinOpIdentity(Opcode, I->getType());
      assert(Identity && "Associative operation without identity!");
-    Ops.push_back(std::make_pair(Identity, APInt(Bitwidth, 1)));
+    Ops.emplace_back(Identity, APInt(Bitwidth, 1));
    }
  
-  return MadeChange;
+  return Changed;
  }
  
-// RewriteExprTree - Now that the operands for this expression tree are
-// linearized and optimized, emit them in-order.
-void Reassociate::RewriteExprTree(BinaryOperator *I,
-                                  SmallVectorImpl<ValueEntry> &Ops) {
+/// Now that the operands for this expression tree are
+/// linearized and optimized, emit them in-order.
+void ReassociatePass::RewriteExprTree(BinaryOperator *I,
+                                      SmallVectorImpl<ValueEntry> &Ops) {
    assert(Ops.size() > 1 && "Single values should be used directly!");
  
    // Since our optimizations should never increase the number of operations, the
@@ -846,7 +749,7 @@ void Reassociate::RewriteExprTree(BinaryOperator *I,
        Constant *Undef = UndefValue::get(I->getType());
        NewOp = BinaryOperator::Create(Instruction::BinaryOps(Opcode),
                                       Undef, Undef, "", I);
-      if (NewOp->getType()->isFloatingPointTy())
+      if (NewOp->getType()->isFPOrFPVectorTy())
          NewOp->setFastMathFlags(I->getFastMathFlags());
      } else {
        NewOp = NodesToRewrite.pop_back_val();
@@ -879,22 +782,28 @@ void Reassociate::RewriteExprTree(BinaryOperator *I,
          break;
        ExpressionChanged->moveBefore(I);
        ExpressionChanged = cast<BinaryOperator>(*ExpressionChanged->user_begin());
-    } while (1);
+    } while (true);
  
    // Throw away any left over nodes from the original expression.
    for (unsigned i = 0, e = NodesToRewrite.size(); i != e; ++i)
      RedoInsts.insert(NodesToRewrite[i]);
  }
  
-/// NegateValue - Insert instructions before the instruction pointed to by BI,
+/// Insert instructions before the instruction pointed to by BI,
  /// that computes the negative version of the value specified.  The negative
  /// version of the value is returned, and BI is left pointing at the instruction
  /// that should be processed next by the reassociation pass.
-static Value *NegateValue(Value *V, Instruction *BI) {
-  if (ConstantFP *C = dyn_cast<ConstantFP>(V))
-    return ConstantExpr::getFNeg(C);
-  if (Constant *C = dyn_cast<Constant>(V))
+/// Also add intermediate instructions to the redo list that are modified while
+/// pushing the negates through adds.  These will be revisited to see if
+/// additional opportunities have been exposed.
+static Value *NegateValue(Value *V, Instruction *BI,
+                          SetVector<AssertingVH<Instruction>> &ToRedo) {
+  if (Constant *C = dyn_cast<Constant>(V)) {
+    if (C->getType()->isFPOrFPVectorTy()) {
+      return ConstantExpr::getFNeg(C);
+    }
      return ConstantExpr::getNeg(C);
+  }
  
    // We are trying to expose opportunity for reassociation.  One of the things
    // that we want to do to achieve this is to push a negation as deep into an
@@ -908,8 +817,12 @@ static Value *NegateValue(Value *V, Instruction *BI) {
    if (BinaryOperator *I =
            isReassociableOp(V, Instruction::Add, Instruction::FAdd)) {
      // Push the negates through the add.
-    I->setOperand(0, NegateValue(I->getOperand(0), BI));
-    I->setOperand(1, NegateValue(I->getOperand(1), BI));
+    I->setOperand(0, NegateValue(I->getOperand(0), BI, ToRedo));
+    I->setOperand(1, NegateValue(I->getOperand(1), BI, ToRedo));
+    if (I->getOpcode() == Instruction::Add) {
+      I->setHasNoUnsignedWrap(false);
+      I->setHasNoSignedWrap(false);
+    }
  
      // We must move the add instruction here, because the neg instructions do
      // not dominate the old add instruction in general.  By moving it, we are
@@ -918,6 +831,10 @@ static Value *NegateValue(Value *V, Instruction *BI) {
      //
      I->moveBefore(BI);
      I->setName(I->getName()+".neg");
+
+    // Add the intermediate negates to the redo list as processing them later
+    // could expose more reassociating opportunities.
+    ToRedo.insert(I);
      return I;
    }
  
@@ -942,29 +859,40 @@ static Value *NegateValue(Value *V, Instruction *BI) {
        if (InvokeInst *II = dyn_cast<InvokeInst>(InstInput)) {
          InsertPt = II->getNormalDest()->begin();
        } else {
-        InsertPt = InstInput;
-        ++InsertPt;
+        InsertPt = ++InstInput->getIterator();
        }
        while (isa<PHINode>(InsertPt)) ++InsertPt;
      } else {
        InsertPt = TheNeg->getParent()->getParent()->getEntryBlock().begin();
      }
-    TheNeg->moveBefore(InsertPt);
+    TheNeg->moveBefore(&*InsertPt);
+    if (TheNeg->getOpcode() == Instruction::Sub) {
+      TheNeg->setHasNoUnsignedWrap(false);
+      TheNeg->setHasNoSignedWrap(false);
+    } else {
+      TheNeg->andIRFlags(BI);
+    }
+    ToRedo.insert(TheNeg);
      return TheNeg;
    }
  
    // Insert a 'neg' instruction that subtracts the value from zero to get the
    // negation.
-  return CreateNeg(V, V->getName() + ".neg", BI, BI);
+  BinaryOperator *NewNeg = CreateNeg(V, V->getName() + ".neg", BI, BI);
+  ToRedo.insert(NewNeg);
+  return NewNeg;
  }
  
-/// ShouldBreakUpSubtract - Return true if we should break up this subtract of
-/// X-Y into (X + -Y).
+/// Return true if we should break up this subtract of X-Y into (X + -Y).
  static bool ShouldBreakUpSubtract(Instruction *Sub) {
    // If this is a negation, we can't split it up!
    if (BinaryOperator::isNeg(Sub) || BinaryOperator::isFNeg(Sub))
      return false;
  
+  // Don't breakup X - undef.
+  if (isa<UndefValue>(Sub->getOperand(1)))
+    return false;
+
    // Don't bother to break this up unless either the LHS is an associable add or
    // subtract or if this is only used by one.
    Value *V0 = Sub->getOperand(0);
@@ -984,17 +912,16 @@ static bool ShouldBreakUpSubtract(Instruction *Sub) {
    return false;
  }
  
-/// BreakUpSubtract - If we have (X-Y), and if either X is an add, or if this is
-/// only used by an add, transform this into (X+(0-Y)) to promote better
-/// reassociation.
-static BinaryOperator *BreakUpSubtract(Instruction *Sub) {
+/// If we have (X-Y), and if either X is an add, or if this is only used by an
+/// add, transform this into (X+(0-Y)) to promote better reassociation.
+static BinaryOperator *
+BreakUpSubtract(Instruction *Sub, SetVector<AssertingVH<Instruction>> &ToRedo) {
    // Convert a subtract into an add and a neg instruction. This allows sub
    // instructions to be commuted with other add instructions.
    //
    // Calculate the negative value of Operand 1 of the sub instruction,
    // and set it as the RHS of the add instruction we just made.
-  //
-  Value *NegVal = NegateValue(Sub->getOperand(1), Sub);
+  Value *NegVal = NegateValue(Sub->getOperand(1), Sub, ToRedo);
    BinaryOperator *New = CreateAdd(Sub->getOperand(0), NegVal, "", Sub, Sub);
    Sub->setOperand(0, Constant::getNullValue(Sub->getType())); // Drop use of op.
    Sub->setOperand(1, Constant::getNullValue(Sub->getType())); // Drop use of op.
@@ -1008,9 +935,8 @@ static BinaryOperator *BreakUpSubtract(Instruction *Sub) {
    return New;
  }
  
-/// ConvertShiftToMul - If this is a shift of a reassociable multiply or is used
-/// by one, change this into a multiply by a constant to assist with further
-/// reassociation.
+/// If this is a shift of a reassociable multiply or is used by one, change
+/// this into a multiply by a constant to assist with further reassociation.
  static BinaryOperator *ConvertShiftToMul(Instruction *Shl) {
    Constant *MulCst = ConstantInt::get(Shl->getType(), 1);
    MulCst = ConstantExpr::getShl(MulCst, cast<Constant>(Shl->getOperand(1)));
@@ -1019,33 +945,53 @@ static BinaryOperator *ConvertShiftToMul(Instruction *Shl) {
      BinaryOperator::CreateMul(Shl->getOperand(0), MulCst, "", Shl);
    Shl->setOperand(0, UndefValue::get(Shl->getType())); // Drop use of op.
    Mul->takeName(Shl);
+
+  // Everyone now refers to the mul instruction.
    Shl->replaceAllUsesWith(Mul);
    Mul->setDebugLoc(Shl->getDebugLoc());
+
+  // We can safely preserve the nuw flag in all cases.  It's also safe to turn a
+  // nuw nsw shl into a nuw nsw mul.  However, nsw in isolation requires special
+  // handling.
+  bool NSW = cast<BinaryOperator>(Shl)->hasNoSignedWrap();
+  bool NUW = cast<BinaryOperator>(Shl)->hasNoUnsignedWrap();
+  if (NSW && NUW)
+    Mul->setHasNoSignedWrap(true);
+  Mul->setHasNoUnsignedWrap(NUW);
    return Mul;
  }
  
-/// FindInOperandList - Scan backwards and forwards among values with the same
-/// rank as element i to see if X exists.  If X does not exist, return i.  This
-/// is useful when scanning for 'x' when we see '-x' because they both get the
-/// same rank.
-static unsigned FindInOperandList(SmallVectorImpl<ValueEntry> &Ops, unsigned i,
-                                  Value *X) {
+/// Scan backwards and forwards among values with the same rank as element i
+/// to see if X exists.  If X does not exist, return i.  This is useful when
+/// scanning for 'x' when we see '-x' because they both get the same rank.
+static unsigned FindInOperandList(const SmallVectorImpl<ValueEntry> &Ops,
+                                  unsigned i, Value *X) {
    unsigned XRank = Ops[i].Rank;
    unsigned e = Ops.size();
-  for (unsigned j = i+1; j != e && Ops[j].Rank == XRank; ++j)
+  for (unsigned j = i+1; j != e && Ops[j].Rank == XRank; ++j) {
      if (Ops[j].Op == X)
        return j;
+    if (Instruction *I1 = dyn_cast<Instruction>(Ops[j].Op))
+      if (Instruction *I2 = dyn_cast<Instruction>(X))
+        if (I1->isIdenticalTo(I2))
+          return j;
+  }
    // Scan backwards.
-  for (unsigned j = i-1; j != ~0U && Ops[j].Rank == XRank; --j)
+  for (unsigned j = i-1; j != ~0U && Ops[j].Rank == XRank; --j) {
      if (Ops[j].Op == X)
        return j;
+    if (Instruction *I1 = dyn_cast<Instruction>(Ops[j].Op))
+      if (Instruction *I2 = dyn_cast<Instruction>(X))
+        if (I1->isIdenticalTo(I2))
+          return j;
+  }
    return i;
  }
  
-/// EmitAddTreeOfValues - Emit a tree of add instructions, summing Ops together
+/// Emit a tree of add instructions, summing Ops together
  /// and returning the result.  Insert the tree before I.
  static Value *EmitAddTreeOfValues(Instruction *I,
-                                  SmallVectorImpl<WeakVH> &Ops){
+                                  SmallVectorImpl<WeakTrackingVH> &Ops) {
    if (Ops.size() == 1) return Ops.back();
  
    Value *V1 = Ops.back();
@@ -1054,10 +1000,10 @@ static Value *EmitAddTreeOfValues(Instruction *I,
    return CreateAdd(V2, V1, "tmp", I, I);
  }
  
-/// RemoveFactorFromExpression - If V is an expression tree that is a
-/// multiplication sequence, and if this sequence contains a multiply by Factor,
+/// If V is an expression tree that is a multiplication sequence,
+/// and if this sequence contains a multiply by Factor,
  /// remove Factor from the tree and return the new tree.
-Value *Reassociate::RemoveFactorFromExpression(Value *V, Value *Factor) {
+Value *ReassociatePass::RemoveFactorFromExpression(Value *V, Value *Factor) {
    BinaryOperator *BO = isReassociableOp(V, Instruction::Mul, Instruction::FMul);
    if (!BO)
      return nullptr;
@@ -1091,7 +1037,7 @@ Value *Reassociate::RemoveFactorFromExpression(Value *V, Value *Factor) {
          }
      } else if (ConstantFP *FC1 = dyn_cast<ConstantFP>(Factor)) {
        if (ConstantFP *FC2 = dyn_cast<ConstantFP>(Factors[i].Op)) {
-        APFloat F1(FC1->getValueAPF());
+        const APFloat &F1 = FC1->getValueAPF();
          APFloat F2(FC2->getValueAPF());
          F2.changeSign();
          if (F1.compare(F2) == APFloat::cmpEqual) {
@@ -1109,7 +1055,7 @@ Value *Reassociate::RemoveFactorFromExpression(Value *V, Value *Factor) {
      return nullptr;
    }
  
-  BasicBlock::iterator InsertPt = BO; ++InsertPt;
+  BasicBlock::iterator InsertPt = ++BO->getIterator();
  
    // If this was just a single multiply, remove the multiply and return the only
    // remaining operand.
@@ -1122,18 +1068,17 @@ Value *Reassociate::RemoveFactorFromExpression(Value *V, Value *Factor) {
    }
  
    if (NeedsNegate)
-    V = CreateNeg(V, "neg", InsertPt, BO);
+    V = CreateNeg(V, "neg", &*InsertPt, BO);
  
    return V;
  }
  
-/// FindSingleUseMultiplyFactors - If V is a single-use multiply, recursively
-/// add its operands as factors, otherwise add V to the list of factors.
+/// If V is a single-use multiply, recursively add its operands as factors,
+/// otherwise add V to the list of factors.
  ///
  /// Ops is the top-level list of add operands we're trying to factor.
  static void FindSingleUseMultiplyFactors(Value *V,
-                                         SmallVectorImpl<Value*> &Factors,
-                                       const SmallVectorImpl<ValueEntry> &Ops) {
+                                         SmallVectorImpl<Value*> &Factors) {
    BinaryOperator *BO = isReassociableOp(V, Instruction::Mul, Instruction::FMul);
    if (!BO) {
      Factors.push_back(V);
@@ -1141,14 +1086,13 @@ static void FindSingleUseMultiplyFactors(Value *V,
    }
  
    // Otherwise, add the LHS and RHS to the list of factors.
-  FindSingleUseMultiplyFactors(BO->getOperand(1), Factors, Ops);
-  FindSingleUseMultiplyFactors(BO->getOperand(0), Factors, Ops);
+  FindSingleUseMultiplyFactors(BO->getOperand(1), Factors);
+  FindSingleUseMultiplyFactors(BO->getOperand(0), Factors);
  }
  
-/// OptimizeAndOrXor - Optimize a series of operands to an 'and', 'or', or 'xor'
-/// instruction.  This optimizes based on identities.  If it can be reduced to
-/// a single Value, it is returned, otherwise the Ops list is mutated as
-/// necessary.
+/// Optimize a series of operands to an 'and', 'or', or 'xor' instruction.
+/// This optimizes based on identities.  If it can be reduced to a single Value,
+/// it is returned, otherwise the Ops list is mutated as necessary.
  static Value *OptimizeAndOrXor(unsigned Opcode,
                                 SmallVectorImpl<ValueEntry> &Ops) {
    // Scan the operand lists looking for X and ~X pairs, along with X,X pairs.
@@ -1194,25 +1138,24 @@ static Value *OptimizeAndOrXor(unsigned Opcode,
    return nullptr;
  }
  
-/// Helper funciton of CombineXorOpnd(). It creates a bitwise-and
+/// Helper function of CombineXorOpnd(). It creates a bitwise-and
  /// instruction with the given two operands, and return the resulting
  /// instruction. There are two special cases: 1) if the constant operand is 0,
  /// it will return NULL. 2) if the constant is ~0, the symbolic operand will
  /// be returned.
-static Value *createAndInstr(Instruction *InsertBefore, Value *Opnd, 
+static Value *createAndInstr(Instruction *InsertBefore, Value *Opnd,
                               const APInt &ConstOpnd) {
-  if (ConstOpnd != 0) {
-    if (!ConstOpnd.isAllOnesValue()) {
-      LLVMContext &Ctx = Opnd->getType()->getContext();
-      Instruction *I;
-      I = BinaryOperator::CreateAnd(Opnd, ConstantInt::get(Ctx, ConstOpnd),
-                                    "and.ra", InsertBefore);
-      I->setDebugLoc(InsertBefore->getDebugLoc());
-      return I;
-    }
+  if (ConstOpnd.isNullValue())
+    return nullptr;
+
+  if (ConstOpnd.isAllOnesValue())
      return Opnd;
-  }
-  return nullptr;
+
+  Instruction *I = BinaryOperator::CreateAnd(
+      Opnd, ConstantInt::get(Opnd->getType(), ConstOpnd), "and.ra",
+      InsertBefore);
+  I->setDebugLoc(InsertBefore->getDebugLoc());
+  return I;
  }
  
  // Helper function of OptimizeXor(). It tries to simplify "Opnd1 ^ ConstOpnd"
@@ -1221,33 +1164,31 @@ static Value *createAndInstr(Instruction *InsertBefore, Value *Opnd,
  // If it was successful, true is returned, and the "R" and "C" is returned
  // via "Res" and "ConstOpnd", respectively; otherwise, false is returned,
  // and both "Res" and "ConstOpnd" remain unchanged.
-//  
-bool Reassociate::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1,
-                                 APInt &ConstOpnd, Value *&Res) {
+bool ReassociatePass::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1,
+                                     APInt &ConstOpnd, Value *&Res) {
    // Xor-Rule 1: (x | c1) ^ c2 = (x | c1) ^ (c1 ^ c1) ^ c2 
    //                       = ((x | c1) ^ c1) ^ (c1 ^ c2)
    //                       = (x & ~c1) ^ (c1 ^ c2)
    // It is useful only when c1 == c2.
-  if (Opnd1->isOrExpr() && Opnd1->getConstPart() != 0) {
-    if (!Opnd1->getValue()->hasOneUse())
-      return false;
+  if (!Opnd1->isOrExpr() || Opnd1->getConstPart().isNullValue())
+    return false;
  
-    const APInt &C1 = Opnd1->getConstPart();
-    if (C1 != ConstOpnd)
-      return false;
+  if (!Opnd1->getValue()->hasOneUse())
+    return false;
  
-    Value *X = Opnd1->getSymbolicPart();
-    Res = createAndInstr(I, X, ~C1);
-    // ConstOpnd was C2, now C1 ^ C2.
-    ConstOpnd ^= C1;
+  const APInt &C1 = Opnd1->getConstPart();
+  if (C1 != ConstOpnd)
+    return false;
  
-    if (Instruction *T = dyn_cast<Instruction>(Opnd1->getValue()))
-      RedoInsts.insert(T);
-    return true;
-  }
-  return false;
-}
+  Value *X = Opnd1->getSymbolicPart();
+  Res = createAndInstr(I, X, ~C1);
+  // ConstOpnd was C2, now C1 ^ C2.
+  ConstOpnd ^= C1;
  
+  if (Instruction *T = dyn_cast<Instruction>(Opnd1->getValue()))
+    RedoInsts.insert(T);
+  return true;
+}
                             
  // Helper function of OptimizeXor(). It tries to simplify
  // "Opnd1 ^ Opnd2 ^ ConstOpnd" into "R ^ C", where C would be 0, and R is a
@@ -1257,8 +1198,9 @@ bool Reassociate::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1,
  // via "Res" and "ConstOpnd", respectively (If the entire expression is
  // evaluated to a constant, the Res is set to NULL); otherwise, false is
  // returned, and both "Res" and "ConstOpnd" remain unchanged.
-bool Reassociate::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1, XorOpnd *Opnd2,
-                                 APInt &ConstOpnd, Value *&Res) {
+bool ReassociatePass::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1,
+                                     XorOpnd *Opnd2, APInt &ConstOpnd,
+                                     Value *&Res) {
    Value *X = Opnd1->getSymbolicPart();
    if (X != Opnd2->getSymbolicPart())
      return false;
@@ -1285,15 +1227,14 @@ bool Reassociate::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1, XorOpnd *Opnd2,
      APInt C3((~C1) ^ C2);
  
      // Do not increase code size!
-    if (C3 != 0 && !C3.isAllOnesValue()) {
-      int NewInstNum = ConstOpnd != 0 ? 1 : 2;
+    if (!C3.isNullValue() && !C3.isAllOnesValue()) {
+      int NewInstNum = ConstOpnd.getBoolValue() ? 1 : 2;
        if (NewInstNum > DeadInstNum)
          return false;
      }
  
      Res = createAndInstr(I, X, C3);
      ConstOpnd ^= C1;
-
    } else if (Opnd1->isOrExpr()) {
      // Xor-Rule 3: (x | c1) ^ (x | c2) = (x & c3) ^ c3 where c3 = c1 ^ c2
      //
@@ -1302,8 +1243,8 @@ bool Reassociate::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1, XorOpnd *Opnd2,
      APInt C3 = C1 ^ C2;
      
      // Do not increase code size
-    if (C3 != 0 && !C3.isAllOnesValue()) {
-      int NewInstNum = ConstOpnd != 0 ? 1 : 2;
+    if (!C3.isNullValue() && !C3.isAllOnesValue()) {
+      int NewInstNum = ConstOpnd.getBoolValue() ? 1 : 2;
        if (NewInstNum > DeadInstNum)
          return false;
      }
@@ -1332,8 +1273,8 @@ bool Reassociate::CombineXorOpnd(Instruction *I, XorOpnd *Opnd1, XorOpnd *Opnd2,
  /// Optimize a series of operands to an 'xor' instruction. If it can be reduced
  /// to a single Value, it is returned, otherwise the Ops list is mutated as
  /// necessary.
-Value *Reassociate::OptimizeXor(Instruction *I,
-                                SmallVectorImpl<ValueEntry> &Ops) {
+Value *ReassociatePass::OptimizeXor(Instruction *I,
+                                    SmallVectorImpl<ValueEntry> &Ops) {
    if (Value *V = OptimizeAndOrXor(Instruction::Xor, Ops))
      return V;
        
@@ -1343,17 +1284,20 @@ Value *Reassociate::OptimizeXor(Instruction *I,
    SmallVector<XorOpnd, 8> Opnds;
    SmallVector<XorOpnd*, 8> OpndPtrs;
    Type *Ty = Ops[0].Op->getType();
-  APInt ConstOpnd(Ty->getIntegerBitWidth(), 0);
+  APInt ConstOpnd(Ty->getScalarSizeInBits(), 0);
  
    // Step 1: Convert ValueEntry to XorOpnd
    for (unsigned i = 0, e = Ops.size(); i != e; ++i) {
      Value *V = Ops[i].Op;
-    if (!isa<ConstantInt>(V)) {
+    const APInt *C;
+    // TODO: Support non-splat vectors.
+    if (match(V, PatternMatch::m_APInt(C))) {
+      ConstOpnd ^= *C;
+    } else {
        XorOpnd O(V);
        O.setSymbolicRank(getRank(O.getSymbolicPart()));
        Opnds.push_back(O);
-    } else
-      ConstOpnd ^= cast<ConstantInt>(V)->getValue();
+    }
    }
  
    // NOTE: From this point on, do *NOT* add/delete element to/from "Opnds".
@@ -1368,7 +1312,19 @@ Value *Reassociate::OptimizeXor(Instruction *I,
    //  the same symbolic value cluster together. For instance, the input operand
    //  sequence ("x | 123", "y & 456", "x & 789") will be sorted into:
    //  ("x | 123", "x & 789", "y & 456").
-  std::stable_sort(OpndPtrs.begin(), OpndPtrs.end(), XorOpnd::PtrSortFunctor());
+  //
+  //  The purpose is twofold:
+  //  1) Cluster together the operands sharing the same symbolic-value.
+  //  2) Operand having smaller symbolic-value-rank is permuted earlier, which
+  //     could potentially shorten crital path, and expose more loop-invariants.
+  //     Note that values' rank are basically defined in RPO order (FIXME).
+  //     So, if Rank(X) < Rank(Y) < Rank(Z), it means X is defined earlier
+  //     than Y which is defined earlier than Z. Permute "x | 1", "Y & 2",
+  //     "z" in the order of X-Y-Z is better than any other orders.
+  std::stable_sort(OpndPtrs.begin(), OpndPtrs.end(),
+                   [](XorOpnd *LHS, XorOpnd *RHS) {
+    return LHS->getSymbolicRank() < RHS->getSymbolicRank();
+  });
  
    // Step 3: Combine adjacent operands
    XorOpnd *PrevOpnd = nullptr;
@@ -1379,7 +1335,8 @@ Value *Reassociate::OptimizeXor(Instruction *I,
      Value *CV;
  
      // Step 3.1: Try simplifying "CurrOpnd ^ ConstOpnd"
-    if (ConstOpnd != 0 && CombineXorOpnd(I, CurrOpnd, ConstOpnd, CV)) {
+    if (!ConstOpnd.isNullValue() &&
+        CombineXorOpnd(I, CurrOpnd, ConstOpnd, CV)) {
        Changed = true;
        if (CV)
          *CurrOpnd = XorOpnd(CV);
@@ -1396,7 +1353,6 @@ Value *Reassociate::OptimizeXor(Instruction *I,
  
      // step 3.2: When previous and current operands share the same symbolic
      //  value, try to simplify "PrevOpnd ^ CurrOpnd ^ ConstOpnd" 
-    //    
      if (CombineXorOpnd(I, CurrOpnd, PrevOpnd, ConstOpnd, CV)) {
        // Remove previous operand
        PrevOpnd->Invalidate();
@@ -1421,28 +1377,28 @@ Value *Reassociate::OptimizeXor(Instruction *I,
        ValueEntry VE(getRank(O.getValue()), O.getValue());
        Ops.push_back(VE);
      }
-    if (ConstOpnd != 0) {
-      Value *C = ConstantInt::get(Ty->getContext(), ConstOpnd);
+    if (!ConstOpnd.isNullValue()) {
+      Value *C = ConstantInt::get(Ty, ConstOpnd);
        ValueEntry VE(getRank(C), C);
        Ops.push_back(VE);
      }
-    int Sz = Ops.size();
+    unsigned Sz = Ops.size();
      if (Sz == 1)
        return Ops.back().Op;
-    else if (Sz == 0) {
-      assert(ConstOpnd == 0);
-      return ConstantInt::get(Ty->getContext(), ConstOpnd);
+    if (Sz == 0) {
+      assert(ConstOpnd.isNullValue());
+      return ConstantInt::get(Ty, ConstOpnd);
      }
    }
  
    return nullptr;
  }
  
-/// OptimizeAdd - Optimize a series of operands to an 'add' instruction.  This
+/// Optimize a series of operands to an 'add' instruction.  This
  /// optimizes based on identities.  If it can be reduced to a single Value, it
  /// is returned, otherwise the Ops list is mutated as necessary.
-Value *Reassociate::OptimizeAdd(Instruction *I,
-                                SmallVectorImpl<ValueEntry> &Ops) {
+Value *ReassociatePass::OptimizeAdd(Instruction *I,
+                                    SmallVectorImpl<ValueEntry> &Ops) {
    // Scan the operand lists looking for X and -X pairs.  If we find any, we
    // can simplify expressions like X+-X == 0 and X+~X ==-1.  While we're at it,
    // scan for any
@@ -1461,13 +1417,13 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
          ++NumFound;
        } while (i != Ops.size() && Ops[i].Op == TheOp);
  
-      DEBUG(errs() << "\nFACTORING [" << NumFound << "]: " << *TheOp << '\n');
+      DEBUG(dbgs() << "\nFACTORING [" << NumFound << "]: " << *TheOp << '\n');
        ++NumFactor;
  
        // Insert a new multiply.
        Type *Ty = TheOp->getType();
-      Constant *C = Ty->isIntegerTy() ? ConstantInt::get(Ty, NumFound)
-                                      : ConstantFP::get(Ty, NumFound);
+      Constant *C = Ty->isIntOrIntVectorTy() ?
+        ConstantInt::get(Ty, NumFound) : ConstantFP::get(Ty, NumFound);
        Instruction *Mul = CreateMul(TheOp, C, "factor", I, I);
  
        // Now that we have inserted a multiply, optimize it. This allows us to
@@ -1550,14 +1506,14 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
  
      // Compute all of the factors of this added value.
      SmallVector<Value*, 8> Factors;
-    FindSingleUseMultiplyFactors(BOp, Factors, Ops);
+    FindSingleUseMultiplyFactors(BOp, Factors);
      assert(Factors.size() > 1 && "Bad linearize!");
  
      // Add one to FactorOccurrences for each unique factor in this op.
      SmallPtrSet<Value*, 8> Duplicates;
      for (unsigned i = 0, e = Factors.size(); i != e; ++i) {
        Value *Factor = Factors[i];
-      if (!Duplicates.insert(Factor))
+      if (!Duplicates.insert(Factor).second)
          continue;
  
        unsigned Occ = ++FactorOccurrences[Factor];
@@ -1572,8 +1528,8 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
        if (ConstantInt *CI = dyn_cast<ConstantInt>(Factor)) {
          if (CI->isNegative() && !CI->isMinValue(true)) {
            Factor = ConstantInt::get(CI->getContext(), -CI->getValue());
-          assert(!Duplicates.count(Factor) &&
-                 "Shouldn't have two constant factors, missed a canonicalize");
+          if (!Duplicates.insert(Factor).second)
+            continue;
            unsigned Occ = ++FactorOccurrences[Factor];
            if (Occ > MaxOcc) {
              MaxOcc = Occ;
@@ -1585,8 +1541,8 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
            APFloat F(CF->getValueAPF());
            F.changeSign();
            Factor = ConstantFP::get(CF->getContext(), F);
-          assert(!Duplicates.count(Factor) &&
-                 "Shouldn't have two constant factors, missed a canonicalize");
+          if (!Duplicates.insert(Factor).second)
+            continue;
            unsigned Occ = ++FactorOccurrences[Factor];
            if (Occ > MaxOcc) {
              MaxOcc = Occ;
@@ -1599,7 +1555,7 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
  
    // If any factor occurred more than one time, we can pull it out.
    if (MaxOcc > 1) {
-    DEBUG(errs() << "\nFACTORING [" << MaxOcc << "]: " << *MaxOccVal << '\n');
+    DEBUG(dbgs() << "\nFACTORING [" << MaxOcc << "]: " << *MaxOccVal << '\n');
      ++NumFactor;
  
      // Create a new instruction that uses the MaxOccVal twice.  If we don't do
@@ -1607,11 +1563,11 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
      // from an expression will drop a use of maxocc, and this can cause
      // RemoveFactorFromExpression on successive values to behave differently.
      Instruction *DummyInst =
-        I->getType()->isIntegerTy()
+        I->getType()->isIntOrIntVectorTy()
              ? BinaryOperator::CreateAdd(MaxOccVal, MaxOccVal)
              : BinaryOperator::CreateFAdd(MaxOccVal, MaxOccVal);
  
-    SmallVector<WeakVH, 4> NewMulOps;
+    SmallVector<WeakTrackingVH, 4> NewMulOps;
      for (unsigned i = 0; i != Ops.size(); ++i) {
        // Only try to remove factors from expressions we're allowed to.
        BinaryOperator *BOp =
@@ -1634,7 +1590,7 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
      }
  
      // No need for extra uses anymore.
-    delete DummyInst;
+    DummyInst->deleteValue();
  
      unsigned NumAddedValues = NewMulOps.size();
      Value *V = EmitAddTreeOfValues(I, NewMulOps);
@@ -1679,8 +1635,8 @@ Value *Reassociate::OptimizeAdd(Instruction *I,
  ///   ((((x*y)*x)*y)*x) -> [(x, 3), (y, 2)]
  ///
  /// \returns Whether any factors have a power greater than one.
-bool Reassociate::collectMultiplyFactors(SmallVectorImpl<ValueEntry> &Ops,
-                                         SmallVectorImpl<Factor> &Factors) {
+static bool collectMultiplyFactors(SmallVectorImpl<ValueEntry> &Ops,
+                                   SmallVectorImpl<Factor> &Factors) {
    // FIXME: Have Ops be (ValueEntry, Multiplicity) pairs, simplifying this.
    // Compute the sum of powers of simplifiable factors.
    unsigned FactorPowerSum = 0;
@@ -1726,7 +1682,10 @@ bool Reassociate::collectMultiplyFactors(SmallVectorImpl<ValueEntry> &Ops,
    // below our mininum of '4'.
    assert(FactorPowerSum >= 4);
  
-  std::stable_sort(Factors.begin(), Factors.end(), Factor::PowerDescendingSorter());
+  std::stable_sort(Factors.begin(), Factors.end(),
+                   [](const Factor &LHS, const Factor &RHS) {
+    return LHS.Power > RHS.Power;
+  });
    return true;
  }
  
@@ -1738,7 +1697,7 @@ static Value *buildMultiplyTree(IRBuilder<> &Builder,
  
    Value *LHS = Ops.pop_back_val();
    do {
-    if (LHS->getType()->isIntegerTy())
+    if (LHS->getType()->isIntOrIntVectorTy())
        LHS = Builder.CreateMul(LHS, Ops.pop_back_val());
      else
        LHS = Builder.CreateFMul(LHS, Ops.pop_back_val());
@@ -1753,8 +1712,9 @@ static Value *buildMultiplyTree(IRBuilder<> &Builder,
  /// equal and the powers are sorted in decreasing order, compute the minimal
  /// DAG of multiplies to compute the final product, and return that product
  /// value.
-Value *Reassociate::buildMinimalMultiplyDAG(IRBuilder<> &Builder,
-                                            SmallVectorImpl<Factor> &Factors) {
+Value *
+ReassociatePass::buildMinimalMultiplyDAG(IRBuilder<> &Builder,
+                                         SmallVectorImpl<Factor> &Factors) {
    assert(Factors[0].Power);
    SmallVector<Value *, 4> OuterProduct;
    for (unsigned LastIdx = 0, Idx = 1, Size = Factors.size();
@@ -1785,7 +1745,9 @@ Value *Reassociate::buildMinimalMultiplyDAG(IRBuilder<> &Builder,
    // Unique factors with equal powers -- we've folded them into the first one's
    // base.
    Factors.erase(std::unique(Factors.begin(), Factors.end(),
-                            Factor::PowerEqual()),
+                            [](const Factor &LHS, const Factor &RHS) {
+                              return LHS.Power == RHS.Power;
+                            }),
                  Factors.end());
  
    // Iteratively collect the base of each factor with an add power into the
@@ -1808,8 +1770,8 @@ Value *Reassociate::buildMinimalMultiplyDAG(IRBuilder<> &Builder,
    return V;
  }
  
-Value *Reassociate::OptimizeMul(BinaryOperator *I,
-                                SmallVectorImpl<ValueEntry> &Ops) {
+Value *ReassociatePass::OptimizeMul(BinaryOperator *I,
+                                    SmallVectorImpl<ValueEntry> &Ops) {
    // We can only optimize the multiplies when there is a chain of more than
    // three, such that a balanced tree might require fewer total multiplies.
    if (Ops.size() < 4)
@@ -1823,6 +1785,12 @@ Value *Reassociate::OptimizeMul(BinaryOperator *I,
      return nullptr; // All distinct factors, so nothing left for us to do.
  
    IRBuilder<> Builder(I);
+  // The reassociate transformation for FP operations is performed only
+  // if unsafe algebra is permitted by FastMathFlags. Propagate those flags
+  // to the newly generated operations.
+  if (auto FPI = dyn_cast<FPMathOperator>(I))
+    Builder.setFastMathFlags(FPI->getFastMathFlags());
+
    Value *V = buildMinimalMultiplyDAG(Builder, Factors);
    if (Ops.empty())
      return V;
@@ -1832,8 +1800,8 @@ Value *Reassociate::OptimizeMul(BinaryOperator *I,
    return nullptr;
  }
  
-Value *Reassociate::OptimizeExpression(BinaryOperator *I,
-                                       SmallVectorImpl<ValueEntry> &Ops) {
+Value *ReassociatePass::OptimizeExpression(BinaryOperator *I,
+                                           SmallVectorImpl<ValueEntry> &Ops) {
    // Now that we have the linearized expression tree, try to optimize it.
    // Start by folding any constants that we found.
    Constant *Cst = nullptr;
@@ -1891,10 +1859,27 @@ Value *Reassociate::OptimizeExpression(BinaryOperator *I,
    return nullptr;
  }
  
-/// EraseInst - Zap the given instruction, adding interesting operands to the
-/// work list.
-void Reassociate::EraseInst(Instruction *I) {
+// Remove dead instructions and if any operands are trivially dead add them to
+// Insts so they will be removed as well.
+void ReassociatePass::RecursivelyEraseDeadInsts(
+    Instruction *I, SetVector<AssertingVH<Instruction>> &Insts) {
+  assert(isInstructionTriviallyDead(I) && "Trivially dead instructions only!");
+  SmallVector<Value *, 4> Ops(I->op_begin(), I->op_end());
+  ValueRankMap.erase(I);
+  Insts.remove(I);
+  RedoInsts.remove(I);
+  I->eraseFromParent();
+  for (auto Op : Ops)
+    if (Instruction *OpInst = dyn_cast<Instruction>(Op))
+      if (OpInst->use_empty())
+        Insts.insert(OpInst);
+}
+
+/// Zap the given instruction, adding interesting operands to the work list.
+void ReassociatePass::EraseInst(Instruction *I) {
    assert(isInstructionTriviallyDead(I) && "Trivially dead instructions only!");
+  DEBUG(dbgs() << "Erasing dead inst: "; I->dump());
+
    SmallVector<Value*, 8> Ops(I->op_begin(), I->op_end());
    // Erase the dead instruction.
    ValueRankMap.erase(I);
@@ -1908,15 +1893,101 @@ void Reassociate::EraseInst(Instruction *I) {
        // and add that since that's where optimization actually happens.
        unsigned Opcode = Op->getOpcode();
        while (Op->hasOneUse() && Op->user_back()->getOpcode() == Opcode &&
-             Visited.insert(Op))
+             Visited.insert(Op).second)
          Op = Op->user_back();
        RedoInsts.insert(Op);
      }
+
+  MadeChange = true;
  }
  
-/// OptimizeInst - Inspect and optimize the given instruction. Note that erasing
+// Canonicalize expressions of the following form:
+//  x + (-Constant * y) -> x - (Constant * y)
+//  x - (-Constant * y) -> x + (Constant * y)
+Instruction *ReassociatePass::canonicalizeNegConstExpr(Instruction *I) {
+  if (!I->hasOneUse() || I->getType()->isVectorTy())
+    return nullptr;
+
+  // Must be a fmul or fdiv instruction.
+  unsigned Opcode = I->getOpcode();
+  if (Opcode != Instruction::FMul && Opcode != Instruction::FDiv)
+    return nullptr;
+
+  auto *C0 = dyn_cast<ConstantFP>(I->getOperand(0));
+  auto *C1 = dyn_cast<ConstantFP>(I->getOperand(1));
+
+  // Both operands are constant, let it get constant folded away.
+  if (C0 && C1)
+    return nullptr;
+
+  ConstantFP *CF = C0 ? C0 : C1;
+
+  // Must have one constant operand.
+  if (!CF)
+    return nullptr;
+
+  // Must be a negative ConstantFP.
+  if (!CF->isNegative())
+    return nullptr;
+
+  // User must be a binary operator with one or more uses.
+  Instruction *User = I->user_back();
+  if (!isa<BinaryOperator>(User) || User->use_empty())
+    return nullptr;
+
+  unsigned UserOpcode = User->getOpcode();
+  if (UserOpcode != Instruction::FAdd && UserOpcode != Instruction::FSub)
+    return nullptr;
+
+  // Subtraction is not commutative. Explicitly, the following transform is
+  // not valid: (-Constant * y) - x  -> x + (Constant * y)
+  if (!User->isCommutative() && User->getOperand(1) != I)
+    return nullptr;
+
+  // Don't canonicalize x + (-Constant * y) -> x - (Constant * y), if the
+  // resulting subtract will be broken up later.  This can get us into an
+  // infinite loop during reassociation.
+  if (UserOpcode == Instruction::FAdd && ShouldBreakUpSubtract(User))
+    return nullptr;
+
+  // Change the sign of the constant.
+  APFloat Val = CF->getValueAPF();
+  Val.changeSign();
+  I->setOperand(C0 ? 0 : 1, ConstantFP::get(CF->getContext(), Val));
+
+  // Canonicalize I to RHS to simplify the next bit of logic. E.g.,
+  // ((-Const*y) + x) -> (x + (-Const*y)).
+  if (User->getOperand(0) == I && User->isCommutative())
+    cast<BinaryOperator>(User)->swapOperands();
+
+  Value *Op0 = User->getOperand(0);
+  Value *Op1 = User->getOperand(1);
+  BinaryOperator *NI;
+  switch (UserOpcode) {
+  default:
+    llvm_unreachable("Unexpected Opcode!");
+  case Instruction::FAdd:
+    NI = BinaryOperator::CreateFSub(Op0, Op1);
+    NI->setFastMathFlags(cast<FPMathOperator>(User)->getFastMathFlags());
+    break;
+  case Instruction::FSub:
+    NI = BinaryOperator::CreateFAdd(Op0, Op1);
+    NI->setFastMathFlags(cast<FPMathOperator>(User)->getFastMathFlags());
+    break;
+  }
+
+  NI->insertBefore(User);
+  NI->setName(User->getName());
+  User->replaceAllUsesWith(NI);
+  NI->setDebugLoc(I->getDebugLoc());
+  RedoInsts.insert(I);
+  MadeChange = true;
+  return NI;
+}
+
+/// Inspect and optimize the given instruction. Note that erasing
  /// instructions is not allowed.
-void Reassociate::OptimizeInst(Instruction *I) {
+void ReassociatePass::OptimizeInst(Instruction *I) {
    // Only consider operations that we understand.
    if (!isa<BinaryOperator>(I))
      return;
@@ -1934,34 +2005,19 @@ void Reassociate::OptimizeInst(Instruction *I) {
        I = NI;
      }
  
-  // Commute floating point binary operators, to canonicalize the order of their
-  // operands.  This can potentially expose more CSE opportunities, and makes
-  // writing other transformations simpler.
-  if (I->getType()->isFloatingPointTy() || I->getType()->isVectorTy()) {
-
-    // FAdd and FMul can be commuted.
-    if (I->getOpcode() == Instruction::FMul ||
-        I->getOpcode() == Instruction::FAdd) {
-      Value *LHS = I->getOperand(0);
-      Value *RHS = I->getOperand(1);
-      unsigned LHSRank = getRank(LHS);
-      unsigned RHSRank = getRank(RHS);
-
-      // Sort the operands by rank.
-      if (RHSRank < LHSRank) {
-        I->setOperand(0, RHS);
-        I->setOperand(1, LHS);
-      }
-    }
+  // Canonicalize negative constants out of expressions.
+  if (Instruction *Res = canonicalizeNegConstExpr(I))
+    I = Res;
  
-    // FIXME: We should commute vector instructions as well.  However, this 
-    // requires further analysis to determine the effect on later passes.
+  // Commute binary operators, to canonicalize the order of their operands.
+  // This can potentially expose more CSE opportunities, and makes writing other
+  // transformations simpler.
+  if (I->isCommutative())
+    canonicalizeOperands(I);
  
-    // Don't try to optimize vector instructions or anything that doesn't have
-    // unsafe algebra.
-    if (I->getType()->isVectorTy() || !I->hasUnsafeAlgebra())
-      return;
-  }
+  // Don't optimize floating-point instructions unless they are 'fast'.
+  if (I->getType()->isFPOrFPVectorTy() && !I->isFast())
+    return;
  
    // Do not reassociate boolean (i1) expressions.  We want to preserve the
    // original order of evaluation for short-circuited comparisons that
@@ -1976,7 +2032,7 @@ void Reassociate::OptimizeInst(Instruction *I) {
    // see if we can convert it to X+-Y.
    if (I->getOpcode() == Instruction::Sub) {
      if (ShouldBreakUpSubtract(I)) {
-      Instruction *NI = BreakUpSubtract(I);
+      Instruction *NI = BreakUpSubtract(I, RedoInsts);
        RedoInsts.insert(I);
        MadeChange = true;
        I = NI;
@@ -1987,6 +2043,12 @@ void Reassociate::OptimizeInst(Instruction *I) {
            (!I->hasOneUse() ||
             !isReassociableOp(I->user_back(), Instruction::Mul))) {
          Instruction *NI = LowerNegateToMultiply(I);
+        // If the negate was simplified, revisit the users to see if we can
+        // reassociate further.
+        for (User *U : NI->users()) {
+          if (BinaryOperator *Tmp = dyn_cast<BinaryOperator>(U))
+            RedoInsts.insert(Tmp);
+        }
          RedoInsts.insert(I);
          MadeChange = true;
          I = NI;
@@ -1994,7 +2056,7 @@ void Reassociate::OptimizeInst(Instruction *I) {
      }
    } else if (I->getOpcode() == Instruction::FSub) {
      if (ShouldBreakUpSubtract(I)) {
-      Instruction *NI = BreakUpSubtract(I);
+      Instruction *NI = BreakUpSubtract(I, RedoInsts);
        RedoInsts.insert(I);
        MadeChange = true;
        I = NI;
@@ -2004,7 +2066,13 @@ void Reassociate::OptimizeInst(Instruction *I) {
        if (isReassociableOp(I->getOperand(1), Instruction::FMul) &&
            (!I->hasOneUse() ||
             !isReassociableOp(I->user_back(), Instruction::FMul))) {
+        // If the negate was simplified, revisit the users to see if we can
+        // reassociate further.
          Instruction *NI = LowerNegateToMultiply(I);
+        for (User *U : NI->users()) {
+          if (BinaryOperator *Tmp = dyn_cast<BinaryOperator>(U))
+            RedoInsts.insert(Tmp);
+        }
          RedoInsts.insert(I);
          MadeChange = true;
          I = NI;
@@ -2019,8 +2087,15 @@ void Reassociate::OptimizeInst(Instruction *I) {
    // If this is an interior node of a reassociable tree, ignore it until we
    // get to the root of the tree, to avoid N^2 analysis.
    unsigned Opcode = BO->getOpcode();
-  if (BO->hasOneUse() && BO->user_back()->getOpcode() == Opcode)
+  if (BO->hasOneUse() && BO->user_back()->getOpcode() == Opcode) {
+    // During the initial run we will get to the root of the tree.
+    // But if we get here while we are redoing instructions, there is no
+    // guarantee that the root will be visited. So Redo later
+    if (BO->user_back() != BO &&
+        BO->getParent() == BO->user_back()->getParent())
+      RedoInsts.insert(BO->user_back());
      return;
+  }
  
    // If this is an add tree that is used by a sub instruction, ignore it
    // until we process the subtract.
@@ -2034,10 +2109,7 @@ void Reassociate::OptimizeInst(Instruction *I) {
    ReassociateExpression(BO);
  }
  
-void Reassociate::ReassociateExpression(BinaryOperator *I) {
-  assert(!I->getType()->isVectorTy() &&
-         "Reassociation of vector instructions is not supported.");
-
+void ReassociatePass::ReassociateExpression(BinaryOperator *I) {
    // First, walk the expression tree, linearizing the tree, collecting the
    // operand information.
    SmallVector<RepeatedValue, 8> Tree;
@@ -2060,7 +2132,7 @@ void Reassociate::ReassociateExpression(BinaryOperator *I) {
    // the vector.
    std::stable_sort(Ops.begin(), Ops.end());
  
-  // OptimizeExpression - Now that we have the expression tree in a convenient
+  // Now that we have the expression tree in a convenient
    // sorted form, optimize it globally if possible.
    if (Value *V = OptimizeExpression(I, Ops)) {
      if (V == I)
@@ -2071,7 +2143,8 @@ void Reassociate::ReassociateExpression(BinaryOperator *I) {
      DEBUG(dbgs() << "Reassoc to scalar: " << *V << '\n');
      I->replaceAllUsesWith(V);
      if (Instruction *VI = dyn_cast<Instruction>(V))
-      VI->setDebugLoc(I->getDebugLoc());
+      if (I->getDebugLoc())
+        VI->setDebugLoc(I->getDebugLoc());
      RedoInsts.insert(I);
      ++NumAnnihil;
      return;
@@ -2085,7 +2158,7 @@ void Reassociate::ReassociateExpression(BinaryOperator *I) {
      if (I->getOpcode() == Instruction::Mul &&
          cast<Instruction>(I->user_back())->getOpcode() == Instruction::Add &&
          isa<ConstantInt>(Ops.back().Op) &&
-        cast<ConstantInt>(Ops.back().Op)->isAllOnesValue()) {
+        cast<ConstantInt>(Ops.back().Op)->isMinusOne()) {
        ValueEntry Tmp = Ops.pop_back_val();
        Ops.insert(Ops.begin(), Tmp);
      } else if (I->getOpcode() == Instruction::FMul &&
@@ -2119,26 +2192,47 @@ void Reassociate::ReassociateExpression(BinaryOperator *I) {
    RewriteExprTree(I, Ops);
  }
  
-bool Reassociate::runOnFunction(Function &F) {
-  if (skipOptnoneFunction(F))
-    return false;
+PreservedAnalyses ReassociatePass::run(Function &F, FunctionAnalysisManager &) {
+  // Get the functions basic blocks in Reverse Post Order. This order is used by
+  // BuildRankMap to pre calculate ranks correctly. It also excludes dead basic
+  // blocks (it has been seen that the analysis in this pass could hang when
+  // analysing dead basic blocks).
+  ReversePostOrderTraversal<Function *> RPOT(&F);
  
-  // Calculate the rank map for F
-  BuildRankMap(F);
+  // Calculate the rank map for F.
+  BuildRankMap(F, RPOT);
  
    MadeChange = false;
-  for (Function::iterator BI = F.begin(), BE = F.end(); BI != BE; ++BI) {
+  // Traverse the same blocks that was analysed by BuildRankMap.
+  for (BasicBlock *BI : RPOT) {
+    assert(RankMap.count(&*BI) && "BB should be ranked.");
      // Optimize every instruction in the basic block.
-    for (BasicBlock::iterator II = BI->begin(), IE = BI->end(); II != IE; )
-      if (isInstructionTriviallyDead(II)) {
-        EraseInst(II++);
+    for (BasicBlock::iterator II = BI->begin(), IE = BI->end(); II != IE;)
+      if (isInstructionTriviallyDead(&*II)) {
+        EraseInst(&*II++);
        } else {
-        OptimizeInst(II);
-        assert(II->getParent() == BI && "Moved to a different block!");
+        OptimizeInst(&*II);
+        assert(II->getParent() == &*BI && "Moved to a different block!");
          ++II;
        }
  
-    // If this produced extra instructions to optimize, handle them now.
+    // Make a copy of all the instructions to be redone so we can remove dead
+    // instructions.
+    SetVector<AssertingVH<Instruction>> ToRedo(RedoInsts);
+    // Iterate over all instructions to be reevaluated and remove trivially dead
+    // instructions. If any operand of the trivially dead instruction becomes
+    // dead mark it for deletion as well. Continue this process until all
+    // trivially dead instructions have been removed.
+    while (!ToRedo.empty()) {
+      Instruction *I = ToRedo.pop_back_val();
+      if (isInstructionTriviallyDead(I)) {
+        RecursivelyEraseDeadInsts(I, ToRedo);
+        MadeChange = true;
+      }
+    }
+
+    // Now that we have removed dead instructions, we can reoptimize the
+    // remaining instructions.
      while (!RedoInsts.empty()) {
        Instruction *I = RedoInsts.pop_back_val();
        if (isInstructionTriviallyDead(I))
@@ -2152,5 +2246,51 @@ bool Reassociate::runOnFunction(Function &F) {
    RankMap.clear();
    ValueRankMap.clear();
  
-  return MadeChange;
+  if (MadeChange) {
+    PreservedAnalyses PA;
+    PA.preserveSet<CFGAnalyses>();
+    PA.preserve<GlobalsAA>();
+    return PA;
+  }
+
+  return PreservedAnalyses::all();
+}
+
+namespace {
+
+  class ReassociateLegacyPass : public FunctionPass {
+    ReassociatePass Impl;
+
+  public:
+    static char ID; // Pass identification, replacement for typeid
+
+    ReassociateLegacyPass() : FunctionPass(ID) {
+      initializeReassociateLegacyPassPass(*PassRegistry::getPassRegistry());
+    }
+
+    bool runOnFunction(Function &F) override {
+      if (skipFunction(F))
+        return false;
+
+      FunctionAnalysisManager DummyFAM;
+      auto PA = Impl.run(F, DummyFAM);
+      return !PA.areAllPreserved();
+    }
+
+    void getAnalysisUsage(AnalysisUsage &AU) const override {
+      AU.setPreservesCFG();
+      AU.addPreserved<GlobalsAAWrapperPass>();
+    }
+  };
+
+} // end anonymous namespace
+
+char ReassociateLegacyPass::ID = 0;
+
+INITIALIZE_PASS(ReassociateLegacyPass, "reassociate",
+                "Reassociate expressions", false, false)
+
+// Public interface to the Reassociate pass
+FunctionPass *llvm::createReassociatePass() {
+  return new ReassociateLegacyPass();
  }