Update Eigen to commit:18018ed013029ca3f28f52a62360999b5a659eac CHANGELOG ========= 18018ed01 - Unwind Block of Blocks 81b48065e - Fix arm32 float division and related bugs PiperOrigin-RevId: 561185435 Change-Id: Ic2f5a3b17ed8f22a4e198cb3766a2f4c72ea00ae
diff --git a/Eigen/src/Core/Block.h b/Eigen/src/Core/Block.h index 6ef26ca..31cd094 100644 --- a/Eigen/src/Core/Block.h +++ b/Eigen/src/Core/Block.h
@@ -17,9 +17,10 @@ namespace Eigen { namespace internal { -template<typename XprType, int BlockRows, int BlockCols, bool InnerPanel> -struct traits<Block<XprType, BlockRows, BlockCols, InnerPanel> > : traits<XprType> +template<typename XprType_, int BlockRows, int BlockCols, bool InnerPanel_> +struct traits<Block<XprType_, BlockRows, BlockCols, InnerPanel_> > : traits<XprType_> { + typedef XprType_ XprType; typedef typename traits<XprType>::Scalar Scalar; typedef typename traits<XprType>::StorageKind StorageKind; typedef typename traits<XprType>::XprKind XprKind; @@ -53,12 +54,13 @@ // FIXME, this traits is rather specialized for dense object and it needs to be cleaned further FlagsLvalueBit = is_lvalue<XprType>::value ? LvalueBit : 0, FlagsRowMajorBit = IsRowMajor ? RowMajorBit : 0, - Flags = (traits<XprType>::Flags & (DirectAccessBit | (InnerPanel?CompressedAccessBit:0))) | FlagsLvalueBit | FlagsRowMajorBit, + Flags = (traits<XprType>::Flags & (DirectAccessBit | (InnerPanel_?CompressedAccessBit:0))) | FlagsLvalueBit | FlagsRowMajorBit, // FIXME DirectAccessBit should not be handled by expressions // // Alignment is needed by MapBase's assertions // We can sefely set it to false here. Internal alignment errors will be detected by an eigen_internal_assert in the respective evaluator - Alignment = 0 + Alignment = 0, + InnerPanel = InnerPanel_ ? 1 : 0 }; }; @@ -107,6 +109,7 @@ : public BlockImpl<XprType, BlockRows, BlockCols, InnerPanel, typename internal::traits<XprType>::StorageKind> { typedef BlockImpl<XprType, BlockRows, BlockCols, InnerPanel, typename internal::traits<XprType>::StorageKind> Impl; + using BlockHelper = internal::block_xpr_helper<Block>; public: //typedef typename Impl::Base Base; typedef Impl Base; @@ -149,9 +152,25 @@ eigen_assert(startRow >= 0 && blockRows >= 0 && startRow <= xpr.rows() - blockRows && startCol >= 0 && blockCols >= 0 && startCol <= xpr.cols() - blockCols); } + + // convert nested blocks (e.g. Block<Block<MatrixType>>) to a simple block expression (Block<MatrixType>) + + using ConstUnwindReturnType = Block<const typename BlockHelper::BaseType, BlockRows, BlockCols, InnerPanel>; + using UnwindReturnType = Block<typename BlockHelper::BaseType, BlockRows, BlockCols, InnerPanel>; + + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE ConstUnwindReturnType unwind() const { + return ConstUnwindReturnType(BlockHelper::base(*this), BlockHelper::row(*this, 0), BlockHelper::col(*this, 0), + this->rows(), this->cols()); + } + + template <typename T = Block, typename EnableIf = std::enable_if_t<!std::is_const<T>::value>> + EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE UnwindReturnType unwind() { + return UnwindReturnType(BlockHelper::base(*this), BlockHelper::row(*this, 0), BlockHelper::col(*this, 0), + this->rows(), this->cols()); + } }; -// The generic default implementation for dense block simplu forward to the internal::BlockImpl_dense +// The generic default implementation for dense block simply forward to the internal::BlockImpl_dense // that must be specialized for direct and non-direct access... template<typename XprType, int BlockRows, int BlockCols, bool InnerPanel> class BlockImpl<XprType, BlockRows, BlockCols, InnerPanel, Dense>
diff --git a/Eigen/src/Core/arch/NEON/PacketMath.h b/Eigen/src/Core/arch/NEON/PacketMath.h index adb1342..e70f8b0 100644 --- a/Eigen/src/Core/arch/NEON/PacketMath.h +++ b/Eigen/src/Core/arch/NEON/PacketMath.h
@@ -956,57 +956,6 @@ vdup_n_u64(vgetq_lane_u64(a, 1)*vgetq_lane_u64(b, 1))); } -template<> EIGEN_STRONG_INLINE Packet2f pdiv<Packet2f>(const Packet2f& a, const Packet2f& b) -{ -#if EIGEN_ARCH_ARM64 - return vdiv_f32(a,b); -#else - Packet2f inv, restep, div; - - // NEON does not offer a divide instruction, we have to do a reciprocal approximation - // However NEON in contrast to other SIMD engines (AltiVec/SSE), offers - // a reciprocal estimate AND a reciprocal step -which saves a few instructions - // vrecpeq_f32() returns an estimate to 1/b, which we will finetune with - // Newton-Raphson and vrecpsq_f32() - inv = vrecpe_f32(b); - - // This returns a differential, by which we will have to multiply inv to get a better - // approximation of 1/b. - restep = vrecps_f32(b, inv); - inv = vmul_f32(restep, inv); - - // Finally, multiply a by 1/b and get the wanted result of the division. - div = vmul_f32(a, inv); - - return div; -#endif -} -template<> EIGEN_STRONG_INLINE Packet4f pdiv<Packet4f>(const Packet4f& a, const Packet4f& b) -{ -#if EIGEN_ARCH_ARM64 - return vdivq_f32(a,b); -#else - Packet4f inv, restep, div; - - // NEON does not offer a divide instruction, we have to do a reciprocal approximation - // However NEON in contrast to other SIMD engines (AltiVec/SSE), offers - // a reciprocal estimate AND a reciprocal step -which saves a few instructions - // vrecpeq_f32() returns an estimate to 1/b, which we will finetune with - // Newton-Raphson and vrecpsq_f32() - inv = vrecpeq_f32(b); - - // This returns a differential, by which we will have to multiply inv to get a better - // approximation of 1/b. - restep = vrecpsq_f32(b, inv); - inv = vmulq_f32(restep, inv); - - // Finally, multiply a by 1/b and get the wanted result of the division. - div = vmulq_f32(a, inv); - - return div; -#endif -} - template<> EIGEN_STRONG_INLINE Packet4c pdiv<Packet4c>(const Packet4c& /*a*/, const Packet4c& /*b*/) { eigen_assert(false && "packet integer division are not supported by NEON"); @@ -3362,26 +3311,115 @@ return res; } +EIGEN_STRONG_INLINE Packet4f prsqrt_float_unsafe(const Packet4f& a) { + // Compute approximate reciprocal sqrt. + // Does not correctly handle +/- 0 or +inf + float32x4_t result = vrsqrteq_f32(a); + result = vmulq_f32(vrsqrtsq_f32(vmulq_f32(a, result), result), result); + result = vmulq_f32(vrsqrtsq_f32(vmulq_f32(a, result), result), result); + return result; +} + +EIGEN_STRONG_INLINE Packet2f prsqrt_float_unsafe(const Packet2f& a) { + // Compute approximate reciprocal sqrt. + // Does not correctly handle +/- 0 or +inf + float32x2_t result = vrsqrte_f32(a); + result = vmul_f32(vrsqrts_f32(vmul_f32(a, result), result), result); + result = vmul_f32(vrsqrts_f32(vmul_f32(a, result), result), result); + return result; +} + +template<typename Packet> Packet prsqrt_float_common(const Packet& a) { + const Packet cst_zero = pzero(a); + const Packet cst_inf = pset1<Packet>(NumTraits<float>::infinity()); + Packet return_zero = pcmp_eq(a, cst_inf); + Packet return_inf = pcmp_eq(a, cst_zero); + Packet result = prsqrt_float_unsafe(a); + result = pselect(return_inf, por(cst_inf, a), result); + result = pandnot(result, return_zero); + return result; +} + template<> EIGEN_STRONG_INLINE Packet4f prsqrt(const Packet4f& a) { - // Do Newton iterations for 1/sqrt(x). - return generic_rsqrt_newton_step<Packet4f, /*Steps=*/2>::run(a, vrsqrteq_f32(a)); + return prsqrt_float_common(a); } template<> EIGEN_STRONG_INLINE Packet2f prsqrt(const Packet2f& a) { - // Compute approximate reciprocal sqrt. - return generic_rsqrt_newton_step<Packet2f, /*Steps=*/2>::run(a, vrsqrte_f32(a)); + return prsqrt_float_common(a); +} + +template<> EIGEN_STRONG_INLINE Packet4f preciprocal<Packet4f>(const Packet4f& a) +{ + // Compute approximate reciprocal. + float32x4_t result = vrecpeq_f32(a); + result = vmulq_f32(vrecpsq_f32(a, result), result); + result = vmulq_f32(vrecpsq_f32(a, result), result); + return result; +} + +template<> EIGEN_STRONG_INLINE Packet2f preciprocal<Packet2f>(const Packet2f& a) +{ + // Compute approximate reciprocal. + float32x2_t result = vrecpe_f32(a); + result = vmul_f32(vrecps_f32(a, result), result); + result = vmul_f32(vrecps_f32(a, result), result); + return result; } // Unfortunately vsqrt_f32 is only available for A64. #if EIGEN_ARCH_ARM64 -template<> EIGEN_STRONG_INLINE Packet4f psqrt(const Packet4f& _x){return vsqrtq_f32(_x);} -template<> EIGEN_STRONG_INLINE Packet2f psqrt(const Packet2f& _x){return vsqrt_f32(_x); } +template<> EIGEN_STRONG_INLINE Packet4f psqrt(const Packet4f& a) { return vsqrtq_f32(a); } + +template<> EIGEN_STRONG_INLINE Packet2f psqrt(const Packet2f& a) { return vsqrt_f32(a); } + +template<> EIGEN_STRONG_INLINE Packet4f pdiv(const Packet4f& a, const Packet4f& b) { return vdivq_f32(a, b); } + +template<> EIGEN_STRONG_INLINE Packet2f pdiv(const Packet2f& a, const Packet2f& b) { return vdiv_f32(a, b); } #else -template<> EIGEN_STRONG_INLINE Packet4f psqrt(const Packet4f& a) { - return generic_sqrt_newton_step<Packet4f>::run(a, prsqrt(a)); +template<typename Packet> +EIGEN_STRONG_INLINE Packet psqrt_float_common(const Packet& a) { + const Packet cst_zero = pzero(a); + const Packet cst_inf = pset1<Packet>(NumTraits<float>::infinity()); + + Packet result = pmul(a, prsqrt_float_unsafe(a)); + Packet a_is_zero = pcmp_eq(a, cst_zero); + Packet a_is_inf = pcmp_eq(a, cst_inf); + Packet return_a = por(a_is_zero, a_is_inf); + + result = pselect(return_a, a, result); + return result; } + +template<> EIGEN_STRONG_INLINE Packet4f psqrt(const Packet4f& a) { + return psqrt_float_common(a); +} + template<> EIGEN_STRONG_INLINE Packet2f psqrt(const Packet2f& a) { - return generic_sqrt_newton_step<Packet2f>::run(a, prsqrt(a)); + return psqrt_float_common(a); +} + +template<typename Packet> +EIGEN_STRONG_INLINE Packet pdiv_float_common(const Packet& a, const Packet& b) { + // if b is large, NEON intrinsics will flush preciprocal(b) to zero + // avoid underflow with the following manipulation: + // a / b = f * (a * reciprocal(f * b)) + + const Packet cst_one = pset1<Packet>(1.0f); + const Packet cst_quarter = pset1<Packet>(0.25f); + const Packet cst_thresh = pset1<Packet>(NumTraits<float>::highest() / 4.0f); + + Packet b_will_underflow = pcmp_le(cst_thresh, pabs(b)); + Packet f = pselect(b_will_underflow, cst_quarter, cst_one); + Packet result = pmul(f, pmul(a, preciprocal(pmul(b, f)))); + return result; +} + +template<> EIGEN_STRONG_INLINE Packet4f pdiv<Packet4f>(const Packet4f& a, const Packet4f& b) { + return pdiv_float_common(a, b); +} + +template<> EIGEN_STRONG_INLINE Packet2f pdiv<Packet2f>(const Packet2f& a, const Packet2f& b) { + return pdiv_float_common(a, b); } #endif
diff --git a/Eigen/src/Core/util/XprHelper.h b/Eigen/src/Core/util/XprHelper.h index dc2b194..bb382a7 100644 --- a/Eigen/src/Core/util/XprHelper.h +++ b/Eigen/src/Core/util/XprHelper.h
@@ -809,6 +809,54 @@ } #endif +template<typename XprType> +struct is_block_xpr : std::false_type {}; + +template<typename XprType, int BlockRows, int BlockCols, bool InnerPanel> +struct is_block_xpr<Block<XprType, BlockRows, BlockCols, InnerPanel>> : std::true_type {}; + +template <typename XprType, int BlockRows, int BlockCols, bool InnerPanel> +struct is_block_xpr<const Block<XprType, BlockRows, BlockCols, InnerPanel>> : std::true_type {}; + +// Helper utility for constructing non-recursive block expressions. +template<typename XprType> +struct block_xpr_helper { + using BaseType = XprType; + + // For regular block expressions, simply forward along the InnerPanel argument, + // which is set when calling row/column expressions. + static constexpr bool is_inner_panel(bool inner_panel) { return inner_panel; }; + + // Only enable non-const base function if XprType is not const (otherwise we get a duplicate definition). + template<typename T = XprType, typename EnableIf=std::enable_if_t<!std::is_const<T>::value>> + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE BaseType& base(XprType& xpr) { return xpr; } + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE const BaseType& base(const XprType& xpr) { return xpr; } + static constexpr EIGEN_ALWAYS_INLINE Index row(const XprType& /*xpr*/, Index r) { return r; } + static constexpr EIGEN_ALWAYS_INLINE Index col(const XprType& /*xpr*/, Index c) { return c; } +}; + +template<typename XprType, int BlockRows, int BlockCols, bool InnerPanel> +struct block_xpr_helper<Block<XprType, BlockRows, BlockCols, InnerPanel>> { + using BlockXprType = Block<XprType, BlockRows, BlockCols, InnerPanel>; + // Recursive helper in case of explicit block-of-block expression. + using NestedXprHelper = block_xpr_helper<XprType>; + using BaseType = typename NestedXprHelper::BaseType; + + // For block-of-block expressions, we need to combine the InnerPannel trait + // with that of the block subexpression. + static constexpr bool is_inner_panel(bool inner_panel) { return InnerPanel && inner_panel; } + + // Only enable non-const base function if XprType is not const (otherwise we get a duplicates definition). + template<typename T = XprType, typename EnableIf=std::enable_if_t<!std::is_const<T>::value>> + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE BaseType& base(BlockXprType& xpr) { return NestedXprHelper::base(xpr.nestedExpression()); } + static EIGEN_DEVICE_FUNC EIGEN_ALWAYS_INLINE const BaseType& base(const BlockXprType& xpr) { return NestedXprHelper::base(xpr.nestedExpression()); } + static constexpr EIGEN_ALWAYS_INLINE Index row(const BlockXprType& xpr, Index r) { return xpr.startRow() + NestedXprHelper::row(xpr.nestedExpression(), r); } + static constexpr EIGEN_ALWAYS_INLINE Index col(const BlockXprType& xpr, Index c) { return xpr.startCol() + NestedXprHelper::col(xpr.nestedExpression(), c); } +}; + +template<typename XprType, int BlockRows, int BlockCols, bool InnerPanel> +struct block_xpr_helper<const Block<XprType, BlockRows, BlockCols, InnerPanel>> : block_xpr_helper<Block<XprType, BlockRows, BlockCols, InnerPanel>> {}; + } // end namespace internal
diff --git a/test/array_cwise.cpp b/test/array_cwise.cpp index 49e6672..058b721 100644 --- a/test/array_cwise.cpp +++ b/test/array_cwise.cpp
@@ -47,7 +47,7 @@ const Scalar sqrt2 = Scalar(std::sqrt(2)); const Scalar inf = Eigen::NumTraits<Scalar>::infinity(); const Scalar nan = Eigen::NumTraits<Scalar>::quiet_NaN(); - const Scalar denorm_min = std::numeric_limits<Scalar>::denorm_min(); + const Scalar denorm_min = EIGEN_ARCH_ARM ? zero : std::numeric_limits<Scalar>::denorm_min(); const Scalar min = (std::numeric_limits<Scalar>::min)(); const Scalar max = (std::numeric_limits<Scalar>::max)(); const Scalar max_exp = (static_cast<Scalar>(int(Eigen::NumTraits<Scalar>::max_exponent())) * Scalar(EIGEN_LN2)) / eps; @@ -97,6 +97,12 @@ for (Index j = 0; j < lhs.cols(); ++j) { Scalar e = static_cast<Scalar>(ref(lhs(i,j), rhs(i,j))); Scalar a = actual(i, j); + #if EIGEN_ARCH_ARM + // Work around NEON flush-to-zero mode + // if ref returns denormalized value and Eigen returns 0, then skip the test + int ref_fpclass = std::fpclassify(e); + if (a == Scalar(0) && ref_fpclass == FP_SUBNORMAL) continue; + #endif bool success = (a==e) || ((numext::isfinite)(e) && internal::isApprox(a, e, tol)) || ((numext::isnan)(a) && (numext::isnan)(e)); if ((a == a) && (e == e)) success &= (bool)numext::signbit(e) == (bool)numext::signbit(a); all_pass &= success; @@ -767,7 +773,12 @@ m3(rows, cols), m4 = m1; - m4 = (m4.abs()==Scalar(0)).select(Scalar(1),m4); + // avoid denormalized values so verification doesn't fail on platforms that don't support them + // denormalized behavior is tested elsewhere (unary_op_test, binary_ops_test) + const Scalar min = (std::numeric_limits<Scalar>::min)(); + m1 = (m1.abs()<min).select(Scalar(0),m1); + m2 = (m2.abs()<min).select(Scalar(0),m2); + m4 = (m4.abs()<min).select(Scalar(1),m4); Scalar s1 = internal::random<Scalar>(); @@ -808,6 +819,7 @@ // avoid inf and NaNs so verification doesn't fail m3 = m4.abs(); + VERIFY_IS_APPROX(m3.sqrt(), sqrt(abs(m3))); VERIFY_IS_APPROX(m3.rsqrt(), Scalar(1)/sqrt(abs(m3))); VERIFY_IS_APPROX(rsqrt(m3), Scalar(1)/sqrt(abs(m3)));
diff --git a/test/block.cpp b/test/block.cpp index aba0896..867b769 100644 --- a/test/block.cpp +++ b/test/block.cpp
@@ -306,6 +306,43 @@ compare_using_data_and_stride(m1.col(c1).transpose()); } + +template <typename BaseXpr, typename Xpr = BaseXpr, int Depth = 0> +struct unwind_test_impl { + static void run(Xpr& xpr) { + Index startRow = internal::random<Index>(0, xpr.rows() / 5); + Index startCol = internal::random<Index>(0, xpr.cols() / 6); + Index rows = xpr.rows() / 3; + Index cols = xpr.cols() / 2; + // test equivalence of const expressions + const Block<const Xpr> constNestedBlock(xpr, startRow, startCol, rows, cols); + const Block<const BaseXpr> constUnwoundBlock = constNestedBlock.unwind(); + VERIFY_IS_CWISE_EQUAL(constNestedBlock, constUnwoundBlock); + // modify a random element in each representation and test equivalence of non-const expressions + Block<Xpr> nestedBlock(xpr, startRow, startCol, rows, cols); + Block<BaseXpr> unwoundBlock = nestedBlock.unwind(); + Index r1 = internal::random<Index>(0, rows - 1); + Index c1 = internal::random<Index>(0, cols - 1); + Index r2 = internal::random<Index>(0, rows - 1); + Index c2 = internal::random<Index>(0, cols - 1); + nestedBlock.coeffRef(r1, c1) = internal::random<typename DenseBase<Xpr>::Scalar>(); + unwoundBlock.coeffRef(r2, c2) = internal::random<typename DenseBase<Xpr>::Scalar>(); + VERIFY_IS_CWISE_EQUAL(nestedBlock, unwoundBlock); + unwind_test_impl<BaseXpr, Block<Xpr>, Depth + 1>::run(nestedBlock); + } +}; + +template <typename BaseXpr, typename Xpr> +struct unwind_test_impl<BaseXpr, Xpr, 4> { + static void run(const Xpr&) {} +}; + +template <typename BaseXpr> +void unwind_test(const BaseXpr&) { + BaseXpr xpr = BaseXpr::Random(100, 100); + unwind_test_impl<BaseXpr>::run(xpr); +} + EIGEN_DECLARE_TEST(block) { for(int i = 0; i < g_repeat; i++) { @@ -320,6 +357,7 @@ CALL_SUBTEST_7( block(Matrix<int,Dynamic,Dynamic,RowMajor>(internal::random(2,50), internal::random(2,50))) ); CALL_SUBTEST_8( block(Matrix<float,Dynamic,4>(3, 4)) ); + CALL_SUBTEST_9( unwind_test(MatrixXf())); #ifndef EIGEN_DEFAULT_TO_ROW_MAJOR CALL_SUBTEST_6( data_and_stride(MatrixXf(internal::random(5,50), internal::random(5,50))) );
diff --git a/test/packetmath.cpp b/test/packetmath.cpp index 5dd4cbc..b5b0401 100644 --- a/test/packetmath.cpp +++ b/test/packetmath.cpp
@@ -754,7 +754,7 @@ } // Test for subnormals. - if (Cond && std::numeric_limits<Scalar>::has_denorm == std::denorm_present) { + if (Cond && std::numeric_limits<Scalar>::has_denorm == std::denorm_present && !EIGEN_ARCH_ARM) { for (int scale = 1; scale < 5; ++scale) { // When EIGEN_FAST_MATH is 1 we relax the conditions slightly, and allow the function @@ -912,12 +912,14 @@ CHECK_CWISE1_BYREF1_IF(PacketTraits::HasExp, REF_FREXP, internal::pfrexp); if (PacketTraits::HasExp) { // Check denormals: + #if !EIGEN_ARCH_ARM for (int j=0; j<3; ++j) { data1[0] = Scalar(std::ldexp(1, NumTraits<Scalar>::min_exponent()-j)); CHECK_CWISE1_BYREF1_IF(PacketTraits::HasExp, REF_FREXP, internal::pfrexp); data1[0] = -data1[0]; CHECK_CWISE1_BYREF1_IF(PacketTraits::HasExp, REF_FREXP, internal::pfrexp); } + #endif // zero data1[0] = Scalar(0);
diff --git a/test/sparse_permutations.cpp b/test/sparse_permutations.cpp index 775c560..3e46bfa 100644 --- a/test/sparse_permutations.cpp +++ b/test/sparse_permutations.cpp
@@ -113,25 +113,6 @@ res_d = p.inverse()*mat_d; VERIFY(res.isApprox(res_d) && "inv(p)*mat"); - // test non-plaintype expressions that require additional temporary - const Scalar alpha(2.34); - - res_d = p * (alpha * mat_d); - VERIFY_TEMPORARY_COUNT( res = p * (alpha * mat), 2); - VERIFY( res.isApprox(res_d) && "p*(alpha*mat)" ); - - res_d = (alpha * mat_d) * p; - VERIFY_TEMPORARY_COUNT( res = (alpha * mat) * p, 2); - VERIFY( res.isApprox(res_d) && "(alpha*mat)*p" ); - - res_d = p.inverse() * (alpha * mat_d); - VERIFY_TEMPORARY_COUNT( res = p.inverse() * (alpha * mat), 2); - VERIFY( res.isApprox(res_d) && "inv(p)*(alpha*mat)" ); - - res_d = (alpha * mat_d) * p.inverse(); - VERIFY_TEMPORARY_COUNT( res = (alpha * mat) * p.inverse(), 2); - VERIFY( res.isApprox(res_d) && "(alpha*mat)*inv(p)" ); - // VERIFY( is_sorted( (p * mat * p.inverse()).eval() ));