Merge changes Iee153445,Iee274471 am: 79df15ea88 am: 10f298fc41 am: 7cb5001398t_frc_odp_330442040 t_frc_odp_330442000 t_frc_ase_330444010 android-wear-13.0.0-gpl_r3 android-wear-13.0.0-gpl_r2 android-wear-13.0.0-gpl_r1 android-vts-13.0_r8 android-vts-13.0_r7 android-vts-13.0_r6 android-vts-13.0_r5 android-vts-13.0_r4 android-vts-13.0_r3 android-vts-13.0_r2 android-t-qpr3-beta-3-gpl android-t-qpr3-beta-1-gpl android-t-qpr2-beta-3-gpl android-t-qpr2-beta-2-gpl android-t-qpr1-beta-3-gpl android-t-qpr1-beta-1-gpl android-cts-13.0_r8 android-cts-13.0_r7 android-cts-13.0_r6 android-cts-13.0_r5 android-cts-13.0_r4 android-cts-13.0_r3 android-cts-13.0_r2 android-13.0.0_r83 android-13.0.0_r82 android-13.0.0_r81 android-13.0.0_r80 android-13.0.0_r79 android-13.0.0_r78 android-13.0.0_r77 android-13.0.0_r76 android-13.0.0_r75 android-13.0.0_r74 android-13.0.0_r73 android-13.0.0_r72 android-13.0.0_r71 android-13.0.0_r70 android-13.0.0_r69 android-13.0.0_r68 android-13.0.0_r67 android-13.0.0_r66 android-13.0.0_r65 android-13.0.0_r64 android-13.0.0_r63 android-13.0.0_r62 android-13.0.0_r61 android-13.0.0_r60 android-13.0.0_r59 android-13.0.0_r58 android-13.0.0_r57 android-13.0.0_r56 android-13.0.0_r55 android-13.0.0_r54 android-13.0.0_r53 android-13.0.0_r52 android-13.0.0_r51 android-13.0.0_r50 android-13.0.0_r49 android-13.0.0_r48 android-13.0.0_r47 android-13.0.0_r46 android-13.0.0_r45 android-13.0.0_r44 android-13.0.0_r43 android-13.0.0_r42 android-13.0.0_r41 android-13.0.0_r40 android-13.0.0_r39 android-13.0.0_r38 android-13.0.0_r37 android-13.0.0_r36 android-13.0.0_r35 android-13.0.0_r34 android-13.0.0_r33 android-13.0.0_r32 android-13.0.0_r30 android-13.0.0_r29 android-13.0.0_r28 android-13.0.0_r27 android-13.0.0_r24 android-13.0.0_r23 android-13.0.0_r22 android-13.0.0_r21 android-13.0.0_r20 android-13.0.0_r19 android-13.0.0_r18 android-13.0.0_r17 android-13.0.0_r16 aml_go_odp_330912000 aml_go_ads_330915100 aml_go_ads_330915000 aml_go_ads_330913000 android13-tests-release android13-tests-dev android13-qpr3-s9-release android13-qpr3-s8-release android13-qpr3-s7-release android13-qpr3-s6-release android13-qpr3-s5-release android13-qpr3-s4-release android13-qpr3-s3-release android13-qpr3-s2-release android13-qpr3-s14-release android13-qpr3-s13-release android13-qpr3-s12-release android13-qpr3-s11-release android13-qpr3-s10-release android13-qpr3-s1-release android13-qpr3-release android13-qpr3-c-s8-release android13-qpr3-c-s7-release android13-qpr3-c-s6-release android13-qpr3-c-s5-release android13-qpr3-c-s4-release android13-qpr3-c-s3-release android13-qpr3-c-s2-release android13-qpr3-c-s12-release android13-qpr3-c-s11-release android13-qpr3-c-s10-release android13-qpr3-c-s1-release android13-qpr2-s9-release android13-qpr2-s8-release android13-qpr2-s7-release android13-qpr2-s6-release android13-qpr2-s5-release android13-qpr2-s3-release android13-qpr2-s2-release android13-qpr2-s12-release android13-qpr2-s11-release android13-qpr2-s10-release android13-qpr2-s1-release android13-qpr2-release android13-qpr2-b-s1-release android13-qpr1-s8-release android13-qpr1-s7-release android13-qpr1-s6-release android13-qpr1-s5-release android13-qpr1-s4-release android13-qpr1-s3-release android13-qpr1-s2-release android13-qpr1-s1-release android13-qpr1-release android13-mainline-go-adservices-release android13-frc-odp-release android13-dev android13-d4-s2-release android13-d4-s1-release android13-d4-release android13-d3-s1-release android13-d2-release android-wear-13.0.0-gpl_r1

Original change: https://android-review.googlesource.com/c/platform/external/eigen/+/1999079 Change-Id: I4c76dc5ddc7fb0ae9fc42436f28bd8bf9de50a97
author: Yi Kong <yikong@google.com> 2022-02-25 16:41:05 +0000
committer: Automerger Merge Worker <android-build-automerger-merge-worker@system.gserviceaccount.com> 2022-02-25 16:41:05 +0000
commit: bc0f5df265caa21a2120c22453655a7fcc941991 (patch)
tree: fb979fb4cf4f8052c8cc66b1ec9516d91fcd859b /unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h
parent: 8fd413e275f78a4c240f1442ce5cf77c73a20a55 (diff)
parent: 7cb50013986f04dce5fac87bebf319bb8db37a36 (diff)
download: eigen-4af9b4d40a11c046f8f762da00edd6f02efb18f2.tar.gz
1 files changed, 52 insertions, 34 deletions
diff --git a/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h b/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h
index 08eb5595a..4a1a0687c 100644
--- a/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h
+++ b/unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h
@@ -10,10 +10,6 @@
 #ifndef EIGEN_CXX11_TENSOR_TENSOR_FFT_H
 #define EIGEN_CXX11_TENSOR_TENSOR_FFT_H
 
-// This code requires the ability to initialize arrays of constant
-// values directly inside a class.
-#if __cplusplus >= 201103L || EIGEN_COMP_MSVC >= 1900
-
 namespace Eigen {
 
 /** \class TensorFFT
@@ -71,6 +67,7 @@ struct traits<TensorFFTOp<FFT, XprType, FFTResultType, FFTDir> > : public traits
   typedef typename remove_reference<Nested>::type _Nested;
   static const int NumDimensions = XprTraits::NumDimensions;
   static const int Layout = XprTraits::Layout;
+  typedef typename traits<XprType>::PointerType PointerType;
 };
 
 template <typename FFT, typename XprType, int FFTResultType, int FFTDirection>
@@ -130,17 +127,24 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
   typedef OutputScalar CoeffReturnType;
   typedef typename PacketType<OutputScalar, Device>::type PacketReturnType;
   static const int PacketSize = internal::unpacket_traits<PacketReturnType>::size;
+    typedef StorageMemory<CoeffReturnType, Device> Storage;
+  typedef typename Storage::Type EvaluatorPointerType;
 
   enum {
     IsAligned = false,
     PacketAccess = true,
     BlockAccess = false,
+    PreferBlockAccess = false,
     Layout = TensorEvaluator<ArgType, Device>::Layout,
     CoordAccess = false,
     RawAccess = false
   };
 
-  EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) : m_fft(op.fft()), m_impl(op.expression(), device), m_data(NULL), m_device(device) {
+  //===- Tensor block evaluation strategy (see TensorBlock.h) -------------===//
+  typedef internal::TensorBlockNotImplemented TensorBlock;
+  //===--------------------------------------------------------------------===//
+
+  EIGEN_STRONG_INLINE TensorEvaluator(const XprType& op, const Device& device) : m_fft(op.fft()), m_impl(op.expression(), device), m_data(NULL), m_device(device) {
     const typename TensorEvaluator<ArgType, Device>::Dimensions& input_dims = m_impl.dimensions();
     for (int i = 0; i < NumDims; ++i) {
       eigen_assert(input_dims[i] > 0);
@@ -165,19 +169,19 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
     return m_dimensions;
   }
 
-  EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(OutputScalar* data) {
+  EIGEN_STRONG_INLINE bool evalSubExprsIfNeeded(EvaluatorPointerType data) {
     m_impl.evalSubExprsIfNeeded(NULL);
     if (data) {
       evalToBuf(data);
       return false;
     } else {
-      m_data = (CoeffReturnType*)m_device.allocate(sizeof(CoeffReturnType) * m_size);
+      m_data = (EvaluatorPointerType)m_device.get((CoeffReturnType*)(m_device.allocate_temp(sizeof(CoeffReturnType) * m_size)));
       evalToBuf(m_data);
       return true;
     }
   }
 
-  EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void cleanup() {
+  EIGEN_STRONG_INLINE void cleanup() {
     if (m_data) {
       m_device.deallocate(m_data);
       m_data = NULL;
@@ -200,11 +204,16 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
     return TensorOpCost(sizeof(CoeffReturnType), 0, 0, vectorized, PacketSize);
   }
 
-  EIGEN_DEVICE_FUNC Scalar* data() const { return m_data; }
-
+  EIGEN_DEVICE_FUNC EvaluatorPointerType data() const { return m_data; }
+#ifdef EIGEN_USE_SYCL
+  // binding placeholder accessors to a command group handler for SYCL
+  EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void bind(cl::sycl::handler &cgh) const {
+    m_data.bind(cgh);
+  }
+#endif
 
  private:
-  EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalToBuf(OutputScalar* data) {
+  EIGEN_DEVICE_FUNC EIGEN_STRONG_INLINE void evalToBuf(EvaluatorPointerType data) {
     const bool write_to_out = internal::is_same<OutputScalar, ComplexScalar>::value;
     ComplexScalar* buf = write_to_out ? (ComplexScalar*)data : (ComplexScalar*)m_device.allocate(sizeof(ComplexScalar) * m_size);
 
@@ -230,20 +239,32 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
         //   t_n = exp(sqrt(-1) * pi * n^2 / line_len)
         // for n = 0, 1,..., line_len-1.
         // For n > 2 we use the recurrence t_n = t_{n-1}^2 / t_{n-2} * t_1^2
-        pos_j_base_powered[0] = ComplexScalar(1, 0);
-        if (line_len > 1) {
-          const RealScalar pi_over_len(EIGEN_PI / line_len);
-          const ComplexScalar pos_j_base = ComplexScalar(
-	       std::cos(pi_over_len), std::sin(pi_over_len));
-          pos_j_base_powered[1] = pos_j_base;
-          if (line_len > 2) {
-            const ComplexScalar pos_j_base_sq = pos_j_base * pos_j_base;
-            for (int j = 2; j < line_len + 1; ++j) {
-              pos_j_base_powered[j] = pos_j_base_powered[j - 1] *
-                                      pos_j_base_powered[j - 1] /
-                                      pos_j_base_powered[j - 2] * pos_j_base_sq;
-            }
-          }
+
+        // The recurrence is correct in exact arithmetic, but causes
+        // numerical issues for large transforms, especially in
+        // single-precision floating point.
+        //
+        // pos_j_base_powered[0] = ComplexScalar(1, 0);
+        // if (line_len > 1) {
+        //   const ComplexScalar pos_j_base = ComplexScalar(
+        //       numext::cos(M_PI / line_len), numext::sin(M_PI / line_len));
+        //   pos_j_base_powered[1] = pos_j_base;
+        //   if (line_len > 2) {
+        //     const ComplexScalar pos_j_base_sq = pos_j_base * pos_j_base;
+        //     for (int i = 2; i < line_len + 1; ++i) {
+        //       pos_j_base_powered[i] = pos_j_base_powered[i - 1] *
+        //           pos_j_base_powered[i - 1] /
+        //           pos_j_base_powered[i - 2] *
+        //           pos_j_base_sq;
+        //     }
+        //   }
+        // }
+        // TODO(rmlarsen): Find a way to use Eigen's vectorized sin
+        // and cosine functions here.
+        for (int j = 0; j < line_len + 1; ++j) {
+          double arg = ((EIGEN_PI * j) * j) / line_len;
+          std::complex<double> tmp(numext::cos(arg), numext::sin(arg));
+          pos_j_base_powered[j] = static_cast<ComplexScalar>(tmp);
         }
       }
 
@@ -253,7 +274,7 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
         // get data into line_buf
         const Index stride = m_strides[dim];
         if (stride == 1) {
-          memcpy(line_buf, &buf[base_offset], line_len*sizeof(ComplexScalar));
+          m_device.memcpy(line_buf, &buf[base_offset], line_len*sizeof(ComplexScalar));
         } else {
           Index offset = base_offset;
           for (int j = 0; j < line_len; ++j, offset += stride) {
@@ -261,7 +282,7 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
           }
         }
 
-        // processs the line
+        // process the line
         if (is_power_of_two) {
           processDataLineCooleyTukey(line_buf, line_len, log_len);
         }
@@ -271,7 +292,7 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
 
         // write back
         if (FFTDir == FFT_FORWARD && stride == 1) {
-          memcpy(&buf[base_offset], line_buf, line_len*sizeof(ComplexScalar));
+          m_device.memcpy(&buf[base_offset], line_buf, line_len*sizeof(ComplexScalar));
         } else {
           Index offset = base_offset;
           const ComplexScalar div_factor =  ComplexScalar(1.0 / line_len, 0);
@@ -562,12 +583,12 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
 
  protected:
   Index m_size;
-  const FFT& m_fft;
+  const FFT EIGEN_DEVICE_REF m_fft;
   Dimensions m_dimensions;
   array<Index, NumDims> m_strides;
   TensorEvaluator<ArgType, Device> m_impl;
-  CoeffReturnType* m_data;
-  const Device& m_device;
+  EvaluatorPointerType m_data;
+  const Device EIGEN_DEVICE_REF m_device;
 
   // This will support a maximum FFT size of 2^32 for each dimension
   // m_sin_PI_div_n_LUT[i] = (-2) * std::sin(M_PI / std::pow(2,i)) ^ 2;
@@ -645,7 +666,4 @@ struct TensorEvaluator<const TensorFFTOp<FFT, ArgType, FFTResultType, FFTDir>, D
 
 }  // end namespace Eigen
 
-#endif  // EIGEN_HAS_CONSTEXPR
-
-
 #endif  // EIGEN_CXX11_TENSOR_TENSOR_FFT_H
author	Yi Kong <yikong@google.com>	2022-02-25 16:41:05 +0000
committer	Automerger Merge Worker <android-build-automerger-merge-worker@system.gserviceaccount.com>	2022-02-25 16:41:05 +0000
commit	bc0f5df265caa21a2120c22453655a7fcc941991 (patch)
tree	fb979fb4cf4f8052c8cc66b1ec9516d91fcd859b /unsupported/Eigen/CXX11/src/Tensor/TensorFFT.h
parent	8fd413e275f78a4c240f1442ce5cf77c73a20a55 (diff)
parent	7cb50013986f04dce5fac87bebf319bb8db37a36 (diff)
download	eigen-4af9b4d40a11c046f8f762da00edd6f02efb18f2.tar.gz