Merge branch 'master' of https://github.com/paboyle/Grid

Conflicts: lib/Make.inc
2025-12-19 12:14:29 +00:00 · 2015-06-09 10:27:10 +01:00
parent 506dfd1517 6b8fe04054
commit d8ddec86f7
19 changed files with 687 additions and 31 deletions
--- a/lib/GridConfig.h
+++ b/lib/GridConfig.h
@@ -1,15 +1,18 @@
 /* lib/GridConfig.h.  Generated from GridConfig.h.in by configure.  */
 /* lib/GridConfig.h.in.  Generated from configure.ac by autoheader.  */

-/* AVX */
+/* AVX Intrinsics */
 /* #undef AVX1 */

-/* AVX2 */
+/* AVX2 Intrinsics */
 /* #undef AVX2 */

-/* AVX512 */
+/* AVX512 Intrinsics for Knights Corner */
 /* #undef AVX512 */

+/* EMPTY_SIMD only for DEBUGGING */
+/* #undef EMPTY_SIMD */
+
 /* GRID_COMMS_MPI */
 /* #undef GRID_COMMS_MPI */

@@ -111,6 +114,9 @@
 /* Define to 1 if you have the <unistd.h> header file. */
 #define HAVE_UNISTD_H 1

+/* NEON ARMv7 Experimental support */
+/* #undef NEONv7 */
+
 /* Name of package */
 #define PACKAGE "grid"

@@ -132,7 +138,7 @@
 /* Define to the version of this package. */
 #define PACKAGE_VERSION "1.0"

-/* SSE4 */
+/* SSE4 Intrinsics */
 #define SSE4 1

 /* Define to 1 if you have the ANSI C header files. */
--- a/lib/GridConfig.h.in
+++ b/lib/GridConfig.h.in
@@ -1,14 +1,17 @@
 /* lib/GridConfig.h.in.  Generated from configure.ac by autoheader.  */

-/* AVX */
+/* AVX Intrinsics */
 #undef AVX1

-/* AVX2 */
+/* AVX2 Intrinsics */
 #undef AVX2

-/* AVX512 */
+/* AVX512 Intrinsics for Knights Corner */
 #undef AVX512

+/* EMPTY_SIMD only for DEBUGGING */
+#undef EMPTY_SIMD
+
 /* GRID_COMMS_MPI */
 #undef GRID_COMMS_MPI

@@ -110,6 +113,9 @@
 /* Define to 1 if you have the <unistd.h> header file. */
 #undef HAVE_UNISTD_H

+/* NEON ARMv7 Experimental support */
+#undef NEONv7
+
 /* Name of package */
 #undef PACKAGE

@@ -131,7 +137,7 @@
 /* Define to the version of this package. */
 #undef PACKAGE_VERSION

-/* SSE4 */
+/* SSE4 Intrinsics */
 #undef SSE4

 /* Define to 1 if you have the ANSI C header files. */
--- a/lib/Make.inc
+++ b/lib/Make.inc
@@ -1,4 +1,4 @@

-HFILES=./algorithms/approx/bigfloat.h ./algorithms/approx/bigfloat_double.h ./algorithms/approx/Chebyshev.h ./algorithms/approx/MultiShiftFunction.h ./algorithms/approx/Remez.h ./algorithms/approx/Zolotarev.h ./algorithms/CoarsenedMatrix.h ./algorithms/iterative/ConjugateGradient.h ./algorithms/iterative/ConjugateGradientMultiShift.h ./algorithms/iterative/ConjugateResidual.h ./algorithms/iterative/NormalEquations.h ./algorithms/iterative/SchurRedBlack.h ./algorithms/LinearOperator.h ./algorithms/SparseMatrix.h ./Algorithms.h ./AlignedAllocator.h ./cartesian/Cartesian_base.h ./cartesian/Cartesian_full.h ./cartesian/Cartesian_red_black.h ./Cartesian.h ./communicator/Communicator_base.h ./Communicator.h ./cshift/Cshift_common.h ./cshift/Cshift_mpi.h ./cshift/Cshift_none.h ./Cshift.h ./Grid.h ./GridConfig.h ./lattice/Lattice_arith.h ./lattice/Lattice_base.h ./lattice/Lattice_comparison.h ./lattice/Lattice_comparison_utils.h ./lattice/Lattice_conformable.h ./lattice/Lattice_coordinate.h ./lattice/Lattice_ET.h ./lattice/Lattice_local.h ./lattice/Lattice_overload.h ./lattice/Lattice_peekpoke.h ./lattice/Lattice_reality.h ./lattice/Lattice_reduction.h ./lattice/Lattice_rng.h ./lattice/Lattice_trace.h ./lattice/Lattice_transfer.h ./lattice/Lattice_transpose.h ./lattice/Lattice_unary.h ./lattice/Lattice_where.h ./Lattice.h ./parallelIO/NerscIO.h ./qcd/action/Actions.h ./qcd/action/DiffAction.h ./qcd/action/fermion/CayleyFermion5D.h ./qcd/action/fermion/ContinuedFractionFermion5D.h ./qcd/action/fermion/DomainWallFermion.h ./qcd/action/fermion/FermionOperator.h ./qcd/action/fermion/g5HermitianLinop.h ./qcd/action/fermion/MobiusFermion.h ./qcd/action/fermion/MobiusZolotarevFermion.h ./qcd/action/fermion/OverlapWilsonCayleyTanhFermion.h ./qcd/action/fermion/OverlapWilsonCayleyZolotarevFermion.h ./qcd/action/fermion/OverlapWilsonContfracTanhFermion.h ./qcd/action/fermion/OverlapWilsonContfracZolotarevFermion.h ./qcd/action/fermion/OverlapWilsonPartialFractionTanhFermion.h ./qcd/action/fermion/OverlapWilsonPartialFractionZolotarevFermion.h ./qcd/action/fermion/PartialFractionFermion5D.h ./qcd/action/fermion/ScaledShamirFermion.h ./qcd/action/fermion/ShamirZolotarevFermion.h ./qcd/action/fermion/WilsonCompressor.h ./qcd/action/fermion/WilsonFermion.h ./qcd/action/fermion/WilsonFermion5D.h ./qcd/action/fermion/WilsonKernels.h ./qcd/action/gauge/GaugeActionBase.h ./qcd/action/gauge/WilsonGaugeAction.h ./qcd/QCD.h ./qcd/spin/Dirac.h ./qcd/spin/TwoSpinor.h ./qcd/utils/CovariantCshift.h ./qcd/utils/LinalgUtils.h ./qcd/utils/SpaceTimeGrid.h ./qcd/utils/WilsonLoops.h ./simd/Grid_avx.h ./simd/Grid_avx512.h ./simd/Grid_qpx.h ./simd/Grid_sse4.h ./simd/Grid_vector_types.h ./simd/Grid_vector_unops.h ./simd/Old/Grid_vComplexD.h ./simd/Old/Grid_vComplexF.h ./simd/Old/Grid_vInteger.h ./simd/Old/Grid_vRealD.h ./simd/Old/Grid_vRealF.h ./Simd.h ./stencil/Lebesgue.h ./Stencil.h ./tensors/Tensor_arith.h ./tensors/Tensor_arith_add.h ./tensors/Tensor_arith_mac.h ./tensors/Tensor_arith_mul.h ./tensors/Tensor_arith_scalar.h ./tensors/Tensor_arith_sub.h ./tensors/Tensor_class.h ./tensors/Tensor_extract_merge.h ./tensors/Tensor_inner.h ./tensors/Tensor_outer.h ./tensors/Tensor_peek.h ./tensors/Tensor_poke.h ./tensors/Tensor_reality.h ./tensors/Tensor_Ta.h ./tensors/Tensor_trace.h ./tensors/Tensor_traits.h ./tensors/Tensor_transpose.h ./tensors/Tensor_unary.h ./Tensors.h ./Threads.h
+HFILES=./algorithms/approx/bigfloat.h ./algorithms/approx/bigfloat_double.h ./algorithms/approx/Chebyshev.h ./algorithms/approx/MultiShiftFunction.h ./algorithms/approx/Remez.h ./algorithms/approx/Zolotarev.h ./algorithms/CoarsenedMatrix.h ./algorithms/iterative/ConjugateGradient.h ./algorithms/iterative/ConjugateGradientMultiShift.h ./algorithms/iterative/ConjugateResidual.h ./algorithms/iterative/NormalEquations.h ./algorithms/iterative/SchurRedBlack.h ./algorithms/LinearOperator.h ./algorithms/SparseMatrix.h ./Algorithms.h ./AlignedAllocator.h ./cartesian/Cartesian_base.h ./cartesian/Cartesian_full.h ./cartesian/Cartesian_red_black.h ./Cartesian.h ./communicator/Communicator_base.h ./Communicator.h ./cshift/Cshift_common.h ./cshift/Cshift_mpi.h ./cshift/Cshift_none.h ./Cshift.h ./Grid.h ./GridConfig.h ./lattice/Lattice_arith.h ./lattice/Lattice_base.h ./lattice/Lattice_comparison.h ./lattice/Lattice_comparison_utils.h ./lattice/Lattice_conformable.h ./lattice/Lattice_coordinate.h ./lattice/Lattice_ET.h ./lattice/Lattice_local.h ./lattice/Lattice_overload.h ./lattice/Lattice_peekpoke.h ./lattice/Lattice_reality.h ./lattice/Lattice_reduction.h ./lattice/Lattice_rng.h ./lattice/Lattice_trace.h ./lattice/Lattice_transfer.h ./lattice/Lattice_transpose.h ./lattice/Lattice_unary.h ./lattice/Lattice_where.h ./Lattice.h ./parallelIO/NerscIO.h ./qcd/action/Actions.h ./qcd/action/DiffAction.h ./qcd/action/fermion/CayleyFermion5D.h ./qcd/action/fermion/ContinuedFractionFermion5D.h ./qcd/action/fermion/DomainWallFermion.h ./qcd/action/fermion/FermionOperator.h ./qcd/action/fermion/g5HermitianLinop.h ./qcd/action/fermion/MobiusFermion.h ./qcd/action/fermion/MobiusZolotarevFermion.h ./qcd/action/fermion/OverlapWilsonCayleyTanhFermion.h ./qcd/action/fermion/OverlapWilsonCayleyZolotarevFermion.h ./qcd/action/fermion/OverlapWilsonContfracTanhFermion.h ./qcd/action/fermion/OverlapWilsonContfracZolotarevFermion.h ./qcd/action/fermion/OverlapWilsonPartialFractionTanhFermion.h ./qcd/action/fermion/OverlapWilsonPartialFractionZolotarevFermion.h ./qcd/action/fermion/PartialFractionFermion5D.h ./qcd/action/fermion/ScaledShamirFermion.h ./qcd/action/fermion/ShamirZolotarevFermion.h ./qcd/action/fermion/WilsonCompressor.h ./qcd/action/fermion/WilsonFermion.h ./qcd/action/fermion/WilsonFermion5D.h ./qcd/action/fermion/WilsonKernels.h ./qcd/action/gauge/GaugeActionBase.h ./qcd/action/gauge/WilsonGaugeAction.h ./qcd/QCD.h ./qcd/spin/Dirac.h ./qcd/spin/TwoSpinor.h ./qcd/utils/CovariantCshift.h ./qcd/utils/LinalgUtils.h ./qcd/utils/SpaceTimeGrid.h ./qcd/utils/WilsonLoops.h ./simd/Grid_avx.h ./simd/Grid_avx512.h ./simd/Grid_empty.h ./simd/Grid_neon.h ./simd/Grid_qpx.h ./simd/Grid_sse4.h ./simd/Grid_vector_types.h ./simd/Grid_vector_unops.h ./simd/Old/Grid_vComplexD.h ./simd/Old/Grid_vComplexF.h ./simd/Old/Grid_vInteger.h ./simd/Old/Grid_vRealD.h ./simd/Old/Grid_vRealF.h ./Simd.h ./stencil/Lebesgue.h ./Stencil.h ./tensors/Tensor_arith.h ./tensors/Tensor_arith_add.h ./tensors/Tensor_arith_mac.h ./tensors/Tensor_arith_mul.h ./tensors/Tensor_arith_scalar.h ./tensors/Tensor_arith_sub.h ./tensors/Tensor_class.h ./tensors/Tensor_extract_merge.h ./tensors/Tensor_inner.h ./tensors/Tensor_outer.h ./tensors/Tensor_peek.h ./tensors/Tensor_poke.h ./tensors/Tensor_reality.h ./tensors/Tensor_Ta.h ./tensors/Tensor_trace.h ./tensors/Tensor_traits.h ./tensors/Tensor_transpose.h ./tensors/Tensor_unary.h ./Tensors.h ./Threads.h

 CCFILES=./algorithms/approx/MultiShiftFunction.cc ./algorithms/approx/Remez.cc ./algorithms/approx/Zolotarev.cc ./GridInit.cc ./qcd/action/fermion/CayleyFermion5D.cc ./qcd/action/fermion/ContinuedFractionFermion5D.cc ./qcd/action/fermion/PartialFractionFermion5D.cc ./qcd/action/fermion/WilsonFermion.cc ./qcd/action/fermion/WilsonFermion5D.cc ./qcd/action/fermion/WilsonKernels.cc ./qcd/action/fermion/WilsonKernelsHand.cc ./qcd/spin/Dirac.cc ./qcd/utils/SpaceTimeGrid.cc ./stencil/Lebesgue.cc ./stencil/Stencil_common.cc
--- a/lib/algorithms/approx/.dirstamp
+++ b/lib/algorithms/approx/.dirstamp
--- a/lib/communicator/.dirstamp
+++ b/lib/communicator/.dirstamp
--- a/lib/qcd/spin/.dirstamp
+++ b/lib/qcd/spin/.dirstamp
--- a/lib/qcd/utils/.dirstamp
+++ b/lib/qcd/utils/.dirstamp
--- a/lib/simd/Grid_avx.h
+++ b/lib/simd/Grid_avx.h
@@ -4,7 +4,7 @@

  Using intrinsics
 */
-// Time-stamp: <2015-05-29 14:13:30 neo>
+// Time-stamp: <2015-06-09 14:26:59 neo>
 //----------------------------------------------------------------------

 #include <immintrin.h>
@@ -383,6 +383,12 @@ namespace Grid {
      _mm_prefetch(ptr+i+512,_MM_HINT_T0);
    }
  }
+  inline void prefetch_HINT_T0(const char *ptr){
+    _mm_prefetch(ptr,_MM_HINT_T0);
+  }
+
+
+
  template < typename VectorSIMD > 
    inline void Gpermute(VectorSIMD &y,const VectorSIMD &b, int perm ) {
    Optimization::permute(y.v,b.v,perm);
--- a/lib/simd/Grid_avx512.h
+++ b/lib/simd/Grid_avx512.h
@@ -4,7 +4,7 @@

  Using intrinsics
 */
-// Time-stamp: <2015-05-27 12:08:50 neo>
+// Time-stamp: <2015-06-09 14:27:28 neo>
 //----------------------------------------------------------------------

 #include <immintrin.h>
@@ -309,6 +309,12 @@ namespace Grid {
      _mm_prefetch(ptr+i+512,_MM_HINT_T0);
    }
  }
+  inline void prefetch_HINT_T0(const char *ptr){
+    _mm_prefetch(ptr,_MM_HINT_T0);
+  }
+
+
+

  // Gpermute utilities consider coalescing into 1 Gpermute
  template < typename VectorSIMD > 
--- a/lib/simd/Grid_empty.h
+++ b/lib/simd/Grid_empty.h
@@ -0,0 +1,289 @@
+//----------------------------------------------------------------------
+/*! @file Grid_sse4.h
+  @brief Empty Optimization libraries for debugging
+
+  Using intrinsics
+*/
+// Time-stamp: <2015-06-09 14:28:02 neo>
+//----------------------------------------------------------------------
+
+namespace Optimization {
+
+  template<class vtype>
+  union uconv {
+    float f;
+    vtype v;
+  };
+
+  union u128f {
+    float v;
+    float f[4];
+  };
+  union u128d {
+    double v;
+    double f[2];
+  };
+  
+  struct Vsplat{
+    //Complex float
+    inline float operator()(float a, float b){
+      return 0;
+    }
+    // Real float
+    inline float operator()(float a){
+      return 0;
+    }
+    //Complex double
+    inline double operator()(double a, double b){
+      return 0;
+    }
+    //Real double
+    inline double operator()(double a){
+      return 0;
+    }
+    //Integer
+    inline int operator()(Integer a){
+      return 0;
+    }
+  };
+
+  struct Vstore{
+    //Float 
+    inline void operator()(float a, float* F){
+      
+    }
+    //Double
+    inline void operator()(double a, double* D){
+     
+    }
+    //Integer
+    inline void operator()(int a, Integer* I){
+      
+    }
+
+  };
+
+  struct Vstream{
+    //Float
+    inline void operator()(float * a, float b){
+     
+    }
+    //Double
+    inline void operator()(double * a, double b){
+     
+    }
+
+
+  };
+
+  struct Vset{
+    // Complex float 
+    inline float operator()(Grid::ComplexF *a){
+      return 0;
+    }
+    // Complex double 
+    inline double operator()(Grid::ComplexD *a){
+      return 0;
+    }
+    // Real float 
+    inline float operator()(float *a){
+      return  0;
+    }
+    // Real double
+    inline double operator()(double *a){
+      return 0;
+    }
+    // Integer
+    inline int operator()(Integer *a){
+      return 0;
+    }
+
+
+  };
+
+  template <typename Out_type, typename In_type>
+  struct Reduce{
+    //Need templated class to overload output type
+    //General form must generate error if compiled
+    inline Out_type operator()(In_type in){
+      printf("Error, using wrong Reduce function\n");
+      exit(1);
+      return 0;
+    }
+  };
+
+  /////////////////////////////////////////////////////
+  // Arithmetic operations
+  /////////////////////////////////////////////////////
+  struct Sum{
+    //Complex/Real float
+    inline float operator()(float a, float b){
+      return 0;
+    }
+    //Complex/Real double
+    inline double operator()(double a, double b){
+      return 0;
+    }
+    //Integer
+    inline int operator()(int a, int b){
+      return 0;
+    }
+  };
+
+  struct Sub{
+    //Complex/Real float
+    inline float operator()(float a, float b){
+      return 0;
+    }
+    //Complex/Real double
+    inline double operator()(double a, double b){
+      return 0;
+    }
+    //Integer
+    inline int operator()(int a, int b){
+      return 0;
+    }
+  };
+
+  struct MultComplex{
+    // Complex float
+    inline float operator()(float a, float b){
+      return 0;
+    }
+    // Complex double
+    inline double operator()(double a, double b){
+      return 0;
+    }
+  };
+
+  struct Mult{
+    // Real float
+    inline float operator()(float a, float b){
+      return 0;
+    }
+    // Real double
+    inline double operator()(double a, double b){
+      return 0;
+    }
+    // Integer
+    inline int operator()(int a, int b){
+      return 0;
+    }
+  };
+
+  struct Conj{
+    // Complex single
+    inline float operator()(float in){
+      return 0;
+    }
+    // Complex double
+    inline double operator()(double in){
+      return 0;
+    }
+    // do not define for integer input
+  };
+
+  struct TimesMinusI{
+    //Complex single
+    inline float operator()(float in, float ret){
+      return 0;
+    }
+    //Complex double
+    inline double operator()(double in, double ret){
+      return 0;
+    }
+
+
+  };
+
+  struct TimesI{
+    //Complex single
+    inline float operator()(float in, float ret){
+      return 0;
+    }
+    //Complex double
+    inline double operator()(double in, double ret){
+      return 0;
+    }
+  };
+
+  //////////////////////////////////////////////
+  // Some Template specialization
+  template < typename vtype > 
+    void permute(vtype &a, vtype b, int perm) {
+   }; 
+
+  //Complex float Reduce
+  template<>
+  inline Grid::ComplexF Reduce<Grid::ComplexF, float>::operator()(float in){
+    return 0;
+  }
+  //Real float Reduce
+  template<>
+  inline Grid::RealF Reduce<Grid::RealF, float>::operator()(float in){
+    return 0;
+  }
+  
+  
+  //Complex double Reduce
+  template<>
+  inline Grid::ComplexD Reduce<Grid::ComplexD, double>::operator()(double in){
+    return 0;
+  }
+  
+  //Real double Reduce
+  template<>
+  inline Grid::RealD Reduce<Grid::RealD, double>::operator()(double in){
+    return 0;
+  }
+
+  //Integer Reduce
+  template<>
+  inline Integer Reduce<Integer, int>::operator()(int in){
+    // FIXME unimplemented
+   printf("Reduce : Missing integer implementation -> FIX\n");
+    assert(0);
+  }
+}
+
+//////////////////////////////////////////////////////////////////////////////////////
+// Here assign types 
+namespace Grid {
+
+  typedef float SIMD_Ftype;  // Single precision type
+  typedef double SIMD_Dtype; // Double precision type
+  typedef int SIMD_Itype; // Integer type
+
+  // prefetch utilities
+  inline void v_prefetch0(int size, const char *ptr){};
+  inline void prefetch_HINT_T0(const char *ptr){};
+
+
+
+  // Gpermute function
+  template < typename VectorSIMD > 
+    inline void Gpermute(VectorSIMD &y,const VectorSIMD &b, int perm ) {
+    Optimization::permute(y.v,b.v,perm);
+  }
+
+
+  // Function name aliases
+  typedef Optimization::Vsplat   VsplatSIMD;
+  typedef Optimization::Vstore   VstoreSIMD;
+  typedef Optimization::Vset     VsetSIMD;
+  typedef Optimization::Vstream  VstreamSIMD;
+  template <typename S, typename T> using ReduceSIMD = Optimization::Reduce<S,T>;
+
+ 
+
+
+  // Arithmetic operations
+  typedef Optimization::Sum         SumSIMD;
+  typedef Optimization::Sub         SubSIMD;
+  typedef Optimization::Mult        MultSIMD;
+  typedef Optimization::MultComplex MultComplexSIMD;
+  typedef Optimization::Conj        ConjSIMD;
+  typedef Optimization::TimesMinusI TimesMinusISIMD;
+  typedef Optimization::TimesI      TimesISIMD;
+
+}
--- a/lib/simd/Grid_neon.h
+++ b/lib/simd/Grid_neon.h
@@ -0,0 +1,308 @@
+//----------------------------------------------------------------------
+/*! @file Grid_sse4.h
+  @brief Optimization libraries for NEON (ARM) instructions set ARMv7
+
+  Experimental - Using intrinsics - DEVELOPING! 
+*/
+// Time-stamp: <2015-06-09 15:25:40 neo>
+//----------------------------------------------------------------------
+
+#include <arm_neon.h>
+
+namespace Optimization {
+
+  template<class vtype>
+  union uconv {
+    float32x4_t f;
+    vtype v;
+  };
+
+  union u128f {
+    float32x4_t v;
+    float f[4];
+  };
+  union u128d {
+    float32x4_t v;
+    float f[4];
+  };
+  
+  struct Vsplat{
+    //Complex float
+    inline float32x4_t operator()(float a, float b){
+      float32x4_t foo;
+      return foo;
+    }
+    // Real float
+    inline float32x4_t operator()(float a){
+      float32x4_t foo;
+      return foo;
+    }
+    //Complex double
+    inline float32x4_t operator()(double a, double b){
+      float32x4_t foo;
+      return foo;
+    }
+    //Real double
+    inline float32x4_t operator()(double a){
+      float32x4_t foo;
+      return foo;
+    }
+    //Integer
+    inline uint32x4_t operator()(Integer a){
+      uint32x4_t foo;
+      return foo;
+    }
+  };
+
+  struct Vstore{
+    //Float 
+    inline void operator()(float32x4_t a, float* F){
+      
+    }
+    //Double
+    inline void operator()(float32x4_t a, double* D){
+      
+    }
+    //Integer
+    inline void operator()(uint32x4_t a, Integer* I){
+     
+    }
+
+  };
+
+  struct Vstream{
+    //Float
+    inline void operator()(float * a, float32x4_t b){
+    
+    }
+    //Double
+    inline void operator()(double * a, float32x4_t b){
+  
+    }
+
+
+  };
+
+  struct Vset{
+    // Complex float 
+    inline float32x4_t operator()(Grid::ComplexF *a){
+      float32x4_t foo;
+      return foo;
+    }
+    // Complex double 
+    inline float32x4_t operator()(Grid::ComplexD *a){
+      float32x4_t foo;
+      return foo;
+    }
+    // Real float 
+    inline float32x4_t operator()(float *a){
+      float32x4_t foo;
+      return foo;
+    }
+    // Real double
+    inline float32x4_t operator()(double *a){
+      float32x4_t foo;
+      return foo;
+    }
+    // Integer
+    inline uint32x4_t operator()(Integer *a){
+      uint32x4_t foo;
+      return foo;
+    }
+
+
+  };
+
+  template <typename Out_type, typename In_type>
+  struct Reduce{
+    //Need templated class to overload output type
+    //General form must generate error if compiled
+    inline Out_type operator()(In_type in){
+      printf("Error, using wrong Reduce function\n");
+      exit(1);
+      return 0;
+    }
+  };
+
+  /////////////////////////////////////////////////////
+  // Arithmetic operations
+  /////////////////////////////////////////////////////
+  struct Sum{
+    //Complex/Real float
+    inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+      float32x4_t foo;
+      return foo;
+    }
+    //Complex/Real double
+    //inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+    //  float32x4_t foo;
+    //  return foo;
+    //}
+    //Integer
+    inline uint32x4_t operator()(uint32x4_t a, uint32x4_t b){
+      uint32x4_t foo;
+      return foo;
+    }
+  };
+
+  struct Sub{
+    //Complex/Real float
+    inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+      float32x4_t foo;
+      return foo;
+    }
+    //Complex/Real double
+    //inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+    //  float32x4_t foo;
+    //  return foo;
+    //}
+    //Integer
+    inline uint32x4_t operator()(uint32x4_t a, uint32x4_t b){
+      uint32x4_t foo;
+      return foo;
+    }
+  };
+
+  struct MultComplex{
+    // Complex float
+    inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+      float32x4_t foo;
+      return foo;
+    }
+    // Complex double
+    //inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+    //  float32x4_t foo;
+    //  return foo;
+    //}
+  };
+
+  struct Mult{
+    // Real float
+    inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+      return a;
+    }
+    // Real double
+    //inline float32x4_t operator()(float32x4_t a, float32x4_t b){
+    //  return 0;
+    //}
+    // Integer
+    inline uint32x4_t operator()(uint32x4_t a, uint32x4_t b){
+      return a;
+    }
+  };
+
+  struct Conj{
+    // Complex single
+    inline float32x4_t operator()(float32x4_t in){
+      return in;
+    }
+    // Complex double
+    //inline float32x4_t operator()(float32x4_t in){
+    // return 0;
+    //}
+    // do not define for integer input
+  };
+
+  struct TimesMinusI{
+    //Complex single
+    inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
+      return in;
+    }
+    //Complex double
+    //inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
+    //  return in;
+    //}
+
+
+  };
+
+  struct TimesI{
+    //Complex single
+    inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
+      return in;
+    }
+    //Complex double
+    //inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
+    //  return 0;
+    //}
+  };
+
+  //////////////////////////////////////////////
+  // Some Template specialization
+  template < typename vtype > 
+    void permute(vtype &a, vtype b, int perm) {
+
+  }; 
+
+  //Complex float Reduce
+  template<>
+  inline Grid::ComplexF Reduce<Grid::ComplexF, float32x4_t>::operator()(float32x4_t in){
+    return 0;
+  }
+  //Real float Reduce
+  template<>
+  inline Grid::RealF Reduce<Grid::RealF, float32x4_t>::operator()(float32x4_t in){
+    return 0;
+  }
+  
+  
+  //Complex double Reduce
+  template<>
+  inline Grid::ComplexD Reduce<Grid::ComplexD, float32x4_t>::operator()(float32x4_t in){
+    return 0;
+  }
+  
+  //Real double Reduce
+  template<>
+  inline Grid::RealD Reduce<Grid::RealD, float32x4_t>::operator()(float32x4_t in){
+    return 0;
+  }
+
+  //Integer Reduce
+  template<>
+  inline Integer Reduce<Integer, uint32x4_t>::operator()(uint32x4_t in){
+    // FIXME unimplemented
+   printf("Reduce : Missing integer implementation -> FIX\n");
+    assert(0);
+  }
+}
+
+//////////////////////////////////////////////////////////////////////////////////////
+// Here assign types 
+namespace Grid {
+
+  typedef float32x4_t  SIMD_Ftype; // Single precision type
+  typedef float32x4_t  SIMD_Dtype; // Double precision type - no double on ARMv7
+  typedef uint32x4_t   SIMD_Itype; // Integer type
+
+  inline void v_prefetch0(int size, const char *ptr){};  // prefetch utilities
+  inline void prefetch_HINT_T0(const char *ptr){};
+
+
+  // Gpermute function
+  template < typename VectorSIMD > 
+    inline void Gpermute(VectorSIMD &y,const VectorSIMD &b, int perm ) {
+    Optimization::permute(y.v,b.v,perm);
+  }
+
+
+  // Function name aliases
+  typedef Optimization::Vsplat   VsplatSIMD;
+  typedef Optimization::Vstore   VstoreSIMD;
+  typedef Optimization::Vset     VsetSIMD;
+  typedef Optimization::Vstream  VstreamSIMD;
+  template <typename S, typename T> using ReduceSIMD = Optimization::Reduce<S,T>;
+
+ 
+
+
+  // Arithmetic operations
+  typedef Optimization::Sum         SumSIMD;
+  typedef Optimization::Sub         SubSIMD;
+  typedef Optimization::Mult        MultSIMD;
+  typedef Optimization::MultComplex MultComplexSIMD;
+  typedef Optimization::Conj        ConjSIMD;
+  typedef Optimization::TimesMinusI TimesMinusISIMD;
+  typedef Optimization::TimesI      TimesISIMD;
+
+}
--- a/lib/simd/Grid_sse4.h
+++ b/lib/simd/Grid_sse4.h
@@ -4,7 +4,7 @@

  Using intrinsics
 */
-// Time-stamp: <2015-05-27 12:02:07 neo>
+// Time-stamp: <2015-06-09 14:24:01 neo>
 //----------------------------------------------------------------------

 #include <pmmintrin.h>
@@ -297,7 +297,12 @@ namespace Grid {
  typedef __m128d SIMD_Dtype; // Double precision type
  typedef __m128i SIMD_Itype; // Integer type

-  inline void v_prefetch0(int size, const char *ptr){};  // prefetch utilities
+  // prefetch utilities
+  inline void v_prefetch0(int size, const char *ptr){};
+  inline void prefetch_HINT_T0(const char *ptr){
+    _mm_prefetch(ptr,_MM_HINT_T0);
+  }
+  

  // Gpermute function
  template < typename VectorSIMD > 
--- a/lib/simd/Grid_vector_types.h
+++ b/lib/simd/Grid_vector_types.h
@@ -2,11 +2,14 @@
 /*! @file Grid_vector_types.h
  @brief Defines templated class Grid_simd to deal with inner vector types
 */
-// Time-stamp: <2015-05-29 14:19:48 neo>
+// Time-stamp: <2015-06-09 15:00:47 neo>
 //---------------------------------------------------------------------------
 #ifndef GRID_VECTOR_TYPES
 #define GRID_VECTOR_TYPES

+#ifdef EMPTY_SIMD
+#include "Grid_empty.h"
+#endif
 #ifdef SSE4
 #include "Grid_sse4.h"
 #endif
@@ -19,6 +22,9 @@
 #if defined QPX
 #include "Grid_qpx.h"
 #endif
+#ifdef NEONv7
+#include "Grid_neon.h"
+#endif

 namespace Grid {

@@ -152,7 +158,7 @@ namespace Grid {
    ///////////////////////
    friend inline void vprefetch(const Grid_simd &v)
    {
-      _mm_prefetch((const char*)&v.v,_MM_HINT_T0);
+      prefetch_HINT_T0((const char*)&v.v);
    }

    ///////////////////////
--- a/lib/stencil/.dirstamp
+++ b/lib/stencil/.dirstamp