1
0
mirror of https://github.com/paboyle/Grid.git synced 2025-06-12 20:27:06 +01:00
Conflicts:
	lib/Make.inc
This commit is contained in:
Peter Boyle
2015-06-09 10:27:10 +01:00
19 changed files with 687 additions and 31 deletions

View File

@ -4,7 +4,7 @@
Using intrinsics
*/
// Time-stamp: <2015-05-29 14:13:30 neo>
// Time-stamp: <2015-06-09 14:26:59 neo>
//----------------------------------------------------------------------
#include <immintrin.h>
@ -383,6 +383,12 @@ namespace Grid {
_mm_prefetch(ptr+i+512,_MM_HINT_T0);
}
}
inline void prefetch_HINT_T0(const char *ptr){
_mm_prefetch(ptr,_MM_HINT_T0);
}
template < typename VectorSIMD >
inline void Gpermute(VectorSIMD &y,const VectorSIMD &b, int perm ) {
Optimization::permute(y.v,b.v,perm);

View File

@ -4,7 +4,7 @@
Using intrinsics
*/
// Time-stamp: <2015-05-27 12:08:50 neo>
// Time-stamp: <2015-06-09 14:27:28 neo>
//----------------------------------------------------------------------
#include <immintrin.h>
@ -309,6 +309,12 @@ namespace Grid {
_mm_prefetch(ptr+i+512,_MM_HINT_T0);
}
}
inline void prefetch_HINT_T0(const char *ptr){
_mm_prefetch(ptr,_MM_HINT_T0);
}
// Gpermute utilities consider coalescing into 1 Gpermute
template < typename VectorSIMD >

289
lib/simd/Grid_empty.h Normal file
View File

@ -0,0 +1,289 @@
//----------------------------------------------------------------------
/*! @file Grid_sse4.h
@brief Empty Optimization libraries for debugging
Using intrinsics
*/
// Time-stamp: <2015-06-09 14:28:02 neo>
//----------------------------------------------------------------------
namespace Optimization {
template<class vtype>
union uconv {
float f;
vtype v;
};
union u128f {
float v;
float f[4];
};
union u128d {
double v;
double f[2];
};
struct Vsplat{
//Complex float
inline float operator()(float a, float b){
return 0;
}
// Real float
inline float operator()(float a){
return 0;
}
//Complex double
inline double operator()(double a, double b){
return 0;
}
//Real double
inline double operator()(double a){
return 0;
}
//Integer
inline int operator()(Integer a){
return 0;
}
};
struct Vstore{
//Float
inline void operator()(float a, float* F){
}
//Double
inline void operator()(double a, double* D){
}
//Integer
inline void operator()(int a, Integer* I){
}
};
struct Vstream{
//Float
inline void operator()(float * a, float b){
}
//Double
inline void operator()(double * a, double b){
}
};
struct Vset{
// Complex float
inline float operator()(Grid::ComplexF *a){
return 0;
}
// Complex double
inline double operator()(Grid::ComplexD *a){
return 0;
}
// Real float
inline float operator()(float *a){
return 0;
}
// Real double
inline double operator()(double *a){
return 0;
}
// Integer
inline int operator()(Integer *a){
return 0;
}
};
template <typename Out_type, typename In_type>
struct Reduce{
//Need templated class to overload output type
//General form must generate error if compiled
inline Out_type operator()(In_type in){
printf("Error, using wrong Reduce function\n");
exit(1);
return 0;
}
};
/////////////////////////////////////////////////////
// Arithmetic operations
/////////////////////////////////////////////////////
struct Sum{
//Complex/Real float
inline float operator()(float a, float b){
return 0;
}
//Complex/Real double
inline double operator()(double a, double b){
return 0;
}
//Integer
inline int operator()(int a, int b){
return 0;
}
};
struct Sub{
//Complex/Real float
inline float operator()(float a, float b){
return 0;
}
//Complex/Real double
inline double operator()(double a, double b){
return 0;
}
//Integer
inline int operator()(int a, int b){
return 0;
}
};
struct MultComplex{
// Complex float
inline float operator()(float a, float b){
return 0;
}
// Complex double
inline double operator()(double a, double b){
return 0;
}
};
struct Mult{
// Real float
inline float operator()(float a, float b){
return 0;
}
// Real double
inline double operator()(double a, double b){
return 0;
}
// Integer
inline int operator()(int a, int b){
return 0;
}
};
struct Conj{
// Complex single
inline float operator()(float in){
return 0;
}
// Complex double
inline double operator()(double in){
return 0;
}
// do not define for integer input
};
struct TimesMinusI{
//Complex single
inline float operator()(float in, float ret){
return 0;
}
//Complex double
inline double operator()(double in, double ret){
return 0;
}
};
struct TimesI{
//Complex single
inline float operator()(float in, float ret){
return 0;
}
//Complex double
inline double operator()(double in, double ret){
return 0;
}
};
//////////////////////////////////////////////
// Some Template specialization
template < typename vtype >
void permute(vtype &a, vtype b, int perm) {
};
//Complex float Reduce
template<>
inline Grid::ComplexF Reduce<Grid::ComplexF, float>::operator()(float in){
return 0;
}
//Real float Reduce
template<>
inline Grid::RealF Reduce<Grid::RealF, float>::operator()(float in){
return 0;
}
//Complex double Reduce
template<>
inline Grid::ComplexD Reduce<Grid::ComplexD, double>::operator()(double in){
return 0;
}
//Real double Reduce
template<>
inline Grid::RealD Reduce<Grid::RealD, double>::operator()(double in){
return 0;
}
//Integer Reduce
template<>
inline Integer Reduce<Integer, int>::operator()(int in){
// FIXME unimplemented
printf("Reduce : Missing integer implementation -> FIX\n");
assert(0);
}
}
//////////////////////////////////////////////////////////////////////////////////////
// Here assign types
namespace Grid {
typedef float SIMD_Ftype; // Single precision type
typedef double SIMD_Dtype; // Double precision type
typedef int SIMD_Itype; // Integer type
// prefetch utilities
inline void v_prefetch0(int size, const char *ptr){};
inline void prefetch_HINT_T0(const char *ptr){};
// Gpermute function
template < typename VectorSIMD >
inline void Gpermute(VectorSIMD &y,const VectorSIMD &b, int perm ) {
Optimization::permute(y.v,b.v,perm);
}
// Function name aliases
typedef Optimization::Vsplat VsplatSIMD;
typedef Optimization::Vstore VstoreSIMD;
typedef Optimization::Vset VsetSIMD;
typedef Optimization::Vstream VstreamSIMD;
template <typename S, typename T> using ReduceSIMD = Optimization::Reduce<S,T>;
// Arithmetic operations
typedef Optimization::Sum SumSIMD;
typedef Optimization::Sub SubSIMD;
typedef Optimization::Mult MultSIMD;
typedef Optimization::MultComplex MultComplexSIMD;
typedef Optimization::Conj ConjSIMD;
typedef Optimization::TimesMinusI TimesMinusISIMD;
typedef Optimization::TimesI TimesISIMD;
}

308
lib/simd/Grid_neon.h Normal file
View File

@ -0,0 +1,308 @@
//----------------------------------------------------------------------
/*! @file Grid_sse4.h
@brief Optimization libraries for NEON (ARM) instructions set ARMv7
Experimental - Using intrinsics - DEVELOPING!
*/
// Time-stamp: <2015-06-09 15:25:40 neo>
//----------------------------------------------------------------------
#include <arm_neon.h>
namespace Optimization {
template<class vtype>
union uconv {
float32x4_t f;
vtype v;
};
union u128f {
float32x4_t v;
float f[4];
};
union u128d {
float32x4_t v;
float f[4];
};
struct Vsplat{
//Complex float
inline float32x4_t operator()(float a, float b){
float32x4_t foo;
return foo;
}
// Real float
inline float32x4_t operator()(float a){
float32x4_t foo;
return foo;
}
//Complex double
inline float32x4_t operator()(double a, double b){
float32x4_t foo;
return foo;
}
//Real double
inline float32x4_t operator()(double a){
float32x4_t foo;
return foo;
}
//Integer
inline uint32x4_t operator()(Integer a){
uint32x4_t foo;
return foo;
}
};
struct Vstore{
//Float
inline void operator()(float32x4_t a, float* F){
}
//Double
inline void operator()(float32x4_t a, double* D){
}
//Integer
inline void operator()(uint32x4_t a, Integer* I){
}
};
struct Vstream{
//Float
inline void operator()(float * a, float32x4_t b){
}
//Double
inline void operator()(double * a, float32x4_t b){
}
};
struct Vset{
// Complex float
inline float32x4_t operator()(Grid::ComplexF *a){
float32x4_t foo;
return foo;
}
// Complex double
inline float32x4_t operator()(Grid::ComplexD *a){
float32x4_t foo;
return foo;
}
// Real float
inline float32x4_t operator()(float *a){
float32x4_t foo;
return foo;
}
// Real double
inline float32x4_t operator()(double *a){
float32x4_t foo;
return foo;
}
// Integer
inline uint32x4_t operator()(Integer *a){
uint32x4_t foo;
return foo;
}
};
template <typename Out_type, typename In_type>
struct Reduce{
//Need templated class to overload output type
//General form must generate error if compiled
inline Out_type operator()(In_type in){
printf("Error, using wrong Reduce function\n");
exit(1);
return 0;
}
};
/////////////////////////////////////////////////////
// Arithmetic operations
/////////////////////////////////////////////////////
struct Sum{
//Complex/Real float
inline float32x4_t operator()(float32x4_t a, float32x4_t b){
float32x4_t foo;
return foo;
}
//Complex/Real double
//inline float32x4_t operator()(float32x4_t a, float32x4_t b){
// float32x4_t foo;
// return foo;
//}
//Integer
inline uint32x4_t operator()(uint32x4_t a, uint32x4_t b){
uint32x4_t foo;
return foo;
}
};
struct Sub{
//Complex/Real float
inline float32x4_t operator()(float32x4_t a, float32x4_t b){
float32x4_t foo;
return foo;
}
//Complex/Real double
//inline float32x4_t operator()(float32x4_t a, float32x4_t b){
// float32x4_t foo;
// return foo;
//}
//Integer
inline uint32x4_t operator()(uint32x4_t a, uint32x4_t b){
uint32x4_t foo;
return foo;
}
};
struct MultComplex{
// Complex float
inline float32x4_t operator()(float32x4_t a, float32x4_t b){
float32x4_t foo;
return foo;
}
// Complex double
//inline float32x4_t operator()(float32x4_t a, float32x4_t b){
// float32x4_t foo;
// return foo;
//}
};
struct Mult{
// Real float
inline float32x4_t operator()(float32x4_t a, float32x4_t b){
return a;
}
// Real double
//inline float32x4_t operator()(float32x4_t a, float32x4_t b){
// return 0;
//}
// Integer
inline uint32x4_t operator()(uint32x4_t a, uint32x4_t b){
return a;
}
};
struct Conj{
// Complex single
inline float32x4_t operator()(float32x4_t in){
return in;
}
// Complex double
//inline float32x4_t operator()(float32x4_t in){
// return 0;
//}
// do not define for integer input
};
struct TimesMinusI{
//Complex single
inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
return in;
}
//Complex double
//inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
// return in;
//}
};
struct TimesI{
//Complex single
inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
return in;
}
//Complex double
//inline float32x4_t operator()(float32x4_t in, float32x4_t ret){
// return 0;
//}
};
//////////////////////////////////////////////
// Some Template specialization
template < typename vtype >
void permute(vtype &a, vtype b, int perm) {
};
//Complex float Reduce
template<>
inline Grid::ComplexF Reduce<Grid::ComplexF, float32x4_t>::operator()(float32x4_t in){
return 0;
}
//Real float Reduce
template<>
inline Grid::RealF Reduce<Grid::RealF, float32x4_t>::operator()(float32x4_t in){
return 0;
}
//Complex double Reduce
template<>
inline Grid::ComplexD Reduce<Grid::ComplexD, float32x4_t>::operator()(float32x4_t in){
return 0;
}
//Real double Reduce
template<>
inline Grid::RealD Reduce<Grid::RealD, float32x4_t>::operator()(float32x4_t in){
return 0;
}
//Integer Reduce
template<>
inline Integer Reduce<Integer, uint32x4_t>::operator()(uint32x4_t in){
// FIXME unimplemented
printf("Reduce : Missing integer implementation -> FIX\n");
assert(0);
}
}
//////////////////////////////////////////////////////////////////////////////////////
// Here assign types
namespace Grid {
typedef float32x4_t SIMD_Ftype; // Single precision type
typedef float32x4_t SIMD_Dtype; // Double precision type - no double on ARMv7
typedef uint32x4_t SIMD_Itype; // Integer type
inline void v_prefetch0(int size, const char *ptr){}; // prefetch utilities
inline void prefetch_HINT_T0(const char *ptr){};
// Gpermute function
template < typename VectorSIMD >
inline void Gpermute(VectorSIMD &y,const VectorSIMD &b, int perm ) {
Optimization::permute(y.v,b.v,perm);
}
// Function name aliases
typedef Optimization::Vsplat VsplatSIMD;
typedef Optimization::Vstore VstoreSIMD;
typedef Optimization::Vset VsetSIMD;
typedef Optimization::Vstream VstreamSIMD;
template <typename S, typename T> using ReduceSIMD = Optimization::Reduce<S,T>;
// Arithmetic operations
typedef Optimization::Sum SumSIMD;
typedef Optimization::Sub SubSIMD;
typedef Optimization::Mult MultSIMD;
typedef Optimization::MultComplex MultComplexSIMD;
typedef Optimization::Conj ConjSIMD;
typedef Optimization::TimesMinusI TimesMinusISIMD;
typedef Optimization::TimesI TimesISIMD;
}

View File

@ -4,7 +4,7 @@
Using intrinsics
*/
// Time-stamp: <2015-05-27 12:02:07 neo>
// Time-stamp: <2015-06-09 14:24:01 neo>
//----------------------------------------------------------------------
#include <pmmintrin.h>
@ -297,7 +297,12 @@ namespace Grid {
typedef __m128d SIMD_Dtype; // Double precision type
typedef __m128i SIMD_Itype; // Integer type
inline void v_prefetch0(int size, const char *ptr){}; // prefetch utilities
// prefetch utilities
inline void v_prefetch0(int size, const char *ptr){};
inline void prefetch_HINT_T0(const char *ptr){
_mm_prefetch(ptr,_MM_HINT_T0);
}
// Gpermute function
template < typename VectorSIMD >

View File

@ -2,11 +2,14 @@
/*! @file Grid_vector_types.h
@brief Defines templated class Grid_simd to deal with inner vector types
*/
// Time-stamp: <2015-05-29 14:19:48 neo>
// Time-stamp: <2015-06-09 15:00:47 neo>
//---------------------------------------------------------------------------
#ifndef GRID_VECTOR_TYPES
#define GRID_VECTOR_TYPES
#ifdef EMPTY_SIMD
#include "Grid_empty.h"
#endif
#ifdef SSE4
#include "Grid_sse4.h"
#endif
@ -19,6 +22,9 @@
#if defined QPX
#include "Grid_qpx.h"
#endif
#ifdef NEONv7
#include "Grid_neon.h"
#endif
namespace Grid {
@ -152,7 +158,7 @@ namespace Grid {
///////////////////////
friend inline void vprefetch(const Grid_simd &v)
{
_mm_prefetch((const char*)&v.v,_MM_HINT_T0);
prefetch_HINT_T0((const char*)&v.v);
}
///////////////////////