1
0
mirror of https://github.com/paboyle/Grid.git synced 2024-11-14 01:35:36 +00:00
Grid/lib/lattice/Grid_lattice_arith.h

157 lines
5.0 KiB
C
Raw Normal View History

2015-04-18 20:44:19 +01:00
#ifndef GRID_LATTICE_ARITH_H
#define GRID_LATTICE_ARITH_H
namespace Grid {
2015-05-03 09:44:47 +01:00
//////////////////////////////////////////////////////////////////////////////////////////////////////
// avoid copy back routines for mult, mac, sub, add
//////////////////////////////////////////////////////////////////////////////////////////////////////
template<class obj1,class obj2,class obj3>
void mult(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const Lattice<obj3> &rhs){
conformable(lhs,rhs);
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<lhs._grid->oSites();ss++){
obj1 tmp;
mult(&tmp,&lhs._odata[ss],&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-05-03 09:44:47 +01:00
}
}
2015-04-18 20:44:19 +01:00
2015-05-03 09:44:47 +01:00
template<class obj1,class obj2,class obj3>
void mac(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const Lattice<obj3> &rhs){
conformable(lhs,rhs);
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<lhs._grid->oSites();ss++){
obj1 tmp;
mac(&tmp,&lhs._odata[ss],&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-05-03 09:44:47 +01:00
}
}
template<class obj1,class obj2,class obj3>
void sub(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const Lattice<obj3> &rhs){
2015-04-18 20:44:19 +01:00
conformable(lhs,rhs);
#pragma omp parallel for
for(int ss=0;ss<lhs._grid->oSites();ss++){
2015-05-05 18:14:09 +01:00
obj1 tmp;
sub(&tmp,&lhs._odata[ss],&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-04-18 20:44:19 +01:00
}
}
2015-05-03 09:44:47 +01:00
template<class obj1,class obj2,class obj3>
void add(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const Lattice<obj3> &rhs){
2015-04-18 20:44:19 +01:00
conformable(lhs,rhs);
#pragma omp parallel for
for(int ss=0;ss<lhs._grid->oSites();ss++){
2015-05-05 18:14:09 +01:00
obj1 tmp;
add(&tmp,&lhs._odata[ss],&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-04-18 20:44:19 +01:00
}
}
2015-05-03 09:44:47 +01:00
//////////////////////////////////////////////////////////////////////////////////////////////////////
// avoid copy back routines for mult, mac, sub, add
//////////////////////////////////////////////////////////////////////////////////////////////////////
template<class obj1,class obj2,class obj3>
void mult(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const obj3 &rhs){
2015-05-05 18:14:09 +01:00
conformable(lhs,ret);
2015-05-03 09:44:47 +01:00
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<lhs._grid->oSites();ss++){
obj1 tmp;
mult(&tmp,&lhs._odata[ss],&rhs);
vstream(ret._odata[ss],tmp);
2015-05-03 09:44:47 +01:00
}
}
2015-04-18 20:44:19 +01:00
template<class obj1,class obj2,class obj3>
2015-05-03 09:44:47 +01:00
void mac(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const obj3 &rhs){
2015-05-05 18:14:09 +01:00
conformable(lhs,ret);
2015-04-18 20:44:19 +01:00
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<lhs._grid->oSites();ss++){
obj1 tmp;
mac(&tmp,&lhs._odata[ss],&rhs);
vstream(ret._odata[ss],tmp);
2015-04-18 20:44:19 +01:00
}
}
template<class obj1,class obj2,class obj3>
2015-05-03 09:44:47 +01:00
void sub(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const obj3 &rhs){
2015-05-05 18:14:09 +01:00
conformable(lhs,ret);
2015-05-03 09:44:47 +01:00
#pragma omp parallel for
for(int ss=0;ss<lhs._grid->oSites();ss++){
2015-05-05 18:14:09 +01:00
obj1 tmp;
sub(&tmp,&lhs._odata[ss],&rhs);
vstream(ret._odata[ss],tmp);
2015-05-03 09:44:47 +01:00
}
}
template<class obj1,class obj2,class obj3>
void add(Lattice<obj1> &ret,const Lattice<obj2> &lhs,const obj3 &rhs){
2015-05-05 18:14:09 +01:00
conformable(lhs,ret);
2015-05-03 09:44:47 +01:00
#pragma omp parallel for
for(int ss=0;ss<lhs._grid->oSites();ss++){
2015-05-05 18:14:09 +01:00
obj1 tmp;
add(&tmp,&lhs._odata[ss],&rhs);
vstream(ret._odata[ss],tmp);
2015-05-03 09:44:47 +01:00
}
}
//////////////////////////////////////////////////////////////////////////////////////////////////////
// avoid copy back routines for mult, mac, sub, add
//////////////////////////////////////////////////////////////////////////////////////////////////////
2015-05-05 18:14:09 +01:00
template<class obj1,class obj2,class obj3>
2015-05-03 09:44:47 +01:00
void mult(Lattice<obj1> &ret,const obj2 &lhs,const Lattice<obj3> &rhs){
2015-05-05 18:14:09 +01:00
conformable(ret,rhs);
2015-04-18 20:44:19 +01:00
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<rhs._grid->oSites();ss++){
obj1 tmp;
mult(&tmp,&lhs,&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-04-18 20:44:19 +01:00
}
}
template<class obj1,class obj2,class obj3>
2015-05-03 09:44:47 +01:00
void mac(Lattice<obj1> &ret,const obj2 &lhs,const Lattice<obj3> &rhs){
2015-05-05 18:14:09 +01:00
conformable(ret,rhs);
2015-05-03 09:44:47 +01:00
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<rhs._grid->oSites();ss++){
obj1 tmp;
mac(&tmp,&lhs,&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-05-03 09:44:47 +01:00
}
}
template<class obj1,class obj2,class obj3>
void sub(Lattice<obj1> &ret,const obj2 &lhs,const Lattice<obj3> &rhs){
2015-05-05 18:14:09 +01:00
conformable(ret,rhs);
2015-04-18 20:44:19 +01:00
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<rhs._grid->oSites();ss++){
obj1 tmp;
sub(&tmp,&lhs,&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-04-18 20:44:19 +01:00
}
}
template<class obj1,class obj2,class obj3>
2015-05-03 09:44:47 +01:00
void add(Lattice<obj1> &ret,const obj2 &lhs,const Lattice<obj3> &rhs){
2015-05-05 18:14:09 +01:00
conformable(ret,rhs);
2015-04-18 20:44:19 +01:00
#pragma omp parallel for
2015-05-05 18:14:09 +01:00
for(int ss=0;ss<rhs._grid->oSites();ss++){
obj1 tmp;
add(&tmp,&lhs,&rhs._odata[ss]);
vstream(ret._odata[ss],tmp);
2015-04-18 20:44:19 +01:00
}
}
2015-05-03 09:44:47 +01:00
template<class sobj,class vobj>
inline void axpy(Lattice<vobj> &ret,sobj a,const Lattice<vobj> &lhs,const Lattice<vobj> &rhs){
conformable(lhs,rhs);
#pragma omp parallel for
for(int ss=0;ss<lhs._grid->oSites();ss++){
2015-05-05 18:14:09 +01:00
vobj tmp = a*lhs._odata[ss];
vstream(ret._odata[ss],tmp+rhs._odata[ss]);
2015-05-03 09:44:47 +01:00
}
}
2015-04-18 20:44:19 +01:00
}
#endif