Merge branch 'develop' into feature/clover

2026-07-22 03:23:28 +01:00 · 2017-08-04 12:19:57 +01:00
parent 62a64d9108 175f393f9d
commit fde71c3c52
296 changed files with 34674 additions and 9453 deletions
@@ -0,0 +1,37 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/DisableWarnings.h
+
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+
+#ifndef DISABLE_WARNINGS_H
+#define DISABLE_WARNINGS_H
+
+ //disables and intel compiler specific warning (in json.hpp)
+#pragma warning disable 488  
+
+
+#endif
@@ -41,7 +41,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
 #include <Grid/GridCore.h>
 #include <Grid/GridQCDcore.h>
 #include <Grid/qcd/action/Action.h>
+#include <Grid/qcd/utils/GaugeFix.h>
 #include <Grid/qcd/smearing/Smearing.h>
+#include <Grid/parallelIO/MetaData.h>
 #include <Grid/qcd/hmc/HMC_aggregate.h>

 #endif
@@ -38,28 +38,7 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
 #ifndef GRID_BASE_H
 #define GRID_BASE_H

-///////////////////
-// Std C++ dependencies
-///////////////////
-#include <cassert>
-#include <complex>
-#include <vector>
-#include <iostream>
-#include <iomanip>
-#include <random>
-#include <functional>
-#include <stdio.h>
-#include <stdlib.h>
-#include <stdio.h>
-#include <signal.h>
-#include <ctime>
-#include <sys/time.h>
-#include <chrono>
-
-///////////////////
-// Grid headers
-///////////////////
-#include "Config.h"
+#include <Grid/GridStd.h>

 #include <Grid/perfmon/Timer.h>
 #include <Grid/perfmon/PerfCount.h>
@@ -0,0 +1,29 @@
+#ifndef GRID_STD_H
+#define GRID_STD_H
+
+///////////////////
+// Std C++ dependencies
+///////////////////
+#include <cassert>
+#include <complex>
+#include <vector>
+#include <string>
+#include <iostream>
+#include <iomanip>
+#include <random>
+#include <functional>
+#include <stdio.h>
+#include <stdlib.h>
+#include <stdio.h>
+#include <signal.h>
+#include <ctime>
+#include <sys/time.h>
+#include <chrono>
+#include <zlib.h>
+
+///////////////////
+// Grid config
+///////////////////
+#include "Config.h"
+
+#endif /* GRID_STD_H */
@@ -0,0 +1,9 @@
+#pragma once
+#if defined __GNUC__
+#pragma GCC diagnostic push
+#pragma GCC diagnostic ignored "-Wdeprecated-declarations"
+#endif
+#include <Grid/Eigen/Dense>
+#if defined __GNUC__
+#pragma GCC diagnostic pop
+#endif
@@ -235,7 +235,7 @@ namespace Grid {
 	Field tmp(in._grid);

 	_Mat.MeooeDag(in,tmp);
-	_Mat.MooeeInvDag(tmp,out);
+        _Mat.MooeeInvDag(tmp,out);
 	_Mat.MeooeDag(out,tmp);

 	_Mat.MooeeDag(in,out);
@@ -197,8 +197,9 @@ namespace Grid {
    void operator() (LinearOperatorBase<Field> &Linop, const Field &in, Field &out) {

      GridBase *grid=in._grid;
-//std::cout << "Chevyshef(): in._grid="<<in._grid<<std::endl;
-//<<" Linop.Grid()="<<Linop.Grid()<<"Linop.RedBlackGrid()="<<Linop.RedBlackGrid()<<std::endl;
+
+      // std::cout << "Chevyshef(): in._grid="<<in._grid<<std::endl;
+      //std::cout <<" Linop.Grid()="<<Linop.Grid()<<"Linop.RedBlackGrid()="<<Linop.RedBlackGrid()<<std::endl;

      int vol=grid->gSites();

@@ -16,7 +16,7 @@
 #define INCLUDED_ALG_REMEZ_H

 #include <stddef.h>
-#include <Config.h>
+#include <Grid/GridStd.h>

 #ifdef HAVE_LIBGMP
 #include "bigfloat.h"
@@ -1,137 +0,0 @@
-    /*************************************************************************************
-
-    Grid physics library, www.github.com/paboyle/Grid 
-
-    Source file: ./lib/algorithms/iterative/DenseMatrix.h
-
-    Copyright (C) 2015
-
-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-Author: paboyle <paboyle@ph.ed.ac.uk>
-
-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
-
-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
-
-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
-#ifndef GRID_DENSE_MATRIX_H
-#define GRID_DENSE_MATRIX_H
-
-namespace Grid {
-    /////////////////////////////////////////////////////////////
-    // Matrix untils
-    /////////////////////////////////////////////////////////////
-
-template<class T> using DenseVector = std::vector<T>;
-template<class T> using DenseMatrix = DenseVector<DenseVector<T> >;
-
-template<class T> void Size(DenseVector<T> & vec, int &N) 
-{ 
-  N= vec.size();
-}
-template<class T> void Size(DenseMatrix<T> & mat, int &N,int &M) 
-{ 
-  N= mat.size();
-  M= mat[0].size();
-}
-
-template<class T> void SizeSquare(DenseMatrix<T> & mat, int &N) 
-{ 
-  int M; Size(mat,N,M);
-  assert(N==M);
-}
-
-template<class T> void Resize(DenseVector<T > & mat, int N) { 
-  mat.resize(N);
-}
-template<class T> void Resize(DenseMatrix<T > & mat, int N, int M) { 
-  mat.resize(N);
-  for(int i=0;i<N;i++){
-    mat[i].resize(M);
-  }
-}
-template<class T> void Fill(DenseMatrix<T> & mat, T&val) { 
-  int N,M;
-  Size(mat,N,M);
-  for(int i=0;i<N;i++){
-  for(int j=0;j<M;j++){
-    mat[i][j] = val;
-  }}
-}
-
-/** Transpose of a matrix **/
-template<class T> DenseMatrix<T> Transpose(DenseMatrix<T> & mat){
-  int N,M;
-  Size(mat,N,M);
-  DenseMatrix<T> C; Resize(C,M,N);
-  for(int i=0;i<M;i++){
-  for(int j=0;j<N;j++){
-    C[i][j] = mat[j][i];
-  }} 
-  return C;
-}
-/** Set DenseMatrix to unit matrix **/
-template<class T> void Unity(DenseMatrix<T> &A){
-  int N;  SizeSquare(A,N);
-  for(int i=0;i<N;i++){
-    for(int j=0;j<N;j++){
-      if ( i==j ) A[i][j] = 1;
-      else        A[i][j] = 0;
-    } 
-  } 
-}
-
-/** Add C * I to matrix **/
-template<class T>
-void PlusUnit(DenseMatrix<T> & A,T c){
-  int dim;  SizeSquare(A,dim);
-  for(int i=0;i<dim;i++){A[i][i] = A[i][i] + c;} 
-}
-
-/** return the Hermitian conjugate of matrix **/
-template<class T>
-DenseMatrix<T> HermitianConj(DenseMatrix<T> &mat){
-
-  int dim; SizeSquare(mat,dim);
-
-  DenseMatrix<T> C; Resize(C,dim,dim);
-
-  for(int i=0;i<dim;i++){
-    for(int j=0;j<dim;j++){
-      C[i][j] = conj(mat[j][i]);
-    } 
-  } 
-  return C;
-}
-/**Get a square submatrix**/
-template <class T>
-DenseMatrix<T> GetSubMtx(DenseMatrix<T> &A,int row_st, int row_end, int col_st, int col_end)
-{
-  DenseMatrix<T> H; Resize(H,row_end - row_st,col_end-col_st);
-
-  for(int i = row_st; i<row_end; i++){
-  for(int j = col_st; j<col_end; j++){
-    H[i-row_st][j-col_st]=A[i][j];
-  }}
-  return H;
-}
-
-}
-
-#include "Householder.h"
-#include "Francis.h"
-
-#endif
-
@@ -1,525 +0,0 @@
-    /*************************************************************************************
-
-    Grid physics library, www.github.com/paboyle/Grid 
-
-    Source file: ./lib/algorithms/iterative/Francis.h
-
-    Copyright (C) 2015
-
-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-
-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
-
-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
-
-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
-#ifndef FRANCIS_H
-#define FRANCIS_H
-
-#include <cstdlib>
-#include <string>
-#include <cmath>
-#include <iostream>
-#include <sstream>
-#include <stdexcept>
-#include <fstream>
-#include <complex>
-#include <algorithm>
-
-//#include <timer.h>
-//#include <lapacke.h>
-//#include <Eigen/Dense>
-
-namespace Grid {
-
-template <class T> int SymmEigensystem(DenseMatrix<T > &Ain, DenseVector<T> &evals, DenseMatrix<T> &evecs, RealD small);
-template <class T> int     Eigensystem(DenseMatrix<T > &Ain, DenseVector<T> &evals, DenseMatrix<T> &evecs, RealD small);
-
-/**
-  Find the eigenvalues of an upper hessenberg matrix using the Francis QR algorithm.
-H =
-      x  x  x  x  x  x  x  x  x
-      x  x  x  x  x  x  x  x  x
-      0  x  x  x  x  x  x  x  x
-      0  0  x  x  x  x  x  x  x
-      0  0  0  x  x  x  x  x  x
-      0  0  0  0  x  x  x  x  x
-      0  0  0  0  0  x  x  x  x
-      0  0  0  0  0  0  x  x  x
-      0  0  0  0  0  0  0  x  x
-Factorization is P T P^H where T is upper triangular (mod cc blocks) and P is orthagonal/unitary.
-**/
-template <class T>
-int QReigensystem(DenseMatrix<T> &Hin, DenseVector<T> &evals, DenseMatrix<T> &evecs, RealD small)
-{
-  DenseMatrix<T> H = Hin; 
-
-  int N ; SizeSquare(H,N);
-  int M = N;
-
-  Fill(evals,0);
-  Fill(evecs,0);
-
-  T s,t,x=0,y=0,z=0;
-  T u,d;
-  T apd,amd,bc;
-  DenseVector<T> p(N,0);
-  T nrm = Norm(H);    ///DenseMatrix Norm
-  int n, m;
-  int e = 0;
-  int it = 0;
-  int tot_it = 0;
-  int l = 0;
-  int r = 0;
-  DenseMatrix<T> P; Resize(P,N,N); Unity(P);
-  DenseVector<int> trows(N,0);
-
-  /// Check if the matrix is really hessenberg, if not abort
-  RealD sth = 0;
-  for(int j=0;j<N;j++){
-    for(int i=j+2;i<N;i++){
-      sth = abs(H[i][j]);
-      if(sth > small){
-	std::cout << "Non hessenberg H = " << sth << " > " << small << std::endl;
-	exit(1);
-      }
-    }
-  }
-
-  do{
-    std::cout << "Francis QR Step N = " << N << std::endl;
-    /** Check for convergence
-      x  x  x  x  x
-      0  x  x  x  x
-      0  0  x  x  x
-      0  0  x  x  x
-      0  0  0  0  x
-      for this matrix l = 4
-     **/
-    do{
-      l = Chop_subdiag(H,nrm,e,small);
-      r = 0;    ///May have converged on more than one eval
-      ///Single eval
-      if(l == N-1){
-        evals[e] = H[l][l];
-        N--; e++; r++; it = 0;
-      }
-      ///RealD eval
-      if(l == N-2){
-        trows[l+1] = 1;    ///Needed for UTSolve
-        apd = H[l][l] + H[l+1][l+1];
-        amd = H[l][l] - H[l+1][l+1];
-        bc =  (T)4.0*H[l+1][l]*H[l][l+1];
-        evals[e]   = (T)0.5*( apd + sqrt(amd*amd + bc) );
-        evals[e+1] = (T)0.5*( apd - sqrt(amd*amd + bc) );
-        N-=2; e+=2; r++; it = 0;
-      }
-    } while(r>0);
-
-    if(N ==0) break;
-
-    DenseVector<T > ck; Resize(ck,3);
-    DenseVector<T> v;   Resize(v,3);
-
-    for(int m = N-3; m >= l; m--){
-      ///Starting vector essentially random shift.
-      if(it%10 == 0 && N >= 3 && it > 0){
-        s = (T)1.618033989*( abs( H[N-1][N-2] ) + abs( H[N-2][N-3] ) );
-        t = (T)0.618033989*( abs( H[N-1][N-2] ) + abs( H[N-2][N-3] ) );
-        x = H[m][m]*H[m][m] + H[m][m+1]*H[m+1][m] - s*H[m][m] + t;
-        y = H[m+1][m]*(H[m][m] + H[m+1][m+1] - s);
-        z = H[m+1][m]*H[m+2][m+1];
-      }
-      ///Starting vector implicit Q theorem
-      else{
-        s = (H[N-2][N-2] + H[N-1][N-1]);
-        t = (H[N-2][N-2]*H[N-1][N-1] - H[N-2][N-1]*H[N-1][N-2]);
-        x = H[m][m]*H[m][m] + H[m][m+1]*H[m+1][m] - s*H[m][m] + t;
-        y = H[m+1][m]*(H[m][m] + H[m+1][m+1] - s);
-        z = H[m+1][m]*H[m+2][m+1];
-      }
-      ck[0] = x; ck[1] = y; ck[2] = z;
-
-      if(m == l) break;
-
-      /** Some stupid thing from numerical recipies, seems to work**/
-      // PAB.. for heaven's sake quote page, purpose, evidence it works.
-      //       what sort of comment is that!?!?!?
-      u=abs(H[m][m-1])*(abs(y)+abs(z));
-      d=abs(x)*(abs(H[m-1][m-1])+abs(H[m][m])+abs(H[m+1][m+1]));
-      if ((T)abs(u+d) == (T)abs(d) ){
-	l = m; break;
-      }
-
-      //if (u < small){l = m; break;}
-    }
-    if(it > 100000){
-     std::cout << "QReigensystem: bugger it got stuck after 100000 iterations" << std::endl;
-     std::cout << "got " << e << " evals " << l << " " << N << std::endl;
-      exit(1);
-    }
-    normalize(ck);    ///Normalization cancels in PHP anyway
-    T beta;
-    Householder_vector<T >(ck, 0, 2, v, beta);
-    Householder_mult<T >(H,v,beta,0,l,l+2,0);
-    Householder_mult<T >(H,v,beta,0,l,l+2,1);
-    ///Accumulate eigenvector
-    Householder_mult<T >(P,v,beta,0,l,l+2,1);
-    int sw = 0;      ///Are we on the last row?
-    for(int k=l;k<N-2;k++){
-      x = H[k+1][k];
-      y = H[k+2][k];
-      z = (T)0.0;
-      if(k+3 <= N-1){
-	z = H[k+3][k];
-      } else{
-	sw = 1; 
-	v[2] = (T)0.0;
-      }
-      ck[0] = x; ck[1] = y; ck[2] = z;
-      normalize(ck);
-      Householder_vector<T >(ck, 0, 2-sw, v, beta);
-      Householder_mult<T >(H,v, beta,0,k+1,k+3-sw,0);
-      Householder_mult<T >(H,v, beta,0,k+1,k+3-sw,1);
-      ///Accumulate eigenvector
-      Householder_mult<T >(P,v, beta,0,k+1,k+3-sw,1);
-    }
-    it++;
-    tot_it++;
-  }while(N > 1);
-  N = evals.size();
-  ///Annoying - UT solves in reverse order;
-  DenseVector<T> tmp; Resize(tmp,N);
-  for(int i=0;i<N;i++){
-    tmp[i] = evals[N-i-1];
-  } 
-  evals = tmp;
-  UTeigenvectors(H, trows, evals, evecs);
-  for(int i=0;i<evals.size();i++){evecs[i] = P*evecs[i]; normalize(evecs[i]);}
-  return tot_it;
-}
-
-template <class T>
-int my_Wilkinson(DenseMatrix<T> &Hin, DenseVector<T> &evals, DenseMatrix<T> &evecs, RealD small)
-{
-  /**
-  Find the eigenvalues of an upper Hessenberg matrix using the Wilkinson QR algorithm.
-  H =
-  x  x  0  0  0  0
-  x  x  x  0  0  0
-  0  x  x  x  0  0
-  0  0  x  x  x  0
-  0  0  0  x  x  x
-  0  0  0  0  x  x
-  Factorization is P T P^H where T is upper triangular (mod cc blocks) and P is orthagonal/unitary.  **/
-  return my_Wilkinson(Hin, evals, evecs, small, small);
-}
-
-template <class T>
-int my_Wilkinson(DenseMatrix<T> &Hin, DenseVector<T> &evals, DenseMatrix<T> &evecs, RealD small, RealD tol)
-{
-  int N; SizeSquare(Hin,N);
-  int M = N;
-
-  ///I don't want to modify the input but matricies must be passed by reference
-  //Scale a matrix by its "norm"
-  //RealD Hnorm = abs( Hin.LargestDiag() ); H =  H*(1.0/Hnorm);
-  DenseMatrix<T> H;  H = Hin;
-  
-  RealD Hnorm = abs(Norm(Hin));
-  H = H * (1.0 / Hnorm);
-
-  // TODO use openmp and memset
-  Fill(evals,0);
-  Fill(evecs,0);
-
-  T s, t, x = 0, y = 0, z = 0;
-  T u, d;
-  T apd, amd, bc;
-  DenseVector<T> p; Resize(p,N); Fill(p,0);
-
-  T nrm = Norm(H);    ///DenseMatrix Norm
-  int n, m;
-  int e = 0;
-  int it = 0;
-  int tot_it = 0;
-  int l = 0;
-  int r = 0;
-  DenseMatrix<T> P; Resize(P,N,N);
-  Unity(P);
-  DenseVector<int> trows(N, 0);
-  /// Check if the matrix is really symm tridiag
-  RealD sth = 0;
-  for(int j = 0; j < N; ++j)
-  {
-    for(int i = j + 2; i < N; ++i)
-    {
-      if(abs(H[i][j]) > tol || abs(H[j][i]) > tol)
-      {
-	std::cout << "Non Tridiagonal H(" << i << ","<< j << ") = |" << Real( real( H[j][i] ) ) << "| > " << tol << std::endl;
-	std::cout << "Warning tridiagonalize and call again" << std::endl;
-        // exit(1); // see what is going on
-        //return;
-      }
-    }
-  }
-
-  do{
-    do{
-      //Jasper
-      //Check if the subdiagonal term is small enough (<small)
-      //if true then it is converged.
-      //check start from H.dim - e - 1
-      //How to deal with more than 2 are converged?
-      //What if Chop_symm_subdiag return something int the middle?
-      //--------------
-      l = Chop_symm_subdiag(H,nrm, e, small);
-      r = 0;    ///May have converged on more than one eval
-      //Jasper
-      //In this case
-      // x  x  0  0  0  0
-      // x  x  x  0  0  0
-      // 0  x  x  x  0  0
-      // 0  0  x  x  x  0
-      // 0  0  0  x  x  0
-      // 0  0  0  0  0  x  <- l
-      //--------------
-      ///Single eval
-      if(l == N - 1)
-      {
-        evals[e] = H[l][l];
-        N--;
-        e++;
-        r++;
-        it = 0;
-      }
-      //Jasper
-      // x  x  0  0  0  0
-      // x  x  x  0  0  0
-      // 0  x  x  x  0  0
-      // 0  0  x  x  0  0
-      // 0  0  0  0  x  x  <- l
-      // 0  0  0  0  x  x
-      //--------------
-      ///RealD eval
-      if(l == N - 2)
-      {
-        trows[l + 1] = 1;    ///Needed for UTSolve
-        apd = H[l][l] + H[l + 1][ l + 1];
-        amd = H[l][l] - H[l + 1][l + 1];
-        bc =  (T) 4.0 * H[l + 1][l] * H[l][l + 1];
-        evals[e] = (T) 0.5 * (apd + sqrt(amd * amd + bc));
-        evals[e + 1] = (T) 0.5 * (apd - sqrt(amd * amd + bc));
-        N -= 2;
-        e += 2;
-        r++;
-        it = 0;
-      }
-    }while(r > 0);
-    //Jasper
-    //Already converged
-    //--------------
-    if(N == 0) break;
-
-    DenseVector<T> ck,v; Resize(ck,2); Resize(v,2);
-
-    for(int m = N - 3; m >= l; m--)
-    {
-      ///Starting vector essentially random shift.
-      if(it%10 == 0 && N >= 3 && it > 0)
-      {
-        t = abs(H[N - 1][N - 2]) + abs(H[N - 2][N - 3]);
-        x = H[m][m] - t;
-        z = H[m + 1][m];
-      } else {
-      ///Starting vector implicit Q theorem
-        d = (H[N - 2][N - 2] - H[N - 1][N - 1]) * (T) 0.5;
-        t =  H[N - 1][N - 1] - H[N - 1][N - 2] * H[N - 1][N - 2] 
-	  / (d + sign(d) * sqrt(d * d + H[N - 1][N - 2] * H[N - 1][N - 2]));
-        x = H[m][m] - t;
-        z = H[m + 1][m];
-      }
-      //Jasper
-      //why it is here????
-      //-----------------------
-      if(m == l)
-        break;
-
-      u = abs(H[m][m - 1]) * (abs(y) + abs(z));
-      d = abs(x) * (abs(H[m - 1][m - 1]) + abs(H[m][m]) + abs(H[m + 1][m + 1]));
-      if ((T)abs(u + d) == (T)abs(d))
-      {
-        l = m;
-        break;
-      }
-    }
-    //Jasper
-    if(it > 1000000)
-    {
-      std::cout << "Wilkinson: bugger it got stuck after 100000 iterations" << std::endl;
-      std::cout << "got " << e << " evals " << l << " " << N << std::endl;
-      exit(1);
-    }
-    //
-    T s, c;
-    Givens_calc<T>(x, z, c, s);
-    Givens_mult<T>(H, l, l + 1, c, -s, 0);
-    Givens_mult<T>(H, l, l + 1, c,  s, 1);
-    Givens_mult<T>(P, l, l + 1, c,  s, 1);
-    //
-    for(int k = l; k < N - 2; ++k)
-    {
-      x = H.A[k + 1][k];
-      z = H.A[k + 2][k];
-      Givens_calc<T>(x, z, c, s);
-      Givens_mult<T>(H, k + 1, k + 2, c, -s, 0);
-      Givens_mult<T>(H, k + 1, k + 2, c,  s, 1);
-      Givens_mult<T>(P, k + 1, k + 2, c,  s, 1);
-    }
-    it++;
-    tot_it++;
-  }while(N > 1);
-
-  N = evals.size();
-  ///Annoying - UT solves in reverse order;
-  DenseVector<T> tmp(N);
-  for(int i = 0; i < N; ++i)
-    tmp[i] = evals[N-i-1];
-  evals = tmp;
-  //
-  UTeigenvectors(H, trows, evals, evecs);
-  //UTSymmEigenvectors(H, trows, evals, evecs);
-  for(int i = 0; i < evals.size(); ++i)
-  {
-    evecs[i] = P * evecs[i];
-    normalize(evecs[i]);
-    evals[i] = evals[i] * Hnorm;
-  }
-  // // FIXME this is to test
-  // Hin.write("evecs3", evecs);
-  // Hin.write("evals3", evals);
-  // // check rsd
-  // for(int i = 0; i < M; i++) {
-  //   vector<T> Aevec = Hin * evecs[i];
-  //   RealD norm2(0.);
-  //   for(int j = 0; j < M; j++) {
-  //     norm2 += (Aevec[j] - evals[i] * evecs[i][j]) * (Aevec[j] - evals[i] * evecs[i][j]);
-  //   }
-  // }
-  return tot_it;
-}
-
-template <class T>
-void Hess(DenseMatrix<T > &A, DenseMatrix<T> &Q, int start){
-
-  /**
-  turn a matrix A =
-  x  x  x  x  x
-  x  x  x  x  x
-  x  x  x  x  x
-  x  x  x  x  x
-  x  x  x  x  x
-  into
-  x  x  x  x  x
-  x  x  x  x  x
-  0  x  x  x  x
-  0  0  x  x  x
-  0  0  0  x  x
-  with householder rotations
-  Slow.
-  */
-  int N ; SizeSquare(A,N);
-  DenseVector<T > p; Resize(p,N); Fill(p,0);
-
-  for(int k=start;k<N-2;k++){
-    //cerr << "hess" << k << std::endl;
-    DenseVector<T > ck,v; Resize(ck,N-k-1); Resize(v,N-k-1);
-    for(int i=k+1;i<N;i++){ck[i-k-1] = A(i,k);}  ///kth column
-    normalize(ck);    ///Normalization cancels in PHP anyway
-    T beta;
-    Householder_vector<T >(ck, 0, ck.size()-1, v, beta);  ///Householder vector
-    Householder_mult<T>(A,v,beta,start,k+1,N-1,0);  ///A -> PA
-    Householder_mult<T >(A,v,beta,start,k+1,N-1,1);  ///PA -> PAP^H
-    ///Accumulate eigenvector
-    Householder_mult<T >(Q,v,beta,start,k+1,N-1,1);  ///Q -> QP^H
-  }
-  /*for(int l=0;l<N-2;l++){
-    for(int k=l+2;k<N;k++){
-    A(0,k,l);
-    }
-    }*/
-}
-
-template <class T>
-void Tri(DenseMatrix<T > &A, DenseMatrix<T> &Q, int start){
-///Tridiagonalize a matrix
-  int N; SizeSquare(A,N);
-  Hess(A,Q,start);
-  /*for(int l=0;l<N-2;l++){
-    for(int k=l+2;k<N;k++){
-    A(0,l,k);
-    }
-    }*/
-}
-
-template <class T>
-void ForceTridiagonal(DenseMatrix<T> &A){
-///Tridiagonalize a matrix
-  int N ; SizeSquare(A,N);
-  for(int l=0;l<N-2;l++){
-    for(int k=l+2;k<N;k++){
-      A[l][k]=0;
-      A[k][l]=0;
-    }
-  }
-}
-
-template <class T>
-int my_SymmEigensystem(DenseMatrix<T > &Ain, DenseVector<T> &evals, DenseVector<DenseVector<T> > &evecs, RealD small){
-  ///Solve a symmetric eigensystem, not necessarily in tridiagonal form
-  int N; SizeSquare(Ain,N);
-  DenseMatrix<T > A; A = Ain;
-  DenseMatrix<T > Q; Resize(Q,N,N); Unity(Q);
-  Tri(A,Q,0);
-  int it = my_Wilkinson<T>(A, evals, evecs, small);
-  for(int k=0;k<N;k++){evecs[k] = Q*evecs[k];}
-  return it;
-}
-
-
-template <class T>
-int Wilkinson(DenseMatrix<T> &Ain, DenseVector<T> &evals, DenseVector<DenseVector<T> > &evecs, RealD small){
-  return my_Wilkinson(Ain, evals, evecs, small);
-}
-
-template <class T>
-int SymmEigensystem(DenseMatrix<T> &Ain, DenseVector<T> &evals, DenseVector<DenseVector<T> > &evecs, RealD small){
-  return my_SymmEigensystem(Ain, evals, evecs, small);
-}
-
-template <class T>
-int Eigensystem(DenseMatrix<T > &Ain, DenseVector<T> &evals, DenseVector<DenseVector<T> > &evecs, RealD small){
-///Solve a general eigensystem, not necessarily in tridiagonal form
-  int N = Ain.dim;
-  DenseMatrix<T > A(N); A = Ain;
-  DenseMatrix<T > Q(N);Q.Unity();
-  Hess(A,Q,0);
-  int it = QReigensystem<T>(A, evals, evecs, small);
-  for(int k=0;k<N;k++){evecs[k] = Q*evecs[k];}
-  return it;
-}
-
-}
-#endif
@@ -1,242 +0,0 @@
-    /*************************************************************************************
-
-    Grid physics library, www.github.com/paboyle/Grid 
-
-    Source file: ./lib/algorithms/iterative/Householder.h
-
-    Copyright (C) 2015
-
-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-
-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
-
-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
-
-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
-#ifndef HOUSEHOLDER_H
-#define HOUSEHOLDER_H
-
-#define TIMER(A) std::cout << GridLogMessage << __FUNC__ << " file "<< __FILE__ <<" line " << __LINE__ << std::endl;
-#define ENTER()  std::cout << GridLogMessage << "ENTRY "<<__FUNC__ << " file "<< __FILE__ <<" line " << __LINE__ << std::endl;
-#define LEAVE()  std::cout << GridLogMessage << "EXIT  "<<__FUNC__ << " file "<< __FILE__ <<" line " << __LINE__ << std::endl;
-
-#include <cstdlib>
-#include <string>
-#include <cmath>
-#include <iostream>
-#include <sstream>
-#include <stdexcept>
-#include <fstream>
-#include <complex>
-#include <algorithm>
-
-namespace Grid {
-/** Comparison function for finding the max element in a vector **/
-template <class T> bool cf(T i, T j) { 
-  return abs(i) < abs(j); 
-}
-
-/** 
-	Calculate a real Givens angle 
- **/
-template <class T> inline void Givens_calc(T y, T z, T &c, T &s){
-
-  RealD mz = (RealD)abs(z);
-  
-  if(mz==0.0){
-    c = 1; s = 0;
-  }
-  if(mz >= (RealD)abs(y)){
-    T t = -y/z;
-    s = (T)1.0 / sqrt ((T)1.0 + t * t);
-    c = s * t;
-  } else {
-    T t = -z/y;
-    c = (T)1.0 / sqrt ((T)1.0 + t * t);
-    s = c * t;
-  }
-}
-
-template <class T> inline void Givens_mult(DenseMatrix<T> &A,  int i, int k, T c, T s, int dir)
-{
-  int q ; SizeSquare(A,q);
-
-  if(dir == 0){
-    for(int j=0;j<q;j++){
-      T nu = A[i][j];
-      T w  = A[k][j];
-      A[i][j] = (c*nu + s*w);
-      A[k][j] = (-s*nu + c*w);
-    }
-  }
-
-  if(dir == 1){
-    for(int j=0;j<q;j++){
-      T nu = A[j][i];
-      T w  = A[j][k];
-      A[j][i] = (c*nu - s*w);
-      A[j][k] = (s*nu + c*w);
-    }
-  }
-}
-
-/**
-	from input = x;
-	Compute the complex Householder vector, v, such that
-	P = (I - b v transpose(v) )
-	b = 2/v.v
-
-	P | x |    | x | k = 0
-	| x |    | 0 | 
-	| x | =  | 0 |
-	| x |    | 0 | j = 3
-	| x |	   | x |
-
-	These are the "Unreduced" Householder vectors.
-
- **/
-template <class T> inline void Householder_vector(DenseVector<T> input, int k, int j, DenseVector<T> &v, T &beta)
-{
-  int N ; Size(input,N);
-  T m = *max_element(input.begin() + k, input.begin() + j + 1, cf<T> );
-
-  if(abs(m) > 0.0){
-    T alpha = 0;
-
-    for(int i=k; i<j+1; i++){
-      v[i] = input[i]/m;
-      alpha = alpha + v[i]*conj(v[i]);
-    }
-    alpha = sqrt(alpha);
-    beta = (T)1.0/(alpha*(alpha + abs(v[k]) ));
-
-    if(abs(v[k]) > 0.0)  v[k] = v[k] + (v[k]/abs(v[k]))*alpha;
-    else                 v[k] = -alpha;
-  } else{
-    for(int i=k; i<j+1; i++){
-      v[i] = 0.0;
-    } 
-  }
-}
-
-/**
-	from input = x;
-	Compute the complex Householder vector, v, such that
-	P = (I - b v transpose(v) )
-	b = 2/v.v
-
-	Px = alpha*e_dir
-
-	These are the "Unreduced" Householder vectors.
-
- **/
-
-template <class T> inline void Householder_vector(DenseVector<T> input, int k, int j, int dir, DenseVector<T> &v, T &beta)
-{
-  int N = input.size();
-  T m = *max_element(input.begin() + k, input.begin() + j + 1, cf);
-  
-  if(abs(m) > 0.0){
-    T alpha = 0;
-
-    for(int i=k; i<j+1; i++){
-      v[i] = input[i]/m;
-      alpha = alpha + v[i]*conj(v[i]);
-    }
-    
-    alpha = sqrt(alpha);
-    beta = 1.0/(alpha*(alpha + abs(v[dir]) ));
-	
-    if(abs(v[dir]) > 0.0) v[dir] = v[dir] + (v[dir]/abs(v[dir]))*alpha;
-    else                  v[dir] = -alpha;
-  }else{
-    for(int i=k; i<j+1; i++){
-      v[i] = 0.0;
-    } 
-  }
-}
-
-/**
-	Compute the product PA if trans = 0
-	AP if trans = 1
-	P = (I - b v transpose(v) )
-	b = 2/v.v
-	start at element l of matrix A
-	v is of length j - k + 1 of v are nonzero
- **/
-
-template <class T> inline void Householder_mult(DenseMatrix<T> &A , DenseVector<T> v, T beta, int l, int k, int j, int trans)
-{
-  int N ; SizeSquare(A,N);
-
-  if(abs(beta) > 0.0){
-    for(int p=l; p<N; p++){
-      T s = 0;
-      if(trans==0){
-	for(int i=k;i<j+1;i++) s += conj(v[i-k])*A[i][p];
-	s *= beta;
-	for(int i=k;i<j+1;i++){ A[i][p] = A[i][p]-s*conj(v[i-k]);}
-      } else {
-	for(int i=k;i<j+1;i++){ s += conj(v[i-k])*A[p][i];}
-	s *= beta;
-	for(int i=k;i<j+1;i++){ A[p][i]=A[p][i]-s*conj(v[i-k]);}
-      }
-    }
-  }
-}
-
-/**
-	Compute the product PA if trans = 0
-	AP if trans = 1
-	P = (I - b v transpose(v) )
-	b = 2/v.v
-	start at element l of matrix A
-	v is of length j - k + 1 of v are nonzero
-	A is tridiagonal
- **/
-template <class T> inline void Householder_mult_tri(DenseMatrix<T> &A , DenseVector<T> v, T beta, int l, int M, int k, int j, int trans)
-{
-  if(abs(beta) > 0.0){
-
-    int N ; SizeSquare(A,N);
-
-    DenseMatrix<T> tmp; Resize(tmp,N,N); Fill(tmp,0); 
-
-    T s;
-    for(int p=l; p<M; p++){
-      s = 0;
-      if(trans==0){
-	for(int i=k;i<j+1;i++) s = s + conj(v[i-k])*A[i][p];
-      }else{
-	for(int i=k;i<j+1;i++) s = s + v[i-k]*A[p][i];
-      }
-      s = beta*s;
-      if(trans==0){
-	for(int i=k;i<j+1;i++) tmp[i][p] = tmp(i,p) - s*v[i-k];
-      }else{
-	for(int i=k;i<j+1;i++) tmp[p][i] = tmp[p][i] - s*conj(v[i-k]);
-      }
-    }
-    for(int p=l; p<M; p++){
-      if(trans==0){
-	for(int i=k;i<j+1;i++) A[i][p] = A[i][p] + tmp[i][p];
-      }else{
-	for(int i=k;i<j+1;i++) A[p][i] = A[p][i] + tmp[p][i];
-      }
-    }
-  }
-}
-}
-#endif
@@ -33,6 +33,8 @@ directory

 namespace Grid {

+enum BlockCGtype { BlockCG, BlockCGrQ, CGmultiRHS };
+
 //////////////////////////////////////////////////////////////////////////
 // Block conjugate gradient. Dimension zero should be the block direction
 //////////////////////////////////////////////////////////////////////////
@@ -40,25 +42,273 @@ template <class Field>
 class BlockConjugateGradient : public OperatorFunction<Field> {
 public:

+
  typedef typename Field::scalar_type scomplex;

-  const int blockDim = 0;
-
+  int blockDim ;
  int Nblock;
+
+  BlockCGtype CGtype;
  bool ErrorOnNoConverge;  // throw an assert when the CG fails to converge.
                           // Defaults true.
  RealD Tolerance;
  Integer MaxIterations;
  Integer IterationsToComplete; //Number of iterations the CG took to finish. Filled in upon completion
  
-  BlockConjugateGradient(RealD tol, Integer maxit, bool err_on_no_conv = true)
-    : Tolerance(tol),
-    MaxIterations(maxit),
-    ErrorOnNoConverge(err_on_no_conv){};
+  BlockConjugateGradient(BlockCGtype cgtype,int _Orthog,RealD tol, Integer maxit, bool err_on_no_conv = true)
+    : Tolerance(tol), CGtype(cgtype),   blockDim(_Orthog),  MaxIterations(maxit), ErrorOnNoConverge(err_on_no_conv)
+  {};

+////////////////////////////////////////////////////////////////////////////////////////////////////
+// Thin QR factorisation (google it)
+////////////////////////////////////////////////////////////////////////////////////////////////////
+void ThinQRfact (Eigen::MatrixXcd &m_rr,
+		 Eigen::MatrixXcd &C,
+		 Eigen::MatrixXcd &Cinv,
+		 Field & Q,
+		 const Field & R)
+{
+  int Orthog = blockDim; // First dimension is block dim; this is an assumption
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  //Dimensions
+  // R_{ferm x Nblock} =  Q_{ferm x Nblock} x  C_{Nblock x Nblock} -> ferm x Nblock
+  //
+  // Rdag R = m_rr = Herm = L L^dag        <-- Cholesky decomposition (LLT routine in Eigen)
+  //
+  //   Q  C = R => Q = R C^{-1}
+  //
+  // Want  Ident = Q^dag Q = C^{-dag} R^dag R C^{-1} = C^{-dag} L L^dag C^{-1} = 1_{Nblock x Nblock} 
+  //
+  // Set C = L^{dag}, and then Q^dag Q = ident 
+  //
+  // Checks:
+  // Cdag C = Rdag R ; passes.
+  // QdagQ  = 1      ; passes
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  sliceInnerProductMatrix(m_rr,R,R,Orthog);
+
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  // Cholesky from Eigen
+  // There exists a ldlt that is documented as more stable
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  Eigen::MatrixXcd L    = m_rr.llt().matrixL(); 
+
+  C    = L.adjoint();
+  Cinv = C.inverse();
+
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  // Q = R C^{-1}
+  //
+  // Q_j  = R_i Cinv(i,j) 
+  //
+  // NB maddMatrix conventions are Right multiplication X[j] a[j,i] already
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  // FIXME:: make a sliceMulMatrix to avoid zero vector
+  sliceMulMatrix(Q,Cinv,R,Orthog);
+}
+////////////////////////////////////////////////////////////////////////////////////////////////////
+// Call one of several implementations
+////////////////////////////////////////////////////////////////////////////////////////////////////
 void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi) 
 {
-  int Orthog = 0; // First dimension is block dim
+  if ( CGtype == BlockCGrQ ) {
+    BlockCGrQsolve(Linop,Src,Psi);
+  } else if (CGtype == BlockCG ) {
+    BlockCGsolve(Linop,Src,Psi);
+  } else if (CGtype == CGmultiRHS ) {
+    CGmultiRHSsolve(Linop,Src,Psi);
+  } else {
+    assert(0);
+  }
+}
+
+////////////////////////////////////////////////////////////////////////////
+// BlockCGrQ implementation:
+//--------------------------
+// X is guess/Solution
+// B is RHS
+// Solve A X_i = B_i    ;        i refers to Nblock index
+////////////////////////////////////////////////////////////////////////////
+void BlockCGrQsolve(LinearOperatorBase<Field> &Linop, const Field &B, Field &X) 
+{
+  int Orthog = blockDim; // First dimension is block dim; this is an assumption
+  Nblock = B._grid->_fdimensions[Orthog];
+
+  std::cout<<GridLogMessage<<" Block Conjugate Gradient : Orthog "<<Orthog<<" Nblock "<<Nblock<<std::endl;
+
+  X.checkerboard = B.checkerboard;
+  conformable(X, B);
+
+  Field tmp(B);
+  Field Q(B);
+  Field D(B);
+  Field Z(B);
+  Field AD(B);
+
+  Eigen::MatrixXcd m_DZ     = Eigen::MatrixXcd::Identity(Nblock,Nblock);
+  Eigen::MatrixXcd m_M      = Eigen::MatrixXcd::Identity(Nblock,Nblock);
+  Eigen::MatrixXcd m_rr     = Eigen::MatrixXcd::Zero(Nblock,Nblock);
+
+  Eigen::MatrixXcd m_C      = Eigen::MatrixXcd::Zero(Nblock,Nblock);
+  Eigen::MatrixXcd m_Cinv   = Eigen::MatrixXcd::Zero(Nblock,Nblock);
+  Eigen::MatrixXcd m_S      = Eigen::MatrixXcd::Zero(Nblock,Nblock);
+  Eigen::MatrixXcd m_Sinv   = Eigen::MatrixXcd::Zero(Nblock,Nblock);
+
+  Eigen::MatrixXcd m_tmp    = Eigen::MatrixXcd::Identity(Nblock,Nblock);
+  Eigen::MatrixXcd m_tmp1   = Eigen::MatrixXcd::Identity(Nblock,Nblock);
+
+  // Initial residual computation & set up
+  std::vector<RealD> residuals(Nblock);
+  std::vector<RealD> ssq(Nblock);
+
+  sliceNorm(ssq,B,Orthog);
+  RealD sssum=0;
+  for(int b=0;b<Nblock;b++) sssum+=ssq[b];
+
+  sliceNorm(residuals,B,Orthog);
+  for(int b=0;b<Nblock;b++){ assert(std::isnan(residuals[b])==0); }
+
+  sliceNorm(residuals,X,Orthog);
+  for(int b=0;b<Nblock;b++){ assert(std::isnan(residuals[b])==0); }
+
+  /************************************************************************
+   * Block conjugate gradient rQ (Sebastien Birk Thesis, after Dubrulle 2001)
+   ************************************************************************
+   * Dimensions:
+   *
+   *   X,B==(Nferm x Nblock)
+   *   A==(Nferm x Nferm)
+   *  
+   * Nferm = Nspin x Ncolour x Ncomplex x Nlattice_site
+   * 
+   * QC = R = B-AX, D = Q     ; QC => Thin QR factorisation (google it)
+   * for k: 
+   *   Z  = AD
+   *   M  = [D^dag Z]^{-1}
+   *   X  = X + D MC
+   *   QS = Q - ZM
+   *   D  = Q + D S^dag
+   *   C  = S C
+   */
+  ///////////////////////////////////////
+  // Initial block: initial search dir is guess
+  ///////////////////////////////////////
+  std::cout << GridLogMessage<<"BlockCGrQ algorithm initialisation " <<std::endl;
+
+  //1.  QC = R = B-AX, D = Q     ; QC => Thin QR factorisation (google it)
+
+  Linop.HermOp(X, AD);
+  tmp = B - AD;  
+  ThinQRfact (m_rr, m_C, m_Cinv, Q, tmp);
+  D=Q;
+
+  std::cout << GridLogMessage<<"BlockCGrQ computed initial residual and QR fact " <<std::endl;
+
+  ///////////////////////////////////////
+  // Timers
+  ///////////////////////////////////////
+  GridStopWatch sliceInnerTimer;
+  GridStopWatch sliceMaddTimer;
+  GridStopWatch QRTimer;
+  GridStopWatch MatrixTimer;
+  GridStopWatch SolverTimer;
+  SolverTimer.Start();
+
+  int k;
+  for (k = 1; k <= MaxIterations; k++) {
+
+    //3. Z  = AD
+    MatrixTimer.Start();
+    Linop.HermOp(D, Z);      
+    MatrixTimer.Stop();
+
+    //4. M  = [D^dag Z]^{-1}
+    sliceInnerTimer.Start();
+    sliceInnerProductMatrix(m_DZ,D,Z,Orthog);
+    sliceInnerTimer.Stop();
+    m_M       = m_DZ.inverse();
+
+    //5. X  = X + D MC
+    m_tmp     = m_M * m_C;
+    sliceMaddTimer.Start();
+    sliceMaddMatrix(X,m_tmp, D,X,Orthog);     
+    sliceMaddTimer.Stop();
+
+    //6. QS = Q - ZM
+    sliceMaddTimer.Start();
+    sliceMaddMatrix(tmp,m_M,Z,Q,Orthog,-1.0);
+    sliceMaddTimer.Stop();
+    QRTimer.Start();
+    ThinQRfact (m_rr, m_S, m_Sinv, Q, tmp);
+    QRTimer.Stop();
+    
+    //7. D  = Q + D S^dag
+    m_tmp = m_S.adjoint();
+    sliceMaddTimer.Start();
+    sliceMaddMatrix(D,m_tmp,D,Q,Orthog);
+    sliceMaddTimer.Stop();
+
+    //8. C  = S C
+    m_C = m_S*m_C;
+    
+    /*********************
+     * convergence monitor
+     *********************
+     */
+    m_rr = m_C.adjoint() * m_C;
+
+    RealD max_resid=0;
+    RealD rrsum=0;
+    RealD rr;
+
+    for(int b=0;b<Nblock;b++) {
+      rrsum+=real(m_rr(b,b));
+      rr = real(m_rr(b,b))/ssq[b];
+      if ( rr > max_resid ) max_resid = rr;
+    }
+
+    std::cout << GridLogIterative << "\titeration "<<k<<" rr_sum "<<rrsum<<" ssq_sum "<< sssum
+	      <<" ave "<<std::sqrt(rrsum/sssum) << " max "<< max_resid <<std::endl;
+
+    if ( max_resid < Tolerance*Tolerance ) { 
+
+      SolverTimer.Stop();
+
+      std::cout << GridLogMessage<<"BlockCGrQ converged in "<<k<<" iterations"<<std::endl;
+
+      for(int b=0;b<Nblock;b++){
+	std::cout << GridLogMessage<< "\t\tblock "<<b<<" computed resid "
+		  << std::sqrt(real(m_rr(b,b))/ssq[b])<<std::endl;
+      }
+      std::cout << GridLogMessage<<"\tMax residual is "<<std::sqrt(max_resid)<<std::endl;
+
+      Linop.HermOp(X, AD);
+      AD = AD-B;
+      std::cout << GridLogMessage <<"\t True residual is " << std::sqrt(norm2(AD)/norm2(B)) <<std::endl;
+
+      std::cout << GridLogMessage << "Time Breakdown "<<std::endl;
+      std::cout << GridLogMessage << "\tElapsed    " << SolverTimer.Elapsed()     <<std::endl;
+      std::cout << GridLogMessage << "\tMatrix     " << MatrixTimer.Elapsed()     <<std::endl;
+      std::cout << GridLogMessage << "\tInnerProd  " << sliceInnerTimer.Elapsed() <<std::endl;
+      std::cout << GridLogMessage << "\tMaddMatrix " << sliceMaddTimer.Elapsed()  <<std::endl;
+      std::cout << GridLogMessage << "\tThinQRfact " << QRTimer.Elapsed()  <<std::endl;
+	    
+      IterationsToComplete = k;
+      return;
+    }
+
+  }
+  std::cout << GridLogMessage << "BlockConjugateGradient(rQ) did NOT converge" << std::endl;
+
+  if (ErrorOnNoConverge) assert(0);
+  IterationsToComplete = k;
+}
+//////////////////////////////////////////////////////////////////////////
+// Block conjugate gradient; Original O'Leary Dimension zero should be the block direction
+//////////////////////////////////////////////////////////////////////////
+void BlockCGsolve(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi) 
+{
+  int Orthog = blockDim; // First dimension is block dim; this is an assumption
  Nblock = Src._grid->_fdimensions[Orthog];

  std::cout<<GridLogMessage<<" Block Conjugate Gradient : Orthog "<<Orthog<<" Nblock "<<Nblock<<std::endl;
@@ -162,8 +412,9 @@ void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi)
     *********************
     */
    RealD max_resid=0;
+    RealD rr;
    for(int b=0;b<Nblock;b++){
-      RealD rr = real(m_rr(b,b))/ssq[b];
+      rr = real(m_rr(b,b))/ssq[b];
      if ( rr > max_resid ) max_resid = rr;
    }
    
@@ -173,13 +424,14 @@ void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi)

      std::cout << GridLogMessage<<"BlockCG converged in "<<k<<" iterations"<<std::endl;
      for(int b=0;b<Nblock;b++){
-	std::cout << GridLogMessage<< "\t\tblock "<<b<<" resid "<< std::sqrt(real(m_rr(b,b))/ssq[b])<<std::endl;
+	std::cout << GridLogMessage<< "\t\tblock "<<b<<" computed resid "
+		  << std::sqrt(real(m_rr(b,b))/ssq[b])<<std::endl;
      }
      std::cout << GridLogMessage<<"\tMax residual is "<<std::sqrt(max_resid)<<std::endl;

      Linop.HermOp(Psi, AP);
      AP = AP-Src;
-      std::cout << GridLogMessage <<"\tTrue residual is " << std::sqrt(norm2(AP)/norm2(Src)) <<std::endl;
+      std::cout << GridLogMessage <<"\t True residual is " << std::sqrt(norm2(AP)/norm2(Src)) <<std::endl;

      std::cout << GridLogMessage << "Time Breakdown "<<std::endl;
      std::cout << GridLogMessage << "\tElapsed    " << SolverTimer.Elapsed()     <<std::endl;
@@ -197,35 +449,13 @@ void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi)
  if (ErrorOnNoConverge) assert(0);
  IterationsToComplete = k;
 }
-};
-
-
 //////////////////////////////////////////////////////////////////////////
 // multiRHS conjugate gradient. Dimension zero should be the block direction
+// Use this for spread out across nodes
 //////////////////////////////////////////////////////////////////////////
-template <class Field>
-class MultiRHSConjugateGradient : public OperatorFunction<Field> {
- public:
-
-  typedef typename Field::scalar_type scomplex;
-
-  const int blockDim = 0;
-
-  int Nblock;
-  bool ErrorOnNoConverge;  // throw an assert when the CG fails to converge.
-                           // Defaults true.
-  RealD Tolerance;
-  Integer MaxIterations;
-  Integer IterationsToComplete; //Number of iterations the CG took to finish. Filled in upon completion
-  
-   MultiRHSConjugateGradient(RealD tol, Integer maxit, bool err_on_no_conv = true)
-    : Tolerance(tol),
-    MaxIterations(maxit),
-    ErrorOnNoConverge(err_on_no_conv){};
-
-void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi) 
+void CGmultiRHSsolve(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi) 
 {
-  int Orthog = 0; // First dimension is block dim
+  int Orthog = blockDim; // First dimension is block dim
  Nblock = Src._grid->_fdimensions[Orthog];

  std::cout<<GridLogMessage<<"MultiRHS Conjugate Gradient : Orthog "<<Orthog<<" Nblock "<<Nblock<<std::endl;
@@ -285,12 +515,10 @@ void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi)
    MatrixTimer.Stop();

    // Alpha
-    //    sliceInnerProductVectorTest(v_pAp_test,P,AP,Orthog);
    sliceInnerTimer.Start();
    sliceInnerProductVector(v_pAp,P,AP,Orthog);
    sliceInnerTimer.Stop();
    for(int b=0;b<Nblock;b++){
-      //      std::cout << " "<< v_pAp[b]<<" "<< v_pAp_test[b]<<std::endl;
      v_alpha[b] = v_rr[b]/real(v_pAp[b]);
    }

@@ -332,7 +560,7 @@ void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi)

      std::cout << GridLogMessage<<"MultiRHS solver converged in " <<k<<" iterations"<<std::endl;
      for(int b=0;b<Nblock;b++){
-	std::cout << GridLogMessage<< "\t\tBlock "<<b<<" resid "<< std::sqrt(v_rr[b]/ssq[b])<<std::endl;
+	std::cout << GridLogMessage<< "\t\tBlock "<<b<<" computed resid "<< std::sqrt(v_rr[b]/ssq[b])<<std::endl;
      }
      std::cout << GridLogMessage<<"\tMax residual is "<<std::sqrt(max_resid)<<std::endl;

@@ -358,9 +586,8 @@ void operator()(LinearOperatorBase<Field> &Linop, const Field &Src, Field &Psi)
  if (ErrorOnNoConverge) assert(0);
  IterationsToComplete = k;
 }
+
 };

-
-
 }
 #endif
@@ -123,8 +123,11 @@ class ConjugateGradient : public OperatorFunction<Field> {
      p = p * b + r;

      LinalgTimer.Stop();
+
      std::cout << GridLogIterative << "ConjugateGradient: Iteration " << k
                << " residual " << cp << " target " << rsq << std::endl;
+      std::cout << GridLogDebug << "a = "<< a << " b_pred = "<< b_pred << "  b = "<< b << std::endl;
+      std::cout << GridLogDebug << "qq = "<< qq << " d = "<< d << "  c = "<< c << std::endl;

      // Stopping condition
      if (cp <= rsq) {
@@ -132,8 +135,6 @@ class ConjugateGradient : public OperatorFunction<Field> {
        Linop.HermOpAndNorm(psi, mmp, d, qq);
        p = mmp - src;

-        RealD mmpnorm = sqrt(norm2(mmp));
-        RealD psinorm = sqrt(norm2(psi));
        RealD srcnorm = sqrt(norm2(src));
        RealD resnorm = sqrt(norm2(p));
        RealD true_residual = resnorm / srcnorm;
@@ -157,8 +158,10 @@ class ConjugateGradient : public OperatorFunction<Field> {
    }
    std::cout << GridLogMessage << "ConjugateGradient did NOT converge"
              << std::endl;
+
    if (ErrorOnNoConverge) assert(0);
    IterationsToComplete = k;
+
  }
 };
 }
@@ -1,81 +0,0 @@
-    /*************************************************************************************
-
-    Grid physics library, www.github.com/paboyle/Grid 
-
-    Source file: ./lib/algorithms/iterative/EigenSort.h
-
-    Copyright (C) 2015
-
-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-
-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
-
-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
-
-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
-#ifndef GRID_EIGENSORT_H
-#define GRID_EIGENSORT_H
-
-
-namespace Grid {
-    /////////////////////////////////////////////////////////////
-    // Eigen sorter to begin with
-    /////////////////////////////////////////////////////////////
-
-template<class Field>
-class SortEigen {
- private:
-  
-//hacking for testing for now
- private:
-  static bool less_lmd(RealD left,RealD right){
-    return left > right;
-  }  
-  static bool less_pair(std::pair<RealD,Field const*>& left,
-                        std::pair<RealD,Field const*>& right){
-    return left.first > (right.first);
-  }  
-  
-  
- public:
-
-  void push(DenseVector<RealD>& lmd,
-            DenseVector<Field>& evec,int N) {
-    DenseVector<Field> cpy(lmd.size(),evec[0]._grid);
-    for(int i=0;i<lmd.size();i++) cpy[i] = evec[i];
-    
-    DenseVector<std::pair<RealD, Field const*> > emod(lmd.size());    
-    for(int i=0;i<lmd.size();++i)
-      emod[i] = std::pair<RealD,Field const*>(lmd[i],&cpy[i]);
-
-    partial_sort(emod.begin(),emod.begin()+N,emod.end(),less_pair);
-
-    typename DenseVector<std::pair<RealD, Field const*> >::iterator it = emod.begin();
-    for(int i=0;i<N;++i){
-      lmd[i]=it->first;
-      evec[i]=*(it->second);
-      ++it;
-    }
-  }
-  void push(DenseVector<RealD>& lmd,int N) {
-    std::partial_sort(lmd.begin(),lmd.begin()+N,lmd.end(),less_lmd);
-  }
-  bool saturated(RealD lmd, RealD thrs) {
-    return fabs(lmd) > fabs(thrs);
-  }
-};
-
-}
-#endif
@@ -98,7 +98,14 @@ public:
 #else
    if ( ptr == (_Tp *) NULL ) ptr = (_Tp *) memalign(128,bytes);
 #endif
-
+    // First touch optimise in threaded loop
+    uint8_t *cp = (uint8_t *)ptr;
+#ifdef GRID_OMP
+#pragma omp parallel for
+#endif
+    for(size_type n=0;n<bytes;n+=4096){
+      cp[n]=0;
+    }
    return ptr;
  }

@@ -186,6 +193,13 @@ public:
 #else
    _Tp * ptr = (_Tp *) memalign(128,__n*sizeof(_Tp));
 #endif
+    size_type bytes = __n*sizeof(_Tp);
+    uint8_t *cp = (uint8_t *)ptr;
+    // One touch per 4k page, static OMP loop to catch same loop order
+#pragma omp parallel for schedule(static)
+    for(size_type n=0;n<bytes;n+=4096){
+      cp[n]=0;
+    }
    return ptr;
  }
  void deallocate(pointer __p, size_type) { 
@@ -6,8 +6,9 @@

    Copyright (C) 2015

-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-Author: paboyle <paboyle@ph.ed.ac.uk>
+    Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+    Author: paboyle <paboyle@ph.ed.ac.uk>
+    Author: Guido Cossu <guido.cossu@ed.ac.uk>

    This program is free software; you can redistribute it and/or modify
    it under the terms of the GNU General Public License as published by
@@ -49,7 +50,6 @@ public:

    GridBase(const std::vector<int> & processor_grid) : CartesianCommunicator(processor_grid) {};

-
    // Physics Grid information.
    std::vector<int> _simd_layout;// Which dimensions get relayed out over simd lanes.
    std::vector<int> _fdimensions;// (full) Global dimensions of array prior to cb removal
@@ -62,13 +62,12 @@ public:
    int _isites;
    int _fsites;                  // _isites*_osites = product(dimensions).
    int _gsites;
-    std::vector<int> _slice_block;   // subslice information
+    std::vector<int> _slice_block;// subslice information
    std::vector<int> _slice_stride;
    std::vector<int> _slice_nblock;

-    // Might need these at some point
-    //    std::vector<int> _lstart;     // local start of array in gcoors. _processor_coor[d]*_ldimensions[d]
-    //    std::vector<int> _lend;       // local end of array in gcoors    _processor_coor[d]*_ldimensions[d]+_ldimensions_[d]-1
+    std::vector<int> _lstart;     // local start of array in gcoors _processor_coor[d]*_ldimensions[d]
+    std::vector<int> _lend  ;     // local end of array in gcoors   _processor_coor[d]*_ldimensions[d]+_ldimensions_[d]-1

 public:

@@ -99,7 +98,7 @@ public:
    virtual int oIndex(std::vector<int> &coor)
    {
        int idx=0;
-	// Works with either global or local coordinates
+        // Works with either global or local coordinates
        for(int d=0;d<_ndimension;d++) idx+=_ostride[d]*(coor[d]%_rdimensions[d]);
        return idx;
    }
@@ -121,6 +120,12 @@ public:
      Lexicographic::CoorFromIndex(coor,Oindex,_rdimensions);
    }

+    inline void InOutCoorToLocalCoor (std::vector<int> &ocoor, std::vector<int> &icoor, std::vector<int> &lcoor) {
+      lcoor.resize(_ndimension);
+      for (int d = 0; d < _ndimension; d++)
+        lcoor[d] = ocoor[d] + _rdimensions[d] * icoor[d];
+    }
+
    //////////////////////////////////////////////////////////
    // SIMD lane addressing
    //////////////////////////////////////////////////////////
@@ -128,6 +133,7 @@ public:
    {
      Lexicographic::CoorFromIndex(coor,lane,_simd_layout);
    }
+
    inline int PermuteDim(int dimension){
      return _simd_layout[dimension]>1;
    }
@@ -145,15 +151,15 @@ public:
      // Distance should be either 0,1,2..
      //
      if ( _simd_layout[dimension] > 2 ) { 
-	for(int d=0;d<_ndimension;d++){
-	  if ( d != dimension ) assert ( (_simd_layout[d]==1)  );
-	}
-	permute_type = RotateBit; // How to specify distance; this is not just direction.
-	return permute_type;
+        for(int d=0;d<_ndimension;d++){
+          if ( d != dimension ) assert ( (_simd_layout[d]==1)  );
+        }
+        permute_type = RotateBit; // How to specify distance; this is not just direction.
+        return permute_type;
      }

      for(int d=_ndimension-1;d>dimension;d--){
-	if (_simd_layout[d]>1 ) permute_type++;
+        if (_simd_layout[d]>1 ) permute_type++;
      }
      return permute_type;
    }
@@ -168,11 +174,30 @@ public:
    inline int gSites(void) const { return _isites*_osites*_Nprocessors; }; 
    inline int Nd    (void) const { return _ndimension;};

+    inline const std::vector<int> LocalStarts(void)             { return _lstart;    };
    inline const std::vector<int> &FullDimensions(void)         { return _fdimensions;};
    inline const std::vector<int> &GlobalDimensions(void)       { return _gdimensions;};
    inline const std::vector<int> &LocalDimensions(void)        { return _ldimensions;};
    inline const std::vector<int> &VirtualLocalDimensions(void) { return _ldimensions;};

+    ////////////////////////////////////////////////////////////////
+    // Utility to print the full decomposition details 
+    ////////////////////////////////////////////////////////////////
+
+    void show_decomposition(){
+      std::cout << GridLogMessage << "Full Dimensions    : " << _fdimensions << std::endl;
+      std::cout << GridLogMessage << "Global Dimensions  : " << _gdimensions << std::endl;
+      std::cout << GridLogMessage << "Local Dimensions   : " << _ldimensions << std::endl;
+      std::cout << GridLogMessage << "Reduced Dimensions : " << _rdimensions << std::endl;
+      std::cout << GridLogMessage << "Outer strides      : " << _ostride << std::endl;
+      std::cout << GridLogMessage << "Inner strides      : " << _istride << std::endl;
+      std::cout << GridLogMessage << "iSites             : " << _isites << std::endl;
+      std::cout << GridLogMessage << "oSites             : " << _osites << std::endl;
+      std::cout << GridLogMessage << "lSites             : " << lSites() << std::endl;        
+      std::cout << GridLogMessage << "gSites             : " << gSites() << std::endl;
+      std::cout << GridLogMessage << "Nd                 : " << _ndimension << std::endl;             
+    } 
+
    ////////////////////////////////////////////////////////////////
    // Global addressing
    ////////////////////////////////////////////////////////////////
@@ -184,12 +209,15 @@ public:
      assert(lidx<lSites());
      Lexicographic::CoorFromIndex(lcoor,lidx,_ldimensions);
    }
+
+
+
    void GlobalCoorToGlobalIndex(const std::vector<int> & gcoor,int & gidx){
      gidx=0;
      int mult=1;
      for(int mu=0;mu<_ndimension;mu++) {
-	gidx+=mult*gcoor[mu];
-	mult*=_gdimensions[mu];
+        gidx+=mult*gcoor[mu];
+        mult*=_gdimensions[mu];
      }
    }
    void GlobalCoorToProcessorCoorLocalCoor(std::vector<int> &pcoor,std::vector<int> &lcoor,const std::vector<int> &gcoor)
@@ -197,9 +225,9 @@ public:
      pcoor.resize(_ndimension);
      lcoor.resize(_ndimension);
      for(int mu=0;mu<_ndimension;mu++){
-	int _fld  = _fdimensions[mu]/_processors[mu];
-	pcoor[mu] = gcoor[mu]/_fld;
-	lcoor[mu] = gcoor[mu]%_fld;
+        int _fld  = _fdimensions[mu]/_processors[mu];
+        pcoor[mu] = gcoor[mu]/_fld;
+        lcoor[mu] = gcoor[mu]%_fld;
      }
    }
    void GlobalCoorToRankIndex(int &rank, int &o_idx, int &i_idx ,const std::vector<int> &gcoor)
@@ -211,9 +239,9 @@ public:
      /*
      std::vector<int> cblcoor(lcoor);
      for(int d=0;d<cblcoor.size();d++){
-	if( this->CheckerBoarded(d) ) {
-	  cblcoor[d] = lcoor[d]/2;
-	}
+        if( this->CheckerBoarded(d) ) {
+          cblcoor[d] = lcoor[d]/2;
+        }
      }
      */
      i_idx= iIndex(lcoor);
@@ -239,7 +267,7 @@ public:
    {
      RankIndexToGlobalCoor(rank,o_idx,i_idx ,fcoor);
      if(CheckerBoarded(0)){
-	fcoor[0] = fcoor[0]*2+cb;
+        fcoor[0] = fcoor[0]*2+cb;
      }
    }
    void ProcessorCoorLocalCoorToGlobalCoor(std::vector<int> &Pcoor,std::vector<int> &Lcoor,std::vector<int> &gcoor)
@@ -76,6 +76,8 @@ public:
        _ldimensions.resize(_ndimension);
        _rdimensions.resize(_ndimension);
        _simd_layout.resize(_ndimension);
+	_lstart.resize(_ndimension);
+	_lend.resize(_ndimension);
            
        _ostride.resize(_ndimension);
        _istride.resize(_ndimension);
@@ -94,8 +96,10 @@ public:
 	  // Use a reduced simd grid
 	  _ldimensions[d]= _gdimensions[d]/_processors[d];  //local dimensions
 	  _rdimensions[d]= _ldimensions[d]/_simd_layout[d]; //overdecomposition
-	  _osites *= _rdimensions[d];
-	  _isites *= _simd_layout[d];
+	  _lstart[d]     = _processor_coor[d]*_ldimensions[d];
+	  _lend[d]       = _processor_coor[d]*_ldimensions[d]+_ldimensions[d]-1;
+	  _osites  *= _rdimensions[d];
+	  _isites  *= _simd_layout[d];
                
 	  // Addressing support
 	  if ( d==0 ) {
@@ -151,6 +151,8 @@ public:
      _ldimensions.resize(_ndimension);
      _rdimensions.resize(_ndimension);
      _simd_layout.resize(_ndimension);
+      _lstart.resize(_ndimension);
+      _lend.resize(_ndimension);
      
      _ostride.resize(_ndimension);
      _istride.resize(_ndimension);
@@ -169,6 +171,8 @@ public:
 	  _gdimensions[d] = _gdimensions[d]/2; // Remove a checkerboard
 	}
 	_ldimensions[d] = _gdimensions[d]/_processors[d];
+	_lstart[d]     = _processor_coor[d]*_ldimensions[d];
+	_lend[d]       = _processor_coor[d]*_ldimensions[d]+_ldimensions[d]-1;

 	// Use a reduced simd grid
 	_simd_layout[d] = simd_layout[d];
@@ -60,6 +60,7 @@ void CartesianCommunicator::ShmBufferFreeAll(void) {
 /////////////////////////////////
 // Grid information queries
 /////////////////////////////////
+int                      CartesianCommunicator::Dimensions(void)         { return _ndimension; };
 int                      CartesianCommunicator::IsBoss(void)            { return _processor==0; };
 int                      CartesianCommunicator::BossRank(void)          { return 0; };
 int                      CartesianCommunicator::ThisRank(void)          { return _processor; };
@@ -91,6 +92,7 @@ void CartesianCommunicator::GlobalSumVector(ComplexD *c,int N)
 #if !defined( GRID_COMMS_MPI3) && !defined (GRID_COMMS_MPI3L)

 int                      CartesianCommunicator::NodeCount(void)    { return ProcessorCount();};
+int                      CartesianCommunicator::RankCount(void)    { return ProcessorCount();};

 double CartesianCommunicator::StencilSendToRecvFromBegin(std::vector<CommsRequest_t> &list,
 						       void *xmit,
@@ -148,6 +148,7 @@ class CartesianCommunicator {
  int  RankFromProcessorCoor(std::vector<int> &coor);
  void ProcessorCoorFromRank(int rank,std::vector<int> &coor);
  
+  int                      Dimensions(void)        ;
  int                      IsBoss(void)            ;
  int                      BossRank(void)          ;
  int                      ThisRank(void)          ;
@@ -155,6 +156,7 @@ class CartesianCommunicator {
  const std::vector<int> & ProcessorGrid(void)     ;
  int                      ProcessorCount(void)    ;
  int                      NodeCount(void)    ;
+  int                      RankCount(void)    ;

  ////////////////////////////////////////////////////////////////////////////////
  // very VERY rarely (Log, serial RNG) we need world without a grid
@@ -175,6 +177,8 @@ class CartesianCommunicator {
  void GlobalSumVector(ComplexF *c,int N);
  void GlobalSum(ComplexD &c);
  void GlobalSumVector(ComplexD *c,int N);
+  void GlobalXOR(uint32_t &);
+  void GlobalXOR(uint64_t &);
  
  template<class obj> void GlobalSum(obj &o){
    typedef typename obj::scalar_type scalar_type;
@@ -83,6 +83,14 @@ void CartesianCommunicator::GlobalSum(uint64_t &u){
  int ierr=MPI_Allreduce(MPI_IN_PLACE,&u,1,MPI_UINT64_T,MPI_SUM,communicator);
  assert(ierr==0);
 }
+void CartesianCommunicator::GlobalXOR(uint32_t &u){
+  int ierr=MPI_Allreduce(MPI_IN_PLACE,&u,1,MPI_UINT32_T,MPI_BXOR,communicator);
+  assert(ierr==0);
+}
+void CartesianCommunicator::GlobalXOR(uint64_t &u){
+  int ierr=MPI_Allreduce(MPI_IN_PLACE,&u,1,MPI_UINT64_T,MPI_BXOR,communicator);
+  assert(ierr==0);
+}
 void CartesianCommunicator::GlobalSum(float &f){
  int ierr=MPI_Allreduce(MPI_IN_PLACE,&f,1,MPI_FLOAT,MPI_SUM,communicator);
  assert(ierr==0);
@@ -37,7 +37,10 @@ Author: Peter Boyle <paboyle@ph.ed.ac.uk>
 #include <sys/ipc.h>
 #include <sys/shm.h>
 #include <sys/mman.h>
-//#include <zlib.h>
+#include <zlib.h>
+#ifdef HAVE_NUMAIF_H
+#include <numaif.h>
+#endif
 #ifndef SHM_HUGETLB
 #define SHM_HUGETLB 04000
 #endif
@@ -65,6 +68,7 @@ std::vector<int> CartesianCommunicator::MyGroup;
 std::vector<void *> CartesianCommunicator::ShmCommBufs;

 int CartesianCommunicator::NodeCount(void)    { return GroupSize;};
+int CartesianCommunicator::RankCount(void)    { return WorldSize;};


 #undef FORCE_COMMS
@@ -213,6 +217,25 @@ void CartesianCommunicator::Init(int *argc, char ***argv) {
      void * ptr =  mmap(NULL,size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0);
      if ( ptr == MAP_FAILED ) {       perror("failed mmap");      assert(0);    }
      assert(((uint64_t)ptr&0x3F)==0);
+
+      // Try to force numa domain on the shm segment if we have numaif.h
+#ifdef HAVE_NUMAIF_H
+	int status;
+	int flags=MPOL_MF_MOVE;
+#ifdef KNL
+	int nodes=1; // numa domain == MCDRAM
+	// Find out if in SNC2,SNC4 mode ?
+#else
+	int nodes=r; // numa domain == MPI ID
+#endif
+	unsigned long count=1;
+	for(uint64_t page=0;page<size;page+=4096){
+	  void *pages = (void *) ( page + (uint64_t)ptr );
+	  uint64_t *cow_it = (uint64_t *)pages;	*cow_it = 1;
+	  ierr= move_pages(0,count, &pages,&nodes,&status,flags);
+	  if (ierr && (page==0)) perror("numa relocate command failed");
+	}
+#endif
      ShmCommBufs[r] =ptr;
      
    }
@@ -509,6 +532,14 @@ void CartesianCommunicator::GlobalSum(uint64_t &u){
  int ierr=MPI_Allreduce(MPI_IN_PLACE,&u,1,MPI_UINT64_T,MPI_SUM,communicator);
  assert(ierr==0);
 }
+void CartesianCommunicator::GlobalXOR(uint32_t &u){
+  int ierr=MPI_Allreduce(MPI_IN_PLACE,&u,1,MPI_UINT32_T,MPI_BXOR,communicator);
+  assert(ierr==0);
+}
+void CartesianCommunicator::GlobalXOR(uint64_t &u){
+  int ierr=MPI_Allreduce(MPI_IN_PLACE,&u,1,MPI_UINT64_T,MPI_BXOR,communicator);
+  assert(ierr==0);
+}
 void CartesianCommunicator::GlobalSum(float &f){
  int ierr=MPI_Allreduce(MPI_IN_PLACE,&f,1,MPI_FLOAT,MPI_SUM,communicator);
  assert(ierr==0);
@@ -59,6 +59,8 @@ void CartesianCommunicator::GlobalSum(double &){}
 void CartesianCommunicator::GlobalSum(uint32_t &){}
 void CartesianCommunicator::GlobalSum(uint64_t &){}
 void CartesianCommunicator::GlobalSumVector(double *,int N){}
+void CartesianCommunicator::GlobalXOR(uint32_t &){}
+void CartesianCommunicator::GlobalXOR(uint64_t &){}

 void CartesianCommunicator::SendRecvPacket(void *xmit,
 					   void *recv,
@@ -30,21 +30,11 @@ Author: Peter Boyle <paboyle@ph.ed.ac.uk>

 namespace Grid {

-template<class vobj>
-class SimpleCompressor {
-public:
-  void Point(int) {};
-
-  vobj operator() (const vobj &arg) {
-    return arg;
-  }
-};
-
 ///////////////////////////////////////////////////////////////////
-// Gather for when there is no need to SIMD split with compression
+// Gather for when there is no need to SIMD split 
 ///////////////////////////////////////////////////////////////////
-template<class vobj,class cobj,class compressor> void 
-Gather_plane_simple (const Lattice<vobj> &rhs,commVector<cobj> &buffer,int dimension,int plane,int cbmask,compressor &compress, int off=0)
+template<class vobj> void 
+Gather_plane_simple (const Lattice<vobj> &rhs,commVector<vobj> &buffer,int dimension,int plane,int cbmask, int off=0)
 {
  int rd = rhs._grid->_rdimensions[dimension];

@@ -62,7 +52,7 @@ Gather_plane_simple (const Lattice<vobj> &rhs,commVector<cobj> &buffer,int dimen
      for(int b=0;b<e2;b++){
 	int o  = n*stride;
 	int bo = n*e2;
-	buffer[off+bo+b]=compress(rhs._odata[so+o+b]);
+	buffer[off+bo+b]=rhs._odata[so+o+b];
      }
    }
  } else { 
@@ -78,17 +68,16 @@ Gather_plane_simple (const Lattice<vobj> &rhs,commVector<cobj> &buffer,int dimen
       }
     }
     parallel_for(int i=0;i<table.size();i++){
-       buffer[off+table[i].first]=compress(rhs._odata[so+table[i].second]);
+       buffer[off+table[i].first]=rhs._odata[so+table[i].second];
     }
  }
 }

-
 ///////////////////////////////////////////////////////////////////
-// Gather for when there *is* need to SIMD split with compression
+// Gather for when there *is* need to SIMD split 
 ///////////////////////////////////////////////////////////////////
-template<class cobj,class vobj,class compressor> void 
-Gather_plane_extract(const Lattice<vobj> &rhs,std::vector<typename cobj::scalar_object *> pointers,int dimension,int plane,int cbmask,compressor &compress)
+template<class vobj> void 
+Gather_plane_extract(const Lattice<vobj> &rhs,std::vector<typename vobj::scalar_object *> pointers,int dimension,int plane,int cbmask)
 {
  int rd = rhs._grid->_rdimensions[dimension];

@@ -109,8 +98,8 @@ Gather_plane_extract(const Lattice<vobj> &rhs,std::vector<typename cobj::scalar_
 	int o      =   n*n1;
 	int offset = b+n*e2;
 	
-	cobj temp =compress(rhs._odata[so+o+b]);
-	extract<cobj>(temp,pointers,offset);
+	vobj temp =rhs._odata[so+o+b];
+	extract<vobj>(temp,pointers,offset);

      }
    }
@@ -127,32 +116,14 @@ Gather_plane_extract(const Lattice<vobj> &rhs,std::vector<typename cobj::scalar_
 	int offset = b+n*e2;

 	if ( ocb & cbmask ) {
-	  cobj temp =compress(rhs._odata[so+o+b]);
-	  extract<cobj>(temp,pointers,offset);
+	  vobj temp =rhs._odata[so+o+b];
+	  extract<vobj>(temp,pointers,offset);
 	}
      }
    }
  }
 }

-//////////////////////////////////////////////////////
-// Gather for when there is no need to SIMD split
-//////////////////////////////////////////////////////
-template<class vobj> void Gather_plane_simple (const Lattice<vobj> &rhs,commVector<vobj> &buffer, int dimension,int plane,int cbmask)
-{
-  SimpleCompressor<vobj> dontcompress;
-  Gather_plane_simple (rhs,buffer,dimension,plane,cbmask,dontcompress);
-}
-
-//////////////////////////////////////////////////////
-// Gather for when there *is* need to SIMD split
-//////////////////////////////////////////////////////
-template<class vobj> void Gather_plane_extract(const Lattice<vobj> &rhs,std::vector<typename vobj::scalar_object *> pointers,int dimension,int plane,int cbmask)
-{
-  SimpleCompressor<vobj> dontcompress;
-  Gather_plane_extract<vobj,vobj,decltype(dontcompress)>(rhs,pointers,dimension,plane,cbmask,dontcompress);
-}
-
 //////////////////////////////////////////////////////
 // Scatter for when there is no need to SIMD split
 //////////////////////////////////////////////////////
@@ -200,7 +171,7 @@ template<class vobj> void Scatter_plane_simple (Lattice<vobj> &rhs,commVector<vo
 //////////////////////////////////////////////////////
 // Scatter for when there *is* need to SIMD split
 //////////////////////////////////////////////////////
- template<class vobj,class cobj> void Scatter_plane_merge(Lattice<vobj> &rhs,std::vector<cobj *> pointers,int dimension,int plane,int cbmask)
+template<class vobj> void Scatter_plane_merge(Lattice<vobj> &rhs,std::vector<typename vobj::scalar_object *> pointers,int dimension,int plane,int cbmask)
 {
  int rd = rhs._grid->_rdimensions[dimension];

@@ -154,13 +154,7 @@ template<class vobj> void Cshift_comms(Lattice<vobj> &ret,const Lattice<vobj> &r
 			   recv_from_rank,
 			   bytes);
      grid->Barrier();
-      /*
-      for(int i=0;i<send_buf.size();i++){
-	assert(recv_buf.size()==buffer_size);
-	assert(send_buf.size()==buffer_size);
-	std::cout << "SendRecv_Cshift_comms ["<<i<<" "<< dimension<<"] snd "<<send_buf[i]<<" rcv " << recv_buf[i] << "  0x" << cbmask<<std::endl;
-      }
-      */
+
      Scatter_plane_simple (ret,recv_buf,dimension,x,cbmask);
    }
  }
@@ -246,13 +240,6 @@ template<class vobj> void  Cshift_comms_simd(Lattice<vobj> &ret,const Lattice<vo
 			     (void *)&recv_buf_extract[i][0],
 			     recv_from_rank,
 			     bytes);
-	/*
-	for(int w=0;w<recv_buf_extract[i].size();w++){
-	  assert(recv_buf_extract[i].size()==buffer_size);
-	  assert(send_buf_extract[i].size()==buffer_size);
-	  std::cout << "SendRecv_Cshift_comms ["<<w<<" "<< dimension<<"] recv "<<recv_buf_extract[i][w]<<" send " << send_buf_extract[nbr_lane][w]  << cbmask<<std::endl;
-	}
-	*/	
 	grid->Barrier();
 	rpointers[i] = &recv_buf_extract[i][0];
      } else { 
@@ -235,64 +235,74 @@ public:
    }
  };

-    //////////////////////////////////////////////////////////////////
-    // Constructor requires "grid" passed.
-    // what about a default grid?
-    //////////////////////////////////////////////////////////////////
-    Lattice(GridBase *grid) : _odata(grid->oSites()) {
-        _grid = grid;
+  //////////////////////////////////////////////////////////////////
+  // Constructor requires "grid" passed.
+  // what about a default grid?
+  //////////////////////////////////////////////////////////////////
+  Lattice(GridBase *grid) : _odata(grid->oSites()) {
+    _grid = grid;
    //        _odata.reserve(_grid->oSites());
    //        _odata.resize(_grid->oSites());
    //      std::cout << "Constructing lattice object with Grid pointer "<<_grid<<std::endl;
-        assert((((uint64_t)&_odata[0])&0xF) ==0);
-        checkerboard=0;
-    }
-
-    Lattice(const Lattice& r){ // copy constructor
-    	_grid = r._grid;
-    	checkerboard = r.checkerboard;
-    	_odata.resize(_grid->oSites());// essential
-	parallel_for(int ss=0;ss<_grid->oSites();ss++){
-            _odata[ss]=r._odata[ss];
-        }  	
-    }
-
-
-
-    virtual ~Lattice(void) = default;
+    assert((((uint64_t)&_odata[0])&0xF) ==0);
+    checkerboard=0;
+  }
+  
+  Lattice(const Lattice& r){ // copy constructor
+    _grid = r._grid;
+    checkerboard = r.checkerboard;
+    _odata.resize(_grid->oSites());// essential
+    parallel_for(int ss=0;ss<_grid->oSites();ss++){
+      _odata[ss]=r._odata[ss];
+    }  	
+  }
+  
+  
+  
+  virtual ~Lattice(void) = default;
    
-    template<class sobj> strong_inline Lattice<vobj> & operator = (const sobj & r){
-      parallel_for(int ss=0;ss<_grid->oSites();ss++){
-            this->_odata[ss]=r;
-        }
-        return *this;
-    }
-    template<class robj> strong_inline Lattice<vobj> & operator = (const Lattice<robj> & r){
-      this->checkerboard = r.checkerboard;
-      conformable(*this,r);
-      
-      parallel_for(int ss=0;ss<_grid->oSites();ss++){
-            this->_odata[ss]=r._odata[ss];
-        }
-        return *this;
+  void reset(GridBase* grid) {
+    if (_grid != grid) {
+      _grid = grid;
+      _odata.resize(grid->oSites());
+      checkerboard = 0;
    }
+  }
+  

-    // *=,+=,-= operators inherit behvour from correspond */+/- operation
-    template<class T> strong_inline Lattice<vobj> &operator *=(const T &r) {
-        *this = (*this)*r;
-        return *this;
+  template<class sobj> strong_inline Lattice<vobj> & operator = (const sobj & r){
+    parallel_for(int ss=0;ss<_grid->oSites();ss++){
+      this->_odata[ss]=r;
    }
-
-    template<class T> strong_inline Lattice<vobj> &operator -=(const T &r) {
-        *this = (*this)-r;
-        return *this;
+    return *this;
+  }
+  
+  template<class robj> strong_inline Lattice<vobj> & operator = (const Lattice<robj> & r){
+    this->checkerboard = r.checkerboard;
+    conformable(*this,r);
+    
+    parallel_for(int ss=0;ss<_grid->oSites();ss++){
+      this->_odata[ss]=r._odata[ss];
    }
-    template<class T> strong_inline Lattice<vobj> &operator +=(const T &r) {
-        *this = (*this)+r;
-        return *this;
-    }
- }; // class Lattice
-
+    return *this;
+  }
+  
+  // *=,+=,-= operators inherit behvour from correspond */+/- operation
+  template<class T> strong_inline Lattice<vobj> &operator *=(const T &r) {
+    *this = (*this)*r;
+    return *this;
+  }
+  
+  template<class T> strong_inline Lattice<vobj> &operator -=(const T &r) {
+    *this = (*this)-r;
+    return *this;
+  }
+  template<class T> strong_inline Lattice<vobj> &operator +=(const T &r) {
+    *this = (*this)+r;
+    return *this;
+  }
+}; // class Lattice
+  
  template<class vobj> std::ostream& operator<< (std::ostream& stream, const Lattice<vobj> &o){
    std::vector<int> gcoor;
    typedef typename vobj::scalar_object sobj;
@@ -310,7 +320,7 @@ public:
    }
    return stream;
  }
-
+  
 }


@@ -1,45 +1,37 @@
-    /*************************************************************************************
-
+/*************************************************************************************
    Grid physics library, www.github.com/paboyle/Grid 
-
    Source file: ./lib/lattice/Lattice_reduction.h
-
    Copyright (C) 2015
-
 Author: Azusa Yamaguchi <ayamaguc@staffmail.ed.ac.uk>
 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
 Author: paboyle <paboyle@ph.ed.ac.uk>
-
    This program is free software; you can redistribute it and/or modify
    it under the terms of the GNU General Public License as published by
    the Free Software Foundation; either version 2 of the License, or
    (at your option) any later version.
-
    This program is distributed in the hope that it will be useful,
    but WITHOUT ANY WARRANTY; without even the implied warranty of
    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
    GNU General Public License for more details.
-
    You should have received a copy of the GNU General Public License along
    with this program; if not, write to the Free Software Foundation, Inc.,
    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
    See the full license in the file "LICENSE" in the top level distribution directory
    *************************************************************************************/
    /*  END LEGAL */
 #ifndef GRID_LATTICE_REDUCTION_H
 #define GRID_LATTICE_REDUCTION_H

-#include <Grid/Eigen/Dense>
+#include <Grid/Grid_Eigen_Dense.h>

 namespace Grid {
 #ifdef GRID_WARN_SUBOPTIMAL
 #warning "Optimisation alert all these reduction loops are NOT threaded "
 #endif     

-    ////////////////////////////////////////////////////////////////////////////////////////////////////
-    // Deterministic Reduction operations
-    ////////////////////////////////////////////////////////////////////////////////////////////////////
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
+  // Deterministic Reduction operations
+  ////////////////////////////////////////////////////////////////////////////////////////////////////
 template<class vobj> inline RealD norm2(const Lattice<vobj> &arg){
  ComplexD nrm = innerProduct(arg,arg);
  return std::real(nrm); 
@@ -336,6 +328,8 @@ static void sliceMaddVector(Lattice<vobj> &R,std::vector<RealD> &a,const Lattice
  typedef typename vobj::vector_type vector_type;
  typedef typename vobj::tensor_reduced tensor_reduced;
  
+  scalar_type zscale(scale);
+
  GridBase *grid  = X._grid;

  int Nsimd  =grid->Nsimd();
@@ -361,7 +355,7 @@ static void sliceMaddVector(Lattice<vobj> &R,std::vector<RealD> &a,const Lattice
      grid->iCoorFromIindex(icoor,l);
      int ldx =r+icoor[orthogdim]*rd;
      scalar_type *as =(scalar_type *)&av;
-      as[l] = scalar_type(a[ldx])*scale;
+      as[l] = scalar_type(a[ldx])*zscale;
    }

    tensor_reduced at; at=av;
@@ -375,74 +369,6 @@ static void sliceMaddVector(Lattice<vobj> &R,std::vector<RealD> &a,const Lattice
  }
 };

-
-/*
-template<class vobj>
-static void sliceMaddVectorSlow (Lattice<vobj> &R,std::vector<RealD> &a,const Lattice<vobj> &X,const Lattice<vobj> &Y,
-			     int Orthog,RealD scale=1.0) 
-{    
-  // FIXME: Implementation is slow
-  // Best base the linear combination by constructing a 
-  // set of vectors of size grid->_rdimensions[Orthog].
-  typedef typename vobj::scalar_object sobj;
-  typedef typename vobj::scalar_type scalar_type;
-  typedef typename vobj::vector_type vector_type;
-  
-  int Nblock = X._grid->GlobalDimensions()[Orthog];
-  
-  GridBase *FullGrid  = X._grid;
-  GridBase *SliceGrid = makeSubSliceGrid(FullGrid,Orthog);
-  
-  Lattice<vobj> Xslice(SliceGrid);
-  Lattice<vobj> Rslice(SliceGrid);
-  // If we based this on Cshift it would work for spread out
-  // but it would be even slower
-  for(int i=0;i<Nblock;i++){
-    ExtractSlice(Rslice,Y,i,Orthog);
-    ExtractSlice(Xslice,X,i,Orthog);
-    Rslice = Rslice + Xslice*(scale*a[i]);
-    InsertSlice(Rslice,R,i,Orthog);
-  }
-};
-
-template<class vobj>
-static void sliceInnerProductVectorSlow( std::vector<ComplexD> & vec, const Lattice<vobj> &lhs,const Lattice<vobj> &rhs,int Orthog) 
-  {
-    // FIXME: Implementation is slow
-    // Look at localInnerProduct implementation,
-    // and do inside a site loop with block strided iterators
-    typedef typename vobj::scalar_object sobj;
-    typedef typename vobj::scalar_type scalar_type;
-    typedef typename vobj::vector_type vector_type;
-    typedef typename vobj::tensor_reduced scalar;
-    typedef typename scalar::scalar_object  scomplex;
-  
-    int Nblock = lhs._grid->GlobalDimensions()[Orthog];
-
-    vec.resize(Nblock);
-    std::vector<scomplex> sip(Nblock);
-    Lattice<scalar> IP(lhs._grid); 
-
-    IP=localInnerProduct(lhs,rhs);
-    sliceSum(IP,sip,Orthog);
-  
-    for(int ss=0;ss<Nblock;ss++){
-      vec[ss] = TensorRemove(sip[ss]);
-    }
-  }
-*/
-
-//////////////////////////////////////////////////////////////////////////////////////////
-// FIXME: Implementation is slow
-// If we based this on Cshift it would work for spread out
-// but it would be even slower
-//
-// Repeated extract slice is inefficient
-//
-// Best base the linear combination by constructing a 
-// set of vectors of size grid->_rdimensions[Orthog].
-//////////////////////////////////////////////////////////////////////////////////////////
-
 inline GridBase         *makeSubSliceGrid(const GridBase *BlockSolverGrid,int Orthog)
 {
  int NN    = BlockSolverGrid->_ndimension;
@@ -462,7 +388,6 @@ inline GridBase         *makeSubSliceGrid(const GridBase *BlockSolverGrid,int Or
  return (GridBase *)new GridCartesian(latt_phys,simd_phys,mpi_phys); 
 }

-
 template<class vobj>
 static void sliceMaddMatrix (Lattice<vobj> &R,Eigen::MatrixXcd &aa,const Lattice<vobj> &X,const Lattice<vobj> &Y,int Orthog,RealD scale=1.0) 
 {    
@@ -471,28 +396,103 @@ static void sliceMaddMatrix (Lattice<vobj> &R,Eigen::MatrixXcd &aa,const Lattice
  typedef typename vobj::vector_type vector_type;

  int Nblock = X._grid->GlobalDimensions()[Orthog];
-  
+
  GridBase *FullGrid  = X._grid;
  GridBase *SliceGrid = makeSubSliceGrid(FullGrid,Orthog);
-  
+
  Lattice<vobj> Xslice(SliceGrid);
  Lattice<vobj> Rslice(SliceGrid);
-  
-  for(int i=0;i<Nblock;i++){
-    ExtractSlice(Rslice,Y,i,Orthog);
-    for(int j=0;j<Nblock;j++){
-      ExtractSlice(Xslice,X,j,Orthog);
-      Rslice = Rslice + Xslice*(scale*aa(j,i));
-    }
-    InsertSlice(Rslice,R,i,Orthog);
+
+  assert( FullGrid->_simd_layout[Orthog]==1);
+  int nh =  FullGrid->_ndimension;
+  int nl = SliceGrid->_ndimension;
+
+  //FIXME package in a convenient iterator
+  //Should loop over a plane orthogonal to direction "Orthog"
+  int stride=FullGrid->_slice_stride[Orthog];
+  int block =FullGrid->_slice_block [Orthog];
+  int nblock=FullGrid->_slice_nblock[Orthog];
+  int ostride=FullGrid->_ostride[Orthog];
+#pragma omp parallel 
+  {
+    std::vector<vobj> s_x(Nblock);
+
+#pragma omp for collapse(2)
+    for(int n=0;n<nblock;n++){
+    for(int b=0;b<block;b++){
+      int o  = n*stride + b;
+
+      for(int i=0;i<Nblock;i++){
+	s_x[i] = X[o+i*ostride];
+      }
+
+      vobj dot;
+      for(int i=0;i<Nblock;i++){
+	dot = Y[o+i*ostride];
+	for(int j=0;j<Nblock;j++){
+	  dot = dot + s_x[j]*(scale*aa(j,i));
+	}
+	R[o+i*ostride]=dot;
+      }
+    }}
  }
 };

+template<class vobj>
+static void sliceMulMatrix (Lattice<vobj> &R,Eigen::MatrixXcd &aa,const Lattice<vobj> &X,int Orthog,RealD scale=1.0) 
+{    
+  typedef typename vobj::scalar_object sobj;
+  typedef typename vobj::scalar_type scalar_type;
+  typedef typename vobj::vector_type vector_type;
+
+  int Nblock = X._grid->GlobalDimensions()[Orthog];
+
+  GridBase *FullGrid  = X._grid;
+  GridBase *SliceGrid = makeSubSliceGrid(FullGrid,Orthog);
+
+  Lattice<vobj> Xslice(SliceGrid);
+  Lattice<vobj> Rslice(SliceGrid);
+
+  assert( FullGrid->_simd_layout[Orthog]==1);
+  int nh =  FullGrid->_ndimension;
+  int nl = SliceGrid->_ndimension;
+
+  //FIXME package in a convenient iterator
+  //Should loop over a plane orthogonal to direction "Orthog"
+  int stride=FullGrid->_slice_stride[Orthog];
+  int block =FullGrid->_slice_block [Orthog];
+  int nblock=FullGrid->_slice_nblock[Orthog];
+  int ostride=FullGrid->_ostride[Orthog];
+#pragma omp parallel 
+  {
+    std::vector<vobj> s_x(Nblock);
+
+#pragma omp for collapse(2)
+    for(int n=0;n<nblock;n++){
+    for(int b=0;b<block;b++){
+      int o  = n*stride + b;
+
+      for(int i=0;i<Nblock;i++){
+	s_x[i] = X[o+i*ostride];
+      }
+
+      vobj dot;
+      for(int i=0;i<Nblock;i++){
+	dot = s_x[0]*(scale*aa(0,i));
+	for(int j=1;j<Nblock;j++){
+	  dot = dot + s_x[j]*(scale*aa(j,i));
+	}
+	R[o+i*ostride]=dot;
+      }
+    }}
+  }
+
+};
+
+
 template<class vobj>
 static void sliceInnerProductMatrix(  Eigen::MatrixXcd &mat, const Lattice<vobj> &lhs,const Lattice<vobj> &rhs,int Orthog) 
 {
-  // FIXME: Implementation is slow
-  // Not sure of best solution.. think about it
  typedef typename vobj::scalar_object sobj;
  typedef typename vobj::scalar_type scalar_type;
  typedef typename vobj::vector_type vector_type;
@@ -506,25 +506,55 @@ static void sliceInnerProductMatrix(  Eigen::MatrixXcd &mat, const Lattice<vobj>
  Lattice<vobj> Rslice(SliceGrid);
  
  mat = Eigen::MatrixXcd::Zero(Nblock,Nblock);
-  
-  for(int i=0;i<Nblock;i++){
-    ExtractSlice(Lslice,lhs,i,Orthog);
-    for(int j=0;j<Nblock;j++){
-      ExtractSlice(Rslice,rhs,j,Orthog);
-      mat(i,j) = innerProduct(Lslice,Rslice);
-    }
+
+  assert( FullGrid->_simd_layout[Orthog]==1);
+  int nh =  FullGrid->_ndimension;
+  int nl = SliceGrid->_ndimension;
+
+  //FIXME package in a convenient iterator
+  //Should loop over a plane orthogonal to direction "Orthog"
+  int stride=FullGrid->_slice_stride[Orthog];
+  int block =FullGrid->_slice_block [Orthog];
+  int nblock=FullGrid->_slice_nblock[Orthog];
+  int ostride=FullGrid->_ostride[Orthog];
+
+  typedef typename vobj::vector_typeD vector_typeD;
+
+#pragma omp parallel 
+  {
+    std::vector<vobj> Left(Nblock);
+    std::vector<vobj> Right(Nblock);
+    Eigen::MatrixXcd  mat_thread = Eigen::MatrixXcd::Zero(Nblock,Nblock);
+
+#pragma omp for collapse(2)
+    for(int n=0;n<nblock;n++){
+    for(int b=0;b<block;b++){
+
+      int o  = n*stride + b;
+
+      for(int i=0;i<Nblock;i++){
+	Left [i] = lhs[o+i*ostride];
+	Right[i] = rhs[o+i*ostride];
+      }
+
+      for(int i=0;i<Nblock;i++){
+      for(int j=0;j<Nblock;j++){
+	auto tmp = innerProduct(Left[i],Right[j]);
+	//	vector_typeD rtmp = TensorRemove(tmp);
+	auto rtmp = TensorRemove(tmp);
+	mat_thread(i,j) += Reduce(rtmp);
+      }}
+    }}
+#pragma omp critical
+    {
+      mat += mat_thread;
+    }  
  }
-#undef FORCE_DIAG
-#ifdef FORCE_DIAG
-  for(int i=0;i<Nblock;i++){
-    for(int j=0;j<Nblock;j++){
-      if ( i != j ) mat(i,j)=0.0;
-    }
-  }
-#endif
  return;
 }

 } /*END NAMESPACE GRID*/
 #endif

+
+
@@ -6,8 +6,8 @@

    Copyright (C) 2015

-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-Author: paboyle <paboyle@ph.ed.ac.uk>
+    Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+    Author: Guido Cossu <guido.cossu@ed.ac.uk>

    This program is free software; you can redistribute it and/or modify
    it under the terms of the GNU General Public License as published by
@@ -75,6 +75,55 @@ namespace Grid {
    return multiplicity;
  }

+  
+// merge of April 11 2017
+//<<<<<<< HEAD
+
+
+  // this function is necessary for the LS vectorised field
+  inline int RNGfillable_general(GridBase *coarse,GridBase *fine)
+  {
+    int rngdims = coarse->_ndimension;
+    
+    // trivially extended in higher dims, with locality guaranteeing RNG state is local to node
+    int lowerdims   = fine->_ndimension - coarse->_ndimension;  assert(lowerdims >= 0);
+    // assumes that the higher dimensions are not using more processors
+    // all further divisions are local
+    for(int d=0;d<lowerdims;d++) assert(fine->_processors[d]==1);
+    for(int d=0;d<rngdims;d++) assert(coarse->_processors[d] == fine->_processors[d+lowerdims]);
+    
+
+    // then divide the number of local sites
+    // check that the total number of sims agree, meanse the iSites are the same
+    assert(fine->Nsimd() == coarse->Nsimd());
+
+    // check that the two grids divide cleanly
+    assert( (fine->lSites() / coarse->lSites() ) * coarse->lSites() == fine->lSites() );
+
+    return fine->lSites() / coarse->lSites();
+  }
+
+  /*
+  // Wrap seed_seq to give common interface with random_device
+  class fixedSeed {
+  public:
+    typedef std::seed_seq::result_type result_type;
+    std::seed_seq src;
+    
+    fixedSeed(const std::vector<int> &seeds) : src(seeds.begin(),seeds.end()) {};
+
+    result_type operator () (void){
+      std::vector<result_type> list(1);
+      src.generate(list.begin(),list.end());
+      return list[0];
+    }
+
+  };
+
+=======
+>>>>>>> develop
+  */
+  
  // real scalars are one component
  template<class scalar,class distribution,class generator> 
  void fillScalar(scalar &s,distribution &dist,generator & gen)
@@ -109,7 +158,7 @@ namespace Grid {
 #ifdef RNG_SITMO
    typedef sitmo::prng_engine 	RngEngine;
    typedef uint64_t    	RngStateType;
-    static const int    	RngStateCount = 4;
+    static const int    	RngStateCount = 13;
 #endif

    std::vector<RngEngine>                             _generators;
@@ -164,7 +213,7 @@ namespace Grid {
      ss<<eng;
      ss.seekg(0,ss.beg);
      for(int i=0;i<RngStateCount;i++){
-	ss>>saved[i];
+        ss>>saved[i];
      }
    }
    void GetState(std::vector<RngStateType> & saved,int gen) {
@@ -174,7 +223,7 @@ namespace Grid {
      assert(saved.size()==RngStateCount);
      std::stringstream ss;
      for(int i=0;i<RngStateCount;i++){
-	ss<< saved[i]<<" ";
+        ss<< saved[i]<<" ";
      }
      ss.seekg(0,ss.beg);
      ss>>eng;
@@ -215,7 +264,7 @@ namespace Grid {

      dist[0].reset();
      for(int idx=0;idx<words;idx++){
-	fillScalar(buf[idx],dist[0],_generators[0]);
+  fillScalar(buf[idx],dist[0],_generators[0]);
      }

      CartesianCommunicator::BroadcastWorld(0,(void *)&l,sizeof(l));
@@ -247,7 +296,7 @@ namespace Grid {
      RealF *pointer=(RealF *)&l;
      dist[0].reset();
      for(int i=0;i<2*vComplexF::Nsimd();i++){
-	fillScalar(pointer[i],dist[0],_generators[0]);
+  fillScalar(pointer[i],dist[0],_generators[0]);
      }
      CartesianCommunicator::BroadcastWorld(0,(void *)&l,sizeof(l));
    }
@@ -255,7 +304,7 @@ namespace Grid {
      RealD *pointer=(RealD *)&l;
      dist[0].reset();
      for(int i=0;i<2*vComplexD::Nsimd();i++){
-	fillScalar(pointer[i],dist[0],_generators[0]);
+  fillScalar(pointer[i],dist[0],_generators[0]);
      }
      CartesianCommunicator::BroadcastWorld(0,(void *)&l,sizeof(l));
    }
@@ -263,7 +312,7 @@ namespace Grid {
      RealF *pointer=(RealF *)&l;
      dist[0].reset();
      for(int i=0;i<vRealF::Nsimd();i++){
-	fillScalar(pointer[i],dist[0],_generators[0]);
+  fillScalar(pointer[i],dist[0],_generators[0]);
      }
      CartesianCommunicator::BroadcastWorld(0,(void *)&l,sizeof(l));
    }
@@ -275,7 +324,7 @@ namespace Grid {
      }
      CartesianCommunicator::BroadcastWorld(0,(void *)&l,sizeof(l));
    }
-
+    
    void SeedFixedIntegers(const std::vector<int> &seeds){
      CartesianCommunicator::BroadcastWorld(0,(void *)&seeds[0],sizeof(int)*seeds.size());
      std::seed_seq src(seeds.begin(),seeds.end());
@@ -284,18 +333,20 @@ namespace Grid {
  };

  class GridParallelRNG : public GridRNGbase {
+
+    double _time_counter;
+
  public:
    GridBase *_grid;
-    int _vol;
-  public:
+    unsigned int _vol;

-    int generator_idx(int os,int is){
+    int generator_idx(int os,int is) {
      return is*_grid->oSites()+os;
    }

    GridParallelRNG(GridBase *grid) : GridRNGbase() {
-      _grid=grid;
-      _vol =_grid->iSites()*_grid->oSites();
+      _grid = grid;
+      _vol  =_grid->iSites()*_grid->oSites();

      _generators.resize(_vol);
      _uniform.resize(_vol,std::uniform_real_distribution<RealD>{0,1});
@@ -309,33 +360,34 @@ namespace Grid {
      typedef typename vobj::scalar_object scalar_object;
      typedef typename vobj::scalar_type scalar_type;
      typedef typename vobj::vector_type vector_type;
-      
-      int multiplicity = RNGfillable(_grid,l._grid);

-      int     Nsimd =_grid->Nsimd();
-      int     osites=_grid->oSites();
-      int words=sizeof(scalar_object)/sizeof(scalar_type);
+      double inner_time_counter = usecond();
+
+      int multiplicity = RNGfillable_general(_grid, l._grid); // l has finer or same grid
+      int Nsimd  = _grid->Nsimd();  // guaranteed to be the same for l._grid too
+      int osites = _grid->oSites();  // guaranteed to be <= l._grid->oSites() by a factor multiplicity
+      int words  = sizeof(scalar_object) / sizeof(scalar_type);

      parallel_for(int ss=0;ss<osites;ss++){
+        std::vector<scalar_object> buf(Nsimd);
+        for (int m = 0; m < multiplicity; m++) {  // Draw from same generator multiplicity times

-	std::vector<scalar_object> buf(Nsimd);
-	for(int m=0;m<multiplicity;m++) {// Draw from same generator multiplicity times
+          int sm = multiplicity * ss + m;  // Maps the generator site to the fine site

-	  int sm=multiplicity*ss+m;      // Maps the generator site to the fine site
-
-	  for(int si=0;si<Nsimd;si++){
-	    int gdx = generator_idx(ss,si); // index of generator state
-	    scalar_type *pointer = (scalar_type *)&buf[si];
-	    dist[gdx].reset();
-	    for(int idx=0;idx<words;idx++){
-	      fillScalar(pointer[idx],dist[gdx],_generators[gdx]);
-	    }
-	  }
-
-	  // merge into SIMD lanes
-	  merge(l._odata[sm],buf);
-	}
+          for (int si = 0; si < Nsimd; si++) {
+            
+            int gdx = generator_idx(ss, si);  // index of generator state
+            scalar_type *pointer = (scalar_type *)&buf[si];
+            dist[gdx].reset();
+            for (int idx = 0; idx < words; idx++) 
+              fillScalar(pointer[idx], dist[gdx], _generators[gdx]);
+          }
+          // merge into SIMD lanes, FIXME suboptimal implementation
+          merge(l._odata[sm], buf);
+        }
      }
+
+      _time_counter += usecond()- inner_time_counter;
    };

    void SeedFixedIntegers(const std::vector<int> &seeds){
@@ -412,6 +464,12 @@ namespace Grid {
      }
 #endif
    }
+
+    void Report(){
+      std::cout << GridLogMessage << "Time spent in the fill() routine by GridParallelRNG: "<< _time_counter/1e3 << " ms" << std::endl;
+    }
+
+
    ////////////////////////////////////////////////////////////////////////
    // Support for rigorous test of RNG's
    // Return uniform random uint32_t from requested site generator
@@ -419,7 +477,6 @@ namespace Grid {
    uint32_t GlobalU01(int gsite){

      uint32_t the_number;
-
      // who
      std::vector<int> gcoor;
      int rank,o_idx,i_idx;
@@ -551,7 +551,10 @@ void Replicate(Lattice<vobj> &coarse,Lattice<vobj> & fine)

 //Copy SIMD-vectorized lattice to array of scalar objects in lexicographic order
 template<typename vobj, typename sobj>
-typename std::enable_if<isSIMDvectorized<vobj>::value && !isSIMDvectorized<sobj>::value, void>::type unvectorizeToLexOrdArray(std::vector<sobj> &out, const Lattice<vobj> &in){
+typename std::enable_if<isSIMDvectorized<vobj>::value && !isSIMDvectorized<sobj>::value, void>::type 
+unvectorizeToLexOrdArray(std::vector<sobj> &out, const Lattice<vobj> &in)
+{
+
  typedef typename vobj::vector_type vtype;
  
  GridBase* in_grid = in._grid;
@@ -590,6 +593,54 @@ typename std::enable_if<isSIMDvectorized<vobj>::value && !isSIMDvectorized<sobj>
    extract1(in_vobj, out_ptrs, 0);
  }
 }
+//Copy SIMD-vectorized lattice to array of scalar objects in lexicographic order
+template<typename vobj, typename sobj>
+typename std::enable_if<isSIMDvectorized<vobj>::value 
+                    && !isSIMDvectorized<sobj>::value, void>::type 
+vectorizeFromLexOrdArray( std::vector<sobj> &in, Lattice<vobj> &out)
+{
+
+  typedef typename vobj::vector_type vtype;
+  
+  GridBase* grid = out._grid;
+  assert(in.size()==grid->lSites());
+  
+  int ndim     = grid->Nd();
+  int nsimd    = vtype::Nsimd();
+
+  std::vector<std::vector<int> > icoor(nsimd);
+      
+  for(int lane=0; lane < nsimd; lane++){
+    icoor[lane].resize(ndim);
+    grid->iCoorFromIindex(icoor[lane],lane);
+  }
+  
+  parallel_for(uint64_t oidx = 0; oidx < grid->oSites(); oidx++){ //loop over outer index
+    //Assemble vector of pointers to output elements
+    std::vector<sobj*> ptrs(nsimd);
+
+    std::vector<int> ocoor(ndim);
+    grid->oCoorFromOindex(ocoor, oidx);
+
+    std::vector<int> lcoor(grid->Nd());
+      
+    for(int lane=0; lane < nsimd; lane++){
+
+      for(int mu=0;mu<ndim;mu++){
+	lcoor[mu] = ocoor[mu] + grid->_rdimensions[mu]*icoor[lane][mu];
+      }
+
+      int lex;
+      Lexicographic::IndexFromCoor(lcoor, lex, grid->_ldimensions);
+      ptrs[lane] = &in[lex];
+    }
+    
+    //pack from those ptrs
+    vobj vecobj;
+    merge1(vecobj, ptrs, 0);
+    out._odata[oidx] = vecobj; 
+  }
+}

 //Convert a Lattice from one precision to another
 template<class VobjOut, class VobjIn>
@@ -615,7 +666,7 @@ void precisionChange(Lattice<VobjOut> &out, const Lattice<VobjIn> &in){
  std::vector<SobjOut> in_slex_conv(in_grid->lSites());
  unvectorizeToLexOrdArray(in_slex_conv, in);
    
-  parallel_for(int out_oidx=0;out_oidx<out_grid->oSites();out_oidx++){
+  parallel_for(uint64_t out_oidx=0;out_oidx<out_grid->oSites();out_oidx++){
    std::vector<int> out_ocoor(ndim);
    out_grid->oCoorFromOindex(out_ocoor, out_oidx);

@@ -62,14 +62,20 @@ namespace Grid {
    return ret;
  }

-  template<class obj> Lattice<obj> expMat(const Lattice<obj> &rhs, ComplexD alpha, Integer Nexp = DEFAULT_MAT_EXP){
+  template<class obj> Lattice<obj> expMat(const Lattice<obj> &rhs, RealD alpha, Integer Nexp = DEFAULT_MAT_EXP){
    Lattice<obj> ret(rhs._grid);
    ret.checkerboard = rhs.checkerboard;
    conformable(ret,rhs);
    parallel_for(int ss=0;ss<rhs._grid->oSites();ss++){
      ret._odata[ss]=Exponentiate(rhs._odata[ss],alpha, Nexp);
    }
+
    return ret;
+
+    
+    
+
+    
  }


@@ -30,6 +30,7 @@ directory
 *************************************************************************************/
 /*  END LEGAL */
 #include <Grid/GridCore.h>
+#include <Grid/util/CompilerCompatible.h>

 #include <cxxabi.h>
 #include <memory>
@@ -0,0 +1,716 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/parallelIO/IldgIO.h
+
+Copyright (C) 2015
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_ILDG_IO_H
+#define GRID_ILDG_IO_H
+
+#ifdef HAVE_LIME
+#include <algorithm>
+#include <fstream>
+#include <iomanip>
+#include <iostream>
+#include <map>
+
+#include <pwd.h>
+#include <sys/utsname.h>
+#include <unistd.h>
+
+//C-Lime is a must have for this functionality
+extern "C" {  
+#include "lime.h"
+}
+
+namespace Grid {
+namespace QCD {
+
+  /////////////////////////////////
+  // Encode word types as strings
+  /////////////////////////////////
+ template<class word> inline std::string ScidacWordMnemonic(void){ return std::string("unknown"); }
+ template<> inline std::string ScidacWordMnemonic<double>  (void){ return std::string("D"); }
+ template<> inline std::string ScidacWordMnemonic<float>   (void){ return std::string("F"); }
+ template<> inline std::string ScidacWordMnemonic< int32_t>(void){ return std::string("I32_t"); }
+ template<> inline std::string ScidacWordMnemonic<uint32_t>(void){ return std::string("U32_t"); }
+ template<> inline std::string ScidacWordMnemonic< int64_t>(void){ return std::string("I64_t"); }
+ template<> inline std::string ScidacWordMnemonic<uint64_t>(void){ return std::string("U64_t"); }
+
+  /////////////////////////////////////////
+  // Encode a generic tensor as a string
+  /////////////////////////////////////////
+ template<class vobj> std::string ScidacRecordTypeString(int &colors, int &spins, int & typesize,int &datacount) { 
+
+   typedef typename getPrecision<vobj>::real_scalar_type stype;
+
+   int _ColourN       = indexRank<ColourIndex,vobj>();
+   int _ColourScalar  =  isScalar<ColourIndex,vobj>();
+   int _ColourVector  =  isVector<ColourIndex,vobj>();
+   int _ColourMatrix  =  isMatrix<ColourIndex,vobj>();
+
+   int _SpinN       = indexRank<SpinIndex,vobj>();
+   int _SpinScalar  =  isScalar<SpinIndex,vobj>();
+   int _SpinVector  =  isVector<SpinIndex,vobj>();
+   int _SpinMatrix  =  isMatrix<SpinIndex,vobj>();
+
+   int _LorentzN       = indexRank<LorentzIndex,vobj>();
+   int _LorentzScalar  =  isScalar<LorentzIndex,vobj>();
+   int _LorentzVector  =  isVector<LorentzIndex,vobj>();
+   int _LorentzMatrix  =  isMatrix<LorentzIndex,vobj>();
+
+   std::stringstream stream;
+
+   stream << "GRID_";
+   stream << ScidacWordMnemonic<stype>();
+
+   //   std::cout << " Lorentz N/S/V/M : " << _LorentzN<<" "<<_LorentzScalar<<"/"<<_LorentzVector<<"/"<<_LorentzMatrix<<std::endl;
+   //   std::cout << " Spin    N/S/V/M : " << _SpinN   <<" "<<_SpinScalar   <<"/"<<_SpinVector   <<"/"<<_SpinMatrix<<std::endl;
+   //   std::cout << " Colour  N/S/V/M : " << _ColourN <<" "<<_ColourScalar <<"/"<<_ColourVector <<"/"<<_ColourMatrix<<std::endl;
+
+   if ( _LorentzVector )   stream << "_LorentzVector"<<_LorentzN;
+   if ( _LorentzMatrix )   stream << "_LorentzMatrix"<<_LorentzN;
+
+   if ( _SpinVector )   stream << "_SpinVector"<<_SpinN;
+   if ( _SpinMatrix )   stream << "_SpinMatrix"<<_SpinN;
+
+   if ( _ColourVector )   stream << "_ColourVector"<<_ColourN;
+   if ( _ColourMatrix )   stream << "_ColourMatrix"<<_ColourN;
+
+   if ( _ColourScalar && _LorentzScalar && _SpinScalar )   stream << "_Complex";
+
+
+   typesize = sizeof(typename vobj::scalar_type);
+
+   if ( _ColourMatrix ) typesize*= _ColourN*_ColourN;
+   else                 typesize*= _ColourN;
+
+   if ( _SpinMatrix )   typesize*= _SpinN*_SpinN;
+   else                 typesize*= _SpinN;
+
+   colors    = _ColourN;
+   spins     = _SpinN;
+   datacount = _LorentzN;
+
+   return stream.str();
+ }
+ 
+ template<class vobj> std::string ScidacRecordTypeString(Lattice<vobj> & lat,int &colors, int &spins, int & typesize,int &datacount) { 
+   return ScidacRecordTypeString<vobj>(colors,spins,typesize,datacount);
+ };
+
+
+ ////////////////////////////////////////////////////////////
+ // Helper to fill out metadata
+ ////////////////////////////////////////////////////////////
+ template<class vobj> void ScidacMetaData(Lattice<vobj> & field,
+					  FieldMetaData &header,
+					  scidacRecord & _scidacRecord,
+					  scidacFile   & _scidacFile) 
+ {
+   typedef typename getPrecision<vobj>::real_scalar_type stype;
+
+   /////////////////////////////////////
+   // Pull Grid's metadata
+   /////////////////////////////////////
+   PrepareMetaData(field,header);
+
+   /////////////////////////////////////
+   // Scidac Private File structure
+   /////////////////////////////////////
+   _scidacFile              = scidacFile(field._grid);
+
+   /////////////////////////////////////
+   // Scidac Private Record structure
+   /////////////////////////////////////
+   scidacRecord sr;
+   sr.datatype   = ScidacRecordTypeString(field,sr.colors,sr.spins,sr.typesize,sr.datacount);
+   sr.date       = header.creation_date;
+   sr.precision  = ScidacWordMnemonic<stype>();
+   sr.recordtype = GRID_IO_FIELD;
+
+   _scidacRecord = sr;
+
+   std::cout << GridLogMessage << "Build SciDAC datatype " <<sr.datatype<<std::endl;
+ }
+ 
+ ///////////////////////////////////////////////////////
+ // Scidac checksum
+ ///////////////////////////////////////////////////////
+ static int scidacChecksumVerify(scidacChecksum &scidacChecksum_,uint32_t scidac_csuma,uint32_t scidac_csumb)
+ {
+   uint32_t scidac_checksuma = stoull(scidacChecksum_.suma,0,16);
+   uint32_t scidac_checksumb = stoull(scidacChecksum_.sumb,0,16);
+   if ( scidac_csuma !=scidac_checksuma) return 0;
+   if ( scidac_csumb !=scidac_checksumb) return 0;
+    return 1;
+ }
+
+////////////////////////////////////////////////////////////////////////////////////
+// Lime, ILDG and Scidac I/O classes
+////////////////////////////////////////////////////////////////////////////////////
+class GridLimeReader : public BinaryIO {
+ public:
+   ///////////////////////////////////////////////////
+   // FIXME: format for RNG? Now just binary out instead
+   ///////////////////////////////////////////////////
+
+   FILE       *File;
+   LimeReader *LimeR;
+   std::string filename;
+
+   /////////////////////////////////////////////
+   // Open the file
+   /////////////////////////////////////////////
+   void open(std::string &_filename) 
+   {
+     filename= _filename;
+     File = fopen(filename.c_str(), "r");
+     LimeR = limeCreateReader(File);
+   }
+   /////////////////////////////////////////////
+   // Close the file
+   /////////////////////////////////////////////
+   void close(void){
+     fclose(File);
+     //     limeDestroyReader(LimeR);
+   }
+
+  ////////////////////////////////////////////
+  // Read a generic lattice field and verify checksum
+  ////////////////////////////////////////////
+  template<class vobj>
+  void readLimeLatticeBinaryObject(Lattice<vobj> &field,std::string record_name)
+  {
+    typedef typename vobj::scalar_object sobj;
+    scidacChecksum scidacChecksum_;
+    uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+
+    std::string format = getFormatString<vobj>();
+
+    while ( limeReaderNextRecord(LimeR) == LIME_SUCCESS ) { 
+
+      std::cout << GridLogMessage << limeReaderType(LimeR) <<std::endl;
+	
+      if ( strncmp(limeReaderType(LimeR), record_name.c_str(),strlen(record_name.c_str()) )  ) {
+
+
+	off_t offset= ftell(File);
+	BinarySimpleMunger<sobj,sobj> munge;
+	BinaryIO::readLatticeObject< sobj, sobj >(field, filename, munge, offset, format,nersc_csum,scidac_csuma,scidac_csumb);
+
+	/////////////////////////////////////////////
+	// Insist checksum is next record
+	/////////////////////////////////////////////
+	readLimeObject(scidacChecksum_,std::string("scidacChecksum"),record_name);
+
+	/////////////////////////////////////////////
+	// Verify checksums
+	/////////////////////////////////////////////
+	scidacChecksumVerify(scidacChecksum_,scidac_csuma,scidac_csumb);
+	return;
+      }
+    }
+  }
+  ////////////////////////////////////////////
+  // Read a generic serialisable object
+  ////////////////////////////////////////////
+  template<class serialisable_object>
+  void readLimeObject(serialisable_object &object,std::string object_name,std::string record_name)
+  {
+    std::string xmlstring;
+    // should this be a do while; can we miss a first record??
+    while ( limeReaderNextRecord(LimeR) == LIME_SUCCESS ) { 
+
+      uint64_t nbytes = limeReaderBytes(LimeR);//size of this record (configuration)
+
+      if ( strncmp(limeReaderType(LimeR), record_name.c_str(),strlen(record_name.c_str()) )  ) {
+	std::vector<char> xmlc(nbytes+1,'\0');
+	limeReaderReadData((void *)&xmlc[0], &nbytes, LimeR);    
+	XmlReader RD(&xmlc[0],"");
+	read(RD,object_name,object);
+	return;
+      }
+
+    }  
+    assert(0);
+  }
+};
+
+class GridLimeWriter : public BinaryIO {
+ public:
+   ///////////////////////////////////////////////////
+   // FIXME: format for RNG? Now just binary out instead
+   ///////////////////////////////////////////////////
+
+   FILE       *File;
+   LimeWriter *LimeW;
+   std::string filename;
+
+   void open(std::string &_filename) { 
+     filename= _filename;
+     File = fopen(filename.c_str(), "w");
+     LimeW = limeCreateWriter(File); assert(LimeW != NULL );
+   }
+   /////////////////////////////////////////////
+   // Close the file
+   /////////////////////////////////////////////
+   void close(void) {
+     fclose(File);
+     //  limeDestroyWriter(LimeW);
+   }
+  ///////////////////////////////////////////////////////
+  // Lime utility functions
+  ///////////////////////////////////////////////////////
+  int createLimeRecordHeader(std::string message, int MB, int ME, size_t PayloadSize)
+  {
+    LimeRecordHeader *h;
+    h = limeCreateHeader(MB, ME, const_cast<char *>(message.c_str()), PayloadSize);
+    assert(limeWriteRecordHeader(h, LimeW) >= 0);
+    limeDestroyHeader(h);
+    return LIME_SUCCESS;
+  }
+  ////////////////////////////////////////////
+  // Write a generic serialisable object
+  ////////////////////////////////////////////
+  template<class serialisable_object>
+  void writeLimeObject(int MB,int ME,serialisable_object &object,std::string object_name,std::string record_name)
+  {
+    std::string xmlstring;
+    {
+      XmlWriter WR("","");
+      write(WR,object_name,object);
+      xmlstring = WR.XmlString();
+    }
+    uint64_t nbytes = xmlstring.size();
+    int err;
+    LimeRecordHeader *h = limeCreateHeader(MB, ME,(char *)record_name.c_str(), nbytes); assert(h!= NULL);
+
+    err=limeWriteRecordHeader(h, LimeW);                    assert(err>=0);
+    err=limeWriteRecordData(&xmlstring[0], &nbytes, LimeW); assert(err>=0);
+    err=limeWriterCloseRecord(LimeW);                       assert(err>=0);
+    limeDestroyHeader(h);
+  }
+  ////////////////////////////////////////////
+  // Write a generic lattice field and csum
+  ////////////////////////////////////////////
+  template<class vobj>
+  void writeLimeLatticeBinaryObject(Lattice<vobj> &field,std::string record_name)
+  {
+    ////////////////////////////////////////////
+    // Create record header
+    ////////////////////////////////////////////
+    typedef typename vobj::scalar_object sobj;
+    int err;
+    uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+    uint64_t PayloadSize = sizeof(sobj) * field._grid->_gsites;
+    createLimeRecordHeader(record_name, 0, 0, PayloadSize);
+
+    ////////////////////////////////////////////////////////////////////
+    // NB: FILE and iostream are jointly writing disjoint sequences in the
+    // the same file through different file handles (integer units).
+    // 
+    // These are both buffered, so why I think this code is right is as follows.
+    //
+    // i)  write record header to FILE *File, telegraphing the size. 
+    // ii) ftell reads the offset from FILE *File .
+    // iii) iostream / MPI Open independently seek this offset. Write sequence direct to disk.
+    //      Closes iostream and flushes.
+    // iv) fseek on FILE * to end of this disjoint section.
+    //  v) Continue writing scidac record.
+    ////////////////////////////////////////////////////////////////////
+    off_t offset = ftell(File);
+    std::string format = getFormatString<vobj>();
+    BinarySimpleMunger<sobj,sobj> munge;
+    BinaryIO::writeLatticeObject<vobj,sobj>(field, filename, munge, offset, format,nersc_csum,scidac_csuma,scidac_csumb);
+    err=limeWriterCloseRecord(LimeW);  assert(err>=0);
+    ////////////////////////////////////////
+    // Write checksum element, propagaing forward from the BinaryIO
+    // Always pair a checksum with a binary object, and close message
+    ////////////////////////////////////////
+    scidacChecksum checksum;
+    std::stringstream streama; streama << std::hex << scidac_csuma;
+    std::stringstream streamb; streamb << std::hex << scidac_csumb;
+    checksum.suma= streama.str();
+    checksum.sumb= streamb.str();
+    std::cout << GridLogMessage<<" writing scidac checksums "<<std::hex<<scidac_csuma<<"/"<<scidac_csumb<<std::dec<<std::endl;
+    writeLimeObject(0,1,checksum,std::string("scidacChecksum"    ),std::string(SCIDAC_CHECKSUM));
+  }
+};
+
+class ScidacWriter : public GridLimeWriter {
+ public:
+
+   template<class SerialisableUserFile>
+   void writeScidacFileRecord(GridBase *grid,SerialisableUserFile &_userFile)
+   {
+     scidacFile    _scidacFile(grid);
+     writeLimeObject(1,0,_scidacFile,_scidacFile.SerialisableClassName(),std::string(SCIDAC_PRIVATE_FILE_XML));
+     writeLimeObject(0,1,_userFile,_userFile.SerialisableClassName(),std::string(SCIDAC_FILE_XML));
+   }
+  ////////////////////////////////////////////////
+  // Write generic lattice field in scidac format
+  ////////////////////////////////////////////////
+   template <class vobj, class userRecord>
+  void writeScidacFieldRecord(Lattice<vobj> &field,userRecord _userRecord) 
+  {
+    typedef typename vobj::scalar_object sobj;
+    uint64_t nbytes;
+    GridBase * grid = field._grid;
+
+    ////////////////////////////////////////
+    // fill the Grid header
+    ////////////////////////////////////////
+    FieldMetaData header;
+    scidacRecord  _scidacRecord;
+    scidacFile    _scidacFile;
+
+    ScidacMetaData(field,header,_scidacRecord,_scidacFile);
+
+    //////////////////////////////////////////////
+    // Fill the Lime file record by record
+    //////////////////////////////////////////////
+    writeLimeObject(1,0,header ,std::string("FieldMetaData"),std::string(GRID_FORMAT)); // Open message 
+    writeLimeObject(0,0,_userRecord,_userRecord.SerialisableClassName(),std::string(SCIDAC_RECORD_XML));
+    writeLimeObject(0,0,_scidacRecord,_scidacRecord.SerialisableClassName(),std::string(SCIDAC_PRIVATE_RECORD_XML));
+    writeLimeLatticeBinaryObject(field,std::string(ILDG_BINARY_DATA));      // Closes message with checksum
+  }
+};
+
+class IldgWriter : public ScidacWriter {
+ public:
+
+  ///////////////////////////////////
+  // A little helper
+  ///////////////////////////////////
+  void writeLimeIldgLFN(std::string &LFN)
+  {
+    uint64_t PayloadSize = LFN.size();
+    int err;
+    createLimeRecordHeader(ILDG_DATA_LFN, 0 , 0, PayloadSize);
+    err=limeWriteRecordData(const_cast<char*>(LFN.c_str()), &PayloadSize,LimeW); assert(err>=0);
+    err=limeWriterCloseRecord(LimeW); assert(err>=0);
+  }
+
+  ////////////////////////////////////////////////////////////////
+  // Special ILDG operations ; gauge configs only.
+  // Don't require scidac records EXCEPT checksum
+  // Use Grid MetaData object if present.
+  ////////////////////////////////////////////////////////////////
+  template <class vsimd>
+  void writeConfiguration(Lattice<iLorentzColourMatrix<vsimd> > &Umu,int sequence,std::string LFN,std::string description) 
+  {
+    GridBase * grid = Umu._grid;
+    typedef Lattice<iLorentzColourMatrix<vsimd> > GaugeField;
+    typedef iLorentzColourMatrix<vsimd> vobj;
+    typedef typename vobj::scalar_object sobj;
+
+    uint64_t nbytes;
+
+    ////////////////////////////////////////
+    // fill the Grid header
+    ////////////////////////////////////////
+    FieldMetaData header;
+    scidacRecord  _scidacRecord;
+    scidacFile    _scidacFile;
+
+    ScidacMetaData(Umu,header,_scidacRecord,_scidacFile);
+
+    std::string format = header.floating_point;
+    header.ensemble_id    = description;
+    header.ensemble_label = description;
+    header.sequence_number = sequence;
+    header.ildg_lfn = LFN;
+
+    assert ( (format == std::string("IEEE32BIG"))  
+           ||(format == std::string("IEEE64BIG")) );
+
+    //////////////////////////////////////////////////////
+    // Fill ILDG header data struct
+    //////////////////////////////////////////////////////
+    ildgFormat ildgfmt ;
+    ildgfmt.field     = std::string("su3gauge");
+
+    if ( format == std::string("IEEE32BIG") ) { 
+      ildgfmt.precision = 32;
+    } else { 
+      ildgfmt.precision = 64;
+    }
+    ildgfmt.version = 1.0;
+    ildgfmt.lx = header.dimension[0];
+    ildgfmt.ly = header.dimension[1];
+    ildgfmt.lz = header.dimension[2];
+    ildgfmt.lt = header.dimension[3];
+    assert(header.nd==4);
+    assert(header.nd==header.dimension.size());
+
+    //////////////////////////////////////////////////////////////////////////////
+    // Fill the USQCD info field
+    //////////////////////////////////////////////////////////////////////////////
+    usqcdInfo info;
+    info.version=1.0;
+    info.plaq   = header.plaquette;
+    info.linktr = header.link_trace;
+
+    std::cout << GridLogMessage << " Writing config; IldgIO "<<std::endl;
+    //////////////////////////////////////////////
+    // Fill the Lime file record by record
+    //////////////////////////////////////////////
+    writeLimeObject(1,0,header ,std::string("FieldMetaData"),std::string(GRID_FORMAT)); // Open message 
+    writeLimeObject(0,0,_scidacFile,_scidacFile.SerialisableClassName(),std::string(SCIDAC_PRIVATE_FILE_XML));
+    writeLimeObject(0,1,info,info.SerialisableClassName(),std::string(SCIDAC_FILE_XML));
+    writeLimeObject(1,0,_scidacRecord,_scidacRecord.SerialisableClassName(),std::string(SCIDAC_PRIVATE_RECORD_XML));
+    writeLimeObject(0,0,info,info.SerialisableClassName(),std::string(SCIDAC_RECORD_XML));
+    writeLimeObject(0,0,ildgfmt,std::string("ildgFormat")   ,std::string(ILDG_FORMAT)); // rec
+    writeLimeIldgLFN(header.ildg_lfn);                                                 // rec
+    writeLimeLatticeBinaryObject(Umu,std::string(ILDG_BINARY_DATA));      // Closes message with checksum
+    //    limeDestroyWriter(LimeW);
+    fclose(File);
+  }
+};
+
+class IldgReader : public GridLimeReader {
+ public:
+
+  ////////////////////////////////////////////////////////////////
+  // Read either Grid/SciDAC/ILDG configuration
+  // Don't require scidac records EXCEPT checksum
+  // Use Grid MetaData object if present.
+  // Else use ILDG MetaData object if present.
+  // Else use SciDAC MetaData object if present.
+  ////////////////////////////////////////////////////////////////
+  template <class vsimd>
+  void readConfiguration(Lattice<iLorentzColourMatrix<vsimd> > &Umu, FieldMetaData &FieldMetaData_) {
+
+    typedef Lattice<iLorentzColourMatrix<vsimd> > GaugeField;
+    typedef typename GaugeField::vector_object  vobj;
+    typedef typename vobj::scalar_object sobj;
+
+    typedef LorentzColourMatrixF fobj;
+    typedef LorentzColourMatrixD dobj;
+
+    GridBase *grid = Umu._grid;
+
+    std::vector<int> dims = Umu._grid->FullDimensions();
+
+    assert(dims.size()==4);
+
+    // Metadata holders
+    ildgFormat     ildgFormat_    ;
+    std::string    ildgLFN_       ;
+    scidacChecksum scidacChecksum_; 
+    usqcdInfo      usqcdInfo_     ;
+
+    // track what we read from file
+    int found_ildgFormat    =0;
+    int found_ildgLFN       =0;
+    int found_scidacChecksum=0;
+    int found_usqcdInfo     =0;
+    int found_ildgBinary =0;
+    int found_FieldMetaData =0;
+
+    uint32_t nersc_csum;
+    uint32_t scidac_csuma;
+    uint32_t scidac_csumb;
+
+    // Binary format
+    std::string format;
+
+    //////////////////////////////////////////////////////////////////////////
+    // Loop over all records
+    // -- Order is poorly guaranteed except ILDG header preceeds binary section.
+    // -- Run like an event loop.
+    // -- Impose trust hierarchy. Grid takes precedence & look for ILDG, and failing
+    //    that Scidac. 
+    // -- Insist on Scidac checksum record.
+    //////////////////////////////////////////////////////////////////////////
+
+    while ( limeReaderNextRecord(LimeR) == LIME_SUCCESS ) { 
+
+      uint64_t nbytes = limeReaderBytes(LimeR);//size of this record (configuration)
+      
+      //////////////////////////////////////////////////////////////////
+      // If not BINARY_DATA read a string and parse
+      //////////////////////////////////////////////////////////////////
+      if ( strncmp(limeReaderType(LimeR), ILDG_BINARY_DATA,strlen(ILDG_BINARY_DATA) )  ) {
+	
+	// Copy out the string
+	std::vector<char> xmlc(nbytes+1,'\0');
+	limeReaderReadData((void *)&xmlc[0], &nbytes, LimeR);    
+	std::cout << GridLogMessage<< "Non binary record :" <<limeReaderType(LimeR) <<std::endl; //<<"\n"<<(&xmlc[0])<<std::endl;
+
+	//////////////////////////////////
+	// ILDG format record
+	if ( !strncmp(limeReaderType(LimeR), ILDG_FORMAT,strlen(ILDG_FORMAT)) ) { 
+
+	  XmlReader RD(&xmlc[0],"");
+	  read(RD,"ildgFormat",ildgFormat_);
+
+	  if ( ildgFormat_.precision == 64 ) format = std::string("IEEE64BIG");
+	  if ( ildgFormat_.precision == 32 ) format = std::string("IEEE32BIG");
+
+	  assert( ildgFormat_.lx == dims[0]);
+	  assert( ildgFormat_.ly == dims[1]);
+	  assert( ildgFormat_.lz == dims[2]);
+	  assert( ildgFormat_.lt == dims[3]);
+
+	  found_ildgFormat = 1;
+	}
+
+	if ( !strncmp(limeReaderType(LimeR), ILDG_DATA_LFN,strlen(ILDG_DATA_LFN)) ) {
+	  FieldMetaData_.ildg_lfn = std::string(&xmlc[0]);
+	  found_ildgLFN = 1;
+	}
+
+	if ( !strncmp(limeReaderType(LimeR), GRID_FORMAT,strlen(ILDG_FORMAT)) ) { 
+
+	  XmlReader RD(&xmlc[0],"");
+	  read(RD,"FieldMetaData",FieldMetaData_);
+
+	  format = FieldMetaData_.floating_point;
+
+	  assert(FieldMetaData_.dimension[0] == dims[0]);
+	  assert(FieldMetaData_.dimension[1] == dims[1]);
+	  assert(FieldMetaData_.dimension[2] == dims[2]);
+	  assert(FieldMetaData_.dimension[3] == dims[3]);
+
+	  found_FieldMetaData = 1;
+	}
+
+	if ( !strncmp(limeReaderType(LimeR), SCIDAC_RECORD_XML,strlen(SCIDAC_RECORD_XML)) ) { 
+	  std::string xmls(&xmlc[0]);
+	  // is it a USQCD info field
+	  if ( xmls.find(std::string("usqcdInfo")) != std::string::npos ) { 
+	    std::cout << GridLogMessage<<"...found a usqcdInfo field"<<std::endl;
+	    XmlReader RD(&xmlc[0],"");
+	    read(RD,"usqcdInfo",usqcdInfo_);
+	    found_usqcdInfo = 1;
+	  }
+	}
+
+	if ( !strncmp(limeReaderType(LimeR), SCIDAC_CHECKSUM,strlen(SCIDAC_CHECKSUM)) ) { 
+	  XmlReader RD(&xmlc[0],"");
+	  read(RD,"scidacChecksum",scidacChecksum_);
+	  found_scidacChecksum = 1;
+	}
+
+      } else {  
+	/////////////////////////////////
+	// Binary data
+	/////////////////////////////////
+	std::cout << GridLogMessage << "ILDG Binary record found : "  ILDG_BINARY_DATA << std::endl;
+	off_t offset= ftell(File);
+
+	if ( format == std::string("IEEE64BIG") ) {
+	  GaugeSimpleMunger<dobj, sobj> munge;
+	  BinaryIO::readLatticeObject< vobj, dobj >(Umu, filename, munge, offset, format,nersc_csum,scidac_csuma,scidac_csumb);
+	} else { 
+	  GaugeSimpleMunger<fobj, sobj> munge;
+	  BinaryIO::readLatticeObject< vobj, fobj >(Umu, filename, munge, offset, format,nersc_csum,scidac_csuma,scidac_csumb);
+	}
+
+	found_ildgBinary = 1;
+      }
+
+    }
+
+    //////////////////////////////////////////////////////
+    // Minimally must find binary segment and checksum
+    // Since this is an ILDG reader require ILDG format
+    //////////////////////////////////////////////////////
+    assert(found_ildgBinary);
+    assert(found_ildgFormat);
+    assert(found_scidacChecksum);
+
+    // Must find something with the lattice dimensions
+    assert(found_FieldMetaData||found_ildgFormat);
+
+    if ( found_FieldMetaData ) {
+
+      std::cout << GridLogMessage<<"Grid MetaData was record found: configuration was probably written by Grid ! Yay ! "<<std::endl;
+
+    } else { 
+
+      assert(found_ildgFormat);
+      assert ( ildgFormat_.field == std::string("su3gauge") );
+
+      ///////////////////////////////////////////////////////////////////////////////////////
+      // Populate our Grid metadata as best we can
+      ///////////////////////////////////////////////////////////////////////////////////////
+
+      std::ostringstream vers; vers << ildgFormat_.version;
+      FieldMetaData_.hdr_version = vers.str();
+      FieldMetaData_.data_type = std::string("4D_SU3_GAUGE_3X3");
+
+      FieldMetaData_.nd=4;
+      FieldMetaData_.dimension.resize(4);
+
+      FieldMetaData_.dimension[0] = ildgFormat_.lx ;
+      FieldMetaData_.dimension[1] = ildgFormat_.ly ;
+      FieldMetaData_.dimension[2] = ildgFormat_.lz ;
+      FieldMetaData_.dimension[3] = ildgFormat_.lt ;
+
+      if ( found_usqcdInfo ) { 
+	FieldMetaData_.plaquette = usqcdInfo_.plaq;
+	FieldMetaData_.link_trace= usqcdInfo_.linktr;
+	std::cout << GridLogMessage <<"This configuration was probably written by USQCD "<<std::endl;
+	std::cout << GridLogMessage <<"USQCD xml record Plaquette : "<<FieldMetaData_.plaquette<<std::endl;
+	std::cout << GridLogMessage <<"USQCD xml record LinkTrace : "<<FieldMetaData_.link_trace<<std::endl;
+      } else { 
+	FieldMetaData_.plaquette = 0.0;
+	FieldMetaData_.link_trace= 0.0;
+	std::cout << GridLogWarning << "This configuration is unsafe with no plaquette records that can verify it !!! "<<std::endl;
+      }
+    }
+
+    ////////////////////////////////////////////////////////////
+    // Really really want to mandate a scidac checksum
+    ////////////////////////////////////////////////////////////
+    if ( found_scidacChecksum ) {
+      FieldMetaData_.scidac_checksuma = stoull(scidacChecksum_.suma,0,16);
+      FieldMetaData_.scidac_checksumb = stoull(scidacChecksum_.sumb,0,16);
+      scidacChecksumVerify(scidacChecksum_,scidac_csuma,scidac_csumb);
+      assert( scidac_csuma ==FieldMetaData_.scidac_checksuma);
+      assert( scidac_csumb ==FieldMetaData_.scidac_checksumb);
+      std::cout << GridLogMessage<<"SciDAC checksums match " << std::endl;
+    } else { 
+      std::cout << GridLogWarning<<"SciDAC checksums not found. This is unsafe. " << std::endl;
+      assert(0); // Can I insist always checksum ?
+    }
+
+    if ( found_FieldMetaData || found_usqcdInfo ) {
+      FieldMetaData checker;
+      GaugeStatistics(Umu,checker);
+      assert(fabs(checker.plaquette  - FieldMetaData_.plaquette )<1.0e-5);
+      assert(fabs(checker.link_trace - FieldMetaData_.link_trace)<1.0e-5);
+      std::cout << GridLogMessage<<"Plaquette and link trace match " << std::endl;
+    }
+  }
+ };
+
+}}
+
+//HAVE_LIME
+#endif
+
+#endif
@@ -0,0 +1,231 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/parallelIO/IldgIO.h
+
+Copyright (C) 2015
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_ILDGTYPES_IO_H
+#define GRID_ILDGTYPES_IO_H
+
+#ifdef HAVE_LIME
+extern "C" { // for linkage
+#include "lime.h"
+}
+
+namespace Grid {
+
+/////////////////////////////////////////////////////////////////////////////////
+// Data representation of records that enter ILDG and SciDac formats
+/////////////////////////////////////////////////////////////////////////////////
+
+#define GRID_FORMAT      "grid-format"
+#define ILDG_FORMAT      "ildg-format"
+#define ILDG_BINARY_DATA "ildg-binary-data"
+#define ILDG_DATA_LFN    "ildg-data-lfn"
+#define SCIDAC_CHECKSUM           "scidac-checksum"
+#define SCIDAC_PRIVATE_FILE_XML   "scidac-private-file-xml"
+#define SCIDAC_FILE_XML           "scidac-file-xml"
+#define SCIDAC_PRIVATE_RECORD_XML "scidac-private-record-xml"
+#define SCIDAC_RECORD_XML         "scidac-record-xml"
+#define SCIDAC_BINARY_DATA        "scidac-binary-data"
+// Unused SCIDAC records names; could move to support this functionality
+#define SCIDAC_SITELIST           "scidac-sitelist"
+
+  ////////////////////////////////////////////////////////////
+  const int GRID_IO_SINGLEFILE = 0; // hardcode lift from QIO compat
+  const int GRID_IO_MULTIFILE  = 1; // hardcode lift from QIO compat
+  const int GRID_IO_FIELD      = 0; // hardcode lift from QIO compat
+  const int GRID_IO_GLOBAL     = 1; // hardcode lift from QIO compat
+  ////////////////////////////////////////////////////////////
+
+/////////////////////////////////////////////////////////////////////////////////
+// QIO uses mandatory "private" records fixed format
+// Private is in principle "opaque" however it can't be changed now because that would break existing 
+// file compatability, so should be correct to assume the undocumented but defacto file structure.
+/////////////////////////////////////////////////////////////////////////////////
+
+////////////////////////
+// Scidac private file xml
+// <?xml version="1.0" encoding="UTF-8"?><scidacFile><version>1.1</version><spacetime>4</spacetime><dims>16 16 16 32 </dims><volfmt>0</volfmt></scidacFile>
+////////////////////////
+struct scidacFile : Serializable {
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(scidacFile,
+                                  double, version,
+                                  int, spacetime,
+				  std::string, dims, // must convert to int
+                                  int, volfmt);
+
+  std::vector<int> getDimensions(void) { 
+    std::stringstream stream(dims);
+    std::vector<int> dimensions;
+    int n;
+    while(stream >> n){
+      dimensions.push_back(n);
+    }
+    return dimensions;
+  }
+
+  void setDimensions(std::vector<int> dimensions) { 
+    char delimiter = ' ';
+    std::stringstream stream;
+    for(int i=0;i<dimensions.size();i++){ 
+      stream << dimensions[i];
+      if ( i != dimensions.size()-1) { 
+	stream << delimiter <<std::endl;
+      }
+    }
+    dims = stream.str();
+  }
+
+  // Constructor provides Grid
+  scidacFile() =default; // default constructor
+  scidacFile(GridBase * grid){
+    version      = 1.0;
+    spacetime    = grid->_ndimension;
+    setDimensions(grid->FullDimensions()); 
+    volfmt       = GRID_IO_SINGLEFILE;
+  }
+
+};
+
+///////////////////////////////////////////////////////////////////////
+// scidac-private-record-xml : example
+// <scidacRecord>
+// <version>1.1</version><date>Tue Jul 26 21:14:44 2011 UTC</date><recordtype>0</recordtype>
+// <datatype>QDP_D3_ColorMatrix</datatype><precision>D</precision><colors>3</colors><spins>4</spins>
+// <typesize>144</typesize><datacount>4</datacount>
+// </scidacRecord>
+///////////////////////////////////////////////////////////////////////
+
+struct scidacRecord : Serializable {
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(scidacRecord,
+                                  double, version,
+                                  std::string, date,
+				  int, recordtype,
+				  std::string, datatype,
+				  std::string, precision,
+				  int, colors,
+				  int, spins,
+				  int, typesize,
+				  int, datacount);
+
+  scidacRecord() { version =1.0; }
+
+};
+
+////////////////////////
+// ILDG format
+////////////////////////
+struct ildgFormat : Serializable {
+public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(ildgFormat,
+				  double, version,
+				  std::string, field,
+				  int, precision,
+				  int, lx,
+				  int, ly,
+				  int, lz,
+				  int, lt);
+  ildgFormat() { version=1.0; };
+};
+////////////////////////
+// USQCD info
+////////////////////////
+struct usqcdInfo : Serializable { 
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(usqcdInfo,
+				  double, version,
+				  double, plaq,
+				  double, linktr,
+				  std::string, info);
+  usqcdInfo() { 
+    version=1.0; 
+  };
+};
+////////////////////////
+// Scidac Checksum
+////////////////////////
+struct scidacChecksum : Serializable { 
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(scidacChecksum,
+				  double, version,
+				  std::string, suma,
+				  std::string, sumb);
+  scidacChecksum() { 
+    version=1.0; 
+  };
+};
+////////////////////////////////////////////////////////////////////////////////////////////////////////////////
+// Type:           scidac-file-xml         <title>MILC ILDG archival gauge configuration</title>
+////////////////////////////////////////////////////////////////////////////////////////////////////////////////
+
+////////////////////////////////////////////////////////////////////////////////////////////////////////////////
+// Type:           
+////////////////////////////////////////////////////////////////////////////////////////////////////////////////
+
+////////////////////////
+// Scidac private file xml 
+// <?xml version="1.0" encoding="UTF-8"?><scidacFile><version>1.1</version><spacetime>4</spacetime><dims>16 16 16 32 </dims><volfmt>0</volfmt></scidacFile> 
+////////////////////////                                                                                                                                                                              
+
+#if 0
+////////////////////////////////////////////////////////////////////////////////////////
+// From http://www.physics.utah.edu/~detar/scidac/qio_2p3.pdf
+////////////////////////////////////////////////////////////////////////////////////////
+struct usqcdPropFile : Serializable { 
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(usqcdPropFile,
+				  double, version,
+				  std::string, type,
+				  std::string, info);
+  usqcdPropFile() { 
+    version=1.0; 
+  };
+};
+struct usqcdSourceInfo : Serializable { 
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(usqcdSourceInfo,
+				  double, version,
+				  std::string, info);
+  usqcdSourceInfo() { 
+    version=1.0; 
+  };
+};
+struct usqcdPropInfo : Serializable { 
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(usqcdPropInfo,
+				  double, version,
+				  int, spin,
+				  int, color,
+				  std::string, info);
+  usqcdPropInfo() { 
+    version=1.0; 
+  };
+};
+#endif
+
+}
+#endif
+#endif
@@ -0,0 +1,325 @@
+/*************************************************************************************
+
+    Grid physics library, www.github.com/paboyle/Grid 
+
+    Source file: ./lib/parallelIO/NerscIO.h
+
+    Copyright (C) 2015
+
+
+    Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+
+    This program is free software; you can redistribute it and/or modify
+    it under the terms of the GNU General Public License as published by
+    the Free Software Foundation; either version 2 of the License, or
+    (at your option) any later version.
+
+    This program is distributed in the hope that it will be useful,
+    but WITHOUT ANY WARRANTY; without even the implied warranty of
+    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+    GNU General Public License for more details.
+
+    You should have received a copy of the GNU General Public License along
+    with this program; if not, write to the Free Software Foundation, Inc.,
+    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+    See the full license in the file "LICENSE" in the top level distribution directory
+*************************************************************************************/
+/*  END LEGAL */
+
+#include <algorithm>
+#include <iostream>
+#include <iomanip>
+#include <fstream>
+#include <map>
+#include <unistd.h>
+#include <sys/utsname.h>
+#include <pwd.h>
+
+namespace Grid {
+
+  ///////////////////////////////////////////////////////
+  // Precision mapping
+  ///////////////////////////////////////////////////////
+  template<class vobj> static std::string getFormatString (void)
+  {
+    std::string format;
+    typedef typename getPrecision<vobj>::real_scalar_type stype;
+    if ( sizeof(stype) == sizeof(float) ) {
+      format = std::string("IEEE32BIG");
+    }
+    if ( sizeof(stype) == sizeof(double) ) {
+      format = std::string("IEEE64BIG");
+    }
+    return format;
+  }
+  ////////////////////////////////////////////////////////////////////////////////
+  // header specification/interpretation
+  ////////////////////////////////////////////////////////////////////////////////
+    class FieldMetaData : Serializable {
+    public:
+
+      GRID_SERIALIZABLE_CLASS_MEMBERS(FieldMetaData,
+				      int, nd,
+				      std::vector<int>, dimension,
+				      std::vector<std::string>, boundary,
+				      int, data_start,
+				      std::string, hdr_version,
+				      std::string, storage_format,
+				      double, link_trace,
+				      double, plaquette,
+				      uint32_t, checksum,
+				      uint32_t, scidac_checksuma,
+				      uint32_t, scidac_checksumb,
+				      unsigned int, sequence_number,
+				      std::string, data_type,
+				      std::string, ensemble_id,
+				      std::string, ensemble_label,
+				      std::string, ildg_lfn,
+				      std::string, creator,
+				      std::string, creator_hardware,
+				      std::string, creation_date,
+				      std::string, archive_date,
+				      std::string, floating_point);
+      FieldMetaData(void) { 
+	nd=4;
+	dimension.resize(4);
+	boundary.resize(4);
+      }
+    };
+
+
+
+  namespace QCD {
+
+    using namespace Grid;
+
+
+    //////////////////////////////////////////////////////////////////////
+    // Bit and Physical Checksumming and QA of data
+    //////////////////////////////////////////////////////////////////////
+    inline void GridMetaData(GridBase *grid,FieldMetaData &header)
+    {
+      int nd = grid->_ndimension;
+      header.nd = nd;
+      header.dimension.resize(nd);
+      header.boundary.resize(nd);
+      for(int d=0;d<nd;d++) {
+	header.dimension[d] = grid->_fdimensions[d];
+      }
+      for(int d=0;d<nd;d++) {
+	header.boundary[d] = std::string("PERIODIC");
+      }
+    }
+
+    inline void MachineCharacteristics(FieldMetaData &header)
+    {
+      // Who
+      struct passwd *pw = getpwuid (getuid());
+      if (pw) header.creator = std::string(pw->pw_name); 
+
+      // When
+      std::time_t t = std::time(nullptr);
+      std::tm tm_ = *std::localtime(&t);
+      std::ostringstream oss; 
+      //      oss << std::put_time(&tm_, "%c %Z");
+      header.creation_date = oss.str();
+      header.archive_date  = header.creation_date;
+
+      // What
+      struct utsname name;  uname(&name);
+      header.creator_hardware = std::string(name.nodename)+"-";
+      header.creator_hardware+= std::string(name.machine)+"-";
+      header.creator_hardware+= std::string(name.sysname)+"-";
+      header.creator_hardware+= std::string(name.release);
+    }
+
+#define dump_meta_data(field, s)					\
+      s << "BEGIN_HEADER"      << std::endl;				\
+      s << "HDR_VERSION = "    << field.hdr_version    << std::endl;	\
+      s << "DATATYPE = "       << field.data_type      << std::endl;	\
+      s << "STORAGE_FORMAT = " << field.storage_format << std::endl;	\
+      for(int i=0;i<4;i++){						\
+	s << "DIMENSION_" << i+1 << " = " << field.dimension[i] << std::endl ; \
+      }									\
+      s << "LINK_TRACE = " << std::setprecision(10) << field.link_trace << std::endl; \
+      s << "PLAQUETTE  = " << std::setprecision(10) << field.plaquette  << std::endl; \
+      for(int i=0;i<4;i++){						\
+	s << "BOUNDARY_"<<i+1<<" = " << field.boundary[i] << std::endl;	\
+      }									\
+									\
+      s << "CHECKSUM = "<< std::hex << std::setw(10) << field.checksum << std::dec<<std::endl; \
+      s << "SCIDAC_CHECKSUMA = "<< std::hex << std::setw(10) << field.scidac_checksuma << std::dec<<std::endl; \
+      s << "SCIDAC_CHECKSUMB = "<< std::hex << std::setw(10) << field.scidac_checksumb << std::dec<<std::endl; \
+      s << "ENSEMBLE_ID = "     << field.ensemble_id      << std::endl;	\
+      s << "ENSEMBLE_LABEL = "  << field.ensemble_label   << std::endl;	\
+      s << "SEQUENCE_NUMBER = " << field.sequence_number  << std::endl;	\
+      s << "CREATOR = "         << field.creator          << std::endl;	\
+      s << "CREATOR_HARDWARE = "<< field.creator_hardware << std::endl;	\
+      s << "CREATION_DATE = "   << field.creation_date    << std::endl;	\
+      s << "ARCHIVE_DATE = "    << field.archive_date     << std::endl;	\
+      s << "FLOATING_POINT = "  << field.floating_point   << std::endl;	\
+      s << "END_HEADER"         << std::endl;
+
+template<class vobj> inline void PrepareMetaData(Lattice<vobj> & field, FieldMetaData &header)
+{
+  GridBase *grid = field._grid;
+  std::string format = getFormatString<vobj>();
+   header.floating_point = format;
+   header.checksum = 0x0; // Nersc checksum unused in ILDG, Scidac
+   GridMetaData(grid,header); 
+   MachineCharacteristics(header);
+ }
+ inline void GaugeStatistics(Lattice<vLorentzColourMatrixF> & data,FieldMetaData &header)
+ {
+   // How to convert data precision etc...
+   header.link_trace=Grid::QCD::WilsonLoops<PeriodicGimplF>::linkTrace(data);
+   header.plaquette =Grid::QCD::WilsonLoops<PeriodicGimplF>::avgPlaquette(data);
+ }
+ inline void GaugeStatistics(Lattice<vLorentzColourMatrixD> & data,FieldMetaData &header)
+ {
+   // How to convert data precision etc...
+   header.link_trace=Grid::QCD::WilsonLoops<PeriodicGimplD>::linkTrace(data);
+   header.plaquette =Grid::QCD::WilsonLoops<PeriodicGimplD>::avgPlaquette(data);
+ }
+ template<> inline void PrepareMetaData<vLorentzColourMatrixF>(Lattice<vLorentzColourMatrixF> & field, FieldMetaData &header)
+ {
+   
+   GridBase *grid = field._grid;
+   std::string format = getFormatString<vLorentzColourMatrixF>();
+   header.floating_point = format;
+   header.checksum = 0x0; // Nersc checksum unused in ILDG, Scidac
+   GridMetaData(grid,header); 
+   GaugeStatistics(field,header);
+   MachineCharacteristics(header);
+ }
+ template<> inline void PrepareMetaData<vLorentzColourMatrixD>(Lattice<vLorentzColourMatrixD> & field, FieldMetaData &header)
+ {
+   GridBase *grid = field._grid;
+   std::string format = getFormatString<vLorentzColourMatrixD>();
+   header.floating_point = format;
+   header.checksum = 0x0; // Nersc checksum unused in ILDG, Scidac
+   GridMetaData(grid,header); 
+   GaugeStatistics(field,header);
+   MachineCharacteristics(header);
+ }
+
+    //////////////////////////////////////////////////////////////////////
+    // Utilities ; these are QCD aware
+    //////////////////////////////////////////////////////////////////////
+    inline void reconstruct3(LorentzColourMatrix & cm)
+    {
+      const int x=0;
+      const int y=1;
+      const int z=2;
+      for(int mu=0;mu<Nd;mu++){
+	cm(mu)()(2,x) = adj(cm(mu)()(0,y)*cm(mu)()(1,z)-cm(mu)()(0,z)*cm(mu)()(1,y)); //x= yz-zy
+	cm(mu)()(2,y) = adj(cm(mu)()(0,z)*cm(mu)()(1,x)-cm(mu)()(0,x)*cm(mu)()(1,z)); //y= zx-xz
+	cm(mu)()(2,z) = adj(cm(mu)()(0,x)*cm(mu)()(1,y)-cm(mu)()(0,y)*cm(mu)()(1,x)); //z= xy-yx
+      }
+    }
+
+    ////////////////////////////////////////////////////////////////////////////////
+    // Some data types for intermediate storage
+    ////////////////////////////////////////////////////////////////////////////////
+    template<typename vtype> using iLorentzColour2x3 = iVector<iVector<iVector<vtype, Nc>, 2>, Nd >;
+
+    typedef iLorentzColour2x3<Complex>  LorentzColour2x3;
+    typedef iLorentzColour2x3<ComplexF> LorentzColour2x3F;
+    typedef iLorentzColour2x3<ComplexD> LorentzColour2x3D;
+
+/////////////////////////////////////////////////////////////////////////////////
+// Simple classes for precision conversion
+/////////////////////////////////////////////////////////////////////////////////
+template <class fobj, class sobj>
+struct BinarySimpleUnmunger {
+  typedef typename getPrecision<fobj>::real_scalar_type fobj_stype;
+  typedef typename getPrecision<sobj>::real_scalar_type sobj_stype;
+  
+  void operator()(sobj &in, fobj &out) {
+    // take word by word and transform accoding to the status
+    fobj_stype *out_buffer = (fobj_stype *)&out;
+    sobj_stype *in_buffer = (sobj_stype *)&in;
+    size_t fobj_words = sizeof(out) / sizeof(fobj_stype);
+    size_t sobj_words = sizeof(in) / sizeof(sobj_stype);
+    assert(fobj_words == sobj_words);
+    
+    for (unsigned int word = 0; word < sobj_words; word++)
+      out_buffer[word] = in_buffer[word];  // type conversion on the fly
+    
+  }
+};
+
+template <class fobj, class sobj>
+struct BinarySimpleMunger {
+  typedef typename getPrecision<fobj>::real_scalar_type fobj_stype;
+  typedef typename getPrecision<sobj>::real_scalar_type sobj_stype;
+
+  void operator()(fobj &in, sobj &out) {
+    // take word by word and transform accoding to the status
+    fobj_stype *in_buffer = (fobj_stype *)&in;
+    sobj_stype *out_buffer = (sobj_stype *)&out;
+    size_t fobj_words = sizeof(in) / sizeof(fobj_stype);
+    size_t sobj_words = sizeof(out) / sizeof(sobj_stype);
+    assert(fobj_words == sobj_words);
+    
+    for (unsigned int word = 0; word < sobj_words; word++)
+      out_buffer[word] = in_buffer[word];  // type conversion on the fly
+    
+  }
+};
+
+
+    template<class fobj,class sobj>
+    struct GaugeSimpleMunger{
+      void operator()(fobj &in, sobj &out) {
+        for (int mu = 0; mu < Nd; mu++) {
+          for (int i = 0; i < Nc; i++) {
+          for (int j = 0; j < Nc; j++) {
+	    out(mu)()(i, j) = in(mu)()(i, j);
+	  }}
+        }
+      };
+    };
+
+    template <class fobj, class sobj>
+    struct GaugeSimpleUnmunger {
+
+      void operator()(sobj &in, fobj &out) {
+        for (int mu = 0; mu < Nd; mu++) {
+          for (int i = 0; i < Nc; i++) {
+          for (int j = 0; j < Nc; j++) {
+	    out(mu)()(i, j) = in(mu)()(i, j);
+	  }}
+        }
+      };
+    };
+
+    template<class fobj,class sobj>
+    struct Gauge3x2munger{
+      void operator() (fobj &in,sobj &out){
+	for(int mu=0;mu<Nd;mu++){
+	  for(int i=0;i<2;i++){
+	  for(int j=0;j<3;j++){
+	    out(mu)()(i,j) = in(mu)(i)(j);
+	  }}
+	}
+	reconstruct3(out);
+      }
+    };
+
+    template<class fobj,class sobj>
+    struct Gauge3x2unmunger{
+      void operator() (sobj &in,fobj &out){
+	for(int mu=0;mu<Nd;mu++){
+	  for(int i=0;i<2;i++){
+	  for(int j=0;j<3;j++){
+	    out(mu)(i)(j) = in(mu)()(i,j);
+	  }}
+	}
+      }
+    };
+  }
+
+
+}
@@ -1,4 +1,4 @@
-    /*************************************************************************************
+/*************************************************************************************

    Grid physics library, www.github.com/paboyle/Grid 

@@ -6,9 +6,9 @@

    Copyright (C) 2015

-Author: Matt Spraggs <matthew.spraggs@gmail.com>
-Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-Author: paboyle <paboyle@ph.ed.ac.uk>
+    Author: Matt Spraggs <matthew.spraggs@gmail.com>
+    Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+    Author: paboyle <paboyle@ph.ed.ac.uk>

    This program is free software; you can redistribute it and/or modify
    it under the terms of the GNU General Public License as published by
@@ -25,253 +25,59 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.

    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
+*************************************************************************************/
+/*  END LEGAL */
 #ifndef GRID_NERSC_IO_H
 #define GRID_NERSC_IO_H

-#include <algorithm>
-#include <iostream>
-#include <iomanip>
-#include <fstream>
-#include <map>
-
-#include <unistd.h>
-#include <sys/utsname.h>
-#include <pwd.h>
-
 namespace Grid {
-namespace QCD {
+  namespace QCD {

-using namespace Grid;
+    using namespace Grid;

-////////////////////////////////////////////////////////////////////////////////
-// Some data types for intermediate storage
-////////////////////////////////////////////////////////////////////////////////
-  template<typename vtype> using iLorentzColour2x3 = iVector<iVector<iVector<vtype, Nc>, 2>, 4 >;
+    ////////////////////////////////////////////////////////////////////////////////
+    // Write and read from fstream; comput header offset for payload
+    ////////////////////////////////////////////////////////////////////////////////
+    class NerscIO : public BinaryIO { 
+    public:

-  typedef iLorentzColour2x3<Complex>  LorentzColour2x3;
-  typedef iLorentzColour2x3<ComplexF> LorentzColour2x3F;
-  typedef iLorentzColour2x3<ComplexD> LorentzColour2x3D;
-
-////////////////////////////////////////////////////////////////////////////////
-// header specification/interpretation
-////////////////////////////////////////////////////////////////////////////////
-class NerscField {
- public:
-    // header strings (not in order)
-    int dimension[4];
-    std::string boundary[4]; 
-    int data_start;
-    std::string hdr_version;
-    std::string storage_format;
-    // Checks on data
-    double link_trace;
-    double plaquette;
-    uint32_t checksum;
-    unsigned int sequence_number;
-    std::string data_type;
-    std::string ensemble_id ;
-    std::string ensemble_label ;
-    std::string creator ;
-    std::string creator_hardware ;
-    std::string creation_date ;
-    std::string archive_date ;
-    std::string floating_point;
-};
-
-//////////////////////////////////////////////////////////////////////
-// Bit and Physical Checksumming and QA of data
-//////////////////////////////////////////////////////////////////////
-
-inline void NerscGrid(GridBase *grid,NerscField &header)
-{
-  assert(grid->_ndimension==4);
-  for(int d=0;d<4;d++) {
-    header.dimension[d] = grid->_fdimensions[d];
-  }
-  for(int d=0;d<4;d++) {
-    header.boundary[d] = std::string("PERIODIC");
-  }
-}
-template<class GaugeField>
-inline void NerscStatistics(GaugeField & data,NerscField &header)
-{
-  // How to convert data precision etc...
-  header.link_trace=Grid::QCD::WilsonLoops<PeriodicGimplR>::linkTrace(data);
-  header.plaquette =Grid::QCD::WilsonLoops<PeriodicGimplR>::avgPlaquette(data);
-}
-
-inline void NerscMachineCharacteristics(NerscField &header)
-{
-  // Who
-  struct passwd *pw = getpwuid (getuid());
-  if (pw) header.creator = std::string(pw->pw_name); 
-
-  // When
-  std::time_t t = std::time(nullptr);
-  std::tm tm = *std::localtime(&t);
-  std::ostringstream oss; 
-  //  oss << std::put_time(&tm, "%c %Z");
-  header.creation_date = oss.str();
-  header.archive_date  = header.creation_date;
-
-  // What
-  struct utsname name;  uname(&name);
-  header.creator_hardware = std::string(name.nodename)+"-";
-  header.creator_hardware+= std::string(name.machine)+"-";
-  header.creator_hardware+= std::string(name.sysname)+"-";
-  header.creator_hardware+= std::string(name.release);
-
-}
-//////////////////////////////////////////////////////////////////////
-// Utilities ; these are QCD aware
-//////////////////////////////////////////////////////////////////////
-    inline void NerscChecksum(uint32_t *buf,uint32_t buf_size_bytes,uint32_t &csum)
-    {
-      BinaryIO::Uint32Checksum(buf,buf_size_bytes,csum);
-    }
-    inline void reconstruct3(LorentzColourMatrix & cm)
-    {
-      const int x=0;
-      const int y=1;
-      const int z=2;
-      for(int mu=0;mu<4;mu++){
-	cm(mu)()(2,x) = adj(cm(mu)()(0,y)*cm(mu)()(1,z)-cm(mu)()(0,z)*cm(mu)()(1,y)); //x= yz-zy
-	cm(mu)()(2,y) = adj(cm(mu)()(0,z)*cm(mu)()(1,x)-cm(mu)()(0,x)*cm(mu)()(1,z)); //y= zx-xz
-	cm(mu)()(2,z) = adj(cm(mu)()(0,x)*cm(mu)()(1,y)-cm(mu)()(0,y)*cm(mu)()(1,x)); //z= xy-yx
+      static inline void truncate(std::string file){
+	std::ofstream fout(file,std::ios::out);
      }
+  
+      static inline unsigned int writeHeader(FieldMetaData &field,std::string file)
+      {
+      std::ofstream fout(file,std::ios::out|std::ios::in);
+      fout.seekp(0,std::ios::beg);
+      dump_meta_data(field, fout);
+      field.data_start = fout.tellp();
+      return field.data_start;
    }

-    template<class fobj,class sobj>
-    struct NerscSimpleMunger{
+      // for the header-reader
+      static inline int readHeader(std::string file,GridBase *grid,  FieldMetaData &field)
+      {
+      int offset=0;
+      std::map<std::string,std::string> header;
+      std::string line;

-      void operator() (fobj &in,sobj &out,uint32_t &csum){
+      //////////////////////////////////////////////////
+      // read the header
+      //////////////////////////////////////////////////
+      std::ifstream fin(file);

-      for(int mu=0;mu<4;mu++){
-      for(int i=0;i<3;i++){
-      for(int j=0;j<3;j++){
-	out(mu)()(i,j) = in(mu)()(i,j);
-      }}}
-      NerscChecksum((uint32_t *)&in,sizeof(in),csum); 
-      };
-    };
+      getline(fin,line); // read one line and insist is 

-    template<class fobj,class sobj>
-    struct NerscSimpleUnmunger{
-      void operator() (sobj &in,fobj &out,uint32_t &csum){
-	for(int mu=0;mu<Nd;mu++){
-	for(int i=0;i<Nc;i++){
-	for(int j=0;j<Nc;j++){
-	  out(mu)()(i,j) = in(mu)()(i,j);
-	}}}
-	NerscChecksum((uint32_t *)&out,sizeof(out),csum); 
-      };
-    };
- 
-    template<class fobj,class sobj>
-    struct Nersc3x2munger{
-      void operator() (fobj &in,sobj &out,uint32_t &csum){
-     
-	NerscChecksum((uint32_t *)&in,sizeof(in),csum); 
+      removeWhitespace(line);
+      std::cout << GridLogMessage << "* " << line << std::endl;

-	for(int mu=0;mu<4;mu++){
-	  for(int i=0;i<2;i++){
-	    for(int j=0;j<3;j++){
-	      out(mu)()(i,j) = in(mu)(i)(j);
-	    }}
-	}
-	reconstruct3(out);
-      }
-    };
+      assert(line==std::string("BEGIN_HEADER"));

-    template<class fobj,class sobj>
-    struct Nersc3x2unmunger{
-
-      void operator() (sobj &in,fobj &out,uint32_t &csum){
-
-
-	for(int mu=0;mu<4;mu++){
-	  for(int i=0;i<2;i++){
-	    for(int j=0;j<3;j++){
-	      out(mu)(i)(j) = in(mu)()(i,j);
-	    }}
-	}
-
-	NerscChecksum((uint32_t *)&out,sizeof(out),csum); 
-
-      }
-    };
-
-
-////////////////////////////////////////////////////////////////////////////////
-// Write and read from fstream; comput header offset for payload
-////////////////////////////////////////////////////////////////////////////////
-class NerscIO : public BinaryIO { 
- public:
-
-  static inline void truncate(std::string file){
-    std::ofstream fout(file,std::ios::out);
-  }
-  
-  #define dump_nersc_header(field, s)\
-  s << "BEGIN_HEADER"      << std::endl;\
-  s << "HDR_VERSION = "    << field.hdr_version    << std::endl;\
-  s << "DATATYPE = "       << field.data_type      << std::endl;\
-  s << "STORAGE_FORMAT = " << field.storage_format << std::endl;\
-  for(int i=0;i<4;i++){\
-    s << "DIMENSION_" << i+1 << " = " << field.dimension[i] << std::endl ;\
-  }\
-  s << "LINK_TRACE = " << std::setprecision(10) << field.link_trace << std::endl;\
-  s << "PLAQUETTE  = " << std::setprecision(10) << field.plaquette  << std::endl;\
-  for(int i=0;i<4;i++){\
-    s << "BOUNDARY_"<<i+1<<" = " << field.boundary[i] << std::endl;\
-  }\
-  \
-  s << "CHECKSUM = "<< std::hex << std::setw(10) << field.checksum << std::dec<<std::endl;\
-  s << "ENSEMBLE_ID = "     << field.ensemble_id      << std::endl;\
-  s << "ENSEMBLE_LABEL = "  << field.ensemble_label   << std::endl;\
-  s << "SEQUENCE_NUMBER = " << field.sequence_number  << std::endl;\
-  s << "CREATOR = "         << field.creator          << std::endl;\
-  s << "CREATOR_HARDWARE = "<< field.creator_hardware << std::endl;\
-  s << "CREATION_DATE = "   << field.creation_date    << std::endl;\
-  s << "ARCHIVE_DATE = "    << field.archive_date     << std::endl;\
-  s << "FLOATING_POINT = "  << field.floating_point   << std::endl;\
-  s << "END_HEADER"         << std::endl;
-  
-  static inline unsigned int writeHeader(NerscField &field,std::string file)
-  {
-    std::ofstream fout(file,std::ios::out|std::ios::in);
-    fout.seekp(0,std::ios::beg);
-    dump_nersc_header(field, fout);
-    field.data_start = fout.tellp();
-    return field.data_start;
-}
-
-// for the header-reader
-static inline int readHeader(std::string file,GridBase *grid,  NerscField &field)
-{
-  int offset=0;
-  std::map<std::string,std::string> header;
-  std::string line;
-
-  //////////////////////////////////////////////////
-  // read the header
-  //////////////////////////////////////////////////
-  std::ifstream fin(file);
-
-  getline(fin,line); // read one line and insist is 
-
-  removeWhitespace(line);
-  std::cout << GridLogMessage << "* " << line << std::endl;
-
-  assert(line==std::string("BEGIN_HEADER"));
-
-  do {
-    getline(fin,line); // read one line
-    std::cout << GridLogMessage << "* "<<line<< std::endl;
-    int eq = line.find("=");
-    if(eq >0) {
+      do {
+      getline(fin,line); // read one line
+      std::cout << GridLogMessage << "* "<<line<< std::endl;
+      int eq = line.find("=");
+      if(eq >0) {
      std::string key=line.substr(0,eq);
      std::string val=line.substr(eq+1);
      removeWhitespace(key);
@@ -279,275 +85,269 @@ static inline int readHeader(std::string file,GridBase *grid,  NerscField &field
      
      header[key] = val;
    }
-  } while( line.find("END_HEADER") == std::string::npos );
+    } while( line.find("END_HEADER") == std::string::npos );

-  field.data_start = fin.tellg();
+      field.data_start = fin.tellg();

-  //////////////////////////////////////////////////
-  // chomp the values
-  //////////////////////////////////////////////////
-  field.hdr_version    = header["HDR_VERSION"];
-  field.data_type      = header["DATATYPE"];
-  field.storage_format = header["STORAGE_FORMAT"];
+      //////////////////////////////////////////////////
+      // chomp the values
+      //////////////////////////////////////////////////
+      field.hdr_version    = header["HDR_VERSION"];
+      field.data_type      = header["DATATYPE"];
+      field.storage_format = header["STORAGE_FORMAT"];
  
-  field.dimension[0] = std::stol(header["DIMENSION_1"]);
-  field.dimension[1] = std::stol(header["DIMENSION_2"]);
-  field.dimension[2] = std::stol(header["DIMENSION_3"]);
-  field.dimension[3] = std::stol(header["DIMENSION_4"]);
+      field.dimension[0] = std::stol(header["DIMENSION_1"]);
+      field.dimension[1] = std::stol(header["DIMENSION_2"]);
+      field.dimension[2] = std::stol(header["DIMENSION_3"]);
+      field.dimension[3] = std::stol(header["DIMENSION_4"]);

-  assert(grid->_ndimension == 4);
-  for(int d=0;d<4;d++){
-    assert(grid->_fdimensions[d]==field.dimension[d]);
-  }
-
-  field.link_trace = std::stod(header["LINK_TRACE"]);
-  field.plaquette  = std::stod(header["PLAQUETTE"]);
-
-  field.boundary[0] = header["BOUNDARY_1"];
-  field.boundary[1] = header["BOUNDARY_2"];
-  field.boundary[2] = header["BOUNDARY_3"];
-  field.boundary[3] = header["BOUNDARY_4"];
-
-  field.checksum = std::stoul(header["CHECKSUM"],0,16);
-  field.ensemble_id      = header["ENSEMBLE_ID"];
-  field.ensemble_label   = header["ENSEMBLE_LABEL"];
-  field.sequence_number  = std::stol(header["SEQUENCE_NUMBER"]);
-  field.creator          = header["CREATOR"];
-  field.creator_hardware = header["CREATOR_HARDWARE"];
-  field.creation_date    = header["CREATION_DATE"];
-  field.archive_date     = header["ARCHIVE_DATE"];
-  field.floating_point   = header["FLOATING_POINT"];
-
-  return field.data_start;
-}
-
-/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////
-// Now the meat: the object readers
-/////////////////////////////////////////////////////////////////////////////////////////////////////////////////////
-#define PARALLEL_READ
-#define PARALLEL_WRITE
-
-template<class vsimd>
-static inline void readConfiguration(Lattice<iLorentzColourMatrix<vsimd> > &Umu,NerscField& header,std::string file)
-{
-  typedef Lattice<iLorentzColourMatrix<vsimd> > GaugeField;
-
-  GridBase *grid = Umu._grid;
-  int offset = readHeader(file,Umu._grid,header);
-
-  NerscField clone(header);
-
-  std::string format(header.floating_point);
-
-  int ieee32big = (format == std::string("IEEE32BIG"));
-  int ieee32    = (format == std::string("IEEE32"));
-  int ieee64big = (format == std::string("IEEE64BIG"));
-  int ieee64    = (format == std::string("IEEE64"));
-
-  uint32_t csum;
-  // depending on datatype, set up munger;
-  // munger is a function of <floating point, Real, data_type>
-  if ( header.data_type == std::string("4D_SU3_GAUGE") ) {
-    if ( ieee32 || ieee32big ) {
-#ifdef PARALLEL_READ
-      csum=BinaryIO::readObjectParallel<iLorentzColourMatrix<vsimd>, LorentzColour2x3F> 
-	(Umu,file,Nersc3x2munger<LorentzColour2x3F,LorentzColourMatrix>(), offset,format);
-#else
-      csum=BinaryIO::readObjectSerial<iLorentzColourMatrix<vsimd>, LorentzColour2x3F> 
-	(Umu,file,Nersc3x2munger<LorentzColour2x3F,LorentzColourMatrix>(), offset,format);
-#endif
+      assert(grid->_ndimension == 4);
+      for(int d=0;d<4;d++){
+      assert(grid->_fdimensions[d]==field.dimension[d]);
    }
-    if ( ieee64 || ieee64big ) {
-#ifdef PARALLEL_READ
-      csum=BinaryIO::readObjectParallel<iLorentzColourMatrix<vsimd>, LorentzColour2x3D> 
-      	(Umu,file,Nersc3x2munger<LorentzColour2x3D,LorentzColourMatrix>(),offset,format);
-#else 
-      csum=BinaryIO::readObjectSerial<iLorentzColourMatrix<vsimd>, LorentzColour2x3D> 
-      	(Umu,file,Nersc3x2munger<LorentzColour2x3D,LorentzColourMatrix>(),offset,format);
-#endif
+
+      field.link_trace = std::stod(header["LINK_TRACE"]);
+      field.plaquette  = std::stod(header["PLAQUETTE"]);
+
+      field.boundary[0] = header["BOUNDARY_1"];
+      field.boundary[1] = header["BOUNDARY_2"];
+      field.boundary[2] = header["BOUNDARY_3"];
+      field.boundary[3] = header["BOUNDARY_4"];
+
+      field.checksum = std::stoul(header["CHECKSUM"],0,16);
+      field.ensemble_id      = header["ENSEMBLE_ID"];
+      field.ensemble_label   = header["ENSEMBLE_LABEL"];
+      field.sequence_number  = std::stol(header["SEQUENCE_NUMBER"]);
+      field.creator          = header["CREATOR"];
+      field.creator_hardware = header["CREATOR_HARDWARE"];
+      field.creation_date    = header["CREATION_DATE"];
+      field.archive_date     = header["ARCHIVE_DATE"];
+      field.floating_point   = header["FLOATING_POINT"];
+
+      return field.data_start;
    }
-  } else if ( header.data_type == std::string("4D_SU3_GAUGE_3x3") ) {
-    if ( ieee32 || ieee32big ) {
-#ifdef PARALLEL_READ
-      csum=BinaryIO::readObjectParallel<iLorentzColourMatrix<vsimd>,LorentzColourMatrixF>
-	(Umu,file,NerscSimpleMunger<LorentzColourMatrixF,LorentzColourMatrix>(),offset,format);
-#else
-      csum=BinaryIO::readObjectSerial<iLorentzColourMatrix<vsimd>,LorentzColourMatrixF>
-	(Umu,file,NerscSimpleMunger<LorentzColourMatrixF,LorentzColourMatrix>(),offset,format);
-#endif
+
+    /////////////////////////////////////////////////////////////////////////////////////////////////////////////////////
+    // Now the meat: the object readers
+    /////////////////////////////////////////////////////////////////////////////////////////////////////////////////////
+
+    template<class vsimd>
+    static inline void readConfiguration(Lattice<iLorentzColourMatrix<vsimd> > &Umu,
+					 FieldMetaData& header,
+					 std::string file)
+    {
+      typedef Lattice<iLorentzColourMatrix<vsimd> > GaugeField;
+
+      GridBase *grid = Umu._grid;
+      int offset = readHeader(file,Umu._grid,header);
+
+      FieldMetaData clone(header);
+
+      std::string format(header.floating_point);
+
+      int ieee32big = (format == std::string("IEEE32BIG"));
+      int ieee32    = (format == std::string("IEEE32"));
+      int ieee64big = (format == std::string("IEEE64BIG"));
+      int ieee64    = (format == std::string("IEEE64"));
+
+      uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+      // depending on datatype, set up munger;
+      // munger is a function of <floating point, Real, data_type>
+      if ( header.data_type == std::string("4D_SU3_GAUGE") ) {
+	if ( ieee32 || ieee32big ) {
+	  BinaryIO::readLatticeObject<iLorentzColourMatrix<vsimd>, LorentzColour2x3F> 
+	    (Umu,file,Gauge3x2munger<LorentzColour2x3F,LorentzColourMatrix>(), offset,format,
+	     nersc_csum,scidac_csuma,scidac_csumb);
+	}
+	if ( ieee64 || ieee64big ) {
+	  BinaryIO::readLatticeObject<iLorentzColourMatrix<vsimd>, LorentzColour2x3D> 
+	    (Umu,file,Gauge3x2munger<LorentzColour2x3D,LorentzColourMatrix>(),offset,format,
+	     nersc_csum,scidac_csuma,scidac_csumb);
+	}
+      } else if ( header.data_type == std::string("4D_SU3_GAUGE_3x3") ) {
+	if ( ieee32 || ieee32big ) {
+	  BinaryIO::readLatticeObject<iLorentzColourMatrix<vsimd>,LorentzColourMatrixF>
+	    (Umu,file,GaugeSimpleMunger<LorentzColourMatrixF,LorentzColourMatrix>(),offset,format,
+	     nersc_csum,scidac_csuma,scidac_csumb);
+	}
+	if ( ieee64 || ieee64big ) {
+	  BinaryIO::readLatticeObject<iLorentzColourMatrix<vsimd>,LorentzColourMatrixD>
+	    (Umu,file,GaugeSimpleMunger<LorentzColourMatrixD,LorentzColourMatrix>(),offset,format,
+	     nersc_csum,scidac_csuma,scidac_csumb);
+	}
+      } else {
+	assert(0);
+      }
+
+      GaugeStatistics(Umu,clone);
+
+      std::cout<<GridLogMessage <<"NERSC Configuration "<<file<<" checksum "<<std::hex<<nersc_csum<< std::dec
+	       <<" header   "<<std::hex<<header.checksum<<std::dec <<std::endl;
+      std::cout<<GridLogMessage <<"NERSC Configuration "<<file<<" plaquette "<<clone.plaquette
+	       <<" header    "<<header.plaquette<<std::endl;
+      std::cout<<GridLogMessage <<"NERSC Configuration "<<file<<" link_trace "<<clone.link_trace
+	       <<" header    "<<header.link_trace<<std::endl;
+
+      if ( fabs(clone.plaquette -header.plaquette ) >=  1.0e-5 ) { 
+	std::cout << " Plaquette mismatch "<<std::endl;
+	std::cout << Umu[0]<<std::endl;
+	std::cout << Umu[1]<<std::endl;
+      }
+      if ( nersc_csum != header.checksum ) { 
+	std::cerr << " checksum mismatch " << std::endl;
+	std::cerr << " plaqs " << clone.plaquette << " " << header.plaquette << std::endl;
+	std::cerr << " trace " << clone.link_trace<< " " << header.link_trace<< std::endl;
+	std::cerr << " nersc_csum  " <<std::hex<< nersc_csum << " " << header.checksum<< std::dec<< std::endl;
+	exit(0);
+      }
+      assert(fabs(clone.plaquette -header.plaquette ) < 1.0e-5 );
+      assert(fabs(clone.link_trace-header.link_trace) < 1.0e-6 );
+      assert(nersc_csum == header.checksum );
+      
+      std::cout<<GridLogMessage <<"NERSC Configuration "<<file<< " and plaquette, link trace, and checksum agree"<<std::endl;
    }
-    if ( ieee64 || ieee64big ) {
-#ifdef PARALLEL_READ
-      csum=BinaryIO::readObjectParallel<iLorentzColourMatrix<vsimd>,LorentzColourMatrixD>
-	(Umu,file,NerscSimpleMunger<LorentzColourMatrixD,LorentzColourMatrix>(),offset,format);
-#else
-      csum=BinaryIO::readObjectSerial<iLorentzColourMatrix<vsimd>,LorentzColourMatrixD>
-	(Umu,file,NerscSimpleMunger<LorentzColourMatrixD,LorentzColourMatrix>(),offset,format);
-#endif
-    }
-  } else {
-    assert(0);
-  }

-  NerscStatistics<GaugeField>(Umu,clone);
+      template<class vsimd>
+      static inline void writeConfiguration(Lattice<iLorentzColourMatrix<vsimd> > &Umu,
+					    std::string file, 
+					    int two_row,
+					    int bits32)
+      {
+	typedef Lattice<iLorentzColourMatrix<vsimd> > GaugeField;

-  std::cout<<GridLogMessage <<"NERSC Configuration "<<file<<" checksum "<<std::hex<<            csum<< std::dec
-	                                                  <<" header   "<<std::hex<<header.checksum<<std::dec <<std::endl;
-  std::cout<<GridLogMessage <<"NERSC Configuration "<<file<<" plaquette "<<clone.plaquette
-	                                                  <<" header    "<<header.plaquette<<std::endl;
-  std::cout<<GridLogMessage <<"NERSC Configuration "<<file<<" link_trace "<<clone.link_trace
-	                                                  <<" header    "<<header.link_trace<<std::endl;
-  assert(fabs(clone.plaquette -header.plaquette ) < 1.0e-5 );
-  assert(fabs(clone.link_trace-header.link_trace) < 1.0e-6 );
-  assert(csum == header.checksum );
+	typedef iLorentzColourMatrix<vsimd> vobj;
+	typedef typename vobj::scalar_object sobj;

-  std::cout<<GridLogMessage <<"NERSC Configuration "<<file<< " and plaquette, link trace, and checksum agree"<<std::endl;
-}
+	FieldMetaData header;
+	///////////////////////////////////////////
+	// Following should become arguments
+	///////////////////////////////////////////
+	header.sequence_number = 1;
+	header.ensemble_id     = "UKQCD";
+	header.ensemble_label  = "DWF";

-template<class vsimd>
-static inline void writeConfiguration(Lattice<iLorentzColourMatrix<vsimd> > &Umu,std::string file, int two_row,int bits32)
-{
-  typedef Lattice<iLorentzColourMatrix<vsimd> > GaugeField;
-
-  typedef iLorentzColourMatrix<vsimd> vobj;
-  typedef typename vobj::scalar_object sobj;
-
-  // Following should become arguments
-  NerscField header;
-  header.sequence_number = 1;
-  header.ensemble_id     = "UKQCD";
-  header.ensemble_label  = "DWF";
-
-  typedef LorentzColourMatrixD fobj3D;
-  typedef LorentzColour2x3D    fobj2D;
-  typedef LorentzColourMatrixF fobj3f;
-  typedef LorentzColour2x3F    fobj2f;
-
-  GridBase *grid = Umu._grid;
-
-  NerscGrid(grid,header);
-  NerscStatistics<GaugeField>(Umu,header);
-  NerscMachineCharacteristics(header);
-
-  uint32_t csum;
-  int offset;
+	typedef LorentzColourMatrixD fobj3D;
+	typedef LorentzColour2x3D    fobj2D;
  
-  truncate(file);
+	GridBase *grid = Umu._grid;

-  if ( two_row ) { 
+	GridMetaData(grid,header);
+	assert(header.nd==4);
+	GaugeStatistics(Umu,header);
+	MachineCharacteristics(header);

-    header.floating_point = std::string("IEEE64BIG");
-    header.data_type      = std::string("4D_SU3_GAUGE");
-    Nersc3x2unmunger<fobj2D,sobj> munge;
-    BinaryIO::Uint32Checksum<vobj,fobj2D>(Umu, munge,header.checksum);
-    offset = writeHeader(header,file);
-#ifdef PARALLEL_WRITE
-    csum=BinaryIO::writeObjectParallel<vobj,fobj2D>(Umu,file,munge,offset,header.floating_point);
-#else
-    csum=BinaryIO::writeObjectSerial<vobj,fobj2D>(Umu,file,munge,offset,header.floating_point);
-#endif
+	int offset;
+  
+	truncate(file);

-  } else { 
-    header.floating_point = std::string("IEEE64BIG");
-    header.data_type      = std::string("4D_SU3_GAUGE_3x3");
-    NerscSimpleUnmunger<fobj3D,sobj> munge;
-    BinaryIO::Uint32Checksum<vobj,fobj3D>(Umu, munge,header.checksum);
-    offset = writeHeader(header,file);
-#ifdef PARALLEL_WRITE
-    csum=BinaryIO::writeObjectParallel<vobj,fobj3D>(Umu,file,munge,offset,header.floating_point);
-#else
-    csum=BinaryIO::writeObjectSerial<vobj,fobj3D>(Umu,file,munge,offset,header.floating_point);
-#endif
-  }
+	// Sod it -- always write 3x3 double
+	header.floating_point = std::string("IEEE64BIG");
+	header.data_type      = std::string("4D_SU3_GAUGE_3x3");
+	GaugeSimpleUnmunger<fobj3D,sobj> munge;
+	offset = writeHeader(header,file);

-  std::cout<<GridLogMessage <<"Written NERSC Configuration "<<file<< " checksum "<<std::hex<<csum<< std::dec<<" plaq "<< header.plaquette <<std::endl;
+	uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+	BinaryIO::writeLatticeObject<vobj,fobj3D>(Umu,file,munge,offset,header.floating_point,
+								  nersc_csum,scidac_csuma,scidac_csumb);
+	header.checksum = nersc_csum;
+	writeHeader(header,file);

- }
+	std::cout<<GridLogMessage <<"Written NERSC Configuration on "<< file << " checksum "
+		 <<std::hex<<header.checksum
+		 <<std::dec<<" plaq "<< header.plaquette <<std::endl;

+      }
+      ///////////////////////////////
+      // RNG state
+      ///////////////////////////////
+      static inline void writeRNGState(GridSerialRNG &serial,GridParallelRNG &parallel,std::string file)
+      {
+	typedef typename GridParallelRNG::RngStateType RngStateType;

-    ///////////////////////////////
-    // RNG state
-    ///////////////////////////////
-static inline void writeRNGState(GridSerialRNG &serial,GridParallelRNG &parallel,std::string file)
-{
-  typedef typename GridParallelRNG::RngStateType RngStateType;
+	// Following should become arguments
+	FieldMetaData header;
+	header.sequence_number = 1;
+	header.ensemble_id     = "UKQCD";
+	header.ensemble_label  = "DWF";

-  // Following should become arguments
-  NerscField header;
-  header.sequence_number = 1;
-  header.ensemble_id     = "UKQCD";
-  header.ensemble_label  = "DWF";
+	GridBase *grid = parallel._grid;

-  GridBase *grid = parallel._grid;
+	GridMetaData(grid,header);
+	assert(header.nd==4);
+	header.link_trace=0.0;
+	header.plaquette=0.0;
+	MachineCharacteristics(header);

-  NerscGrid(grid,header);
-  header.link_trace=0.0;
-  header.plaquette=0.0;
-  NerscMachineCharacteristics(header);
-
-  uint32_t csum;
-  int offset;
+	int offset;
  
 #ifdef RNG_RANLUX
-    header.floating_point = std::string("UINT64");
-    header.data_type      = std::string("RANLUX48");
+	header.floating_point = std::string("UINT64");
+	header.data_type      = std::string("RANLUX48");
 #endif
 #ifdef RNG_MT19937
-    header.floating_point = std::string("UINT32");
-    header.data_type      = std::string("MT19937");
+	header.floating_point = std::string("UINT32");
+	header.data_type      = std::string("MT19937");
 #endif
 #ifdef RNG_SITMO
-    header.floating_point = std::string("UINT64");
-    header.data_type      = std::string("SITMO");
+	header.floating_point = std::string("UINT64");
+	header.data_type      = std::string("SITMO");
 #endif

-  truncate(file);
-  offset = writeHeader(header,file);
-  csum=BinaryIO::writeRNGSerial(serial,parallel,file,offset);
-  header.checksum = csum;
-  offset = writeHeader(header,file);
+	truncate(file);
+	offset = writeHeader(header,file);
+	uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+	BinaryIO::writeRNG(serial,parallel,file,offset,nersc_csum,scidac_csuma,scidac_csumb);
+	header.checksum = nersc_csum;
+	offset = writeHeader(header,file);

-  std::cout<<GridLogMessage <<"Written NERSC RNG STATE "<<file<< " checksum "<<std::hex<<csum<<std::dec<<std::endl;
+	std::cout<<GridLogMessage 
+		 <<"Written NERSC RNG STATE "<<file<< " checksum "
+		 <<std::hex<<header.checksum
+		 <<std::dec<<std::endl;

- }
+      }
    
-static inline void readRNGState(GridSerialRNG &serial,GridParallelRNG & parallel,NerscField& header,std::string file)
-{
-  typedef typename GridParallelRNG::RngStateType RngStateType;
+      static inline void readRNGState(GridSerialRNG &serial,GridParallelRNG & parallel,FieldMetaData& header,std::string file)
+      {
+	typedef typename GridParallelRNG::RngStateType RngStateType;

-  GridBase *grid = parallel._grid;
+	GridBase *grid = parallel._grid;

-  int offset = readHeader(file,grid,header);
+	int offset = readHeader(file,grid,header);

-  NerscField clone(header);
+	FieldMetaData clone(header);

-  std::string format(header.floating_point);
-  std::string data_type(header.data_type);
+	std::string format(header.floating_point);
+	std::string data_type(header.data_type);

 #ifdef RNG_RANLUX
-  assert(format == std::string("UINT64"));
-  assert(data_type == std::string("RANLUX48"));
+	assert(format == std::string("UINT64"));
+	assert(data_type == std::string("RANLUX48"));
 #endif
 #ifdef RNG_MT19937
-  assert(format == std::string("UINT32"));
-  assert(data_type == std::string("MT19937"));
+	assert(format == std::string("UINT32"));
+	assert(data_type == std::string("MT19937"));
 #endif
 #ifdef RNG_SITMO
-  assert(format == std::string("UINT64"));
-  assert(data_type == std::string("SITMO"));
+	assert(format == std::string("UINT64"));
+	assert(data_type == std::string("SITMO"));
 #endif

-  // depending on datatype, set up munger;
-  // munger is a function of <floating point, Real, data_type>
-  uint32_t csum=BinaryIO::readRNGSerial(serial,parallel,file,offset);
+	// depending on datatype, set up munger;
+	// munger is a function of <floating point, Real, data_type>
+	uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+	BinaryIO::readRNG(serial,parallel,file,offset,nersc_csum,scidac_csuma,scidac_csumb);

-  assert(csum == header.checksum );
+	if ( nersc_csum != header.checksum ) { 
+	  std::cerr << "checksum mismatch "<<std::hex<< nersc_csum <<" "<<header.checksum<<std::dec<<std::endl;
+	  exit(0);
+	}
+	assert(nersc_csum == header.checksum );

-  std::cout<<GridLogMessage <<"Read NERSC RNG file "<<file<< " format "<< data_type <<std::endl;
-}
+	std::cout<<GridLogMessage <<"Read NERSC RNG file "<<file<< " format "<< data_type <<std::endl;
+      }

-};
+    };

-
-}}
+  }}
 #endif
@@ -40,7 +40,7 @@ const PerformanceCounter::PerformanceCounterConfig PerformanceCounter::Performan
  { PERF_TYPE_HARDWARE, PERF_COUNT_HW_CPU_CYCLES          ,  "CPUCYCLES.........." , INSTRUCTIONS},
  { PERF_TYPE_HARDWARE, PERF_COUNT_HW_INSTRUCTIONS        ,  "INSTRUCTIONS......." , CPUCYCLES   },
    // 4
-#ifdef AVX512
+#ifdef KNL
    { PERF_TYPE_RAW, RawConfig(0x40,0x04), "ALL_LOADS..........", CPUCYCLES    },
    { PERF_TYPE_RAW, RawConfig(0x01,0x04), "L1_MISS_LOADS......", L1D_READ_ACCESS  },
    { PERF_TYPE_RAW, RawConfig(0x40,0x04), "ALL_LOADS..........", L1D_READ_ACCESS    },
@@ -205,13 +205,14 @@ public:
  void Stop(void) {
    count=0;
    cycles=0;
-    size_t ign;
 #ifdef __linux__
+    ssize_t ign;
    if ( fd!= -1) {
      ::ioctl(fd, PERF_EVENT_IOC_DISABLE, 0);
      ::ioctl(cyclefd, PERF_EVENT_IOC_DISABLE, 0);
      ign=::read(fd, &count, sizeof(long long));
-      ign=::read(cyclefd, &cycles, sizeof(long long));
+      ign+=::read(cyclefd, &cycles, sizeof(long long));
+      assert(ign=2*sizeof(long long));
    }
    elapsed = cyclecount() - begin;
 #else
@@ -0,0 +1,124 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/QCD.h
+
+Copyright (C) 2015
+
+Author: Azusa Yamaguchi <ayamaguc@staffmail.ed.ac.uk>
+Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+Author: Peter Boyle <peterboyle@Peters-MacBook-Pro-2.local>
+Author: neo <cossu@post.kek.jp>
+Author: paboyle <paboyle@ph.ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_LT_H
+#define GRID_LT_H
+namespace Grid{
+
+// First steps in the complete generalization of the Physics part
+// Design not final
+namespace LatticeTheories {
+
+template <int Dimensions>
+struct LatticeTheory {
+  static const int Nd = Dimensions;
+  static const int Nds = Dimensions * 2;  // double stored field
+  template <typename vtype>
+  using iSinglet = iScalar<iScalar<iScalar<vtype> > >;
+};
+
+template <int Dimensions, int Colours>
+struct LatticeGaugeTheory : public LatticeTheory<Dimensions> {
+  static const int Nds = Dimensions * 2;
+  static const int Nd = Dimensions;
+  static const int Nc = Colours;
+
+  template <typename vtype> 
+  using iColourMatrix = iScalar<iScalar<iMatrix<vtype, Nc> > >;
+  template <typename vtype>
+  using iLorentzColourMatrix = iVector<iScalar<iMatrix<vtype, Nc> >, Nd>;
+  template <typename vtype>
+  using iDoubleStoredColourMatrix = iVector<iScalar<iMatrix<vtype, Nc> >, Nds>;
+  template <typename vtype>
+  using iColourVector = iScalar<iScalar<iVector<vtype, Nc> > >;
+};
+
+template <int Dimensions, int Colours, int Spin>
+struct FermionicLatticeGaugeTheory
+    : public LatticeGaugeTheory<Dimensions, Colours> {
+  static const int Nd = Dimensions;
+  static const int Nds = Dimensions * 2;
+  static const int Nc = Colours;
+  static const int Ns = Spin;
+
+  template <typename vtype>
+  using iSpinMatrix = iScalar<iMatrix<iScalar<vtype>, Ns> >;
+  template <typename vtype>
+  using iSpinColourMatrix = iScalar<iMatrix<iMatrix<vtype, Nc>, Ns> >;
+  template <typename vtype>
+  using iSpinVector = iScalar<iVector<iScalar<vtype>, Ns> >;
+  template <typename vtype>
+  using iSpinColourVector = iScalar<iVector<iVector<vtype, Nc>, Ns> >;
+  // These 2 only if Spin is a multiple of 2
+  static const int Nhs = Spin / 2;
+  template <typename vtype>
+  using iHalfSpinVector = iScalar<iVector<iScalar<vtype>, Nhs> >;
+  template <typename vtype>
+  using iHalfSpinColourVector = iScalar<iVector<iVector<vtype, Nc>, Nhs> >;
+
+  //tests
+  typedef iColourMatrix<Complex> ColourMatrix;
+  typedef iColourMatrix<ComplexF> ColourMatrixF;
+  typedef iColourMatrix<ComplexD> ColourMatrixD;
+
+
+};
+
+// Examples, not complete now.
+struct QCD : public FermionicLatticeGaugeTheory<4, 3, 4> {
+    static const int Xp = 0;
+    static const int Yp = 1;
+    static const int Zp = 2;
+    static const int Tp = 3;
+    static const int Xm = 4;
+    static const int Ym = 5;
+    static const int Zm = 6;
+    static const int Tm = 7;
+
+    typedef FermionicLatticeGaugeTheory FLGT;
+
+    typedef FLGT::iSpinMatrix<Complex  >          SpinMatrix;
+    typedef FLGT::iSpinMatrix<ComplexF >          SpinMatrixF;
+    typedef FLGT::iSpinMatrix<ComplexD >          SpinMatrixD;
+
+};
+struct QED : public FermionicLatticeGaugeTheory<4, 1, 4> {//fill
+};
+
+template <int Dimensions>
+struct Scalar : public LatticeTheory<Dimensions> {};
+
+};  // LatticeTheories
+
+} // Grid
+
+#endif
@@ -32,14 +32,14 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
 #ifndef GRID_QCD_BASE_H
 #define GRID_QCD_BASE_H
 namespace Grid{
-
 namespace QCD {

    static const int Xdir = 0;
    static const int Ydir = 1;
    static const int Zdir = 2;
    static const int Tdir = 3;
-    
+
+  
    static const int Xp = 0;
    static const int Yp = 1;
    static const int Zp = 2;
@@ -358,36 +358,36 @@ namespace QCD {
    //////////////////////////////////////////////
    template<class vobj> 
      void pokeColour(Lattice<vobj> &lhs,
-		      const Lattice<decltype(peekIndex<ColourIndex>(lhs._odata[0],0))> & rhs,
-		      int i)
+              const Lattice<decltype(peekIndex<ColourIndex>(lhs._odata[0],0))> & rhs,
+              int i)
    {
      PokeIndex<ColourIndex>(lhs,rhs,i);
    }
    template<class vobj> 
      void pokeColour(Lattice<vobj> &lhs,
-		      const Lattice<decltype(peekIndex<ColourIndex>(lhs._odata[0],0,0))> & rhs,
-		      int i,int j)
+              const Lattice<decltype(peekIndex<ColourIndex>(lhs._odata[0],0,0))> & rhs,
+              int i,int j)
    {
      PokeIndex<ColourIndex>(lhs,rhs,i,j);
    }
    template<class vobj> 
      void pokeSpin(Lattice<vobj> &lhs,
-		      const Lattice<decltype(peekIndex<SpinIndex>(lhs._odata[0],0))> & rhs,
-		      int i)
+              const Lattice<decltype(peekIndex<SpinIndex>(lhs._odata[0],0))> & rhs,
+              int i)
    {
      PokeIndex<SpinIndex>(lhs,rhs,i);
    }
    template<class vobj> 
      void pokeSpin(Lattice<vobj> &lhs,
-		      const Lattice<decltype(peekIndex<SpinIndex>(lhs._odata[0],0,0))> & rhs,
-		      int i,int j)
+              const Lattice<decltype(peekIndex<SpinIndex>(lhs._odata[0],0,0))> & rhs,
+              int i,int j)
    {
      PokeIndex<SpinIndex>(lhs,rhs,i,j);
    }
    template<class vobj> 
      void pokeLorentz(Lattice<vobj> &lhs,
-		      const Lattice<decltype(peekIndex<LorentzIndex>(lhs._odata[0],0))> & rhs,
-		      int i)
+              const Lattice<decltype(peekIndex<LorentzIndex>(lhs._odata[0],0))> & rhs,
+              int i)
    {
      PokeIndex<LorentzIndex>(lhs,rhs,i);
    }
@@ -4,10 +4,11 @@ Grid physics library, www.github.com/paboyle/Grid

 Source file: ./lib/qcd/action/ActionBase.h

-Copyright (C) 2015
+Copyright (C) 2015-2016

 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
 Author: neo <cossu@post.kek.jp>
+Author: Guido Cossu <guido.cossu@ed.ac.uk>

 This program is free software; you can redistribute it and/or modify
 it under the terms of the GNU General Public License as published by
@@ -27,128 +28,29 @@ See the full license in the file "LICENSE" in the top level distribution
 directory
 *************************************************************************************/
 /*  END LEGAL */
-#ifndef QCD_ACTION_BASE
-#define QCD_ACTION_BASE
+
+#ifndef ACTION_BASE_H
+#define ACTION_BASE_H
+
 namespace Grid {
 namespace QCD {

-template <class GaugeField>
-class Action {
+template <class GaugeField >
+class Action 
+{
+
 public:
  bool is_smeared = false;
-  // Boundary conditions? // Heatbath?
-  virtual void refresh(const GaugeField& U,
-                       GridParallelRNG& pRNG) = 0;  // refresh pseudofermions
-  virtual RealD S(const GaugeField& U) = 0;         // evaluate the action
-  virtual void deriv(const GaugeField& U,
-                     GaugeField& dSdU) = 0;  // evaluate the action derivative
-  virtual ~Action(){};
+  // Heatbath?
+  virtual void refresh(const GaugeField& U, GridParallelRNG& pRNG) = 0; // refresh pseudofermions
+  virtual RealD S(const GaugeField& U) = 0;                             // evaluate the action
+  virtual void deriv(const GaugeField& U, GaugeField& dSdU) = 0;        // evaluate the action derivative
+  virtual std::string action_name()    = 0;                             // return the action name
+  virtual std::string LogParameters()  = 0;                             // prints action parameters
+  virtual ~Action(){}
 };

-// Indexing of tuple types
-template <class T, class Tuple>
-struct Index;
-
-template <class T, class... Types>
-struct Index<T, std::tuple<T, Types...>> {
-  static const std::size_t value = 0;
-};
-
-template <class T, class U, class... Types>
-struct Index<T, std::tuple<U, Types...>> {
-  static const std::size_t value = 1 + Index<T, std::tuple<Types...>>::value;
-};
-
-/*
-template <class GaugeField>
-struct ActionLevel {
- public:
-  typedef Action<GaugeField>*
-      ActPtr;  // now force the same colours as the rest of the code
-
-  //Add supported representations here
-
-
-  unsigned int multiplier;
-
-  std::vector<ActPtr> actions;
-
-  ActionLevel(unsigned int mul = 1) : actions(0), multiplier(mul) {
-    assert(mul >= 1);
-  };
-
-  void push_back(ActPtr ptr) { actions.push_back(ptr); }
-};
-*/
-
-template <class GaugeField, class Repr = NoHirep >
-struct ActionLevel {
- public:
-  unsigned int multiplier; 
-
-  // Fundamental repr actions separated because of the smearing
-  typedef Action<GaugeField>* ActPtr;
-
-  // construct a tuple of vectors of the actions for the corresponding higher
-  // representation fields
-  typedef typename AccessTypes<Action, Repr>::VectorCollection action_collection;
-  action_collection actions_hirep;
-  typedef typename  AccessTypes<Action, Repr>::FieldTypeCollection action_hirep_types;
-
-  std::vector<ActPtr>& actions;
-
-  // Temporary conversion between ActionLevel and ActionLevelHirep
-  //ActionLevelHirep(ActionLevel<GaugeField>& AL ):actions(AL.actions), multiplier(AL.multiplier){}
-
-  ActionLevel(unsigned int mul = 1) : actions(std::get<0>(actions_hirep)), multiplier(mul) {
-    // initialize the hirep vectors to zero.
-    //apply(this->resize, actions_hirep, 0); //need a working resize
-    assert(mul >= 1);
-  };
-
-  //void push_back(ActPtr ptr) { actions.push_back(ptr); }
-
-
-
-  template < class Field >
-  void push_back(Action<Field>* ptr) {
-    // insert only in the correct vector
-    std::get< Index < Field, action_hirep_types>::value >(actions_hirep).push_back(ptr);
-  };
-
- 
-
-  template < class ActPtr>
-  static void resize(ActPtr ap, unsigned int n){
-    ap->resize(n);
-
-  }
-
-  //template <std::size_t I>
-  //auto getRepresentation(Repr& R)->decltype(std::get<I>(R).U)  {return std::get<I>(R).U;}
-
-  // Loop on tuple for a callable function
-  template <std::size_t I = 1, typename Callable, typename ...Args>
-  inline typename std::enable_if<I == std::tuple_size<action_collection>::value, void>::type apply(
-      Callable, Repr& R,Args&...) const {}
-
-  template <std::size_t I = 1, typename Callable, typename ...Args>
-  inline typename std::enable_if<I < std::tuple_size<action_collection>::value, void>::type apply(
-      Callable fn, Repr& R, Args&... arguments) const {
-    fn(std::get<I>(actions_hirep), std::get<I>(R.rep), arguments...);
-    apply<I + 1>(fn, R, arguments...);
-  }  
-
-};
-
-
-//template <class GaugeField>
-//using ActionSet = std::vector<ActionLevel<GaugeField> >;
-
-template <class GaugeField, class R>
-using ActionSet = std::vector<ActionLevel<GaugeField, R> >;
-
 }
 }

-#endif
+#endif // ACTION_BASE_H
@@ -31,15 +31,31 @@ directory
 #define QCD_ACTION_CORE

 #include <Grid/qcd/action/ActionBase.h>
+#include <Grid/qcd/action/ActionSet.h>
 #include <Grid/qcd/action/ActionParams.h>

 ////////////////////////////////////////////
 // Gauge Actions
 ////////////////////////////////////////////
 #include <Grid/qcd/action/gauge/Gauge.h>
+
 ////////////////////////////////////////////
 // Fermion prereqs
 ////////////////////////////////////////////
 #include <Grid/qcd/action/fermion/FermionCore.h>

+////////////////////////////////////////////
+// Scalar Actions
+////////////////////////////////////////////
+#include <Grid/qcd/action/scalar/Scalar.h>
+
+////////////////////////////////////////////
+// Utility functions
+////////////////////////////////////////////
+#include <Grid/qcd/utils/Metric.h>
+#include <Grid/qcd/utils/CovariantLaplacian.h>
+
+
+
+
 #endif
@@ -1,67 +1,92 @@
-    /*************************************************************************************
+/*************************************************************************************

-    Grid physics library, www.github.com/paboyle/Grid 
+Grid physics library, www.github.com/paboyle/Grid

-    Source file: ./lib/qcd/action/ActionParams.h
+Source file: ./lib/qcd/action/ActionParams.h

-    Copyright (C) 2015
+Copyright (C) 2015

 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
 Author: paboyle <paboyle@ph.ed.ac.uk>
+Author: Guido Cossu <guido.cossu@ed.ac.uk>

-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.

-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.

-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */

-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
 #ifndef GRID_QCD_ACTION_PARAMS_H
 #define GRID_QCD_ACTION_PARAMS_H

 namespace Grid {
 namespace QCD {

-    // These can move into a params header and be given MacroMagic serialisation
-    struct GparityWilsonImplParams {
-      bool overlapCommsCompute;
-      std::vector<int> twists; 
-      GparityWilsonImplParams () : twists(Nd,0), overlapCommsCompute(false) {};
-
+  // These can move into a params header and be given MacroMagic serialisation
+  struct GparityWilsonImplParams {
+    bool overlapCommsCompute;
+    std::vector<int> twists;
+    GparityWilsonImplParams() : twists(Nd, 0), overlapCommsCompute(false){};
+  };
+  
+  struct WilsonImplParams {
+    bool overlapCommsCompute;
+    std::vector<Complex> boundary_phases;
+    WilsonImplParams() : overlapCommsCompute(false) {
+      boundary_phases.resize(Nd, 1.0);
    };
+    WilsonImplParams(const std::vector<Complex> phi)
+      : boundary_phases(phi), overlapCommsCompute(false) {}
+  };

-    struct WilsonImplParams {
-      bool overlapCommsCompute;
-      WilsonImplParams() : overlapCommsCompute(false) {};
-    };
+  struct StaggeredImplParams {
+    StaggeredImplParams()  {};
+  };
+  
+  struct OneFlavourRationalParams : Serializable {
+    GRID_SERIALIZABLE_CLASS_MEMBERS(OneFlavourRationalParams, 
+				    RealD, lo, 
+				    RealD, hi, 
+				    int,   MaxIter, 
+				    RealD, tolerance, 
+				    int,   degree, 
+				    int,   precision);
+    
+    // MaxIter and tolerance, vectors??
+    
+    // constructor 
+    OneFlavourRationalParams(	RealD _lo      = 0.0, 
+				RealD _hi      = 1.0, 
+				int _maxit     = 1000,
+				RealD tol      = 1.0e-8, 
+                           	int _degree    = 10,
+				int _precision = 64)
+      : lo(_lo),
+	hi(_hi),
+	MaxIter(_maxit),
+	tolerance(tol),
+	degree(_degree),
+	precision(_precision){};
+  };
+  
+  
+}
+}

-    struct StaggeredImplParams {
-      StaggeredImplParams()  {};
-    };

-    struct OneFlavourRationalParams { 
-      RealD  lo;
-      RealD  hi;
-      int MaxIter;   // Vector?
-      RealD tolerance; // Vector? 
-      int    degree=10;
-      int precision=64;

-      OneFlavourRationalParams (RealD _lo,RealD _hi,int _maxit,RealD tol=1.0e-8,int _degree = 10,int _precision=64) :
-        lo(_lo), hi(_hi), MaxIter(_maxit), tolerance(tol), degree(_degree), precision(_precision)
-      {};
-    };
-
-}}

 #endif
@@ -0,0 +1,116 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/action/ActionSet.h
+
+Copyright (C) 2015
+
+Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+Author: neo <cossu@post.kek.jp>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef ACTION_SET_H
+#define ACTION_SET_H
+
+namespace Grid {
+
+// Should drop this namespace here
+namespace QCD {
+
+//////////////////////////////////
+// Indexing of tuple types
+//////////////////////////////////
+
+template <class T, class Tuple>
+struct Index;
+
+template <class T, class... Types>
+struct Index<T, std::tuple<T, Types...>> {
+  static const std::size_t value = 0;
+};
+
+template <class T, class U, class... Types>
+struct Index<T, std::tuple<U, Types...>> {
+  static const std::size_t value = 1 + Index<T, std::tuple<Types...>>::value;
+};
+
+
+////////////////////////////////////////////
+// Action Level
+// Action collection 
+// in a integration level
+// (for multilevel integration schemes)
+////////////////////////////////////////////
+
+template <class Field, class Repr = NoHirep >
+struct ActionLevel {
+ public:
+  unsigned int multiplier;
+
+  // Fundamental repr actions separated because of the smearing
+  typedef Action<Field>* ActPtr;
+
+  // construct a tuple of vectors of the actions for the corresponding higher
+  // representation fields
+  typedef typename AccessTypes<Action, Repr>::VectorCollection action_collection;
+  typedef typename  AccessTypes<Action, Repr>::FieldTypeCollection action_hirep_types;
+
+  action_collection actions_hirep;
+  std::vector<ActPtr>& actions;
+
+  explicit ActionLevel(unsigned int mul = 1) : 
+  actions(std::get<0>(actions_hirep)), multiplier(mul) {
+    // initialize the hirep vectors to zero.
+    // apply(this->resize, actions_hirep, 0); //need a working resize
+    assert(mul >= 1);
+  }
+
+  template < class GenField >
+  void push_back(Action<GenField>* ptr) {
+    // insert only in the correct vector
+    std::get< Index < GenField, action_hirep_types>::value >(actions_hirep).push_back(ptr);
+  };
+
+  template <class ActPtr>
+  static void resize(ActPtr ap, unsigned int n) {
+    ap->resize(n);
+  }
+
+  // Loop on tuple for a callable function
+  template <std::size_t I = 1, typename Callable, typename ...Args>
+  inline typename std::enable_if<I == std::tuple_size<action_collection>::value, void>::type apply(Callable, Repr& R,Args&...) const {}
+
+  template <std::size_t I = 1, typename Callable, typename ...Args>
+  inline typename std::enable_if<I < std::tuple_size<action_collection>::value, void>::type apply(Callable fn, Repr& R, Args&... arguments) const {
+    fn(std::get<I>(actions_hirep), std::get<I>(R.rep), arguments...);
+    apply<I + 1>(fn, R, arguments...);
+  }  
+
+};
+
+// Define the ActionSet
+template <class GaugeField, class R>
+using ActionSet = std::vector<ActionLevel<GaugeField, R> >;
+
+} // QCD
+} // Grid
+
+#endif  // ACTION_SET_H
@@ -29,7 +29,7 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
    *************************************************************************************/
    /*  END LEGAL */

-#include <Grid/Eigen/Dense>
+#include <Grid/Grid_Eigen_Dense.h>
 #include <Grid/qcd/action/fermion/FermionCore.h>
 #include <Grid/qcd/action/fermion/CayleyFermion5D.h>

@@ -320,7 +320,7 @@ void CayleyFermion5D<Impl>::MDeriv  (GaugeField &mat,const FermionField &U,const
    this->DhopDeriv(mat,U,Din,dag);
  } else {
    //      U d/du [D_w D5]^dag V = U D5^dag d/du DW^dag Y // implicit adj on U in call
-    MeooeDag5D(U,Din);
+    Meooe5D(U,Din);
    this->DhopDeriv(mat,Din,V,dag);
  }
 };
@@ -335,8 +335,8 @@ void CayleyFermion5D<Impl>::MoeDeriv(GaugeField &mat,const FermionField &U,const
    this->DhopDerivOE(mat,U,Din,dag);
  } else {
    //      U d/du [D_w D5]^dag V = U D5^dag d/du DW^dag Y // implicit adj on U in call
-      MeooeDag5D(U,Din);
-      this->DhopDerivOE(mat,Din,V,dag);
+    Meooe5D(U,Din);
+    this->DhopDerivOE(mat,Din,V,dag);
  }
 };
 template<class Impl>
@@ -350,7 +350,7 @@ void CayleyFermion5D<Impl>::MeoDeriv(GaugeField &mat,const FermionField &U,const
    this->DhopDerivEO(mat,U,Din,dag);
  } else {
    //      U d/du [D_w D5]^dag V = U D5^dag d/du DW^dag Y // implicit adj on U in call
-    MeooeDag5D(U,Din);
+    Meooe5D(U,Din);
    this->DhopDerivEO(mat,Din,V,dag);
  }
 };
@@ -380,6 +380,8 @@ void CayleyFermion5D<Impl>::SetCoefficientsInternal(RealD zolo_hi,std::vector<Co
  ///////////////////////////////////////////////////////////
  // The Cayley coeffs (unprec)
  ///////////////////////////////////////////////////////////
+  assert(gamma.size()==Ls);
+
  omega.resize(Ls);
  bs.resize(Ls);
  cs.resize(Ls);
@@ -412,10 +414,11 @@ void CayleyFermion5D<Impl>::SetCoefficientsInternal(RealD zolo_hi,std::vector<Co
  for(int i=0; i < Ls; i++){
    as[i] = 1.0;
    omega[i] = gamma[i]*zolo_hi; //NB reciprocal relative to Chroma NEF code
+    //    assert(fabs(omega[i])>0.0);
    bs[i] = 0.5*(bpc/omega[i] + bmc);
    cs[i] = 0.5*(bpc/omega[i] - bmc);
  }
-  
+
  ////////////////////////////////////////////////////////
  // Constants for the preconditioned matrix Cayley form
  ////////////////////////////////////////////////////////
@@ -425,12 +428,12 @@ void CayleyFermion5D<Impl>::SetCoefficientsInternal(RealD zolo_hi,std::vector<Co
  ceo.resize(Ls);
  
  for(int i=0;i<Ls;i++){
-    bee[i]=as[i]*(bs[i]*(4.0-this->M5) +1.0);
+    bee[i]=as[i]*(bs[i]*(4.0-this->M5) +1.0);     
+    //    assert(fabs(bee[i])>0.0);
    cee[i]=as[i]*(1.0-cs[i]*(4.0-this->M5));
    beo[i]=as[i]*bs[i];
    ceo[i]=-as[i]*cs[i];
  }
-  
  aee.resize(Ls);
  aeo.resize(Ls);
  for(int i=0;i<Ls;i++){
@@ -474,14 +477,16 @@ void CayleyFermion5D<Impl>::SetCoefficientsInternal(RealD zolo_hi,std::vector<Co
 	
  { 
    Coeff_t delta_d=mass*cee[Ls-1];
-    for(int j=0;j<Ls-1;j++) delta_d *= cee[j]/bee[j];
+    for(int j=0;j<Ls-1;j++) {
+      //      assert(fabs(bee[j])>0.0);
+      delta_d *= cee[j]/bee[j];
+    }
    dee[Ls-1] += delta_d;
  }  

  int inv=1;
  this->MooeeInternalCompute(0,inv,MatpInv,MatmInv);
  this->MooeeInternalCompute(1,inv,MatpInvDag,MatmInvDag);
-
 }


@@ -495,7 +500,9 @@ void CayleyFermion5D<Impl>::MooeeInternalCompute(int dag, int inv,
  GridBase *grid = this->FermionRedBlackGrid();
  int LLs = grid->_rdimensions[0];

-  if ( LLs == Ls ) return; // Not vectorised in 5th direction
+  if ( LLs == Ls ) {
+    return; // Not vectorised in 5th direction
+  }

  Eigen::MatrixXcd Pplus  = Eigen::MatrixXcd::Zero(Ls,Ls);
  Eigen::MatrixXcd Pminus = Eigen::MatrixXcd::Zero(Ls,Ls);
@@ -237,6 +237,13 @@ void CayleyFermion5D<Impl>::MooeeInvDag (const FermionField &psi, FermionField &
  INSTANTIATE_DPERP(GparityWilsonImplD);
  INSTANTIATE_DPERP(ZWilsonImplF);
  INSTANTIATE_DPERP(ZWilsonImplD);
+
+  INSTANTIATE_DPERP(WilsonImplFH);
+  INSTANTIATE_DPERP(WilsonImplDF);
+  INSTANTIATE_DPERP(GparityWilsonImplFH);
+  INSTANTIATE_DPERP(GparityWilsonImplDF);
+  INSTANTIATE_DPERP(ZWilsonImplFH);
+  INSTANTIATE_DPERP(ZWilsonImplDF);
 #endif

 }}
@@ -29,7 +29,7 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
    *************************************************************************************/
    /*  END LEGAL */

-#include <Grid/Eigen/Dense>
+#include <Grid/Grid_Eigen_Dense.h>
 #include <Grid/qcd/action/fermion/FermionCore.h>
 #include <Grid/qcd/action/fermion/CayleyFermion5D.h>

@@ -137,6 +137,20 @@ template void CayleyFermion5D<WilsonImplF>::MooeeInternal(const FermionField &ps
 template void CayleyFermion5D<WilsonImplD>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
 template void CayleyFermion5D<ZWilsonImplF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
 template void CayleyFermion5D<ZWilsonImplD>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+
+INSTANTIATE_DPERP(GparityWilsonImplFH);
+INSTANTIATE_DPERP(GparityWilsonImplDF);
+INSTANTIATE_DPERP(WilsonImplFH);
+INSTANTIATE_DPERP(WilsonImplDF);
+INSTANTIATE_DPERP(ZWilsonImplFH);
+INSTANTIATE_DPERP(ZWilsonImplDF);
+
+template void CayleyFermion5D<GparityWilsonImplFH>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<GparityWilsonImplDF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<WilsonImplFH>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<WilsonImplDF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<ZWilsonImplFH>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<ZWilsonImplDF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
 #endif

 }}
@@ -37,7 +37,6 @@ namespace Grid {
 namespace QCD {

  // FIXME -- make a version of these routines with site loop outermost for cache reuse.
-
  // Pminus fowards
  // Pplus  backwards
 template<class Impl>  
@@ -152,6 +151,13 @@ void CayleyFermion5D<Impl>::MooeeInvDag (const FermionField &psi, FermionField &
  INSTANTIATE_DPERP(GparityWilsonImplD);
  INSTANTIATE_DPERP(ZWilsonImplF);
  INSTANTIATE_DPERP(ZWilsonImplD);
+
+  INSTANTIATE_DPERP(WilsonImplFH);
+  INSTANTIATE_DPERP(WilsonImplDF);
+  INSTANTIATE_DPERP(GparityWilsonImplFH);
+  INSTANTIATE_DPERP(GparityWilsonImplDF);
+  INSTANTIATE_DPERP(ZWilsonImplFH);
+  INSTANTIATE_DPERP(ZWilsonImplDF);
 #endif

 }
@@ -808,10 +808,21 @@ INSTANTIATE_DPERP(DomainWallVec5dImplF);
 INSTANTIATE_DPERP(ZDomainWallVec5dImplD);
 INSTANTIATE_DPERP(ZDomainWallVec5dImplF);

+INSTANTIATE_DPERP(DomainWallVec5dImplDF);
+INSTANTIATE_DPERP(DomainWallVec5dImplFH);
+INSTANTIATE_DPERP(ZDomainWallVec5dImplDF);
+INSTANTIATE_DPERP(ZDomainWallVec5dImplFH);
+
 template void CayleyFermion5D<DomainWallVec5dImplF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
 template void CayleyFermion5D<DomainWallVec5dImplD>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
 template void CayleyFermion5D<ZDomainWallVec5dImplF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
 template void CayleyFermion5D<ZDomainWallVec5dImplD>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);

+template void CayleyFermion5D<DomainWallVec5dImplFH>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<DomainWallVec5dImplDF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<ZDomainWallVec5dImplFH>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+template void CayleyFermion5D<ZDomainWallVec5dImplDF>::MooeeInternal(const FermionField &psi, FermionField &chi,int dag, int inv);
+
+

 }}
@@ -68,7 +68,7 @@ namespace Grid {
 	Approx::zolotarev_data *zdata = Approx::higham(eps,this->Ls);// eps is ignored for higham
 	assert(zdata->n==this->Ls);
 	
-	//	std::cout<<GridLogMessage << "DomainWallFermion with Ls="<<this->Ls<<std::endl;
+	std::cout<<GridLogMessage << "DomainWallFermion with Ls="<<this->Ls<<std::endl;
 	// Call base setter
 	this->SetCoefficientsTanh(zdata,1.0,0.0);

@@ -1,4 +1,4 @@
-    /*************************************************************************************
+/*************************************************************************************

    Grid physics library, www.github.com/paboyle/Grid 

@@ -91,6 +91,10 @@ typedef WilsonFermion<WilsonImplR> WilsonFermionR;
 typedef WilsonFermion<WilsonImplF> WilsonFermionF;
 typedef WilsonFermion<WilsonImplD> WilsonFermionD;

+typedef WilsonFermion<WilsonImplRL> WilsonFermionRL;
+typedef WilsonFermion<WilsonImplFH> WilsonFermionFH;
+typedef WilsonFermion<WilsonImplDF> WilsonFermionDF;
+
 typedef WilsonFermion<WilsonAdjImplR> WilsonAdjFermionR;
 typedef WilsonFermion<WilsonAdjImplF> WilsonAdjFermionF;
 typedef WilsonFermion<WilsonAdjImplD> WilsonAdjFermionD;
@@ -113,27 +117,50 @@ typedef DomainWallFermion<WilsonImplR> DomainWallFermionR;
 typedef DomainWallFermion<WilsonImplF> DomainWallFermionF;
 typedef DomainWallFermion<WilsonImplD> DomainWallFermionD;

+typedef DomainWallFermion<WilsonImplRL> DomainWallFermionRL;
+typedef DomainWallFermion<WilsonImplFH> DomainWallFermionFH;
+typedef DomainWallFermion<WilsonImplDF> DomainWallFermionDF;
+
 typedef MobiusFermion<WilsonImplR> MobiusFermionR;
 typedef MobiusFermion<WilsonImplF> MobiusFermionF;
 typedef MobiusFermion<WilsonImplD> MobiusFermionD;

+typedef MobiusFermion<WilsonImplRL> MobiusFermionRL;
+typedef MobiusFermion<WilsonImplFH> MobiusFermionFH;
+typedef MobiusFermion<WilsonImplDF> MobiusFermionDF;
+
 typedef ZMobiusFermion<ZWilsonImplR> ZMobiusFermionR;
 typedef ZMobiusFermion<ZWilsonImplF> ZMobiusFermionF;
 typedef ZMobiusFermion<ZWilsonImplD> ZMobiusFermionD;

+typedef ZMobiusFermion<ZWilsonImplRL> ZMobiusFermionRL;
+typedef ZMobiusFermion<ZWilsonImplFH> ZMobiusFermionFH;
+typedef ZMobiusFermion<ZWilsonImplDF> ZMobiusFermionDF;
+
 // Ls vectorised 
 typedef DomainWallFermion<DomainWallVec5dImplR> DomainWallFermionVec5dR;
 typedef DomainWallFermion<DomainWallVec5dImplF> DomainWallFermionVec5dF;
 typedef DomainWallFermion<DomainWallVec5dImplD> DomainWallFermionVec5dD;

+typedef DomainWallFermion<DomainWallVec5dImplRL> DomainWallFermionVec5dRL;
+typedef DomainWallFermion<DomainWallVec5dImplFH> DomainWallFermionVec5dFH;
+typedef DomainWallFermion<DomainWallVec5dImplDF> DomainWallFermionVec5dDF;
+
 typedef MobiusFermion<DomainWallVec5dImplR> MobiusFermionVec5dR;
 typedef MobiusFermion<DomainWallVec5dImplF> MobiusFermionVec5dF;
 typedef MobiusFermion<DomainWallVec5dImplD> MobiusFermionVec5dD;

+typedef MobiusFermion<DomainWallVec5dImplRL> MobiusFermionVec5dRL;
+typedef MobiusFermion<DomainWallVec5dImplFH> MobiusFermionVec5dFH;
+typedef MobiusFermion<DomainWallVec5dImplDF> MobiusFermionVec5dDF;
+
 typedef ZMobiusFermion<ZDomainWallVec5dImplR> ZMobiusFermionVec5dR;
 typedef ZMobiusFermion<ZDomainWallVec5dImplF> ZMobiusFermionVec5dF;
 typedef ZMobiusFermion<ZDomainWallVec5dImplD> ZMobiusFermionVec5dD;

+typedef ZMobiusFermion<ZDomainWallVec5dImplRL> ZMobiusFermionVec5dRL;
+typedef ZMobiusFermion<ZDomainWallVec5dImplFH> ZMobiusFermionVec5dFH;
+typedef ZMobiusFermion<ZDomainWallVec5dImplDF> ZMobiusFermionVec5dDF;

 typedef ScaledShamirFermion<WilsonImplR> ScaledShamirFermionR;
 typedef ScaledShamirFermion<WilsonImplF> ScaledShamirFermionF;
@@ -174,17 +201,35 @@ typedef OverlapWilsonPartialFractionZolotarevFermion<WilsonImplD> OverlapWilsonP
 typedef WilsonFermion<GparityWilsonImplR>     GparityWilsonFermionR;
 typedef WilsonFermion<GparityWilsonImplF>     GparityWilsonFermionF;
 typedef WilsonFermion<GparityWilsonImplD>     GparityWilsonFermionD;
+
+typedef WilsonFermion<GparityWilsonImplRL>     GparityWilsonFermionRL;
+typedef WilsonFermion<GparityWilsonImplFH>     GparityWilsonFermionFH;
+typedef WilsonFermion<GparityWilsonImplDF>     GparityWilsonFermionDF;
+
 typedef DomainWallFermion<GparityWilsonImplR> GparityDomainWallFermionR;
 typedef DomainWallFermion<GparityWilsonImplF> GparityDomainWallFermionF;
 typedef DomainWallFermion<GparityWilsonImplD> GparityDomainWallFermionD;

+typedef DomainWallFermion<GparityWilsonImplRL> GparityDomainWallFermionRL;
+typedef DomainWallFermion<GparityWilsonImplFH> GparityDomainWallFermionFH;
+typedef DomainWallFermion<GparityWilsonImplDF> GparityDomainWallFermionDF;
+
 typedef WilsonTMFermion<GparityWilsonImplR> GparityWilsonTMFermionR;
 typedef WilsonTMFermion<GparityWilsonImplF> GparityWilsonTMFermionF;
 typedef WilsonTMFermion<GparityWilsonImplD> GparityWilsonTMFermionD;
+
+typedef WilsonTMFermion<GparityWilsonImplRL> GparityWilsonTMFermionRL;
+typedef WilsonTMFermion<GparityWilsonImplFH> GparityWilsonTMFermionFH;
+typedef WilsonTMFermion<GparityWilsonImplDF> GparityWilsonTMFermionDF;
+
 typedef MobiusFermion<GparityWilsonImplR> GparityMobiusFermionR;
 typedef MobiusFermion<GparityWilsonImplF> GparityMobiusFermionF;
 typedef MobiusFermion<GparityWilsonImplD> GparityMobiusFermionD;

+typedef MobiusFermion<GparityWilsonImplRL> GparityMobiusFermionRL;
+typedef MobiusFermion<GparityWilsonImplFH> GparityMobiusFermionFH;
+typedef MobiusFermion<GparityWilsonImplDF> GparityMobiusFermionDF;
+
 typedef ImprovedStaggeredFermion<StaggeredImplR> ImprovedStaggeredFermionR;
 typedef ImprovedStaggeredFermion<StaggeredImplF> ImprovedStaggeredFermionF;
 typedef ImprovedStaggeredFermion<StaggeredImplD> ImprovedStaggeredFermionD;
@@ -200,4 +245,11 @@ typedef ImprovedStaggeredFermion5D<StaggeredVec5dImplD> ImprovedStaggeredFermion

  }}

+////////////////////
+// Scalar QED actions
+// TODO: this needs to move to another header after rename to Fermion.h
+////////////////////
+#include <Grid/qcd/action/scalar/Scalar.h>
+#include <Grid/qcd/action/gauge/Photon.h>
+
 #endif
@@ -55,7 +55,14 @@ Author: Peter Boyle <pabobyle@ph.ed.ac.uk>
  template class A<ZWilsonImplF>;		\
  template class A<ZWilsonImplD>;		\
  template class A<GparityWilsonImplF>;		\
-  template class A<GparityWilsonImplD>;		
+  template class A<GparityWilsonImplD>;		\
+  template class A<WilsonImplFH>;		\
+  template class A<WilsonImplDF>;		\
+  template class A<ZWilsonImplFH>;		\
+  template class A<ZWilsonImplDF>;		\
+  template class A<GparityWilsonImplFH>;		\
+  template class A<GparityWilsonImplDF>;		
+

 #define AdjointFermOpTemplateInstantiate(A) \
  template class A<WilsonAdjImplF>; \
@@ -69,7 +76,11 @@ Author: Peter Boyle <pabobyle@ph.ed.ac.uk>
  template class A<DomainWallVec5dImplF>;	\
  template class A<DomainWallVec5dImplD>;	\
  template class A<ZDomainWallVec5dImplF>;	\
-  template class A<ZDomainWallVec5dImplD>;	
+  template class A<ZDomainWallVec5dImplD>;	\
+  template class A<DomainWallVec5dImplFH>;	\
+  template class A<DomainWallVec5dImplDF>;	\
+  template class A<ZDomainWallVec5dImplFH>;	\
+  template class A<ZDomainWallVec5dImplDF>;	

 #define FermOpTemplateInstantiate(A) \
 FermOp4dVecTemplateInstantiate(A) \
@@ -35,7 +35,6 @@ directory
 namespace Grid {
 namespace QCD {

-
  //////////////////////////////////////////////
  // Template parameter class constructs to package
  // externally control Fermion implementations
@@ -44,7 +43,7 @@ namespace QCD {
  // Ultimately need Impl to always define types where XXX is opaque
  //
  //    typedef typename XXX               Simd;
-  //    typedef typename XXX     GaugeLinkField;	
+  //    typedef typename XXX     GaugeLinkField;        
  //    typedef typename XXX         GaugeField;
  //    typedef typename XXX      GaugeActField;
  //    typedef typename XXX       FermionField;
@@ -89,7 +88,53 @@ namespace QCD {
  //    
  //  }
  //////////////////////////////////////////////
-  
+
+  template <class T> struct SamePrecisionMapper {
+    typedef T HigherPrecVector ;
+    typedef T LowerPrecVector ;
+  };
+  template <class T> struct LowerPrecisionMapper {  };
+  template <> struct LowerPrecisionMapper<vRealF> {
+    typedef vRealF HigherPrecVector ;
+    typedef vRealH LowerPrecVector ;
+  };
+  template <> struct LowerPrecisionMapper<vRealD> {
+    typedef vRealD HigherPrecVector ;
+    typedef vRealF LowerPrecVector ;
+  };
+  template <> struct LowerPrecisionMapper<vComplexF> {
+    typedef vComplexF HigherPrecVector ;
+    typedef vComplexH LowerPrecVector ;
+  };
+  template <> struct LowerPrecisionMapper<vComplexD> {
+    typedef vComplexD HigherPrecVector ;
+    typedef vComplexF LowerPrecVector ;
+  };
+
+  struct CoeffReal {
+  public:
+    typedef RealD _Coeff_t;
+    static const int Nhcs = 2;
+    template<class Simd> using PrecisionMapper = SamePrecisionMapper<Simd>;
+  };
+  struct CoeffRealHalfComms {
+  public:
+    typedef RealD _Coeff_t;
+    static const int Nhcs = 1;
+    template<class Simd> using PrecisionMapper = LowerPrecisionMapper<Simd>;
+  };
+  struct CoeffComplex {
+  public:
+    typedef ComplexD _Coeff_t;
+    static const int Nhcs = 2;
+    template<class Simd> using PrecisionMapper = SamePrecisionMapper<Simd>;
+  };
+  struct CoeffComplexHalfComms {
+  public:
+    typedef ComplexD _Coeff_t;
+    static const int Nhcs = 1;
+    template<class Simd> using PrecisionMapper = LowerPrecisionMapper<Simd>;
+  };

  ////////////////////////////////////////////////////////////////////////
  // Implementation dependent fermion types
@@ -108,58 +153,63 @@ namespace QCD {
  typedef typename Impl::Coeff_t                     Coeff_t;           \
  
 #define INHERIT_IMPL_TYPES(Base) \
-  INHERIT_GIMPL_TYPES(Base)	 \
+  INHERIT_GIMPL_TYPES(Base)      \
  INHERIT_FIMPL_TYPES(Base)
  
  /////////////////////////////////////////////////////////////////////////////
  // Single flavour four spinors with colour index
  /////////////////////////////////////////////////////////////////////////////
-  template <class S, class Representation = FundamentalRepresentation,class _Coeff_t = RealD >
+  template <class S, class Representation = FundamentalRepresentation,class Options = CoeffReal >
  class WilsonImpl : public PeriodicGaugeImpl<GaugeImplTypes<S, Representation::Dimension > > {
-
    public:

    static const int Dimension = Representation::Dimension;
+    static const bool LsVectorised=false;
+    static const int Nhcs = Options::Nhcs;
+
    typedef PeriodicGaugeImpl<GaugeImplTypes<S, Dimension > > Gimpl;
+    INHERIT_GIMPL_TYPES(Gimpl);
      
    //Necessary?
    constexpr bool is_fundamental() const{return Dimension == Nc ? 1 : 0;}
    
-    const bool LsVectorised=false;
-    typedef _Coeff_t Coeff_t;
-
-    INHERIT_GIMPL_TYPES(Gimpl);
+    typedef typename Options::_Coeff_t Coeff_t;
+    typedef typename Options::template PrecisionMapper<Simd>::LowerPrecVector SimdL;
      
    template <typename vtype> using iImplSpinor            = iScalar<iVector<iVector<vtype, Dimension>, Ns> >;
    template <typename vtype> using iImplPropagator        = iScalar<iMatrix<iMatrix<vtype, Dimension>, Ns> >;
    template <typename vtype> using iImplHalfSpinor        = iScalar<iVector<iVector<vtype, Dimension>, Nhs> >;
+    template <typename vtype> using iImplHalfCommSpinor    = iScalar<iVector<iVector<vtype, Dimension>, Nhcs> >;
    template <typename vtype> using iImplDoubledGaugeField = iVector<iScalar<iMatrix<vtype, Dimension> >, Nds>;
    
    typedef iImplSpinor<Simd>            SiteSpinor;
    typedef iImplPropagator<Simd>        SitePropagator;
    typedef iImplHalfSpinor<Simd>        SiteHalfSpinor;
+    typedef iImplHalfCommSpinor<SimdL>   SiteHalfCommSpinor;
    typedef iImplDoubledGaugeField<Simd> SiteDoubledGaugeField;
    
    typedef Lattice<SiteSpinor>            FermionField;
    typedef Lattice<SitePropagator>        PropagatorField;
    typedef Lattice<SiteDoubledGaugeField> DoubledGaugeField;
    
-    typedef WilsonCompressor<SiteHalfSpinor, SiteSpinor> Compressor;
+    typedef WilsonCompressor<SiteHalfCommSpinor,SiteHalfSpinor, SiteSpinor> Compressor;
    typedef WilsonImplParams ImplParams;
    typedef WilsonStencil<SiteSpinor, SiteHalfSpinor> StencilImpl;
    
    ImplParams Params;
    
-    WilsonImpl(const ImplParams &p = ImplParams()) : Params(p){};
+    WilsonImpl(const ImplParams &p = ImplParams()) : Params(p){
+      assert(Params.boundary_phases.size() == Nd);
+    };
      
    bool overlapCommsCompute(void) { return Params.overlapCommsCompute; };
      
    inline void multLink(SiteHalfSpinor &phi,
-			 const SiteDoubledGaugeField &U,
-			 const SiteHalfSpinor &chi,
-			 int mu,
-			 StencilEntry *SE,
-			 StencilImpl &St) {
+                         const SiteDoubledGaugeField &U,
+                         const SiteHalfSpinor &chi,
+                         int mu,
+                         StencilEntry *SE,
+                         StencilImpl &St) {
      mult(&phi(), &U(mu), &chi());
    }
      
@@ -169,16 +219,34 @@ namespace QCD {
    }
      
    inline void DoubleStore(GridBase *GaugeGrid,
-			    DoubledGaugeField &Uds,
-			    const GaugeField &Umu) {
+                            DoubledGaugeField &Uds,
+                            const GaugeField &Umu) 
+    {
+      typedef typename Simd::scalar_type scalar_type;
+
      conformable(Uds._grid, GaugeGrid);
      conformable(Umu._grid, GaugeGrid);
+
      GaugeLinkField U(GaugeGrid);
+      GaugeLinkField tmp(GaugeGrid);
+
+      Lattice<iScalar<vInteger> > coor(GaugeGrid);
      for (int mu = 0; mu < Nd; mu++) {
-	U = PeekIndex<LorentzIndex>(Umu, mu);
-	PokeIndex<LorentzIndex>(Uds, U, mu);
-	U = adj(Cshift(U, mu, -1));
-	PokeIndex<LorentzIndex>(Uds, U, mu + 4);
+
+	      auto pha = Params.boundary_phases[mu];
+	      scalar_type phase( real(pha),imag(pha) );
+
+        int Lmu = GaugeGrid->GlobalDimensions()[mu] - 1;
+
+        LatticeCoordinate(coor, mu);
+
+        U = PeekIndex<LorentzIndex>(Umu, mu);
+        tmp = where(coor == Lmu, phase * U, U);
+        PokeIndex<LorentzIndex>(Uds, tmp, mu);
+
+        U = adj(Cshift(U, mu, -1));
+        U = where(coor == 0, conjugate(phase) * U, U); 
+        PokeIndex<LorentzIndex>(Uds, U, mu + 4);
      }
    }

@@ -195,11 +263,11 @@ namespace QCD {
      tmp = zero;
      
      parallel_for(int sss=0;sss<tmp._grid->oSites();sss++){
-	int sU=sss;
-	for(int s=0;s<Ls;s++){
-	  int sF = s+Ls*sU;
-	  tmp[sU] = tmp[sU]+ traceIndex<SpinIndex>(outerProduct(Btilde[sF],Atilde[sF])); // ordering here
-	}
+	    int sU=sss;
+	    for(int s=0;s<Ls;s++){
+	        int sF = s+Ls*sU;
+	        tmp[sU] = tmp[sU]+ traceIndex<SpinIndex>(outerProduct(Btilde[sF],Atilde[sF])); // ordering here
+	    }
      }
      PokeIndex<LorentzIndex>(mat,tmp,mu);
      
@@ -209,31 +277,34 @@ namespace QCD {
  ////////////////////////////////////////////////////////////////////////////////////
  // Single flavour four spinors with colour index, 5d redblack
  ////////////////////////////////////////////////////////////////////////////////////
-
-template<class S,int Nrepresentation=Nc,class _Coeff_t = RealD>
+template<class S,int Nrepresentation=Nc, class Options=CoeffReal>
 class DomainWallVec5dImpl :  public PeriodicGaugeImpl< GaugeImplTypes< S,Nrepresentation> > { 
  public:
-      
-  static const int Dimension = Nrepresentation;
-  const bool LsVectorised=true;
-  typedef _Coeff_t Coeff_t;      
+
  typedef PeriodicGaugeImpl<GaugeImplTypes<S, Nrepresentation> > Gimpl;
-  
  INHERIT_GIMPL_TYPES(Gimpl);
+
+  static const int Dimension = Nrepresentation;
+  static const bool LsVectorised=true;
+  static const int Nhcs = Options::Nhcs;
+      
+  typedef typename Options::_Coeff_t Coeff_t;      
+  typedef typename Options::template PrecisionMapper<Simd>::LowerPrecVector SimdL;
  
  template <typename vtype> using iImplSpinor            = iScalar<iVector<iVector<vtype, Nrepresentation>, Ns> >;
  template <typename vtype> using iImplPropagator        = iScalar<iMatrix<iMatrix<vtype, Nrepresentation>, Ns> >;
  template <typename vtype> using iImplHalfSpinor        = iScalar<iVector<iVector<vtype, Nrepresentation>, Nhs> >;
+  template <typename vtype> using iImplHalfCommSpinor    = iScalar<iVector<iVector<vtype, Nrepresentation>, Nhcs> >;
  template <typename vtype> using iImplDoubledGaugeField = iVector<iScalar<iMatrix<vtype, Nrepresentation> >, Nds>;
  template <typename vtype> using iImplGaugeField        = iVector<iScalar<iMatrix<vtype, Nrepresentation> >, Nd>;
  template <typename vtype> using iImplGaugeLink         = iScalar<iScalar<iMatrix<vtype, Nrepresentation> > >;
  
-  typedef iImplSpinor<Simd> SiteSpinor;
-  typedef iImplPropagator<Simd> SitePropagator;
-  typedef iImplHalfSpinor<Simd> SiteHalfSpinor;
-  typedef Lattice<SiteSpinor> FermionField;
-  typedef Lattice<SitePropagator> PropagatorField;
-  
+  typedef iImplSpinor<Simd>            SiteSpinor;
+  typedef iImplPropagator<Simd>        SitePropagator;
+  typedef iImplHalfSpinor<Simd>        SiteHalfSpinor;
+  typedef iImplHalfCommSpinor<SimdL>   SiteHalfCommSpinor;
+  typedef Lattice<SiteSpinor>          FermionField;
+  typedef Lattice<SitePropagator>      PropagatorField;

  /////////////////////////////////////////////////
  // Make the doubled gauge field a *scalar*
@@ -241,9 +312,9 @@ class DomainWallVec5dImpl :  public PeriodicGaugeImpl< GaugeImplTypes< S,Nrepres
  typedef iImplDoubledGaugeField<typename Simd::scalar_type>  SiteDoubledGaugeField;  // This is a scalar
  typedef iImplGaugeField<typename Simd::scalar_type>         SiteScalarGaugeField;  // scalar
  typedef iImplGaugeLink<typename Simd::scalar_type>          SiteScalarGaugeLink;  // scalar
-  typedef Lattice<SiteDoubledGaugeField> DoubledGaugeField;
+  typedef Lattice<SiteDoubledGaugeField>                      DoubledGaugeField;
      
-  typedef WilsonCompressor<SiteHalfSpinor, SiteSpinor> Compressor;
+  typedef WilsonCompressor<SiteHalfCommSpinor,SiteHalfSpinor, SiteSpinor> Compressor;
  typedef WilsonImplParams ImplParams;
  typedef WilsonStencil<SiteSpinor, SiteHalfSpinor> StencilImpl;
  
@@ -259,12 +330,12 @@ class DomainWallVec5dImpl :  public PeriodicGaugeImpl< GaugeImplTypes< S,Nrepres
  }

  inline void multLink(SiteHalfSpinor &phi, const SiteDoubledGaugeField &U,
-		       const SiteHalfSpinor &chi, int mu, StencilEntry *SE,
-		       StencilImpl &St) {
+                       const SiteHalfSpinor &chi, int mu, StencilEntry *SE,
+                       StencilImpl &St) {
    SiteGaugeLink UU;
    for (int i = 0; i < Nrepresentation; i++) {
      for (int j = 0; j < Nrepresentation; j++) {
-	vsplat(UU()()(i, j), U(mu)()(i, j));
+        vsplat(UU()()(i, j), U(mu)()(i, j));
      }
    }
    mult(&phi(), &UU(), &chi());
@@ -301,45 +372,90 @@ class DomainWallVec5dImpl :  public PeriodicGaugeImpl< GaugeImplTypes< S,Nrepres
  {
    assert(0);
  }
-      
-  inline void InsertForce5D(GaugeField &mat, FermionField &Btilde,FermionField &Atilde, int mu) 
-  {
-	assert(0);
+
+  inline void InsertForce5D(GaugeField &mat, FermionField &Btilde, FermionField &Atilde, int mu) {
+
+    assert(0);
+    // Following lines to be revised after Peter's addition of half prec
+    // missing put lane...
+    /*
+    typedef decltype(traceIndex<SpinIndex>(outerProduct(Btilde[0], Atilde[0]))) result_type;
+    unsigned int LLs = Btilde._grid->_rdimensions[0];
+    conformable(Atilde._grid,Btilde._grid);
+    GridBase* grid = mat._grid;
+    GridBase* Bgrid = Btilde._grid;
+    unsigned int dimU = grid->Nd();
+    unsigned int dimF = Bgrid->Nd();
+    GaugeLinkField tmp(grid); 
+    tmp = zero;
+    
+    // FIXME 
+    // Current implementation works, thread safe, probably suboptimal
+    // Passing through the local coordinate for grid transformation
+    // the force grid is in general very different from the Ls vectorized grid
+
+    PARALLEL_FOR_LOOP
+    for (int so = 0; so < grid->oSites(); so++) {
+      std::vector<typename result_type::scalar_object> vres(Bgrid->Nsimd());
+      std::vector<int> ocoor;  grid->oCoorFromOindex(ocoor,so); 
+      for (int si = 0; si < tmp._grid->iSites(); si++){
+        typename result_type::scalar_object scalar_object; scalar_object = zero;
+        std::vector<int> local_coor;      
+        std::vector<int> icoor; grid->iCoorFromIindex(icoor,si);
+        grid->InOutCoorToLocalCoor(ocoor, icoor, local_coor);
+        for (int s = 0; s < LLs; s++) {
+          std::vector<int> slocal_coor(dimF);
+          slocal_coor[0] = s;
+          for (int s4d = 1; s4d< dimF; s4d++) slocal_coor[s4d] = local_coor[s4d-1];
+          int sF = Bgrid->oIndexReduced(slocal_coor);  
+          assert(sF < Bgrid->oSites());
+
+          extract(traceIndex<SpinIndex>(outerProduct(Btilde[sF], Atilde[sF])), vres); 
+          // sum across the 5d dimension
+          for (auto v : vres) scalar_object += v;  
+        }
+        tmp._odata[so].putlane(scalar_object, si);
+      }
+    }
+    PokeIndex<LorentzIndex>(mat, tmp, mu);
+    */
  }
 };
    
    ////////////////////////////////////////////////////////////////////////////////////////
    // Flavour doubled spinors; is Gparity the only? what about C*?
    ////////////////////////////////////////////////////////////////////////////////////////
-    
-template <class S, int Nrepresentation,class _Coeff_t = RealD>
+template <class S, int Nrepresentation, class Options=CoeffReal>
 class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresentation> > {
 public:

 static const int Dimension = Nrepresentation;
+ static const int Nhcs = Options::Nhcs;
+ static const bool LsVectorised=false;

- const bool LsVectorised=false;
-
- typedef _Coeff_t Coeff_t;
 typedef ConjugateGaugeImpl< GaugeImplTypes<S,Nrepresentation> > Gimpl;
- 
 INHERIT_GIMPL_TYPES(Gimpl);
+
+ typedef typename Options::_Coeff_t Coeff_t;
+ typedef typename Options::template PrecisionMapper<Simd>::LowerPrecVector SimdL;
      
- template <typename vtype> using iImplSpinor            = iVector<iVector<iVector<vtype, Nrepresentation>, Ns>, Ngp>;
- template <typename vtype> using iImplPropagator        = iVector<iMatrix<iMatrix<vtype, Nrepresentation>, Ns>, Ngp >;
- template <typename vtype> using iImplHalfSpinor        = iVector<iVector<iVector<vtype, Nrepresentation>, Nhs>, Ngp>;
+ template <typename vtype> using iImplSpinor            = iVector<iVector<iVector<vtype, Nrepresentation>, Ns>,   Ngp>;
+ template <typename vtype> using iImplPropagator        = iVector<iMatrix<iMatrix<vtype, Nrepresentation>, Ns>,   Ngp>;
+ template <typename vtype> using iImplHalfSpinor        = iVector<iVector<iVector<vtype, Nrepresentation>, Nhs>,  Ngp>;
+ template <typename vtype> using iImplHalfCommSpinor    = iVector<iVector<iVector<vtype, Nrepresentation>, Nhcs>, Ngp>;
 template <typename vtype> using iImplDoubledGaugeField = iVector<iVector<iScalar<iMatrix<vtype, Nrepresentation> >, Nds>, Ngp>;
-      
- typedef iImplSpinor<Simd> SiteSpinor;
- typedef iImplPropagator<Simd> SitePropagator;
- typedef iImplHalfSpinor<Simd> SiteHalfSpinor;
+
+ typedef iImplSpinor<Simd>            SiteSpinor;
+ typedef iImplPropagator<Simd>        SitePropagator;
+ typedef iImplHalfSpinor<Simd>        SiteHalfSpinor;
+ typedef iImplHalfCommSpinor<SimdL>   SiteHalfCommSpinor;
 typedef iImplDoubledGaugeField<Simd> SiteDoubledGaugeField;

 typedef Lattice<SiteSpinor> FermionField;
 typedef Lattice<SitePropagator> PropagatorField;
 typedef Lattice<SiteDoubledGaugeField> DoubledGaugeField;
 
- typedef WilsonCompressor<SiteHalfSpinor, SiteSpinor> Compressor;
+ typedef WilsonCompressor<SiteHalfCommSpinor,SiteHalfSpinor, SiteSpinor> Compressor;
 typedef WilsonStencil<SiteSpinor, SiteHalfSpinor> StencilImpl;
 
 typedef GparityWilsonImplParams ImplParams;
@@ -353,19 +469,19 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
 // provide the multiply by link that is differentiated between Gparity (with
 // flavour index) and non-Gparity
 inline void multLink(SiteHalfSpinor &phi, const SiteDoubledGaugeField &U,
-		      const SiteHalfSpinor &chi, int mu, StencilEntry *SE,
-		      StencilImpl &St) {
+                      const SiteHalfSpinor &chi, int mu, StencilEntry *SE,
+                      StencilImpl &St) {

-  typedef SiteHalfSpinor vobj;
-  typedef typename SiteHalfSpinor::scalar_object sobj;
+   typedef SiteHalfSpinor vobj;
+   typedef typename SiteHalfSpinor::scalar_object sobj;
 	
   vobj vtmp;
   sobj stmp;
-	
+        
   GridBase *grid = St._grid;
-	
+        
   const int Nsimd = grid->Nsimd();
-	
+        
   int direction = St._directions[mu];
   int distance = St._distances[mu];
   int ptype = St._permute_type[mu];
@@ -373,13 +489,13 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
   
   // Fixme X.Y.Z.T hardcode in stencil
   int mmu = mu % Nd;
-	
+        
   // assert our assumptions
   assert((distance == 1) || (distance == -1));  // nearest neighbour stencil hard code
   assert((sl == 1) || (sl == 2));
   
   std::vector<int> icoor;
-	
+        
   if ( SE->_around_the_world && Params.twists[mmu] ) {

     if ( sl == 2 ) {
@@ -389,25 +505,25 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
       extract(chi,vals);
       for(int s=0;s<Nsimd;s++){

-	 grid->iCoorFromIindex(icoor,s);
-	      
-	 assert((icoor[direction]==0)||(icoor[direction]==1));
-	      
-	 int permute_lane;
-	 if ( distance == 1) {
-	   permute_lane = icoor[direction]?1:0;
-	 } else {
-	   permute_lane = icoor[direction]?0:1;
-	 }
-	      
-	 if ( permute_lane ) { 
-	   stmp(0) = vals[s](1);
-	   stmp(1) = vals[s](0);
-	   vals[s] = stmp;
-	      }
+         grid->iCoorFromIindex(icoor,s);
+              
+         assert((icoor[direction]==0)||(icoor[direction]==1));
+              
+         int permute_lane;
+         if ( distance == 1) {
+           permute_lane = icoor[direction]?1:0;
+         } else {
+           permute_lane = icoor[direction]?0:1;
+         }
+              
+         if ( permute_lane ) { 
+           stmp(0) = vals[s](1);
+           stmp(1) = vals[s](0);
+           vals[s] = stmp;
+              }
       }
       merge(vtmp,vals);
-	    
+            
     } else { 
       vtmp(0) = chi(1);
       vtmp(1) = chi(0);
@@ -432,11 +548,11 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
   GaugeLinkField Uconj(GaugeGrid);
   
   Lattice<iScalar<vInteger> > coor(GaugeGrid);
-	
+        
   for(int mu=0;mu<Nd;mu++){
-	  
+          
     LatticeCoordinate(coor,mu);
-	  
+          
     U     = PeekIndex<LorentzIndex>(Umu,mu);
     Uconj = conjugate(U);
     
@@ -450,7 +566,7 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
       Uds[ss](0)(mu) = U[ss]();
       Uds[ss](1)(mu) = Uconj[ss]();
     }
-	  
+          
     U     = adj(Cshift(U    ,mu,-1));      // correct except for spanning the boundary
     Uconj = adj(Cshift(Uconj,mu,-1));
 
@@ -458,11 +574,12 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
     if ( Params.twists[mu] ) { 
       Utmp = where(coor==0,Uconj,Utmp);
     }
+
 	  
     parallel_for(auto ss=U.begin();ss<U.end();ss++){
       Uds[ss](0)(mu+4) = Utmp[ss]();
     }
-	  
+          
     Utmp = Uconj;
     if ( Params.twists[mu] ) { 
       Utmp = where(coor==0,U,Utmp);
@@ -471,11 +588,10 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
     parallel_for(auto ss=U.begin();ss<U.end();ss++){
       Uds[ss](1)(mu+4) = Utmp[ss]();
     }
-	  
+          
   }
 }
      
-      
 inline void InsertForce4D(GaugeField &mat, FermionField &Btilde, FermionField &A, int mu) {

   // DhopDir provides U or Uconj depending on coor/flavour.
@@ -483,7 +599,7 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
   // use lorentz for flavour as hack.
   auto tmp = TraceIndex<SpinIndex>(outerProduct(Btilde, A));
   parallel_for(auto ss = tmp.begin(); ss < tmp.end(); ss++) {
-     link[ss]() = tmp[ss](0, 0) - conjugate(tmp[ss](1, 1));
+     link[ss]() = tmp[ss](0, 0) + conjugate(tmp[ss](1, 1));
   }
   PokeIndex<LorentzIndex>(mat, link, mu);
   return;
@@ -492,7 +608,7 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
 inline void InsertForce5D(GaugeField &mat, FermionField &Btilde, FermionField &Atilde, int mu) {

   int Ls = Btilde._grid->_fdimensions[0];
-	
+        
   GaugeLinkField tmp(mat._grid);
   tmp = zero;
   parallel_for(int ss = 0; ss < tmp._grid->oSites(); ss++) {
@@ -508,40 +624,36 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent

 };

-
-  /////////////////////////////////////////////////////////////////////////////
-  // Single flavour one component spinors with colour index
-  /////////////////////////////////////////////////////////////////////////////
-  template <class S, class Representation = FundamentalRepresentation >
-  class StaggeredImpl : public PeriodicGaugeImpl<GaugeImplTypes<S, Representation::Dimension > > {
+/////////////////////////////////////////////////////////////////////////////
+// Single flavour one component spinors with colour index
+/////////////////////////////////////////////////////////////////////////////
+template <class S, class Representation = FundamentalRepresentation >
+class StaggeredImpl : public PeriodicGaugeImpl<GaugeImplTypes<S, Representation::Dimension > > {

    public:

    typedef RealD  _Coeff_t ;
    static const int Dimension = Representation::Dimension;
+    static const bool LsVectorised=false;
    typedef PeriodicGaugeImpl<GaugeImplTypes<S, Dimension > > Gimpl;
      
    //Necessary?
    constexpr bool is_fundamental() const{return Dimension == Nc ? 1 : 0;}
    
-    const bool LsVectorised=false;
    typedef _Coeff_t Coeff_t;

    INHERIT_GIMPL_TYPES(Gimpl);
      
-    template <typename vtype> using iImplScalar            = iScalar<iScalar<iScalar<vtype> > >;
    template <typename vtype> using iImplSpinor            = iScalar<iScalar<iVector<vtype, Dimension> > >;
    template <typename vtype> using iImplHalfSpinor        = iScalar<iScalar<iVector<vtype, Dimension> > >;
    template <typename vtype> using iImplDoubledGaugeField = iVector<iScalar<iMatrix<vtype, Dimension> >, Nds>;
    template <typename vtype> using iImplPropagator        = iScalar<iScalar<iMatrix<vtype, Dimension> > >;
    
-    typedef iImplScalar<Simd>            SiteComplex;
    typedef iImplSpinor<Simd>            SiteSpinor;
    typedef iImplHalfSpinor<Simd>        SiteHalfSpinor;
    typedef iImplDoubledGaugeField<Simd> SiteDoubledGaugeField;
    typedef iImplPropagator<Simd>        SitePropagator;
    
-    typedef Lattice<SiteComplex>           ComplexField;
    typedef Lattice<SiteSpinor>            FermionField;
    typedef Lattice<SiteDoubledGaugeField> DoubledGaugeField;
    typedef Lattice<SitePropagator> PropagatorField;
@@ -641,8 +753,6 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
    }
  };

-
-
  /////////////////////////////////////////////////////////////////////////////
  // Single flavour one component spinors with colour index. 5d vec
  /////////////////////////////////////////////////////////////////////////////
@@ -651,20 +761,17 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent

    public:

-    typedef RealD  _Coeff_t ;
    static const int Dimension = Representation::Dimension;
+    static const bool LsVectorised=true;
+    typedef RealD   Coeff_t ;
    typedef PeriodicGaugeImpl<GaugeImplTypes<S, Dimension > > Gimpl;
      
    //Necessary?
    constexpr bool is_fundamental() const{return Dimension == Nc ? 1 : 0;}
-    
-    const bool LsVectorised=true;

-    typedef _Coeff_t Coeff_t;

    INHERIT_GIMPL_TYPES(Gimpl);

-    template <typename vtype> using iImplScalar            = iScalar<iScalar<iScalar<vtype> > >;
    template <typename vtype> using iImplSpinor            = iScalar<iScalar<iVector<vtype, Dimension> > >;
    template <typename vtype> using iImplHalfSpinor        = iScalar<iScalar<iVector<vtype, Dimension> > >;
    template <typename vtype> using iImplDoubledGaugeField = iVector<iScalar<iMatrix<vtype, Dimension> >, Nds>;
@@ -681,12 +788,10 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
    typedef Lattice<SiteDoubledGaugeField> DoubledGaugeField;
    typedef Lattice<SitePropagator> PropagatorField;
    
-    typedef iImplScalar<Simd>            SiteComplex;
    typedef iImplSpinor<Simd>            SiteSpinor;
    typedef iImplHalfSpinor<Simd>        SiteHalfSpinor;

    
-    typedef Lattice<SiteComplex>           ComplexField;
    typedef Lattice<SiteSpinor>            FermionField;
    
    typedef SimpleCompressor<SiteSpinor> Compressor;
@@ -823,43 +928,61 @@ class GparityWilsonImpl : public ConjugateGaugeImpl<GaugeImplTypes<S, Nrepresent
    }
  };

+typedef WilsonImpl<vComplex,  FundamentalRepresentation, CoeffReal > WilsonImplR;  // Real.. whichever prec
+typedef WilsonImpl<vComplexF, FundamentalRepresentation, CoeffReal > WilsonImplF;  // Float
+typedef WilsonImpl<vComplexD, FundamentalRepresentation, CoeffReal > WilsonImplD;  // Double

+typedef WilsonImpl<vComplex,  FundamentalRepresentation, CoeffRealHalfComms > WilsonImplRL;  // Real.. whichever prec
+typedef WilsonImpl<vComplexF, FundamentalRepresentation, CoeffRealHalfComms > WilsonImplFH;  // Float
+typedef WilsonImpl<vComplexD, FundamentalRepresentation, CoeffRealHalfComms > WilsonImplDF;  // Double

- typedef WilsonImpl<vComplex,  FundamentalRepresentation > WilsonImplR;   // Real.. whichever prec
- typedef WilsonImpl<vComplexF, FundamentalRepresentation > WilsonImplF;  // Float
- typedef WilsonImpl<vComplexD, FundamentalRepresentation > WilsonImplD;  // Double
+typedef WilsonImpl<vComplex,  FundamentalRepresentation, CoeffComplex > ZWilsonImplR; // Real.. whichever prec
+typedef WilsonImpl<vComplexF, FundamentalRepresentation, CoeffComplex > ZWilsonImplF; // Float
+typedef WilsonImpl<vComplexD, FundamentalRepresentation, CoeffComplex > ZWilsonImplD; // Double

- typedef WilsonImpl<vComplex,  FundamentalRepresentation, ComplexD > ZWilsonImplR; // Real.. whichever prec
- typedef WilsonImpl<vComplexF, FundamentalRepresentation, ComplexD > ZWilsonImplF; // Float
- typedef WilsonImpl<vComplexD, FundamentalRepresentation, ComplexD > ZWilsonImplD; // Double
+typedef WilsonImpl<vComplex,  FundamentalRepresentation, CoeffComplexHalfComms > ZWilsonImplRL; // Real.. whichever prec
+typedef WilsonImpl<vComplexF, FundamentalRepresentation, CoeffComplexHalfComms > ZWilsonImplFH; // Float
+typedef WilsonImpl<vComplexD, FundamentalRepresentation, CoeffComplexHalfComms > ZWilsonImplDF; // Double
 
- typedef WilsonImpl<vComplex,  AdjointRepresentation > WilsonAdjImplR;   // Real.. whichever prec
- typedef WilsonImpl<vComplexF, AdjointRepresentation > WilsonAdjImplF;  // Float
- typedef WilsonImpl<vComplexD, AdjointRepresentation > WilsonAdjImplD;  // Double
+typedef WilsonImpl<vComplex,  AdjointRepresentation, CoeffReal > WilsonAdjImplR;   // Real.. whichever prec
+typedef WilsonImpl<vComplexF, AdjointRepresentation, CoeffReal > WilsonAdjImplF;  // Float
+typedef WilsonImpl<vComplexD, AdjointRepresentation, CoeffReal > WilsonAdjImplD;  // Double
 
- typedef WilsonImpl<vComplex,  TwoIndexSymmetricRepresentation > WilsonTwoIndexSymmetricImplR;   // Real.. whichever prec
- typedef WilsonImpl<vComplexF, TwoIndexSymmetricRepresentation > WilsonTwoIndexSymmetricImplF;  // Float
- typedef WilsonImpl<vComplexD, TwoIndexSymmetricRepresentation > WilsonTwoIndexSymmetricImplD;  // Double
+typedef WilsonImpl<vComplex,  TwoIndexSymmetricRepresentation, CoeffReal > WilsonTwoIndexSymmetricImplR;   // Real.. whichever prec
+typedef WilsonImpl<vComplexF, TwoIndexSymmetricRepresentation, CoeffReal > WilsonTwoIndexSymmetricImplF;  // Float
+typedef WilsonImpl<vComplexD, TwoIndexSymmetricRepresentation, CoeffReal > WilsonTwoIndexSymmetricImplD;  // Double
 
- typedef DomainWallVec5dImpl<vComplex ,Nc> DomainWallVec5dImplR; // Real.. whichever prec
- typedef DomainWallVec5dImpl<vComplexF,Nc> DomainWallVec5dImplF; // Float
- typedef DomainWallVec5dImpl<vComplexD,Nc> DomainWallVec5dImplD; // Double
+typedef DomainWallVec5dImpl<vComplex ,Nc, CoeffReal> DomainWallVec5dImplR; // Real.. whichever prec
+typedef DomainWallVec5dImpl<vComplexF,Nc, CoeffReal> DomainWallVec5dImplF; // Float
+typedef DomainWallVec5dImpl<vComplexD,Nc, CoeffReal> DomainWallVec5dImplD; // Double
 
- typedef DomainWallVec5dImpl<vComplex ,Nc,ComplexD> ZDomainWallVec5dImplR; // Real.. whichever prec
- typedef DomainWallVec5dImpl<vComplexF,Nc,ComplexD> ZDomainWallVec5dImplF; // Float
- typedef DomainWallVec5dImpl<vComplexD,Nc,ComplexD> ZDomainWallVec5dImplD; // Double
+typedef DomainWallVec5dImpl<vComplex ,Nc, CoeffRealHalfComms> DomainWallVec5dImplRL; // Real.. whichever prec
+typedef DomainWallVec5dImpl<vComplexF,Nc, CoeffRealHalfComms> DomainWallVec5dImplFH; // Float
+typedef DomainWallVec5dImpl<vComplexD,Nc, CoeffRealHalfComms> DomainWallVec5dImplDF; // Double
 
- typedef GparityWilsonImpl<vComplex , Nc> GparityWilsonImplR;  // Real.. whichever prec
- typedef GparityWilsonImpl<vComplexF, Nc> GparityWilsonImplF;  // Float
- typedef GparityWilsonImpl<vComplexD, Nc> GparityWilsonImplD;  // Double
+typedef DomainWallVec5dImpl<vComplex ,Nc,CoeffComplex> ZDomainWallVec5dImplR; // Real.. whichever prec
+typedef DomainWallVec5dImpl<vComplexF,Nc,CoeffComplex> ZDomainWallVec5dImplF; // Float
+typedef DomainWallVec5dImpl<vComplexD,Nc,CoeffComplex> ZDomainWallVec5dImplD; // Double
+ 
+typedef DomainWallVec5dImpl<vComplex ,Nc,CoeffComplexHalfComms> ZDomainWallVec5dImplRL; // Real.. whichever prec
+typedef DomainWallVec5dImpl<vComplexF,Nc,CoeffComplexHalfComms> ZDomainWallVec5dImplFH; // Float
+typedef DomainWallVec5dImpl<vComplexD,Nc,CoeffComplexHalfComms> ZDomainWallVec5dImplDF; // Double
+ 
+typedef GparityWilsonImpl<vComplex , Nc,CoeffReal> GparityWilsonImplR;  // Real.. whichever prec
+typedef GparityWilsonImpl<vComplexF, Nc,CoeffReal> GparityWilsonImplF;  // Float
+typedef GparityWilsonImpl<vComplexD, Nc,CoeffReal> GparityWilsonImplD;  // Double
+ 
+typedef GparityWilsonImpl<vComplex , Nc,CoeffRealHalfComms> GparityWilsonImplRL;  // Real.. whichever prec
+typedef GparityWilsonImpl<vComplexF, Nc,CoeffRealHalfComms> GparityWilsonImplFH;  // Float
+typedef GparityWilsonImpl<vComplexD, Nc,CoeffRealHalfComms> GparityWilsonImplDF;  // Double

- typedef StaggeredImpl<vComplex,  FundamentalRepresentation > StaggeredImplR;   // Real.. whichever prec
- typedef StaggeredImpl<vComplexF, FundamentalRepresentation > StaggeredImplF;  // Float
- typedef StaggeredImpl<vComplexD, FundamentalRepresentation > StaggeredImplD;  // Double
+typedef StaggeredImpl<vComplex,  FundamentalRepresentation > StaggeredImplR;   // Real.. whichever prec
+typedef StaggeredImpl<vComplexF, FundamentalRepresentation > StaggeredImplF;  // Float
+typedef StaggeredImpl<vComplexD, FundamentalRepresentation > StaggeredImplD;  // Double

- typedef StaggeredVec5dImpl<vComplex,  FundamentalRepresentation > StaggeredVec5dImplR;   // Real.. whichever prec
- typedef StaggeredVec5dImpl<vComplexF, FundamentalRepresentation > StaggeredVec5dImplF;  // Float
- typedef StaggeredVec5dImpl<vComplexD, FundamentalRepresentation > StaggeredVec5dImplD;  // Double
+typedef StaggeredVec5dImpl<vComplex,  FundamentalRepresentation > StaggeredVec5dImplR;   // Real.. whichever prec
+typedef StaggeredVec5dImpl<vComplexF, FundamentalRepresentation > StaggeredVec5dImplF;  // Float
+typedef StaggeredVec5dImpl<vComplexD, FundamentalRepresentation > StaggeredVec5dImplD;  // Double

 }}

@@ -33,228 +33,321 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
 namespace Grid {
 namespace QCD {

-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonCompressor {
-  public:
-    int mu;
-    int dag;
+/////////////////////////////////////////////////////////////////////////////////////////////
+// optimised versions supporting half precision too
+/////////////////////////////////////////////////////////////////////////////////////////////

-    WilsonCompressor(int _dag){
-      mu=0;
-      dag=_dag;
-      assert((dag==0)||(dag==1));
-    }
-    void Point(int p) { 
-      mu=p;
-    };
+template<class _HCspinor,class _Hspinor,class _Spinor, class projector,typename SFINAE = void >
+class WilsonCompressorTemplate;

-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      int mudag=mu;
-      if (!dag) {
-	mudag=(mu+Nd)%(2*Nd);
-      }
-      switch(mudag) {
-      case Xp:
-	spProjXp(ret,in);
-	break;
-      case Yp:
-	spProjYp(ret,in);
-	break;
-      case Zp:
-	spProjZp(ret,in);
-	break;
-      case Tp:
-	spProjTp(ret,in);
-	break;
-      case Xm:
-	spProjXm(ret,in);
-	break;
-      case Ym:
-	spProjYm(ret,in);
-	break;
-      case Zm:
-	spProjZm(ret,in);
-	break;
-      case Tm:
-	spProjTm(ret,in);
-	break;
-      default: 
-	assert(0);
-	break;
-      }
-      return ret;
-    }
-  };

-  /////////////////////////
-  // optimised versions
-  /////////////////////////
+template<class _HCspinor,class _Hspinor,class _Spinor, class projector>
+class WilsonCompressorTemplate< _HCspinor, _Hspinor, _Spinor, projector,
+  typename std::enable_if<std::is_same<_HCspinor,_Hspinor>::value>::type >
+{
+ public:
+  
+  int mu,dag;  

-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonXpCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjXp(ret,in);
-      return ret;
-    }
-  };
-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonYpCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjYp(ret,in);
-      return ret;
-    }
-  };
-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonZpCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjZp(ret,in);
-      return ret;
-    }
-  };
-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonTpCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjTp(ret,in);
-      return ret;
-    }
-  };
+  void Point(int p) { mu=p; };

-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonXmCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjXm(ret,in);
-      return ret;
-    }
-  };
-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonYmCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjYm(ret,in);
-      return ret;
-    }
-  };
-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonZmCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjZm(ret,in);
-      return ret;
-    }
-  };
-  template<class SiteHalfSpinor,class SiteSpinor>
-  class WilsonTmCompressor {
-  public:
-    inline SiteHalfSpinor operator () (const SiteSpinor &in) {
-      SiteHalfSpinor ret;
-      spProjTm(ret,in);
-      return ret;
-    }
-  };
+  WilsonCompressorTemplate(int _dag=0){
+    dag = _dag;
+  }

-    // Fast comms buffer manipulation which should inline right through (avoid direction
-    // dependent logic that prevents inlining
-  template<class vobj,class cobj>
-  class WilsonStencil : public CartesianStencil<vobj,cobj> {
-  public:
+  typedef _Spinor         SiteSpinor;
+  typedef _Hspinor     SiteHalfSpinor;
+  typedef _HCspinor SiteHalfCommSpinor;
+  typedef typename SiteHalfCommSpinor::vector_type vComplexLow;
+  typedef typename SiteHalfSpinor::vector_type     vComplexHigh;
+  constexpr static int Nw=sizeof(SiteHalfSpinor)/sizeof(vComplexHigh);

-    typedef CartesianCommunicator::CommsRequest_t CommsRequest_t;
+  inline int CommDatumSize(void) {
+    return sizeof(SiteHalfCommSpinor);
+  }

-    WilsonStencil(GridBase *grid,
+  /*****************************************************/
+  /* Compress includes precision change if mpi data is not same */
+  /*****************************************************/
+  inline void Compress(SiteHalfSpinor *buf,Integer o,const SiteSpinor &in) {
+    projector::Proj(buf[o],in,mu,dag);
+  }
+
+  /*****************************************************/
+  /* Exchange includes precision change if mpi data is not same */
+  /*****************************************************/
+  inline void Exchange(SiteHalfSpinor *mp,
+                       SiteHalfSpinor *vp0,
+                       SiteHalfSpinor *vp1,
+		       Integer type,Integer o){
+    exchange(mp[2*o],mp[2*o+1],vp0[o],vp1[o],type);
+  }
+
+  /*****************************************************/
+  /* Have a decompression step if mpi data is not same */
+  /*****************************************************/
+  inline void Decompress(SiteHalfSpinor *out,
+			 SiteHalfSpinor *in, Integer o) {    
+    assert(0);
+  }
+
+  /*****************************************************/
+  /* Compress Exchange                                 */
+  /*****************************************************/
+  inline void CompressExchange(SiteHalfSpinor *out0,
+			       SiteHalfSpinor *out1,
+			       const SiteSpinor *in,
+			       Integer j,Integer k, Integer m,Integer type){
+    SiteHalfSpinor temp1, temp2,temp3,temp4;
+    projector::Proj(temp1,in[k],mu,dag);
+    projector::Proj(temp2,in[m],mu,dag);
+    exchange(out0[j],out1[j],temp1,temp2,type);
+  }
+
+  /*****************************************************/
+  /* Pass the info to the stencil */
+  /*****************************************************/
+  inline bool DecompressionStep(void) { return false; }
+
+};
+
+template<class _HCspinor,class _Hspinor,class _Spinor, class projector>
+class WilsonCompressorTemplate< _HCspinor, _Hspinor, _Spinor, projector,
+  typename std::enable_if<!std::is_same<_HCspinor,_Hspinor>::value>::type >
+{
+ public:
+  
+  int mu,dag;  
+
+  void Point(int p) { mu=p; };
+
+  WilsonCompressorTemplate(int _dag=0){
+    dag = _dag;
+  }
+
+  typedef _Spinor         SiteSpinor;
+  typedef _Hspinor     SiteHalfSpinor;
+  typedef _HCspinor SiteHalfCommSpinor;
+  typedef typename SiteHalfCommSpinor::vector_type vComplexLow;
+  typedef typename SiteHalfSpinor::vector_type     vComplexHigh;
+  constexpr static int Nw=sizeof(SiteHalfSpinor)/sizeof(vComplexHigh);
+
+  inline int CommDatumSize(void) {
+    return sizeof(SiteHalfCommSpinor);
+  }
+
+  /*****************************************************/
+  /* Compress includes precision change if mpi data is not same */
+  /*****************************************************/
+  inline void Compress(SiteHalfSpinor *buf,Integer o,const SiteSpinor &in) {
+    SiteHalfSpinor hsp;
+    SiteHalfCommSpinor *hbuf = (SiteHalfCommSpinor *)buf;
+    projector::Proj(hsp,in,mu,dag);
+    precisionChange((vComplexLow *)&hbuf[o],(vComplexHigh *)&hsp,Nw);
+  }
+
+  /*****************************************************/
+  /* Exchange includes precision change if mpi data is not same */
+  /*****************************************************/
+  inline void Exchange(SiteHalfSpinor *mp,
+                       SiteHalfSpinor *vp0,
+                       SiteHalfSpinor *vp1,
+		       Integer type,Integer o){
+    SiteHalfSpinor vt0,vt1;
+    SiteHalfCommSpinor *vpp0 = (SiteHalfCommSpinor *)vp0;
+    SiteHalfCommSpinor *vpp1 = (SiteHalfCommSpinor *)vp1;
+    precisionChange((vComplexHigh *)&vt0,(vComplexLow *)&vpp0[o],Nw);
+    precisionChange((vComplexHigh *)&vt1,(vComplexLow *)&vpp1[o],Nw);
+    exchange(mp[2*o],mp[2*o+1],vt0,vt1,type);
+  }
+
+  /*****************************************************/
+  /* Have a decompression step if mpi data is not same */
+  /*****************************************************/
+  inline void Decompress(SiteHalfSpinor *out,
+			 SiteHalfSpinor *in, Integer o){
+    SiteHalfCommSpinor *hin=(SiteHalfCommSpinor *)in;
+    precisionChange((vComplexHigh *)&out[o],(vComplexLow *)&hin[o],Nw);
+  }
+
+  /*****************************************************/
+  /* Compress Exchange                                 */
+  /*****************************************************/
+  inline void CompressExchange(SiteHalfSpinor *out0,
+			       SiteHalfSpinor *out1,
+			       const SiteSpinor *in,
+			       Integer j,Integer k, Integer m,Integer type){
+    SiteHalfSpinor temp1, temp2,temp3,temp4;
+    SiteHalfCommSpinor *hout0 = (SiteHalfCommSpinor *)out0;
+    SiteHalfCommSpinor *hout1 = (SiteHalfCommSpinor *)out1;
+    projector::Proj(temp1,in[k],mu,dag);
+    projector::Proj(temp2,in[m],mu,dag);
+    exchange(temp3,temp4,temp1,temp2,type);
+    precisionChange((vComplexLow *)&hout0[j],(vComplexHigh *)&temp3,Nw);
+    precisionChange((vComplexLow *)&hout1[j],(vComplexHigh *)&temp4,Nw);
+  }
+
+  /*****************************************************/
+  /* Pass the info to the stencil */
+  /*****************************************************/
+  inline bool DecompressionStep(void) { return true; }
+
+};
+
+#define DECLARE_PROJ(Projector,Compressor,spProj)			\
+  class Projector {							\
+  public:								\
+    template<class hsp,class fsp>					\
+    static void Proj(hsp &result,const fsp &in,int mu,int dag){			\
+      spProj(result,in);						\
+    }									\
+  };									\
+template<typename HCS,typename HS,typename S> using Compressor = WilsonCompressorTemplate<HCS,HS,S,Projector>;
+
+DECLARE_PROJ(WilsonXpProjector,WilsonXpCompressor,spProjXp);
+DECLARE_PROJ(WilsonYpProjector,WilsonYpCompressor,spProjYp);
+DECLARE_PROJ(WilsonZpProjector,WilsonZpCompressor,spProjZp);
+DECLARE_PROJ(WilsonTpProjector,WilsonTpCompressor,spProjTp);
+DECLARE_PROJ(WilsonXmProjector,WilsonXmCompressor,spProjXm);
+DECLARE_PROJ(WilsonYmProjector,WilsonYmCompressor,spProjYm);
+DECLARE_PROJ(WilsonZmProjector,WilsonZmCompressor,spProjZm);
+DECLARE_PROJ(WilsonTmProjector,WilsonTmCompressor,spProjTm);
+
+class WilsonProjector {
+ public:
+  template<class hsp,class fsp>
+  static void Proj(hsp &result,const fsp &in,int mu,int dag){
+    int mudag=dag? mu : (mu+Nd)%(2*Nd);
+    switch(mudag) {
+    case Xp:	spProjXp(result,in);	break;
+    case Yp:	spProjYp(result,in);	break;
+    case Zp:	spProjZp(result,in);	break;
+    case Tp:	spProjTp(result,in);	break;
+    case Xm:	spProjXm(result,in);	break;
+    case Ym:	spProjYm(result,in);	break;
+    case Zm:	spProjZm(result,in);	break;
+    case Tm:	spProjTm(result,in);	break;
+    default: 	assert(0);	        break;
+    }
+  }
+};
+template<typename HCS,typename HS,typename S> using WilsonCompressor = WilsonCompressorTemplate<HCS,HS,S,WilsonProjector>;
+
+// Fast comms buffer manipulation which should inline right through (avoid direction
+// dependent logic that prevents inlining
+template<class vobj,class cobj>
+class WilsonStencil : public CartesianStencil<vobj,cobj> {
+public:
+
+  typedef CartesianCommunicator::CommsRequest_t CommsRequest_t;
+
+  std::vector<int> same_node;
+  std::vector<int> surface_list;
+
+  WilsonStencil(GridBase *grid,
 		int npoints,
 		int checkerboard,
 		const std::vector<int> &directions,
-		const std::vector<int> &distances)  : CartesianStencil<vobj,cobj> (grid,npoints,checkerboard,directions,distances) 
-      {    };
-
-    template < class compressor>
-    void HaloExchangeOpt(const Lattice<vobj> &source,compressor &compress) 
-    {
-      std::vector<std::vector<CommsRequest_t> > reqs;
-      HaloExchangeOptGather(source,compress);
-      this->CommunicateBegin(reqs);
-      this->calls++;
-      this->CommunicateComplete(reqs);
-      this->CommsMerge();
-    }
-
-    template < class compressor>
-    void HaloExchangeOptGather(const Lattice<vobj> &source,compressor &compress) 
-    {
-      this->calls++;
-      this->Mergers.resize(0); 
-      this->Packets.resize(0);
-      this->HaloGatherOpt(source,compress);
-    }
-
-
-    template < class compressor>
-    void HaloGatherOpt(const Lattice<vobj> &source,compressor &compress)
-    {
-      this->_grid->StencilBarrier();
-      // conformable(source._grid,_grid);
-      assert(source._grid==this->_grid);
-      this->halogtime-=usecond();
-      
-      this->u_comm_offset=0;
-      
-      int dag = compress.dag;
-      
-      WilsonXpCompressor<cobj,vobj> XpCompress; 
-      WilsonYpCompressor<cobj,vobj> YpCompress; 
-      WilsonZpCompressor<cobj,vobj> ZpCompress; 
-      WilsonTpCompressor<cobj,vobj> TpCompress;
-      WilsonXmCompressor<cobj,vobj> XmCompress;
-      WilsonYmCompressor<cobj,vobj> YmCompress;
-      WilsonZmCompressor<cobj,vobj> ZmCompress;
-      WilsonTmCompressor<cobj,vobj> TmCompress;
-
-      // Gather all comms buffers
-      //    for(int point = 0 ; point < _npoints; point++) {
-      //      compress.Point(point);
-      //      HaloGatherDir(source,compress,point,face_idx);
-      //    }
-      int face_idx=0;
-      if ( dag ) { 
-	//	std::cout << " Optimised Dagger compress " <<std::endl;
-	this->HaloGatherDir(source,XpCompress,Xp,face_idx);
-	this->HaloGatherDir(source,YpCompress,Yp,face_idx);
-	this->HaloGatherDir(source,ZpCompress,Zp,face_idx);
-	this->HaloGatherDir(source,TpCompress,Tp,face_idx);
-	this->HaloGatherDir(source,XmCompress,Xm,face_idx);
-	this->HaloGatherDir(source,YmCompress,Ym,face_idx);
-	this->HaloGatherDir(source,ZmCompress,Zm,face_idx);
-	this->HaloGatherDir(source,TmCompress,Tm,face_idx);
-      } else {
-	this->HaloGatherDir(source,XmCompress,Xp,face_idx);
-	this->HaloGatherDir(source,YmCompress,Yp,face_idx);
-	this->HaloGatherDir(source,ZmCompress,Zp,face_idx);
-	this->HaloGatherDir(source,TmCompress,Tp,face_idx);
-	this->HaloGatherDir(source,XpCompress,Xm,face_idx);
-	this->HaloGatherDir(source,YpCompress,Ym,face_idx);
-	this->HaloGatherDir(source,ZpCompress,Zm,face_idx);
-	this->HaloGatherDir(source,TpCompress,Tm,face_idx);
-      }
-      this->face_table_computed=1;
-      assert(this->u_comm_offset==this->_unified_buffer_size);
-      this->halogtime+=usecond();
-    }
-
+		const std::vector<int> &distances)  
+    : CartesianStencil<vobj,cobj> (grid,npoints,checkerboard,directions,distances) ,
+    same_node(npoints)
+  { 
+    surface_list.resize(0);
  };

+  void BuildSurfaceList(int Ls,int vol4){
+
+    // find same node for SHM
+    // Here we know the distance is 1 for WilsonStencil
+    for(int point=0;point<this->_npoints;point++){
+      same_node[point] = this->SameNode(point);
+      //      std::cout << " dir " <<point<<" same_node " <<same_node[point]<<std::endl;
+    }
+    
+    for(int site = 0 ;site< vol4;site++){
+      int local = 1;
+      for(int point=0;point<this->_npoints;point++){
+	if( (!this->GetNodeLocal(site*Ls,point)) && (!same_node[point]) ){ 
+	  local = 0;
+	}
+      }
+      if(local == 0) { 
+	surface_list.push_back(site);
+      }
+    }
+  }
+
+  template < class compressor>
+  void HaloExchangeOpt(const Lattice<vobj> &source,compressor &compress) 
+  {
+    std::vector<std::vector<CommsRequest_t> > reqs;
+    this->HaloExchangeOptGather(source,compress);
+    this->CommunicateBegin(reqs);
+    this->CommunicateComplete(reqs);
+    this->CommsMerge(compress);
+    this->CommsMergeSHM(compress);
+  }
+  
+  template <class compressor>
+  void HaloExchangeOptGather(const Lattice<vobj> &source,compressor &compress) 
+  {
+    this->Prepare();
+    this->HaloGatherOpt(source,compress);
+  }
+
+  template <class compressor>
+  void HaloGatherOpt(const Lattice<vobj> &source,compressor &compress)
+  {
+    // Strategy. Inherit types from Compressor.
+    // Use types to select the write direction by directon compressor
+    typedef typename compressor::SiteSpinor         SiteSpinor;
+    typedef typename compressor::SiteHalfSpinor     SiteHalfSpinor;
+    typedef typename compressor::SiteHalfCommSpinor SiteHalfCommSpinor;
+
+    this->_grid->StencilBarrier();
+
+    assert(source._grid==this->_grid);
+    this->halogtime-=usecond();
+    
+    this->u_comm_offset=0;
+      
+    WilsonXpCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> XpCompress; 
+    WilsonYpCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> YpCompress; 
+    WilsonZpCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> ZpCompress; 
+    WilsonTpCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> TpCompress;
+    WilsonXmCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> XmCompress; 
+    WilsonYmCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> YmCompress; 
+    WilsonZmCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> ZmCompress; 
+    WilsonTmCompressor<SiteHalfCommSpinor,SiteHalfSpinor,SiteSpinor> TmCompress;
+
+    int dag = compress.dag;
+    int face_idx=0;
+    if ( dag ) { 
+      //	std::cout << " Optimised Dagger compress " <<std::endl;
+      assert(same_node[Xp]==this->HaloGatherDir(source,XpCompress,Xp,face_idx));
+      assert(same_node[Yp]==this->HaloGatherDir(source,YpCompress,Yp,face_idx));
+      assert(same_node[Zp]==this->HaloGatherDir(source,ZpCompress,Zp,face_idx));
+      assert(same_node[Tp]==this->HaloGatherDir(source,TpCompress,Tp,face_idx));
+      assert(same_node[Xm]==this->HaloGatherDir(source,XmCompress,Xm,face_idx));
+      assert(same_node[Ym]==this->HaloGatherDir(source,YmCompress,Ym,face_idx));
+      assert(same_node[Zm]==this->HaloGatherDir(source,ZmCompress,Zm,face_idx));
+      assert(same_node[Tm]==this->HaloGatherDir(source,TmCompress,Tm,face_idx));
+    } else {
+      assert(same_node[Xp]==this->HaloGatherDir(source,XmCompress,Xp,face_idx));
+      assert(same_node[Yp]==this->HaloGatherDir(source,YmCompress,Yp,face_idx));
+      assert(same_node[Zp]==this->HaloGatherDir(source,ZmCompress,Zp,face_idx));
+      assert(same_node[Tp]==this->HaloGatherDir(source,TmCompress,Tp,face_idx));
+      assert(same_node[Xm]==this->HaloGatherDir(source,XpCompress,Xm,face_idx));
+      assert(same_node[Ym]==this->HaloGatherDir(source,YpCompress,Ym,face_idx));
+      assert(same_node[Zm]==this->HaloGatherDir(source,ZpCompress,Zm,face_idx));
+      assert(same_node[Tm]==this->HaloGatherDir(source,TpCompress,Tm,face_idx));
+    }
+    this->face_table_computed=1;
+    assert(this->u_comm_offset==this->_unified_buffer_size);
+    this->halogtime+=usecond();
+  }
+
+ };

 }} // namespace close
 #endif
@@ -230,8 +230,7 @@ void WilsonFermion<Impl>::DerivInternal(StencilImpl &st, DoubledGaugeField &U,
 }

 template <class Impl>
-void WilsonFermion<Impl>::DhopDeriv(GaugeField &mat, const FermionField &U,
-                                    const FermionField &V, int dag) {
+void WilsonFermion<Impl>::DhopDeriv(GaugeField &mat, const FermionField &U, const FermionField &V, int dag) {
  conformable(U._grid, _grid);
  conformable(U._grid, V._grid);
  conformable(U._grid, mat._grid);
@@ -242,12 +241,12 @@ void WilsonFermion<Impl>::DhopDeriv(GaugeField &mat, const FermionField &U,
 }

 template <class Impl>
-void WilsonFermion<Impl>::DhopDerivOE(GaugeField &mat, const FermionField &U,
-                                      const FermionField &V, int dag) {
+void WilsonFermion<Impl>::DhopDerivOE(GaugeField &mat, const FermionField &U, const FermionField &V, int dag) {
  conformable(U._grid, _cbgrid);
  conformable(U._grid, V._grid);
-  conformable(U._grid, mat._grid);
-
+  //conformable(U._grid, mat._grid); not general, leaving as a comment (Guido)
+  // Motivation: look at the SchurDiff operator
+  
  assert(V.checkerboard == Even);
  assert(U.checkerboard == Odd);
  mat.checkerboard = Odd;
@@ -256,11 +255,10 @@ void WilsonFermion<Impl>::DhopDerivOE(GaugeField &mat, const FermionField &U,
 }

 template <class Impl>
-void WilsonFermion<Impl>::DhopDerivEO(GaugeField &mat, const FermionField &U,
-                                      const FermionField &V, int dag) {
+void WilsonFermion<Impl>::DhopDerivEO(GaugeField &mat, const FermionField &U, const FermionField &V, int dag) {
  conformable(U._grid, _cbgrid);
  conformable(U._grid, V._grid);
-  conformable(U._grid, mat._grid);
+  //conformable(U._grid, mat._grid);

  assert(V.checkerboard == Odd);
  assert(U.checkerboard == Even);
@@ -11,6 +11,7 @@ Author: Peter Boyle <pabobyle@ph.ed.ac.uk>
 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
 Author: Peter Boyle <peterboyle@Peters-MacBook-Pro-2.local>
 Author: paboyle <paboyle@ph.ed.ac.uk>
+Author: Guido Cossu <guido.cossu@ed.ac.uk>

    This program is free software; you can redistribute it and/or modify
    it under the terms of the GNU General Public License as published by
@@ -117,49 +118,19 @@ WilsonFermion5D<Impl>::WilsonFermion5D(GaugeField &_Umu,
    
  // Allocate the required comms buffer
  ImportGauge(_Umu);
+  // Build lists of exterior only nodes
+  int LLs = FiveDimGrid._rdimensions[0];
+  int vol4;
+  vol4=FourDimGrid.oSites();
+  Stencil.BuildSurfaceList(LLs,vol4);
+  vol4=FourDimRedBlackGrid.oSites();
+  StencilEven.BuildSurfaceList(LLs,vol4);
+   StencilOdd.BuildSurfaceList(LLs,vol4);
+
+  std::cout << GridLogMessage << " SurfaceLists "<< Stencil.surface_list.size()
+                       <<" " << StencilEven.surface_list.size()<<std::endl;
+
 }
-  /*
-template<class Impl>
-WilsonFermion5D<Impl>::WilsonFermion5D(int simd,GaugeField &_Umu,
-               GridCartesian         &FiveDimGrid,
-               GridRedBlackCartesian &FiveDimRedBlackGrid,
-               GridCartesian         &FourDimGrid,
-               RealD _M5,const ImplParams &p) :
-{
-  int nsimd = Simd::Nsimd();
-
-  // some assertions
-  assert(FiveDimGrid._ndimension==5);
-  assert(FiveDimRedBlackGrid._ndimension==5);
-  assert(FiveDimRedBlackGrid._checker_dim==0); // Checkerboard the s-direction
-  assert(FourDimGrid._ndimension==4);
-
-  // Dimension zero of the five-d is the Ls direction
-  Ls=FiveDimGrid._fdimensions[0];
-  assert(FiveDimGrid._processors[0]         ==1);
-  assert(FiveDimGrid._simd_layout[0]        ==nsimd);
-
-  assert(FiveDimRedBlackGrid._fdimensions[0]==Ls);
-  assert(FiveDimRedBlackGrid._processors[0] ==1);
-  assert(FiveDimRedBlackGrid._simd_layout[0]==nsimd);
-
-  // Other dimensions must match the decomposition of the four-D fields 
-  for(int d=0;d<4;d++){
-    assert(FiveDimRedBlackGrid._fdimensions[d+1]==FourDimGrid._fdimensions[d]);
-    assert(FiveDimRedBlackGrid._processors[d+1] ==FourDimGrid._processors[d]);
-
-    assert(FourDimGrid._simd_layout[d]=1);
-    assert(FiveDimRedBlackGrid._simd_layout[d+1]==1);
-
-    assert(FiveDimGrid._fdimensions[d+1]        ==FourDimGrid._fdimensions[d]);
-    assert(FiveDimGrid._processors[d+1]         ==FourDimGrid._processors[d]);
-    assert(FiveDimGrid._simd_layout[d+1]        ==FourDimGrid._simd_layout[d]);
-  }
-
-  {
-  }
-}  
-  */
     
 template<class Impl>
 void WilsonFermion5D<Impl>::Report(void)
@@ -296,6 +267,8 @@ void WilsonFermion5D<Impl>::DerivInternal(StencilImpl & st,
  DerivCommTime+=usecond();

  Atilde=A;
+  int LLs = B._grid->_rdimensions[0];
+

  DerivComputeTime-=usecond();
  for (int mu = 0; mu < Nd; mu++) {
@@ -325,6 +298,9 @@ void WilsonFermion5D<Impl>::DerivInternal(StencilImpl & st,
        ////////////////////////////
      }
    }
+    ////////////////////////////
+    // spin trace outer product
+    ////////////////////////////
    DerivDhopComputeTime += usecond();
    Impl::InsertForce5D(mat, Btilde, Atilde, mu);
  }
@@ -333,13 +309,14 @@ void WilsonFermion5D<Impl>::DerivInternal(StencilImpl & st,

 template<class Impl>
 void WilsonFermion5D<Impl>::DhopDeriv(GaugeField &mat,
-				      const FermionField &A,
-				      const FermionField &B,
-				      int dag)
+                                      const FermionField &A,
+                                      const FermionField &B,
+                                      int dag)
 {
  conformable(A._grid,FermionGrid());  
  conformable(A._grid,B._grid);
-  conformable(GaugeGrid(),mat._grid);
+
+  //conformable(GaugeGrid(),mat._grid);// this is not general! leaving as a comment

  mat.checkerboard = A.checkerboard;

@@ -348,12 +325,11 @@ void WilsonFermion5D<Impl>::DhopDeriv(GaugeField &mat,

 template<class Impl>
 void WilsonFermion5D<Impl>::DhopDerivEO(GaugeField &mat,
-					const FermionField &A,
-					const FermionField &B,
-					int dag)
+                                        const FermionField &A,
+                                        const FermionField &B,
+                                        int dag)
 {
  conformable(A._grid,FermionRedBlackGrid());
-  conformable(GaugeRedBlackGrid(),mat._grid);
  conformable(A._grid,B._grid);

  assert(B.checkerboard==Odd);
@@ -366,12 +342,11 @@ void WilsonFermion5D<Impl>::DhopDerivEO(GaugeField &mat,

 template<class Impl>
 void WilsonFermion5D<Impl>::DhopDerivOE(GaugeField &mat,
-					const FermionField &A,
-					const FermionField &B,
-					int dag)
+                                        const FermionField &A,
+                                        const FermionField &B,
+                                        int dag)
 {
  conformable(A._grid,FermionRedBlackGrid());
-  conformable(GaugeRedBlackGrid(),mat._grid);
  conformable(A._grid,B._grid);

  assert(B.checkerboard==Even);
@@ -383,8 +358,8 @@ void WilsonFermion5D<Impl>::DhopDerivOE(GaugeField &mat,

 template<class Impl>
 void WilsonFermion5D<Impl>::DhopInternal(StencilImpl & st, LebesgueOrder &lo,
-					 DoubledGaugeField & U,
-					 const FermionField &in, FermionField &out,int dag)
+                                         DoubledGaugeField & U,
+                                         const FermionField &in, FermionField &out,int dag)
 {
  DhopTotalTime-=usecond();
 #ifdef GRID_OMP
@@ -396,6 +371,7 @@ void WilsonFermion5D<Impl>::DhopInternal(StencilImpl & st, LebesgueOrder &lo,
  DhopTotalTime+=usecond();
 }

+
 template<class Impl>
 void WilsonFermion5D<Impl>::DhopInternalOverlappedComms(StencilImpl & st, LebesgueOrder &lo,
 							DoubledGaugeField & U,
@@ -409,12 +385,21 @@ void WilsonFermion5D<Impl>::DhopInternalOverlappedComms(StencilImpl & st, Lebesg

  int LLs = in._grid->_rdimensions[0];
  int len =  U._grid->oSites();
-  
+
  DhopFaceTime-=usecond();
  st.HaloExchangeOptGather(in,compressor);
  DhopFaceTime+=usecond();
  std::vector<std::vector<CommsRequest_t> > reqs;

+  // Rely on async comms; start comms before merge of local data
+  DhopCommTime-=usecond();
+  st.CommunicateBegin(reqs);
+
+  DhopFaceTime-=usecond();
+  st.CommsMergeSHM(compressor);
+  DhopFaceTime+=usecond();
+
+  // Perhaps use omp task and region
 #pragma omp parallel 
  { 
    int nthreads = omp_get_num_threads();
@@ -425,8 +410,6 @@ void WilsonFermion5D<Impl>::DhopInternalOverlappedComms(StencilImpl & st, Lebesg
    int sF = LLs * myoff;

    if ( me == 0 ) {
-      DhopCommTime-=usecond();
-      st.CommunicateBegin(reqs);
      st.CommunicateComplete(reqs);
      DhopCommTime+=usecond();
    } else { 
@@ -439,28 +422,37 @@ void WilsonFermion5D<Impl>::DhopInternalOverlappedComms(StencilImpl & st, Lebesg
  }

  DhopFaceTime-=usecond();
-  st.CommsMerge();
+  st.CommsMerge(compressor);
  DhopFaceTime+=usecond();

-#pragma omp parallel 
-  {
-    int nthreads = omp_get_num_threads();
-    int me = omp_get_thread_num();
-    int myoff, mywork;
-
-    GridThread::GetWork(len,me,mywork,myoff,nthreads);
-    int sF = LLs * myoff;
-
-    // Exterior links in stencil
-    if ( me==0 ) DhopComputeTime2-=usecond();
-    if (dag == DaggerYes) Kernels::DhopSiteDag(st,lo,U,st.CommBuf(),sF,myoff,LLs,mywork,in,out,0,1);
-    else                  Kernels::DhopSite   (st,lo,U,st.CommBuf(),sF,myoff,LLs,mywork,in,out,0,1);
-    if ( me==0 ) DhopComputeTime2+=usecond();
-  }// end parallel region
+  // Load imbalance alert. Should use dynamic schedule OMP for loop
+  // Perhaps create a list of only those sites with face work, and 
+  // load balance process the list.
+  DhopComputeTime2-=usecond();
+  if (dag == DaggerYes) {
+    int sz=st.surface_list.size();
+    parallel_for (int ss = 0; ss < sz; ss++) {
+      int sU = st.surface_list[ss];
+      int sF = LLs * sU;
+      Kernels::DhopSiteDag(st,lo,U,st.CommBuf(),sF,sU,LLs,1,in,out,0,1);
+    }
+  } else {
+    int sz=st.surface_list.size();
+    parallel_for (int ss = 0; ss < sz; ss++) {
+      int sU = st.surface_list[ss];
+      int sF = LLs * sU;
+      Kernels::DhopSite(st,lo,U,st.CommBuf(),sF,sU,LLs,1,in,out,0,1);
+    }
+  }
+  DhopComputeTime2+=usecond();
 #else 
  assert(0);
 #endif
+
 }
+
+
+
 template<class Impl>
 void WilsonFermion5D<Impl>::DhopInternalSerialComms(StencilImpl & st, LebesgueOrder &lo,
 					 DoubledGaugeField & U,
@@ -679,7 +671,6 @@ void WilsonFermion5D<Impl>::MomentumSpacePropagatorHw(FermionField &out,const Fe

 }

-
 FermOpTemplateInstantiate(WilsonFermion5D);
 GparityFermOpTemplateInstantiate(WilsonFermion5D);
  
@@ -33,52 +33,8 @@ directory
 namespace Grid {
 namespace QCD {

-  int WilsonKernelsStatic::Opt   = WilsonKernelsStatic::OptGeneric;
-  int WilsonKernelsStatic::Comms = WilsonKernelsStatic::CommsAndCompute;
-
-#ifdef QPX
-#include <spi/include/kernel/location.h>
-#include <spi/include/l1p/types.h>
-#include <hwi/include/bqc/l1p_mmio.h>
-#include <hwi/include/bqc/A2_inlines.h>
-#endif
-
-void bgq_l1p_optimisation(int mode)
-{
-#ifdef QPX
-#undef L1P_CFG_PF_USR
-#define L1P_CFG_PF_USR  (0x3fde8000108ll)   /*  (64 bit reg, 23 bits wide, user/unpriv) */
-
-  uint64_t cfg_pf_usr;
-  if ( mode ) { 
-    cfg_pf_usr =
-        L1P_CFG_PF_USR_ifetch_depth(0)       
-      | L1P_CFG_PF_USR_ifetch_max_footprint(1)   
-      | L1P_CFG_PF_USR_pf_stream_est_on_dcbt 
-      | L1P_CFG_PF_USR_pf_stream_establish_enable
-      | L1P_CFG_PF_USR_pf_stream_optimistic
-      | L1P_CFG_PF_USR_pf_adaptive_throttle(0xF) ;
-    //    if ( sizeof(Float) == sizeof(double) ) {
-      cfg_pf_usr |=  L1P_CFG_PF_USR_dfetch_depth(2)| L1P_CFG_PF_USR_dfetch_max_footprint(3)   ;
-      //    } else {
-      //      cfg_pf_usr |=  L1P_CFG_PF_USR_dfetch_depth(1)| L1P_CFG_PF_USR_dfetch_max_footprint(2)   ;
-      //    }
-  } else { 
-    cfg_pf_usr = L1P_CFG_PF_USR_dfetch_depth(1)
-      | L1P_CFG_PF_USR_dfetch_max_footprint(2)   
-      | L1P_CFG_PF_USR_ifetch_depth(0)       
-      | L1P_CFG_PF_USR_ifetch_max_footprint(1)   
-      | L1P_CFG_PF_USR_pf_stream_est_on_dcbt 
-      | L1P_CFG_PF_USR_pf_stream_establish_enable
-      | L1P_CFG_PF_USR_pf_stream_optimistic
-      | L1P_CFG_PF_USR_pf_stream_prefetch_enable;
-  }
-  *((uint64_t *)L1P_CFG_PF_USR) = cfg_pf_usr;
-
-#endif
-
-}
-
+int WilsonKernelsStatic::Opt   = WilsonKernelsStatic::OptGeneric;
+int WilsonKernelsStatic::Comms = WilsonKernelsStatic::CommsAndCompute;

 template <class Impl>
 WilsonKernels<Impl>::WilsonKernels(const ImplParams &p) : Base(p){};
@@ -86,12 +42,72 @@ WilsonKernels<Impl>::WilsonKernels(const ImplParams &p) : Base(p){};
 ////////////////////////////////////////////
 // Generic implementation; move to different file?
 ////////////////////////////////////////////
+  
+#define GENERIC_STENCIL_LEG(Dir,spProj,Recon)			\
+  SE = st.GetEntry(ptype, Dir, sF);				\
+  if (SE->_is_local) {						\
+    chi_p = &chi;						\
+    if (SE->_permute) {						\
+      spProj(tmp, in._odata[SE->_offset]);			\
+      permute(chi, tmp, ptype);					\
+    } else {							\
+      spProj(chi, in._odata[SE->_offset]);			\
+    }								\
+  } else {							\
+    chi_p = &buf[SE->_offset];					\
+  }								\
+  Impl::multLink(Uchi, U._odata[sU], *chi_p, Dir, SE, st);	\
+  Recon(result, Uchi);
+  
+#define GENERIC_STENCIL_LEG_INT(Dir,spProj,Recon)		\
+  SE = st.GetEntry(ptype, Dir, sF);				\
+  if (SE->_is_local) {						\
+    chi_p = &chi;						\
+    if (SE->_permute) {						\
+      spProj(tmp, in._odata[SE->_offset]);			\
+      permute(chi, tmp, ptype);					\
+    } else {							\
+      spProj(chi, in._odata[SE->_offset]);			\
+    }								\
+  } else if ( st.same_node[Dir] ) {				\
+      chi_p = &buf[SE->_offset];				\
+  }								\
+  if (SE->_is_local || st.same_node[Dir] ) {			\
+    Impl::multLink(Uchi, U._odata[sU], *chi_p, Dir, SE, st);	\
+    Recon(result, Uchi);					\
+  }

+#define GENERIC_STENCIL_LEG_EXT(Dir,spProj,Recon)		\
+  SE = st.GetEntry(ptype, Dir, sF);				\
+  if ((!SE->_is_local) && (!st.same_node[Dir]) ) {		\
+    chi_p = &buf[SE->_offset];					\
+    Impl::multLink(Uchi, U._odata[sU], *chi_p, Dir, SE, st);	\
+    Recon(result, Uchi);					\
+    nmu++;							\
+  }
+
+#define GENERIC_DHOPDIR_LEG(Dir,spProj,Recon)			\
+  if (gamma == Dir) {						\
+    if (SE->_is_local && SE->_permute) {			\
+      spProj(tmp, in._odata[SE->_offset]);			\
+      permute(chi, tmp, ptype);					\
+    } else if (SE->_is_local) {					\
+      spProj(chi, in._odata[SE->_offset]);			\
+    } else {							\
+      chi = buf[SE->_offset];					\
+    }								\
+    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);	\
+    Recon(result, Uchi);					\
+  }
+
+  ////////////////////////////////////////////////////////////////////
+  // All legs kernels ; comms then compute
+  ////////////////////////////////////////////////////////////////////
 template <class Impl>
 void WilsonKernels<Impl>::GenericDhopSiteDag(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U,
-						     SiteHalfSpinor *buf, int sF,
-						     int sU, const FermionField &in, FermionField &out,
-						     int interior,int exterior) {
+					     SiteHalfSpinor *buf, int sF,
+					     int sU, const FermionField &in, FermionField &out)
+{
  SiteHalfSpinor tmp;
  SiteHalfSpinor chi;
  SiteHalfSpinor *chi_p;
@@ -100,174 +116,22 @@ void WilsonKernels<Impl>::GenericDhopSiteDag(StencilImpl &st, LebesgueOrder &lo,
  StencilEntry *SE;
  int ptype;

-  ///////////////////////////
-  // Xp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Xp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjXp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjXp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Xp, SE, st);
-  spReconXp(result, Uchi);
-
-  ///////////////////////////
-  // Yp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Yp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjYp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjYp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Yp, SE, st);
-  accumReconYp(result, Uchi);
-
-  ///////////////////////////
-  // Zp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Zp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjZp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjZp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Zp, SE, st);
-  accumReconZp(result, Uchi);
-
-  ///////////////////////////
-  // Tp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Tp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjTp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjTp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Tp, SE, st);
-  accumReconTp(result, Uchi);
-
-  ///////////////////////////
-  // Xm
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Xm, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjXm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjXm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Xm, SE, st);
-  accumReconXm(result, Uchi);
-
-  ///////////////////////////
-  // Ym
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Ym, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjYm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjYm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Ym, SE, st);
-  accumReconYm(result, Uchi);
-
-  ///////////////////////////
-  // Zm
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Zm, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjZm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjZm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Zm, SE, st);
-  accumReconZm(result, Uchi);
-
-  ///////////////////////////
-  // Tm
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Tm, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjTm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjTm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Tm, SE, st);
-  accumReconTm(result, Uchi);
-
+  GENERIC_STENCIL_LEG(Xp,spProjXp,spReconXp);
+  GENERIC_STENCIL_LEG(Yp,spProjYp,accumReconYp);
+  GENERIC_STENCIL_LEG(Zp,spProjZp,accumReconZp);
+  GENERIC_STENCIL_LEG(Tp,spProjTp,accumReconTp);
+  GENERIC_STENCIL_LEG(Xm,spProjXm,accumReconXm);
+  GENERIC_STENCIL_LEG(Ym,spProjYm,accumReconYm);
+  GENERIC_STENCIL_LEG(Zm,spProjZm,accumReconZm);
+  GENERIC_STENCIL_LEG(Tm,spProjTm,accumReconTm);
  vstream(out._odata[sF], result);
 };

-// Need controls to do interior, exterior, or both
 template <class Impl>
 void WilsonKernels<Impl>::GenericDhopSite(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U,
-						  SiteHalfSpinor *buf, int sF,
-						  int sU, const FermionField &in, FermionField &out,int interior,int exterior) {
+					  SiteHalfSpinor *buf, int sF,
+					  int sU, const FermionField &in, FermionField &out) 
+{
  SiteHalfSpinor tmp;
  SiteHalfSpinor chi;
  SiteHalfSpinor *chi_p;
@@ -276,168 +140,123 @@ void WilsonKernels<Impl>::GenericDhopSite(StencilImpl &st, LebesgueOrder &lo, Do
  StencilEntry *SE;
  int ptype;

-  ///////////////////////////
-  // Xp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Xm, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjXp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjXp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Xm, SE, st);
-  spReconXp(result, Uchi);
-
-  ///////////////////////////
-  // Yp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Ym, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjYp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjYp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Ym, SE, st);
-  accumReconYp(result, Uchi);
-
-  ///////////////////////////
-  // Zp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Zm, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjZp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjZp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Zm, SE, st);
-  accumReconZp(result, Uchi);
-
-  ///////////////////////////
-  // Tp
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Tm, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjTp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjTp(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Tm, SE, st);
-  accumReconTp(result, Uchi);
-
-  ///////////////////////////
-  // Xm
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Xp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjXm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjXm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Xp, SE, st);
-  accumReconXm(result, Uchi);
-
-  ///////////////////////////
-  // Ym
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Yp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjYm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjYm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Yp, SE, st);
-  accumReconYm(result, Uchi);
-
-  ///////////////////////////
-  // Zm
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Zp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjZm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjZm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Zp, SE, st);
-  accumReconZm(result, Uchi);
-
-  ///////////////////////////
-  // Tm
-  ///////////////////////////
-  SE = st.GetEntry(ptype, Tp, sF);
-
-  if (SE->_is_local) {
-    chi_p = &chi;
-    if (SE->_permute) {
-      spProjTm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else {
-      spProjTm(chi, in._odata[SE->_offset]);
-    }
-  } else {
-    chi_p = &buf[SE->_offset];
-  }
-
-  Impl::multLink(Uchi, U._odata[sU], *chi_p, Tp, SE, st);
-  accumReconTm(result, Uchi);
-
+  GENERIC_STENCIL_LEG(Xm,spProjXp,spReconXp);
+  GENERIC_STENCIL_LEG(Ym,spProjYp,accumReconYp);
+  GENERIC_STENCIL_LEG(Zm,spProjZp,accumReconZp);
+  GENERIC_STENCIL_LEG(Tm,spProjTp,accumReconTp);
+  GENERIC_STENCIL_LEG(Xp,spProjXm,accumReconXm);
+  GENERIC_STENCIL_LEG(Yp,spProjYm,accumReconYm);
+  GENERIC_STENCIL_LEG(Zp,spProjZm,accumReconZm);
+  GENERIC_STENCIL_LEG(Tp,spProjTm,accumReconTm);
  vstream(out._odata[sF], result);
 };
+  ////////////////////////////////////////////////////////////////////
+  // Interior kernels
+  ////////////////////////////////////////////////////////////////////
+template <class Impl>
+void WilsonKernels<Impl>::GenericDhopSiteDagInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U,
+						SiteHalfSpinor *buf, int sF,
+						int sU, const FermionField &in, FermionField &out)
+{
+  SiteHalfSpinor tmp;
+  SiteHalfSpinor chi;
+  SiteHalfSpinor *chi_p;
+  SiteHalfSpinor Uchi;
+  SiteSpinor result;
+  StencilEntry *SE;
+  int ptype;
+
+  result=zero;
+  GENERIC_STENCIL_LEG_INT(Xp,spProjXp,accumReconXp);
+  GENERIC_STENCIL_LEG_INT(Yp,spProjYp,accumReconYp);
+  GENERIC_STENCIL_LEG_INT(Zp,spProjZp,accumReconZp);
+  GENERIC_STENCIL_LEG_INT(Tp,spProjTp,accumReconTp);
+  GENERIC_STENCIL_LEG_INT(Xm,spProjXm,accumReconXm);
+  GENERIC_STENCIL_LEG_INT(Ym,spProjYm,accumReconYm);
+  GENERIC_STENCIL_LEG_INT(Zm,spProjZm,accumReconZm);
+  GENERIC_STENCIL_LEG_INT(Tm,spProjTm,accumReconTm);
+  vstream(out._odata[sF], result);
+};
+
+template <class Impl>
+void WilsonKernels<Impl>::GenericDhopSiteInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U,
+					     SiteHalfSpinor *buf, int sF,
+					     int sU, const FermionField &in, FermionField &out) 
+{
+  SiteHalfSpinor tmp;
+  SiteHalfSpinor chi;
+  SiteHalfSpinor *chi_p;
+  SiteHalfSpinor Uchi;
+  SiteSpinor result;
+  StencilEntry *SE;
+  int ptype;
+  result=zero;
+  GENERIC_STENCIL_LEG_INT(Xm,spProjXp,accumReconXp);
+  GENERIC_STENCIL_LEG_INT(Ym,spProjYp,accumReconYp);
+  GENERIC_STENCIL_LEG_INT(Zm,spProjZp,accumReconZp);
+  GENERIC_STENCIL_LEG_INT(Tm,spProjTp,accumReconTp);
+  GENERIC_STENCIL_LEG_INT(Xp,spProjXm,accumReconXm);
+  GENERIC_STENCIL_LEG_INT(Yp,spProjYm,accumReconYm);
+  GENERIC_STENCIL_LEG_INT(Zp,spProjZm,accumReconZm);
+  GENERIC_STENCIL_LEG_INT(Tp,spProjTm,accumReconTm);
+  vstream(out._odata[sF], result);
+};
+////////////////////////////////////////////////////////////////////
+// Exterior kernels
+////////////////////////////////////////////////////////////////////
+template <class Impl>
+void WilsonKernels<Impl>::GenericDhopSiteDagExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U,
+						SiteHalfSpinor *buf, int sF,
+						int sU, const FermionField &in, FermionField &out)
+{
+  SiteHalfSpinor tmp;
+  SiteHalfSpinor chi;
+  SiteHalfSpinor *chi_p;
+  SiteHalfSpinor Uchi;
+  SiteSpinor result;
+  StencilEntry *SE;
+  int ptype;
+  int nmu=0;
+  result=zero;
+  GENERIC_STENCIL_LEG_EXT(Xp,spProjXp,accumReconXp);
+  GENERIC_STENCIL_LEG_EXT(Yp,spProjYp,accumReconYp);
+  GENERIC_STENCIL_LEG_EXT(Zp,spProjZp,accumReconZp);
+  GENERIC_STENCIL_LEG_EXT(Tp,spProjTp,accumReconTp);
+  GENERIC_STENCIL_LEG_EXT(Xm,spProjXm,accumReconXm);
+  GENERIC_STENCIL_LEG_EXT(Ym,spProjYm,accumReconYm);
+  GENERIC_STENCIL_LEG_EXT(Zm,spProjZm,accumReconZm);
+  GENERIC_STENCIL_LEG_EXT(Tm,spProjTm,accumReconTm);
+  if ( nmu ) { 
+    out._odata[sF] = out._odata[sF] + result; 
+  }
+};
+
+template <class Impl>
+void WilsonKernels<Impl>::GenericDhopSiteExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U,
+					     SiteHalfSpinor *buf, int sF,
+					     int sU, const FermionField &in, FermionField &out) 
+{
+  SiteHalfSpinor tmp;
+  SiteHalfSpinor chi;
+  SiteHalfSpinor *chi_p;
+  SiteHalfSpinor Uchi;
+  SiteSpinor result;
+  StencilEntry *SE;
+  int ptype;
+  int nmu=0;
+  result=zero;
+  GENERIC_STENCIL_LEG_EXT(Xm,spProjXp,accumReconXp);
+  GENERIC_STENCIL_LEG_EXT(Ym,spProjYp,accumReconYp);
+  GENERIC_STENCIL_LEG_EXT(Zm,spProjZp,accumReconZp);
+  GENERIC_STENCIL_LEG_EXT(Tm,spProjTp,accumReconTp);
+  GENERIC_STENCIL_LEG_EXT(Xp,spProjXm,accumReconXm);
+  GENERIC_STENCIL_LEG_EXT(Yp,spProjYm,accumReconYm);
+  GENERIC_STENCIL_LEG_EXT(Zp,spProjZm,accumReconZm);
+  GENERIC_STENCIL_LEG_EXT(Tp,spProjTm,accumReconTm);
+  if ( nmu ) { 
+    out._odata[sF] = out._odata[sF] + result; 
+  }
+};

 template <class Impl>
 void WilsonKernels<Impl>::DhopDir( StencilImpl &st, DoubledGaugeField &U,SiteHalfSpinor *buf, int sF,
@@ -451,119 +270,14 @@ void WilsonKernels<Impl>::DhopDir( StencilImpl &st, DoubledGaugeField &U,SiteHal
  int ptype;

  SE = st.GetEntry(ptype, dir, sF);
-
-  // Xp
-  if (gamma == Xp) {
-    if (SE->_is_local && SE->_permute) {
-      spProjXp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjXp(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconXp(result, Uchi);
-  }
-
-  // Yp
-  if (gamma == Yp) {
-    if (SE->_is_local && SE->_permute) {
-      spProjYp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjYp(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconYp(result, Uchi);
-  }
-
-  // Zp
-  if (gamma == Zp) {
-    if (SE->_is_local && SE->_permute) {
-      spProjZp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjZp(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconZp(result, Uchi);
-  }
-
-  // Tp
-  if (gamma == Tp) {
-    if (SE->_is_local && SE->_permute) {
-      spProjTp(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjTp(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconTp(result, Uchi);
-  }
-
-  // Xm
-  if (gamma == Xm) {
-    if (SE->_is_local && SE->_permute) {
-      spProjXm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjXm(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconXm(result, Uchi);
-  }
-
-  // Ym
-  if (gamma == Ym) {
-    if (SE->_is_local && SE->_permute) {
-      spProjYm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjYm(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconYm(result, Uchi);
-  }
-
-  // Zm
-  if (gamma == Zm) {
-    if (SE->_is_local && SE->_permute) {
-      spProjZm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjZm(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconZm(result, Uchi);
-  }
-
-  // Tm
-  if (gamma == Tm) {
-    if (SE->_is_local && SE->_permute) {
-      spProjTm(tmp, in._odata[SE->_offset]);
-      permute(chi, tmp, ptype);
-    } else if (SE->_is_local) {
-      spProjTm(chi, in._odata[SE->_offset]);
-    } else {
-      chi = buf[SE->_offset];
-    }
-    Impl::multLink(Uchi, U._odata[sU], chi, dir, SE, st);
-    spReconTm(result, Uchi);
-  }
-
+  GENERIC_DHOPDIR_LEG(Xp,spProjXp,spReconXp);
+  GENERIC_DHOPDIR_LEG(Yp,spProjYp,spReconYp);
+  GENERIC_DHOPDIR_LEG(Zp,spProjZp,spReconZp);
+  GENERIC_DHOPDIR_LEG(Tp,spProjTp,spReconTp);
+  GENERIC_DHOPDIR_LEG(Xm,spProjXm,spReconXm);
+  GENERIC_DHOPDIR_LEG(Ym,spProjYm,spReconYm);
+  GENERIC_DHOPDIR_LEG(Zm,spProjZm,spReconZm);
+  GENERIC_DHOPDIR_LEG(Tm,spProjTm,spReconTm);
  vstream(out._odata[sF], result);
 }

@@ -34,8 +34,6 @@ directory
 namespace Grid {
 namespace QCD {

-void bgq_l1p_optimisation(int mode);
-
  ////////////////////////////////////////////////////////////////////////////////////////////////////////////////
  // Helper routines that implement Wilson stencil for a single site.
  // Common to both the WilsonFermion and WilsonFermion5D
@@ -44,9 +42,8 @@ class WilsonKernelsStatic {
 public:
  enum { OptGeneric, OptHandUnroll, OptInlineAsm };
  enum { CommsAndCompute, CommsThenCompute };
-  // S-direction is INNERMOST and takes no part in the parity.
-  static int Opt;  // these are a temporary hack
-  static int Comms;  // these are a temporary hack
+  static int Opt;  
+  static int Comms;
 };
 
 template<class Impl> class WilsonKernels : public FermionOperator<Impl> , public WilsonKernelsStatic { 
@@ -66,7 +63,7 @@ public:
    switch(Opt) {
 #if defined(AVX512) || defined (QPX)
    case OptInlineAsm:
-      if(interior&&exterior) WilsonKernels<Impl>::AsmDhopSite(st,lo,U,buf,sF,sU,Ls,Ns,in,out);
+      if(interior&&exterior) WilsonKernels<Impl>::AsmDhopSite   (st,lo,U,buf,sF,sU,Ls,Ns,in,out);
      else if (interior)     WilsonKernels<Impl>::AsmDhopSiteInt(st,lo,U,buf,sF,sU,Ls,Ns,in,out);
      else if (exterior)     WilsonKernels<Impl>::AsmDhopSiteExt(st,lo,U,buf,sF,sU,Ls,Ns,in,out);
      else assert(0);
@@ -75,7 +72,9 @@ public:
    case OptHandUnroll:
      for (int site = 0; site < Ns; site++) {
 	for (int s = 0; s < Ls; s++) {
-	  if( exterior) WilsonKernels<Impl>::HandDhopSite(st,lo,U,buf,sF,sU,in,out,interior,exterior);
+	  if(interior&&exterior) WilsonKernels<Impl>::HandDhopSite(st,lo,U,buf,sF,sU,in,out);
+	  else if (interior)     WilsonKernels<Impl>::HandDhopSiteInt(st,lo,U,buf,sF,sU,in,out);
+	  else if (exterior)     WilsonKernels<Impl>::HandDhopSiteExt(st,lo,U,buf,sF,sU,in,out);
 	  sF++;
 	}
 	sU++;
@@ -84,7 +83,10 @@ public:
    case OptGeneric:
      for (int site = 0; site < Ns; site++) {
 	for (int s = 0; s < Ls; s++) {
-	  if( exterior) WilsonKernels<Impl>::GenericDhopSite(st,lo,U,buf,sF,sU,in,out,interior,exterior);
+	  if(interior&&exterior) WilsonKernels<Impl>::GenericDhopSite(st,lo,U,buf,sF,sU,in,out);
+	  else if (interior)     WilsonKernels<Impl>::GenericDhopSiteInt(st,lo,U,buf,sF,sU,in,out);
+	  else if (exterior)     WilsonKernels<Impl>::GenericDhopSiteExt(st,lo,U,buf,sF,sU,in,out);
+	  else assert(0);
 	  sF++;
 	}
 	sU++;
@@ -99,11 +101,14 @@ public:
  template <bool EnableBool = true>
  typename std::enable_if<(Impl::Dimension != 3 || (Impl::Dimension == 3 && Nc != 3)) && EnableBool, void>::type
  DhopSite(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-		   int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out,int interior=1,int exterior=1 ) {
+	   int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out,int interior=1,int exterior=1 ) {
    // no kernel choice  
    for (int site = 0; site < Ns; site++) {
      for (int s = 0; s < Ls; s++) {
-	if( exterior) WilsonKernels<Impl>::GenericDhopSite(st, lo, U, buf, sF, sU, in, out,interior,exterior);
+	if(interior&&exterior) WilsonKernels<Impl>::GenericDhopSite(st,lo,U,buf,sF,sU,in,out);
+	else if (interior)     WilsonKernels<Impl>::GenericDhopSiteInt(st,lo,U,buf,sF,sU,in,out);
+	else if (exterior)     WilsonKernels<Impl>::GenericDhopSiteExt(st,lo,U,buf,sF,sU,in,out);
+	else assert(0);
 	sF++;
      }
      sU++;
@@ -113,13 +118,13 @@ public:
  template <bool EnableBool = true>
  typename std::enable_if<Impl::Dimension == 3 && Nc == 3 && EnableBool,void>::type
  DhopSiteDag(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-		      int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out,int interior=1,int exterior=1) {
-
+	      int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out,int interior=1,int exterior=1) 
+{
    bgq_l1p_optimisation(1);
    switch(Opt) {
 #if defined(AVX512) || defined (QPX)
    case OptInlineAsm:
-      if(interior&&exterior) WilsonKernels<Impl>::AsmDhopSiteDag(st,lo,U,buf,sF,sU,Ls,Ns,in,out);
+      if(interior&&exterior) WilsonKernels<Impl>::AsmDhopSiteDag   (st,lo,U,buf,sF,sU,Ls,Ns,in,out);
      else if (interior)     WilsonKernels<Impl>::AsmDhopSiteDagInt(st,lo,U,buf,sF,sU,Ls,Ns,in,out);
      else if (exterior)     WilsonKernels<Impl>::AsmDhopSiteDagExt(st,lo,U,buf,sF,sU,Ls,Ns,in,out);
      else assert(0);
@@ -128,7 +133,10 @@ public:
    case OptHandUnroll:
      for (int site = 0; site < Ns; site++) {
 	for (int s = 0; s < Ls; s++) {
-	  if( exterior) WilsonKernels<Impl>::HandDhopSiteDag(st,lo,U,buf,sF,sU,in,out,interior,exterior);
+	  if(interior&&exterior) WilsonKernels<Impl>::HandDhopSiteDag(st,lo,U,buf,sF,sU,in,out);
+	  else if (interior)     WilsonKernels<Impl>::HandDhopSiteDagInt(st,lo,U,buf,sF,sU,in,out);
+	  else if (exterior)     WilsonKernels<Impl>::HandDhopSiteDagExt(st,lo,U,buf,sF,sU,in,out);
+	  else assert(0);
 	  sF++;
 	}
 	sU++;
@@ -137,7 +145,10 @@ public:
    case OptGeneric:
      for (int site = 0; site < Ns; site++) {
 	for (int s = 0; s < Ls; s++) {
-	  if( exterior) WilsonKernels<Impl>::GenericDhopSiteDag(st,lo,U,buf,sF,sU,in,out,interior,exterior);
+	  if(interior&&exterior) WilsonKernels<Impl>::GenericDhopSiteDag(st,lo,U,buf,sF,sU,in,out);
+	  else if (interior)     WilsonKernels<Impl>::GenericDhopSiteDagInt(st,lo,U,buf,sF,sU,in,out);
+	  else if (exterior)     WilsonKernels<Impl>::GenericDhopSiteDagExt(st,lo,U,buf,sF,sU,in,out);
+	  else assert(0);
 	  sF++;
 	}
 	sU++;
@@ -156,7 +167,10 @@ public:

    for (int site = 0; site < Ns; site++) {
      for (int s = 0; s < Ls; s++) {
-	if( exterior) WilsonKernels<Impl>::GenericDhopSiteDag(st,lo,U,buf,sF,sU,in,out,interior,exterior);
+	if(interior&&exterior) WilsonKernels<Impl>::GenericDhopSiteDag(st,lo,U,buf,sF,sU,in,out);
+	else if (interior)     WilsonKernels<Impl>::GenericDhopSiteDagInt(st,lo,U,buf,sF,sU,in,out);
+	else if (exterior)     WilsonKernels<Impl>::GenericDhopSiteDagExt(st,lo,U,buf,sF,sU,in,out);
+	else assert(0);
 	sF++;
      }
      sU++;
@@ -169,36 +183,60 @@ public:
 private:
     // Specialised variants
  void GenericDhopSite(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			       int sF, int sU, const FermionField &in, FermionField &out,int interior,int exterior);
+		       int sF, int sU, const FermionField &in, FermionField &out);
      
  void GenericDhopSiteDag(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-				  int sF, int sU, const FermionField &in, FermionField &out,int interior,int exterior);
+			  int sF, int sU, const FermionField &in, FermionField &out);
+
+  void GenericDhopSiteInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+			  int sF, int sU, const FermionField &in, FermionField &out);
+      
+  void GenericDhopSiteDagInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+			     int sF, int sU, const FermionField &in, FermionField &out);
+
+  void GenericDhopSiteExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+			  int sF, int sU, const FermionField &in, FermionField &out);
+      
+  void GenericDhopSiteDagExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+			     int sF, int sU, const FermionField &in, FermionField &out);

  void AsmDhopSite(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			   int sF, int sU, int Ls, int Ns, const FermionField &in,FermionField &out);
+		   int sF, int sU, int Ls, int Ns, const FermionField &in,FermionField &out);

  void AsmDhopSiteDag(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			      int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out);
+		      int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out);

  void AsmDhopSiteInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			   int sF, int sU, int Ls, int Ns, const FermionField &in,FermionField &out);
+		      int sF, int sU, int Ls, int Ns, const FermionField &in,FermionField &out);

  void AsmDhopSiteDagInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			      int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out);
+			 int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out);

  void AsmDhopSiteExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			      int sF, int sU, int Ls, int Ns, const FermionField &in,FermionField &out);
+		      int sF, int sU, int Ls, int Ns, const FermionField &in,FermionField &out);

  void AsmDhopSiteDagExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-				 int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out);
+			 int sF, int sU, int Ls, int Ns, const FermionField &in, FermionField &out);


  void HandDhopSite(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			    int sF, int sU, const FermionField &in, FermionField &out,int interior,int exterior);
+		    int sF, int sU, const FermionField &in, FermionField &out);

  void HandDhopSiteDag(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
-			       int sF, int sU, const FermionField &in, FermionField &out,int interior,int exterior);
+		       int sF, int sU, const FermionField &in, FermionField &out);
      
+  void HandDhopSiteInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+		       int sF, int sU, const FermionField &in, FermionField &out);
+  
+  void HandDhopSiteDagInt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+			  int sF, int sU, const FermionField &in, FermionField &out);
+  
+  void HandDhopSiteExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+		       int sF, int sU, const FermionField &in, FermionField &out);
+  
+  void HandDhopSiteDagExt(StencilImpl &st, LebesgueOrder &lo, DoubledGaugeField &U, SiteHalfSpinor * buf,
+			  int sF, int sU, const FermionField &in, FermionField &out);
+  
 public:

  WilsonKernels(const ImplParams &p = ImplParams());
@@ -112,5 +112,16 @@ INSTANTIATE_ASM(DomainWallVec5dImplD);
 INSTANTIATE_ASM(ZDomainWallVec5dImplF);
 INSTANTIATE_ASM(ZDomainWallVec5dImplD);

+INSTANTIATE_ASM(WilsonImplFH);
+INSTANTIATE_ASM(WilsonImplDF);
+INSTANTIATE_ASM(ZWilsonImplFH);
+INSTANTIATE_ASM(ZWilsonImplDF);
+INSTANTIATE_ASM(GparityWilsonImplFH);
+INSTANTIATE_ASM(GparityWilsonImplDF);
+INSTANTIATE_ASM(DomainWallVec5dImplFH);
+INSTANTIATE_ASM(DomainWallVec5dImplDF);
+INSTANTIATE_ASM(ZDomainWallVec5dImplFH);
+INSTANTIATE_ASM(ZDomainWallVec5dImplDF);
+
 }}

@@ -71,6 +71,16 @@ WilsonKernels<ZWilsonImplF>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,Doub
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplFH>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
+template<> void 
+WilsonKernels<ZWilsonImplFH>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -84,6 +94,16 @@ WilsonKernels<ZWilsonImplF>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,D
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplFH>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
+template<> void 
+WilsonKernels<ZWilsonImplFH>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+

 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
@@ -97,6 +117,16 @@ template<> void
 WilsonKernels<ZWilsonImplF>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
+template<> void 
+WilsonKernels<WilsonImplFH>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
+template<> void 
+WilsonKernels<ZWilsonImplFH>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
      
 /////////////////////////////////////////////////////////////////
 // XYZT vectorised, dag Kernel, single
@@ -115,6 +145,16 @@ WilsonKernels<ZWilsonImplF>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,D
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplFH>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
+template<> void 
+WilsonKernels<ZWilsonImplFH>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -128,6 +168,16 @@ WilsonKernels<ZWilsonImplF>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & l
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplFH>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
+template<> void 
+WilsonKernels<ZWilsonImplFH>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -141,6 +191,16 @@ WilsonKernels<ZWilsonImplF>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & l
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>
 				    
+template<> void 
+WilsonKernels<WilsonImplFH>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+				    
+template<> void 
+WilsonKernels<ZWilsonImplFH>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+				    
 #undef MAYBEPERM
 #undef MULT_2SPIN
 #define MAYBEPERM(A,B) 
@@ -162,6 +222,15 @@ WilsonKernels<ZDomainWallVec5dImplF>::AsmDhopSite(StencilImpl &st,LebesgueOrder
 							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplFH>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplFH>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -174,6 +243,15 @@ WilsonKernels<ZDomainWallVec5dImplF>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrd
 							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplFH>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplFH>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -189,6 +267,16 @@ WilsonKernels<ZDomainWallVec5dImplF>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrd
 							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>
 				    
+template<> void 
+WilsonKernels<DomainWallVec5dImplFH>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+				    
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplFH>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+				    
 /////////////////////////////////////////////////////////////////
 // Ls vectorised, dag Kernel, single
 /////////////////////////////////////////////////////////////////
@@ -205,6 +293,15 @@ WilsonKernels<ZDomainWallVec5dImplF>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrd
 							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplFH>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplFH>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -217,6 +314,15 @@ WilsonKernels<ZDomainWallVec5dImplF>::AsmDhopSiteDagInt(StencilImpl &st,Lebesgue
 							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplFH>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplFH>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -229,6 +335,15 @@ WilsonKernels<ZDomainWallVec5dImplF>::AsmDhopSiteDagExt(StencilImpl &st,Lebesgue
 							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplFH>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplFH>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef COMPLEX_SIGNS
 #undef MAYBEPERM
 #undef MULT_2SPIN
@@ -269,6 +384,15 @@ WilsonKernels<ZWilsonImplD>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,Doub
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplDF>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZWilsonImplDF>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -281,6 +405,15 @@ WilsonKernels<ZWilsonImplD>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,D
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplDF>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZWilsonImplDF>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -293,6 +426,15 @@ WilsonKernels<ZWilsonImplD>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,D
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>
      
+template<> void 
+WilsonKernels<WilsonImplDF>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZWilsonImplDF>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+      
 /////////////////////////////////////////////////////////////////
 // XYZT vectorised, dag Kernel, single
 /////////////////////////////////////////////////////////////////
@@ -309,6 +451,15 @@ WilsonKernels<ZWilsonImplD>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,D
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplDF>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZWilsonImplDF>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -321,6 +472,15 @@ WilsonKernels<ZWilsonImplD>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & l
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<WilsonImplDF>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZWilsonImplDF>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -333,6 +493,15 @@ WilsonKernels<ZWilsonImplD>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & l
 						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>
 				    
+template<> void 
+WilsonKernels<WilsonImplDF>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZWilsonImplDF>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+						int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+				    
 #undef MAYBEPERM
 #undef MULT_2SPIN
 #define MAYBEPERM(A,B) 
@@ -354,6 +523,15 @@ WilsonKernels<ZDomainWallVec5dImplD>::AsmDhopSite(StencilImpl &st,LebesgueOrder
 							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplDF>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplDF>::AsmDhopSite(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -366,6 +544,15 @@ WilsonKernels<ZDomainWallVec5dImplD>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrd
 							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplDF>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplDF>::AsmDhopSiteInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -380,6 +567,15 @@ WilsonKernels<ZDomainWallVec5dImplD>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrd
 							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>
 				    
+template<> void 
+WilsonKernels<DomainWallVec5dImplDF>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplDF>::AsmDhopSiteExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U, SiteHalfSpinor *buf,
+							 int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+				    
 /////////////////////////////////////////////////////////////////
 // Ls vectorised, dag Kernel, single
 /////////////////////////////////////////////////////////////////
@@ -396,6 +592,15 @@ WilsonKernels<ZDomainWallVec5dImplD>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrd
 							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplDF>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplDF>::AsmDhopSiteDag(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #define INTERIOR
 #undef EXTERIOR
@@ -408,6 +613,15 @@ WilsonKernels<ZDomainWallVec5dImplD>::AsmDhopSiteDagInt(StencilImpl &st,Lebesgue
 							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplDF>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplDF>::AsmDhopSiteDagInt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef INTERIOR_AND_EXTERIOR
 #undef INTERIOR
 #define EXTERIOR
@@ -420,6 +634,15 @@ WilsonKernels<ZDomainWallVec5dImplD>::AsmDhopSiteDagExt(StencilImpl &st,Lebesgue
 							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
 #include <qcd/action/fermion/WilsonKernelsAsmBody.h>

+template<> void 
+WilsonKernels<DomainWallVec5dImplDF>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+template<> void 
+WilsonKernels<ZDomainWallVec5dImplDF>::AsmDhopSiteDagExt(StencilImpl &st,LebesgueOrder & lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+							    int ss,int ssU,int Ls,int Ns,const FermionField &in, FermionField &out)
+#include <qcd/action/fermion/WilsonKernelsAsmBody.h>
+
 #undef COMPLEX_SIGNS
 #undef MAYBEPERM
 #undef MULT_2SPIN
@@ -39,24 +39,26 @@
 ////////////////////////////////////////////////////////////////////////////////
 #ifdef INTERIOR_AND_EXTERIOR

-#define ZERO_NMU(A) 
-#define INTERIOR_BLOCK_XP(a,b,PERMUTE_DIR,PROJMEM,RECON) INTERIOR_BLOCK(a,b,PERMUTE_DIR,PROJMEM,RECON)
-#define EXTERIOR_BLOCK_XP(a,b,RECON) EXTERIOR_BLOCK(a,b,RECON)
+#define ASM_LEG(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON)			\
+      basep = st.GetPFInfo(nent,plocal); nent++;			\
+      if ( local ) {							\
+	LOAD64(%r10,isigns);						\
+	PROJ(base);							\
+	MAYBEPERM(PERMUTE_DIR,perm);					\
+      } else {								\
+	LOAD_CHI(base);							\
+      }									\
+      base = st.GetInfo(ptype,local,perm,NxtDir,ent,plocal); ent++;	\
+      PREFETCH_CHIMU(base);						\
+      MULT_2SPIN_DIR_PF(Dir,basep);					\
+      LOAD64(%r10,isigns);						\
+      RECON;								\

-#define INTERIOR_BLOCK(a,b,PERMUTE_DIR,PROJMEM,RECON)	\
-  LOAD64(%r10,isigns);                                  \
-  PROJMEM(base);                                        \
-  MAYBEPERM(PERMUTE_DIR,perm);                                  
-
-#define EXTERIOR_BLOCK(a,b,RECON)             \
-  LOAD_CHI(base);
-
-#define COMMON_BLOCK(a,b,RECON)               \
-  base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;     \
-  PREFETCH_CHIMU(base);                                         \
-  MULT_2SPIN_DIR_PF(a,basep);					\
-  LOAD64(%r10,isigns);                                          \
-  RECON;                                                        
+#define ASM_LEG_XP(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON)			\
+  base = st.GetInfo(ptype,local,perm,Dir,ent,plocal); ent++;		\
+  PF_GAUGE(Xp);								\
+  PREFETCH1_CHIMU(base);						\
+  ASM_LEG(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON) 

 #define RESULT(base,basep) SAVE_RESULT(base,basep);

@@ -67,62 +69,62 @@
 ////////////////////////////////////////////////////////////////////////////////
 #ifdef INTERIOR

-#define COMMON_BLOCK(a,b,RECON)       
-#define ZERO_NMU(A) 
+#define ASM_LEG(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON)			\
+      basep = st.GetPFInfo(nent,plocal); nent++;			\
+      if ( local ) {							\
+	LOAD64(%r10,isigns);						\
+	PROJ(base);							\
+	MAYBEPERM(PERMUTE_DIR,perm);					\
+      }else if ( st.same_node[Dir] ) {LOAD_CHI(base);}			\
+      if ( local || st.same_node[Dir] ) {				\
+	MULT_2SPIN_DIR_PF(Dir,basep);					\
+	LOAD64(%r10,isigns);						\
+	RECON;								\
+      }									\
+      base = st.GetInfo(ptype,local,perm,NxtDir,ent,plocal); ent++;	\
+      PREFETCH_CHIMU(base);						\

-// No accumulate for DIR0
-#define EXTERIOR_BLOCK_XP(a,b,RECON)				\
-  ZERO_PSI;							\
-  base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;	
-
-#define EXTERIOR_BLOCK(a,b,RECON)  \
-  base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;     
-
-#define INTERIOR_BLOCK_XP(a,b,PERMUTE_DIR,PROJMEM,RECON) INTERIOR_BLOCK(a,b,PERMUTE_DIR,PROJMEM,RECON)
-
-#define INTERIOR_BLOCK(a,b,PERMUTE_DIR,PROJMEM,RECON)		\
-  LOAD64(%r10,isigns);						\
-  PROJMEM(base);                                                \
-  MAYBEPERM(PERMUTE_DIR,perm);                                  \
-  base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;	\
-  PREFETCH_CHIMU(base);						\
-  MULT_2SPIN_DIR_PF(a,basep);					\
-  LOAD64(%r10,isigns);                                          \
-  RECON;                                                        
+#define ASM_LEG_XP(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON)			\
+  base = st.GetInfo(ptype,local,perm,Dir,ent,plocal); ent++;		\
+  PF_GAUGE(Xp);								\
+  PREFETCH1_CHIMU(base);						\
+  { ZERO_PSI; }								\
+  ASM_LEG(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON) 

 #define RESULT(base,basep) SAVE_RESULT(base,basep);

 #endif
-
 ////////////////////////////////////////////////////////////////////////////////
 // Post comms kernel
 ////////////////////////////////////////////////////////////////////////////////
 #ifdef EXTERIOR

-#define ZERO_NMU(A) nmu=0;

-#define INTERIOR_BLOCK_XP(a,b,PERMUTE_DIR,PROJMEM,RECON) \
-  ZERO_PSI;   base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;		
+#define ASM_LEG(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON)			\
+  base = st.GetInfo(ptype,local,perm,Dir,ent,plocal); ent++;		\
+  if((!local)&&(!st.same_node[Dir]) ) {					\
+    LOAD_CHI(base);							\
+    MULT_2SPIN_DIR_PF(Dir,base);					\
+    LOAD64(%r10,isigns);						\
+    RECON;								\
+    nmu++;								\
+  }									

-#define EXTERIOR_BLOCK_XP(a,b,RECON) EXTERIOR_BLOCK(a,b,RECON)
+#define ASM_LEG_XP(Dir,NxtDir,PERMUTE_DIR,PROJ,RECON)			\
+  nmu=0;								\
+  { ZERO_PSI;}								\
+  base = st.GetInfo(ptype,local,perm,Dir,ent,plocal); ent++;		\
+  if((!local)&&(!st.same_node[Dir]) ) {					\
+    LOAD_CHI(base);							\
+    MULT_2SPIN_DIR_PF(Dir,base);					\
+    LOAD64(%r10,isigns);						\
+    RECON;								\
+    nmu++;								\
+  }

-#define INTERIOR_BLOCK(a,b,PERMUTE_DIR,PROJMEM,RECON)			\
-  base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;		
-
-#define EXTERIOR_BLOCK(a,b,RECON)				\
-    nmu++;							\
-    LOAD_CHI(base);						\
-    MULT_2SPIN_DIR_PF(a,base);					\
-    base = st.GetInfo(ptype,local,perm,b,ent,plocal); ent++;	\
-    LOAD64(%r10,isigns);					\
-    RECON;                                                        
-
-#define COMMON_BLOCK(a,b,RECON)			
-
-#define RESULT(base,basep) if (nmu){  ADD_RESULT(base,base);}
+#define RESULT(base,basep) if (nmu){ ADD_RESULT(base,base);}

 #endif
-
 {
  int nmu;
  int local,perm, ptype;
@@ -134,11 +136,15 @@
  MASK_REGS;
  int nmax=U._grid->oSites();
  for(int site=0;site<Ns;site++) {
+#ifndef EXTERIOR
    int sU =lo.Reorder(ssU);
    int ssn=ssU+1;     if(ssn>=nmax) ssn=0;
    int sUn=lo.Reorder(ssn);
-#ifndef EXTERIOR
    LOCK_GAUGE(0);
+#else
+    int sU =ssU;
+    int ssn=ssU+1;     if(ssn>=nmax) ssn=0;
+    int sUn=ssn;
 #endif
    for(int s=0;s<Ls;s++) {
      ss =sU*Ls+s;
@@ -146,93 +152,20 @@
      int  ent=ss*8;// 2*Ndim
      int nent=ssn*8;

-      ZERO_NMU(0);
-      base  = st.GetInfo(ptype,local,perm,Xp,ent,plocal); ent++;
-#ifndef EXTERIOR
-      PF_GAUGE(Xp); 
-      PREFETCH1_CHIMU(base);
-#endif
-      ////////////////////////////////
-      // Xp
-      ////////////////////////////////
-      basep = st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK_XP(Xp,Yp,PERMUTE_DIR3,DIR0_PROJMEM,DIR0_RECON);
-      } else { 
-	EXTERIOR_BLOCK_XP(Xp,Yp,DIR0_RECON);
-      }
-      COMMON_BLOCK(Xp,Yp,DIR0_RECON);
-      ////////////////////////////////
-      // Yp
-      ////////////////////////////////
-      basep = st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Yp,Zp,PERMUTE_DIR2,DIR1_PROJMEM,DIR1_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Yp,Zp,DIR1_RECON);
-      }
-      COMMON_BLOCK(Yp,Zp,DIR1_RECON);
-      ////////////////////////////////
-      // Zp
-      ////////////////////////////////
-      basep = st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Zp,Tp,PERMUTE_DIR1,DIR2_PROJMEM,DIR2_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Zp,Tp,DIR2_RECON);
-      }
-      COMMON_BLOCK(Zp,Tp,DIR2_RECON);
-      ////////////////////////////////
-      // Tp
-      ////////////////////////////////
-      basep = st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Tp,Xm,PERMUTE_DIR0,DIR3_PROJMEM,DIR3_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Tp,Xm,DIR3_RECON);
-      }
-      COMMON_BLOCK(Tp,Xm,DIR3_RECON);
-      ////////////////////////////////
-      // Xm
-      ////////////////////////////////
-      //  basep= st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Xm,Ym,PERMUTE_DIR3,DIR4_PROJMEM,DIR4_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Xm,Ym,DIR4_RECON);
-      }
-      COMMON_BLOCK(Xm,Ym,DIR4_RECON);
-      ////////////////////////////////
-      // Ym
-      ////////////////////////////////
-      basep= st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Ym,Zm,PERMUTE_DIR2,DIR5_PROJMEM,DIR5_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Ym,Zm,DIR5_RECON);
-      }
-      COMMON_BLOCK(Ym,Zm,DIR5_RECON);
-      ////////////////////////////////
-      // Zm
-      ////////////////////////////////
-      basep= st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Zm,Tm,PERMUTE_DIR1,DIR6_PROJMEM,DIR6_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Zm,Tm,DIR6_RECON);
-      }
-      COMMON_BLOCK(Zm,Tm,DIR6_RECON);
-      ////////////////////////////////
-      // Tm
-      ////////////////////////////////
-      basep= st.GetPFInfo(nent,plocal); nent++;
-      if ( local ) {
-	INTERIOR_BLOCK(Tm,Xp,PERMUTE_DIR0,DIR7_PROJMEM,DIR7_RECON);
-      } else { 
-	EXTERIOR_BLOCK(Tm,Xp,DIR7_RECON);
-      }
-      COMMON_BLOCK(Tm,Xp,DIR7_RECON);
+   ASM_LEG_XP(Xp,Yp,PERMUTE_DIR3,DIR0_PROJMEM,DIR0_RECON);
+      ASM_LEG(Yp,Zp,PERMUTE_DIR2,DIR1_PROJMEM,DIR1_RECON);
+      ASM_LEG(Zp,Tp,PERMUTE_DIR1,DIR2_PROJMEM,DIR2_RECON);
+      ASM_LEG(Tp,Xm,PERMUTE_DIR0,DIR3_PROJMEM,DIR3_RECON);

+      ASM_LEG(Xm,Ym,PERMUTE_DIR3,DIR4_PROJMEM,DIR4_RECON);
+      ASM_LEG(Ym,Zm,PERMUTE_DIR2,DIR5_PROJMEM,DIR5_RECON);
+      ASM_LEG(Zm,Tm,PERMUTE_DIR1,DIR6_PROJMEM,DIR6_RECON);
+      ASM_LEG(Tm,Xp,PERMUTE_DIR0,DIR7_PROJMEM,DIR7_RECON);
+
+#ifdef EXTERIOR
+      if (nmu==0) break;
+      //      if (nmu!=0) std::cout << "EXT "<<sU<<std::endl;
+#endif
      base = (uint64_t) &out._odata[ss];
      basep= st.GetPFInfo(nent,plocal); nent++;
      RESULT(base,basep);
@@ -258,10 +191,6 @@
 #undef DIR5_RECON
 #undef DIR6_RECON
 #undef DIR7_RECON
-#undef EXTERIOR_BLOCK
-#undef INTERIOR_BLOCK
-#undef EXTERIOR_BLOCK_XP
-#undef INTERIOR_BLOCK_XP
-#undef COMMON_BLOCK
-#undef ZERO_NMU
+#undef ASM_LEG
+#undef ASM_LEG_XP
 #undef RESULT
@@ -31,7 +31,7 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
 #define REGISTER

 #define LOAD_CHIMU \
-  const SiteSpinor & ref (in._odata[offset]);	\
+  {const SiteSpinor & ref (in._odata[offset]);	\
    Chimu_00=ref()(0)(0);\
    Chimu_01=ref()(0)(1);\
    Chimu_02=ref()(0)(2);\
@@ -43,20 +43,20 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
    Chimu_22=ref()(2)(2);\
    Chimu_30=ref()(3)(0);\
    Chimu_31=ref()(3)(1);\
-    Chimu_32=ref()(3)(2);
+    Chimu_32=ref()(3)(2);}

 #define LOAD_CHI\
-  const SiteHalfSpinor &ref(buf[offset]);	\
+  {const SiteHalfSpinor &ref(buf[offset]);	\
    Chi_00 = ref()(0)(0);\
    Chi_01 = ref()(0)(1);\
    Chi_02 = ref()(0)(2);\
    Chi_10 = ref()(1)(0);\
    Chi_11 = ref()(1)(1);\
-    Chi_12 = ref()(1)(2);
+    Chi_12 = ref()(1)(2);}

 // To splat or not to splat depends on the implementation
 #define MULT_2SPIN(A)\
-   auto & ref(U._odata[sU](A));	\
+  {auto & ref(U._odata[sU](A));			\
   Impl::loadLinkElement(U_00,ref()(0,0));	\
   Impl::loadLinkElement(U_10,ref()(1,0));	\
   Impl::loadLinkElement(U_20,ref()(2,0));	\
@@ -83,7 +83,7 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
    UChi_01+= U_10*Chi_02;\
    UChi_11+= U_10*Chi_12;\
    UChi_02+= U_20*Chi_02;\
-    UChi_12+= U_20*Chi_12;
+    UChi_12+= U_20*Chi_12;}


 #define PERMUTE_DIR(dir)			\
@@ -307,55 +307,132 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
  result_31-= UChi_11;	\
  result_32-= UChi_12;

-namespace Grid {
-namespace QCD {
+#define HAND_STENCIL_LEG(PROJ,PERM,DIR,RECON)	\
+  SE=st.GetEntry(ptype,DIR,ss);			\
+  offset = SE->_offset;				\
+  local  = SE->_is_local;			\
+  perm   = SE->_permute;			\
+  if ( local ) {				\
+    LOAD_CHIMU;					\
+    PROJ;					\
+    if ( perm) {				\
+      PERMUTE_DIR(PERM);			\
+    }						\
+  } else {					\
+    LOAD_CHI;					\
+  }						\
+  MULT_2SPIN(DIR);				\
+  RECON;					
+
+#define HAND_STENCIL_LEG_INT(PROJ,PERM,DIR,RECON)	\
+  SE=st.GetEntry(ptype,DIR,ss);			\
+  offset = SE->_offset;				\
+  local  = SE->_is_local;			\
+  perm   = SE->_permute;			\
+  if ( local ) {				\
+    LOAD_CHIMU;					\
+    PROJ;					\
+    if ( perm) {				\
+      PERMUTE_DIR(PERM);			\
+    }						\
+  } else if ( st.same_node[DIR] ) {		\
+    LOAD_CHI;					\
+  }						\
+  if (local || st.same_node[DIR] ) {		\
+    MULT_2SPIN(DIR);				\
+    RECON;					\
+  }
+
+#define HAND_STENCIL_LEG_EXT(PROJ,PERM,DIR,RECON)	\
+  SE=st.GetEntry(ptype,DIR,ss);			\
+  offset = SE->_offset;				\
+  if((!SE->_is_local)&&(!st.same_node[DIR]) ) {	\
+    LOAD_CHI;					\
+    MULT_2SPIN(DIR);				\
+    RECON;					\
+    nmu++;					\
+  }
+
+#define HAND_RESULT(ss)				\
+  {						\
+    SiteSpinor & ref (out._odata[ss]);		\
+    vstream(ref()(0)(0),result_00);		\
+    vstream(ref()(0)(1),result_01);		\
+    vstream(ref()(0)(2),result_02);		\
+    vstream(ref()(1)(0),result_10);		\
+    vstream(ref()(1)(1),result_11);		\
+    vstream(ref()(1)(2),result_12);		\
+    vstream(ref()(2)(0),result_20);		\
+    vstream(ref()(2)(1),result_21);		\
+    vstream(ref()(2)(2),result_22);		\
+    vstream(ref()(3)(0),result_30);		\
+    vstream(ref()(3)(1),result_31);		\
+    vstream(ref()(3)(2),result_32);		\
+  }
+
+#define HAND_RESULT_EXT(ss)			\
+  if (nmu){					\
+    SiteSpinor & ref (out._odata[ss]);		\
+    ref()(0)(0)+=result_00;		\
+    ref()(0)(1)+=result_01;		\
+    ref()(0)(2)+=result_02;		\
+    ref()(1)(0)+=result_10;		\
+    ref()(1)(1)+=result_11;		\
+    ref()(1)(2)+=result_12;		\
+    ref()(2)(0)+=result_20;		\
+    ref()(2)(1)+=result_21;		\
+    ref()(2)(2)+=result_22;		\
+    ref()(3)(0)+=result_30;		\
+    ref()(3)(1)+=result_31;		\
+    ref()(3)(2)+=result_32;		\
+  }


-template<class Impl> void 
-WilsonKernels<Impl>::HandDhopSite(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor  *buf,
-					  int ss,int sU,const FermionField &in, FermionField &out,int interior,int exterior)
-{
-  typedef typename Simd::scalar_type S;
-  typedef typename Simd::vector_type V;
+#define HAND_DECLARATIONS(a)			\
+  Simd result_00;				\
+  Simd result_01;				\
+  Simd result_02;				\
+  Simd result_10;				\
+  Simd result_11;				\
+  Simd result_12;				\
+  Simd result_20;				\
+  Simd result_21;				\
+  Simd result_22;				\
+  Simd result_30;				\
+  Simd result_31;				\
+  Simd result_32;				\
+  Simd Chi_00;					\
+  Simd Chi_01;					\
+  Simd Chi_02;					\
+  Simd Chi_10;					\
+  Simd Chi_11;					\
+  Simd Chi_12;					\
+  Simd UChi_00;					\
+  Simd UChi_01;					\
+  Simd UChi_02;					\
+  Simd UChi_10;					\
+  Simd UChi_11;					\
+  Simd UChi_12;					\
+  Simd U_00;					\
+  Simd U_10;					\
+  Simd U_20;					\
+  Simd U_01;					\
+  Simd U_11;					\
+  Simd U_21;

-  REGISTER Simd result_00; // 12 regs on knc
-  REGISTER Simd result_01;
-  REGISTER Simd result_02;
-
-  REGISTER Simd result_10;
-  REGISTER Simd result_11;
-  REGISTER Simd result_12;
-
-  REGISTER Simd result_20;
-  REGISTER Simd result_21;
-  REGISTER Simd result_22;
-
-  REGISTER Simd result_30;
-  REGISTER Simd result_31;
-  REGISTER Simd result_32; // 20 left
-
-  REGISTER Simd Chi_00;    // two spinor; 6 regs
-  REGISTER Simd Chi_01;
-  REGISTER Simd Chi_02;
-
-  REGISTER Simd Chi_10;
-  REGISTER Simd Chi_11;
-  REGISTER Simd Chi_12;   // 14 left
-
-  REGISTER Simd UChi_00;  // two spinor; 6 regs
-  REGISTER Simd UChi_01;
-  REGISTER Simd UChi_02;
-
-  REGISTER Simd UChi_10;
-  REGISTER Simd UChi_11;
-  REGISTER Simd UChi_12;  // 8 left
-
-  REGISTER Simd U_00;  // two rows of U matrix
-  REGISTER Simd U_10;
-  REGISTER Simd U_20;  
-  REGISTER Simd U_01;
-  REGISTER Simd U_11;
-  REGISTER Simd U_21;  // 2 reg left.
+#define ZERO_RESULT				\
+  result_00=zero;				\
+  result_01=zero;				\
+  result_02=zero;				\
+  result_10=zero;				\
+  result_11=zero;				\
+  result_12=zero;				\
+  result_20=zero;				\
+  result_21=zero;				\
+  result_22=zero;				\
+  result_30=zero;				\
+  result_31=zero;				\
+  result_32=zero;			

 #define Chimu_00 Chi_00
 #define Chimu_01 Chi_01
@@ -370,475 +447,225 @@ WilsonKernels<Impl>::HandDhopSite(StencilImpl &st,LebesgueOrder &lo,DoubledGauge
 #define Chimu_31 UChi_11
 #define Chimu_32 UChi_12

+namespace Grid {
+namespace QCD {
+
+template<class Impl> void 
+WilsonKernels<Impl>::HandDhopSite(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor  *buf,
+					  int ss,int sU,const FermionField &in, FermionField &out)
+{
+// T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
+  typedef typename Simd::scalar_type S;
+  typedef typename Simd::vector_type V;
+
+  HAND_DECLARATIONS(ignore);

  int offset,local,perm, ptype;
  StencilEntry *SE;

-  // Xp
-  SE=st.GetEntry(ptype,Xp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    XM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(3); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Xp);
-  }
-  XM_RECON;
-  
-  // Yp
-  SE=st.GetEntry(ptype,Yp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    YM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(2); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Yp);
-  }
-  YM_RECON_ACCUM;
-
-
-  // Zp
-  SE=st.GetEntry(ptype,Zp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    ZM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(1); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Zp);
-  }
-  ZM_RECON_ACCUM;
-
-  // Tp
-  SE=st.GetEntry(ptype,Tp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    TM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(0); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Tp);
-  }
-  TM_RECON_ACCUM;
-  
-  // Xm
-  SE=st.GetEntry(ptype,Xm,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    XP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(3); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Xm);
-  }
-  XP_RECON_ACCUM;
-  
-  
-  // Ym
-  SE=st.GetEntry(ptype,Ym,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    YP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(2); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Ym);
-  }
-  YP_RECON_ACCUM;
-
-  // Zm
-  SE=st.GetEntry(ptype,Zm,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    ZP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(1); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Zm);
-  }
-  ZP_RECON_ACCUM;
-
-  // Tm
-  SE=st.GetEntry(ptype,Tm,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    TP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(0); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Tm);
-  }
-  TP_RECON_ACCUM;
-
-  {
-    SiteSpinor & ref (out._odata[ss]);
-    vstream(ref()(0)(0),result_00);
-    vstream(ref()(0)(1),result_01);
-    vstream(ref()(0)(2),result_02);
-    vstream(ref()(1)(0),result_10);
-    vstream(ref()(1)(1),result_11);
-    vstream(ref()(1)(2),result_12);
-    vstream(ref()(2)(0),result_20);
-    vstream(ref()(2)(1),result_21);
-    vstream(ref()(2)(2),result_22);
-    vstream(ref()(3)(0),result_30);
-    vstream(ref()(3)(1),result_31);
-    vstream(ref()(3)(2),result_32);
-  }
+  HAND_STENCIL_LEG(XM_PROJ,3,Xp,XM_RECON);
+  HAND_STENCIL_LEG(YM_PROJ,2,Yp,YM_RECON_ACCUM);
+  HAND_STENCIL_LEG(ZM_PROJ,1,Zp,ZM_RECON_ACCUM);
+  HAND_STENCIL_LEG(TM_PROJ,0,Tp,TM_RECON_ACCUM);
+  HAND_STENCIL_LEG(XP_PROJ,3,Xm,XP_RECON_ACCUM);
+  HAND_STENCIL_LEG(YP_PROJ,2,Ym,YP_RECON_ACCUM);
+  HAND_STENCIL_LEG(ZP_PROJ,1,Zm,ZP_RECON_ACCUM);
+  HAND_STENCIL_LEG(TP_PROJ,0,Tm,TP_RECON_ACCUM);
+  HAND_RESULT(ss);
 }

 template<class Impl>
 void WilsonKernels<Impl>::HandDhopSiteDag(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
-						  int ss,int sU,const FermionField &in, FermionField &out,int interior,int exterior)
+						  int ss,int sU,const FermionField &in, FermionField &out)
 {
-  //  std::cout << "Hand op Dhop "<<std::endl;
  typedef typename Simd::scalar_type S;
  typedef typename Simd::vector_type V;

-  REGISTER Simd result_00; // 12 regs on knc
-  REGISTER Simd result_01;
-  REGISTER Simd result_02;
-  
-  REGISTER Simd result_10;
-  REGISTER Simd result_11;
-  REGISTER Simd result_12;
-
-  REGISTER Simd result_20;
-  REGISTER Simd result_21;
-  REGISTER Simd result_22;
-
-  REGISTER Simd result_30;
-  REGISTER Simd result_31;
-  REGISTER Simd result_32; // 20 left
-
-  REGISTER Simd Chi_00;    // two spinor; 6 regs
-  REGISTER Simd Chi_01;
-  REGISTER Simd Chi_02;
-
-  REGISTER Simd Chi_10;
-  REGISTER Simd Chi_11;
-  REGISTER Simd Chi_12;   // 14 left
-
-  REGISTER Simd UChi_00;  // two spinor; 6 regs
-  REGISTER Simd UChi_01;
-  REGISTER Simd UChi_02;
-
-  REGISTER Simd UChi_10;
-  REGISTER Simd UChi_11;
-  REGISTER Simd UChi_12;  // 8 left
-
-  REGISTER Simd U_00;  // two rows of U matrix
-  REGISTER Simd U_10;
-  REGISTER Simd U_20;  
-  REGISTER Simd U_01;
-  REGISTER Simd U_11;
-  REGISTER Simd U_21;  // 2 reg left.
-
-#define Chimu_00 Chi_00
-#define Chimu_01 Chi_01
-#define Chimu_02 Chi_02
-#define Chimu_10 Chi_10
-#define Chimu_11 Chi_11
-#define Chimu_12 Chi_12
-#define Chimu_20 UChi_00
-#define Chimu_21 UChi_01
-#define Chimu_22 UChi_02
-#define Chimu_30 UChi_10
-#define Chimu_31 UChi_11
-#define Chimu_32 UChi_12
-
+  HAND_DECLARATIONS(ignore);

  StencilEntry *SE;
  int offset,local,perm, ptype;
  
-  // Xp
-  SE=st.GetEntry(ptype,Xp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    XP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(3); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
+  HAND_STENCIL_LEG(XP_PROJ,3,Xp,XP_RECON);
+  HAND_STENCIL_LEG(YP_PROJ,2,Yp,YP_RECON_ACCUM);
+  HAND_STENCIL_LEG(ZP_PROJ,1,Zp,ZP_RECON_ACCUM);
+  HAND_STENCIL_LEG(TP_PROJ,0,Tp,TP_RECON_ACCUM);
+  HAND_STENCIL_LEG(XM_PROJ,3,Xm,XM_RECON_ACCUM);
+  HAND_STENCIL_LEG(YM_PROJ,2,Ym,YM_RECON_ACCUM);
+  HAND_STENCIL_LEG(ZM_PROJ,1,Zm,ZM_RECON_ACCUM);
+  HAND_STENCIL_LEG(TM_PROJ,0,Tm,TM_RECON_ACCUM);
+  HAND_RESULT(ss);
+}

-  {
-    MULT_2SPIN(Xp);
-  }
-  XP_RECON;
+template<class Impl> void 
+WilsonKernels<Impl>::HandDhopSiteInt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor  *buf,
+					  int ss,int sU,const FermionField &in, FermionField &out)
+{
+// T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
+  typedef typename Simd::scalar_type S;
+  typedef typename Simd::vector_type V;

-  // Yp
-  SE=st.GetEntry(ptype,Yp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    YP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(2); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Yp);
-  }
-  YP_RECON_ACCUM;
+  HAND_DECLARATIONS(ignore);

+  int offset,local,perm, ptype;
+  StencilEntry *SE;
+  ZERO_RESULT;
+  HAND_STENCIL_LEG_INT(XM_PROJ,3,Xp,XM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(YM_PROJ,2,Yp,YM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(ZM_PROJ,1,Zp,ZM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(TM_PROJ,0,Tp,TM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(XP_PROJ,3,Xm,XP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(YP_PROJ,2,Ym,YP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(ZP_PROJ,1,Zm,ZP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(TP_PROJ,0,Tm,TP_RECON_ACCUM);
+  HAND_RESULT(ss);
+}

-  // Zp
-  SE=st.GetEntry(ptype,Zp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    ZP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(1); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Zp);
-  }
-  ZP_RECON_ACCUM;
+template<class Impl>
+void WilsonKernels<Impl>::HandDhopSiteDagInt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+						  int ss,int sU,const FermionField &in, FermionField &out)
+{
+  typedef typename Simd::scalar_type S;
+  typedef typename Simd::vector_type V;

-  // Tp
-  SE=st.GetEntry(ptype,Tp,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    TP_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(0); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Tp);
-  }
-  TP_RECON_ACCUM;
-  
-  // Xm
-  SE=st.GetEntry(ptype,Xm,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    XM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(3); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Xm);
-  }
-  XM_RECON_ACCUM;
-  
-  // Ym
-  SE=st.GetEntry(ptype,Ym,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
-  
-  if ( local ) {
-    LOAD_CHIMU;
-    YM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(2); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Ym);
-  }
-  YM_RECON_ACCUM;
+  HAND_DECLARATIONS(ignore);

-  // Zm
-  SE=st.GetEntry(ptype,Zm,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
+  StencilEntry *SE;
+  int offset,local,perm, ptype;
+  ZERO_RESULT;
+  HAND_STENCIL_LEG_INT(XP_PROJ,3,Xp,XP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(YP_PROJ,2,Yp,YP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(ZP_PROJ,1,Zp,ZP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(TP_PROJ,0,Tp,TP_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(XM_PROJ,3,Xm,XM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(YM_PROJ,2,Ym,YM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(ZM_PROJ,1,Zm,ZM_RECON_ACCUM);
+  HAND_STENCIL_LEG_INT(TM_PROJ,0,Tm,TM_RECON_ACCUM);
+  HAND_RESULT(ss);
+}

-  if ( local ) {
-    LOAD_CHIMU;
-    ZM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(1); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Zm);
-  }
-  ZM_RECON_ACCUM;
+template<class Impl> void 
+WilsonKernels<Impl>::HandDhopSiteExt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor  *buf,
+					  int ss,int sU,const FermionField &in, FermionField &out)
+{
+// T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
+  typedef typename Simd::scalar_type S;
+  typedef typename Simd::vector_type V;

-  // Tm
-  SE=st.GetEntry(ptype,Tm,ss);
-  offset = SE->_offset;
-  local  = SE->_is_local;
-  perm   = SE->_permute;
+  HAND_DECLARATIONS(ignore);

-  if ( local ) {
-    LOAD_CHIMU;
-    TM_PROJ;
-    if ( perm) {
-      PERMUTE_DIR(0); // T==0, Z==1, Y==2, Z==3 expect 1,2,2,2 simd layout etc...
-    }
-  } else { 
-    LOAD_CHI;
-  }
-  {
-    MULT_2SPIN(Tm);
-  }
-  TM_RECON_ACCUM;
+  int offset,local,perm, ptype;
+  StencilEntry *SE;
+  int nmu=0;
+  ZERO_RESULT;
+  HAND_STENCIL_LEG_EXT(XM_PROJ,3,Xp,XM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(YM_PROJ,2,Yp,YM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(ZM_PROJ,1,Zp,ZM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(TM_PROJ,0,Tp,TM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(XP_PROJ,3,Xm,XP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(YP_PROJ,2,Ym,YP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(ZP_PROJ,1,Zm,ZP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(TP_PROJ,0,Tm,TP_RECON_ACCUM);
+  HAND_RESULT_EXT(ss);
+}

-  {
-    SiteSpinor & ref (out._odata[ss]);
-    vstream(ref()(0)(0),result_00);
-    vstream(ref()(0)(1),result_01);
-    vstream(ref()(0)(2),result_02);
-    vstream(ref()(1)(0),result_10);
-    vstream(ref()(1)(1),result_11);
-    vstream(ref()(1)(2),result_12);
-    vstream(ref()(2)(0),result_20);
-    vstream(ref()(2)(1),result_21);
-    vstream(ref()(2)(2),result_22);
-    vstream(ref()(3)(0),result_30);
-    vstream(ref()(3)(1),result_31);
-    vstream(ref()(3)(2),result_32);
-  }
+template<class Impl>
+void WilsonKernels<Impl>::HandDhopSiteDagExt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
+						  int ss,int sU,const FermionField &in, FermionField &out)
+{
+  typedef typename Simd::scalar_type S;
+  typedef typename Simd::vector_type V;
+
+  HAND_DECLARATIONS(ignore);
+
+  StencilEntry *SE;
+  int offset,local,perm, ptype;
+  int nmu=0;
+  ZERO_RESULT;
+  HAND_STENCIL_LEG_EXT(XP_PROJ,3,Xp,XP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(YP_PROJ,2,Yp,YP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(ZP_PROJ,1,Zp,ZP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(TP_PROJ,0,Tp,TP_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(XM_PROJ,3,Xm,XM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(YM_PROJ,2,Ym,YM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(ZM_PROJ,1,Zm,ZM_RECON_ACCUM);
+  HAND_STENCIL_LEG_EXT(TM_PROJ,0,Tm,TM_RECON_ACCUM);
+  HAND_RESULT_EXT(ss);
 }

  ////////////////////////////////////////////////
  // Specialise Gparity to simple implementation
  ////////////////////////////////////////////////
-template<> void 
-WilsonKernels<GparityWilsonImplF>::HandDhopSite(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,
-							SiteHalfSpinor *buf,
-							int sF,int sU,const FermionField &in, FermionField &out,int internal,int external)
-{
-  assert(0);
-}
-
-template<> void 
-WilsonKernels<GparityWilsonImplF>::HandDhopSiteDag(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,
-							   SiteHalfSpinor *buf,
-							   int sF,int sU,const FermionField &in, FermionField &out,int internal,int external)
-{
-  assert(0);
-}
-
-template<> void 
-WilsonKernels<GparityWilsonImplD>::HandDhopSite(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
-							int sF,int sU,const FermionField &in, FermionField &out,int internal,int external)
-{
-  assert(0);
-}
-
-template<> void 
-WilsonKernels<GparityWilsonImplD>::HandDhopSiteDag(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,
-							   int sF,int sU,const FermionField &in, FermionField &out,int internal,int external)
-{
-  assert(0);
-}
-
+#define HAND_SPECIALISE_EMPTY(IMPL)					\
+  template<> void							\
+  WilsonKernels<IMPL>::HandDhopSite(StencilImpl &st,			\
+				    LebesgueOrder &lo,			\
+				    DoubledGaugeField &U,		\
+				    SiteHalfSpinor *buf,		\
+				    int sF,int sU,			\
+				    const FermionField &in,		\
+				    FermionField &out){ assert(0); }	\
+  template<> void							\
+  WilsonKernels<IMPL>::HandDhopSiteDag(StencilImpl &st,			\
+				    LebesgueOrder &lo,			\
+				    DoubledGaugeField &U,		\
+				    SiteHalfSpinor *buf,		\
+				    int sF,int sU,			\
+				    const FermionField &in,		\
+				    FermionField &out){ assert(0); }	\
+  template<> void							\
+  WilsonKernels<IMPL>::HandDhopSiteInt(StencilImpl &st,			\
+				    LebesgueOrder &lo,			\
+				    DoubledGaugeField &U,		\
+				    SiteHalfSpinor *buf,		\
+				    int sF,int sU,			\
+				    const FermionField &in,		\
+				    FermionField &out){ assert(0); }	\
+  template<> void							\
+  WilsonKernels<IMPL>::HandDhopSiteExt(StencilImpl &st,			\
+				    LebesgueOrder &lo,			\
+				    DoubledGaugeField &U,		\
+				    SiteHalfSpinor *buf,		\
+				    int sF,int sU,			\
+				    const FermionField &in,		\
+				    FermionField &out){ assert(0); }	\
+  template<> void							\
+  WilsonKernels<IMPL>::HandDhopSiteDagInt(StencilImpl &st,	       	\
+				    LebesgueOrder &lo,			\
+				    DoubledGaugeField &U,		\
+				    SiteHalfSpinor *buf,		\
+				    int sF,int sU,			\
+				    const FermionField &in,		\
+				    FermionField &out){ assert(0); }	\
+  template<> void							\
+  WilsonKernels<IMPL>::HandDhopSiteDagExt(StencilImpl &st,	       	\
+				    LebesgueOrder &lo,			\
+				    DoubledGaugeField &U,		\
+				    SiteHalfSpinor *buf,		\
+				    int sF,int sU,			\
+				    const FermionField &in,		\
+				    FermionField &out){ assert(0); }	\

+  HAND_SPECIALISE_EMPTY(GparityWilsonImplF);
+  HAND_SPECIALISE_EMPTY(GparityWilsonImplD);
+  HAND_SPECIALISE_EMPTY(GparityWilsonImplFH);
+  HAND_SPECIALISE_EMPTY(GparityWilsonImplDF);

 ////////////// Wilson ; uses this implementation /////////////////////
-// Need Nc=3 though //

 #define INSTANTIATE_THEM(A) \
 template void WilsonKernels<A>::HandDhopSite(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,\
-						     int ss,int sU,const FermionField &in, FermionField &out,int interior,int exterior); \
-template void WilsonKernels<A>::HandDhopSiteDag(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,\
-							int ss,int sU,const FermionField &in, FermionField &out,int interior,int exterior);
+					     int ss,int sU,const FermionField &in, FermionField &out); \
+template void WilsonKernels<A>::HandDhopSiteDag(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf, \
+						int ss,int sU,const FermionField &in, FermionField &out);\
+template void WilsonKernels<A>::HandDhopSiteInt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,\
+						int ss,int sU,const FermionField &in, FermionField &out); \
+template void WilsonKernels<A>::HandDhopSiteDagInt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf, \
+						   int ss,int sU,const FermionField &in, FermionField &out); \
+template void WilsonKernels<A>::HandDhopSiteExt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf,\
+						int ss,int sU,const FermionField &in, FermionField &out); \
+template void WilsonKernels<A>::HandDhopSiteDagExt(StencilImpl &st,LebesgueOrder &lo,DoubledGaugeField &U,SiteHalfSpinor *buf, \
+						   int ss,int sU,const FermionField &in, FermionField &out); 

 INSTANTIATE_THEM(WilsonImplF);
 INSTANTIATE_THEM(WilsonImplD);
@@ -850,5 +677,15 @@ INSTANTIATE_THEM(DomainWallVec5dImplF);
 INSTANTIATE_THEM(DomainWallVec5dImplD);
 INSTANTIATE_THEM(ZDomainWallVec5dImplF);
 INSTANTIATE_THEM(ZDomainWallVec5dImplD);
+INSTANTIATE_THEM(WilsonImplFH);
+INSTANTIATE_THEM(WilsonImplDF);
+INSTANTIATE_THEM(ZWilsonImplFH);
+INSTANTIATE_THEM(ZWilsonImplDF);
+INSTANTIATE_THEM(GparityWilsonImplFH);
+INSTANTIATE_THEM(GparityWilsonImplDF);
+INSTANTIATE_THEM(DomainWallVec5dImplFH);
+INSTANTIATE_THEM(DomainWallVec5dImplDF);
+INSTANTIATE_THEM(ZDomainWallVec5dImplFH);
+INSTANTIATE_THEM(ZDomainWallVec5dImplDF);

 }}
@@ -29,7 +29,7 @@ directory
 #ifndef GRID_QCD_GAUGE_H
 #define GRID_QCD_GAUGE_H

-#include <Grid/qcd/action/gauge/GaugeImpl.h>
+#include <Grid/qcd/action/gauge/GaugeImplementations.h>
 #include <Grid/qcd/utils/WilsonLoops.h>
 #include <Grid/qcd/action/gauge/WilsonGaugeAction.h>
 #include <Grid/qcd/action/gauge/PlaqPlusRectangleAction.h>
@@ -0,0 +1,153 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/action/gauge/GaugeImpl.h
+
+Copyright (C) 2015
+
+Author: paboyle <paboyle@ph.ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_GAUGE_IMPL_TYPES_H
+#define GRID_GAUGE_IMPL_TYPES_H
+
+namespace Grid {
+namespace QCD {
+
+////////////////////////////////////////////////////////////////////////
+// Implementation dependent gauge types
+////////////////////////////////////////////////////////////////////////
+
+#define INHERIT_GIMPL_TYPES(GImpl)                  \
+  typedef typename GImpl::Simd Simd;                \
+  typedef typename GImpl::LinkField GaugeLinkField; \
+  typedef typename GImpl::Field GaugeField;         \
+  typedef typename GImpl::ComplexField ComplexField;\
+  typedef typename GImpl::SiteField SiteGaugeField; \
+  typedef typename GImpl::SiteComplex SiteComplex;  \
+  typedef typename GImpl::SiteLink SiteGaugeLink;
+
+#define INHERIT_FIELD_TYPES(Impl)		    \
+  typedef typename Impl::Simd Simd;		    \
+  typedef typename Impl::ComplexField ComplexField; \
+  typedef typename Impl::SiteField SiteField;	    \
+  typedef typename Impl::Field Field;
+
+// hardcodes the exponential approximation in the template
+template <class S, int Nrepresentation = Nc, int Nexp = 12 > class GaugeImplTypes {
+public:
+  typedef S Simd;
+
+  template <typename vtype> using iImplScalar     = iScalar<iScalar<iScalar<vtype> > >;
+  template <typename vtype> using iImplGaugeLink  = iScalar<iScalar<iMatrix<vtype, Nrepresentation> > >;
+  template <typename vtype> using iImplGaugeField = iVector<iScalar<iMatrix<vtype, Nrepresentation> >, Nd>;
+
+  typedef iImplScalar<Simd>     SiteComplex;
+  typedef iImplGaugeLink<Simd>  SiteLink;
+  typedef iImplGaugeField<Simd> SiteField;
+
+  typedef Lattice<SiteComplex> ComplexField;
+  typedef Lattice<SiteLink>    LinkField; 
+  typedef Lattice<SiteField>   Field;
+
+  // Guido: we can probably separate the types from the HMC functions
+  // this will create 2 kind of implementations
+  // probably confusing the users
+  // Now keeping only one class
+
+
+  // Move this elsewhere? FIXME
+  static inline void AddLink(Field &U, LinkField &W,
+                                  int mu) { // U[mu] += W
+    PARALLEL_FOR_LOOP
+    for (auto ss = 0; ss < U._grid->oSites(); ss++) {
+      U._odata[ss]._internal[mu] =
+          U._odata[ss]._internal[mu] + W._odata[ss]._internal;
+    }
+  }
+
+  ///////////////////////////////////////////////////////////
+  // Move these to another class
+  // HMC auxiliary functions
+  static inline void generate_momenta(Field &P, GridParallelRNG &pRNG) {
+    // specific for SU gauge fields
+    LinkField Pmu(P._grid);
+    Pmu = zero;
+    for (int mu = 0; mu < Nd; mu++) {
+      SU<Nrepresentation>::GaussianFundamentalLieAlgebraMatrix(pRNG, Pmu);
+      PokeIndex<LorentzIndex>(P, Pmu, mu);
+    }
+  }
+
+  static inline Field projectForce(Field &P) { return Ta(P); }
+
+  static inline void update_field(Field& P, Field& U, double ep){
+    //static std::chrono::duration<double> diff;
+
+    //auto start = std::chrono::high_resolution_clock::now();
+    parallel_for(int ss=0;ss<P._grid->oSites();ss++){
+      for (int mu = 0; mu < Nd; mu++) 
+        U[ss]._internal[mu] = ProjectOnGroup(Exponentiate(P[ss]._internal[mu], ep, Nexp) * U[ss]._internal[mu]);
+    }
+    
+    //auto end = std::chrono::high_resolution_clock::now();
+   // diff += end - start;
+   // std::cout << "Time to exponentiate matrix " << diff.count() << " s\n";
+  }
+
+  static inline RealD FieldSquareNorm(Field& U){
+    LatticeComplex Hloc(U._grid);
+    Hloc = zero;
+    for (int mu = 0; mu < Nd; mu++) {
+      auto Umu = PeekIndex<LorentzIndex>(U, mu);
+      Hloc += trace(Umu * Umu);
+    }
+    Complex Hsum = sum(Hloc);
+    return Hsum.real();
+  }
+
+  static inline void HotConfiguration(GridParallelRNG &pRNG, Field &U) {
+    SU<Nc>::HotConfiguration(pRNG, U);
+  }
+
+  static inline void TepidConfiguration(GridParallelRNG &pRNG, Field &U) {
+    SU<Nc>::TepidConfiguration(pRNG, U);
+  }
+
+  static inline void ColdConfiguration(GridParallelRNG &pRNG, Field &U) {
+    SU<Nc>::ColdConfiguration(pRNG, U);
+  }
+};
+
+
+typedef GaugeImplTypes<vComplex, Nc> GimplTypesR;
+typedef GaugeImplTypes<vComplexF, Nc> GimplTypesF;
+typedef GaugeImplTypes<vComplexD, Nc> GimplTypesD;
+
+typedef GaugeImplTypes<vComplex, SU<Nc>::AdjointDimension> GimplAdjointTypesR;
+typedef GaugeImplTypes<vComplexF, SU<Nc>::AdjointDimension> GimplAdjointTypesF;
+typedef GaugeImplTypes<vComplexD, SU<Nc>::AdjointDimension> GimplAdjointTypesD;
+
+
+} // QCD
+} // Grid
+
+#endif // GRID_GAUGE_IMPL_TYPES_H
@@ -2,7 +2,7 @@

 Grid physics library, www.github.com/paboyle/Grid

-Source file: ./lib/qcd/action/gauge/GaugeImpl.h
+Source file: ./lib/qcd/action/gauge/GaugeImplementations.h

 Copyright (C) 2015

@@ -26,53 +26,14 @@ See the full license in the file "LICENSE" in the top level distribution
 directory
 *************************************************************************************/
 /*  END LEGAL */
-#ifndef GRID_QCD_GAUGE_IMPL_H
-#define GRID_QCD_GAUGE_IMPL_H
+#ifndef GRID_QCD_GAUGE_IMPLEMENTATIONS_H
+#define GRID_QCD_GAUGE_IMPLEMENTATIONS_H
+
+#include "GaugeImplTypes.h"

 namespace Grid {
 namespace QCD {

-////////////////////////////////////////////////////////////////////////
-// Implementation dependent gauge types
-////////////////////////////////////////////////////////////////////////
-
-template <class Gimpl> class WilsonLoops;
-
-#define INHERIT_GIMPL_TYPES(GImpl)                                             \
-  typedef typename GImpl::Simd Simd;                                           \
-  typedef typename GImpl::GaugeLinkField GaugeLinkField;                       \
-  typedef typename GImpl::GaugeField GaugeField;                               \
-  typedef typename GImpl::SiteGaugeField SiteGaugeField;                       \
-  typedef typename GImpl::SiteGaugeLink SiteGaugeLink;
-
-//
-template <class S, int Nrepresentation = Nc> class GaugeImplTypes {
-public:
-  typedef S Simd;
-
-  template <typename vtype>
-  using iImplGaugeLink  = iScalar<iScalar<iMatrix<vtype, Nrepresentation>>>;
-  template <typename vtype>
-  using iImplGaugeField = iVector<iScalar<iMatrix<vtype, Nrepresentation>>, Nd>;
-
-  typedef iImplGaugeLink<Simd> SiteGaugeLink;
-  typedef iImplGaugeField<Simd> SiteGaugeField;
-
-  typedef Lattice<SiteGaugeLink> GaugeLinkField; // bit ugly naming; polarised
-                                                 // gauge field, lorentz... all
-                                                 // ugly
-  typedef Lattice<SiteGaugeField> GaugeField;
-
-  // Move this elsewhere? FIXME
-  static inline void AddGaugeLink(GaugeField &U, GaugeLinkField &W,
-                                  int mu) { // U[mu] += W
-    parallel_for (auto ss = 0; ss < U._grid->oSites(); ss++) {
-      U._odata[ss]._internal[mu] =
-          U._odata[ss]._internal[mu] + W._odata[ss]._internal;
-    }
-  }
-};
-
 // Composition with smeared link, bc's etc.. probably need multiple inheritance
 // Variable precision "S" and variable Nc
 template <class GimplTypes> class PeriodicGaugeImpl : public GimplTypes {
@@ -168,14 +129,6 @@ public:
  static inline bool isPeriodicGaugeField(void) { return false; }
 };

-typedef GaugeImplTypes<vComplex, Nc> GimplTypesR;
-typedef GaugeImplTypes<vComplexF, Nc> GimplTypesF;
-typedef GaugeImplTypes<vComplexD, Nc> GimplTypesD;
-
-typedef GaugeImplTypes<vComplex, SU<Nc>::AdjointDimension> GimplAdjointTypesR;
-typedef GaugeImplTypes<vComplexF, SU<Nc>::AdjointDimension> GimplAdjointTypesF;
-typedef GaugeImplTypes<vComplexD, SU<Nc>::AdjointDimension> GimplAdjointTypesD;
-
 typedef PeriodicGaugeImpl<GimplTypesR> PeriodicGimplR; // Real.. whichever prec
 typedef PeriodicGaugeImpl<GimplTypesF> PeriodicGimplF; // Float
 typedef PeriodicGaugeImpl<GimplTypesD> PeriodicGimplD; // Double
@@ -187,6 +140,8 @@ typedef PeriodicGaugeImpl<GimplAdjointTypesD> PeriodicGimplAdjD; // Double
 typedef ConjugateGaugeImpl<GimplTypesR> ConjugateGimplR; // Real.. whichever prec
 typedef ConjugateGaugeImpl<GimplTypesF> ConjugateGimplF; // Float
 typedef ConjugateGaugeImpl<GimplTypesD> ConjugateGimplD; // Double
+
+
 }
 }

@@ -0,0 +1,286 @@
+/*************************************************************************************
+ 
+ Grid physics library, www.github.com/paboyle/Grid
+ 
+ Source file: ./lib/qcd/action/gauge/Photon.h
+ 
+ Copyright (C) 2015
+ 
+ Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+ 
+ This program is free software; you can redistribute it and/or modify
+ it under the terms of the GNU General Public License as published by
+ the Free Software Foundation; either version 2 of the License, or
+ (at your option) any later version.
+ 
+ This program is distributed in the hope that it will be useful,
+ but WITHOUT ANY WARRANTY; without even the implied warranty of
+ MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+ GNU General Public License for more details.
+ 
+ You should have received a copy of the GNU General Public License along
+ with this program; if not, write to the Free Software Foundation, Inc.,
+ 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+ 
+ See the full license in the file "LICENSE" in the top level distribution directory
+ *************************************************************************************/
+/*  END LEGAL */
+#ifndef QCD_PHOTON_ACTION_H
+#define QCD_PHOTON_ACTION_H
+
+namespace Grid{
+namespace QCD{
+  template <class S>
+  class QedGimpl
+  {
+  public:
+    typedef S Simd;
+    
+    template <typename vtype>
+    using iImplGaugeLink  = iScalar<iScalar<iScalar<vtype>>>;
+    template <typename vtype>
+    using iImplGaugeField = iVector<iScalar<iScalar<vtype>>, Nd>;
+    
+    typedef iImplGaugeLink<Simd>  SiteLink;
+    typedef iImplGaugeField<Simd> SiteField;
+    typedef SiteField             SiteComplex;
+    
+    typedef Lattice<SiteLink>  LinkField;
+    typedef Lattice<SiteField> Field;
+    typedef Field              ComplexField;
+  };
+  
+  typedef QedGimpl<vComplex> QedGimplR;
+  
+  template<class Gimpl>
+  class Photon
+  {
+  public:
+    INHERIT_GIMPL_TYPES(Gimpl);
+    GRID_SERIALIZABLE_ENUM(Gauge, undef, feynman, 1, coulomb, 2, landau, 3);
+    GRID_SERIALIZABLE_ENUM(ZmScheme, undef, qedL, 1, qedTL, 2);
+  public:
+    Photon(Gauge gauge, ZmScheme zmScheme);
+    virtual ~Photon(void) = default;
+    void FreePropagator(const GaugeField &in, GaugeField &out);
+    void MomentumSpacePropagator(const GaugeField &in, GaugeField &out);
+    void StochasticWeight(GaugeLinkField &weight);
+    void StochasticField(GaugeField &out, GridParallelRNG &rng);
+    void StochasticField(GaugeField &out, GridParallelRNG &rng,
+                         const GaugeLinkField &weight);
+  private:
+    void invKHatSquared(GaugeLinkField &out);
+    void zmSub(GaugeLinkField &out);
+  private:
+    Gauge    gauge_;
+    ZmScheme zmScheme_;
+  };
+
+  typedef Photon<QedGimplR>  PhotonR;
+  
+  template<class Gimpl>
+  Photon<Gimpl>::Photon(Gauge gauge, ZmScheme zmScheme)
+  : gauge_(gauge), zmScheme_(zmScheme)
+  {}
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::FreePropagator (const GaugeField &in,GaugeField &out)
+  {
+    FFT theFFT(in._grid);
+    
+    GaugeField in_k(in._grid);
+    GaugeField prop_k(in._grid);
+    
+    theFFT.FFT_all_dim(in_k,in,FFT::forward);
+    MomentumSpacePropagator(prop_k,in_k);
+    theFFT.FFT_all_dim(out,prop_k,FFT::backward);
+  }
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::invKHatSquared(GaugeLinkField &out)
+  {
+    GridBase           *grid = out._grid;
+    GaugeLinkField     kmu(grid), one(grid);
+    const unsigned int nd    = grid->_ndimension;
+    std::vector<int>   &l    = grid->_fdimensions;
+    std::vector<int>   zm(nd,0);
+    TComplex           Tone = Complex(1.0,0.0);
+    TComplex           Tzero= Complex(0.0,0.0);
+    
+    one = Complex(1.0,0.0);
+    out = zero;
+    for(int mu = 0; mu < nd; mu++)
+    {
+      Real twoPiL = M_PI*2./l[mu];
+      
+      LatticeCoordinate(kmu,mu);
+      kmu = 2.*sin(.5*twoPiL*kmu);
+      out = out + kmu*kmu;
+    }
+    pokeSite(Tone, out, zm);
+    out = one/out;
+    pokeSite(Tzero, out, zm);
+  }
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::zmSub(GaugeLinkField &out)
+  {
+    GridBase           *grid = out._grid;
+    const unsigned int nd    = grid->_ndimension;
+    
+    switch (zmScheme_)
+    {
+      case ZmScheme::qedTL:
+      {
+        std::vector<int> zm(nd,0);
+        TComplex         Tzero = Complex(0.0,0.0);
+        
+        pokeSite(Tzero, out, zm);
+        
+        break;
+      }
+      case ZmScheme::qedL:
+      {
+        LatticeInteger spNrm(grid), coor(grid);
+        GaugeLinkField z(grid);
+        
+        spNrm = zero;
+        for(int d = 0; d < grid->_ndimension - 1; d++)
+        {
+          LatticeCoordinate(coor,d);
+          spNrm = spNrm + coor*coor;
+        }
+        out = where(spNrm == Integer(0), 0.*out, out);
+        
+        break;
+      }
+      default:
+        break;
+    }
+  }
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::MomentumSpacePropagator(const GaugeField &in,
+                                               GaugeField &out)
+  {
+    GridBase           *grid = out._grid;
+    LatticeComplex     k2Inv(grid);
+    
+    invKHatSquared(k2Inv);
+    zmSub(k2Inv);
+    
+    out = in*k2Inv;
+  }
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::StochasticWeight(GaugeLinkField &weight)
+  {
+    auto               *grid     = dynamic_cast<GridCartesian *>(weight._grid);
+    const unsigned int nd        = grid->_ndimension;
+    std::vector<int>   latt_size = grid->_fdimensions;
+    
+    Integer vol = 1;
+    for(int d = 0; d < nd; d++)
+    {
+      vol = vol * latt_size[d];
+    }
+    invKHatSquared(weight);
+    weight = sqrt(vol*real(weight));
+    zmSub(weight);
+  }
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::StochasticField(GaugeField &out, GridParallelRNG &rng)
+  {
+    auto           *grid = dynamic_cast<GridCartesian *>(out._grid);
+    GaugeLinkField weight(grid);
+    
+    StochasticWeight(weight);
+    StochasticField(out, rng, weight);
+  }
+  
+  template<class Gimpl>
+  void Photon<Gimpl>::StochasticField(GaugeField &out, GridParallelRNG &rng,
+                                      const GaugeLinkField &weight)
+  {
+    auto               *grid = dynamic_cast<GridCartesian *>(out._grid);
+    const unsigned int nd = grid->_ndimension;
+    GaugeLinkField     r(grid);
+    GaugeField         aTilde(grid);
+    FFT                fft(grid);
+    
+    for(int mu = 0; mu < nd; mu++)
+    {
+      gaussian(rng, r);
+      r = weight*r;
+      pokeLorentz(aTilde, r, mu);
+    }
+    fft.FFT_all_dim(out, aTilde, FFT::backward);
+    
+    out = real(out);
+  }
+//  template<class Gimpl>
+//  void Photon<Gimpl>::FeynmanGaugeMomentumSpacePropagator_L(GaugeField &out,
+//                                                            const GaugeField &in)
+//  {
+//    
+//    FeynmanGaugeMomentumSpacePropagator_TL(out,in);
+//    
+//    GridBase *grid = out._grid;
+//    LatticeInteger     coor(grid);
+//    GaugeField zz(grid); zz=zero;
+//    
+//    // xyzt
+//    for(int d = 0; d < grid->_ndimension-1;d++){
+//      LatticeCoordinate(coor,d);
+//      out = where(coor==Integer(0),zz,out);
+//    }
+//  }
+//  
+//  template<class Gimpl>
+//  void Photon<Gimpl>::FeynmanGaugeMomentumSpacePropagator_TL(GaugeField &out,
+//                                                             const GaugeField &in)
+//  {
+//    
+//    // what type LatticeComplex
+//    GridBase *grid = out._grid;
+//    int nd = grid->_ndimension;
+//    
+//    typedef typename GaugeField::vector_type vector_type;
+//    typedef typename GaugeField::scalar_type ScalComplex;
+//    typedef Lattice<iSinglet<vector_type> > LatComplex;
+//    
+//    std::vector<int> latt_size   = grid->_fdimensions;
+//    
+//    LatComplex denom(grid); denom= zero;
+//    LatComplex   one(grid); one = ScalComplex(1.0,0.0);
+//    LatComplex   kmu(grid);
+//    
+//    ScalComplex ci(0.0,1.0);
+//    // momphase = n * 2pi / L
+//    for(int mu=0;mu<Nd;mu++) {
+//      
+//      LatticeCoordinate(kmu,mu);
+//      
+//      RealD TwoPiL =  M_PI * 2.0/ latt_size[mu];
+//      
+//      kmu = TwoPiL * kmu ;
+//      
+//      denom = denom + 4.0*sin(kmu*0.5)*sin(kmu*0.5); // Wilson term
+//    }
+//    std::vector<int> zero_mode(nd,0);
+//    TComplexD Tone = ComplexD(1.0,0.0);
+//    TComplexD Tzero= ComplexD(0.0,0.0);
+//    
+//    pokeSite(Tone,denom,zero_mode);
+//    
+//    denom= one/denom;
+//    
+//    pokeSite(Tzero,denom,zero_mode);
+//    
+//    out = zero;
+//    out = in*denom;
+//  };
+  
+}}
+#endif
@@ -47,9 +47,19 @@ namespace Grid{

    public:
    PlaqPlusRectangleAction(RealD b,RealD c): c_plaq(b),c_rect(c){};
+
+      virtual std::string action_name(){return "PlaqPlusRectangleAction";}
      
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {}; // noop as no pseudoferms
      
+      virtual std::string LogParameters(){
+      	std::stringstream sstream;
+      	sstream << GridLogMessage << "["<<action_name() <<"] c_plaq: " << c_plaq << std::endl;
+      	sstream << GridLogMessage << "["<<action_name() <<"] c_rect: " << c_rect << std::endl;
+      	return sstream.str();
+      }
+
+
      virtual RealD S(const GaugeField &U) {
 	RealD vol = U._grid->gSites();

@@ -108,32 +118,32 @@ namespace Grid{
    class RBCGaugeAction : public PlaqPlusRectangleAction<Gimpl> {
    public:
      INHERIT_GIMPL_TYPES(Gimpl);
-      RBCGaugeAction(RealD beta,RealD c1) : PlaqPlusRectangleAction<Gimpl>(beta*(1.0-8.0*c1), beta*c1) {
-      };
+      RBCGaugeAction(RealD beta,RealD c1) : PlaqPlusRectangleAction<Gimpl>(beta*(1.0-8.0*c1), beta*c1) {};
+      virtual std::string action_name(){return "RBCGaugeAction";}
    };

    template<class Gimpl>
    class IwasakiGaugeAction : public RBCGaugeAction<Gimpl> {
    public:
      INHERIT_GIMPL_TYPES(Gimpl);
-      IwasakiGaugeAction(RealD beta) : RBCGaugeAction<Gimpl>(beta,-0.331) {
-      };
+      IwasakiGaugeAction(RealD beta) : RBCGaugeAction<Gimpl>(beta,-0.331) {};
+      virtual std::string action_name(){return "IwasakiGaugeAction";}
    };

    template<class Gimpl>
    class SymanzikGaugeAction : public RBCGaugeAction<Gimpl> {
    public:
      INHERIT_GIMPL_TYPES(Gimpl);
-      SymanzikGaugeAction(RealD beta) : RBCGaugeAction<Gimpl>(beta,-1.0/12.0) {
-      };
+      SymanzikGaugeAction(RealD beta) : RBCGaugeAction<Gimpl>(beta,-1.0/12.0) {};
+      virtual std::string action_name(){return "SymanzikGaugeAction";}
    };

    template<class Gimpl>
    class DBW2GaugeAction : public RBCGaugeAction<Gimpl> {
    public:
      INHERIT_GIMPL_TYPES(Gimpl);
-      DBW2GaugeAction(RealD beta) : RBCGaugeAction<Gimpl>(beta,-1.4067) {
-      };
+      DBW2GaugeAction(RealD beta) : RBCGaugeAction<Gimpl>(beta,-1.4067) {};
+      virtual std::string action_name(){return "DBW2GaugeAction";}
    };

  }
@@ -1,86 +1,99 @@
-    /*************************************************************************************
+/*************************************************************************************

-    Grid physics library, www.github.com/paboyle/Grid 
+Grid physics library, www.github.com/paboyle/Grid

-    Source file: ./lib/qcd/action/gauge/WilsonGaugeAction.h
+Source file: ./lib/qcd/action/gauge/WilsonGaugeAction.h

-    Copyright (C) 2015
+Copyright (C) 2015

 Author: Azusa Yamaguchi <ayamaguc@staffmail.ed.ac.uk>
 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
 Author: neo <cossu@post.kek.jp>
 Author: paboyle <paboyle@ph.ed.ac.uk>
+Author: Guido Cossu <guido.cossu@ed.ac.uk>

-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.

-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.

-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.

-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
 #ifndef QCD_WILSON_GAUGE_ACTION_H
 #define QCD_WILSON_GAUGE_ACTION_H

-namespace Grid{
-  namespace QCD{
-    
-    ////////////////////////////////////////////////////////////////////////
-    // Wilson Gauge Action .. should I template the Nc etc..
-    ////////////////////////////////////////////////////////////////////////
-    template<class Gimpl>
-    class WilsonGaugeAction : public Action<typename Gimpl::GaugeField> {
-    public:
+namespace Grid {
+namespace QCD {

-      INHERIT_GIMPL_TYPES(Gimpl);
+////////////////////////////////////////////////////////////////////////
+// Wilson Gauge Action .. should I template the Nc etc..
+////////////////////////////////////////////////////////////////////////
+template <class Gimpl>
+class WilsonGaugeAction : public Action<typename Gimpl::GaugeField> {
+ public:  
+  INHERIT_GIMPL_TYPES(Gimpl);

-      //      typedef LorentzScalar<GaugeField> GaugeLinkField;
+  /////////////////////////// constructors
+  explicit WilsonGaugeAction(RealD beta_):beta(beta_){};

-    private:
-      RealD beta;
-    public:
-    WilsonGaugeAction(RealD b):beta(b){};
-      
-      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {}; // noop as no pseudoferms
-      
-      virtual RealD S(const GaugeField &U) {
-	RealD plaq = WilsonLoops<Gimpl>::avgPlaquette(U);
-	RealD vol = U._grid->gSites();
-	RealD action=beta*(1.0 -plaq)*(Nd*(Nd-1.0))*vol*0.5;
-	return action;
-      };
+  virtual std::string action_name() {return "WilsonGaugeAction";}

-      virtual void deriv(const GaugeField &U,GaugeField & dSdU) {
-	//not optimal implementation FIXME
-	//extend Ta to include Lorentz indexes
-
-	//RealD factor = 0.5*beta/RealD(Nc);
-	RealD factor = 0.5*beta/RealD(Nc);
-
-	GaugeLinkField Umu(U._grid);
-	GaugeLinkField dSdU_mu(U._grid);
-	for (int mu=0; mu < Nd; mu++){
-
-	  Umu = PeekIndex<LorentzIndex>(U,mu);
-
-	  // Staple in direction mu
-	  WilsonLoops<Gimpl>::Staple(dSdU_mu,U,mu);
-	  dSdU_mu = Ta(Umu*dSdU_mu)*factor;
-	  PokeIndex<LorentzIndex>(dSdU, dSdU_mu, mu);
-	}
-      };
-    };
-    
+  virtual std::string LogParameters(){
+    std::stringstream sstream;
+    sstream << GridLogMessage << "[WilsonGaugeAction] Beta: " << beta << std::endl;
+    return sstream.str();
  }
+
+  virtual void refresh(const GaugeField &U,
+                       GridParallelRNG &pRNG){};  // noop as no pseudoferms
+
+  virtual RealD S(const GaugeField &U) {
+    RealD plaq = WilsonLoops<Gimpl>::avgPlaquette(U);
+    RealD vol = U._grid->gSites();
+    RealD action = beta * (1.0 - plaq) * (Nd * (Nd - 1.0)) * vol * 0.5;
+    return action;
+  };
+
+  virtual void deriv(const GaugeField &U, GaugeField &dSdU) {
+    // not optimal implementation FIXME
+    // extend Ta to include Lorentz indexes
+
+    RealD factor = 0.5 * beta / RealD(Nc);
+
+    //GaugeLinkField Umu(U._grid);
+    GaugeLinkField dSdU_mu(U._grid);
+    for (int mu = 0; mu < Nd; mu++) {
+      //Umu = PeekIndex<LorentzIndex>(U, mu);
+
+      // Staple in direction mu
+      //WilsonLoops<Gimpl>::Staple(dSdU_mu, U, mu);
+      //dSdU_mu = Ta(Umu * dSdU_mu) * factor;
+
+  
+      WilsonLoops<Gimpl>::StapleMult(dSdU_mu, U, mu);
+      dSdU_mu = Ta(dSdU_mu) * factor;
+
+      PokeIndex<LorentzIndex>(dSdU, dSdU_mu, mu);
+    }
+  }
+private:
+  RealD beta;  
+};
+
+
+
+}
 }

 #endif
@@ -7,6 +7,7 @@
    Copyright (C) 2015

 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+Author: Guido Cossu <guido.cossu@ed.ac.uk>

    This program is free software; you can redistribute it and/or modify
    it under the terms of the GNU General Public License as published by
@@ -45,92 +46,97 @@ namespace Grid{
      public:
      INHERIT_IMPL_TYPES(Impl);

- 	typedef FermionOperator<Impl> Matrix;
+        typedef FermionOperator<Impl> Matrix;

-	SchurDifferentiableOperator (Matrix &Mat) : SchurDiagMooeeOperator<Matrix,FermionField>(Mat) {};
+        SchurDifferentiableOperator (Matrix &Mat) : SchurDiagMooeeOperator<Matrix,FermionField>(Mat) {};

-	void MpcDeriv(GaugeField &Force,const FermionField &U,const FermionField &V) {
-	
-	  GridBase *fgrid   = this->_Mat.FermionGrid();
-	  GridBase *fcbgrid = this->_Mat.FermionRedBlackGrid();
-	  GridBase *ugrid   = this->_Mat.GaugeGrid();
-	  GridBase *ucbgrid = this->_Mat.GaugeRedBlackGrid();
+        void MpcDeriv(GaugeField &Force,const FermionField &U,const FermionField &V) {
+        
+          GridBase *fgrid   = this->_Mat.FermionGrid();
+          GridBase *fcbgrid = this->_Mat.FermionRedBlackGrid();

-	  Real coeff = 1.0;
+          FermionField tmp1(fcbgrid);
+          FermionField tmp2(fcbgrid);

-	  FermionField tmp1(fcbgrid);
-	  FermionField tmp2(fcbgrid);
+          conformable(fcbgrid,U._grid);
+          conformable(fcbgrid,V._grid);

-	  conformable(fcbgrid,U._grid);
-	  conformable(fcbgrid,V._grid);
+          // Assert the checkerboard?? or code for either
+          assert(U.checkerboard==Odd);
+          assert(V.checkerboard==U.checkerboard);

-	  // Assert the checkerboard?? or code for either
-	  assert(U.checkerboard==Odd);
-	  assert(V.checkerboard==U.checkerboard);
+          // NOTE Guido: WE DO NOT WANT TO USE THE ucbgrid GRID FOR THE FORCE
+          // it is not conformable with the HMC force field
+	  // Case: Ls vectorised fields
+          // INHERIT FROM THE Force field instead
+          GridRedBlackCartesian* forcecb = new GridRedBlackCartesian(Force._grid);
+          GaugeField ForceO(forcecb);
+          GaugeField ForceE(forcecb);

-	  GaugeField ForceO(ucbgrid);
-	  GaugeField ForceE(ucbgrid);

-	  //  X^dag Der_oe MeeInv Meo Y
-	  // Use Mooee as nontrivial but gauge field indept
-	  this->_Mat.Meooe   (V,tmp1);      // odd->even -- implicit -0.5 factor to be applied
+          //  X^dag Der_oe MeeInv Meo Y
+          // Use Mooee as nontrivial but gauge field indept
+          this->_Mat.Meooe   (V,tmp1);      // odd->even -- implicit -0.5 factor to be applied
 	  this->_Mat.MooeeInv(tmp1,tmp2);   // even->even 
-	  this->_Mat.MoeDeriv(ForceO,U,tmp2,DaggerNo);
-	  
-	  //  Accumulate X^dag M_oe MeeInv Der_eo Y
-	  this->_Mat.MeooeDag   (U,tmp1);    // even->odd -- implicit -0.5 factor to be applied
-	  this->_Mat.MooeeInvDag(tmp1,tmp2); // even->even 
-	  this->_Mat.MeoDeriv(ForceE,tmp2,V,DaggerNo);
-	  
-	  assert(ForceE.checkerboard==Even);
-	  assert(ForceO.checkerboard==Odd);
+          this->_Mat.MoeDeriv(ForceO,U,tmp2,DaggerNo);
+          //  Accumulate X^dag M_oe MeeInv Der_eo Y
+          this->_Mat.MeooeDag   (U,tmp1);    // even->odd -- implicit -0.5 factor to be applied
+          this->_Mat.MooeeInvDag(tmp1,tmp2); // even->even 
+          this->_Mat.MeoDeriv(ForceE,tmp2,V,DaggerNo);
+          
+          assert(ForceE.checkerboard==Even);
+          assert(ForceO.checkerboard==Odd);

-	  setCheckerboard(Force,ForceE); 
-	  setCheckerboard(Force,ForceO);
-	  Force=-Force;
-	}
+          setCheckerboard(Force,ForceE); 
+          setCheckerboard(Force,ForceO);
+          Force=-Force;
+
+          delete forcecb;
+        }


-	void MpcDagDeriv(GaugeField &Force,const FermionField &U,const FermionField &V) {
-	
-	  GridBase *fgrid   = this->_Mat.FermionGrid();
-	  GridBase *fcbgrid = this->_Mat.FermionRedBlackGrid();
-	  GridBase *ugrid   = this->_Mat.GaugeGrid();
-	  GridBase *ucbgrid = this->_Mat.GaugeRedBlackGrid();
+        void MpcDagDeriv(GaugeField &Force,const FermionField &U,const FermionField &V) {
+        
+          GridBase *fgrid   = this->_Mat.FermionGrid();
+          GridBase *fcbgrid = this->_Mat.FermionRedBlackGrid();

-	  Real coeff = 1.0;
+          FermionField tmp1(fcbgrid);
+          FermionField tmp2(fcbgrid);

-	  FermionField tmp1(fcbgrid);
-	  FermionField tmp2(fcbgrid);
+          conformable(fcbgrid,U._grid);
+          conformable(fcbgrid,V._grid);

-	  conformable(fcbgrid,U._grid);
-	  conformable(fcbgrid,V._grid);
+          // Assert the checkerboard?? or code for either
+          assert(V.checkerboard==Odd);
+          assert(V.checkerboard==V.checkerboard);

-	  // Assert the checkerboard?? or code for either
-	  assert(V.checkerboard==Odd);
-	  assert(V.checkerboard==V.checkerboard);
+          // NOTE Guido: WE DO NOT WANT TO USE THE ucbgrid GRID FOR THE FORCE
+          // it is not conformable with the HMC force field
+          // INHERIT FROM THE Force field instead
+	  GridRedBlackCartesian* forcecb = new GridRedBlackCartesian(Force._grid);
+          GaugeField ForceO(forcecb);
+          GaugeField ForceE(forcecb);

-	  GaugeField ForceO(ucbgrid);
-	  GaugeField ForceE(ucbgrid);
+          //  X^dag Der_oe MeeInv Meo Y
+          // Use Mooee as nontrivial but gauge field indept
+          this->_Mat.MeooeDag   (V,tmp1);      // odd->even -- implicit -0.5 factor to be applied
+          this->_Mat.MooeeInvDag(tmp1,tmp2);   // even->even 
+          this->_Mat.MoeDeriv(ForceO,U,tmp2,DaggerYes);
+          
+          //  Accumulate X^dag M_oe MeeInv Der_eo Y
+          this->_Mat.Meooe   (U,tmp1);    // even->odd -- implicit -0.5 factor to be applied
+          this->_Mat.MooeeInv(tmp1,tmp2); // even->even 
+          this->_Mat.MeoDeriv(ForceE,tmp2,V,DaggerYes);

-	  //  X^dag Der_oe MeeInv Meo Y
-	  // Use Mooee as nontrivial but gauge field indept
-	  this->_Mat.MeooeDag   (V,tmp1);      // odd->even -- implicit -0.5 factor to be applied
-	  this->_Mat.MooeeInvDag(tmp1,tmp2);   // even->even 
-	  this->_Mat.MoeDeriv(ForceO,U,tmp2,DaggerYes);
-	  
-	  //  Accumulate X^dag M_oe MeeInv Der_eo Y
-	  this->_Mat.Meooe   (U,tmp1);    // even->odd -- implicit -0.5 factor to be applied
-	  this->_Mat.MooeeInv(tmp1,tmp2); // even->even 
-	  this->_Mat.MeoDeriv(ForceE,tmp2,V,DaggerYes);
+          assert(ForceE.checkerboard==Even);
+          assert(ForceO.checkerboard==Odd);

-	  assert(ForceE.checkerboard==Even);
-	  assert(ForceO.checkerboard==Odd);
+          setCheckerboard(Force,ForceE); 
+          setCheckerboard(Force,ForceO);
+          Force=-Force;

-	  setCheckerboard(Force,ForceE); 
-	  setCheckerboard(Force,ForceO);
-	  Force=-Force;
-	}
+          delete forcecb;
+        }

    };

@@ -1,3 +1,4 @@
+
 /*************************************************************************************

 Grid physics library, www.github.com/paboyle/Grid
@@ -90,6 +91,19 @@ class OneFlavourEvenOddRationalPseudoFermionAction
    PowerNegQuarter.Init(remez, param.tolerance, true);
  };

+  virtual std::string action_name(){return "OneFlavourEvenOddRationalPseudoFermionAction";}
+
+  virtual std::string LogParameters(){
+    std::stringstream sstream;
+    sstream << GridLogMessage << "["<<action_name()<<"] Low            :" << param.lo <<  std::endl;
+    sstream << GridLogMessage << "["<<action_name()<<"] High           :" << param.hi <<  std::endl;
+    sstream << GridLogMessage << "["<<action_name()<<"] Max iterations :" << param.MaxIter <<  std::endl;
+    sstream << GridLogMessage << "["<<action_name()<<"] Tolerance      :" << param.tolerance <<  std::endl;
+    sstream << GridLogMessage << "["<<action_name()<<"] Degree         :" << param.degree <<  std::endl;
+    sstream << GridLogMessage << "["<<action_name()<<"] Precision      :" << param.precision <<  std::endl;
+    return sstream.str();
+  }
+  
  virtual void refresh(const GaugeField &U, GridParallelRNG &pRNG) {
    // P(phi) = e^{- phi^dag (MpcdagMpc)^-1/2 phi}
    //        = e^{- phi^dag (MpcdagMpc)^-1/4 (MpcdagMpc)^-1/4 phi}
@@ -87,6 +87,20 @@ namespace Grid{
   	PowerQuarter.Init(remez,param.tolerance,false);
 	PowerNegQuarter.Init(remez,param.tolerance,true);
      };
+
+      virtual std::string action_name(){return "OneFlavourEvenOddRatioRationalPseudoFermionAction";}
+
+      virtual std::string LogParameters(){
+	std::stringstream sstream;
+	sstream << GridLogMessage << "["<<action_name()<<"] Low            :" << param.lo <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] High           :" << param.hi <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Max iterations :" << param.MaxIter <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Tolerance      :" << param.tolerance <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Degree         :" << param.degree <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Precision      :" << param.precision <<  std::endl;
+	return sstream.str();
+      }
+      
      
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {

@@ -83,9 +83,25 @@ namespace Grid{
   	PowerQuarter.Init(remez,param.tolerance,false);
 	PowerNegQuarter.Init(remez,param.tolerance,true);
      };
+
+      virtual std::string action_name(){return "OneFlavourRationalPseudoFermionAction";}
+
+      virtual std::string LogParameters(){
+	std::stringstream sstream;
+	sstream << GridLogMessage << "["<<action_name()<<"] Low            :" << param.lo <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] High           :" << param.hi <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Max iterations :" << param.MaxIter <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Tolerance      :" << param.tolerance <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Degree         :" << param.degree <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Precision      :" << param.precision <<  std::endl;
+	return sstream.str();
+      }  
+
+
      
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {

+	
 	// P(phi) = e^{- phi^dag (MdagM)^-1/2 phi}
 	//        = e^{- phi^dag (MdagM)^-1/4 (MdagM)^-1/4 phi}
 	// Phi = Mdag^{1/4} eta 
@@ -81,7 +81,21 @@ namespace Grid{
   	PowerQuarter.Init(remez,param.tolerance,false);
 	PowerNegQuarter.Init(remez,param.tolerance,true);
      };
+
+      virtual std::string action_name(){return "OneFlavourRatioRationalPseudoFermionAction";}
      
+      virtual std::string LogParameters(){
+	std::stringstream sstream;
+	sstream << GridLogMessage << "["<<action_name()<<"] Low            :" << param.lo <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] High           :" << param.hi <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Max iterations :" << param.MaxIter <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Tolerance      :" << param.tolerance <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Degree         :" << param.degree <<  std::endl;
+	sstream << GridLogMessage << "["<<action_name()<<"] Precision      :" << param.precision <<  std::endl;
+	return sstream.str();
+      }
+      
+
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {

 	// S_f = chi^dag* P(V^dag*V)/Q(V^dag*V)* N(M^dag*M)/D(M^dag*M)* P(V^dag*V)/Q(V^dag*V)* chi       
@@ -62,6 +62,15 @@ class TwoFlavourPseudoFermionAction : public Action<typename Impl::GaugeField> {
        ActionSolver(AS),
        Phi(Op.FermionGrid()){};

+
+  virtual std::string action_name(){return "TwoFlavourPseudoFermionAction";}
+
+  virtual std::string LogParameters(){
+    std::stringstream sstream;
+    sstream << GridLogMessage << "["<<action_name()<<"] has no parameters" << std::endl;
+    return sstream.str();
+  }  
+  
  //////////////////////////////////////////////////////////////////////////////////////
  // Push the gauge field in to the dops. Assume any BC's and smearing already applied
  //////////////////////////////////////////////////////////////////////////////////////
@@ -80,7 +89,9 @@ class TwoFlavourPseudoFermionAction : public Action<typename Impl::GaugeField> {
    //         in the Phi integral, and thus is only an irrelevant prefactor for
    //         the partition function.
    //
+
    RealD scale = std::sqrt(0.5);
+
    FermionField eta(FermOp.FermionGrid());

    gaussian(pRNG, eta);
@@ -31,80 +31,89 @@ directory
 #define QCD_PSEUDOFERMION_TWO_FLAVOUR_EVEN_ODD_H

 namespace Grid {
-namespace QCD {
+  namespace QCD {

-////////////////////////////////////////////////////////////////////////
-// Two flavour pseudofermion action for any EO prec dop
-////////////////////////////////////////////////////////////////////////
-template <class Impl>
-class TwoFlavourEvenOddPseudoFermionAction
-    : public Action<typename Impl::GaugeField> {
- public:
-  INHERIT_IMPL_TYPES(Impl);
+    ////////////////////////////////////////////////////////////////////////
+    // Two flavour pseudofermion action for any EO prec dop
+    ////////////////////////////////////////////////////////////////////////
+    template <class Impl>
+    class TwoFlavourEvenOddPseudoFermionAction
+      : public Action<typename Impl::GaugeField> {
+    public:
+      INHERIT_IMPL_TYPES(Impl);

- private:
-  FermionOperator<Impl> &FermOp;  // the basic operator
+    private:
+      FermionOperator<Impl> &FermOp;  // the basic operator

-  OperatorFunction<FermionField> &DerivativeSolver;
-  OperatorFunction<FermionField> &ActionSolver;
+      OperatorFunction<FermionField> &DerivativeSolver;
+      OperatorFunction<FermionField> &ActionSolver;

-  FermionField PhiOdd;   // the pseudo fermion field for this trajectory
-  FermionField PhiEven;  // the pseudo fermion field for this trajectory
+      FermionField PhiOdd;   // the pseudo fermion field for this trajectory
+      FermionField PhiEven;  // the pseudo fermion field for this trajectory

- public:
-  /////////////////////////////////////////////////
-  // Pass in required objects.
-  /////////////////////////////////////////////////
-  TwoFlavourEvenOddPseudoFermionAction(FermionOperator<Impl> &Op,
-                                       OperatorFunction<FermionField> &DS,
-                                       OperatorFunction<FermionField> &AS)
-      : FermOp(Op),
-        DerivativeSolver(DS),
-        ActionSolver(AS),
-        PhiEven(Op.FermionRedBlackGrid()),
-	PhiOdd(Op.FermionRedBlackGrid())
-		  {};
+    public:
+      /////////////////////////////////////////////////
+      // Pass in required objects.
+      /////////////////////////////////////////////////
+      TwoFlavourEvenOddPseudoFermionAction(FermionOperator<Impl> &Op,
+					   OperatorFunction<FermionField> &DS,
+					   OperatorFunction<FermionField> &AS)
+	: FermOp(Op),
+	  DerivativeSolver(DS),
+	  ActionSolver(AS),
+	  PhiEven(Op.FermionRedBlackGrid()),
+	  PhiOdd(Op.FermionRedBlackGrid())
+      {};
+  
+      virtual std::string action_name(){return "TwoFlavourEvenOddPseudoFermionAction";}
      
+      virtual std::string LogParameters(){
+	std::stringstream sstream;
+	sstream << GridLogMessage << "["<<action_name()<<"] has no parameters" << std::endl;
+	return sstream.str();
+      }  
+
+
      //////////////////////////////////////////////////////////////////////////////////////
      // Push the gauge field in to the dops. Assume any BC's and smearing already applied
      //////////////////////////////////////////////////////////////////////////////////////
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {
-
+    
 	// P(phi) = e^{- phi^dag (MpcdagMpc)^-1 phi}
 	// Phi = McpDag eta 
 	// P(eta) = e^{- eta^dag eta}
 	//
 	// e^{x^2/2 sig^2} => sig^2 = 0.5.
-
+    
 	RealD scale = std::sqrt(0.5);
-
+    
 	FermionField eta    (FermOp.FermionGrid());
 	FermionField etaOdd (FermOp.FermionRedBlackGrid());
 	FermionField etaEven(FermOp.FermionRedBlackGrid());
-
+    
 	gaussian(pRNG,eta);
 	pickCheckerboard(Even,etaEven,eta);
 	pickCheckerboard(Odd,etaOdd,eta);
-
+    
 	FermOp.ImportGauge(U);
 	SchurDifferentiableOperator<Impl> PCop(FermOp);
-	
-
+    
+    
 	PCop.MpcDag(etaOdd,PhiOdd);
-
+    
 	FermOp.MooeeDag(etaEven,PhiEven);
-
+    
 	PhiOdd =PhiOdd*scale;
 	PhiEven=PhiEven*scale;
-
+    
      };
-
+  
      //////////////////////////////////////////////////////
      // S = phi^dag (Mdag M)^-1 phi  (odd)
      //   + phi^dag (Mdag M)^-1 phi  (even)
      //////////////////////////////////////////////////////
      virtual RealD S(const GaugeField &U) {
-
+	
 	FermOp.ImportGauge(U);

 	FermionField X(FermOp.FermionRedBlackGrid());
@@ -135,7 +144,6 @@ class TwoFlavourEvenOddPseudoFermionAction
      //
      //////////////////////////////////////////////////////
      virtual void deriv(const GaugeField &U,GaugeField & dSdU) {
-
 	FermOp.ImportGauge(U);

 	FermionField X(FermOp.FermionRedBlackGrid());
@@ -150,8 +158,8 @@ class TwoFlavourEvenOddPseudoFermionAction
 	X=zero;
 	DerivativeSolver(Mpc,PhiOdd,X);
 	Mpc.Mpc(X,Y);
-  	Mpc.MpcDeriv(tmp , Y, X );    dSdU=tmp;
-	Mpc.MpcDagDeriv(tmp , X, Y);  dSdU=dSdU+tmp;
+  Mpc.MpcDeriv(tmp , Y, X );    dSdU=tmp;
+  Mpc.MpcDagDeriv(tmp , X, Y);  dSdU=dSdU+tmp;

 	// Treat the EE case. (MdagM)^-1 = Minv Minvdag
 	// Deriv defaults to zero.
@@ -163,10 +171,10 @@ class TwoFlavourEvenOddPseudoFermionAction
 	assert(FermOp.ConstEE() == 1);

 	/*
-        FermOp.MooeeInvDag(PhiOdd,Y);
-        FermOp.MooeeInv(Y,X);
-  	FermOp.MeeDeriv(tmp , Y, X,DaggerNo );    dSdU=tmp;
-	FermOp.MeeDeriv(tmp , X, Y,DaggerYes);  dSdU=dSdU+tmp;
+	  FermOp.MooeeInvDag(PhiOdd,Y);
+	  FermOp.MooeeInv(Y,X);
+	  FermOp.MeeDeriv(tmp , Y, X,DaggerNo );    dSdU=tmp;
+	  FermOp.MeeDeriv(tmp , X, Y,DaggerYes);  dSdU=dSdU+tmp;
 	*/
 	
 	//dSdU = Ta(dSdU);
@@ -52,66 +52,75 @@ namespace Grid{

    public:
      TwoFlavourEvenOddRatioPseudoFermionAction(FermionOperator<Impl>  &_NumOp, 
-						FermionOperator<Impl>  &_DenOp, 
-						OperatorFunction<FermionField> & DS,
-						OperatorFunction<FermionField> & AS) :
+                                                FermionOperator<Impl>  &_DenOp, 
+                                                OperatorFunction<FermionField> & DS,
+                                                OperatorFunction<FermionField> & AS) :
      NumOp(_NumOp), 
      DenOp(_DenOp), 
      DerivativeSolver(DS), 
      ActionSolver(AS),
      PhiEven(_NumOp.FermionRedBlackGrid()),
      PhiOdd(_NumOp.FermionRedBlackGrid()) 
-	{
-	  conformable(_NumOp.FermionGrid(), _DenOp.FermionGrid());
-	  conformable(_NumOp.FermionRedBlackGrid(), _DenOp.FermionRedBlackGrid());
-	  conformable(_NumOp.GaugeGrid(), _DenOp.GaugeGrid());
-	  conformable(_NumOp.GaugeRedBlackGrid(), _DenOp.GaugeRedBlackGrid());
-	};
+        {
+          conformable(_NumOp.FermionGrid(), _DenOp.FermionGrid());
+          conformable(_NumOp.FermionRedBlackGrid(), _DenOp.FermionRedBlackGrid());
+          conformable(_NumOp.GaugeGrid(), _DenOp.GaugeGrid());
+          conformable(_NumOp.GaugeRedBlackGrid(), _DenOp.GaugeRedBlackGrid());
+        };
+
+      virtual std::string action_name(){return "TwoFlavourEvenOddRatioPseudoFermionAction";}
+
+      virtual std::string LogParameters(){
+	std::stringstream sstream;
+	sstream << GridLogMessage << "["<<action_name()<<"] has no parameters" << std::endl;
+	return sstream.str();
+      } 
+
      
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {

-	// P(phi) = e^{- phi^dag Vpc (MpcdagMpc)^-1 Vpcdag phi}
-	//
-	// NumOp == V
-	// DenOp == M
-	//
-	// Take phi_o = Vpcdag^{-1} Mpcdag eta_o  ; eta_o = Mpcdag^{-1} Vpcdag Phi
-	//
-	// P(eta_o) = e^{- eta_o^dag eta_o}
-	//
-	// e^{x^2/2 sig^2} => sig^2 = 0.5.
-	// 
-	RealD scale = std::sqrt(0.5);
+        // P(phi) = e^{- phi^dag Vpc (MpcdagMpc)^-1 Vpcdag phi}
+        //
+        // NumOp == V
+        // DenOp == M
+        //
+        // Take phi_o = Vpcdag^{-1} Mpcdag eta_o  ; eta_o = Mpcdag^{-1} Vpcdag Phi
+        //
+        // P(eta_o) = e^{- eta_o^dag eta_o}
+        //
+        // e^{x^2/2 sig^2} => sig^2 = 0.5.
+        // 
+        RealD scale = std::sqrt(0.5);

-	FermionField eta    (NumOp.FermionGrid());
-	FermionField etaOdd (NumOp.FermionRedBlackGrid());
-	FermionField etaEven(NumOp.FermionRedBlackGrid());
-	FermionField tmp    (NumOp.FermionRedBlackGrid());
+        FermionField eta    (NumOp.FermionGrid());
+        FermionField etaOdd (NumOp.FermionRedBlackGrid());
+        FermionField etaEven(NumOp.FermionRedBlackGrid());
+        FermionField tmp    (NumOp.FermionRedBlackGrid());

-	gaussian(pRNG,eta);
+        gaussian(pRNG,eta);

-	pickCheckerboard(Even,etaEven,eta);
-	pickCheckerboard(Odd,etaOdd,eta);
+        pickCheckerboard(Even,etaEven,eta);
+        pickCheckerboard(Odd,etaOdd,eta);

-	NumOp.ImportGauge(U);
-	DenOp.ImportGauge(U);
+        NumOp.ImportGauge(U);
+        DenOp.ImportGauge(U);

-	SchurDifferentiableOperator<Impl> Mpc(DenOp);
-	SchurDifferentiableOperator<Impl> Vpc(NumOp);
+        SchurDifferentiableOperator<Impl> Mpc(DenOp);
+        SchurDifferentiableOperator<Impl> Vpc(NumOp);

-	// Odd det factors
-	Mpc.MpcDag(etaOdd,PhiOdd);
-	tmp=zero;
-	ActionSolver(Vpc,PhiOdd,tmp);
-	Vpc.Mpc(tmp,PhiOdd);            
+        // Odd det factors
+        Mpc.MpcDag(etaOdd,PhiOdd);
+        tmp=zero;
+        ActionSolver(Vpc,PhiOdd,tmp);
+        Vpc.Mpc(tmp,PhiOdd);            

-	// Even det factors
-	DenOp.MooeeDag(etaEven,tmp);
-	NumOp.MooeeInvDag(tmp,PhiEven);
+        // Even det factors
+        DenOp.MooeeDag(etaEven,tmp);
+        NumOp.MooeeInvDag(tmp,PhiEven);

-	PhiOdd =PhiOdd*scale;
-	PhiEven=PhiEven*scale;
-	
+        PhiOdd =PhiOdd*scale;
+        PhiEven=PhiEven*scale;
+        
      };

      //////////////////////////////////////////////////////
@@ -119,33 +128,33 @@ namespace Grid{
      //////////////////////////////////////////////////////
      virtual RealD S(const GaugeField &U) {

-	NumOp.ImportGauge(U);
-	DenOp.ImportGauge(U);
+        NumOp.ImportGauge(U);
+        DenOp.ImportGauge(U);

-	SchurDifferentiableOperator<Impl> Mpc(DenOp);
-	SchurDifferentiableOperator<Impl> Vpc(NumOp);
+        SchurDifferentiableOperator<Impl> Mpc(DenOp);
+        SchurDifferentiableOperator<Impl> Vpc(NumOp);

-	FermionField X(NumOp.FermionRedBlackGrid());
-	FermionField Y(NumOp.FermionRedBlackGrid());
+        FermionField X(NumOp.FermionRedBlackGrid());
+        FermionField Y(NumOp.FermionRedBlackGrid());

-	Vpc.MpcDag(PhiOdd,Y);           // Y= Vdag phi
-	X=zero;
-	ActionSolver(Mpc,Y,X);          // X= (MdagM)^-1 Vdag phi
-	//Mpc.Mpc(X,Y);                   // Y=  Mdag^-1 Vdag phi
-	// Multiply by Ydag
-	RealD action = real(innerProduct(Y,X));
+        Vpc.MpcDag(PhiOdd,Y);           // Y= Vdag phi
+        X=zero;
+        ActionSolver(Mpc,Y,X);          // X= (MdagM)^-1 Vdag phi
+        //Mpc.Mpc(X,Y);                   // Y=  Mdag^-1 Vdag phi
+        // Multiply by Ydag
+        RealD action = real(innerProduct(Y,X));

-	//RealD action = norm2(Y);
+        //RealD action = norm2(Y);

-	// The EE factorised block; normally can replace with zero if det is constant (gauge field indept)
-	// Only really clover term that creates this. Leave the EE portion as a future to do to make most
-	// rapid progresss on DWF for now.
-	//
-	NumOp.MooeeDag(PhiEven,X);
-	DenOp.MooeeInvDag(X,Y);
-	action = action + norm2(Y);
+        // The EE factorised block; normally can replace with zero if det is constant (gauge field indept)
+        // Only really clover term that creates this. Leave the EE portion as a future to do to make most
+        // rapid progresss on DWF for now.
+        //
+        NumOp.MooeeDag(PhiEven,X);
+        DenOp.MooeeInvDag(X,Y);
+        action = action + norm2(Y);

-	return action;
+        return action;
      };

      //////////////////////////////////////////////////////
@@ -155,44 +164,44 @@ namespace Grid{
      //////////////////////////////////////////////////////
      virtual void deriv(const GaugeField &U,GaugeField & dSdU) {

-	NumOp.ImportGauge(U);
-	DenOp.ImportGauge(U);
+        NumOp.ImportGauge(U);
+        DenOp.ImportGauge(U);

-	SchurDifferentiableOperator<Impl> Mpc(DenOp);
-	SchurDifferentiableOperator<Impl> Vpc(NumOp);
+        SchurDifferentiableOperator<Impl> Mpc(DenOp);
+        SchurDifferentiableOperator<Impl> Vpc(NumOp);

-	FermionField  X(NumOp.FermionRedBlackGrid());
-	FermionField  Y(NumOp.FermionRedBlackGrid());
+        FermionField  X(NumOp.FermionRedBlackGrid());
+        FermionField  Y(NumOp.FermionRedBlackGrid());

-	GaugeField   force(NumOp.GaugeGrid());	
+        // This assignment is necessary to be compliant with the HMC grids
+	GaugeField force(dSdU._grid);

-	//Y=Vdag phi
-	//X = (Mdag M)^-1 V^dag phi
-	//Y = (Mdag)^-1 V^dag  phi
-	Vpc.MpcDag(PhiOdd,Y);          // Y= Vdag phi
-	X=zero;
-	DerivativeSolver(Mpc,Y,X);     // X= (MdagM)^-1 Vdag phi
-	Mpc.Mpc(X,Y);                  // Y=  Mdag^-1 Vdag phi
+        //Y=Vdag phi
+        //X = (Mdag M)^-1 V^dag phi
+        //Y = (Mdag)^-1 V^dag  phi
+        Vpc.MpcDag(PhiOdd,Y);          // Y= Vdag phi
+        X=zero;
+        DerivativeSolver(Mpc,Y,X);     // X= (MdagM)^-1 Vdag phi
+        Mpc.Mpc(X,Y);                  // Y=  Mdag^-1 Vdag phi

-	// phi^dag V (Mdag M)^-1 dV^dag  phi
-	Vpc.MpcDagDeriv(force , X, PhiOdd );  dSdU=force;
+        // phi^dag V (Mdag M)^-1 dV^dag  phi
+        Vpc.MpcDagDeriv(force , X, PhiOdd );   dSdU = force;
  
-	// phi^dag dV (Mdag M)^-1 V^dag  phi
-	Vpc.MpcDeriv(force , PhiOdd, X );  dSdU=dSdU+force;
+        // phi^dag dV (Mdag M)^-1 V^dag  phi
+        Vpc.MpcDeriv(force , PhiOdd, X );      dSdU = dSdU+force;

-	//    -    phi^dag V (Mdag M)^-1 Mdag dM   (Mdag M)^-1 V^dag  phi
-	//    -    phi^dag V (Mdag M)^-1 dMdag M   (Mdag M)^-1 V^dag  phi
-	Mpc.MpcDeriv(force,Y,X);   dSdU=dSdU-force;
-	Mpc.MpcDagDeriv(force,X,Y);  dSdU=dSdU-force;
+        //    -    phi^dag V (Mdag M)^-1 Mdag dM   (Mdag M)^-1 V^dag  phi
+        //    -    phi^dag V (Mdag M)^-1 dMdag M   (Mdag M)^-1 V^dag  phi
+        Mpc.MpcDeriv(force,Y,X);              dSdU = dSdU-force;
+        Mpc.MpcDagDeriv(force,X,Y);           dSdU = dSdU-force;

-	// FIXME No force contribution from EvenEven assumed here
-	// Needs a fix for clover.
-	assert(NumOp.ConstEE() == 1);
-	assert(DenOp.ConstEE() == 1);
+        // FIXME No force contribution from EvenEven assumed here
+        // Needs a fix for clover.
+        assert(NumOp.ConstEE() == 1);
+        assert(DenOp.ConstEE() == 1);

-	//dSdU = -Ta(dSdU);
-	dSdU = -dSdU;
-	
+        dSdU = -dSdU;
+        
      };
    };
  }
@@ -57,6 +57,14 @@ namespace Grid{
 					 OperatorFunction<FermionField> & AS
 					 ) : NumOp(_NumOp), DenOp(_DenOp), DerivativeSolver(DS), ActionSolver(AS), Phi(_NumOp.FermionGrid()) {};
      
+      virtual std::string action_name(){return "TwoFlavourRatioPseudoFermionAction";}
+
+      virtual std::string LogParameters(){
+	std::stringstream sstream;
+	sstream << GridLogMessage << "["<<action_name()<<"] has no parameters" << std::endl;
+	return sstream.str();
+      }  
+      
      virtual void refresh(const GaugeField &U, GridParallelRNG& pRNG) {

 	// P(phi) = e^{- phi^dag V (MdagM)^-1 Vdag phi}
@@ -0,0 +1,50 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/action/gauge/Scalar.h
+
+Copyright (C) 2017
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_QCD_SCALAR_H
+#define GRID_QCD_SCALAR_H
+
+#include <Grid/qcd/action/scalar/ScalarImpl.h>
+#include <Grid/qcd/action/scalar/ScalarAction.h>
+#include <Grid/qcd/action/scalar/ScalarInteractionAction.h>
+
+namespace Grid {
+namespace QCD {
+
+  typedef ScalarAction<ScalarImplR>                 ScalarActionR;
+  typedef ScalarAction<ScalarImplF>                 ScalarActionF;
+  typedef ScalarAction<ScalarImplD>                 ScalarActionD;
+
+  template <int Colours, int Dimensions> using ScalarAdjActionR = ScalarInteractionAction<ScalarNxNAdjImplR<Colours>, Dimensions>;
+  template <int Colours, int Dimensions> using ScalarAdjActionF = ScalarInteractionAction<ScalarNxNAdjImplF<Colours>, Dimensions>;
+  template <int Colours, int Dimensions> using ScalarAdjActionD = ScalarInteractionAction<ScalarNxNAdjImplD<Colours>, Dimensions>;
+  
+}
+}
+
+#endif  // GRID_QCD_SCALAR_H
@@ -0,0 +1,83 @@
+/*************************************************************************************
+
+  Grid physics library, www.github.com/paboyle/Grid
+
+  Source file: ./lib/qcd/action/gauge/WilsonGaugeAction.h
+
+  Copyright (C) 2015
+
+  Author: Azusa Yamaguchi <ayamaguc@staffmail.ed.ac.uk>
+  Author: Peter Boyle <paboyle@ph.ed.ac.uk>
+  Author: neo <cossu@post.kek.jp>
+  Author: paboyle <paboyle@ph.ed.ac.uk>
+
+  This program is free software; you can redistribute it and/or modify
+  it under the terms of the GNU General Public License as published by
+  the Free Software Foundation; either version 2 of the License, or
+  (at your option) any later version.
+
+  This program is distributed in the hope that it will be useful,
+  but WITHOUT ANY WARRANTY; without even the implied warranty of
+  MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+  GNU General Public License for more details.
+
+  You should have received a copy of the GNU General Public License along
+  with this program; if not, write to the Free Software Foundation, Inc.,
+  51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+  See the full license in the file "LICENSE" in the top level distribution
+directory
+  *************************************************************************************/
+/*  END LEGAL */
+
+#ifndef SCALAR_ACTION_H
+#define SCALAR_ACTION_H
+
+namespace Grid {
+  // FIXME drop the QCD namespace everywhere here
+
+template <class Impl>
+class ScalarAction : public QCD::Action<typename Impl::Field> {
+ public:
+    INHERIT_FIELD_TYPES(Impl);
+
+ private:
+    RealD mass_square;
+    RealD lambda;
+
+ public:
+    ScalarAction(RealD ms, RealD l) : mass_square(ms), lambda(l) {}
+
+    virtual std::string LogParameters() {
+      std::stringstream sstream;
+      sstream << GridLogMessage << "[ScalarAction] lambda      : " << lambda      << std::endl;
+      sstream << GridLogMessage << "[ScalarAction] mass_square : " << mass_square << std::endl;
+      return sstream.str();
+    }
+    virtual std::string action_name() {return "ScalarAction";}
+
+    virtual void refresh(const Field &U, GridParallelRNG &pRNG) {}  // noop as no pseudoferms
+
+    virtual RealD S(const Field &p) {
+      return (mass_square * 0.5 + QCD::Nd) * ScalarObs<Impl>::sumphisquared(p) +
+    (lambda / 24.) * ScalarObs<Impl>::sumphifourth(p) +
+    ScalarObs<Impl>::sumphider(p);
+    };
+
+    virtual void deriv(const Field &p,
+                       Field &force) {
+      Field tmp(p._grid);
+      Field p2(p._grid);
+      ScalarObs<Impl>::phisquared(p2, p);
+      tmp = -(Cshift(p, 0, -1) + Cshift(p, 0, 1));
+      for (int mu = 1; mu < QCD::Nd; mu++) tmp -= Cshift(p, mu, -1) + Cshift(p, mu, 1);
+
+      force =+(mass_square + 2. * QCD::Nd) * p + (lambda / 6.) * p2 * p + tmp;
+    }
+};
+
+
+
+}  // namespace Grid
+
+#endif // SCALAR_ACTION_H
@@ -0,0 +1,162 @@
+#ifndef SCALAR_IMPL
+#define SCALAR_IMPL
+
+
+namespace Grid {
+  //namespace QCD {
+
+template <class S>
+class ScalarImplTypes {
+ public:
+    typedef S Simd;
+
+    template <typename vtype>
+    using iImplField = iScalar<iScalar<iScalar<vtype> > >;
+
+    typedef iImplField<Simd> SiteField;
+    typedef SiteField        SitePropagator;
+    typedef SiteField        SiteComplex;
+    
+    typedef Lattice<SiteField> Field;
+    typedef Field              ComplexField;
+    typedef Field              FermionField;
+    typedef Field              PropagatorField;
+    
+    static inline void generate_momenta(Field& P, GridParallelRNG& pRNG){
+      gaussian(pRNG, P);
+    }
+
+    static inline Field projectForce(Field& P){return P;}
+
+    static inline void update_field(Field& P, Field& U, double ep) {
+      U += P*ep;
+    }
+
+    static inline RealD FieldSquareNorm(Field& U) {
+      return (- sum(trace(U*U))/2.0);
+    }
+
+    static inline void HotConfiguration(GridParallelRNG &pRNG, Field &U) {
+      gaussian(pRNG, U);
+    }
+
+    static inline void TepidConfiguration(GridParallelRNG &pRNG, Field &U) {
+      gaussian(pRNG, U);
+    }
+
+    static inline void ColdConfiguration(GridParallelRNG &pRNG, Field &U) {
+      U = 1.0;
+    }
+    
+    static void MomentumSpacePropagator(Field &out, RealD m)
+    {
+      GridBase           *grid = out._grid;
+      Field              kmu(grid), one(grid);
+      const unsigned int nd    = grid->_ndimension;
+      std::vector<int>   &l    = grid->_fdimensions;
+      
+      one = Complex(1.0,0.0);
+      out = m*m;
+      for(int mu = 0; mu < nd; mu++)
+      {
+        Real twoPiL = M_PI*2./l[mu];
+        
+        LatticeCoordinate(kmu,mu);
+        kmu = 2.*sin(.5*twoPiL*kmu);
+        out = out + kmu*kmu;
+      }
+      out = one/out;
+    }
+    
+    static void FreePropagator(const Field &in, Field &out,
+                               const Field &momKernel)
+    {
+      FFT   fft((GridCartesian *)in._grid);
+      Field inFT(in._grid);
+      
+      fft.FFT_all_dim(inFT, in, FFT::forward);
+      inFT = inFT*momKernel;
+      fft.FFT_all_dim(out, inFT, FFT::backward);
+    }
+    
+    static void FreePropagator(const Field &in, Field &out, RealD m)
+    {
+      Field momKernel(in._grid);
+      
+      MomentumSpacePropagator(momKernel, m);
+      FreePropagator(in, out, momKernel);
+    }
+    
+  };
+
+  template <class S, unsigned int N>
+  class ScalarAdjMatrixImplTypes {
+  public:
+    typedef S Simd;
+    typedef QCD::SU<N> Group;
+    
+    template <typename vtype>
+    using iImplField   = iScalar<iScalar<iMatrix<vtype, N>>>;
+    template <typename vtype>
+    using iImplComplex = iScalar<iScalar<iScalar<vtype>>>;
+
+    typedef iImplField<Simd>   SiteField;
+    typedef SiteField          SitePropagator;
+    typedef iImplComplex<Simd> SiteComplex;
+    
+    typedef Lattice<SiteField>   Field;
+    typedef Lattice<SiteComplex> ComplexField;
+    typedef Field                FermionField;
+    typedef Field                PropagatorField;
+
+    static inline void generate_momenta(Field& P, GridParallelRNG& pRNG) {
+      Group::GaussianFundamentalLieAlgebraMatrix(pRNG, P);
+    }
+
+    static inline Field projectForce(Field& P) {return P;}
+
+    static inline void update_field(Field& P, Field& U, double ep) {
+      U += P*ep;
+    }
+
+    static inline RealD FieldSquareNorm(Field& U) {
+      return (TensorRemove(sum(trace(U*U))).real());
+    }
+
+    static inline void HotConfiguration(GridParallelRNG &pRNG, Field &U) {
+      Group::GaussianFundamentalLieAlgebraMatrix(pRNG, U);
+    }
+
+    static inline void TepidConfiguration(GridParallelRNG &pRNG, Field &U) {
+      Group::GaussianFundamentalLieAlgebraMatrix(pRNG, U, 0.01);
+    }
+
+    static inline void ColdConfiguration(GridParallelRNG &pRNG, Field &U) {
+      U = zero;
+    }
+
+  };
+
+
+
+
+  typedef ScalarImplTypes<vReal> ScalarImplR;
+  typedef ScalarImplTypes<vRealF> ScalarImplF;
+  typedef ScalarImplTypes<vRealD> ScalarImplD;
+  typedef ScalarImplTypes<vComplex> ScalarImplCR;
+  typedef ScalarImplTypes<vComplexF> ScalarImplCF;
+  typedef ScalarImplTypes<vComplexD> ScalarImplCD;
+    
+  // Hardcoding here the size of the matrices
+  typedef ScalarAdjMatrixImplTypes<vComplex,  QCD::Nc> ScalarAdjImplR;
+  typedef ScalarAdjMatrixImplTypes<vComplexF, QCD::Nc> ScalarAdjImplF;
+  typedef ScalarAdjMatrixImplTypes<vComplexD, QCD::Nc> ScalarAdjImplD;
+
+  template <int Colours > using ScalarNxNAdjImplR = ScalarAdjMatrixImplTypes<vComplex,   Colours >;
+  template <int Colours > using ScalarNxNAdjImplF = ScalarAdjMatrixImplTypes<vComplexF,  Colours >;
+  template <int Colours > using ScalarNxNAdjImplD = ScalarAdjMatrixImplTypes<vComplexD,  Colours >;
+  
+  //}
+}
+
+#endif
@@ -0,0 +1,148 @@
+/*************************************************************************************
+
+  Grid physics library, www.github.com/paboyle/Grid
+
+  Source file: ./lib/qcd/action/gauge/WilsonGaugeAction.h
+
+  Copyright (C) 2015
+
+  Author: Guido Cossu <guido,cossu@ed.ac.uk>
+
+  This program is free software; you can redistribute it and/or modify
+  it under the terms of the GNU General Public License as published by
+  the Free Software Foundation; either version 2 of the License, or
+  (at your option) any later version.
+
+  This program is distributed in the hope that it will be useful,
+  but WITHOUT ANY WARRANTY; without even the implied warranty of
+  MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+  GNU General Public License for more details.
+
+  You should have received a copy of the GNU General Public License along
+  with this program; if not, write to the Free Software Foundation, Inc.,
+  51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+  See the full license in the file "LICENSE" in the top level distribution
+directory
+  *************************************************************************************/
+/*  END LEGAL */
+
+#ifndef SCALAR_INT_ACTION_H
+#define SCALAR_INT_ACTION_H
+
+
+// Note: this action can completely absorb the ScalarAction for real float fields
+// use the scalarObjs to generalise the structure
+
+namespace Grid {
+  // FIXME drop the QCD namespace everywhere here
+
+  template <class Impl, int Ndim >
+  class ScalarInteractionAction : public QCD::Action<typename Impl::Field> {
+  public:
+    INHERIT_FIELD_TYPES(Impl);
+  private:
+    RealD mass_square;
+    RealD lambda;
+
+
+    typedef typename Field::vector_object vobj;
+    typedef CartesianStencil<vobj,vobj> Stencil;
+
+    SimpleCompressor<vobj> compressor;
+    int npoint = 2*Ndim;
+    std::vector<int> directions;//    = {0,1,2,3,0,1,2,3};  // forcing 4 dimensions
+    std::vector<int> displacements;//  = {1,1,1,1, -1,-1,-1,-1};
+
+
+  public:
+
+    ScalarInteractionAction(RealD ms, RealD l) : mass_square(ms), lambda(l), displacements(2*Ndim,0), directions(2*Ndim,0){
+      for (int mu = 0 ; mu < Ndim; mu++){
+		directions[mu]         = mu; directions[mu+Ndim]    = mu;
+		displacements[mu]      =  1; displacements[mu+Ndim] = -1;
+      }
+    }
+
+    virtual std::string LogParameters() {
+      std::stringstream sstream;
+      sstream << GridLogMessage << "[ScalarAction] lambda      : " << lambda      << std::endl;
+      sstream << GridLogMessage << "[ScalarAction] mass_square : " << mass_square << std::endl;
+      return sstream.str();
+    }
+
+    virtual std::string action_name() {return "ScalarAction";}
+
+    virtual void refresh(const Field &U, GridParallelRNG &pRNG) {}
+
+    virtual RealD S(const Field &p) {
+      assert(p._grid->Nd() == Ndim);
+      static Stencil phiStencil(p._grid, npoint, 0, directions, displacements);
+      phiStencil.HaloExchange(p, compressor);
+      Field action(p._grid), pshift(p._grid), phisquared(p._grid);
+      phisquared = p*p;
+      action = (2.0*Ndim + mass_square)*phisquared - lambda/24.*phisquared*phisquared;
+      for (int mu = 0; mu < Ndim; mu++) {
+	//  pshift = Cshift(p, mu, +1);  // not efficient, implement with stencils
+	parallel_for (int i = 0; i < p._grid->oSites(); i++) {
+	  int permute_type;
+	  StencilEntry *SE;
+	  vobj temp2;
+	  const vobj *temp, *t_p;
+	    
+	  SE = phiStencil.GetEntry(permute_type, mu, i);
+	  t_p  = &p._odata[i];
+	  if ( SE->_is_local ) {
+	    temp = &p._odata[SE->_offset];
+	    if ( SE->_permute ) {
+	      permute(temp2, *temp, permute_type);
+	      action._odata[i] -= temp2*(*t_p) + (*t_p)*temp2;
+	    } else {
+	      action._odata[i] -= (*temp)*(*t_p) + (*t_p)*(*temp);
+	    }
+	  } else {
+	    action._odata[i] -= phiStencil.CommBuf()[SE->_offset]*(*t_p) + (*t_p)*phiStencil.CommBuf()[SE->_offset];
+	  }
+	}
+	//  action -= pshift*p + p*pshift;
+      }
+      // NB the trace in the algebra is normalised to 1/2
+      // minus sign coming from the antihermitian fields
+      return -(TensorRemove(sum(trace(action)))).real();
+    };
+
+    virtual void deriv(const Field &p, Field &force) {
+      assert(p._grid->Nd() == Ndim);
+      force = (2.0*Ndim + mass_square)*p - lambda/12.*p*p*p;
+      // move this outside
+      static Stencil phiStencil(p._grid, npoint, 0, directions, displacements);
+      phiStencil.HaloExchange(p, compressor);
+      
+      //for (int mu = 0; mu < QCD::Nd; mu++) force -= Cshift(p, mu, -1) + Cshift(p, mu, 1);
+      for (int point = 0; point < npoint; point++) {
+	parallel_for (int i = 0; i < p._grid->oSites(); i++) {
+	  const vobj *temp;
+	  vobj temp2;
+	  int permute_type;
+	  StencilEntry *SE;
+	  SE = phiStencil.GetEntry(permute_type, point, i);
+	  
+	  if ( SE->_is_local ) {
+	    temp = &p._odata[SE->_offset];
+	    if ( SE->_permute ) {
+	      permute(temp2, *temp, permute_type);
+	      force._odata[i] -= temp2;
+	    } else {
+	      force._odata[i] -= *temp;
+	    }
+	  } else {
+	    force._odata[i] -= phiStencil.CommBuf()[SE->_offset];
+	  }
+	}
+      }
+    }
+  };
+  
+}  // namespace Grid
+
+#endif  // SCALAR_INT_ACTION_H
@@ -0,0 +1,219 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/GenericHmcRunner.h
+
+Copyright (C) 2015
+
+Author: paboyle <paboyle@ph.ed.ac.uk>
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+  See the full license in the file "LICENSE" in the top level distribution
+  directory
+  *************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_GENERIC_HMC_RUNNER
+#define GRID_GENERIC_HMC_RUNNER
+
+#include <unordered_map>
+
+namespace Grid {
+namespace QCD {
+
+
+// very ugly here but possibly resolved if we had a base Reader class
+template < class ReaderClass >
+class HMCRunnerBase {
+public:
+  virtual void Run() = 0;
+  virtual void initialize(ReaderClass& ) = 0;
+};
+
+
+template <class Implementation,
+          template <typename, typename, typename> class Integrator,
+          class RepresentationsPolicy = NoHirep, class ReaderClass = XmlReader>
+class HMCWrapperTemplate: public HMCRunnerBase<ReaderClass> {
+ public:
+  INHERIT_FIELD_TYPES(Implementation);
+  typedef Implementation ImplPolicy;  // visible from outside
+  template <typename S = NoSmearing<Implementation> >
+  using IntegratorType = Integrator<Implementation, S, RepresentationsPolicy>;
+
+  HMCparameters Parameters;
+  std::string ParameterFile;
+  HMCResourceManager<Implementation> Resources;
+
+  // The set of actions (keep here for lower level users, for now)
+  ActionSet<Field, RepresentationsPolicy> TheAction;
+
+  HMCWrapperTemplate() = default;
+
+  HMCWrapperTemplate(HMCparameters Par){
+    Parameters = Par;
+  }
+
+  void initialize(ReaderClass & TheReader){
+    std::cout  << "Initialization of the HMC" << std::endl;
+    Resources.initialize(TheReader);
+
+    // eventually add smearing
+
+    Resources.GetActionSet(TheAction);    
+  }
+
+
+  void ReadCommandLine(int argc, char **argv) {
+    std::string arg;
+
+    if (GridCmdOptionExists(argv, argv + argc, "--StartingType")) {
+      arg = GridCmdOptionPayload(argv, argv + argc, "--StartingType");
+
+      if (arg != "HotStart" && arg != "ColdStart" && arg != "TepidStart" &&
+          arg != "CheckpointStart") {
+        std::cout << GridLogError << "Unrecognized option in --StartingType\n";
+        std::cout
+            << GridLogError
+            << "Valid [HotStart, ColdStart, TepidStart, CheckpointStart]\n";
+        exit(1);
+      }
+      Parameters.StartingType = arg;
+    }
+
+    if (GridCmdOptionExists(argv, argv + argc, "--StartingTrajectory")) {
+      arg = GridCmdOptionPayload(argv, argv + argc, "--StartingTrajectory");
+      std::vector<int> ivec(0);
+      GridCmdOptionIntVector(arg, ivec);
+      Parameters.StartTrajectory = ivec[0];
+    }
+
+    if (GridCmdOptionExists(argv, argv + argc, "--Trajectories")) {
+      arg = GridCmdOptionPayload(argv, argv + argc, "--Trajectories");
+      std::vector<int> ivec(0);
+      GridCmdOptionIntVector(arg, ivec);
+      Parameters.Trajectories = ivec[0];
+    }
+
+    if (GridCmdOptionExists(argv, argv + argc, "--Thermalizations")) {
+      arg = GridCmdOptionPayload(argv, argv + argc, "--Thermalizations");
+      std::vector<int> ivec(0);
+      GridCmdOptionIntVector(arg, ivec);
+      Parameters.NoMetropolisUntil = ivec[0];
+    }
+    if (GridCmdOptionExists(argv, argv + argc, "--ParameterFile")) {
+      arg = GridCmdOptionPayload(argv, argv + argc, "--ParameterFile");
+      ParameterFile = arg;
+    }
+  }
+
+
+  template <class SmearingPolicy>
+  void Run(SmearingPolicy &S) {
+    Runner(S);
+  }
+
+  void Run(){
+    NoSmearing<Implementation> S;
+    Runner(S);
+  }
+
+  //////////////////////////////////////////////////////////////////
+
+ private:
+  template <class SmearingPolicy>
+  void Runner(SmearingPolicy &Smearing) {
+    auto UGrid = Resources.GetCartesian();
+    Resources.AddRNGs();
+    Field U(UGrid);
+
+    // Can move this outside?
+    typedef IntegratorType<SmearingPolicy> TheIntegrator;
+    TheIntegrator MDynamics(UGrid, Parameters.MD, TheAction, Smearing);
+
+    if (Parameters.StartingType == "HotStart") {
+      // Hot start
+      Resources.SeedFixedIntegers();
+      Implementation::HotConfiguration(Resources.GetParallelRNG(), U);
+    } else if (Parameters.StartingType == "ColdStart") {
+      // Cold start
+      Resources.SeedFixedIntegers();
+      Implementation::ColdConfiguration(Resources.GetParallelRNG(), U);
+    } else if (Parameters.StartingType == "TepidStart") {
+      // Tepid start
+      Resources.SeedFixedIntegers();
+      Implementation::TepidConfiguration(Resources.GetParallelRNG(), U);
+    } else if (Parameters.StartingType == "CheckpointStart") {
+      // CheckpointRestart
+      Resources.GetCheckPointer()->CheckpointRestore(Parameters.StartTrajectory, U,
+                                   Resources.GetSerialRNG(),
+                                   Resources.GetParallelRNG());
+    }
+
+    Smearing.set_Field(U);
+
+    HybridMonteCarlo<TheIntegrator> HMC(Parameters, MDynamics,
+                                        Resources.GetSerialRNG(),
+                                        Resources.GetParallelRNG(), 
+                                        Resources.GetObservables(), U);
+
+    // Run it
+    HMC.evolve();
+  }
+};
+
+// These are for gauge fields, default integrator MinimumNorm2
+template <template <typename, typename, typename> class Integrator>
+using GenericHMCRunner = HMCWrapperTemplate<PeriodicGimplR, Integrator>;
+template <template <typename, typename, typename> class Integrator>
+using GenericHMCRunnerF = HMCWrapperTemplate<PeriodicGimplF, Integrator>;
+template <template <typename, typename, typename> class Integrator>
+using GenericHMCRunnerD = HMCWrapperTemplate<PeriodicGimplD, Integrator>;
+
+
+// These are for gauge fields, default integrator MinimumNorm2
+template <template <typename, typename, typename> class Integrator>
+using ConjugateHMCRunner = HMCWrapperTemplate<ConjugateGimplR, Integrator>;
+template <template <typename, typename, typename> class Integrator>
+using ConjugateHMCRunnerF = HMCWrapperTemplate<ConjugateGimplF, Integrator>;
+template <template <typename, typename, typename> class Integrator>
+using ConjugateHMCRunnerD = HMCWrapperTemplate<ConjugateGimplD, Integrator>;
+
+
+
+template <class RepresentationsPolicy,
+          template <typename, typename, typename> class Integrator>
+using GenericHMCRunnerHirep =
+    HMCWrapperTemplate<PeriodicGimplR, Integrator, RepresentationsPolicy>;
+
+template <class Implementation, class RepresentationsPolicy, 
+          template <typename, typename, typename> class Integrator>
+using GenericHMCRunnerTemplate = HMCWrapperTemplate<Implementation, Integrator, RepresentationsPolicy>;
+
+typedef HMCWrapperTemplate<ScalarImplR, MinimumNorm2, ScalarFields>
+    ScalarGenericHMCRunner;
+
+typedef HMCWrapperTemplate<ScalarAdjImplR, MinimumNorm2, ScalarMatrixFields>
+    ScalarAdjGenericHMCRunner;
+
+template <int Colours> 
+using ScalarNxNAdjGenericHMCRunner = HMCWrapperTemplate < ScalarNxNAdjImplR<Colours>, MinimumNorm2, ScalarNxNMatrixFields<Colours> >;
+
+}  // namespace QCD
+}  // namespace Grid
+
+#endif  // GRID_GENERIC_HMC_RUNNER
@@ -34,13 +34,15 @@ directory
 * @brief Classes for Hybrid Monte Carlo update
 *
 * @author Guido Cossu
- * Time-stamp: <2015-07-30 16:58:26 neo>
 */
 //--------------------------------------------------------------------
 #ifndef HMC_INCLUDED
 #define HMC_INCLUDED

 #include <string>
+#include <list>
+
+

 #include <Grid/qcd/hmc/integrators/Integrator.h>
 #include <Grid/qcd/hmc/integrators/Integrator_algorithm.h>
@@ -48,91 +50,64 @@ directory
 namespace Grid {
 namespace QCD {

-struct HMCparameters {
-  Integer StartTrajectory;
-  Integer Trajectories; /* @brief Number of sweeps in this run */
-  bool MetropolisTest;
-  Integer NoMetropolisUntil;
+struct HMCparameters: Serializable {
+	GRID_SERIALIZABLE_CLASS_MEMBERS(HMCparameters,
+                                  Integer, StartTrajectory,
+                                  Integer, Trajectories, /* @brief Number of sweeps in this run */
+                                  bool, MetropolisTest,
+                                  Integer, NoMetropolisUntil,
+                                  std::string, StartingType,
+                                  IntegratorParameters, MD)

  HMCparameters() {
    ////////////////////////////// Default values
-    MetropolisTest = true;
+    MetropolisTest    = true;
    NoMetropolisUntil = 10;
-    StartTrajectory = 0;
-    Trajectories = 200;
+    StartTrajectory   = 0;
+    Trajectories      = 10;
+    StartingType      = "HotStart";
    /////////////////////////////////
  }

-  void print() const {
-    std::cout << GridLogMessage << "[HMC parameter] Trajectories            : " << Trajectories << "\n";
-    std::cout << GridLogMessage << "[HMC parameter] Start trajectory        : " << StartTrajectory << "\n";
-    std::cout << GridLogMessage << "[HMC parameter] Metropolis test (on/off): " << MetropolisTest << "\n";
-    std::cout << GridLogMessage << "[HMC parameter] Thermalization trajs    : " << NoMetropolisUntil << "\n";
+  template <class ReaderClass >
+  HMCparameters(Reader<ReaderClass> & TheReader){
+  	initialize(TheReader);
+  }
+
+  template < class ReaderClass > 
+  void initialize(Reader<ReaderClass> &TheReader){
+  	std::cout << GridLogMessage << "Reading HMC\n";
+  	read(TheReader, "HMC", *this);
+  }
+
+
+  void print_parameters() const {
+    std::cout << GridLogMessage << "[HMC parameters] Trajectories            : " << Trajectories << "\n";
+    std::cout << GridLogMessage << "[HMC parameters] Start trajectory        : " << StartTrajectory << "\n";
+    std::cout << GridLogMessage << "[HMC parameters] Metropolis test (on/off): " << std::boolalpha << MetropolisTest << "\n";
+    std::cout << GridLogMessage << "[HMC parameters] Thermalization trajs    : " << NoMetropolisUntil << "\n";
+    std::cout << GridLogMessage << "[HMC parameters] Starting type           : " << StartingType << "\n";
+    MD.print_parameters();
  }
  
 };
-
-template <class GaugeField>
-class HmcObservable {
- public:
-  virtual void TrajectoryComplete(int traj, GaugeField &U, GridSerialRNG &sRNG,
-                                  GridParallelRNG &pRNG) = 0;
-};
-
-template <class Gimpl>
-class PlaquetteLogger : public HmcObservable<typename Gimpl::GaugeField> {
- private:
-  std::string Stem;
-
- public:
-  INHERIT_GIMPL_TYPES(Gimpl);
-  PlaquetteLogger(std::string cf) { Stem = cf; };
-
-  void TrajectoryComplete(int traj, GaugeField &U, GridSerialRNG &sRNG,
-                          GridParallelRNG &pRNG) {
-    std::string file;
-    {
-      std::ostringstream os;
-      os << Stem << "." << traj;
-      file = os.str();
-    }
-    std::ofstream of(file);
-
-    RealD peri_plaq = WilsonLoops<PeriodicGimplR>::avgPlaquette(U);
-    RealD peri_rect = WilsonLoops<PeriodicGimplR>::avgRectangle(U);
-
-    RealD impl_plaq = WilsonLoops<Gimpl>::avgPlaquette(U);
-    RealD impl_rect = WilsonLoops<Gimpl>::avgRectangle(U);
-
-    of << traj << " " << impl_plaq << " " << impl_rect << "  " << peri_plaq
-       << " " << peri_rect << std::endl;
-    std::cout << GridLogMessage << "traj"
-              << " "
-              << "plaq "
-              << " "
-              << " rect  "
-              << "  "
-              << "peri_plaq"
-              << " "
-              << "peri_rect" << std::endl;
-    std::cout << GridLogMessage << traj << " " << impl_plaq << " " << impl_rect
-              << "  " << peri_plaq << " " << peri_rect << std::endl;
-  }
-};
-
-//    template <class GaugeField, class Integrator, class Smearer, class
-//    Boundary>
-template <class GaugeField, class IntegratorType>
+	
+template <class IntegratorType>
 class HybridMonteCarlo {
 private:
  const HMCparameters Params;

-  GridSerialRNG &sRNG;    // Fixme: need a RNG management strategy.
-  GridParallelRNG &pRNG;  // Fixme: need a RNG management strategy.
-  GaugeField &Ucur;
+  typedef typename IntegratorType::Field Field;
+  typedef std::vector< HmcObservable<Field> * > ObsListType;
+  
+  	//pass these from the resource manager
+  GridSerialRNG &sRNG;   
+  GridParallelRNG &pRNG; 

+  Field &Ucur;
+  
  IntegratorType &TheIntegrator;
-  std::vector<HmcObservable<GaugeField> *> Observables;
+	ObsListType Observables;

  /////////////////////////////////////////////////////////
  // Metropolis step
@@ -167,13 +142,13 @@ class HybridMonteCarlo {
  /////////////////////////////////////////////////////////
  // Evolution
  /////////////////////////////////////////////////////////
-  RealD evolve_step(GaugeField &U) {
+  RealD evolve_hmc_step(Field &U) {
    TheIntegrator.refresh(U, pRNG);  // set U and initialize P and phi's

    RealD H0 = TheIntegrator.S(U);  // initial state action

    std::streamsize current_precision = std::cout.precision();
-    std::cout.precision(17);
+    std::cout.precision(15);
    std::cout << GridLogMessage << "Total H before trajectory = " << H0 << "\n";
    std::cout.precision(current_precision);

@@ -181,64 +156,96 @@ class HybridMonteCarlo {

    RealD H1 = TheIntegrator.S(U);  // updated state action

-    std::cout.precision(17);
-    std::cout << GridLogMessage << "Total H after trajectory  = " << H1
-              << "  dH = " << H1 - H0 << "\n";
-    std::cout.precision(current_precision);
+    ///////////////////////////////////////////////////////////
+    if(0){
+      std::cout << "------------------------- Reversibility test" << std::endl;
+      TheIntegrator.reverse_momenta();
+      TheIntegrator.integrate(U);

+      H1 = TheIntegrator.S(U);  // updated state action
+      std::cout << "--------------------------------------------" << std::endl;
+    }
+    ///////////////////////////////////////////////////////////
+
+
+    std::cout.precision(15);
+    std::cout << GridLogMessage << "Total H after trajectory  = " << H1
+	      << "  dH = " << H1 - H0 << "\n";
+    std::cout.precision(current_precision);
+    
    return (H1 - H0);
  }
+  
+
+  

 public:
  /////////////////////////////////////////
  // Constructor
  /////////////////////////////////////////
-  HybridMonteCarlo(HMCparameters Pams, IntegratorType &_Int,
-                   GridSerialRNG &_sRNG, GridParallelRNG &_pRNG, GaugeField &_U)
-      : Params(Pams), TheIntegrator(_Int), sRNG(_sRNG), pRNG(_pRNG), Ucur(_U) {}
+  HybridMonteCarlo(HMCparameters _Pams, IntegratorType &_Int,
+                   GridSerialRNG &_sRNG, GridParallelRNG &_pRNG, 
+                   ObsListType _Obs, Field &_U)
+    : Params(_Pams), TheIntegrator(_Int), sRNG(_sRNG), pRNG(_pRNG), Observables(_Obs), Ucur(_U) {}
  ~HybridMonteCarlo(){};

-  void AddObservable(HmcObservable<GaugeField> *obs) {
-    Observables.push_back(obs);
-  }
-
  void evolve(void) {
    Real DeltaH;

-    GaugeField Ucopy(Ucur._grid);
+    Field Ucopy(Ucur._grid);

-    Params.print();
+    Params.print_parameters();
+    TheIntegrator.print_actions();

    // Actual updates (evolve a copy Ucopy then copy back eventually)
-    for (int traj = Params.StartTrajectory;
-         traj < Params.Trajectories + Params.StartTrajectory; ++traj) {
+    unsigned int FinalTrajectory = Params.Trajectories + Params.NoMetropolisUntil + Params.StartTrajectory;
+    for (int traj = Params.StartTrajectory; traj < FinalTrajectory; ++traj) {
      std::cout << GridLogMessage << "-- # Trajectory = " << traj << "\n";
+      if (traj < Params.StartTrajectory + Params.NoMetropolisUntil) {
+      	std::cout << GridLogMessage << "-- Thermalization" << std::endl;
+      }
+      
+      double t0=usecond();
      Ucopy = Ucur;

-      DeltaH = evolve_step(Ucopy);
-
+      DeltaH = evolve_hmc_step(Ucopy);
+      // Metropolis-Hastings test
      bool accept = true;
-      if (traj >= Params.NoMetropolisUntil) {
+      if (traj >= Params.StartTrajectory + Params.NoMetropolisUntil) {
        accept = metropolis_test(DeltaH);
+      } else {
+      	std::cout << GridLogMessage << "Skipping Metropolis test" << std::endl;
      }

-      if (accept) {
-        Ucur = Ucopy;
-      }
+      if (accept)
+        Ucur = Ucopy; 
+      
+     
+      
+      double t1=usecond();
+      std::cout << GridLogMessage << "Total time for trajectory (s): " << (t1-t0)/1e6 << std::endl;
+

      for (int obs = 0; obs < Observables.size(); obs++) {
+      	std::cout << GridLogDebug << "Observables # " << obs << std::endl;
+      	std::cout << GridLogDebug << "Observables total " << Observables.size() << std::endl;
+      	std::cout << GridLogDebug << "Observables pointer " << Observables[obs] << std::endl;
        Observables[obs]->TrajectoryComplete(traj + 1, Ucur, sRNG, pRNG);
      }
+      std::cout << GridLogMessage << ":::::::::::::::::::::::::::::::::::::::::::" << std::endl;
    }
  }
+
 };


 }  // QCD
 }  // Grid

-#include <Grid/parallelIO/NerscIO.h>
-#include <Grid/qcd/hmc/NerscCheckpointer.h>
-#include <Grid/qcd/hmc/HmcRunner.h>
+
+// april 11 2017 merge, Guido, commenting out
+//#include <Grid/parallelIO/NerscIO.h>
+//#include <Grid/qcd/hmc/NerscCheckpointer.h>
+//#include <Grid/qcd/hmc/HmcRunner.h>

 #endif
@@ -0,0 +1,111 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/GenericHmcRunner.h
+
+Copyright (C) 2015
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef GRID_HMC_MODULES
+#define GRID_HMC_MODULES
+
+
+#include "HMC_GridModules.h"
+
+namespace Grid {
+namespace QCD {
+
+////////////////////////////////////////////////////////////////////
+struct RNGModuleParameters: Serializable {
+  GRID_SERIALIZABLE_CLASS_MEMBERS(RNGModuleParameters,
+  std::string, serial_seeds,
+  std::string, parallel_seeds,);
+
+  std::vector<int> getSerialSeeds(){return strToVec<int>(serial_seeds);}
+  std::vector<int> getParallelSeeds(){return strToVec<int>(parallel_seeds);}
+
+  RNGModuleParameters(): serial_seeds("1"), parallel_seeds("1"){}
+
+  template <class ReaderClass >
+  RNGModuleParameters(Reader<ReaderClass>& Reader){
+    read(Reader, "RandomNumberGenerator", *this); 
+  }
+  
+};
+
+// Random number generators module
+class RNGModule{
+   GridSerialRNG sRNG_;
+   std::unique_ptr<GridParallelRNG> pRNG_;
+   RNGModuleParameters Params_;
+
+public:
+
+  RNGModule(){};
+
+  void set_pRNG(GridParallelRNG* pRNG){
+    pRNG_.reset(pRNG);
+  }
+
+  void set_RNGSeeds(RNGModuleParameters& Params) {
+    Params_ = Params;
+  }
+
+  GridSerialRNG& get_sRNG() { return sRNG_; }
+  GridParallelRNG& get_pRNG() { return *pRNG_.get(); }
+
+  void seed() {
+    auto SerialSeeds   = Params_.getSerialSeeds();
+    auto ParallelSeeds = Params_.getParallelSeeds();
+    if (SerialSeeds.size() == 0 && ParallelSeeds.size() == 0) {
+      std::cout << GridLogError << "Seeds not initialized" << std::endl;
+      exit(1);
+    }
+    sRNG_.SeedFixedIntegers(SerialSeeds);
+    pRNG_->SeedFixedIntegers(ParallelSeeds);
+  }
+};
+
+
+/*
+///////////////////////////////////////////////////////////////////
+/// Smearing module
+template <class ImplementationPolicy>
+class SmearingModule{
+   virtual void get_smearing();
+};
+
+template <class ImplementationPolicy>
+class StoutSmearingModule: public SmearingModule<ImplementationPolicy>{
+   SmearedConfiguration<ImplementationPolicy> SmearingPolicy;
+};
+
+*/
+
+
+
+}  // namespace QCD
+}  // namespace Grid
+
+#endif  // GRID_HMC_MODULES
@@ -0,0 +1,301 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/GenericHmcRunner.h
+
+Copyright (C) 2015
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+  See the full license in the file "LICENSE" in the top level distribution
+  directory
+  *************************************************************************************/
+/*  END LEGAL */
+#ifndef HMC_RESOURCE_MANAGER_H
+#define HMC_RESOURCE_MANAGER_H
+
+#include <unordered_map>
+
+// One function per Checkpointer, use a macro to simplify
+#define RegisterLoadCheckPointerFunction(NAME)                           \
+  void Load##NAME##Checkpointer(const CheckpointerParameters& Params_) { \
+    if (!have_CheckPointer) {                                            \
+      std::cout << GridLogDebug << "Loading Checkpointer " << #NAME      \
+                << std::endl;                                            \
+      CP = std::unique_ptr<CheckpointerBaseModule>(                      \
+        new NAME##CPModule<ImplementationPolicy>(Params_));              \
+      have_CheckPointer = true;                                          \
+    } else {                                                             \
+      std::cout << GridLogError << "Checkpointer already loaded "        \
+                << std::endl;                                            \
+      exit(1);                                                           \
+    }                                                                    \
+  }
+
+namespace Grid {
+namespace QCD {
+
+// HMC Resource manager
+template <class ImplementationPolicy>
+class HMCResourceManager {
+  typedef HMCModuleBase< QCD::BaseHmcCheckpointer<ImplementationPolicy> > CheckpointerBaseModule;
+  typedef HMCModuleBase< QCD::HmcObservable<typename ImplementationPolicy::Field> > ObservableBaseModule;
+  typedef ActionModuleBase< QCD::Action<typename ImplementationPolicy::Field>, GridModule > ActionBaseModule;
+
+  // Named storage for grid pairs (std + red-black)
+  std::unordered_map<std::string, GridModule> Grids;
+  RNGModule RNGs;
+
+  // SmearingModule<ImplementationPolicy> Smearing;
+  std::unique_ptr<CheckpointerBaseModule> CP;
+
+  // A vector of HmcObservable modules
+  std::vector<std::unique_ptr<ObservableBaseModule> > ObservablesList;
+
+
+  // A vector of HmcObservable modules
+  std::multimap<int, std::unique_ptr<ActionBaseModule> > ActionsList;
+  std::vector<int> multipliers;
+
+  bool have_RNG;
+  bool have_CheckPointer;
+
+  // NOTE: operator << is not overloaded for std::vector<string> 
+  // so thsi function is necessary
+  void output_vector_string(const std::vector<std::string> &vs){
+    for (auto &i: vs)
+      std::cout << i << " ";
+    std::cout << std::endl;
+  }
+
+
+ public:
+  HMCResourceManager() : have_RNG(false), have_CheckPointer(false) {}
+
+  template <class ReaderClass, class vector_type = vComplex >
+  void initialize(ReaderClass &Read){
+    // assumes we are starting from the main node
+
+    // Geometry
+    GridModuleParameters GridPar(Read);
+    GridFourDimModule<vector_type> GridMod( GridPar) ;
+    AddGrid("gauge", GridMod);
+
+    // Checkpointer
+    auto &CPfactory = HMC_CPModuleFactory<cp_string, ImplementationPolicy, ReaderClass >::getInstance();
+    Read.push("Checkpointer");
+    std::string cp_type;
+    read(Read,"name", cp_type);
+    std::cout << "Registered types " << std::endl;
+    output_vector_string(CPfactory.getBuilderList());
+
+
+    CP = CPfactory.create(cp_type, Read);
+    CP->print_parameters();
+    Read.pop();    
+    have_CheckPointer = true;  
+
+    RNGModuleParameters RNGpar(Read);
+    SetRNGSeeds(RNGpar);
+
+    // Observables
+    auto &ObsFactory = HMC_ObservablesModuleFactory<observable_string, typename ImplementationPolicy::Field, ReaderClass>::getInstance(); 
+    Read.push(observable_string);// here must check if existing...
+    do {
+      std::string obs_type;
+      read(Read,"name", obs_type);
+      std::cout << "Registered types " << std::endl;
+      output_vector_string(ObsFactory.getBuilderList() );
+
+      ObservablesList.emplace_back(ObsFactory.create(obs_type, Read));
+      ObservablesList[ObservablesList.size() - 1]->print_parameters();
+    } while (Read.nextElement(observable_string));
+    Read.pop();
+
+    // Loop on levels
+    if(!Read.push("Actions")){
+      std::cout << "Actions not found" << std::endl; 
+      exit(1);
+    }
+
+    if(!Read.push("Level")){// push must check if the node exist
+         std::cout << "Level not found" << std::endl; 
+      exit(1);
+    }
+    do
+    {
+      fill_ActionsLevel(Read); 
+    }
+    while(Read.push("Level"));
+
+    Read.pop();
+  }
+
+
+ 
+  template <class RepresentationPolicy>
+  void GetActionSet(ActionSet<typename ImplementationPolicy::Field, RepresentationPolicy>& Aset){
+    Aset.resize(multipliers.size());
+ 
+    for(auto it = ActionsList.begin(); it != ActionsList.end(); it++){
+      (*it).second->acquireResource(Grids["gauge"]);
+      Aset[(*it).first-1].push_back((*it).second->getPtr());
+    }
+  }
+
+
+
+  //////////////////////////////////////////////////////////////
+  // Grids
+  //////////////////////////////////////////////////////////////
+
+  void AddGrid(std::string s, GridModule& M) {
+    // Check for name clashes
+    auto search = Grids.find(s);
+    if (search != Grids.end()) {
+      std::cout << GridLogError << "Grid with name \"" << search->first
+                << "\" already present. Terminating\n";
+      exit(1);
+    }
+    Grids[s] = std::move(M);
+  }
+
+  // Add a named grid set, 4d shortcut
+  void AddFourDimGrid(std::string s) {
+    GridFourDimModule<vComplex> Mod;
+    AddGrid(s, Mod);
+  }
+
+
+
+  GridCartesian* GetCartesian(std::string s = "") {
+    if (s.empty()) s = Grids.begin()->first;
+    std::cout << GridLogDebug << "Getting cartesian grid from: " << s
+              << std::endl;
+    return Grids[s].get_full();
+  }
+
+  GridRedBlackCartesian* GetRBCartesian(std::string s = "") {
+    if (s.empty()) s = Grids.begin()->first;
+    std::cout << GridLogDebug << "Getting rb-cartesian grid from: " << s
+              << std::endl;
+    return Grids[s].get_rb();
+  }
+
+  //////////////////////////////////////////////////////
+  // Random number generators
+  //////////////////////////////////////////////////////
+
+  void AddRNGs(std::string s = "") {
+    // Couple the RNGs to the GridModule tagged by s
+    // the default is the first grid registered
+    assert(Grids.size() > 0 && !have_RNG);
+    if (s.empty()) s = Grids.begin()->first;
+    std::cout << GridLogDebug << "Adding RNG to grid: " << s << std::endl;
+    RNGs.set_pRNG(new GridParallelRNG(GetCartesian(s)));
+    have_RNG = true;
+  }
+
+  void SetRNGSeeds(RNGModuleParameters& Params) { RNGs.set_RNGSeeds(Params); }
+
+  GridSerialRNG& GetSerialRNG() { return RNGs.get_sRNG(); }
+
+  GridParallelRNG& GetParallelRNG() {
+    assert(have_RNG);
+    return RNGs.get_pRNG();
+  }
+
+  void SeedFixedIntegers() {
+    assert(have_RNG);
+    RNGs.seed();
+  }
+
+  //////////////////////////////////////////////////////
+  // Checkpointers
+  //////////////////////////////////////////////////////
+
+  BaseHmcCheckpointer<ImplementationPolicy>* GetCheckPointer() {
+    if (have_CheckPointer)
+      return CP->getPtr();
+    else {
+      std::cout << GridLogError << "Error: no checkpointer defined"
+                << std::endl;
+      exit(1);
+    }
+  }
+
+  RegisterLoadCheckPointerFunction(Binary);
+  RegisterLoadCheckPointerFunction(Nersc);
+  #ifdef HAVE_LIME
+  RegisterLoadCheckPointerFunction(ILDG);
+  #endif
+
+  ////////////////////////////////////////////////////////
+  // Observables
+  ////////////////////////////////////////////////////////
+
+  template<class T, class... Types>
+  void AddObservable(Types&&... Args){
+    ObservablesList.push_back(std::unique_ptr<T>(new T(std::forward<Types>(Args)...)));
+    ObservablesList.back()->print_parameters();
+  }
+
+  std::vector<HmcObservable<typename ImplementationPolicy::Field>* > GetObservables(){
+    std::vector<HmcObservable<typename ImplementationPolicy::Field>* > out;
+    for (auto &i : ObservablesList){
+      out.push_back(i->getPtr());
+    }
+
+    // Add the checkpointer to the observables
+    out.push_back(GetCheckPointer());
+    return out;
+  }
+
+
+
+private:
+   // this private
+  template <class ReaderClass >
+  void fill_ActionsLevel(ReaderClass &Read){
+    // Actions set
+    int m;
+    Read.readDefault("multiplier",m);
+    multipliers.push_back(m);
+    std::cout << "Level : " << multipliers.size()  << " with multiplier : " << m << std::endl; 
+    // here gauge
+    Read.push("Action");
+    do{
+      auto &ActionFactory = HMC_ActionModuleFactory<gauge_string, typename ImplementationPolicy::Field, ReaderClass>::getInstance(); 
+      std::string action_type;
+      Read.readDefault("name", action_type); 
+      output_vector_string(ActionFactory.getBuilderList() );
+      ActionsList.emplace(m, ActionFactory.create(action_type, Read));
+    } while (Read.nextElement("Action"));
+    ActionsList.find(m)->second->print_parameters();    
+    Read.pop();
+
+  }
+
+
+
+};
+}
+}
+
+#endif  // HMC_RESOURCE_MANAGER_H
@@ -0,0 +1,137 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/GenericHmcRunner.h
+
+Copyright (C) 2015
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef HMC_RUNNER_MODULE
+#define HMC_RUNNER_MODULE
+
+namespace Grid {
+
+// the reader class is necessary here for the automatic initialization of the resources
+// if we had a virtual reader would have been unecessary
+template <class HMCType, class ReaderClass >
+class HMCModule
+    : public Parametrized< QCD::HMCparameters >,
+      public HMCModuleBase< QCD::HMCRunnerBase<ReaderClass> > {
+ public:
+  typedef HMCModuleBase< QCD::HMCRunnerBase<ReaderClass> > Base;
+  typedef typename Base::Product Product;
+
+  std::unique_ptr<HMCType> HMCPtr;
+
+  HMCModule(QCD::HMCparameters Par) : Parametrized<QCD::HMCparameters>(Par) {}
+
+  template <class ReaderCl>
+  HMCModule(Reader<ReaderCl>& R) : Parametrized<QCD::HMCparameters>(R, "HMC"){};
+
+  Product* getPtr() {
+    if (!HMCPtr) initialize();
+ 
+    return HMCPtr.get();
+  }
+
+ private:
+  virtual void initialize() = 0;
+};
+
+// Factory
+template <char const *str, class ReaderClass >
+class HMCRunnerModuleFactory
+    : public Factory < HMCModuleBase< QCD::HMCRunnerBase<ReaderClass> > ,	Reader<ReaderClass> > {
+ public:
+ 	typedef Reader<ReaderClass> TheReader; 
+ 	// use SINGLETON FUNCTOR MACRO HERE
+  HMCRunnerModuleFactory(const HMCRunnerModuleFactory& e) = delete;
+  void operator=(const HMCRunnerModuleFactory& e) = delete;
+  static HMCRunnerModuleFactory& getInstance(void) {
+    static HMCRunnerModuleFactory e;
+    return e;
+  }
+
+ private:
+  HMCRunnerModuleFactory(void) = default;
+  std::string obj_type() const {
+  	return std::string(str);
+  }
+};
+
+
+
+
+
+///////////////
+// macro for these
+
+template < class ImplementationPolicy, class RepresentationPolicy, class ReaderClass >
+class HMCLeapFrog: public HMCModule< QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::LeapFrog>, ReaderClass >{
+  typedef HMCModule< QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::LeapFrog>, ReaderClass  > HMCBaseMod;
+  using HMCBaseMod::HMCBaseMod;
+
+  // aquire resource
+  virtual void initialize(){
+    this->HMCPtr.reset(new QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::LeapFrog>(this->Par_) );
+  }
+};
+
+template < class ImplementationPolicy, class RepresentationPolicy, class ReaderClass >
+class HMCMinimumNorm2: public HMCModule< QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::MinimumNorm2>, ReaderClass  >{
+  typedef HMCModule< QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::MinimumNorm2>, ReaderClass  > HMCBaseMod;
+  using HMCBaseMod::HMCBaseMod;
+
+  // aquire resource
+  virtual void initialize(){
+    this->HMCPtr.reset(new QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::MinimumNorm2>(this->Par_));
+  }
+};
+
+
+template < class ImplementationPolicy, class RepresentationPolicy, class ReaderClass >
+class HMCForceGradient: public HMCModule< QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::ForceGradient>, ReaderClass  >{
+  typedef HMCModule< QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::ForceGradient>, ReaderClass   > HMCBaseMod;
+  using HMCBaseMod::HMCBaseMod;
+
+  // aquire resource
+  virtual void initialize(){
+    this->HMCPtr.reset(new QCD::GenericHMCRunnerTemplate<ImplementationPolicy, RepresentationPolicy, QCD::ForceGradient>(this->Par_) );
+  }
+};
+
+extern char hmc_string[];
+
+
+
+
+
+
+//////////////////////////////////////////////////////////////
+
+
+
+}
+
+#endif
@@ -0,0 +1,133 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/HMC_GridModules.h
+
+Copyright (C) 2015
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef HMC_GRID_MODULES
+#define HMC_GRID_MODULES
+
+namespace Grid {
+
+// Resources
+// Modules for grids 
+
+// Introduce another namespace HMCModules?
+
+class GridModuleParameters: Serializable{   
+public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(GridModuleParameters,
+  std::string, lattice,
+  std::string, mpi);
+
+  std::vector<int> getLattice(){return strToVec<int>(lattice);}
+  std::vector<int> getMpi()    {return strToVec<int>(mpi);}
+
+  void check(){
+    if (getLattice().size() != getMpi().size()) {
+      std::cout << GridLogError 
+                << "Error in GridModuleParameters: lattice and mpi dimensions "
+                   "do not match"
+                << std::endl;
+      exit(1);
+    }
+  }    
+
+  template <class ReaderClass>
+  GridModuleParameters(Reader<ReaderClass>& Reader, std::string n = "LatticeGrid"):name(n) {
+    read(Reader, name, *this);
+    check();
+  }
+
+  // Save on file
+  template< class WriterClass>
+  void save(Writer<WriterClass>& Writer){
+    check();
+    write(Writer, name, *this);
+  }
+private:
+    std::string name;
+};
+
+// Lower level class
+class GridModule {
+ public:
+  GridCartesian* get_full() { 
+    std::cout << GridLogDebug << "Getting cartesian in module"<< std::endl;
+    return grid_.get(); }
+  GridRedBlackCartesian* get_rb() { 
+    std::cout << GridLogDebug << "Getting rb-cartesian in module"<< std::endl;
+    return rbgrid_.get(); }
+
+  void set_full(GridCartesian* grid) { grid_.reset(grid); }
+  void set_rb(GridRedBlackCartesian* rbgrid) { rbgrid_.reset(rbgrid); }
+
+ protected:
+  std::unique_ptr<GridCartesian> grid_;
+  std::unique_ptr<GridRedBlackCartesian> rbgrid_;
+  
+};
+
+////////////////////////////////////
+// Classes for the user
+////////////////////////////////////
+// Note: the space time grid should be out of the QCD namespace
+template< class vector_type>
+class GridFourDimModule : public GridModule {
+ public:
+  GridFourDimModule() {
+    using namespace QCD;
+    set_full(SpaceTimeGrid::makeFourDimGrid(
+        GridDefaultLatt(), GridDefaultSimd(4, vector_type::Nsimd()),
+        GridDefaultMpi()));
+    set_rb(SpaceTimeGrid::makeFourDimRedBlackGrid(grid_.get()));
+  }
+
+  GridFourDimModule(GridModuleParameters Params) {
+    using namespace QCD;
+    Params.check();
+    std::vector<int> lattice_v = Params.getLattice();
+    std::vector<int> mpi_v = Params.getMpi();
+    if (lattice_v.size() == 4) {
+      set_full(SpaceTimeGrid::makeFourDimGrid(
+          lattice_v, GridDefaultSimd(4, vector_type::Nsimd()),
+          mpi_v));
+      set_rb(SpaceTimeGrid::makeFourDimRedBlackGrid(grid_.get()));
+    } else {
+      std::cout << GridLogError 
+          << "Error in GridFourDimModule: lattice dimension different from 4"
+          << std::endl;
+      exit(1);
+    }
+  }
+};
+
+typedef GridFourDimModule<vComplex> GridDefaultFourDimModule;
+
+
+}  // namespace Grid
+
+#endif  // HMC_GRID_MODULES
@@ -33,10 +33,20 @@ directory

 #include <string>

+#include <Grid/qcd/observables/hmc_observable.h>
 #include <Grid/qcd/hmc/HMC.h>
+
+
 // annoying location; should move this ?
+#include <Grid/parallelIO/IldgIOtypes.h>
+#include <Grid/parallelIO/IldgIO.h>
 #include <Grid/parallelIO/NerscIO.h>
-#include <Grid/qcd/hmc/NerscCheckpointer.h>
-#include <Grid/qcd/hmc/HmcRunner.h>
+
+#include <Grid/qcd/hmc/checkpointers/CheckPointers.h>
+#include <Grid/qcd/hmc/HMCModules.h>
+#include <Grid/qcd/modules/mods.h>
+#include <Grid/qcd/hmc/HMCResourceManager.h>
+#include <Grid/qcd/hmc/GenericHMCrunner.h>
+#include <Grid/qcd/hmc/HMCRunnerModule.h>

 #endif
@@ -1,191 +0,0 @@
-/*************************************************************************************
-
-Grid physics library, www.github.com/paboyle/Grid
-
-Source file: ./lib/qcd/hmc/HmcRunner.h
-
-Copyright (C) 2015
-
-Author: paboyle <paboyle@ph.ed.ac.uk>
-
-This program is free software; you can redistribute it and/or modify
-it under the terms of the GNU General Public License as published by
-the Free Software Foundation; either version 2 of the License, or
-(at your option) any later version.
-
-This program is distributed in the hope that it will be useful,
-but WITHOUT ANY WARRANTY; without even the implied warranty of
-MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-GNU General Public License for more details.
-
-You should have received a copy of the GNU General Public License along
-with this program; if not, write to the Free Software Foundation, Inc.,
-51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
-See the full license in the file "LICENSE" in the top level distribution
-directory
-*************************************************************************************/
-/*  END LEGAL */
-#ifndef HMC_RUNNER
-#define HMC_RUNNER
-
-namespace Grid {
-namespace QCD {
-
-template <class Gimpl, class RepresentationsPolicy = NoHirep >
-class NerscHmcRunnerTemplate {
- public:
-  INHERIT_GIMPL_TYPES(Gimpl);
-
-  enum StartType_t { ColdStart, HotStart, TepidStart, CheckpointStart };
-
-  ActionSet<GaugeField, RepresentationsPolicy> TheAction;
-
-  GridCartesian *UGrid;
-  GridCartesian *FGrid;
-  GridRedBlackCartesian *UrbGrid;
-  GridRedBlackCartesian *FrbGrid;
-
-  virtual void BuildTheAction(int argc, char **argv) = 0;  // necessary?
-
-  void Run(int argc, char **argv) {
-    StartType_t StartType = HotStart;
-
-    std::string arg;
-
-    if (GridCmdOptionExists(argv, argv + argc, "--StartType")) {
-      arg = GridCmdOptionPayload(argv, argv + argc, "--StartType");
-      if (arg == "HotStart") {
-        StartType = HotStart;
-      } else if (arg == "ColdStart") {
-        StartType = ColdStart;
-      } else if (arg == "TepidStart") {
-        StartType = TepidStart;
-      } else if (arg == "CheckpointStart") {
-        StartType = CheckpointStart;
-      } else {
-        std::cout << GridLogError << "Unrecognized option in --StartType\n";
-        std::cout << GridLogError << "Valid [HotStart, ColdStart, TepidStart, CheckpointStart]\n";
-        assert(0);
-      }
-    }
-
-    int StartTraj = 0;
-    if (GridCmdOptionExists(argv, argv + argc, "--StartTrajectory")) {
-      arg = GridCmdOptionPayload(argv, argv + argc, "--StartTrajectory");
-      std::vector<int> ivec(0);
-      GridCmdOptionIntVector(arg, ivec);
-      StartTraj = ivec[0];
-    }
-
-    int NumTraj = 1;
-    if (GridCmdOptionExists(argv, argv + argc, "--Trajectories")) {
-      arg = GridCmdOptionPayload(argv, argv + argc, "--Trajectories");
-      std::vector<int> ivec(0);
-      GridCmdOptionIntVector(arg, ivec);
-      NumTraj = ivec[0];
-    }
-
-    int NumThermalizations = 10;
-    if (GridCmdOptionExists(argv, argv + argc, "--Thermalizations")) {
-      arg = GridCmdOptionPayload(argv, argv + argc, "--Thermalizations");
-      std::vector<int> ivec(0);
-      GridCmdOptionIntVector(arg, ivec);
-      NumThermalizations = ivec[0];
-    }
-
-    GridSerialRNG sRNG;
-    GridParallelRNG pRNG(UGrid);
-    LatticeGaugeField U(UGrid);  // change this to an extended field (smearing class)?
-
-    std::vector<int> SerSeed({1, 2, 3, 4, 5});
-    std::vector<int> ParSeed({6, 7, 8, 9, 10});
-
-    // Create integrator, including the smearing policy
-    // Smearing policy, only defined for Nc=3
-    /*
-    std::cout << GridLogDebug << " Creating the Stout class\n";
-    double rho = 0.1;  // smearing parameter, now hardcoded
-    int Nsmear = 1;    // number of smearing levels
-    Smear_Stout<Gimpl> Stout(rho);
-    std::cout << GridLogDebug << " Creating the SmearedConfiguration class\n";
-    //SmearedConfiguration<Gimpl> SmearingPolicy(UGrid, Nsmear, Stout);
-    std::cout << GridLogDebug << " done\n";
-    */
-    //////////////
-    NoSmearing<Gimpl> SmearingPolicy;
-    // change here to change the algorithm
-    typedef MinimumNorm2<GaugeField, NoSmearing<Gimpl>, RepresentationsPolicy >  IntegratorType;  
-    IntegratorParameters MDpar(40, 1.0);
-    IntegratorType MDynamics(UGrid, MDpar, TheAction, SmearingPolicy);
-
-    // Checkpoint strategy
-    NerscHmcCheckpointer<Gimpl> Checkpoint(std::string("ckpoint_lat"),
-                                           std::string("ckpoint_rng"), 1);
-    PlaquetteLogger<Gimpl> PlaqLog(std::string("plaq"));
-
-    HMCparameters HMCpar;
-    HMCpar.StartTrajectory = StartTraj;
-    HMCpar.Trajectories = NumTraj;
-    HMCpar.NoMetropolisUntil = NumThermalizations;
-
-    if (StartType == HotStart) {
-      // Hot start
-      HMCpar.MetropolisTest = true;
-      sRNG.SeedFixedIntegers(SerSeed);
-      pRNG.SeedFixedIntegers(ParSeed);
-      SU<Nc>::HotConfiguration(pRNG, U);
-    } else if (StartType == ColdStart) {
-      // Cold start
-      HMCpar.MetropolisTest = true;
-      sRNG.SeedFixedIntegers(SerSeed);
-      pRNG.SeedFixedIntegers(ParSeed);
-      SU<Nc>::ColdConfiguration(pRNG, U);
-    } else if (StartType == TepidStart) {
-      // Tepid start
-      HMCpar.MetropolisTest = true;
-      sRNG.SeedFixedIntegers(SerSeed);
-      pRNG.SeedFixedIntegers(ParSeed);
-      SU<Nc>::TepidConfiguration(pRNG, U);
-    } else if (StartType == CheckpointStart) {
-      HMCpar.MetropolisTest = true;
-      // CheckpointRestart
-      Checkpoint.CheckpointRestore(StartTraj, U, sRNG, pRNG);
-    }
-
-    // Attach the gauge field to the smearing Policy and create the fill the
-    // smeared set
-    // notice that the unit configuration is singular in this procedure
-    std::cout << GridLogMessage << "Filling the smeared set\n";
-    SmearingPolicy.set_GaugeField(U);
-
-    HybridMonteCarlo<GaugeField, IntegratorType> HMC(HMCpar, MDynamics, sRNG,
-                                                     pRNG, U);
-    HMC.AddObservable(&Checkpoint);
-    HMC.AddObservable(&PlaqLog);
-
-    // Run it
-    HMC.evolve();
-  }
-};
-
-typedef NerscHmcRunnerTemplate<PeriodicGimplR> NerscHmcRunner;
-typedef NerscHmcRunnerTemplate<PeriodicGimplF> NerscHmcRunnerF;
-typedef NerscHmcRunnerTemplate<PeriodicGimplD> NerscHmcRunnerD;
-
-typedef NerscHmcRunnerTemplate<PeriodicGimplR> PeriodicNerscHmcRunner;
-typedef NerscHmcRunnerTemplate<PeriodicGimplF> PeriodicNerscHmcRunnerF;
-typedef NerscHmcRunnerTemplate<PeriodicGimplD> PeriodicNerscHmcRunnerD;
-
-typedef NerscHmcRunnerTemplate<ConjugateGimplR> ConjugateNerscHmcRunner;
-typedef NerscHmcRunnerTemplate<ConjugateGimplF> ConjugateNerscHmcRunnerF;
-typedef NerscHmcRunnerTemplate<ConjugateGimplD> ConjugateNerscHmcRunnerD;
-
-template <class RepresentationsPolicy>
-using NerscHmcRunnerHirep = NerscHmcRunnerTemplate<PeriodicGimplR, RepresentationsPolicy>;
-
-
-
-}
-}
-#endif
@@ -1,80 +0,0 @@
-    /*************************************************************************************
-
-    Grid physics library, www.github.com/paboyle/Grid 
-
-    Source file: ./lib/qcd/hmc/NerscCheckpointer.h
-
-    Copyright (C) 2015
-
-Author: paboyle <paboyle@ph.ed.ac.uk>
-
-    This program is free software; you can redistribute it and/or modify
-    it under the terms of the GNU General Public License as published by
-    the Free Software Foundation; either version 2 of the License, or
-    (at your option) any later version.
-
-    This program is distributed in the hope that it will be useful,
-    but WITHOUT ANY WARRANTY; without even the implied warranty of
-    MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
-    GNU General Public License for more details.
-
-    You should have received a copy of the GNU General Public License along
-    with this program; if not, write to the Free Software Foundation, Inc.,
-    51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
-
-    See the full license in the file "LICENSE" in the top level distribution directory
-    *************************************************************************************/
-    /*  END LEGAL */
-#ifndef NERSC_CHECKPOINTER
-#define NERSC_CHECKPOINTER
-
-#include <string>
-#include <iostream>
-#include <sstream>
-
-
-namespace Grid{
-  namespace QCD{
-    
-
-    template<class Gimpl> 
-    class NerscHmcCheckpointer : public HmcObservable<typename Gimpl::GaugeField> {
-    private:
-      std::string configStem;
-      std::string rngStem;
-      int SaveInterval;
-    public:
-      INHERIT_GIMPL_TYPES(Gimpl);
-
-      NerscHmcCheckpointer(std::string cf, std::string rn,int savemodulo) {
-        configStem  = cf;
-        rngStem     = rn;
-        SaveInterval= savemodulo;
-      };
-
-      void TrajectoryComplete(int traj, GaugeField &U, GridSerialRNG &sRNG, GridParallelRNG & pRNG )
-      {
-	if ( (traj % SaveInterval)== 0 ) {
-	  std::string rng;   { std::ostringstream os; os << rngStem     <<"."<< traj; rng = os.str(); }
-	  std::string config;{ std::ostringstream os; os << configStem  <<"."<< traj; config = os.str();}
-
-	  int precision32=1;
-	  int tworow     =0;
-	  NerscIO::writeRNGState(sRNG,pRNG,rng);
-	  NerscIO::writeConfiguration(U,config,tworow,precision32);
-	}
-      };
-
-      void CheckpointRestore(int traj, GaugeField &U, GridSerialRNG &sRNG, GridParallelRNG & pRNG ){
-
-	std::string rng;   { std::ostringstream os; os << rngStem     <<"."<< traj; rng = os.str(); }
-	std::string config;{ std::ostringstream os; os << configStem  <<"."<< traj; config = os.str();}
-
-	NerscField header;
-	NerscIO::readRNGState(sRNG,pRNG,header,rng);
-	NerscIO::readConfiguration(U,header,config);
-      };
-
-    };
-}}
-#endif
@@ -0,0 +1,107 @@
+Using HMC in Grid version 0.5.1
+
+These are the instructions to use the Generalised HMC on Grid version 0.5.1.
+Disclaimer: GRID is still under active development so any information here can be changed in future releases.
+
+
+Command line options
+===================
+(relevant file GenericHMCrunner.h)
+The initial configuration can be changed at the command line using 
+--StartType <your choice>
+valid choices, one among these
+HotStart, ColdStart, TepidStart, CheckpointStart
+default: HotStart
+
+example
+./My_hmc_exec  --StartType HotStart
+
+The CheckpointStart option uses the prefix for the configurations and rng seed files defined in your executable and the initial configuration is specified by
+--StartTrajectory <integer>
+default: 0
+
+The number of trajectories for a specific run are specified at command line by
+--Trajectories <integer>
+default: 1
+
+The number of thermalization steps (i.e. steps when the Metropolis acceptance check is turned off) is specified by
+--Thermalizations <integer>
+default: 10
+
+
+Any other parameter is defined in the source for the executable.
+
+HMC controls
+===========
+
+The lines 
+
+  std::vector<int> SerSeed({1, 2, 3, 4, 5});
+  std::vector<int> ParSeed({6, 7, 8, 9, 10});
+
+define the seeds for the serial and the parallel RNG.
+
+The line 
+
+  TheHMC.MDparameters.set(20, 1.0);// MDsteps, traj length
+
+declares the number of molecular dynamics steps and the total trajectory length.
+
+
+Actions
+======
+
+Action names are defined in the file
+lib/qcd/Actions.h
+
+Gauge actions list:
+
+WilsonGaugeActionR;
+WilsonGaugeActionF;
+WilsonGaugeActionD;
+PlaqPlusRectangleActionR;
+PlaqPlusRectangleActionF;
+PlaqPlusRectangleActionD;
+IwasakiGaugeActionR;
+IwasakiGaugeActionF;
+IwasakiGaugeActionD;
+SymanzikGaugeActionR;
+SymanzikGaugeActionF;
+SymanzikGaugeActionD;
+
+
+ConjugateWilsonGaugeActionR;
+ConjugateWilsonGaugeActionF;
+ConjugateWilsonGaugeActionD;
+ConjugatePlaqPlusRectangleActionR;
+ConjugatePlaqPlusRectangleActionF;
+ConjugatePlaqPlusRectangleActionD;
+ConjugateIwasakiGaugeActionR;
+ConjugateIwasakiGaugeActionF;
+ConjugateIwasakiGaugeActionD;
+ConjugateSymanzikGaugeActionR;
+ConjugateSymanzikGaugeActionF;
+ConjugateSymanzikGaugeActionD;
+
+
+ScalarActionR;
+ScalarActionF;
+ScalarActionD;
+
+
+each of these action accept one single parameter at creation time (beta).
+Example for creating a Symanzik action with beta=4.0
+
+	SymanzikGaugeActionR(4.0)
+
+The suffixes R,F,D in the action names refer to the Real
+(the precision is defined at compile time by the --enable-precision flag in the configure),
+Float and Double, that force the precision of the action to be 32, 64 bit respectively.
+
+
+
+
+
+
+
+
@@ -0,0 +1,89 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/BaseCheckpointer.h
+
+Copyright (C) 2015
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef BASE_CHECKPOINTER
+#define BASE_CHECKPOINTER
+
+namespace Grid {
+namespace QCD {
+
+class CheckpointerParameters : Serializable {
+ public:
+  GRID_SERIALIZABLE_CLASS_MEMBERS(CheckpointerParameters, 
+  	std::string, config_prefix, 
+  	std::string, rng_prefix, 
+  	int, saveInterval, 
+  	std::string, format, );
+
+  CheckpointerParameters(std::string cf = "cfg", std::string rn = "rng",
+   		      int savemodulo = 1, const std::string &f = "IEEE64BIG")
+      : config_prefix(cf),
+        rng_prefix(rn),
+        saveInterval(savemodulo),
+        format(f){};
+
+
+  template <class ReaderClass >
+  CheckpointerParameters(Reader<ReaderClass> &Reader) {
+    read(Reader, "Checkpointer", *this);
+  }
+ 
+
+};
+
+//////////////////////////////////////////////////////////////////////////////
+// Base class for checkpointers
+template <class Impl>
+class BaseHmcCheckpointer : public HmcObservable<typename Impl::Field> {
+ public:
+  void build_filenames(int traj, CheckpointerParameters &Params,
+                       std::string &conf_file, std::string &rng_file) {
+    {
+      std::ostringstream os;
+      os << Params.rng_prefix << "." << traj;
+      rng_file = os.str();
+    }
+
+    {
+      std::ostringstream os;
+      os << Params.config_prefix << "." << traj;
+      conf_file = os.str();
+    }
+ 	} 
+
+  virtual void initialize(const CheckpointerParameters &Params) = 0;
+
+  virtual void CheckpointRestore(int traj, typename Impl::Field &U,
+                                 GridSerialRNG &sRNG,
+                                 GridParallelRNG &pRNG) = 0;
+
+};  // class BaseHmcCheckpointer
+///////////////////////////////////////////////////////////////////////////////
+}
+}
+#endif
@@ -0,0 +1,113 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/BinaryCheckpointer.h
+
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef BINARY_CHECKPOINTER
+#define BINARY_CHECKPOINTER
+
+#include <iostream>
+#include <sstream>
+#include <string>
+
+namespace Grid {
+namespace QCD {
+
+// Simple checkpointer, only binary file
+template <class Impl>
+class BinaryHmcCheckpointer : public BaseHmcCheckpointer<Impl> {
+ private:
+  CheckpointerParameters Params;
+
+ public:
+  INHERIT_FIELD_TYPES(Impl);  // Gets the Field type, a Lattice object
+
+  // Extract types from the Field
+  typedef typename Field::vector_object vobj;
+  typedef typename vobj::scalar_object sobj;
+  typedef typename getPrecision<sobj>::real_scalar_type sobj_stype;
+  typedef typename sobj::DoublePrecision sobj_double;
+
+  BinaryHmcCheckpointer(const CheckpointerParameters &Params_) {
+    initialize(Params_);
+  }
+
+  void initialize(const CheckpointerParameters &Params_) { Params = Params_; }
+
+  void truncate(std::string file) {
+    std::ofstream fout(file, std::ios::out);
+    fout.close();
+  }
+
+  void TrajectoryComplete(int traj, Field &U, GridSerialRNG &sRNG, GridParallelRNG &pRNG) {
+
+    if ((traj % Params.saveInterval) == 0) {
+      std::string config, rng;
+      this->build_filenames(traj, Params, config, rng);
+
+      uint32_t nersc_csum;
+      uint32_t scidac_csuma;
+      uint32_t scidac_csumb;
+      
+      BinarySimpleUnmunger<sobj_double, sobj> munge;
+      truncate(rng);
+      BinaryIO::writeRNG(sRNG, pRNG, rng, 0,nersc_csum,scidac_csuma,scidac_csumb);
+      truncate(config);
+
+      BinaryIO::writeLatticeObject<vobj, sobj_double>(U, config, munge, 0, Params.format,
+						      nersc_csum,scidac_csuma,scidac_csumb);
+
+      std::cout << GridLogMessage << "Written Binary Configuration " << config
+                << " checksum " << std::hex 
+		<< nersc_csum   <<"/"
+		<< scidac_csuma   <<"/"
+		<< scidac_csumb 
+		<< std::dec << std::endl;
+    }
+
+  };
+
+  void CheckpointRestore(int traj, Field &U, GridSerialRNG &sRNG, GridParallelRNG &pRNG) {
+    std::string config, rng;
+    this->build_filenames(traj, Params, config, rng);
+
+    BinarySimpleMunger<sobj_double, sobj> munge;
+
+    uint32_t nersc_csum;
+    uint32_t scidac_csuma;
+    uint32_t scidac_csumb;
+    BinaryIO::readRNG(sRNG, pRNG, rng, 0,nersc_csum,scidac_csuma,scidac_csumb);
+    BinaryIO::readLatticeObject<vobj, sobj_double>(U, config, munge, 0, Params.format,
+						   nersc_csum,scidac_csuma,scidac_csumb);
+    
+    std::cout << GridLogMessage << "Read Binary Configuration " << config
+              << " checksums " << std::hex << nersc_csum<<"/"<<scidac_csuma<<"/"<<scidac_csumb 
+	      << std::dec << std::endl;
+  };
+};
+}
+}
+#endif
@@ -0,0 +1,158 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/action/gauge/WilsonGaugeAction.h
+
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+
+#ifndef CP_MODULES_H
+#define CP_MODULES_H
+
+
+// FIXME  Reorganize QCD namespace 
+
+
+namespace Grid {
+
+////////////////////////////////////////////////////////////////////////
+// Checkpoint module, owns the Checkpointer
+////////////////////////////////////////////////////////////////////////
+
+template <class ImplementationPolicy>
+class CheckPointerModule: public Parametrized<QCD::CheckpointerParameters>, public HMCModuleBase< QCD::BaseHmcCheckpointer<ImplementationPolicy> >  {
+ public:
+ 	std::unique_ptr<QCD::BaseHmcCheckpointer<ImplementationPolicy> > CheckPointPtr;
+ 	typedef QCD::CheckpointerParameters APar;
+  typedef HMCModuleBase< QCD::BaseHmcCheckpointer<ImplementationPolicy> > Base;
+  typedef typename Base::Product Product;
+
+  CheckPointerModule(APar Par): Parametrized<APar>(Par) {}
+  template <class ReaderClass>
+  CheckPointerModule(Reader<ReaderClass>& Reader) : Parametrized<APar>(Reader){};
+
+  virtual void print_parameters(){
+  	std::cout << this->Par_ << std::endl;
+  }
+
+  Product* getPtr() {
+    if (!CheckPointPtr) initialize();
+
+    return CheckPointPtr.get();
+  }
+
+ private:
+  virtual void initialize() = 0;
+
+};
+
+
+
+template <char const *str, class ImplementationPolicy, class ReaderClass >
+class HMC_CPModuleFactory
+    : public Factory < HMCModuleBase< QCD::BaseHmcCheckpointer<ImplementationPolicy> > ,	Reader<ReaderClass> > {
+ public:
+ 	typedef Reader<ReaderClass> TheReader; 
+ 	// use SINGLETON FUNCTOR MACRO HERE
+  HMC_CPModuleFactory(const HMC_CPModuleFactory& e) = delete;
+  void operator=(const HMC_CPModuleFactory& e) = delete;
+  static HMC_CPModuleFactory& getInstance(void) {
+    static HMC_CPModuleFactory e;
+    return e;
+  }
+
+ private:
+  HMC_CPModuleFactory(void) = default;
+  std::string obj_type() const {
+  	return std::string(str);
+  }
+};
+
+
+
+/////////////////////////////////////////////////////////////////////
+// Concrete classes
+/////////////////////////////////////////////////////////////////////
+namespace QCD{
+
+template<class ImplementationPolicy>
+class BinaryCPModule: public CheckPointerModule< ImplementationPolicy> {
+  typedef CheckPointerModule< ImplementationPolicy> CPBase;
+  using CPBase::CPBase; // for constructors
+
+  // acquire resource
+  virtual void initialize(){
+    this->CheckPointPtr.reset(new BinaryHmcCheckpointer<ImplementationPolicy>(this->Par_));
+  }
+
+};
+
+
+template<class ImplementationPolicy>
+class NerscCPModule: public CheckPointerModule< ImplementationPolicy> {
+  typedef CheckPointerModule< ImplementationPolicy> CPBase;
+  using CPBase::CPBase; // for constructors inheritance
+
+  // acquire resource
+  virtual void initialize(){
+     this->CheckPointPtr.reset(new NerscHmcCheckpointer<ImplementationPolicy>(this->Par_));
+  }
+
+};
+
+
+#ifdef HAVE_LIME
+  
+template<class ImplementationPolicy>
+class ILDGCPModule: public CheckPointerModule< ImplementationPolicy> {
+  typedef CheckPointerModule< ImplementationPolicy> CPBase;
+  using CPBase::CPBase; // for constructors
+
+  // acquire resource
+  virtual void initialize(){
+     this->CheckPointPtr.reset(new ILDGHmcCheckpointer<ImplementationPolicy>(this->Par_));
+  }
+
+};
+
+#endif
+
+
+}// QCD temporarily here
+
+
+extern char cp_string[];
+
+/*
+// use macros?
+static Registrar<QCD::BinaryCPModule<QCD::PeriodicGimplR>, HMC_CPModuleFactory<cp_string, QCD::PeriodicGimplR, XmlReader> > __CPBinarymodXMLInit("Binary");
+static Registrar<QCD::NerscCPModule<QCD::PeriodicGimplR> , HMC_CPModuleFactory<cp_string, QCD::PeriodicGimplR, XmlReader> > __CPNerscmodXMLInit("Nersc");
+
+#ifdef HAVE_LIME
+static Registrar<QCD::ILDGCPModule<QCD::PeriodicGimplR>  , HMC_CPModuleFactory<cp_string, QCD::PeriodicGimplR, XmlReader> > __CPILDGmodXMLInit("ILDG");
+#endif
+*/
+
+}// Grid
+#endif //CP_MODULES_H
@@ -0,0 +1,40 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/action/gauge/WilsonGaugeAction.h
+
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+
+#ifndef CHECKPOINTERS_H
+#define CHECKPOINTERS_H
+
+#include <Grid/qcd/hmc/checkpointers/BaseCheckpointer.h>
+#include <Grid/qcd/hmc/checkpointers/NerscCheckpointer.h>
+#include <Grid/qcd/hmc/checkpointers/BinaryCheckpointer.h>
+#include <Grid/qcd/hmc/checkpointers/ILDGCheckpointer.h>
+//#include <Grid/qcd/hmc/checkpointers/CheckPointerModules.h>
+
+
+#endif // CHECKPOINTERS_H
@@ -0,0 +1,120 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/ILDGCheckpointer.h
+
+Copyright (C) 2016
+
+Author: Guido Cossu <guido.cossu@ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef ILDG_CHECKPOINTER
+#define ILDG_CHECKPOINTER
+
+#ifdef HAVE_LIME
+
+#include <iostream>
+#include <sstream>
+#include <string>
+
+namespace Grid {
+namespace QCD {
+
+// Only for Gauge fields
+template <class Implementation>
+class ILDGHmcCheckpointer : public BaseHmcCheckpointer<Implementation> {
+ private:
+  CheckpointerParameters Params;
+
+ public:
+  INHERIT_GIMPL_TYPES(Implementation);
+
+  ILDGHmcCheckpointer(const CheckpointerParameters &Params_) { initialize(Params_); }
+
+  void initialize(const CheckpointerParameters &Params_) {
+    Params = Params_;
+
+    // check here that the format is valid
+    int ieee32big = (Params.format == std::string("IEEE32BIG"));
+    int ieee32    = (Params.format == std::string("IEEE32"));
+    int ieee64big = (Params.format == std::string("IEEE64BIG"));
+    int ieee64    = (Params.format == std::string("IEEE64"));
+
+    if (!(ieee64big || ieee32 || ieee32big || ieee64)) {
+      std::cout << GridLogError << "Unrecognized file format " << Params.format
+                << std::endl;
+      std::cout << GridLogError
+                << "Allowed: IEEE32BIG | IEEE32 | IEEE64BIG | IEEE64"
+                << std::endl;
+
+      exit(1);
+    }
+  }
+
+  void TrajectoryComplete(int traj, GaugeField &U, GridSerialRNG &sRNG,
+                          GridParallelRNG &pRNG) {
+    if ((traj % Params.saveInterval) == 0) {
+      std::string config, rng;
+      this->build_filenames(traj, Params, config, rng);
+      
+      uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+      BinaryIO::writeRNG(sRNG, pRNG, rng, 0,nersc_csum,scidac_csuma,scidac_csumb);
+      IldgWriter _IldgWriter;
+      _IldgWriter.open(config);
+      _IldgWriter.writeConfiguration(U, traj, config, config);
+      _IldgWriter.close();
+
+      std::cout << GridLogMessage << "Written ILDG Configuration on " << config
+                << " checksum " << std::hex 
+		<< nersc_csum<<"/"
+		<< scidac_csuma<<"/"
+		<< scidac_csumb
+		<< std::dec << std::endl;
+    }
+  };
+
+  void CheckpointRestore(int traj, GaugeField &U, GridSerialRNG &sRNG,
+                         GridParallelRNG &pRNG) {
+    std::string config, rng;
+    this->build_filenames(traj, Params, config, rng);
+
+    uint32_t nersc_csum,scidac_csuma,scidac_csumb;
+    BinaryIO::readRNG(sRNG, pRNG, rng, 0,nersc_csum,scidac_csuma,scidac_csumb);
+
+    FieldMetaData header;
+    IldgReader _IldgReader;
+    _IldgReader.open(config);
+    _IldgReader.readConfiguration(U,header);  // format from the header
+    _IldgReader.close();
+
+    std::cout << GridLogMessage << "Read ILDG Configuration from " << config
+              << " checksum " << std::hex 
+	      << nersc_csum<<"/"
+	      << scidac_csuma<<"/"
+	      << scidac_csumb
+	      << std::dec << std::endl;
+  };
+};
+}
+}
+
+#endif  // HAVE_LIME
+#endif  // ILDG_CHECKPOINTER
@@ -0,0 +1,80 @@
+/*************************************************************************************
+
+Grid physics library, www.github.com/paboyle/Grid
+
+Source file: ./lib/qcd/hmc/NerscCheckpointer.h
+
+Copyright (C) 2015
+
+Author: paboyle <paboyle@ph.ed.ac.uk>
+
+This program is free software; you can redistribute it and/or modify
+it under the terms of the GNU General Public License as published by
+the Free Software Foundation; either version 2 of the License, or
+(at your option) any later version.
+
+This program is distributed in the hope that it will be useful,
+but WITHOUT ANY WARRANTY; without even the implied warranty of
+MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+GNU General Public License for more details.
+
+You should have received a copy of the GNU General Public License along
+with this program; if not, write to the Free Software Foundation, Inc.,
+51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
+
+See the full license in the file "LICENSE" in the top level distribution
+directory
+*************************************************************************************/
+/*  END LEGAL */
+#ifndef NERSC_CHECKPOINTER
+#define NERSC_CHECKPOINTER
+
+#include <iostream>
+#include <sstream>
+#include <string>
+
+namespace Grid {
+namespace QCD {
+
+// Only for Gauge fields
+template <class Gimpl>
+class NerscHmcCheckpointer : public BaseHmcCheckpointer<Gimpl> {
+ private:
+  CheckpointerParameters Params;
+
+ public:
+  INHERIT_GIMPL_TYPES(Gimpl);  // only for gauge configurations
+
+  NerscHmcCheckpointer(const CheckpointerParameters &Params_) { initialize(Params_); }
+
+  void initialize(const CheckpointerParameters &Params_) {
+    Params = Params_;
+    Params.format = "IEEE64BIG";  // fixed, overwrite any other choice
+  }
+
+  void TrajectoryComplete(int traj, GaugeField &U, GridSerialRNG &sRNG,
+                          GridParallelRNG &pRNG) {
+    if ((traj % Params.saveInterval) == 0) {
+      std::string config, rng;
+      this->build_filenames(traj, Params, config, rng);
+
+      int precision32 = 1;
+      int tworow = 0;
+      NerscIO::writeRNGState(sRNG, pRNG, rng);
+      NerscIO::writeConfiguration(U, config, tworow, precision32);
+    }
+  };
+
+  void CheckpointRestore(int traj, GaugeField &U, GridSerialRNG &sRNG,
+                         GridParallelRNG &pRNG) {
+    std::string config, rng;
+    this->build_filenames(traj, Params, config, rng);
+
+    FieldMetaData header;
+    NerscIO::readRNGState(sRNG, pRNG, header, rng);
+    NerscIO::readConfiguration(U, header, config);
+  };
+};
+}
+}
+#endif
@@ -8,8 +8,7 @@ Copyright (C) 2015

 Author: Azusa Yamaguchi <ayamaguc@staffmail.ed.ac.uk>
 Author: Peter Boyle <paboyle@ph.ed.ac.uk>
-Author: neo <cossu@post.kek.jp>
-Author: paboyle <paboyle@ph.ed.ac.uk>
+Author: Guido Cossu <cossu@post.kek.jp>

 This program is free software; you can redistribute it and/or modify
 it under the terms of the GNU General Public License as published by
@@ -30,79 +29,62 @@ directory
 *************************************************************************************/
 /*  END LEGAL */
 //--------------------------------------------------------------------
-/*! @file Integrator.h
- * @brief Classes for the Molecular Dynamics integrator
- *
- * @author Guido Cossu
- * Time-stamp: <2015-07-30 16:21:29 neo>
- */
-//--------------------------------------------------------------------
-
 #ifndef INTEGRATOR_INCLUDED
 #define INTEGRATOR_INCLUDED

-// class Observer;
-
 #include <memory>

 namespace Grid {
 namespace QCD {

-struct IntegratorParameters {
-  int Nexp;
-  int MDsteps;  // number of outer steps
-  RealD trajL;  // trajectory length
-  RealD stepsize;
+class IntegratorParameters: Serializable {
+public:
+        GRID_SERIALIZABLE_CLASS_MEMBERS(IntegratorParameters,
+                std::string, name,      // name of the integrator
+    unsigned int, MDsteps,  // number of outer steps
+    RealD, trajL,           // trajectory length
+  )

-  IntegratorParameters(int MDsteps_, RealD trajL_ = 1.0, int Nexp_ = 12)
-      : Nexp(Nexp_),
-        MDsteps(MDsteps_),
-        trajL(trajL_),
-        stepsize(trajL / MDsteps){
-            // empty body constructor
-        };
+  IntegratorParameters(int MDsteps_ = 10, RealD trajL_ = 1.0)
+      : MDsteps(MDsteps_),
+        trajL(trajL_){
+        // empty body constructor
+  };
+
+
+  template <class ReaderClass, typename std::enable_if<isReader<ReaderClass>::value, int >::type = 0 >
+  IntegratorParameters(ReaderClass & Reader){
+    std::cout << "Reading integrator\n";
+        read(Reader, "Integrator", *this);
+  }
+
+  void print_parameters() const {
+    std::cout << GridLogMessage << "[Integrator] Type               : " << name << std::endl;
+    std::cout << GridLogMessage << "[Integrator] Trajectory length  : " << trajL << std::endl;
+    std::cout << GridLogMessage << "[Integrator] Number of MD steps : " << MDsteps << std::endl;
+    std::cout << GridLogMessage << "[Integrator] Step size          : " << trajL/MDsteps << std::endl;
+  }
 };

 /*! @brief Class for Molecular Dynamics management */
-template <class GaugeField, class SmearingPolicy, class RepresentationPolicy>
+template <class FieldImplementation, class SmearingPolicy, class RepresentationPolicy>
 class Integrator {
 protected:
-  typedef IntegratorParameters ParameterType;
+  typedef typename FieldImplementation::Field MomentaField;  //for readability
+  typedef typename FieldImplementation::Field Field;

+  int levels;  // number of integration levels
+  double t_U;  // Track time passing on each level and for U and for P
+  std::vector<double> t_P;  
+
+  MomentaField P;
+  SmearingPolicy& Smearer;
+  RepresentationPolicy Representations;
  IntegratorParameters Params;

-  const ActionSet<GaugeField, RepresentationPolicy> as;
+  const ActionSet<Field, RepresentationPolicy> as;

-  int levels;  //
-  double t_U;  // Track time passing on each level and for U and for P
-  std::vector<double> t_P;  //
-
-  GaugeField P;
-
-  SmearingPolicy& Smearer;
-
-  RepresentationPolicy Representations;
-
-  // Should match any legal (SU(n)) gauge field
-  // Need to use this template to match Ncol to pass to SU<N> class
-  template <int Ncol, class vec>
-  void generate_momenta(Lattice<iVector<iScalar<iMatrix<vec, Ncol> >, Nd> >& P,
-                        GridParallelRNG& pRNG) {
-    typedef Lattice<iScalar<iScalar<iMatrix<vec, Ncol> > > > GaugeLinkField;
-    GaugeLinkField Pmu(P._grid);
-    Pmu = zero;
-    for (int mu = 0; mu < Nd; mu++) {
-      SU<Ncol>::GaussianFundamentalLieAlgebraMatrix(pRNG, Pmu);
-      PokeIndex<LorentzIndex>(P, Pmu, mu);
-    }
-  }
-
-  // ObserverList observers; // not yet
-  //      typedef std::vector<Observer*> ObserverList;
-  //      void register_observers();
-  //      void notify_observers();
-
-  void update_P(GaugeField& U, int level, double ep) {
+  void update_P(Field& U, int level, double ep) {
    t_P[level] += ep;
    update_P(P, U, level, ep);

@@ -120,67 +102,60 @@ class Integrator {
        FieldType forceR(U._grid);
        // Implement smearing only for the fundamental representation now
        repr_set.at(a)->deriv(Rep.U, forceR);
-        GF force =
-            Rep.RtoFundamentalProject(forceR);  // Ta for the fundamental rep
-        std::cout << GridLogIntegrator << "Hirep Force average: "
-                  << norm2(force) / (U._grid->gSites()) << std::endl;
+        GF force = Rep.RtoFundamentalProject(forceR);  // Ta for the fundamental rep
+        Real force_abs = std::sqrt(norm2(force)/(U._grid->gSites()));
+        std::cout << GridLogIntegrator << "Hirep Force average: " << force_abs << std::endl;
        Mom -= force * ep ;
      }
    }
  } update_P_hireps{};

-  void update_P(GaugeField& Mom, GaugeField& U, int level, double ep) {
+  void update_P(MomentaField& Mom, Field& U, int level, double ep) {
    // input U actually not used in the fundamental case
    // Fundamental updates, include smearing
-    for (int a = 0; a < as[level].actions.size(); ++a) {
-      GaugeField force(U._grid);
-      GaugeField& Us = Smearer.get_U(as[level].actions.at(a)->is_smeared);
+
+   for (int a = 0; a < as[level].actions.size(); ++a) {
+      Field force(U._grid);
+      conformable(U._grid, Mom._grid);
+      Field& Us = Smearer.get_U(as[level].actions.at(a)->is_smeared);
      as[level].actions.at(a)->deriv(Us, force);  // deriv should NOT include Ta

-      std::cout << GridLogIntegrator
-                << "Smearing (on/off): " << as[level].actions.at(a)->is_smeared
-                << std::endl;
+      std::cout << GridLogIntegrator << "Smearing (on/off): " << as[level].actions.at(a)->is_smeared << std::endl;
      if (as[level].actions.at(a)->is_smeared) Smearer.smeared_force(force);
-      force = Ta(force);
-      std::cout << GridLogIntegrator
-                << "Force average: " << norm2(force) / (U._grid->gSites())
-                << std::endl;
-      Mom -= force * ep;
+      force = FieldImplementation::projectForce(force); // Ta for gauge fields
+      Real force_abs = std::sqrt(norm2(force)/U._grid->gSites());
+      std::cout << GridLogIntegrator << "Force average: " << force_abs << std::endl;
+      Mom -= force * ep; 
    }

    // Force from the other representations
    as[level].apply(update_P_hireps, Representations, Mom, U, ep);
  }

-  void update_U(GaugeField& U, double ep) {
+  void update_U(Field& U, double ep) {
    update_U(P, U, ep);

    t_U += ep;
    int fl = levels - 1;
-    std::cout << GridLogIntegrator << "   "
-              << "[" << fl << "] U "
-              << " dt " << ep << " : t_U " << t_U << std::endl;
+    std::cout << GridLogIntegrator << "   " << "[" << fl << "] U " << " dt " << ep << " : t_U " << t_U << std::endl;
  }
-  void update_U(GaugeField& Mom, GaugeField& U, double ep) {
-    // rewrite exponential to deal internally with the lorentz index?
-    for (int mu = 0; mu < Nd; mu++) {
-      auto Umu = PeekIndex<LorentzIndex>(U, mu);
-      auto Pmu = PeekIndex<LorentzIndex>(Mom, mu);
-      Umu = expMat(Pmu, ep, Params.Nexp) * Umu;
-      PokeIndex<LorentzIndex>(U, ProjectOnGroup(Umu), mu);
-    }
-    
+  
+  void update_U(MomentaField& Mom, Field& U, double ep) {
+    // exponential of Mom*U in the gauge fields case
+    FieldImplementation::update_field(Mom, U, ep);
+
    // Update the smeared fields, can be implemented as observer
-    Smearer.set_GaugeField(U);
+    Smearer.set_Field(U);
+
    // Update the higher representations fields
    Representations.update(U);  // void functions if fundamental representation
  }

-  virtual void step(GaugeField& U, int level, int first, int last) = 0;
+  virtual void step(Field& U, int level, int first, int last) = 0;

 public:
  Integrator(GridBase* grid, IntegratorParameters Par,
-             ActionSet<GaugeField, RepresentationPolicy>& Aset,
+             ActionSet<Field, RepresentationPolicy>& Aset,
             SmearingPolicy& Sm)
      : Params(Par),
        as(Aset),
@@ -195,6 +170,31 @@ class Integrator {

  virtual ~Integrator() {}

+  virtual std::string integrator_name() = 0;
+
+  void print_parameters(){
+    std::cout << GridLogMessage << "[Integrator] Name : "<< integrator_name() << std::endl;
+    Params.print_parameters();
+  }
+
+  void print_actions(){
+        std::cout << GridLogMessage << ":::::::::::::::::::::::::::::::::::::::::" << std::endl;
+        std::cout << GridLogMessage << "[Integrator] Action summary: "<<std::endl;
+        for (int level = 0; level < as.size(); ++level) {
+                std::cout << GridLogMessage << "[Integrator] ---- Level: "<< level << std::endl;
+                for (int actionID = 0; actionID < as[level].actions.size(); ++actionID) {
+                        std::cout << GridLogMessage << "["<< as[level].actions.at(actionID)->action_name() << "] ID: " << actionID << std::endl;
+                        std::cout << as[level].actions.at(actionID)->LogParameters();
+                }
+        }
+        std::cout << GridLogMessage << ":::::::::::::::::::::::::::::::::::::::::"<< std::endl;
+
+  }
+
+  void reverse_momenta(){
+    P *= -1.0;
+  }
+
  // to be used by the actionlevel class to iterate
  // over the representations
  struct _refresh {
@@ -210,14 +210,16 @@ class Integrator {
  } refresh_hireps{};

  // Initialization of momenta and actions
-  void refresh(GaugeField& U, GridParallelRNG& pRNG) {
+  void refresh(Field& U, GridParallelRNG& pRNG) {
+    assert(P._grid == U._grid);
    std::cout << GridLogIntegrator << "Integrator refresh\n";
-    generate_momenta(P, pRNG);
+
+    FieldImplementation::generate_momenta(P, pRNG);

    // Update the smeared fields, can be implemented as observer
    // necessary to keep the fields updated even after a reject
    // of the Metropolis
-    Smearer.set_GaugeField(U);
+    Smearer.set_Field(U);
    // Set the (eventual) representations gauge fields
    Representations.update(U);

@@ -228,7 +230,7 @@ class Integrator {
      for (int actionID = 0; actionID < as[level].actions.size(); ++actionID) {
        // get gauge field from the SmearingPolicy and
        // based on the boolean is_smeared in actionID
-        GaugeField& Us =
+        Field& Us =
            Smearer.get_U(as[level].actions.at(actionID)->is_smeared);
        as[level].actions.at(actionID)->refresh(Us, pRNG);
      }
@@ -256,18 +258,9 @@ class Integrator {
  } S_hireps{};

  // Calculate action
-  RealD S(GaugeField& U) {  // here also U not used
+  RealD S(Field& U) {  // here also U not used

-    LatticeComplex Hloc(U._grid);
-    Hloc = zero;
-    // Momenta
-    for (int mu = 0; mu < Nd; mu++) {
-      auto Pmu = PeekIndex<LorentzIndex>(P, mu);
-      Hloc -= trace(Pmu * Pmu);
-    }
-    Complex Hsum = sum(Hloc);
-
-    RealD H = Hsum.real();
+    RealD H = - FieldImplementation::FieldSquareNorm(P); // - trace (P*P)
    RealD Hterm;
    std::cout << GridLogMessage << "Momentum action H_p = " << H << "\n";

@@ -276,7 +269,7 @@ class Integrator {
      for (int actionID = 0; actionID < as[level].actions.size(); ++actionID) {
        // get gauge field from the SmearingPolicy and
        // based on the boolean is_smeared in actionID
-        GaugeField& Us =
+        Field& Us =
            Smearer.get_U(as[level].actions.at(actionID)->is_smeared);
        Hterm = as[level].actions.at(actionID)->S(Us);
        std::cout << GridLogMessage << "S Level " << level << " term "
@@ -289,7 +282,7 @@ class Integrator {
    return H;
  }

-  void integrate(GaugeField& U) {
+  void integrate(Field& U) {
    // reset the clocks
    t_U = 0;
    for (int level = 0; level < as.size(); ++level) {
@@ -311,8 +304,14 @@ class Integrator {

    // and that we indeed got to the end of the trajectory
    assert(fabs(t_U - Params.trajL) < 1.0e-6);
+
  }
+
+
+  
+
 };
 }
 }
 #endif  // INTEGRATOR_INCLUDED
+
--- a/Show More
+++ b/Show More