mirror of
https://github.com/paboyle/Grid.git
synced 2026-10-07 00:08:06 +01:00
Preparing for multigrid parameter consolidation and clean up of code, rationalise the different variants.
This commit is contained in:
1 parent
482f3cbaa2
commit
a07545adc4
21 files changed
+262
-2435
No files matched your search
@@ -244,7 +244,7 @@ int main(int argc, char **argv)
|
||||
// T7 : the SAME shape as T6, but assembled by C sequential MPI_Bcast --
|
||||
// one broadcast per contributing rank -- instead of one MPI_Allgatherv.
|
||||
//
|
||||
// This is the transport of DENSE_GATHER=2. Bcast takes no count vector,
|
||||
// This is the transport of the retired chunked-Bcast gather. Bcast takes no count vector,
|
||||
// so the zero-count asymmetry that makes T6 run at ~0.18 MB/s and trip
|
||||
// mpir_request.h:508 cannot arise. It costs C collectives rather than 1
|
||||
// and the roots do not transmit concurrently, so the byte cost is about
|
||||
|
||||
@@ -49,8 +49,8 @@ using namespace Grid;
|
||||
static int failures = 0;
|
||||
|
||||
// Portable |z|: ComplexD is std::complex on CPU builds and thrust::complex
|
||||
// under HIP, where std::abs does not resolve (same trap RecursiveSchurInverse
|
||||
// documents at FrobNorm2Local). Member real()/imag() work on both.
|
||||
// under HIP, where std::abs does not resolve. Member real()/imag() work
|
||||
// on both.
|
||||
static double Cabs(const ComplexD &z)
|
||||
{
|
||||
double re = z.real(), im = z.imag();
|
||||
|
||||
@@ -23,7 +23,7 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
//
|
||||
// 1D rank-major rows -> block cyclic -> Invert -> back to 1D rows
|
||||
//
|
||||
// which is exactly what DENSE_SCHUR2D runs inside DenseCoarseMatrix.
|
||||
// which is exactly what the dense inverse runs inside DenseCoarseMatrix.
|
||||
// CPU build under mpirun at n = 1,2,3,4.
|
||||
//
|
||||
// T1 : RowsToCyclic against a direct ImportGlobal of the same matrix --
|
||||
@@ -31,14 +31,9 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
// T2 : round trip rows -> 2D -> rows -- BITWISE, uniform AND non-uniform
|
||||
// rowStart, layouts with ragged trailing blocks.
|
||||
// T3 : full pipeline inverse against a host Gauss-Jordan reference.
|
||||
// T4 : CROSS-IMPLEMENTATION: the same matrix inverted by the 1D
|
||||
// RecursiveSchurInverse and by the 2D pipeline; results compared
|
||||
// element-wise. Two independent implementations, two independent
|
||||
// decompositions, one answer.
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#include <Grid/algorithms/multigrid/RecursiveSchurInverse.h>
|
||||
#include <Grid/algorithms/multigrid/BlockCyclicSchurInverse.h>
|
||||
#include <Grid/algorithms/multigrid/BlockCyclicRedistribute.h>
|
||||
|
||||
@@ -47,8 +42,8 @@ using namespace Grid;
|
||||
static int failures = 0;
|
||||
|
||||
// Portable |z|: ComplexD is std::complex on CPU builds and thrust::complex
|
||||
// under HIP, where std::abs does not resolve (same trap RecursiveSchurInverse
|
||||
// documents at FrobNorm2Local). Member real()/imag() work on both.
|
||||
// under HIP, where std::abs does not resolve. Member real()/imag() work
|
||||
// on both.
|
||||
static double Cabs(const ComplexD &z)
|
||||
{
|
||||
double re = z.real(), im = z.imag();
|
||||
@@ -190,12 +185,11 @@ int main(int argc, char **argv)
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
// T3 + T4 : the DENSE_SCHUR2D pipeline against the host reference and
|
||||
// against the INDEPENDENT 1D RecursiveSchurInverse.
|
||||
// T3 : the 2D pipeline against the host reference.
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
{
|
||||
bool ok3 = true, ok4 = true;
|
||||
double worst3 = 0.0, worst4 = 0.0;
|
||||
bool ok3 = true;
|
||||
double worst3 = 0.0;
|
||||
BlockCyclicSchurInverse RSI2;
|
||||
for(auto &g : grids){
|
||||
for(auto &c : cfgs){
|
||||
@@ -232,27 +226,9 @@ int main(int argc, char **argv)
|
||||
worst3 = std::max(worst3,d);
|
||||
if ( d > 1.0e-9 ) ok3 = false;
|
||||
}
|
||||
|
||||
// ---- 1D RecursiveSchurInverse on the same matrix ----
|
||||
{
|
||||
BlockRows Ar; Ar.Resize(myrows, N);
|
||||
acceleratorCopyToDevice(&h[0], &Ar.data[0], h.size()*sizeof(ComplexD));
|
||||
std::vector<int64_t> rs = rowStart;
|
||||
RecursiveSchurInverse RSI1(grid, N, rs, 1<<20);
|
||||
RSI1.Invert(Ar);
|
||||
std::vector<ComplexD> h1d(h.size());
|
||||
acceleratorCopyFromDevice(&Ar.data[0], &h1d[0], h1d.size()*sizeof(ComplexD));
|
||||
for(int64_t j=0;j<N;j++)
|
||||
for(int64_t i=0;i<myrows;i++){
|
||||
double d = Cabs(h2d[i+j*myrows]-h1d[i+j*myrows])/mxref;
|
||||
worst4 = std::max(worst4,d);
|
||||
if ( d > 1.0e-9 ) ok4 = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Report("T3 2D pipeline vs host reference", ok3, "worst "+std::to_string(worst3));
|
||||
Report("T4 2D pipeline vs 1D RecursiveSchurInverse", ok4, "worst "+std::to_string(worst4));
|
||||
}
|
||||
|
||||
{
|
||||
|
||||
@@ -20,7 +20,7 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
// SCALE rehearsal for the 2D distributed dense inverse: the full
|
||||
// DENSE_SCHUR2D pipeline -- 1D rows -> redistribute -> invert ->
|
||||
// 2D block-cyclic pipeline -- 1D rows -> redistribute -> invert ->
|
||||
// redistribute back -> certificate -- on a SYNTHETIC matrix of any size,
|
||||
// with no multigrid machinery, no configuration and no subspace file.
|
||||
//
|
||||
@@ -31,8 +31,8 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
// rank; at N=138240 on 288 ranks it is the production problem shape
|
||||
// exactly, in a driver that runs in minutes.
|
||||
//
|
||||
// S2D_N : global dimension (default 720, laptop friendly)
|
||||
// S2D_NB : block size (default N/P rows-per-rank if that
|
||||
// --schur2d-global-dimension <n> : N (default 720, laptop friendly)
|
||||
// --schur2d-block-size <n> : nb (default N/P rows-per-rank if that
|
||||
// is exact, else 48)
|
||||
//
|
||||
// The matrix is diagonally dominant (the recursion does not pivot); its
|
||||
@@ -55,8 +55,8 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
using namespace Grid;
|
||||
|
||||
// Portable |z|: ComplexD is std::complex on CPU builds and thrust::complex
|
||||
// under HIP, where std::abs does not resolve (same trap RecursiveSchurInverse
|
||||
// documents at FrobNorm2Local). Member real()/imag() work on both.
|
||||
// under HIP, where std::abs does not resolve. Member real()/imag() work
|
||||
// on both.
|
||||
static double Cabs(const ComplexD &z)
|
||||
{
|
||||
double re = z.real(), im = z.imag();
|
||||
@@ -76,13 +76,6 @@ static ComplexD Fill(int64_t i, int64_t j, int64_t N)
|
||||
|
||||
int main(int argc, char **argv)
|
||||
{
|
||||
// Environment walk (2026-08-27, systems/Frontier/schur2d_env.job): knobs to
|
||||
// reproduce the example's environment here, one at a time. Result: thread
|
||||
// level, OMP_NUM_THREADS, device residency and sustained load all NIL; only
|
||||
// "first job step on fresh nodes" (+3 s) is real.
|
||||
// S2D_BALLAST_GB=x x GB of Lattice fields made device-resident before the invert
|
||||
// S2D_PREHEAT_S=x x seconds of back-to-back zgemm before the invert
|
||||
// OMP_NUM_THREADS set in the job, read by nothing here but the runtime
|
||||
Grid_init(&argc, &argv);
|
||||
|
||||
GridCartesian *grid = SpaceTimeGrid::makeFourDimGrid(GridDefaultLatt(),
|
||||
@@ -91,9 +84,12 @@ int main(int argc, char **argv)
|
||||
const int P = grid->ProcessorCount();
|
||||
const int me = grid->ThisRank();
|
||||
|
||||
int64_t N = getenv("S2D_N") ? atol(getenv("S2D_N")) : 720;
|
||||
int64_t N = 720;
|
||||
if ( GridCmdOptionExists(argv,argv+argc,"--schur2d-global-dimension") )
|
||||
N = atol(GridCmdOptionPayload(argv,argv+argc,"--schur2d-global-dimension").c_str());
|
||||
int64_t nb;
|
||||
if ( getenv("S2D_NB") ) nb = atol(getenv("S2D_NB"));
|
||||
if ( GridCmdOptionExists(argv,argv+argc,"--schur2d-block-size") )
|
||||
nb = atol(GridCmdOptionPayload(argv,argv+argc,"--schur2d-block-size").c_str());
|
||||
else if ( N % P == 0 ) nb = N/P;
|
||||
else nb = 48;
|
||||
GRID_ASSERT( N >= 1 ); GRID_ASSERT( nb >= 1 );
|
||||
@@ -125,56 +121,13 @@ int main(int argc, char **argv)
|
||||
double t1 = usecond();
|
||||
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
// The DENSE_SCHUR2D pipeline, phase-timed. A0 keeps the original for
|
||||
// The 2D pipeline, phase-timed. A0 keeps the original for
|
||||
// the certificate.
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
BlockCyclicMatrix A (grid,N,nb,Pr,Pc);
|
||||
BlockCyclicMatrix A0(grid,N,nb,Pr,Pc);
|
||||
BlockCyclicSchurInverse RSI2;
|
||||
|
||||
// Pre-heat: drive the GCD with back-to-back zgemm for S2D_PREHEAT_S seconds
|
||||
// before the invert. The example calls the inverse after ~100 s of full
|
||||
// load on all 288 GCDs and its LOCAL kernels run 20-40% slower than the
|
||||
// idle-start harness (GEMM 3.1 vs 2.5 s, leaf 0.76 vs 0.24 s) with the
|
||||
// wires unchanged; thread level / OMP / residency (E1-E5) did not reproduce
|
||||
// that. If sustained load does, it is clock/power management, not code.
|
||||
if ( getenv("S2D_PREHEAT_S") ) {
|
||||
double secs = atof(getenv("S2D_PREHEAT_S"));
|
||||
const int64_t W = 4320;
|
||||
deviceVector<ComplexD> M((uint64_t)W*W), C((uint64_t)W*W);
|
||||
{ ComplexD *m = &M[0]; accelerator_for(idx,(uint64_t)W*W,1,{ m[idx] = ComplexD(1.0e-3*(idx%97),1.0e-3*(idx%89)); }); accelerator_barrier(); }
|
||||
deviceVector<ComplexD*> ap(1),bp(1),cp(1); std::vector<ComplexD*> ptr(1);
|
||||
ptr[0]=&M[0]; acceleratorCopyToDevice(&ptr[0],&ap[0],sizeof(ComplexD*)); acceleratorCopyToDevice(&ptr[0],&bp[0],sizeof(ComplexD*));
|
||||
ptr[0]=&C[0]; acceleratorCopyToDevice(&ptr[0],&cp[0],sizeof(ComplexD*));
|
||||
double t0=usecond(); int n=0; double tlast=0;
|
||||
while ( (usecond()-t0)/1.0e6 < secs ) {
|
||||
double t1=usecond();
|
||||
RSI2.SUMMA.BLAS.gemmBatched(GridBLAS_OP_N,GridBLAS_OP_N,(int)W,(int)W,(int)W,ComplexD(1.0,0.0),ap,(int)W,bp,(int)W,ComplexD(0.0,0.0),cp,(int)W);
|
||||
RSI2.SUMMA.BLAS.synchronise(); tlast=usecond()-t1; n++;
|
||||
}
|
||||
double tfirst = 0; (void)tfirst;
|
||||
std::cout << GridLogMessage << "Test_schur2d_scale: pre-heat " << (usecond()-t0)/1.0e6 << " s, " << n << " zgemm W=" << W
|
||||
<< ", last zgemm " << tlast/1.0e6 << " s (" << 8.0*W*W*W/tlast/1.0e6 << " TF/s; idle-start rate 23.7)" << std::endl;
|
||||
}
|
||||
|
||||
// Device ballast: Lattice fields written on the accelerator so they sit in
|
||||
// the MemoryManager's device LRU exactly as the example's fine-grid state does.
|
||||
typedef Lattice<iVector<iVector<vComplexD,Nc>,Ns> > BallastField;
|
||||
std::vector<BallastField> ballast;
|
||||
if ( getenv("S2D_BALLAST_GB") ) {
|
||||
double gb = atof(getenv("S2D_BALLAST_GB"));
|
||||
uint64_t fbytes = (uint64_t)grid->oSites()*sizeof(BallastField::vector_object);
|
||||
int nf = (int)(gb*1.0e9/(double)fbytes + 0.5);
|
||||
ballast.reserve(nf);
|
||||
for(int i=0;i<nf;i++){
|
||||
ballast.emplace_back(grid);
|
||||
autoView(v, ballast[i], AcceleratorWriteDiscard);
|
||||
accelerator_for(ss, grid->oSites(), 1, { v[ss] = Zero(); });
|
||||
}
|
||||
std::cout << GridLogMessage << "Test_schur2d_scale: device ballast " << nf << " fields x " << fbytes/1.0e6
|
||||
<< " MB = " << nf*fbytes/1.0e9 << " GB resident (S2D_BALLAST_GB=" << gb << ")" << std::endl;
|
||||
}
|
||||
|
||||
BlockCyclicRedistribute::RowsToCyclic(grid,rowStart,&rows1d[0],myrows,A);
|
||||
double t2 = usecond();
|
||||
if ( A.data.size() )
|
||||
|
||||
@@ -64,9 +64,10 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
// libblaspp shadows the ROCm one via LD_LIBRARY_PATH and throws
|
||||
// "device BLAS not available" from host_malloc_pinned.
|
||||
//
|
||||
// S2D_N, S2D_NB as in Test_schur2d_scale (default nb = N/P).
|
||||
// S2D_SKIP_GETRI=1 skips the getri leg (host loop; ~4 min at N=138240).
|
||||
// S2D_NOWARM=1 skips the warm-up.
|
||||
// --schur2d-global-dimension, --schur2d-block-size as in
|
||||
// Test_schur2d_scale (default nb = N/P).
|
||||
// --schur2d-skip-getri skips the getri leg (host loop; ~4 min at N=138240).
|
||||
// --schur2d-nowarm skips the warm-up.
|
||||
// A third leg, getrf+getrs(I), is SLATE's device-resident inverse route.
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -138,8 +139,12 @@ int main(int argc, char **argv)
|
||||
const int P = grid->ProcessorCount();
|
||||
const int me = grid->ThisRank();
|
||||
|
||||
int64_t N = getenv("S2D_N") ? atol(getenv("S2D_N")) : 720;
|
||||
int64_t nb = getenv("S2D_NB") ? atol(getenv("S2D_NB")) : ( (N%P==0) ? N/P : 48 );
|
||||
int64_t N = 720;
|
||||
if ( GridCmdOptionExists(argv,argv+argc,"--schur2d-global-dimension") )
|
||||
N = atol(GridCmdOptionPayload(argv,argv+argc,"--schur2d-global-dimension").c_str());
|
||||
int64_t nb = (N%P==0) ? N/P : 48;
|
||||
if ( GridCmdOptionExists(argv,argv+argc,"--schur2d-block-size") )
|
||||
nb = atol(GridCmdOptionPayload(argv,argv+argc,"--schur2d-block-size").c_str());
|
||||
int Pr,Pc; BlockCyclicLayout::ChooseProcessGrid(P,Pr,Pc);
|
||||
|
||||
std::vector<int64_t> rowStart(P+1); rowStart[0]=0;
|
||||
@@ -161,10 +166,10 @@ int main(int argc, char **argv)
|
||||
// first. Run a small throwaway inverse through BOTH paths so the timed
|
||||
// legs below measure hot code. Not reported.
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
// S2D_NOWARM=1 skips it (hang localisation). Stage markers are flushed so
|
||||
// --schur2d-nowarm skips it (hang localisation). Stage markers are flushed so
|
||||
// a hang shows WHERE even through block-buffered stdout.
|
||||
auto Stage = [&](const char *s){ std::cout << GridLogMessage << "stage: " << s << std::endl << std::flush; };
|
||||
if ( !getenv("S2D_NOWARM") ) {
|
||||
if ( !GridCmdOptionExists(argv,argv+argc,"--schur2d-nowarm") ) {
|
||||
// Fixed tiny size independent of P: the purpose is handle creation and
|
||||
// kernel loading, not work. (8*P at P=288 was N=2304 -> a 122 s SLATE
|
||||
// warm-up dominated by 288-way tile broadcasts.) Ranks beyond the first
|
||||
@@ -207,7 +212,7 @@ int main(int argc, char **argv)
|
||||
#endif
|
||||
std::cout << GridLogMessage << "warm-up done (both paths, N=" << Nw << ")" << std::endl << std::flush;
|
||||
} else {
|
||||
std::cout << GridLogMessage << "warm-up SKIPPED (S2D_NOWARM)" << std::endl << std::flush;
|
||||
std::cout << GridLogMessage << "warm-up SKIPPED (--schur2d-nowarm)" << std::endl << std::flush;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
@@ -240,7 +245,7 @@ int main(int argc, char **argv)
|
||||
// LEG 2: SLATE, every layout step timed and charged.
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
#ifdef HAVE_SLATE
|
||||
if ( !getenv("S2D_SKIP_GETRI") ) { // S2D_SKIP_GETRI=1: getri is a host loop, minutes at N=138240
|
||||
if ( !GridCmdOptionExists(argv,argv+argc,"--schur2d-skip-getri") ) { // getri is a host loop, minutes at N=138240
|
||||
typedef std::complex<double> scalar_t;
|
||||
acceleratorCopyToDevice(&h[0], &rows1d[0], h.size()*sizeof(ComplexD));
|
||||
BlockCyclicMatrix A(grid,N,nb,Pr,Pc), A0(grid,N,nb,Pr,Pc);
|
||||
|
||||
@@ -27,25 +27,19 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
/* END LEGAL */
|
||||
|
||||
//
|
||||
// T6 of the RecursiveSchurInverse regression chain
|
||||
// (schur_recursive_inverse_plan.txt 4B.5): the DenseCoarseMatrix GLUE,
|
||||
// on a real (tiny) lattice coarse operator, CPU laptop build.
|
||||
// The DenseCoarseMatrix GLUE test, on a real (tiny) lattice coarse
|
||||
// operator, CPU laptop build.
|
||||
//
|
||||
// Builds a genuine GeneralCoarsenedMatrix (DWF MdagM + 0.5 shift for a
|
||||
// guaranteed-invertible Galerkin coarse op, random aggregation basis,
|
||||
// nbasis=8, 4^4 x Ls/1 blocking) and constructs DenseCoarseMatrix in
|
||||
// DENSE_SCHUR=2 AUDIT mode with small DENSE_PANEL_BYTES (multi-panel
|
||||
// gathers exercised through the glue). The constructor then runs, in
|
||||
// order, all the certificates this stage exists to check:
|
||||
// - fresh ImportDense (no SLAB_FILE) + IMPORT CERTIFICATE vs Op.M
|
||||
// - InvertDenseSingle (the oracle)
|
||||
// - InvertDenseSchur: self-certifying rank-major map, fp64 diagonal
|
||||
// import certificate vs the fp32 slab, distributed recursion,
|
||||
// growth telemetry
|
||||
// - AUDIT: max|Ainv_schur - Ainv_single| over the full slab
|
||||
// - VERIFY ||A Ainv x - x||/||x|| through the SCHUR result
|
||||
// This program adds asserts on the audit number and a random-vector
|
||||
// round trip.
|
||||
// guaranteed-invertible Galerkin coarse op, random aggregation basis)
|
||||
// and runs the whole Import certificate chain through the glue:
|
||||
// - fresh ImportDense + IMPORT CERTIFICATE vs Op.M
|
||||
// - fp64 rank-major import certificate vs the fp32 slab
|
||||
// - the 2D block-cyclic recursion + growth telemetry
|
||||
// - VERIFY ||A Ainv x - x||/||x|| through the device split-K apply
|
||||
// This program adds an INDEPENDENT Eigen fp64 host-inverse oracle at
|
||||
// small N (built from applies of M to unit vectors, so it shares no
|
||||
// code with the import) and a random-vector round trip.
|
||||
//
|
||||
// Uniform local volume 12.12.12.12 (fine), per-dim blocks {4,4,3,3},
|
||||
// coarse 3.3.4.4/rank, nbasis 4 (N = 576n):
|
||||
@@ -193,13 +187,18 @@ int main (int argc, char ** argv)
|
||||
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
// Full-matrix conditioning probe at small N: dense columns by
|
||||
// applying M to unit vectors, fp64 Eigen SVD.
|
||||
// applying M to unit vectors, fp64 Eigen SVD. eA is kept: it is the
|
||||
// INDEPENDENT oracle for the inverse below (built from applies of M,
|
||||
// sharing no code with the stencil->dense import).
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
int64_t Nprobe = Coarse5d->gSites() * nbasis;
|
||||
Eigen::MatrixXcd eA;
|
||||
bool haveOracle = false;
|
||||
{
|
||||
int64_t Nprobe = Coarse5d->gSites() * nbasis;
|
||||
if ( Nprobe <= 700 )
|
||||
{
|
||||
Eigen::MatrixXcd eA(Nprobe, Nprobe);
|
||||
eA.resize(Nprobe, Nprobe);
|
||||
haveOracle = true;
|
||||
CoarseVector e(Coarse5d);
|
||||
CoarseVector Me(Coarse5d);
|
||||
for(int64_t j=0; j<Nprobe; j++)
|
||||
@@ -255,26 +254,15 @@ int main (int argc, char ** argv)
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
// T6: AUDIT mode, fresh import, multi-panel gathers. The constructor
|
||||
// runs every certificate in the chain (see banner).
|
||||
// The glue under test: Import runs the whole certificate chain
|
||||
// (import certificate, fp64 certificate, 2D inverse, VERIFY).
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
setenv("DENSE_SCHUR","2",1);
|
||||
setenv("DENSE_PANEL_BYTES","65536",1);
|
||||
unsetenv("SLAB_FILE");
|
||||
|
||||
// Restructured interface: DenseCoarseMatrix<CComplex,nbasis>, constructed
|
||||
// on the grid and fed by Import (which runs the certificate chain).
|
||||
typedef DenseCoarseMatrix<vTComplex,nbasis> DenseCC;
|
||||
DenseCC dcm(Coarse5d);
|
||||
dcm.Import(LittleDiracOp);
|
||||
|
||||
std::cout << GridLogMessage << "T6 audit relative slab difference (schur vs single) = "
|
||||
<< dcm.schurAuditRel << std::endl;
|
||||
GRID_ASSERT( dcm.schurAuditRel >= 0.0 ); // audit actually ran
|
||||
GRID_ASSERT( dcm.schurAuditRel < 1.0e-3 );
|
||||
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
// Random-vector round trip through the SCHUR inverse
|
||||
// Random-vector round trip through the inverse
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
CoarseVector x(Coarse5d);
|
||||
CoarseVector y(Coarse5d);
|
||||
@@ -288,6 +276,38 @@ int main (int argc, char ** argv)
|
||||
<< rel << std::endl;
|
||||
GRID_ASSERT( rel < 1.0e-2 );
|
||||
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
// Independent oracle at small N: y from dcm must match the Eigen fp64
|
||||
// solve of eA (dense columns of M itself) on the same x. Catches an
|
||||
// inverse that is self-consistent with a WRONG import, which the round
|
||||
// trip above cannot (dense and M would share the error).
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
if ( haveOracle )
|
||||
{
|
||||
Eigen::VectorXcd xs(Nprobe), ys(Nprobe);
|
||||
typedef typename CoarseVector::vector_object::scalar_object csobj;
|
||||
for(int64_t j=0; j<Nprobe; j++)
|
||||
{
|
||||
int64_t gsite = j / nbasis;
|
||||
int b = j % nbasis;
|
||||
Coordinate gcoor(Coarse5d->_ndimension);
|
||||
Lexicographic::CoorFromIndex(gcoor, gsite, Coarse5d->GlobalDimensions());
|
||||
csobj s;
|
||||
peekSite(s, x, gcoor);
|
||||
ComplexD zz = ((ComplexD *)&s)[b];
|
||||
xs(j) = std::complex<double>(zz.real(), zz.imag());
|
||||
peekSite(s, y, gcoor);
|
||||
zz = ((ComplexD *)&s)[b];
|
||||
ys(j) = std::complex<double>(zz.real(), zz.imag());
|
||||
}
|
||||
Eigen::VectorXcd yref = eA.fullPivLu().solve(xs);
|
||||
double dev = (ys - yref).cwiseAbs().maxCoeff();
|
||||
double ymax = yref.cwiseAbs().maxCoeff();
|
||||
std::cout << GridLogMessage << "T6 oracle max|Ainv x - eigen solve|/max|y| = "
|
||||
<< dev/ymax << " (fp32 slab vs fp64 host solve)" << std::endl;
|
||||
GRID_ASSERT( dev/ymax < 1.0e-4 );
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage << "Test_schur_dense_coarse: T6 ALL PASS" << std::endl;
|
||||
|
||||
Grid_finalize();
|
||||
|
||||
@@ -1,769 +0,0 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: Test_schur_inverse.cc
|
||||
|
||||
Copyright (C) 2026
|
||||
|
||||
Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
|
||||
//
|
||||
// Staged regression gate for RecursiveSchurInverse (distributed dense
|
||||
// inversion by recursive Schur complement) -- the laptop-side certificate
|
||||
// chain of schur_recursive_inverse_plan.txt section 4B.5. Runs on a
|
||||
// CPU-only build (Eigen BLAS backends) under mpirun:
|
||||
//
|
||||
// mpirun -n 1 ./Test_schur_inverse --grid 8.8.8.8 --mpi 1.1.1.1
|
||||
// mpirun -n 2 ./Test_schur_inverse --grid 8.8.8.8 --mpi 1.1.1.2
|
||||
// mpirun -n 3 ./Test_schur_inverse --grid 8.8.8.12 --mpi 1.1.1.3
|
||||
// mpirun -n 4 ./Test_schur_inverse --grid 8.8.8.8 --mpi 1.1.1.4
|
||||
//
|
||||
// (n=3 exercises uneven row splits throughout.) The lattice exists only
|
||||
// to furnish the communicator; no field is ever constructed.
|
||||
//
|
||||
// PRECISION: the inversion runs ENTIRELY in fp64 (decision 2026-08-14,
|
||||
// superseding the fp32-merge design); certificates are eps64-scaled.
|
||||
// The single terminal fp32 rounding belongs to the caller (tested at the
|
||||
// glue level, Test_schur_dense_coarse).
|
||||
//
|
||||
// Stages present (cumulative -- earlier tests are never removed):
|
||||
// T1a : ownership tables -- CheckRowStart on synthetic uneven partitions,
|
||||
// MakeRowStart allgather vs closed form on the live communicator.
|
||||
// T1b : STORAGE-CONVENTION PIN -- column-major + ld + window-offset
|
||||
// semantics fixed once via identity multiplies through the
|
||||
// explicit-ld gemmBatched, on INTEGER-VALUED data so all three
|
||||
// cases below are EXACT (values well within the mantissa):
|
||||
// (1) alpha=1,beta=0 read from an input column window
|
||||
// (2) alpha=-1,beta=1 accumulate (the S-formation case)
|
||||
// (3) write INTO an output column window, neighbours untouched
|
||||
// No later failure can be a transposition/convention ambiguity.
|
||||
// T2 : GatherGemm vs naive fp64 oracle (owner sub-ranges, alpha-beta
|
||||
// cases, tiny+huge panels, half-participation call shape).
|
||||
// T3 : LeafInvert in-place residual certificate.
|
||||
// T4 : full recursive Invert vs Eigen fp64 oracle, growth-scaled
|
||||
// certification, adversarial near-singular-A11 family with
|
||||
// telemetry-spike assertion.
|
||||
//
|
||||
// Hard asserts throughout; thresholds pre-registered in the plan.
|
||||
//
|
||||
#include <Grid/Grid.h>
|
||||
#include <Grid/Grid_Eigen_Dense.h>
|
||||
#include <Grid/algorithms/multigrid/RecursiveSchurInverse.h>
|
||||
|
||||
using namespace std;
|
||||
using namespace Grid;
|
||||
|
||||
int main (int argc, char ** argv)
|
||||
{
|
||||
Grid_init(&argc,&argv);
|
||||
|
||||
GridCartesian Comm(GridDefaultLatt(),
|
||||
GridDefaultSimd(Nd,vComplex::Nsimd()),
|
||||
GridDefaultMpi());
|
||||
GridBase *grid = &Comm;
|
||||
|
||||
////////////////////////////////////////////////////////////////
|
||||
// T1a : ownership tables
|
||||
////////////////////////////////////////////////////////////////
|
||||
{
|
||||
// Synthetic partitions of N=97 (prime: every P>1 is uneven)
|
||||
const int64_t N = 97;
|
||||
for(int P=1; P<=4; P++)
|
||||
{
|
||||
std::vector<int64_t> table(P+1);
|
||||
table[0] = 0;
|
||||
for(int r=0; r<P; r++)
|
||||
{
|
||||
int64_t nr = N/P + ( (r < (int)(N%P)) ? 1 : 0 );
|
||||
table[r+1] = table[r] + nr;
|
||||
}
|
||||
RecursiveSchurInverse::CheckRowStart(table, N);
|
||||
}
|
||||
|
||||
// Live allgather: deliberately uneven local counts, closed-form oracle
|
||||
int P = grid->ProcessorCount();
|
||||
int me = grid->ThisRank();
|
||||
|
||||
int64_t myNrows = 3 + me;
|
||||
std::vector<int64_t> table = RecursiveSchurInverse::MakeRowStart(grid, myNrows);
|
||||
|
||||
std::vector<int64_t> expect(P+1);
|
||||
expect[0] = 0;
|
||||
for(int r=0; r<P; r++)
|
||||
{
|
||||
expect[r+1] = expect[r] + (3 + r);
|
||||
}
|
||||
GRID_ASSERT( (int)table.size() == P+1 );
|
||||
for(int r=0; r<=P; r++)
|
||||
{
|
||||
GRID_ASSERT( table[r] == expect[r] );
|
||||
}
|
||||
|
||||
// Constructor smoke: derived ownership matches
|
||||
RecursiveSchurInverse RSI(grid, table[P], table, 1024*1024);
|
||||
GRID_ASSERT( RSI.P == P );
|
||||
GRID_ASSERT( RSI.me == me );
|
||||
GRID_ASSERT( RSI.myRow0 == expect[me] );
|
||||
GRID_ASSERT( RSI.myNrows == myNrows );
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "T1a ownership tables (synthetic P=1..4, live allgather, ctor) PASS"
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////
|
||||
// T1b : storage-convention pin (every rank, local, exact)
|
||||
////////////////////////////////////////////////////////////////
|
||||
{
|
||||
const int64_t rows = 5;
|
||||
const int64_t cols = 13;
|
||||
const int64_t col0 = 6; // input window start
|
||||
const int64_t w = 4; // window width
|
||||
|
||||
// f(i,j): integer-valued, unique per element
|
||||
auto f = [](int64_t i, int64_t j) -> ComplexD
|
||||
{
|
||||
return ComplexD( (RealD)(1 + i + 10*j), (RealD)(i - j) );
|
||||
};
|
||||
|
||||
BlockRows A;
|
||||
A.Resize(rows, cols);
|
||||
{
|
||||
std::vector<ComplexD> Ahost((uint64_t)rows*cols);
|
||||
for(int64_t j=0; j<cols; j++)
|
||||
{
|
||||
for(int64_t i=0; i<rows; i++)
|
||||
{
|
||||
Ahost[(uint64_t)(i + j*rows)] = f(i,j);
|
||||
}
|
||||
}
|
||||
acceleratorCopyToDevice(&Ahost[0], &A.data[0], (uint64_t)rows*cols*sizeof(ComplexD));
|
||||
}
|
||||
|
||||
// Identity I_w, column major
|
||||
deviceVector<ComplexD> Idev((uint64_t)w*w);
|
||||
{
|
||||
std::vector<ComplexD> Ihost((uint64_t)w*w, ComplexD(0.0,0.0));
|
||||
for(int64_t d=0; d<w; d++)
|
||||
{
|
||||
Ihost[(uint64_t)(d + d*w)] = ComplexD(1.0,0.0);
|
||||
}
|
||||
acceleratorCopyToDevice(&Ihost[0], &Idev[0], (uint64_t)w*w*sizeof(ComplexD));
|
||||
}
|
||||
|
||||
GridBLAS BLAS;
|
||||
ComplexD one ( 1.0,0.0);
|
||||
ComplexD minus (-1.0,0.0);
|
||||
ComplexD zero ( 0.0,0.0);
|
||||
|
||||
deviceVector<ComplexD*> Ap(1);
|
||||
deviceVector<ComplexD*> Bp(1);
|
||||
deviceVector<ComplexD*> Cp(1);
|
||||
std::vector<ComplexD*> ptr_h(1);
|
||||
|
||||
auto setptr = [&](deviceVector<ComplexD*> &d, ComplexD *p)
|
||||
{
|
||||
ptr_h[0] = p;
|
||||
acceleratorCopyToDevice(&ptr_h[0], &d[0], sizeof(ComplexD*));
|
||||
};
|
||||
|
||||
////////////////////////////////////////////////////////////
|
||||
// Case 1: C = A(:, col0:col0+w) . I_w (alpha=1, beta=0)
|
||||
////////////////////////////////////////////////////////////
|
||||
{
|
||||
deviceVector<ComplexD> Cdev((uint64_t)rows*w);
|
||||
setptr(Ap, A.ColumnWindow(col0));
|
||||
setptr(Bp, &Idev[0]);
|
||||
setptr(Cp, &Cdev[0]);
|
||||
|
||||
BLAS.gemmBatched(GridBLAS_OP_N, GridBLAS_OP_N,
|
||||
(int)rows, (int)w, (int)w,
|
||||
one, Ap, (int)A.ld,
|
||||
Bp, (int)w,
|
||||
zero, Cp, (int)rows);
|
||||
BLAS.synchronise();
|
||||
|
||||
std::vector<ComplexD> Chost((uint64_t)rows*w);
|
||||
acceleratorCopyFromDevice(&Cdev[0], &Chost[0], (uint64_t)rows*w*sizeof(ComplexD));
|
||||
for(int64_t j=0; j<w; j++)
|
||||
{
|
||||
for(int64_t i=0; i<rows; i++)
|
||||
{
|
||||
GRID_ASSERT( Chost[(uint64_t)(i + j*rows)] == f(i, col0+j) );
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////
|
||||
// Case 2: C = C0 - A(:, col0:col0+w) . I_w (alpha=-1, beta=1)
|
||||
// -- the S-formation accumulate; exact on integer data
|
||||
////////////////////////////////////////////////////////////
|
||||
{
|
||||
auto g = [](int64_t i, int64_t j) -> ComplexD
|
||||
{
|
||||
return ComplexD( (RealD)(100 + i + j), (RealD)7 );
|
||||
};
|
||||
deviceVector<ComplexD> Cdev((uint64_t)rows*w);
|
||||
{
|
||||
std::vector<ComplexD> Chost((uint64_t)rows*w);
|
||||
for(int64_t j=0; j<w; j++)
|
||||
{
|
||||
for(int64_t i=0; i<rows; i++)
|
||||
{
|
||||
Chost[(uint64_t)(i + j*rows)] = g(i,j);
|
||||
}
|
||||
}
|
||||
acceleratorCopyToDevice(&Chost[0], &Cdev[0], (uint64_t)rows*w*sizeof(ComplexD));
|
||||
}
|
||||
setptr(Ap, A.ColumnWindow(col0));
|
||||
setptr(Bp, &Idev[0]);
|
||||
setptr(Cp, &Cdev[0]);
|
||||
|
||||
BLAS.gemmBatched(GridBLAS_OP_N, GridBLAS_OP_N,
|
||||
(int)rows, (int)w, (int)w,
|
||||
minus, Ap, (int)A.ld,
|
||||
Bp, (int)w,
|
||||
one, Cp, (int)rows);
|
||||
BLAS.synchronise();
|
||||
|
||||
std::vector<ComplexD> Chost((uint64_t)rows*w);
|
||||
acceleratorCopyFromDevice(&Cdev[0], &Chost[0], (uint64_t)rows*w*sizeof(ComplexD));
|
||||
for(int64_t j=0; j<w; j++)
|
||||
{
|
||||
for(int64_t i=0; i<rows; i++)
|
||||
{
|
||||
ComplexD expect = g(i,j) - f(i, col0+j);
|
||||
GRID_ASSERT( Chost[(uint64_t)(i + j*rows)] == expect );
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////
|
||||
// Case 3: write INTO a column window of a wider C;
|
||||
// columns outside the window must be untouched
|
||||
////////////////////////////////////////////////////////////
|
||||
{
|
||||
const int64_t ccols = 6;
|
||||
const int64_t cw0 = 2; // output window start
|
||||
BlockRows C;
|
||||
C.Resize(rows, ccols);
|
||||
{
|
||||
std::vector<ComplexD> Chost((uint64_t)rows*ccols, ComplexD(-999.0, 999.0));
|
||||
acceleratorCopyToDevice(&Chost[0], &C.data[0], (uint64_t)rows*ccols*sizeof(ComplexD));
|
||||
}
|
||||
setptr(Ap, A.ColumnWindow(col0));
|
||||
setptr(Bp, &Idev[0]);
|
||||
setptr(Cp, C.ColumnWindow(cw0));
|
||||
|
||||
BLAS.gemmBatched(GridBLAS_OP_N, GridBLAS_OP_N,
|
||||
(int)rows, (int)w, (int)w,
|
||||
one, Ap, (int)A.ld,
|
||||
Bp, (int)w,
|
||||
zero, Cp, (int)C.ld);
|
||||
BLAS.synchronise();
|
||||
|
||||
std::vector<ComplexD> Chost((uint64_t)rows*ccols);
|
||||
acceleratorCopyFromDevice(&C.data[0], &Chost[0], (uint64_t)rows*ccols*sizeof(ComplexD));
|
||||
for(int64_t j=0; j<ccols; j++)
|
||||
{
|
||||
for(int64_t i=0; i<rows; i++)
|
||||
{
|
||||
ComplexD got = Chost[(uint64_t)(i + j*rows)];
|
||||
if ( (j >= cw0) && (j < cw0+w) )
|
||||
{
|
||||
GRID_ASSERT( got == f(i, col0 + (j-cw0)) );
|
||||
}
|
||||
else
|
||||
{
|
||||
GRID_ASSERT( got == ComplexD(-999.0, 999.0) );
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "T1b storage-convention pin (window read / S-accumulate / window write, exact) PASS"
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////
|
||||
// T2 : GatherGemm vs naive double-precision oracle.
|
||||
//
|
||||
// Every rank generates the SAME full N x N random fp64 operands
|
||||
// from a fixed seed (no comms needed for the oracle), keeps only
|
||||
// its own rows in BlockRows form, and after each GatherGemm call
|
||||
// checks its output window element-by-element against a plain
|
||||
// triple-loop ComplexD accumulation over the same entries.
|
||||
//
|
||||
// Sweep: N in {8, 96, 97}; owner ranges full/upper-half/single;
|
||||
// (alpha,beta) in {(1,0), (-1,1)}; panelBytes tiny (ragged
|
||||
// many-chunk gathers) and huge (single panel). Sentinel columns
|
||||
// outside the output window must be untouched. Finally, a
|
||||
// HALF-PARTICIPATION case rehearses the recursion call pattern:
|
||||
// lower ranks own B but pass EMPTY A/C (collectives only).
|
||||
////////////////////////////////////////////////////////////////
|
||||
{
|
||||
int P = grid->ProcessorCount();
|
||||
int me = grid->ThisRank();
|
||||
|
||||
std::mt19937 rng(777);
|
||||
std::uniform_real_distribution<double> dist(-1.0,1.0);
|
||||
|
||||
const int64_t nout = 5; // output width
|
||||
const int64_t colB = 3; // B window offset
|
||||
const int64_t colC = 2; // C window offset
|
||||
|
||||
for(int64_t N : {8L, 96L, 97L})
|
||||
{
|
||||
// Ownership: uneven for any P not dividing N
|
||||
std::vector<int64_t> table(P+1);
|
||||
table[0] = 0;
|
||||
for(int r=0; r<P; r++)
|
||||
{
|
||||
int64_t nr = N/P + ( (r < (int)(N%P)) ? 1 : 0 );
|
||||
table[r+1] = table[r] + nr;
|
||||
}
|
||||
int64_t r0 = table[me];
|
||||
int64_t myNr = table[me+1] - table[me];
|
||||
|
||||
// Identical full operands on every rank
|
||||
std::vector<ComplexD> Aglob((uint64_t)N*N);
|
||||
std::vector<ComplexD> Bglob((uint64_t)N*N);
|
||||
for(uint64_t i=0; i<(uint64_t)N*N; i++) Aglob[i] = ComplexD(dist(rng),dist(rng));
|
||||
for(uint64_t i=0; i<(uint64_t)N*N; i++) Bglob[i] = ComplexD(dist(rng),dist(rng));
|
||||
|
||||
// My rows of a full-matrix operand as a BlockRows
|
||||
auto fillRows = [&](BlockRows &X, std::vector<ComplexD> &glob,
|
||||
int64_t row0, int64_t nr)
|
||||
{
|
||||
X.Resize(nr, N);
|
||||
if ( nr == 0 ) return;
|
||||
std::vector<ComplexD> h((uint64_t)nr*N);
|
||||
for(int64_t j=0; j<N; j++)
|
||||
{
|
||||
for(int64_t i=0; i<nr; i++)
|
||||
{
|
||||
h[(uint64_t)(i + j*nr)] = glob[(uint64_t)((row0+i) + j*N)];
|
||||
}
|
||||
}
|
||||
acceleratorCopyToDevice(&h[0], &X.data[0], (uint64_t)nr*N*sizeof(ComplexD));
|
||||
};
|
||||
|
||||
// Owner-range cases: full span, upper half, single interior rank
|
||||
std::vector<std::pair<int,int> > ranges;
|
||||
ranges.push_back(std::make_pair(0, P));
|
||||
if ( P > 1 ) ranges.push_back(std::make_pair(P/2, P));
|
||||
if ( P > 1 ) ranges.push_back(std::make_pair(1, 2));
|
||||
|
||||
for(auto range : ranges)
|
||||
{
|
||||
int rB0 = range.first;
|
||||
int rB1 = range.second;
|
||||
int64_t ka0 = table[rB0]; // A-column window start = B row span
|
||||
int64_t k = table[rB1] - table[rB0];
|
||||
|
||||
for(int acase=0; acase<2; acase++)
|
||||
{
|
||||
ComplexD alpha = ( acase==0 ) ? ComplexD( 1.0,0.0) : ComplexD(-1.0,0.0);
|
||||
ComplexD beta = ( acase==0 ) ? ComplexD( 0.0,0.0) : ComplexD( 1.0,0.0);
|
||||
|
||||
for(int64_t panelBytes : {64L, 1L<<30})
|
||||
{
|
||||
RecursiveSchurInverse RSI(grid, N, table, panelBytes);
|
||||
|
||||
BlockRows A;
|
||||
BlockRows B;
|
||||
BlockRows C;
|
||||
fillRows(A, Aglob, r0, myNr);
|
||||
fillRows(B, Bglob, r0, myNr);
|
||||
|
||||
// Output: sentinel-filled, window at colC
|
||||
const ComplexD sentinel(-999.0, 999.0);
|
||||
const int64_t ccols = colC + nout + 2;
|
||||
C.Resize(myNr, ccols);
|
||||
std::vector<ComplexD> C0((uint64_t)myNr*ccols, sentinel);
|
||||
if ( acase == 1 )
|
||||
{
|
||||
// beta=1 needs defined window content: g(i,j), integer-valued
|
||||
for(int64_t j=0; j<nout; j++)
|
||||
{
|
||||
for(int64_t i=0; i<myNr; i++)
|
||||
{
|
||||
C0[(uint64_t)(i + (colC+j)*myNr)] = ComplexD((RealD)(50+i+j), (RealD)-3);
|
||||
}
|
||||
}
|
||||
}
|
||||
if ( myNr > 0 )
|
||||
{
|
||||
acceleratorCopyToDevice(&C0[0], &C.data[0], (uint64_t)myNr*ccols*sizeof(ComplexD));
|
||||
}
|
||||
|
||||
RSI.GatherGemm(alpha, A, ka0, k,
|
||||
rB0, rB1,
|
||||
B, colB, nout,
|
||||
beta, C, colC);
|
||||
|
||||
std::vector<ComplexD> Chost((uint64_t)myNr*ccols);
|
||||
if ( myNr > 0 )
|
||||
{
|
||||
acceleratorCopyFromDevice(&C.data[0], &Chost[0], (uint64_t)myNr*ccols*sizeof(ComplexD));
|
||||
}
|
||||
|
||||
double tol = 1.0e-14 * (double)k;
|
||||
for(int64_t j=0; j<ccols; j++)
|
||||
{
|
||||
for(int64_t i=0; i<myNr; i++)
|
||||
{
|
||||
ComplexD got = Chost[(uint64_t)(i + j*myNr)];
|
||||
if ( (j >= colC) && (j < colC+nout) )
|
||||
{
|
||||
int64_t jj = j - colC;
|
||||
ComplexD acc(0.0,0.0);
|
||||
if ( acase == 1 )
|
||||
{
|
||||
acc = C0[(uint64_t)(i + j*myNr)];
|
||||
}
|
||||
for(int64_t t=0; t<k; t++)
|
||||
{
|
||||
acc += alpha
|
||||
* Aglob[(uint64_t)((r0+i) + (ka0+t)*N)]
|
||||
* Bglob[(uint64_t)((ka0+t) + (colB+jj)*N)];
|
||||
}
|
||||
GRID_ASSERT( abs(got - acc) < tol );
|
||||
}
|
||||
else
|
||||
{
|
||||
GRID_ASSERT( got == sentinel );
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////
|
||||
// Half-participation: owners = [0,ph) hold B; participants
|
||||
// = [ph,P) hold A/C; owners pass EMPTY A/C and column
|
||||
// offset 0 (collectives only) -- the recursion call shape.
|
||||
////////////////////////////////////////////////////////////
|
||||
if ( P > 1 )
|
||||
{
|
||||
int ph = ( P+1 ) / 2;
|
||||
int64_t ka0 = table[0];
|
||||
int64_t k = table[ph] - table[0];
|
||||
int participant = ( me >= ph );
|
||||
|
||||
RecursiveSchurInverse RSI(grid, N, table, 64);
|
||||
|
||||
BlockRows A;
|
||||
BlockRows B;
|
||||
BlockRows C;
|
||||
fillRows(B, Bglob, r0, myNr);
|
||||
if ( participant )
|
||||
{
|
||||
fillRows(A, Aglob, r0, myNr);
|
||||
C.Resize(myNr, nout);
|
||||
}
|
||||
|
||||
ComplexD one (1.0,0.0);
|
||||
ComplexD zero(0.0,0.0);
|
||||
int64_t cA = participant ? ka0 : 0;
|
||||
RSI.GatherGemm(one, A, cA, k,
|
||||
0, ph,
|
||||
B, colB, nout, // owners deposit from their B window
|
||||
zero, C, 0);
|
||||
|
||||
if ( participant )
|
||||
{
|
||||
std::vector<ComplexD> Chost((uint64_t)myNr*nout);
|
||||
acceleratorCopyFromDevice(&C.data[0], &Chost[0], (uint64_t)myNr*nout*sizeof(ComplexD));
|
||||
double tol = 1.0e-14 * (double)k;
|
||||
for(int64_t j=0; j<nout; j++)
|
||||
{
|
||||
for(int64_t i=0; i<myNr; i++)
|
||||
{
|
||||
ComplexD acc(0.0,0.0);
|
||||
for(int64_t t=0; t<k; t++)
|
||||
{
|
||||
acc += Aglob[(uint64_t)((r0+i) + (ka0+t)*N)]
|
||||
* Bglob[(uint64_t)((ka0+t) + (colB+j)*N)];
|
||||
}
|
||||
GRID_ASSERT( abs(Chost[(uint64_t)(i + j*myNr)] - acc) < tol );
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "T2 GatherGemm vs oracle (N=8/96/97, 3 owner ranges, 2 alpha-beta, tiny+huge panels, half-participation) PASS"
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////
|
||||
// T3 : LeafInvert -- in-place fp64 inversion of the contiguous
|
||||
// leaf window. Purely local, every rank runs its own
|
||||
// uneven-size leaf; residual certificate in ComplexD.
|
||||
////////////////////////////////////////////////////////////////
|
||||
{
|
||||
int P = grid->ProcessorCount();
|
||||
int me = grid->ThisRank();
|
||||
|
||||
int64_t w = 17 + 3*me;
|
||||
uint64_t len = (uint64_t)w*w;
|
||||
|
||||
std::vector<int64_t> table = RecursiveSchurInverse::MakeRowStart(grid, w);
|
||||
RecursiveSchurInverse RSI(grid, table[P], table, 1<<20);
|
||||
|
||||
// A = w I + R : well conditioned
|
||||
std::mt19937 rng(31 + me);
|
||||
std::uniform_real_distribution<double> dist(-1.0,1.0);
|
||||
std::vector<ComplexD> Ahost(len);
|
||||
for(uint64_t i=0; i<len; i++) Ahost[i] = ComplexD(dist(rng),dist(rng));
|
||||
for(int64_t d=0; d<w; d++) Ahost[(uint64_t)(d + d*w)] += ComplexD((RealD)w, 0.0);
|
||||
|
||||
BlockRows Ar;
|
||||
Ar.Resize(w, w);
|
||||
acceleratorCopyToDevice(&Ahost[0], &Ar.data[0], len*sizeof(ComplexD));
|
||||
RSI.LeafInvert(0, w, Ar);
|
||||
std::vector<ComplexD> X(len);
|
||||
acceleratorCopyFromDevice(&Ar.data[0], &X[0], len*sizeof(ComplexD));
|
||||
|
||||
double maxdev = 0.0;
|
||||
for(int64_t j=0; j<w; j++)
|
||||
{
|
||||
for(int64_t i=0; i<w; i++)
|
||||
{
|
||||
ComplexD acc(0.0,0.0);
|
||||
for(int64_t t=0; t<w; t++)
|
||||
{
|
||||
acc += Ahost[(uint64_t)(i + t*w)] * X[(uint64_t)(t + j*w)];
|
||||
}
|
||||
if ( i==j ) acc -= ComplexD(1.0,0.0);
|
||||
maxdev = std::max(maxdev, abs(acc));
|
||||
}
|
||||
}
|
||||
GRID_ASSERT( maxdev < 1.0e-13 );
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "T3 LeafInvert in-place fp64 (residual " << maxdev << ") PASS" << std::endl;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////
|
||||
// T4 : full recursive Invert vs Eigen fp64 oracle.
|
||||
//
|
||||
// Every rank builds the SAME N x N fp64 matrix from a fixed seed,
|
||||
// keeps its rows, inverts through the full SPMD recursion, then
|
||||
// the test gathers the complete inverse (zero-fill GlobalSum) and
|
||||
// checks BOTH certificates:
|
||||
// cert1 = || A X - I ||_max (ComplexD accumulation)
|
||||
// cert2 = max|X - Xref| / max|Xref| (Xref = Eigen fp64 inverse)
|
||||
//
|
||||
// Families (eps64-scaled tolerances; the fp32-era growth data
|
||||
// rescales by eps64/eps32 ~ 1.9e-9):
|
||||
// kappa-moderate : A = R + 3 sqrt(N) I
|
||||
// kappa-large : A = R + 0.3 sqrt(N) I
|
||||
// adversarial : leading block (rank 0's whole leaf) REPLACED by
|
||||
// 1e-2 * (R' + 3 sqrt(b) I) inside a well-conditioned
|
||||
// A -- the growth spike must REGISTER in telemetry
|
||||
// (asserted > 10 when P > 1); at fp64 the certificate
|
||||
// barely notices it: that insensitivity IS the point
|
||||
// of the fp64 conversion.
|
||||
//
|
||||
// N=64 runs with panelBytes=128 (ragged many-chunk gathers inside
|
||||
// the recursion); larger N with 1 MB panels.
|
||||
////////////////////////////////////////////////////////////////
|
||||
{
|
||||
int P = grid->ProcessorCount();
|
||||
int me = grid->ThisRank();
|
||||
|
||||
std::mt19937 rng(2026);
|
||||
std::uniform_real_distribution<double> dist(-1.0,1.0);
|
||||
|
||||
for(int64_t N : {64L, 200L, 513L})
|
||||
{
|
||||
std::vector<int64_t> table(P+1);
|
||||
table[0] = 0;
|
||||
for(int r=0; r<P; r++)
|
||||
{
|
||||
int64_t nr = N/P + ( (r < (int)(N%P)) ? 1 : 0 );
|
||||
table[r+1] = table[r] + nr;
|
||||
}
|
||||
int64_t r0 = table[me];
|
||||
int64_t myNr = table[me+1] - table[me];
|
||||
|
||||
for(int fam=0; fam<3; fam++)
|
||||
{
|
||||
const char *famname = (fam==0) ? "kappa-moderate" :
|
||||
(fam==1) ? "kappa-large" : "adversarial-A11";
|
||||
double shift = (fam==1) ? 0.3*std::sqrt((double)N) : 3.0*std::sqrt((double)N);
|
||||
double tol = (fam==0) ? 1.0e-12 :
|
||||
(fam==1) ? 1.0e-11 : 5.0e-11;
|
||||
|
||||
// Identical operand on every rank (all draws rank-independent)
|
||||
std::vector<ComplexD> Aglob((uint64_t)N*N);
|
||||
for(uint64_t i=0; i<(uint64_t)N*N; i++) Aglob[i] = ComplexD(dist(rng),dist(rng));
|
||||
for(int64_t d=0; d<N; d++) Aglob[(uint64_t)(d + d*N)] += ComplexD(shift, 0.0);
|
||||
if ( fam == 2 )
|
||||
{
|
||||
// Leading block = rank 0's whole leaf, scaled down 100x but
|
||||
// internally well conditioned (shift scales as sqrt(b): a
|
||||
// FIXED shift makes A11 itself near-singular at large b).
|
||||
int64_t b = ( P > 1 ) ? table[1] : N/4;
|
||||
for(int64_t j=0; j<b; j++)
|
||||
{
|
||||
for(int64_t i=0; i<b; i++)
|
||||
{
|
||||
Aglob[(uint64_t)(i + j*N)] = ComplexD(0.01,0.0)*ComplexD(dist(rng),dist(rng));
|
||||
}
|
||||
}
|
||||
RealD bshift = (RealD)(0.03*std::sqrt((double)b));
|
||||
for(int64_t d=0; d<b; d++) Aglob[(uint64_t)(d + d*N)] += ComplexD(bshift,0.0);
|
||||
}
|
||||
|
||||
// Eigen fp64 oracle. Explicit re/im conversion at the boundary:
|
||||
// on HIP builds ComplexD is thrust::complex, which has no
|
||||
// operators against Eigen's std::complex.
|
||||
auto toStd = [](const ComplexD &z) -> std::complex<double>
|
||||
{
|
||||
return std::complex<double>(z.real(), z.imag());
|
||||
};
|
||||
Eigen::MatrixXcd eA(N,N);
|
||||
for(int64_t j=0; j<N; j++)
|
||||
{
|
||||
for(int64_t i=0; i<N; i++)
|
||||
{
|
||||
eA(i,j) = toStd(Aglob[(uint64_t)(i + j*N)]);
|
||||
}
|
||||
}
|
||||
Eigen::MatrixXcd Xref = eA.inverse();
|
||||
|
||||
// Distribute, invert
|
||||
int64_t panelBytes = ( N == 64 ) ? 128 : (1<<20);
|
||||
RecursiveSchurInverse RSI(grid, N, table, panelBytes);
|
||||
|
||||
BlockRows Arows;
|
||||
Arows.Resize(myNr, N);
|
||||
{
|
||||
std::vector<ComplexD> h((uint64_t)myNr*N);
|
||||
for(int64_t j=0; j<N; j++)
|
||||
{
|
||||
for(int64_t i=0; i<myNr; i++)
|
||||
{
|
||||
h[(uint64_t)(i + j*myNr)] = Aglob[(uint64_t)((r0+i) + j*N)];
|
||||
}
|
||||
}
|
||||
acceleratorCopyToDevice(&h[0], &Arows.data[0], (uint64_t)myNr*N*sizeof(ComplexD));
|
||||
}
|
||||
|
||||
RSI.Invert(Arows);
|
||||
|
||||
// Gather the full inverse: zero-fill + GlobalSum
|
||||
std::vector<ComplexD> Xfull((uint64_t)N*N, ComplexD(0.0,0.0));
|
||||
{
|
||||
std::vector<ComplexD> h((uint64_t)myNr*N);
|
||||
acceleratorCopyFromDevice(&Arows.data[0], &h[0], (uint64_t)myNr*N*sizeof(ComplexD));
|
||||
for(int64_t j=0; j<N; j++)
|
||||
{
|
||||
for(int64_t i=0; i<myNr; i++)
|
||||
{
|
||||
Xfull[(uint64_t)((r0+i) + j*N)] = h[(uint64_t)(i + j*myNr)];
|
||||
}
|
||||
}
|
||||
}
|
||||
grid->GlobalSumVector(&Xfull[0], (int)(N*N));
|
||||
|
||||
// cert1 = ||A X - I||_max
|
||||
double cert1 = 0.0;
|
||||
for(int64_t j=0; j<N; j++)
|
||||
{
|
||||
for(int64_t i=0; i<N; i++)
|
||||
{
|
||||
ComplexD acc(0.0,0.0);
|
||||
for(int64_t t=0; t<N; t++)
|
||||
{
|
||||
acc += Aglob[(uint64_t)(i + t*N)] * Xfull[(uint64_t)(t + j*N)];
|
||||
}
|
||||
if ( i==j ) acc -= ComplexD(1.0,0.0);
|
||||
cert1 = std::max(cert1, abs(acc));
|
||||
}
|
||||
}
|
||||
|
||||
// cert2 = max|X - Xref| / max|Xref|
|
||||
double maxref = 0.0;
|
||||
double maxdif = 0.0;
|
||||
for(int64_t j=0; j<N; j++)
|
||||
{
|
||||
for(int64_t i=0; i<N; i++)
|
||||
{
|
||||
maxref = std::max(maxref, std::abs(Xref(i,j)));
|
||||
maxdif = std::max(maxdif, std::abs(toStd(Xfull[(uint64_t)(i + j*N)]) - Xref(i,j)));
|
||||
}
|
||||
}
|
||||
double cert2 = maxdif / maxref;
|
||||
|
||||
double maxNormB = 0.0;
|
||||
for(uint64_t i=0; i<RSI.telNormB.size(); i++)
|
||||
{
|
||||
maxNormB = std::max(maxNormB, RSI.telNormB[i]);
|
||||
}
|
||||
|
||||
// GROWTH-SCALED certification, eps64 (the fp32-era model with
|
||||
// eps swapped: cert2 ~ (10-12) ||B||_F sqrt(N) eps; threshold =
|
||||
// 3x margin, floored at the family tolerance). ||B||_F capped
|
||||
// per family so growth cannot silently excuse a logic error.
|
||||
// cert1 remains a loose absolute bound (an O(1) logic error
|
||||
// gives cert1 ~ 1e2-1e3; fp64 rounding gives ~1e-10).
|
||||
double eps64 = 2.3e-16;
|
||||
double tolModel = 30.0 * std::max(1.0, maxNormB) * std::sqrt((double)N) * eps64;
|
||||
double tolEff = std::max(tol, tolModel);
|
||||
double capB = (fam==0) ? 100.0 : (fam==1) ? 2000.0 : 10000.0;
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "T4 N=" << N << " " << famname
|
||||
<< " ||AX-I||_max " << cert1
|
||||
<< " |X-Xref|/|Xref| " << cert2
|
||||
<< " max||B||_F " << maxNormB
|
||||
<< " tolEff " << tolEff
|
||||
<< ( (cert2 < tolEff) && (cert1 < 1.0e-6) ? " PASS" : " FAIL" )
|
||||
<< std::endl;
|
||||
|
||||
GRID_ASSERT( cert2 < tolEff );
|
||||
GRID_ASSERT( cert1 < 1.0e-6 );
|
||||
GRID_ASSERT( maxNormB < capB );
|
||||
if ( (fam == 2) && (P > 1) )
|
||||
{
|
||||
GRID_ASSERT( maxNormB > 10.0 ); // the spike must REGISTER
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "T4 recursive Invert vs Eigen oracle (N=64/200/513, 3 families) PASS"
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< "Test_schur_inverse: ALL STAGES PASS" << std::endl;
|
||||
|
||||
Grid_finalize();
|
||||
}
|
||||
@@ -50,8 +50,8 @@ using namespace Grid;
|
||||
static int failures = 0;
|
||||
|
||||
// Portable |z|: ComplexD is std::complex on CPU builds and thrust::complex
|
||||
// under HIP, where std::abs does not resolve (same trap RecursiveSchurInverse
|
||||
// documents at FrobNorm2Local). Member real()/imag() work on both.
|
||||
// under HIP, where std::abs does not resolve. Member real()/imag() work
|
||||
// on both.
|
||||
static double Cabs(const ComplexD &z)
|
||||
{
|
||||
double re = z.real(), im = z.imag();
|
||||
|
||||
@@ -37,9 +37,9 @@ Zero zero;
|
||||
// serialization strategy of Grid?
|
||||
|
||||
// clang-format off
|
||||
struct MultiGridParams : Serializable {
|
||||
struct WilsonMGParams : Serializable {
|
||||
public:
|
||||
GRID_SERIALIZABLE_CLASS_MEMBERS(MultiGridParams,
|
||||
GRID_SERIALIZABLE_CLASS_MEMBERS(WilsonMGParams,
|
||||
int, nLevels,
|
||||
std::vector<std::vector<int>>, blockSizes, // size == nLevels - 1
|
||||
std::vector<double>, smootherTol, // size == nLevels - 1
|
||||
@@ -54,7 +54,7 @@ public:
|
||||
int, coarseSolverMaxInnerIter);
|
||||
|
||||
// constructor with default values
|
||||
MultiGridParams(int _nLevels = 2,
|
||||
WilsonMGParams(int _nLevels = 2,
|
||||
std::vector<std::vector<int>> _blockSizes = {{4, 4, 4, 4}},
|
||||
std::vector<double> _smootherTol = {1e-14},
|
||||
std::vector<int> _smootherMaxOuterIter = {4},
|
||||
@@ -82,7 +82,7 @@ public:
|
||||
};
|
||||
// clang-format on
|
||||
|
||||
void checkParameterValidity(MultiGridParams const ¶ms) {
|
||||
void checkParameterValidity(WilsonMGParams const ¶ms) {
|
||||
|
||||
auto correctSize = params.nLevels - 1;
|
||||
|
||||
@@ -101,7 +101,7 @@ public:
|
||||
std::vector<GridCartesian *> Grids;
|
||||
std::vector<GridParallelRNG> PRNGs;
|
||||
|
||||
LevelInfo(GridCartesian *FineGrid, MultiGridParams const &mgParams) {
|
||||
LevelInfo(GridCartesian *FineGrid, WilsonMGParams const &mgParams) {
|
||||
|
||||
auto nCoarseLevels = mgParams.blockSizes.size();
|
||||
|
||||
@@ -176,7 +176,7 @@ public:
|
||||
int _CurrentLevel;
|
||||
int _NextCoarserLevel;
|
||||
|
||||
MultiGridParams &_MultiGridParams;
|
||||
WilsonMGParams &_WilsonMGParams;
|
||||
LevelInfo & _LevelInfo;
|
||||
|
||||
FineDiracMatrix & _FineMatrix;
|
||||
@@ -201,10 +201,10 @@ public:
|
||||
// Member Functions
|
||||
/////////////////////////////////////////////
|
||||
|
||||
MultiGridPreconditioner(MultiGridParams &mgParams, LevelInfo &LvlInfo, FineDiracMatrix &FineMat, FineDiracMatrix &SmootherMat)
|
||||
MultiGridPreconditioner(WilsonMGParams &mgParams, LevelInfo &LvlInfo, FineDiracMatrix &FineMat, FineDiracMatrix &SmootherMat)
|
||||
: _CurrentLevel(mgParams.nLevels - (nCoarserLevels + 1)) // _Level = 0 corresponds to finest
|
||||
, _NextCoarserLevel(_CurrentLevel + 1) // incremented for instances on coarser levels
|
||||
, _MultiGridParams(mgParams)
|
||||
, _WilsonMGParams(mgParams)
|
||||
, _LevelInfo(LvlInfo)
|
||||
, _FineMatrix(FineMat)
|
||||
, _SmootherMatrix(SmootherMat)
|
||||
@@ -212,7 +212,7 @@ public:
|
||||
, _CoarseMatrix(*_LevelInfo.Grids[_NextCoarserLevel]) {
|
||||
|
||||
_NextPreconditionerLevel
|
||||
= std::unique_ptr<NextPreconditionerLevel>(new NextPreconditionerLevel(_MultiGridParams, _LevelInfo, _CoarseMatrix, _CoarseMatrix));
|
||||
= std::unique_ptr<NextPreconditionerLevel>(new NextPreconditionerLevel(_WilsonMGParams, _LevelInfo, _CoarseMatrix, _CoarseMatrix));
|
||||
|
||||
resetTimers();
|
||||
}
|
||||
@@ -261,7 +261,7 @@ public:
|
||||
conformable(in, out);
|
||||
|
||||
// TODO: implement a W-cycle
|
||||
if(_MultiGridParams.kCycle)
|
||||
if(_WilsonMGParams.kCycle)
|
||||
kCycle(in, out);
|
||||
else
|
||||
vCycle(in, out);
|
||||
@@ -279,13 +279,13 @@ public:
|
||||
|
||||
FineVector fineTmp(in.Grid());
|
||||
|
||||
auto maxSmootherIter = _MultiGridParams.smootherMaxOuterIter[_CurrentLevel] * _MultiGridParams.smootherMaxInnerIter[_CurrentLevel];
|
||||
auto maxSmootherIter = _WilsonMGParams.smootherMaxOuterIter[_CurrentLevel] * _WilsonMGParams.smootherMaxInnerIter[_CurrentLevel];
|
||||
|
||||
TrivialPrecon<FineVector> fineTrivialPreconditioner;
|
||||
FlexibleGeneralisedMinimalResidual<FineVector> fineFGMRES(_MultiGridParams.smootherTol[_CurrentLevel],
|
||||
FlexibleGeneralisedMinimalResidual<FineVector> fineFGMRES(_WilsonMGParams.smootherTol[_CurrentLevel],
|
||||
maxSmootherIter,
|
||||
fineTrivialPreconditioner,
|
||||
_MultiGridParams.smootherMaxInnerIter[_CurrentLevel],
|
||||
_WilsonMGParams.smootherMaxInnerIter[_CurrentLevel],
|
||||
false);
|
||||
|
||||
MdagMLinearOperator<FineDiracMatrix, FineVector> fineMdagMOp(_FineMatrix);
|
||||
@@ -336,19 +336,19 @@ public:
|
||||
|
||||
FineVector fineTmp(in.Grid());
|
||||
|
||||
auto smootherMaxIter = _MultiGridParams.smootherMaxOuterIter[_CurrentLevel] * _MultiGridParams.smootherMaxInnerIter[_CurrentLevel];
|
||||
auto kCycleMaxIter = _MultiGridParams.kCycleMaxOuterIter[_CurrentLevel] * _MultiGridParams.kCycleMaxInnerIter[_CurrentLevel];
|
||||
auto smootherMaxIter = _WilsonMGParams.smootherMaxOuterIter[_CurrentLevel] * _WilsonMGParams.smootherMaxInnerIter[_CurrentLevel];
|
||||
auto kCycleMaxIter = _WilsonMGParams.kCycleMaxOuterIter[_CurrentLevel] * _WilsonMGParams.kCycleMaxInnerIter[_CurrentLevel];
|
||||
|
||||
TrivialPrecon<FineVector> fineTrivialPreconditioner;
|
||||
FlexibleGeneralisedMinimalResidual<FineVector> fineFGMRES(_MultiGridParams.smootherTol[_CurrentLevel],
|
||||
FlexibleGeneralisedMinimalResidual<FineVector> fineFGMRES(_WilsonMGParams.smootherTol[_CurrentLevel],
|
||||
smootherMaxIter,
|
||||
fineTrivialPreconditioner,
|
||||
_MultiGridParams.smootherMaxInnerIter[_CurrentLevel],
|
||||
_WilsonMGParams.smootherMaxInnerIter[_CurrentLevel],
|
||||
false);
|
||||
FlexibleGeneralisedMinimalResidual<CoarseVector> coarseFGMRES(_MultiGridParams.kCycleTol[_CurrentLevel],
|
||||
FlexibleGeneralisedMinimalResidual<CoarseVector> coarseFGMRES(_WilsonMGParams.kCycleTol[_CurrentLevel],
|
||||
kCycleMaxIter,
|
||||
*_NextPreconditionerLevel,
|
||||
_MultiGridParams.kCycleMaxInnerIter[_CurrentLevel],
|
||||
_WilsonMGParams.kCycleMaxInnerIter[_CurrentLevel],
|
||||
false);
|
||||
|
||||
MdagMLinearOperator<FineDiracMatrix, FineVector> fineMdagMOp(_FineMatrix);
|
||||
@@ -581,7 +581,7 @@ public:
|
||||
|
||||
int _CurrentLevel;
|
||||
|
||||
MultiGridParams &_MultiGridParams;
|
||||
WilsonMGParams &_WilsonMGParams;
|
||||
LevelInfo & _LevelInfo;
|
||||
|
||||
FineDiracMatrix &_FineMatrix;
|
||||
@@ -594,9 +594,9 @@ public:
|
||||
// Member Functions
|
||||
/////////////////////////////////////////////
|
||||
|
||||
MultiGridPreconditioner(MultiGridParams &mgParams, LevelInfo &LvlInfo, FineDiracMatrix &FineMat, FineDiracMatrix &SmootherMat)
|
||||
MultiGridPreconditioner(WilsonMGParams &mgParams, LevelInfo &LvlInfo, FineDiracMatrix &FineMat, FineDiracMatrix &SmootherMat)
|
||||
: _CurrentLevel(mgParams.nLevels - (0 + 1))
|
||||
, _MultiGridParams(mgParams)
|
||||
, _WilsonMGParams(mgParams)
|
||||
, _LevelInfo(LvlInfo)
|
||||
, _FineMatrix(FineMat)
|
||||
, _SmootherMatrix(SmootherMat) {
|
||||
@@ -613,12 +613,12 @@ public:
|
||||
conformable(_LevelInfo.Grids[_CurrentLevel], in.Grid());
|
||||
conformable(in, out);
|
||||
|
||||
auto coarseSolverMaxIter = _MultiGridParams.coarseSolverMaxOuterIter * _MultiGridParams.coarseSolverMaxInnerIter;
|
||||
auto coarseSolverMaxIter = _WilsonMGParams.coarseSolverMaxOuterIter * _WilsonMGParams.coarseSolverMaxInnerIter;
|
||||
|
||||
// On the coarsest level we only have what I above call the fine level, no coarse one
|
||||
TrivialPrecon<FineVector> fineTrivialPreconditioner;
|
||||
FlexibleGeneralisedMinimalResidual<FineVector> fineFGMRES(
|
||||
_MultiGridParams.coarseSolverTol, coarseSolverMaxIter, fineTrivialPreconditioner, _MultiGridParams.coarseSolverMaxInnerIter, false);
|
||||
_WilsonMGParams.coarseSolverTol, coarseSolverMaxIter, fineTrivialPreconditioner, _WilsonMGParams.coarseSolverMaxInnerIter, false);
|
||||
|
||||
MdagMLinearOperator<FineDiracMatrix, FineVector> fineMdagMOp(_FineMatrix);
|
||||
|
||||
@@ -651,7 +651,7 @@ using NLevelMGPreconditioner = MultiGridPreconditioner<Fobj, CComplex, nBasis, n
|
||||
|
||||
template<class Fobj, class CComplex, int nBasis, class Matrix>
|
||||
std::unique_ptr<MultiGridPreconditionerBase<Lattice<Fobj>>>
|
||||
createMGInstance(MultiGridParams &mgParams, LevelInfo &levelInfo, Matrix &FineMat, Matrix &SmootherMat) {
|
||||
createMGInstance(WilsonMGParams &mgParams, LevelInfo &levelInfo, Matrix &FineMat, Matrix &SmootherMat) {
|
||||
|
||||
#define CASE_FOR_N_LEVELS(nLevels) \
|
||||
case nLevels: \
|
||||
|
||||
@@ -51,7 +51,7 @@ int main(int argc, char **argv) {
|
||||
|
||||
RealD mass = -0.25;
|
||||
|
||||
MultiGridParams mgParams;
|
||||
WilsonMGParams mgParams;
|
||||
std::string inputXml{"./mg_params.xml"};
|
||||
|
||||
if(GridCmdOptionExists(argv, argv + argc, "--inputxml")) {
|
||||
|
||||
@@ -58,7 +58,7 @@ int main(int argc, char **argv) {
|
||||
|
||||
RealD mass = -0.25;
|
||||
|
||||
MultiGridParams mgParams;
|
||||
WilsonMGParams mgParams;
|
||||
std::string inputXml{"./mg_params.xml"};
|
||||
|
||||
if(GridCmdOptionExists(argv, argv + argc, "--inputxml")) {
|
||||
|
||||
@@ -54,7 +54,7 @@ int main(int argc, char **argv) {
|
||||
RealD csw_r = 1.0;
|
||||
RealD csw_t = 1.0;
|
||||
|
||||
MultiGridParams mgParams;
|
||||
WilsonMGParams mgParams;
|
||||
std::string inputXml{"./mg_params.xml"};
|
||||
|
||||
if(GridCmdOptionExists(argv, argv + argc, "--inputxml")) {
|
||||
|
||||
@@ -84,7 +84,7 @@ int main(int argc, char **argv) {
|
||||
RealD csw_r = 1.0;
|
||||
RealD csw_t = 1.0;
|
||||
|
||||
MultiGridParams mgParams;
|
||||
WilsonMGParams mgParams;
|
||||
std::string inputXml{"./mg_params.xml"};
|
||||
|
||||
if(GridCmdOptionExists(argv, argv + argc, "--inputxml")) {
|
||||
|
||||
@@ -60,7 +60,7 @@ int main(int argc, char **argv) {
|
||||
RealD csw_r = 1.0;
|
||||
RealD csw_t = 1.0;
|
||||
|
||||
MultiGridParams mgParams;
|
||||
WilsonMGParams mgParams;
|
||||
std::string inputXml{"./mg_params.xml"};
|
||||
|
||||
if(GridCmdOptionExists(argv, argv + argc, "--inputxml")) {
|
||||
|
||||
Reference in new issue
Block a user