mirror of
https://github.com/paboyle/Grid.git
synced 2026-10-07 00:08:06 +01:00
238 lines
9.3 KiB
C++
238 lines
9.3 KiB
C++
/*************************************************************************************
|
|
|
|
Grid physics library, www.github.com/paboyle/Grid
|
|
|
|
Source file: ./tests/solver/Test_split_mobius_batched.cc
|
|
|
|
Copyright (C) 2026
|
|
|
|
Author: Peter Boyle <paboyle@ph.ed.ac.uk>
|
|
|
|
This program is free software; you can redistribute it and/or modify
|
|
it under the terms of the GNU General Public License as published by
|
|
the Free Software Foundation; either version 2 of the License, or
|
|
(at your option) any later version.
|
|
|
|
This program is distributed in the hope that it will be useful,
|
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
GNU General Public License for more details.
|
|
|
|
You should have received a copy of the GNU General Public License along
|
|
with this program; if not, write to the Free Software Foundation, Inc.,
|
|
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
|
|
|
See the full license in the file "LICENSE" in the top level distribution directory
|
|
*************************************************************************************/
|
|
/* END LEGAL */
|
|
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
// Production-size timing of MixedPrecisionConjugateGradientBatched for Mobius with the
|
|
// Hadrons default SchurDiagMooeeOperator: the same batch solved without and then with
|
|
// split inner solves (--batched-solver-split), reporting wall clock, per-rhs iterations
|
|
// and true residuals for each.
|
|
//
|
|
// --Ls 12 --mass 0.026 --M5 1.8 --b 1.5 --c 0.5 --nbatch 4 --tol 1e-8
|
|
// --config <NERSC file> (omit for a hot configuration: timing only, not physics)
|
|
// --nounsplit (skip the reference unsplit solve)
|
|
// --repeat N (split solve N times; host RSS must not grow between them)
|
|
//
|
|
// MEMORY lines report host RSS (current and peak) and allocator cache sizes, maximum over
|
|
// ranks, at each phase: with --enable-unified=no every Lattice lives in host memory.
|
|
//
|
|
// Only one solution vector is kept, so the driver's own footprint is two batches of
|
|
// double red-black 5d fields; the solver adds about as much again.
|
|
/////////////////////////////////////////////////////////////////////////////////////////////
|
|
#include <Grid/Grid.h>
|
|
#include <sys/resource.h>
|
|
#ifdef __APPLE__
|
|
#include <mach/mach.h>
|
|
#endif
|
|
|
|
using namespace std;
|
|
using namespace Grid;
|
|
|
|
// Host memory of this process in GB: current resident set and its high-water mark
|
|
void HostRSS(RealD ¤t,RealD &peak)
|
|
{
|
|
struct rusage ru;
|
|
getrusage(RUSAGE_SELF,&ru);
|
|
#ifdef __APPLE__
|
|
peak = ru.ru_maxrss/1.0e9; // bytes on macOS
|
|
mach_task_basic_info_data_t info;
|
|
mach_msg_type_number_t count = MACH_TASK_BASIC_INFO_COUNT;
|
|
task_info(mach_task_self(),MACH_TASK_BASIC_INFO,(task_info_t)&info,&count);
|
|
current = info.resident_size/1.0e9;
|
|
#else
|
|
peak = ru.ru_maxrss*1024.0/1.0e9; // kilobytes on Linux
|
|
long pages = 0;
|
|
long resident = 0;
|
|
FILE *f = fopen("/proc/self/statm","r");
|
|
if ( f ) {
|
|
if ( fscanf(f,"%ld %ld",&pages,&resident) != 2 ) {
|
|
resident = 0;
|
|
}
|
|
fclose(f);
|
|
}
|
|
current = resident*(RealD)sysconf(_SC_PAGESIZE)/1.0e9;
|
|
#endif
|
|
}
|
|
|
|
// Largest values over ranks: host RSS now and at peak, and the allocator caches
|
|
void ReportMemory(GridBase *grid,const std::string &phase)
|
|
{
|
|
RealD rss;
|
|
RealD peak;
|
|
HostRSS(rss,peak);
|
|
RealD hostcache = MemoryManager::HostCacheBytes()/1.0e9;
|
|
RealD devcache = MemoryManager::DeviceCacheBytes()/1.0e9;
|
|
grid->GlobalMax(rss);
|
|
grid->GlobalMax(peak);
|
|
grid->GlobalMax(hostcache);
|
|
grid->GlobalMax(devcache);
|
|
std::cout << GridLogMessage << "MEMORY " << phase
|
|
<< " : host RSS " << rss << " GB, peak " << peak
|
|
<< " GB; allocator cache host " << hostcache << " GB, device " << devcache
|
|
<< " GB (max over ranks)" << std::endl;
|
|
HostMemoryReport(grid,GridLogMessage,phase);
|
|
}
|
|
|
|
typedef LatticeFermionD FieldD;
|
|
typedef LatticeFermionF FieldF;
|
|
|
|
template<class T>
|
|
T CmdOption(int argc,char **argv,const std::string &name,T def)
|
|
{
|
|
T val = def;
|
|
if ( GridCmdOptionExists(argv,argv+argc,name) ) {
|
|
std::stringstream ss(GridCmdOptionPayload(argv,argv+argc,name));
|
|
ss >> val;
|
|
}
|
|
return val;
|
|
}
|
|
|
|
void SolveAndReport(const std::string &label,
|
|
MixedPrecisionConjugateGradientBatched<FieldD,FieldF> &mCG,
|
|
LinearOperatorBase<FieldD> &Linop_d,
|
|
std::vector<FieldD> &src,
|
|
std::vector<FieldD> &sol)
|
|
{
|
|
int nbatch = src.size();
|
|
for(int i=0;i<nbatch;i++){
|
|
sol[i].Checkerboard() = src[i].Checkerboard();
|
|
sol[i] = Zero();
|
|
}
|
|
|
|
std::cout << GridLogMessage << "==================================================" << std::endl;
|
|
std::cout << GridLogMessage << label << " batched solve, nbatch " << nbatch << std::endl;
|
|
std::cout << GridLogMessage << "==================================================" << std::endl;
|
|
|
|
RealD t0 = usecond();
|
|
mCG(src,sol);
|
|
RealD t1 = usecond();
|
|
|
|
FieldD Msol(src[0].Grid());
|
|
Msol.Checkerboard() = src[0].Checkerboard();
|
|
RealD worst = 0.0;
|
|
for(int i=0;i<nbatch;i++){
|
|
Linop_d.HermOp(sol[i],Msol);
|
|
Msol = Msol - src[i];
|
|
RealD resid = std::sqrt(norm2(Msol)/norm2(src[i]));
|
|
worst = std::max(worst,resid);
|
|
std::cout << GridLogMessage << label << ": rhs " << i << " true residual " << resid << std::endl;
|
|
}
|
|
std::cout << GridLogMessage << label << ": SUMMARY wall clock " << (t1-t0)/1.0e6
|
|
<< " s for " << nbatch << " rhs, " << (t1-t0)/1.0e6/nbatch
|
|
<< " s/rhs, worst true residual " << worst << std::endl;
|
|
}
|
|
|
|
int main (int argc, char ** argv)
|
|
{
|
|
Grid_init(&argc,&argv);
|
|
|
|
int Ls = CmdOption<int> (argc,argv,"--Ls",12);
|
|
RealD mass = CmdOption<RealD> (argc,argv,"--mass",0.026);
|
|
RealD M5 = CmdOption<RealD> (argc,argv,"--M5",1.8);
|
|
RealD b = CmdOption<RealD> (argc,argv,"--b",1.5);
|
|
RealD c = CmdOption<RealD> (argc,argv,"--c",0.5);
|
|
int nbatch = CmdOption<int> (argc,argv,"--nbatch",4);
|
|
RealD tol = CmdOption<RealD> (argc,argv,"--tol",1.0e-8);
|
|
std::string config = CmdOption<std::string>(argc,argv,"--config",std::string(""));
|
|
bool unsplit = !GridCmdOptionExists(argv,argv+argc,"--nounsplit");
|
|
int repeat = CmdOption<int> (argc,argv,"--repeat",1);
|
|
|
|
std::cout << GridLogMessage << "Mobius Ls " << Ls << " mass " << mass << " M5 " << M5
|
|
<< " b " << b << " c " << c << " nbatch " << nbatch << " tol " << tol << std::endl;
|
|
|
|
GridCartesian *UGrid_d = SpaceTimeGrid::makeFourDimGrid(GridDefaultLatt(), GridDefaultSimd(Nd,vComplexD::Nsimd()), GridDefaultMpi());
|
|
GridRedBlackCartesian *UrbGrid_d = SpaceTimeGrid::makeFourDimRedBlackGrid(UGrid_d);
|
|
GridCartesian *FGrid_d = SpaceTimeGrid::makeFiveDimGrid(Ls,UGrid_d);
|
|
GridRedBlackCartesian *FrbGrid_d = SpaceTimeGrid::makeFiveDimRedBlackGrid(Ls,UGrid_d);
|
|
|
|
GridCartesian *UGrid_f = SpaceTimeGrid::makeFourDimGrid(GridDefaultLatt(), GridDefaultSimd(Nd,vComplexF::Nsimd()), GridDefaultMpi());
|
|
GridRedBlackCartesian *UrbGrid_f = SpaceTimeGrid::makeFourDimRedBlackGrid(UGrid_f);
|
|
GridCartesian *FGrid_f = SpaceTimeGrid::makeFiveDimGrid(Ls,UGrid_f);
|
|
GridRedBlackCartesian *FrbGrid_f = SpaceTimeGrid::makeFiveDimRedBlackGrid(Ls,UGrid_f);
|
|
|
|
std::vector<int> seeds4({1,2,3,4});
|
|
std::vector<int> seeds5({5,6,7,8});
|
|
GridParallelRNG RNG4(UGrid_d);
|
|
GridParallelRNG RNG5(FGrid_d);
|
|
RNG4.SeedFixedIntegers(seeds4);
|
|
RNG5.SeedFixedIntegers(seeds5);
|
|
|
|
LatticeGaugeFieldD Umu_d(UGrid_d);
|
|
LatticeGaugeFieldF Umu_f(UGrid_f);
|
|
if ( config.size() ) {
|
|
FieldMetaData header;
|
|
NerscIO::readConfiguration(Umu_d,header,config);
|
|
} else {
|
|
std::cout << GridLogMessage << "No --config: hot configuration, timing only" << std::endl;
|
|
SU<Nc>::HotConfiguration(RNG4,Umu_d);
|
|
}
|
|
precisionChange(Umu_f,Umu_d);
|
|
ReportMemory(UGrid_d,"gauge field ready");
|
|
|
|
// Antiperiodic in time, as in production
|
|
WilsonImplParams params;
|
|
params.boundary_phases[Nd-1] = -1.0;
|
|
|
|
MobiusFermionD Dd(Umu_d,*FGrid_d,*FrbGrid_d,*UGrid_d,*UrbGrid_d,mass,M5,b,c,params);
|
|
MobiusFermionF Df(Umu_f,*FGrid_f,*FrbGrid_f,*UGrid_f,*UrbGrid_f,mass,M5,b,c,params);
|
|
SchurDiagMooeeOperator<MobiusFermionD,FieldD> Linop_d(Dd);
|
|
SchurDiagMooeeOperator<MobiusFermionF,FieldF> Linop_f(Df);
|
|
|
|
std::vector<FieldD> src(nbatch,FrbGrid_d);
|
|
std::vector<FieldD> sol(nbatch,FrbGrid_d);
|
|
{
|
|
FieldD tmp(FGrid_d);
|
|
for(int i=0;i<nbatch;i++){
|
|
random(RNG5,tmp);
|
|
pickCheckerboard(Odd,src[i],tmp);
|
|
}
|
|
}
|
|
|
|
ReportMemory(UGrid_d,"operators and sources ready");
|
|
|
|
MixedPrecisionConjugateGradientBatched<FieldD,FieldF> mCG(tol,10000,50,10000,FrbGrid_f,Linop_f,Linop_d);
|
|
|
|
Coordinate split = mCG.BatchedSplit;
|
|
bool splitnode = mCG.BatchedSplitNode;
|
|
|
|
if ( unsplit ) {
|
|
mCG.BatchedSplit = Coordinate();
|
|
mCG.BatchedSplitNode = false;
|
|
SolveAndReport("UNSPLIT",mCG,Linop_d,src,sol);
|
|
ReportMemory(UGrid_d,"after unsplit solve");
|
|
}
|
|
|
|
mCG.BatchedSplit = split;
|
|
mCG.BatchedSplitNode = splitnode;
|
|
// Repeated split solves expose allocations not released between calls
|
|
for(int r=0;r<repeat;r++){
|
|
SolveAndReport("SPLIT",mCG,Linop_d,src,sol);
|
|
ReportMemory(UGrid_d,"after split solve "+std::to_string(r));
|
|
}
|
|
|
|
Grid_finalize();
|
|
}
|