/************************************************************************************* Grid physics library, www.github.com/paboyle/Grid Source file: ./tests/solver/Test_split_mobius_batched.cc Copyright (C) 2026 Author: Peter Boyle This program is free software; you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free Software Foundation; either version 2 of the License, or (at your option) any later version. This program is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for more details. You should have received a copy of the GNU General Public License along with this program; if not, write to the Free Software Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA. See the full license in the file "LICENSE" in the top level distribution directory *************************************************************************************/ /* END LEGAL */ ///////////////////////////////////////////////////////////////////////////////////////////// // Production-size timing of MixedPrecisionConjugateGradientBatched for Mobius with the // Hadrons default SchurDiagMooeeOperator: the same batch solved without and then with // split inner solves (--batched-solver-split), reporting wall clock, per-rhs iterations // and true residuals for each. // // --Ls 12 --mass 0.026 --M5 1.8 --b 1.5 --c 0.5 --nbatch 4 --tol 1e-8 // --config (omit for a hot configuration: timing only, not physics) // --nounsplit (skip the reference unsplit solve) // --repeat N (split solve N times; host RSS must not grow between them) // // MEMORY lines report host RSS (current and peak) and allocator cache sizes, maximum over // ranks, at each phase: with --enable-unified=no every Lattice lives in host memory. // // Only one solution vector is kept, so the driver's own footprint is two batches of // double red-black 5d fields; the solver adds about as much again. ///////////////////////////////////////////////////////////////////////////////////////////// #include #include #ifdef __APPLE__ #include #endif using namespace std; using namespace Grid; // Host memory of this process in GB: current resident set and its high-water mark void HostRSS(RealD ¤t,RealD &peak) { struct rusage ru; getrusage(RUSAGE_SELF,&ru); #ifdef __APPLE__ peak = ru.ru_maxrss/1.0e9; // bytes on macOS mach_task_basic_info_data_t info; mach_msg_type_number_t count = MACH_TASK_BASIC_INFO_COUNT; task_info(mach_task_self(),MACH_TASK_BASIC_INFO,(task_info_t)&info,&count); current = info.resident_size/1.0e9; #else peak = ru.ru_maxrss*1024.0/1.0e9; // kilobytes on Linux long pages = 0; long resident = 0; FILE *f = fopen("/proc/self/statm","r"); if ( f ) { if ( fscanf(f,"%ld %ld",&pages,&resident) != 2 ) { resident = 0; } fclose(f); } current = resident*(RealD)sysconf(_SC_PAGESIZE)/1.0e9; #endif } // Largest values over ranks: host RSS now and at peak, and the allocator caches void ReportMemory(GridBase *grid,const std::string &phase) { RealD rss; RealD peak; HostRSS(rss,peak); RealD hostcache = MemoryManager::HostCacheBytes()/1.0e9; RealD devcache = MemoryManager::DeviceCacheBytes()/1.0e9; grid->GlobalMax(rss); grid->GlobalMax(peak); grid->GlobalMax(hostcache); grid->GlobalMax(devcache); std::cout << GridLogMessage << "MEMORY " << phase << " : host RSS " << rss << " GB, peak " << peak << " GB; allocator cache host " << hostcache << " GB, device " << devcache << " GB (max over ranks)" << std::endl; HostMemoryReport(grid,GridLogMessage,phase); } typedef LatticeFermionD FieldD; typedef LatticeFermionF FieldF; template T CmdOption(int argc,char **argv,const std::string &name,T def) { T val = def; if ( GridCmdOptionExists(argv,argv+argc,name) ) { std::stringstream ss(GridCmdOptionPayload(argv,argv+argc,name)); ss >> val; } return val; } void SolveAndReport(const std::string &label, MixedPrecisionConjugateGradientBatched &mCG, LinearOperatorBase &Linop_d, std::vector &src, std::vector &sol) { int nbatch = src.size(); for(int i=0;i (argc,argv,"--Ls",12); RealD mass = CmdOption (argc,argv,"--mass",0.026); RealD M5 = CmdOption (argc,argv,"--M5",1.8); RealD b = CmdOption (argc,argv,"--b",1.5); RealD c = CmdOption (argc,argv,"--c",0.5); int nbatch = CmdOption (argc,argv,"--nbatch",4); RealD tol = CmdOption (argc,argv,"--tol",1.0e-8); std::string config = CmdOption(argc,argv,"--config",std::string("")); bool unsplit = !GridCmdOptionExists(argv,argv+argc,"--nounsplit"); int repeat = CmdOption (argc,argv,"--repeat",1); std::cout << GridLogMessage << "Mobius Ls " << Ls << " mass " << mass << " M5 " << M5 << " b " << b << " c " << c << " nbatch " << nbatch << " tol " << tol << std::endl; GridCartesian *UGrid_d = SpaceTimeGrid::makeFourDimGrid(GridDefaultLatt(), GridDefaultSimd(Nd,vComplexD::Nsimd()), GridDefaultMpi()); GridRedBlackCartesian *UrbGrid_d = SpaceTimeGrid::makeFourDimRedBlackGrid(UGrid_d); GridCartesian *FGrid_d = SpaceTimeGrid::makeFiveDimGrid(Ls,UGrid_d); GridRedBlackCartesian *FrbGrid_d = SpaceTimeGrid::makeFiveDimRedBlackGrid(Ls,UGrid_d); GridCartesian *UGrid_f = SpaceTimeGrid::makeFourDimGrid(GridDefaultLatt(), GridDefaultSimd(Nd,vComplexF::Nsimd()), GridDefaultMpi()); GridRedBlackCartesian *UrbGrid_f = SpaceTimeGrid::makeFourDimRedBlackGrid(UGrid_f); GridCartesian *FGrid_f = SpaceTimeGrid::makeFiveDimGrid(Ls,UGrid_f); GridRedBlackCartesian *FrbGrid_f = SpaceTimeGrid::makeFiveDimRedBlackGrid(Ls,UGrid_f); std::vector seeds4({1,2,3,4}); std::vector seeds5({5,6,7,8}); GridParallelRNG RNG4(UGrid_d); GridParallelRNG RNG5(FGrid_d); RNG4.SeedFixedIntegers(seeds4); RNG5.SeedFixedIntegers(seeds5); LatticeGaugeFieldD Umu_d(UGrid_d); LatticeGaugeFieldF Umu_f(UGrid_f); if ( config.size() ) { FieldMetaData header; NerscIO::readConfiguration(Umu_d,header,config); } else { std::cout << GridLogMessage << "No --config: hot configuration, timing only" << std::endl; SU::HotConfiguration(RNG4,Umu_d); } precisionChange(Umu_f,Umu_d); ReportMemory(UGrid_d,"gauge field ready"); // Antiperiodic in time, as in production WilsonImplParams params; params.boundary_phases[Nd-1] = -1.0; MobiusFermionD Dd(Umu_d,*FGrid_d,*FrbGrid_d,*UGrid_d,*UrbGrid_d,mass,M5,b,c,params); MobiusFermionF Df(Umu_f,*FGrid_f,*FrbGrid_f,*UGrid_f,*UrbGrid_f,mass,M5,b,c,params); SchurDiagMooeeOperator Linop_d(Dd); SchurDiagMooeeOperator Linop_f(Df); std::vector src(nbatch,FrbGrid_d); std::vector sol(nbatch,FrbGrid_d); { FieldD tmp(FGrid_d); for(int i=0;i mCG(tol,10000,50,10000,FrbGrid_f,Linop_f,Linop_d); Coordinate split = mCG.BatchedSplit; bool splitnode = mCG.BatchedSplitNode; if ( unsplit ) { mCG.BatchedSplit = Coordinate(); mCG.BatchedSplitNode = false; SolveAndReport("UNSPLIT",mCG,Linop_d,src,sol); ReportMemory(UGrid_d,"after unsplit solve"); } mCG.BatchedSplit = split; mCG.BatchedSplitNode = splitnode; // Repeated split solves expose allocations not released between calls for(int r=0;r