diff --git a/Grid/GridCore.h b/Grid/GridCore.h index 51bcecd3a..251d32691 100644 --- a/Grid/GridCore.h +++ b/Grid/GridCore.h @@ -56,6 +56,7 @@ Author: paboyle #include #include #include +#include #include #include #include diff --git a/Grid/algorithms/iterative/ConjugateGradientMixedPrecBatched.h b/Grid/algorithms/iterative/ConjugateGradientMixedPrecBatched.h index e6b6cc82b..807973e70 100644 --- a/Grid/algorithms/iterative/ConjugateGradientMixedPrecBatched.h +++ b/Grid/algorithms/iterative/ConjugateGradientMixedPrecBatched.h @@ -151,6 +151,8 @@ public: SplitCloneTimer.Stop(); if ( split == nullptr ) { std::cout << GridLogMessage << "MixedPrecisionConjugateGradientBatched: operator cannot be split; serial inner solves" << std::endl; + } else { + HostMemoryReport(DoublePrecGrid,GridLogMessage,"MixedPrecisionConjugateGradientBatched: after split clone"); } } @@ -159,6 +161,7 @@ public: for(outer_iter = 0; outer_iter < MaxOuterIterations; outer_iter++){ std::cout << GridLogMessage << std::endl; std::cout << GridLogMessage << "Outer iteration " << outer_iter << std::endl; + HostMemoryReport(DoublePrecGrid,GridLogMessage,"MixedPrecisionConjugateGradientBatched: outer iteration "+std::to_string(outer_iter)); bool allConverged = true; diff --git a/Grid/lattice/Lattice_transfer.h b/Grid/lattice/Lattice_transfer.h index 2273ab3a3..d485827cc 100644 --- a/Grid/lattice/Lattice_transfer.h +++ b/Grid/lattice/Lattice_transfer.h @@ -1741,6 +1741,10 @@ void Grid_split(std::vector > & full,Lattice & split) << " alltoall " << t_a2a/1.0e6 << " reorder " << t_reorder/1.0e6 << " vectorise " << t_vec/1.0e6 << std::endl; + // Staging vectors are still live here, so this is the high-water point of the call + if ( GridLogPerformance.isActive() ) { + HostMemoryReport(full_grid,GridLogPerformance,"Grid_split"); + } } template @@ -1906,6 +1910,10 @@ void Grid_unsplit(std::vector > & full,Lattice & split) << " alltoall " << t_a2a/1.0e6 << " reorder " << t_reorder/1.0e6 << " vectorise " << t_vec/1.0e6 << std::endl; + // Staging vectors are still live here, so this is the high-water point of the call + if ( GridLogPerformance.isActive() ) { + HostMemoryReport(full_grid,GridLogPerformance,"Grid_unsplit"); + } } ////////////////////////////////////////////////////// diff --git a/tests/solver/Test_split_mobius_batched.cc b/tests/solver/Test_split_mobius_batched.cc index faf5eadfb..a31017f08 100644 --- a/tests/solver/Test_split_mobius_batched.cc +++ b/tests/solver/Test_split_mobius_batched.cc @@ -93,6 +93,7 @@ void ReportMemory(GridBase *grid,const std::string &phase) << " : host RSS " << rss << " GB, peak " << peak << " GB; allocator cache host " << hostcache << " GB, device " << devcache << " GB (max over ranks)" << std::endl; + HostMemoryReport(grid,GridLogMessage,phase); } typedef LatticeFermionD FieldD;