mirror of
https://github.com/paboyle/Grid.git
synced 2026-08-15 15:09:36 +01:00
Compare commits
217
Commits
82cfff2990
...
develop
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c4b089cf15 | ||
|
|
85e1bbf4bc | ||
|
|
032f111c8b | ||
|
|
e38528888a | ||
|
|
7c9a6d354e | ||
|
|
3ec90803ac | ||
|
|
c22473f15d | ||
|
|
55c064de28 | ||
|
|
06ce057920 | ||
|
|
76b4bd6d12 | ||
|
|
b5541aab55 | ||
|
|
a7160ac513 | ||
|
|
02d0301c9f | ||
|
|
a6cdf20c18 | ||
|
|
ad9a413892 | ||
|
|
1fddd2c29b | ||
|
|
2f75067569 | ||
|
|
6e8a00f215 | ||
|
|
702773e5fb | ||
|
|
4dfbd850ff | ||
|
|
b039e659af | ||
|
|
d16d44dda0 | ||
|
|
1c19389ba6 | ||
|
|
02fdff674c | ||
|
|
fd8b6a23a6 | ||
|
|
9e3a51d078 | ||
|
|
6f7a2ad7c7 | ||
|
|
499d656949 | ||
|
|
ba68f09026 | ||
|
|
3bdeeb73ef | ||
|
|
19868a800f | ||
|
|
df908ee872 | ||
|
|
84715ff4b9 | ||
|
|
5792195073 | ||
|
|
fb5662a449 | ||
|
|
6b2ad3db80 | ||
|
|
f1a969f0c3 | ||
|
|
f18320a152 | ||
|
|
bfcde59199 | ||
|
|
f23e9e70cd | ||
|
|
06fcd31da0 | ||
|
|
7e4fe99b1e | ||
|
|
159ea3d64d | ||
|
|
1f55c13bc4 | ||
|
|
d68b111d06 | ||
|
|
41f5a02204 | ||
|
|
7647576863 | ||
|
|
f11ba18df2 | ||
|
|
cf8587e401 | ||
|
|
7dd35ef749 | ||
|
|
41e570ddce | ||
|
|
a452131b50 | ||
|
|
4e49ca55ab | ||
|
|
c3f4474401 | ||
|
|
3d3eff86f3 | ||
|
|
fc9f154ac1 | ||
|
|
4aa0bca4dc | ||
|
|
905da6f083 | ||
|
|
86c7f29183 | ||
|
|
b0c99f876e | ||
|
|
bf5fcdc860 | ||
|
|
b58a1508fa | ||
|
|
4d527e81fa | ||
|
|
7803580aa6 | ||
|
|
32654db366 | ||
|
|
cd340cfab3 | ||
|
|
f32866b2ff | ||
|
|
1cd1dc091e | ||
|
|
0493656e86 | ||
|
|
66fd504c4d | ||
|
|
be4dd2b52f | ||
|
|
707d059766 | ||
|
|
f08c755ae6 | ||
|
|
dbbfdd4e4b | ||
|
|
f967fb40bf | ||
|
|
74e0f846cb | ||
|
|
303a4d26e5 | ||
|
|
119888653c | ||
|
|
a9f42c08f9 | ||
|
|
e79adc9d31 | ||
|
|
5a9056cd93 | ||
|
|
012c36ab5a | ||
|
|
5c4574f9aa | ||
|
|
a424775884 | ||
|
|
d6b1388741 | ||
|
|
796c6cae4e | ||
|
|
1a8064d6d9 | ||
|
|
43648924c3 | ||
|
|
bf2140e74d | ||
|
|
a1119266c1 | ||
|
|
a0f00c0eca | ||
|
|
d358954a84 | ||
|
|
aee00bdfb5 | ||
|
|
cf324b0fa1 | ||
|
|
b314dc224d | ||
|
|
1bbd62498e | ||
|
|
f3c3b1c04b | ||
|
|
069f98b253 | ||
|
|
dfd0503eae | ||
|
|
c629b2e87e | ||
|
|
7c8462abd1 | ||
|
|
95a6a0bde7 | ||
|
|
bba328fac5 | ||
|
|
41362349f3 | ||
|
|
12e3499b6d | ||
|
|
9576011011 | ||
|
|
155b34c1aa | ||
|
|
982ffe9ebe | ||
|
|
0251ecaeab | ||
|
|
372a27d645 | ||
|
|
72b4a061f3 | ||
|
|
29198efabe | ||
|
|
50aa51f93a | ||
|
|
79ccc81a86 | ||
|
|
3f0fdbb597 | ||
|
|
ea57bd8f03 | ||
|
|
bdba5b8403 | ||
|
|
58cc6ca9c0 | ||
|
|
e5996b440d | ||
|
|
ad9d03fd85 | ||
|
|
4de160ce20 | ||
|
|
fc8c8ce6e7 | ||
|
|
ddbb7f07c8 | ||
|
|
a5a04929fb | ||
|
|
1e29c59bcc | ||
|
|
b6abdc3845 | ||
|
|
77b8657fcc | ||
|
|
2fadd8bb62 | ||
|
|
60df2dd5d0 | ||
|
|
66b529b345 | ||
|
|
1304172a93 | ||
|
|
1315d4604d | ||
|
|
a31af31328 | ||
|
|
26c3c7d8f9 | ||
|
|
0650d7c7eb | ||
|
|
068f95ad2d | ||
|
|
f4fbf7c9ca | ||
|
|
843d6497b2 | ||
|
|
747c167658 | ||
|
|
fca2c5dba0 | ||
|
|
e12bc7f07c | ||
|
|
dc6ae51cab | ||
|
|
baa70d8ec9 | ||
|
|
c93b338bdd | ||
|
|
c0472aa0ec | ||
|
|
09552cfd73 | ||
|
|
003fec509c | ||
|
|
773a82d87f | ||
|
|
286c29d6fb | ||
|
|
969b0a3922 | ||
|
|
f8b2eacf99 | ||
|
|
6140ac6864 | ||
|
|
c6c2834e03 | ||
|
|
856545a1db | ||
|
|
e2d607f6c7 | ||
|
|
66da4e0657 | ||
|
|
b37390bb5a | ||
|
|
829dc8cceb | ||
|
|
13cc2c39f5 | ||
|
|
66ea3b271c | ||
|
|
d293b58a20 | ||
|
|
ce093b2bf3 | ||
|
|
e4404efe5a | ||
|
|
5ce270f1de | ||
|
|
af43b067a0 | ||
|
|
34b44d1fee | ||
|
|
595ceaac37 | ||
|
|
daf5834e8e | ||
|
|
0d8658a039 | ||
|
|
095e004d01 | ||
|
|
0acabee7f6 | ||
|
|
76fbcffb60 | ||
|
|
a0a62d7ead | ||
|
|
c5038ea6a5 | ||
|
|
a5120903eb | ||
|
|
00b286a08a | ||
|
|
24a9759353 | ||
|
|
1b56f6f46d | ||
|
|
2a8084d569 | ||
|
|
6ff29f9d4f | ||
|
|
c4d3e79193 | ||
|
|
7cd3f21e6b | ||
|
|
4a0aaf0786 | ||
|
|
9c3835524c | ||
|
|
549351bb8a | ||
|
|
b650b89682 | ||
|
|
74e6b19f83 | ||
|
|
2e684028de | ||
|
|
c54d87a472 | ||
|
|
4304245c1b | ||
|
|
6165931afa | ||
|
|
1d1fd3bcaf | ||
|
|
23581333e6 | ||
|
|
e5fa3d887f | ||
|
|
583fa7bb0a | ||
|
|
fe0db53842 | ||
|
|
76c0ada1e1 | ||
|
|
92f49e9194 | ||
|
|
44c8057b5f | ||
|
|
0ad837f595 | ||
|
|
bd2103c746 | ||
|
|
9c18d2ddb0 | ||
|
|
1245a8c151 | ||
|
|
07113dc8ba | ||
|
|
a3420e6fa9 | ||
|
|
732836d9f8 | ||
|
|
87658f7b53 | ||
|
|
e7f51e5fb1 | ||
|
|
1ce5f70dd1 | ||
|
|
473635f401 | ||
|
|
5adf2657dd | ||
|
|
c646d91527 | ||
|
|
a2b98d82e1 | ||
|
|
7b9415c088 | ||
|
|
cb7110f492 | ||
|
|
0c7af66490 | ||
|
|
496d1b914a |
@@ -0,0 +1,183 @@
|
||||
# CLAUDE.md
|
||||
|
||||
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
||||
|
||||
## What This Is
|
||||
|
||||
Grid is a data-parallel C++ library for lattice QCD. It provides SIMD-vectorised lattice containers, MPI-based domain decomposition, GPU acceleration (CUDA/HIP/SYCL), and a full suite of QCD algorithms including HMC.
|
||||
|
||||
## Build
|
||||
|
||||
Uses GNU Autotools. The bootstrap step only needs to run once (or after `configure.ac` changes).
|
||||
|
||||
```bash
|
||||
./bootstrap.sh # downloads Eigen 3.4.0, generates configure
|
||||
mkdir build && cd build
|
||||
../configure [options]
|
||||
make -j$(nproc)
|
||||
make check # run root-level tests
|
||||
make install
|
||||
```
|
||||
|
||||
Key configure options:
|
||||
|
||||
| Option | Common values |
|
||||
|--------|---------------|
|
||||
| `--enable-simd=` | `AVX2`, `AVX512`, `KNL`, `A64FX`, `NEONv8`, `GPU` |
|
||||
| `--enable-comms=` | `mpi-auto`, `mpi3-auto`, `none` |
|
||||
| `--enable-accelerator=` | `cuda`, `hip`, `sycl` |
|
||||
| `--enable-shm=` | `shmopen`, `hugetlbfs`, `nvlink` |
|
||||
| `--enable-Nc=` | `3` (default), `2`, `4`, `5` |
|
||||
| `--with-gmp=`, `--with-mpfr=`, `--with-fftw=`, `--with-lime=` | paths to libs |
|
||||
| `--enable-hdf5`, `--enable-mkl`, `--enable-lapack` | optional features |
|
||||
|
||||
GPU builds additionally need `--enable-gen-simd-width=64` (sets 512-bit SIMD width for GPU warp/wavefront sizing) and `--enable-unified=no --enable-shm=nvlink` for multi-GPU runs.
|
||||
|
||||
To speed up compilation, `--disable-fermion-reps --disable-gparity` skips instantiating G-parity and higher-representation fermion operators.
|
||||
|
||||
Platform recipes from `README.md`:
|
||||
- **KNL**: `--enable-simd=KNL --enable-comms=mpi3-auto --enable-mkl`
|
||||
- **Skylake/Haswell**: `--enable-simd=AVX512` or `AVX2` + `--enable-comms=mpi3-auto`
|
||||
- **AMD EPYC**: `--enable-simd=AVX2 --enable-comms=mpi3`
|
||||
- **A64FX (Fugaku)**: `--enable-simd=A64FX --enable-comms=mpi3 --enable-shm=shmget` (see `SVE_README.txt`)
|
||||
|
||||
Complete, working `configure` invocations for specific HPC systems (Frontier/ROCm, Perlmutter/CUDA, Summit, SDCC-A100, etc.) live in `systems/<platform>/config-command`. These are the canonical references for production builds.
|
||||
|
||||
Required external libs: GMP, MPFR, OpenSSL, zlib.
|
||||
|
||||
### Use `systems/` for real machines
|
||||
|
||||
`systems/<machine>/` holds the known-good build for each production platform (`Frontier`, `Aurora`, `Perlmutter`, `Summit`, `Tursa`, `Lumi`, `Booster`, `Crusher`, `SDCC-*`, `mac-arm`, …). Each contains a `config-command` (the exact `../../configure` invocation) and a `sourceme.sh` (module loads and env). **Prefer copying/adapting these over hand-rolling configure flags** — they encode compiler workarounds, `LDFLAGS`, and shared-memory settings that are easy to get wrong. `systems/WorkArounds.txt` records known vendor bugs.
|
||||
|
||||
Note the GPU builds use `--enable-simd=GPU --enable-gen-simd-width=64`, so `Nsimd` is *not* 1 on device (it is `64/sizeof(scalar)`).
|
||||
|
||||
### Regenerating `Make.inc` — required after adding or deleting source files
|
||||
|
||||
`Make.inc` files are generated, not tracked in git (`.gitignore`d). `scripts/filelist` walks `Grid/`, `tests/*`, `benchmarks/`, `examples/`, and `HMC/` and writes the file lists and per-test `bin_PROGRAMS` rules. Every new `.cc`/`.h` in `Grid/`, and every new `Test_*.cc` / `Benchmark_*.cc` / `Example_*.cc`, is invisible to the build until you run:
|
||||
|
||||
```bash
|
||||
./scripts/filelist # from the source root, then re-run configure/make
|
||||
```
|
||||
|
||||
`bootstrap.sh` runs it for you on the first setup.
|
||||
|
||||
## Running Tests and Benchmarks
|
||||
|
||||
```bash
|
||||
# From build directory
|
||||
make check # root-level tests (Test_simd, Test_cshift, etc.)
|
||||
make -C tests/<subdir> tests # build tests in a subdirectory
|
||||
make tests # build all tests across all subdirectories
|
||||
./tests/core/Test_simd # run a single test binary directly
|
||||
mpirun -n 4 ./tests/core/Test_cshift --grid 16.16.16.16 --mpi 1.1.1.4
|
||||
```
|
||||
|
||||
`make check` is a thin smoke test — building a subdirectory with `make -C tests/<subdir> tests` and running the relevant binaries directly is the normal development loop. Test binaries take Grid's standard command-line arguments (`--grid`, `--mpi`, `--accelerator-threads`, `--threads`, `--debug-signals`, `--log`); see `Grid/util/Init.cc`.
|
||||
|
||||
Test subdirectories and their focus: `core` (SIMD, stencil, comms), `solver` (CG, GMRES, eigensolvers), `hmc` (MD integrators), `forces` (fermion forces), `lanczos`, `IO`, `smearing`, `sp2n`, `debug`.
|
||||
|
||||
Tests and benchmarks that need optional fermion representations are guarded by `disable_tests_without_instantiations.h` / `disable_benchmarks_without_instantiations.h`, so a `--disable-fermion-reps --disable-gparity` build silently compiles them to no-ops.
|
||||
|
||||
## Architecture
|
||||
|
||||
### Layer stack (bottom to top)
|
||||
|
||||
1. **SIMD layer** (`Grid/simd/`) — platform-specific intrinsics wrapped into `vRealF`, `vComplexD`, etc. The SIMD width and layout are compile-time constants controlled by `--enable-simd`.
|
||||
|
||||
2. **Tensor layer** (`Grid/tensors/`) — Lorentz/colour/spin tensor algebra built on top of SIMD types. `iMatrix`, `iVector`, `iScalar` templates compose into QCD types like `ColourMatrix`, `SpinColourVector`.
|
||||
|
||||
3. **Lattice layer** (`Grid/lattice/`) — `Lattice<T>` container: a site-local tensor replicated across a distributed Cartesian grid. All arithmetic is site-parallel and expression-template-fused.
|
||||
|
||||
4. **Cartesian/comms layer** (`Grid/cartesian/`, `Grid/communicator/`) — `GridCartesian` holds the MPI topology and local/global geometry. `Grid/cshift/` implements nearest-neighbour halo exchange; `Grid/stencil/` is the optimised multi-hop stencil used by Dirac operators.
|
||||
|
||||
5. **Algorithm layer** (`Grid/algorithms/`) — iterative solvers (CG, GMRES, BiCGSTAB, mixed-precision), eigensolvers (Lanczos, LAPACK), FFT, smearing, and multigrid.
|
||||
|
||||
6. **QCD layer** (`Grid/qcd/`) — gauge and fermion actions, HMC integrators, observables.
|
||||
|
||||
### QCD subsystem (`Grid/qcd/`)
|
||||
|
||||
- `action/fermion/` — Wilson, Clover, DWF (Mobius), Staggered, twisted-mass, G-parity variants
|
||||
- `action/gauge/` — Wilson gauge, Symanzik, Iwasaki, DBW2, plaquette+rect
|
||||
- `representations/` — Fundamental, Adjoint, Two-index, Sp(2n)
|
||||
- `hmc/` — Leapfrog, OMF2/OMF4 integrators; pseudofermion refreshment; Metropolis accept/reject
|
||||
- `smearing/` — APE, Stout, HEX, gradient flow
|
||||
- `observables/` — Polyakov loop, plaquette, topological charge
|
||||
|
||||
### GPU acceleration and the view/memory-manager discipline
|
||||
### Multigrid (`Grid/algorithms/multigrid/`)
|
||||
|
||||
Aggregation-based algebraic multigrid for Wilson-type fermions. Key files: `CoarsenedMatrix.h` (coarse operator), `GeneralCoarsenedMatrix.h` and `GeneralCoarsenedMatrixMultiRHS.h` (general coarsening supporting multi-RHS solves), `Aggregates.h` (near-null vector construction), `Geometry.h` (coarse-grid geometry). `MultiGrid.h` is the top-level include.
|
||||
|
||||
### GPU acceleration
|
||||
|
||||
GPU support is injected via macros in `Grid/threads/Accelerator.h` — `accelerator_for(i, n, nsimd, {...})`, `accelerator_forNB` (non-blocking, must be followed by `accelerator_barrier()`), `accelerator_for2dNB`, and `accelerator_inline`. On a CPU build these degrade to `thread_for` (OpenMP). Unified virtual memory is on by default (`--enable-unified=yes`); device-aware MPI (`--enable-accelerator-aware-mpi`) avoids device→host copies on transfers.
|
||||
|
||||
Lattice data is **not** directly addressable inside a kernel. You must open a view with the correct access mode so `Grid/allocator/MemoryManager.h` can move/mark the data:
|
||||
|
||||
```cpp
|
||||
autoView(out_v, out, AcceleratorWriteDiscard); // RAII; closes at end of scope
|
||||
autoView(in_v, in, AcceleratorRead);
|
||||
accelerator_for(ss, grid->oSites(), Nsimd, {
|
||||
coalescedWrite(out_v[ss], coalescedRead(in_v[ss]));
|
||||
});
|
||||
```
|
||||
|
||||
Modes are `AcceleratorRead/Write/WriteDiscard` and `CpuRead/Write/WriteDiscard`. Getting the mode wrong (e.g. `AcceleratorRead` on a field you write) produces stale-data bugs that only appear on GPU builds. Inside kernels use `coalescedRead`/`coalescedWrite` rather than raw `operator[]` — they map the SIMD lane onto `threadIdx.x` so accesses stay coalesced.
|
||||
|
||||
### Repo-local debugging skills (`skills/`)
|
||||
|
||||
`skills/` contains hard-won, Grid-specific playbooks written as invocable skill files. Consult them before debugging in these areas rather than reasoning from first principles:
|
||||
|
||||
| File | Covers |
|
||||
|---|---|
|
||||
| `gpu-memory-performance.md` | `acceleratorThreads()`, LambdaApply thread mapping, `coalescedRead` idiom, fused vs staged HBM access |
|
||||
| `gpu-runtime-correctness.md` | GPU runtime returning early from sync, silent wrong answers |
|
||||
| `communication-overlap.md` | 7-phase halo pipeline, per-packet events, host-staging vs GPU-direct RDMA |
|
||||
| `mpi-heterogeneous.md` | `MPI_Sendrecv` device-buffer aliasing, deterministic reductions |
|
||||
| `compiler-validation.md` | Isolating GPU compiler codegen bugs, minimal reproducers |
|
||||
| `correctness-verification.md` | Double-run fingerprinting, per-packet checksums, flight recorder |
|
||||
| `hang-diagnosis.md` | Diagnosing MPI/accelerator hangs |
|
||||
|
||||
The key loop macros (defined in `Grid/threads/Accelerator.h`) are:
|
||||
- `accelerator_for(iter, num, nsimd, {...})` — maps to CUDA/HIP kernel or OpenMP loop; `nsimd` is the innermost SIMD lane count
|
||||
- `accelerator_forNB(...)` — non-blocking variant (no implicit barrier)
|
||||
- `accelerator_for2dNB(iter1, num1, iter2, num2, nsimd, {...})` — 2D kernel launch
|
||||
- `thread_for(iter, num, {...})` — CPU OpenMP loop (never dispatches to GPU)
|
||||
|
||||
On CPU builds, `accelerator_for` aliases to `thread_for`.
|
||||
|
||||
### Solver patterns
|
||||
|
||||
`SchurRedBlack` (`Grid/algorithms/iterative/SchurRedBlack.h`) implements red-black (even/odd) preconditioning for fermion operators. Most production fermion solves use `SchurRedBlackDiagMooeeSolve` or similar wrappers that internally call a `ConjugateGradient` on the Schur complement.
|
||||
|
||||
Mixed-precision solvers (`ConjugateGradientMixedPrec`, `BiCGSTABMixedPrec`) drive a double-precision outer loop with single-precision inner solves.
|
||||
|
||||
### Memory and I/O
|
||||
|
||||
- `Grid/allocator/` — aligned/NUMA-aware allocators; caching allocator via `--enable-alloc-cache`
|
||||
- `Grid/parallelIO/` — distributed parallel reader/writer for ILDG (via LIME), SciDAC, and native binary formats
|
||||
- `Grid/serialisation/` — text, binary, HDF5, XML/JSON serialisation of arbitrary Grid objects
|
||||
|
||||
### Executables
|
||||
|
||||
- `HMC/` — production HMC driver programmes (e.g. `Mobius2p1f.cc`, `DWF_plus_DSDR_nf2plus1_Shamir_Gparity.cc`)
|
||||
- `benchmarks/` — `Benchmark_dwf`, `Benchmark_ITT`, `Benchmark_comms`, `Benchmark_memory_bandwidth`, … used to qualify a new machine
|
||||
- `examples/` — small, readable programmes (`Example_plaquette.cc`, `Example_Mobius_spectrum.cc`) that are the best starting point for learning the API
|
||||
|
||||
Each of these directories auto-builds every top-level `.cc` as its own binary via `scripts/filelist`.
|
||||
|
||||
Every programme is wrapped in `Grid_init(&argc, &argv)` / `Grid_finalize()` (`Grid/util/Init.h`).
|
||||
|
||||
## Key Conventions
|
||||
|
||||
- **C++17** is required throughout.
|
||||
- Template structure: most classes are templated on `<_FImpl>` (fermion impl) or `<Gimpl>` (gauge impl), which encode the representation and precision. Instantiation is controlled by `--enable-fermion-instantiations`.
|
||||
- **Tensor indices are positional, not labelled.** The `Grid/tensors/` arithmetic recurses structurally over the `iScalar`/`iVector`/`iMatrix` nest: each level defines only the {scalar,vector,matrix}² products at its own level, with element types resolved by automatic type deduction, so every colour/spin/lorentz combination composes from ~200 lines (versus the pre-C++11 QDP++/PETE approach of machine-generating every case). An index's meaning derives entirely from its nesting depth counted from the outside; `iScalar` is the identity/broadcast case at every level. Never insert or remove a nesting level casually — the multiplication tables contract by position.
|
||||
- **Multigrid coarsening deepens the tensor nest by one level.** A coarse site vector is `iVector<CComplex,nbasis>`, and `innerProduct` on it returns `iScalar<CComplex>` — one level deeper than the fine block scalar. So the block-inner-product scalar type gains one `iScalar` wrapper per MG level (fine: `vTComplex`; level 2: `iScalar<vTComplex>`; see `examples/Example_pvdagm_3level.cc`). When calling `blockInnerProduct`/`blockZAXPY`/`blockOrthogonalise` on coarse fields, the coarse scalar type must match `decltype(innerProduct(siteVector(),siteVector()))` exactly; a wrong depth fails to compile (no viable `operator=` deep in the instantiation chain) rather than mis-contracting.
|
||||
- The `RealD`/`RealF`/`ComplexD`/`ComplexF` typedefs are used everywhere; avoid raw `double`/`float`.
|
||||
- Use `GRID_ASSERT(cond)` (defined in `Grid/GridStd.h`), not bare `assert` — it prints a Grid-formatted message and aborts cleanly under MPI.
|
||||
- Logging is stream-based, not macro-based: `std::cout << GridLogMessage << ... << std::endl;`. Channels declared in `Grid/log/Log.h` include `GridLogError`, `GridLogWarning`, `GridLogDebug`, `GridLogPerformance`, `GridLogIterative`, `GridLogSolver`, `GridLogHMC`, `GridLogComms`, `GridLogMemory`, `GridLogDslash`, `GridLogIRL`, `GridLogMG`. A subset is switched on at runtime with e.g. `--log Error,Warning,Message,Performance,Iterative,Integrator,Debug,Colours` (names given without the `GridLog` prefix).
|
||||
- Performance-critical paths use `GRID_TRACE(name)` from `Grid/perfmon/Tracing.h` (compiled out unless `--enable-tracing` selects a backend) and the `GridStopWatch` timers in `Grid/perfmon/Timer.h`.
|
||||
- Reductions across MPI ranks go through `GridBase::GlobalSum` / `GlobalMax`; never reduce with bare MPI calls inside library code.
|
||||
- Everything lives in `NAMESPACE_BEGIN(Grid)` / `NAMESPACE_END(Grid)` macros; follow the surrounding file rather than writing `namespace Grid { }`.
|
||||
- British spelling is used in identifiers and comments (`colour`, `neighbour`, `serialisation`).
|
||||
+3
-1
@@ -23,6 +23,7 @@ extern void * Grid_backtrace_buffer[_NBACKTRACE];
|
||||
#include <random>
|
||||
#include <functional>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <stdlib.h>
|
||||
#include <unistd.h>
|
||||
#include <strings.h>
|
||||
@@ -38,9 +39,10 @@ extern void * Grid_backtrace_buffer[_NBACKTRACE];
|
||||
|
||||
void GridAbort(void);
|
||||
|
||||
#define ASSLOG(A) ::write(STDERR_FILENO,A,strlen(A));
|
||||
#define ASSLOG(A) ::write(STDERR_FILENO,A,::strlen(A));
|
||||
#ifdef HAVE_EXECINFO_H
|
||||
#define GRID_ASSERT(b) if(!(b)) { \
|
||||
fflush(stdout); \
|
||||
ASSLOG(" GRID_ASSERT failure: "); \
|
||||
ASSLOG(__FILE__); \
|
||||
ASSLOG(" : "); \
|
||||
|
||||
+11
-7
@@ -54,21 +54,25 @@ Version.h: version-cache
|
||||
include Make.inc
|
||||
include Eigen.inc
|
||||
|
||||
extra_sources+=$(WILS_FERMION_FILES)
|
||||
extra_sources+=$(STAG_FERMION_FILES)
|
||||
if BUILD_FERMION_INSTANTIATIONS
|
||||
extra_sources+=$(WILS_FERMION_FILES)
|
||||
extra_sources+=$(STAG_FERMION_FILES)
|
||||
if BUILD_ZMOBIUS
|
||||
extra_sources+=$(ZWILS_FERMION_FILES)
|
||||
extra_sources+=$(ZWILS_FERMION_FILES)
|
||||
endif
|
||||
if BUILD_GPARITY
|
||||
extra_sources+=$(GP_FERMION_FILES)
|
||||
extra_sources+=$(GP_FERMION_FILES)
|
||||
endif
|
||||
if BUILD_FERMION_REPS
|
||||
extra_sources+=$(ADJ_FERMION_FILES)
|
||||
extra_sources+=$(TWOIND_FERMION_FILES)
|
||||
extra_sources+=$(ADJ_FERMION_FILES)
|
||||
extra_sources+=$(TWOIND_FERMION_FILES)
|
||||
endif
|
||||
if BUILD_SP
|
||||
extra_sources+=$(SP_FERMION_FILES)
|
||||
extra_sources+=$(SP_TWOIND_FERMION_FILES)
|
||||
if BUILD_FERMION_REPS
|
||||
extra_sources+=$(SP_TWOIND_FERMION_FILES)
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
lib_LIBRARIES = libGrid.a
|
||||
|
||||
+427
-235
@@ -1,6 +1,6 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: ./lib/Cshift.h
|
||||
|
||||
@@ -28,6 +28,15 @@ Author: Peter Boyle <paboyle@ph.ed.ac.uk>
|
||||
#ifndef _GRID_FFT_H_
|
||||
#define _GRID_FFT_H_
|
||||
|
||||
#ifdef GRID_CUDA
|
||||
#include <cufft.h>
|
||||
#endif
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#include <hipfft/hipfft.h>
|
||||
#endif
|
||||
|
||||
#if !defined(GRID_CUDA) && !defined(GRID_HIP)
|
||||
#ifdef HAVE_FFTW
|
||||
#if defined(USE_MKL) || defined(GRID_SYCL)
|
||||
#include <fftw/fftw3.h>
|
||||
@@ -35,266 +44,449 @@ Author: Peter Boyle <paboyle@ph.ed.ac.uk>
|
||||
#include <fftw3.h>
|
||||
#endif
|
||||
#endif
|
||||
#endif
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
template<class scalar> struct FFTW { };
|
||||
|
||||
#ifdef HAVE_FFTW
|
||||
template<> struct FFTW<ComplexD> {
|
||||
public:
|
||||
|
||||
typedef fftw_complex FFTW_scalar;
|
||||
typedef fftw_plan FFTW_plan;
|
||||
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, const int *n,int howmany,
|
||||
FFTW_scalar *in, const int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, const int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
return ::fftw_plan_many_dft(rank,n,howmany,in,inembed,istride,idist,out,onembed,ostride,odist,sign,flags);
|
||||
}
|
||||
|
||||
static void fftw_flops(const FFTW_plan p,double *add, double *mul, double *fmas){
|
||||
::fftw_flops(p,add,mul,fmas);
|
||||
}
|
||||
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out) {
|
||||
::fftw_execute_dft(p,in,out);
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) {
|
||||
::fftw_destroy_plan(p);
|
||||
}
|
||||
};
|
||||
|
||||
template<> struct FFTW<ComplexF> {
|
||||
public:
|
||||
|
||||
typedef fftwf_complex FFTW_scalar;
|
||||
typedef fftwf_plan FFTW_plan;
|
||||
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, const int *n,int howmany,
|
||||
FFTW_scalar *in, const int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, const int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
return ::fftwf_plan_many_dft(rank,n,howmany,in,inembed,istride,idist,out,onembed,ostride,odist,sign,flags);
|
||||
}
|
||||
|
||||
static void fftw_flops(const FFTW_plan p,double *add, double *mul, double *fmas){
|
||||
::fftwf_flops(p,add,mul,fmas);
|
||||
}
|
||||
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out) {
|
||||
::fftwf_execute_dft(p,in,out);
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) {
|
||||
::fftwf_destroy_plan(p);
|
||||
}
|
||||
};
|
||||
|
||||
#endif
|
||||
|
||||
#ifndef FFTW_FORWARD
|
||||
#define FFTW_FORWARD (-1)
|
||||
#define FFTW_BACKWARD (+1)
|
||||
#define FFTW_ESTIMATE (0)
|
||||
#endif
|
||||
|
||||
class FFT {
|
||||
private:
|
||||
|
||||
GridCartesian *vgrid;
|
||||
GridCartesian *sgrid;
|
||||
|
||||
int Nd;
|
||||
double flops;
|
||||
double flops_call;
|
||||
uint64_t usec;
|
||||
|
||||
Coordinate dimensions;
|
||||
Coordinate processors;
|
||||
Coordinate processor_coor;
|
||||
|
||||
template<class scalar> struct FFTW {
|
||||
};
|
||||
|
||||
#ifdef GRID_HIP
|
||||
template<> struct FFTW<ComplexD> {
|
||||
public:
|
||||
|
||||
static const int forward=FFTW_FORWARD;
|
||||
static const int backward=FFTW_BACKWARD;
|
||||
|
||||
double Flops(void) {return flops;}
|
||||
double MFlops(void) {return flops/usec;}
|
||||
double USec(void) {return (double)usec;}
|
||||
|
||||
FFT ( GridCartesian * grid ) :
|
||||
vgrid(grid),
|
||||
Nd(grid->_ndimension),
|
||||
dimensions(grid->_fdimensions),
|
||||
processors(grid->_processors),
|
||||
processor_coor(grid->_processor_coor)
|
||||
{
|
||||
flops=0;
|
||||
usec =0;
|
||||
Coordinate layout(Nd,1);
|
||||
sgrid = new GridCartesian(dimensions,layout,processors,*grid);
|
||||
};
|
||||
|
||||
~FFT ( void) {
|
||||
delete sgrid;
|
||||
typedef hipfftDoubleComplex FFTW_scalar;
|
||||
typedef hipfftHandle FFTW_plan;
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, int *n,int howmany,
|
||||
FFTW_scalar *in, int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
FFTW_plan p;
|
||||
auto rv = hipfftPlanMany(&p,rank,n,n,istride,idist,n,ostride,odist,HIPFFT_Z2Z,howmany);
|
||||
GRID_ASSERT(rv==HIPFFT_SUCCESS);
|
||||
return p;
|
||||
}
|
||||
|
||||
template<class vobj>
|
||||
void FFT_dim_mask(Lattice<vobj> &result,const Lattice<vobj> &source,Coordinate mask,int sign){
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out, int sign) {
|
||||
hipfftResult rv;
|
||||
if ( sign == forward ) rv =hipfftExecZ2Z(p,in,out,HIPFFT_FORWARD);
|
||||
else rv =hipfftExecZ2Z(p,in,out,HIPFFT_BACKWARD);
|
||||
accelerator_barrier();
|
||||
GRID_ASSERT(rv==HIPFFT_SUCCESS);
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) { hipfftDestroy(p); }
|
||||
};
|
||||
template<> struct FFTW<ComplexF> {
|
||||
public:
|
||||
static const int forward=FFTW_FORWARD;
|
||||
static const int backward=FFTW_BACKWARD;
|
||||
typedef hipfftComplex FFTW_scalar;
|
||||
typedef hipfftHandle FFTW_plan;
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, int *n,int howmany,
|
||||
FFTW_scalar *in, int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
FFTW_plan p;
|
||||
auto rv = hipfftPlanMany(&p,rank,n,n,istride,idist,n,ostride,odist,HIPFFT_C2C,howmany);
|
||||
GRID_ASSERT(rv==HIPFFT_SUCCESS);
|
||||
return p;
|
||||
}
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out, int sign) {
|
||||
hipfftResult rv;
|
||||
if ( sign == forward ) rv =hipfftExecC2C(p,in,out,HIPFFT_FORWARD);
|
||||
else rv =hipfftExecC2C(p,in,out,HIPFFT_BACKWARD);
|
||||
accelerator_barrier();
|
||||
GRID_ASSERT(rv==HIPFFT_SUCCESS);
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) { hipfftDestroy(p); }
|
||||
};
|
||||
#endif
|
||||
|
||||
conformable(result.Grid(),vgrid);
|
||||
conformable(source.Grid(),vgrid);
|
||||
Lattice<vobj> tmp(vgrid);
|
||||
tmp = source;
|
||||
for(int d=0;d<Nd;d++){
|
||||
if( mask[d] ) {
|
||||
FFT_dim(result,tmp,d,sign);
|
||||
tmp=result;
|
||||
#ifdef GRID_CUDA
|
||||
template<> struct FFTW<ComplexD> {
|
||||
public:
|
||||
static const int forward=FFTW_FORWARD;
|
||||
static const int backward=FFTW_BACKWARD;
|
||||
typedef cufftDoubleComplex FFTW_scalar;
|
||||
typedef cufftHandle FFTW_plan;
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, int *n,int howmany,
|
||||
FFTW_scalar *in, int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
FFTW_plan p;
|
||||
cufftPlanMany(&p,rank,n,n,istride,idist,n,ostride,odist,CUFFT_Z2Z,howmany);
|
||||
return p;
|
||||
}
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out, int sign) {
|
||||
if ( sign == forward ) cufftExecZ2Z(p,in,out,CUFFT_FORWARD);
|
||||
else cufftExecZ2Z(p,in,out,CUFFT_INVERSE);
|
||||
accelerator_barrier();
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) { cufftDestroy(p); }
|
||||
};
|
||||
template<> struct FFTW<ComplexF> {
|
||||
public:
|
||||
static const int forward=FFTW_FORWARD;
|
||||
static const int backward=FFTW_BACKWARD;
|
||||
typedef cufftComplex FFTW_scalar;
|
||||
typedef cufftHandle FFTW_plan;
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, int *n,int howmany,
|
||||
FFTW_scalar *in, int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
FFTW_plan p;
|
||||
cufftPlanMany(&p,rank,n,n,istride,idist,n,ostride,odist,CUFFT_C2C,howmany);
|
||||
return p;
|
||||
}
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out, int sign) {
|
||||
if ( sign == forward ) cufftExecC2C(p,in,out,CUFFT_FORWARD);
|
||||
else cufftExecC2C(p,in,out,CUFFT_INVERSE);
|
||||
accelerator_barrier();
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) { cufftDestroy(p); }
|
||||
};
|
||||
#endif
|
||||
|
||||
#if !defined(GRID_CUDA) && !defined(GRID_HIP)
|
||||
#ifdef HAVE_FFTW
|
||||
template<> struct FFTW<ComplexD> {
|
||||
public:
|
||||
typedef fftw_complex FFTW_scalar;
|
||||
typedef fftw_plan FFTW_plan;
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, int *n,int howmany,
|
||||
FFTW_scalar *in, int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
return ::fftw_plan_many_dft(rank,n,howmany,in,inembed,istride,idist,out,onembed,ostride,odist,sign,flags);
|
||||
}
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out, int sign) {
|
||||
::fftw_execute_dft(p,in,out);
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) { ::fftw_destroy_plan(p); }
|
||||
};
|
||||
template<> struct FFTW<ComplexF> {
|
||||
public:
|
||||
typedef fftwf_complex FFTW_scalar;
|
||||
typedef fftwf_plan FFTW_plan;
|
||||
static FFTW_plan fftw_plan_many_dft(int rank, int *n,int howmany,
|
||||
FFTW_scalar *in, int *inembed,
|
||||
int istride, int idist,
|
||||
FFTW_scalar *out, int *onembed,
|
||||
int ostride, int odist,
|
||||
int sign, unsigned flags) {
|
||||
return ::fftwf_plan_many_dft(rank,n,howmany,in,inembed,istride,idist,out,onembed,ostride,odist,sign,flags);
|
||||
}
|
||||
inline static void fftw_execute_dft(const FFTW_plan p,FFTW_scalar *in,FFTW_scalar *out, int sign) {
|
||||
::fftwf_execute_dft(p,in,out);
|
||||
}
|
||||
inline static void fftw_destroy_plan(const FFTW_plan p) { ::fftwf_destroy_plan(p); }
|
||||
};
|
||||
#endif
|
||||
#endif
|
||||
|
||||
struct FFTbase {
|
||||
double flops;
|
||||
double flops_call;
|
||||
uint64_t usec;
|
||||
GridCartesian *_grid;
|
||||
|
||||
static const int forward = FFTW_FORWARD;
|
||||
static const int backward = FFTW_BACKWARD;
|
||||
|
||||
double Flops(void) { return flops; }
|
||||
double MFlops(void) { return flops / usec; }
|
||||
double USec(void) { return (double)usec; }
|
||||
|
||||
FFTbase(GridCartesian *grid) : _grid(grid), flops(0), flops_call(0), usec(0) {}
|
||||
};
|
||||
|
||||
// Barrel-shift gather, FFT execute, and insert. Called by both FFT and PlannedFFT.
|
||||
// The caller is responsible for plan acquisition and destruction.
|
||||
template<class vobj>
|
||||
static void FFT_dim_execute(
|
||||
Lattice<vobj> &result,
|
||||
const Lattice<vobj> &source,
|
||||
int dim, int sign,
|
||||
typename FFTW<typename vobj::scalar_type>::FFTW_plan p,
|
||||
GridCartesian *grid,
|
||||
double &flops, double &flops_call, uint64_t &usec)
|
||||
{
|
||||
typedef typename vobj::scalar_type scalar;
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::scalar_type scalar_type;
|
||||
typedef typename vobj::vector_type vector_type;
|
||||
typedef typename FFTW<scalar>::FFTW_scalar FFTW_scalar;
|
||||
|
||||
const int Ndim = grid->Nd();
|
||||
int L = grid->_ldimensions[dim];
|
||||
int G = grid->_fdimensions[dim];
|
||||
int Ncomp = sizeof(sobj) / sizeof(scalar);
|
||||
int64_t Nlow = 1, Nhigh = 1;
|
||||
for (int d = 0; d < dim; d++) Nlow *= grid->_ldimensions[d];
|
||||
for (int d = dim+1; d < Ndim; d++) Nhigh *= grid->_ldimensions[d];
|
||||
int64_t Nperp = Nlow * Nhigh;
|
||||
|
||||
deviceVector<scalar> pgbuf(Nperp * Ncomp * G);
|
||||
scalar *pgbuf_v = &pgbuf[0];
|
||||
int howmany = Ncomp * Nperp;
|
||||
|
||||
scalar div;
|
||||
if (sign == FFTW_BACKWARD) div = 1.0 / G;
|
||||
else if (sign == FFTW_FORWARD) div = 1.0;
|
||||
else GRID_ASSERT(0);
|
||||
|
||||
double t_pencil = 0, t_fft = 0, t_copy = 0, t_shift = 0;
|
||||
double t_total = -usecond();
|
||||
|
||||
result = source;
|
||||
int pc = grid->_processor_coor[dim];
|
||||
|
||||
const Coordinate ldims = grid->_ldimensions;
|
||||
const Coordinate rdims = grid->_rdimensions;
|
||||
const Coordinate sdims = grid->_simd_layout;
|
||||
const Coordinate processors = grid->_processors;
|
||||
|
||||
Coordinate pgdims(Ndim);
|
||||
pgdims[0] = G;
|
||||
for (int d = 0, dd = 1; d < Ndim; d++)
|
||||
if (d != dim) pgdims[dd++] = ldims[d];
|
||||
int64_t pgvol = 1;
|
||||
for (int d = 0; d < Ndim; d++) pgvol *= pgdims[d];
|
||||
|
||||
const int Nsimd = vobj::Nsimd();
|
||||
t_pencil = -usecond();
|
||||
for (int p_idx = 0; p_idx < processors[dim]; p_idx++) {
|
||||
t_copy -= usecond();
|
||||
autoView(r_v, result, AcceleratorRead);
|
||||
accelerator_for(idx, grid->oSites(), vobj::Nsimd(), {
|
||||
#ifdef GRID_SIMT
|
||||
{
|
||||
int lane = acceleratorSIMTlane(Nsimd);
|
||||
#else
|
||||
for (int lane = 0; lane < Nsimd; lane++) {
|
||||
#endif
|
||||
Coordinate icoor, ocoor, pgcoor;
|
||||
Lexicographic::CoorFromIndex(icoor, lane, sdims);
|
||||
Lexicographic::CoorFromIndex(ocoor, idx, rdims);
|
||||
pgcoor[0] = ocoor[dim] + icoor[dim]*rdims[dim] + ((pc+p_idx)%processors[dim])*L;
|
||||
for (int d = 0, dd = 1; d < Ndim; d++)
|
||||
if (d != dim) { pgcoor[dd] = ocoor[d] + icoor[d]*rdims[d]; dd++; }
|
||||
int64_t pgidx;
|
||||
Lexicographic::IndexFromCoor(pgcoor, pgidx, pgdims);
|
||||
vector_type *from = (vector_type *)&r_v[idx];
|
||||
scalar_type stmp;
|
||||
for (int w = 0; w < Ncomp; w++) {
|
||||
stmp = getlane(from[w], lane);
|
||||
pgbuf_v[pgidx + w*pgvol] = stmp;
|
||||
}
|
||||
#ifdef GRID_SIMT
|
||||
}
|
||||
#else
|
||||
}
|
||||
#endif
|
||||
});
|
||||
t_copy += usecond();
|
||||
if (p_idx != processors[dim] - 1) {
|
||||
Lattice<vobj> temp(grid);
|
||||
t_shift -= usecond();
|
||||
temp = Cshift(result, dim, L); result = temp;
|
||||
t_shift += usecond();
|
||||
}
|
||||
}
|
||||
t_pencil += usecond();
|
||||
|
||||
FFTW_scalar *in = (FFTW_scalar *)pgbuf_v;
|
||||
FFTW_scalar *out = (FFTW_scalar *)pgbuf_v;
|
||||
t_fft = -usecond();
|
||||
FFTW<scalar>::fftw_execute_dft(p, in, out, sign);
|
||||
t_fft += usecond();
|
||||
|
||||
flops_call = 5.0 * howmany * G * log2(G);
|
||||
usec = t_fft;
|
||||
flops = flops_call;
|
||||
|
||||
result = Zero();
|
||||
double t_insert = -usecond();
|
||||
{
|
||||
autoView(r_v, result, AcceleratorWrite);
|
||||
accelerator_for(idx, grid->oSites(), Nsimd, {
|
||||
#ifdef GRID_SIMT
|
||||
{
|
||||
int lane = acceleratorSIMTlane(Nsimd);
|
||||
#else
|
||||
for (int lane = 0; lane < Nsimd; lane++) {
|
||||
#endif
|
||||
Coordinate icoor(Ndim), ocoor(Ndim), pgcoor(Ndim);
|
||||
Lexicographic::CoorFromIndex(icoor, lane, sdims);
|
||||
Lexicographic::CoorFromIndex(ocoor, idx, rdims);
|
||||
pgcoor[0] = ocoor[dim] + icoor[dim]*rdims[dim] + pc*L;
|
||||
for (int d = 0, dd = 1; d < Ndim; d++)
|
||||
if (d != dim) { pgcoor[dd] = ocoor[d] + icoor[d]*rdims[d]; dd++; }
|
||||
int64_t pgidx;
|
||||
Lexicographic::IndexFromCoor(pgcoor, pgidx, pgdims);
|
||||
vector_type *to = (vector_type *)&r_v[idx];
|
||||
scalar_type stmp;
|
||||
for (int w = 0; w < Ncomp; w++) {
|
||||
stmp = pgbuf_v[pgidx + w*pgvol];
|
||||
putlane(to[w], stmp, lane);
|
||||
}
|
||||
#ifdef GRID_SIMT
|
||||
}
|
||||
#else
|
||||
}
|
||||
#endif
|
||||
});
|
||||
}
|
||||
result = result * div;
|
||||
t_insert += usecond();
|
||||
t_total += usecond();
|
||||
|
||||
std::cout << GridLogPerformance << " FFT took " << t_total/1.0e6 << " s" << std::endl;
|
||||
std::cout << GridLogPerformance << " FFT pencil " << t_pencil/1.0e6 << " s" << std::endl;
|
||||
std::cout << GridLogPerformance << " of which copy " << t_copy/1.0e6 << " s" << std::endl;
|
||||
std::cout << GridLogPerformance << " of which shift" << t_shift/1.0e6 << " s" << std::endl;
|
||||
std::cout << GridLogPerformance << " FFT kernels " << t_fft/1.0e6 << " s" << std::endl;
|
||||
std::cout << GridLogPerformance << " FFT insert " << t_insert/1.0e6 << " s" << std::endl;
|
||||
}
|
||||
|
||||
class FFT : public FFTbase {
|
||||
public:
|
||||
FFT(GridCartesian *grid) : FFTbase(grid) {}
|
||||
~FFT() {}
|
||||
|
||||
template<class vobj>
|
||||
void FFT_dim_mask(Lattice<vobj> &result, const Lattice<vobj> &source, Coordinate mask, int sign) {
|
||||
const int Ndim = _grid->Nd();
|
||||
Lattice<vobj> tmp = source;
|
||||
for (int d = 0; d < Ndim; d++) {
|
||||
if (mask[d]) {
|
||||
FFT_dim(result, tmp, d, sign);
|
||||
tmp = result;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<class vobj>
|
||||
void FFT_all_dim(Lattice<vobj> &result,const Lattice<vobj> &source,int sign){
|
||||
Coordinate mask(Nd,1);
|
||||
FFT_dim_mask(result,source,mask,sign);
|
||||
void FFT_all_dim(Lattice<vobj> &result, const Lattice<vobj> &source, int sign) {
|
||||
Coordinate mask(_grid->Nd(), 1);
|
||||
FFT_dim_mask(result, source, mask, sign);
|
||||
}
|
||||
|
||||
|
||||
template<class vobj>
|
||||
void FFT_dim(Lattice<vobj> &result,const Lattice<vobj> &source,int dim, int sign){
|
||||
#ifndef HAVE_FFTW
|
||||
std::cerr << "FFTW is not compiled but is called"<<std::endl;
|
||||
GRID_ASSERT(0);
|
||||
#else
|
||||
conformable(result.Grid(),vgrid);
|
||||
conformable(source.Grid(),vgrid);
|
||||
void FFT_dim(Lattice<vobj> &result, const Lattice<vobj> &source, int dim, int sign) {
|
||||
GRID_ASSERT(source.Grid() == _grid);
|
||||
GRID_ASSERT(result.Grid() == _grid);
|
||||
conformable(result.Grid(), source.Grid());
|
||||
|
||||
int L = vgrid->_ldimensions[dim];
|
||||
int G = vgrid->_fdimensions[dim];
|
||||
|
||||
Coordinate layout(Nd,1);
|
||||
Coordinate pencil_gd(vgrid->_fdimensions);
|
||||
|
||||
pencil_gd[dim] = G*processors[dim];
|
||||
|
||||
// Pencil global vol LxLxGxLxL per node
|
||||
GridCartesian pencil_g(pencil_gd,layout,processors,*vgrid);
|
||||
|
||||
// Construct pencils
|
||||
typedef typename vobj::scalar_type scalar;
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename sobj::scalar_type scalar;
|
||||
|
||||
Lattice<sobj> pgbuf(&pencil_g);
|
||||
autoView(pgbuf_v , pgbuf, CpuWrite);
|
||||
//std::cout << "CPU view" << std::endl;
|
||||
|
||||
typedef typename FFTW<scalar>::FFTW_scalar FFTW_scalar;
|
||||
typedef typename FFTW<scalar>::FFTW_plan FFTW_plan;
|
||||
|
||||
int Ncomp = sizeof(sobj)/sizeof(scalar);
|
||||
int Nlow = 1;
|
||||
for(int d=0;d<dim;d++){
|
||||
Nlow*=vgrid->_ldimensions[d];
|
||||
}
|
||||
|
||||
int rank = 1; /* 1d transforms */
|
||||
int n[] = {G}; /* 1d transforms of length G */
|
||||
int howmany = Ncomp;
|
||||
int odist,idist,istride,ostride;
|
||||
idist = odist = 1; /* Distance between consecutive FT's */
|
||||
istride = ostride = Ncomp*Nlow; /* distance between two elements in the same FT */
|
||||
int *inembed = n, *onembed = n;
|
||||
|
||||
scalar div;
|
||||
if ( sign == backward ) div = 1.0/G;
|
||||
else if ( sign == forward ) div = 1.0;
|
||||
else GRID_ASSERT(0);
|
||||
|
||||
//std::cout << GridLogPerformance<<"Making FFTW plan" << std::endl;
|
||||
FFTW_plan p;
|
||||
{
|
||||
FFTW_scalar *in = (FFTW_scalar *)&pgbuf_v[0];
|
||||
FFTW_scalar *out= (FFTW_scalar *)&pgbuf_v[0];
|
||||
p = FFTW<scalar>::fftw_plan_many_dft(rank,n,howmany,
|
||||
in,inembed,
|
||||
istride,idist,
|
||||
out,onembed,
|
||||
ostride, odist,
|
||||
sign,FFTW_ESTIMATE);
|
||||
}
|
||||
|
||||
// Barrel shift and collect global pencil
|
||||
//std::cout << GridLogPerformance<<"Making pencil" << std::endl;
|
||||
Coordinate lcoor(Nd), gcoor(Nd);
|
||||
result = source;
|
||||
int pc = processor_coor[dim];
|
||||
for(int p=0;p<processors[dim];p++) {
|
||||
{
|
||||
autoView(r_v,result,CpuRead);
|
||||
autoView(p_v,pgbuf,CpuWrite);
|
||||
thread_for(idx, sgrid->lSites(),{
|
||||
Coordinate cbuf(Nd);
|
||||
sobj s;
|
||||
sgrid->LocalIndexToLocalCoor(idx,cbuf);
|
||||
peekLocalSite(s,r_v,cbuf);
|
||||
cbuf[dim]+=((pc+p) % processors[dim])*L;
|
||||
pokeLocalSite(s,p_v,cbuf);
|
||||
});
|
||||
}
|
||||
if (p != processors[dim] - 1) {
|
||||
result = Cshift(result,dim,L);
|
||||
}
|
||||
}
|
||||
|
||||
//std::cout <<GridLogPerformance<< "Looping orthog" << std::endl;
|
||||
// Loop over orthog coords
|
||||
int NN=pencil_g.lSites();
|
||||
GridStopWatch timer;
|
||||
timer.Start();
|
||||
thread_for( idx,NN,{
|
||||
Coordinate cbuf(Nd);
|
||||
pencil_g.LocalIndexToLocalCoor(idx, cbuf);
|
||||
if ( cbuf[dim] == 0 ) { // restricts loop to plane at lcoor[dim]==0
|
||||
FFTW_scalar *in = (FFTW_scalar *)&pgbuf_v[idx];
|
||||
FFTW_scalar *out= (FFTW_scalar *)&pgbuf_v[idx];
|
||||
FFTW<scalar>::fftw_execute_dft(p,in,out);
|
||||
}
|
||||
});
|
||||
timer.Stop();
|
||||
|
||||
// performance counting
|
||||
double add,mul,fma;
|
||||
FFTW<scalar>::fftw_flops(p,&add,&mul,&fma);
|
||||
flops_call = add+mul+2.0*fma;
|
||||
usec += timer.useconds();
|
||||
flops+= flops_call*NN;
|
||||
|
||||
//std::cout <<GridLogPerformance<< "Writing back results " << std::endl;
|
||||
// writing out result
|
||||
{
|
||||
autoView(pgbuf_v,pgbuf,CpuRead);
|
||||
autoView(result_v,result,CpuWrite);
|
||||
thread_for(idx,sgrid->lSites(),{
|
||||
Coordinate clbuf(Nd), cgbuf(Nd);
|
||||
sobj s;
|
||||
sgrid->LocalIndexToLocalCoor(idx,clbuf);
|
||||
cgbuf = clbuf;
|
||||
cgbuf[dim] = clbuf[dim]+L*pc;
|
||||
peekLocalSite(s,pgbuf_v,cgbuf);
|
||||
pokeLocalSite(s,result_v,clbuf);
|
||||
});
|
||||
}
|
||||
result = result*div;
|
||||
|
||||
//std::cout <<GridLogPerformance<< "Destroying plan " << std::endl;
|
||||
// destroying plan
|
||||
|
||||
const int Ndim = _grid->Nd();
|
||||
int G = _grid->_fdimensions[dim];
|
||||
int Ncomp = sizeof(sobj) / sizeof(scalar);
|
||||
int64_t Nperp = 1;
|
||||
for (int d = 0; d < Ndim; d++)
|
||||
if (d != dim) Nperp *= _grid->_ldimensions[d];
|
||||
int n[] = {G};
|
||||
int howmany = Ncomp * Nperp;
|
||||
|
||||
deviceVector<scalar> dummy(2);
|
||||
FFTW_scalar *buf = (FFTW_scalar *)&dummy[0];
|
||||
FFTW_plan p = FFTW<scalar>::fftw_plan_many_dft(1, n, howmany,
|
||||
buf, n, 1, G,
|
||||
buf, n, 1, G,
|
||||
sign, FFTW_ESTIMATE);
|
||||
FFT_dim_execute(result, source, dim, sign, p, _grid, flops, flops_call, usec);
|
||||
FFTW<scalar>::fftw_destroy_plan(p);
|
||||
#endif
|
||||
}
|
||||
};
|
||||
|
||||
template<class vobj>
|
||||
class PlannedFFT : public FFTbase {
|
||||
private:
|
||||
typedef typename vobj::scalar_type scalar;
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::vector_type vector_type;
|
||||
typedef typename FFTW<scalar>::FFTW_scalar FFTW_scalar;
|
||||
typedef typename FFTW<scalar>::FFTW_plan FFTW_plan;
|
||||
|
||||
std::vector<FFTW_plan> forward_plans;
|
||||
std::vector<FFTW_plan> backward_plans;
|
||||
|
||||
void PlanCreate() {
|
||||
const int Ndim = _grid->Nd();
|
||||
forward_plans.resize(Ndim);
|
||||
backward_plans.resize(Ndim);
|
||||
|
||||
for (int d = 0; d < Ndim; d++) {
|
||||
int G = _grid->_fdimensions[d];
|
||||
int Ncomp = sizeof(sobj) / sizeof(scalar);
|
||||
int64_t Nperp = 1;
|
||||
for (int dd = 0; dd < Ndim; dd++)
|
||||
if (dd != d) Nperp *= _grid->_ldimensions[dd];
|
||||
int howmany = Ncomp * (int)Nperp;
|
||||
int n[] = {G};
|
||||
|
||||
deviceVector<scalar> dummy(2);
|
||||
FFTW_scalar *buf = (FFTW_scalar *)&dummy[0];
|
||||
|
||||
forward_plans[d] = FFTW<scalar>::fftw_plan_many_dft(1, n, howmany, buf, n, 1, G, buf, n, 1, G, FFTW_FORWARD, FFTW_ESTIMATE);
|
||||
backward_plans[d] = FFTW<scalar>::fftw_plan_many_dft(1, n, howmany, buf, n, 1, G, buf, n, 1, G, FFTW_BACKWARD, FFTW_ESTIMATE);
|
||||
}
|
||||
}
|
||||
|
||||
void PlanDestroy() {
|
||||
for (auto p : forward_plans) FFTW<scalar>::fftw_destroy_plan(p);
|
||||
for (auto p : backward_plans) FFTW<scalar>::fftw_destroy_plan(p);
|
||||
forward_plans.clear();
|
||||
backward_plans.clear();
|
||||
}
|
||||
|
||||
public:
|
||||
PlannedFFT(GridCartesian *grid) : FFTbase(grid) { PlanCreate(); }
|
||||
~PlannedFFT() { PlanDestroy(); }
|
||||
|
||||
void FFT_dim_mask(Lattice<vobj> &result, const Lattice<vobj> &source, Coordinate mask, int sign) {
|
||||
const int Ndim = _grid->Nd();
|
||||
Lattice<vobj> tmp = source;
|
||||
for (int d = 0; d < Ndim; d++) {
|
||||
if (mask[d]) {
|
||||
FFT_dim(result, tmp, d, sign);
|
||||
tmp = result;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void FFT_all_dim(Lattice<vobj> &result, const Lattice<vobj> &source, int sign) {
|
||||
Coordinate mask(_grid->Nd(), 1);
|
||||
FFT_dim_mask(result, source, mask, sign);
|
||||
}
|
||||
|
||||
void FFT_dim(Lattice<vobj> &result, const Lattice<vobj> &source, int dim, int sign) {
|
||||
GRID_ASSERT(source.Grid() == _grid);
|
||||
GRID_ASSERT(result.Grid() == _grid);
|
||||
GRID_ASSERT((int)forward_plans.size() == _grid->Nd());
|
||||
conformable(result.Grid(), source.Grid());
|
||||
FFTW_plan p = (sign == forward ? forward_plans : backward_plans)[dim];
|
||||
FFT_dim_execute(result, source, dim, sign, p, _grid, flops, flops_call, usec);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -28,6 +28,7 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
#pragma once
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#include <hip/hip_version.h>
|
||||
#include <hipblas/hipblas.h>
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -67,6 +68,59 @@ NAMESPACE_BEGIN(Grid);
|
||||
enum GridBLASOperation_t { GridBLAS_OP_N, GridBLAS_OP_T, GridBLAS_OP_C } ;
|
||||
enum GridBLASPrecision_t { GridBLAS_PRECISION_DEFAULT, GridBLAS_PRECISION_16F, GridBLAS_PRECISION_16BF, GridBLAS_PRECISION_TF32 };
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// BLAS scalar constants: the put() wrapper OWNS the residency policy so
|
||||
// call sites just pass values (scalars-live-on-the-host rule).
|
||||
//
|
||||
// Policy per backend:
|
||||
// - CUDA / HIP : handles are put in HOST pointer mode at Init (cuBLAS docs
|
||||
// 2.2.7: host mode is the documented default; 2.1.5: host-mode scalars
|
||||
// are consumed AT CALL TIME, "can be freed just after the return of the
|
||||
// call even though the kernel launch is asynchronous"). put() stores
|
||||
// the value in persistent host memory and returns its address: ZERO
|
||||
// host->device copies.
|
||||
// - SYCL : the oneMKL group-API alpha/beta arrays are dereferenced
|
||||
// USM-side (spec is silent for the group API; implementation observed
|
||||
// to require USM-accessible storage -- host stack pointers fault).
|
||||
// put() keeps a device-resident copy with VALUE CACHING: the copy is
|
||||
// issued only when the value changes (accumulation pattern
|
||||
// beta = (p==0 ? 0 : 1) costs two copies per Mult instead of npoint).
|
||||
//
|
||||
// Motivation (rocprof, Frontier, 2026-08-13): per-call alpha/beta device
|
||||
// staging generated ~92k tiny staged hipMemcpys in a 12s solve window
|
||||
// (~26% of host API time) at the latency-bound coarse level.
|
||||
// NB not thread safe -- matches the single-threaded host BLAS call
|
||||
// pattern of the per-call staging it replaces.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
template<class T>
|
||||
class GridBLASDeviceConstant {
|
||||
#ifdef GRID_SYCL
|
||||
deviceVector<T> dev;
|
||||
T host;
|
||||
int valid;
|
||||
public:
|
||||
GridBLASDeviceConstant() : dev(1), valid(0) {};
|
||||
T * put(T v) {
|
||||
if ( (!valid) || (v != host) ) {
|
||||
acceleratorCopyToDevice((void *)&v,(void *)&dev[0],sizeof(T));
|
||||
host = v;
|
||||
valid = 1;
|
||||
}
|
||||
return &dev[0];
|
||||
}
|
||||
#else
|
||||
// CUDA / HIP in HOST pointer mode (and CPU/Eigen, where the pointer is
|
||||
// unused): persistent host storage, no copies ever.
|
||||
T host;
|
||||
public:
|
||||
GridBLASDeviceConstant() {};
|
||||
T * put(T v) {
|
||||
host = v;
|
||||
return &host;
|
||||
}
|
||||
#endif
|
||||
};
|
||||
|
||||
class GridBLAS {
|
||||
public:
|
||||
|
||||
@@ -80,11 +134,31 @@ public:
|
||||
#ifdef GRID_CUDA
|
||||
std::cout << "cublasCreate"<<std::endl;
|
||||
cublasCreate(&gridblasHandle);
|
||||
cublasSetPointerMode(gridblasHandle, CUBLAS_POINTER_MODE_DEVICE);
|
||||
// HOST pointer mode: scalars consumed at call time from host memory
|
||||
// (cuBLAS docs 2.1.5/2.2.7) -- no device staging of alpha/beta.
|
||||
// DEVICE mode would be a deliberate opt-in for device-produced
|
||||
// scalars (e.g. a future graph-captured solver).
|
||||
cublasSetPointerMode(gridblasHandle, CUBLAS_POINTER_MODE_HOST);
|
||||
{
|
||||
cublasPointerMode_t pm;
|
||||
cublasGetPointerMode(gridblasHandle,&pm);
|
||||
std::cout << "GridBLAS: cuBLAS pointer mode "
|
||||
<< ((pm==CUBLAS_POINTER_MODE_DEVICE)?"DEVICE":"HOST") <<std::endl;
|
||||
}
|
||||
#endif
|
||||
#ifdef GRID_HIP
|
||||
std::cout << "hipblasCreate"<<std::endl;
|
||||
hipblasCreate(&gridblasHandle);
|
||||
// Explicit HOST mode: the hipBLAS default is UNDOCUMENTED in the
|
||||
// headers (enum 0 == HOST by cuBLAS-mirroring convention only);
|
||||
// set it and print it so every log carries the ground truth.
|
||||
hipblasSetPointerMode(gridblasHandle, HIPBLAS_POINTER_MODE_HOST);
|
||||
{
|
||||
hipblasPointerMode_t pm;
|
||||
hipblasGetPointerMode(gridblasHandle,&pm);
|
||||
std::cout << "GridBLAS: hipBLAS pointer mode "
|
||||
<< ((pm==HIPBLAS_POINTER_MODE_DEVICE)?"DEVICE":"HOST") <<std::endl;
|
||||
}
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
gridblasHandle = theGridAccelerator;
|
||||
@@ -111,6 +185,7 @@ public:
|
||||
default:
|
||||
GRID_ASSERT(0);
|
||||
}
|
||||
return CUBLAS_COMPUTE_32F_FAST_16F;
|
||||
}
|
||||
#endif
|
||||
// Force construct once
|
||||
@@ -238,11 +313,11 @@ public:
|
||||
if(OpB!=GridBLAS_OP_N)
|
||||
ldb = n;
|
||||
|
||||
static deviceVector<ComplexD> alpha_p(1);
|
||||
static deviceVector<ComplexD> beta_p(1);
|
||||
// can prestore the 1 and the zero on device
|
||||
acceleratorCopyToDevice((void *)&alpha,(void *)&alpha_p[0],sizeof(ComplexD));
|
||||
acceleratorCopyToDevice((void *)&beta ,(void *)&beta_p[0],sizeof(ComplexD));
|
||||
// Cached device constants: copy only on value change (see GridBLASDeviceConstant)
|
||||
static GridBLASDeviceConstant<ComplexD> alpha_c;
|
||||
static GridBLASDeviceConstant<ComplexD> beta_c;
|
||||
ComplexD *alpha_p = alpha_c.put(alpha);
|
||||
ComplexD *beta_p = beta_c.put(beta);
|
||||
RealD t0=usecond();
|
||||
// std::cout << "ZgemmBatched mnk "<<m<<","<<n<<","<<k<<" count "<<batchCount<<std::endl;
|
||||
#ifdef GRID_HIP
|
||||
@@ -254,17 +329,30 @@ public:
|
||||
if ( OpB == GridBLAS_OP_N ) hOpB = HIPBLAS_OP_N;
|
||||
if ( OpB == GridBLAS_OP_T ) hOpB = HIPBLAS_OP_T;
|
||||
if ( OpB == GridBLAS_OP_C ) hOpB = HIPBLAS_OP_C;
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasZgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipblasDoubleComplex *) &alpha_p[0],
|
||||
(hipblasDoubleComplex **)&Amk[0], lda,
|
||||
(hipblasDoubleComplex **)&Bkn[0], ldb,
|
||||
(hipblasDoubleComplex *) &beta_p[0],
|
||||
(hipblasDoubleComplex **)&Cmn[0], ldc,
|
||||
(hipDoubleComplex *) &alpha_p[0],
|
||||
(hipDoubleComplex **)&Amk[0], lda,
|
||||
(hipDoubleComplex **)&Bkn[0], ldb,
|
||||
(hipDoubleComplex *) &beta_p[0],
|
||||
(hipDoubleComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
// std::cout << " hipblas return code " <<(int)err<<std::endl;
|
||||
#else
|
||||
auto err = hipblasZgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipblasDoubleComplex *) &alpha_p[0],
|
||||
(hipblasDoubleComplex **)&Amk[0], lda,
|
||||
(hipblasDoubleComplex **)&Bkn[0], ldb,
|
||||
(hipblasDoubleComplex *) &beta_p[0],
|
||||
(hipblasDoubleComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
#endif
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -483,11 +571,11 @@ public:
|
||||
lda = k;
|
||||
if(OpB!=GridBLAS_OP_N)
|
||||
ldb = n;
|
||||
static deviceVector<ComplexF> alpha_p(1);
|
||||
static deviceVector<ComplexF> beta_p(1);
|
||||
// can prestore the 1 and the zero on device
|
||||
acceleratorCopyToDevice((void *)&alpha,(void *)&alpha_p[0],sizeof(ComplexF));
|
||||
acceleratorCopyToDevice((void *)&beta ,(void *)&beta_p[0],sizeof(ComplexF));
|
||||
// Cached device constants: copy only on value change (see GridBLASDeviceConstant)
|
||||
static GridBLASDeviceConstant<ComplexF> alpha_c;
|
||||
static GridBLASDeviceConstant<ComplexF> beta_c;
|
||||
ComplexF *alpha_p = alpha_c.put(alpha);
|
||||
ComplexF *beta_p = beta_c.put(beta);
|
||||
RealD t0=usecond();
|
||||
|
||||
GRID_ASSERT(Bkn.size()==batchCount);
|
||||
@@ -502,17 +590,31 @@ public:
|
||||
if ( OpB == GridBLAS_OP_N ) hOpB = HIPBLAS_OP_N;
|
||||
if ( OpB == GridBLAS_OP_T ) hOpB = HIPBLAS_OP_T;
|
||||
if ( OpB == GridBLAS_OP_C ) hOpB = HIPBLAS_OP_C;
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasCgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipblasComplex *) &alpha_p[0],
|
||||
(hipblasComplex **)&Amk[0], lda,
|
||||
(hipblasComplex **)&Bkn[0], ldb,
|
||||
(hipblasComplex *) &beta_p[0],
|
||||
(hipblasComplex **)&Cmn[0], ldc,
|
||||
(hipComplex *) &alpha_p[0],
|
||||
(hipComplex **)&Amk[0], lda,
|
||||
(hipComplex **)&Bkn[0], ldb,
|
||||
(hipComplex *) &beta_p[0],
|
||||
(hipComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
#else
|
||||
auto err = hipblasCgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipblasComplex *) &alpha_p[0],
|
||||
(hipblasComplex **)&Amk[0], lda,
|
||||
(hipblasComplex **)&Bkn[0], ldb,
|
||||
(hipblasComplex *) &beta_p[0],
|
||||
(hipblasComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
|
||||
#endif
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -549,6 +651,7 @@ public:
|
||||
(void **)&Cmn[0], CUDA_C_32F, ldc,
|
||||
batchCount, compute_precision, CUBLAS_GEMM_DEFAULT);
|
||||
}
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==CUBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
@@ -665,6 +768,456 @@ public:
|
||||
RealD bytes = 1.0*sizeof(ComplexF)*(m*k+k*n+m*n)*batchCount;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
// Explicit-leading-dimension complex single GEMM.
|
||||
//
|
||||
// A,B,C may be SLICES of larger parent allocations: lda/ldb/ldc are the
|
||||
// PARENT strides (>= the compact values the ten-argument overload derives).
|
||||
// Motivating use: software split-K for tiny-output/huge-K dense multiplies
|
||||
// (arXiv:2409.03904 fig 11; cf MultiRHSBlockCGLinalg) -- batch over K-chunks
|
||||
// of a dense slab by pointer offset j*Kchunk with lda = the full K extent,
|
||||
// then reduce the partial C's. Backends pass lda straight through; only
|
||||
// the compact overload invented them.
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
void gemmBatched(GridBLASOperation_t OpA,
|
||||
GridBLASOperation_t OpB,
|
||||
int m,int n, int k,
|
||||
ComplexF alpha,
|
||||
deviceVector<ComplexF*> &Amk, int lda,
|
||||
deviceVector<ComplexF*> &Bkn, int ldb,
|
||||
ComplexF beta,
|
||||
deviceVector<ComplexF*> &Cmn, int ldc,
|
||||
GridBLASPrecision_t precision = GridBLAS_PRECISION_DEFAULT)
|
||||
{
|
||||
RealD t2=usecond();
|
||||
int32_t batchCount = Amk.size();
|
||||
|
||||
GRID_ASSERT( lda >= ((OpA==GridBLAS_OP_N) ? m : k) );
|
||||
GRID_ASSERT( ldb >= ((OpB==GridBLAS_OP_N) ? k : n) );
|
||||
GRID_ASSERT( ldc >= m );
|
||||
|
||||
// Cached device constants: copy only on value change (see GridBLASDeviceConstant)
|
||||
static GridBLASDeviceConstant<ComplexF> alpha_c;
|
||||
static GridBLASDeviceConstant<ComplexF> beta_c;
|
||||
ComplexF *alpha_p = alpha_c.put(alpha);
|
||||
ComplexF *beta_p = beta_c.put(beta);
|
||||
RealD t0=usecond();
|
||||
|
||||
GRID_ASSERT(Bkn.size()==batchCount);
|
||||
GRID_ASSERT(Cmn.size()==batchCount);
|
||||
#ifdef GRID_HIP
|
||||
GRID_ASSERT(precision == GridBLAS_PRECISION_DEFAULT);
|
||||
hipblasOperation_t hOpA;
|
||||
hipblasOperation_t hOpB;
|
||||
if ( OpA == GridBLAS_OP_N ) hOpA = HIPBLAS_OP_N;
|
||||
if ( OpA == GridBLAS_OP_T ) hOpA = HIPBLAS_OP_T;
|
||||
if ( OpA == GridBLAS_OP_C ) hOpA = HIPBLAS_OP_C;
|
||||
if ( OpB == GridBLAS_OP_N ) hOpB = HIPBLAS_OP_N;
|
||||
if ( OpB == GridBLAS_OP_T ) hOpB = HIPBLAS_OP_T;
|
||||
if ( OpB == GridBLAS_OP_C ) hOpB = HIPBLAS_OP_C;
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasCgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipComplex *) &alpha_p[0],
|
||||
(hipComplex **)&Amk[0], lda,
|
||||
(hipComplex **)&Bkn[0], ldb,
|
||||
(hipComplex *) &beta_p[0],
|
||||
(hipComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
#else
|
||||
auto err = hipblasCgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipblasComplex *) &alpha_p[0],
|
||||
(hipblasComplex **)&Amk[0], lda,
|
||||
(hipblasComplex **)&Bkn[0], ldb,
|
||||
(hipblasComplex *) &beta_p[0],
|
||||
(hipblasComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
#endif
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
cublasOperation_t hOpA;
|
||||
cublasOperation_t hOpB;
|
||||
if ( OpA == GridBLAS_OP_N ) hOpA = CUBLAS_OP_N;
|
||||
if ( OpA == GridBLAS_OP_T ) hOpA = CUBLAS_OP_T;
|
||||
if ( OpA == GridBLAS_OP_C ) hOpA = CUBLAS_OP_C;
|
||||
if ( OpB == GridBLAS_OP_N ) hOpB = CUBLAS_OP_N;
|
||||
if ( OpB == GridBLAS_OP_T ) hOpB = CUBLAS_OP_T;
|
||||
if ( OpB == GridBLAS_OP_C ) hOpB = CUBLAS_OP_C;
|
||||
cublasStatus_t err;
|
||||
if (precision == GridBLAS_PRECISION_DEFAULT) {
|
||||
err = cublasCgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(cuComplex *) &alpha_p[0],
|
||||
(cuComplex **)&Amk[0], lda,
|
||||
(cuComplex **)&Bkn[0], ldb,
|
||||
(cuComplex *) &beta_p[0],
|
||||
(cuComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
} else {
|
||||
cublasComputeType_t compute_precision = toDataType(precision);
|
||||
err = cublasGemmBatchedEx(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(void *) &alpha_p[0],
|
||||
(void **)&Amk[0], CUDA_C_32F, lda,
|
||||
(void **)&Bkn[0], CUDA_C_32F, ldb,
|
||||
(void *) &beta_p[0],
|
||||
(void **)&Cmn[0], CUDA_C_32F, ldc,
|
||||
batchCount, compute_precision, CUBLAS_GEMM_DEFAULT);
|
||||
}
|
||||
GRID_ASSERT(err==CUBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
GRID_ASSERT(precision == GridBLAS_PRECISION_DEFAULT);
|
||||
int64_t m64=m;
|
||||
int64_t n64=n;
|
||||
int64_t k64=k;
|
||||
int64_t lda64=lda;
|
||||
int64_t ldb64=ldb;
|
||||
int64_t ldc64=ldc;
|
||||
int64_t batchCount64=batchCount;
|
||||
|
||||
oneapi::mkl::transpose iOpA;
|
||||
oneapi::mkl::transpose iOpB;
|
||||
|
||||
if ( OpA == GridBLAS_OP_N ) iOpA = oneapi::mkl::transpose::N;
|
||||
if ( OpA == GridBLAS_OP_T ) iOpA = oneapi::mkl::transpose::T;
|
||||
if ( OpA == GridBLAS_OP_C ) iOpA = oneapi::mkl::transpose::C;
|
||||
if ( OpB == GridBLAS_OP_N ) iOpB = oneapi::mkl::transpose::N;
|
||||
if ( OpB == GridBLAS_OP_T ) iOpB = oneapi::mkl::transpose::T;
|
||||
if ( OpB == GridBLAS_OP_C ) iOpB = oneapi::mkl::transpose::C;
|
||||
|
||||
oneapi::mkl::blas::column_major::gemm_batch(*gridblasHandle,
|
||||
&iOpA,
|
||||
&iOpB,
|
||||
&m64,&n64,&k64,
|
||||
(ComplexF *) &alpha_p[0],
|
||||
(const ComplexF **)&Amk[0], (const int64_t *)&lda64,
|
||||
(const ComplexF **)&Bkn[0], (const int64_t *)&ldb64,
|
||||
(ComplexF *) &beta_p[0],
|
||||
(ComplexF **)&Cmn[0], (const int64_t *)&ldc64,
|
||||
(int64_t)1,&batchCount64,std::vector<sycl::event>());
|
||||
synchronise();
|
||||
#endif
|
||||
#if !defined(GRID_SYCL) && !defined(GRID_CUDA) && !defined(GRID_HIP)
|
||||
GRID_ASSERT(precision == GridBLAS_PRECISION_DEFAULT);
|
||||
// Reference implementation: Eigen with explicit outer stride
|
||||
typedef Eigen::Map<Eigen::MatrixXcf,0,Eigen::OuterStride<> > eMat;
|
||||
if ( (OpA == GridBLAS_OP_N ) && (OpB == GridBLAS_OP_N) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],m,k,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],k,n,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk * eBkn ;
|
||||
else
|
||||
eCmn = alpha * eAmk * eBkn ;
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_C ) && (OpB == GridBLAS_OP_N) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],k,n,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk.adjoint() * eBkn ;
|
||||
else
|
||||
eCmn = alpha * eAmk.adjoint() * eBkn ;
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_T ) && (OpB == GridBLAS_OP_N) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],k,n,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk.transpose() * eBkn ;
|
||||
else
|
||||
eCmn = alpha * eAmk.transpose() * eBkn ;
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_N ) && (OpB == GridBLAS_OP_C) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],m,k,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk * eBkn.adjoint() ;
|
||||
else
|
||||
eCmn = alpha * eAmk * eBkn.adjoint() ;
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_N ) && (OpB == GridBLAS_OP_T) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],m,k,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk * eBkn.transpose() ;
|
||||
else
|
||||
eCmn = alpha * eAmk * eBkn.transpose() ;
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_C ) && (OpB == GridBLAS_OP_C) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk.adjoint() * eBkn.adjoint() ;
|
||||
else
|
||||
eCmn = alpha * eAmk.adjoint() * eBkn.adjoint() ;
|
||||
} );
|
||||
} else if ( (OpA == GridBLAS_OP_T ) && (OpB == GridBLAS_OP_T) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
eCmn = beta * eCmn + alpha * eAmk.transpose() * eBkn.transpose() ;
|
||||
else
|
||||
eCmn = alpha * eAmk.transpose() * eBkn.transpose() ;
|
||||
} );
|
||||
} else {
|
||||
assert(0);
|
||||
}
|
||||
#endif
|
||||
RealD t1=usecond();
|
||||
RealD flops = 8.0*m*n*k*batchCount;
|
||||
RealD bytes = 1.0*sizeof(ComplexF)*(m*k+k*n+m*n)*batchCount;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
// Explicit-leading-dimension complex double GEMM. Mirror of the ComplexF
|
||||
// overload above; motivating use is the fp64 distributed recursive Schur
|
||||
// inversion (RecursiveSchurInverse), whose operands are column windows of
|
||||
// larger row-slab allocations.
|
||||
///////////////////////////////////////////////////////////////////////////////////
|
||||
void gemmBatched(GridBLASOperation_t OpA,
|
||||
GridBLASOperation_t OpB,
|
||||
int m,int n, int k,
|
||||
ComplexD alpha,
|
||||
deviceVector<ComplexD*> &Amk, int lda,
|
||||
deviceVector<ComplexD*> &Bkn, int ldb,
|
||||
ComplexD beta,
|
||||
deviceVector<ComplexD*> &Cmn, int ldc)
|
||||
{
|
||||
RealD t2=usecond();
|
||||
int32_t batchCount = Amk.size();
|
||||
|
||||
GRID_ASSERT( lda >= ((OpA==GridBLAS_OP_N) ? m : k) );
|
||||
GRID_ASSERT( ldb >= ((OpB==GridBLAS_OP_N) ? k : n) );
|
||||
GRID_ASSERT( ldc >= m );
|
||||
|
||||
// Cached device constants: copy only on value change (see GridBLASDeviceConstant)
|
||||
static GridBLASDeviceConstant<ComplexD> alpha_c;
|
||||
static GridBLASDeviceConstant<ComplexD> beta_c;
|
||||
ComplexD *alpha_p = alpha_c.put(alpha);
|
||||
ComplexD *beta_p = beta_c.put(beta);
|
||||
RealD t0=usecond();
|
||||
|
||||
GRID_ASSERT(Bkn.size()==batchCount);
|
||||
GRID_ASSERT(Cmn.size()==batchCount);
|
||||
#ifdef GRID_HIP
|
||||
hipblasOperation_t hOpA;
|
||||
hipblasOperation_t hOpB;
|
||||
if ( OpA == GridBLAS_OP_N ) hOpA = HIPBLAS_OP_N;
|
||||
if ( OpA == GridBLAS_OP_T ) hOpA = HIPBLAS_OP_T;
|
||||
if ( OpA == GridBLAS_OP_C ) hOpA = HIPBLAS_OP_C;
|
||||
if ( OpB == GridBLAS_OP_N ) hOpB = HIPBLAS_OP_N;
|
||||
if ( OpB == GridBLAS_OP_T ) hOpB = HIPBLAS_OP_T;
|
||||
if ( OpB == GridBLAS_OP_C ) hOpB = HIPBLAS_OP_C;
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasZgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipDoubleComplex *) &alpha_p[0],
|
||||
(hipDoubleComplex **)&Amk[0], lda,
|
||||
(hipDoubleComplex **)&Bkn[0], ldb,
|
||||
(hipDoubleComplex *) &beta_p[0],
|
||||
(hipDoubleComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
#else
|
||||
auto err = hipblasZgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(hipblasDoubleComplex *) &alpha_p[0],
|
||||
(hipblasDoubleComplex **)&Amk[0], lda,
|
||||
(hipblasDoubleComplex **)&Bkn[0], ldb,
|
||||
(hipblasDoubleComplex *) &beta_p[0],
|
||||
(hipblasDoubleComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
#endif
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
cublasOperation_t hOpA;
|
||||
cublasOperation_t hOpB;
|
||||
if ( OpA == GridBLAS_OP_N ) hOpA = CUBLAS_OP_N;
|
||||
if ( OpA == GridBLAS_OP_T ) hOpA = CUBLAS_OP_T;
|
||||
if ( OpA == GridBLAS_OP_C ) hOpA = CUBLAS_OP_C;
|
||||
if ( OpB == GridBLAS_OP_N ) hOpB = CUBLAS_OP_N;
|
||||
if ( OpB == GridBLAS_OP_T ) hOpB = CUBLAS_OP_T;
|
||||
if ( OpB == GridBLAS_OP_C ) hOpB = CUBLAS_OP_C;
|
||||
auto err = cublasZgemmBatched(gridblasHandle,
|
||||
hOpA,
|
||||
hOpB,
|
||||
m,n,k,
|
||||
(cuDoubleComplex *) &alpha_p[0],
|
||||
(cuDoubleComplex **)&Amk[0], lda,
|
||||
(cuDoubleComplex **)&Bkn[0], ldb,
|
||||
(cuDoubleComplex *) &beta_p[0],
|
||||
(cuDoubleComplex **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
GRID_ASSERT(err==CUBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
int64_t m64=m;
|
||||
int64_t n64=n;
|
||||
int64_t k64=k;
|
||||
int64_t lda64=lda;
|
||||
int64_t ldb64=ldb;
|
||||
int64_t ldc64=ldc;
|
||||
int64_t batchCount64=batchCount;
|
||||
|
||||
oneapi::mkl::transpose iOpA;
|
||||
oneapi::mkl::transpose iOpB;
|
||||
|
||||
if ( OpA == GridBLAS_OP_N ) iOpA = oneapi::mkl::transpose::N;
|
||||
if ( OpA == GridBLAS_OP_T ) iOpA = oneapi::mkl::transpose::T;
|
||||
if ( OpA == GridBLAS_OP_C ) iOpA = oneapi::mkl::transpose::C;
|
||||
if ( OpB == GridBLAS_OP_N ) iOpB = oneapi::mkl::transpose::N;
|
||||
if ( OpB == GridBLAS_OP_T ) iOpB = oneapi::mkl::transpose::T;
|
||||
if ( OpB == GridBLAS_OP_C ) iOpB = oneapi::mkl::transpose::C;
|
||||
|
||||
oneapi::mkl::blas::column_major::gemm_batch(*gridblasHandle,
|
||||
&iOpA,
|
||||
&iOpB,
|
||||
&m64,&n64,&k64,
|
||||
(ComplexD *) &alpha_p[0],
|
||||
(const ComplexD **)&Amk[0], (const int64_t *)&lda64,
|
||||
(const ComplexD **)&Bkn[0], (const int64_t *)&ldb64,
|
||||
(ComplexD *) &beta_p[0],
|
||||
(ComplexD **)&Cmn[0], (const int64_t *)&ldc64,
|
||||
(int64_t)1,&batchCount64,std::vector<sycl::event>());
|
||||
synchronise();
|
||||
#endif
|
||||
#if !defined(GRID_SYCL) && !defined(GRID_CUDA) && !defined(GRID_HIP)
|
||||
// Reference implementation: Eigen with explicit outer stride
|
||||
typedef Eigen::Map<Eigen::MatrixXcd,0,Eigen::OuterStride<> > eMat;
|
||||
if ( (OpA == GridBLAS_OP_N ) && (OpB == GridBLAS_OP_N) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],m,k,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],k,n,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk * eBkn;
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk * eBkn;
|
||||
}
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_C ) && (OpB == GridBLAS_OP_N) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],k,n,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk.adjoint() * eBkn;
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk.adjoint() * eBkn;
|
||||
}
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_T ) && (OpB == GridBLAS_OP_N) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],k,n,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk.transpose() * eBkn;
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk.transpose() * eBkn;
|
||||
}
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_N ) && (OpB == GridBLAS_OP_C) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],m,k,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk * eBkn.adjoint();
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk * eBkn.adjoint();
|
||||
}
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_N ) && (OpB == GridBLAS_OP_T) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],m,k,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk * eBkn.transpose();
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk * eBkn.transpose();
|
||||
}
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_C ) && (OpB == GridBLAS_OP_C) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk.adjoint() * eBkn.adjoint();
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk.adjoint() * eBkn.adjoint();
|
||||
}
|
||||
});
|
||||
} else if ( (OpA == GridBLAS_OP_T ) && (OpB == GridBLAS_OP_T) ) {
|
||||
thread_for (p, batchCount, {
|
||||
eMat eAmk(Amk[p],k,m,Eigen::OuterStride<>(lda));
|
||||
eMat eBkn(Bkn[p],n,k,Eigen::OuterStride<>(ldb));
|
||||
eMat eCmn(Cmn[p],m,n,Eigen::OuterStride<>(ldc));
|
||||
if (std::abs(beta) != 0.0)
|
||||
{
|
||||
eCmn = beta * eCmn + alpha * eAmk.transpose() * eBkn.transpose();
|
||||
}
|
||||
else
|
||||
{
|
||||
eCmn = alpha * eAmk.transpose() * eBkn.transpose();
|
||||
}
|
||||
});
|
||||
} else {
|
||||
assert(0);
|
||||
}
|
||||
#endif
|
||||
RealD t1=usecond();
|
||||
RealD flops = 8.0*m*n*k*batchCount;
|
||||
RealD bytes = 1.0*sizeof(ComplexD)*(m*k+k*n+m*n)*batchCount;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Single precision real GEMM
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
@@ -691,11 +1244,11 @@ public:
|
||||
lda = k;
|
||||
if(OpB!=GridBLAS_OP_N)
|
||||
ldb = n;
|
||||
static deviceVector<RealF> alpha_p(1);
|
||||
static deviceVector<RealF> beta_p(1);
|
||||
// can prestore the 1 and the zero on device
|
||||
acceleratorCopyToDevice((void *)&alpha,(void *)&alpha_p[0],sizeof(RealF));
|
||||
acceleratorCopyToDevice((void *)&beta ,(void *)&beta_p[0],sizeof(RealF));
|
||||
// Cached device constants: copy only on value change (see GridBLASDeviceConstant)
|
||||
static GridBLASDeviceConstant<RealF> alpha_c;
|
||||
static GridBLASDeviceConstant<RealF> beta_c;
|
||||
RealF *alpha_p = alpha_c.put(alpha);
|
||||
RealF *beta_p = beta_c.put(beta);
|
||||
RealD t0=usecond();
|
||||
|
||||
GRID_ASSERT(Bkn.size()==batchCount);
|
||||
@@ -719,6 +1272,7 @@ public:
|
||||
(float *) &beta_p[0],
|
||||
(float **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -851,11 +1405,11 @@ public:
|
||||
if(OpB!=GridBLAS_OP_N)
|
||||
ldb = n;
|
||||
|
||||
static deviceVector<RealD> alpha_p(1);
|
||||
static deviceVector<RealD> beta_p(1);
|
||||
// can prestore the 1 and the zero on device
|
||||
acceleratorCopyToDevice((void *)&alpha,(void *)&alpha_p[0],sizeof(RealD));
|
||||
acceleratorCopyToDevice((void *)&beta ,(void *)&beta_p[0],sizeof(RealD));
|
||||
// Cached device constants: copy only on value change (see GridBLASDeviceConstant)
|
||||
static GridBLASDeviceConstant<RealD> alpha_c;
|
||||
static GridBLASDeviceConstant<RealD> beta_c;
|
||||
RealD *alpha_p = alpha_c.put(alpha);
|
||||
RealD *beta_p = beta_c.put(beta);
|
||||
RealD t0=usecond();
|
||||
|
||||
GRID_ASSERT(Bkn.size()==batchCount);
|
||||
@@ -879,6 +1433,7 @@ public:
|
||||
(double *) &beta_p[0],
|
||||
(double **)&Cmn[0], ldc,
|
||||
batchCount);
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -1093,11 +1648,20 @@ public:
|
||||
GRID_ASSERT(info.size()==batchCount);
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasZgetrfBatched(gridblasHandle,(int)n,
|
||||
(hipblasDoubleComplex **)&Ann[0], (int)n,
|
||||
(hipDoubleComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#else
|
||||
auto err = hipblasZgetrfBatched(gridblasHandle,(int)n,
|
||||
(hipblasDoubleComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#endif
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -1123,11 +1687,21 @@ public:
|
||||
GRID_ASSERT(info.size()==batchCount);
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasCgetrfBatched(gridblasHandle,(int)n,
|
||||
(hipblasComplex **)&Ann[0], (int)n,
|
||||
(hipComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#else
|
||||
auto err = hipblasCgetrfBatched(gridblasHandle,(int)n,
|
||||
(hipblasComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#endif
|
||||
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -1200,12 +1774,23 @@ public:
|
||||
GRID_ASSERT(Cnn.size()==batchCount);
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasZgetriBatched(gridblasHandle,(int)n,
|
||||
(hipblasDoubleComplex **)&Ann[0], (int)n,
|
||||
(hipDoubleComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(hipblasDoubleComplex **)&Cnn[0], (int)n,
|
||||
(hipDoubleComplex **)&Cnn[0], (int)n,
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#else
|
||||
auto err = hipblasZgetriBatched(gridblasHandle,(int)n,
|
||||
(hipblasDoubleComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(hipblasDoubleComplex **)&Cnn[0], (int)n,
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
|
||||
#endif
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
@@ -1234,12 +1819,22 @@ public:
|
||||
GRID_ASSERT(Cnn.size()==batchCount);
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#if defined(HIP_VERSION_MAJOR) && (HIP_VERSION_MAJOR >=7)
|
||||
auto err = hipblasCgetriBatched(gridblasHandle,(int)n,
|
||||
(hipblasComplex **)&Ann[0], (int)n,
|
||||
(hipComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(hipblasComplex **)&Cnn[0], (int)n,
|
||||
(hipComplex **)&Cnn[0], (int)n,
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#else
|
||||
auto err = hipblasCgetriBatched(gridblasHandle,(int)n,
|
||||
(hipblasComplex **)&Ann[0], (int)n,
|
||||
(int*) &ipiv[0],
|
||||
(hipblasComplex **)&Cnn[0], (int)n,
|
||||
(int*) &info[0],
|
||||
(int)batchCount);
|
||||
#endif
|
||||
// std::cout << " hipblas return code " <<(int)err<<" "<<__LINE__<<std::endl;
|
||||
GRID_ASSERT(err==HIPBLAS_STATUS_SUCCESS);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: BatchedInverse.h
|
||||
|
||||
Copyright (C) 2026
|
||||
|
||||
Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#pragma once
|
||||
|
||||
#include <Grid/algorithms/blas/BatchedBlas.h>
|
||||
|
||||
#ifdef GRID_HIP
|
||||
#include <rocsolver/rocsolver.h>
|
||||
#endif
|
||||
// GRID_CUDA: batched LU inversion lives in cuBLAS (getrfBatched/getriBatched);
|
||||
// cublas_v2.h already included via BatchedBlas.h.
|
||||
// GRID_SYCL: oneapi/mkl.hpp already included via BatchedBlas.h (lapack::getrf/getri).
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// GridBLASInverse: cross-platform batched dense matrix inversion.
|
||||
//
|
||||
// HIGH LEVEL contract (deliberately NOT a getrf/getrs interface): invert a
|
||||
// batch of dense N x N matrices IN PLACE,
|
||||
//
|
||||
// A[i] <- A[i]^{-1} i = 0 .. batchCount-1
|
||||
//
|
||||
// Layout: column major, lda = N, contiguous per batch element; pointer list
|
||||
// exactly as GridBLAS::gemmBatched (deviceVector<T*> of device pointers).
|
||||
// Each backend chooses HOW:
|
||||
// HIP : rocSOLVER getrf_batched + getri_batched
|
||||
// CUDA : cuBLAS getrfBatched + getriBatched (out-of-place getri; workspace
|
||||
// hidden here, result copied back so the surface stays in-place)
|
||||
// SYCL : oneMKL LAPACK getrf + getri per batch element (USM, in-order queue)
|
||||
// CPU : Eigen PartialPivLU (the correctness oracle for all of the above)
|
||||
//
|
||||
// The int32 vendor-batched entry points bound N < 2^31 (asserted); the huge
|
||||
// single-matrix ILP64 path (getrf_64 + blocked identity-getrs harvest, proven
|
||||
// in the dense coarse-coarse setup at N=69120) migrates here as a batch==1
|
||||
// large-N dispatch in a follow-up -- the recursive Schur leaves are the
|
||||
// batched consumers this surface is shaped for.
|
||||
//
|
||||
// NB GPU-backend call signatures are written to vendor documentation but the
|
||||
// air-gapped development loop compiles only the CPU/Eigen path; verify the
|
||||
// rocSOLVER/cuBLAS/oneMKL calls against headers on first device compile.
|
||||
// Semantics are locked by the CPU unit test (Test_batched_blas).
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
class GridBLASInverse {
|
||||
public:
|
||||
|
||||
#ifdef GRID_HIP
|
||||
// rocSOLVER runs on a rocblas_handle (distinct type from hipblasHandle_t)
|
||||
static rocblas_handle & Handle(void) {
|
||||
static rocblas_handle h;
|
||||
static int init = 0;
|
||||
if ( !init ) {
|
||||
auto st = rocblas_create_handle(&h);
|
||||
GRID_ASSERT(st == rocblas_status_success);
|
||||
init = 1;
|
||||
}
|
||||
return h;
|
||||
}
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
// cuBLAS batched LU shares the GridBLAS handle
|
||||
static cublasHandle_t & Handle(void) {
|
||||
GridBLAS::Init();
|
||||
return GridBLAS::gridblasHandle;
|
||||
}
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
static sycl::queue * & Handle(void) {
|
||||
GridBLAS::Init();
|
||||
return GridBLAS::gridblasHandle;
|
||||
}
|
||||
#endif
|
||||
|
||||
GridBLASInverse() {};
|
||||
~GridBLASInverse() {};
|
||||
|
||||
void inverseBatched(int64_t N, deviceVector<ComplexF*> &Amat)
|
||||
{
|
||||
int32_t batchCount = Amat.size();
|
||||
GRID_ASSERT(batchCount > 0);
|
||||
|
||||
#ifdef GRID_HIP
|
||||
GRID_ASSERT( N < 2147483647L );
|
||||
rocblas_int n = (rocblas_int)N;
|
||||
rocblas_int lda = (rocblas_int)N;
|
||||
|
||||
deviceVector<rocblas_int> ipiv((uint64_t)batchCount*N);
|
||||
deviceVector<rocblas_int> info(batchCount);
|
||||
|
||||
auto st1 = rocsolver_cgetrf_batched(Handle(), n, n,
|
||||
(rocblas_float_complex *const *)&Amat[0], lda,
|
||||
&ipiv[0], (rocblas_stride)N,
|
||||
&info[0], batchCount);
|
||||
GRID_ASSERT(st1 == rocblas_status_success);
|
||||
auto st2 = rocsolver_cgetri_batched(Handle(), n,
|
||||
(rocblas_float_complex *const *)&Amat[0], lda,
|
||||
&ipiv[0], (rocblas_stride)N,
|
||||
&info[0], batchCount);
|
||||
GRID_ASSERT(st2 == rocblas_status_success);
|
||||
accelerator_barrier();
|
||||
std::vector<rocblas_int> info_h(batchCount);
|
||||
acceleratorCopyFromDevice(&info[0],&info_h[0],batchCount*sizeof(rocblas_int));
|
||||
for(int i=0;i<batchCount;i++) GRID_ASSERT(info_h[i]==0); // singular pivot => abort loudly
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
GRID_ASSERT( N < 2147483647L );
|
||||
int n = (int)N;
|
||||
|
||||
deviceVector<int> ipiv((uint64_t)batchCount*N);
|
||||
deviceVector<int> info(batchCount);
|
||||
|
||||
auto st1 = cublasCgetrfBatched(Handle(), n,
|
||||
(cuComplex **)&Amat[0], n,
|
||||
&ipiv[0], &info[0], batchCount);
|
||||
GRID_ASSERT(st1 == CUBLAS_STATUS_SUCCESS);
|
||||
|
||||
// getri is OUT of place: hidden workspace keeps the surface in-place
|
||||
deviceVector<ComplexF> work((uint64_t)batchCount*N*N);
|
||||
deviceVector<ComplexF*> Cptr(batchCount);
|
||||
std::vector<ComplexF*> Cptr_h(batchCount);
|
||||
std::vector<ComplexF*> Aptr_h(batchCount);
|
||||
for(int i=0;i<batchCount;i++) Cptr_h[i] = &work[(uint64_t)i*N*N];
|
||||
acceleratorCopyToDevice(&Cptr_h[0],&Cptr[0],batchCount*sizeof(ComplexF*));
|
||||
acceleratorCopyFromDevice(&Amat[0],&Aptr_h[0],batchCount*sizeof(ComplexF*));
|
||||
|
||||
auto st2 = cublasCgetriBatched(Handle(), n,
|
||||
(const cuComplex *const *)&Amat[0], n,
|
||||
&ipiv[0],
|
||||
(cuComplex **)&Cptr[0], n,
|
||||
&info[0], batchCount);
|
||||
GRID_ASSERT(st2 == CUBLAS_STATUS_SUCCESS);
|
||||
accelerator_barrier();
|
||||
std::vector<int> info_h(batchCount);
|
||||
acceleratorCopyFromDevice(&info[0],&info_h[0],batchCount*sizeof(int));
|
||||
for(int i=0;i<batchCount;i++) GRID_ASSERT(info_h[i]==0);
|
||||
for(int i=0;i<batchCount;i++)
|
||||
acceleratorCopyDeviceToDevice(Cptr_h[i],Aptr_h[i],(uint64_t)N*N*sizeof(ComplexF));
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
// Per-element oneMKL LAPACK on the in-order queue; group API optimisation later.
|
||||
sycl::queue *q = Handle();
|
||||
std::vector<ComplexF*> Aptr_h(batchCount);
|
||||
acceleratorCopyFromDevice(&Amat[0],&Aptr_h[0],batchCount*sizeof(ComplexF*));
|
||||
|
||||
int64_t lwf = oneapi::mkl::lapack::getrf_scratchpad_size<std::complex<float> >(*q,N,N,N);
|
||||
int64_t lwi = oneapi::mkl::lapack::getri_scratchpad_size<std::complex<float> >(*q,N,N);
|
||||
deviceVector<ComplexF> scratchf(lwf);
|
||||
deviceVector<ComplexF> scratchi(lwi);
|
||||
deviceVector<int64_t> ipiv(N);
|
||||
for(int i=0;i<batchCount;i++){
|
||||
oneapi::mkl::lapack::getrf(*q,N,N,(std::complex<float>*)Aptr_h[i],N,&ipiv[0],
|
||||
(std::complex<float>*)&scratchf[0],lwf);
|
||||
oneapi::mkl::lapack::getri(*q,N, (std::complex<float>*)Aptr_h[i],N,&ipiv[0],
|
||||
(std::complex<float>*)&scratchi[0],lwi);
|
||||
}
|
||||
q->wait();
|
||||
#endif
|
||||
#if !defined(GRID_SYCL) && !defined(GRID_CUDA) && !defined(GRID_HIP)
|
||||
// Reference implementation; the oracle the unit test locks semantics with.
|
||||
thread_for (p, batchCount, {
|
||||
Eigen::Map<Eigen::MatrixXcf> eA(Amat[p],N,N);
|
||||
Eigen::PartialPivLU<Eigen::MatrixXcf> lu(eA);
|
||||
eA = lu.inverse();
|
||||
});
|
||||
#endif
|
||||
}
|
||||
|
||||
void inverseBatched(int64_t N, deviceVector<ComplexD*> &Amat)
|
||||
{
|
||||
int32_t batchCount = Amat.size();
|
||||
GRID_ASSERT(batchCount > 0);
|
||||
|
||||
#ifdef GRID_HIP
|
||||
GRID_ASSERT( N < 2147483647L );
|
||||
rocblas_int n = (rocblas_int)N;
|
||||
rocblas_int lda = (rocblas_int)N;
|
||||
|
||||
deviceVector<rocblas_int> ipiv((uint64_t)batchCount*N);
|
||||
deviceVector<rocblas_int> info(batchCount);
|
||||
|
||||
auto st1 = rocsolver_zgetrf_batched(Handle(), n, n,
|
||||
(rocblas_double_complex *const *)&Amat[0], lda,
|
||||
&ipiv[0], (rocblas_stride)N,
|
||||
&info[0], batchCount);
|
||||
GRID_ASSERT(st1 == rocblas_status_success);
|
||||
auto st2 = rocsolver_zgetri_batched(Handle(), n,
|
||||
(rocblas_double_complex *const *)&Amat[0], lda,
|
||||
&ipiv[0], (rocblas_stride)N,
|
||||
&info[0], batchCount);
|
||||
GRID_ASSERT(st2 == rocblas_status_success);
|
||||
accelerator_barrier();
|
||||
std::vector<rocblas_int> info_h(batchCount);
|
||||
acceleratorCopyFromDevice(&info[0],&info_h[0],batchCount*sizeof(rocblas_int));
|
||||
for(int i=0;i<batchCount;i++) GRID_ASSERT(info_h[i]==0);
|
||||
#endif
|
||||
#ifdef GRID_CUDA
|
||||
GRID_ASSERT( N < 2147483647L );
|
||||
int n = (int)N;
|
||||
|
||||
deviceVector<int> ipiv((uint64_t)batchCount*N);
|
||||
deviceVector<int> info(batchCount);
|
||||
|
||||
auto st1 = cublasZgetrfBatched(Handle(), n,
|
||||
(cuDoubleComplex **)&Amat[0], n,
|
||||
&ipiv[0], &info[0], batchCount);
|
||||
GRID_ASSERT(st1 == CUBLAS_STATUS_SUCCESS);
|
||||
|
||||
deviceVector<ComplexD> work((uint64_t)batchCount*N*N);
|
||||
deviceVector<ComplexD*> Cptr(batchCount);
|
||||
std::vector<ComplexD*> Cptr_h(batchCount);
|
||||
std::vector<ComplexD*> Aptr_h(batchCount);
|
||||
for(int i=0;i<batchCount;i++) Cptr_h[i] = &work[(uint64_t)i*N*N];
|
||||
acceleratorCopyToDevice(&Cptr_h[0],&Cptr[0],batchCount*sizeof(ComplexD*));
|
||||
acceleratorCopyFromDevice(&Amat[0],&Aptr_h[0],batchCount*sizeof(ComplexD*));
|
||||
|
||||
auto st2 = cublasZgetriBatched(Handle(), n,
|
||||
(const cuDoubleComplex *const *)&Amat[0], n,
|
||||
&ipiv[0],
|
||||
(cuDoubleComplex **)&Cptr[0], n,
|
||||
&info[0], batchCount);
|
||||
GRID_ASSERT(st2 == CUBLAS_STATUS_SUCCESS);
|
||||
accelerator_barrier();
|
||||
std::vector<int> info_h(batchCount);
|
||||
acceleratorCopyFromDevice(&info[0],&info_h[0],batchCount*sizeof(int));
|
||||
for(int i=0;i<batchCount;i++) GRID_ASSERT(info_h[i]==0);
|
||||
for(int i=0;i<batchCount;i++)
|
||||
acceleratorCopyDeviceToDevice(Cptr_h[i],Aptr_h[i],(uint64_t)N*N*sizeof(ComplexD));
|
||||
#endif
|
||||
#ifdef GRID_SYCL
|
||||
sycl::queue *q = Handle();
|
||||
std::vector<ComplexD*> Aptr_h(batchCount);
|
||||
acceleratorCopyFromDevice(&Amat[0],&Aptr_h[0],batchCount*sizeof(ComplexD*));
|
||||
|
||||
int64_t lwf = oneapi::mkl::lapack::getrf_scratchpad_size<std::complex<double> >(*q,N,N,N);
|
||||
int64_t lwi = oneapi::mkl::lapack::getri_scratchpad_size<std::complex<double> >(*q,N,N);
|
||||
deviceVector<ComplexD> scratchf(lwf);
|
||||
deviceVector<ComplexD> scratchi(lwi);
|
||||
deviceVector<int64_t> ipiv(N);
|
||||
for(int i=0;i<batchCount;i++){
|
||||
oneapi::mkl::lapack::getrf(*q,N,N,(std::complex<double>*)Aptr_h[i],N,&ipiv[0],
|
||||
(std::complex<double>*)&scratchf[0],lwf);
|
||||
oneapi::mkl::lapack::getri(*q,N, (std::complex<double>*)Aptr_h[i],N,&ipiv[0],
|
||||
(std::complex<double>*)&scratchi[0],lwi);
|
||||
}
|
||||
q->wait();
|
||||
#endif
|
||||
#if !defined(GRID_SYCL) && !defined(GRID_CUDA) && !defined(GRID_HIP)
|
||||
thread_for (p, batchCount, {
|
||||
Eigen::Map<Eigen::MatrixXcd> eA(Amat[p],N,N);
|
||||
Eigen::PartialPivLU<Eigen::MatrixXcd> lu(eA);
|
||||
eA = lu.inverse();
|
||||
});
|
||||
#endif
|
||||
}
|
||||
};
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
@@ -228,6 +228,11 @@ public:
|
||||
//
|
||||
void Project(Field &data,std::vector< typename Field::scalar_object > & projected_gdata)
|
||||
{
|
||||
double t_import=0;
|
||||
double t_export=0;
|
||||
double t_gemm =0;
|
||||
double t_allreduce=0;
|
||||
t_import-=usecond();
|
||||
this->ImportVector(data);
|
||||
|
||||
std::vector< typename Field::scalar_object > projected_planes;
|
||||
@@ -243,12 +248,14 @@ public:
|
||||
acceleratorPut(Vd[0],Vh);
|
||||
acceleratorPut(Md[0],Mh);
|
||||
acceleratorPut(Pd[0],Ph);
|
||||
t_import+=usecond();
|
||||
|
||||
GridBLAS BLAS;
|
||||
|
||||
/////////////////////////////////////////
|
||||
// P_im = VMmx . Vxi
|
||||
/////////////////////////////////////////
|
||||
t_gemm-=usecond();
|
||||
BLAS.gemmBatched(GridBLAS_OP_N,GridBLAS_OP_N,
|
||||
words*nt,nmom,nxyz,
|
||||
scalar(1.0),
|
||||
@@ -257,8 +264,11 @@ public:
|
||||
scalar(0.0), // wipe out result
|
||||
Pd);
|
||||
BLAS.synchronise();
|
||||
t_gemm+=usecond();
|
||||
|
||||
t_export-=usecond();
|
||||
ExportMomentumProjection(projected_planes); // resizes
|
||||
t_export+=usecond();
|
||||
|
||||
/////////////////////////////////
|
||||
// Reduce across MPI ranks
|
||||
@@ -275,7 +285,15 @@ public:
|
||||
int st = grid->LocalStarts()[nd-1];
|
||||
projected_gdata[t+st + gt*m] = projected_planes[t+lt*m];
|
||||
}}
|
||||
t_allreduce-=usecond();
|
||||
grid->GlobalSumVector((scalar *)&projected_gdata[0],gt*nmom*words);
|
||||
t_allreduce+=usecond();
|
||||
|
||||
std::cout << GridLogPerformance<<" MomentumProject t_import "<<t_import<<"us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<" MomentumProject t_export "<<t_export<<"us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<" MomentumProject t_gemm "<<t_gemm<<"us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<" MomentumProject t_reduce "<<t_allreduce<<"us"<<std::endl;
|
||||
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
@@ -53,7 +53,22 @@ class TwoLevelCG : public LinearFunction<Field>
|
||||
// Fine operator, Smoother, CoarseSolver
|
||||
LinearOperatorBase<Field> &_FineLinop;
|
||||
LinearFunction<Field> &_Smoother;
|
||||
|
||||
|
||||
GridStopWatch ProjectTimer;
|
||||
GridStopWatch PromoteTimer;
|
||||
GridStopWatch CoarseTimer;
|
||||
GridStopWatch SmoothTimer;
|
||||
GridStopWatch MatrixTimer;
|
||||
GridStopWatch M3Timer;
|
||||
GridStopWatch LinalgTimer;
|
||||
|
||||
int64_t M3Calls;
|
||||
int64_t SmoothCalls;
|
||||
int64_t MatrixCalls;
|
||||
int64_t ProjectCalls;
|
||||
int64_t CoarseCalls;
|
||||
int64_t PromoteCalls;
|
||||
|
||||
// more most opertor functions
|
||||
TwoLevelCG(RealD tol,
|
||||
Integer maxit,
|
||||
@@ -103,12 +118,20 @@ class TwoLevelCG : public LinearFunction<Field>
|
||||
RealD tn;
|
||||
|
||||
GridStopWatch HDCGTimer;
|
||||
ProjectTimer.Reset();
|
||||
PromoteTimer.Reset();
|
||||
CoarseTimer.Reset();
|
||||
SmoothTimer.Reset();
|
||||
MatrixTimer.Reset();
|
||||
M3Timer.Reset();
|
||||
LinalgTimer.Reset();
|
||||
M3Calls = SmoothCalls = MatrixCalls = ProjectCalls = CoarseCalls = PromoteCalls = 0;
|
||||
HDCGTimer.Start();
|
||||
//////////////////////////
|
||||
// x0 = Vstart -- possibly modify guess
|
||||
//////////////////////////
|
||||
Vstart(x,src);
|
||||
|
||||
|
||||
// r0 = b -A x0
|
||||
_FineLinop.HermOp(x,mmp[0]);
|
||||
axpy (r, -1.0,mmp[0], src); // Recomputes r=src-Ax0
|
||||
@@ -145,33 +168,40 @@ class TwoLevelCG : public LinearFunction<Field>
|
||||
int peri_kp = (k+1) % mmax;
|
||||
|
||||
rtz=rtzp;
|
||||
M3Timer.Start();
|
||||
d= PcgM3(p[peri_k],mmp[peri_k]);
|
||||
M3Timer.Stop();
|
||||
M3Calls++;
|
||||
a = rtz/d;
|
||||
|
||||
|
||||
// Memorise this
|
||||
pAp[peri_k] = d;
|
||||
|
||||
|
||||
LinalgTimer.Start();
|
||||
axpy(x,a,p[peri_k],x);
|
||||
RealD rn = axpy_norm(r,-a,mmp[peri_k],r);
|
||||
LinalgTimer.Stop();
|
||||
|
||||
// Compute z = M x
|
||||
PcgM1(r,z);
|
||||
|
||||
|
||||
{
|
||||
RealD n1,n2;
|
||||
n1=norm2(r);
|
||||
n2=norm2(z);
|
||||
std::cout << GridLogMessage<<"HDCG::fPcg iteration "<<k<<" : vector r,z "<<n1<<" "<<n2<<"\n";
|
||||
}
|
||||
LinalgTimer.Start();
|
||||
rtzp =real(innerProduct(r,z));
|
||||
LinalgTimer.Stop();
|
||||
std::cout << GridLogMessage<<"HDCG::fPcg iteration "<<k<<" : inner rtzp "<<rtzp<<"\n";
|
||||
|
||||
// PcgM2(z,p[0]);
|
||||
PcgM2(z,mu); // ADEF-2 this is identity. Axpy possible to eliminate
|
||||
|
||||
|
||||
p[peri_kp]=mu;
|
||||
|
||||
// Standard search direction p -> z + b p
|
||||
// Standard search direction p -> z + b p
|
||||
b = (rtzp)/rtz;
|
||||
|
||||
int northog;
|
||||
@@ -202,8 +232,25 @@ class TwoLevelCG : public LinearFunction<Field>
|
||||
if ( rn <= rsq ) {
|
||||
|
||||
HDCGTimer.Stop();
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg converged in "<<k<<" iterations and "<<HDCGTimer.Elapsed()<<std::endl;;
|
||||
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg converged in "<<k<<" iterations and "<<HDCGTimer.Elapsed()<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg breakdown"<<std::endl;
|
||||
auto mspc = [](GridStopWatch &sw, int64_t n) -> double {
|
||||
return (n > 0) ? sw.useconds() * 1e-3 / n : 0.0;
|
||||
};
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg M3 (fine MVM) "<<M3Timer.Elapsed()
|
||||
<<" "<<M3Calls<<" calls "<<mspc(M3Timer,M3Calls)<<" ms/call"<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg linalg "<<LinalgTimer.Elapsed()<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg smoother "<<SmoothTimer.Elapsed()
|
||||
<<" "<<SmoothCalls<<" calls "<<mspc(SmoothTimer,SmoothCalls)<<" ms/call"<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg matrix (in M1) "<<MatrixTimer.Elapsed()
|
||||
<<" "<<MatrixCalls<<" calls "<<mspc(MatrixTimer,MatrixCalls)<<" ms/call"<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg project "<<ProjectTimer.Elapsed()
|
||||
<<" "<<ProjectCalls<<" calls "<<mspc(ProjectTimer,ProjectCalls)<<" ms/call"<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg coarse "<<CoarseTimer.Elapsed()
|
||||
<<" "<<CoarseCalls<<" calls "<<mspc(CoarseTimer,CoarseCalls)<<" ms/call"<<std::endl;
|
||||
std::cout<<GridLogMessage<<"HDCG: fPcg promote "<<PromoteTimer.Elapsed()
|
||||
<<" "<<PromoteCalls<<" calls "<<mspc(PromoteTimer,PromoteCalls)<<" ms/call"<<std::endl;
|
||||
|
||||
_FineLinop.HermOp(x,mmp[0]);
|
||||
axpy(tmp,-1.0,src,mmp[0]);
|
||||
|
||||
@@ -475,35 +522,29 @@ class TwoLevelADEF2 : public TwoLevelCG<Field>
|
||||
CoarseField PleftProj(this->coarsegrid);
|
||||
CoarseField PleftMss_proj(this->coarsegrid);
|
||||
|
||||
GridStopWatch SmootherTimer;
|
||||
GridStopWatch MatrixTimer;
|
||||
SmootherTimer.Start();
|
||||
this->SmoothTimer.Start();
|
||||
this->_Smoother(in,Min);
|
||||
SmootherTimer.Stop();
|
||||
this->SmoothTimer.Stop();
|
||||
this->SmoothCalls++;
|
||||
|
||||
MatrixTimer.Start();
|
||||
this->MatrixTimer.Start();
|
||||
this->_FineLinop.HermOp(Min,out);
|
||||
MatrixTimer.Stop();
|
||||
this->MatrixTimer.Stop();
|
||||
this->MatrixCalls++;
|
||||
axpy(tmp,-1.0,out,in); // tmp = in - A Min
|
||||
|
||||
GridStopWatch ProjTimer;
|
||||
GridStopWatch CoarseTimer;
|
||||
GridStopWatch PromTimer;
|
||||
ProjTimer.Start();
|
||||
this->_Aggregates.ProjectToSubspace(PleftProj,tmp);
|
||||
ProjTimer.Stop();
|
||||
CoarseTimer.Start();
|
||||
this->ProjectTimer.Start();
|
||||
this->_Aggregates.ProjectToSubspace(PleftProj,tmp);
|
||||
this->ProjectTimer.Stop();
|
||||
this->ProjectCalls++;
|
||||
this->CoarseTimer.Start();
|
||||
this->_CoarseSolver(PleftProj,PleftMss_proj); // Ass^{-1} [in - A Min]_s
|
||||
CoarseTimer.Stop();
|
||||
PromTimer.Start();
|
||||
this->_Aggregates.PromoteFromSubspace(PleftMss_proj,tmp);// tmp = Q[in - A Min]
|
||||
PromTimer.Stop();
|
||||
std::cout << GridLogPerformance << "PcgM1 breakdown "<<std::endl;
|
||||
std::cout << GridLogPerformance << "\tSmoother " << SmootherTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\tMatrix " << MatrixTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\tProj " << ProjTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\tCoarse " << CoarseTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\tProm " << PromTimer.Elapsed() <<std::endl;
|
||||
this->CoarseTimer.Stop();
|
||||
this->CoarseCalls++;
|
||||
this->PromoteTimer.Start();
|
||||
this->_Aggregates.PromoteFromSubspace(PleftMss_proj,tmp);// tmp = Q[in - A Min]
|
||||
this->PromoteTimer.Stop();
|
||||
this->PromoteCalls++;
|
||||
|
||||
axpy(out,1.0,Min,tmp); // Min+tmp
|
||||
}
|
||||
|
||||
@@ -92,8 +92,8 @@ class TwoLevelCGmrhs
|
||||
// Vector case
|
||||
virtual void operator() (std::vector<Field> &src, std::vector<Field> &x)
|
||||
{
|
||||
// SolveSingleSystem(src,x);
|
||||
SolvePrecBlockCG(src,x);
|
||||
SolveSingleSystem(src,x);
|
||||
// SolvePrecBlockCG(src,x);
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -212,13 +212,17 @@ public:
|
||||
<< "\tTarget " << Tolerance << std::endl;
|
||||
|
||||
// std::cout << GridLogMessage << "\tPreamble " << PreambleTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tSolver Elapsed " << SolverTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "Time breakdown "<<std::endl;
|
||||
std::cout << GridLogPerformance << "\tMatrix " << MatrixTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\tLinalg " << LinalgTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\t\tInner " << InnerTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\t\tAxpyNorm " << AxpyNormTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogPerformance << "\t\tLinearComb " << LinearCombTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tPreamble " << PreambleTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tConstruct " << ConstructTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tNorm " << NormTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tAssign " << AssignTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tSolver " << SolverTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "Solver breakdown "<<std::endl;
|
||||
std::cout << GridLogMessage << "\tMatrix " << MatrixTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\tLinalg " << LinalgTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\t\tInner " << InnerTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\t\tAxpyNorm " << AxpyNormTimer.Elapsed() <<std::endl;
|
||||
std::cout << GridLogMessage << "\t\tLinearComb " << LinearCombTimer.Elapsed() <<std::endl;
|
||||
|
||||
std::cout << GridLogDebug << "\tMobius flop rate " << DwfFlops/ usecs<< " Gflops " <<std::endl;
|
||||
|
||||
|
||||
@@ -236,4 +236,5 @@ public:
|
||||
}
|
||||
};
|
||||
NAMESPACE_END(Grid);
|
||||
#undef GCRLogLevel
|
||||
#endif
|
||||
|
||||
@@ -38,13 +38,14 @@ Author: Peter Boyle <paboyle@ph.ed.ac.uk>
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
#define GCRLogLevel std::cout << GridLogMessage <<std::string(level,'\t')<< " Level "<<level<<" "
|
||||
#define GCRLogLevel std::cout << GridLogMessage <<std::string(level,'\t')<< name<<" "
|
||||
|
||||
template<class Field>
|
||||
class PrecGeneralisedConjugateResidualNonHermitian : public LinearFunction<Field> {
|
||||
public:
|
||||
using LinearFunction<Field>::operator();
|
||||
RealD Tolerance;
|
||||
RealD SSQ;
|
||||
Integer MaxIterations;
|
||||
int verbose;
|
||||
int mmax;
|
||||
@@ -54,11 +55,18 @@ public:
|
||||
GridStopWatch PrecTimer;
|
||||
GridStopWatch MatTimer;
|
||||
GridStopWatch LinalgTimer;
|
||||
std::string name;
|
||||
int ZeroGuess = 0; // caller contract: guess is always zero => first-cycle r0 = src, skip the apply
|
||||
int FirstCycle = 0;
|
||||
|
||||
LinearFunction<Field> &Preconditioner;
|
||||
LinearOperatorBase<Field> &Linop;
|
||||
|
||||
void Level(int lv) { level=lv; };
|
||||
void Name(std::string _name) { name = _name; };
|
||||
|
||||
void Level(int n) { Name("Level " + std::to_string(n)); level = n; }
|
||||
|
||||
void SetZeroGuess(int z) { ZeroGuess = z; };
|
||||
|
||||
PrecGeneralisedConjugateResidualNonHermitian(RealD tol,Integer maxit,LinearOperatorBase<Field> &_Linop,LinearFunction<Field> &Prec,int _mmax,int _nstep) :
|
||||
Tolerance(tol),
|
||||
@@ -67,8 +75,8 @@ public:
|
||||
Preconditioner(Prec),
|
||||
mmax(_mmax),
|
||||
nstep(_nstep)
|
||||
{
|
||||
level=1;
|
||||
{
|
||||
Level(1);
|
||||
verbose=1;
|
||||
};
|
||||
|
||||
@@ -77,6 +85,7 @@ public:
|
||||
// psi=Zero();
|
||||
RealD cp, ssq,rsq;
|
||||
ssq=norm2(src);
|
||||
SSQ=ssq;
|
||||
rsq=Tolerance*Tolerance*ssq;
|
||||
|
||||
Field r(src.Grid());
|
||||
@@ -89,11 +98,12 @@ public:
|
||||
SolverTimer.Start();
|
||||
|
||||
steps=0;
|
||||
FirstCycle=1;
|
||||
for(int k=0;k<MaxIterations;k++){
|
||||
|
||||
cp=GCRnStep(src,psi,rsq);
|
||||
|
||||
GCRLogLevel <<"PGCR("<<mmax<<","<<nstep<<") "<< steps <<" steps cp = "<<cp<<" target "<<rsq <<std::endl;
|
||||
GCRLogLevel <<"PGCR("<<mmax<<","<<nstep<<") "<< steps <<" steps cp = "<<sqrt(cp/ssq)<<" target "<<sqrt(rsq/ssq) <<std::endl;
|
||||
|
||||
if(cp<rsq) {
|
||||
|
||||
@@ -142,21 +152,25 @@ public:
|
||||
GCRLogLevel<< "PGCR nStep("<<nstep<<")"<<std::endl;
|
||||
|
||||
//////////////////////////////////
|
||||
// initial guess x0 is taken as nonzero.
|
||||
// r0=src-A x0 = src
|
||||
// r0 = src - A x0. ZeroGuess: on the first cycle x0==0 by caller
|
||||
// contract (enforced here), so r0 = src exactly; skip the apply.
|
||||
// Restart cycles (psi!=0) always do the full computation.
|
||||
//////////////////////////////////
|
||||
MatTimer.Start();
|
||||
Linop.Op(psi,Az);
|
||||
// zAz = innerProduct(Az,psi);
|
||||
zAAz= norm2(Az);
|
||||
MatTimer.Stop();
|
||||
|
||||
if (ZeroGuess && FirstCycle) {
|
||||
psi = Zero();
|
||||
LinalgTimer.Start();
|
||||
r = src;
|
||||
LinalgTimer.Stop();
|
||||
} else {
|
||||
MatTimer.Start();
|
||||
Linop.Op(psi,Az);
|
||||
MatTimer.Stop();
|
||||
LinalgTimer.Start();
|
||||
r=src-Az;
|
||||
LinalgTimer.Stop();
|
||||
}
|
||||
FirstCycle=0;
|
||||
|
||||
LinalgTimer.Start();
|
||||
r=src-Az;
|
||||
LinalgTimer.Stop();
|
||||
GCRLogLevel<< "PGCR true residual r = src - A psi "<<norm2(r) <<std::endl;
|
||||
|
||||
/////////////////////
|
||||
// p = Prec(r)
|
||||
/////////////////////
|
||||
@@ -181,6 +195,7 @@ public:
|
||||
|
||||
cp =norm2(r);
|
||||
LinalgTimer.Stop();
|
||||
GCRLogLevel<< "PGCR true residual "<< sqrt(cp/SSQ) <<std::endl;
|
||||
|
||||
for(int k=0;k<nstep;k++){
|
||||
|
||||
@@ -199,13 +214,12 @@ public:
|
||||
cp = axpy_norm(r,-a,q[peri_k],r);
|
||||
LinalgTimer.Stop();
|
||||
|
||||
GCRLogLevel<< "PGCR step["<<steps<<"] resid " << cp << " target " <<rsq<<std::endl;
|
||||
GCRLogLevel<< "PGCR step["<<steps<<"] resid " << sqrt(cp/SSQ)<<std::endl;
|
||||
|
||||
if((k==nstep-1)||(cp<rsq)){
|
||||
return cp;
|
||||
}
|
||||
|
||||
|
||||
PrecTimer.Start();
|
||||
Preconditioner(r,z);// solve Az = r
|
||||
PrecTimer.Stop();
|
||||
@@ -239,4 +253,6 @@ public:
|
||||
}
|
||||
};
|
||||
NAMESPACE_END(Grid);
|
||||
|
||||
#undef GCRLogLevel
|
||||
#endif
|
||||
|
||||
@@ -66,7 +66,21 @@ public:
|
||||
{
|
||||
};
|
||||
|
||||
|
||||
void GlobalOrthonormalise(void)
|
||||
{
|
||||
// Normalise all vectors
|
||||
for(int i=0;i<nbasis; i++){
|
||||
RealD scale = std::pow(norm2(subspace[i]),-0.5);
|
||||
subspace[i] = subspace[i]*scale;
|
||||
}
|
||||
for(int i=0;i<nbasis; i++){
|
||||
for(int j=0;j<i; j++){
|
||||
basisOrthogonalize(subspace,subspace[i],j);
|
||||
}
|
||||
RealD scale = std::pow(norm2(subspace[i]),-0.5);
|
||||
subspace[i] = subspace[i]*scale;
|
||||
}
|
||||
}
|
||||
void Orthogonalise(void){
|
||||
CoarseScalar InnerProd(CoarseGrid);
|
||||
// std::cout << GridLogMessage <<" Block Gramm-Schmidt pass 1"<<std::endl;
|
||||
@@ -97,7 +111,7 @@ public:
|
||||
|
||||
RealD scale;
|
||||
|
||||
ConjugateGradient<FineField> CG(1.0e-3,400,false);
|
||||
ConjugateGradient<FineField> CG(1.0e-4,2000,false);
|
||||
FineField noise(FineGrid);
|
||||
FineField Mn(FineGrid);
|
||||
|
||||
@@ -110,14 +124,16 @@ public:
|
||||
|
||||
hermop.Op(noise,Mn); std::cout<<GridLogMessage << "noise ["<<b<<"] <n|MdagM|n> "<<norm2(Mn)<<std::endl;
|
||||
|
||||
for(int i=0;i<4;i++){
|
||||
for(int i=0;i<2;i++){
|
||||
|
||||
CG(hermop,noise,subspace[b]);
|
||||
|
||||
noise = subspace[b];
|
||||
scale = std::pow(norm2(noise),-0.5);
|
||||
noise=noise*scale;
|
||||
|
||||
|
||||
hermop.Op(noise,Mn); std::cout<<GridLogMessage << "intermediate["<<i<<"] <i|MdagM|i> "<<norm2(Mn)<<std::endl;
|
||||
|
||||
}
|
||||
|
||||
hermop.Op(noise,Mn); std::cout<<GridLogMessage << "filtered["<<b<<"] <f|MdagM|f> "<<norm2(Mn)<<std::endl;
|
||||
@@ -131,7 +147,11 @@ public:
|
||||
RealD scale;
|
||||
|
||||
TrivialPrecon<FineField> simple_fine;
|
||||
PrecGeneralisedConjugateResidualNonHermitian<FineField> GCR(0.001,30,DiracOp,simple_fine,12,12);
|
||||
// PrecGeneralisedConjugateResidualNonHermitian<FineField> GCR(0.001,10,DiracOp,simple_fine,30,30);
|
||||
// PrecGeneralisedConjugateResidualNonHermitian<FineField> GCR(0.001,10,DiracOp,simple_fine,12,12);
|
||||
// PrecGeneralisedConjugateResidualNonHermitian<FineField> GCR(0.001,30,DiracOp,simple_fine,12,12);
|
||||
// PrecGeneralisedConjugateResidualNonHermitian<FineField> GCR(0.0005,30,DiracOp,simple_fine,20,20);
|
||||
PrecGeneralisedConjugateResidualNonHermitian<FineField> GCR(0.0005,30,DiracOp,simple_fine,10,10);
|
||||
FineField noise(FineGrid);
|
||||
FineField src(FineGrid);
|
||||
FineField guess(FineGrid);
|
||||
@@ -146,16 +166,16 @@ public:
|
||||
|
||||
DiracOp.Op(noise,Mn); std::cout<<GridLogMessage << "noise ["<<b<<"] <n|Op|n> "<<innerProduct(noise,Mn)<<std::endl;
|
||||
|
||||
for(int i=0;i<2;i++){
|
||||
for(int i=0;i<3;i++){
|
||||
// void operator() (const Field &src, Field &psi){
|
||||
#if 1
|
||||
std::cout << GridLogMessage << " inverting on noise "<<std::endl;
|
||||
if (i==0)std::cout << GridLogMessage << " inverting on noise "<<std::endl;
|
||||
src = noise;
|
||||
guess=Zero();
|
||||
GCR(src,guess);
|
||||
subspace[b] = guess;
|
||||
#else
|
||||
std::cout << GridLogMessage << " inverting on zero "<<std::endl;
|
||||
if (i==0)std::cout << GridLogMessage << " inverting on zero "<<std::endl;
|
||||
src=Zero();
|
||||
guess = noise;
|
||||
GCR(src,guess);
|
||||
@@ -164,13 +184,16 @@ public:
|
||||
noise = subspace[b];
|
||||
scale = std::pow(norm2(noise),-0.5);
|
||||
noise=noise*scale;
|
||||
|
||||
DiracOp.Op(noise,Mn); std::cout<<GridLogMessage << "intermediate["<<i<<"] <f|Op|f> "<<innerProduct(noise,Mn)<<" <f|OpDagOp|f>"<<norm2(Mn)<<std::endl;
|
||||
|
||||
}
|
||||
|
||||
DiracOp.Op(noise,Mn); std::cout<<GridLogMessage << "filtered["<<b<<"] <f|Op|f> "<<innerProduct(noise,Mn)<<std::endl;
|
||||
DiracOp.Op(noise,Mn); std::cout<<GridLogMessage << "filtered["<<b<<"] <f|Op|f> "<<innerProduct(noise,Mn)<<" <f|OpDagOp|f>"<<norm2(Mn)<<std::endl;
|
||||
subspace[b] = noise;
|
||||
|
||||
}
|
||||
GlobalOrthonormalise();
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -31,6 +31,7 @@ Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
#include <Grid/lattice/PaddedCell.h>
|
||||
#include <Grid/stencil/GeneralLocalStencil.h>
|
||||
#include <Grid/algorithms/deflation/MultiRHSBlockProject.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
@@ -66,6 +67,10 @@ public:
|
||||
std::vector<CoarseMatrix> _Adag;
|
||||
std::vector<CoarseVector> MultTemporaries;
|
||||
|
||||
int64_t MultCalls;
|
||||
double MultFlopsAccum;
|
||||
double MultUsecAccum;
|
||||
|
||||
///////////////////////
|
||||
// Interface
|
||||
///////////////////////
|
||||
@@ -104,19 +109,20 @@ public:
|
||||
}
|
||||
*/
|
||||
|
||||
GeneralCoarsenedMatrix(NonLocalStencilGeometry &_geom,GridBase *FineGrid, GridCartesian * CoarseGrid)
|
||||
GeneralCoarsenedMatrix(NonLocalStencilGeometry &_geom,GridBase *FineGrid, GridCartesian * CoarseGrid,int _herm=1)
|
||||
: geom(_geom),
|
||||
_FineGrid(FineGrid),
|
||||
_CoarseGrid(CoarseGrid),
|
||||
hermitian(1),
|
||||
hermitian(_herm),
|
||||
Cell(_geom.Depth(),_CoarseGrid),
|
||||
Stencil(Cell.grids.back(),geom.shifts)
|
||||
Stencil(Cell.grids.back(),geom.shifts),
|
||||
MultCalls(0), MultFlopsAccum(0.0), MultUsecAccum(0.0)
|
||||
{
|
||||
{
|
||||
int npoint = _geom.npoint;
|
||||
}
|
||||
_A.resize(geom.npoint,CoarseGrid);
|
||||
// _Adag.resize(geom.npoint,CoarseGrid);
|
||||
if ( !hermitian ) _Adag.resize(geom.npoint,CoarseGrid);
|
||||
}
|
||||
void M (const CoarseVector &in, CoarseVector &out)
|
||||
{
|
||||
@@ -124,10 +130,10 @@ public:
|
||||
}
|
||||
void Mdag (const CoarseVector &in, CoarseVector &out)
|
||||
{
|
||||
GRID_ASSERT(hermitian);
|
||||
Mult(_A,in,out);
|
||||
// if ( hermitian ) M(in,out);
|
||||
// else Mult(_Adag,in,out);
|
||||
if(hermitian)
|
||||
Mult(_A,in,out);
|
||||
else
|
||||
Mult(_Adag,in,out);
|
||||
}
|
||||
void Mult (std::vector<CoarseMatrix> &A,const CoarseVector &in, CoarseVector &out)
|
||||
{
|
||||
@@ -227,29 +233,28 @@ public:
|
||||
text+=usecond();
|
||||
ttot+=usecond();
|
||||
|
||||
std::cout << GridLogPerformance<<"Coarse 1rhs Mult Aviews "<<tviews<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Mult exch "<<texch<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Mult mult "<<tmult<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<" of which mult2 "<<tmult2<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Mult ext "<<text<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Mult temps "<<ttemps<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Mult copy "<<tcopy<<" us"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Mult tot "<<ttot<<" us"<<std::endl;
|
||||
// std::cout << GridLogPerformance<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Kernel flops "<< flops<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Kernel flop/s "<< flops/tmult<<" mflop/s"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse Kernel bytes/s "<< bytes/tmult<<" MB/s"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse overall flops/s "<< flops/ttot<<" mflop/s"<<std::endl;
|
||||
std::cout << GridLogPerformance<<"Coarse total bytes "<< bytes/1e6<<" MB"<<std::endl;
|
||||
MultCalls++;
|
||||
MultFlopsAccum += flops;
|
||||
MultUsecAccum += ttot;
|
||||
std::cout << GridLogPerformance
|
||||
<< "Coarse Mult call " << MultCalls
|
||||
<< " tot " << ttot << " us"
|
||||
<< " kernel " << tmult << " us"
|
||||
<< " kernel " << flops/tmult*1e-3 << " GFlop/s"
|
||||
<< " overall " << MultFlopsAccum/MultUsecAccum*1e-3 << " GFlop/s (cumul)"
|
||||
<< " bw " << bytes/tmult*1e-3 << " GB/s"
|
||||
<< std::endl;
|
||||
|
||||
};
|
||||
|
||||
void PopulateAdag(void)
|
||||
{
|
||||
#if 0
|
||||
// Serial global peek/poke reference implementation
|
||||
for(int64_t bidx=0;bidx<CoarseGrid()->gSites() ;bidx++){
|
||||
Coordinate bcoor;
|
||||
CoarseGrid()->GlobalIndexToGlobalCoor(bidx,bcoor);
|
||||
|
||||
|
||||
for(int p=0;p<geom.npoint;p++){
|
||||
Coordinate scoor = bcoor;
|
||||
for(int mu=0;mu<bcoor.size();mu++){
|
||||
@@ -262,6 +267,36 @@ public:
|
||||
pokeSite(adj(link),_Adag[pp],bcoor);
|
||||
}
|
||||
}
|
||||
#else
|
||||
// Parallel: _Adag[pp](x) = adj( _A[p](x + s_pp) ), pp = Reverse(p), s_pp = -s_p.
|
||||
// The neighbour fetch reuses the same padded-cell + stencil machinery as Mult,
|
||||
// reading one matrix element per coalesced access so no whole site matrix
|
||||
// (230KB at nbasis=60) ever lands on a GPU thread stack (HIP limit 128KB).
|
||||
// Halo sites compute garbage neighbours; Cell.Extract discards them.
|
||||
// Must run on the unpadded _A, i.e. before ExchangeCoarseLinks.
|
||||
const int Nsimd = CComplex::Nsimd();
|
||||
for(int p=0;p<geom.npoint;p++){
|
||||
int pp = geom.Reverse(p);
|
||||
CoarseMatrix Apad = Cell.ExchangePeriodic(_A[p]);
|
||||
CoarseMatrix Dpad(Apad.Grid());
|
||||
int64_t osites = Apad.Grid()->oSites();
|
||||
{
|
||||
autoView( Apad_v , Apad, AcceleratorRead);
|
||||
autoView( Dpad_v , Dpad, AcceleratorWriteDiscard);
|
||||
autoView( Stencil_v, Stencil, AcceleratorRead);
|
||||
accelerator_for(sj, osites*nbasis, Nsimd, {
|
||||
int32_t ss = sj/nbasis;
|
||||
int32_t j = sj%nbasis;
|
||||
auto SE = Stencil_v.GetEntry(pp,ss);
|
||||
for(int i=0;i<nbasis;i++){
|
||||
auto z = coalescedReadGeneralPermute(Apad_v[SE->_offset](i,j),SE->_permute,Nd);
|
||||
coalescedWrite(Dpad_v[ss](j,i),conjugate(z));
|
||||
}
|
||||
});
|
||||
}
|
||||
_Adag[pp] = Cell.Extract(Dpad);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
/////////////////////////////////////////////////////////////
|
||||
//
|
||||
@@ -417,10 +452,19 @@ public:
|
||||
int osites=CoarseGrid()->oSites();
|
||||
autoView( A_v , _A[k], AcceleratorWrite);
|
||||
autoView( FT_v , FT[k], AcceleratorRead);
|
||||
accelerator_for(sss, osites, 1, {
|
||||
accelerator_for(sss, osites, nbasis, {
|
||||
#ifdef GRID_SIMT
|
||||
int j = acceleratorSIMTlane(nbasis);
|
||||
A_v[sss](i,j) = FT_v[sss](j);
|
||||
#else
|
||||
// CPU build: acceleratorSIMTlane()==0 -- an un-looped SIMT tensor
|
||||
// index writes ONLY j=0 and silently drops the other nbasis-1
|
||||
// columns (caught by Test_schur_dense_coarse import certificate,
|
||||
// 2026-08-14). Loop explicitly.
|
||||
for(int j=0;j<nbasis;j++){
|
||||
A_v[sss](i,j) = FT_v[sss](j);
|
||||
}
|
||||
#endif
|
||||
});
|
||||
}
|
||||
tinv+=usecond();
|
||||
@@ -428,8 +472,8 @@ public:
|
||||
|
||||
// Only needed if nonhermitian
|
||||
if ( ! hermitian ) {
|
||||
// std::cout << GridLogMessage<<"PopulateAdag "<<std::endl;
|
||||
// PopulateAdag();
|
||||
std::cout << GridLogMessage<<"PopulateAdag "<<std::endl;
|
||||
PopulateAdag();
|
||||
}
|
||||
|
||||
// Need to write something to populate Adag from A
|
||||
@@ -517,13 +561,9 @@ public:
|
||||
// Now compute the matrix elements of linop between the orthonormal
|
||||
// set of vectors.
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
FineField phaV(grid); // Phased block basis vector
|
||||
FineField MphaV(grid);// Matrix applied
|
||||
std::vector<FineComplexField> phaF(npoint,grid);
|
||||
std::vector<CoarseComplexField> pha(npoint,CoarseGrid());
|
||||
|
||||
CoarseVector coarseInner(CoarseGrid());
|
||||
|
||||
|
||||
typedef typename CComplex::scalar_type SComplex;
|
||||
FineComplexField one(grid); one=SComplex(1.0);
|
||||
FineComplexField zz(grid); zz = Zero();
|
||||
@@ -542,37 +582,52 @@ public:
|
||||
pha[p] =exp(pha[p]*ci);
|
||||
|
||||
blockZAXPY(phaF[p],pha[p],one,zz);
|
||||
|
||||
|
||||
}
|
||||
tphase+=usecond();
|
||||
|
||||
std::vector<CoarseVector> ComputeProj(npoint,CoarseGrid());
|
||||
std::vector<CoarseVector> FT(npoint,CoarseGrid());
|
||||
|
||||
// Import basis into BLAS layout once; blockProject then reads it once per
|
||||
// basis vector rather than once per (i,p) as in scalar blockProject.
|
||||
// Process all npoint in a single batch.
|
||||
MultiRHSBlockProject<FineField> Projector;
|
||||
Projector.Allocate(nbasis, grid, CoarseGrid());
|
||||
Projector.ImportBasis(U.subspace);
|
||||
|
||||
std::vector<FineField> phaV_batch(npoint, grid);
|
||||
std::vector<FineField> MphaV_batch(npoint, grid);
|
||||
std::vector<CoarseVector> proj_batch(npoint, CoarseGrid());
|
||||
std::vector<CoarseVector> ComputeProj(npoint, CoarseGrid());
|
||||
std::vector<CoarseVector> FT(npoint, CoarseGrid());
|
||||
|
||||
// Pre-allocate BLAS_F and BLAS_C to avoid repeated hipMalloc/hipFree of
|
||||
// ~5.6 GB per blockProject call, which hangs on ROCm for large allocations.
|
||||
Projector.BLAS_F.resize(Projector.fine_vol * Projector.words * npoint);
|
||||
Projector.BLAS_C.resize(Projector.coarse_vol * nbasis * npoint);
|
||||
|
||||
for(int i=0;i<nbasis;i++){// Loop over basis vectors
|
||||
accelerator_barrier(); // ensure prior iteration's async writes are retired
|
||||
std::cout << GridLogMessage<< "CoarsenMatrixColoured vec "<<i<<"/"<<nbasis<< std::endl;
|
||||
for(int p=0;p<npoint;p++){ // Loop over momenta in npoint
|
||||
tphaseBZ-=usecond();
|
||||
phaV = phaF[p]*V.subspace[i];
|
||||
tphaseBZ+=usecond();
|
||||
|
||||
/////////////////////////////////////////////////////////////////////
|
||||
// Multiple phased subspace vector by matrix and project to subspace
|
||||
// Remove local bulk phase to leave relative phases
|
||||
/////////////////////////////////////////////////////////////////////
|
||||
tmat-=usecond();
|
||||
linop.Op(phaV,MphaV);
|
||||
tmat+=usecond();
|
||||
// std::cout << i << " " <<p << " MphaV "<<norm2(MphaV)<<" "<<norm2(phaV)<<std::endl;
|
||||
tphaseBZ-=usecond();
|
||||
for(int p=0;p<npoint;p++)
|
||||
phaV_batch[p] = phaF[p] * V.subspace[i];
|
||||
tphaseBZ+=usecond();
|
||||
std::cout << GridLogMessage<< "CoarsenMatrixColoured vec "<<i<<" phaseBZ done"<< std::endl;
|
||||
|
||||
tproj-=usecond();
|
||||
blockProject(coarseInner,MphaV,U.subspace);
|
||||
coarseInner = conjugate(pha[p]) * coarseInner;
|
||||
tmat-=usecond();
|
||||
for(int p=0;p<npoint;p++)
|
||||
linop.Op(phaV_batch[p], MphaV_batch[p]);
|
||||
tmat+=usecond();
|
||||
std::cout << GridLogMessage<< "CoarsenMatrixColoured vec "<<i<<" mat done"<< std::endl;
|
||||
|
||||
ComputeProj[p] = coarseInner;
|
||||
tproj+=usecond();
|
||||
// std::cout << i << " " <<p << " ComputeProj "<<norm2(ComputeProj[p])<<std::endl;
|
||||
|
||||
}
|
||||
// One batched GEMM reads BLAS_V once for all npoint vectors.
|
||||
tproj-=usecond();
|
||||
Projector.blockProject(MphaV_batch, proj_batch);
|
||||
std::cout << GridLogMessage<< "CoarsenMatrixColoured vec "<<i<<" blockProject done"<< std::endl;
|
||||
for(int p=0;p<npoint;p++)
|
||||
ComputeProj[p] = conjugate(pha[p]) * proj_batch[p];
|
||||
tproj+=usecond();
|
||||
std::cout << GridLogMessage<< "CoarsenMatrixColoured vec "<<i<<" proj done"<< std::endl;
|
||||
|
||||
tinv-=usecond();
|
||||
for(int k=0;k<npoint;k++){
|
||||
@@ -580,14 +635,23 @@ public:
|
||||
for(int l=0;l<npoint;l++){
|
||||
FT[k]= FT[k]+ invMkl(l,k)*ComputeProj[l];
|
||||
}
|
||||
|
||||
|
||||
int osites=CoarseGrid()->oSites();
|
||||
autoView( A_v , _A[k], AcceleratorWrite);
|
||||
autoView( FT_v , FT[k], AcceleratorRead);
|
||||
accelerator_for(sss, osites, 1, {
|
||||
accelerator_for(sss, osites, nbasis, {
|
||||
#ifdef GRID_SIMT
|
||||
int j = acceleratorSIMTlane(nbasis);
|
||||
A_v[sss](i,j) = FT_v[sss](j);
|
||||
#else
|
||||
// CPU build: acceleratorSIMTlane()==0 -- an un-looped SIMT tensor
|
||||
// index writes ONLY j=0 and silently drops the other nbasis-1
|
||||
// columns (caught by Test_schur_dense_coarse import certificate,
|
||||
// 2026-08-14). Loop explicitly.
|
||||
for(int j=0;j<nbasis;j++){
|
||||
A_v[sss](i,j) = FT_v[sss](j);
|
||||
}
|
||||
#endif
|
||||
});
|
||||
}
|
||||
tinv+=usecond();
|
||||
@@ -595,13 +659,13 @@ public:
|
||||
|
||||
// Only needed if nonhermitian
|
||||
if ( ! hermitian ) {
|
||||
// std::cout << GridLogMessage<<"PopulateAdag "<<std::endl;
|
||||
// PopulateAdag();
|
||||
std::cout << GridLogMessage<<"PopulateAdag "<<std::endl;
|
||||
PopulateAdag();
|
||||
}
|
||||
|
||||
for(int p=0;p<geom.npoint;p++){
|
||||
std::cout << " _A["<<p<<"] "<<norm2(_A[p])<<std::endl;
|
||||
}
|
||||
// for(int p=0;p<geom.npoint;p++){
|
||||
// std::cout << " _A["<<p<<"] "<<norm2(_A[p])<<std::endl;
|
||||
// }
|
||||
|
||||
// Need to write something to populate Adag from A
|
||||
ExchangeCoarseLinks();
|
||||
@@ -616,7 +680,7 @@ public:
|
||||
void ExchangeCoarseLinks(void){
|
||||
for(int p=0;p<geom.npoint;p++){
|
||||
_A[p] = Cell.ExchangePeriodic(_A[p]);
|
||||
// _Adag[p]= Cell.ExchangePeriodic(_Adag[p]);
|
||||
if ( !hermitian ) _Adag[p]= Cell.ExchangePeriodic(_Adag[p]);
|
||||
}
|
||||
}
|
||||
virtual void Mdiag (const Field &in, Field &out){ GRID_ASSERT(0);};
|
||||
|
||||
@@ -0,0 +1,662 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: RecursiveSchurInverse.h
|
||||
|
||||
Copyright (C) 2026
|
||||
|
||||
Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#pragma once
|
||||
|
||||
#include <Grid/algorithms/blas/BatchedBlas.h>
|
||||
#include <Grid/algorithms/blas/BatchedInverse.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// RecursiveSchurInverse: distributed dense inversion by recursive Schur
|
||||
// complement over a binary rank-range tree.
|
||||
//
|
||||
// CONTRACT: the caller presents an N x N matrix in RANK-MAJOR row ordering,
|
||||
// distributed by rows -- rank r owns global rows [rowStart[r], rowStart[r+1])
|
||||
// -- and receives its rows of the INVERSE in the same layout. This class
|
||||
// knows nothing of lattices or coarse operators; it consumes a GridBase for
|
||||
// world collectives, GridBLAS for GEMMs and GridBLASInverse for the leaf
|
||||
// inversions (all of which have Eigen reference backends, so the whole
|
||||
// algorithm unit-tests on a CPU-only laptop build under mpirun).
|
||||
//
|
||||
// PRECISION (decision 2026-08-14, superseding the fp32-merge design): the
|
||||
// ENTIRE inversion runs in fp64 (ComplexD). The apply-side fp32 gain is
|
||||
// taken where it matters -- inside the iterative process -- by rounding the
|
||||
// finished inverse ONCE when the caller stores it in the fp32 apply slab.
|
||||
// Consequences: merge-growth error accumulates in eps64 and the terminal
|
||||
// rounding gives representation-only ~eps32 accuracy independent of growth;
|
||||
// the Newton-Schulz refinement and the fp32 escalation ladder are DELETED
|
||||
// (resurrectable from git history if a future scale forces reduced-precision
|
||||
// merges). Setup cost: ~2x panel-gather bytes and ~2x transient memory,
|
||||
// once per setup; fp64 GEMM runs at fp32 rate on CDNA2/PVC.
|
||||
//
|
||||
// EXECUTION MODEL: SPMD full-tree walk. Every rank executes the identical
|
||||
// recursion call sequence; participation in DATA is ownership-gated, and
|
||||
// every collective is a world-communicator zero-fill GlobalSumVector. No
|
||||
// sub-communicators exist, so no deadlock surface exists.
|
||||
//
|
||||
// STORAGE CONVENTION (pinned by unit test T1b, Test_schur_inverse.cc):
|
||||
// BlockRows is COLUMN-MAJOR with ld = rows, matching the BLAS world:
|
||||
// element (i,j) lives at data[ i + j*ld ]; a column window [col0, col0+w)
|
||||
// is the contiguous slice starting at data[ col0*ld ].
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// My rows of a distributed dense matrix: rows x cols, column major, ld = rows.
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
class BlockRows
|
||||
{
|
||||
public:
|
||||
deviceVector<ComplexD> data;
|
||||
int64_t rows;
|
||||
int64_t cols;
|
||||
int64_t ld;
|
||||
|
||||
BlockRows()
|
||||
{
|
||||
rows = 0;
|
||||
cols = 0;
|
||||
ld = 0;
|
||||
}
|
||||
void Resize(int64_t r, int64_t c)
|
||||
{
|
||||
rows = r;
|
||||
cols = c;
|
||||
ld = r;
|
||||
data.resize((uint64_t)r*c);
|
||||
}
|
||||
ComplexD *ColumnWindow(int64_t col0)
|
||||
{
|
||||
GRID_ASSERT( col0 >= 0 );
|
||||
GRID_ASSERT( col0 <= cols );
|
||||
return &data[(uint64_t)col0*ld];
|
||||
}
|
||||
};
|
||||
|
||||
class RecursiveSchurInverse
|
||||
{
|
||||
public:
|
||||
GridBase *grid; // world collectives only
|
||||
int64_t N; // global matrix dimension
|
||||
int P; // ranks
|
||||
int me; // this rank
|
||||
std::vector<int64_t> rowStart; // P+1 entries: rank-major row ownership
|
||||
int64_t myRow0;
|
||||
int64_t myNrows;
|
||||
int64_t panelBytes; // gather panel budget (DENSE_PANEL_BYTES)
|
||||
|
||||
GridBLAS BLAS;
|
||||
GridBLASInverse INV;
|
||||
|
||||
// Growth telemetry (diagnostic, not load-bearing at fp64): one entry per
|
||||
// merge node, walk order
|
||||
std::vector<double> telNormB; // ||B||_F = ||A11inv A12||_F
|
||||
std::vector<double> telSratio; // ||S||_F / ||A22||_F
|
||||
double telLeafMaxInv; // max |(leaf inverse)_ij| over leaves
|
||||
|
||||
// Phase timers/counters (this rank), accumulated across the whole walk:
|
||||
// where does the setup wall actually go? Reported by ReportTelemetry.
|
||||
double tStage; // owner B-window device->host
|
||||
double tMemset; // panel zero-fill (host)
|
||||
double tDeposit; // owner rows -> panel (host memcpy)
|
||||
double tAllreduce; // GlobalSumVector on panels
|
||||
double tH2D; // panel host->device
|
||||
double tGemm; // strided gemm + synchronise
|
||||
double tLeaf; // leaf inversions
|
||||
double tARmin; // fastest single panel collective
|
||||
double tARmax; // slowest single panel collective
|
||||
uint64_t bytesAllreduce;
|
||||
uint64_t nAllreduce; // panel collectives
|
||||
uint64_t nGatherGemm; // GatherGemm calls
|
||||
|
||||
// PERSISTENT comms/staging buffers, grow-only across the whole walk.
|
||||
// Fresh per-call host allocations defeat the MPI registration cache
|
||||
// (every page unpinned every call => re-registration or bounce-buffer
|
||||
// copies inside MPI on ~GB panels) -- the same reason the halo
|
||||
// exchange uses allocate-once buffers. Measured motivation: 0.48 GB/s
|
||||
// effective allreduce payload at N=138k with per-call vectors.
|
||||
std::vector<ComplexD> panelBuf;
|
||||
deviceVector<ComplexD> dPanelBuf;
|
||||
std::vector<ComplexD> stageBuf;
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Ownership-table validation: a proper partition of [0,N).
|
||||
// Static and communicator-free so synthetic tables unit-test directly.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
static void CheckRowStart(const std::vector<int64_t> &table, int64_t N)
|
||||
{
|
||||
int P = (int)table.size() - 1;
|
||||
GRID_ASSERT( P >= 1 );
|
||||
GRID_ASSERT( table[0] == 0 );
|
||||
GRID_ASSERT( table[P] == N );
|
||||
for(int r=0; r<P; r++)
|
||||
{
|
||||
GRID_ASSERT( table[r+1] >= table[r] ); // zero-row ranks permitted
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Build the ownership table from each rank's local row count: zero-fill
|
||||
// allgather (the standing comms idiom) then prefix sum. Every rank
|
||||
// returns the identical table.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
static std::vector<int64_t> MakeRowStart(GridBase *g, int64_t myNrows)
|
||||
{
|
||||
int P = g->ProcessorCount();
|
||||
int me = g->ThisRank();
|
||||
|
||||
std::vector<uint64_t> counts(P, 0);
|
||||
counts[me] = (uint64_t)myNrows;
|
||||
g->GlobalSumVector(&counts[0], P);
|
||||
|
||||
std::vector<int64_t> table(P+1);
|
||||
table[0] = 0;
|
||||
for(int r=0; r<P; r++)
|
||||
{
|
||||
table[r+1] = table[r] + (int64_t)counts[r];
|
||||
}
|
||||
CheckRowStart(table, table[P]);
|
||||
return table;
|
||||
}
|
||||
|
||||
RecursiveSchurInverse(GridBase *g,
|
||||
int64_t N_,
|
||||
std::vector<int64_t> &rowStart_,
|
||||
int64_t panelBytes_)
|
||||
{
|
||||
grid = g;
|
||||
N = N_;
|
||||
P = g->ProcessorCount();
|
||||
me = g->ThisRank();
|
||||
rowStart = rowStart_;
|
||||
panelBytes = panelBytes_;
|
||||
|
||||
GRID_ASSERT( (int)rowStart.size() == P+1 );
|
||||
CheckRowStart(rowStart, N);
|
||||
|
||||
myRow0 = rowStart[me];
|
||||
myNrows = rowStart[me+1] - rowStart[me];
|
||||
|
||||
telLeafMaxInv = 0.0;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// THE communication primitive (plan 3.4 / 4B.3).
|
||||
//
|
||||
// C(:, colC : colC+widthB) <- beta * C(:, colC : colC+widthB)
|
||||
// + alpha * A(:, colA : colA+widthA) * Bsub
|
||||
//
|
||||
// Bsub is the widthA x widthB sub-block of a row-distributed operand
|
||||
// owned by ranks [rB0, rB1): owner r contributes its rows of
|
||||
// B(:, colB : colB+widthB) at sub-block row offset
|
||||
// rowStart[r] - rowStart[rB0]. The sub-block is gathered in panelBytes
|
||||
// row-chunks by host zero-fill + world GlobalSumVector.
|
||||
//
|
||||
// SPMD rules: EVERY rank calls (the collectives are world-wide);
|
||||
// non-owners of B add zeros; ranks with A.rows == 0 skip all local
|
||||
// compute but still make every collective call. Column offsets are
|
||||
// LOCAL buffer offsets -- non-participants pass 0.
|
||||
//
|
||||
// Owners stage their whole B window device->host ONCE (ld == rows makes
|
||||
// the window contiguous); per-chunk deposits are host memcpy runs.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void GatherGemm(ComplexD alpha,
|
||||
BlockRows &A, int64_t colA, int64_t widthA,
|
||||
int rB0, int rB1,
|
||||
BlockRows &B, int64_t colB, int64_t widthB,
|
||||
ComplexD beta,
|
||||
BlockRows &C, int64_t colC)
|
||||
{
|
||||
GRID_ASSERT( rB0 >= 0 );
|
||||
GRID_ASSERT( rB1 > rB0 );
|
||||
GRID_ASSERT( rB1 <= P );
|
||||
|
||||
int64_t k = rowStart[rB1] - rowStart[rB0];
|
||||
int64_t m = A.rows;
|
||||
int64_t n = widthB;
|
||||
GRID_ASSERT( widthA == k );
|
||||
GRID_ASSERT( n >= 1 );
|
||||
|
||||
int owner = ( me >= rB0 ) && ( me < rB1 ) && ( B.rows > 0 );
|
||||
int64_t myOff = 0;
|
||||
if ( owner )
|
||||
{
|
||||
myOff = rowStart[me] - rowStart[rB0];
|
||||
}
|
||||
|
||||
if ( m > 0 )
|
||||
{
|
||||
GRID_ASSERT( colA + widthA <= A.cols );
|
||||
GRID_ASSERT( colC + widthB <= C.cols );
|
||||
GRID_ASSERT( C.rows == m );
|
||||
}
|
||||
|
||||
nGatherGemm++;
|
||||
GRID_TRACE("GatherGemm");
|
||||
|
||||
std::vector<ComplexD> &stage = stageBuf;
|
||||
if ( owner )
|
||||
{
|
||||
GRID_ASSERT( colB + widthB <= B.cols );
|
||||
tStage -= usecond();
|
||||
if ( stage.size() < (uint64_t)B.rows*n ) stage.resize((uint64_t)B.rows*n);
|
||||
acceleratorCopyFromDevice(B.ColumnWindow(colB), &stage[0],
|
||||
(uint64_t)B.rows*n*sizeof(ComplexD));
|
||||
tStage += usecond();
|
||||
}
|
||||
|
||||
int64_t kc = panelBytes / ( (int64_t)sizeof(ComplexD) * n );
|
||||
if ( kc < 1 ) kc = 1;
|
||||
if ( kc > k ) kc = k;
|
||||
GRID_ASSERT( kc*n < 2147483647L ); // GlobalSumVector count is int
|
||||
|
||||
std::vector<ComplexD> &panel = panelBuf;
|
||||
deviceVector<ComplexD> &dPanel = dPanelBuf;
|
||||
if ( panel.size() < (uint64_t)kc*n ) panel.resize((uint64_t)kc*n);
|
||||
if ( dPanel.size() < (uint64_t)kc*n ) dPanel.resize((uint64_t)kc*n);
|
||||
deviceVector<ComplexD*> ap(1);
|
||||
deviceVector<ComplexD*> bp(1);
|
||||
deviceVector<ComplexD*> cp(1);
|
||||
std::vector<ComplexD*> ptr(1);
|
||||
|
||||
for(int64_t k0=0; k0<k; k0+=kc)
|
||||
{
|
||||
int64_t kchunk = std::min(kc, k-k0);
|
||||
|
||||
// PLANNED OPTIMISATION (not yet): single-threaded memset zero-fills
|
||||
// the WHOLE panel; owners then overwrite their segment. A threaded
|
||||
// zero of only the non-owned rows (thread_for over columns, memset
|
||||
// per column run) halves the host traffic and parallelises it.
|
||||
// Deliberately deferred until the simple version is proven.
|
||||
tMemset -= usecond();
|
||||
memset(&panel[0], 0, (uint64_t)kchunk*n*sizeof(ComplexD));
|
||||
tMemset += usecond();
|
||||
if ( owner )
|
||||
{
|
||||
int64_t i0 = std::max(k0, myOff);
|
||||
int64_t i1 = std::min(k0+kchunk, myOff+B.rows);
|
||||
if ( i1 > i0 )
|
||||
{
|
||||
int64_t len = i1-i0;
|
||||
tDeposit -= usecond();
|
||||
thread_for(j, n, {
|
||||
memcpy(&panel[(uint64_t)((i0-k0) + j*kchunk)],
|
||||
&stage[(uint64_t)((i0-myOff) + j*B.rows)],
|
||||
len*sizeof(ComplexD));
|
||||
});
|
||||
tDeposit += usecond();
|
||||
}
|
||||
}
|
||||
double tar = -usecond();
|
||||
grid->GlobalSumVector(&panel[0], (int)(kchunk*n));
|
||||
tar += usecond();
|
||||
tAllreduce += tar;
|
||||
tARmin = std::min(tARmin, tar);
|
||||
tARmax = std::max(tARmax, tar);
|
||||
bytesAllreduce += (uint64_t)kchunk*n*sizeof(ComplexD);
|
||||
nAllreduce++;
|
||||
|
||||
if ( m > 0 )
|
||||
{
|
||||
tH2D -= usecond();
|
||||
acceleratorCopyToDevice(&panel[0], &dPanel[0],
|
||||
(uint64_t)kchunk*n*sizeof(ComplexD));
|
||||
tH2D += usecond();
|
||||
|
||||
ComplexD beta_use = ( k0==0 ) ? beta : ComplexD(1.0,0.0);
|
||||
|
||||
ptr[0] = A.ColumnWindow(colA + k0);
|
||||
acceleratorCopyToDevice(&ptr[0], &ap[0], sizeof(ComplexD*));
|
||||
ptr[0] = &dPanel[0];
|
||||
acceleratorCopyToDevice(&ptr[0], &bp[0], sizeof(ComplexD*));
|
||||
ptr[0] = C.ColumnWindow(colC);
|
||||
acceleratorCopyToDevice(&ptr[0], &cp[0], sizeof(ComplexD*));
|
||||
|
||||
tGemm -= usecond();
|
||||
BLAS.gemmBatched(GridBLAS_OP_N, GridBLAS_OP_N,
|
||||
(int)m, (int)n, (int)kchunk,
|
||||
alpha, ap, (int)A.ld,
|
||||
bp, (int)kchunk,
|
||||
beta_use, cp, (int)C.ld);
|
||||
BLAS.synchronise();
|
||||
tGemm += usecond();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Local Frobenius norm-squared of a full-height column window.
|
||||
// NO comms; callers GlobalSum the result. Host staging, setup-scale.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
double FrobNorm2Local(BlockRows &X, int64_t col0, int64_t w)
|
||||
{
|
||||
if ( X.rows == 0 ) return 0.0;
|
||||
GRID_ASSERT( col0 + w <= X.cols );
|
||||
uint64_t len = (uint64_t)X.rows*w;
|
||||
std::vector<ComplexD> h(len);
|
||||
acceleratorCopyFromDevice(X.ColumnWindow(col0), &h[0], len*sizeof(ComplexD));
|
||||
// Member real()/imag(): portable across std::complex (CPU) and
|
||||
// thrust::complex (HIP), where std::norm does not resolve.
|
||||
double s = 0.0;
|
||||
for(uint64_t i=0; i<len; i++)
|
||||
{
|
||||
double re = h[i].real();
|
||||
double im = h[i].imag();
|
||||
s += re*re + im*im;
|
||||
}
|
||||
return s;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// dst(:, dcol0 : dcol0+w) = - src(:, 0:w). Both operands have ld == rows
|
||||
// so full-height windows are contiguous: flat elementwise device copy.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void NegateCopy(BlockRows &src, BlockRows &dst, int64_t dcol0, int64_t w)
|
||||
{
|
||||
GRID_ASSERT( src.rows == dst.rows );
|
||||
GRID_ASSERT( w <= src.cols );
|
||||
GRID_ASSERT( dcol0 + w <= dst.cols );
|
||||
if ( src.rows == 0 ) return;
|
||||
uint64_t len = (uint64_t)src.rows*w;
|
||||
ComplexD *s = &src.data[0];
|
||||
ComplexD *d = dst.ColumnWindow(dcol0);
|
||||
accelerator_for(i, len, 1, {
|
||||
d[i] = -s[i];
|
||||
});
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Leaf inversion. Purely LOCAL -- the calling rank owns the whole
|
||||
// width x width leaf (width == my row count); no collectives, so the
|
||||
// SPMD walk stays uniform with other ranks doing nothing. The window
|
||||
// is contiguous (ld == rows == width): invert IN PLACE via
|
||||
// GridBLASInverse. Everything is already fp64; no promote/demote.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void LeafInvert(int64_t col0, int64_t width, BlockRows &Arows)
|
||||
{
|
||||
GRID_TRACE("SchurLeaf");
|
||||
GRID_ASSERT( width == Arows.rows );
|
||||
GRID_ASSERT( col0 + width <= Arows.cols );
|
||||
int64_t w = width;
|
||||
uint64_t len = (uint64_t)w*w;
|
||||
tLeaf -= usecond();
|
||||
|
||||
deviceVector<ComplexD*> bp(1);
|
||||
std::vector<ComplexD*> ptr(1);
|
||||
ptr[0] = Arows.ColumnWindow(col0);
|
||||
acceleratorCopyToDevice(&ptr[0], &bp[0], sizeof(ComplexD*));
|
||||
INV.inverseBatched(w, bp);
|
||||
|
||||
// Telemetry: max |element| of the leaf inverse
|
||||
{
|
||||
std::vector<ComplexD> h(len);
|
||||
acceleratorCopyFromDevice(Arows.ColumnWindow(col0), &h[0], len*sizeof(ComplexD));
|
||||
double mx = 0.0;
|
||||
for(uint64_t i=0; i<len; i++)
|
||||
{
|
||||
double re = h[i].real();
|
||||
double im = h[i].imag();
|
||||
mx = std::max(mx, re*re + im*im);
|
||||
}
|
||||
telLeafMaxInv = std::max(telLeafMaxInv, std::sqrt(mx));
|
||||
}
|
||||
tLeaf += usecond();
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// The recursion (plan 3.5 / 4B.3). Inverts the diagonal block of the
|
||||
// rank-major matrix spanned by ranks [r0, r1), living in every member
|
||||
// rank's column window [col0, col0+width) -- IN PLACE.
|
||||
//
|
||||
// SPMD: every rank calls with IDENTICAL (r0, r1, width) and its own
|
||||
// local (col0, Arows); ranks outside [r0, r1) participate in the
|
||||
// collectives only (dummy operands, zero contributions). The collective
|
||||
// sequence -- 5 GatherGemm calls + 3 scalar GlobalSums per merge node --
|
||||
// is identical on every rank by construction.
|
||||
//
|
||||
// I = [r0, mid) J = [mid, r1) widths WI, WJ
|
||||
// 1. recurse I: A11 -> A11inv
|
||||
// 2. B = A11inv.A12 (I rows)
|
||||
// 3. C = A21.A11inv (J rows)
|
||||
// 4. S = A22 - A21.B in place (J rows) [alpha=-1, beta=1]
|
||||
// 5. recurse J: S -> Sinv
|
||||
// 6. T = Sinv.C (J rows)
|
||||
// 7. U = B.Sinv (I rows)
|
||||
// 8. X11 = A11inv + U.C in place (I rows) [beta=1]
|
||||
// 9. X12 = -U, X21 = -T local negates; X22 = Sinv already in place
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void SchurNode(int r0, int r1, int64_t col0, int64_t width, BlockRows &Arows)
|
||||
{
|
||||
int span = r1 - r0;
|
||||
GRID_ASSERT( span >= 1 );
|
||||
GRID_ASSERT( width == rowStart[r1] - rowStart[r0] );
|
||||
|
||||
if ( span == 1 )
|
||||
{
|
||||
if ( ( me == r0 ) && ( myNrows > 0 ) )
|
||||
{
|
||||
LeafInvert(col0, width, Arows);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
int mid = ( r0 + r1 ) / 2;
|
||||
int64_t WI = rowStart[mid] - rowStart[r0];
|
||||
int64_t WJ = rowStart[r1] - rowStart[mid];
|
||||
|
||||
// Zero-width child ranges (all ranks of a half owning no rows) are a
|
||||
// KNOWN LIMITATION: fail loudly rather than divide mysteriously.
|
||||
GRID_ASSERT( WI > 0 );
|
||||
GRID_ASSERT( WJ > 0 );
|
||||
|
||||
int inI = ( me >= r0 ) && ( me < mid );
|
||||
int inJ = ( me >= mid ) && ( me < r1 );
|
||||
|
||||
ComplexD one ( 1.0,0.0);
|
||||
ComplexD mone (-1.0,0.0);
|
||||
ComplexD zero ( 0.0,0.0);
|
||||
|
||||
BlockRows dummy;
|
||||
|
||||
// 1. A11 -> A11inv
|
||||
SchurNode(r0, mid, col0, WI, Arows);
|
||||
|
||||
// 2. B = A11inv . A12 (I rows; gather A12 from I owners)
|
||||
BlockRows Bbuf;
|
||||
if ( inI ) Bbuf.Resize(myNrows, WJ);
|
||||
{
|
||||
BlockRows &Aop = inI ? Arows : dummy;
|
||||
BlockRows &Cop = inI ? Bbuf : dummy;
|
||||
int64_t cA = inI ? col0 : 0;
|
||||
GatherGemm(one, Aop, cA, WI,
|
||||
r0, mid,
|
||||
Arows, col0+WI, WJ,
|
||||
zero, Cop, 0);
|
||||
}
|
||||
double nB = FrobNorm2Local(Bbuf, 0, inI ? WJ : 0);
|
||||
grid->GlobalSumVector(&nB, 1);
|
||||
telNormB.push_back(std::sqrt(nB));
|
||||
|
||||
// 3. C = A21 . A11inv (J rows; gather A11inv from I owners)
|
||||
BlockRows Cbuf;
|
||||
if ( inJ ) Cbuf.Resize(myNrows, WI);
|
||||
{
|
||||
BlockRows &Aop = inJ ? Arows : dummy;
|
||||
BlockRows &Cop = inJ ? Cbuf : dummy;
|
||||
int64_t cA = inJ ? col0 : 0;
|
||||
GatherGemm(one, Aop, cA, WI,
|
||||
r0, mid,
|
||||
Arows, col0, WI,
|
||||
zero, Cop, 0);
|
||||
}
|
||||
|
||||
// 4. S = A22 - A21 . B in place on my A22 window (J rows)
|
||||
double nA22 = FrobNorm2Local( inJ ? Arows : dummy, inJ ? col0+WI : 0, inJ ? WJ : 0 );
|
||||
grid->GlobalSumVector(&nA22, 1);
|
||||
{
|
||||
BlockRows &Aop = inJ ? Arows : dummy;
|
||||
BlockRows &Cop = inJ ? Arows : dummy;
|
||||
int64_t cA = inJ ? col0 : 0;
|
||||
int64_t cC = inJ ? col0+WI : 0;
|
||||
GatherGemm(mone, Aop, cA, WI,
|
||||
r0, mid,
|
||||
Bbuf, 0, WJ,
|
||||
one, Cop, cC);
|
||||
}
|
||||
double nS = FrobNorm2Local( inJ ? Arows : dummy, inJ ? col0+WI : 0, inJ ? WJ : 0 );
|
||||
grid->GlobalSumVector(&nS, 1);
|
||||
telSratio.push_back( std::sqrt(nS) / ( std::sqrt(nA22) + 1.0e-300 ) );
|
||||
|
||||
// 5. S -> Sinv
|
||||
SchurNode(mid, r1, col0+WI, WJ, Arows);
|
||||
|
||||
// 6. T = Sinv . C (J rows; gather C from J owners)
|
||||
BlockRows Tbuf;
|
||||
if ( inJ ) Tbuf.Resize(myNrows, WI);
|
||||
{
|
||||
BlockRows &Aop = inJ ? Arows : dummy;
|
||||
BlockRows &Cop = inJ ? Tbuf : dummy;
|
||||
int64_t cA = inJ ? col0+WI : 0;
|
||||
GatherGemm(one, Aop, cA, WJ,
|
||||
mid, r1,
|
||||
Cbuf, 0, WI,
|
||||
zero, Cop, 0);
|
||||
}
|
||||
|
||||
// 7. U = B . Sinv (I rows; gather Sinv from J owners)
|
||||
BlockRows Ubuf;
|
||||
if ( inI ) Ubuf.Resize(myNrows, WJ);
|
||||
{
|
||||
BlockRows &Aop = inI ? Bbuf : dummy;
|
||||
BlockRows &Cop = inI ? Ubuf : dummy;
|
||||
GatherGemm(one, Aop, 0, WJ,
|
||||
mid, r1,
|
||||
Arows, col0+WI, WJ,
|
||||
zero, Cop, 0);
|
||||
}
|
||||
|
||||
// 8. X11 = A11inv + U . C in place (I rows; gather C from J owners)
|
||||
{
|
||||
BlockRows &Aop = inI ? Ubuf : dummy;
|
||||
BlockRows &Cop = inI ? Arows : dummy;
|
||||
int64_t cC = inI ? col0 : 0;
|
||||
GatherGemm(one, Aop, 0, WJ,
|
||||
mid, r1,
|
||||
Cbuf, 0, WI,
|
||||
one, Cop, cC);
|
||||
}
|
||||
|
||||
// 9. Off-diagonal signs, local
|
||||
if ( inI ) NegateCopy(Ubuf, Arows, col0+WI, WJ);
|
||||
if ( inJ ) NegateCopy(Tbuf, Arows, col0, WI);
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// PUBLIC ENTRY. Arows: my rows of the rank-major N x N matrix (fp64).
|
||||
// On exit Arows holds my rows of the inverse, still fp64; the caller
|
||||
// owns the single terminal rounding into its fp32 apply storage.
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void Invert(BlockRows &Arows)
|
||||
{
|
||||
GRID_ASSERT( Arows.rows == myNrows );
|
||||
GRID_ASSERT( Arows.cols == N );
|
||||
|
||||
telNormB.resize(0);
|
||||
telSratio.resize(0);
|
||||
telLeafMaxInv = 0.0;
|
||||
|
||||
tStage = 0.0;
|
||||
tMemset = 0.0;
|
||||
tDeposit = 0.0;
|
||||
tAllreduce = 0.0;
|
||||
tH2D = 0.0;
|
||||
tGemm = 0.0;
|
||||
tLeaf = 0.0;
|
||||
tARmin = 1.0e30;
|
||||
tARmax = 0.0;
|
||||
bytesAllreduce = 0;
|
||||
nAllreduce = 0;
|
||||
nGatherGemm = 0;
|
||||
|
||||
SchurNode(0, P, 0, N, Arows);
|
||||
|
||||
RealD mx = telLeafMaxInv;
|
||||
grid->GlobalMax(mx);
|
||||
telLeafMaxInv = mx;
|
||||
}
|
||||
|
||||
// All telemetry values are globally reduced or boss-local; safe to
|
||||
// stream on every rank (Grid quiesces stdout to the boss unless
|
||||
// --debug-stdout). NOTE: tAllreduce INCLUDES wait/imbalance -- a rank
|
||||
// arriving early books its wait here; the min/max spread across ranks
|
||||
// separates true wire time (~min) from skew (max-min).
|
||||
void ReportTelemetry(void)
|
||||
{
|
||||
for(uint64_t i=0; i<telNormB.size(); i++)
|
||||
{
|
||||
std::cout << GridLogPerformance
|
||||
<< "SchurNode " << i
|
||||
<< " ||B||_F " << telNormB[i]
|
||||
<< " ||S||/||A22|| " << telSratio[i]
|
||||
<< std::endl;
|
||||
}
|
||||
std::cout << GridLogPerformance
|
||||
<< "Schur leaves max|Ainv| " << telLeafMaxInv
|
||||
<< std::endl;
|
||||
|
||||
RealD armax = tAllreduce;
|
||||
RealD armin = -tAllreduce;
|
||||
grid->GlobalMax(armax);
|
||||
grid->GlobalMax(armin);
|
||||
armin = -armin;
|
||||
|
||||
std::cout << GridLogMessage << "Schur phases (boss rank, seconds):"
|
||||
<< " stage " << tStage/1.0e6
|
||||
<< " memset " << tMemset/1.0e6
|
||||
<< " deposit " << tDeposit/1.0e6
|
||||
<< " allreduce " << tAllreduce/1.0e6
|
||||
<< " H2D " << tH2D/1.0e6
|
||||
<< " gemm " << tGemm/1.0e6
|
||||
<< " leaf " << tLeaf/1.0e6
|
||||
<< std::endl;
|
||||
std::cout << GridLogMessage << "Schur comms:"
|
||||
<< " GatherGemm calls " << nGatherGemm
|
||||
<< " panel allreduces " << nAllreduce
|
||||
<< " allreduce GB " << bytesAllreduce/1024./1024./1024.
|
||||
<< " allreduce s min/max over ranks " << armin/1.0e6
|
||||
<< " / " << armax/1.0e6
|
||||
<< std::endl;
|
||||
std::cout << GridLogMessage << "Schur comms per-call (boss):"
|
||||
<< " min " << tARmin/1.0e3 << " ms"
|
||||
<< " avg " << (nAllreduce ? tAllreduce/nAllreduce/1.0e3 : 0.0) << " ms"
|
||||
<< " max " << tARmax/1.0e3 << " ms"
|
||||
<< std::endl;
|
||||
}
|
||||
};
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
@@ -63,12 +63,10 @@ void MemoryManager::PrintBytes(void)
|
||||
std::cout << " MemoryManager : "<<(total_device>>20)<<" accelerator Mbytes "<<std::endl;
|
||||
std::cout << " MemoryManager : "<<(total_host>>20) <<" cpu Mbytes "<<std::endl;
|
||||
uint64_t cacheBytes;
|
||||
cacheBytes = CacheBytes[Cpu];
|
||||
std::cout << " MemoryManager : "<<(cacheBytes>>20) <<" cpu cache Mbytes "<<std::endl;
|
||||
cacheBytes = CacheBytes[Acc];
|
||||
std::cout << " MemoryManager : "<<(cacheBytes>>20) <<" acc cache Mbytes "<<std::endl;
|
||||
cacheBytes = CacheBytes[Shared];
|
||||
std::cout << " MemoryManager : "<<(cacheBytes>>20) <<" shared cache Mbytes "<<std::endl;
|
||||
cacheBytes = HostCacheBytes();
|
||||
std::cout << " MemoryManager : "<<(cacheBytes>>20) <<" cpu alloc cache Mbytes "<<std::endl;
|
||||
cacheBytes = DeviceCacheBytes();
|
||||
std::cout << " MemoryManager : "<<(cacheBytes>>20) <<" acc alloc cache Mbytes "<<std::endl;
|
||||
|
||||
#ifdef GRID_CUDA
|
||||
cuda_mem();
|
||||
|
||||
@@ -113,7 +113,7 @@ private:
|
||||
static void *Insert(void *ptr,size_t bytes,AllocationCacheEntry *entries,int ncache,int &victim,uint64_t &cbytes) ;
|
||||
static void *Lookup(size_t bytes,AllocationCacheEntry *entries,int ncache,uint64_t &cbytes) ;
|
||||
|
||||
public:
|
||||
public:
|
||||
static void PrintBytes(void);
|
||||
static void Audit(std::string s);
|
||||
static void Init(void);
|
||||
@@ -215,6 +215,7 @@ private:
|
||||
static void NotifyDeletion(void * CpuPtr);
|
||||
static void Print(void);
|
||||
static void PrintAll(void);
|
||||
static void EvictAll(void);
|
||||
static void PrintState( void* CpuPtr);
|
||||
static int isOpen (void* CpuPtr);
|
||||
static void ViewClose(void* CpuPtr,ViewMode mode);
|
||||
|
||||
@@ -79,6 +79,25 @@ void MemoryManager::EntryErase(uint64_t CpuPtr)
|
||||
auto AccCache = EntryLookup(CpuPtr);
|
||||
AccViewTable.erase(CpuPtr);
|
||||
}
|
||||
/////////////////////////////////////////////////////////////////////////////////
|
||||
// LRU membership invariant:
|
||||
//
|
||||
// LRU_valid == 1 <=> AccPtr != NULL && accLock == 0 && cpuLock == 0
|
||||
//
|
||||
// i.e. the LRU queue contains exactly the device-resident, completely unlocked
|
||||
// entries -- the evictable set. Membership is maintained EAGERLY at the lock
|
||||
// 0<->1 edges, O(1) via the stored LRU_entry iterator:
|
||||
//
|
||||
// AcceleratorViewOpen lock 0->1 : LRUremove (gated on LRU_valid)
|
||||
// AcceleratorViewClose accLock->0: LRUinsert (AccPtr necessarily exists)
|
||||
// CpuViewOpen lock 0->1 : LRUremove (gated on LRU_valid)
|
||||
// CpuViewClose cpuLock->0: LRUinsert (iff AccPtr exists)
|
||||
// Evict/AccDiscard : LRUremove (frees the device copy)
|
||||
//
|
||||
// Consequences: victims taken from LRU.back() are evictable by construction;
|
||||
// Evict() on a locked entry is an invariant violation (asserted), and the
|
||||
// eviction loops (EvictVictims/EvictAll) cannot spin.
|
||||
/////////////////////////////////////////////////////////////////////////////////
|
||||
void MemoryManager::LRUinsert(AcceleratorViewEntry &AccCache)
|
||||
{
|
||||
GRID_ASSERT(AccCache.LRU_valid==0);
|
||||
@@ -130,21 +149,21 @@ void MemoryManager::Evict(AcceleratorViewEntry &AccCache)
|
||||
{
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Make CPU consistent, remove from Accelerator, remove from LRU, LEAVE CPU only entry
|
||||
// Cannot be acclocked. If allocated must be in LRU pool.
|
||||
// Cannot be locked. If allocated must be in LRU pool.
|
||||
//
|
||||
// Nov 2022... Felix issue: Allocating two CpuPtrs, can have an entry in LRU-q with CPUlock.
|
||||
// and require to evict the AccPtr copy. Eviction was a mistake in CpuViewOpen
|
||||
// but there is a weakness where CpuLock entries are attempted for erase
|
||||
// Take these OUT LRU queue when CPU locked?
|
||||
// Cannot take out the table as cpuLock data is important.
|
||||
// (Historical: a Nov 2022 incident (two CpuPtrs; eviction called from
|
||||
// CpuViewOpen -- since excised) could present a cpuLocked entry here, and
|
||||
// silent-return guards were added. The LRU membership invariant (see
|
||||
// LRUinsert) now excludes ALL locked entries from the queue eagerly at the
|
||||
// lock edges, so a locked victim is an invariant violation: asserted.)
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
GRID_ASSERT(AccCache.state!=Empty);
|
||||
|
||||
mprintf("MemoryManager: Evict CpuPtr %lx AccPtr %lx cpuLock %ld accLock %ld",
|
||||
(uint64_t)AccCache.CpuPtr,(uint64_t)AccCache.AccPtr,
|
||||
(uint64_t)AccCache.cpuLock,(uint64_t)AccCache.accLock);
|
||||
if (AccCache.accLock!=0) return;
|
||||
if (AccCache.cpuLock!=0) return;
|
||||
GRID_ASSERT(AccCache.accLock==0);
|
||||
GRID_ASSERT(AccCache.cpuLock==0);
|
||||
if(AccCache.state==AccDirty) {
|
||||
Flush(AccCache);
|
||||
}
|
||||
@@ -250,6 +269,19 @@ void MemoryManager::EvictVictims(uint64_t bytes)
|
||||
}
|
||||
}
|
||||
}
|
||||
void MemoryManager::EvictAll(void)
|
||||
{
|
||||
while(LRU.size()>0){
|
||||
if ( DeviceLRUBytes > 0){
|
||||
uint64_t victim = LRU.back(); // From the LRU
|
||||
auto AccCacheIterator = EntryLookup(victim);
|
||||
auto & AccCache = AccCacheIterator->second;
|
||||
Evict(AccCache);
|
||||
} else {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
uint64_t MemoryManager::AcceleratorViewOpen(uint64_t CpuPtr,size_t bytes,ViewMode mode,ViewAdvise hint)
|
||||
{
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
@@ -383,6 +415,13 @@ void MemoryManager::CpuViewClose(uint64_t CpuPtr)
|
||||
GRID_ASSERT(AccCache.accLock==0);
|
||||
|
||||
AccCache.cpuLock--;
|
||||
// Return to LRU queue when fully unlocked -- mirrors AcceleratorViewClose.
|
||||
// Asymmetry vs the Acc side: a device copy need not exist for a host view;
|
||||
// only device-resident entries belong in the (evictable) LRU queue.
|
||||
if( (AccCache.cpuLock==0) && (AccCache.AccPtr!=(uint64_t)NULL) ) {
|
||||
dprintf("CpuViewClose %lx cpuLock decremented to zero, move to LRU queue",(uint64_t)CpuPtr);
|
||||
LRUinsert(AccCache);
|
||||
}
|
||||
}
|
||||
/*
|
||||
* Action State StateNext Flush Clone
|
||||
@@ -449,6 +488,14 @@ uint64_t MemoryManager::CpuViewOpen(uint64_t CpuPtr,size_t bytes,ViewMode mode,V
|
||||
GRID_ASSERT(0); // should be unreachable
|
||||
}
|
||||
|
||||
GRID_ASSERT(AccCache.cpuLock>0);
|
||||
// If view is opened on host must remove from LRU -- mirrors AcceleratorViewOpen.
|
||||
// LRU_valid==1 here implies this is the 0->1 lock edge of a device-resident entry.
|
||||
if(AccCache.LRU_valid==1){
|
||||
dprintf("CpuViewOpen: entry removed from LRU ");
|
||||
LRUremove(AccCache);
|
||||
}
|
||||
|
||||
AccCache.transient= transient? EvictNext : 0;
|
||||
|
||||
return AccCache.CpuPtr;
|
||||
|
||||
@@ -108,7 +108,7 @@ public:
|
||||
// very VERY rarely (Log, serial RNG) we need world without a grid
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
static int RankWorld(void) ;
|
||||
static void BroadcastWorld(int root,void* data, int bytes);
|
||||
static void BroadcastWorld(int root,void* data, uint64_t bytes);
|
||||
static void BarrierWorld(void);
|
||||
|
||||
////////////////////////////////////////////////////////////
|
||||
@@ -175,27 +175,27 @@ public:
|
||||
int dest,
|
||||
void *recv,
|
||||
int from,
|
||||
int bytes,int dir);
|
||||
uint64_t bytes,int dir);
|
||||
|
||||
void SendToRecvFrom(void *xmit,
|
||||
int xmit_to_rank,
|
||||
void *recv,
|
||||
int recv_from_rank,
|
||||
int bytes);
|
||||
uint64_t bytes);
|
||||
|
||||
int IsOffNode(int rank);
|
||||
double StencilSendToRecvFrom(void *xmit,
|
||||
int xmit_to_rank,int do_xmit,
|
||||
void *recv,
|
||||
int recv_from_rank,int do_recv,
|
||||
int bytes,int dir);
|
||||
uint64_t bytes,int dir);
|
||||
|
||||
double StencilSendToRecvFromPrepare(std::vector<CommsRequest_t> &list,
|
||||
void *xmit,
|
||||
int xmit_to_rank,int do_xmit,
|
||||
void *recv,
|
||||
int recv_from_rank,int do_recv,
|
||||
int xbytes,int rbytes,int dir);
|
||||
uint64_t xbytes,uint64_t rbytes,int dir);
|
||||
|
||||
// Could do a PollHtoD and have a CommsMerge dependence
|
||||
void StencilSendToRecvFromPollDtoH (std::vector<CommsRequest_t> &list);
|
||||
@@ -206,7 +206,7 @@ public:
|
||||
int xmit_to_rank,int do_xmit,
|
||||
void *recv,void *recv_comp,
|
||||
int recv_from_rank,int do_recv,
|
||||
int xbytes,int rbytes,int dir);
|
||||
uint64_t xbytes,uint64_t rbytes,int dir);
|
||||
|
||||
|
||||
void StencilSendToRecvFromComplete(std::vector<CommsRequest_t> &waitall,int i);
|
||||
@@ -220,7 +220,7 @@ public:
|
||||
////////////////////////////////////////////////////////////
|
||||
// Broadcast a buffer and composite larger
|
||||
////////////////////////////////////////////////////////////
|
||||
void Broadcast(int root,void* data, int bytes);
|
||||
void Broadcast(int root,void* data, uint64_t bytes);
|
||||
|
||||
////////////////////////////////////////////////////////////
|
||||
// All2All down one dimension
|
||||
@@ -238,6 +238,16 @@ public:
|
||||
}
|
||||
void AllToAll(int dim ,void *in,void *out,uint64_t words,uint64_t bytes);
|
||||
void AllToAll(void *in,void *out,uint64_t words ,uint64_t bytes);
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
// Variable count all to all. Counts and displacements are in units of
|
||||
// "bytes" sized words and are indexed by rank within this communicator.
|
||||
// For exchanges that are a permutation but do not divide evenly between
|
||||
// ranks; AllToAll above is the uniform count special case.
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
void AllToAllV(void *in ,const std::vector<int> &sendcounts,const std::vector<int> &senddispls,
|
||||
void *out,const std::vector<int> &recvcounts,const std::vector<int> &recvdispls,
|
||||
uint64_t bytes);
|
||||
|
||||
template<class obj> void Broadcast(int root,obj &data)
|
||||
{
|
||||
|
||||
@@ -342,23 +342,22 @@ void CartesianCommunicator::SendToRecvFromBegin(std::vector<MpiCommsRequest_t> &
|
||||
int dest,
|
||||
void *recv,
|
||||
int from,
|
||||
int bytes,int dir)
|
||||
uint64_t bytes,int dir)
|
||||
{
|
||||
MPI_Request xrq;
|
||||
MPI_Request rrq;
|
||||
|
||||
GRID_ASSERT(dest != _processor);
|
||||
GRID_ASSERT(from != _processor);
|
||||
|
||||
int tag;
|
||||
|
||||
tag= dir+from*32;
|
||||
int ierr=MPI_Irecv(recv, bytes, MPI_CHAR,from,tag,communicator,&rrq);
|
||||
int ierr=MPI_Irecv(recv,(int)( bytes/sizeof(int32_t)), MPI_INT32_T,from,tag,communicator,&rrq);
|
||||
GRID_ASSERT(ierr==0);
|
||||
list.push_back(rrq);
|
||||
|
||||
tag= dir+_processor*32;
|
||||
ierr =MPI_Isend(xmit, bytes, MPI_CHAR,dest,tag,communicator,&xrq);
|
||||
ierr =MPI_Isend(xmit,(int)(bytes/sizeof(int32_t)), MPI_INT32_T,dest,tag,communicator,&xrq);
|
||||
GRID_ASSERT(ierr==0);
|
||||
list.push_back(xrq);
|
||||
}
|
||||
@@ -379,7 +378,7 @@ void CartesianCommunicator::SendToRecvFrom(void *xmit,
|
||||
int dest,
|
||||
void *recv,
|
||||
int from,
|
||||
int bytes)
|
||||
uint64_t bytes)
|
||||
{
|
||||
std::vector<MpiCommsRequest_t> reqs(0);
|
||||
|
||||
@@ -392,8 +391,8 @@ void CartesianCommunicator::SendToRecvFrom(void *xmit,
|
||||
|
||||
// Give the CPU to MPI immediately; can use threads to overlap optionally
|
||||
// printf("proc %d SendToRecvFrom %d bytes Sendrecv \n",_processor,bytes);
|
||||
ierr=MPI_Sendrecv(xmit,bytes,MPI_CHAR,dest,myrank,
|
||||
recv,bytes,MPI_CHAR,from, from,
|
||||
ierr=MPI_Sendrecv(xmit,(int)(bytes/sizeof(int32_t)),MPI_INT32_T,dest,myrank,
|
||||
recv,(int)(bytes/sizeof(int32_t)),MPI_INT32_T,from, from,
|
||||
communicator,MPI_STATUS_IGNORE);
|
||||
GRID_ASSERT(ierr==0);
|
||||
|
||||
@@ -403,7 +402,7 @@ double CartesianCommunicator::StencilSendToRecvFrom( void *xmit,
|
||||
int dest, int dox,
|
||||
void *recv,
|
||||
int from, int dor,
|
||||
int bytes,int dir)
|
||||
uint64_t bytes,int dir)
|
||||
{
|
||||
std::vector<CommsRequest_t> list;
|
||||
double offbytes = StencilSendToRecvFromPrepare(list,xmit,dest,dox,recv,from,dor,bytes,bytes,dir);
|
||||
@@ -426,7 +425,7 @@ double CartesianCommunicator::StencilSendToRecvFromPrepare(std::vector<CommsRequ
|
||||
int dest,int dox,
|
||||
void *recv,
|
||||
int from,int dor,
|
||||
int xbytes,int rbytes,int dir)
|
||||
uint64_t xbytes,uint64_t rbytes,int dir)
|
||||
{
|
||||
return 0.0; // Do nothing -- no preparation required
|
||||
}
|
||||
@@ -435,7 +434,7 @@ double CartesianCommunicator::StencilSendToRecvFromBegin(std::vector<CommsReques
|
||||
int dest,int dox,
|
||||
void *recv,void *recv_comp,
|
||||
int from,int dor,
|
||||
int xbytes,int rbytes,int dir)
|
||||
uint64_t xbytes,uint64_t rbytes,int dir)
|
||||
{
|
||||
int ncomm =communicator_halo.size();
|
||||
int commdir=dir%ncomm;
|
||||
@@ -458,7 +457,7 @@ double CartesianCommunicator::StencilSendToRecvFromBegin(std::vector<CommsReques
|
||||
if ( (gfrom ==MPI_UNDEFINED) || Stencil_force_mpi ) {
|
||||
tag= dir+from*32;
|
||||
// std::cout << " StencilSendToRecvFrom "<<dir<<" MPI_Irecv "<<std::hex<<recv<<std::dec<<std::endl;
|
||||
ierr=MPI_Irecv(recv_comp, rbytes, MPI_CHAR,from,tag,communicator_halo[commdir],&rrq);
|
||||
ierr=MPI_Irecv(recv_comp,(int)(rbytes/sizeof(int32_t)), MPI_INT32_T,from,tag,communicator_halo[commdir],&rrq);
|
||||
GRID_ASSERT(ierr==0);
|
||||
list.push_back(rrq);
|
||||
off_node_bytes+=rbytes;
|
||||
@@ -476,7 +475,7 @@ double CartesianCommunicator::StencilSendToRecvFromBegin(std::vector<CommsReques
|
||||
if (dox) {
|
||||
if ( (gdest == MPI_UNDEFINED) || Stencil_force_mpi ) {
|
||||
tag= dir+_processor*32;
|
||||
ierr =MPI_Isend(xmit_comp, xbytes, MPI_CHAR,dest,tag,communicator_halo[commdir],&xrq);
|
||||
ierr =MPI_Isend(xmit_comp,(int)(xbytes/sizeof(int32_t)), MPI_INT32_T,dest,tag,communicator_halo[commdir],&xrq);
|
||||
GRID_ASSERT(ierr==0);
|
||||
list.push_back(xrq);
|
||||
off_node_bytes+=xbytes;
|
||||
@@ -543,7 +542,7 @@ double CartesianCommunicator::StencilSendToRecvFromPrepare(std::vector<CommsRequ
|
||||
int dest,int dox,
|
||||
void *recv,
|
||||
int from,int dor,
|
||||
int xbytes,int rbytes,int dir)
|
||||
uint64_t xbytes,uint64_t rbytes,int dir)
|
||||
{
|
||||
/*
|
||||
* Bring sequence from Stencil.h down to lower level.
|
||||
@@ -584,7 +583,7 @@ double CartesianCommunicator::StencilSendToRecvFromPrepare(std::vector<CommsRequ
|
||||
if ( (gfrom ==MPI_UNDEFINED) || Stencil_force_mpi ) {
|
||||
tag= dir+from*32;
|
||||
host_recv = this->HostBufferMalloc(rbytes);
|
||||
ierr=MPI_Irecv(host_recv, rbytes, MPI_CHAR,from,tag,communicator_halo[commdir],&rrq);
|
||||
ierr=MPI_Irecv(host_recv,(int)(rbytes/sizeof(int32_t)), MPI_INT32_T,from,tag,communicator_halo[commdir],&rrq);
|
||||
GRID_ASSERT(ierr==0);
|
||||
CommsRequest_t srq;
|
||||
srq.PacketType = InterNodeRecv;
|
||||
@@ -686,7 +685,7 @@ void CartesianCommunicator::StencilSendToRecvFromPollDtoH(std::vector<CommsReque
|
||||
if ( acceleratorEventIsComplete(list[idx].ev) ) {
|
||||
|
||||
void *host_xmit = list[idx].host_buf;
|
||||
uint32_t xbytes = list[idx].bytes;
|
||||
uint64_t xbytes = list[idx].bytes;
|
||||
int dest = list[idx].dest;
|
||||
int tag = list[idx].tag;
|
||||
int commdir = list[idx].commdir;
|
||||
@@ -697,7 +696,7 @@ void CartesianCommunicator::StencilSendToRecvFromPollDtoH(std::vector<CommsReque
|
||||
// std::cout << " DtoH is complete for index "<<idx<<" calling MPI_Isend "<<std::endl;
|
||||
|
||||
MPI_Request xrq;
|
||||
int ierr =MPI_Isend(host_xmit, xbytes, MPI_CHAR,dest,tag,communicator_halo[commdir],&xrq);
|
||||
int ierr =MPI_Isend(host_xmit, (int)(xbytes/sizeof(int32_t)), MPI_INT32_T,dest,tag,communicator_halo[commdir],&xrq);
|
||||
GRID_ASSERT(ierr==0);
|
||||
|
||||
list[idx].req = xrq; // Update the MPI request in the list
|
||||
@@ -718,7 +717,7 @@ double CartesianCommunicator::StencilSendToRecvFromBegin(std::vector<CommsReques
|
||||
int dest,int dox,
|
||||
void *recv,void *recv_comp,
|
||||
int from,int dor,
|
||||
int xbytes,int rbytes,int dir)
|
||||
uint64_t xbytes,uint64_t rbytes,int dir)
|
||||
{
|
||||
int ncomm =communicator_halo.size();
|
||||
int commdir=dir%ncomm;
|
||||
@@ -884,11 +883,11 @@ void CartesianCommunicator::Barrier(void)
|
||||
int ierr = MPI_Barrier(communicator);
|
||||
GRID_ASSERT(ierr==0);
|
||||
}
|
||||
void CartesianCommunicator::Broadcast(int root,void* data, int bytes)
|
||||
void CartesianCommunicator::Broadcast(int root,void* data,uint64_t bytes)
|
||||
{
|
||||
FlightRecorder::StepLog("Broadcast");
|
||||
int ierr=MPI_Bcast(data,
|
||||
bytes,
|
||||
(int)bytes,
|
||||
MPI_BYTE,
|
||||
root,
|
||||
communicator);
|
||||
@@ -904,11 +903,11 @@ void CartesianCommunicator::BarrierWorld(void){
|
||||
int ierr = MPI_Barrier(communicator_world);
|
||||
GRID_ASSERT(ierr==0);
|
||||
}
|
||||
void CartesianCommunicator::BroadcastWorld(int root,void* data, int bytes)
|
||||
void CartesianCommunicator::BroadcastWorld(int root,void* data, uint64_t bytes)
|
||||
{
|
||||
FlightRecorder::StepLog("BroadcastWorld");
|
||||
int ierr= MPI_Bcast(data,
|
||||
bytes,
|
||||
(int)bytes,
|
||||
MPI_BYTE,
|
||||
root,
|
||||
communicator_world);
|
||||
@@ -946,5 +945,25 @@ void CartesianCommunicator::AllToAll(void *in,void *out,uint64_t words,uint64_t
|
||||
MPI_Alltoall(in,iwords,object,out,iwords,object,communicator);
|
||||
MPI_Type_free(&object);
|
||||
}
|
||||
void CartesianCommunicator::AllToAllV(void *in ,const std::vector<int> &sendcounts,const std::vector<int> &senddispls,
|
||||
void *out,const std::vector<int> &recvcounts,const std::vector<int> &recvdispls,
|
||||
uint64_t bytes)
|
||||
{
|
||||
FlightRecorder::StepLog("AllToAllV");
|
||||
GRID_ASSERT(sendcounts.size()==(size_t)_Nprocessors);
|
||||
GRID_ASSERT(senddispls.size()==(size_t)_Nprocessors);
|
||||
GRID_ASSERT(recvcounts.size()==(size_t)_Nprocessors);
|
||||
GRID_ASSERT(recvdispls.size()==(size_t)_Nprocessors);
|
||||
// MPI counts are "int"; the caller sizes the word to keep them in range
|
||||
int ibytes = bytes;
|
||||
GRID_ASSERT(bytes == (uint64_t)ibytes);
|
||||
MPI_Datatype object;
|
||||
MPI_Type_contiguous(ibytes,MPI_BYTE,&object);
|
||||
MPI_Type_commit(&object);
|
||||
int ierr = MPI_Alltoallv(in ,(int *)&sendcounts[0],(int *)&senddispls[0],object,
|
||||
out,(int *)&recvcounts[0],(int *)&recvdispls[0],object,communicator);
|
||||
GRID_ASSERT(ierr==0);
|
||||
MPI_Type_free(&object);
|
||||
}
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
|
||||
@@ -27,6 +27,8 @@ Author: Peter Boyle <paboyle@ph.ed.ac.uk>
|
||||
/* END LEGAL */
|
||||
#include <Grid/GridCore.h>
|
||||
|
||||
void GridAbort(void) { abort(); }
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
@@ -34,7 +36,6 @@ NAMESPACE_BEGIN(Grid);
|
||||
///////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
Grid_MPI_Comm CartesianCommunicator::communicator_world;
|
||||
|
||||
void GridAbort(void) { abort(); }
|
||||
|
||||
void CartesianCommunicator::Init(int *argc, char *** arv)
|
||||
{
|
||||
@@ -89,7 +90,7 @@ void CartesianCommunicator::SendToRecvFrom(void *xmit,
|
||||
int dest,
|
||||
void *recv,
|
||||
int from,
|
||||
int bytes)
|
||||
uint64_t bytes)
|
||||
{
|
||||
GRID_ASSERT(0);
|
||||
}
|
||||
@@ -99,7 +100,7 @@ void CartesianCommunicator::SendToRecvFromBegin(std::vector<CommsRequest_t> &lis
|
||||
int dest,
|
||||
void *recv,
|
||||
int from,
|
||||
int bytes,int dir)
|
||||
uint64_t bytes,int dir)
|
||||
{
|
||||
GRID_ASSERT(0);
|
||||
}
|
||||
@@ -112,11 +113,22 @@ void CartesianCommunicator::AllToAll(void *in,void *out,uint64_t words,uint64_t
|
||||
{
|
||||
bcopy(in,out,bytes*words);
|
||||
}
|
||||
void CartesianCommunicator::AllToAllV(void *in ,const std::vector<int> &sendcounts,const std::vector<int> &senddispls,
|
||||
void *out,const std::vector<int> &recvcounts,const std::vector<int> &recvdispls,
|
||||
uint64_t bytes)
|
||||
{
|
||||
// Single rank: the exchange degenerates to a copy of our own segment
|
||||
GRID_ASSERT(sendcounts.size()==1);
|
||||
GRID_ASSERT(recvcounts.size()==1);
|
||||
GRID_ASSERT(sendcounts[0]==recvcounts[0]);
|
||||
bcopy((char *)in +(uint64_t)senddispls[0]*bytes,
|
||||
(char *)out+(uint64_t)recvdispls[0]*bytes,bytes*(uint64_t)sendcounts[0]);
|
||||
}
|
||||
|
||||
int CartesianCommunicator::RankWorld(void){return 0;}
|
||||
void CartesianCommunicator::Barrier(void){}
|
||||
void CartesianCommunicator::Broadcast(int root,void* data, int bytes) {}
|
||||
void CartesianCommunicator::BroadcastWorld(int root,void* data, int bytes) { }
|
||||
void CartesianCommunicator::Broadcast(int root,void* data, uint64_t bytes) {}
|
||||
void CartesianCommunicator::BroadcastWorld(int root,void* data, uint64_t bytes) { }
|
||||
void CartesianCommunicator::BarrierWorld(void) { }
|
||||
int CartesianCommunicator::RankFromProcessorCoor(Coordinate &coor) { return 0;}
|
||||
void CartesianCommunicator::ProcessorCoorFromRank(int rank, Coordinate &coor){ coor = _processor_coor; }
|
||||
@@ -132,7 +144,7 @@ double CartesianCommunicator::StencilSendToRecvFrom( void *xmit,
|
||||
int xmit_to_rank,int dox,
|
||||
void *recv,
|
||||
int recv_from_rank,int dor,
|
||||
int bytes, int dir)
|
||||
uint64_t bytes, int dir)
|
||||
{
|
||||
return 2.0*bytes;
|
||||
}
|
||||
@@ -143,16 +155,16 @@ double CartesianCommunicator::StencilSendToRecvFromPrepare(std::vector<CommsRequ
|
||||
int xmit_to_rank,int dox,
|
||||
void *recv,
|
||||
int recv_from_rank,int dor,
|
||||
int xbytes,int rbytes, int dir)
|
||||
uint64_t xbytes,uint64_t rbytes, int dir)
|
||||
{
|
||||
return 0.0;
|
||||
}
|
||||
double CartesianCommunicator::StencilSendToRecvFromBegin(std::vector<CommsRequest_t> &list,
|
||||
void *xmit,
|
||||
void *xmit, void *xmit_comp,
|
||||
int xmit_to_rank,int dox,
|
||||
void *recv,
|
||||
void *recv, void *recv_comp,
|
||||
int recv_from_rank,int dor,
|
||||
int xbytes,int rbytes, int dir)
|
||||
uint64_t xbytes,uint64_t rbytes, int dir)
|
||||
{
|
||||
return xbytes+rbytes;
|
||||
}
|
||||
|
||||
@@ -49,6 +49,20 @@ template<class vobj> Lattice<vobj> Cshift(const Lattice<vobj> &rhs,int dimension
|
||||
// Map to always positive shift modulo global full dimension.
|
||||
shift = (shift+fd)%fd;
|
||||
|
||||
if( shift ==0 ) {
|
||||
ret = rhs;
|
||||
return ret;
|
||||
}
|
||||
//
|
||||
// Potential easy fast cases:
|
||||
// Shift is a multiple of the local lattice extent.
|
||||
// Then need only to shift whole subvolumes
|
||||
int L = rhs.Grid()->_ldimensions[dimension];
|
||||
if ( (shift%L )==0 && !rhs.Grid()->CheckerBoarded(dimension) ) {
|
||||
Cshift_simple(ret,rhs,dimension,shift);
|
||||
return ret;
|
||||
}
|
||||
|
||||
ret.Checkerboard() = rhs.Grid()->CheckerBoardDestination(rhs.Checkerboard(),shift,dimension);
|
||||
|
||||
// the permute type
|
||||
@@ -73,6 +87,55 @@ template<class vobj> Lattice<vobj> Cshift(const Lattice<vobj> &rhs,int dimension
|
||||
return ret;
|
||||
}
|
||||
|
||||
template<class vobj> void Cshift_simple(Lattice<vobj>& ret,const Lattice<vobj> &rhs,int dimension,int shift)
|
||||
{
|
||||
GridBase *grid=rhs.Grid();
|
||||
int comm_proc, xmit_to_rank, recv_from_rank;
|
||||
|
||||
int fd = rhs.Grid()->_fdimensions[dimension];
|
||||
int rd = rhs.Grid()->_rdimensions[dimension];
|
||||
int ld = rhs.Grid()->_ldimensions[dimension];
|
||||
int pd = rhs.Grid()->_processors[dimension];
|
||||
int simd_layout = rhs.Grid()->_simd_layout[dimension];
|
||||
int comm_dim = rhs.Grid()->_processors[dimension] >1 ;
|
||||
|
||||
comm_proc = ((shift)/ld)%pd;
|
||||
|
||||
grid->ShiftedRanks(dimension,comm_proc,xmit_to_rank,recv_from_rank);
|
||||
if(comm_dim) {
|
||||
|
||||
int64_t bytes = sizeof(vobj) * grid->oSites();
|
||||
|
||||
autoView(rhs_v , rhs, AcceleratorRead);
|
||||
autoView(ret_v , ret, AcceleratorWrite);
|
||||
void *send_buf = (void *)&rhs_v[0];
|
||||
void *recv_buf = (void *)&ret_v[0];
|
||||
|
||||
#ifdef ACCELERATOR_AWARE_MPI
|
||||
grid->SendToRecvFrom(send_buf,
|
||||
xmit_to_rank,
|
||||
recv_buf,
|
||||
recv_from_rank,
|
||||
bytes);
|
||||
#else
|
||||
static hostVector<vobj> hrhs; hrhs.resize(grid->oSites());
|
||||
static hostVector<vobj> hret; hret.resize(grid->oSites());
|
||||
|
||||
void *hsend_buf = (void *)&hrhs[0];
|
||||
void *hrecv_buf = (void *)&hret[0];
|
||||
|
||||
acceleratorCopyFromDevice(send_buf,hsend_buf,bytes);
|
||||
|
||||
grid->SendToRecvFrom(hsend_buf,
|
||||
xmit_to_rank,
|
||||
hrecv_buf,
|
||||
recv_from_rank,
|
||||
bytes);
|
||||
|
||||
acceleratorCopyToDevice(hrecv_buf,recv_buf,bytes);
|
||||
#endif
|
||||
}
|
||||
}
|
||||
template<class vobj> void Cshift_comms(Lattice<vobj>& ret,const Lattice<vobj> &rhs,int dimension,int shift)
|
||||
{
|
||||
int sshift[2];
|
||||
|
||||
@@ -289,7 +289,7 @@ public:
|
||||
///////////////////////////////////////////
|
||||
// move constructor
|
||||
///////////////////////////////////////////
|
||||
Lattice(Lattice && r){
|
||||
Lattice(Lattice && r) noexcept {
|
||||
this->_grid = r.Grid();
|
||||
this->_odata = r._odata;
|
||||
this->_odata_size = r._odata_size;
|
||||
@@ -330,7 +330,7 @@ public:
|
||||
///////////////////////////////////////////
|
||||
// Move assignment possible if same type
|
||||
///////////////////////////////////////////
|
||||
inline Lattice<vobj> & operator = (Lattice<vobj> && r){
|
||||
inline Lattice<vobj> & operator = (Lattice<vobj> && r) noexcept {
|
||||
|
||||
resize(0); // deletes if appropriate
|
||||
this->_grid = r.Grid();
|
||||
|
||||
@@ -197,12 +197,15 @@ __global__ void reduceKernel(const vobj *lat, sobj *buffer, Iterator n) {
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
// Possibly promote to double and sum
|
||||
/////////////////////////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
#undef GRID_REDUCTION_TIMING
|
||||
|
||||
template <class vobj>
|
||||
inline typename vobj::scalar_objectD sumD_gpu_small(const vobj *lat, Integer osites)
|
||||
inline typename vobj::scalar_objectD sumD_gpu_small(const vobj *lat, Integer osites)
|
||||
{
|
||||
typedef typename vobj::scalar_objectD sobj;
|
||||
typedef decltype(lat) Iterator;
|
||||
|
||||
|
||||
Integer nsimd= vobj::Nsimd();
|
||||
Integer size = osites*nsimd;
|
||||
|
||||
@@ -211,41 +214,188 @@ inline typename vobj::scalar_objectD sumD_gpu_small(const vobj *lat, Integer osi
|
||||
GRID_ASSERT(ok);
|
||||
|
||||
Integer smemSize = numThreads * sizeof(sobj);
|
||||
// Move out of UVM
|
||||
// Turns out I had messed up the synchronise after move to compute stream
|
||||
// as running this on the default stream fools the synchronise
|
||||
deviceVector<sobj> buffer(numBlocks);
|
||||
sobj *buffer_v = &buffer[0];
|
||||
sobj result;
|
||||
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
RealD t_kernel = -usecond();
|
||||
#endif
|
||||
reduceKernel<<< numBlocks, numThreads, smemSize, computeStream >>>(lat, buffer_v, size);
|
||||
accelerator_barrier();
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
t_kernel += usecond();
|
||||
RealD t_d2h = -usecond();
|
||||
#endif
|
||||
acceleratorCopyFromDevice(buffer_v,&result,sizeof(result));
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
t_d2h += usecond();
|
||||
std::cout << GridLogDebug << " sumD_gpu_small"
|
||||
<< " sizeof(sobj)=" << sizeof(sobj)
|
||||
<< " blocks=" << numBlocks << " threads=" << numThreads
|
||||
<< " kernel+barrier=" << t_kernel << " us"
|
||||
<< " D2H=" << t_d2h << " us" << std::endl;
|
||||
#endif
|
||||
return result;
|
||||
}
|
||||
|
||||
// Fused pack+reduce: reads R words of each vobj at word offset 'base',
|
||||
// accumulates directly into iVector<iScalar<scalarD>,R> without staging
|
||||
// through an intermediate bundle buffer. One HBM pass instead of three.
|
||||
template <int R, class vobj, class sobj, class Iterator>
|
||||
__device__ void packReduceBlocks(
|
||||
const iScalar<typename vobj::vector_type> *idat,
|
||||
sobj *g_odata, Iterator osites, int base, int words)
|
||||
{
|
||||
constexpr Iterator nsimd = vobj::Nsimd();
|
||||
Iterator blockSize = blockDim.x;
|
||||
|
||||
extern __shared__ __align__(COALESCE_GRANULARITY) unsigned char shmem_pointer[];
|
||||
sobj *sdata = (sobj *)shmem_pointer;
|
||||
|
||||
Iterator tid = threadIdx.x;
|
||||
Iterator i = blockIdx.x * (blockSize * 2) + threadIdx.x;
|
||||
Iterator gridSize = blockSize * 2 * gridDim.x;
|
||||
sobj mySum = Zero();
|
||||
|
||||
while (i < osites * nsimd) {
|
||||
Iterator lane = i % nsimd;
|
||||
Iterator ss = i / nsimd;
|
||||
sobj tmpD; zeroit(tmpD);
|
||||
for (int k = 0; k < R; k++) {
|
||||
auto w = extractLane(lane, idat[ss * words + base + k]);
|
||||
iScalar<typename vobj::scalar_typeD> wd; wd = w;
|
||||
tmpD._internal[k] = wd;
|
||||
}
|
||||
mySum += tmpD;
|
||||
|
||||
if (i + blockSize < osites * nsimd) {
|
||||
lane = (i + blockSize) % nsimd;
|
||||
ss = (i + blockSize) / nsimd;
|
||||
sobj tmpD2; zeroit(tmpD2);
|
||||
for (int k = 0; k < R; k++) {
|
||||
auto w = extractLane(lane, idat[ss * words + base + k]);
|
||||
iScalar<typename vobj::scalar_typeD> wd; wd = w;
|
||||
tmpD2._internal[k] = wd;
|
||||
}
|
||||
mySum += tmpD2;
|
||||
}
|
||||
i += gridSize;
|
||||
}
|
||||
|
||||
reduceBlock(sdata, mySum, tid);
|
||||
if (tid == 0) g_odata[blockIdx.x] = sdata[0];
|
||||
}
|
||||
|
||||
template <int R, class vobj, class sobj, class Iterator>
|
||||
__global__ void packReduceKernel(
|
||||
const iScalar<typename vobj::vector_type> *idat,
|
||||
sobj *buffer, Iterator osites, int base, int words)
|
||||
{
|
||||
Iterator blockSize = blockDim.x;
|
||||
|
||||
packReduceBlocks<R, vobj, sobj>(idat, buffer, osites, base, words);
|
||||
|
||||
if (gridDim.x > 1) {
|
||||
const Iterator tid = threadIdx.x;
|
||||
__shared__ bool amLast;
|
||||
extern __shared__ __align__(COALESCE_GRANULARITY) unsigned char shmem_pointer[];
|
||||
sobj *smem = (sobj *)shmem_pointer;
|
||||
|
||||
acceleratorFence();
|
||||
|
||||
if (tid == 0) {
|
||||
unsigned int ticket = atomicInc(&retirementCount, gridDim.x);
|
||||
amLast = (ticket == gridDim.x - 1);
|
||||
}
|
||||
acceleratorSynchroniseAll();
|
||||
|
||||
if (amLast) {
|
||||
Iterator i = tid;
|
||||
sobj mySum = Zero();
|
||||
while (i < (Iterator)gridDim.x) {
|
||||
mySum += buffer[i];
|
||||
i += blockSize;
|
||||
}
|
||||
reduceBlock(smem, mySum, tid);
|
||||
if (tid == 0) {
|
||||
buffer[0] = smem[0];
|
||||
retirementCount = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
template<int R, class vobj>
|
||||
inline void sumD_gpu_reduce_words(const vobj *lat, Integer osites,
|
||||
typename vobj::scalar_typeD *ret_p, int base)
|
||||
{
|
||||
typedef typename vobj::vector_type vector;
|
||||
typedef typename vobj::scalar_typeD scalarD;
|
||||
using BundleScalarD = iVector<iScalar<scalarD>, R>;
|
||||
|
||||
constexpr int Nsimd = vobj::Nsimd();
|
||||
const int words = sizeof(vobj) / sizeof(vector);
|
||||
const iScalar<vector> *idat = (const iScalar<vector> *)lat;
|
||||
|
||||
Integer size = (Integer)osites * Nsimd;
|
||||
Integer numThreads, numBlocks;
|
||||
int ok = getNumBlocksAndThreads(size, sizeof(BundleScalarD), numThreads, numBlocks);
|
||||
GRID_ASSERT(ok);
|
||||
|
||||
Integer smemSize = numThreads * sizeof(BundleScalarD);
|
||||
deviceVector<BundleScalarD> buffer(numBlocks);
|
||||
BundleScalarD *buffer_v = &buffer[0];
|
||||
BundleScalarD result;
|
||||
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
RealD t_kernel = -usecond();
|
||||
#endif
|
||||
packReduceKernel<R, vobj, BundleScalarD, Integer>
|
||||
<<<numBlocks, numThreads, smemSize, computeStream>>>
|
||||
(idat, buffer_v, osites, base, words);
|
||||
accelerator_barrier();
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
t_kernel += usecond();
|
||||
RealD t_d2h = -usecond();
|
||||
#endif
|
||||
acceleratorCopyFromDevice(buffer_v, &result, sizeof(result));
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
t_d2h += usecond();
|
||||
std::cout << GridLogDebug << " sumD_gpu_reduce_words R=" << R
|
||||
<< " base=" << base
|
||||
<< " kernel=" << t_kernel << " D2H=" << t_d2h << " us" << std::endl;
|
||||
#endif
|
||||
|
||||
for (int k = 0; k < R; k++)
|
||||
ret_p[base + k] = TensorRemove(result._internal[k]);
|
||||
}
|
||||
|
||||
template <class vobj>
|
||||
inline typename vobj::scalar_objectD sumD_gpu_large(const vobj *lat, Integer osites)
|
||||
{
|
||||
typedef typename vobj::vector_type vector;
|
||||
typedef typename vobj::scalar_typeD scalarD;
|
||||
typedef typename vobj::scalar_objectD sobj;
|
||||
sobj ret;
|
||||
typedef typename vobj::vector_type vector;
|
||||
typedef typename vobj::scalar_typeD scalarD;
|
||||
typedef typename vobj::scalar_objectD sobjD;
|
||||
|
||||
const int words = sizeof(vobj) / sizeof(vector);
|
||||
sobjD ret; zeroit(ret);
|
||||
scalarD *ret_p = (scalarD *)&ret;
|
||||
|
||||
const int words = sizeof(vobj)/sizeof(vector);
|
||||
|
||||
deviceVector<vector> buffer(osites);
|
||||
vector *dat = (vector *)lat;
|
||||
vector *buf = &buffer[0];
|
||||
iScalar<vector> *tbuf =(iScalar<vector> *) &buffer[0];
|
||||
for(int w=0;w<words;w++) {
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
RealD t_large = -usecond();
|
||||
#endif
|
||||
int w = 0;
|
||||
while (w + 12 <= words) { sumD_gpu_reduce_words<12>(lat, osites, ret_p, w); w += 12; }
|
||||
while (w + 4 <= words) { sumD_gpu_reduce_words< 4>(lat, osites, ret_p, w); w += 4; }
|
||||
while (w < words) { sumD_gpu_reduce_words< 1>(lat, osites, ret_p, w); w += 1; }
|
||||
#ifdef GRID_REDUCTION_TIMING
|
||||
t_large += usecond();
|
||||
std::cout << GridLogDebug << "sumD_gpu_large"
|
||||
<< " sizeof(sobjD)=" << sizeof(sobjD)
|
||||
<< " words=" << words << " total=" << t_large << " us" << std::endl;
|
||||
#endif
|
||||
|
||||
accelerator_for(ss,osites,1,{
|
||||
buf[ss] = dat[ss*words+w];
|
||||
});
|
||||
|
||||
ret_p[w] = sumD_gpu_small(tbuf,osites);
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
@@ -288,5 +438,11 @@ inline typename vobj::scalar_object sum_gpu_large(const vobj *lat, Integer osite
|
||||
result = sumD_gpu_large(lat,osites);
|
||||
return result;
|
||||
}
|
||||
template<class Word> Word checksum_gpu(Word *vec,uint64_t L)
|
||||
{
|
||||
Word w;
|
||||
bzero(&w,sizeof(w));
|
||||
return w;
|
||||
}
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
|
||||
@@ -6,28 +6,27 @@ NAMESPACE_BEGIN(Grid);
|
||||
|
||||
|
||||
template <class vobj>
|
||||
inline typename vobj::scalar_objectD sumD_gpu_tensor(const vobj *lat, Integer osites)
|
||||
inline typename vobj::scalar_objectD sumD_gpu_tensor(const vobj *lat, Integer osites)
|
||||
{
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::scalar_objectD sobjD;
|
||||
|
||||
sobj identity; zeroit(identity);
|
||||
sobj ret; zeroit(ret);
|
||||
Integer nsimd= vobj::Nsimd();
|
||||
{
|
||||
sycl::buffer<sobj, 1> abuff(&ret, {1});
|
||||
sobjD identity; zeroit(identity);
|
||||
sobjD ret; zeroit(ret);
|
||||
{
|
||||
sycl::buffer<sobjD, 1> abuff(&ret, {1});
|
||||
theGridAccelerator->submit([&](sycl::handler &cgh) {
|
||||
auto Reduction = sycl::reduction(abuff,cgh,identity,std::plus<>());
|
||||
cgh.parallel_for(sycl::range<1>{osites},
|
||||
Reduction,
|
||||
[=] (sycl::id<1> item, auto &sum) {
|
||||
auto osite = item[0];
|
||||
sum +=Reduce(lat[osite]);
|
||||
});
|
||||
auto Reduction = sycl::reduction(abuff, cgh, identity, std::plus<>());
|
||||
cgh.parallel_for(sycl::range<1>{(size_t)osites},
|
||||
Reduction,
|
||||
[=](sycl::id<1> item, auto &sum) {
|
||||
sobj s = Reduce(lat[item[0]]);
|
||||
sobjD sd; sd = s;
|
||||
sum += sd;
|
||||
});
|
||||
});
|
||||
}
|
||||
sobjD dret; convertType(dret,ret);
|
||||
return dret;
|
||||
return ret;
|
||||
}
|
||||
|
||||
template <class vobj>
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#pragma once
|
||||
|
||||
#if defined(GRID_CUDA)
|
||||
|
||||
#include <cub/cub.cuh>
|
||||
#define gpucub cub
|
||||
#define gpuError_t cudaError_t
|
||||
@@ -57,8 +56,13 @@ inline void sliceSumReduction_cub_small(const vobj *Data,
|
||||
//copy offsets to device
|
||||
acceleratorCopyToDeviceAsynch(&offsets[0],d_offsets,sizeof(int)*(rd+1),computeStream);
|
||||
|
||||
#if defined(__CUDACC__) && (__CUDACC_VER_MAJOR__ >= 13)
|
||||
#define GRID_CUB_SUM_OP ::cuda::std::plus<>{}
|
||||
#else
|
||||
#define GRID_CUB_SUM_OP ::gpucub::Sum()
|
||||
#endif
|
||||
|
||||
gpuError_t gpuErr = gpucub::DeviceSegmentedReduce::Reduce(temp_storage_array, temp_storage_bytes, rb_p,d_out, rd, d_offsets, d_offsets+1, ::gpucub::Sum(), zero_init, computeStream);
|
||||
gpuError_t gpuErr = gpucub::DeviceSegmentedReduce::Reduce(temp_storage_array, temp_storage_bytes, rb_p,d_out, rd, d_offsets, d_offsets+1, GRID_CUB_SUM_OP, zero_init, computeStream);
|
||||
if (gpuErr!=gpuSuccess) {
|
||||
std::cout << GridLogError << "Lattice_slicesum_gpu.h: Encountered error during gpucub::DeviceSegmentedReduce::Reduce (setup)! Error: " << gpuErr <<std::endl;
|
||||
exit(EXIT_FAILURE);
|
||||
@@ -82,11 +86,13 @@ inline void sliceSumReduction_cub_small(const vobj *Data,
|
||||
});
|
||||
|
||||
//issue segmented reductions in computeStream
|
||||
gpuErr = gpucub::DeviceSegmentedReduce::Reduce(temp_storage_array, temp_storage_bytes, rb_p, d_out, rd, d_offsets, d_offsets+1,::gpucub::Sum(), zero_init, computeStream);
|
||||
gpuErr = gpucub::DeviceSegmentedReduce::Reduce(temp_storage_array, temp_storage_bytes, rb_p, d_out, rd, d_offsets, d_offsets+1, GRID_CUB_SUM_OP, zero_init, computeStream);
|
||||
if (gpuErr!=gpuSuccess) {
|
||||
std::cout << GridLogError << "Lattice_slicesum_gpu.h: Encountered error during gpucub::DeviceSegmentedReduce::Reduce! Error: " << gpuErr <<std::endl;
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
#undef GRID_CUB_SUM_OP
|
||||
|
||||
acceleratorCopyFromDeviceAsynch(d_out,&lvSum[0],rd*sizeof(vobj),computeStream);
|
||||
|
||||
|
||||
@@ -124,6 +124,68 @@ Lattice<iScalar<iScalar<iMatrix<vComplexD, N> > > > Inverse(const Lattice<iScala
|
||||
return ret;
|
||||
}
|
||||
|
||||
template<int N>
|
||||
Lattice<iMatrix<iScalar<iScalar<iScalar<vComplexD> > > , N> > Inverse(const Lattice<iMatrix<iScalar<iScalar<iScalar<vComplexD> > >, N> > &Umu)
|
||||
{
|
||||
GridBase *grid=Umu.Grid();
|
||||
auto lvol = grid->lSites();
|
||||
Lattice<iMatrix<iScalar<iScalar<iScalar<vComplexD> > >, N > > ret(grid);
|
||||
|
||||
autoView(Umu_v,Umu,CpuRead);
|
||||
autoView(ret_v,ret,CpuWrite);
|
||||
thread_for(site,lvol,{
|
||||
Eigen::MatrixXcd EigenU = Eigen::MatrixXcd::Zero(N,N);
|
||||
Coordinate lcoor;
|
||||
grid->LocalIndexToLocalCoor(site, lcoor);
|
||||
iMatrix<iScalar<iScalar<iScalar<ComplexD> > >, N > Us;
|
||||
iMatrix<iScalar<iScalar<iScalar<ComplexD> > >, N > Ui;
|
||||
peekLocalSite(Us, Umu_v, lcoor);
|
||||
for(int i=0;i<N;i++){
|
||||
for(int j=0;j<N;j++){
|
||||
EigenU(i,j) = Us(i,j)()()();
|
||||
}}
|
||||
Eigen::MatrixXcd EigenUinv = EigenU.inverse();
|
||||
for(int i=0;i<N;i++){
|
||||
for(int j=0;j<N;j++){
|
||||
Ui(i,j)()()() = EigenUinv(i,j);
|
||||
}}
|
||||
pokeLocalSite(Ui,ret_v,lcoor);
|
||||
});
|
||||
return ret;
|
||||
}
|
||||
|
||||
template<int N>
|
||||
Lattice<iMatrix<iScalar<iScalar<iScalar<iScalar<vComplexD> > > > , N> > Inverse(const Lattice<iMatrix<iScalar<iScalar<iScalar<iScalar<vComplexD> > > >, N> > &Umu)
|
||||
{
|
||||
GridBase *grid=Umu.Grid();
|
||||
auto lvol = grid->lSites();
|
||||
Lattice<iMatrix<iScalar<iScalar<iScalar<iScalar<vComplexD> > > >, N > > ret(grid);
|
||||
|
||||
autoView(Umu_v,Umu,CpuRead);
|
||||
autoView(ret_v,ret,CpuWrite);
|
||||
thread_for(site,lvol,{
|
||||
Eigen::MatrixXcd EigenU = Eigen::MatrixXcd::Zero(N,N);
|
||||
Coordinate lcoor;
|
||||
grid->LocalIndexToLocalCoor(site, lcoor);
|
||||
iMatrix<iScalar<iScalar<iScalar<iScalar<ComplexD> > > >, N > Us;
|
||||
iMatrix<iScalar<iScalar<iScalar<iScalar<ComplexD> > > >, N > Ui;
|
||||
peekLocalSite(Us, Umu_v, lcoor);
|
||||
for(int i=0;i<N;i++){
|
||||
for(int j=0;j<N;j++){
|
||||
EigenU(i,j) = Us(i,j)()()()();
|
||||
}}
|
||||
Eigen::MatrixXcd EigenUinv = EigenU.inverse();
|
||||
for(int i=0;i<N;i++){
|
||||
for(int j=0;j<N;j++){
|
||||
Ui(i,j)()()()() = EigenUinv(i,j);
|
||||
}}
|
||||
pokeLocalSite(Ui,ret_v,lcoor);
|
||||
});
|
||||
return ret;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
#endif
|
||||
|
||||
@@ -102,7 +102,7 @@ template<class vobj> inline void acceleratorPickCheckerboard(int cb,Lattice<vobj
|
||||
int linear=0;
|
||||
|
||||
Lexicographic::CoorFromIndex(coor,ss,rdim_full);
|
||||
GRID_ASSERT(coor.size()==ndim_half);
|
||||
assert(coor.size()==ndim_half);
|
||||
|
||||
for(int d=0;d<ndim_half;d++){
|
||||
if(checker_dim_mask_half[d]) linear += coor[d];
|
||||
@@ -136,7 +136,7 @@ template<class vobj> inline void acceleratorSetCheckerboard(Lattice<vobj> &full,
|
||||
int linear=0;
|
||||
|
||||
Lexicographic::CoorFromIndex(coor,ss,rdim_full);
|
||||
GRID_ASSERT(coor.size()==ndim_half);
|
||||
assert(coor.size()==ndim_half);
|
||||
|
||||
for(int d=0;d<ndim_half;d++){
|
||||
if(checker_dim_mask_half[d]) linear += coor[d];
|
||||
|
||||
@@ -10,7 +10,7 @@ class LatticeBase {};
|
||||
/////////////////////////////////////////////////////////////////////////////////////////
|
||||
void accelerator_inline conformable(GridBase *lhs,GridBase *rhs)
|
||||
{
|
||||
GRID_ASSERT(lhs == rhs);
|
||||
assert(lhs == rhs);
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
@@ -2,3 +2,7 @@
|
||||
|
||||
int Grid::BinaryIO::latticeWriteMaxRetry = -1;
|
||||
Grid::BinaryIO::IoPerf Grid::BinaryIO::lastPerf;
|
||||
|
||||
// Target size of a single contiguous file extent under BINARYIO_AGGREGATE.
|
||||
// 4MB is around the knee for Lustre; exposed so it can be swept at runtime.
|
||||
uint64_t Grid::BinaryIO::aggregateTargetBytes = 4*1024*1024;
|
||||
|
||||
+457
-14
@@ -39,6 +39,7 @@
|
||||
#endif
|
||||
|
||||
#include <arpa/inet.h>
|
||||
#include <sys/stat.h>
|
||||
#include <algorithm>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
@@ -87,6 +88,7 @@ class BinaryIO {
|
||||
|
||||
static IoPerf lastPerf;
|
||||
static int latticeWriteMaxRetry;
|
||||
static uint64_t aggregateTargetBytes;
|
||||
|
||||
/////////////////////////////////////////////////////////////////////////////
|
||||
// more byte manipulation helpers
|
||||
@@ -253,12 +255,392 @@ class BinaryIO {
|
||||
// Read or Write distributed lexico array of ANY object to a specific location in file
|
||||
//////////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
static const int BINARYIO_AGGREGATE = 0x20;
|
||||
static const int BINARYIO_MASTER_APPEND = 0x10;
|
||||
static const int BINARYIO_UNORDERED = 0x08;
|
||||
static const int BINARYIO_LEXICOGRAPHIC = 0x04;
|
||||
static const int BINARYIO_READ = 0x02;
|
||||
static const int BINARYIO_WRITE = 0x01;
|
||||
|
||||
#ifdef USE_MPI_IO
|
||||
/////////////////////////////////////////////////////////////////////////////
|
||||
// Aggregation: self controlled transposition onto an I/O friendly layout.
|
||||
//
|
||||
// Under BINARYIO_LEXICOGRAPHIC the subarray file view handed to MPI-IO has
|
||||
// contiguous runs of only lLattice[0]*sizeof(fobj) bytes -- a few KB for
|
||||
// typical local volumes. Rather than rely on collective buffering to repair
|
||||
// that, redistribute the payload ourselves so every rank owns a contiguous
|
||||
// range of the global lexicographic site ordering, then issue large plain
|
||||
// contiguous writes.
|
||||
//
|
||||
// "Un-splitting" the nunsplit fastest dimensions means the row of ranks
|
||||
// sharing the remaining process coordinates collectively owns whole global
|
||||
// hyperplanes. All data movement is then confined to that row communicator.
|
||||
// Every rank still owns exactly lSites() sites afterwards, so the exchange is
|
||||
// a pure permutation and needs no divisibility condition on the process grid.
|
||||
/////////////////////////////////////////////////////////////////////////////
|
||||
struct AggregationPlan {
|
||||
int nunsplit{0}; // number of fastest dimensions un-split
|
||||
int rowsize{0}; // ranks in the aggregation (row) communicator
|
||||
int rowrank{0}; // our logical (lexicographic) index within the row
|
||||
uint64_t lsites{0}; // sites per rank -- invariant under the permutation
|
||||
uint64_t chunk{0}; // sites in one globally contiguous run owned by the row
|
||||
std::unique_ptr<CartesianCommunicator> rowcomm;
|
||||
// counts and displacements are indexed by rank within rowcomm
|
||||
std::vector<int> sendcounts, senddispls, recvcounts, recvdispls;
|
||||
std::vector<uint64_t> scatter; // recv slot -> slot in the aggregated buffer
|
||||
std::vector<uint64_t> extentGsite; // global lex site index of extent start
|
||||
std::vector<uint64_t> extentLocal; // offset of extent within aggregated buffer
|
||||
std::vector<uint64_t> extentSites; // sites in this extent
|
||||
};
|
||||
|
||||
static inline void BuildAggregationPlan(GridBase *grid,uint64_t fobjSize,AggregationPlan &p)
|
||||
{
|
||||
int ndim = grid->Dimensions();
|
||||
Coordinate psizes = grid->ProcessorGrid();
|
||||
Coordinate pcoor = grid->ThisProcessorCoor();
|
||||
Coordinate gLattice= grid->GlobalDimensions();
|
||||
Coordinate lLattice= grid->LocalDimensions();
|
||||
Coordinate lstart = grid->LocalStarts();
|
||||
|
||||
uint64_t lsites = grid->lSites();
|
||||
p.lsites = lsites;
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// Un-splitting dims 0..k-1 gives the row a contiguous run of
|
||||
// chunk(k) = prod_{d<k} gLattice[d] * lLattice[k]
|
||||
// sites, and each rank writes extents of min(chunk,lsites). Take the
|
||||
// smallest k that reaches the target so we disturb as few dimensions --
|
||||
// and move as little data -- as possible.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
int k = ndim-1;
|
||||
for(int trial=1; trial<ndim; trial++){
|
||||
uint64_t chunk = lLattice[trial];
|
||||
for(int d=0; d<trial; d++) chunk *= gLattice[d];
|
||||
if ( std::min(chunk,lsites)*fobjSize >= aggregateTargetBytes ) { k = trial; break; }
|
||||
}
|
||||
p.nunsplit = k;
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// The box the row collectively owns, expressed in global coordinates.
|
||||
// Restricting the global lexicographic order to this box preserves the
|
||||
// ordering, so the row index below is monotone in the global index.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
Coordinate B(ndim), S(ndim);
|
||||
for(int d=0; d<ndim; d++){
|
||||
if ( d<k ) { B[d] = gLattice[d]; S[d] = 0; }
|
||||
else { B[d] = lLattice[d]; S[d] = lstart[d]; }
|
||||
}
|
||||
|
||||
uint64_t chunk = lLattice[k];
|
||||
for(int d=0; d<k; d++) chunk *= gLattice[d];
|
||||
p.chunk = chunk;
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// Row communicator: the ranks sharing the process coordinates of the slow
|
||||
// (still split) dimensions. This is the sub-division the Cartesian
|
||||
// communicator already performs for AllToAll(dim,...), widened from one
|
||||
// dimension to the k fastest.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
Coordinate row(ndim,1);
|
||||
for(int d=0; d<k; d++) row[d] = psizes[d];
|
||||
int srank;
|
||||
p.rowcomm.reset(new CartesianCommunicator(row,*grid,srank));
|
||||
p.rowsize = p.rowcomm->ProcessorCount();
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// Our logical index in the row is the forward lexicographic index of the
|
||||
// un-split process coordinates, so that increasing logical index means
|
||||
// increasing global lexicographic position in the file. The communicator
|
||||
// numbers its own ranks by the reversed (MPI) convention, so build the map
|
||||
// between the two rather than assuming either.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
int64_t logical=0, lstride=1;
|
||||
for(int d=0; d<k; d++){ logical += pcoor[d]*lstride; lstride *= psizes[d]; }
|
||||
GRID_ASSERT(lstride == (int64_t)p.rowsize);
|
||||
p.rowrank = (int)logical;
|
||||
|
||||
std::vector<uint64_t> commOf(p.rowsize,0);
|
||||
commOf[p.rowrank] = (uint64_t)p.rowcomm->ThisRank();
|
||||
p.rowcomm->GlobalSumVector(&commOf[0],p.rowsize);
|
||||
|
||||
uint64_t mystart = (uint64_t)p.rowrank * lsites;
|
||||
uint64_t myend = mystart + lsites;
|
||||
|
||||
Coordinate lcoor(ndim), bcoor(ndim), gcoor(ndim);
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// Send side. Walking our local sites in local lexicographic order walks
|
||||
// the row index monotonically, so the send buffer is iodata untouched and
|
||||
// we need only the per destination counts.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
std::vector<int> sendLogical(p.rowsize,0);
|
||||
for(uint64_t L=0; L<lsites; L++){
|
||||
Lexicographic::CoorFromIndex(lcoor,L,lLattice);
|
||||
for(int d=0; d<ndim; d++) bcoor[d] = (d<k) ? (lstart[d]+lcoor[d]) : lcoor[d];
|
||||
int64_t ri; Lexicographic::IndexFromCoor(bcoor,ri,B);
|
||||
sendLogical[ ri/(int64_t)lsites ]++;
|
||||
}
|
||||
p.sendcounts.assign(p.rowsize,0);
|
||||
p.senddispls.assign(p.rowsize,0);
|
||||
{ int64_t disp=0;
|
||||
for(int d=0; d<p.rowsize; d++){ // send buffer is in logical order
|
||||
int c = (int)commOf[d];
|
||||
p.sendcounts[c] = sendLogical[d];
|
||||
p.senddispls[c] = (int)disp;
|
||||
disp += sendLogical[d];
|
||||
}
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// Receive side. For each slot of our aggregated range work out which rank
|
||||
// of the row owns it. Within one source the slots arrive in increasing row
|
||||
// index order, which is the order the source sends them in.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
std::vector<int> recvLogical(p.rowsize,0), recvDisplLogical(p.rowsize,0);
|
||||
std::vector<int> source(lsites);
|
||||
for(uint64_t pos=0; pos<lsites; pos++){
|
||||
Lexicographic::CoorFromIndex(bcoor,(int64_t)(mystart+pos),B);
|
||||
int64_t j=0, jstride=1;
|
||||
for(int d=0; d<k; d++){ j += (bcoor[d]/lLattice[d])*jstride; jstride *= psizes[d]; }
|
||||
source[pos] = (int)j;
|
||||
recvLogical[j]++;
|
||||
}
|
||||
p.recvcounts.assign(p.rowsize,0);
|
||||
p.recvdispls.assign(p.rowsize,0);
|
||||
{ int64_t disp=0;
|
||||
for(int s=0; s<p.rowsize; s++){ // recv buffer is in logical order
|
||||
int c = (int)commOf[s];
|
||||
recvDisplLogical[s] = (int)disp;
|
||||
p.recvcounts[c] = recvLogical[s];
|
||||
p.recvdispls[c] = (int)disp;
|
||||
disp += recvLogical[s];
|
||||
}
|
||||
}
|
||||
p.scatter.resize(lsites);
|
||||
{
|
||||
std::vector<int> fill(p.rowsize,0);
|
||||
for(uint64_t pos=0; pos<lsites; pos++){
|
||||
int j = source[pos];
|
||||
p.scatter[ recvDisplLogical[j] + fill[j]++ ] = pos;
|
||||
}
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// The two sides are derived independently; make them check each other.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
{
|
||||
std::vector<uint64_t> sendc(p.rowsize),recvc(p.rowsize);
|
||||
for(int c=0;c<p.rowsize;c++) sendc[c]=(uint64_t)p.sendcounts[c];
|
||||
p.rowcomm->AllToAll(&sendc[0],&recvc[0],1,sizeof(uint64_t));
|
||||
for(int c=0;c<p.rowsize;c++) GRID_ASSERT((int)recvc[c]==p.recvcounts[c]);
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// Decompose our range into globally contiguous file extents.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
for(uint64_t c = mystart/chunk; c <= (myend-1)/chunk; c++){
|
||||
uint64_t lo = std::max(mystart, c*chunk);
|
||||
uint64_t hi = std::min(myend, (c+1)*chunk);
|
||||
Lexicographic::CoorFromIndex(bcoor,(int64_t)(c*chunk),B);
|
||||
for(int d=0;d<ndim;d++) gcoor[d] = (d<k) ? bcoor[d] : bcoor[d]+S[d];
|
||||
int64_t gbase; Lexicographic::IndexFromCoor(gcoor,gbase,gLattice);
|
||||
p.extentGsite.push_back( (uint64_t)gbase + (lo - c*chunk) );
|
||||
p.extentLocal.push_back( lo - mystart );
|
||||
p.extentSites.push_back( hi - lo );
|
||||
}
|
||||
}
|
||||
|
||||
static inline void ReportAggregationPlan(GridBase *grid,const AggregationPlan &p,uint64_t fobjSize,const char *what)
|
||||
{
|
||||
if ( !grid->IsBoss() ) return;
|
||||
std::cout << GridLogMessage << "IOobject: aggregate " << what
|
||||
<< " un-splitting " << p.nunsplit << " fastest dimensions, row of "
|
||||
<< p.rowsize << " ranks" << std::endl;
|
||||
std::cout << GridLogMessage << "IOobject: aggregate " << p.extentSites.size()
|
||||
<< " extent(s)/rank, first " << p.extentSites[0]*fobjSize/1024./1024. << " MB"
|
||||
<< " (target " << aggregateTargetBytes/1024./1024. << " MB)" << std::endl;
|
||||
std::cout << GridLogMessage << "IOobject: aggregate buffer overhead "
|
||||
<< p.lsites*fobjSize/1024./1024. << " MB/rank" << std::endl;
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
// Stage timings. The interesting quantity is the slowest rank, since every
|
||||
// stage is followed sooner or later by a synchronisation, so reduce with
|
||||
// GlobalMax rather than reporting whatever the boss happened to see.
|
||||
////////////////////////////////////////////////////////////////////////////
|
||||
static inline void ReportStages(GridBase *grid,const char *what,
|
||||
const std::vector<const char *> &names,
|
||||
std::vector<RealD> &useconds)
|
||||
{
|
||||
GRID_ASSERT(names.size()==useconds.size());
|
||||
for(uint64_t i=0;i<useconds.size();i++) grid->GlobalMax(useconds[i]);
|
||||
if ( grid->IsBoss() ) {
|
||||
std::cout << GridLogMessage << "IOobject: aggregate " << what << " stages (max over ranks, s):";
|
||||
for(uint64_t i=0;i<names.size();i++)
|
||||
std::cout << " " << names[i] << " " << useconds[i]/1.0e6;
|
||||
std::cout << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
template<class fobj>
|
||||
static inline void AggregateExchange(GridBase *grid,AggregationPlan &p,std::vector<fobj> &iodata,
|
||||
std::vector<fobj> &aggregated,int forward)
|
||||
{
|
||||
uint64_t lsites = p.lsites;
|
||||
GridStopWatch talloc,tperm,tcomm;
|
||||
|
||||
talloc.Start();
|
||||
std::vector<fobj> tmp(lsites);
|
||||
talloc.Stop();
|
||||
|
||||
if ( forward ) { // iodata (local order) -> aggregated (lexicographic order)
|
||||
tcomm.Start();
|
||||
p.rowcomm->AllToAllV(&iodata[0],p.sendcounts,p.senddispls,
|
||||
&tmp[0], p.recvcounts,p.recvdispls,sizeof(fobj));
|
||||
tcomm.Stop();
|
||||
tperm.Start();
|
||||
thread_for(s,lsites,{ aggregated[p.scatter[s]] = tmp[s]; });
|
||||
tperm.Stop();
|
||||
} else { // aggregated -> iodata, the exact mirror
|
||||
tperm.Start();
|
||||
thread_for(s,lsites,{ tmp[s] = aggregated[p.scatter[s]]; });
|
||||
tperm.Stop();
|
||||
tcomm.Start();
|
||||
p.rowcomm->AllToAllV(&tmp[0], p.recvcounts,p.recvdispls,
|
||||
&iodata[0],p.sendcounts,p.senddispls,sizeof(fobj));
|
||||
tcomm.Stop();
|
||||
}
|
||||
|
||||
std::vector<RealD> us = { (RealD)talloc.useconds(), (RealD)tperm.useconds(), (RealD)tcomm.useconds() };
|
||||
ReportStages(grid,forward?"exchange (write)":"exchange (read)",
|
||||
{"alloc","permute","alltoallv"},us);
|
||||
}
|
||||
|
||||
template<class fobj>
|
||||
static inline void AggregateWrite(GridBase *grid,AggregationPlan &p,std::vector<fobj> &aggregated,
|
||||
std::string file,uint64_t offset)
|
||||
{
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
// All ranks write concurrently into a shared file, so the file must exist
|
||||
// before any of them open it for update, but it does NOT have to be the
|
||||
// right length first: the extents tile the record exactly, so writing them
|
||||
// extends a short file to precisely offset+payload.
|
||||
//
|
||||
// Records are created in sequence, so this payload ends the file: the
|
||||
// length must end up precisely offset+payload. Anything beyond is left
|
||||
// over from whatever the file previously held and must not survive -- a
|
||||
// shorter new record written over a longer old one would otherwise leave
|
||||
// a trailing fragment of the previous contents masquerading as data.
|
||||
// That is the only case needing a truncate, so stat first and truncate
|
||||
// afterwards only when the size actually came out wrong. Measured on
|
||||
// Frontier, an unconditional truncate up front cost 0.22 to 5.4 s per
|
||||
// record -- 15 to 25% of a 19 GB write and 100% of a small one -- while
|
||||
// create, open and close together cost a few milliseconds. It is per
|
||||
// record, so multi record files do not amortise it away.
|
||||
//
|
||||
// ::truncate is used because the C++ standard library cannot express this.
|
||||
// std::filebuf has no length operation at all; ios::trunc only truncates to
|
||||
// zero; seeking past the end and writing a byte can grow a file but never
|
||||
// shrink one; and there is no portable way to recover a descriptor from a
|
||||
// stream in order to call ftruncate. C++17 does finally offer
|
||||
// std::filesystem::resize_file, but that would be Grid's first <filesystem>
|
||||
// dependency and needs -lstdc++fs on the older toolchains still in use.
|
||||
//////////////////////////////////////////////////////////////////////////
|
||||
GridStopWatch tcreate,ttrunc,tbar,topen,twrite,tclose,tskew;
|
||||
uint64_t need = offset + (uint64_t)grid->_gsites*sizeof(fobj);
|
||||
|
||||
tcreate.Start();
|
||||
if ( grid->IsBoss() ) {
|
||||
// opening for update needs the file to exist; create one only if not
|
||||
std::fstream probe(file,std::ios::binary|std::ios::out|std::ios::in);
|
||||
if ( !probe.is_open() ) {
|
||||
std::ofstream create(file,std::ios::binary|std::ios::out);
|
||||
create.close();
|
||||
}
|
||||
}
|
||||
tcreate.Stop();
|
||||
|
||||
tbar.Start();
|
||||
grid->Barrier();
|
||||
tbar.Stop();
|
||||
|
||||
std::ofstream fout;
|
||||
fout.exceptions( std::fstream::failbit | std::fstream::badbit );
|
||||
try {
|
||||
topen.Start();
|
||||
fout.open(file,std::ios::binary|std::ios::out|std::ios::in);
|
||||
topen.Stop();
|
||||
twrite.Start();
|
||||
for(uint64_t e=0;e<p.extentSites.size();e++){
|
||||
fout.seekp(offset + p.extentGsite[e]*sizeof(fobj));
|
||||
fout.write((char *)&aggregated[p.extentLocal[e]],p.extentSites[e]*sizeof(fobj));
|
||||
}
|
||||
twrite.Stop();
|
||||
tclose.Start();
|
||||
fout.close(); // flushes the stream buffer; does not force writeback
|
||||
tclose.Stop();
|
||||
} catch (const std::fstream::failure& exc) {
|
||||
std::cout << GridLogError << "Error in aggregate write to " << file << std::endl;
|
||||
std::cout << GridLogError << "Exception description: " << exc.what() << std::endl;
|
||||
GridAbort();
|
||||
}
|
||||
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
// Timed apart from the truncate that follows it. seek+write above is the
|
||||
// slowest rank; this barrier is what the fastest rank then waits, so the
|
||||
// pair separates the write cost from the spread across ranks. Folding it
|
||||
// into the truncate makes a millisecond stat look like a second.
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
tskew.Start();
|
||||
grid->Barrier(); // every extent must be on its way first
|
||||
tskew.Stop();
|
||||
|
||||
ttrunc.Start();
|
||||
if ( grid->IsBoss() ) {
|
||||
struct stat sb;
|
||||
int ierr = ::stat(file.c_str(),&sb);
|
||||
GRID_ASSERT(ierr==0);
|
||||
if ( (uint64_t)sb.st_size != need ) { // only when a longer record preceded us
|
||||
ierr = ::truncate(file.c_str(),(off_t)need);
|
||||
GRID_ASSERT(ierr==0);
|
||||
}
|
||||
}
|
||||
grid->Barrier();
|
||||
ttrunc.Stop();
|
||||
|
||||
std::vector<RealD> us = { (RealD)tcreate.useconds(), (RealD)tbar.useconds(),
|
||||
(RealD)topen.useconds(), (RealD)twrite.useconds(),
|
||||
(RealD)tclose.useconds(), (RealD)tskew.useconds(),
|
||||
(RealD)ttrunc.useconds() };
|
||||
ReportStages(grid,"write",{"create","barrier","open","seek+write","close","skew","stat+truncate"},us);
|
||||
}
|
||||
|
||||
template<class fobj>
|
||||
static inline void AggregateRead(GridBase *grid,AggregationPlan &p,std::vector<fobj> &aggregated,
|
||||
std::string file,uint64_t offset)
|
||||
{
|
||||
GridStopWatch topen,tread,tclose;
|
||||
std::ifstream fin;
|
||||
topen.Start();
|
||||
fin.open(file,std::ios::binary|std::ios::in);
|
||||
topen.Stop();
|
||||
tread.Start();
|
||||
for(uint64_t e=0;e<p.extentSites.size();e++){
|
||||
fin.seekg(offset + p.extentGsite[e]*sizeof(fobj));
|
||||
fin.read((char *)&aggregated[p.extentLocal[e]],p.extentSites[e]*sizeof(fobj));
|
||||
GRID_ASSERT(fin.fail()==0);
|
||||
}
|
||||
tread.Stop();
|
||||
tclose.Start();
|
||||
fin.close();
|
||||
tclose.Stop();
|
||||
|
||||
std::vector<RealD> us = { (RealD)topen.useconds(), (RealD)tread.useconds(), (RealD)tclose.useconds() };
|
||||
ReportStages(grid,"read",{"open","seek+read","close"},us);
|
||||
}
|
||||
#endif
|
||||
|
||||
template<class word,class fobj>
|
||||
static inline void IOobject(word w,
|
||||
GridBase *grid,
|
||||
@@ -302,6 +684,18 @@ class BinaryIO {
|
||||
lStart[d] = 0;
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
// Aggregate the lexicographic layout onto contiguous per rank extents
|
||||
// ourselves rather than leaving it to MPI-IO collective buffering
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
int aggregate = (control & BINARYIO_AGGREGATE)
|
||||
&& (control & BINARYIO_LEXICOGRAPHIC)
|
||||
&& !(control & BINARYIO_MASTER_APPEND)
|
||||
&& (nrank > 1);
|
||||
#ifndef USE_MPI_IO
|
||||
GRID_ASSERT(aggregate==0); // BINARYIO_AGGREGATE requires MPI
|
||||
#endif
|
||||
|
||||
#ifdef USE_MPI_IO
|
||||
std::vector<int> distribs(ndim,MPI_DISTRIBUTE_BLOCK);
|
||||
std::vector<int> dargs (ndim,MPI_DISTRIBUTE_DFLT_DARG);
|
||||
@@ -329,6 +723,8 @@ class BinaryIO {
|
||||
ierr = MPI_Type_contiguous(numword,mpiword,&mpiObject); GRID_ASSERT(ierr==0);
|
||||
ierr = MPI_Type_commit(&mpiObject);
|
||||
|
||||
// The subarray view is what aggregation exists to avoid; do not build it
|
||||
if ( !aggregate ) {
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
// File global array data type
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
@@ -340,6 +736,7 @@ class BinaryIO {
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
ierr=MPI_Type_create_subarray(ndim,&lLattice[0],&lLattice[0],&lStart[0],MPI_ORDER_FORTRAN, mpiObject,&localArray); GRID_ASSERT(ierr==0);
|
||||
ierr=MPI_Type_commit(&localArray); GRID_ASSERT(ierr==0);
|
||||
}
|
||||
#endif
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
@@ -358,7 +755,19 @@ class BinaryIO {
|
||||
|
||||
timer.Start();
|
||||
|
||||
if ( (control & BINARYIO_LEXICOGRAPHIC) && (nrank > 1) ) {
|
||||
if ( aggregate ) {
|
||||
#ifdef USE_MPI_IO
|
||||
std::cout<< GridLogMessage<<"IOobject: aggregate read I/O "<< file<< std::endl;
|
||||
AggregationPlan plan;
|
||||
BuildAggregationPlan(grid,sizeof(fobj),plan);
|
||||
ReportAggregationPlan(grid,plan,sizeof(fobj),"read");
|
||||
std::vector<fobj> aggregated(lsites);
|
||||
AggregateRead(grid,plan,aggregated,file,offset);
|
||||
AggregateExchange(grid,plan,iodata,aggregated,0);
|
||||
#else
|
||||
GRID_ASSERT(0);
|
||||
#endif
|
||||
} else if ( (control & BINARYIO_LEXICOGRAPHIC) && (nrank > 1) ) {
|
||||
#ifdef USE_MPI_IO
|
||||
std::cout<< GridLogMessage<<"IOobject: MPI read I/O "<< file<< std::endl;
|
||||
ierr=MPI_File_open(grid->communicator,(char *) file.c_str(), MPI_MODE_RDONLY, MPI_INFO_NULL, &fh); GRID_ASSERT(ierr==0);
|
||||
@@ -387,10 +796,11 @@ class BinaryIO {
|
||||
GRID_ASSERT(fin.fail() == 0);
|
||||
fin.close();
|
||||
}
|
||||
timer.Stop();
|
||||
|
||||
|
||||
grid->Barrier();
|
||||
|
||||
timer.Stop();
|
||||
|
||||
bstimer.Start();
|
||||
ScidacChecksum(grid,iodata,scidac_csuma,scidac_csumb);
|
||||
if (ieee32big) be32toh_v((void *)&iodata[0], sizeof(fobj)*iodata.size());
|
||||
@@ -415,7 +825,25 @@ class BinaryIO {
|
||||
grid->Barrier();
|
||||
|
||||
timer.Start();
|
||||
if ( (control & BINARYIO_LEXICOGRAPHIC) && (nrank > 1) ) {
|
||||
if ( aggregate ) {
|
||||
#ifdef USE_MPI_IO
|
||||
std::cout << GridLogMessage <<"IOobject: aggregate write I/O " << file << std::endl;
|
||||
AggregationPlan plan;
|
||||
BuildAggregationPlan(grid,sizeof(fobj),plan);
|
||||
ReportAggregationPlan(grid,plan,sizeof(fobj),"write");
|
||||
std::vector<fobj> aggregated(lsites);
|
||||
AggregateExchange(grid,plan,iodata,aggregated,1);
|
||||
AggregateWrite(grid,plan,aggregated,file,offset);
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
// Not every rank ends at the end of the payload, so the position can
|
||||
// not be recovered from a file handle. Callers (Lime record chaining)
|
||||
// rely on this being the first byte past the record.
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
offset = offset + (uint64_t)grid->_gsites*sizeof(fobj);
|
||||
#else
|
||||
GRID_ASSERT(0);
|
||||
#endif
|
||||
} else if ( (control & BINARYIO_LEXICOGRAPHIC) && (nrank > 1) ) {
|
||||
#ifdef USE_MPI_IO
|
||||
std::cout << GridLogMessage <<"IOobject: MPI write I/O " << file << std::endl;
|
||||
ierr = MPI_File_open(grid->communicator, (char *)file.c_str(), MPI_MODE_RDWR | MPI_MODE_CREATE, MPI_INFO_NULL, &fh);
|
||||
@@ -460,12 +888,26 @@ class BinaryIO {
|
||||
|
||||
std::ofstream fout;
|
||||
fout.exceptions ( std::fstream::failbit | std::fstream::badbit );
|
||||
|
||||
////////////////////////////////////////////////////////////////////
|
||||
// Grid's model is that the boss rank performs the metadata
|
||||
// operations and every other rank only seeks and writes into a file
|
||||
// that already exists. Opening with ios::out on all ranks broke that:
|
||||
// it is O_TRUNC, so a rank opening late truncated the file back to
|
||||
// zero after an earlier rank had written its segment, leaving a hole
|
||||
// in its place. The barriers around this block are outside it and do
|
||||
// not order the opens against the writes. Let the boss create and
|
||||
// empty the file, then everyone opens for update only. Same resulting
|
||||
// length, one metadata operation instead of one per rank, no race.
|
||||
////////////////////////////////////////////////////////////////////
|
||||
if ( !offset && grid->IsBoss() ) { // offset zero: this record starts the file
|
||||
std::ofstream create(file,std::ios::binary|std::ios::out);
|
||||
create.close();
|
||||
}
|
||||
grid->Barrier();
|
||||
|
||||
try {
|
||||
if (offset) { // Must already exist and contain data
|
||||
fout.open(file,std::ios::binary|std::ios::out|std::ios::in);
|
||||
} else { // Allow create
|
||||
fout.open(file,std::ios::binary|std::ios::out);
|
||||
}
|
||||
fout.open(file,std::ios::binary|std::ios::out|std::ios::in);
|
||||
} catch (const std::fstream::failure& exc) {
|
||||
std::cout << GridLogError << "Error in opening the file " << file << " for output" <<std::endl;
|
||||
std::cout << GridLogError << "Exception description: " << exc.what() << std::endl;
|
||||
@@ -476,7 +918,7 @@ class BinaryIO {
|
||||
exit(1);
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
if ( control & BINARYIO_MASTER_APPEND ) {
|
||||
try {
|
||||
fout.seekp(0,fout.end);
|
||||
@@ -506,6 +948,7 @@ class BinaryIO {
|
||||
offset = fout.tellp();
|
||||
fout.close();
|
||||
}
|
||||
grid->Barrier();
|
||||
timer.Stop();
|
||||
}
|
||||
|
||||
@@ -546,7 +989,7 @@ class BinaryIO {
|
||||
uint32_t &nersc_csum,
|
||||
uint32_t &scidac_csuma,
|
||||
uint32_t &scidac_csumb,
|
||||
int control=BINARYIO_LEXICOGRAPHIC
|
||||
int control=BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE
|
||||
)
|
||||
{
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
@@ -585,7 +1028,7 @@ class BinaryIO {
|
||||
uint32_t &nersc_csum,
|
||||
uint32_t &scidac_csuma,
|
||||
uint32_t &scidac_csumb,
|
||||
int control=BINARYIO_LEXICOGRAPHIC)
|
||||
int control=BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE)
|
||||
{
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::Realified::scalar_type word; word w=0;
|
||||
@@ -672,7 +1115,7 @@ class BinaryIO {
|
||||
std::cout << GridLogMessage << "RNG read I/O on file " << file << std::endl;
|
||||
|
||||
std::vector<RNGstate> iodata(lsites);
|
||||
IOobject(w,grid,iodata,file,offset,format,BINARYIO_READ|BINARYIO_LEXICOGRAPHIC,
|
||||
IOobject(w,grid,iodata,file,offset,format,BINARYIO_READ|BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE,
|
||||
nersc_csum,scidac_csuma,scidac_csumb);
|
||||
|
||||
timer.Start();
|
||||
@@ -751,7 +1194,7 @@ class BinaryIO {
|
||||
});
|
||||
timer.Stop();
|
||||
|
||||
IOobject(w,grid,iodata,file,offset,format,BINARYIO_WRITE|BINARYIO_LEXICOGRAPHIC,
|
||||
IOobject(w,grid,iodata,file,offset,format,BINARYIO_WRITE|BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE,
|
||||
nersc_csum,scidac_csuma,scidac_csumb);
|
||||
iodata.resize(1);
|
||||
{
|
||||
|
||||
@@ -212,7 +212,7 @@ class GridLimeReader : public BinaryIO {
|
||||
// Read a generic lattice field and verify checksum
|
||||
////////////////////////////////////////////
|
||||
template<class vobj>
|
||||
void readLimeLatticeBinaryObject(Lattice<vobj> &field,std::string record_name,int control=BINARYIO_LEXICOGRAPHIC)
|
||||
void readLimeLatticeBinaryObject(Lattice<vobj> &field,std::string record_name,int control=BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE)
|
||||
{
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
scidacChecksum scidacChecksum_;
|
||||
@@ -414,7 +414,7 @@ class GridLimeWriter : public BinaryIO
|
||||
// in communicator used by the field.Grid()
|
||||
////////////////////////////////////////////////////
|
||||
template<class vobj>
|
||||
void writeLimeLatticeBinaryObject(Lattice<vobj> &field,std::string record_name,int control=BINARYIO_LEXICOGRAPHIC)
|
||||
void writeLimeLatticeBinaryObject(Lattice<vobj> &field,std::string record_name,int control=BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE)
|
||||
{
|
||||
////////////////////////////////////////////////////////////////////
|
||||
// NB: FILE and iostream are jointly writing disjoint sequences in the
|
||||
@@ -519,7 +519,7 @@ class ScidacWriter : public GridLimeWriter {
|
||||
template <class vobj, class userRecord>
|
||||
void writeScidacFieldRecord(Lattice<vobj> &field,userRecord _userRecord,
|
||||
const unsigned int recordScientificPrec = 0,
|
||||
int control=BINARYIO_LEXICOGRAPHIC)
|
||||
int control=BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE)
|
||||
{
|
||||
GridBase * grid = field.Grid();
|
||||
|
||||
@@ -561,7 +561,7 @@ class ScidacReader : public GridLimeReader {
|
||||
////////////////////////////////////////////////
|
||||
template <class vobj, class userRecord>
|
||||
void readScidacFieldRecord(Lattice<vobj> &field,userRecord &_userRecord,
|
||||
int control=BINARYIO_LEXICOGRAPHIC)
|
||||
int control=BINARYIO_LEXICOGRAPHIC|BINARYIO_AGGREGATE)
|
||||
{
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
GridBase * grid = field.Grid();
|
||||
|
||||
@@ -93,8 +93,7 @@ inline uint64_t cyclecount(void){
|
||||
}
|
||||
#elif defined __x86_64__
|
||||
inline uint64_t cyclecount(void){
|
||||
uint64_t ret = __rdtsc();
|
||||
return (uint64_t)ret;
|
||||
return (uint64_t)0;
|
||||
}
|
||||
#else
|
||||
|
||||
|
||||
@@ -60,12 +60,16 @@ inline std::ostream& operator<< (std::ostream & stream, const GridSecs & time)
|
||||
}
|
||||
inline std::ostream& operator<< (std::ostream & stream, const GridMillisecs & now)
|
||||
{
|
||||
double secs = 1.0*now.count()*1.0e-3;
|
||||
stream << secs<<" s";
|
||||
/*
|
||||
GridSecs second(1);
|
||||
auto secs = now/second ;
|
||||
auto subseconds = now%second ;
|
||||
auto fill = stream.fill();
|
||||
stream << secs<<"."<<std::setw(3)<<std::setfill('0')<<subseconds.count()<<" s";
|
||||
stream.fill(fill);
|
||||
*/
|
||||
return stream;
|
||||
}
|
||||
inline std::ostream& operator<< (std::ostream & stream, const GridUsecs & now)
|
||||
|
||||
@@ -596,16 +596,32 @@ template<int Index,class vobj> inline vobj transposeColour(const vobj &lhs){
|
||||
//////////////////////////////////////////
|
||||
// Trace lattice and non-lattice
|
||||
//////////////////////////////////////////
|
||||
#define GRID_UNOP(name) name
|
||||
#define GRID_DEF_UNOP(op, name) \
|
||||
template <typename T1, typename std::enable_if<is_lattice<T1>::value||is_lattice_expr<T1>::value,T1>::type * = nullptr> \
|
||||
inline auto op(const T1 &arg) ->decltype(LatticeUnaryExpression<GRID_UNOP(name),T1>(GRID_UNOP(name)(), arg)) \
|
||||
{ \
|
||||
return LatticeUnaryExpression<GRID_UNOP(name),T1>(GRID_UNOP(name)(), arg); \
|
||||
}
|
||||
|
||||
template<int Index,class vobj>
|
||||
inline auto traceSpin(const Lattice<vobj> &lhs) -> Lattice<decltype(traceIndex<SpinIndex>(vobj()))>
|
||||
{
|
||||
return traceIndex<SpinIndex>(lhs);
|
||||
}
|
||||
|
||||
GridUnopClass(UnaryTraceSpin, traceIndex<SpinIndex>(a));
|
||||
GRID_DEF_UNOP(traceSpin, UnaryTraceSpin);
|
||||
|
||||
template<int Index,class vobj>
|
||||
inline auto traceColour(const Lattice<vobj> &lhs) -> Lattice<decltype(traceIndex<ColourIndex>(vobj()))>
|
||||
{
|
||||
return traceIndex<ColourIndex>(lhs);
|
||||
}
|
||||
|
||||
GridUnopClass(UnaryTraceColour, traceIndex<ColourIndex>(a));
|
||||
GRID_DEF_UNOP(traceColour, UnaryTraceColour);
|
||||
|
||||
template<int Index,class vobj>
|
||||
inline auto traceSpin(const vobj &lhs) -> Lattice<decltype(traceIndex<SpinIndex>(lhs))>
|
||||
{
|
||||
@@ -617,6 +633,8 @@ inline auto traceColour(const vobj &lhs) -> Lattice<decltype(traceIndex<ColourIn
|
||||
return traceIndex<ColourIndex>(lhs);
|
||||
}
|
||||
|
||||
#undef GRID_UNOP
|
||||
#undef GRID_DEF_UNOP
|
||||
//////////////////////////////////////////
|
||||
// Current types
|
||||
//////////////////////////////////////////
|
||||
|
||||
@@ -222,7 +222,7 @@ public:
|
||||
|
||||
#if 0
|
||||
static accelerator_inline typename SiteCloverTriangle::vector_type triangle_elem(const SiteCloverTriangle& triangle, int block, int i, int j) {
|
||||
GRID_ASSERT(i != j);
|
||||
assert(i != j);
|
||||
if(i < j) {
|
||||
return triangle()(block)(triangle_index(i, j));
|
||||
} else { // i > j
|
||||
@@ -232,7 +232,7 @@ public:
|
||||
#else
|
||||
template<typename vobj>
|
||||
static accelerator_inline vobj triangle_elem(const iImplCloverTriangle<vobj>& triangle, int block, int i, int j) {
|
||||
GRID_ASSERT(i != j);
|
||||
assert(i != j);
|
||||
if(i < j) {
|
||||
return triangle()(block)(triangle_index(i, j));
|
||||
} else { // i > j
|
||||
|
||||
@@ -411,7 +411,7 @@ void WilsonKernels<Impl>::DhopDirKernel( StencilImpl &st, DoubledGaugeField &U,S
|
||||
#undef LoopBody
|
||||
}
|
||||
|
||||
#ifdef GRID_SYCL
|
||||
#if 0
|
||||
extern "C" {
|
||||
ulong SYCL_EXTERNAL __attribute__((overloadable)) intel_get_cycle_counter( void );
|
||||
uint SYCL_EXTERNAL __attribute__((overloadable)) intel_get_active_channel_mask( void );
|
||||
|
||||
@@ -138,10 +138,13 @@ public:
|
||||
//auto start = std::chrono::high_resolution_clock::now();
|
||||
autoView(U_v,U,AcceleratorWrite);
|
||||
autoView(P_v,P,AcceleratorRead);
|
||||
accelerator_for(ss, P.Grid()->oSites(),1,{
|
||||
typedef typename Field::vector_object vobj;
|
||||
const int Nsimd = vobj::Nsimd();
|
||||
accelerator_for(ss, P.Grid()->oSites(),Nsimd,{
|
||||
for (int mu = 0; mu < Nd; mu++) {
|
||||
U_v[ss](mu) = Exponentiate(P_v[ss](mu), ep, Nexp) * U_v[ss](mu);
|
||||
U_v[ss](mu) = Group::ProjectOnGeneralGroup(U_v[ss](mu));
|
||||
auto tmp = Exponentiate(P_v(ss)(mu), ep, Nexp) * U_v(ss)(mu);
|
||||
tmp = Group::ProjectOnGeneralGroup(tmp);
|
||||
coalescedWrite(U_v[ss](mu),tmp);
|
||||
}
|
||||
});
|
||||
//auto end = std::chrono::high_resolution_clock::now();
|
||||
@@ -176,6 +179,8 @@ public:
|
||||
Group::ColdConfiguration(pRNG, U);
|
||||
}
|
||||
|
||||
static const int num_colours = Group::Dimension;
|
||||
|
||||
};
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: ./lib/qcd/action/pseudofermion/TwoFlavourBosonPseudoFermion.h
|
||||
|
||||
Copyright (C) 2026
|
||||
|
||||
Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#pragma once
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// Two flavour BOSON (wrong-sign) pseudofermion for any FermionOperator B:
|
||||
//
|
||||
// S2 = chi^dag Bdag B chi = |B chi|^2
|
||||
//
|
||||
// integral ==> det( Bdag B )^-1 = |det B|^-2
|
||||
//
|
||||
// A compensator monomial: supplies an INVERSE determinant with NO solve in
|
||||
// the force or the action -- both are matrix multiplies. The only solve is
|
||||
// the heatbath chi = B^-1 eta, once per trajectory (for B = the
|
||||
// Pauli-Villars operator this is a mass-one solve, trivially cheap).
|
||||
//
|
||||
// Primary use: two instances with B = PV cancel the |det PV|^2 excess of
|
||||
// TwoFlavourPVdagMPseudoFermionAction down to the DWF quotient
|
||||
// |det M|^2/|det PV|^2 (two unsquared instances rather than one squared
|
||||
// kernel: first powers of PV in the force, milder). Being generic in B it
|
||||
// also serves Hasenbusch-chain compensation at intermediate masses, or any
|
||||
// future inverse-det bookkeeping. (Sibling of the domain-decomposed boson
|
||||
// in DomainDecomposedBoundaryTwoFlavourBosonPseudoFermion.h, without the
|
||||
// boundary machinery.)
|
||||
//
|
||||
// Heatbath exact by construction: S2 after refresh = |B B^-1 eta|^2 = |eta|^2.
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template<class Impl>
|
||||
class TwoFlavourBosonPseudoFermionAction : public Action<typename Impl::GaugeField> {
|
||||
public:
|
||||
INHERIT_IMPL_TYPES(Impl);
|
||||
|
||||
private:
|
||||
FermionOperator<Impl> & BOp; // the operator whose |det|^-2 is supplied
|
||||
|
||||
LinearFunction<FermionField> &HeatbathSolver; // b -> B^-1 b (heatbath only)
|
||||
|
||||
FermionField Chi; // the pseudo fermion field for this trajectory
|
||||
|
||||
public:
|
||||
TwoFlavourBosonPseudoFermionAction(FermionOperator<Impl> &_BOp,
|
||||
LinearFunction<FermionField> & HS
|
||||
) : BOp(_BOp),
|
||||
HeatbathSolver(HS),
|
||||
Chi(_BOp.FermionGrid())
|
||||
{};
|
||||
|
||||
virtual std::string action_name(){return "TwoFlavourBosonPseudoFermionAction";}
|
||||
|
||||
virtual std::string LogParameters(){
|
||||
std::stringstream sstream;
|
||||
sstream << GridLogMessage << "["<<action_name()<<"] has no parameters" << std::endl;
|
||||
return sstream.str();
|
||||
}
|
||||
|
||||
virtual void refresh(const GaugeField &U, GridSerialRNG &sRNG, GridParallelRNG& pRNG) {
|
||||
// P(chi) = e^{- chi^dag BdagB chi} ; chi = B^-1 eta ; P(eta) = e^{-eta^dag eta}
|
||||
// e^{-x^2/2 sig^2} => sig^2 = 0.5 ; eta enters with width 1/sqrt(2).
|
||||
RealD scale = std::sqrt(0.5);
|
||||
FermionField eta(BOp.FermionGrid());
|
||||
gaussian(pRNG,eta);
|
||||
eta = eta * scale;
|
||||
refresh(U,eta);
|
||||
}
|
||||
|
||||
// Deterministic-noise variant (test hook):
|
||||
// after this, S(U) == norm2(eta) exactly (to solver tolerance).
|
||||
void refresh(const GaugeField &U, const FermionField &eta) {
|
||||
BOp.ImportGauge(U);
|
||||
Chi = Zero();
|
||||
HeatbathSolver(eta,Chi); // Chi = B^-1 eta : the ONLY solve
|
||||
std::cout << GridLogMessage << action_name() << " refresh |Chi|^2 = "<< norm2(Chi)<<std::endl;
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////
|
||||
// S2 = |B chi|^2 -- matrix multiply only
|
||||
//////////////////////////////////////////////////////
|
||||
virtual RealD S(const GaugeField &U) {
|
||||
BOp.ImportGauge(U);
|
||||
|
||||
FermionField w(BOp.FermionGrid());
|
||||
BOp.M(Chi,w); // w = B chi
|
||||
RealD action = norm2(w);
|
||||
return action;
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////
|
||||
// dS2 = chi^dag dBdag w + w^dag dB chi , w = B chi
|
||||
// NO solves.
|
||||
//////////////////////////////////////////////////////
|
||||
virtual void deriv(const GaugeField &U,GaugeField & dSdU) {
|
||||
BOp.ImportGauge(U);
|
||||
|
||||
FermionField w(BOp.FermionGrid());
|
||||
GaugeField force(BOp.GaugeGrid());
|
||||
|
||||
BOp.M(Chi,w); // w = B chi
|
||||
|
||||
BOp.MDeriv(force, Chi, w, DaggerYes); dSdU = force;
|
||||
BOp.MDeriv(force, w, Chi, DaggerNo ); dSdU = dSdU+force;
|
||||
|
||||
dSdU *= -1.0; // Grid action sign convention (cf TwoFlavourRatio.h)
|
||||
};
|
||||
};
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
@@ -0,0 +1,264 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: ./lib/qcd/action/pseudofermion/TwoFlavourRatio4DPseudoFermion.h
|
||||
|
||||
Copyright (C) 2026
|
||||
|
||||
Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#pragma once
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// Two flavour ratio with FOUR dimensional pseudofermion, UNpreconditioned
|
||||
// (full grid) solves.
|
||||
//
|
||||
// Companion to TwoFlavourRatioEO4DPseudoFermion.h but with the solver
|
||||
// plumbing exposed as LinearFunction<FermionField> objects that already
|
||||
// know their operator -- the natural interface for the non-Hermitian
|
||||
// multigrid GCR stack (PVdagM), which solves M and Mdag DIRECTLY rather
|
||||
// than through SchurRedBlack normal equations.
|
||||
//
|
||||
// Why: with 5D pseudofermions the squared-operator formulation hands
|
||||
// normal-equation solvers (MdagM)^-1 phi AND Mdag^-1 phi from ONE Krylov
|
||||
// space; a direct solver must solve twice, halving its per-solve gain.
|
||||
// The 4D pseudofermion action needs one M^-1 and one M^-dag solve per
|
||||
// force evaluation FOR BOTH solver families, so the direct-solver gain
|
||||
// carries through undiluted. In addition phi4 is Ls-agnostic, so the
|
||||
// force can be evaluated with a reduced-Ls operator pair while the
|
||||
// accept/reject uses full Ls (inexact force, exact action).
|
||||
//
|
||||
// Solver slots (all full-grid 5D LinearFunctions, solution overwritten,
|
||||
// zero guess imposed internally):
|
||||
// DerivMinvSolver : x = M^-1 b (DenOp)
|
||||
// DerivMdagInvSolver : x = M^-dag b (DenOp). For G5R5-hermitian
|
||||
// actions this may be implemented by the caller as
|
||||
// G5R5 . DerivMinvSolver . G5R5 -- no adjoint
|
||||
// multigrid needed.
|
||||
// ActionMinvSolver : x = M^-1 b (DenOp, accept/reject tolerance)
|
||||
// HeatbathVinvSolver : x = V^-1 b (NumOp)
|
||||
//
|
||||
// 4D <-> 5D wall maps: the action is S = | P (M^-1 V) Pdag phi4 |^2 where
|
||||
// (P,Pdag) MUST be a mutually adjoint pair for S and deriv to be
|
||||
// consistent. Two candidate conventions, selected by solution_walls:
|
||||
// true : P = P_- psi(0) + P_+ psi(Ls-1) (solution walls, matches
|
||||
// ExportPhysicalFermionSolution) and Pdag its literal adjoint.
|
||||
// false : P = P_+ psi(0) + P_- psi(Ls-1) (source walls, Pdag matches
|
||||
// ImportUnphysicalFermion).
|
||||
// The heatbath is exact iff [P M^-1 V Pdag][P V^-1 M Pdag] = 1 (the 4D
|
||||
// effective-operator composition identity); which convention satisfies it
|
||||
// is settled numerically by the refresh test S == 0.5*|eta4|^2 exactly.
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template<class Impl>
|
||||
class TwoFlavourRatio4DPseudoFermionAction : public Action<typename Impl::GaugeField> {
|
||||
public:
|
||||
INHERIT_IMPL_TYPES(Impl);
|
||||
|
||||
private:
|
||||
typedef FermionOperator<Impl> FermOp;
|
||||
FermionOperator<Impl> & NumOp;// the basic operator (V)
|
||||
FermionOperator<Impl> & DenOp;// the basic operator (M)
|
||||
|
||||
LinearFunction<FermionField> &DerivMinvSolver;
|
||||
LinearFunction<FermionField> &DerivMdagInvSolver;
|
||||
LinearFunction<FermionField> &ActionMinvSolver;
|
||||
LinearFunction<FermionField> &HeatbathVinvSolver;
|
||||
|
||||
FermionField phi4; // the pseudo fermion field for this trajectory
|
||||
|
||||
int solution_walls; // wall convention for the (P,Pdag) pair; see header
|
||||
|
||||
////////////////////////////////////////////////////////////////////
|
||||
// The mutually adjoint 4D <-> 5D pair.
|
||||
// Wall4D : q4 = P psi5 (extract)
|
||||
// Wall4DAdj : psi5 = Pdag q4 (insert; literal adjoint of Wall4D)
|
||||
////////////////////////////////////////////////////////////////////
|
||||
void Wall4D(const FermionField &psi5, FermionField &q4)
|
||||
{
|
||||
int Ls = NumOp.FermionGrid()->_fdimensions[0];
|
||||
FermionField tmp(NumOp.FermionGrid());
|
||||
if ( solution_walls ) {
|
||||
// q4 = P_- psi(0) + P_+ psi(Ls-1)
|
||||
axpby_ssp_pminus(tmp, 0., psi5, 1., psi5, 0, 0);
|
||||
axpby_ssp_pplus (tmp, 1., tmp , 1., psi5, 0, Ls-1);
|
||||
} else {
|
||||
// q4 = P_+ psi(0) + P_- psi(Ls-1)
|
||||
axpby_ssp_pplus (tmp, 0., psi5, 1., psi5, 0, 0);
|
||||
axpby_ssp_pminus(tmp, 1., tmp , 1., psi5, 0, Ls-1);
|
||||
}
|
||||
ExtractSlice(q4, tmp, 0, 0);
|
||||
}
|
||||
void Wall4DAdj(const FermionField &q4, FermionField &psi5)
|
||||
{
|
||||
int Ls = NumOp.FermionGrid()->_fdimensions[0];
|
||||
FermionField tmp(NumOp.FermionGrid());
|
||||
tmp = Zero();
|
||||
InsertSlice(q4, tmp, 0 , 0);
|
||||
InsertSlice(q4, tmp, Ls-1, 0);
|
||||
if ( solution_walls ) {
|
||||
// psi(0) = P_- q4 ; psi(Ls-1) = P_+ q4
|
||||
axpby_ssp_pminus(tmp, 0., tmp, 1., tmp, 0 , 0);
|
||||
axpby_ssp_pplus (tmp, 0., tmp, 1., tmp, Ls-1, Ls-1);
|
||||
} else {
|
||||
// psi(0) = P_+ q4 ; psi(Ls-1) = P_- q4
|
||||
axpby_ssp_pplus (tmp, 0., tmp, 1., tmp, 0 , 0);
|
||||
axpby_ssp_pminus(tmp, 0., tmp, 1., tmp, Ls-1, Ls-1);
|
||||
}
|
||||
psi5 = tmp;
|
||||
}
|
||||
|
||||
public:
|
||||
TwoFlavourRatio4DPseudoFermionAction(FermionOperator<Impl> &_NumOp,
|
||||
FermionOperator<Impl> &_DenOp,
|
||||
LinearFunction<FermionField> & DMS,
|
||||
LinearFunction<FermionField> & DMDS,
|
||||
LinearFunction<FermionField> & AMS,
|
||||
LinearFunction<FermionField> & HVS,
|
||||
int _solution_walls = 1
|
||||
) : NumOp(_NumOp),
|
||||
DenOp(_DenOp),
|
||||
DerivMinvSolver(DMS),
|
||||
DerivMdagInvSolver(DMDS),
|
||||
ActionMinvSolver(AMS),
|
||||
HeatbathVinvSolver(HVS),
|
||||
phi4(_NumOp.GaugeGrid()),
|
||||
solution_walls(_solution_walls)
|
||||
{};
|
||||
|
||||
virtual std::string action_name(){return "TwoFlavourRatio4DPseudoFermionAction";}
|
||||
|
||||
virtual std::string LogParameters(){
|
||||
std::stringstream sstream;
|
||||
sstream << GridLogMessage << "["<<action_name()<<"] solution_walls " << solution_walls << std::endl;
|
||||
return sstream.str();
|
||||
}
|
||||
|
||||
virtual void refresh(const GaugeField &U, GridSerialRNG &sRNG, GridParallelRNG& pRNG) {
|
||||
|
||||
// P(phi4) = e^{- phi4^dag Beff^dag Beff phi4} ; Beff = P M^-1 V Pdag
|
||||
//
|
||||
// NumOp == V
|
||||
// DenOp == M
|
||||
//
|
||||
// Take phi4 = P V^-1 M Pdag eta4 ( = Beff^-1 eta4 by the composition
|
||||
// identity; verified numerically by S == 0.5 |eta4|^2 after refresh )
|
||||
//
|
||||
// P(eta) = e^{- eta^dag eta} ; e^{-x^2/2 sig^2} => sig^2 = 0.5
|
||||
// so eta enters with width 1/sqrt(2).
|
||||
//
|
||||
RealD scale = std::sqrt(0.5);
|
||||
|
||||
FermionField eta4(NumOp.GaugeGrid());
|
||||
FermionField eta5(NumOp.FermionGrid());
|
||||
FermionField tmp (NumOp.FermionGrid());
|
||||
FermionField phi5(NumOp.FermionGrid());
|
||||
|
||||
gaussian(pRNG,eta4);
|
||||
|
||||
NumOp.ImportGauge(U);
|
||||
DenOp.ImportGauge(U);
|
||||
|
||||
Wall4DAdj(eta4,eta5); // eta5 = Pdag eta4
|
||||
DenOp.M(eta5,tmp); // tmp = M eta5
|
||||
phi5 = Zero();
|
||||
HeatbathVinvSolver(tmp,phi5); // phi5 = V^-1 M eta5
|
||||
Wall4D(phi5,phi4); // phi4 = P phi5
|
||||
phi4 = phi4*scale;
|
||||
|
||||
std::cout << GridLogMessage << "4d pf (non-EO) refresh "<< norm2(phi4)<<"\n";
|
||||
};
|
||||
|
||||
//////////////////////////////////////////////////////
|
||||
// S = phi4^dag (Pdag^dag V^dag M^-dag P^dag) (P M^-1 V Pdag) phi4
|
||||
// = | P M^-1 V Pdag phi4 |^2
|
||||
//////////////////////////////////////////////////////
|
||||
virtual RealD S(const GaugeField &U) {
|
||||
|
||||
NumOp.ImportGauge(U);
|
||||
DenOp.ImportGauge(U);
|
||||
|
||||
FermionField Y4 (NumOp.GaugeGrid());
|
||||
FermionField phi5(NumOp.FermionGrid());
|
||||
FermionField X (NumOp.FermionGrid());
|
||||
FermionField Y (NumOp.FermionGrid());
|
||||
|
||||
Wall4DAdj(phi4,phi5); // phi5 = Pdag phi4
|
||||
NumOp.M(phi5,X); // X = V phi5
|
||||
Y = Zero();
|
||||
ActionMinvSolver(X,Y); // Y = M^-1 V phi5
|
||||
Wall4D(Y,Y4); // Y4 = P Y
|
||||
|
||||
RealD action = norm2(Y4);
|
||||
|
||||
return action;
|
||||
};
|
||||
|
||||
//////////////////////////////////////////////////////
|
||||
// dS/du = 2 Re [ (M^-dag Pdag w4)^dag dV Pdag phi4 ]
|
||||
// - 2 Re [ (M^-dag Pdag w4)^dag dM (M^-1 V Pdag phi4) ]
|
||||
// with w4 = P M^-1 V Pdag phi4.
|
||||
// Two first-power solves: one M^-1, one M^-dag.
|
||||
//////////////////////////////////////////////////////
|
||||
virtual void deriv(const GaugeField &U,GaugeField & dSdU) {
|
||||
|
||||
NumOp.ImportGauge(U);
|
||||
DenOp.ImportGauge(U);
|
||||
|
||||
FermionField phi5 (NumOp.FermionGrid());
|
||||
FermionField Vphi (NumOp.FermionGrid());
|
||||
FermionField MinvVphi (NumOp.FermionGrid());
|
||||
FermionField w4 (NumOp.GaugeGrid());
|
||||
FermionField Y (NumOp.FermionGrid());
|
||||
FermionField MdagInvPdagW (NumOp.FermionGrid());
|
||||
|
||||
GaugeField force(NumOp.GaugeGrid());
|
||||
|
||||
Wall4DAdj(phi4,phi5); // phi5 = Pdag phi4
|
||||
NumOp.M(phi5,Vphi); // Vphi = V phi5
|
||||
MinvVphi = Zero();
|
||||
DerivMinvSolver(Vphi,MinvVphi); // MinvVphi = M^-1 V phi5
|
||||
std::cout << GridLogMessage << "4d pf (non-EO) deriv solve "<< norm2(MinvVphi)<<"\n";
|
||||
|
||||
// Project onto the physical 4D subspace and back: Y = Pdag P MinvVphi.
|
||||
// Pdag here MUST be the literal adjoint of the P used in S, else the
|
||||
// force is inconsistent with the action.
|
||||
Wall4D(MinvVphi,w4); // w4 = P MinvVphi
|
||||
Wall4DAdj(w4,Y); // Y = Pdag w4
|
||||
|
||||
MdagInvPdagW = Zero();
|
||||
DerivMdagInvSolver(Y,MdagInvPdagW); // = M^-dag Pdag w4 (adjoint solve)
|
||||
std::cout << GridLogMessage << "4d pf (non-EO) deriv solve dag "<< norm2(MdagInvPdagW)<<"\n";
|
||||
|
||||
// phi^dag (Pdag' Vdag Mdag^-1 P') (dV) Pdag phi + h.c.
|
||||
NumOp.MDeriv(force, MdagInvPdagW, phi5, DaggerNo ); dSdU=force;
|
||||
NumOp.MDeriv(force, phi5, MdagInvPdagW, DaggerYes); dSdU=dSdU+force;
|
||||
|
||||
// - phi^dag ( ... Mdag^-1 ) dM ( M^-1 V ... ) phi + h.c.
|
||||
DenOp.MDeriv(force, MdagInvPdagW, MinvVphi, DaggerNo ); dSdU=dSdU-force;
|
||||
DenOp.MDeriv(force, MinvVphi, MdagInvPdagW, DaggerYes); dSdU=dSdU-force;
|
||||
|
||||
dSdU *= -1.0;
|
||||
};
|
||||
};
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
@@ -0,0 +1,206 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: ./lib/qcd/action/pseudofermion/TwoFlavourRatioLeftPrec.h
|
||||
|
||||
Copyright (C) 2026
|
||||
|
||||
Author: Peter Boyle <pboyle@bnl.gov>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#pragma once
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// Two flavour ratio with LEFT-PRECONDITIONED solves.
|
||||
//
|
||||
// Same action content as TwoFlavourRatio.h:
|
||||
//
|
||||
// S = phi^dag V (Mdag M)^-1 Vdag phi ==> det[ Mdag M / Vdag V ]
|
||||
//
|
||||
// (V = NumOp the heavier / Pauli-Villars operator, M = DenOp the lighter),
|
||||
// but organised around the composite
|
||||
//
|
||||
// F = Vdag M
|
||||
//
|
||||
// which is the 2-hop-coarsenable operator the non-Hermitian multigrid
|
||||
// serves. Solving M X = b as F X = Vdag b is LEFT PRECONDITIONING by
|
||||
// Vdag; the determinant/action layer is the standard quotient, and all
|
||||
// novelty is confined to the solver contract.
|
||||
//
|
||||
// TwoFlavourRatio.h is tied to a normal-equations solver: one (MdagM)^-1
|
||||
// solve, then Y = M X gives Mdag^-1 Vdag phi almost free. The left-
|
||||
// preconditioned idiom is DIFFERENT: the chain
|
||||
//
|
||||
// b = Vdag phi
|
||||
// z : Fdag z = b (adjoint F solve)
|
||||
// Y = V z (= Mdag^-1 Vdag phi -- harvested from solve 1)
|
||||
// s = Vdag Y (= Vdag V z)
|
||||
// X : F X = s (forward F solve; X = (MdagM)^-1 Vdag phi)
|
||||
//
|
||||
// yields Y BEFORE X (so S(U) needs only the adjoint solve), with Y's
|
||||
// accuracy independent of the second solve. Force terms are then the
|
||||
// standard four MDeriv insertions of TwoFlavourRatio.
|
||||
//
|
||||
// Solver slots are LinearFunctions with the F-SOLVE contract (solution
|
||||
// overwritten, zero guess imposed internally):
|
||||
// ForwardSolver(b,x) : F x = b
|
||||
// AdjointSolver(b,z) : Fdag z = b
|
||||
// implemented in production by the multigrid-GCR stack (forward cycle and
|
||||
// adjoint cycle); in tests by CG on the composite normal equations.
|
||||
// HeatbathSolver(b,x) : x = (Vdag V)^-1 b -- heavy operator, plain CG.
|
||||
//
|
||||
// Heatbath is exact by operator algebra: phi = V (VdagV)^-1 Mdag eta
|
||||
// ==> S = | Mdag^-1 Vdag phi |^2 = |eta|^2 (to solver tolerance); the
|
||||
// deterministic refresh(U,eta) hook below is the test point.
|
||||
//
|
||||
// Hasenbusch: nothing requires V to have mass one; any (heavier,lighter)
|
||||
// pair works, F(V,M) = Vdag M coarsenable by the same machinery, rungs'
|
||||
// solves are F-family (mrhs-batchable, mass-shared coarse space).
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
template<class Impl>
|
||||
class TwoFlavourRatioLeftPrecPseudoFermionAction : public Action<typename Impl::GaugeField> {
|
||||
public:
|
||||
INHERIT_IMPL_TYPES(Impl);
|
||||
|
||||
private:
|
||||
FermionOperator<Impl> & NumOp;// V
|
||||
FermionOperator<Impl> & DenOp;// M
|
||||
|
||||
LinearFunction<FermionField> &DerivForwardSolver; // F x = b, MD tolerance
|
||||
LinearFunction<FermionField> &DerivAdjointSolver; // Fdag z = b, MD tolerance
|
||||
LinearFunction<FermionField> &ActionAdjointSolver; // Fdag z = b, accept/reject tolerance
|
||||
LinearFunction<FermionField> &HeatbathSolver; // (VdagV)^-1 b, heavy op
|
||||
|
||||
FermionField Phi; // the pseudo fermion field for this trajectory
|
||||
|
||||
public:
|
||||
TwoFlavourRatioLeftPrecPseudoFermionAction(FermionOperator<Impl> &_NumOp,
|
||||
FermionOperator<Impl> &_DenOp,
|
||||
LinearFunction<FermionField> & DFS,
|
||||
LinearFunction<FermionField> & DAS,
|
||||
LinearFunction<FermionField> & AAS,
|
||||
LinearFunction<FermionField> & HS
|
||||
) : NumOp(_NumOp),
|
||||
DenOp(_DenOp),
|
||||
DerivForwardSolver(DFS),
|
||||
DerivAdjointSolver(DAS),
|
||||
ActionAdjointSolver(AAS),
|
||||
HeatbathSolver(HS),
|
||||
Phi(_NumOp.FermionGrid())
|
||||
{};
|
||||
|
||||
virtual std::string action_name(){return "TwoFlavourRatioLeftPrecPseudoFermionAction";}
|
||||
|
||||
virtual std::string LogParameters(){
|
||||
std::stringstream sstream;
|
||||
sstream << GridLogMessage << "["<<action_name()<<"] has no parameters" << std::endl;
|
||||
return sstream.str();
|
||||
}
|
||||
|
||||
virtual void refresh(const GaugeField &U, GridSerialRNG &sRNG, GridParallelRNG& pRNG) {
|
||||
// P(phi) = e^{- phi^dag V (MdagM)^-1 Vdag phi} ; phi = Vdag^-1 Mdag eta
|
||||
// e^{-x^2/2 sig^2} => sig^2 = 0.5 ; eta enters with width 1/sqrt(2).
|
||||
RealD scale = std::sqrt(0.5);
|
||||
FermionField eta(NumOp.FermionGrid());
|
||||
gaussian(pRNG,eta);
|
||||
eta = eta * scale;
|
||||
refresh(U,eta);
|
||||
}
|
||||
|
||||
// Deterministic-noise variant (test hook):
|
||||
// after this, S(U) == norm2(eta) exactly (to solver tolerance).
|
||||
void refresh(const GaugeField &U, const FermionField &eta) {
|
||||
NumOp.ImportGauge(U);
|
||||
DenOp.ImportGauge(U);
|
||||
|
||||
FermionField tmp(NumOp.FermionGrid());
|
||||
FermionField w (NumOp.FermionGrid());
|
||||
|
||||
DenOp.Mdag(eta,tmp); // tmp = Mdag eta
|
||||
w = Zero();
|
||||
HeatbathSolver(tmp,w); // w = (VdagV)^-1 Mdag eta
|
||||
NumOp.M(w,Phi); // Phi = V (VdagV)^-1 Mdag eta = Vdag^-1 Mdag eta
|
||||
std::cout << GridLogMessage << action_name() << " refresh |Phi|^2 = "<< norm2(Phi)<<std::endl;
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////
|
||||
// S = phi^dag V (MdagM)^-1 Vdag phi = | Mdag^-1 Vdag phi |^2
|
||||
// ONE adjoint F solve: Y = V Fdag^-1 Vdag phi = Mdag^-1 Vdag phi
|
||||
//////////////////////////////////////////////////////
|
||||
virtual RealD S(const GaugeField &U) {
|
||||
NumOp.ImportGauge(U);
|
||||
DenOp.ImportGauge(U);
|
||||
|
||||
FermionField b(NumOp.FermionGrid());
|
||||
FermionField z(NumOp.FermionGrid());
|
||||
FermionField Y(NumOp.FermionGrid());
|
||||
|
||||
NumOp.Mdag(Phi,b); // b = Vdag phi
|
||||
z = Zero();
|
||||
ActionAdjointSolver(b,z); // Fdag z = b
|
||||
NumOp.M(z,Y); // Y = V z = Mdag^-1 Vdag phi
|
||||
|
||||
RealD action = norm2(Y);
|
||||
return action;
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////
|
||||
// dS/du = phi^dag dV (MdagM)^-1 Vdag phi
|
||||
// - phi^dag V (MdagM)^-1 [ Mdag dM + dMdag M ] (MdagM)^-1 Vdag phi
|
||||
// + phi^dag V (MdagM)^-1 dVdag phi
|
||||
// Identical force insertions to TwoFlavourRatio.h; X and Y from the
|
||||
// left-preconditioned chain (Y harvested from the adjoint solve).
|
||||
//////////////////////////////////////////////////////
|
||||
virtual void deriv(const GaugeField &U,GaugeField & dSdU) {
|
||||
NumOp.ImportGauge(U);
|
||||
DenOp.ImportGauge(U);
|
||||
|
||||
FermionField b(NumOp.FermionGrid());
|
||||
FermionField z(NumOp.FermionGrid());
|
||||
FermionField Y(NumOp.FermionGrid());
|
||||
FermionField s(NumOp.FermionGrid());
|
||||
FermionField X(NumOp.FermionGrid());
|
||||
|
||||
GaugeField force(NumOp.GaugeGrid());
|
||||
|
||||
NumOp.Mdag(Phi,b); // b = Vdag phi
|
||||
z = Zero();
|
||||
DerivAdjointSolver(b,z); // Fdag z = b
|
||||
NumOp.M(z,Y); // Y = V z = Mdag^-1 Vdag phi (solve-1 harvest)
|
||||
NumOp.Mdag(Y,s); // s = Vdag V z
|
||||
X = Zero();
|
||||
DerivForwardSolver(s,X); // F X = s ==> X = (MdagM)^-1 Vdag phi
|
||||
|
||||
// phi^dag V (MdagM)^-1 dVdag phi
|
||||
NumOp.MDeriv(force , X, Phi, DaggerYes); dSdU = force;
|
||||
// phi^dag dV (MdagM)^-1 Vdag phi
|
||||
NumOp.MDeriv(force , Phi, X, DaggerNo ); dSdU = dSdU+force;
|
||||
// - phi^dag V (MdagM)^-1 Mdag dM (MdagM)^-1 Vdag phi
|
||||
// - phi^dag V (MdagM)^-1 dMdag M (MdagM)^-1 Vdag phi
|
||||
DenOp.MDeriv(force, Y, X, DaggerNo ); dSdU = dSdU-force;
|
||||
DenOp.MDeriv(force, X, Y, DaggerYes); dSdU = dSdU-force;
|
||||
|
||||
dSdU *= -1.0;
|
||||
};
|
||||
};
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
@@ -103,6 +103,18 @@ class PolyakovMod: public ObservableModule<PolyakovLogger<Impl>, NoParameters>{
|
||||
PolyakovMod(): ObsBase(NoParameters()){}
|
||||
};
|
||||
|
||||
template < class Impl >
|
||||
class SpatialPolyakovMod: public ObservableModule<SpatialPolyakovLogger<Impl>, NoParameters>{
|
||||
typedef ObservableModule<SpatialPolyakovLogger<Impl>, NoParameters> ObsBase;
|
||||
using ObsBase::ObsBase; // for constructors
|
||||
|
||||
// acquire resource
|
||||
virtual void initialize(){
|
||||
this->ObservablePtr.reset(new SpatialPolyakovLogger<Impl>());
|
||||
}
|
||||
public:
|
||||
SpatialPolyakovMod(): ObsBase(NoParameters()){}
|
||||
};
|
||||
|
||||
template < class Impl >
|
||||
class TopologicalChargeMod: public ObservableModule<TopologicalCharge<Impl>, TopologyObsParameters>{
|
||||
|
||||
@@ -2,11 +2,12 @@
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: ./lib/qcd/modules/polyakov_line.h
|
||||
Source file: ./Grid/qcd/observables/polyakov_loop.h
|
||||
|
||||
Copyright (C) 2017
|
||||
Copyright (C) 2025
|
||||
|
||||
Author: David Preti <david.preti@csic.es>
|
||||
Author: Alexis Verney-Provatas <2414441@swansea.ac.uk>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
@@ -60,4 +61,43 @@ class PolyakovLogger : public HmcObservable<typename Impl::Field> {
|
||||
}
|
||||
};
|
||||
|
||||
template <class Impl>
|
||||
class SpatialPolyakovLogger : public HmcObservable<typename Impl::Field> {
|
||||
public:
|
||||
// here forces the Impl to be of gauge fields
|
||||
// if not the compiler will complain
|
||||
INHERIT_GIMPL_TYPES(Impl);
|
||||
|
||||
// necessary for HmcObservable compatibility
|
||||
typedef typename Impl::Field Field;
|
||||
|
||||
void TrajectoryComplete(int traj,
|
||||
Field &U,
|
||||
GridSerialRNG &sRNG,
|
||||
GridParallelRNG &pRNG) {
|
||||
|
||||
// Save current numerical output precision
|
||||
int def_prec = std::cout.precision();
|
||||
|
||||
// Assume that the dimensions are D=3+1
|
||||
int Ndim = 3;
|
||||
ComplexD polyakov;
|
||||
|
||||
// Iterate over the spatial directions and print the average spatial polyakov loop
|
||||
// over them
|
||||
for (int idx=0; idx<Ndim; idx++) {
|
||||
polyakov = WilsonLoops<Impl>::avgPolyakovLoop(U, idx);
|
||||
|
||||
std::cout << GridLogMessage
|
||||
<< std::setprecision(std::numeric_limits<Real>::digits10 + 1)
|
||||
<< "Polyakov Loop in the " << idx << " spatial direction : [ " << traj << " ] "<< polyakov << std::endl;
|
||||
|
||||
}
|
||||
|
||||
// Return to original output precision
|
||||
std::cout.precision(def_prec);
|
||||
|
||||
}
|
||||
};
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
|
||||
@@ -291,8 +291,8 @@ public:
|
||||
int idx=0;
|
||||
for(int mu=0;mu<4;mu++){
|
||||
for(int nu=0;nu<4;nu++){
|
||||
if ( mu!=nu) GRID_ASSERT(this->StoutSmearing->SmearRho[idx]==rho);
|
||||
else GRID_ASSERT(this->StoutSmearing->SmearRho[idx]==0.0);
|
||||
if ( mu!=nu) assert(this->StoutSmearing->SmearRho[idx]==rho);
|
||||
else assert(this->StoutSmearing->SmearRho[idx]==0.0);
|
||||
idx++;
|
||||
}}
|
||||
//////////////////////////////////////////////////////////////////
|
||||
@@ -825,6 +825,7 @@ public:
|
||||
virtual void fill_smearedSet(GaugeField &U)
|
||||
{
|
||||
this->ThinLinks = &U; // attach the smearing routine to the field U
|
||||
std::cout << GridLogMessage << " fill_smearedSet " << WilsonLoops<PeriodicGimplR>::avgPlaquette(U) << std::endl;
|
||||
|
||||
// check the pointer is not null
|
||||
if (this->ThinLinks == NULL)
|
||||
@@ -846,6 +847,8 @@ public:
|
||||
ApplyMask(smeared_A,smearLvl);
|
||||
smeared_B = previous_u;
|
||||
ApplyMask(smeared_B,smearLvl);
|
||||
std::cout << GridLogMessage << " smeared_A " << norm2(smeared_A) << std::endl;
|
||||
std::cout << GridLogMessage << " smeared_B " << norm2(smeared_B) << std::endl;
|
||||
// Replace only the masked portion
|
||||
this->SmearedSet[smearLvl] = previous_u-smeared_B + smeared_A;
|
||||
previous_u = this->SmearedSet[smearLvl];
|
||||
@@ -934,10 +937,10 @@ public:
|
||||
SmearedConfigurationMasked(GridCartesian* _UGrid, unsigned int Nsmear, Smear_Stout<Gimpl>& Stout)
|
||||
: SmearedConfiguration<Gimpl>(_UGrid, Nsmear,Stout)
|
||||
{
|
||||
GRID_ASSERT(Nsmear%(2*Nd)==0); // Or multiply by 8??
|
||||
assert(Nsmear%(2*Nd)==0); // Or multiply by 8??
|
||||
|
||||
// was resized in base class
|
||||
GRID_ASSERT(this->SmearedSet.size()==Nsmear);
|
||||
assert(this->SmearedSet.size()==Nsmear);
|
||||
|
||||
GridRedBlackCartesian * UrbGrid;
|
||||
UrbGrid = SpaceTimeGrid::makeFourDimRedBlackGrid(_UGrid);
|
||||
|
||||
@@ -54,7 +54,7 @@ public:
|
||||
// Usual cases are not used
|
||||
//////////////////////////////////
|
||||
virtual void refresh(const GaugeField &U, GridSerialRNG &sRNG, GridParallelRNG &pRNG){ GRID_ASSERT(0);};
|
||||
virtual RealD S(const GaugeField &U) { GRID_ASSERT(0); }
|
||||
virtual RealD S(const GaugeField &U) { GRID_ASSERT(0); return 0; }
|
||||
virtual void deriv(const GaugeField &U, GaugeField &dSdU) { GRID_ASSERT(0); }
|
||||
|
||||
//////////////////////////////////
|
||||
|
||||
@@ -51,10 +51,16 @@ protected:
|
||||
|
||||
public:
|
||||
|
||||
//Define the action used to evolve the plaquettes
|
||||
//(Lüscher: https://arxiv.org/pdf/1006.4518 eq. 1.4)
|
||||
//V'(t) = -g^2 * ( d/dVt S[Vt](g) ) * Vt
|
||||
// = -g^2 * ( d/dVt (1/g^2 * sum_p Re tr{ 1 - Vt(p) } ) ) * Vt
|
||||
// = - d/dVt ( sum_p ( Nc - Re tr Vt(p) ) * Vt
|
||||
// = - d/dVt ( Nc * sum_p ( 1 - Re tr Vt(p)/Nc ) ) * Vt
|
||||
// = - d/dVt SG[Vt](Nc) * Vt
|
||||
explicit WilsonFlowBase(unsigned int meas_interval =1) {
|
||||
|
||||
SG = (ActionBase *) new WilsonGaugeAction<Gimpl>(3.0);
|
||||
// WilsonGaugeAction with beta 3.0
|
||||
SG = (ActionBase *) new WilsonGaugeAction<Gimpl>(Gimpl::num_colours);
|
||||
setDefaultMeasurements(meas_interval);
|
||||
}
|
||||
|
||||
@@ -149,9 +155,17 @@ public:
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Implementations
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
|
||||
//Compute t^2 <E(t)> for time from the plaquette form
|
||||
//(Lüscher: https://arxiv.org/pdf/1006.4518 eq. 3.1)
|
||||
//E(t) = 2 * sum_p Retr{ 1 - Vt(p) } =
|
||||
// = 2 * sum_p ( Nc - Retr Vt(p) ) =
|
||||
// = 2 * Nc * sum_p ( 1 - Retr Vt(p)/Nc )
|
||||
// = 2 * SG[Vt](Nc)
|
||||
//We divide by the volume to get an energy density per site, as is convention
|
||||
template <class Gimpl>
|
||||
RealD WilsonFlowBase<Gimpl>::energyDensityPlaquette(const RealD t, const GaugeField& U){
|
||||
static WilsonGaugeAction<Gimpl> SG(3.0);
|
||||
static WilsonGaugeAction<Gimpl> SG(Gimpl::num_colours);
|
||||
return 2.0 * t * t * SG.S(U)/U.Grid()->gSites();
|
||||
}
|
||||
|
||||
|
||||
+60
-387
@@ -62,16 +62,37 @@ public:
|
||||
const FermionField *rhs_vj,
|
||||
std::vector<Gamma::Algebra> gammas,
|
||||
const std::vector<ComplexField > &mom,
|
||||
int orthogdim, double *t_kernel = nullptr, double *t_gsum = nullptr);
|
||||
int orthogdim);
|
||||
template <typename TensorType>
|
||||
static void MesonField(TensorType &mat,
|
||||
const FermionField *lhs_wi,
|
||||
const FermionField *rhs_vj,
|
||||
std::vector<Gamma::Algebra> gammas,
|
||||
const std::vector<ComplexField > &mom,
|
||||
int orthogdim,double *timer)
|
||||
{
|
||||
MesonField(mat,lhs_wi,rhs_vj,gammas,mom,orthogdim);
|
||||
}
|
||||
|
||||
template <typename TensorType> // output: rank 5 tensor, e.g. Eigen::Tensor<ComplexD, 5>
|
||||
static void AslashField(TensorType &mat,
|
||||
const FermionField *lhs_wi,
|
||||
const FermionField *rhs_vj,
|
||||
const std::vector<ComplexField> &emB0,
|
||||
const std::vector<ComplexField> &emB1,
|
||||
int orthogdim, double *t_kernel = nullptr, double *t_gsum = nullptr);
|
||||
const FermionField *lhs_wi,
|
||||
const FermionField *rhs_vj,
|
||||
const std::vector<ComplexField> &emB0,
|
||||
const std::vector<ComplexField> &emB1,
|
||||
int orthogdim);
|
||||
|
||||
template <typename TensorType> // output: rank 5 tensor, e.g. Eigen::Tensor<ComplexD, 5>
|
||||
static void AslashField(TensorType &mat,
|
||||
const FermionField *lhs_wi,
|
||||
const FermionField *rhs_vj,
|
||||
const std::vector<ComplexField> &emB0,
|
||||
const std::vector<ComplexField> &emB1,
|
||||
int orthogdim,double *timer)
|
||||
{
|
||||
AslashField(mat,lhs_wi,rhs_vj,emB0,emB1,orthogdim);
|
||||
}
|
||||
|
||||
template <typename TensorType>
|
||||
typename std::enable_if<(std::is_same<Eigen::Tensor<ComplexD,3>, TensorType>::value ||
|
||||
std::is_same<Eigen::TensorMap<Eigen::Tensor<Complex, 3, Eigen::RowMajor>>, TensorType>::value),
|
||||
@@ -136,7 +157,7 @@ typedef iVecComplex<vComplex > vVecComplex;
|
||||
typedef Lattice<vVecComplex> LatticeVecComplex;
|
||||
|
||||
#define A2A_GPU_KERNELS
|
||||
#ifdef A2A_GPU_KERNELS
|
||||
|
||||
template <class FImpl>
|
||||
template <typename TensorType>
|
||||
void A2Autils<FImpl>::MesonField(TensorType &mat,
|
||||
@@ -144,7 +165,7 @@ void A2Autils<FImpl>::MesonField(TensorType &mat,
|
||||
const FermionField *rhs_vj,
|
||||
std::vector<Gamma::Algebra> gammas,
|
||||
const std::vector<ComplexField > &mom,
|
||||
int orthogdim, double *t_kernel, double *t_gsum)
|
||||
int orthogdim)
|
||||
{
|
||||
const int block=A2Ablocking;
|
||||
typedef typename FImpl::SiteSpinor vobj;
|
||||
@@ -173,24 +194,34 @@ void A2Autils<FImpl>::MesonField(TensorType &mat,
|
||||
|
||||
std::cout <<GridLogMessage<< "A2A Meson Field"<<std::endl;
|
||||
MomentumProject<LatticeVecSpinMatrix,ComplexField> MP;
|
||||
std::cout <<GridLogMessage<< "Momentum project constructed"<<std::endl;
|
||||
MP.Allocate(Nmom,grid);
|
||||
std::cout <<GridLogMessage<< "Momentum project allocated"<<std::endl;
|
||||
MP.ImportMomenta(mom);
|
||||
std::cout <<GridLogMessage<< "Momentum project momenta imported"<<std::endl;
|
||||
|
||||
|
||||
double t_view, t_gamma, t_kernel, t_momproj;
|
||||
t_view=0;
|
||||
t_gamma=0;
|
||||
t_kernel=0;
|
||||
t_momproj=0;
|
||||
|
||||
|
||||
std::vector<VecSpinMatrix> sliced;
|
||||
for(int i=0;i<Lblock;i++){
|
||||
t_view -= usecond();
|
||||
autoView(SpinMat_v,SpinMat,AcceleratorWrite);
|
||||
autoView(lhs_v,lhs_wi[i],AcceleratorRead);
|
||||
t_view += usecond();
|
||||
for(int jo=0;jo<Rblock;jo+=block){
|
||||
for(int j=jo;j<MIN(Rblock,jo+block);j++){
|
||||
int jj=j%block;
|
||||
t_view -= usecond();
|
||||
autoView(rhs_v,rhs_vj[j],AcceleratorRead); // Create a vector of views
|
||||
t_view += usecond();
|
||||
//////////////////////////////////////////
|
||||
// Should write a SpinOuterColorTrace
|
||||
//////////////////////////////////////////
|
||||
|
||||
t_kernel -= usecond();
|
||||
accelerator_for(ss,grid->oSites(),(size_t)Nsimd,{
|
||||
auto left = conjugate(lhs_v(ss));
|
||||
auto right = rhs_v(ss);
|
||||
@@ -203,48 +234,38 @@ void A2Autils<FImpl>::MesonField(TensorType &mat,
|
||||
}}
|
||||
coalescedWrite(SpinMat_v[ss],vv);
|
||||
});
|
||||
t_kernel += usecond();
|
||||
|
||||
}// j within block
|
||||
// After getting the sitewise product do the mom phase loop
|
||||
#if 1
|
||||
std::cout <<GridLogMessage<< "A2A contract "<<std::endl;
|
||||
|
||||
assert(orthogdim==Nd-1);
|
||||
t_momproj -= usecond();
|
||||
MP.Project(SpinMat,sliced);
|
||||
std::cout <<GridLogMessage<< "A2A MP Project "<<std::endl;
|
||||
for(int m=0;m<Nmom;m++){
|
||||
for(int t=0;t<Nt;t++){
|
||||
t_momproj += usecond();
|
||||
|
||||
t_gamma -= usecond();
|
||||
thread_for2d( m, Nmom,t,Nt,{
|
||||
// for(int m=0;m<Nmom;m++)
|
||||
// for(int t=0;t<Nt;t++)
|
||||
int idx = t+m*Nt;
|
||||
for(int j=jo;j<MIN(Rblock,jo+block);j++){
|
||||
int jj=j%block;
|
||||
auto tmp = peekIndex<LorentzIndex>(sliced[idx],jj);
|
||||
for(int mu=0;mu<Ngamma;mu++){
|
||||
auto trSG = trace(tmp*Gamma(gammas[mu]));
|
||||
mat(m,mu,t,i,j) = trSG()();
|
||||
mat((long)m,mu,(long)t,i,j) = trSG()();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
for(int m=0;m<Nmom;m++){
|
||||
|
||||
MomSpinMat = SpinMat * mom[m];
|
||||
|
||||
sliceSum(MomSpinMat,sliced,orthogdim);
|
||||
|
||||
for(int mu=0;mu<Ngamma;mu++){
|
||||
for(int t=0;t<sliced.size();t++){
|
||||
for(int j=jo;j<MIN(Rblock,jo+block);j++){
|
||||
int jj=j%block;
|
||||
auto tmp = peekIndex<LorentzIndex>(sliced[t],jj);
|
||||
auto trSG = trace(tmp*Gamma(gammas[mu]));
|
||||
mat(m,mu,t,i,j) = trSG()();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
});
|
||||
t_gamma += usecond();
|
||||
}//jo
|
||||
}
|
||||
std::cout << GridLogMessage<<" A2A::MesonField t_view "<<t_view/1e6<<"s"<<std::endl;
|
||||
std::cout << GridLogMessage<<" A2A::MesonField t_momproj "<<t_momproj/1e6<<"s"<<std::endl;
|
||||
std::cout << GridLogMessage<<" A2A::MesonField t_kernel "<<t_kernel/1e6<<"s"<<std::endl;
|
||||
std::cout << GridLogMessage<<" A2A::MesonField t_gamma "<<t_gamma/1e6<<"s"<<std::endl;
|
||||
|
||||
}
|
||||
|
||||
// "A-slash" field w_i(x)^dag * i * A_mu * gamma_mu * v_j(x)
|
||||
@@ -268,7 +289,7 @@ void A2Autils<FImpl>::AslashField(TensorType &mat,
|
||||
const FermionField *rhs_vj,
|
||||
const std::vector<ComplexField> &emB0,
|
||||
const std::vector<ComplexField> &emB1,
|
||||
int orthogdim, double *t_kernel, double *t_gsum)
|
||||
int orthogdim)
|
||||
{
|
||||
const int block=A2Ablocking;
|
||||
typedef typename FImpl::SiteSpinor vobj;
|
||||
@@ -355,354 +376,6 @@ void A2Autils<FImpl>::AslashField(TensorType &mat,
|
||||
}
|
||||
}
|
||||
|
||||
#else
|
||||
template <class FImpl>
|
||||
template <typename TensorType>
|
||||
void A2Autils<FImpl>::MesonField(TensorType &mat,
|
||||
const FermionField *lhs_wi,
|
||||
const FermionField *rhs_vj,
|
||||
std::vector<Gamma::Algebra> gammas,
|
||||
const std::vector<ComplexField > &mom,
|
||||
int orthogdim, double *t_kernel, double *t_gsum)
|
||||
{
|
||||
typedef typename FImpl::SiteSpinor vobj;
|
||||
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::scalar_type scalar_type;
|
||||
typedef typename vobj::vector_type vector_type;
|
||||
|
||||
typedef iSpinMatrix<vector_type> SpinMatrix_v;
|
||||
typedef iSpinMatrix<scalar_type> SpinMatrix_s;
|
||||
|
||||
int Lblock = mat.dimension(3);
|
||||
int Rblock = mat.dimension(4);
|
||||
|
||||
GridBase *grid = lhs_wi[0].Grid();
|
||||
|
||||
const int Nd = grid->_ndimension;
|
||||
const int Nsimd = grid->Nsimd();
|
||||
|
||||
int Nt = grid->GlobalDimensions()[orthogdim];
|
||||
int Ngamma = gammas.size();
|
||||
int Nmom = mom.size();
|
||||
|
||||
int fd=grid->_fdimensions[orthogdim];
|
||||
int ld=grid->_ldimensions[orthogdim];
|
||||
int rd=grid->_rdimensions[orthogdim];
|
||||
|
||||
// will locally sum vectors first
|
||||
// sum across these down to scalars
|
||||
// splitting the SIMD
|
||||
int MFrvol = rd*Lblock*Rblock*Nmom;
|
||||
int MFlvol = ld*Lblock*Rblock*Nmom;
|
||||
|
||||
std::vector<SpinMatrix_v > lvSum(MFrvol);
|
||||
for(int r=0;r<MFrvol;r++){
|
||||
lvSum[r] = Zero();
|
||||
}
|
||||
|
||||
std::vector<SpinMatrix_s > lsSum(MFlvol);
|
||||
for(int r=0;r<MFlvol;r++){
|
||||
lsSum[r]=scalar_type(0.0);
|
||||
}
|
||||
|
||||
int e1= grid->_slice_nblock[orthogdim];
|
||||
int e2= grid->_slice_block [orthogdim];
|
||||
int stride=grid->_slice_stride[orthogdim];
|
||||
|
||||
// potentially wasting cores here if local time extent too small
|
||||
if (t_kernel) *t_kernel = -usecond();
|
||||
for(int r=0;r<rd;r++) {
|
||||
|
||||
int so=r*grid->_ostride[orthogdim]; // base offset for start of plane
|
||||
|
||||
for(int n=0;n<e1;n++){
|
||||
for(int b=0;b<e2;b++){
|
||||
|
||||
int ss= so+n*stride+b;
|
||||
|
||||
for(int i=0;i<Lblock;i++){
|
||||
|
||||
// Recreate view potentially expensive outside fo UVM mode
|
||||
autoView(lhs_v,lhs_wi[i],CpuRead);
|
||||
auto left = conjugate(lhs_v[ss]);
|
||||
for(int j=0;j<Rblock;j++){
|
||||
|
||||
SpinMatrix_v vv;
|
||||
// Recreate view potentially expensive outside fo UVM mode
|
||||
autoView(rhs_v,rhs_vj[j],CpuRead);
|
||||
auto right = rhs_v[ss];
|
||||
for(int s1=0;s1<Ns;s1++){
|
||||
for(int s2=0;s2<Ns;s2++){
|
||||
vv()(s1,s2)() = left()(s2)(0) * right()(s1)(0)
|
||||
+ left()(s2)(1) * right()(s1)(1)
|
||||
+ left()(s2)(2) * right()(s1)(2);
|
||||
}}
|
||||
|
||||
// After getting the sitewise product do the mom phase loop
|
||||
int base = Nmom*i+Nmom*Lblock*j+Nmom*Lblock*Rblock*r;
|
||||
for ( int m=0;m<Nmom;m++){
|
||||
int idx = m+base;
|
||||
autoView(mom_v,mom[m],CpuRead);
|
||||
auto phase = mom_v[ss];
|
||||
mac(&lvSum[idx],&vv,&phase);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Sum across simd lanes in the plane, breaking out orthog dir.
|
||||
for(int rt=0;rt<rd;rt++){
|
||||
|
||||
Coordinate icoor(Nd);
|
||||
ExtractBuffer<SpinMatrix_s> extracted(Nsimd);
|
||||
|
||||
for(int i=0;i<Lblock;i++){
|
||||
for(int j=0;j<Rblock;j++){
|
||||
for(int m=0;m<Nmom;m++){
|
||||
|
||||
int ij_rdx = m+Nmom*i+Nmom*Lblock*j+Nmom*Lblock*Rblock*rt;
|
||||
|
||||
extract(lvSum[ij_rdx],extracted);
|
||||
|
||||
for(int idx=0;idx<Nsimd;idx++){
|
||||
|
||||
grid->iCoorFromIindex(icoor,idx);
|
||||
|
||||
int ldx = rt+icoor[orthogdim]*rd;
|
||||
|
||||
int ij_ldx = m+Nmom*i+Nmom*Lblock*j+Nmom*Lblock*Rblock*ldx;
|
||||
|
||||
lsSum[ij_ldx]=lsSum[ij_ldx]+extracted[idx];
|
||||
|
||||
}
|
||||
}}}
|
||||
}
|
||||
if (t_kernel) *t_kernel += usecond();
|
||||
GRID_ASSERT(mat.dimension(0) == Nmom);
|
||||
GRID_ASSERT(mat.dimension(1) == Ngamma);
|
||||
GRID_ASSERT(mat.dimension(2) == Nt);
|
||||
|
||||
// ld loop and local only??
|
||||
int pd = grid->_processors[orthogdim];
|
||||
int pc = grid->_processor_coor[orthogdim];
|
||||
thread_for_collapse(2,lt,ld,{
|
||||
for(int pt=0;pt<pd;pt++){
|
||||
int t = lt + pt*ld;
|
||||
if (pt == pc){
|
||||
for(int i=0;i<Lblock;i++){
|
||||
for(int j=0;j<Rblock;j++){
|
||||
for(int m=0;m<Nmom;m++){
|
||||
int ij_dx = m+Nmom*i + Nmom*Lblock * j + Nmom*Lblock * Rblock * lt;
|
||||
for(int mu=0;mu<Ngamma;mu++){
|
||||
// this is a bit slow
|
||||
mat(m,mu,t,i,j) = trace(lsSum[ij_dx]*Gamma(gammas[mu]))()()();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
const scalar_type zz(0.0);
|
||||
for(int i=0;i<Lblock;i++){
|
||||
for(int j=0;j<Rblock;j++){
|
||||
for(int mu=0;mu<Ngamma;mu++){
|
||||
for(int m=0;m<Nmom;m++){
|
||||
mat(m,mu,t,i,j) =zz;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
////////////////////////////////////////////////////////////////////
|
||||
// This global sum is taking as much as 50% of time on 16 nodes
|
||||
// Vector size is 7 x 16 x 32 x 16 x 16 x sizeof(complex) = 2MB - 60MB depending on volume
|
||||
// Healthy size that should suffice
|
||||
////////////////////////////////////////////////////////////////////
|
||||
if (t_gsum) *t_gsum = -usecond();
|
||||
grid->GlobalSumVector(&mat(0,0,0,0,0),Nmom*Ngamma*Nt*Lblock*Rblock);
|
||||
if (t_gsum) *t_gsum += usecond();
|
||||
}
|
||||
|
||||
template <class FImpl>
|
||||
template <typename TensorType>
|
||||
void A2Autils<FImpl>::AslashField(TensorType &mat,
|
||||
const FermionField *lhs_wi,
|
||||
const FermionField *rhs_vj,
|
||||
const std::vector<ComplexField> &emB0,
|
||||
const std::vector<ComplexField> &emB1,
|
||||
int orthogdim, double *t_kernel, double *t_gsum)
|
||||
{
|
||||
typedef typename FermionField::vector_object vobj;
|
||||
typedef typename vobj::scalar_object sobj;
|
||||
typedef typename vobj::scalar_type scalar_type;
|
||||
typedef typename vobj::vector_type vector_type;
|
||||
|
||||
typedef iSpinMatrix<vector_type> SpinMatrix_v;
|
||||
typedef iSpinMatrix<scalar_type> SpinMatrix_s;
|
||||
typedef iSinglet<vector_type> Singlet_v;
|
||||
typedef iSinglet<scalar_type> Singlet_s;
|
||||
|
||||
int Lblock = mat.dimension(3);
|
||||
int Rblock = mat.dimension(4);
|
||||
|
||||
GridBase *grid = lhs_wi[0].Grid();
|
||||
|
||||
const int Nd = grid->_ndimension;
|
||||
const int Nsimd = grid->Nsimd();
|
||||
|
||||
int Nt = grid->GlobalDimensions()[orthogdim];
|
||||
int Nem = emB0.size();
|
||||
GRID_ASSERT(emB1.size() == Nem);
|
||||
|
||||
int fd=grid->_fdimensions[orthogdim];
|
||||
int ld=grid->_ldimensions[orthogdim];
|
||||
int rd=grid->_rdimensions[orthogdim];
|
||||
|
||||
// will locally sum vectors first
|
||||
// sum across these down to scalars
|
||||
// splitting the SIMD
|
||||
int MFrvol = rd*Lblock*Rblock*Nem;
|
||||
int MFlvol = ld*Lblock*Rblock*Nem;
|
||||
|
||||
std::vector<vector_type> lvSum(MFrvol);
|
||||
thread_for(r,MFrvol,
|
||||
{
|
||||
lvSum[r] = Zero();
|
||||
});
|
||||
|
||||
std::vector<scalar_type> lsSum(MFlvol);
|
||||
thread_for(r,MFlvol,
|
||||
{
|
||||
lsSum[r] = scalar_type(0.0);
|
||||
});
|
||||
|
||||
int e1= grid->_slice_nblock[orthogdim];
|
||||
int e2= grid->_slice_block [orthogdim];
|
||||
int stride=grid->_slice_stride[orthogdim];
|
||||
|
||||
// Nested parallelism would be ok
|
||||
// Wasting cores here. Test case r
|
||||
if (t_kernel) *t_kernel = -usecond();
|
||||
for(int r=0;r<rd;r++)
|
||||
{
|
||||
int so=r*grid->_ostride[orthogdim]; // base offset for start of plane
|
||||
|
||||
for(int n=0;n<e1;n++)
|
||||
for(int b=0;b<e2;b++)
|
||||
{
|
||||
int ss= so+n*stride+b;
|
||||
|
||||
for(int i=0;i<Lblock;i++)
|
||||
{
|
||||
autoView(wi_v,lhs_wi[i],CpuRead);
|
||||
auto left = conjugate(wi_v[ss]);
|
||||
|
||||
for(int j=0;j<Rblock;j++)
|
||||
{
|
||||
SpinMatrix_v vv;
|
||||
autoView(vj_v,rhs_vj[j],CpuRead);
|
||||
auto right = vj_v[ss];
|
||||
|
||||
for(int s1=0;s1<Ns;s1++)
|
||||
for(int s2=0;s2<Ns;s2++)
|
||||
{
|
||||
vv()(s1,s2)() = left()(s2)(0) * right()(s1)(0)
|
||||
+ left()(s2)(1) * right()(s1)(1)
|
||||
+ left()(s2)(2) * right()(s1)(2);
|
||||
}
|
||||
|
||||
// After getting the sitewise product do the mom phase loop
|
||||
int base = Nem*i+Nem*Lblock*j+Nem*Lblock*Rblock*r;
|
||||
|
||||
for ( int m=0;m<Nem;m++)
|
||||
{
|
||||
autoView(emB0_v,emB0[m],CpuRead);
|
||||
autoView(emB1_v,emB1[m],CpuRead);
|
||||
int idx = m+base;
|
||||
auto b0 = emB0_v[ss];
|
||||
auto b1 = emB1_v[ss];
|
||||
auto cb0 = conjugate(b0);
|
||||
auto cb1 = conjugate(b1);
|
||||
|
||||
lvSum[idx] += - vv()(3,0)()*b0()()() - vv()(2,0)()*cb1()()()
|
||||
+ vv()(3,1)()*b1()()() - vv()(2,1)()*cb0()()()
|
||||
+ vv()(0,2)()*b1()()() + vv()(1,2)()*b0()()()
|
||||
+ vv()(0,3)()*cb0()()() - vv()(1,3)()*cb1()()();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Sum across simd lanes in the plane, breaking out orthog dir.
|
||||
thread_for(rt,rd,
|
||||
{
|
||||
Coordinate icoor(Nd);
|
||||
ExtractBuffer<scalar_type> extracted(Nsimd);
|
||||
|
||||
for(int i=0;i<Lblock;i++)
|
||||
for(int j=0;j<Rblock;j++)
|
||||
for(int m=0;m<Nem;m++)
|
||||
{
|
||||
|
||||
int ij_rdx = m+Nem*i+Nem*Lblock*j+Nem*Lblock*Rblock*rt;
|
||||
|
||||
extract<vector_type,scalar_type>(lvSum[ij_rdx],extracted);
|
||||
for(int idx=0;idx<Nsimd;idx++)
|
||||
{
|
||||
grid->iCoorFromIindex(icoor,idx);
|
||||
|
||||
int ldx = rt+icoor[orthogdim]*rd;
|
||||
int ij_ldx = m+Nem*i+Nem*Lblock*j+Nem*Lblock*Rblock*ldx;
|
||||
|
||||
lsSum[ij_ldx]=lsSum[ij_ldx]+extracted[idx];
|
||||
}
|
||||
}
|
||||
});
|
||||
if (t_kernel) *t_kernel += usecond();
|
||||
|
||||
// ld loop and local only??
|
||||
int pd = grid->_processors[orthogdim];
|
||||
int pc = grid->_processor_coor[orthogdim];
|
||||
thread_for_collapse(2,lt,ld,
|
||||
{
|
||||
for(int pt=0;pt<pd;pt++)
|
||||
{
|
||||
int t = lt + pt*ld;
|
||||
if (pt == pc)
|
||||
{
|
||||
for(int i=0;i<Lblock;i++)
|
||||
for(int j=0;j<Rblock;j++)
|
||||
for(int m=0;m<Nem;m++)
|
||||
{
|
||||
int ij_dx = m+Nem*i + Nem*Lblock * j + Nem*Lblock * Rblock * lt;
|
||||
|
||||
mat(m,0,t,i,j) = lsSum[ij_dx];
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const scalar_type zz(0.0);
|
||||
|
||||
for(int i=0;i<Lblock;i++)
|
||||
for(int j=0;j<Rblock;j++)
|
||||
for(int m=0;m<Nem;m++)
|
||||
{
|
||||
mat(m,0,t,i,j) = zz;
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
if (t_gsum) *t_gsum = -usecond();
|
||||
grid->GlobalSumVector(&mat(0,0,0,0,0),Nem*Nt*Lblock*Rblock);
|
||||
if (t_gsum) *t_gsum += usecond();
|
||||
}
|
||||
#endif
|
||||
////////////////////////////////////////////
|
||||
// Schematic thoughts about more generalised four quark insertion
|
||||
//
|
||||
|
||||
@@ -119,7 +119,7 @@ static void generatorDiagonal(int diagIndex, iGroupMatrix<cplx> &ta) {
|
||||
// Map a su2 subgroup number to the pair of rows that are non zero
|
||||
////////////////////////////////////////////////////////////////////////
|
||||
static accelerator_inline void su2SubGroupIndex(int &i1, int &i2, int su2_index, GroupName::SU) {
|
||||
GRID_ASSERT((su2_index >= 0) && (su2_index < (ncolour * (ncolour - 1)) / 2));
|
||||
assert((su2_index >= 0) && (su2_index < (ncolour * (ncolour - 1)) / 2));
|
||||
|
||||
int spare = su2_index;
|
||||
for (i1 = 0; spare >= (ncolour - 1 - i1); i1++) {
|
||||
|
||||
@@ -254,9 +254,9 @@ static void testGenerators(GroupName::Sp) {
|
||||
}
|
||||
}
|
||||
|
||||
template <int N>
|
||||
static Lattice<iScalar<iScalar<iMatrix<vComplexD, N> > > >
|
||||
ProjectOnGeneralGroup(const Lattice<iScalar<iScalar<iMatrix<vComplexD, N> > > > &Umu, GroupName::Sp) {
|
||||
template <class vtype, int N>
|
||||
static Lattice<iScalar<iScalar<iMatrix<vtype, N> > > >
|
||||
ProjectOnGeneralGroup(const Lattice<iScalar<iScalar<iMatrix<vtype, N> > > > &Umu, GroupName::Sp) {
|
||||
return ProjectOnSpGroup(Umu);
|
||||
}
|
||||
|
||||
|
||||
@@ -177,25 +177,43 @@ public:
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////
|
||||
// average over all x,y,z the temporal loop
|
||||
// average Polyakov loop in mu direction over all directions != mu
|
||||
//////////////////////////////////////////////////
|
||||
static ComplexD avgPolyakovLoop(const GaugeField &Umu) { //assume Nd=4
|
||||
GaugeMat Ut(Umu.Grid()), P(Umu.Grid());
|
||||
static ComplexD avgPolyakovLoop(const GaugeField &Umu, const int mu) { //assume Nd=4
|
||||
|
||||
// Protect against bad value of mu [0, 3]
|
||||
if ((mu < 0 ) || (mu > 3)) {
|
||||
std::cout << GridLogError << "Index is not an integer inclusively between 0 and 3." << std::endl;
|
||||
exit(1);
|
||||
}
|
||||
|
||||
// U_loop is U_{mu}
|
||||
GaugeMat U_loop(Umu.Grid()), P(Umu.Grid());
|
||||
ComplexD out;
|
||||
int T = Umu.Grid()->GlobalDimensions()[3];
|
||||
int X = Umu.Grid()->GlobalDimensions()[0];
|
||||
int Y = Umu.Grid()->GlobalDimensions()[1];
|
||||
int Z = Umu.Grid()->GlobalDimensions()[2];
|
||||
|
||||
Ut = peekLorentz(Umu,3); //Select temporal direction
|
||||
P = Ut;
|
||||
for (int t=1;t<T;t++){
|
||||
P = Gimpl::CovShiftForward(Ut,3,P);
|
||||
// Number of sites in mu direction
|
||||
int N_mu = Umu.Grid()->GlobalDimensions()[mu];
|
||||
|
||||
U_loop = peekLorentz(Umu, mu); //Select direction
|
||||
P = U_loop;
|
||||
for (int t=1;t<N_mu;t++){
|
||||
P = Gimpl::CovShiftForward(U_loop,mu,P);
|
||||
}
|
||||
RealD norm = 1.0/(Nc*X*Y*Z*T);
|
||||
out = sum(trace(P))*norm;
|
||||
return out;
|
||||
}
|
||||
}
|
||||
|
||||
/////////////////////////////////////////////////
|
||||
// overload for temporal Polyakov loop
|
||||
/////////////////////////////////////////////////
|
||||
static ComplexD avgPolyakovLoop(const GaugeField &Umu) {
|
||||
return avgPolyakovLoop(Umu, 3);
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////
|
||||
// average over traced single links
|
||||
|
||||
@@ -113,6 +113,14 @@ accelerator_inline RealD adj(const RealD & r){ return r; }
|
||||
accelerator_inline ComplexD adj(const ComplexD& r){ return(conjugate(r)); }
|
||||
accelerator_inline ComplexF adj(const ComplexF& r ){ return(conjugate(r)); }
|
||||
|
||||
#if defined(GRID_CUDA) || defined(GRID_HIP)
|
||||
//Provide for convenience
|
||||
inline std::complex<double> conjugate(const std::complex<double>& r){ return(conj(r)); }
|
||||
inline std::complex<float> conjugate(const std::complex<float>& r) { return(conj(r)); }
|
||||
inline std::complex<double> adj(const std::complex<double>& r) { return(conj(r)); }
|
||||
inline std::complex<float> adj(const std::complex<float>& r) { return(conj(r)); }
|
||||
#endif
|
||||
|
||||
accelerator_inline RealF real(const RealF & r){ return r; }
|
||||
accelerator_inline RealD real(const RealD & r){ return r; }
|
||||
accelerator_inline RealF real(const ComplexF & r){ return r.real(); }
|
||||
|
||||
@@ -52,6 +52,10 @@ class GeneralLocalStencilView {
|
||||
return & this->_entries_p[point+this->_npoints*osite];
|
||||
}
|
||||
void ViewClose(void){};
|
||||
#ifdef GRID_LOG_VIEWS
|
||||
size_t size() { return 0; };
|
||||
uint64_t & operator[](size_t i) { static uint64_t v=0; return v; };
|
||||
#endif
|
||||
};
|
||||
////////////////////////////////////////
|
||||
// The Stencil Class itself
|
||||
|
||||
@@ -751,7 +751,7 @@ public:
|
||||
obj.xbytes = xbytes;
|
||||
obj.rbytes = rbytes;
|
||||
obj.cb = cb;
|
||||
|
||||
|
||||
for(int i=0;i<CachedTransfers.size();i++){
|
||||
if ( (CachedTransfers[i].direction ==direction)
|
||||
&&(CachedTransfers[i].OrthogPlane==OrthogPlane)
|
||||
@@ -763,11 +763,13 @@ public:
|
||||
){
|
||||
// FIXME worry about duplicate with partial compression
|
||||
// Wont happen as DWF has no duplicates, but...
|
||||
AddCopy(CachedTransfers[i].recv_buf,recv_buf,rbytes);
|
||||
return 1;
|
||||
// AddCopy(CachedTransfers[i].recv_buf,recv_buf,rbytes);
|
||||
// std::cout << "Duplicate dir " <<direction<<" "<<" OrthogPlane "<<OrthogPlane<<" Dest"<<DestProc <<" xbytes " <<xbytes<<" lane "<< lane<<" cb "<<cb<<std::endl;
|
||||
return 0;
|
||||
|
||||
// return 1;
|
||||
}
|
||||
}
|
||||
|
||||
CachedTransfers.push_back(obj);
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
int world_rank; // Use to control world rank for print guarding
|
||||
int acceleratorAbortOnGpuError=1;
|
||||
uint32_t accelerator_threads=2;
|
||||
uint32_t accelerator_threads=8;
|
||||
uint32_t acceleratorThreads(void) {return accelerator_threads;};
|
||||
void acceleratorThreads(uint32_t t) {accelerator_threads = t;};
|
||||
|
||||
|
||||
+16
-16
@@ -96,7 +96,9 @@ void acceleratorInit(void);
|
||||
|
||||
#ifdef GRID_CUDA
|
||||
|
||||
NAMESPACE_END(Grid);
|
||||
#include <cuda.h>
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
#define GRID_SIMT
|
||||
@@ -432,22 +434,20 @@ accelerator_inline int acceleratorSIMTlane(int Nsimd) {
|
||||
|
||||
#define accelerator_for2dNB( iter1, num1, iter2, num2, nsimd, ... ) \
|
||||
{ \
|
||||
typedef uint64_t Iterator; \
|
||||
auto lambda = [=] accelerator \
|
||||
(Iterator iter1,Iterator iter2,Iterator lane ) mutable { \
|
||||
{ __VA_ARGS__;} \
|
||||
}; \
|
||||
int nt=acceleratorThreads(); \
|
||||
dim3 hip_threads(nsimd, nt, 1); \
|
||||
dim3 hip_blocks ((num1+nt-1)/nt,num2,1); \
|
||||
if(hip_threads.x * hip_threads.y * hip_threads.z <= 64){ \
|
||||
hipLaunchKernelGGL(LambdaApply64,hip_blocks,hip_threads, \
|
||||
0,computeStream, \
|
||||
num1,num2,nsimd, lambda); \
|
||||
} else { \
|
||||
hipLaunchKernelGGL(LambdaApply,hip_blocks,hip_threads, \
|
||||
0,computeStream, \
|
||||
num1,num2,nsimd, lambda); \
|
||||
if (num1*num2) { \
|
||||
typedef uint64_t Iterator; \
|
||||
auto lambda = [=] accelerator \
|
||||
(Iterator iter1,Iterator iter2,Iterator lane ) mutable { \
|
||||
{ __VA_ARGS__;} \
|
||||
}; \
|
||||
int nt=acceleratorThreads(); \
|
||||
dim3 hip_threads(nsimd, nt, 1); \
|
||||
dim3 hip_blocks ((num1+nt-1)/nt,num2,1); \
|
||||
if(hip_threads.x * hip_threads.y * hip_threads.z <= 64){ \
|
||||
LambdaApply64<<<hip_blocks,hip_threads,0,computeStream>>>(num1,num2,nsimd,lambda); \
|
||||
} else { \
|
||||
LambdaApply<<<hip_blocks,hip_threads,0,computeStream>>>(num1,num2,nsimd,lambda); \
|
||||
} \
|
||||
} \
|
||||
}
|
||||
|
||||
|
||||
@@ -31,7 +31,7 @@ namespace Grid{
|
||||
static accelerator_inline void IndexFromCoor (const coor_t& coor,int &index,const coor_t &dims){
|
||||
int64_t index64;
|
||||
IndexFromCoor(coor,index64,dims);
|
||||
GRID_ASSERT(index64<2*1024*1024*1024LL);
|
||||
assert(index64<2*1024*1024*1024LL);
|
||||
index = (int) index64;
|
||||
}
|
||||
|
||||
|
||||
+6
-1
@@ -24,7 +24,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
#if Nc == 3
|
||||
#include <Grid/qcd/smearing/GaugeConfigurationMasked.h>
|
||||
@@ -230,3 +234,4 @@ int main(int argc, char **argv)
|
||||
#endif
|
||||
} // main
|
||||
|
||||
#endif
|
||||
|
||||
@@ -25,7 +25,11 @@ directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
#if Nc == 3
|
||||
#include <Grid/qcd/smearing/GaugeConfigurationMasked.h>
|
||||
@@ -231,5 +235,4 @@ int main(int argc, char **argv)
|
||||
#endif
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
+6
-3
@@ -24,7 +24,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
#if Nc == 3
|
||||
#include <Grid/qcd/smearing/GaugeConfigurationMasked.h>
|
||||
@@ -230,5 +234,4 @@ int main(int argc, char **argv)
|
||||
#endif
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
+6
-3
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
using namespace Grid;
|
||||
@@ -195,5 +199,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -28,7 +28,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
#ifdef GRID_DEFAULT_PRECISION_DOUBLE
|
||||
#define MIXED_PRECISION
|
||||
@@ -449,5 +453,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -28,7 +28,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
#ifdef GRID_DEFAULT_PRECISION_DOUBLE
|
||||
#define MIXED_PRECISION
|
||||
@@ -442,5 +446,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -28,7 +28,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
using namespace Grid;
|
||||
|
||||
@@ -918,3 +922,5 @@ int main(int argc, char **argv) {
|
||||
return 0;
|
||||
#endif
|
||||
} // main
|
||||
|
||||
#endif
|
||||
|
||||
@@ -28,7 +28,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
using namespace Grid;
|
||||
|
||||
@@ -873,3 +877,5 @@ int main(int argc, char **argv) {
|
||||
return 0;
|
||||
#endif
|
||||
} // main
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
using namespace Grid;
|
||||
@@ -193,5 +197,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
@@ -512,5 +516,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
using namespace Grid;
|
||||
@@ -345,5 +349,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
@@ -516,5 +520,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
@@ -567,5 +571,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
using namespace Grid;
|
||||
@@ -263,5 +267,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
using namespace Grid;
|
||||
@@ -417,5 +421,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
@@ -452,5 +456,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
NAMESPACE_BEGIN(Grid);
|
||||
|
||||
@@ -462,5 +466,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -27,7 +27,11 @@ See the full license in the file "LICENSE" in the top level distribution
|
||||
directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include<Grid/Grid.h>
|
||||
|
||||
|
||||
|
||||
@@ -264,5 +268,4 @@ int main(int argc, char **argv) {
|
||||
Grid_finalize();
|
||||
} // main
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
#include <Grid/Grid.h>
|
||||
#pragma once
|
||||
|
||||
|
||||
#ifndef ENABLE_FERMION_INSTANTIATIONS
|
||||
#include <iostream>
|
||||
|
||||
int main(void) {
|
||||
std::cout << "This build of Grid was configured to exclude fermion instantiations, "
|
||||
<< "which this example relies on. "
|
||||
<< "Please reconfigure and rebuild Grid with --enable-fermion-instantiations"
|
||||
<< "to run this example."
|
||||
<< std::endl;
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
@@ -1 +1,4 @@
|
||||
mpicxx -fsycl halo_mpi.cc -o halo_mpi
|
||||
mpicxx -fsycl halo_mpi.cc -o halo_mpi
|
||||
mpicxx -O2 -std=c++11 io_mpi.cc -o io_mpi
|
||||
# Frontier: (hipcc via mpicxx wrapper; ACC_HIP is the default in-file)
|
||||
mpicxx -O2 -x hip gather_mpi.cc -o gather_mpi -L${ROCM_PATH}/lib -lamdhip64
|
||||
|
||||
@@ -0,0 +1,373 @@
|
||||
#include <cassert>
|
||||
#include <complex>
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <string>
|
||||
#include <sstream>
|
||||
#include <cstring>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <strings.h>
|
||||
#include <ctime>
|
||||
#include <sys/time.h>
|
||||
|
||||
#include <mpi.h>
|
||||
|
||||
/**************************************************************
|
||||
* Distributed dense-panel gather benchmark / reproducer.
|
||||
*
|
||||
* Pattern under test: P ranks each OWN a contiguous block of rows
|
||||
* of a (krows x ncols) fp64-complex panel; every rank must end
|
||||
* holding the WHOLE panel. This is the setup-time collective of a
|
||||
* distributed recursive Schur inversion (row-distributed dense
|
||||
* matrices, rank-major ordering => owned rows contiguous).
|
||||
*
|
||||
* Three implementations of identical semantics:
|
||||
* A zero-fill + MPI_Allreduce(SUM) (simple; non-owners send zeros)
|
||||
* B MPI_Allgatherv (owners-only send)
|
||||
* C ring allgather via MPI_Sendrecv (halo-exchange primitive)
|
||||
*
|
||||
* Measured motivation (Frontier, 288 ranks, ~340MB panels): pattern A
|
||||
* on host buffers delivered 0.48 GB/s effective payload -- ~8% of its
|
||||
* own per-node wire floor, ~13x below the efficiency the Sendrecv
|
||||
* halo exchange achieves on the same NICs (185 GB/s bidirectional,
|
||||
* see halo_mpi.cc).
|
||||
*
|
||||
* Config: what is the target
|
||||
**************************************************************
|
||||
*/
|
||||
#undef ACC_CUDA
|
||||
#define ACC_HIP
|
||||
#undef ACC_SYCL
|
||||
#undef ACC_NONE
|
||||
|
||||
/**************************************************************
|
||||
* Some MPI globals
|
||||
**************************************************************
|
||||
*/
|
||||
MPI_Comm WorldComm;
|
||||
MPI_Comm WorldShmComm;
|
||||
|
||||
int WorldSize;
|
||||
int WorldRank;
|
||||
|
||||
int WorldShmSize;
|
||||
int WorldShmRank;
|
||||
|
||||
/**************************************************************
|
||||
* Allocate buffers on the GPU, SYCL needs an init call and context
|
||||
**************************************************************
|
||||
*/
|
||||
#ifdef ACC_CUDA
|
||||
#include <cuda.h>
|
||||
void acceleratorInit(void){}
|
||||
void *acceleratorAllocDevice(size_t bytes)
|
||||
{
|
||||
void *ptr=NULL;
|
||||
auto err = cudaMalloc((void **)&ptr,bytes);
|
||||
assert(err==cudaSuccess);
|
||||
return ptr;
|
||||
}
|
||||
void acceleratorFreeDevice(void *ptr){ cudaFree(ptr);}
|
||||
void acceleratorMemSet(void *ptr,int val,size_t bytes){ cudaMemset(ptr,val,bytes);}
|
||||
void acceleratorCopyToDevice(const void *from,void *to,size_t bytes){ cudaMemcpy(to,from,bytes,cudaMemcpyHostToDevice);}
|
||||
#endif
|
||||
#ifdef ACC_HIP
|
||||
#include <hip/hip_runtime.h>
|
||||
void acceleratorInit(void){}
|
||||
inline void *acceleratorAllocDevice(size_t bytes)
|
||||
{
|
||||
void *ptr=NULL;
|
||||
auto err = hipMalloc((void **)&ptr,bytes);
|
||||
if( err != hipSuccess ) {
|
||||
ptr = (void *) NULL;
|
||||
printf(" hipMalloc failed for %ld %s \n",bytes,hipGetErrorString(err));
|
||||
}
|
||||
return ptr;
|
||||
};
|
||||
inline void acceleratorFreeDevice(void *ptr){ auto r=hipFree(ptr);};
|
||||
inline void acceleratorMemSet(void *ptr,int val,size_t bytes){ auto r=hipMemset(ptr,val,bytes);};
|
||||
inline void acceleratorCopyToDevice(const void *from,void *to,size_t bytes){ auto r=hipMemcpy(to,from,bytes,hipMemcpyHostToDevice);};
|
||||
#endif
|
||||
#ifdef ACC_SYCL
|
||||
#include <sycl/CL/sycl.hpp>
|
||||
#include <sycl/usm.hpp>
|
||||
cl::sycl::queue *theAccelerator;
|
||||
void acceleratorInit(void)
|
||||
{
|
||||
cl::sycl::gpu_selector selector;
|
||||
cl::sycl::device selectedDevice { selector };
|
||||
theAccelerator = new sycl::queue (selectedDevice);
|
||||
auto name = theAccelerator->get_device().get_info<sycl::info::device::name>();
|
||||
printf("AcceleratorSyclInit: Selected device is %s\n",name.c_str()); fflush(stdout);
|
||||
}
|
||||
inline void *acceleratorAllocDevice(size_t bytes){ return malloc_device(bytes,*theAccelerator);};
|
||||
inline void acceleratorFreeDevice(void *ptr){free(ptr,*theAccelerator);};
|
||||
inline void acceleratorMemSet(void *ptr,int val,size_t bytes){ theAccelerator->memset(ptr,val,bytes).wait();};
|
||||
inline void acceleratorCopyToDevice(const void *from,void *to,size_t bytes){ theAccelerator->memcpy(to,from,bytes).wait();};
|
||||
#endif
|
||||
#ifdef ACC_NONE
|
||||
void acceleratorInit(void){}
|
||||
inline void *acceleratorAllocDevice(size_t bytes){ return malloc(bytes);};
|
||||
inline void acceleratorFreeDevice(void *ptr){free(ptr);};
|
||||
inline void acceleratorMemSet(void *ptr,int val,size_t bytes){ memset(ptr,val,bytes);};
|
||||
inline void acceleratorCopyToDevice(const void *from,void *to,size_t bytes){ memcpy(to,from,bytes);};
|
||||
#endif
|
||||
|
||||
/**************************************************************
|
||||
* Microsecond timer
|
||||
**************************************************************
|
||||
*/
|
||||
inline double usecond(void) {
|
||||
struct timeval tv;
|
||||
gettimeofday(&tv,NULL);
|
||||
return 1.0e6*tv.tv_sec + 1.0*tv.tv_usec;
|
||||
}
|
||||
|
||||
/**************************************************************
|
||||
* Main benchmark routine.
|
||||
*
|
||||
* panel = krows x ncols complex<double>, column major, krows = P*myrows.
|
||||
* Rank r owns rows [r*myrows, (r+1)*myrows): with column-major layout
|
||||
* the owned data is strided; for B and C the owners' contribution is
|
||||
* packed contiguously (rank-major block layout), which is exactly how
|
||||
* the production code would call it. Effective payload = full panel
|
||||
* bytes; every rank must hold it all at the end.
|
||||
**************************************************************
|
||||
*/
|
||||
void Benchmark(size_t panel_bytes,bool use_device,int ncall)
|
||||
{
|
||||
size_t words = panel_bytes/sizeof(double); // treat as flat doubles
|
||||
size_t mywords = words/WorldSize;
|
||||
words = mywords*WorldSize; // exact division
|
||||
size_t bytes = words*sizeof(double);
|
||||
size_t mybytes= mywords*sizeof(double);
|
||||
|
||||
double *panel;
|
||||
double *contrib;
|
||||
if ( use_device ) {
|
||||
panel = (double *)acceleratorAllocDevice(bytes);
|
||||
contrib = (double *)acceleratorAllocDevice(mybytes);
|
||||
if ( panel==NULL || contrib==NULL ) { printf("alloc failed\n"); return; }
|
||||
} else {
|
||||
panel = (double *)malloc(bytes);
|
||||
contrib = (double *)malloc(mybytes);
|
||||
}
|
||||
// Owners deposit non-trivial data once (content is irrelevant to timing)
|
||||
std::vector<double> init(mywords,1.0*WorldRank);
|
||||
acceleratorCopyToDevice(&init[0],contrib,mybytes);
|
||||
if ( !use_device ) memcpy(contrib,&init[0],mybytes);
|
||||
|
||||
double tA,tB,tC;
|
||||
|
||||
/*********************************************************
|
||||
* A: zero-fill + Allreduce(SUM) -- the simple idiom
|
||||
*********************************************************/
|
||||
{
|
||||
MPI_Barrier(WorldComm);
|
||||
double t0=usecond();
|
||||
for(int n=0;n<ncall;n++){
|
||||
if ( use_device ) acceleratorMemSet(panel,0,bytes);
|
||||
else memset(panel,0,bytes);
|
||||
// owner deposits its block at its rank-major offset
|
||||
if ( use_device ) {
|
||||
// device-to-device deposit stands in for the real kernel
|
||||
#ifdef ACC_HIP
|
||||
auto r=hipMemcpy(panel+WorldRank*mywords,contrib,mybytes,hipMemcpyDeviceToDevice);
|
||||
#else
|
||||
acceleratorCopyToDevice(contrib,panel+WorldRank*mywords,mybytes);
|
||||
#endif
|
||||
} else {
|
||||
memcpy(panel+WorldRank*mywords,contrib,mybytes);
|
||||
}
|
||||
int ierr=MPI_Allreduce(MPI_IN_PLACE,panel,words,MPI_DOUBLE,MPI_SUM,WorldComm);
|
||||
assert(ierr==0);
|
||||
}
|
||||
MPI_Barrier(WorldComm);
|
||||
tA=(usecond()-t0)/ncall;
|
||||
}
|
||||
|
||||
/*********************************************************
|
||||
* A2: as A but with a FRESH buffer every call -- the
|
||||
* registration-cache killer (unpinned pages each call).
|
||||
* Host-memory case only; isolates the production bug.
|
||||
*********************************************************/
|
||||
double tA2 = 0.0;
|
||||
if ( !use_device ) {
|
||||
MPI_Barrier(WorldComm);
|
||||
double t0=usecond();
|
||||
for(int n=0;n<ncall;n++){
|
||||
double *fresh = (double *)malloc(bytes);
|
||||
memset(fresh,0,bytes);
|
||||
memcpy(fresh+WorldRank*mywords,contrib,mybytes);
|
||||
int ierr=MPI_Allreduce(MPI_IN_PLACE,fresh,words,MPI_DOUBLE,MPI_SUM,WorldComm);
|
||||
assert(ierr==0);
|
||||
free(fresh);
|
||||
}
|
||||
MPI_Barrier(WorldComm);
|
||||
tA2=(usecond()-t0)/ncall;
|
||||
}
|
||||
|
||||
/*********************************************************
|
||||
* D: chunked Allreduce on the persistent buffer -- hand
|
||||
* decomposition of the primitive (the split-K analogue):
|
||||
* does MPI fail to pipeline internally?
|
||||
*********************************************************/
|
||||
double tD;
|
||||
{
|
||||
size_t chunkwords = (32ULL<<20)/sizeof(double); // 32MB chunks
|
||||
MPI_Barrier(WorldComm);
|
||||
double t0=usecond();
|
||||
for(int n=0;n<ncall;n++){
|
||||
if ( use_device ) acceleratorMemSet(panel,0,bytes);
|
||||
else memset(panel,0,bytes);
|
||||
if ( use_device ) {
|
||||
#ifdef ACC_HIP
|
||||
auto r=hipMemcpy(panel+WorldRank*mywords,contrib,mybytes,hipMemcpyDeviceToDevice);
|
||||
#else
|
||||
acceleratorCopyToDevice(contrib,panel+WorldRank*mywords,mybytes);
|
||||
#endif
|
||||
} else {
|
||||
memcpy(panel+WorldRank*mywords,contrib,mybytes);
|
||||
}
|
||||
for(size_t off=0; off<words; off+=chunkwords){
|
||||
size_t cw = std::min(chunkwords, words-off);
|
||||
int ierr=MPI_Allreduce(MPI_IN_PLACE,panel+off,(int)cw,MPI_DOUBLE,MPI_SUM,WorldComm);
|
||||
assert(ierr==0);
|
||||
}
|
||||
}
|
||||
MPI_Barrier(WorldComm);
|
||||
tD=(usecond()-t0)/ncall;
|
||||
}
|
||||
|
||||
/*********************************************************
|
||||
* B: Allgatherv -- owners-only send, identical result
|
||||
*********************************************************/
|
||||
{
|
||||
std::vector<int> counts(WorldSize,(int)mywords);
|
||||
std::vector<int> displs(WorldSize);
|
||||
for(int r=0;r<WorldSize;r++) displs[r]=r*(int)mywords;
|
||||
|
||||
MPI_Barrier(WorldComm);
|
||||
double t0=usecond();
|
||||
for(int n=0;n<ncall;n++){
|
||||
int ierr=MPI_Allgatherv(contrib,(int)mywords,MPI_DOUBLE,
|
||||
panel,&counts[0],&displs[0],MPI_DOUBLE,WorldComm);
|
||||
assert(ierr==0);
|
||||
}
|
||||
MPI_Barrier(WorldComm);
|
||||
tB=(usecond()-t0)/ncall;
|
||||
}
|
||||
|
||||
/*********************************************************
|
||||
* C: ring allgather from Sendrecv -- the halo primitive.
|
||||
* P-1 steps; step s passes the block received at step s-1 to
|
||||
* the right neighbour. Per-step message = panel/P.
|
||||
*********************************************************/
|
||||
{
|
||||
int right=(WorldRank+1)%WorldSize;
|
||||
int left =(WorldRank-1+WorldSize)%WorldSize;
|
||||
|
||||
MPI_Barrier(WorldComm);
|
||||
double t0=usecond();
|
||||
for(int n=0;n<ncall;n++){
|
||||
// start with my own block resident at my slot
|
||||
if ( use_device ) {
|
||||
#ifdef ACC_HIP
|
||||
auto r=hipMemcpy(panel+WorldRank*mywords,contrib,mybytes,hipMemcpyDeviceToDevice);
|
||||
#else
|
||||
acceleratorCopyToDevice(contrib,panel+WorldRank*mywords,mybytes);
|
||||
#endif
|
||||
} else {
|
||||
memcpy(panel+WorldRank*mywords,contrib,mybytes);
|
||||
}
|
||||
for(int s=0;s<WorldSize-1;s++){
|
||||
int sendblock=(WorldRank-s+WorldSize)%WorldSize;
|
||||
int recvblock=(WorldRank-s-1+WorldSize)%WorldSize;
|
||||
int ierr=MPI_Sendrecv(panel+sendblock*mywords,(int)mywords,MPI_DOUBLE,right,s,
|
||||
panel+recvblock*mywords,(int)mywords,MPI_DOUBLE,left, s,
|
||||
WorldComm,MPI_STATUS_IGNORE);
|
||||
assert(ierr==0);
|
||||
}
|
||||
}
|
||||
MPI_Barrier(WorldComm);
|
||||
tC=(usecond()-t0)/ncall;
|
||||
}
|
||||
|
||||
if ( !WorldRank ) {
|
||||
double GB=bytes/1024./1024./1024.;
|
||||
printf("\t%10.3f GB\t A allreduce %8.2f\t A2 fresh-buf %8.2f\t D chunked %8.2f\t B allgatherv %8.2f\t C ring %8.2f GB/s\n",
|
||||
GB,
|
||||
GB/(tA/1.0e6),
|
||||
(tA2>0.0) ? GB/(tA2/1.0e6) : 0.0,
|
||||
GB/(tD/1.0e6),
|
||||
GB/(tB/1.0e6),
|
||||
GB/(tC/1.0e6));
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
if ( use_device ) {
|
||||
acceleratorFreeDevice(panel);
|
||||
acceleratorFreeDevice(contrib);
|
||||
} else {
|
||||
free(panel);
|
||||
free(contrib);
|
||||
}
|
||||
}
|
||||
|
||||
/**************************************
|
||||
* Command line junk
|
||||
**************************************/
|
||||
int main(int argc, char **argv)
|
||||
{
|
||||
acceleratorInit();
|
||||
|
||||
MPI_Init(&argc,&argv);
|
||||
|
||||
WorldComm = MPI_COMM_WORLD;
|
||||
|
||||
MPI_Comm_split_type(WorldComm, MPI_COMM_TYPE_SHARED, 0, MPI_INFO_NULL,&WorldShmComm);
|
||||
|
||||
MPI_Comm_rank(WorldComm ,&WorldRank);
|
||||
MPI_Comm_size(WorldComm ,&WorldSize);
|
||||
|
||||
MPI_Comm_rank(WorldShmComm ,&WorldShmRank);
|
||||
MPI_Comm_size(WorldShmComm ,&WorldShmSize);
|
||||
|
||||
if( !WorldRank ) {
|
||||
printf("***********************************\n");
|
||||
printf("%d ranks\n",WorldSize);
|
||||
printf("%d ranks-per-node\n",WorldShmSize);
|
||||
printf("%d nodes\n",WorldSize/WorldShmSize);fflush(stdout);
|
||||
printf("***********************************\n");
|
||||
printf("Panel gather: every rank owns 1/%d of the rows;\n",WorldSize);
|
||||
printf("every rank must finish holding the whole panel.\n");
|
||||
printf("Effective payload rate = panel bytes / wall. Same semantics x3:\n");
|
||||
printf(" A zero-fill+Allreduce B Allgatherv C ring Sendrecv\n");
|
||||
}
|
||||
|
||||
std::vector<size_t> sizes({ (size_t)8<<20, (size_t)64<<20, (size_t)256<<20, (size_t)1<<30 });
|
||||
|
||||
if( !WorldRank ) {
|
||||
printf("=========================================================\n");
|
||||
printf("= HOST memory \n");
|
||||
printf("=========================================================\n");fflush(stdout);
|
||||
}
|
||||
for( auto sz : sizes ) Benchmark(sz,false,5);
|
||||
|
||||
if( !WorldRank ) {
|
||||
printf("=========================================================\n");
|
||||
printf("= DEVICE memory \n");
|
||||
printf("=========================================================\n");fflush(stdout);
|
||||
}
|
||||
for( auto sz : sizes ) Benchmark(sz,true,5);
|
||||
|
||||
if( !WorldRank ) {
|
||||
printf("=========================================================\n");
|
||||
printf("= DONE \n");
|
||||
printf("=========================================================\n");
|
||||
}
|
||||
MPI_Finalize();
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
#!/bin/bash
|
||||
|
||||
# Standalone MPI-only I/O reproducer on Aurora. Host only -- no SYCL, no
|
||||
# gpu_tile_compact.sh -- so unlike halo_mpi this needs nothing but MPI:
|
||||
#
|
||||
# mpicxx -O2 -std=c++11 io_mpi.cc -o io_mpi
|
||||
#
|
||||
# 12 ranks per node, one per tile, which is what the machine is. That is a
|
||||
# deliberate difference from io_frontier.slurm (8 per node, one per GCD), so
|
||||
# read the comparison carefully -- see WHAT IS AND IS NOT COMPARABLE below.
|
||||
|
||||
# Run io_aurora_debug.pbs first. If the cross validation fails there, this
|
||||
# scan is 2 hours of 128 nodes producing numbers for a broken file.
|
||||
|
||||
#PBS -q prod
|
||||
#PBS -l filesystems=flare
|
||||
#PBS -l filesystems=home
|
||||
#PBS -l select=128
|
||||
#PBS -l walltime=02:00:00
|
||||
#PBS -A 15479
|
||||
|
||||
cd $PBS_O_WORKDIR
|
||||
cp $PBS_NODEFILE nodefile
|
||||
|
||||
# Only if mpiexec is not already in the environment. io_mpi is host only
|
||||
# and needs no part of the Grid build environment.
|
||||
#source ../../sourceme.sh
|
||||
|
||||
##########################################################################
|
||||
# Environment. io_mpi has no OpenMP and never touches a GPU, so one thread
|
||||
# per rank and a NUMA NIC policy rather than a GPU one.
|
||||
#
|
||||
# The MPICH_DBG_* variables are deliberately absent: at 1536 ranks they
|
||||
# produce gigabytes of log and perturb the timings they would explain.
|
||||
# MPICH_MPIIO_STATS/TIMERS are also off here -- they are per collective and
|
||||
# 1536 ranks x 6 rungs x 3 reps is unreadable. Get them from the debug run.
|
||||
##########################################################################
|
||||
export OMP_NUM_THREADS=1
|
||||
export MPICH_CH4_SHM=XPMEM
|
||||
export MPICH_OFI_NIC_POLICY=NUMA
|
||||
|
||||
##########################################################################
|
||||
# WHAT IS AND IS NOT COMPARABLE WITH THE FRONTIER SCAN
|
||||
#
|
||||
# Held identical at every rung of both scans:
|
||||
# local volume 8.8.32.128 = 262144 sites x 576 B = 151 MB/rank
|
||||
# file view 32768 contiguous runs of 4608 B per rank
|
||||
# aggregation k=2, row of 16, 8 extents of 18 MB
|
||||
# (verified: 4.4.3.1 at 48 ranks and 4.4.3.2 at 96 ranks give exactly the
|
||||
# same plan as Frontier's 4.4.2.1 at 32 ranks.)
|
||||
#
|
||||
# NOT identical, because 12 ranks/node is 1.5x the clients per node:
|
||||
# record size at a given NODE count is 1.5x Frontier's
|
||||
# client count at a given NODE count is 1.5x Frontier's
|
||||
#
|
||||
# So compare the two machines at equal RANK count (Aurora 4 nodes vs
|
||||
# Frontier 6, and so on) if what you want is equal client count and equal
|
||||
# record size; compare at equal NODE count if what you want is each machine
|
||||
# used as it is meant to be used. Both are legitimate, they answer
|
||||
# different questions, and a table that does not say which one it is
|
||||
# reporting is worthless. The quantity that carries the MPI-IO pathology --
|
||||
# per rank local volume and the resulting file view -- is invariant either
|
||||
# way, which is the point.
|
||||
##########################################################################
|
||||
|
||||
##########################################################################
|
||||
# WHICH FILESYSTEM. Point this at Lustre for the like-for-like comparison
|
||||
# with Frontier's Orion. DAOS is a different architecture -- its numbers
|
||||
# are interesting but they are NOT a reproduction of the Frontier result,
|
||||
# and mixing them into one table would misrepresent both. Label every set
|
||||
# of numbers with the filesystem it came from.
|
||||
##########################################################################
|
||||
# PROJECT is the flare project DIRECTORY name, which is not the -A account
|
||||
# number. Set it once; the mkdir below fails loudly rather than writing
|
||||
# somewhere unintended.
|
||||
PROJECT=LatticeQCD_aesp_CNDA
|
||||
WORK=/lus/flare/projects/$PROJECT/$USER/iompi.$PBS_JOBID
|
||||
mkdir -p $WORK || { echo "cannot create $WORK -- set PROJECT correctly"; exit 1; }
|
||||
cd $WORK
|
||||
|
||||
# Match Frontier's default: no explicit striping. Record what was inherited.
|
||||
lfs getstripe -d $WORK 2>/dev/null || echo "(no lfs getstripe -- not Lustre?)"
|
||||
|
||||
# The largest rung writes three files of 232 GB, so budget ~700 GB and check
|
||||
# the quota before submitting. Each run unlinks the three files first, so
|
||||
# that is peak usage, not cumulative.
|
||||
|
||||
NRANKS=12 # one per tile
|
||||
|
||||
BIN=$PBS_O_WORKDIR/io_mpi
|
||||
ARGS="--target 4194304 --reps 3"
|
||||
|
||||
run () { # run <nodes> <grid> <mpi> <comment> [extra args...]
|
||||
local nodes=$1 gr=$2 mp=$3 note=$4
|
||||
local ntot=$(( nodes * NRANKS ))
|
||||
shift 4
|
||||
echo
|
||||
echo "==================================================================="
|
||||
echo "=== nodes=$nodes ranks=$ntot grid=$gr mpi=$mp $note"
|
||||
echo "=== extra: $@"
|
||||
echo "==================================================================="
|
||||
mpiexec -np $ntot -ppn $NRANKS -envall $BIN --grid $gr --mpi $mp $ARGS "$@"
|
||||
echo "=== exit $?"
|
||||
}
|
||||
|
||||
#####################################################################
|
||||
# Phase 0. Correctness. Both count branches of MPI_Alltoallv are
|
||||
# covered; the labels below were checked, not assumed. The whole-file
|
||||
# crc32 is serial, so keep these small.
|
||||
#####################################################################
|
||||
run 1 16.16.16.24 2.2.1.3 "correctness, UNIFORM counts, row of 4" --reps 0 --serial-crc
|
||||
run 2 24.12.8.8 3.2.2.2 "correctness, NON-UNIFORM counts, row of 12" --reps 0 --serial-crc
|
||||
run 4 16.16.32.24 2.2.4.3 "correctness, NON-UNIFORM, non-zero offset" --reps 0 --serial-crc --offset 1024
|
||||
|
||||
#####################################################################
|
||||
# Phase 1. Weak scan at 151 MB/rank. Identical plan at every rung:
|
||||
# k=2, row of 16, 8 extents of 18 MB, 32768 runs of 4608 B in the view.
|
||||
#####################################################################
|
||||
# nodes global lattice mpi record
|
||||
run 4 32.32.96.128 4.4.3.1 "7.2 GB" --no-validate
|
||||
run 8 32.32.96.256 4.4.3.2 "14.5 GB" --no-validate
|
||||
run 16 32.32.96.512 4.4.3.4 "29.0 GB" --no-validate
|
||||
run 32 32.32.192.512 4.4.6.4 "58.0 GB" --no-validate
|
||||
run 64 32.32.192.1024 4.4.6.8 "116.0 GB" --no-validate
|
||||
run 128 32.32.384.1024 4.4.12.8 "231.9 GB" --no-validate
|
||||
|
||||
#####################################################################
|
||||
# Phase 2. The three questions a reviewer asks immediately.
|
||||
#####################################################################
|
||||
run 128 32.32.384.1024 4.4.12.8 "231.9 GB, durable" --no-validate --fsync --drop-cache
|
||||
run 128 32.32.384.1024 4.4.12.8 "231.9 GB, cb hints" --no-validate \
|
||||
--hints romio_cb_write=enable,romio_cb_read=enable,cb_nodes=128,cb_buffer_size=16777216
|
||||
run 128 32.32.384.1024 4.4.12.8 "231.9 GB, mem subarray" --no-validate --mem-subarray
|
||||
|
||||
echo
|
||||
echo "=== done. Files left in $WORK"
|
||||
ls -l $WORK
|
||||
@@ -0,0 +1,109 @@
|
||||
#!/bin/bash -l
|
||||
|
||||
# Standalone MPI-only I/O reproducer on Frontier. No Grid, no accelerator,
|
||||
# so no GCD/NUMA wrapper is needed -- the point of the exercise is that this
|
||||
# depends on nothing but an MPI installation and a filesystem.
|
||||
#
|
||||
# mpicxx -O2 -std=c++11 io_mpi.cc -o io_mpi
|
||||
#
|
||||
# Weak scan: the local volume, and therefore the file view structure, is held
|
||||
# identical at every rung and only the number of Lustre clients changes:
|
||||
#
|
||||
# local volume 8.8.32.128 = 262144 sites x 576 B = 151 MB/rank
|
||||
# file view 32768 contiguous runs of 4608 B per rank, at every rung
|
||||
# aggregate k=2, row of 16, 8 extents of 18 MB, at every rung
|
||||
#
|
||||
# so any change in the relative bandwidth of the two lexicographic paths is a
|
||||
# property of the client count alone.
|
||||
#
|
||||
# The PERF lines are MiB/s (bytes/1024/1024/s), which is what BinaryIO.h
|
||||
# computes for lastPerf.mbytesPerSecond and prints as "MB/s", so the two
|
||||
# tools can be compared directly. Grid's timed region is used here too:
|
||||
# barrier, start, [plan build + exchange + I/O], barrier, stop, quoting the
|
||||
# boss rank's stopwatch. --reuse-plan hoists the plan build out, which is
|
||||
# how to show it is not where the time goes; do not use it when comparing
|
||||
# against Grid's own numbers.
|
||||
#
|
||||
# io_aurora.pbs runs 12 ranks per node, one per tile, because that is what
|
||||
# that machine is. The per rank local volume and the file view are the same
|
||||
# there as here, but the record size and client count at a given NODE count
|
||||
# are 1.5x. See the header of that script before tabulating the two
|
||||
# together.
|
||||
|
||||
#SBATCH --job-name=ioMPI
|
||||
#SBATCH --nodes=128
|
||||
#SBATCH --ntasks-per-node=8
|
||||
#SBATCH --cpus-per-task=7
|
||||
#SBATCH --time=02:00:00
|
||||
#SBATCH --account=phy157_dwf
|
||||
#SBATCH --exclusive
|
||||
#SBATCH --mem=0
|
||||
|
||||
module load cce/21.0.0
|
||||
module load cpe/26.03
|
||||
|
||||
WORK=/lustre/orion/phy157/proj-shared/phy157_dwf/paboyle/iompi.$SLURM_JOB_ID
|
||||
mkdir -p $WORK
|
||||
cd $WORK
|
||||
|
||||
# Do NOT stripe by default. Wide striping is what rescues the collective at
|
||||
# scale and costs every other path a factor of 1.2-2.2; the default layout is
|
||||
# what a user gets without knowing to ask. Uncomment to reproduce that
|
||||
# interaction, and record which one you ran.
|
||||
#lfs setstripe -c -1 -S 8M $WORK
|
||||
lfs getstripe -d $WORK
|
||||
|
||||
BIN=$SLURM_SUBMIT_DIR/io_mpi
|
||||
ARGS="--target 4194304 --reps 3"
|
||||
|
||||
# ROMIO's own view of what it did. Verbose, but the first thing anyone
|
||||
# reading the report will ask for.
|
||||
# export MPICH_MPIIO_STATS=1
|
||||
# export MPICH_MPIIO_TIMERS=1
|
||||
|
||||
run () { # run <nodes> <grid> <mpi> <comment> [extra args...]
|
||||
local nodes=$1 gr=$2 mp=$3 note=$4
|
||||
local nranks=$(( nodes * 8 ))
|
||||
shift 4
|
||||
echo
|
||||
echo "==================================================================="
|
||||
echo "=== nodes=$nodes ranks=$nranks grid=$gr mpi=$mp $note"
|
||||
echo "=== extra: $@"
|
||||
echo "==================================================================="
|
||||
srun -N$nodes -n$nranks --ntasks-per-node=8 $BIN --grid $gr --mpi $mp $ARGS "$@"
|
||||
echo "=== exit $?"
|
||||
}
|
||||
|
||||
#####################################################################
|
||||
# Phase 0. Correctness, including the non-uniform Alltoallv branch
|
||||
# (odd process factor in an un-split dimension). Small, and the
|
||||
# whole-file crc32 is serial, so keep the volume down here.
|
||||
#####################################################################
|
||||
run 1 12.12.8.8 2.2.2.1 "correctness, uniform counts" --reps 0 --serial-crc
|
||||
run 3 24.12.8.8 3.2.2.2 "correctness, NON-uniform counts" --reps 0 --serial-crc
|
||||
run 4 16.16.16.32 2.2.2.4 "correctness, non-zero offset" --reps 0 --serial-crc --offset 1024
|
||||
|
||||
#####################################################################
|
||||
# Phase 1. Weak scan, 151 MB/rank. Timing only.
|
||||
#####################################################################
|
||||
run 4 32.32.64.128 4.4.2.1 "4.8 GB" --no-validate
|
||||
run 8 32.32.64.256 4.4.2.2 "9.7 GB" --no-validate
|
||||
run 16 32.32.64.512 4.4.2.4 "19.3 GB" --no-validate
|
||||
run 32 32.32.128.512 4.4.4.4 "38.6 GB" --no-validate
|
||||
run 64 32.32.128.1024 4.4.4.8 "77.3 GB" --no-validate
|
||||
run 128 32.32.256.1024 4.4.8.8 "154.6 GB" --no-validate
|
||||
|
||||
#####################################################################
|
||||
# Phase 2. Answer the two questions a reviewer will ask immediately.
|
||||
#####################################################################
|
||||
# Is the gap an artefact of measuring cache rather than the filesystem?
|
||||
run 128 32.32.256.1024 4.4.8.8 "154.6 GB, durable" --no-validate --fsync --drop-cache
|
||||
# Does the collective recover if it is given the hints it wants?
|
||||
run 128 32.32.256.1024 4.4.8.8 "154.6 GB, cb hints" --no-validate \
|
||||
--hints romio_cb_write=enable,romio_cb_read=enable,cb_nodes=128,cb_buffer_size=16777216
|
||||
# Does the degenerate memory subarray matter?
|
||||
run 128 32.32.256.1024 4.4.8.8 "154.6 GB, mem subarray" --no-validate --mem-subarray
|
||||
|
||||
echo
|
||||
echo "=== done. Files left in $WORK"
|
||||
ls -l $WORK
|
||||
File diff suppressed because it is too large
Load Diff
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace Grid;
|
||||
@@ -731,3 +734,5 @@ int main (int argc, char ** argv)
|
||||
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -20,6 +20,9 @@
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#ifdef GRID_CUDA
|
||||
#define CUDA_PROFILE
|
||||
@@ -439,3 +442,4 @@ void Benchmark(int Ls, Coordinate Dirichlet,bool sloppy)
|
||||
GRID_ASSERT(norm2(src_e)<1.0e-4);
|
||||
GRID_ASSERT(norm2(src_o)<1.0e-4);
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -20,6 +20,10 @@
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#ifdef GRID_CUDA
|
||||
#define CUDA_PROFILE
|
||||
@@ -439,3 +443,5 @@ void Benchmark(int Ls, Coordinate Dirichlet,bool sloppy)
|
||||
GRID_ASSERT(norm2(src_e)<1.0e-4);
|
||||
GRID_ASSERT(norm2(src_o)<1.0e-4);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -20,6 +20,9 @@
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#ifdef GRID_CUDA
|
||||
#define CUDA_PROFILE
|
||||
@@ -385,3 +388,5 @@ int main (int argc, char ** argv)
|
||||
Grid_finalize();
|
||||
exit(0);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -238,5 +241,4 @@ void benchDw(std::vector<int> & latt4, int Ls, int threads,int report )
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,3 +1,7 @@
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#include <sstream>
|
||||
using namespace std;
|
||||
@@ -155,3 +159,4 @@ int main (int argc, char ** argv)
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -20,6 +20,9 @@
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#ifdef GRID_CUDA
|
||||
#define CUDA_PROFILE
|
||||
@@ -129,3 +132,5 @@ int main (int argc, char ** argv)
|
||||
Grid_finalize();
|
||||
exit(0);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -149,3 +152,5 @@ int main (int argc, char ** argv)
|
||||
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -172,5 +175,4 @@ void benchDw(std::vector<int> & latt4, int Ls)
|
||||
// Dw.Report();
|
||||
}
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -110,3 +113,5 @@ int main (int argc, char ** argv)
|
||||
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -112,3 +115,5 @@ int main (int argc, char ** argv)
|
||||
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,10 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
#include <Grid/algorithms/blas/BatchedBlas.h>
|
||||
|
||||
@@ -978,3 +982,5 @@ int main (int argc, char ** argv)
|
||||
Grid_finalize();
|
||||
fclose(FP);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -26,6 +26,9 @@ Author: paboyle <paboyle@ph.ed.ac.uk>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -258,3 +261,5 @@ int main (int argc, char ** argv)
|
||||
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -19,6 +19,9 @@ Author: Richard Rollins <rprollins@users.noreply.github.com>
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include "disable_benchmarks_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -161,3 +164,5 @@ void bench_wilson_eo (
|
||||
double flops = (single_site_flops * volume * ncall)/2.0;
|
||||
std::cout << flops/(t1-t0) << "\t\t";
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
#pragma once
|
||||
|
||||
#ifndef ENABLE_FERMION_INSTANTIATIONS
|
||||
#include <iostream>
|
||||
|
||||
int main(void) {
|
||||
std::cout << "This build of Grid was configured to exclude fermion instantiations, "
|
||||
<< "which this benchmark relies on. "
|
||||
<< "Please reconfigure and rebuild Grid with --enable-fermion-instantiations"
|
||||
<< "to run this benchmark."
|
||||
<< std::endl;
|
||||
return 1;
|
||||
}
|
||||
#endif
|
||||
+11
-1
@@ -172,6 +172,12 @@ case ${ac_TRACING} in
|
||||
esac
|
||||
|
||||
############### fermions
|
||||
AC_ARG_ENABLE([fermion-instantiations],
|
||||
[AS_HELP_STRING([--enable-fermion-instantiations=yes|no],[enable fermion instantiations])],
|
||||
[ac_FERMION_REPS=${enable_fermion_instantiations}], [ac_FERMION_INSTANTIATIONS=yes])
|
||||
|
||||
AM_CONDITIONAL(BUILD_FERMION_INSTANTIATIONS, [ test "${ac_FERMION_INSTANTIATIONS}X" == "yesX" ])
|
||||
|
||||
AC_ARG_ENABLE([fermion-reps],
|
||||
[AS_HELP_STRING([--enable-fermion-reps=yes|no],[enable extra fermion representation support])],
|
||||
[ac_FERMION_REPS=${enable_fermion_reps}], [ac_FERMION_REPS=yes])
|
||||
@@ -194,6 +200,9 @@ AM_CONDITIONAL(BUILD_ZMOBIUS, [ test "${ac_ZMOBIUS}X" == "yesX" ])
|
||||
case ${ac_FERMION_REPS} in
|
||||
yes) AC_DEFINE([ENABLE_FERMION_REPS],[1],[non QCD fermion reps]);;
|
||||
esac
|
||||
case ${ac_FERMION_INSTANTIATIONS} in
|
||||
yes) AC_DEFINE([ENABLE_FERMION_INSTANTIATIONS],[1],[enable fermions]);;
|
||||
esac
|
||||
case ${ac_GPARITY} in
|
||||
yes) AC_DEFINE([ENABLE_GPARITY],[1],[fermion actions with GPARITY BCs]);;
|
||||
esac
|
||||
@@ -292,13 +301,14 @@ AC_ARG_ENABLE([accelerator],
|
||||
case ${ac_ACCELERATOR} in
|
||||
cuda)
|
||||
echo CUDA acceleration
|
||||
LIBS="${LIBS} -lcuda"
|
||||
LIBS="${LIBS} -lcuda -lcublas -lcufft"
|
||||
AC_DEFINE([GRID_CUDA],[1],[Use CUDA offload]);;
|
||||
sycl)
|
||||
echo SYCL acceleration
|
||||
AC_DEFINE([GRID_SYCL],[1],[Use SYCL offload]);;
|
||||
hip)
|
||||
echo HIP acceleration
|
||||
LIBS="${LIBS} -lhipblas -lrocblas -lhipfft"
|
||||
AC_DEFINE([GRID_HIP],[1],[Use HIP offload]);;
|
||||
none)
|
||||
echo NO acceleration ;;
|
||||
|
||||
@@ -3,6 +3,9 @@
|
||||
* without regression / tests being applied
|
||||
*/
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -310,5 +313,4 @@ int main (int argc, char ** argv)
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -3,6 +3,9 @@
|
||||
* without regression / tests being applied
|
||||
*/
|
||||
|
||||
#include "disable_examples_without_instantiations.h"
|
||||
#ifdef ENABLE_FERMION_INSTANTIATIONS
|
||||
|
||||
#include <Grid/Grid.h>
|
||||
|
||||
using namespace std;
|
||||
@@ -432,5 +435,4 @@ int main (int argc, char ** argv)
|
||||
Grid_finalize();
|
||||
}
|
||||
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,803 @@
|
||||
/*************************************************************************************
|
||||
|
||||
Grid physics library, www.github.com/paboyle/Grid
|
||||
|
||||
Source file: ./tests/Test_padded_cell.cc
|
||||
|
||||
Copyright (C) 2023
|
||||
|
||||
Author: Peter Boyle <paboyle@ph.ed.ac.uk>
|
||||
|
||||
This program is free software; you can redistribute it and/or modify
|
||||
it under the terms of the GNU General Public License as published by
|
||||
the Free Software Foundation; either version 2 of the License, or
|
||||
(at your option) any later version.
|
||||
|
||||
This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
|
||||
You should have received a copy of the GNU General Public License along
|
||||
with this program; if not, write to the Free Software Foundation, Inc.,
|
||||
51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
||||
|
||||
See the full license in the file "LICENSE" in the top level distribution directory
|
||||
*************************************************************************************/
|
||||
/* END LEGAL */
|
||||
#include <Grid/Grid.h>
|
||||
#include <Grid/lattice/PaddedCell.h>
|
||||
#include <Grid/stencil/GeneralLocalStencil.h>
|
||||
|
||||
#include <Grid/algorithms/iterative/PrecGeneralisedConjugateResidual.h>
|
||||
#include <Grid/algorithms/iterative/PrecGeneralisedConjugateResidualNonHermitian.h>
|
||||
#include <Grid/algorithms/iterative/BiCGSTAB.h>
|
||||
|
||||
using namespace std;
|
||||
using namespace Grid;
|
||||
RealD FineSmootherShift = 0.1;
|
||||
int FineSmootherOrder = 8;
|
||||
int FineSmootherTol = 0;
|
||||
//RealD CoarseSmootherShift = 0.1;
|
||||
//int CoarseSmootherOrder = 8;
|
||||
//int CoarseSmootherTol = 0;
|
||||
RealD CoarseSolverShift = 0.002;
|
||||
RealD CoarseSolverTol = 0.03;
|
||||
int CoarseSolverOrder = 200;
|
||||
int CoarseMmax = 20; // coarse GCR restart length (was hardcoded 20)
|
||||
RealD mass=0.00078;
|
||||
void ParseEnvironment(void)
|
||||
{
|
||||
|
||||
if(getenv("MASS") ) mass = atof(getenv("MASS"));
|
||||
if(getenv("FineSmootherShift")) FineSmootherShift = atof(getenv("FineSmootherShift"));
|
||||
if(getenv("FineSmootherOrder")) FineSmootherOrder = atoi(getenv("FineSmootherOrder"));
|
||||
if(getenv("CoarseSolverShift")) CoarseSolverShift = atof(getenv("CoarseSolverShift"));
|
||||
if(getenv("CoarseSolverTol")) CoarseSolverTol = atof(getenv("CoarseSolverTol"));
|
||||
if(getenv("CoarseSolverOrder")) CoarseSolverOrder = atoi(getenv("CoarseSolverOrder"));
|
||||
if(getenv("CoarseMmax")) CoarseMmax = atoi(getenv("CoarseMmax"));
|
||||
if(getenv("DiagInvPrec"))
|
||||
{
|
||||
std::cout << GridLogMessage << "WARNING: DiagInvPrec option REMOVED (diagonal-inverse preconditioning wrecks fine->coarse null-vector inheritance); IGNORED" << std::endl;
|
||||
}
|
||||
|
||||
// if(getenv("CoarseSmootherShift")) CoarseSmootherShift = atof(getenv("CoarseSmootherShift"));
|
||||
// if(getenv("CoarseSmootherOrder")) CoarseSmootherOrder = atoi(getenv("CoarseSmootherOrder"));
|
||||
|
||||
std::cout << GridLogMessage << "PARAM: FineSmootherShift "<<FineSmootherShift<<std::endl;
|
||||
std::cout << GridLogMessage << "PARAM: FineSmootherOrder "<<FineSmootherOrder<<std::endl;
|
||||
// std::cout << GridLogMessage << "PARAM: CoarseSmootherShift "<<CoarseSmootherShift<<std::endl;
|
||||
// std::cout << GridLogMessage << "PARAM: CoarseSmootherOrder "<<CoarseSmootherOrder<<std::endl;
|
||||
std::cout << GridLogMessage << "PARAM: CoarseSolverShift "<<CoarseSolverShift<<std::endl;
|
||||
std::cout << GridLogMessage << "PARAM: CoarseSolverTol "<<CoarseSolverTol<<std::endl;
|
||||
std::cout << GridLogMessage << "PARAM: CoarseSolverOrder "<<CoarseSolverOrder<<std::endl;
|
||||
std::cout << GridLogMessage << "PARAM: CoarseMmax "<<CoarseMmax<<std::endl;
|
||||
|
||||
std::cout << GridLogMessage << "PARAM: MASS "<<mass<<std::endl;
|
||||
}
|
||||
|
||||
template <class T> void readFile(T& out, std::string const fname){
|
||||
#ifdef HAVE_LIME
|
||||
// Ref: https://github.com/paboyle/Grid/blob/feature/scidac-wp1/tests/debug/Test_general_coarse_hdcg_phys48.cc#L111
|
||||
std::cout << Grid::GridLogMessage << "Reads at: " << fname << std::endl;
|
||||
Grid::emptyUserRecord record;
|
||||
// Grid::ScidacReader SR(out.Grid()->IsBoss());
|
||||
Grid::ScidacReader SR;
|
||||
SR.open(fname);
|
||||
SR.readScidacFieldRecord(out, record);
|
||||
SR.close();
|
||||
#endif
|
||||
}
|
||||
template <class Field>
|
||||
void saveSubspace(std::vector<Field> &subspace, std::string const fname){
|
||||
#ifdef HAVE_LIME
|
||||
std::cout << Grid::GridLogMessage << "Saving subspace (" << subspace.size() << " vectors) to: " << fname << std::endl;
|
||||
Grid::emptyUserRecord record;
|
||||
Grid::ScidacWriter SW(subspace[0].Grid()->IsBoss());
|
||||
SW.open(fname);
|
||||
for (int k = 0; k < (int)subspace.size(); k++)
|
||||
SW.writeScidacFieldRecord(subspace[k], record);
|
||||
SW.close();
|
||||
#endif
|
||||
}
|
||||
|
||||
template <class Field>
|
||||
void loadSubspace(std::vector<Field> &subspace, std::string const fname){
|
||||
#ifdef HAVE_LIME
|
||||
std::cout << Grid::GridLogMessage << "Loading subspace (" << subspace.size() << " vectors) from: " << fname << std::endl;
|
||||
Grid::emptyUserRecord record;
|
||||
Grid::ScidacReader SR;
|
||||
SR.open(fname);
|
||||
for (int k = 0; k < (int)subspace.size(); k++)
|
||||
SR.readScidacFieldRecord(subspace[k], record);
|
||||
SR.close();
|
||||
#endif
|
||||
}
|
||||
|
||||
template<class Matrix,class Field>
|
||||
class PVdagMLinearOperator : public LinearOperatorBase<Field> {
|
||||
Matrix &_Mat;
|
||||
Matrix &_PV;
|
||||
int nApp;
|
||||
int nAppDag;
|
||||
public:
|
||||
PVdagMLinearOperator(Matrix &Mat,Matrix &PV): _Mat(Mat),_PV(PV), nApp(0), nAppDag(0) {};
|
||||
|
||||
void OpDiag (const Field &in, Field &out) { assert(0); }
|
||||
void OpDir (const Field &in, Field &out,int dir,int disp) { assert(0); }
|
||||
void OpDirAll (const Field &in, std::vector<Field> &out){ assert(0); };
|
||||
void Op (const Field &in, Field &out){
|
||||
// std::cout << GridLogMessage<< "Op: PVdag M "<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
_Mat.M(in,tmp);
|
||||
_PV.Mdag(tmp,out);
|
||||
nApp++;
|
||||
}
|
||||
void AdjOp (const Field &in, Field &out){
|
||||
// std::cout << GridLogMessage<<"AdjOp: Mdag PV "<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
_PV.M(in,tmp);
|
||||
_Mat.Mdag(tmp,out);
|
||||
nAppDag++;
|
||||
}
|
||||
|
||||
void clear() {
|
||||
nApp = 0;
|
||||
nAppDag = 0;
|
||||
}
|
||||
|
||||
void getApplications() {
|
||||
std::cout << GridLogMessage << "# applications of PVdagM: " << nApp << std::endl;
|
||||
std::cout << GridLogMessage << "# applications of PVdagM^dag: " << nAppDag << std::endl;
|
||||
std::cout << GridLogMessage << "# applications total: " << nApp + nAppDag << std::endl;
|
||||
}
|
||||
|
||||
void HermOpAndNorm(const Field &in, Field &out,RealD &n1,RealD &n2){
|
||||
HermOp(in,out);
|
||||
ComplexD dot = innerProduct(in,out);
|
||||
n1=real(dot);
|
||||
n2=norm2(out);
|
||||
}
|
||||
void HermOp(const Field &in, Field &out){
|
||||
// std::cout <<GridLogMessage<< "HermOp: Mdag PV PVdag M"<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
Op(in,tmp);
|
||||
AdjOp(tmp,out);
|
||||
// std::cout << "HermOp done "<<norm2(out)<<std::endl;
|
||||
}
|
||||
};
|
||||
template<class Matrix,class Field>
|
||||
class MdagPVLinearOperator : public LinearOperatorBase<Field> {
|
||||
Matrix &_Mat;
|
||||
Matrix &_PV;
|
||||
public:
|
||||
MdagPVLinearOperator(Matrix &Mat,Matrix &PV): _Mat(Mat),_PV(PV){};
|
||||
|
||||
void OpDiag (const Field &in, Field &out) { assert(0); }
|
||||
void OpDir (const Field &in, Field &out,int dir,int disp) { assert(0); }
|
||||
void OpDirAll (const Field &in, std::vector<Field> &out){ assert(0); };
|
||||
void Op (const Field &in, Field &out){
|
||||
Field tmp(in.Grid());
|
||||
// std::cout <<GridLogMessage<< "Op: PVdag M "<<std::endl;
|
||||
_PV.M(in,tmp);
|
||||
_Mat.Mdag(tmp,out);
|
||||
}
|
||||
void AdjOp (const Field &in, Field &out){
|
||||
// std::cout <<GridLogMessage<< "AdjOp: Mdag PV "<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
_Mat.M(in,tmp);
|
||||
_PV.Mdag(tmp,out);
|
||||
}
|
||||
void HermOpAndNorm(const Field &in, Field &out,RealD &n1,RealD &n2){
|
||||
ComplexD dot = innerProduct(in,out);
|
||||
n1=real(dot);
|
||||
n2=norm2(out);
|
||||
}
|
||||
void HermOp(const Field &in, Field &out){
|
||||
// std::cout << GridLogMessage<<"HermOp: PVdag M Mdag PV "<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
Op(in,tmp);
|
||||
AdjOp(tmp,out);
|
||||
// std::cout << "HermOp done "<<norm2(out)<<std::endl;
|
||||
}
|
||||
};
|
||||
template<class Matrix,class Field>
|
||||
class ShiftedPVdagMLinearOperator : public LinearOperatorBase<Field> {
|
||||
Matrix &_Mat;
|
||||
Matrix &_PV;
|
||||
int nApp;
|
||||
int nAppDag;
|
||||
public:
|
||||
RealD shift;
|
||||
ShiftedPVdagMLinearOperator(RealD _shift,Matrix &Mat,Matrix &PV): shift(_shift),_Mat(Mat),_PV(PV) , nApp(0), nAppDag(0){};
|
||||
|
||||
void OpDiag (const Field &in, Field &out) { assert(0); }
|
||||
void OpDir (const Field &in, Field &out,int dir,int disp) { assert(0); }
|
||||
void OpDirAll (const Field &in, std::vector<Field> &out){ assert(0); };
|
||||
void Op (const Field &in, Field &out){
|
||||
// std::cout << "Op: PVdag M "<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
_Mat.M(in,tmp);
|
||||
_PV.Mdag(tmp,out);
|
||||
nApp++;
|
||||
out = out + shift * in;
|
||||
}
|
||||
void AdjOp (const Field &in, Field &out){
|
||||
// std::cout << "AdjOp: Mdag PV "<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
_PV.M(tmp,out);
|
||||
_Mat.Mdag(in,tmp);
|
||||
nAppDag++;
|
||||
out = out + shift * in;
|
||||
}
|
||||
void HermOpAndNorm(const Field &in, Field &out,RealD &n1,RealD &n2){ assert(0); }
|
||||
void HermOp(const Field &in, Field &out){
|
||||
// std::cout << "HermOp: Mdag PV PVdag M"<<std::endl;
|
||||
Field tmp(in.Grid());
|
||||
Op(in,tmp);
|
||||
AdjOp(tmp,out);
|
||||
}
|
||||
void clear() {
|
||||
nApp = 0;
|
||||
nAppDag = 0;
|
||||
}
|
||||
void getApplications() {
|
||||
std::cout << GridLogMessage << "# applications of ShiftedPVdagM: " << nApp << std::endl;
|
||||
std::cout << GridLogMessage << "# applications of ShiftedPVdagM^dag: " << nAppDag << std::endl;
|
||||
std::cout << GridLogMessage << "# applications total: " << nApp + nAppDag << std::endl;
|
||||
}
|
||||
};
|
||||
template<class Fobj,class CComplex,int nbasis>
|
||||
class MGPreconditioner : public LinearFunction< Lattice<Fobj> > {
|
||||
public:
|
||||
|
||||
using LinearFunction<Lattice<Fobj> >::operator();
|
||||
|
||||
typedef Aggregation<Fobj,CComplex,nbasis> Aggregates;
|
||||
typedef typename Aggregation<Fobj,CComplex,nbasis>::FineField FineField;
|
||||
typedef typename Aggregation<Fobj,CComplex,nbasis>::CoarseVector CoarseVector;
|
||||
typedef typename Aggregation<Fobj,CComplex,nbasis>::CoarseMatrix CoarseMatrix;
|
||||
typedef LinearOperatorBase<FineField> FineOperator;
|
||||
typedef LinearFunction <FineField> FineSmoother;
|
||||
typedef LinearOperatorBase<CoarseVector> CoarseOperator;
|
||||
typedef LinearFunction <CoarseVector> CoarseSolver;
|
||||
Aggregates & _Aggregates;
|
||||
FineOperator & _FineOperator;
|
||||
FineSmoother & _PreSmoother;
|
||||
FineSmoother & _PostSmoother;
|
||||
CoarseOperator & _CoarseOperator;
|
||||
CoarseSolver & _CoarseSolve;
|
||||
std::string name;
|
||||
|
||||
int level; void Level(int lv) {level = lv; };
|
||||
|
||||
MGPreconditioner(Aggregates &Agg,
|
||||
FineOperator &Fine,
|
||||
FineSmoother &PreSmoother,
|
||||
FineSmoother &PostSmoother,
|
||||
CoarseOperator &CoarseOperator_,
|
||||
CoarseSolver &CoarseSolve_,
|
||||
std::string _name = std::string("unnamed"))
|
||||
: _Aggregates(Agg),
|
||||
_FineOperator(Fine),
|
||||
_PreSmoother(PreSmoother),
|
||||
_PostSmoother(PostSmoother),
|
||||
_CoarseOperator(CoarseOperator_),
|
||||
_CoarseSolve(CoarseSolve_),
|
||||
name(_name),
|
||||
level(1) { }
|
||||
|
||||
virtual void operator()(const FineField &in, FineField & out)
|
||||
{
|
||||
GridBase *CoarseGrid = _Aggregates.CoarseGrid;
|
||||
// auto CoarseGrid = _CoarseOperator.Grid();
|
||||
CoarseVector Csrc(CoarseGrid);
|
||||
CoarseVector Csol(CoarseGrid);
|
||||
FineField vec1(in.Grid());
|
||||
FineField vec2(in.Grid());
|
||||
|
||||
std::cout<<GridLogMessage << "Calling PreSmoother " <<std::endl;
|
||||
|
||||
// std::cout<<GridLogMessage << "Calling PreSmoother input residual "<<norm2(in) <<std::endl;
|
||||
double t;
|
||||
// Fine Smoother
|
||||
// out = in;
|
||||
out = Zero();
|
||||
t=-usecond();
|
||||
_PreSmoother(in,out);
|
||||
t+=usecond();
|
||||
|
||||
std::cout<<GridLogMessage << "PreSmoother took "<< t/1000.0<< "ms" <<std::endl;
|
||||
|
||||
// Update the residual
|
||||
_FineOperator.Op(out,vec1); sub(vec1, in ,vec1);
|
||||
// std::cout<<GridLogMessage <<"Residual-1 now " <<norm2(vec1)<<std::endl;
|
||||
|
||||
// Fine to Coarse
|
||||
t=-usecond();
|
||||
_Aggregates.ProjectToSubspace (Csrc,vec1);
|
||||
t+=usecond();
|
||||
std::cout<<GridLogMessage << "Project to coarse took "<< t/1000.0<< "ms" <<std::endl;
|
||||
|
||||
// Coarse correction
|
||||
t=-usecond();
|
||||
Csol = Zero();
|
||||
_CoarseSolve(Csrc,Csol);
|
||||
//Csol=Zero();
|
||||
t+=usecond();
|
||||
std::cout<<GridLogMessage << "Coarse solve took "<< t/1000.0<< "ms" <<std::endl;
|
||||
|
||||
// Coarse to Fine
|
||||
t=-usecond();
|
||||
// _CoarseOperator.PromoteFromSubspace(_Aggregates,Csol,vec1);
|
||||
_Aggregates.PromoteFromSubspace(Csol,vec1);
|
||||
add(out,out,vec1);
|
||||
t+=usecond();
|
||||
std::cout<<GridLogMessage << "Promote to this level took "<< t/1000.0<< "ms" <<std::endl;
|
||||
|
||||
// Residual
|
||||
_FineOperator.Op(out,vec1); sub(vec1 ,in , vec1);
|
||||
// std::cout<<GridLogMessage <<"Residual-2 now " <<norm2(vec1)<<std::endl;
|
||||
|
||||
// Fine Smoother
|
||||
t=-usecond();
|
||||
// vec2=vec1;
|
||||
vec2=Zero();
|
||||
_PostSmoother(vec1,vec2);
|
||||
t+=usecond();
|
||||
std::cout<<GridLogMessage << "PostSmoother took "<< t/1000.0<< "ms" <<std::endl;
|
||||
|
||||
add( out,out,vec2);
|
||||
std::cout<<GridLogMessage << "Done " <<std::endl;
|
||||
}
|
||||
};
|
||||
|
||||
template<class PVdagM_t, class ShiftedPVdagM_t, class Subspace, class LittleDiracOperator, class CoarseVector, class TwoLevelMG>
|
||||
void runMG(
|
||||
GridCartesian *FGrid,
|
||||
GridCartesian *Coarse5d,
|
||||
NextToNearestStencilGeometry5D geom,
|
||||
PVdagM_t PVdagM,
|
||||
ShiftedPVdagM_t ShiftedPVdagM,
|
||||
// std::vector<LatticeFermion> subspace
|
||||
Subspace AggregatesPD
|
||||
) {
|
||||
|
||||
// typedef Aggregation<vSpinColourVector,vTComplex,nbasis> Subspace;
|
||||
// typedef GeneralCoarsenedMatrix<vSpinColourVector,vTComplex,nbasis> LittleDiracOperator;
|
||||
// typedef LittleDiracOperator::CoarseVector CoarseVector;
|
||||
ParseEnvironment();
|
||||
|
||||
std::vector<LatticeFermion> subspace = AggregatesPD.subspace;
|
||||
int nbasis = subspace.size();
|
||||
const int cb = 0 ;
|
||||
|
||||
LatticeFermion err(FGrid);
|
||||
LatticeFermion prom(FGrid);
|
||||
LatticeFermion tmp(FGrid);
|
||||
|
||||
CoarseVector c_src (Coarse5d);
|
||||
CoarseVector c_res (Coarse5d);
|
||||
CoarseVector c_proj(Coarse5d);
|
||||
Complex one(1.0);
|
||||
|
||||
LatticeFermionD f_src(FGrid);
|
||||
LatticeFermionD f_res(FGrid);
|
||||
// typedef MGPreconditioner<vSpinColourVector, vTComplex,nbasis> TwoLevelMG;
|
||||
TrivialPrecon<CoarseVector> simple;
|
||||
TrivialPrecon<LatticeFermionD> simple_fine;
|
||||
|
||||
// Subspace AggregatesPD(Coarse5d,FGrid,cb);
|
||||
|
||||
// Orthonormalize subspace and compute nulliness
|
||||
|
||||
ShiftedPVdagM.shift = CoarseSolverShift;
|
||||
int nonherm = 0;
|
||||
LittleDiracOperator LittleDiracOpPV(geom,FGrid,Coarse5d,nonherm);
|
||||
LittleDiracOpPV.CoarsenOperator(ShiftedPVdagM, AggregatesPD);
|
||||
ShiftedPVdagM.shift = FineSmootherShift;
|
||||
|
||||
std::cout<<GridLogMessage<<std::endl;
|
||||
std::cout<<GridLogMessage<<"*******************************************"<<std::endl;
|
||||
std::cout<<GridLogMessage<<std::endl;
|
||||
std::cout<<GridLogMessage<<"Testing coarsened operator "<<std::endl;
|
||||
|
||||
c_src = one; // 1 in every element for vector 1.
|
||||
blockPromote(c_src,err,subspace);
|
||||
|
||||
prom=Zero();
|
||||
for(int b=0;b<nbasis;b++){
|
||||
prom=prom+subspace[b];
|
||||
}
|
||||
err=err-prom;
|
||||
std::cout<<GridLogMessage<<"Promoted back from subspace: err "<<norm2(err)<<std::endl;
|
||||
std::cout<<GridLogMessage<<"c_src "<<norm2(c_src)<<std::endl;
|
||||
std::cout<<GridLogMessage<<"prom "<<norm2(prom)<<std::endl;
|
||||
|
||||
// PVdagM.Op(prom,tmp);
|
||||
// blockProject(c_proj,tmp,subspace);
|
||||
// std::cout<<GridLogMessage<<" Called Big Dirac Op "<<norm2(tmp)<<std::endl;
|
||||
|
||||
// LittleDiracOpPV.M(c_src,c_res);
|
||||
// std::cout<<GridLogMessage<<" Called Little Dirac Op c_src "<< norm2(c_src) << " c_res "<< norm2(c_res) <<std::endl;
|
||||
|
||||
// std::cout<<GridLogMessage<<"Little dop : "<<norm2(c_res)<<std::endl;
|
||||
// // std::cout<<GridLogMessage<<" Little "<< c_res<<std::endl;
|
||||
// std::cout<<GridLogMessage<<"Big dop in subspace : "<<norm2(c_proj)<<std::endl;
|
||||
// // std::cout<<GridLogMessage<<" Big "<< c_proj<<std::endl;
|
||||
// c_proj = c_proj - c_res;
|
||||
// std::cout<<GridLogMessage<<" ldop error: "<<norm2(c_proj)<<std::endl;
|
||||
// // std::cout<<GridLogMessage<<" error "<< c_proj<<std::endl;
|
||||
|
||||
///////////////////////////////////////
|
||||
// Coarse grid solver test
|
||||
///////////////////////////////////////
|
||||
|
||||
std::cout<<GridLogMessage<<"******************* "<<std::endl;
|
||||
std::cout<<GridLogMessage<<" Coarse Grid Solve -- Level 2 "<<std::endl;
|
||||
std::cout<<GridLogMessage<<"******************* "<<std::endl;
|
||||
NonHermitianLinearOperator<LittleDiracOperator,CoarseVector> LinOpCoarse(LittleDiracOpPV);
|
||||
// DiagonalInverse preconditioning REMOVED (library support withdrawn: it
|
||||
// wrecks the collinearity that makes fine->coarse null-vector inheritance
|
||||
// free). TrivialPrecon reproduces the former DiagInvPrec=0 path exactly.
|
||||
PrecGeneralisedConjugateResidualNonHermitian<CoarseVector> L2PGCR(CoarseSolverTol, (CoarseSolverOrder+CoarseMmax-1)/CoarseMmax, LinOpCoarse,simple,CoarseMmax,CoarseMmax);
|
||||
L2PGCR.SetZeroGuess(1); // callers zero Csol / c_res
|
||||
L2PGCR.Level(2);
|
||||
L2PGCR.Name("Couter");
|
||||
c_res=Zero();
|
||||
L2PGCR(c_src,c_res);
|
||||
|
||||
|
||||
////////////////////////////////////////
|
||||
// Fine grid smoother
|
||||
////////////////////////////////////////
|
||||
// NonHermitianLinearOperator<PVdagM_t,LatticeFermionD> LinOpSmooth(PVdagM);
|
||||
|
||||
// PrecGeneralisedConjugateResidualNonHermitian<LatticeFermionD> SmootherGCR(0.05,1,ShiftedPVdagM,simple_fine,8,8);
|
||||
// Force 10 iters exactly, no early termination
|
||||
PrecGeneralisedConjugateResidualNonHermitian<LatticeFermionD> SmootherGCR(FineSmootherTol,1,
|
||||
ShiftedPVdagM,simple_fine,
|
||||
FineSmootherOrder,FineSmootherOrder);
|
||||
SmootherGCR.Level(1);
|
||||
SmootherGCR.Name("Fsmoother");
|
||||
SmootherGCR.SetZeroGuess(1); // pre/post slots + direct call all zero their guess
|
||||
|
||||
f_src = one; // 1 in every element for vector 1.
|
||||
f_res=Zero();
|
||||
SmootherGCR(f_src,f_res);
|
||||
|
||||
TwoLevelMG TwoLevelPrecon(AggregatesPD,
|
||||
PVdagM,
|
||||
simple_fine,
|
||||
SmootherGCR,
|
||||
LinOpCoarse,
|
||||
L2PGCR,
|
||||
"PVdagM");
|
||||
|
||||
PrecGeneralisedConjugateResidualNonHermitian<LatticeFermion> L1PGCR(1.0e-8,1000,PVdagM,TwoLevelPrecon,32,32);
|
||||
L1PGCR.SetZeroGuess(1); // f_res=Zero() before the solve
|
||||
L1PGCR.Level(1);
|
||||
L1PGCR.Name("Fouter");
|
||||
|
||||
std::cout<<GridLogMessage<<"******************* "<<std::endl;
|
||||
std::cout<<GridLogMessage<<" Running Multi Grid Solver "<<std::endl;
|
||||
std::cout<<GridLogMessage<<"******************* "<<std::endl;
|
||||
f_res=Zero();
|
||||
L1PGCR(f_src,f_res);
|
||||
|
||||
std::cout << GridLogMessage << "Fine Grid Smoother -- Level 2 operator uses: " << std::endl;
|
||||
PVdagM.getApplications();
|
||||
PVdagM.clear();
|
||||
ShiftedPVdagM.getApplications();
|
||||
ShiftedPVdagM.clear();
|
||||
|
||||
}
|
||||
|
||||
int main (int argc, char ** argv)
|
||||
{
|
||||
Grid_init(&argc,&argv);
|
||||
|
||||
// TODO read in more parameters: nbasis, GCR iters, smoother order, m
|
||||
// Might be impossible because nbasis needs to be a constant to be a template parameter
|
||||
// Usage : $ ./Example_pvdagm <nbasis> <smooth> <outerIters> <m>
|
||||
// std::string nbasisStr = argv[1];
|
||||
// std::string smoothStr = argv[2];
|
||||
// std::string outerStr = argv[3];
|
||||
// std::string mStr = argv[4];
|
||||
// int nbasis = std::stoi(nbasisStr);
|
||||
// int smooth = std::stoi(smoothStr);
|
||||
|
||||
const int Ls=24;
|
||||
RealD M5=1.8;
|
||||
|
||||
|
||||
// const int nbasis = 40;
|
||||
const int nbasis = 60;
|
||||
|
||||
std::cout << GridLogMessage << "Mass: " << mass << ", Ls: " << Ls << ", running Mobius kernel with b=1.5, c=0.5" << std::endl;
|
||||
std::cout << GridLogMessage << "nbasis: " << nbasis << std::endl;
|
||||
|
||||
std::vector<int> lat_size {48, 48, 48, 96};
|
||||
|
||||
GridCartesian * UGrid = SpaceTimeGrid::makeFourDimGrid(lat_size, GridDefaultSimd(Nd,vComplex::Nsimd()),GridDefaultMpi());
|
||||
GridRedBlackCartesian * UrbGrid = SpaceTimeGrid::makeFourDimRedBlackGrid(UGrid);
|
||||
|
||||
GridCartesian * FGrid = SpaceTimeGrid::makeFiveDimGrid(Ls,UGrid);
|
||||
GridRedBlackCartesian * FrbGrid = SpaceTimeGrid::makeFiveDimRedBlackGrid(Ls,UGrid);
|
||||
|
||||
// Construct a coarsened grid
|
||||
// Coordinate clatt = GridDefaultLatt();
|
||||
Coordinate clatt = lat_size;
|
||||
Coordinate Block({4,4,4,4});
|
||||
std::cout << GridLogMessage << "Lattice size: " << lat_size << std::endl;
|
||||
for(int d=0;d<clatt.size();d++){
|
||||
clatt[d] = lat_size[d]/Block[d];
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage << "constructing coarse grid" << std::endl;
|
||||
GridCartesian *Coarse4d = SpaceTimeGrid::makeFourDimGrid(clatt, GridDefaultSimd(Nd,vComplex::Nsimd()),GridDefaultMpi());
|
||||
GridCartesian *Coarse5d = SpaceTimeGrid::makeFiveDimGrid(1,Coarse4d);
|
||||
|
||||
std::vector<int> seeds4({1,2,3,4});
|
||||
std::vector<int> seeds5({5,6,7,8});
|
||||
std::vector<int> cseeds({5,6,7,8});
|
||||
GridParallelRNG RNG5(FGrid); RNG5.SeedFixedIntegers(seeds5);
|
||||
GridParallelRNG RNG4(UGrid); RNG4.SeedFixedIntegers(seeds4);
|
||||
GridParallelRNG CRNG(Coarse5d);CRNG.SeedFixedIntegers(cseeds);
|
||||
|
||||
LatticeFermion src(FGrid); random(RNG5,src);
|
||||
LatticeFermion result(FGrid); result=Zero();
|
||||
LatticeFermion ref(FGrid); ref=Zero();
|
||||
LatticeFermion tmp(FGrid);
|
||||
LatticeFermion err(FGrid);
|
||||
LatticeGaugeField Umu(UGrid);
|
||||
|
||||
|
||||
std::cout << GridLogMessage << "Reading in gauge field" << std::endl;
|
||||
FieldMetaData header;
|
||||
// std::string file("/sdcc/u/poare/PETSc-Grid/ckpoint_lat.4000");
|
||||
std::string file("/ccs/home/poare/ckpoint_lat.1000");
|
||||
NerscIO::readConfiguration(Umu,header,file);
|
||||
|
||||
/*
|
||||
|
||||
// DWF, m=0.01
|
||||
// std::string eigenPath = "/hpcgpfs01/work/lqcd/staging/RBC/ckpoint_lat.4000/ks_evecs/PVdagM_Nm80_Nk40_Niter5000_337342/";
|
||||
|
||||
// DWF, m=0.001
|
||||
// std::string eigenPath = "/hpcgpfs01/work/lqcd/staging/RBC/ckpoint_lat.4000/ks_evecs/PVdagM_Nm80_Nk40_Niter5000_m0p001_339143/";
|
||||
|
||||
// Mobius, m=0.001
|
||||
// std::string eigenPath = "/hpcgpfs01/work/lqcd/staging/RBC/ckpoint_lat.4000/ks_evecs/PVdagM_Nm80_Nk40_Niter5000_346851/";
|
||||
|
||||
// Frontier path
|
||||
std::string eigenPath = "/ccs/home/poare/lqcd/multigrid/spectra/ckpoint_lat.1000/...";
|
||||
|
||||
|
||||
std::cout << GridLogMessage << "Loading eigenvalues" << std::endl;
|
||||
std::ifstream evalFile(eigenPath + "evals.txt");
|
||||
std::string str;
|
||||
std::vector<ComplexD> evals;
|
||||
while (std::getline(evalFile, str)) {
|
||||
std::cout << GridLogMessage << "Reading line: " << str << std::endl;
|
||||
int i1 = str.find("(") + 1;
|
||||
int i2 = str.find(",") + 1;
|
||||
int i3 = str.find(")");
|
||||
std::cout << "i1,i2,i3 = " << i1 << "," << i2 << "," << i3 << std::endl;
|
||||
std::string reStr = str.substr(i1, i2 - i1);
|
||||
std::string imStr = str.substr(i2, i3 - i2);
|
||||
std::cout << GridLogMessage << "Parsed re = " << reStr << " and im = " << imStr << std::endl;
|
||||
// ComplexD z (std::stof(reStr), std::stof(imStr));
|
||||
ComplexD z (std::stod(reStr), std::stod(imStr));
|
||||
evals.push_back(z);
|
||||
}
|
||||
std::cout << GridLogMessage << "Eigenvalues: " << evals << std::endl;
|
||||
|
||||
int Nevecs = 20;
|
||||
std::vector<LatticeFermion> evecs;
|
||||
LatticeFermion evec (FGrid);
|
||||
for (int i = 0; i < Nevecs; i++) {
|
||||
std::string evecPath = eigenPath + "evec" + std::to_string(i);
|
||||
readFile(evec, evecPath);
|
||||
evecs.push_back(evec);
|
||||
}
|
||||
std::cout << GridLogMessage << "Evecs loaded" << std::endl;
|
||||
|
||||
*/
|
||||
// TODO uncomment when evecs are computed!
|
||||
|
||||
// DomainWallFermionD Ddwf(Umu,*FGrid,*FrbGrid,*UGrid,*UrbGrid,mass,M5);
|
||||
// DomainWallFermionD Dpv(Umu,*FGrid,*FrbGrid,*UGrid,*UrbGrid,1.0,M5);
|
||||
|
||||
// Mobius
|
||||
RealD b=1.5;// Scale factor b+c=2, b-c=1
|
||||
RealD c=0.5;
|
||||
MobiusFermionD Ddwf(Umu,*FGrid,*FrbGrid,*UGrid,*UrbGrid,mass,M5,b,c);
|
||||
MobiusFermionD Dpv(Umu,*FGrid,*FrbGrid,*UGrid,*UrbGrid,1.0,M5,b,c);
|
||||
|
||||
const int cb = 0 ;
|
||||
LatticeFermion prom(FGrid);
|
||||
|
||||
// assert(nbasis <= Nevecs); // need to have enough evecs
|
||||
|
||||
typedef GeneralCoarsenedMatrix<vSpinColourVector,vTComplex,nbasis> LittleDiracOperator;
|
||||
typedef LittleDiracOperator::CoarseVector CoarseVector;
|
||||
|
||||
NextToNearestStencilGeometry5D geom(Coarse5d);
|
||||
|
||||
std::cout<<GridLogMessage<<std::endl;
|
||||
std::cout<<GridLogMessage<<"*******************************************"<<std::endl;
|
||||
std::cout<<GridLogMessage<<std::endl;
|
||||
|
||||
// typedef PVdagMLinearOperator<DomainWallFermionD,LatticeFermionD> PVdagM_t;
|
||||
// typedef MdagPVLinearOperator<DomainWallFermionD,LatticeFermionD> MdagPV_t;
|
||||
// typedef ShiftedPVdagMLinearOperator<DomainWallFermionD,LatticeFermionD> ShiftedPVdagM_t;
|
||||
typedef PVdagMLinearOperator<MobiusFermionD,LatticeFermionD> PVdagM_t;
|
||||
typedef MdagPVLinearOperator<MobiusFermionD,LatticeFermionD> MdagPV_t;
|
||||
typedef ShiftedPVdagMLinearOperator<MobiusFermionD,LatticeFermionD> ShiftedPVdagM_t;
|
||||
|
||||
PVdagM_t PVdagM(Ddwf,Dpv);
|
||||
MdagPV_t MdagPV(Ddwf,Dpv);
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(2.0,Ddwf,Dpv); // 355
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(1.0,Ddwf,Dpv); // 246
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.5,Ddwf,Dpv); // 183
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.25,Ddwf,Dpv); // 145
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 134
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 127 -- NULL space via inverse iteration
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 57 -- NULL space via inverse iteration; 3 iterations
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.25,Ddwf,Dpv); // 57 , tighter inversion
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.25,Ddwf,Dpv); // nbasis 20 -- 49 iters
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.25,Ddwf,Dpv); // nbasis 20 -- 70 iters; asymmetric
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.25,Ddwf,Dpv); // 58; Loosen coarse, tighten fine
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 56 ...
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 51 ... with 24 vecs
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 31 ... with 24 vecs and 2^4 blocking
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 43 ... with 16 vecs and 2^4 blocking, sloppier
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 35 ... with 20 vecs and 2^4 blocking
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 35 ... with 20 vecs and 2^4 blocking, looser coarse
|
||||
// ShiftedPVdagM_t ShiftedPVdagM(0.1,Ddwf,Dpv); // 64 ... with 20 vecs, Christoph setup, and 2^4 blocking, looser coarse
|
||||
|
||||
ShiftedPVdagM_t ShiftedPVdagM(FineSmootherShift,Ddwf,Dpv); //
|
||||
|
||||
// Run power method on HOA??
|
||||
PowerMethod<LatticeFermion> PM;
|
||||
|
||||
CoarseVector c_src (Coarse5d);
|
||||
CoarseVector c_res (Coarse5d);
|
||||
CoarseVector c_proj(Coarse5d);
|
||||
Complex one(1.0);
|
||||
|
||||
std::vector<LatticeFermion> subspace(nbasis,FGrid);
|
||||
|
||||
LatticeFermionD f_src(FGrid);
|
||||
LatticeFermionD f_res(FGrid);
|
||||
typedef MGPreconditioner<vSpinColourVector, vTComplex,nbasis> TwoLevelMG;
|
||||
TrivialPrecon<CoarseVector> simple;
|
||||
TrivialPrecon<LatticeFermionD> simple_fine;
|
||||
|
||||
// Warning: This routine calls PVdagM.Op, not PVdagM.HermOp
|
||||
typedef Aggregation<vSpinColourVector,vTComplex,nbasis> Subspace;
|
||||
|
||||
// Breeds right singular vectors with call to HermOp (V)
|
||||
// int chebyOrd = 500;
|
||||
// V.CreateSubspaceChebyshev(RNG5,PVdagM,
|
||||
// nbasis,
|
||||
// 4000.0,0.003,
|
||||
// chebyOrd);
|
||||
// AggregatesPD.CreateSubspaceChebyshev(RNG5,
|
||||
// PVdagM,
|
||||
// nbasis,
|
||||
// 4000.0,
|
||||
// 0.003,
|
||||
// chebyOrd);
|
||||
|
||||
// Subspace testing (uncomment blocks when needed)
|
||||
|
||||
// - nbasis = 20, m=0.01, 35 outer iterations
|
||||
// - nbasis = 40, m=0.01, 23 outer iterations
|
||||
std::cout << GridLogMessage << "*** GCR setup ***" << std::endl;
|
||||
|
||||
// Subspace cache: save after generation, reload on subsequent runs to skip expensive setup.
|
||||
// Set SUBSPACE_FILE to override the default path.
|
||||
std::string subspace_file = "/lustre/orion/phy157/proj-shared/phy157_dwf/paboyle/subspace_nb"
|
||||
+ std::to_string(nbasis) + ".scidac";
|
||||
if ( getenv("SUBSPACE_FILE") ) subspace_file = std::string(getenv("SUBSPACE_FILE"));
|
||||
|
||||
// Check if subspace file exists (boss rank checks, result broadcast via GlobalSum).
|
||||
uint64_t file_exists = 0;
|
||||
if ( UGrid->IsBoss() ) {
|
||||
std::ifstream f(subspace_file);
|
||||
file_exists = f.good() ? 1 : 0;
|
||||
}
|
||||
UGrid->GlobalSum(file_exists);
|
||||
|
||||
Subspace AggregatesGCR(Coarse5d,FGrid,cb);
|
||||
|
||||
if ( file_exists ) {
|
||||
std::cout << GridLogMessage << "*** Loading subspace from disk ***" << std::endl;
|
||||
loadSubspace(AggregatesGCR.subspace, subspace_file);
|
||||
// Insurance: GLOBAL (whole-lattice) orthonormalise, matching what
|
||||
// CreateSubspaceGCR applies to generated subspaces (Aggregates.h:196), so a
|
||||
// reloaded file ends in the same state. This replaces the block
|
||||
// Orthogonalise() previously called here -- that is redundant (CoarsenOperator
|
||||
// block-GS's the subspace internally) and would leave a loaded file block-
|
||||
// orthonormal while a generated one is globally orthonormal. Global GS is
|
||||
// span-preserving, so the coarse operator is unchanged.
|
||||
AggregatesGCR.GlobalOrthonormalise();
|
||||
std::cout << GridLogMessage << "Subspace loaded and globally orthonormalised." << std::endl;
|
||||
} else {
|
||||
std::cout << GridLogMessage << "*** GCR subspace generation ***" << std::endl;
|
||||
AggregatesGCR.CreateSubspaceGCR(RNG5,PVdagM,nbasis);
|
||||
std::cout << GridLogMessage << "Subspace generation: PVdagM operator uses:" << std::endl;
|
||||
PVdagM.getApplications();
|
||||
PVdagM.clear();
|
||||
saveSubspace(AggregatesGCR.subspace, subspace_file);
|
||||
std::cout << GridLogMessage << "Subspace saved to: " << subspace_file << std::endl;
|
||||
}
|
||||
|
||||
std::cout << GridLogMessage << "Basis construction operator uses: " << std::endl;
|
||||
PVdagM.getApplications();
|
||||
PVdagM.clear();
|
||||
|
||||
std::cout << GridLogMessage << "Calling runMG " << std::endl;
|
||||
runMG<PVdagM_t, ShiftedPVdagM_t, Subspace, LittleDiracOperator, CoarseVector, TwoLevelMG>(
|
||||
FGrid,
|
||||
Coarse5d,
|
||||
geom,
|
||||
PVdagM,
|
||||
ShiftedPVdagM,
|
||||
AggregatesGCR
|
||||
);
|
||||
|
||||
//////////////////////////////////
|
||||
// Standard CG
|
||||
//////////////////////////////////
|
||||
#if 0
|
||||
{
|
||||
std::cout << "**************************************"<<std::endl;
|
||||
std::cout << "Calling red black CG"<<std::endl;
|
||||
std::cout << "**************************************"<<std::endl;
|
||||
ConjugateGradient<LatticeFermionD> CGfine(1.0e-8,30000,false);
|
||||
SchurDiagMooeeOperator<MobiusFermionD, LatticeFermion> HermOpEO(Ddwf);
|
||||
|
||||
LatticeFermion result(FrbGrid); result=Zero();
|
||||
LatticeFermion src(FrbGrid); random(RNG5,src);
|
||||
result=Zero();
|
||||
|
||||
CGfine(HermOpEO, src, result);
|
||||
}
|
||||
{
|
||||
std::cout << "**************************************"<<std::endl;
|
||||
std::cout << "Calling MdagM CG"<<std::endl;
|
||||
std::cout << "**************************************"<<std::endl;
|
||||
|
||||
LatticeFermion result(FGrid); result=Zero();
|
||||
LatticeFermion src(FGrid); random(RNG5,src);
|
||||
result=Zero();
|
||||
|
||||
MdagMLinearOperator<MobiusFermionD, LatticeFermionD> HermOp(Ddwf);
|
||||
ConjugateGradient<LatticeFermionD> CGfine(1.0e-8,100000,false);
|
||||
CGfine(HermOp, src, result);
|
||||
}
|
||||
{
|
||||
std::cout << "**************************************"<<std::endl;
|
||||
std::cout << "Calling PVdagM GCR"<<std::endl;
|
||||
std::cout << "**************************************"<<std::endl;
|
||||
|
||||
LatticeFermion result(FGrid); result=Zero();
|
||||
LatticeFermion src(FGrid); random(RNG5,src);
|
||||
result=Zero();
|
||||
PrecGeneralisedConjugateResidualNonHermitian<LatticeFermionD> GCR(1.0e-8,3000,PVdagM,simple_fine,50,50);
|
||||
GCR.Name("Fbaseline");
|
||||
GCR.SetZeroGuess(1); // result=Zero() above
|
||||
GCR(src,result);
|
||||
}
|
||||
#endif
|
||||
|
||||
|
||||
std::cout<<GridLogMessage<<std::endl;
|
||||
std::cout<<GridLogMessage << "Done "<< std::endl;
|
||||
|
||||
Grid_finalize();
|
||||
return 0;
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user