diff --git a/Cxx11/transpose-put-mpi.cc b/Cxx11/transpose-put-mpi.cc new file mode 100644 index 000000000..eb86b0703 --- /dev/null +++ b/Cxx11/transpose-put-mpi.cc @@ -0,0 +1,234 @@ +/// +/// Copyright (c) 2020, Intel Corporation +/// Copyright (c) 2025, NVIDIA +/// +/// Redistribution and use in source and binary forms, with or without +/// modification, are permitted provided that the following conditions +/// are met: +/// +/// * Redistributions of source code must retain the above copyright +/// notice, this list of conditions and the following disclaimer. +/// * Redistributions in binary form must reproduce the above +/// copyright notice, this list of conditions and the following +/// disclaimer in the documentation and/or other materials provided +/// with the distribution. +/// * Neither the name of Intel Corporation nor the names of its +/// contributors may be used to endorse or promote products +/// derived from this software without specific prior written +/// permission. +/// +/// THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS +/// "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT +/// LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS +/// FOR A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE +/// COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, +/// INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, +/// BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; +/// LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER +/// CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT +/// LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN +/// ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE +/// POSSIBILITY OF SUCH DAMAGE. + +////////////////////////////////////////////////////////////////////// +/// +/// NAME: transpose +/// +/// PURPOSE: This program measures the time for the transpose of a +/// column-major stored matrix into a row-major stored matrix. +/// +/// USAGE: Program input is the matrix order and the number of times to +/// repeat the operation: +/// +/// transpose <# iterations> [tile size] +/// +/// An optional parameter specifies the tile size used to divide the +/// individual matrix blocks for improved cache and TLB performance. +/// +/// The output consists of diagnostics to make sure the +/// transpose worked and timing statistics. +/// +/// HISTORY: Written by Rob Van der Wijngaart, February 2009. +/// Converted to C++11 by Jeff Hammond, February 2016 and May 2017. +/// +////////////////////////////////////////////////////////////////////// + +#include "prk_util.h" +#include "prk_mpi.h" +#include "transpose-kernel.h" + +int main(int argc, char * argv[]) +{ + { + prk::MPI::state mpi(&argc,&argv); + + int np = prk::MPI::size(); + int me = prk::MPI::rank(); + + ////////////////////////////////////////////////////////////////////// + /// Read and test input parameters + ////////////////////////////////////////////////////////////////////// + + int iterations; + size_t order, block_order, tile_size; + + if (me == 0) { + std::cout << "Parallel Research Kernels" << std::endl; + std::cout << "C++11/MPI Matrix transpose (PUT-based): B = A^T" << std::endl; + + try { + if (argc < 3) { + throw "Usage: <# iterations> [tile size]"; + } + + iterations = std::atoi(argv[1]); + if (iterations < 1) { + throw "ERROR: iterations must be >= 1"; + } + + order = std::atol(argv[2]); + if (order <= 0) { + throw "ERROR: Matrix Order must be greater than 0"; + // } else if (order > prk::get_max_matrix_size()) { + // throw "ERROR: matrix dimension too large - overflow risk"; + } + + if (order % np != 0) { + throw "ERROR: Matrix order must be an integer multiple of the number of MPI processes"; + } + + // default tile size for tiling of local transpose + tile_size = (argc>3) ? std::atoi(argv[3]) : 32; + // a negative tile size means no tiling of the local transpose + if (tile_size <= 0) tile_size = order; + } + catch (const char * e) { + std::cout << e << std::endl; + prk::MPI::abort(1); + return 1; + } + + std::cout << "Number of iterations = " << iterations << std::endl; + std::cout << "Matrix order = " << order << std::endl; + std::cout << "Tile size = " << tile_size << std::endl; + } + + prk::MPI::bcast(&iterations); + prk::MPI::bcast(&order); + prk::MPI::bcast(&tile_size); + + block_order = order / np; + + //std::cout << "@" << me << " order=" << order << " block_order=" << block_order << std::endl; + + ////////////////////////////////////////////////////////////////////// + // Allocate space for the input and transpose matrix + ////////////////////////////////////////////////////////////////////// + + double trans_time{0}; + + // WA_A = window for A, LA_A = local address for A + auto [WA_A,LA_A] = prk::MPI::win_allocate(order * block_order); + // WB_B = window for B, LB_B = local address for B (B must be in MPI window for PUT operations) + auto [WA_B,LA_B] = prk::MPI::win_allocate(order * block_order); + + //A[order][block_order] + prk::vector A(LA_A, order * block_order, 0.0); + prk::vector B(LA_B, order * block_order, 0.0); + prk::vector T(block_order * block_order, 0.0); + + // fill A with the sequence 0 to order^2-1 as doubles + for (size_t i=0; i <# iterations> [tile size] +/// +/// An optional parameter specifies the tile size used to divide the +/// individual matrix blocks for improved cache and TLB performance. +/// +/// The output consists of diagnostics to make sure the +/// transpose worked and timing statistics. +/// +/// HISTORY: Written by Rob Van der Wijngaart, February 2009. +/// Converted to C++11 by Jeff Hammond, February 2016 and May 2017. +/// +////////////////////////////////////////////////////////////////////// + +#include "prk_util.h" +#include "prk_oshmem.h" +#include "transpose-kernel.h" + +int main(int argc, char * argv[]) +{ + { + prk::SHMEM::state shmem(&argc,&argv); + + int np = prk::SHMEM::size(); + int me = prk::SHMEM::rank(); + + ////////////////////////////////////////////////////////////////////// + /// Read and test input parameters + ////////////////////////////////////////////////////////////////////// + + int iterations; + size_t order, block_order, tile_size; + + if (me == 0) { + std::cout << "Parallel Research Kernels" << std::endl; + std::cout << "C++11/SHMEM Matrix transpose (PUT-based): B = A^T" << std::endl; + + try { + if (argc < 3) { + throw "Usage: <# iterations> [tile size]"; + } + + iterations = std::atoi(argv[1]); + if (iterations < 1) { + throw "ERROR: iterations must be >= 1"; + } + + order = std::atol(argv[2]); + if (order <= 0) { + throw "ERROR: Matrix Order must be greater than 0"; + } + + if (order % np != 0) { + throw "ERROR: Matrix order must be an integer multiple of the number of SHMEM PEs"; + } + + // default tile size for tiling of local transpose + tile_size = (argc>3) ? std::atoi(argv[3]) : 32; + // a negative tile size means no tiling of the local transpose + if (tile_size <= 0) tile_size = order; + } + catch (const char * e) { + std::cout << e << std::endl; + prk::SHMEM::abort(1); + return 1; + } + + std::cout << "Number of iterations = " << iterations << std::endl; + std::cout << "Matrix order = " << order << std::endl; + std::cout << "Tile size = " << tile_size << std::endl; + } + + prk::SHMEM::broadcast(&iterations); + prk::SHMEM::broadcast(&order); + prk::SHMEM::broadcast(&tile_size); + + block_order = order / np; + + ////////////////////////////////////////////////////////////////////// + // Allocate space for the input and transpose matrix + ////////////////////////////////////////////////////////////////////// + + double trans_time{0}; + + // A[order][block_order] - source matrix in symmetric memory + auto LA = prk::SHMEM::allocate(order * block_order); + prk::vector A(LA, order * block_order, 0.0); + + // B[order][block_order] - destination matrix in symmetric memory for PUT operations + auto LB = prk::SHMEM::allocate(order * block_order); + prk::vector B(LB, order * block_order, 0.0); + + // T[block_order][block_order] - temporary workspace for local transpose + prk::vector T(block_order * block_order, 0.0); + + // fill A with the sequence 0 to order^2-1 as doubles + for (size_t i=0; i