PeriDEM 0.3.0
PeriDEM -- Peridynamics-based high-fidelity model for granular media
Loading...
Searching...
No Matches
testParallelCompLib.cpp
Go to the documentation of this file.
1/*
2 * -------------------------------------------
3 * Copyright (c) 2021 - 2026 Prashant K. Jha
4 * -------------------------------------------
5 * PeriDEM https://github.com/prashjha/PeriDEM
6 *
7 * Distributed under the Boost Software License, Version 1.0. (See accompanying
8 * file LICENSE)
9 */
10
11#include "testParallelCompLib.h"
12#include "util/vecMethods.h"
13#include "util/randomDist.h"
14#include "util/point.h"
15#include "util/io.h"
16#include "mesh/mesh.h"
18#include "mesh/meshUtil.h"
19#include "geom/geomIncludes.h"
20#include <mpi.h>
21#include <format>
22#include <fstream>
23#include <iostream>
24#include <random>
25#include <vector>
26
27typedef std::mt19937 RandGenerator;
28typedef std::uniform_real_distribution<> UniformDistribution;
29
30#include <taskflow/taskflow/taskflow.hpp>
31#include <taskflow/taskflow/algorithm/for_each.hpp>
32
33namespace {
34
35 void printMsg(std::string msg, int mpiRank, int printMpiRank) {
36 if (printMpiRank < 0) {
37 std::cout << msg;
38 } else {
39 if (mpiRank == printMpiRank)
40 std::cout << msg;
41 }
42 }
43
44 double f1(const double &x){
45 return x*x*x + std::exp(x) - std::sin(x);
46 }
47
48 double f2(const double &x){
49 return 2*(x-0.5)*(x-0.5)*(x-0.5) + std::exp(x-0.5) - std::cos(x-0.5);
50 }
51
52 void setupOwnerAndGhost(size_t mpiSize,
53 size_t mpiRank,
54 const std::vector<size_t> &nodePartition,
55 const std::vector<std::vector<size_t>> &nodeNeighs,
56 std::vector<size_t> &ownedNodes,
57 std::vector<size_t> &ownedInternalNodes,
58 std::vector<size_t> &ownedBdryNodes,
59 std::vector<std::pair<std::vector<size_t>, std::vector<size_t>>> &ghostData) {
60
61 auto numNodes = nodePartition.size();
62
63 // clear data
64 ownedNodes.clear();
65 ownedInternalNodes.clear();
66 ownedBdryNodes.clear();
67 ghostData.resize(mpiSize);
68 for (size_t i_proc=0; i_proc<mpiSize; i_proc++) {
69 ghostData[i_proc].first.clear();
70 ghostData[i_proc].second.clear();
71 }
72
73 // setup
74 for (size_t i=0; i<numNodes; i++) {
75 if (nodePartition[i] == mpiRank) {
76 // this processor owns this node
77 ownedNodes.push_back(i);
78
79 // ascertain if this node has neighboring nodes owned by other processors
80 bool ghostExist = false;
81 for (auto j : nodeNeighs[i]) {
82 auto j_proc = nodePartition[j];
83 if (j_proc != mpiRank) {
84 ghostExist = true;
85
86 // add j to ghost node list (to receive from j_proc)
87 ghostData[j_proc].first.push_back(j);
88
89 // add i to ghost node list (to send to j_proc)
90 ghostData[j_proc].second.push_back(i);
91 }
92 }
93
94 if (ghostExist)
95 ownedBdryNodes.push_back(i);
96 else
97 ownedInternalNodes.push_back(i);
98 } // loop over neighboring nodes
99 } // loop over nodes
100
102 bool debugGhostData = false;
103 if (debugGhostData and mpiRank == 0) {
104 std::vector<std::vector<std::pair<std::vector<size_t>, std::vector<size_t>>>> ghostDataAllProc(
105 mpiSize);
106 std::vector<std::vector<size_t>> numGhostDataAllProc(mpiSize);
107 for (size_t i_proc = 0; i_proc < mpiSize; i_proc++) {
108 numGhostDataAllProc[i_proc].resize(mpiSize);
109 ghostDataAllProc[i_proc].resize(mpiSize);
110 }
111
112 for (size_t i_proc = 0; i_proc < mpiSize; i_proc++) {
113 // assume we are i_proc and then create ghostData for i_proc
114 for (size_t i = 0; i < numNodes; i++) {
115 if (nodePartition[i] == i_proc) {
116 // i_proc processor owns this node
117 for (auto j: nodeNeighs[i]) {
118 auto j_proc = nodePartition[j];
119 if (j_proc != i_proc) {
120 // add to ghost node list
121 ghostDataAllProc[i_proc][j_proc].first.push_back(j);
122 }
123 }
124 } // loop over neighboring nodes
125 } // loop over nodes
126
127 // total number of ghost nodes from neighboring processors
128 for (size_t j_proc = 0; j_proc < mpiSize; j_proc++)
129 numGhostDataAllProc[i_proc][j_proc] = ghostDataAllProc[i_proc][j_proc].first.size();
130 } // loop over i_proc
131
132 // print the information
133 std::cout << "\n\nGhost data debug output\n\n";
134 bool found_asym = false;
135 for (size_t i_proc = 0; i_proc < mpiSize; i_proc++) {
136 for (size_t j_proc = 0; j_proc < mpiSize; j_proc++) {
137 std::cout << std::format("(i,j) = ({}, {}), num data = {}\n",
138 i_proc, j_proc,
139 numGhostDataAllProc[i_proc][j_proc]);
140
141 if (j_proc > i_proc) {
142 if (numGhostDataAllProc[i_proc][j_proc] == numGhostDataAllProc[j_proc][i_proc])
143 std::cout << std::format(" symmetric: data ({}, {}) = data ({}, {})\n",
144 i_proc, j_proc, j_proc, i_proc);
145 else {
146 found_asym = true;
147 std::cout << std::format(
148 " asymmetric: data ({}, {}) != data ({}, {})\n",
149 i_proc, j_proc, j_proc, i_proc);
150 }
151 }
152 }
153 }
154 if (found_asym)
155 std::cout << "Found asymetric ghost data\n";
156 else
157 std::cout << "No asymetric ghost data\n";
158
159 }
161 } // setupOwnerAndGhost()
162
163 void exchangeDispData(size_t mpiSize,
164 size_t mpiRank,
165 const std::vector<std::pair<std::vector<size_t>, std::vector<size_t>>> &ghostData,
166 std::vector<std::pair<std::vector<util::Point>, std::vector<util::Point>>> &dispGhostData,
167 std::vector<util::Point> &dispNodes) {
168 // resize dispGhostData if not done
169 dispGhostData.resize(mpiSize);
170 for (size_t j_proc = 0; j_proc < mpiSize; j_proc++) {
171 auto j_data_size = ghostData[j_proc].first.size(); // same size for second
172 dispGhostData[j_proc].first.resize(j_data_size);
173 dispGhostData[j_proc].second.resize(j_data_size);
174 }
175
176 // exchange data
177 util::io::print("\n\nBegin exchange data\n\n");
178 util::io::print(std::format("\n\nThis processor's rank = {}\n\n", mpiRank), util::io::print_default_tab, -1);
179 MPI_Request mpiRequests[2*(mpiSize-1)];
180 size_t requestCounter = 0;
181 for (size_t j_proc=0; j_proc<mpiSize; j_proc++) {
182 auto & sendIds = ghostData[j_proc].second;
183 auto & recvIds = ghostData[j_proc].first;
184
185 if (j_proc != mpiRank and recvIds.size() != 0) {
186
187 util::io::print(std::format("\n\n Processing j_proc = {}\n\n", j_proc), util::io::print_default_tab, -1);
188
189 // fill in ghost displacement data that we are sending from nodal displacement data
190 for (size_t k = 0; k<sendIds.size(); k++)
191 dispGhostData[j_proc].second[k] = dispNodes[sendIds[k]];
192
193 // send data from this process to j_proc
194 MPI_Isend(dispGhostData[j_proc].second.data(), 3*sendIds.size(),
195 MPI_DOUBLE, j_proc, 0, MPI_COMM_WORLD,
196 &mpiRequests[requestCounter++]);
197
198 // receive data from j_proc
199 MPI_Irecv(dispGhostData[j_proc].first.data(), 3*recvIds.size(),
200 MPI_DOUBLE, j_proc, 0, MPI_COMM_WORLD,
201 &mpiRequests[requestCounter++]);
202 }
203 } // loop over j_proc
204
205 util::io::print("\n\nCalling MPI_Waitall\n\n");
206 MPI_Waitall(requestCounter, mpiRequests, MPI_STATUSES_IGNORE);
207
208 util::io::print("\n\nUpdate dispNodes data\n\n");
209
210 for (size_t j_proc=0; j_proc<mpiSize; j_proc++) {
211 auto & recvIds = ghostData[j_proc].first;
212 if (j_proc != mpiRank and recvIds.size() != 0) {
213 for (size_t k = 0; k<recvIds.size(); k++)
214 dispNodes[recvIds[k]] = dispGhostData[j_proc].first[k];
215 }
216 } // loop over j_proc
217 } // exchangeDispData()
218} // namespace
219
220
221std::string test::testTaskflow(size_t N, int seed) {
222
223 auto nThreads = util::parallel::getNThreads();
224 util::io::print(std::format("\n\ntestTaskflow(): Number of threads = {}\n\n", nThreads));
225
226 // task: perform N computations in serial and using taskflow for_each
227
228 // generate vector of random numbers
230 auto dist = util::DistributionSample<UniformDistribution>(0., 1., seed);
231
232 std::vector<double> x(N);
233 std::vector<double> y1(N);
234 std::vector<double> y2(N);
235 std::generate(std::begin(x), std::end(x), [&] { return dist(); });
236
237 // now do serial calculation
238 auto t1 = steady_clock::now();
239 for (size_t i = 0; i < N; i++) {
240 if (x[i] < 0.5)
241 y1[i] = f1(x[i]);
242 else
243 y1[i] = f2(x[i]);
244 }
245 auto t2 = steady_clock::now();
246 auto dt12 = util::methods::timeDiff(t1, t2, "microseconds");
247
248 // now do parallel calculation using taskflow
249 tf::Executor executor(nThreads);
250 tf::Taskflow taskflow;
251
252 taskflow.for_each_index((std::size_t) 0, N, (std::size_t) 1, [&x, &y2](std::size_t i) {
253 if (x[i] < 0.5)
254 y2[i] = f1(x[i]);
255 else
256 y2[i] = f2(x[i]);
257 }
258 ); // for_each
259
260 executor.run(taskflow).get();
261 auto t3 = steady_clock::now();
262 auto dt23 = util::methods::timeDiff(t2, t3, "microseconds");
263
264 // compare results
265 double y_err = 0.;
266 for (size_t i=0; i<N; i++)
267 y_err += std::pow(y1[i] - y2[i], 2);
268
269 if (y_err > 1.e-10) {
270 std::cerr << std::format("Error: Serial and taskflow computation results do not match (squared error = {})\n",
271 y_err);
272 exit(1);
273 }
274
275 // get time
276 std::ostringstream msg;
277 msg << std::format(" Serial computation took = {}ms\n", dt12);
278 msg << std::format(" Taskflow computation took = {}ms\n", dt23);
279 msg << std::format(" Speed-up factor = {}\n\n\n", dt12/dt23);
280
281 return msg.str();
282}
283
284void test::testMPI(size_t nGrid, size_t mHorizon,
285 size_t testOption, std::string meshFilename) {
286 int mpiSize, mpiRank;
287 MPI_Comm_size(MPI_COMM_WORLD, &mpiSize);
288 MPI_Comm_rank(MPI_COMM_WORLD, &mpiRank);
289
290 // number of partitions
291 size_t nPart(mpiSize);
292
293 // create uniform mesh
294 size_t dim(2);
295 auto mesh = mesh::Mesh(dim);
296 mesh.d_spatialDiscretization = "finite_difference";
297
298 // create mesh
299 std::string outMeshFilename = "";
300 if (testOption == 1) {
301 // set geometry details
302 std::pair<std::vector<double>, std::vector<double>> box;
303 std::vector<size_t> nGridVec;
304 for (size_t i=0; i<dim; i++) {
305 box.first.push_back(0.);
306 box.second.push_back(1.);
307 nGridVec.push_back(nGrid);
308 }
309
310 // call utility function to create mesh
311 util::io::print("\n\nCreating uniform mesh\n\n");
312 mesh::createUniformMesh(&mesh, dim, box, nGridVec);
313
314 // filename for outputting
315 outMeshFilename = std::format("uniform_mesh_Lx_{}_Ly_{}_Nx_{}_Ny_{}",
316 box.second[0], box.second[1],
317 nGridVec[0], nGridVec[1]);
318 }
319 else if (testOption == 2) {
320 if (meshFilename.empty()) {
321 std::cerr << "testGraphPartitioning(): mesh filename is empty.\n";
322 exit(1);
323 }
324
325 // call in-built function of mesh to create data from file
326 util::io::print("\n\nReading mesh\n\n");
327 mesh.createData(meshFilename);
328
329 // find the name of mesh file excluding path and extension
331 }
332 else {
333 std::cerr << "testMPI() accepts either 1 or 2 for testOption. The value "
334 << testOption << " is invalid.\n";
335 exit(1);
336 }
337
338 // print mesh data and write mesh to a file
339 util::io::print(mesh.printStr());
340
341 // calculate nonlocal neighborhood
342 double horizon = mHorizon*mesh.d_h;
343 std::vector<std::vector<size_t>> nodeNeighs(mesh.d_numNodes);
344 geom::computeNonlocalNeighborhood(mesh.d_nodes, horizon, nodeNeighs);
345
346 // partition the mesh on root processor and broadcast to other processors
347 mesh.d_nodePartition.resize(mesh.d_numNodes);
348 util::io::print("\n\nCreating partition of mesh\n\n");
349 if (mpiRank == 0)
350 mesh::metisGraphPartition("metis_kway", &mesh, nodeNeighs, nPart);
351
352 util::io::print("\n\nBroadcasting partition to all processors\n\n");
353 MPI_Bcast(mesh.d_nodePartition.data(), mesh.d_numNodes,
354 MPI_UNSIGNED_LONG, 0, MPI_COMM_WORLD);
355
356
357 // Tasks:
358 // 1. Create list of ghost nodes associated to neighboring processors
359 // 2. Create a method that updates displacement of ghost nodes via MPI communication
360 // 3. Create two set of nodes owned by this processor; internal set will have nodes
361 // that do not depend on ghost nodes and boundary set that depend on ghost nodes
362
363 // store id of nodes owned by this processor
364 std::vector<size_t> ownedNodes, ownedInternalNodes, ownedBdryNodes;
365
366 // for each neighboring processor, store id of ghost nodes owned by that processor
367 std::vector<std::pair<std::vector<size_t>, std::vector<size_t>>> ghostData;
368
369 // fill the owned and ghost node vectors
370 util::io::print("\n\nCalling setupOwnerAndGhost()\n\n");
371 setupOwnerAndGhost(mpiSize, mpiRank,
372 mesh.d_nodePartition, nodeNeighs,
373 ownedNodes, ownedInternalNodes, ownedBdryNodes,
374 ghostData);
375
376 // create dummy displacement vector with random values
377 std::vector<util::Point> dispNodes(mesh.d_numNodes, util::Point(-1., -1., -1.));
378 {
379 // int seed = mpiRank;
380 // RandGenerator gen(util::get_rd_gen(seed));
381 // auto dist = util::DistributionSample<UniformDistribution>(0., 1., seed);
382 // for (auto i: ownedNodes)
383 // dispNodes[i] = util::Point(dist(), dist(), dist());
384 // assign value to owned nodes
385 for (auto i: ownedNodes)
386 dispNodes[i] = util::Point(mpiRank + 1, (mpiRank + 1)*100, (mpiRank + 1)*10000);
387 }
388
389 // MPI communication to send and receive ghost nodes data
390 printMsg("\n\nCalling exchangeDispData()\n\n", mpiRank, 0);
391 std::vector<std::pair<std::vector<util::Point>, std::vector<util::Point>>> dispGhostData;
392 exchangeDispData(mpiSize, mpiRank, ghostData, dispGhostData, dispNodes);
393
395 if (true) {
396 printMsg("\n\nDebugging dispGhostData()\n\n", mpiRank, -1);
397 bool debug_failed = false;
398 for (size_t j_proc = 0; j_proc < mpiSize; j_proc++) {
399 auto &recvIds = ghostData[j_proc].first;
400
401 if (j_proc != mpiRank and recvIds.size() != 0) {
402 for (size_t k = 0; k < recvIds.size(); k++) {
403 auto &uk = dispNodes[recvIds[k]];
404
405 // verify we received correct value of uk
406 if (uk[0] != j_proc + 1 or uk[1] != 100 * (j_proc + 1)
407 or uk[2] != 10000 * (j_proc + 1)) {
408 debug_failed = true;
409 util::io::print(std::format(" MPI exchange error: j_proc = {}, "
410 "uk = ({}, {}, {})\n", j_proc,
411 uk[0], uk[1], uk[2]),
413 }
414 }
415 }
416 } // loop over j_proc
417
418 if (debug_failed)
419 util::io::print(std::format("\n\nDEBUG failed for processor = {}\n\n", mpiRank), util::io::print_default_tab, -1);
420 else
421 util::io::print(std::format("\n\nDEBUG passed for processor = {}\n\n", mpiRank), util::io::print_default_tab, -1);
422 }
424}
A class for mesh data.
Definition mesh.h:53
Templated probability distribution.
Definition randomDist.h:90
void printMsg(std::string msg, int mpiRank, int printMpiRank)
void exchangeDispData(size_t mpiSize, size_t mpiRank, const std::vector< std::pair< std::vector< size_t >, std::vector< size_t > > > &ghostData, std::vector< std::pair< std::vector< util::Point >, std::vector< util::Point > > > &dispGhostData, std::vector< util::Point > &dispNodes)
void setupOwnerAndGhost(size_t mpiSize, size_t mpiRank, const std::vector< size_t > &nodePartition, const std::vector< std::vector< size_t > > &nodeNeighs, std::vector< size_t > &ownedNodes, std::vector< size_t > &ownedInternalNodes, std::vector< size_t > &ownedBdryNodes, std::vector< std::pair< std::vector< size_t >, std::vector< size_t > > > &ghostData)
void computeNonlocalNeighborhood(const std::vector< util::Point > &nodes, double horizon, std::vector< std::vector< size_t > > &nodeNeighs)
Partitions the nodes based on node neighborlist supplied. Function first creates a graph with nodes a...
Collection of methods and data related to finite element and mesh.
Definition mesh.cpp:29
void metisGraphPartition(std::string partitionMethod, const std::vector< std::vector< size_t > > &nodeNeighs, std::vector< size_t > &nodePartition, size_t nPartitions)
Partitions the nodes based on node neighborlist supplied. Function first creates a graph with nodes a...
void createUniformMesh(mesh::Mesh *mesh_p, size_t dim, std::pair< std::vector< double >, std::vector< double > > box, std::vector< size_t > nGrid)
Creates uniform mesh for rectangle/cuboid domain.
Definition meshUtil.cpp:64
std::string testTaskflow(size_t N, int seed)
Perform test on taskflow.
void testMPI(size_t nGrid=10, size_t mHorizon=3, size_t testOption=0, std::string meshFilename="")
Perform parallelization test using MPI on mesh partition based on metis.
const int print_default_tab
Default value of tab used in outputting formatted information.
Definition constants.h:18
void print(const T &msg, int nt=print_default_tab, int printMpiRank=print_default_mpi_rank)
Prints formatted information.
Definition io.h:171
std::string removeExtensionFromFile(std::string const &filename)
Remove extension from the filename Source - https://stackoverflow.com/a/24386991.
Definition io.h:339
std::string getFilenameFromPath(std::string const &path, std::string const &delims="/\\")
Get filename removing path from the string Source - https://stackoverflow.com/a/24386991.
Definition io.h:327
float timeDiff(std::chrono::steady_clock::time_point begin, std::chrono::steady_clock::time_point end, std::string unit="microseconds")
Returns difference between two times.
Definition vecMethods.h:309
unsigned int getNThreads()
Get number of threads to be used by taskflow.
RandGenerator get_rd_gen(int seed=-1)
Return random number generator.
Definition randomDist.h:30
std::mt19937 RandGenerator
Definition randomDist.h:16
A structure to represent 3d vectors.
Definition point.h:30
std::uniform_real_distribution UniformDistribution
std::mt19937 RandGenerator