The Quantum Exact Simulation Toolkit v4.3.0
Loading...
Searching...
No Matches
environment.cpp
1/** @file
2 * API definitions for managing QuESTEnv instances, which
3 * themselves control and query the deployment environment.
4 *
5 * @author Tyson Jones
6 */
7
8#include "quest/include/config.h"
9#include "quest/include/environment.h"
10#include "quest/include/precision.h"
11#include "quest/include/modes.h"
12
13#include "quest/src/core/errors.hpp"
14#include "quest/src/core/memory.hpp"
15#include "quest/src/core/parser.hpp"
16#include "quest/src/core/printer.hpp"
17#include "quest/src/core/envvars.hpp"
18#include "quest/src/core/autodeployer.hpp"
19#include "quest/src/core/validation.hpp"
20#include "quest/src/core/randomiser.hpp"
21#include "quest/src/comm/comm_config.hpp"
22#include "quest/src/cpu/cpu_config.hpp"
23#include "quest/src/gpu/gpu_config.hpp"
24
25#include <iostream>
26#include <typeinfo>
27#include <cstring>
28#include <cstdio>
29#include <string>
30#include <thread>
31#include <vector>
32#include <tuple>
33
34using std::string;
35
36
37
38/*
39 * PRIVATE QUESTENV SINGLETON
40 *
41 * Global to this file, accessible to other files only through
42 * getQuESTEnv() which returns a copy, which also has const fields.
43 * The use of static ensures we never accidentally expose the "true"
44 * runtime single instance to other files. We allocate the env
45 * in heap memory (hence the pointer) so that we can defer
46 * initialisation of the const fields. The address being nullptr
47 * indicates the QuESTEnv is not currently initialised; perhaps never,
48 * or it was but has since been finalized.
49 */
50
51
52static QuESTEnv* global_envPtr = nullptr;
53
54
55
56/*
57 * PRIVATE QUESTENV INITIALISATION HISTORY
58 *
59 * indicating whether QuEST has ever been finalized. This is important, since
60 * the QuEST environment can only ever be initialised once, and can never
61 * be re-initialised after finalisation, due to re-initialisation of MPI
62 * being undefined behaviour.
63 */
64
65
66static bool global_hasEnvBeenFinalized = false;
67
68
69
70/*
71 * PRIVATE QUESTENV INITIALISATION INNER FUNCTIONS
72 */
73
74
75void validateAndInitCustomQuESTEnv(int useDistrib, bool userOwnsMpi, int useGpuAccel, int useMultithread, const char* caller) {
76
77 // ensure that we are never re-initialising QuEST (even after finalize) because
78 // this leads to undefined behaviour in distributed mode, as per the MPI std,
79 // regardless of whether the user owns MPI
80 validate_envNeverInit(global_envPtr != nullptr, global_hasEnvBeenFinalized, caller);
81
82 // load env-vars before validating deployment mode, because some env vars can
83 // affect validation (such as QUEST_PERMIT_NODES_TO_SHARE_GPU). note that
84 // some env-vars (like QUEST_DEFAULT_NUM_GPU_THREADS_PER_BLOCK) will be here
85 // validated to have a correct format (like an int), but the validity of its
86 // actual value will be checked later (since it requires deciding GPU-accel).
87 envvars_validateAndLoadEnvVars(caller);
88 validateconfig_setEpsilonToDefault();
89
90 // ensure the chosen deployment is compiled and supported by hardware.
91 // note that these error messages will be printed by every node because
92 // validation occurs before comm_init() below, so all processes spawned
93 // by mpirun believe they are each the main rank. This seems unavoidable.
94 validate_newEnvDeploymentMode(useDistrib, useGpuAccel, useMultithread, caller);
95
96 // overwrite deployments (left as modeflag::USE_AUTO=-1) with 0,1 (a bool),
97 // which crucially, resolves useDistrib, permitting its consultation below
98 autodep_chooseQuESTEnvDeployment(useDistrib, useGpuAccel, useMultithread);
99
100 // ensure that current state of MPI is valid
101 validate_mpiInitStatus(useDistrib, userOwnsMpi, caller);
102
103 // optionally initialise MPI; necessary before completing validation,
104 // and before any GPU initialisation and validation, since we will
105 // perform that specifically upon the MPI-process-bound GPU(s). Further,
106 // we can make sure validation errors are reported only by the root node.
107 if (useDistrib)
108 comm_init(userOwnsMpi);
109
110 validate_newEnvDistributedBetweenPower2Nodes(caller);
111
112 /// @todo
113 /// consider immediately disabling MPI here if comm_numNodes() == 1
114 /// (also overwriting useDistrib = 0)
115
116 // bind MPI nodes to unique GPUs; even when not distributed,
117 // and before we have validated local GPUs are compatible
118 if (useGpuAccel)
119 gpu_bindLocalGPUsToNodes();
120
121 // consult environment variable to decide whether to allow GPU sharing
122 // (default = false) which informs whether below validation is triggered
123 bool permitGpuSharing = envvars_getWhetherGpuSharingIsPermitted();
124
125 // each MPI process should ordinarily use a unique GPU. This is
126 // critical when initializing cuQuantum so that we don't re-init
127 // cuStateVec on any paticular GPU (which can apparently cause a
128 // so-far-unwitnessed runtime error), but is otherwise essential
129 // for good performance. GPU sharing is useful for unit testing
130 // however permitting a single GPU to test CUDA+MPI deployment
131 if (useGpuAccel && useDistrib && ! permitGpuSharing)
132 validate_newEnvNodesEachHaveUniqueGpu(caller);
133
134 /// @todo
135 /// should we warn here if each machine contains
136 /// more GPUs than deployed MPI-processes (some GPUs idle)?
137
138 // validate the initial numTPB env-var (if specified) is valid
139 int initNumThreadsPerBlock = envvars_getDefaultNumGpuThreadsPerBlock();
140 validate_numGpuThreadsPerBlock(initNumThreadsPerBlock, useGpuAccel, caller);
141 gpu_setNumThreadsPerBlock(initNumThreadsPerBlock);
142
143 // cuQuantum is always used in GPU-accelerated envs when available
144 bool useCuQuantum = useGpuAccel && gpu_isCuQuantumCompiled();
145 if (useCuQuantum) {
146 validate_gpuIsCuQuantumCompatible(caller); // assesses above bound GPU
147 gpu_initCuQuantum();
148 }
149
150 // MPI GPU-awareness detection is platform specific; sometimes it is
151 // known at compile-time, other times according to env-vars
152 bool isMpiGpuAware = comm_isMpiGpuAware();
153
154 // initialise RNG, used by measurements and random-state generation
155 rand_setSeedsToDefault();
156
157 // allocate space for the global QuESTEnv singleton (overwriting nullptr, unless malloc fails)
158 global_envPtr = (QuESTEnv*) malloc(sizeof(QuESTEnv));
159
160 // pedantically check that teeny tiny malloc just succeeded
161 if (global_envPtr == nullptr)
162 error_allocOfQuESTEnvFailed();
163
164 // bind deployment info to global instance (autocasting int to bool)
165 global_envPtr->isMultithreaded = useMultithread;
166 global_envPtr->isGpuAccelerated = useGpuAccel;
167 global_envPtr->isDistributed = useDistrib;
168 global_envPtr->isMpiUserOwned = userOwnsMpi;
169 global_envPtr->isMpiGpuAware = isMpiGpuAware;
170 global_envPtr->isCuQuantumEnabled = useCuQuantum;
171 global_envPtr->isGpuSharingEnabled = permitGpuSharing;
172
173 // bind distributed info
174 global_envPtr->rank = (useDistrib)? comm_getRank() : 0;
175 global_envPtr->numNodes = (useDistrib)? comm_getNumNodes() : 1;
176}
177
178
179
180/*
181 * PRIVATE QUESTENV REPORTING INNER FUNCTIONS
182 */
183
184
185void printPrecisionInfo() {
186
187 /// @todo
188 /// - report MPI qcomp type?
189 /// - report CUDA qcomp type?
190 /// - report CUDA kernel qcomp type?
191
192 print_table(
193 "precision", {
194 {"qreal", printer_getQrealType() + " (" + printer_getMemoryWithUnitStr(sizeof(qreal)) + ")"},
195
196 /// @todo this is showing the backend C++ qcomp type, rather than that actually wieldable
197 /// by the user which could the C-type. No idea how to solve this however!
198 {"qcomp", printer_getQcompType() + " (" + printer_getMemoryWithUnitStr(sizeof(qcomp)) + ")"},
199
200 {"qindex", printer_getQindexType() + " (" + printer_getMemoryWithUnitStr(sizeof(qindex)) + ")"},
201
202 /// @todo this currently prints 0 when epsilon is inf (encoded by zero), i.e. disabled
203 {"validationEpsilon", printer_toStr(validateconfig_getEpsilon())},
204 });
205}
206
207
208void printCompilationInfo() {
209
210 print_table(
211 "compilation", {
212 {"isOmpCompiled", cpu_isOpenmpCompiled()},
213 {"isMpiCompiled", comm_isMpiCompiled()},
214 {"isMpiSubCommCompiled", comm_isMpiSubCommCompiled()},
215 {"isGpuCompiled", gpu_isGpuCompiled()},
216 {"isHipCompiled", gpu_isHipCompiled()},
217 {"isCuQuantumCompiled", gpu_isCuQuantumCompiled()},
218 {"isCheckpointingCompiled", QUEST_COMPILE_ADIOS2},
219 });
220}
221
222
223void printDeploymentInfo() {
224
225 print_table(
226 "deployment", {
227 {"isOmpEnabled", global_envPtr->isMultithreaded},
228 {"isMpiEnabled", global_envPtr->isDistributed},
229 {"isGpuEnabled", global_envPtr->isGpuAccelerated},
230 {"isCuQuantumEnabled", global_envPtr->isCuQuantumEnabled},
231 });
232}
233
234
235void printCpuInfo() {
236
237 using namespace printer_substrings;
238
239 // assume RAM is unknown unless it can be queried
240 string ram = un;
241 try {
242 ram = printer_getMemoryWithUnitStr(mem_tryGetLocalRamCapacityInBytes()) + pm;
243 } catch(mem::COULD_NOT_QUERY_RAM e){};
244
245 /// @todo
246 /// - CPU info e.g. speeds/caches?
247
248 print_table(
249 "cpu", {
250 {"numCpuCores", printer_toStr(std::thread::hardware_concurrency()) + pm},
251 {"numOmpProcs", (cpu_isOpenmpCompiled())? printer_toStr(cpu_getNumOpenmpProcessors()) + pm : na},
252 {"numOmpThrds", (cpu_isOpenmpCompiled())? printer_toStr(cpu_getAvailableNumThreads()) + pn : na},
253 {"cpuMemory", ram},
254 {"cpuMemoryFree", un},
255 });
256}
257
258
259void printGpuInfo() {
260
261 using namespace printer_substrings;
262
263 /// @todo below:
264 /// - GPU compute capability
265 /// - GPU #SVMs etc
266
267 // must not query any GPU facilities unless confirmed compiled and available
268 bool isComp = gpu_isGpuCompiled();
269 bool isGpu = isComp && gpu_isGpuAvailable();
270
271 print_table(
272 "gpu", {
273 {"numGpus", isComp? printer_toStr(gpu_getNumberOfLocalGpus()) : na},
274 {"gpuDirect", isGpu? printer_toStr(gpu_isDirectGpuCommPossible()) : na},
275 {"gpuMemPools", isGpu? printer_toStr(gpu_doesGpuSupportMemPools()) : na},
276 {"gpuMemory", isGpu? printer_getMemoryWithUnitStr(gpu_getTotalMemoryInBytes()) + pg : na},
277 {"gpuMemoryFree", isGpu? printer_getMemoryWithUnitStr(gpu_getCurrentAvailableMemoryInBytes()) + pg : na},
278 {"gpuCache", isGpu? printer_getMemoryWithUnitStr(gpu_getCacheMemoryInBytes()) + pg : na},
279 {"numThreadsPerBlock", isGpu? printer_toStr(gpu_getNumThreadsPerBlock()) : na},
280 });
281}
282
283
284void printDistributionInfo() {
285
286 using namespace printer_substrings;
287
288 bool comm = global_envPtr->isDistributed;
289 bool gpu = global_envPtr->isGpuAccelerated;
290 bool both = comm && gpu;
291
292 print_table(
293 "distribution", {
294 {"isMpiUserOwned", comm? printer_toStr(global_envPtr->isMpiUserOwned) : na},
295 {"isMpiGpuAware", comm? printer_toStr(global_envPtr->isMpiGpuAware ) : na},
296 {"isGpuSharingEnabled", both? printer_toStr(global_envPtr->isGpuSharingEnabled) : na},
297 {"numMpiNodes", printer_toStr(global_envPtr->numNodes)},
298 });
299}
300
301
302void printQuregSizeLimits(bool isDensMatr) {
303
304 using namespace printer_substrings;
305
306 // for brevity
307 int numNodes = global_envPtr->numNodes;
308
309 // by default, CPU limits are unknown (because memory query might fail)
310 string maxQbForCpu = un;
311 string maxQbForMpiCpu = un;
312
313 // max CPU registers are only determinable if RAM query succeeds
314 try {
315 qindex cpuMem = mem_tryGetLocalRamCapacityInBytes();
316 maxQbForCpu = printer_toStr(mem_getMaxNumQuregQubitsWhichCanFitInMemory(isDensMatr, 1, cpuMem));
317
318 // and the max MPI sizes are only relevant when env is distributed
319 if (global_envPtr->isDistributed)
320 maxQbForMpiCpu = printer_toStr(mem_getMaxNumQuregQubitsWhichCanFitInMemory(isDensMatr, numNodes, cpuMem));
321
322 // when MPI irrelevant, change their status from "unknown" to "N/A"
323 else
324 maxQbForMpiCpu = na;
325
326 // no problem if we can't query RAM; we simply don't report relevant limits
327 } catch(mem::COULD_NOT_QUERY_RAM e) {};
328
329 // GPU limits are default N/A because they're always determinable when relevant
330 string maxQbForGpu = na;
331 string maxQbForMpiGpu = na;
332
333 // max GPU registers only relevant if env is GPU-accelerated
334 if (global_envPtr->isGpuAccelerated) {
335 qindex gpuMem = gpu_getCurrentAvailableMemoryInBytes();
336 maxQbForGpu = printer_toStr(mem_getMaxNumQuregQubitsWhichCanFitInMemory(isDensMatr, 1, gpuMem));
337
338 // and the max MPI sizes are further only relevant when env is distributed
339 if (global_envPtr->isDistributed)
340 maxQbForMpiGpu = printer_toStr(mem_getMaxNumQuregQubitsWhichCanFitInMemory(isDensMatr, numNodes, gpuMem));
341 }
342
343 // tailor table title to type of Qureg
344 string prefix = (isDensMatr)? "density matrix" : "statevector";
345 string title = prefix + " limits";
346
347 print_table(
348 title, {
349 {"minQubitsForMpi", (numNodes>1)? printer_toStr(mem_getMinNumQubitsForDistribution(numNodes)) : na},
350 {"maxQubitsForCpu", maxQbForCpu},
351 {"maxQubitsForGpu", maxQbForGpu},
352 {"maxQubitsForMpiCpu", maxQbForMpiCpu},
353 {"maxQubitsForMpiGpu", maxQbForMpiGpu},
354 {"maxQubitsForMemOverflow", printer_toStr(mem_getMaxNumQuregQubitsBeforeGlobalMemSizeofOverflow(isDensMatr, numNodes))},
355 {"maxQubitsForIndOverflow", printer_toStr(mem_getMaxNumQuregQubitsBeforeIndexOverflow(isDensMatr))},
356 });
357}
358
359
360void printQuregAutoDeployments(bool isDensMatr) {
361
362 // build all table rows dynamically before print
363 std::vector<std::tuple<string, string>> rows;
364
365 // we will get auto-deployment for every possible number of qubits; silly but cheap and robust!
366 int useDistrib, useGpuAccel, useMulti;
367 int prevDistrib, prevGpuAccel, prevMulti;
368
369 // assume all deployments disabled for 1 qubit
370 prevDistrib = 0;
371 prevGpuAccel = 0;
372 prevMulti = 0;
373
374 // test to theoretically max #qubits, surpassing max that can fit in RAM and GPUs, because
375 // auto-deploy will still try to deploy there to (then subsequent validation will fail)
376 int maxQubits = mem_getMaxNumQuregQubitsBeforeGlobalMemSizeofOverflow(isDensMatr, global_envPtr->numNodes);
377
378 for (int numQubits=1; numQubits<maxQubits; numQubits++) {
379
380 // re-choose auto deployment
381 useDistrib = modeflag::USE_AUTO;
382 useGpuAccel = modeflag::USE_AUTO;
383 useMulti = modeflag::USE_AUTO;;
384 autodep_chooseQuregDeployment(numQubits, isDensMatr, useDistrib, useGpuAccel, useMulti, *global_envPtr);
385
386 // skip if deployments are unchanged
387 if (useDistrib == prevDistrib &&
388 useGpuAccel == prevGpuAccel &&
389 useMulti == prevMulti)
390 continue;
391
392 // else prepare string summarising the new deployments (trailing space is fine)
393 string value = "";
394 if (useMulti)
395 value += "[omp] "; // ordered by #qubits to attempt consistent printed columns
396 if (useGpuAccel)
397 value += "[gpu] ";
398 if (useDistrib)
399 value += "[mpi] ";
400
401 // log the #qubits of the deployment change
402 rows.push_back({printer_toStr(numQubits) + " qubits", value});
403
404 // skip subsequent qubits with the same deployments
405 prevDistrib = useDistrib;
406 prevGpuAccel = useGpuAccel;
407 prevMulti = useMulti;
408 }
409
410 // tailor table title to type of Qureg
411 string prefix = (isDensMatr)? "density matrix" : "statevector";
412 string title = prefix + " autodeployment";
413 rows.empty()?
414 print_table(title, "(no parallelisations available)"):
415 print_table(title, rows);
416}
417
418
419
420/*
421 * API FUNCTIONS
422 */
423
424
425// enable invocation by both C and C++ binaries
426extern "C" {
427
428
429void initCustomQuESTEnv(int useDistrib, int useGpuAccel, int useMultithread) {
430
431 const bool userOwnsMpi = false;
432 validateAndInitCustomQuESTEnv(useDistrib, userOwnsMpi, useGpuAccel, useMultithread, __func__);
433}
434
435
437
438 const bool userOwnsMpi = false;
439 validateAndInitCustomQuESTEnv(modeflag::USE_AUTO, userOwnsMpi, modeflag::USE_AUTO, modeflag::USE_AUTO, __func__);
440}
441
442
444
445 return (int) (global_envPtr != nullptr);
446}
447
448
450 validate_envIsInit(__func__);
451
452 // returns a copy, so cheeky users calling memcpy() upon const struct still won't mutate
453 return *global_envPtr;
454}
455
456
458 validate_envIsInit(__func__);
459
460 // NOTE:
461 // calling this will not automatically
462 // free the memory of existing Quregs
463
464 if (global_envPtr->isGpuAccelerated)
465 gpu_clearCache(); // syncs first
466
467 if (global_envPtr->isGpuAccelerated && gpu_isCuQuantumCompiled())
468 gpu_finalizeCuQuantum();
469
470 if (global_envPtr->isDistributed) {
471 comm_sync();
472 comm_end();
473 }
474
475 // free global env's heap memory and flag it as unallocated
476 free(global_envPtr);
477 global_envPtr = nullptr;
478
479 // flag that the environment was finalised, to ensure it is never re-initialised
480 global_hasEnvBeenFinalized = true;
481}
482
483
485 validate_envIsInit(__func__);
486
487 if (global_envPtr->isGpuAccelerated)
488 gpu_sync();
489
490 if (global_envPtr->isDistributed)
491 comm_sync();
492}
493
494
496 validate_envIsInit(__func__);
497 validate_numReportedNewlinesAboveZero(__func__); // because trailing newline mandatory
498
499 /// @todo add function to write this output to file (useful for HPC debugging)
500
501 printer_sync();
502
503 print_label("QuEST execution environment");
504
505 bool statevec = false;
506 bool densmatr = true;
507
508 // we attempt to report properties of available hardware facilities
509 // (e.g. number of CPU cores, number of GPUs) even if the environment is not
510 // making use of them, to inform the user how they might change deployment.
511 printPrecisionInfo();
512 printCompilationInfo();
513 printDeploymentInfo();
514 printCpuInfo();
515 printGpuInfo();
516 printDistributionInfo();
517 printQuregSizeLimits(statevec);
518 printQuregSizeLimits(densmatr);
519 printQuregAutoDeployments(statevec);
520 printQuregAutoDeployments(densmatr);
521
522 // exclude mandatory newline above
523 print_oneFewerNewlines();
524
525 printer_sync();
526}
527
528
529void getQuESTEnvironmentString(char str[200]) {
530 validate_envIsInit(__func__);
531
532 int numThreads = cpu_isOpenmpCompiled()? cpu_getAvailableNumThreads() : 1;
533 int cuQuantum = global_envPtr->isGpuAccelerated && gpu_isCuQuantumCompiled();
534 int gpuDirect = global_envPtr->isGpuAccelerated && gpu_isDirectGpuCommPossible();
535
536 snprintf(str, 200, "CUDA=%d OpenMP=%d MPI=%d userOwnsMPI=%d threads=%d ranks=%d cuQuantum=%d gpuDirect=%d",
537 global_envPtr->isGpuAccelerated,
538 global_envPtr->isMultithreaded,
539 global_envPtr->isDistributed,
540 global_envPtr->isMpiUserOwned,
541 numThreads,
542 global_envPtr->numNodes,
543 cuQuantum,
544 gpuDirect);
545}
546
547
548// end de-mangler
549}
void getQuESTEnvironmentString(char str[200])
void reportQuESTEnv()
void finalizeQuESTEnv()
void initCustomQuESTEnv(int useDistrib, int useGpuAccel, int useMultithread)
QuESTEnv getQuESTEnv()
int isQuESTEnvInit()
void syncQuESTEnv()
void initQuESTEnv()