1 /*
2 * Copyright (c) 2024, Oracle and/or its affiliates. All rights reserved.
3 * DO NOT ALTER OR REMOVE COPYRIGHT NOTICES OR THIS FILE HEADER.
4 *
5 * This code is free software; you can redistribute it and/or modify it
6 * under the terms of the GNU General Public License version 2 only, as
7 * published by the Free Software Foundation. Oracle designates this
8 * particular file as subject to the "Classpath" exception as provided
9 * by Oracle in the LICENSE file that accompanied this code.
10 *
11 * This code is distributed in the hope that it will be useful, but WITHOUT
12 * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
13 * FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License
14 * version 2 for more details (a copy is included in the LICENSE file that
15 * accompanied this code).
16 *
17 * You should have received a copy of the GNU General Public License version
18 * 2 along with this work; if not, write to the Free Software Foundation,
19 * Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA.
20 *
21 * Please contact Oracle, 500 Oracle Parkway, Redwood Shores, CA 94065 USA
22 * or visit www.oracle.com if you need additional information or have any
23 * questions.
24 */
25
26 #include <sys/wait.h>
27 #include <chrono>
28 #include <thread>
29 #include "cuda_backend.h"
30
31 CudaBackend::CudaQueue::CudaQueue(Backend *backend)
32 : Backend::Queue(backend),cuStream(),streamCreationThread() {
33 }
34 void CudaBackend::CudaQueue::init(){
35 streamCreationThread = std::this_thread::get_id();
36 if (backend->config->traceCalls){
37 std::cout << "init() 0x"
38 << " thread=" <<streamCreationThread
39 << std::endl;
40 }
41
42 WHERE{.f=__FILE__ , .l=__LINE__,
43 .e=cuStreamCreate(&cuStream,CU_STREAM_DEFAULT),
44 .t= "cuStreamCreate"
45 }.report();
46
47 if (backend->config->traceCalls){
48 std::cout << "exiting init() 0x"
49 << " custream=" <<std::hex<<streamCreationThread <<std::dec
50 << std::endl;
51 }
52 }
53
54 void CudaBackend::CudaQueue::wait(){
55 CUDA_CHECK(cuStreamSynchronize(cuStream), "cuStreamSynchronize");
56 }
57
58
59 void CudaBackend::CudaQueue::computeStart() {
60 if (backend->config->alwaysCopy) {
61 wait();
62 release();
63 }
64 }
65
66 void CudaBackend::CudaQueue::computeEnd() {
67
68 }
69
70 void CudaBackend::CudaQueue::release() {
71
72 }
73
74 CudaBackend::CudaQueue::~CudaQueue() {
75 CUDA_CHECK(cuStreamDestroy(cuStream), "cuStreamDestroy");
76 }
77
78 void CudaBackend::CudaQueue::copyToDevice(Buffer *buffer) {
79 const auto *cudaBuffer = dynamic_cast<CudaBuffer *>(buffer);
80 const std::thread::id thread_id = std::this_thread::get_id();
81 if (thread_id != streamCreationThread){
82 std::cout << "copyToDevice() thread=" <<thread_id<< " != "<< streamCreationThread<< std::endl;
83 }
84 if (backend->config->traceCalls) {
85
86 std::cout << "copyToDevice() 0x"
87 << std::hex<<cudaBuffer->bufferState->length<<std::dec << "/"
88 << cudaBuffer->bufferState->length << " "
89 << "devptr=" << std::hex<< static_cast<long>(cudaBuffer->devicePtr) <<std::dec
90 << " thread=" <<thread_id
91 << std::endl;
92 }
93
94 CUDA_CHECK(cuMemcpyHtoDAsync(cudaBuffer->devicePtr,
95 cudaBuffer->bufferState->ptr,
96 cudaBuffer->bufferState->length,
97 dynamic_cast<CudaQueue*>(backend->queue)->cuStream), "cuMemcpyHtoDAsync");
98 }
99
100 void CudaBackend::CudaQueue::copyFromDevice(Buffer *buffer) {
101 const auto *cudaBuffer = dynamic_cast<CudaBuffer *>(buffer);
102 const std::thread::id thread_id = std::this_thread::get_id();
103 if (thread_id != streamCreationThread){
104 std::cout << "copyFromDevice() thread=" <<thread_id<< " != "<< streamCreationThread<< std::endl;
105 }
106 if (backend->config->traceCalls) {
107
108 std::cout << "copyFromDevice() 0x"
109 << std::hex<<cudaBuffer->bufferState->length<<std::dec << "/"
110 << cudaBuffer->bufferState->length << " "
111 << "devptr=" << std::hex<< static_cast<long>(cudaBuffer->devicePtr) <<std::dec
112 << " thread=" <<thread_id
113 << std::endl;
114 }
115
116 CUDA_CHECK(cuMemcpyDtoHAsync(cudaBuffer->bufferState->ptr,
117 cudaBuffer->devicePtr,
118 cudaBuffer->bufferState->length,
119 dynamic_cast<CudaQueue*>(backend->queue)->cuStream),
120 "cuMemcpyDtoHAsync");
121
122 }
123
124 // TODO: Improve heuristics to decide a better block size, if possible.
125 // The following is just a rough number to fit into a modern NVIDIA GPU.
126 int CudaBackend::CudaQueue::estimateThreadsPerBlock(int dimensions) {
127 switch (dimensions) {
128 case 1: return 256;
129 case 2: return 16;
130 case 3: return 16;
131 default: return 1;
132 }
133 }
134
135 int CudaBackend::CudaQueue::estimateThreadsPerBlock(int dimensions, int globalSizePerDimension, int localSize) {
136 int threadsPerBlock = 1;
137 if (localSize > 0) {
138 threadsPerBlock = localSize;
139 } else if (globalSizePerDimension > 1) {
140 threadsPerBlock = estimateThreadsPerBlock(dimensions);
141 // Check if we are running a small range
142 while (globalSizePerDimension < threadsPerBlock) {
143 threadsPerBlock /= 2;
144 }
145 }
146 return threadsPerBlock;
147 }
148
149 void CudaBackend::CudaQueue::dispatch(DispatchContext *dispatchContext, CompilationUnit::Kernel *kernel) {
150 const auto cudaKernel = dynamic_cast<CudaModule::CudaKernel *>(kernel);
151
152 int threadsPerBlockX = 1;
153 int threadsPerBlockY = 1;
154 int threadsPerBlockZ = 1;
155 int blocksPerGridX = 1;
156 int blocksPerGridY = 1;
157 int blocksPerGridZ = 1;
158 if (dispatchContext->type > 0) {
159 // Local Size must be > 0. The corresponding check happens in the Java side.
160 blocksPerGridX = ceil_div(dispatchContext->gsx, dispatchContext->lsx);
161 blocksPerGridY = ceil_div(dispatchContext->gsy, dispatchContext->lsy);
162 blocksPerGridZ = ceil_div(dispatchContext->gsz, dispatchContext->lsz);
163
164 if (dispatchContext->wsx != 0) {
165 blocksPerGridX = blocksPerGridX * blocksPerGridY;
166 blocksPerGridY = 1;
167 blocksPerGridZ = 1;
168 } else if (dispatchContext->wsy != 0 || dispatchContext->wsz != 0) {
169 // Check from the Java side we never execute this. From the Java
170 // side we can throw an exception.
171 std::cerr << "NOT SUPPORTED = " << std::endl;
172 }
173
174 } else {
175 threadsPerBlockX = estimateThreadsPerBlock(dispatchContext->dimensions, dispatchContext->gsx,
176 dispatchContext->lsx);
177 threadsPerBlockY = estimateThreadsPerBlock(dispatchContext->dimensions, dispatchContext->gsy,
178 dispatchContext->lsy);
179 threadsPerBlockZ = estimateThreadsPerBlock(dispatchContext->dimensions, dispatchContext->gsz,
180 dispatchContext->lsz);
181
182 int warpFactor[3] = {1, 1, 1};
183 if (dispatchContext->wsx != 0) {
184 warpFactor[0] = 32;
185 }
186 if (dispatchContext->wsy != 0) {
187 warpFactor[1] = 32;
188 }
189 if (dispatchContext->wsz != 0) {
190 warpFactor[2] = 32;
191 }
192
193 int globalSize[3] = {dispatchContext->gsx, dispatchContext->gsy, dispatchContext->gsz};
194 globalSize[0] = dispatchContext->tlx
195 ? ceil_div(dispatchContext->gsx, dispatchContext->tlx) * warpFactor[0]
196 : dispatchContext->gsx;
197 globalSize[1] = dispatchContext->tly
198 ? ceil_div(dispatchContext->gsy, dispatchContext->tly) * warpFactor[1]
199 : dispatchContext->gsy;
200 globalSize[2] = dispatchContext->tlz
201 ? ceil_div(dispatchContext->gsz, dispatchContext->tlz) * warpFactor[2]
202 : dispatchContext->gsz;
203
204 blocksPerGridX = ceil_div(globalSize[0], threadsPerBlockX);
205 blocksPerGridY = 1;
206 blocksPerGridZ = 1;
207 if (dispatchContext->dimensions > 1) {
208 blocksPerGridY = ceil_div(globalSize[1], threadsPerBlockY);
209 }
210 if (dispatchContext->dimensions > 2) {
211 blocksPerGridZ = ceil_div(globalSize[2], threadsPerBlockZ);
212 }
213 }
214
215 // Enable debug information with info: HAT=INFO
216 if (backend->config->info) {
217 backend->shortDeviceInfo();
218 std::cout << "[INFO] Dispatching the CUDA kernel" << std::endl;
219 std::cout << " \\_ BlocksPerGrid = [" << blocksPerGridX << "," << blocksPerGridY << "," << blocksPerGridZ << "]" << std::endl;
220 std::cout << " \\_ ThreadsPerBlock = [" << threadsPerBlockX << "," << threadsPerBlockY << "," << threadsPerBlockZ << "]" << std::endl;
221 }
222
223 const std::thread::id thread_id = std::this_thread::get_id();
224 if (thread_id != streamCreationThread) {
225 std::cout << "dispatch() thread=" << thread_id << " != " << streamCreationThread << std::endl;
226 }
227
228 // // CUDA events for timing
229 // cudaEvent_t start, stop;
230 // cuEventCreate(&start, cudaEventDefault);
231 // cuEventCreate(&stop, cudaEventDefault);
232 // cuEventRecord(start, 0);
233
234 const auto status = cuLaunchKernel(cudaKernel->function, //
235 blocksPerGridX, blocksPerGridY, blocksPerGridZ, //
236 threadsPerBlockX, threadsPerBlockY, threadsPerBlockZ, //
237 0, //
238 cuStream, //
239 cudaKernel->argslist, //
240 nullptr);
241 // cuEventRecord(stop, 0);
242 // cuEventSynchronize(stop);
243 // float elapsedTimeMs = 0.0f;
244 // cuEventElapsedTime(&elapsedTimeMs, start, stop);
245 // std::cout << "Kernel Elapsed Time: " << elapsedTimeMs << " ms\n";
246
247 CUDA_CHECK(status, "cuLaunchKernel");
248 }