1 /*
  2  * Copyright (c) 2024, Oracle and/or its affiliates. All rights reserved.
  3  * DO NOT ALTER OR REMOVE COPYRIGHT NOTICES OR THIS FILE HEADER.
  4  *
  5  * This code is free software; you can redistribute it and/or modify it
  6  * under the terms of the GNU General Public License version 2 only, as
  7  * published by the Free Software Foundation.  Oracle designates this
  8  * particular file as subject to the "Classpath" exception as provided
  9  * by Oracle in the LICENSE file that accompanied this code.
 10  *
 11  * This code is distributed in the hope that it will be useful, but WITHOUT
 12  * ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
 13  * FITNESS FOR A PARTICULAR PURPOSE.  See the GNU General Public License
 14  * version 2 for more details (a copy is included in the LICENSE file that
 15  * accompanied this code).
 16  *
 17  * You should have received a copy of the GNU General Public License version
 18  * 2 along with this work; if not, write to the Free Software Foundation,
 19  * Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA.
 20  *
 21  * Please contact Oracle, 500 Oracle Parkway, Redwood Shores, CA 94065 USA
 22  * or visit www.oracle.com if you need additional information or have any
 23  * questions.
 24  */
 25 
 26 #include <sys/wait.h>
 27 #include <chrono>
 28 #include <thread>
 29 #include "cuda_backend.h"
 30 
 31 CudaBackend::CudaQueue::CudaQueue(Backend *backend)
 32         : Backend::Queue(backend),cuStream(),streamCreationThread() {
 33 }
 34 void CudaBackend::CudaQueue::init(){
 35     streamCreationThread = std::this_thread::get_id();
 36     if (backend->config->traceCalls){
 37         std::cout << "init() 0x"
 38                   << " thread=" <<streamCreationThread
 39                   << std::endl;
 40     }
 41 
 42     WHERE{.f=__FILE__ , .l=__LINE__,
 43           .e=cuStreamCreate(&cuStream,CU_STREAM_DEFAULT),
 44           .t= "cuStreamCreate"
 45     }.report();
 46 
 47     if (backend->config->traceCalls){
 48         std::cout << "exiting init() 0x"
 49                   << " custream=" <<std::hex<<streamCreationThread <<std::dec
 50                   << std::endl;
 51     }
 52 }
 53 
 54 void CudaBackend::CudaQueue::wait(){
 55     CUDA_CHECK(cuStreamSynchronize(cuStream), "cuStreamSynchronize");
 56 }
 57 
 58 
 59 void CudaBackend::CudaQueue::computeStart() {
 60     if (backend->config->alwaysCopy) {
 61         wait();
 62         release();
 63     }
 64 }
 65 
 66 void CudaBackend::CudaQueue::computeEnd() {
 67 
 68 }
 69 
 70 void CudaBackend::CudaQueue::release() {
 71 
 72 }
 73 
 74 CudaBackend::CudaQueue::~CudaQueue() {
 75     CUDA_CHECK(cuStreamDestroy(cuStream), "cuStreamDestroy");
 76 }
 77 
 78 void CudaBackend::CudaQueue::copyToDevice(Buffer *buffer) {
 79     const auto *cudaBuffer = dynamic_cast<CudaBuffer *>(buffer);
 80     const std::thread::id thread_id = std::this_thread::get_id();
 81     if (thread_id != streamCreationThread){
 82         std::cout << "copyToDevice()  thread=" <<thread_id<< " != "<< streamCreationThread<< std::endl;
 83     }
 84     if (backend->config->traceCalls) {
 85 
 86         std::cout << "copyToDevice() 0x"
 87                 << std::hex<<cudaBuffer->bufferState->length<<std::dec << "/"
 88                 << cudaBuffer->bufferState->length << " "
 89                 << "devptr=" << std::hex<<  static_cast<long>(cudaBuffer->devicePtr) <<std::dec
 90                 << " thread=" <<thread_id
 91                   << std::endl;
 92     }
 93 
 94     CUDA_CHECK(cuMemcpyHtoDAsync(cudaBuffer->devicePtr,
 95                     cudaBuffer->bufferState->ptr,
 96                     cudaBuffer->bufferState->length,
 97                     dynamic_cast<CudaQueue*>(backend->queue)->cuStream), "cuMemcpyHtoDAsync");
 98 }
 99 
100 void CudaBackend::CudaQueue::copyFromDevice(Buffer *buffer) {
101     const auto *cudaBuffer = dynamic_cast<CudaBuffer *>(buffer);
102     const std::thread::id thread_id = std::this_thread::get_id();
103     if (thread_id != streamCreationThread){
104         std::cout << "copyFromDevice()  thread=" <<thread_id<< " != "<< streamCreationThread<< std::endl;
105     }
106     if (backend->config->traceCalls) {
107 
108         std::cout << "copyFromDevice() 0x"
109                   << std::hex<<cudaBuffer->bufferState->length<<std::dec << "/"
110                   << cudaBuffer->bufferState->length << " "
111                   << "devptr=" << std::hex<<  static_cast<long>(cudaBuffer->devicePtr) <<std::dec
112                   << " thread=" <<thread_id
113                   << std::endl;
114     }
115 
116     CUDA_CHECK(cuMemcpyDtoHAsync(cudaBuffer->bufferState->ptr,
117                                 cudaBuffer->devicePtr,
118                                 cudaBuffer->bufferState->length,
119                                 dynamic_cast<CudaQueue*>(backend->queue)->cuStream),
120                                 "cuMemcpyDtoHAsync");
121 
122 }
123 
124 // TODO: Improve heuristics to decide a better block size, if possible.
125 // The following is just a rough number to fit into a modern NVIDIA GPU.
126 int CudaBackend::CudaQueue::estimateThreadsPerBlock(int dimensions) {
127     switch (dimensions) {
128         case 1: return 256;
129         case 2: return 16;
130         case 3: return 16;
131         default: return 1;
132     }
133 }
134 
135 int CudaBackend::CudaQueue::estimateThreadsPerBlock(int dimensions, int globalSizePerDimension, int localSize) {
136     int threadsPerBlock = 1;
137     if (localSize > 0) {
138         threadsPerBlock = localSize;
139     } else if (globalSizePerDimension > 1) {
140         threadsPerBlock = estimateThreadsPerBlock(dimensions);
141         // Check if we are running a small range
142         while (globalSizePerDimension < threadsPerBlock) {
143             threadsPerBlock /= 2;
144         }
145     }
146     return threadsPerBlock;
147 }
148 
149 void CudaBackend::CudaQueue::dispatch(DispatchContext *dispatchContext, CompilationUnit::Kernel *kernel) {
150     const auto cudaKernel = dynamic_cast<CudaModule::CudaKernel *>(kernel);
151 
152     int threadsPerBlockX = 1;
153     int threadsPerBlockY = 1;
154     int threadsPerBlockZ = 1;
155     int blocksPerGridX = 1;
156     int blocksPerGridY = 1;
157     int blocksPerGridZ = 1;
158     if (dispatchContext->type > 0) {
159         // Local Size must be > 0. The corresponding check happens in the Java side.
160         blocksPerGridX = ceil_div(dispatchContext->gsx, dispatchContext->lsx);
161         blocksPerGridY = ceil_div(dispatchContext->gsy, dispatchContext->lsy);
162         blocksPerGridZ = ceil_div(dispatchContext->gsz, dispatchContext->lsz);
163 
164         if (dispatchContext->wsx != 0) {
165             blocksPerGridX = blocksPerGridX * blocksPerGridY;
166             blocksPerGridY = 1;
167             blocksPerGridZ = 1;
168         } else if (dispatchContext->wsy != 0 || dispatchContext->wsz != 0) {
169             // Check from the Java side we never execute this. From the Java
170             // side we can throw an exception.
171             std::cerr << "NOT SUPPORTED = " << std::endl;
172         }
173 
174     } else {
175         threadsPerBlockX = estimateThreadsPerBlock(dispatchContext->dimensions, dispatchContext->gsx,
176                                                    dispatchContext->lsx);
177         threadsPerBlockY = estimateThreadsPerBlock(dispatchContext->dimensions, dispatchContext->gsy,
178                                                    dispatchContext->lsy);
179         threadsPerBlockZ = estimateThreadsPerBlock(dispatchContext->dimensions, dispatchContext->gsz,
180                                                    dispatchContext->lsz);
181 
182         int warpFactor[3] = {1, 1, 1};
183         if (dispatchContext->wsx != 0) {
184             warpFactor[0] = 32;
185         }
186         if (dispatchContext->wsy != 0) {
187             warpFactor[1] = 32;
188         }
189         if (dispatchContext->wsz != 0) {
190             warpFactor[2] = 32;
191         }
192 
193         int globalSize[3] = {dispatchContext->gsx, dispatchContext->gsy, dispatchContext->gsz};
194         globalSize[0] = dispatchContext->tlx
195                             ? ceil_div(dispatchContext->gsx, dispatchContext->tlx) * warpFactor[0]
196                             : dispatchContext->gsx;
197         globalSize[1] = dispatchContext->tly
198                             ? ceil_div(dispatchContext->gsy, dispatchContext->tly) * warpFactor[1]
199                             : dispatchContext->gsy;
200         globalSize[2] = dispatchContext->tlz
201                             ? ceil_div(dispatchContext->gsz, dispatchContext->tlz) * warpFactor[2]
202                             : dispatchContext->gsz;
203 
204         blocksPerGridX = ceil_div(globalSize[0], threadsPerBlockX);
205         blocksPerGridY = 1;
206         blocksPerGridZ = 1;
207         if (dispatchContext->dimensions > 1) {
208             blocksPerGridY = ceil_div(globalSize[1], threadsPerBlockY);
209         }
210         if (dispatchContext->dimensions > 2) {
211             blocksPerGridZ = ceil_div(globalSize[2], threadsPerBlockZ);
212         }
213     }
214 
215     // Enable debug information with info: HAT=INFO
216     if (backend->config->info) {
217         backend->shortDeviceInfo();
218         std::cout << "[INFO] Dispatching the CUDA kernel" << std::endl;
219         std::cout << "        \\_ BlocksPerGrid   = [" << blocksPerGridX << "," << blocksPerGridY << "," << blocksPerGridZ << "]" << std::endl;
220         std::cout << "        \\_ ThreadsPerBlock = [" << threadsPerBlockX << "," << threadsPerBlockY << "," << threadsPerBlockZ << "]" << std::endl;
221     }
222 
223     const std::thread::id thread_id = std::this_thread::get_id();
224     if (thread_id != streamCreationThread) {
225         std::cout << "dispatch()  thread=" << thread_id << " != " << streamCreationThread << std::endl;
226     }
227 
228     //     // CUDA events for timing
229     //     cudaEvent_t start, stop;
230     //     cuEventCreate(&start, cudaEventDefault);
231     //     cuEventCreate(&stop, cudaEventDefault);
232     //     cuEventRecord(start, 0);
233 
234     const auto status = cuLaunchKernel(cudaKernel->function, //
235                                        blocksPerGridX, blocksPerGridY, blocksPerGridZ, //
236                                        threadsPerBlockX, threadsPerBlockY, threadsPerBlockZ, //
237                                        0, //
238                                        cuStream, //
239                                        cudaKernel->argslist, //
240                                        nullptr);
241     //     cuEventRecord(stop, 0);
242     //     cuEventSynchronize(stop);
243     //     float elapsedTimeMs = 0.0f;
244     //     cuEventElapsedTime(&elapsedTimeMs, start, stop);
245     //     std::cout << "Kernel Elapsed Time: " << elapsedTimeMs << " ms\n";
246 
247     CUDA_CHECK(status, "cuLaunchKernel");
248 }