From 800f4fe48eba1f36c0aba66a5a28521ab38f971d Mon Sep 17 00:00:00 2001 From: Alan Gray Date: Mon, 22 Apr 2024 04:50:39 -0700 Subject: [PATCH] Tidied to now only use CUDA runtime (not mixed with driver calls) --- ggml-cuda.cu | 22 ++++++++-------------- 1 file changed, 8 insertions(+), 14 deletions(-) diff --git a/ggml-cuda.cu b/ggml-cuda.cu index 670ba78a0..7da061240 100644 --- a/ggml-cuda.cu +++ b/ggml-cuda.cu @@ -2418,8 +2418,7 @@ struct ggml_cudaGraph { size_t numNodes = 0; int softmax_ne0 = 0; cudaGraphNode_t nodes[MAX_NODES_IN_CUDA_GRAPH]; - CUDA_KERNEL_NODE_PARAMS_v2 paramsDriver[MAX_NODES_IN_CUDA_GRAPH]; - cudaKernelNodeParams paramsRuntime[MAX_NODES_IN_CUDA_GRAPH]; + cudaKernelNodeParams params[MAX_NODES_IN_CUDA_GRAPH]; }; #endif @@ -2523,12 +2522,10 @@ GGML_CALL static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t // Loop over nodes, and extract kernel parameters fro each node for(size_t i=0; i