fallback to CPU buffer if host buffer alloc fails (#4610)

author slaren <redacted>

Sat, 23 Dec 2023 15:10:51 +0000 (16:10 +0100)

committer GitHub <redacted>

Sat, 23 Dec 2023 15:10:51 +0000 (16:10 +0100)
author slaren <redacted>
Sat, 23 Dec 2023 15:10:51 +0000 (16:10 +0100)
committer GitHub <redacted>
Sat, 23 Dec 2023 15:10:51 +0000 (16:10 +0100)
diff --git a/ggml-cuda.cu b/ggml-cuda.cu

index 490081cac8c1b656aa1401230144cf53cb25b27b..f9830328be51bbf8f1d283e90d9eed0c213b6b50 100644 (file)
--- a/ggml-cuda.cu
+++ b/ggml-cuda.cu
@@ -6729,8 +6729,7 @@ void * ggml_cuda_host_malloc(size_t size) {
      void * ptr = nullptr;
      cudaError_t err = cudaMallocHost((void **) &ptr, size);
      if (err != cudaSuccess) {
-        // The allocation error can be bypassed. A null ptr will assigned out of this function.
-        // This can fixed the OOM error in WSL.
+        // clear the error
          cudaGetLastError();
          fprintf(stderr, "WARNING: failed to allocate %.2f MB of pinned memory: %s\n",
              size/1024.0/1024.0, cudaGetErrorString(err));
@@ -9674,12 +9673,14 @@ ggml_backend_buffer_type_t ggml_backend_cuda_buffer_type(int device) {
  // host buffer type
  
  static void ggml_backend_cuda_host_buffer_free_buffer(ggml_backend_buffer_t buffer) {
-    CUDA_CHECK(cudaFreeHost(buffer->context));
+    ggml_cuda_host_free(buffer->context);
  }
  
  static ggml_backend_buffer_t ggml_backend_cuda_host_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) {
-    void * ptr;
-    CUDA_CHECK(cudaMallocHost(&ptr, size));
+    void * ptr = ggml_cuda_host_malloc(size);
+    if (ptr == nullptr) {
+        return nullptr;
+    }
  
      // FIXME: this is a hack to avoid having to implement a new buffer type
      ggml_backend_buffer_t buffer = ggml_backend_cpu_buffer_from_ptr(ptr, size);
diff --git a/llama.cpp b/llama.cpp

index 4e4495739bbbdd23be2803211b810678aa9c62e0..5699a0fcf3495e02f975bd7882742fd6be7f5477 100644 (file)
--- a/llama.cpp
+++ b/llama.cpp
@@ -1177,21 +1177,27 @@ static std::string llama_token_to_piece(const struct llama_context * ctx, llama_
  }
  
  static ggml_backend_buffer_type_t llama_default_buffer_type(int n_gpu_layers) {
+    ggml_backend_buffer_type_t buft = nullptr;
+
  #ifdef GGML_USE_METAL
      if (n_gpu_layers > 0) {
-        return ggml_backend_metal_buffer_type();
+        buft = ggml_backend_metal_buffer_type();
      }
  #elif defined(GGML_USE_CUBLAS) && defined(LLAMA_GGML_BACKEND_CUDA_TEST)
      if (n_gpu_layers > 0) {
-        return ggml_backend_cuda_buffer_type(0);
+        buft = ggml_backend_cuda_buffer_type(0);
      }
  #elif defined(GGML_USE_CUBLAS)
-    return ggml_backend_cuda_host_buffer_type();
+    buft = ggml_backend_cuda_host_buffer_type();
  #elif defined(GGML_USE_CPU_HBM)
-    return ggml_backend_cpu_hbm_buffer_type();
+    buft = ggml_backend_cpu_hbm_buffer_type();
  #endif
  
-    return ggml_backend_cpu_buffer_type();
+    if (buft == nullptr) {
+        buft = ggml_backend_cpu_buffer_type();
+    }
+
+    return buft;
  
      GGML_UNUSED(n_gpu_layers);
  }
author	slaren <redacted>
	Sat, 23 Dec 2023 15:10:51 +0000 (16:10 +0100)
committer	GitHub <redacted>
	Sat, 23 Dec 2023 15:10:51 +0000 (16:10 +0100)
ggml-cuda.cu		patch \| blob \| history
llama.cpp		patch \| blob \| history