Use runners for GPU discovery (#12090)

This revamps how we discover GPUs in the system by leveraging the Ollama runner. This should eliminate inconsistency between our GPU discovery and the runners capabilities at runtime, particularly for cases where we try to filter out unsupported GPUs. Now the runner does that implicitly based on the actual device list. In some cases free VRAM reporting can be unreliable which can leaad to scheduling mistakes, so this also includes a patch to leverage more reliable VRAM reporting libraries if available. Automatic workarounds have been removed as only one GPU leveraged this, which is now documented. This GPU will soon fall off the support matrix with the next ROCm bump. Additional cleanup of the scheduler and discovery packages can be done in the future once we have switched on the new memory management code, and removed support for the llama runner.

Use runners for GPU discovery (#12090)
This revamps how we discover GPUs in the system by leveraging the Ollama runner. This should eliminate inconsistency between our GPU discovery and the runners capabilities at runtime, particularly for cases where we try to filter out unsupported GPUs. Now the runner does that implicitly based on the actual device list. In some cases free VRAM reporting can be unreliable which can leaad to scheduling mistakes, so this also includes a patch to leverage more reliable VRAM reporting libraries if available. Automatic workarounds have been removed as only one GPU leveraged this, which is now documented. This GPU will soon fall off the support matrix with the next ROCm bump. Additional cleanup of the scheduler and discovery packages can be done in the future once we have switched on the new memory management code, and removed support for the llama runner.
bc8909fb · Daniel Hiltgen · GitHub · 6b50f2b9 · bc8909fb · bc8909fb
Unverified Commit bc8909fb authored Oct 01, 2025 by Daniel Hiltgen Committed by GitHub Oct 01, 2025
17 changed files
--- a/ml/backend/ggml/ggml/src/ggml-cuda/vendors/hip.h
+++ b/ml/backend/ggml/ggml/src/ggml-cuda/vendors/hip.h
@@ -42,6 +42,7 @@
 #define cudaDeviceProp hipDeviceProp_t
 #define cudaDeviceReset hipDeviceReset
 #define cudaDeviceSynchronize hipDeviceSynchronize
+#define cudaDriverGetVersion hipDriverGetVersion
 #define cudaError_t hipError_t
 #define cudaErrorPeerAccessAlreadyEnabled hipErrorPeerAccessAlreadyEnabled
 #define cudaErrorPeerAccessNotEnabled hipErrorPeerAccessNotEnabled

--- a/ml/backend/ggml/ggml/src/ggml-impl.h
+++ b/ml/backend/ggml/ggml/src/ggml-impl.h
@@ -602,6 +602,14 @@ static inline bool ggml_can_fuse(const struct ggml_cgraph * cgraph, int node_idx
    return true;
 }

+// Management libraries for fetching more accurate free VRAM data
+GGML_API int ggml_nvml_init();
+GGML_API int ggml_nvml_get_device_memory(const char *uuid, size_t *free, size_t *total);
+GGML_API void ggml_nvml_release();
+GGML_API int ggml_hip_mgmt_init();
+GGML_API int ggml_hip_get_device_memory(int pci_bus_id, int pci_device_id, size_t *free, size_t *total);
+GGML_API void ggml_hip_mgmt_release();
+
 #ifdef __cplusplus
 }
 #endif

--- a/ml/backend/ggml/ggml/src/ggml-metal/ggml-metal.m
+++ b/ml/backend/ggml/ggml/src/ggml-metal/ggml-metal.m
@@ -6523,12 +6523,14 @@ static enum ggml_backend_dev_type ggml_backend_metal_device_get_type(ggml_backen
    GGML_UNUSED(dev);
 }

+#define GGML_METAL_NAME "Metal"
 static void ggml_backend_metal_device_get_props(ggml_backend_dev_t dev, struct ggml_backend_dev_props * props) {
    props->name        = ggml_backend_metal_device_get_name(dev);
    props->description = ggml_backend_metal_device_get_description(dev);
    props->id          = "0";
    props->type        = ggml_backend_metal_device_get_type(dev);
    ggml_backend_metal_device_get_memory(dev, &props->memory_free, &props->memory_total);
+    props->library = GGML_METAL_NAME;
    props->caps = (struct ggml_backend_dev_caps) {
        /* .async                 = */ false,
        /* .host_buffer           = */ false,

--- a/ml/backend/ggml/ggml/src/ggml.go
+++ b/ml/backend/ggml/ggml/src/ggml.go
@@ -75,9 +75,9 @@ var OnceLoad = sync.OnceFunc(func() {
 		paths = value
 	}

-	split := filepath.SplitList(paths)
-	visited := make(map[string]struct{}, len(split))
-	for _, path := range split {
+	libPaths = filepath.SplitList(paths)
+	visited := make(map[string]struct{}, len(libPaths))
+	for _, path := range libPaths {
 		abspath, err := filepath.Abs(path)
 		if err != nil {
 			slog.Error("failed to get absolute path", "error", err)
@@ -104,6 +104,12 @@ var OnceLoad = sync.OnceFunc(func() {
 	slog.Info("system", "", system{})
 })

+var libPaths []string
+
+func LibPaths() []string {
+	return libPaths
+}
+
 type system struct{}

 func (system) LogValue() slog.Value {

--- a/ml/backend/ggml/ggml/src/mem_hip.cpp
+++ b/ml/backend/ggml/ggml/src/mem_hip.cpp
--- a/ml/backend/ggml/ggml/src/mem_nvml.cpp
+++ b/ml/backend/ggml/ggml/src/mem_nvml.cpp
--- a/ml/device.go
+++ b/ml/device.go
--- a/ml/nn/pooling/pooling_test.go
+++ b/ml/nn/pooling/pooling_test.go
@@ -3,11 +3,9 @@ package pooling_test
 import (
 	"bytes"
 	"os"
-	"slices"
 	"testing"

 	"github.com/google/go-cmp/cmp"
-	"github.com/ollama/ollama/discover"
 	fsggml "github.com/ollama/ollama/fs/ggml"
 	"github.com/ollama/ollama/ml"
 	"github.com/ollama/ollama/ml/backend/ggml"
@@ -32,20 +30,7 @@ func setup(tb testing.TB, n int) ml.Backend {
 		tb.Fatal(err)
 	}

-	var gpuLayers ml.GPULayersList
-	if gpus := discover.GetGPUInfo(); len(gpus) > 0 {
-		gpuLayers = append(gpuLayers, ml.GPULayers{
-			ID: gpus[0].ID,
-			Layers: slices.Collect(func(yield func(int) bool) {
-				for i := range n {
-					if !yield(i) {
-						return
-					}
-				}
-			}),
-		})
-	}
-	b, err := ggml.New(f.Name(), ml.BackendParams{AllocMemory: true, GPULayers: gpuLayers})
+	b, err := ggml.New(f.Name(), ml.BackendParams{AllocMemory: true})
 	if err != nil {
 		tb.Fatal(err)
 	}

--- a/runner/llamarunner/runner.go
+++ b/runner/llamarunner/runner.go
--- a/runner/ollamarunner/runner.go
+++ b/runner/ollamarunner/runner.go
--- a/scripts/build_windows.ps1
+++ b/scripts/build_windows.ps1
--- a/server/routes.go
+++ b/server/routes.go
--- a/server/routes_debug_test.go
+++ b/server/routes_debug_test.go
--- a/server/routes_generate_test.go
+++ b/server/routes_generate_test.go
--- a/server/routes_harmony_streaming_test.go
+++ b/server/routes_harmony_streaming_test.go
--- a/server/sched.go
+++ b/server/sched.go
--- a/server/sched_test.go
+++ b/server/sched_test.go