diff --git a/docs/backend/OPENVINO.md b/docs/backend/OPENVINO.md index 89d00968c549..b877f1434b7d 100644 --- a/docs/backend/OPENVINO.md +++ b/docs/backend/OPENVINO.md @@ -625,7 +625,7 @@ $env:GGML_OPENVINO_DEVICE = "NPU" build\ReleaseOV\bin\llama-cli.exe -m "C:\models\Llama-3.2-1B-Instruct-Q4_K_M.gguf" -c 512 ``` > [!NOTE] -> On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html) for more details. +> On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. A device that is not available is an error (no fallback to CPU), and the error message lists the available OpenVINO devices with their names. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html) for more details. ### 5. Docker Build @@ -713,7 +713,7 @@ Boolean flags follow a uniform convention: set to a **positive integer** (e.g. ` | Variable | Type | Default | Description | |-----------------------------------|-----------|------------|-------------------------------------------------------------------------------------------------------------| -| `GGML_OPENVINO_DEVICE` | String | `CPU` | Specify the target device (CPU, GPU, NPU). On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html). When set to **NPU**, static compilation mode is enabled for optimal performance. | +| `GGML_OPENVINO_DEVICE` | String | `CPU` | Specify the target device (CPU, GPU, NPU). On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. A device that is not available is an error (no fallback to CPU), and the error message lists the available OpenVINO devices with their names. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html). When set to **NPU**, static compilation mode is enabled for optimal performance. | | `GGML_OPENVINO_CACHE_DIR` | String | `not set` | Directory for OpenVINO's separate plugin cache. On NPU, this sets `NPUW_CACHE_DIR`. | | `GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR` | String | `not set` | Directory for standalone compiled blobs with weights. Dynamic CPU/GPU graphs can import matching blobs on later runs. | | `GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY` | Boolean | `0` | Require an existing compiled blob and skip weight uploads and compilation. Requires Linux or Windows mmap loading and a full dynamic CPU/GPU graph on OpenVINO. | diff --git a/ggml/src/ggml-openvino/ggml-openvino-extra.cpp b/ggml/src/ggml-openvino/ggml-openvino-extra.cpp index f420e0adfcdb..27414ebb7e9f 100644 --- a/ggml/src/ggml-openvino/ggml-openvino-extra.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino-extra.cpp @@ -4,11 +4,13 @@ #include "ggml.h" #include "model-cache.h" +#include #include #include #include #include #include +#include #include ov::Core & ov_singleton_core() { @@ -16,14 +18,88 @@ ov::Core & ov_singleton_core() { return core; } +static bool has_prefix(const std::string & s, const std::string & prefix) { + return s.size() >= prefix.size() && std::equal(prefix.begin(), prefix.end(), s.begin()); +} + +static bool is_virtual_routing_device(const std::string & device_name) { + return has_prefix(device_name, "AUTO") || has_prefix(device_name, "MULTI") || has_prefix(device_name, "HETERO"); +} + +static std::vector ov_enumerate_devices() { + std::vector result; + + for (const auto & device : ov_singleton_core().get_available_devices()) { + if (!is_virtual_routing_device(device)) { + result.push_back(device); + } + } + + if (result.empty()) { + result.push_back("CPU"); + } + + std::sort(result.begin(), result.end()); + result.erase(std::unique(result.begin(), result.end()), result.end()); + return result; +} + +std::string ggml_openvino_get_device_description(const std::string & device_name) { + std::string description = device_name; + try { + description = ov_singleton_core().get_property(device_name, ov::device::full_name); + } catch (...) { + return device_name; + } + + if (has_prefix(device_name, "NPU")) { + try { + const std::string arch = ov_singleton_core().get_property(device_name, "DEVICE_ARCHITECTURE").as(); + if (!arch.empty()) { + description += " (NPU " + arch + ")"; + } + } catch (...) { + } + } + + return description; +} + +// requested: GGML_OPENVINO_DEVICE, nullptr if unset. available_devices is never empty (see ov_enumerate_devices) +static std::string resolve_openvino_device_name(const std::vector & available_devices, + const char * requested) { + auto available = [&](const std::string & name) { + return std::find(available_devices.begin(), available_devices.end(), name) != available_devices.end(); + }; + if (requested == nullptr) { + return available("CPU") ? "CPU" : available_devices.front(); + } + if (!available(requested)) { + // No fallback to CPU (easy to miss) and no GPU -> GPU.0 alias (with iGPU + dGPU, GPU.0 is often the + // wrong one). List the devices here: --list-devices initializes this backend and would abort too. + std::string list; + for (const std::string & name : available_devices) { + list += "\n " + name + ": " + ggml_openvino_get_device_description(name); + } + GGML_ABORT("GGML OpenVINO Backend: GGML_OPENVINO_DEVICE=%s is not available. " + "Set it to one of the available OpenVINO devices:%s", + requested, list.c_str()); + } + return requested; +} + // ===================================================== // Device Configuration Implementations // ===================================================== void ggml_openvino_device_config::init() { + static std::mutex mutex; + std::lock_guard lock(mutex); if (initialized) { return; } + // Set up front: a failed OpenCL setup below is not retried on every call + initialized = true; // All recognized GGML_OPENVINO_* env vars. Their values are cached here // once at backend init time and read back via ggml_openvino_getenv_str() @@ -76,18 +152,14 @@ void ggml_openvino_device_config::init() { } } - device_name = ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE", "CPU"); - auto available_devices = ov_singleton_core().get_available_devices(); - if (std::find(available_devices.begin(), available_devices.end(), device_name) == available_devices.end()) { - GGML_LOG_WARN("GGML OpenVINO Backend: device %s is not available, fallback to CPU\n", device_name.c_str()); - device_name = "CPU"; - } - is_npu = (device_name == "NPU"); + available_devices = ov_enumerate_devices(); + device_name = resolve_openvino_device_name(available_devices, ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE")); + is_npu = has_prefix(device_name, "NPU"); ggml_openvino_model_cache_init(); const char * cache_dir = ggml_openvino_getenv_str("GGML_OPENVINO_CACHE_DIR"); - if (device_name == "NPU") { + if (has_prefix(device_name, "NPU")) { compile_config = { {"NPU_COMPILER_DYNAMIC_QUANTIZATION", "YES" }, {"NPU_USE_NPUW", "YES" }, @@ -118,18 +190,20 @@ void ggml_openvino_device_config::init() { } // Initialize remote context with queue sharing for GPU - if (device_name == "GPU") { - // Create OpenCL context and queue - cl_int err; - cl_platform_id platform; - err = clGetPlatformIDs(1, &platform, nullptr); - if (err != CL_SUCCESS) { - GGML_LOG_ERROR("Failed to get OpenCL platform: %d\n", err); + if (has_prefix(device_name, "GPU")) { + // Use the OpenCL context OpenVINO created for this device, so GPU.N gets its own device + cl_context cl_ctx; + try { + auto ov_ctx = ov_singleton_core().get_default_context(device_name).as(); + cl_ctx = ov_ctx.get(); + } catch (const std::exception & e) { + GGML_LOG_ERROR("Failed to get OpenCL context for %s: %s\n", device_name.c_str(), e.what()); return; } + cl_int err; cl_device_id cl_device; - err = clGetDeviceIDs(platform, CL_DEVICE_TYPE_GPU, 1, &cl_device, nullptr); + err = clGetContextInfo(cl_ctx, CL_CONTEXT_DEVICES, sizeof(cl_device), &cl_device, nullptr); if (err != CL_SUCCESS) { GGML_LOG_ERROR("Failed to get OpenCL device: %d\n", err); return; @@ -145,12 +219,6 @@ void ggml_openvino_device_config::init() { GGML_LOG_WARN("Failed to get OpenCL max allocation size: %d\n", err); } - cl_context cl_ctx = clCreateContext(nullptr, 1, &cl_device, nullptr, nullptr, &err); - if (err != CL_SUCCESS) { - GGML_LOG_ERROR("Failed to create OpenCL context: %d\n", err); - return; - } - const cl_queue_properties profiling_properties[] = { CL_QUEUE_PROPERTIES, CL_QUEUE_PROFILING_ENABLE, @@ -161,21 +229,15 @@ void ggml_openvino_device_config::init() { cl_queue = clCreateCommandQueueWithProperties(cl_ctx, cl_device, queue_properties, &err); if (err != CL_SUCCESS) { GGML_LOG_ERROR("Failed to create OpenCL command queue: %d\n", err); - clReleaseContext(cl_ctx); return; } // Create OpenVINO remote context with queue sharing remote_context = ov::intel_gpu::ocl::ClContext(ov_singleton_core(), cl_queue); - - // Release the context (queue keeps a reference) - clReleaseContext(cl_ctx); - } else if (device_name == "NPU") { + } else if (has_prefix(device_name, "NPU")) { // remote tensor is not used for NPU yet // remote_context = ov_singleton_core().get_default_context(device_name); } - - initialized = true; } ggml_openvino_device_config::~ggml_openvino_device_config() { @@ -201,6 +263,12 @@ const std::string & ggml_openvino_get_device_name() { return ggml_openvino_get_device_config().device_name; } +std::vector ggml_openvino_get_available_devices() { + auto & config = ggml_openvino_get_device_config(); + config.init(); + return config.available_devices; +} + // Get the value of a GGML_OPENVINO_* env var as a string. Returns // default_value when the var is unset or set to an empty string. const char * ggml_openvino_getenv_str(const char * var, const char * default_value) { @@ -226,12 +294,12 @@ bool ggml_openvino_reduce_compile_mem_enabled() { return ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0; } -bool ggml_openvino_release_weights_enabled(const std::string & device) { +bool ggml_openvino_release_weights_enabled() { const char * release_weights = ggml_openvino_getenv_str("GGML_OPENVINO_RELEASE_WEIGHTS"); if (release_weights != nullptr) { - return device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_RELEASE_WEIGHTS") != 0; + return ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_RELEASE_WEIGHTS") != 0; } - return device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0; + return ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0; } // Check if running on NPU @@ -239,6 +307,10 @@ bool ggml_openvino_is_npu() { return ggml_openvino_get_device_config().is_npu; } +bool ggml_openvino_is_gpu() { + return has_prefix(ggml_openvino_get_device_name(), "GPU"); +} + size_t ggml_openvino_max_alloc_size() { return ggml_openvino_get_device_config().max_alloc_size; } @@ -358,7 +430,7 @@ std::optional ggml_openvino_get_requant_type(const ggml_tensor * // already requantize to per-channel Q8_0_C (grouped=0). Sending these to grouped 4 bit // avoids the broken layout and restores correct output. // Opt out with GGML_OPENVINO_REQUANT_KQUANT=native. - if (ggml_openvino_get_device_name() == "GPU" && !is_opt("native")) { + if (ggml_openvino_is_gpu() && !is_opt("native")) { return ExtraQuantType::Q4_0_64; } } @@ -590,12 +662,11 @@ ggml_openvino_tensor_extra * ggml_openvino_create_tensor_extra(const ggml_tensor return nullptr; } - const auto & device_name = ggml_openvino_get_device_name(); auto remote_context = ggml_openvino_get_remote_context(); std::shared_ptr ov_tensor; if (is_remote) { - GGML_ASSERT(device_name == "GPU"); + GGML_ASSERT(ggml_openvino_is_gpu()); auto gpu_context = remote_context->as(); auto usm_tensor = gpu_context.create_tensor(element_type, shape, tensor->data); ov_tensor = std::make_shared(std::move(usm_tensor)); diff --git a/ggml/src/ggml-openvino/ggml-openvino-extra.h b/ggml/src/ggml-openvino/ggml-openvino-extra.h index b41153352770..8521e978d2b7 100644 --- a/ggml/src/ggml-openvino/ggml-openvino-extra.h +++ b/ggml/src/ggml-openvino/ggml-openvino-extra.h @@ -63,6 +63,7 @@ clEnqueueMemcpyINTEL_fn ggml_openvino_get_clEnqueueMemcpyINTEL(); struct ggml_openvino_device_config { std::string device_name = "CPU"; + std::vector available_devices; bool is_npu = false; bool initialized = false; std::optional remote_context; @@ -84,6 +85,12 @@ void ggml_openvino_init_device_config(); // Get the device name const std::string & ggml_openvino_get_device_name(); +// Get all available physical OpenVINO devices +std::vector ggml_openvino_get_available_devices(); + +// Human-readable device name, e.g. "Intel(R) AI Boost (NPU 4000)"; the device id if unavailable +std::string ggml_openvino_get_device_description(const std::string & device_name); + // Environment variable accessors. All GGML_OPENVINO_* env vars are read once // during backend init and cached on the device config; consumers must go // through these helpers (never call ::getenv directly) so behavior stays @@ -103,11 +110,14 @@ int ggml_openvino_getenv_int(const char * var, int default_value = 0); // Memory optimization toggles. GGML_OPENVINO_MEMORY_OPTIMIZE is an umbrella // switch; the fine-grained env vars still override it when explicitly set. bool ggml_openvino_reduce_compile_mem_enabled(); -bool ggml_openvino_release_weights_enabled(const std::string & device); +bool ggml_openvino_release_weights_enabled(); // Check if running on NPU bool ggml_openvino_is_npu(); +// Check if running on a GPU (GPU, GPU.0, GPU.1, ...) +bool ggml_openvino_is_gpu(); + // Largest single memory object the device can allocate, SIZE_MAX when there is no known limit size_t ggml_openvino_max_alloc_size(); diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp index 9b94bcc27c9c..e9b402bf4592 100644 --- a/ggml/src/ggml-openvino/ggml-openvino.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino.cpp @@ -25,7 +25,10 @@ #include #include #include +#include #include +#include +#include #include #include #include @@ -102,7 +105,7 @@ struct ggml_backend_openvino_buffer_context { const auto & device_name = ggml_openvino_get_device_name(); if (is_remote) { - GGML_ASSERT(device_name == "GPU"); + GGML_ASSERT(ggml_openvino_is_gpu()); auto remote_context = ggml_openvino_get_remote_context(); auto gpu_context = remote_context->as(); ov::intel_gpu::ocl::USMTensor usm_tensor = @@ -318,7 +321,7 @@ static enum ggml_status ggml_backend_openvino_buffer_init_tensor(ggml_backend_bu ggml_backend_openvino_buffer_context * ctx = (ggml_backend_openvino_buffer_context *) buffer->context; // Put kvcache on device memory for GPU (NPU memory is too small even for kvcache) - if (strncmp(tensor->name, "cache_", 6) == 0 && !ctx->is_remote && ggml_openvino_get_device_name() == "GPU" && + if (strncmp(tensor->name, "cache_", 6) == 0 && !ctx->is_remote && ggml_openvino_is_gpu() && !is_stateful_enabled()) { GGML_ASSERT(ctx->tensor_extras.empty()); auto device = ctx->device; @@ -873,7 +876,7 @@ static const ggml_backend_i ggml_backend_openvino_interface = { }; int ggml_backend_openvino_get_device_count() { - return 1; + return (int) ggml_openvino_get_available_devices().size(); } static ggml_guid_t ggml_backend_openvino_guid(void) { @@ -932,10 +935,122 @@ namespace { struct ggml_backend_openvino_device_context { int device; std::string name; + std::string ov_name; // OpenVINO device id: CPU, GPU, GPU.1, NPU, ... std::string description; + size_t total_memory; }; } +static bool ov_device_has_prefix(const std::string & s, const std::string & prefix) { + return s.size() >= prefix.size() && std::equal(prefix.begin(), prefix.end(), s.begin()); +} + +static bool ov_try_get_size_t_property(const std::string & device, const std::string & property, size_t & out) { + try { + const ov::Any value = ov_singleton_core().get_property(device, property); + if (value.is()) { + out = value.as(); + return true; + } + if (value.is()) { + out = (size_t) value.as(); + return true; + } + if (value.is()) { + out = (size_t) value.as(); + return true; + } + if (value.is()) { + const int64_t v = value.as(); + if (v >= 0) { + out = (size_t) v; + return true; + } + } + } catch (...) { + } + return false; +} + +// System memory available to new allocations (MemAvailable on Linux), SIZE_MAX if unknown +static size_t ov_system_available_memory() { +#ifdef _WIN32 + MEMORYSTATUSEX status; + status.dwLength = sizeof(status); + if (GlobalMemoryStatusEx(&status)) { + return (size_t) status.ullAvailPhys; + } +#else + if (FILE * f = fopen("/proc/meminfo", "r")) { + char line[256]; + unsigned long long kb = 0; + bool found = false; + while (!found && fgets(line, sizeof(line), f)) { + found = sscanf(line, "MemAvailable: %llu kB", &kb) == 1; + } + fclose(f); + if (found) { + return (size_t) std::min(kb * 1024, SIZE_MAX); + } + } +#endif + return SIZE_MAX; +} + +// iGPU and NPU allocate from system RAM, so their free memory can't exceed what the OS has available +static bool ov_device_shares_system_memory(const std::string & device) { + if (ov_device_has_prefix(device, "NPU")) { + return true; + } + if (!ov_device_has_prefix(device, "GPU")) { + return false; + } + try { + return ov_singleton_core().get_property(device, ov::device::type) == ov::device::Type::INTEGRATED; + } catch (...) { + return false; + } +} + +// usm_host / usm_shared allocations live in system RAM on a discrete GPU +static bool ov_gpu_stat_is_host_memory(const std::string & key) { + return key == "usm_host" || key == "usm_shared"; +} + +static bool ov_try_get_gpu_used_memory(const std::string & device, size_t & out) { + out = 0; + try { + const ov::Any stats_any = ov_singleton_core().get_property(device, "GPU_MEMORY_STATISTICS"); + if (stats_any.is>()) { + const auto stats = stats_any.as>(); + for (const auto & kv : stats) { + if (!ov_gpu_stat_is_host_memory(kv.first)) { + out += (size_t) kv.second; + } + } + return true; + } + if (stats_any.is()) { + const auto stats = stats_any.as(); + for (const auto & kv : stats) { + if (ov_gpu_stat_is_host_memory(kv.first)) { + continue; + } + if (kv.second.is()) { + out += kv.second.as(); + } else if (kv.second.is()) { + out += (size_t) kv.second.as(); + } else if (kv.second.is()) { + out += (size_t) kv.second.as(); + } + } + return true; + } + } catch (...) { + } + return false; +} + static const char * ggml_backend_openvino_device_get_name(ggml_backend_dev_t dev) { ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context; return ctx->name.c_str(); @@ -947,27 +1062,45 @@ static const char * ggml_backend_openvino_device_get_description(ggml_backend_de } static void ggml_backend_openvino_device_get_memory(ggml_backend_dev_t dev, size_t * free, size_t * total) { + ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context; + + // total_memory is only set for GPU/NPU; used = this process's OpenVINO allocations on the device + size_t used = 0; + const bool known = ctx->total_memory > 0 && + (ov_device_has_prefix(ctx->ov_name, "GPU") ? + ov_try_get_gpu_used_memory(ctx->ov_name, used) : + ov_try_get_size_t_property(ctx->ov_name, "NPU_DEVICE_ALLOC_MEM_SIZE", used)); + if (known) { + *total = ctx->total_memory; + *free = (used >= *total) ? 0 : (*total - used); + } else { + // CPU, or a plugin without memory properties: report system memory #ifdef _WIN32 - MEMORYSTATUSEX status; - status.dwLength = sizeof(status); - GlobalMemoryStatusEx(&status); - *total = status.ullTotalPhys; - *free = status.ullAvailPhys; + MEMORYSTATUSEX status; + status.dwLength = sizeof(status); + GlobalMemoryStatusEx(&status); + *total = status.ullTotalPhys; + *free = status.ullAvailPhys; #else - long pages = sysconf(_SC_PHYS_PAGES); - long page_size = sysconf(_SC_PAGE_SIZE); - *total = pages * page_size; + long pages = sysconf(_SC_PHYS_PAGES); + long page_size = sysconf(_SC_PAGE_SIZE); + *total = pages * page_size; - // "free" system memory is ill-defined, for practical purposes assume that all of it is free: - *free = *total; + // "free" system memory is ill-defined, for practical purposes assume that all of it is free: + *free = *total; #endif // _WIN32 + } - GGML_UNUSED(dev); + if (ov_device_shares_system_memory(ctx->ov_name)) { + *free = std::min(*free, ov_system_available_memory()); + } } static enum ggml_backend_dev_type ggml_backend_openvino_device_get_type(ggml_backend_dev_t dev) { - GGML_UNUSED(dev); - return GGML_BACKEND_DEVICE_TYPE_GPU; + ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context; + // Only the device selected by GGML_OPENVINO_DEVICE is offered for offload. The others are + // registered for discovery (--list-devices) only; llama.cpp skips IGPU devices when a GPU exists. + return ctx->ov_name == ggml_openvino_get_device_name() ? GGML_BACKEND_DEVICE_TYPE_GPU : GGML_BACKEND_DEVICE_TYPE_IGPU; } static void ggml_backend_openvino_device_get_props(ggml_backend_dev_t dev, ggml_backend_dev_props * props) { @@ -988,6 +1121,12 @@ static void ggml_backend_openvino_device_get_props(ggml_backend_dev_t dev, ggml_ static ggml_backend_t ggml_backend_openvino_device_init(ggml_backend_dev_t dev, const char * params) { GGML_UNUSED(params); ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context; + if (ctx->ov_name != ggml_openvino_get_device_name()) { + // Not an error: test-backend-ops initializes every device + GGML_LOG_WARN("%s: %s (OpenVINO %s) is not the selected device, no ops will run on it; " + "set GGML_OPENVINO_DEVICE=%s to use it\n", + __func__, ctx->name.c_str(), ctx->ov_name.c_str(), ctx->ov_name.c_str()); + } return ggml_backend_openvino_init(ctx->device); } @@ -1207,7 +1346,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { if (op->type == GGML_TYPE_I64) { return {false, "CONCAT with I64 type is not supported"}; } - if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_BF16 && has_view_op_input(op)) { + if (ggml_openvino_is_gpu() && op->type == GGML_TYPE_BF16 && has_view_op_input(op)) { return {false, "CONCAT with BF16 type and VIEW input is not supported on GPU"}; } break; @@ -1235,7 +1374,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { op->src[0]->view_src != nullptr && op->src[0]->view_offs != 0) { return {false, "GET_ROWS with a nonzero quantized src0 view offset is not supported"}; } - if (op->op == GGML_OP_GET_ROWS && ggml_openvino_get_device_name() == "GPU" && + if (op->op == GGML_OP_GET_ROWS && ggml_openvino_is_gpu() && op->src[0]->type == GGML_TYPE_BF16) { return {false, "GET_ROWS with BF16 src0 is not supported on GPU"}; } @@ -1288,22 +1427,21 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { // The GPU plugin can fuse broadcast DIV into the preceding FFN GEMM path // and produce infs for per-channel scale vectors. Keep those DIVs on CPU // until the fused GPU kernel is reliable. (falied case llama-arch-test mpt) - if (ggml_openvino_get_device_name() == "GPU" && op->src[1]->ne[0] == op->ne[0] && + if (ggml_openvino_is_gpu() && op->src[1]->ne[0] == op->ne[0] && op->src[1]->ne[1] == 1 && op->src[1]->ne[2] == 1 && op->src[1]->ne[3] == 1) { return {false, "DIV per-channel scale broadcast is not supported on GPU"}; } break; } case GGML_OP_POOL_2D: { - const auto& name = ggml_openvino_get_device_name(); - if (name == "GPU") { + if (ggml_openvino_is_gpu()) { const int32_t * params = op->op_params; const int k0 = params[1]; const int k1 = params[2]; const int p0 = params[5]; const int p1 = params[6]; if ((p0 > 0 || p1 > 0) && (k0 < 3 || k1 < 3)) { - return {false, "POOL_2D with padding and kernel size < 3 is not supported on " + name}; + return {false, "POOL_2D with padding and kernel size < 3 is not supported on " + ggml_openvino_get_device_name()}; } } break; @@ -1345,7 +1483,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { break; } case GGML_OP_PERMUTE: { - if (op->type == GGML_TYPE_BF16 && ggml_openvino_get_device_name() == "GPU") { + if (op->type == GGML_TYPE_BF16 && ggml_openvino_is_gpu()) { return {false, "PERMUTE with BF16 type is not supported on GPU"}; } break; @@ -1354,7 +1492,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { if (op->src[0]->type != GGML_TYPE_BF16 && op->src[1]->type == GGML_TYPE_BF16) { return {false, "CPY with BF16 src[1] type is not supported"}; } - if (ggml_openvino_get_device_name() == "NPU" && (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) { + if (ggml_openvino_is_npu() && (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) { return {false, "CPY with BF16 is not supported is not supported on NPU"}; } // CPY to a quantized destination (e.g. f32 -> q4_0) is numerically unstable with OpenVINO backend. @@ -1379,13 +1517,13 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { break; } case GGML_OP_MUL_MAT: { - if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[1] != nullptr && + if (ggml_openvino_is_gpu() && op->src[0] != nullptr && op->src[1] != nullptr && ggml_is_quantized(op->src[0]->type) && strcmp(op->src[0]->name, "a") == 0 && strcmp(op->src[1]->name, "b") == 0 && op->src[0]->ne[1] == 1 && op->src[1]->ne[1] == 64 && op->src[0]->ne[0] == 256 && op->src[1]->ne[0] == 256) { return {false, "MUL_MAT quantized benchmark test case on GPU is not supported"}; } - if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_F32 && op->ne[0] == 1 && op->ne[1] == 1 && + if (ggml_openvino_is_gpu() && op->type == GGML_TYPE_F32 && op->ne[0] == 1 && op->ne[1] == 1 && (op->src[0]->buffer == nullptr || op->src[0]->buffer->usage != GGML_BACKEND_BUFFER_USAGE_WEIGHTS)) { return {false, "MUL_MAT scalar dot product with non-weight src[0] on GPU is not supported"}; } @@ -1405,7 +1543,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { return {false, "MUL_MAT_ID with single-expert or empty ne[2] <= 1 (ne[2]=" + std::to_string(op->src[0]->ne[2]) + ") is not supported"}; } - if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && !ggml_is_quantized(op->src[0]->type)) { + if (ggml_openvino_is_gpu() && op->src[0] != nullptr && !ggml_is_quantized(op->src[0]->type)) { return {false, "MUL_MAT_ID with non-quantized weights on GPU is not supported"}; } // The GPU plugin's GatherMatmul returns wrong values for the layouts test-backend-ops @@ -1414,12 +1552,12 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { // The same graph is correct on the CPU plugin, and correct on GPU for every real model, // which always feeds experts from a bound tensor buffer. Standalone op-test tensors have // no buffer at all, so use that to exclude them and let the scheduler run them on CPU. - if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->buffer == nullptr) { + if (ggml_openvino_is_gpu() && op->src[0] != nullptr && op->src[0]->buffer == nullptr) { return {false, "MUL_MAT_ID with unbound expert tensors on GPU is not supported"}; } // Only MXFP4 still needs the large-temporary guard; every other quantized type goes // through GatherMatmul, which never materializes the selected expert weights. - if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->type == GGML_TYPE_MXFP4 && + if (ggml_openvino_is_gpu() && op->src[0] != nullptr && op->src[0]->type == GGML_TYPE_MXFP4 && mul_mat_id_requires_large_tmp(op)) { return {false, "MUL_MAT_ID with MXFP4 weights requires large temporary on GPU"}; } @@ -1472,7 +1610,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { break; } case GGML_OP_REPEAT: { - if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_BF16) { + if (ggml_openvino_is_gpu() && op->type == GGML_TYPE_BF16) { return {false, "REPEAT with BF16 type is not supported on GPU"}; } break; @@ -1480,7 +1618,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { case GGML_OP_GATED_DELTA_NET: { // enable after https://github.com/openvinotoolkit/openvino/pull/35917 is included in OV release // return true; - // if (ggml_openvino_get_device_name() == "GPU" && op->src[0]->ne[2] > 1) { + // if (ggml_openvino_is_gpu() && op->src[0]->ne[2] > 1) { // // CVS-186471 // return true; // } @@ -1520,6 +1658,11 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { static ggml_openvino_op_support ggml_backend_openvino_device_supports_op_impl(ggml_backend_dev_t dev, const ggml_tensor * op) { GGML_ASSERT(dev->reg != nullptr); + ggml_backend_openvino_device_context * dev_ctx = (ggml_backend_openvino_device_context *) dev->context; + if (dev_ctx->ov_name != ggml_openvino_get_device_name()) { + return {false, "device is not the selected OpenVINO device"}; + } + static std::unordered_set supported_types{ GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16, GGML_TYPE_I64, GGML_TYPE_I32, GGML_TYPE_Q4_0, GGML_TYPE_Q4_1, GGML_TYPE_Q4_K, GGML_TYPE_Q5_1, GGML_TYPE_Q5_K, GGML_TYPE_Q8_0, GGML_TYPE_Q6_K, @@ -1707,15 +1850,24 @@ GGML_BACKEND_API ggml_backend_reg_t ggml_backend_openvino_reg(void) { std::lock_guard lock(mutex); if (!initialized) { ggml_openvino_init(); + const std::vector openvino_devices = ggml_openvino_get_available_devices(); ggml_backend_openvino_reg_context * ctx = new ggml_backend_openvino_reg_context; for (int i = 0; i < ggml_backend_openvino_get_device_count(); i++) { ggml_backend_openvino_device_context * dev_ctx = new ggml_backend_openvino_device_context; dev_ctx->device = i; + // Not the raw OpenVINO id: "CPU" would shadow the ggml CPU backend in ggml_backend_dev_by_name dev_ctx->name = GGML_OPENVINO_NAME + std::to_string(i); - - dev_ctx->description = ov::get_openvino_version().description; + dev_ctx->ov_name = openvino_devices[i]; + dev_ctx->description = + ggml_openvino_get_device_description(dev_ctx->ov_name) + " (OpenVINO " + dev_ctx->ov_name + ")"; + dev_ctx->total_memory = 0; + if (ov_device_has_prefix(dev_ctx->ov_name, "GPU")) { + ov_try_get_size_t_property(dev_ctx->ov_name, "GPU_DEVICE_TOTAL_MEM_SIZE", dev_ctx->total_memory); + } else if (ov_device_has_prefix(dev_ctx->ov_name, "NPU")) { + ov_try_get_size_t_property(dev_ctx->ov_name, "NPU_DEVICE_TOTAL_MEM_SIZE", dev_ctx->total_memory); + } ggml_backend_dev_t dev = new ggml_backend_device{/* .interface = */ ggml_backend_openvino_device_interface, diff --git a/ggml/src/ggml-openvino/model-cache.cpp b/ggml/src/ggml-openvino/model-cache.cpp index 0c67e2779fc1..c8c5fb5adb8e 100644 --- a/ggml/src/ggml-openvino/model-cache.cpp +++ b/ggml/src/ggml-openvino/model-cache.cpp @@ -354,7 +354,7 @@ uint64_t ggml_openvino_model_fingerprint(const ggml_cgraph * cgraph, h = fnv1a(h, "GGML_OPENVINO_DEBUG_NODE", sizeof("GGML_OPENVINO_DEBUG_NODE")); h = fnv1a(h, debug_nodes, strlen(debug_nodes) + 1); } - if (device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_MOE_OP", 1) == 0) { + if (ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_MOE_OP", 1) == 0) { h = fnv1a(h, "GGML_OPENVINO_MOE_OP=0", sizeof("GGML_OPENVINO_MOE_OP=0")); } diff --git a/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp b/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp index b06d01dcace0..44c6dbc03ddb 100644 --- a/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp +++ b/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp @@ -126,8 +126,7 @@ OutputVector translate_flash_attn_ext(const NodeContext & context) { if (env != nullptr) { return ggml_openvino_getenv_int("GGML_OPENVINO_MANUAL_GQA_ATTN") > 0; } - const char * dev = ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE"); - return dev != nullptr && std::string(dev) == "GPU"; + return ggml_openvino_is_gpu(); }(); const bool use_manual_gqa_attention = manual_gqa_enabled && factor > 1 && num_heads_kv > 1 && !context.is_stateful(); diff --git a/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp b/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp index 1f9b1f6baf47..9cedf68116bf 100644 --- a/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp +++ b/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp @@ -118,7 +118,7 @@ OutputVector translate_gated_delta_net(const NodeContext & context) { Output raw_q, raw_k; float q_eps = 1e-6f, k_eps = 1e-6f; - const bool fuse_qk_l2norm = ggml_openvino_get_device_name() == "GPU" && + const bool fuse_qk_l2norm = ggml_openvino_is_gpu() && match_gdn_l2_norm(q, raw_q, q_eps) && match_gdn_l2_norm(k, raw_k, k_eps); if (fuse_qk_l2norm) { // Keep head tiling below; GDN applies normalization per head and keeps its attention scale. diff --git a/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp b/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp index 9b708eeb46db..4f1979b1047c 100644 --- a/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp +++ b/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp @@ -204,7 +204,7 @@ OutputVector translate_mul_mat_id(const NodeContext & context) { } const auto output_type = context.get_output_type(); - const auto activations_type = ggml_openvino_get_device_name() == "GPU" ? ov::element::f16 : ov::element::f32; + const auto activations_type = ggml_openvino_is_gpu() ? ov::element::f16 : ov::element::f32; if (activations.get_element_type() != activations_type) { activations = std::make_shared(activations, activations_type); } diff --git a/ggml/src/ggml-openvino/openvino/translate_session.cpp b/ggml/src/ggml-openvino/openvino/translate_session.cpp index 9867ea544830..87cfaf90a1f1 100644 --- a/ggml/src/ggml-openvino/openvino/translate_session.cpp +++ b/ggml/src/ggml-openvino/openvino/translate_session.cpp @@ -488,7 +488,7 @@ std::shared_ptr TranslateSession::apply_transformations(std::shared_ptr(); manager.register_pass(); diff --git a/ggml/src/ggml-openvino/utils.cpp b/ggml/src/ggml-openvino/utils.cpp index 858152420153..9e3e2dd301f7 100644 --- a/ggml/src/ggml-openvino/utils.cpp +++ b/ggml/src/ggml-openvino/utils.cpp @@ -108,11 +108,11 @@ std::optional try_make_kv_sliced_tensor(const std::shared_ptrdata); } -static uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, bool stateful, bool recurrent) { +static uint64_t ggml_openvino_model_cache_extra_cfg(bool stateful, bool recurrent) { const char * manual_gqa_env = ggml_openvino_getenv_str("GGML_OPENVINO_MANUAL_GQA_ATTN"); const bool manual_gqa_enabled = manual_gqa_env != nullptr ? ggml_openvino_getenv_int("GGML_OPENVINO_MANUAL_GQA_ATTN") > 0 : - device == "GPU"; + ggml_openvino_is_gpu(); uint64_t extra_cfg = 1; // Graph-ordinal port names (invalidate older disk-cache blobs). extra_cfg = extra_cfg * 131 + (stateful ? (recurrent ? 2u : 1u) : 0u); @@ -1034,7 +1034,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, const std::share std::string blob_path, manifest_path; if (!model_cache_dir.empty() && !model_is_splitted) { const uint64_t extra_cfg = - ggml_openvino_model_cache_extra_cfg(device, r_ctx->stateful, stateful_recurrent); + ggml_openvino_model_cache_extra_cfg(r_ctx->stateful, stateful_recurrent); model_fp = ggml_openvino_model_fingerprint(cgraph, device, /*fa=*/true, m_params.rope_params, 16, extra_cfg, dynamic_graph_signature(cgraph, *ggml_decoder, m_params)); @@ -1377,7 +1377,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, const std::share // be reading host weights during conversion/compilation. Pin the shared compiled // models across backend teardown; a later context can create its own request without // reading the dropped pages. A new, uncached graph still fails fast above. - if (cache_hit && ggml_openvino_release_weights_enabled(device)) { + if (cache_hit && ggml_openvino_release_weights_enabled()) { std::lock_guard compile_lock(r_ctx->compiled_cache->mutex); if (!ggml_openvino_weight_buffers_released()) { ggml_openvino_release_weight_buffers();