Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions docs/backend/OPENVINO.md
Original file line number Diff line number Diff line change
Expand Up @@ -625,7 +625,7 @@ $env:GGML_OPENVINO_DEVICE = "NPU"
build\ReleaseOV\bin\llama-cli.exe -m "C:\models\Llama-3.2-1B-Instruct-Q4_K_M.gguf" -c 512
```
> [!NOTE]
> On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html) for more details.
> On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. A device that is not available is an error (no fallback to CPU), and the error message lists the available OpenVINO devices with their names. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html) for more details.

### 5. Docker Build

Expand Down Expand Up @@ -713,7 +713,7 @@ Boolean flags follow a uniform convention: set to a **positive integer** (e.g. `

| Variable | Type | Default | Description |
|-----------------------------------|-----------|------------|-------------------------------------------------------------------------------------------------------------|
| `GGML_OPENVINO_DEVICE` | String | `CPU` | Specify the target device (CPU, GPU, NPU). On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html). When set to **NPU**, static compilation mode is enabled for optimal performance. |
| `GGML_OPENVINO_DEVICE` | String | `CPU` | Specify the target device (CPU, GPU, NPU). On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. A device that is not available is an error (no fallback to CPU), and the error message lists the available OpenVINO devices with their names. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html). When set to **NPU**, static compilation mode is enabled for optimal performance. |
| `GGML_OPENVINO_CACHE_DIR` | String | `not set` | Directory for OpenVINO's separate plugin cache. On NPU, this sets `NPUW_CACHE_DIR`. |
| `GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR` | String | `not set` | Directory for standalone compiled blobs with weights. Dynamic CPU/GPU graphs can import matching blobs on later runs. |
| `GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY` | Boolean | `0` | Require an existing compiled blob and skip weight uploads and compilation. Requires Linux or Windows mmap loading and a full dynamic CPU/GPU graph on OpenVINO. |
Expand Down
141 changes: 106 additions & 35 deletions ggml/src/ggml-openvino/ggml-openvino-extra.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -4,26 +4,102 @@
#include "ggml.h"
#include "model-cache.h"

#include <algorithm>
#include <cstdlib>
#include <cstring>
#include <openvino/runtime/intel_gpu/ocl/ocl.hpp>
#include <openvino/runtime/intel_npu/level_zero/level_zero.hpp>
#include <openvino/runtime/properties.hpp>
#include <mutex>
#include <optional>

ov::Core & ov_singleton_core() {
static ov::Core core;
return core;
}

static bool has_prefix(const std::string & s, const std::string & prefix) {
return s.size() >= prefix.size() && std::equal(prefix.begin(), prefix.end(), s.begin());
}

static bool is_virtual_routing_device(const std::string & device_name) {
return has_prefix(device_name, "AUTO") || has_prefix(device_name, "MULTI") || has_prefix(device_name, "HETERO");
}

static std::vector<std::string> ov_enumerate_devices() {
std::vector<std::string> result;

for (const auto & device : ov_singleton_core().get_available_devices()) {
if (!is_virtual_routing_device(device)) {
result.push_back(device);
}
}

if (result.empty()) {
result.push_back("CPU");
}

std::sort(result.begin(), result.end());
result.erase(std::unique(result.begin(), result.end()), result.end());
return result;
}

std::string ggml_openvino_get_device_description(const std::string & device_name) {
std::string description = device_name;
try {
description = ov_singleton_core().get_property(device_name, ov::device::full_name);
} catch (...) {
return device_name;
}

if (has_prefix(device_name, "NPU")) {
try {
const std::string arch = ov_singleton_core().get_property(device_name, "DEVICE_ARCHITECTURE").as<std::string>();
if (!arch.empty()) {
description += " (NPU " + arch + ")";
}
} catch (...) {
}
}

return description;
}

// requested: GGML_OPENVINO_DEVICE, nullptr if unset. available_devices is never empty (see ov_enumerate_devices)
static std::string resolve_openvino_device_name(const std::vector<std::string> & available_devices,
const char * requested) {
auto available = [&](const std::string & name) {
return std::find(available_devices.begin(), available_devices.end(), name) != available_devices.end();
};
if (requested == nullptr) {
return available("CPU") ? "CPU" : available_devices.front();
}
if (!available(requested)) {
// No fallback to CPU (easy to miss) and no GPU -> GPU.0 alias (with iGPU + dGPU, GPU.0 is often the
// wrong one). List the devices here: --list-devices initializes this backend and would abort too.
std::string list;
for (const std::string & name : available_devices) {
list += "\n " + name + ": " + ggml_openvino_get_device_description(name);
}
GGML_ABORT("GGML OpenVINO Backend: GGML_OPENVINO_DEVICE=%s is not available. "
"Set it to one of the available OpenVINO devices:%s",
requested, list.c_str());
}
return requested;
}

// =====================================================
// Device Configuration Implementations
// =====================================================

void ggml_openvino_device_config::init() {
static std::mutex mutex;
std::lock_guard<std::mutex> lock(mutex);
if (initialized) {
return;
}
// Set up front: a failed OpenCL setup below is not retried on every call
initialized = true;

// All recognized GGML_OPENVINO_* env vars. Their values are cached here
// once at backend init time and read back via ggml_openvino_getenv_str()
Expand Down Expand Up @@ -76,18 +152,14 @@ void ggml_openvino_device_config::init() {
}
}

device_name = ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE", "CPU");
auto available_devices = ov_singleton_core().get_available_devices();
if (std::find(available_devices.begin(), available_devices.end(), device_name) == available_devices.end()) {
GGML_LOG_WARN("GGML OpenVINO Backend: device %s is not available, fallback to CPU\n", device_name.c_str());
device_name = "CPU";
}
is_npu = (device_name == "NPU");
available_devices = ov_enumerate_devices();
device_name = resolve_openvino_device_name(available_devices, ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE"));
is_npu = has_prefix(device_name, "NPU");

ggml_openvino_model_cache_init();

const char * cache_dir = ggml_openvino_getenv_str("GGML_OPENVINO_CACHE_DIR");
if (device_name == "NPU") {
if (has_prefix(device_name, "NPU")) {
compile_config = {
{"NPU_COMPILER_DYNAMIC_QUANTIZATION", "YES" },
{"NPU_USE_NPUW", "YES" },
Expand Down Expand Up @@ -118,18 +190,20 @@ void ggml_openvino_device_config::init() {
}

// Initialize remote context with queue sharing for GPU
if (device_name == "GPU") {
// Create OpenCL context and queue
cl_int err;
cl_platform_id platform;
err = clGetPlatformIDs(1, &platform, nullptr);
if (err != CL_SUCCESS) {
GGML_LOG_ERROR("Failed to get OpenCL platform: %d\n", err);
if (has_prefix(device_name, "GPU")) {
// Use the OpenCL context OpenVINO created for this device, so GPU.N gets its own device
cl_context cl_ctx;
try {
auto ov_ctx = ov_singleton_core().get_default_context(device_name).as<ov::intel_gpu::ocl::ClContext>();
cl_ctx = ov_ctx.get();
} catch (const std::exception & e) {
GGML_LOG_ERROR("Failed to get OpenCL context for %s: %s\n", device_name.c_str(), e.what());
return;
}

cl_int err;
cl_device_id cl_device;
err = clGetDeviceIDs(platform, CL_DEVICE_TYPE_GPU, 1, &cl_device, nullptr);
err = clGetContextInfo(cl_ctx, CL_CONTEXT_DEVICES, sizeof(cl_device), &cl_device, nullptr);
if (err != CL_SUCCESS) {
GGML_LOG_ERROR("Failed to get OpenCL device: %d\n", err);
return;
Expand All @@ -145,12 +219,6 @@ void ggml_openvino_device_config::init() {
GGML_LOG_WARN("Failed to get OpenCL max allocation size: %d\n", err);
}

cl_context cl_ctx = clCreateContext(nullptr, 1, &cl_device, nullptr, nullptr, &err);
if (err != CL_SUCCESS) {
GGML_LOG_ERROR("Failed to create OpenCL context: %d\n", err);
return;
}

const cl_queue_properties profiling_properties[] = {
CL_QUEUE_PROPERTIES,
CL_QUEUE_PROFILING_ENABLE,
Expand All @@ -161,21 +229,15 @@ void ggml_openvino_device_config::init() {
cl_queue = clCreateCommandQueueWithProperties(cl_ctx, cl_device, queue_properties, &err);
if (err != CL_SUCCESS) {
GGML_LOG_ERROR("Failed to create OpenCL command queue: %d\n", err);
clReleaseContext(cl_ctx);
return;
}

// Create OpenVINO remote context with queue sharing
remote_context = ov::intel_gpu::ocl::ClContext(ov_singleton_core(), cl_queue);

// Release the context (queue keeps a reference)
clReleaseContext(cl_ctx);
} else if (device_name == "NPU") {
} else if (has_prefix(device_name, "NPU")) {
// remote tensor is not used for NPU yet
// remote_context = ov_singleton_core().get_default_context(device_name);
}

initialized = true;
}

ggml_openvino_device_config::~ggml_openvino_device_config() {
Expand All @@ -201,6 +263,12 @@ const std::string & ggml_openvino_get_device_name() {
return ggml_openvino_get_device_config().device_name;
}

std::vector<std::string> ggml_openvino_get_available_devices() {
auto & config = ggml_openvino_get_device_config();
config.init();
return config.available_devices;
}

// Get the value of a GGML_OPENVINO_* env var as a string. Returns
// default_value when the var is unset or set to an empty string.
const char * ggml_openvino_getenv_str(const char * var, const char * default_value) {
Expand All @@ -226,19 +294,23 @@ bool ggml_openvino_reduce_compile_mem_enabled() {
return ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0;
}

bool ggml_openvino_release_weights_enabled(const std::string & device) {
bool ggml_openvino_release_weights_enabled() {
const char * release_weights = ggml_openvino_getenv_str("GGML_OPENVINO_RELEASE_WEIGHTS");
if (release_weights != nullptr) {
return device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_RELEASE_WEIGHTS") != 0;
return ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_RELEASE_WEIGHTS") != 0;
}
return device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0;
return ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0;
}

// Check if running on NPU
bool ggml_openvino_is_npu() {
return ggml_openvino_get_device_config().is_npu;
}

bool ggml_openvino_is_gpu() {
return has_prefix(ggml_openvino_get_device_name(), "GPU");
}

size_t ggml_openvino_max_alloc_size() {
return ggml_openvino_get_device_config().max_alloc_size;
}
Expand Down Expand Up @@ -358,7 +430,7 @@ std::optional<ExtraQuantType> ggml_openvino_get_requant_type(const ggml_tensor *
// already requantize to per-channel Q8_0_C (grouped=0). Sending these to grouped 4 bit
// avoids the broken layout and restores correct output.
// Opt out with GGML_OPENVINO_REQUANT_KQUANT=native.
if (ggml_openvino_get_device_name() == "GPU" && !is_opt("native")) {
if (ggml_openvino_is_gpu() && !is_opt("native")) {
return ExtraQuantType::Q4_0_64;
}
}
Expand Down Expand Up @@ -590,12 +662,11 @@ ggml_openvino_tensor_extra * ggml_openvino_create_tensor_extra(const ggml_tensor
return nullptr;
}

const auto & device_name = ggml_openvino_get_device_name();
auto remote_context = ggml_openvino_get_remote_context();

std::shared_ptr<ov::Tensor> ov_tensor;
if (is_remote) {
GGML_ASSERT(device_name == "GPU");
GGML_ASSERT(ggml_openvino_is_gpu());
auto gpu_context = remote_context->as<ov::intel_gpu::ocl::ClContext>();
auto usm_tensor = gpu_context.create_tensor(element_type, shape, tensor->data);
ov_tensor = std::make_shared<ov::intel_gpu::ocl::USMTensor>(std::move(usm_tensor));
Expand Down
12 changes: 11 additions & 1 deletion ggml/src/ggml-openvino/ggml-openvino-extra.h
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,7 @@ clEnqueueMemcpyINTEL_fn ggml_openvino_get_clEnqueueMemcpyINTEL();

struct ggml_openvino_device_config {
std::string device_name = "CPU";
std::vector<std::string> available_devices;
bool is_npu = false;
bool initialized = false;
std::optional<ov::RemoteContext> remote_context;
Expand All @@ -84,6 +85,12 @@ void ggml_openvino_init_device_config();
// Get the device name
const std::string & ggml_openvino_get_device_name();

// Get all available physical OpenVINO devices
std::vector<std::string> ggml_openvino_get_available_devices();

// Human-readable device name, e.g. "Intel(R) AI Boost (NPU 4000)"; the device id if unavailable
std::string ggml_openvino_get_device_description(const std::string & device_name);

// Environment variable accessors. All GGML_OPENVINO_* env vars are read once
// during backend init and cached on the device config; consumers must go
// through these helpers (never call ::getenv directly) so behavior stays
Expand All @@ -103,11 +110,14 @@ int ggml_openvino_getenv_int(const char * var, int default_value = 0);
// Memory optimization toggles. GGML_OPENVINO_MEMORY_OPTIMIZE is an umbrella
// switch; the fine-grained env vars still override it when explicitly set.
bool ggml_openvino_reduce_compile_mem_enabled();
bool ggml_openvino_release_weights_enabled(const std::string & device);
bool ggml_openvino_release_weights_enabled();

// Check if running on NPU
bool ggml_openvino_is_npu();

// Check if running on a GPU (GPU, GPU.0, GPU.1, ...)
bool ggml_openvino_is_gpu();

// Largest single memory object the device can allocate, SIZE_MAX when there is no known limit
size_t ggml_openvino_max_alloc_size();

Expand Down
Loading