From 5420c46f3864db7fa2da691f1b05797e8016b305 Mon Sep 17 00:00:00 2001 From: Amir Balwel Date: Wed, 23 Sep 2026 22:06:27 +0800 Subject: [PATCH 1/5] [Kubernetes] Add Intel GPU discovery and resource selection Signed-off-by: Amir Balwel --- Dockerfile | 2 +- docs/source/compute/gpus.rst | 6 + .../source/reference/kubernetes/intel-gpu.rst | 75 ++++++++++ sky/clouds/kubernetes.py | 21 +-- sky/provision/kubernetes/utils.py | 129 ++++++++++++++---- 5 files changed, 193 insertions(+), 40 deletions(-) create mode 100644 docs/source/reference/kubernetes/intel-gpu.rst diff --git a/Dockerfile b/Dockerfile index d4717209367..443df2651ae 100644 --- a/Dockerfile +++ b/Dockerfile @@ -25,7 +25,7 @@ RUN if [ "$INSTALL_FROM_SOURCE" = "true" ]; then \ echo "Installing NPM and Node.js for dashboard build" && \ apt-get update -y && \ apt-get install --no-install-recommends -y git curl ca-certificates gnupg && \ - curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \ + curl -fsSL https://deb.nodesource.com/setup_22.x | bash - && \ apt-get install -y nodejs && \ npm install -g npm@latest; \ fi diff --git a/docs/source/compute/gpus.rst b/docs/source/compute/gpus.rst index f3a22350987..0a9c844aadf 100644 --- a/docs/source/compute/gpus.rst +++ b/docs/source/compute/gpus.rst @@ -95,9 +95,15 @@ AMD GPUs See :ref:`kubernetes-amd-gpu`. +Intel GPUs +---------- + +See :ref:`kubernetes-intel-gpu`. + .. toctree:: :maxdepth: 1 :hidden: Using Google TPUs <../../reference/tpu> Using AMD GPUs <../../reference/kubernetes/amd-gpu> + Using Intel GPUs <../../reference/kubernetes/intel-gpu> \ No newline at end of file diff --git a/docs/source/reference/kubernetes/intel-gpu.rst b/docs/source/reference/kubernetes/intel-gpu.rst new file mode 100644 index 00000000000..e578e4b971d --- /dev/null +++ b/docs/source/reference/kubernetes/intel-gpu.rst @@ -0,0 +1,75 @@ +.. _kubernetes-intel-gpu: + +Using Intel GPUs on Kubernetes +============================== + +SkyPilot discovers Intel GPUs through Kubernetes node resources and labels. +Only the ``gpu.intel.com/xe`` resource is supported for Intel GPUs. Monitoring +resources are not counted as GPUs. + +Cluster setup +------------- + +Install the host ``xe`` GPU driver, the `Intel GPU device plugin +`_, +and its Node Feature Discovery (NFD) rules. Use ``shared-dev-num=1`` if you want +reported counts to correspond to unshared GPU devices. With sharing enabled, +Kubernetes advertises allocation slots rather than physical GPU counts. + +Use nodes exposing a single GPU model and a single GPU resource type. +A node containing different GPU models, such as an integrated GPU plus an Arc +card, must have the unwanted devices excluded from the device plugin before +being labeled with a single model name. + +GPU labels +---------- + +Intel NFD product labels are recognized automatically: + +.. code-block:: text + + gpu.intel.com/product=Flex_170 -> Intel-Flex-170 + gpu.intel.com/product=Max_1550 -> Intel-Max-1550 + +For Arc or integrated GPUs without an NFD product label, add a lowercase +SkyPilot label identifying the GPU actually exposed by the device plugin: + +.. code-block:: bash + + kubectl label node skypilot.co/accelerator=intel-arc-a770 --overwrite + +PCI device-ID labels alone do not identify a model in SkyPilot yet. Do not use +``xe`` as a model name: it identifies a driver, not a GPU model. + +SkyPilot selects ``gpu.intel.com/xe`` from the capacity of nodes matching the +accelerator label. Intel scale-from-zero resource selection requires an explicit +``CUSTOM_GPU_RESOURCE_KEY=gpu.intel.com/xe`` on the API server; automatic +detection requires an existing node advertising capacity. + +Workloads need a container image with the appropriate Intel userspace drivers +and compute runtime. Discovery and resource selection alone do not establish +compatibility with a particular framework or workload. + +Manual verification +------------------- + +After installing this version, restart the API server to pick up the changes: + +.. code-block:: bash + + sky api stop + sky api start + kubectl get nodes -o json + sky check kubernetes + sky show-gpus --infra kubernetes + +Verify that each Intel node advertises ``gpu.intel.com/xe`` and has a product +or SkyPilot accelerator label. The GPU +listing should show the corresponding model and allocatable count. Repeat with +NVIDIA and AMD nodes in the same cluster and confirm their counts are unchanged. + +For a launch using an Intel-compatible image and the discovered accelerator, +inspect the resulting pod with ``kubectl get pod -o yaml``. Its GPU +request and limit should use ``gpu.intel.com/xe``. Check dashboard +free counts before and during the workload: one allocated GPU should reduce +availability by one. Monitoring resources must not increase GPU counts. \ No newline at end of file diff --git a/sky/clouds/kubernetes.py b/sky/clouds/kubernetes.py index d2308177448..d93ea30d092 100644 --- a/sky/clouds/kubernetes.py +++ b/sky/clouds/kubernetes.py @@ -587,23 +587,12 @@ def _get_image_id(resources: 'resources_lib.Resources') -> str: tpu_requested = True k8s_resource_key = kubernetes_utils.TPU_RESOURCE_KEY else: - # Derive resource key from the matched label key. - # AMD device plugin labels start with 'amd.com/'; all other - # recognized GPU label formatters (GFD, SkyPilot, GKE, - # Karpenter, CoreWeave, Nebius) are for NVIDIA GPUs. - # We must NOT fall back to get_gpu_resource_key(context) here: - # in a mixed cluster it scans nodes and returns the first - # vendor key found (amd.com/gpu before nvidia.com/gpu), - # which would put an amd.com/gpu request on an NVIDIA pod. - if (k8s_acc_label_key is not None and - k8s_acc_label_key.startswith('amd.com/')): + if k8s_acc_label_key is not None: k8s_resource_key = ( - kubernetes_utils.SUPPORTED_GPU_RESOURCE_KEYS['amd']) - elif k8s_acc_label_key is not None: - k8s_resource_key = ( - kubernetes_utils.SUPPORTED_GPU_RESOURCE_KEYS['nvidia']) + kubernetes_utils.get_gpu_resource_key_for_labels( + context, k8s_acc_label_key, + k8s_acc_label_values or [])) else: - # Fallback if no label formatter matched (unusual). k8s_resource_key = kubernetes_utils.get_gpu_resource_key( context) else: @@ -1400,4 +1389,4 @@ def _detect_network_type( 'instance_type': 'a3-ultragpu-8g' }) - return KubernetesHighPerformanceNetworkType.NONE, None + return KubernetesHighPerformanceNetworkType.NONE, None \ No newline at end of file diff --git a/sky/provision/kubernetes/utils.py b/sky/provision/kubernetes/utils.py index 8fd1e53261b..232eb314502 100644 --- a/sky/provision/kubernetes/utils.py +++ b/sky/provision/kubernetes/utils.py @@ -172,16 +172,19 @@ def requires_tcpxo_daemon(self) -> bool: 'P': 2**50, } -# The resource keys used by Kubernetes to track NVIDIA GPUs and Google TPUs on +# The resource keys used by Kubernetes to track GPUs and Google TPUs on # nodes. These keys are typically used in the node's status.allocatable # or status.capacity fields to indicate the available resources on the node. -SUPPORTED_GPU_RESOURCE_KEYS = {'amd': 'amd.com/gpu', 'nvidia': 'nvidia.com/gpu'} +SUPPORTED_GPU_RESOURCE_KEYS = { + 'amd': 'amd.com/gpu', + 'nvidia': 'nvidia.com/gpu', + 'intel_xe': 'gpu.intel.com/xe', +} TPU_RESOURCE_KEY = 'google.com/tpu' NO_ACCELERATOR_HELP_MESSAGE = ( 'If your cluster contains GPUs or TPUs, make sure ' - f'one of {SUPPORTED_GPU_RESOURCE_KEYS["amd"]}, ' - f'{SUPPORTED_GPU_RESOURCE_KEYS["nvidia"]} or ' + f'one of {", ".join(SUPPORTED_GPU_RESOURCE_KEYS.values())} or ' f'{TPU_RESOURCE_KEY} resource is available ' 'on the nodes and the node labels for identifying GPUs/TPUs ' '(e.g., skypilot.co/accelerator) are setup correctly. ') @@ -791,6 +794,45 @@ def _normalize(cls, raw: str) -> str: return name.replace(' ', '') +class IntelGPULabelFormatter(GPULabelFormatter): + """Intel NFD product labels on nodes exposing a single GPU model. + + Arc and integrated GPUs without a product label can use an explicit + skypilot.co/accelerator label instead. + """ + + LABEL_KEY = 'gpu.intel.com/product' + + @classmethod + def match_label_key(cls, label_key: str) -> bool: + return label_key == cls.LABEL_KEY + + @classmethod + def get_label_key(cls, accelerator: Optional[str] = None) -> str: + return cls.LABEL_KEY + + @classmethod + def get_label_keys(cls) -> List[str]: + return [cls.LABEL_KEY] + + @classmethod + def get_label_values(cls, accelerator: str) -> List[str]: + raise NotImplementedError + + @classmethod + def validate_label_value(cls, value: str) -> Tuple[bool, str]: + valid = bool(value and value.strip()) + return valid, (f'Intel GPU label value {value!r} must be non-empty.' + if not valid else '') + + @classmethod + def get_accelerator_from_label_value(cls, value: str) -> str: + name = value.strip().replace('_', '-').replace(' ', '-') + if name.lower().startswith('intel-'): + name = name[6:] + return f'Intel-{name}' + + def _accelerator_name_matches(requested_acc: str, viable_names: List[str]) -> bool: """Check if requested accelerator matches any viable name. @@ -882,8 +924,8 @@ def validate_label_value(cls, value: str) -> Tuple[bool, str]: # auto-detecting the GPU label type. LABEL_FORMATTER_REGISTRY = [ SkyPilotLabelFormatter, GKELabelFormatter, KarpenterLabelFormatter, - GFDLabelFormatter, AMDGPULabelFormatter, CoreWeaveLabelFormatter, - NebiusLabelFormatter + GFDLabelFormatter, AMDGPULabelFormatter, IntelGPULabelFormatter, + CoreWeaveLabelFormatter, NebiusLabelFormatter ] @@ -1360,10 +1402,8 @@ def detect_accelerator_resource( context: Optional[str]) -> Tuple[bool, Set[str]]: """Checks if the Kubernetes cluster has GPU/TPU resource. - Three types of accelerator resources are available which are each checked - with amd.com/gpu, nvidia.com/gpu and google.com/tpu. If amd.com/gpu or nvidia.com/gpu resource is - missing, that typically means that the Kubernetes cluster does not have - GPUs or the amd/nvidia GPU operator and/or device drivers are not installed. + Checks all supported GPU resources and Google TPUs. Missing resources + typically mean the device plugin or drivers are not installed. Returns: bool: True if the cluster has GPU_RESOURCE_KEY or TPU_RESOURCE_KEY @@ -1374,8 +1414,11 @@ def detect_accelerator_resource( nodes = get_kubernetes_nodes(context=context) for node in nodes: cluster_resources.update(node.status.allocatable.keys()) - has_accelerator = (get_gpu_resource_key(context) in cluster_resources or - TPU_RESOURCE_KEY in cluster_resources) + resource_keys = {TPU_RESOURCE_KEY, *SUPPORTED_GPU_RESOURCE_KEYS.values()} + custom_key = os.getenv('CUSTOM_GPU_RESOURCE_KEY') + if custom_key: + resource_keys.add(custom_key) + has_accelerator = bool(resource_keys.intersection(cluster_resources)) return has_accelerator, cluster_resources @@ -2058,15 +2101,14 @@ def get_accelerator_label_key_values( 'and re-run ' f'`sky ssh up --infra {context_display_name}`. {suffix}') else: + gpu_resources = ', '.join(SUPPORTED_GPU_RESOURCE_KEYS.values()) msg = ( - f'Could not detect GPU/TPU resources ({SUPPORTED_GPU_RESOURCE_KEYS["amd"]!r}, ' - f'{SUPPORTED_GPU_RESOURCE_KEYS["nvidia"]!r} or ' + f'Could not detect GPU/TPU resources ({gpu_resources} or ' f'{TPU_RESOURCE_KEY!r}) in Kubernetes cluster. If this cluster' ' contains GPUs, please ensure GPU drivers are installed on ' 'the node. Check if the GPUs are setup correctly by running ' '`kubectl describe nodes` and looking for the ' - f'{SUPPORTED_GPU_RESOURCE_KEYS["amd"]!r}, ' - f'{SUPPORTED_GPU_RESOURCE_KEYS["nvidia"]!r} or ' + f'{gpu_resources} or ' f'{TPU_RESOURCE_KEY!r} resource. ' 'Please refer to the documentation on how to set up GPUs.' f'{suffix}') @@ -3809,12 +3851,14 @@ def get_node_accelerator_count(context: Optional[str], Number of accelerators allocated or available from the node. If no resource is found, it returns 0. """ - gpu_resource_name = get_gpu_resource_key(context) - assert not (gpu_resource_name in attribute_dict and - TPU_RESOURCE_KEY in attribute_dict) - for gpu_resource in ["nvidia.com/gpu", "amd.com/gpu"]: - if gpu_resource in attribute_dict: - return int(attribute_dict[gpu_resource]) + resource_keys = set(SUPPORTED_GPU_RESOURCE_KEYS.values()) + custom_key = os.getenv('CUSTOM_GPU_RESOURCE_KEY') + if custom_key: + resource_keys.add(custom_key) + gpu_count = sum(int(attribute_dict.get(key, 0)) for key in resource_keys) + assert not (gpu_count and TPU_RESOURCE_KEY in attribute_dict) + if gpu_count: + return gpu_count if TPU_RESOURCE_KEY in attribute_dict: return int(attribute_dict[TPU_RESOURCE_KEY]) return 0 @@ -4039,6 +4083,45 @@ def process_skypilot_pods( return list(clusters.values()), jobs_controllers, serve_controllers +def get_gpu_resource_key_for_labels(context: Optional[str], + label_key: str, + label_values: List[str]) -> str: + """Resolve a resource from matching nodes, rather than the whole cluster. + + A pod can request only one resource key. Refuse ambiguous selections rather + than requesting one driver's resources for nodes using another driver. + """ + custom_key = os.getenv('CUSTOM_GPU_RESOURCE_KEY') + if custom_key: + return custom_key + resource_keys = set() + for node in get_kubernetes_nodes(context=context): + if (node.metadata.labels or {}).get(label_key) not in label_values: + continue + capacity = node.status.capacity or {} + resource_keys.update( + key for key in SUPPORTED_GPU_RESOURCE_KEYS.values() + if int(capacity.get(key, 0)) > 0) + if len(resource_keys) == 1: + return next(iter(resource_keys)) + if resource_keys: + raise exceptions.ResourcesUnavailableError( + f'Nodes matching {label_key}={label_values} expose multiple GPU ' + f'resource types: {sorted(resource_keys)}. Use a distinct accelerator ' + 'label for nodes with a single GPU resource type.') + # Preserve scale-from-zero behavior for existing label formats. Intel + # product labels do not identify which kernel driver will be used. + if (label_key.startswith('gpu.intel.com/') or + (label_key == SkyPilotLabelFormatter.LABEL_KEY and + any(value.lower().startswith('intel-') for value in label_values))): + raise exceptions.ResourcesUnavailableError( + 'Cannot determine the Intel GPU resource type from matching nodes. ' + 'Install the Intel GPU device plugin and verify node capacity.') + if label_key.startswith('amd.com/'): + return SUPPORTED_GPU_RESOURCE_KEYS['amd'] + return SUPPORTED_GPU_RESOURCE_KEYS['nvidia'] + + def _gpu_resource_key_helper(context: Optional[str]) -> str: """Helper function to get the GPU resource key.""" gpu_resource_key = SUPPORTED_GPU_RESOURCE_KEYS['nvidia'] @@ -4286,4 +4369,4 @@ def get_pvc_events(context: Optional[str], return sorted(pvc_events.items, key=lambda e: (e.last_timestamp or e.metadata.creation_timestamp), - reverse=reverse) + reverse=reverse) \ No newline at end of file From ba52c974dd2091564eaab28b7b15660a77964ecd Mon Sep 17 00:00:00 2001 From: Amir Balwel Date: Wed, 23 Sep 2026 22:11:27 +0800 Subject: [PATCH 2/5] update readme Signed-off-by: Amir Balwel --- README.md | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/README.md b/README.md index 65919501cd3..6f13794efe6 100644 --- a/README.md +++ b/README.md @@ -317,6 +317,38 @@ deserialize() rather than reimplementing it. + + [Kubernetes] Add Intel GPU discovery and resource selection + 5420c46 + + sky/provision/kubernetes/utils.py
+ sky/clouds/kubernetes.py
+ docs/source/reference/kubernetes/intel-gpu.rst
+ docs/source/compute/gpus.rst
+ Dockerfile + + + Adds Intel GPU discovery and counting through + gpu.intel.com/xe. The new + IntelGPULabelFormatter reads + gpu.intel.com/product labels and normalizes model names + (e.g. Flex_170 → Intel-Flex-170). + Arc and integrated GPUs without a product label can use an explicit + skypilot.co/accelerator label. Each node must expose a + single GPU model and resource type; monitoring resources are not + counted as GPUs.
+ Replaces the formatter-category resource selection from + f65b71f with a lookup of capacity on nodes matching the + accelerator label, supporting mixed Intel + NVIDIA + AMD clusters + and rejecting ambiguous resource selections. Automatic Intel resource + selection requires an existing node advertising capacity; an explicit + CUSTOM_GPU_RESOURCE_KEY=gpu.intel.com/xe supports + resource selection when scaling from zero.
+ Adds an Intel GPU setup and manual verification guide, links it from + the GPU documentation, and updates the Dockerfile's dashboard build + stage from Node.js 20 to 22. + + @@ -396,6 +428,7 @@ git cherry-pick c6f8f23 # Raise per-controller service capacity for k8s git cherry-pick 78fe751 # Pin uv pip to runtime venv via --python git cherry-pick 69b0a69 # Exclude kubernetes==36.0.0 (in-cluster auth regression) git cherry-pick 827ff42 # Accept PEP 585 dict[K,V] type strings in pod_config validator +git cherry-pick 5420c46 # Intel xe GPU discovery and resource selection # Resolve any conflicts if upstream changed the same files # 4. Push new branch From bb8cbc962901175de184b11ac860c568e43b51b9 Mon Sep 17 00:00:00 2001 From: Amir Balwel Date: Thu, 24 Sep 2026 20:42:35 +0800 Subject: [PATCH 3/5] update Signed-off-by: Amir Balwel --- Dockerfile | 2 +- README.md | 20 +++--- .../source/reference/kubernetes/intel-gpu.rst | 11 +++- sky/clouds/kubernetes.py | 19 ++++-- sky/provision/kubernetes/utils.py | 66 +++---------------- 5 files changed, 41 insertions(+), 77 deletions(-) diff --git a/Dockerfile b/Dockerfile index 443df2651ae..d4717209367 100644 --- a/Dockerfile +++ b/Dockerfile @@ -25,7 +25,7 @@ RUN if [ "$INSTALL_FROM_SOURCE" = "true" ]; then \ echo "Installing NPM and Node.js for dashboard build" && \ apt-get update -y && \ apt-get install --no-install-recommends -y git curl ca-certificates gnupg && \ - curl -fsSL https://deb.nodesource.com/setup_22.x | bash - && \ + curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \ apt-get install -y nodejs && \ npm install -g npm@latest; \ fi diff --git a/README.md b/README.md index 6f13794efe6..b521fb3f958 100644 --- a/README.md +++ b/README.md @@ -324,8 +324,7 @@ sky/provision/kubernetes/utils.py
sky/clouds/kubernetes.py
docs/source/reference/kubernetes/intel-gpu.rst
- docs/source/compute/gpus.rst
- Dockerfile + docs/source/compute/gpus.rst Adds Intel GPU discovery and counting through @@ -337,16 +336,13 @@ skypilot.co/accelerator label. Each node must expose a single GPU model and resource type; monitoring resources are not counted as GPUs.
- Replaces the formatter-category resource selection from - f65b71f with a lookup of capacity on nodes matching the - accelerator label, supporting mixed Intel + NVIDIA + AMD clusters - and rejecting ambiguous resource selections. Automatic Intel resource - selection requires an existing node advertising capacity; an explicit - CUSTOM_GPU_RESOURCE_KEY=gpu.intel.com/xe supports - resource selection when scaling from zero.
- Adds an Intel GPU setup and manual verification guide, links it from - the GPU documentation, and updates the Dockerfile's dashboard build - stage from Node.js 20 to 22. + Selects the Xe resource from Intel product labels or generic SkyPilot + accelerator labels beginning with intel-, supporting mixed + Intel + NVIDIA + AMD clusters. Generic autoscaling can select the + resource without an existing GPU node or a custom-resource override; + the autoscaler must have a matching Xe node pool configured.
+ Adds an Intel GPU setup and manual verification guide and links it + from the GPU documentation. diff --git a/docs/source/reference/kubernetes/intel-gpu.rst b/docs/source/reference/kubernetes/intel-gpu.rst index e578e4b971d..79c0d488be4 100644 --- a/docs/source/reference/kubernetes/intel-gpu.rst +++ b/docs/source/reference/kubernetes/intel-gpu.rst @@ -5,7 +5,8 @@ Using Intel GPUs on Kubernetes SkyPilot discovers Intel GPUs through Kubernetes node resources and labels. Only the ``gpu.intel.com/xe`` resource is supported for Intel GPUs. Monitoring -resources are not counted as GPUs. +resources are not counted as GPUs. Nodes using ``gpu.intel.com/i915`` are not +supported by this integration. Cluster setup ------------- @@ -31,6 +32,10 @@ Intel NFD product labels are recognized automatically: gpu.intel.com/product=Flex_170 -> Intel-Flex-170 gpu.intel.com/product=Max_1550 -> Intel-Max-1550 +These examples describe label formatting, not driver compatibility. A product +label does not identify the kernel driver; the node must also expose +``gpu.intel.com/xe`` to be usable with this integration. + For Arc or integrated GPUs without an NFD product label, add a lowercase SkyPilot label identifying the GPU actually exposed by the device plugin: @@ -64,8 +69,8 @@ After installing this version, restart the API server to pick up the changes: sky show-gpus --infra kubernetes Verify that each Intel node advertises ``gpu.intel.com/xe`` and has a product -or SkyPilot accelerator label. The GPU -listing should show the corresponding model and allocatable count. Repeat with +or SkyPilot accelerator label. The GPU listing should show the corresponding +model and allocatable count. Repeat with NVIDIA and AMD nodes in the same cluster and confirm their counts are unchanged. For a launch using an Intel-compatible image and the discovered accelerator, diff --git a/sky/clouds/kubernetes.py b/sky/clouds/kubernetes.py index d93ea30d092..eacf8018cd0 100644 --- a/sky/clouds/kubernetes.py +++ b/sky/clouds/kubernetes.py @@ -587,11 +587,20 @@ def _get_image_id(resources: 'resources_lib.Resources') -> str: tpu_requested = True k8s_resource_key = kubernetes_utils.TPU_RESOURCE_KEY else: - if k8s_acc_label_key is not None: + if (k8s_acc_label_key is not None and + k8s_acc_label_key.startswith('amd.com/')): k8s_resource_key = ( - kubernetes_utils.get_gpu_resource_key_for_labels( - context, k8s_acc_label_key, - k8s_acc_label_values or [])) + kubernetes_utils.SUPPORTED_GPU_RESOURCE_KEYS['amd']) + elif (k8s_acc_label_key + == kubernetes_utils.IntelGPULabelFormatter.LABEL_KEY or + (k8s_acc_label_key + == kubernetes_utils.SkyPilotLabelFormatter.LABEL_KEY and + acc_type.lower().startswith('intel-'))): + k8s_resource_key = (kubernetes_utils. + SUPPORTED_GPU_RESOURCE_KEYS['intel_xe']) + elif k8s_acc_label_key is not None: + k8s_resource_key = ( + kubernetes_utils.SUPPORTED_GPU_RESOURCE_KEYS['nvidia']) else: k8s_resource_key = kubernetes_utils.get_gpu_resource_key( context) @@ -1389,4 +1398,4 @@ def _detect_network_type( 'instance_type': 'a3-ultragpu-8g' }) - return KubernetesHighPerformanceNetworkType.NONE, None \ No newline at end of file + return KubernetesHighPerformanceNetworkType.NONE, None diff --git a/sky/provision/kubernetes/utils.py b/sky/provision/kubernetes/utils.py index 232eb314502..869989168e2 100644 --- a/sky/provision/kubernetes/utils.py +++ b/sky/provision/kubernetes/utils.py @@ -828,8 +828,6 @@ def validate_label_value(cls, value: str) -> Tuple[bool, str]: @classmethod def get_accelerator_from_label_value(cls, value: str) -> str: name = value.strip().replace('_', '-').replace(' ', '-') - if name.lower().startswith('intel-'): - name = name[6:] return f'Intel-{name}' @@ -1414,11 +1412,8 @@ def detect_accelerator_resource( nodes = get_kubernetes_nodes(context=context) for node in nodes: cluster_resources.update(node.status.allocatable.keys()) - resource_keys = {TPU_RESOURCE_KEY, *SUPPORTED_GPU_RESOURCE_KEYS.values()} - custom_key = os.getenv('CUSTOM_GPU_RESOURCE_KEY') - if custom_key: - resource_keys.add(custom_key) - has_accelerator = bool(resource_keys.intersection(cluster_resources)) + has_accelerator = (get_gpu_resource_key(context) in cluster_resources or + TPU_RESOURCE_KEY in cluster_resources) return has_accelerator, cluster_resources @@ -3386,7 +3381,8 @@ def get_unlabeled_accelerator_nodes(context: Optional[str] = None) -> List[Any]: continue node_label_keys = set(node.metadata.labels or {}) labeled = any( - fmt.match_label_key(lk) for lk in node_label_keys + fmt.match_label_key(lk) + for lk in node_label_keys for fmt in LABEL_FORMATTER_REGISTRY) if not labeled: unlabeled_nodes.append(node) @@ -3837,6 +3833,7 @@ def is_tpu_on_gke(accelerator: str, normalize: bool = True) -> bool: def get_node_accelerator_count(context: Optional[str], attribute_dict: dict) -> int: + # pylint: disable=unused-argument """Retrieves the count of accelerators from a node's resource dictionary. This method checks the node's allocatable resources or the accelerators @@ -3851,14 +3848,10 @@ def get_node_accelerator_count(context: Optional[str], Number of accelerators allocated or available from the node. If no resource is found, it returns 0. """ - resource_keys = set(SUPPORTED_GPU_RESOURCE_KEYS.values()) - custom_key = os.getenv('CUSTOM_GPU_RESOURCE_KEY') - if custom_key: - resource_keys.add(custom_key) - gpu_count = sum(int(attribute_dict.get(key, 0)) for key in resource_keys) - assert not (gpu_count and TPU_RESOURCE_KEY in attribute_dict) - if gpu_count: - return gpu_count + for gpu_resource in SUPPORTED_GPU_RESOURCE_KEYS.values(): + if gpu_resource in attribute_dict: + assert TPU_RESOURCE_KEY not in attribute_dict + return int(attribute_dict[gpu_resource]) if TPU_RESOURCE_KEY in attribute_dict: return int(attribute_dict[TPU_RESOURCE_KEY]) return 0 @@ -4083,45 +4076,6 @@ def process_skypilot_pods( return list(clusters.values()), jobs_controllers, serve_controllers -def get_gpu_resource_key_for_labels(context: Optional[str], - label_key: str, - label_values: List[str]) -> str: - """Resolve a resource from matching nodes, rather than the whole cluster. - - A pod can request only one resource key. Refuse ambiguous selections rather - than requesting one driver's resources for nodes using another driver. - """ - custom_key = os.getenv('CUSTOM_GPU_RESOURCE_KEY') - if custom_key: - return custom_key - resource_keys = set() - for node in get_kubernetes_nodes(context=context): - if (node.metadata.labels or {}).get(label_key) not in label_values: - continue - capacity = node.status.capacity or {} - resource_keys.update( - key for key in SUPPORTED_GPU_RESOURCE_KEYS.values() - if int(capacity.get(key, 0)) > 0) - if len(resource_keys) == 1: - return next(iter(resource_keys)) - if resource_keys: - raise exceptions.ResourcesUnavailableError( - f'Nodes matching {label_key}={label_values} expose multiple GPU ' - f'resource types: {sorted(resource_keys)}. Use a distinct accelerator ' - 'label for nodes with a single GPU resource type.') - # Preserve scale-from-zero behavior for existing label formats. Intel - # product labels do not identify which kernel driver will be used. - if (label_key.startswith('gpu.intel.com/') or - (label_key == SkyPilotLabelFormatter.LABEL_KEY and - any(value.lower().startswith('intel-') for value in label_values))): - raise exceptions.ResourcesUnavailableError( - 'Cannot determine the Intel GPU resource type from matching nodes. ' - 'Install the Intel GPU device plugin and verify node capacity.') - if label_key.startswith('amd.com/'): - return SUPPORTED_GPU_RESOURCE_KEYS['amd'] - return SUPPORTED_GPU_RESOURCE_KEYS['nvidia'] - - def _gpu_resource_key_helper(context: Optional[str]) -> str: """Helper function to get the GPU resource key.""" gpu_resource_key = SUPPORTED_GPU_RESOURCE_KEYS['nvidia'] @@ -4369,4 +4323,4 @@ def get_pvc_events(context: Optional[str], return sorted(pvc_events.items, key=lambda e: (e.last_timestamp or e.metadata.creation_timestamp), - reverse=reverse) \ No newline at end of file + reverse=reverse) From 0e986b2cc50c3a0b7ad2133d4fe15b668968c6b1 Mon Sep 17 00:00:00 2001 From: Amir Balwel Date: Thu, 24 Sep 2026 20:46:35 +0800 Subject: [PATCH 4/5] update Signed-off-by: Amir Balwel --- sky/clouds/kubernetes.py | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/sky/clouds/kubernetes.py b/sky/clouds/kubernetes.py index eacf8018cd0..774c470b9a8 100644 --- a/sky/clouds/kubernetes.py +++ b/sky/clouds/kubernetes.py @@ -592,10 +592,7 @@ def _get_image_id(resources: 'resources_lib.Resources') -> str: k8s_resource_key = ( kubernetes_utils.SUPPORTED_GPU_RESOURCE_KEYS['amd']) elif (k8s_acc_label_key - == kubernetes_utils.IntelGPULabelFormatter.LABEL_KEY or - (k8s_acc_label_key - == kubernetes_utils.SkyPilotLabelFormatter.LABEL_KEY and - acc_type.lower().startswith('intel-'))): + == kubernetes_utils.IntelGPULabelFormatter.LABEL_KEY): k8s_resource_key = (kubernetes_utils. SUPPORTED_GPU_RESOURCE_KEYS['intel_xe']) elif k8s_acc_label_key is not None: From f57943872d969a6479781b4f3def0a9012758dbe Mon Sep 17 00:00:00 2001 From: Hoipang_ Date: Thu, 24 Sep 2026 21:02:02 +0800 Subject: [PATCH 5/5] [Kubernetes] Clean up Intel GPU docs and minimize diff - Docs/README: xe driver only (i915 GPUs unsupported), B-series examples, require NFD rules v0.37.0+, drop manual-label and CUSTOM_GPU_RESOURCE_KEY sections. - Restore resource-key comments in clouds/kubernetes.py. - Restore original assert in get_node_accelerator_count; drop pylint disable. - Revert unrelated reformat; restore trailing newlines. --- README.md | 20 +++--- docs/source/compute/gpus.rst | 2 +- .../source/reference/kubernetes/intel-gpu.rst | 69 ++++++++----------- sky/clouds/kubernetes.py | 10 +++ sky/provision/kubernetes/utils.py | 12 ++-- 5 files changed, 52 insertions(+), 61 deletions(-) diff --git a/README.md b/README.md index b521fb3f958..3c163de1227 100644 --- a/README.md +++ b/README.md @@ -328,19 +328,15 @@ Adds Intel GPU discovery and counting through - gpu.intel.com/xe. The new - IntelGPULabelFormatter reads + gpu.intel.com/xe (xe driver only; + i915 GPUs are not supported). The new + IntelGPULabelFormatter reads NFD gpu.intel.com/product labels and normalizes model names - (e.g. Flex_170 → Intel-Flex-170). - Arc and integrated GPUs without a product label can use an explicit - skypilot.co/accelerator label. Each node must expose a - single GPU model and resource type; monitoring resources are not - counted as GPUs.
- Selects the Xe resource from Intel product labels or generic SkyPilot - accelerator labels beginning with intel-, supporting mixed - Intel + NVIDIA + AMD clusters. Generic autoscaling can select the - resource without an existing GPU node or a custom-resource override; - the autoscaler must have a matching Xe node pool configured.
+ (e.g. Arc_Pro_B60 → Intel-Arc-Pro-B60); + B-series product labels need Intel device plugin NFD rules v0.37.0+.
+ Pods for Intel product labels request gpu.intel.com/xe, + alongside the existing AMD/NVIDIA label-key selection, so mixed + clusters and autoscaling without an existing GPU node work.
Adds an Intel GPU setup and manual verification guide and links it from the GPU documentation. diff --git a/docs/source/compute/gpus.rst b/docs/source/compute/gpus.rst index 0a9c844aadf..9e1456d4a17 100644 --- a/docs/source/compute/gpus.rst +++ b/docs/source/compute/gpus.rst @@ -106,4 +106,4 @@ See :ref:`kubernetes-intel-gpu`. Using Google TPUs <../../reference/tpu> Using AMD GPUs <../../reference/kubernetes/amd-gpu> - Using Intel GPUs <../../reference/kubernetes/intel-gpu> \ No newline at end of file + Using Intel GPUs <../../reference/kubernetes/intel-gpu> diff --git a/docs/source/reference/kubernetes/intel-gpu.rst b/docs/source/reference/kubernetes/intel-gpu.rst index 79c0d488be4..164f87acb9d 100644 --- a/docs/source/reference/kubernetes/intel-gpu.rst +++ b/docs/source/reference/kubernetes/intel-gpu.rst @@ -3,57 +3,42 @@ Using Intel GPUs on Kubernetes ============================== -SkyPilot discovers Intel GPUs through Kubernetes node resources and labels. -Only the ``gpu.intel.com/xe`` resource is supported for Intel GPUs. Monitoring -resources are not counted as GPUs. Nodes using ``gpu.intel.com/i915`` are not -supported by this integration. +SkyPilot supports Intel GPUs that use the ``xe`` kernel driver and are exposed +through the ``gpu.intel.com/xe`` resource, such as Arc Pro B-series cards. +GPUs using the ``i915`` driver (``gpu.intel.com/i915``), including Arc +A-series, Data Center GPU Flex and Max, are not supported. Monitoring +resources are not counted as GPUs. Cluster setup ------------- Install the host ``xe`` GPU driver, the `Intel GPU device plugin `_, -and its Node Feature Discovery (NFD) rules. Use ``shared-dev-num=1`` if you want -reported counts to correspond to unshared GPU devices. With sharing enabled, -Kubernetes advertises allocation slots rather than physical GPU counts. +and its Node Feature Discovery (NFD) rules, v0.37.0 or later. Use +``shared-dev-num=1`` if you want reported counts to correspond to unshared GPU +devices. With sharing enabled, Kubernetes advertises allocation slots rather +than physical GPU counts. -Use nodes exposing a single GPU model and a single GPU resource type. -A node containing different GPU models, such as an integrated GPU plus an Arc -card, must have the unwanted devices excluded from the device plugin before -being labeled with a single model name. +Each node must expose a single GPU model and a single GPU resource type. GPU labels ---------- -Intel NFD product labels are recognized automatically: +SkyPilot identifies Intel GPUs from the NFD ``gpu.intel.com/product`` label, +which the NFD rules set automatically: .. code-block:: text - gpu.intel.com/product=Flex_170 -> Intel-Flex-170 - gpu.intel.com/product=Max_1550 -> Intel-Max-1550 + gpu.intel.com/product=Arc_Pro_B60 -> Intel-Arc-Pro-B60 + gpu.intel.com/product=Arc_B580 -> Intel-Arc-B580 -These examples describe label formatting, not driver compatibility. A product -label does not identify the kernel driver; the node must also expose -``gpu.intel.com/xe`` to be usable with this integration. +Request the GPU by its SkyPilot name, e.g. ``--gpus Intel-Arc-Pro-B60:1``. +Pods requesting an Intel GPU use the ``gpu.intel.com/xe`` resource. Because the +resource is derived from the label, this also works with autoscalers that +create Intel nodes on demand. -For Arc or integrated GPUs without an NFD product label, add a lowercase -SkyPilot label identifying the GPU actually exposed by the device plugin: - -.. code-block:: bash - - kubectl label node skypilot.co/accelerator=intel-arc-a770 --overwrite - -PCI device-ID labels alone do not identify a model in SkyPilot yet. Do not use -``xe`` as a model name: it identifies a driver, not a GPU model. - -SkyPilot selects ``gpu.intel.com/xe`` from the capacity of nodes matching the -accelerator label. Intel scale-from-zero resource selection requires an explicit -``CUSTOM_GPU_RESOURCE_KEY=gpu.intel.com/xe`` on the API server; automatic -detection requires an existing node advertising capacity. - -Workloads need a container image with the appropriate Intel userspace drivers -and compute runtime. Discovery and resource selection alone do not establish -compatibility with a particular framework or workload. +Workloads need a container image with the Intel userspace drivers and compute +runtime. Manual verification ------------------- @@ -68,13 +53,13 @@ After installing this version, restart the API server to pick up the changes: sky check kubernetes sky show-gpus --infra kubernetes -Verify that each Intel node advertises ``gpu.intel.com/xe`` and has a product -or SkyPilot accelerator label. The GPU listing should show the corresponding -model and allocatable count. Repeat with -NVIDIA and AMD nodes in the same cluster and confirm their counts are unchanged. +Verify that each Intel node advertises ``gpu.intel.com/xe`` and has a +``gpu.intel.com/product`` label. The GPU listing should show the corresponding +model and allocatable count. Repeat with NVIDIA and AMD nodes in the same +cluster and confirm their counts are unchanged. For a launch using an Intel-compatible image and the discovered accelerator, inspect the resulting pod with ``kubectl get pod -o yaml``. Its GPU -request and limit should use ``gpu.intel.com/xe``. Check dashboard -free counts before and during the workload: one allocated GPU should reduce -availability by one. Monitoring resources must not increase GPU counts. \ No newline at end of file +request and limit should use ``gpu.intel.com/xe``. Check dashboard free counts +before and during the workload: one allocated GPU should reduce availability +by one. diff --git a/sky/clouds/kubernetes.py b/sky/clouds/kubernetes.py index 774c470b9a8..b53acc6a0a6 100644 --- a/sky/clouds/kubernetes.py +++ b/sky/clouds/kubernetes.py @@ -587,6 +587,15 @@ def _get_image_id(resources: 'resources_lib.Resources') -> str: tpu_requested = True k8s_resource_key = kubernetes_utils.TPU_RESOURCE_KEY else: + # Derive resource key from the matched label key. + # AMD device plugin labels start with 'amd.com/'; Intel NFD + # product labels map to the Xe resource; all other + # recognized GPU label formatters (GFD, SkyPilot, GKE, + # Karpenter, CoreWeave, Nebius) are for NVIDIA GPUs. + # We must NOT fall back to get_gpu_resource_key(context) here: + # in a mixed cluster it scans nodes and returns the first + # vendor key found (amd.com/gpu before nvidia.com/gpu), + # which would put an amd.com/gpu request on an NVIDIA pod. if (k8s_acc_label_key is not None and k8s_acc_label_key.startswith('amd.com/')): k8s_resource_key = ( @@ -599,6 +608,7 @@ def _get_image_id(resources: 'resources_lib.Resources') -> str: k8s_resource_key = ( kubernetes_utils.SUPPORTED_GPU_RESOURCE_KEYS['nvidia']) else: + # Fallback if no label formatter matched (unusual). k8s_resource_key = kubernetes_utils.get_gpu_resource_key( context) else: diff --git a/sky/provision/kubernetes/utils.py b/sky/provision/kubernetes/utils.py index 869989168e2..3fe586eef62 100644 --- a/sky/provision/kubernetes/utils.py +++ b/sky/provision/kubernetes/utils.py @@ -797,8 +797,8 @@ def _normalize(cls, raw: str) -> str: class IntelGPULabelFormatter(GPULabelFormatter): """Intel NFD product labels on nodes exposing a single GPU model. - Arc and integrated GPUs without a product label can use an explicit - skypilot.co/accelerator label instead. + e.g. gpu.intel.com/product=Arc_Pro_B60 -> Intel-Arc-Pro-B60. Only nodes + exposing the gpu.intel.com/xe resource are supported. """ LABEL_KEY = 'gpu.intel.com/product' @@ -3381,8 +3381,7 @@ def get_unlabeled_accelerator_nodes(context: Optional[str] = None) -> List[Any]: continue node_label_keys = set(node.metadata.labels or {}) labeled = any( - fmt.match_label_key(lk) - for lk in node_label_keys + fmt.match_label_key(lk) for lk in node_label_keys for fmt in LABEL_FORMATTER_REGISTRY) if not labeled: unlabeled_nodes.append(node) @@ -3833,7 +3832,6 @@ def is_tpu_on_gke(accelerator: str, normalize: bool = True) -> bool: def get_node_accelerator_count(context: Optional[str], attribute_dict: dict) -> int: - # pylint: disable=unused-argument """Retrieves the count of accelerators from a node's resource dictionary. This method checks the node's allocatable resources or the accelerators @@ -3848,9 +3846,11 @@ def get_node_accelerator_count(context: Optional[str], Number of accelerators allocated or available from the node. If no resource is found, it returns 0. """ + gpu_resource_name = get_gpu_resource_key(context) + assert not (gpu_resource_name in attribute_dict and + TPU_RESOURCE_KEY in attribute_dict) for gpu_resource in SUPPORTED_GPU_RESOURCE_KEYS.values(): if gpu_resource in attribute_dict: - assert TPU_RESOURCE_KEY not in attribute_dict return int(attribute_dict[gpu_resource]) if TPU_RESOURCE_KEY in attribute_dict: return int(attribute_dict[TPU_RESOURCE_KEY])