From fa250448835c7f6359cd907a2f22a0adfa1b090c Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Wed, 30 Sep 2026 22:09:53 +0300 Subject: [PATCH 1/9] feat(vllm): install ROCm 10.1 wheels from staging with a cp314 runtime ROCm 10.x vLLM, flash-attn, and amd-aiter wheels are cp314-only, and ROCm 10.1's frameworks are staged on a separate index than 10.0's production one. The discover route now selects an index, torch index, and Python tag per row keyed on major.minor rather than major alone, so 10.0 and 10.1 resolve through distinct rows instead of sharing one. Torch is now discovered like the other three packages instead of pinned as a literal, since ROCm 10.1's local version rotates too fast to express as a fixed pin. Every resolved pin carrying a +rocmX.Y local version is checked against the SDK's own line before installing, so an index serving more than one ROCm line can't silently install the wrong one's wheel. ROCm CLI now derives the required Python tag (cp312 vs cp314) from the requested source layout before the interpreter is chosen, since that is knowable from the request itself ahead of resolution. A wrong-tag interpreter is rejected with a message naming the required tag, split out from the unrelated venv-creation failure it used to be reported as. The same gate now also applies when reusing a manifest-recorded interpreter on update, closing a bypass that let a stale cp312 venv carry into a layout change. Ref ROCMAI-439 Signed-off-by: Juho Vainio --- apps/rocm/src/therock.rs | 206 ++++++++++++-- docs/vllm.md | 47 ++-- engines/vllm/src/install.rs | 262 +++++++++++++++--- .../features/therock_next_generation.feature | 9 +- tests/e2e-cucumber/tests/e2e/therock_steps.rs | 23 ++ 5 files changed, 459 insertions(+), 88 deletions(-) diff --git a/apps/rocm/src/therock.rs b/apps/rocm/src/therock.rs index d8c236eae..ddc29f3cc 100644 --- a/apps/rocm/src/therock.rs +++ b/apps/rocm/src/therock.rs @@ -53,6 +53,9 @@ const THEROCK_NEXT_LAYOUT_GENERATION: &str = "next-v1"; /// therefore the only thing that selects [`SourceLayout::Next`]. const THEROCK_NEXT_MIN_MAJOR: u32 = 10; const DEFAULT_MANAGED_PYTHON_VERSION: &str = "3.12"; +/// Interpreter `uv` provisions for a [`SourceLayout::Next`] install. See +/// [`python_requirement`] for why this differs from the canonical default. +const NEXT_MANAGED_PYTHON_VERSION: &str = "3.14"; const STARTUP_UPDATE_CHECK_INTERVAL_MS: u128 = 12 * 60 * 60 * 1_000; const STARTUP_UPDATE_CHECK_TIMEOUT_SECS: u64 = 2; /// Timeout for the best-effort HEAD probe that sizes a download before starting it. @@ -1482,18 +1485,24 @@ fn resolve_latest_for_manifest( let layout = manifest_source_layout(manifest)?; match manifest.format.as_str() { "wheel" => { + // A runtime updated across a layout change needs the new layout's + // interpreter, so the recorded one is re-checked rather than trusted + // on `is_file()` alone; otherwise a canonical cp312 venv would carry + // into a next-layout update and resolve nothing. + let requirement = python_requirement(layout); let manifest_python = manifest .python_executable .as_deref() .map(PathBuf::from) .filter(|path| path.is_file()) + .filter(|path| python_launcher_install_ready(path, requirement).is_ok()) .map(|executable| PythonLauncher { executable, source: "manifest", }); let python_executable = match manifest_python { Some(python) => python, - None => resolve_python_launcher(paths)?, + None => resolve_python_launcher(paths, requirement)?, }; let wheel_compatibility = wheel_compatibility_for_python(&python_executable.executable)?; @@ -1724,11 +1733,16 @@ fn install_wheel_runtime( device_target: device_target_override, layout: layout_override, } = source_override; + // The interpreter has to be picked before `resolve_pip_runtime`, which + // selects wheels using that interpreter's own tags, so the requirement is + // read from the *request* rather than from the resolved version. + let python_requirement = + python_requirement(requested_source_layout(layout_override, version_selector)); progress_line(format!( "Checking Python for the ROCm install; if needed, ROCm CLI will prepare Python {}.", - managed_python_version() + managed_python_version(python_requirement) )); - let python_launcher = resolve_python_launcher(paths)?; + let python_launcher = resolve_python_launcher(paths, python_requirement)?; progress_line(match python_launcher.source { "path" => format!( "Using Python from PATH: {}.", @@ -5699,15 +5713,66 @@ fn managed_python_bootstrap_disabled() -> bool { }) } -fn managed_python_version() -> String { +/// The interpreter a runtime's wheels require. +/// +/// ROCm >= [`THEROCK_NEXT_MIN_MAJOR`] publishes its framework wheels (notably +/// vLLM, flash-attn and amd-aiter) for one CPython version only, and it is not +/// the one the canonical stream uses. Both fields move together, so they travel +/// together: `version` is what `uv` provisions, `tag` is what an already-present +/// interpreter is checked against. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct PythonRequirement { + tag: &'static str, + version: &'static str, +} + +/// The interpreter required to install from `layout`. +/// +/// Keyed on the layout rather than on a resolved ROCm version because the +/// interpreter has to be chosen *before* resolution: `resolve_pip_runtime` picks +/// wheels using the interpreter's own tags, so the resolved version is not +/// available yet. The layout is an honest stand-in, because +/// [`next_layout_requested`] means an explicit stable pin at ROCm >= +/// [`THEROCK_NEXT_MIN_MAJOR`] is the only way to reach `Next` at all. +const fn python_requirement(layout: SourceLayout) -> PythonRequirement { + match layout { + SourceLayout::Next => PythonRequirement { + tag: "cp314", + version: NEXT_MANAGED_PYTHON_VERSION, + }, + SourceLayout::Canonical => PythonRequirement { + tag: "cp312", + version: DEFAULT_MANAGED_PYTHON_VERSION, + }, + } +} + +/// The layout an install or update will read from, as far as is knowable before +/// an interpreter has been chosen. An update passes its manifest's recorded +/// layout; a fresh install passes its request and lets the pin decide. +fn requested_source_layout( + layout_override: Option, + version_selector: Option<&RuntimeVersionSelector>, +) -> SourceLayout { + match layout_override { + Some(layout) => layout, + None if version_selector.is_some_and(next_layout_requested) => SourceLayout::Next, + None => SourceLayout::Canonical, + } +} + +fn managed_python_version(requirement: PythonRequirement) -> String { std::env::var("ROCM_CLI_MANAGED_PYTHON_VERSION") .ok() .filter(|value| !value.trim().is_empty()) - .unwrap_or_else(|| DEFAULT_MANAGED_PYTHON_VERSION.to_owned()) + .unwrap_or_else(|| requirement.version.to_owned()) } -fn ensure_managed_python(paths: &AppPaths) -> Result { - let version = managed_python_version(); +fn ensure_managed_python( + paths: &AppPaths, + requirement: PythonRequirement, +) -> Result { + let version = managed_python_version(requirement); progress_line(format!("Preparing Python {version}...")); let uv = ensure_uv_binary(paths)?; @@ -5716,7 +5781,7 @@ fn ensure_managed_python(paths: &AppPaths) -> Result { if let Ok(Some(manifest)) = load_managed_python_manifest(paths) && manifest.version == version && manifest.executable.is_file() - && python_launcher_install_ready(&manifest.executable).is_ok() + && python_launcher_install_ready(&manifest.executable, requirement).is_ok() { progress_line(format!( "Using existing Python {version} at {}.", @@ -5779,9 +5844,9 @@ fn ensure_managed_python(paths: &AppPaths) -> Result { ); } - python_launcher_install_ready(&executable).with_context(|| { + python_launcher_install_ready(&executable, requirement).with_context(|| { format!( - "Python {version} at {} could not create a virtual environment", + "Python {version} at {} is not usable for this ROCm install", executable.display() ) })?; @@ -5828,13 +5893,20 @@ impl PythonResolverEnv { } } -fn resolve_python_launcher(paths: &AppPaths) -> Result { - resolve_python_launcher_in(paths, &PythonResolverEnv::from_process_env()) +fn resolve_python_launcher( + paths: &AppPaths, + requirement: PythonRequirement, +) -> Result { + resolve_python_launcher_in(paths, &PythonResolverEnv::from_process_env(), requirement) } -fn resolve_python_launcher_in(paths: &AppPaths, env: &PythonResolverEnv) -> Result { +fn resolve_python_launcher_in( + paths: &AppPaths, + env: &PythonResolverEnv, + requirement: PythonRequirement, +) -> Result { if let Some(value) = env.python_override.as_deref() { - python_launcher_install_ready(Path::new(value)) + python_launcher_install_ready(Path::new(value), requirement) .with_context(|| format!("ROCM_CLI_PYTHON is not usable for ROCm setup: {value}"))?; return Ok(PythonLauncher { executable: PathBuf::from(value), @@ -5844,7 +5916,7 @@ fn resolve_python_launcher_in(paths: &AppPaths, env: &PythonResolverEnv) -> Resu let mut skipped_path_python = false; for candidate in python_path_candidates(&env.search_dirs) { - match python_launcher_install_ready(&candidate) { + match python_launcher_install_ready(&candidate, requirement) { Ok(()) => { return Ok(PythonLauncher { executable: candidate, @@ -5857,23 +5929,25 @@ fn resolve_python_launcher_in(paths: &AppPaths, env: &PythonResolverEnv) -> Resu } } if skipped_path_python { - progress_line( - "Python from PATH cannot create a virtual environment; using ROCm CLI's managed Python.", - ); + progress_line(format!( + "Python from PATH is not usable for this ROCm install ({} is required); using ROCm CLI's managed Python.", + requirement.tag + )); } if let Some(manifest) = load_managed_python_manifest(paths)? && manifest.executable.is_file() { - if python_launcher_install_ready(&manifest.executable).is_ok() { + if python_launcher_install_ready(&manifest.executable, requirement).is_ok() { return Ok(PythonLauncher { executable: manifest.executable, source: "managed", }); } - progress_line( - "Saved managed Python cannot create a virtual environment; preparing Python again.", - ); + progress_line(format!( + "Saved managed Python is not usable for this ROCm install ({} is required); preparing Python again.", + requirement.tag + )); } if managed_python_bootstrap_disabled() { @@ -5881,7 +5955,7 @@ fn resolve_python_launcher_in(paths: &AppPaths, env: &PythonResolverEnv) -> Resu "unable to locate Python, and managed Python bootstrap is disabled by ROCM_CLI_DISABLE_MANAGED_PYTHON_BOOTSTRAP" ); } - ensure_managed_python(paths) + ensure_managed_python(paths, requirement) } fn python_path_candidates(search_dirs: &[PathBuf]) -> Vec { @@ -5930,15 +6004,16 @@ fn program_path_candidates(program: &str) -> Vec { names } -fn python_launcher_install_ready(program: &Path) -> Result<()> { +fn python_launcher_install_ready(program: &Path, requirement: PythonRequirement) -> Result<()> { let compatibility = wheel_compatibility_for_python(program)?; - if compatibility.python_tag != "cp312" { + if compatibility.python_tag != requirement.tag { bail!( - "Python wheel tag {} is not supported; cp312 is required", - compatibility.python_tag + "Python wheel tag {} is not supported by this ROCm install; {} is required", + compatibility.python_tag, + requirement.tag ); } - verify_python_can_create_venv(program) + verify_python_can_create_venv(program).context("Python could not create a virtual environment") } fn verify_python_can_create_venv(program: &Path) -> Result<()> { @@ -6867,8 +6942,8 @@ mod tests { // python is a non-executable stub fails // `python_launcher_install_ready`, so `resolve_python_launcher_in` // falls back to the managed-Python path and calls - // `progress_line("Python from PATH cannot create a virtual - // environment; using ROCm CLI's managed Python.")` before bailing out + // `progress_line("Python from PATH is not usable for this ROCm + // install ...; using ROCm CLI's managed Python.")` before bailing out // (managed bootstrap is disabled here, so the whole thing stays // network-free). If `render_update_json` stopped installing the // guard, that call would land in `PROGRESS_LINE_SINK` instead of @@ -8729,6 +8804,75 @@ mod tests { assert_eq!(DEFAULT_MANAGED_PYTHON_VERSION, "3.12"); } + /// ROCm 10.x framework wheels (vLLM, flash-attn, amd-aiter) are published + /// cp314-only, so the next layout must provision 3.14 while everything else + /// stays on 3.12. + #[test] + fn python_requirement_follows_the_source_layout() { + let canonical = python_requirement(SourceLayout::Canonical); + assert_eq!(canonical.tag, "cp312"); + assert_eq!(canonical.version, "3.12"); + + let next = python_requirement(SourceLayout::Next); + assert_eq!(next.tag, "cp314"); + assert_eq!(next.version, "3.14"); + } + + /// The requirement is read before `resolve_pip_runtime` runs, so it has to + /// be derivable from the request alone. Only an explicit stable ROCm >= 10 + /// pin reaches the next layout; a build-date pin, an older pin, or no + /// selector at all must stay canonical so nothing that installs today + /// starts demanding a different interpreter. + #[test] + fn requested_source_layout_reads_the_request_not_the_resolution() { + let next_pin = RuntimeVersionSelector::Version("10.1.0".to_owned()); + assert_eq!( + requested_source_layout(None, Some(&next_pin)), + SourceLayout::Next + ); + + let old_pin = RuntimeVersionSelector::Version("7.2.3".to_owned()); + assert_eq!( + requested_source_layout(None, Some(&old_pin)), + SourceLayout::Canonical + ); + assert_eq!(requested_source_layout(None, None), SourceLayout::Canonical); + + // An update supplies its manifest's layout, which wins over the (absent) + // selector so the runtime keeps the interpreter its stream needs. + assert_eq!( + requested_source_layout(Some(SourceLayout::Next), None), + SourceLayout::Next + ); + } + + /// A tag mismatch and a broken venv are different failures with different + /// fixes, so the gate must not report one as the other. + #[test] + fn python_gate_rejects_the_wrong_tag_without_blaming_the_venv() -> Result<()> { + if runtime_is_windows() { + return Ok(()); + } + let (root, _paths) = test_paths("python-gate-tag-mismatch"); + let bin_dir = root.join("bin"); + fs::create_dir_all(&bin_dir)?; + // A real, working cp312 interpreter: the venv probe would succeed. + let python = write_fake_python_with_venv(&bin_dir, "python")?; + + python_launcher_install_ready(&python, python_requirement(SourceLayout::Canonical)) + .expect("a cp312 python satisfies a canonical install"); + + let err = python_launcher_install_ready(&python, python_requirement(SourceLayout::Next)) + .expect_err("a cp312 python cannot serve a next-layout install"); + let rendered = format!("{err:#}"); + assert!(rendered.contains("cp314"), "{rendered}"); + assert!( + !rendered.contains("virtual environment"), + "a tag mismatch must not be reported as a venv failure: {rendered}" + ); + Ok(()) + } + #[test] fn managed_python_manifest_round_trips() -> Result<()> { let (root, paths) = test_paths("managed-python-manifest"); @@ -8783,6 +8927,7 @@ mod tests { python_override: None, search_dirs: vec![bin_dir], }, + python_requirement(SourceLayout::Canonical), )?; assert_eq!(launcher.source, "path"); assert!( @@ -8884,6 +9029,7 @@ mod tests { python_override: None, search_dirs: vec![bin_dir], }, + python_requirement(SourceLayout::Canonical), )?; assert_eq!(launcher.source, "path"); diff --git a/docs/vllm.md b/docs/vllm.md index 75c32059b..55eb84110 100644 --- a/docs/vllm.md +++ b/docs/vllm.md @@ -64,22 +64,37 @@ For most ROCm SDK versions, `rocm engines install vllm` pins a fixed vLLM wheel and index URL. Any ROCm SDK 10.x version is different: AMD publishes vLLM, flash-attn, and amd-aiter there under a rotating dev-tag filename (for example `vllm-0.27.1.dev5+rocm10.0.0.gf46a9dfe2.d20260826-cp314-cp314-linux_x86_64.whl`), -so there is no fixed filename to pin in the adapter. This route is selected by -major version alone, so `10.0.0`, `10.1.0`, and any other `10.x` all discover -through it rather than only `10.0.0`. - -Instead, the install resolves each package's current wheel from AMD's index -with `uv pip install --dry-run --reinstall`, parses the version it reports it -would install, then reinstalls pinned to that exact version (torch and -tensorizer stay pinned as usual). If AMD's index has no compatible build for a -package, the resolver fails and the install fails rather than falling back to -an unpinned or CPU install. Every other ROCm SDK version, including 7.2.3, -keeps using the static pin table; an SDK version with no matching row there -falls back to the table's default pin, *unless* its major release matches a -discovery-table entry, in which case guessing the default pin would very -likely install an ABI-incompatible build, so the install fails closed instead -with a message naming the detected version and pointing at -`ROCM_CLI_VLLM_ROCM_INDEX_URL` as the way to install anyway. +so there is no fixed filename to pin in the adapter. The wheel's own tag shows +the other thing that changes on this route: ROCm 10.x's vLLM, flash-attn, and +amd-aiter are published for **`cp314` only**, unlike every earlier ROCm SDK +version's `cp312` wheels. ROCm CLI provisions a matching interpreter for a +10.x install automatically; a Python already on `PATH` or already installed +as ROCm CLI's managed Python is rejected if it is not `cp314`, with a message +naming the required tag. + +This route is selected by `major.minor`, so `10.0.0` and `10.1.0` discover +through *different* rows: they use different index URLs and different vLLM +minors, because AMD stages ROCm 10.1's frameworks on a separate host from +10.0's production index. Patch and any dev/pre-release suffix are still +ignored within a row, since AMD rotates those constantly. + +The install resolves each package's current wheel, including torch, from the +row's index with `uv pip install --dry-run --reinstall`, parses the version it +reports it would install, then reinstalls pinned to that exact version +(tensorizer is the one package that stays a plain literal pin, since it has no +ROCm-specific build to discover). Every resolved pin that carries a +`+rocmX.Y` local version is checked against the SDK's own major.minor before +installing, so an index that happens to serve more than one ROCm line at once +cannot silently install the wrong line's wheel onto this SDK. If AMD's index +has no compatible build for a package, the resolver fails and the install +fails rather than falling back to an unpinned or CPU install. Every other +ROCm SDK version, including 7.2.3, keeps using the static pin table; an SDK +version with no matching row there falls back to the table's default pin, +*unless* its major release matches a discovery-table entry, in which case +guessing the default pin would very likely install an ABI-incompatible build, +so the install fails closed instead with a message naming the detected +version and pointing at `ROCM_CLI_VLLM_ROCM_INDEX_URL` as the way to install +anyway. Supported discovery paths: diff --git a/engines/vllm/src/install.rs b/engines/vllm/src/install.rs index ce17e327d..425eee7ab 100644 --- a/engines/vllm/src/install.rs +++ b/engines/vllm/src/install.rs @@ -49,47 +49,86 @@ const VLLM_ROCM_INDEX_PREFIX: &str = "https://wheels.vllm.ai/rocm"; /// pinned in [`VLLM_ROCM_BUILD_TABLE`]. pub(crate) struct VllmRocmDiscoverBuild { rocm_sdk_version: &'static str, + /// PEP 425 interpreter tag this row's wheels are published for. AMD builds + /// ROCm 10.x `vllm`, `flash-attn` and `amd-aiter` for exactly one CPython + /// version, so a venv on any other interpreter resolves nothing at all. + python_tag: &'static str, + /// Index publishing this row's rotating-dev-tag vLLM/flash-attn/amd-aiter + /// wheels. Per-row because ROCm 10.1 frameworks are staged on a different + /// host than 10.0's. + vllm_index_url: &'static str, + /// Index publishing this row's torch build. + torch_index_url: &'static str, vllm_version_prefix: &'static str, flash_attn_version_prefix: &'static str, amd_aiter_version_prefix: &'static str, - torch_requirement: &'static str, + torch_version_prefix: &'static str, + /// Plain PyPI pin: tensorizer publishes no ROCm-specific build, so there is + /// nothing to discover. tensorizer_requirement: &'static str, } -const VLLM_ROCM_DISCOVER_BUILD_TABLE: &[VllmRocmDiscoverBuild] = &[VllmRocmDiscoverBuild { - rocm_sdk_version: "10.0.0", - vllm_version_prefix: "0.27", - flash_attn_version_prefix: "2.8", - amd_aiter_version_prefix: "0.1", - torch_requirement: "torch==2.12.0+rocm10.0.0", - tensorizer_requirement: "tensorizer==2.12.1", -}]; -/// Index that publishes the rotating-dev-tag vLLM/flash-attn/amd-aiter -/// wheels for [`VLLM_ROCM_DISCOVER_BUILD_TABLE`] rows. -const VLLM_ROCM_DISCOVER_INDEX_URL: &str = "https://rocm.frameworks.amd.com/whl-multi-arch/vllm/"; -/// Index that publishes the pinned torch build for -/// [`VLLM_ROCM_DISCOVER_BUILD_TABLE`] rows. -const VLLM_ROCM_DISCOVER_TORCH_INDEX_URL: &str = "https://stable.repo.amd.com/rocm/whl-next/"; +const VLLM_ROCM_DISCOVER_BUILD_TABLE: &[VllmRocmDiscoverBuild] = &[ + VllmRocmDiscoverBuild { + rocm_sdk_version: "10.0.0", + python_tag: "cp314", + vllm_index_url: "https://rocm.frameworks.amd.com/whl-multi-arch/vllm/", + torch_index_url: "https://stable.repo.amd.com/rocm/whl-next/", + vllm_version_prefix: "0.27", + flash_attn_version_prefix: "2.8", + amd_aiter_version_prefix: "0.1", + torch_version_prefix: "2.12", + tensorizer_requirement: "tensorizer==2.12.1", + }, + VllmRocmDiscoverBuild { + rocm_sdk_version: "10.1.0", + python_tag: "cp314", + vllm_index_url: "https://rocm.frameworks-prereleases.amd.com/whl-multi-arch-staging/vllm/", + torch_index_url: "https://rocm.frameworks-prereleases.amd.com/whl-multi-arch-staging/", + vllm_version_prefix: "0.29", + flash_attn_version_prefix: "2.8", + amd_aiter_version_prefix: "0.1", + torch_version_prefix: "2.12", + tensorizer_requirement: "tensorizer==2.12.1", + }, +]; /// Looks up the discovery build recipe for a ROCm SDK version, if any. /// -/// Matched on major version only: unlike [`VLLM_ROCM_BUILD_TABLE`], where a -/// row pins one exact release's wheel filename, a discover row is a live -/// resolver recipe AMD's index applies uniformly across an entire ROCm major -/// line. AMD's preview wheels are tagged with the real target release -/// (`whl-multi-arch/torch/` carries `+rocm7.13.0`, `+rocm7.14.0`, and -/// `+rocm7.14.1` as genuinely distinct, coexisting builds), so once ROCm -/// 10.x's target moves past `10.0.0` the same rotation will happen here; a -/// row keyed to an exact string would then silently stop matching. `10.0.0` -/// and `10.1.0` should both discover through the same `"10.0.0"` row. This -/// intentionally differs from `apps/rocm/src/therock.rs`'s SDK layout -/// selection, which avoids major-only gating for unrelated reasons (on-disk -/// layout, not wheel availability). +/// Matched on `major.minor`, ignoring patch and any dev/pre-release suffix. A +/// discover row is a live resolver recipe rather than one pinned filename, so +/// it does span a release line, but only a `major.minor` one: 10.0 and 10.1 +/// publish genuinely different wheels (vLLM `0.27` from +/// `rocm.frameworks.amd.com` versus `0.29` from the staging host), so a +/// major-only row would serve one SDK's wheels to the other. Patch is still +/// ignored, because AMD does rotate the patch and dev tag within a line +/// (`whl-multi-arch/torch/` carries `+rocm7.13.0`, `+rocm7.14.0` and +/// `+rocm7.14.1` as coexisting builds) and a row keyed to an exact string +/// would silently stop matching. An unknown line matches nothing and fails +/// closed in [`resolve_vllm_install_target`] rather than falling back to a +/// stale static pin. pub(crate) fn vllm_rocm_discover_build( rocm_sdk_version: &str, ) -> Option<&'static VllmRocmDiscoverBuild> { VLLM_ROCM_DISCOVER_BUILD_TABLE .iter() - .find(|build| rocm_sdk_major_matches(rocm_sdk_version, build.rocm_sdk_version)) + .find(|build| rocm_sdk_series_matches(rocm_sdk_version, build.rocm_sdk_version)) +} +/// Whether `recorded` and `table_key` name the same ROCm `major.minor` line, +/// ignoring patch and any dev/pre-release suffix. See +/// [`vllm_rocm_discover_build`] for why the line, not the exact release, is +/// what a discover row covers. +fn rocm_sdk_series_matches(recorded: &str, table_key: &str) -> bool { + fn series(version: &str) -> Option<(u64, u64)> { + let mut parts = version.trim().split('.'); + let major = parts.next()?.parse().ok()?; + let minor: String = parts + .next()? + .chars() + .take_while(char::is_ascii_digit) + .collect(); + Some((major, minor.parse().ok()?)) + } + series(recorded).is_some_and(|version| Some(version) == series(table_key)) } /// Whether `recorded` (a runtime manifest's live `rocm_sdk.__version__` probe) /// and `table_key` (a literal key in [`VLLM_ROCM_DISCOVER_BUILD_TABLE`]) share @@ -611,6 +650,7 @@ fn dry_run_resolved_pin(stdout: &str, pkg: &str) -> Option { fn vllm_rocm10_discover_install_args( python: &Path, reinstall: bool, + build: &VllmRocmDiscoverBuild, pins: &[String], ) -> Vec { let mut args = uv_pip_install_base(python); @@ -621,11 +661,81 @@ fn vllm_rocm10_discover_install_args( args.push("--prerelease".to_owned()); args.push("allow".to_owned()); args.push("--extra-index-url".to_owned()); - args.push(VLLM_ROCM_DISCOVER_INDEX_URL.to_owned()); + args.push(build.vllm_index_url.to_owned()); args.push("--extra-index-url".to_owned()); - args.push(VLLM_ROCM_DISCOVER_TORCH_INDEX_URL.to_owned()); + args.push(build.torch_index_url.to_owned()); args } +/// Rejects a discovered pin whose `+rocm` local version names a +/// different ROCm line than the SDK being installed for. +/// +/// Discovery constrains the *release* (`vllm==0.29.*`) but cannot constrain the +/// local version, because PEP 440 has no local-version wildcard. Both discovery +/// indexes serve more than one ROCm line at once (staging carries 10.0 and 10.1 +/// vLLM side by side), so without this a mis-stocked or re-pointed index could +/// resolve a 10.0 wheel onto a 10.1 SDK and only fail at import time. +/// `flash-attn` and `amd-aiter` publish no rocm local version at all, so a pin +/// without one passes. +fn ensure_rocm_local_version_matches(pin: &str, rocm_sdk_version: &str) -> Result<()> { + let Some((_, version)) = pin.split_once("==") else { + return Ok(()); + }; + let Some(local) = split_local_version(version).1 else { + return Ok(()); + }; + let Some(local_rocm) = local.strip_prefix("rocm") else { + return Ok(()); + }; + if rocm_sdk_series_matches(local_rocm, rocm_sdk_version) { + return Ok(()); + } + bail!( + "discovered `{pin}` for ROCm SDK {rocm_sdk_version}, but its `+{local}` build targets a \ + different ROCm release; the index for this ROCm line is serving another SDK's wheels" + ) +} +/// The PEP 425 interpreter tag of `python`, e.g. `cp314`. +fn python_interpreter_tag(python: &Path) -> Result { + let output = ProcessCommand::new(python) + .args([ + "-c", + "import sys; print(f'cp{sys.version_info[0]}{sys.version_info[1]}')", + ]) + .output() + .with_context(|| format!("failed to launch {} to read its version", python.display()))?; + if !output.status.success() { + bail!( + "{} could not report its version: {}", + python.display(), + String::from_utf8_lossy(&output.stderr).trim() + ); + } + Ok(String::from_utf8_lossy(&output.stdout).trim().to_owned()) +} +/// Fails before discovery when `python` is not the interpreter this row's +/// wheels are built for. +/// +/// Checked up front rather than left to `uv`: every package in the row is +/// published for one tag only, so a mismatched interpreter surfaces as three +/// separate "no compatible version found" resolver dumps that name the +/// requirement but never the interpreter, which is the thing to fix. +fn ensure_discover_python_tag(python: &Path, build: &VllmRocmDiscoverBuild) -> Result<()> { + let tag = python_interpreter_tag(python)?; + if tag == build.python_tag { + return Ok(()); + } + bail!( + "vLLM for ROCm {} is published for {} only, but {} is {}.\n\ + Re-create this runtime's Python environment on {}, for example by re-running \ + `rocm install sdk --version {}` so ROCm CLI provisions a matching interpreter.", + build.rocm_sdk_version, + build.python_tag, + python.display(), + tag, + build.python_tag, + build.rocm_sdk_version + ) +} /// Discovers and installs the current vLLM/flash-attn/amd-aiter wheels for a /// [`VllmRocmDiscoverBuild`] row, pinning each to the exact version `uv pip /// install --dry-run` resolved so the real install can never silently drift @@ -637,11 +747,20 @@ fn install_vllm_rocm10_discover( reinstall: bool, build: &VllmRocmDiscoverBuild, ) -> Result> { + ensure_discover_python_tag(python, build)?; + let torch = discover_pinned_requirement( + uv, + paths, + python, + build.torch_index_url, + "torch", + build.torch_version_prefix, + )?; let vllm = discover_pinned_requirement( uv, paths, python, - VLLM_ROCM_DISCOVER_INDEX_URL, + build.vllm_index_url, "vllm", build.vllm_version_prefix, )?; @@ -649,7 +768,7 @@ fn install_vllm_rocm10_discover( uv, paths, python, - VLLM_ROCM_DISCOVER_INDEX_URL, + build.vllm_index_url, "flash-attn", build.flash_attn_version_prefix, )?; @@ -657,20 +776,23 @@ fn install_vllm_rocm10_discover( uv, paths, python, - VLLM_ROCM_DISCOVER_INDEX_URL, + build.vllm_index_url, "amd-aiter", build.amd_aiter_version_prefix, )?; let pins = vec![ - build.torch_requirement.to_owned(), + torch, vllm, flash_attn, amd_aiter, build.tensorizer_requirement.to_owned(), ]; + for pin in &pins { + ensure_rocm_local_version_matches(pin, build.rocm_sdk_version)?; + } - let args = vllm_rocm10_discover_install_args(python, reinstall, &pins); + let args = vllm_rocm10_discover_install_args(python, reinstall, build, &pins); let output = ProcessCommand::new(uv) .args(args) .envs(uv_command_env(paths)) @@ -1429,18 +1551,80 @@ mod tests { "tensorizer==2.12.1".to_owned(), ]; let python = PathBuf::from("/opt/venv/bin/python"); + let build = vllm_rocm_discover_build("10.0.0").expect("10.0.0 has a discover row"); - let args = vllm_rocm10_discover_install_args(&python, false, &pins); + let args = vllm_rocm10_discover_install_args(&python, false, build, &pins); assert!(!args.contains(&"--reinstall".to_owned())); for pin in &pins { assert!(args.contains(pin), "{args:?} should contain {pin}"); } assert!(args.contains(&"--prerelease".to_owned())); assert!(args.contains(&"allow".to_owned())); - assert!(args.contains(&VLLM_ROCM_DISCOVER_INDEX_URL.to_owned())); - assert!(args.contains(&VLLM_ROCM_DISCOVER_TORCH_INDEX_URL.to_owned())); + assert!(args.contains(&build.vllm_index_url.to_owned())); + assert!(args.contains(&build.torch_index_url.to_owned())); - let args = vllm_rocm10_discover_install_args(&python, true, &pins); + let args = vllm_rocm10_discover_install_args(&python, true, build, &pins); assert!(args.contains(&"--reinstall".to_owned())); } + /// The 10.1 row must reach the staging host for both vLLM and torch; 10.0 + /// must keep using the production frameworks index and `whl-next`. This is + /// the offline half of the 10.1 verification: no ROCm 10.1 SDK is published + /// yet, so the live install cannot be exercised. + #[test] + fn discover_rows_select_their_own_indexes() { + let ten_zero = vllm_rocm_discover_build("10.0.0").expect("10.0.0 has a discover row"); + assert_eq!( + ten_zero.vllm_index_url, + "https://rocm.frameworks.amd.com/whl-multi-arch/vllm/" + ); + assert_eq!( + ten_zero.torch_index_url, + "https://stable.repo.amd.com/rocm/whl-next/" + ); + assert_eq!(ten_zero.vllm_version_prefix, "0.27"); + + // A real probe carries a patch and dev suffix the table key never does. + let ten_one = vllm_rocm_discover_build("10.1.0a20260928").expect("10.1 has a discover row"); + assert_eq!( + ten_one.vllm_index_url, + "https://rocm.frameworks-prereleases.amd.com/whl-multi-arch-staging/vllm/" + ); + assert_eq!( + ten_one.torch_index_url, + "https://rocm.frameworks-prereleases.amd.com/whl-multi-arch-staging/" + ); + assert_eq!(ten_one.vllm_version_prefix, "0.29"); + + // An unpublished line must not borrow another line's wheels. + assert!(vllm_rocm_discover_build("10.2.0").is_none()); + assert!(vllm_rocm_discover_build("7.2.3").is_none()); + } + /// An unknown 10.x line fails closed rather than silently resolving the + /// static table's 7.2.3 pin, which is the whole point of keeping + /// [`rocm_sdk_major_matches`] alongside the narrower row lookup. + #[test] + fn unknown_rocm10_line_still_fails_closed() { + let err = install_target(None, Some("10.2.0")).expect_err("10.2.0 has no vLLM build"); + let rendered = format!("{err:#}"); + assert!( + !rendered.contains("7.2.3"), + "should not fall back to the static pin: {rendered}" + ); + } + #[test] + fn rocm_local_version_guard_rejects_another_lines_wheel() { + // No local version at all (flash-attn, amd-aiter) is fine. + assert!(ensure_rocm_local_version_matches("flash-attn==2.8.3", "10.1.0").is_ok()); + // Same line, different patch and rc suffix, is fine. + assert!(ensure_rocm_local_version_matches("torch==2.12.0+rocm10.1.0rc3", "10.1.0").is_ok()); + // A 10.0 wheel resolved for a 10.1 SDK is not. + let err = ensure_rocm_local_version_matches( + "vllm==0.27.1.dev5+rocm10.0.0.gf46a9dfe2.d20260826", + "10.1.0", + ) + .expect_err("a 10.0 wheel must not install onto a 10.1 SDK"); + let rendered = format!("{err:#}"); + assert!(rendered.contains("10.1.0"), "{rendered}"); + assert!(rendered.contains("rocm10.0.0"), "{rendered}"); + } } diff --git a/tests/e2e-cucumber/features/therock_next_generation.feature b/tests/e2e-cucumber/features/therock_next_generation.feature index fe4362cb5..95fd943c9 100644 --- a/tests/e2e-cucumber/features/therock_next_generation.feature +++ b/tests/e2e-cucumber/features/therock_next_generation.feature @@ -132,9 +132,11 @@ Feature: TheRock "next" ROCm 10 install layout # `uv pip install --dry-run --reinstall` before installing pinned to what # that reported. No fixture can serve a rotating dev-tag filename and stay # meaningful, so this is the only place that mechanism runs against the - # real index at all. Provisions its own ROCm 10 runtime rather than reusing - # therock-next-07's, so that scenario's arch-detection coverage still runs - # on hosts that can't start vLLM. + # real index at all. Also the only place that proves ROCm CLI provisions the + # cp314 interpreter this route's wheels require, rather than the cp312 + # every earlier SDK version uses. Provisions its own ROCm 10 runtime rather + # than reusing therock-next-07's, so that scenario's arch-detection coverage + # still runs on hosts that can't start vLLM. @id:therock-next-09-live-install-reports-vllm-rocm10x-discovery-pins @requires-gpu @requires-engine:vllm @nightly Scenario: therock-next-09 - Installing vLLM against a live ROCm 10 preview runtime reports the discovery pins Given a machine with no CLI-managed runtimes @@ -142,6 +144,7 @@ Feature: TheRock "next" ROCm 10 install layout Then a runtime is registered And the runtime is set as active And the runtime includes an inference engine + And the ROCm 10 runtime provisioned a cp314 Python interpreter When the user reinstalls vllm Then the install reports the vLLM ROCm 10.x discovery pins diff --git a/tests/e2e-cucumber/tests/e2e/therock_steps.rs b/tests/e2e-cucumber/tests/e2e/therock_steps.rs index 07e14ae45..abf58d36b 100644 --- a/tests/e2e-cucumber/tests/e2e/therock_steps.rs +++ b/tests/e2e-cucumber/tests/e2e/therock_steps.rs @@ -1085,6 +1085,29 @@ print(json.dumps({"rocm_sdk": rocm_sdk.__version__, "torch": torch.__version__, ); } +#[then("the ROCm 10 runtime provisioned a cp314 Python interpreter")] +async fn assert_next_runtime_provisioned_cp314(world: &mut E2eWorld) { + let install_output = stdout(world); + let python = install_output + .lines() + .find_map(|line| line.trim().strip_prefix("python_executable: ")) + .expect("live install did not report its managed Python executable"); + let result = std::process::Command::new(python) + .args([ + "-c", + "import sys; print(f'cp{sys.version_info[0]}{sys.version_info[1]}')", + ]) + .output() + .unwrap_or_else(|error| panic!("failed to launch managed Python {python}: {error}")); + assert!(result.status.success(), "failed to read {python}'s tag"); + let tag = String::from_utf8_lossy(&result.stdout).trim().to_owned(); + assert_eq!( + tag, "cp314", + "ROCm 10.x's vLLM/flash-attn/amd-aiter wheels are cp314-only, but the runtime \ + provisioned {python} as {tag}" + ); +} + #[when("the user reinstalls vllm")] async fn user_reinstalls_vllm(world: &mut E2eWorld) { // `--reinstall` so the adapter's ROCm 10.x discovery route (dry-run From 1e0a295e902ddc4e99306dc96810e3fa15f5cbbd Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Thu, 1 Oct 2026 08:45:33 +0300 Subject: [PATCH 2/9] fix(vllm): gate windows-only test compilation with cfg(unix) python_gate_rejects_the_wrong_tag_without_blaming_the_venv called the cfg(unix)-gated write_fake_python_with_venv helper but only carried a runtime guard, which doesn't stop Windows from trying to compile the call to a function that doesn't exist there. Signed-off-by: Juho Vainio --- apps/rocm/src/therock.rs | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/apps/rocm/src/therock.rs b/apps/rocm/src/therock.rs index ddc29f3cc..12be7b1e3 100644 --- a/apps/rocm/src/therock.rs +++ b/apps/rocm/src/therock.rs @@ -8848,11 +8848,9 @@ mod tests { /// A tag mismatch and a broken venv are different failures with different /// fixes, so the gate must not report one as the other. + #[cfg(unix)] #[test] fn python_gate_rejects_the_wrong_tag_without_blaming_the_venv() -> Result<()> { - if runtime_is_windows() { - return Ok(()); - } let (root, _paths) = test_paths("python-gate-tag-mismatch"); let bin_dir = root.join("bin"); fs::create_dir_all(&bin_dir)?; From 3e716576006f46fb75370e6de66ed623dd9bdc72 Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Thu, 1 Oct 2026 09:31:29 +0300 Subject: [PATCH 3/9] fix(vllm): validate channel/arch before provisioning Python 3.14 A request select_source_layout would reject anyway (wrong channel, or a grouped family with no exact arch) used to fail before any work happened. Since the interpreter requirement started depending on layout, the Python provisioning step ran first, so the same doomed request now pays for a full `uv python install` of 3.14 before failing. Hoist the validation ahead of interpreter resolution. Also note the managed-Python manifest's single-slot cache thrashes when alternating between canonical and Next layouts, and fix a test comment left describing the pre-PR major-only matching semantics. Signed-off-by: Juho Vainio --- apps/rocm/src/therock.rs | 27 +++++++++++++++++++++------ engines/vllm/src/install.rs | 6 +++--- 2 files changed, 24 insertions(+), 9 deletions(-) diff --git a/apps/rocm/src/therock.rs b/apps/rocm/src/therock.rs index 12be7b1e3..8923f2349 100644 --- a/apps/rocm/src/therock.rs +++ b/apps/rocm/src/therock.rs @@ -1733,6 +1733,22 @@ fn install_wheel_runtime( device_target: device_target_override, layout: layout_override, } = source_override; + // `device_target_override` (an update apply's exact recorded arch) is only + // otherwise consulted after `resolve_pip_runtime` returns — too late for the + // channel/arch check below, which needs an exact arch before it can run at + // all. Recover it into the family override up front, same as `resolve_pip_runtime` + // does internally. + let recovered_family_override = family_override + .map(|family| family_override_or_recovered_arch(family, device_target_override)); + // Fail fast on a request `resolve_pip_runtime` could only ever reject (wrong + // channel, or a grouped family with no exact arch) before paying for Python + // provisioning below, which can mean a real download for a layout this host + // or channel can never actually use. + select_source_layout( + channel, + &resolve_family(paths, recovered_family_override.as_deref())?, + version_selector, + )?; // The interpreter has to be picked before `resolve_pip_runtime`, which // selects wheels using that interpreter's own tags, so the requirement is // read from the *request* rather than from the resolved version. @@ -1766,12 +1782,6 @@ fn install_wheel_runtime( "Checking TheRock {} packages for this AMD GPU...", channel.as_str() )); - // `device_target_override` (an update apply's exact recorded arch) is only - // otherwise consulted below, after `resolve_pip_runtime` returns — too late - // for the next layout, which needs an exact arch before it can query - // package metadata at all. Recover it into the family override up front. - let recovered_family_override = family_override - .map(|family| family_override_or_recovered_arch(family, device_target_override)); let resolution = resolve_pip_runtime( paths, channel, @@ -5778,6 +5788,11 @@ fn ensure_managed_python( let uv = ensure_uv_binary(paths)?; // Check the manifest first — if the recorded executable is still usable, skip the install. + // ponytail: this manifest holds one version at a time, so alternating between a + // canonical (3.12) and a `Next` (3.14) install re-runs `uv python install` on every + // switch instead of keeping both cached — `uv` itself still short-circuits the + // re-download, so this costs a subprocess round trip, not a real reinstall. Upgrade + // to a version-keyed manifest if that alternation turns out to be a common pattern. if let Ok(Some(manifest)) = load_managed_python_manifest(paths) && manifest.version == version && manifest.executable.is_file() diff --git a/engines/vllm/src/install.rs b/engines/vllm/src/install.rs index 425eee7ab..102b98489 100644 --- a/engines/vllm/src/install.rs +++ b/engines/vllm/src/install.rs @@ -1491,9 +1491,9 @@ mod tests { assert!(vllm_rocm_discover_build("10.0.0").is_some()); // AMD tags preview wheels with the real target release (verified via // whl-multi-arch/torch/'s coexisting +rocm7.13.0/7.14.0/7.14.1 - // builds), so ROCm 10.x's tag will move past 10.0.0 the same way; - // the discovery recipe must keep firing across the whole major line, - // not just the exact version it happened to be added for. + // builds), so a pre-release suffix on a known major.minor must still + // match its row (here the dedicated 10.1.0 one), not just an exact + // version string. assert!(vllm_rocm_discover_build("10.1.0a20260822").is_some()); assert!(vllm_rocm_discover_build("999.0.0").is_none()); } From 8bd88a0e1077ac1411cc5b7b13dd35eafbdc8269 Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Thu, 1 Oct 2026 14:50:43 +0300 Subject: [PATCH 4/9] fix(therock): provision cp314 for ROCm 10.x auto-select and nightly pins requested_source_layout only routes to SourceLayout::Next (and its cp314 python_requirement) when a stable version is pinned at ROCm >= 10. A plain auto-select or a nightly build-date/prerelease pin on the canonical channel can also land on a ROCm >= 10 build without ever going through Next-layout routing, so python_requirement stayed on cp312 and the install failed against cp314-only wheels. canonical_auto_select_needs_next_python peeks the channel's live rocm package listing and checks the highest matching candidate's major version, so these routes get the cp314 interpreter too. ensure_uv_venv also now checks the existing venv's wheel-compatibility python tag against what's required, not just that python --version succeeds, so a stale cp312 venv from before this fix gets recreated instead of silently reused once cp314 is required. Signed-off-by: Juho Vainio --- apps/rocm/src/therock.rs | 201 ++++++++++++++++++++++++++++++++++++--- 1 file changed, 190 insertions(+), 11 deletions(-) diff --git a/apps/rocm/src/therock.rs b/apps/rocm/src/therock.rs index 8923f2349..a1d2945c1 100644 --- a/apps/rocm/src/therock.rs +++ b/apps/rocm/src/therock.rs @@ -1750,10 +1750,23 @@ fn install_wheel_runtime( version_selector, )?; // The interpreter has to be picked before `resolve_pip_runtime`, which - // selects wheels using that interpreter's own tags, so the requirement is - // read from the *request* rather than from the resolved version. - let python_requirement = - python_requirement(requested_source_layout(layout_override, version_selector)); + // selects wheels using that interpreter's own tags. `requested_source_layout` + // only answers from the *request*: an explicit stable ROCm 10+ pin already + // routes to `Next` and gets `cp314` that way, but a plain auto-select or a + // nightly build-date/prerelease pin always reports `Canonical`, even once + // the canonical/nightly feed's own "latest" has rotated past that major on + // its own, with no `Next` routing involved. `canonical_auto_select_needs_next_python` + // peeks the channel's own `rocm` listing to catch that case. + let python_requirement = match requested_source_layout(layout_override, version_selector) { + SourceLayout::Next => python_requirement(SourceLayout::Next), + SourceLayout::Canonical => { + if canonical_auto_select_needs_next_python(paths, channel, version_selector)? { + python_requirement(SourceLayout::Next) + } else { + python_requirement(SourceLayout::Canonical) + } + } + }; progress_line(format!( "Checking Python for the ROCm install; if needed, ROCm CLI will prepare Python {}.", managed_python_version(python_requirement) @@ -4860,19 +4873,29 @@ fn ensure_uv_venv( ) -> Result<()> { let env_python = venv_python_path(install_root); if env_python.is_file() { - if run_command( + let reusable = run_command( &env_python, &["--version"], "verify existing managed TheRock runtime Python", ) .is_ok() - { + && wheel_compatibility_for_python(&env_python).is_ok_and(|existing| { + wheel_compatibility_for_python(python_launcher) + .is_ok_and(|required| existing.python_tag == required.python_tag) + }); + if reusable { return Ok(()); } - progress_line("Existing Python environment is incomplete; recreating it."); + // Same `install_root` can be reused across reinstalls (it's keyed on + // resolved version/build date, not on the interpreter), so an existing + // venv left over from a lower-tag requirement must not be mistaken for + // a compatible one just because it still runs. + progress_line( + "Existing Python environment does not match the required interpreter; recreating it.", + ); fs::remove_dir_all(install_root).with_context(|| { format!( - "failed to remove incomplete Python environment at {}", + "failed to remove incompatible Python environment at {}", install_root.display() ) })?; @@ -5741,9 +5764,11 @@ struct PythonRequirement { /// Keyed on the layout rather than on a resolved ROCm version because the /// interpreter has to be chosen *before* resolution: `resolve_pip_runtime` picks /// wheels using the interpreter's own tags, so the resolved version is not -/// available yet. The layout is an honest stand-in, because -/// [`next_layout_requested`] means an explicit stable pin at ROCm >= -/// [`THEROCK_NEXT_MIN_MAJOR`] is the only way to reach `Next` at all. +/// available yet. The layout is an honest stand-in only when it was decided +/// from an explicit stable pin (`next_layout_requested`); a fresh install with +/// no pin, or a nightly build-date/prerelease pin, needs +/// [`canonical_auto_select_needs_next_python`] to check whether the channel's +/// own "latest" has rotated past [`THEROCK_NEXT_MIN_MAJOR`] on its own. const fn python_requirement(layout: SourceLayout) -> PythonRequirement { match layout { SourceLayout::Next => PythonRequirement { @@ -5771,6 +5796,51 @@ fn requested_source_layout( } } +/// Whether a `Canonical`-layout install (auto-select, or a nightly +/// build-date/prerelease pin) will actually resolve a ROCm build at major >= +/// [`THEROCK_NEXT_MIN_MAJOR`], regardless of [`VersionStage`]. +/// +/// `requested_source_layout` only answers from the shape of the request: an +/// explicit *stable* pin at ROCm >= `THEROCK_NEXT_MIN_MAJOR` routes to `Next` +/// and gets `cp314` that way, but the canonical/nightly feed can resolve a +/// ROCm >= `THEROCK_NEXT_MIN_MAJOR` build (stable or prerelease) on its own, +/// with no `Next` routing involved — and the `cp314` requirement is a property +/// of the build itself, not of which index served it. Fetching the channel's +/// own interpreter-agnostic `rocm` listing ahead of interpreter selection is +/// the only way to know; `resolve_pip_runtime` re-fetches the same URL for the +/// real resolution afterwards and shares this call's cache entry, so this +/// costs no extra network round trip. +fn canonical_auto_select_needs_next_python( + paths: &AppPaths, + channel: TheRockChannel, + version_selector: Option<&RuntimeVersionSelector>, +) -> Result { + let source = resolve_source(channel, SourceLayout::Canonical); + let rocm_versions = load_simple_index_versions(paths, &source.wheel_index, "rocm", None, None)?; + Ok(highest_candidate_needs_next_python( + &rocm_versions, + channel, + version_selector, + )) +} + +/// Pure decision core of [`canonical_auto_select_needs_next_python`], split out +/// so it can be tested without a network fetch. +fn highest_candidate_needs_next_python( + rocm_versions: &[String], + channel: TheRockChannel, + version_selector: Option<&RuntimeVersionSelector>, +) -> bool { + let mut candidates = channel_rocm_candidates(rocm_versions, channel); + if let Some(selector) = version_selector { + candidates.retain(|version| selector.matches_version(version)); + } + candidates + .last() + .and_then(|version| parse_version(version)) + .is_some_and(|parsed| parsed.major >= THEROCK_NEXT_MIN_MAJOR) +} + fn managed_python_version(requirement: PythonRequirement) -> String { std::env::var("ROCM_CLI_MANAGED_PYTHON_VERSION") .ok() @@ -7574,6 +7644,25 @@ mod tests { assert_eq!(selected.rocm, "10.1.0a20260822"); } + #[test] + fn auto_select_landing_on_next_major_needs_next_python() { + // Same feed as `nightly_accepts_future_prerelease_major_without_cli_changes`: + // auto-select reaches ROCm 10.1.0a20260822, which needs cp314, not cp312, + // even though it's a prerelease served from the ordinary canonical/nightly + // index rather than `Next`'s separately hosted one. + assert!(highest_candidate_needs_next_python( + &["10.1.0a20260822".to_owned()], + TheRockChannel::Nightly, + None, + )); + + assert!(!highest_candidate_needs_next_python( + &["7.14.0".to_owned()], + TheRockChannel::Nightly, + None, + )); + } + #[test] fn tarball_selection_never_crosses_channels() { let platform = platform_tarball_token(); @@ -9108,6 +9197,96 @@ echo Python 3.12.10 Ok(path) } + #[cfg(unix)] + fn write_fake_python_reporting( + dir: &Path, + name: &str, + tag: &str, + version: &str, + ) -> Result { + use std::os::unix::fs::PermissionsExt; + let path = dir.join(name); + let script = format!( + "#!/bin/sh\nif [ \"$1\" = \"-c\" ]; then\n echo {tag}\n exit 0\nfi\necho {version}\n" + ); + fs::write(&path, script)?; + fs::set_permissions(&path, fs::Permissions::from_mode(0o755))?; + Ok(path) + } + + #[cfg(unix)] + fn write_fake_uv_creating_venv( + dir: &Path, + name: &str, + tag: &str, + version: &str, + ) -> Result { + use std::os::unix::fs::PermissionsExt; + let path = dir.join(name); + let script = format!( + r#"#!/bin/sh +if [ "$1" = "venv" ] && [ "$2" = "--python" ]; then + /bin/mkdir -p "$4/bin" + /bin/cat > "$4/bin/python" < Date: Thu, 1 Oct 2026 14:50:59 +0300 Subject: [PATCH 5/9] fix(vllm): tolerate 403s from vllm's PyPI deps, realign torch after install The ROCm 10.x discover route's final install resolves vllm/flash-attn/ amd-aiter with full dependencies, which also pulls in vllm's own plain-PyPI transitive dependencies (e.g. lm-format-enforcer) that AMD's index doesn't host. uv treats a 403 from any configured index as fatal by default, and a CLI --extra-index-url can't opt a non-pytorch index into ignore-error-codes tolerance alongside it: that only works via a uv.toml [[index]] entry passed with --config-file, and --config-file and --extra-index-url for the same URL don't merge. write_vllm_index_uv_config now generates that uv.toml so the install can fall through to PyPI instead of failing on the first 403. That same full-dependency resolve can pull in an unconstrained torch from PyPI, undoing the exact ROCm-pinned torch installed just before it. The install now runs an unconditional, forced torch realignment immediately afterward to put the pinned build back. Torch itself is still installed in its own call scoped to only the torch index, with --no-deps: mixing indexes makes uv probe torch under the vllm index too (403s fatally on AMD's CDN for a path it doesn't serve), and torch's wheel metadata pins an exact rocm[libraries] dependency that may not match the SDK's actual (e.g. nightly) version. tensorizer is no longer independently pinned: it has no ROCm-specific build, and vllm's own wheel metadata already declares an exact dependency on it, so a second independent pin risked conflicting with that. Signed-off-by: Juho Vainio --- docs/vllm.md | 18 +- engines/vllm/src/install.rs | 339 ++++++++++++++++++++++++++++++------ 2 files changed, 301 insertions(+), 56 deletions(-) diff --git a/docs/vllm.md b/docs/vllm.md index 55eb84110..c805babaf 100644 --- a/docs/vllm.md +++ b/docs/vllm.md @@ -80,14 +80,24 @@ ignored within a row, since AMD rotates those constantly. The install resolves each package's current wheel, including torch, from the row's index with `uv pip install --dry-run --reinstall`, parses the version it -reports it would install, then reinstalls pinned to that exact version -(tensorizer is the one package that stays a plain literal pin, since it has no -ROCm-specific build to discover). Every resolved pin that carries a +reports it would install, then reinstalls pinned to that exact version. +tensorizer is not discovered or pinned this way: it has no ROCm-specific +build, and vllm's own wheel metadata already declares an exact tensorizer +dependency, so it is left to vllm's own dependency resolution rather than +risk a conflicting pin of its own. Every resolved pin that carries a `+rocmX.Y` local version is checked against the SDK's own major.minor before installing, so an index that happens to serve more than one ROCm line at once cannot silently install the wrong line's wheel onto this SDK. If AMD's index has no compatible build for a package, the resolver fails and the install -fails rather than falling back to an unpinned or CPU install. Every other +fails rather than falling back to an unpinned or CPU install. The final install +of vllm/flash-attn/amd-aiter also resolves vllm's own plain-PyPI transitive +dependencies (e.g. `lm-format-enforcer`), which AMD's index doesn't host; a +generated `uv.toml` sets `ignore-error-codes = [403]` for that index so `uv` +falls through to PyPI for those instead of treating the index's 403 as fatal. +That same full-dependency resolve can also pull in an unconstrained `torch` +from PyPI, undoing the exact ROCm pin just installed; the install re-pins +torch back to it immediately afterwards. +Every other ROCm SDK version, including 7.2.3, keeps using the static pin table; an SDK version with no matching row there falls back to the table's default pin, *unless* its major release matches a discovery-table entry, in which case diff --git a/engines/vllm/src/install.rs b/engines/vllm/src/install.rs index 102b98489..e8b30a03a 100644 --- a/engines/vllm/src/install.rs +++ b/engines/vllm/src/install.rs @@ -63,9 +63,6 @@ pub(crate) struct VllmRocmDiscoverBuild { flash_attn_version_prefix: &'static str, amd_aiter_version_prefix: &'static str, torch_version_prefix: &'static str, - /// Plain PyPI pin: tensorizer publishes no ROCm-specific build, so there is - /// nothing to discover. - tensorizer_requirement: &'static str, } const VLLM_ROCM_DISCOVER_BUILD_TABLE: &[VllmRocmDiscoverBuild] = &[ @@ -78,7 +75,6 @@ const VLLM_ROCM_DISCOVER_BUILD_TABLE: &[VllmRocmDiscoverBuild] = &[ flash_attn_version_prefix: "2.8", amd_aiter_version_prefix: "0.1", torch_version_prefix: "2.12", - tensorizer_requirement: "tensorizer==2.12.1", }, VllmRocmDiscoverBuild { rocm_sdk_version: "10.1.0", @@ -89,7 +85,6 @@ const VLLM_ROCM_DISCOVER_BUILD_TABLE: &[VllmRocmDiscoverBuild] = &[ flash_attn_version_prefix: "2.8", amd_aiter_version_prefix: "0.1", torch_version_prefix: "2.12", - tensorizer_requirement: "tensorizer==2.12.1", }, ]; /// Looks up the discovery build recipe for a ROCm SDK version, if any. @@ -645,27 +640,129 @@ fn dry_run_resolved_pin(stdout: &str, pkg: &str) -> Option { (!version.is_empty()).then(|| format!("{pkg}=={version}")) }) } -/// Builds the `uv pip install` argv for a ROCm 10.x discovery install: the -/// resolved `pins` plus `--prerelease allow` and both discovery indexes. -fn vllm_rocm10_discover_install_args( +/// Builds the `uv pip install` argv for pinned torch alone, scoped to only +/// `torch_index_url`. +/// +/// Mixing `vllm_index_url` into the same call (as a prior version of this +/// code did) makes `uv` probe torch's package name under the wrong index +/// too; AMD's CloudFront/S3-backed discovery hosts answer a nonexistent +/// sub-path with a fatal 403 rather than a fall-through-safe 404, which +/// aborts the whole resolution before `uv` ever reaches the index that +/// actually has torch. +/// +/// `--no-deps` matters just as much here as it does in +/// [`discover_pinned_requirement`]: this torch wheel's own metadata pins an +/// exact `rocm[libraries]==` dependency, naming the ROCm release +/// its ROCm libs were built against. That release is managed separately by +/// `rocm install sdk`, which can legitimately be a different version track +/// (e.g. a nightly build) than this discovery row's pin, so the `rocm` +/// project is never published under this index at that exact version; +/// resolving it here fails the whole install instead of leaving the +/// already-installed SDK alone. +fn vllm_rocm10_discover_torch_install_args( python: &Path, reinstall: bool, build: &VllmRocmDiscoverBuild, - pins: &[String], + torch_pin: &str, ) -> Vec { let mut args = uv_pip_install_base(python); if reinstall { - args.push("--reinstall".to_owned()); + args.push("--reinstall-package".to_owned()); + args.push("torch".to_owned()); } - args.extend(pins.iter().cloned()); + args.push(torch_pin.to_owned()); + args.push("--no-deps".to_owned()); args.push("--prerelease".to_owned()); args.push("allow".to_owned()); args.push("--extra-index-url".to_owned()); - args.push(build.vllm_index_url.to_owned()); - args.push("--extra-index-url".to_owned()); args.push(build.torch_index_url.to_owned()); args } +/// Builds the `uv pip install` argv for pinned vllm/flash-attn/amd-aiter, +/// scoped to only `vllm_index_url`. tensorizer is not pinned here: vllm's own +/// wheel metadata already declares an exact tensorizer dependency, and a +/// second, independently chosen pin for the same package risks conflicting +/// with it. +/// +/// See [`vllm_rocm10_discover_torch_install_args`] for why torch is never +/// mixed into this same call. This call resolves full dependencies (unlike +/// torch's `--no-deps` call), which can pull in an unconstrained `torch` from +/// PyPI and silently undo the exact pin just installed; callers must re-run +/// [`vllm_rocm10_discover_torch_install_args`] afterwards to realign it. +/// +/// `vllm_index_url` is supplied via `--config-file`, not `--extra-index-url`, +/// here: `uv` does not merge a CLI `--extra-index-url` with a config-file +/// `[[index]]` entry for the same URL, so a CLI copy would keep 403ing fatally +/// even with the config file's `ignore-error-codes` also active. +fn vllm_rocm10_discover_remaining_install_args( + python: &Path, + reinstall: bool, + pins: &[String], +) -> Vec { + let mut args = uv_pip_install_base(python); + if reinstall { + for pin in pins { + if let Some((name, _)) = pin.split_once("==") { + args.push("--reinstall-package".to_owned()); + args.push(name.to_owned()); + } + } + } + args.extend(pins.iter().cloned()); + args.push("--prerelease".to_owned()); + args.push("allow".to_owned()); + args +} +/// Writes a scratch `uv.toml` that keeps `uv` searching other indexes when +/// `vllm_index_url` 403s on a package it doesn't host, instead of aborting +/// resolution outright. +/// +/// `uv`'s default index-strategy treats a 403 from any configured index as +/// fatal, not as "not found here, try the next index" — the docs note this is +/// normally only special-cased for the pytorch index. AMD's staging index is a +/// CDN-backed static index with the same 403-for-missing-package behavior, so +/// vllm's own plain-PyPI transitive dependencies (e.g. `lm-format-enforcer`) +/// need the same `ignore-error-codes` treatment here. There is no CLI flag or +/// env var for it, only this config file. +fn write_vllm_index_uv_config(paths: &AppPaths, build: &VllmRocmDiscoverBuild) -> Result { + std::fs::create_dir_all(&paths.cache_dir) + .with_context(|| format!("failed to create {}", paths.cache_dir.display()))?; + let path = paths.cache_dir.join("vllm-rocm10-uv.toml"); + let contents = format!( + "[[index]]\nurl = \"{}\"\nignore-error-codes = [403]\n", + build.vllm_index_url + ); + std::fs::write(&path, contents) + .with_context(|| format!("failed to write {}", path.display()))?; + Ok(path) +} +/// Runs a `uv pip install` invocation, surfacing a failure with the full +/// command and `uv`'s own output. +fn run_uv_pip_install(uv: &Path, paths: &AppPaths, python: &Path, args: Vec) -> Result<()> { + let output = ProcessCommand::new(uv) + .args(&args) + .envs(uv_command_env(paths)) + .output() + .context("failed to launch uv pip install for vLLM (ROCm 10.x discovery)")?; + if output.status.success() { + return Ok(()); + } + let stderr = String::from_utf8_lossy(&output.stderr).trim().to_owned(); + let stdout = String::from_utf8_lossy(&output.stdout).trim().to_owned(); + let detail = if !stderr.is_empty() { + stderr + } else if !stdout.is_empty() { + stdout + } else { + "no output".to_owned() + }; + bail!( + "`uv pip install {}` failed for {}: {}", + args.join(" "), + python.display(), + detail + ) +} /// Rejects a discovered pin whose `+rocm` local version names a /// different ROCm line than the SDK being installed for. /// @@ -781,41 +878,30 @@ fn install_vllm_rocm10_discover( build.amd_aiter_version_prefix, )?; - let pins = vec![ - torch, - vllm, - flash_attn, - amd_aiter, - build.tensorizer_requirement.to_owned(), - ]; + let pins = vec![torch, vllm, flash_attn, amd_aiter]; for pin in &pins { ensure_rocm_local_version_matches(pin, build.rocm_sdk_version)?; } - let args = vllm_rocm10_discover_install_args(python, reinstall, build, &pins); - let output = ProcessCommand::new(uv) - .args(args) - .envs(uv_command_env(paths)) - .output() - .context("failed to launch uv pip install for vLLM (ROCm 10.x discovery)")?; - if output.status.success() { - return Ok(pins); - } - let stderr = String::from_utf8_lossy(&output.stderr).trim().to_owned(); - let stdout = String::from_utf8_lossy(&output.stdout).trim().to_owned(); - let detail = if !stderr.is_empty() { - stderr - } else if !stdout.is_empty() { - stdout - } else { - "no output".to_owned() - }; - bail!( - "`uv pip install {}` failed for {}: {}", - pins.join(" "), - python.display(), - detail - ) + // Two separate `uv pip install` calls, each scoped to exactly one index: + // see `vllm_rocm10_discover_torch_install_args` for why combining both + // indexes into one call is unsafe. + let torch_args = vllm_rocm10_discover_torch_install_args(python, reinstall, build, &pins[0]); + run_uv_pip_install(uv, paths, python, torch_args)?; + + let mut remaining_args = + vllm_rocm10_discover_remaining_install_args(python, reinstall, &pins[1..]); + let config_file = write_vllm_index_uv_config(paths, build)?; + remaining_args.push("--config-file".to_owned()); + remaining_args.push(config_file.display().to_string()); + run_uv_pip_install(uv, paths, python, remaining_args)?; + + // The full-dependency resolve above can replace torch with an + // unconstrained PyPI build; force it back to the exact ROCm pin. + let torch_realign_args = vllm_rocm10_discover_torch_install_args(python, true, build, &pins[0]); + run_uv_pip_install(uv, paths, python, torch_realign_args)?; + + Ok(pins) } /// Wheel index and exact requirement for one `uv pip install vllm`. /// @@ -1542,9 +1628,33 @@ mod tests { assert_eq!(vllm_install_route(None, None), VllmInstallRoute::Static); } #[test] - fn vllm_rocm10_discover_install_args_includes_pins_and_both_indexes() { + fn vllm_rocm10_discover_torch_install_args_use_only_the_torch_index() { + let python = PathBuf::from("/opt/venv/bin/python"); + let build = vllm_rocm_discover_build("10.0.0").expect("10.0.0 has a discover row"); + let torch_pin = "torch==2.12.0+rocm10.0.0"; + + let args = vllm_rocm10_discover_torch_install_args(&python, false, build, torch_pin); + assert!(!args.contains(&"--reinstall".to_owned())); + assert!(!args.contains(&"--reinstall-package".to_owned())); + assert!(args.contains(&torch_pin.to_owned())); + assert!(args.contains(&"--no-deps".to_owned())); + assert!(args.contains(&"--prerelease".to_owned())); + assert!(args.contains(&"allow".to_owned())); + assert!(args.contains(&build.torch_index_url.to_owned())); + assert!( + !args.contains(&build.vllm_index_url.to_owned()), + "{args:?} must never reference vllm_index_url, or `uv` may probe torch under it \ + and hit a fatal 403 from a nonexistent sub-path" + ); + + let args = vllm_rocm10_discover_torch_install_args(&python, true, build, torch_pin); + assert!(args.contains(&"--reinstall-package".to_owned())); + assert!(args.contains(&"torch".to_owned())); + assert!(!args.contains(&"--reinstall".to_owned())); + } + #[test] + fn vllm_rocm10_discover_remaining_install_args_use_only_the_vllm_index() { let pins = vec![ - "torch==2.12.0+rocm10.0.0".to_owned(), "vllm==0.27.1.dev5+rocm10.0.0".to_owned(), "flash-attn==2.8.3".to_owned(), "amd-aiter==0.1.4".to_owned(), @@ -1553,18 +1663,53 @@ mod tests { let python = PathBuf::from("/opt/venv/bin/python"); let build = vllm_rocm_discover_build("10.0.0").expect("10.0.0 has a discover row"); - let args = vllm_rocm10_discover_install_args(&python, false, build, &pins); - assert!(!args.contains(&"--reinstall".to_owned())); + let args = vllm_rocm10_discover_remaining_install_args(&python, false, &pins); + assert!(!args.contains(&"--reinstall-package".to_owned())); for pin in &pins { assert!(args.contains(pin), "{args:?} should contain {pin}"); } assert!(args.contains(&"--prerelease".to_owned())); assert!(args.contains(&"allow".to_owned())); - assert!(args.contains(&build.vllm_index_url.to_owned())); - assert!(args.contains(&build.torch_index_url.to_owned())); + assert!( + !args.contains(&build.vllm_index_url.to_owned()), + "{args:?} must supply vllm_index_url via --config-file, not --extra-index-url, or \ + `uv` 403s fatally on a package the index doesn't host despite ignore-error-codes" + ); + assert!( + !args.contains(&build.torch_index_url.to_owned()), + "{args:?} must never reference torch_index_url, or `uv` may probe flash-attn/\ + amd-aiter under it and hit a fatal 403 from a nonexistent sub-path" + ); + + let args = vllm_rocm10_discover_remaining_install_args(&python, true, &pins); + for pin in &pins { + let name = pin.split_once("==").unwrap().0; + assert!(args.contains(&"--reinstall-package".to_owned())); + assert!( + args.contains(&name.to_owned()), + "{args:?} should reinstall {name}" + ); + } + assert!(!args.contains(&"--reinstall".to_owned())); + } + #[test] + fn write_vllm_index_uv_config_writes_an_ignore_error_codes_entry() -> Result<()> { + let root = tempfile::tempdir()?; + let paths = AppPaths { + config_dir: root.path().join("config"), + data_dir: root.path().join("data"), + cache_dir: root.path().join("cache"), + }; + let build = vllm_rocm_discover_build("10.0.0").expect("10.0.0 has a discover row"); - let args = vllm_rocm10_discover_install_args(&python, true, build, &pins); - assert!(args.contains(&"--reinstall".to_owned())); + let config_file = write_vllm_index_uv_config(&paths, build)?; + let contents = std::fs::read_to_string(&config_file)?; + assert!(contents.contains(build.vllm_index_url), "{contents}"); + assert!( + contents.contains("ignore-error-codes = [403]"), + "{contents}" + ); + Ok(()) } /// The 10.1 row must reach the staging host for both vLLM and torch; 10.0 /// must keep using the production frameworks index and `whl-next`. This is @@ -1627,4 +1772,94 @@ mod tests { assert!(rendered.contains("10.1.0"), "{rendered}"); assert!(rendered.contains("rocm10.0.0"), "{rendered}"); } + /// Pins `install_vllm_rocm10_discover`'s own call sequencing, not just its + /// arg-builder helpers in isolation: the remaining-deps install must use + /// `--config-file` (never `--extra-index-url`, which cannot tolerate the + /// 403 `lm-format-enforcer` needs), and a torch realign call must follow + /// it so that install's full-dependency resolve cannot leave an + /// unconstrained torch in place. Runs offline against fake `uv`/`python` + /// shims, same pattern as `crates/rocm-core/src/examine.rs`. + #[test] + #[cfg(unix)] + fn install_vllm_rocm10_discover_realigns_torch_after_the_config_file_install() -> Result<()> { + use std::os::unix::fs::PermissionsExt; + + let root = tempfile::tempdir()?; + let paths = AppPaths { + config_dir: root.path().join("config"), + data_dir: root.path().join("data"), + cache_dir: root.path().join("cache"), + }; + let log = root.path().join("uv-calls.log"); + + let python = root.path().join("python"); + std::fs::write(&python, "#!/bin/sh\necho cp314\n")?; + std::fs::set_permissions(&python, std::fs::Permissions::from_mode(0o755))?; + + // Answers every `--dry-run` discovery call with a resolved pin for + // whatever package the requirement names, and otherwise just records + // the call (the real installs) for the assertions below. + let uv = root.path().join("uv"); + std::fs::write( + &uv, + format!( + r#"#!/bin/sh +echo "$@" >> "{log}" +last="" +for a in "$@"; do last="$a"; done +case "$*" in + *--dry-run*) + pkg=${{last%%==*}} + echo " + ${{pkg}}==9.9.9" >&2 + ;; +esac +exit 0 +"#, + log = log.display() + ), + )?; + std::fs::set_permissions(&uv, std::fs::Permissions::from_mode(0o755))?; + + let build = vllm_rocm_discover_build("10.1.0").expect("10.1.0 has a discover row"); + let pins = install_vllm_rocm10_discover(&uv, &paths, &python, true, build)?; + assert_eq!(pins.len(), 4); + + let calls: Vec> = std::fs::read_to_string(&log)? + .lines() + .map(|line| line.split_whitespace().map(ToOwned::to_owned).collect()) + .collect(); + let real_installs: Vec<&Vec> = calls + .iter() + .filter(|args| !args.contains(&"--dry-run".to_owned())) + .collect(); + assert_eq!( + real_installs.len(), + 3, + "expected torch install, remaining-deps install, torch realign: {calls:?}" + ); + + let torch_install = real_installs[0]; + assert!(torch_install.contains(&"--extra-index-url".to_owned())); + assert!(!torch_install.contains(&"--config-file".to_owned())); + + let remaining_install = real_installs[1]; + assert!( + remaining_install.contains(&"--config-file".to_owned()), + "the remaining-deps install must use --config-file, not --extra-index-url, \ + for the 403-tolerant index: {remaining_install:?}" + ); + assert!(!remaining_install.contains(&"--extra-index-url".to_owned())); + + let torch_realign = real_installs[2]; + assert!( + torch_realign.contains(&"--reinstall-package".to_owned()) + && torch_realign.contains(&"torch".to_owned()), + "the full-dependency install can silently replace torch with an unconstrained \ + PyPI build; a forced reinstall must follow it: {torch_realign:?}" + ); + assert!(torch_realign.contains(&"--extra-index-url".to_owned())); + assert!(!torch_realign.contains(&"--config-file".to_owned())); + + Ok(()) + } } From 123ae6b58672c55d6dd8807cf017e50badf02bcf Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Thu, 1 Oct 2026 14:51:24 +0300 Subject: [PATCH 6/9] test(e2e): make therock-next-09 a per-PR lane, prove serve+inference too Drop @nightly: install_vllm_rocm10_discover always runs its 403-tolerant config-file install and forced torch realignment regardless of which discover-table row is newest, so this is the live regression guard for both ROCMAI-439 fixes as well as the discovery mechanism itself, not just nightly-cadence coverage. Also extend the scenario to serve the model it just installed and send it a chat completion, reusing the same serve/chat steps every other engine-canary scenario uses. serve-vllm-inference (model_serving.feature) doesn't cover this route: its runtime comes from a plain unpinned install, never the ROCm 10 preview source. Without this, a wheel that installs cleanly but can't actually load or answer a request would pass every assertion in this scenario and still be unservable. Signed-off-by: Juho Vainio --- .../features/therock_next_generation.feature | 23 ++++++++++++++++++- 1 file changed, 22 insertions(+), 1 deletion(-) diff --git a/tests/e2e-cucumber/features/therock_next_generation.feature b/tests/e2e-cucumber/features/therock_next_generation.feature index 95fd943c9..afd0db4f6 100644 --- a/tests/e2e-cucumber/features/therock_next_generation.feature +++ b/tests/e2e-cucumber/features/therock_next_generation.feature @@ -137,7 +137,24 @@ Feature: TheRock "next" ROCm 10 install layout # every earlier SDK version uses. Provisions its own ROCm 10 runtime rather # than reusing therock-next-07's, so that scenario's arch-detection coverage # still runs on hosts that can't start vLLM. - @id:therock-next-09-live-install-reports-vllm-rocm10x-discovery-pins @requires-gpu @requires-engine:vllm @nightly + # + # Not `@nightly`: `install_vllm_rocm10_discover` always runs its 403-tolerant + # `--config-file` install and forced torch realignment regardless of which + # row (10.0.x production, 10.1.x staging) the live preview index currently + # serves as newest, so this is the live per-PR regression guard for both + # fixes (ROCMAI-439) as well as the discovery mechanism itself — accepted as + # a real fresh-SDK-install cost on every vLLM-capable self-hosted GPU lane. + # + # The serve+inference tail matters because `serve-vllm-inference` + # (model_serving.feature) does not cover this route: its runtime comes from + # a plain unpinned `rocm install sdk`, never the ROCm 10 preview source. A + # discovered wheel that installs cleanly but can't actually load or answer a + # request (bad torch realignment, cp314 ABI mismatch, a `uv.toml` fallback + # pulling an incompatible transitive dep) would pass every assertion above + # this line and still be unservable, so this scenario reuses the same + # serve/chat steps every other engine-canary scenario does rather than + # stopping at "install reported the right pins". + @id:therock-next-09-live-install-reports-vllm-rocm10x-discovery-pins @requires-gpu @requires-engine:vllm Scenario: therock-next-09 - Installing vLLM against a live ROCm 10 preview runtime reports the discovery pins Given a machine with no CLI-managed runtimes When the user installs the SDK from the ROCm 10 preview source with no family override @@ -147,6 +164,10 @@ Feature: TheRock "next" ROCm 10 install layout And the ROCm 10 runtime provisioned a cp314 Python interpreter When the user reinstalls vllm Then the install reports the vLLM ROCm 10.x discovery pins + And a model is being served on GPU + When the user sends a chat completion request + Then the response contains a model reply + And the response identifies the correct model # The opt-in half of therock-next-02. Both polarities run here, on the mock # lane, because this is the only place the flag's effect on the real install From 171b4a5558facdb97887a51549a277cee06547c7 Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Fri, 2 Oct 2026 09:58:29 +0300 Subject: [PATCH 7/9] fix(e2e): fix keyword inheritance bug skipping therock-next-09's serve step `And a model is being served on GPU` inherited `Then` from the preceding step, but that step text is only registered as a `#[given(...)]` step definition. Cucumber found no match in the `Then` collection and silently skipped the step, which the reconciliation layer then flagged as an unexpected failure on every self-hosted GPU lane (MI300X, MI350P, Strix Halo WSL2). Every other scenario reusing this step precedes it with `Given`, matching the registration. Signed-off-by: Juho Vainio --- tests/e2e-cucumber/features/therock_next_generation.feature | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/e2e-cucumber/features/therock_next_generation.feature b/tests/e2e-cucumber/features/therock_next_generation.feature index afd0db4f6..663fce350 100644 --- a/tests/e2e-cucumber/features/therock_next_generation.feature +++ b/tests/e2e-cucumber/features/therock_next_generation.feature @@ -164,7 +164,7 @@ Feature: TheRock "next" ROCm 10 install layout And the ROCm 10 runtime provisioned a cp314 Python interpreter When the user reinstalls vllm Then the install reports the vLLM ROCm 10.x discovery pins - And a model is being served on GPU + Given a model is being served on GPU When the user sends a chat completion request Then the response contains a model reply And the response identifies the correct model From 103064d3593d82a881f3d47c75a03b21d9e96afe Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Fri, 2 Oct 2026 13:16:01 +0300 Subject: [PATCH 8/9] fix(vllm): realign torchvision/torchaudio after ROCm 10.x discover install The full-dependency resolve in install_vllm_rocm10_discover can replace torchvision/torchaudio with unconstrained PyPI builds ABI-incompatible with the realigned torch, causing RuntimeError: operator torchvision::nms does not exist at serve time. Snapshot whatever torchvision/torchaudio the SDK install already wrote via `uv pip freeze` before the resolve runs, then restore them alongside torch in the final realignment step. Signed-off-by: Juho Vainio --- docs/vllm.md | 6 +- engines/vllm/src/install.rs | 111 +++++++++++++++++++++++++++++++----- 2 files changed, 100 insertions(+), 17 deletions(-) diff --git a/docs/vllm.md b/docs/vllm.md index c805babaf..c572276de 100644 --- a/docs/vllm.md +++ b/docs/vllm.md @@ -95,8 +95,10 @@ dependencies (e.g. `lm-format-enforcer`), which AMD's index doesn't host; a generated `uv.toml` sets `ignore-error-codes = [403]` for that index so `uv` falls through to PyPI for those instead of treating the index's 403 as fatal. That same full-dependency resolve can also pull in an unconstrained `torch` -from PyPI, undoing the exact ROCm pin just installed; the install re-pins -torch back to it immediately afterwards. +from PyPI, undoing the exact ROCm pin just installed, and transitively an +unconstrained `torchvision`/`torchaudio` with it; the install re-pins torch, +plus whatever of torchvision/torchaudio was already installed by the SDK +install, back to their prior exact builds immediately afterwards. Every other ROCm SDK version, including 7.2.3, keeps using the static pin table; an SDK version with no matching row there falls back to the table's default pin, diff --git a/engines/vllm/src/install.rs b/engines/vllm/src/install.rs index e8b30a03a..49edd9f1b 100644 --- a/engines/vllm/src/install.rs +++ b/engines/vllm/src/install.rs @@ -5,7 +5,8 @@ use anyhow::{Context, Result, anyhow, bail}; use rocm_core::{ AppPaths, DependencyViolation, check_dependencies, ensure_uv_binary, split_local_version, - uv_command_env, uv_pip_install_base, violation_subject, violations_requiring, + uv_command_env, uv_pip_freeze_args, uv_pip_install_base, violation_subject, + violations_requiring, }; use rocm_engine_protocol::{InstallRequest, InstallResponse}; use std::path::{Path, PathBuf}; @@ -678,6 +679,67 @@ fn vllm_rocm10_discover_torch_install_args( args.push(build.torch_index_url.to_owned()); args } +/// Builds the `uv pip install` argv that realigns torch and, if present, +/// torchvision/torchaudio back to `pins` after the full-dependency install, +/// all `--no-deps` and scoped to `torch_index_url`. +/// +/// `pins` is always `torch_pin` plus whatever of torchvision/torchaudio +/// [`installed_stack_pins`] found before the full-dependency install ran. A +/// single call realigns the whole stack together rather than one `uv` +/// invocation per package. +fn vllm_rocm10_discover_realign_install_args( + python: &Path, + build: &VllmRocmDiscoverBuild, + pins: &[String], +) -> Vec { + let mut args = uv_pip_install_base(python); + for pin in pins { + if let Some((name, _)) = pin.split_once("==") { + args.push("--reinstall-package".to_owned()); + args.push(name.to_owned()); + } + } + args.extend(pins.iter().cloned()); + args.push("--no-deps".to_owned()); + args.push("--prerelease".to_owned()); + args.push("allow".to_owned()); + args.push("--extra-index-url".to_owned()); + args.push(build.torch_index_url.to_owned()); + args +} +/// Reads the exact `torchvision`/`torchaudio` pins already installed in +/// `python`, if any, via `uv pip freeze`. +/// +/// Called before the full-dependency vllm/flash-attn/amd-aiter install, which +/// can otherwise pull in a plain-PyPI torchvision/torchaudio build that is +/// ABI-incompatible with the ROCm torch the SDK install already wrote (same +/// failure mode torch itself needed realignment for). The SDK install always +/// writes a coherent torch/torchvision/torchaudio triplet from the same +/// index as `torch_index_url` (see `apps/rocm/src/therock.rs`), so restoring +/// whatever was already there is correct without discovering a fresh pin. +fn installed_stack_pins(uv: &Path, paths: &AppPaths, python: &Path) -> Result> { + let output = ProcessCommand::new(uv) + .args(uv_pip_freeze_args(python)) + .envs(uv_command_env(paths)) + .output() + .context("failed to launch uv pip freeze")?; + if !output.status.success() { + bail!( + "`uv pip freeze` for {} failed: {}", + python.display(), + String::from_utf8_lossy(&output.stderr).trim() + ); + } + let stdout = String::from_utf8_lossy(&output.stdout).into_owned(); + Ok(stdout + .lines() + .filter(|line| { + line.split_once("==") + .is_some_and(|(name, _)| name == "torchvision" || name == "torchaudio") + }) + .map(ToOwned::to_owned) + .collect()) +} /// Builds the `uv pip install` argv for pinned vllm/flash-attn/amd-aiter, /// scoped to only `vllm_index_url`. tensorizer is not pinned here: vllm's own /// wheel metadata already declares an exact tensorizer dependency, and a @@ -845,6 +907,7 @@ fn install_vllm_rocm10_discover( build: &VllmRocmDiscoverBuild, ) -> Result> { ensure_discover_python_tag(python, build)?; + let stack_pins = installed_stack_pins(uv, paths, python)?; let torch = discover_pinned_requirement( uv, paths, @@ -896,10 +959,13 @@ fn install_vllm_rocm10_discover( remaining_args.push(config_file.display().to_string()); run_uv_pip_install(uv, paths, python, remaining_args)?; - // The full-dependency resolve above can replace torch with an - // unconstrained PyPI build; force it back to the exact ROCm pin. - let torch_realign_args = vllm_rocm10_discover_torch_install_args(python, true, build, &pins[0]); - run_uv_pip_install(uv, paths, python, torch_realign_args)?; + // The full-dependency resolve above can replace torch (and, pulling it in + // transitively, torchvision/torchaudio) with unconstrained PyPI builds; + // force the whole stack back to what was installed before that resolve ran. + let mut realign_pins = vec![pins[0].clone()]; + realign_pins.extend(stack_pins); + let realign_args = vllm_rocm10_discover_realign_install_args(python, build, &realign_pins); + run_uv_pip_install(uv, paths, python, realign_args)?; Ok(pins) } @@ -1797,8 +1863,10 @@ mod tests { std::fs::set_permissions(&python, std::fs::Permissions::from_mode(0o755))?; // Answers every `--dry-run` discovery call with a resolved pin for - // whatever package the requirement names, and otherwise just records - // the call (the real installs) for the assertions below. + // whatever package the requirement names, `pip freeze` with the + // torchvision/torchaudio pins the SDK install already wrote, and + // otherwise just records the call (the real installs) for the + // assertions below. let uv = root.path().join("uv"); std::fs::write( &uv, @@ -1812,6 +1880,10 @@ case "$*" in pkg=${{last%%==*}} echo " + ${{pkg}}==9.9.9" >&2 ;; + *freeze*) + echo "torchvision==1.2.3+rocm10.1.0" + echo "torchaudio==4.5.6+rocm10.1.0" + ;; esac exit 0 "#, @@ -1830,12 +1902,14 @@ exit 0 .collect(); let real_installs: Vec<&Vec> = calls .iter() - .filter(|args| !args.contains(&"--dry-run".to_owned())) + .filter(|args| { + !args.contains(&"--dry-run".to_owned()) && !args.contains(&"freeze".to_owned()) + }) .collect(); assert_eq!( real_installs.len(), 3, - "expected torch install, remaining-deps install, torch realign: {calls:?}" + "expected torch install, remaining-deps install, stack realign: {calls:?}" ); let torch_install = real_installs[0]; @@ -1850,15 +1924,22 @@ exit 0 ); assert!(!remaining_install.contains(&"--extra-index-url".to_owned())); - let torch_realign = real_installs[2]; + let stack_realign = real_installs[2]; assert!( - torch_realign.contains(&"--reinstall-package".to_owned()) - && torch_realign.contains(&"torch".to_owned()), + stack_realign.contains(&"--reinstall-package".to_owned()) + && stack_realign.contains(&"torch".to_owned()), "the full-dependency install can silently replace torch with an unconstrained \ - PyPI build; a forced reinstall must follow it: {torch_realign:?}" + PyPI build; a forced reinstall must follow it: {stack_realign:?}" + ); + assert!( + stack_realign.contains(&"torchvision==1.2.3+rocm10.1.0".to_owned()) + && stack_realign.contains(&"torchaudio==4.5.6+rocm10.1.0".to_owned()), + "torchvision/torchaudio installed by the SDK must be restored alongside torch, \ + or the full-dependency resolve can leave an ABI-incompatible build in place: \ + {stack_realign:?}" ); - assert!(torch_realign.contains(&"--extra-index-url".to_owned())); - assert!(!torch_realign.contains(&"--config-file".to_owned())); + assert!(stack_realign.contains(&"--extra-index-url".to_owned())); + assert!(!stack_realign.contains(&"--config-file".to_owned())); Ok(()) } From b6066d30031c391b29e1e9c16691b173bce844f9 Mon Sep 17 00:00:00 2001 From: Juho Vainio Date: Fri, 2 Oct 2026 15:59:54 +0300 Subject: [PATCH 9/9] fix(vllm): use --index-url, not --extra-index-url, for ROCm 10.x torch realign uv keeps --extra-index-url pip-compatible: it is checked after the implicit default PyPI index, not before it. Torch's discovered pin always carries a +rocmX.Y local version absent from PyPI, so it was never affected, but the torchvision/torchaudio pins captured from `uv pip freeze` carry plain version numbers that can collide with a same-numbered public PyPI release. uv was silently installing that generic wheel over AMD's build, reproducing the torchvision::nms ABI mismatch the realignment step exists to prevent even after the prior fix. Both calls are single-pin --no-deps installs with nothing legitimate for PyPI to supply, so --index-url (replacing the default index outright) closes the loophole. Signed-off-by: Juho Vainio --- engines/vllm/src/install.rs | 31 +++++++++++++++++++++++++++---- 1 file changed, 27 insertions(+), 4 deletions(-) diff --git a/engines/vllm/src/install.rs b/engines/vllm/src/install.rs index 49edd9f1b..c3c4b2fe4 100644 --- a/engines/vllm/src/install.rs +++ b/engines/vllm/src/install.rs @@ -651,6 +651,13 @@ fn dry_run_resolved_pin(stdout: &str, pkg: &str) -> Option { /// aborts the whole resolution before `uv` ever reaches the index that /// actually has torch. /// +/// Uses `--index-url`, not `--extra-index-url`: the latter is kept +/// pip-compatible by `uv`, meaning it is checked *after* the implicit +/// default PyPI index, not before it. This call installs a single exact +/// pin with `--no-deps`, so there is nothing legitimate for PyPI to supply; +/// leaving it reachable only risks `uv` silently satisfying the pin from a +/// same-numbered public PyPI release instead of AMD's build. +/// /// `--no-deps` matters just as much here as it does in /// [`discover_pinned_requirement`]: this torch wheel's own metadata pins an /// exact `rocm[libraries]==` dependency, naming the ROCm release @@ -675,7 +682,7 @@ fn vllm_rocm10_discover_torch_install_args( args.push("--no-deps".to_owned()); args.push("--prerelease".to_owned()); args.push("allow".to_owned()); - args.push("--extra-index-url".to_owned()); + args.push("--index-url".to_owned()); args.push(build.torch_index_url.to_owned()); args } @@ -687,6 +694,14 @@ fn vllm_rocm10_discover_torch_install_args( /// [`installed_stack_pins`] found before the full-dependency install ran. A /// single call realigns the whole stack together rather than one `uv` /// invocation per package. +/// +/// Uses `--index-url`, for the same reason +/// [`vllm_rocm10_discover_torch_install_args`] does: torch's pin always +/// carries a `+rocmX.Y` local version with no PyPI equivalent, but +/// torchvision/torchaudio do not, so `--extra-index-url`'s lower-than-default +/// priority would let a same-numbered public PyPI wheel silently win over +/// AMD's build here, reproducing the exact ABI mismatch this realign step +/// exists to prevent. fn vllm_rocm10_discover_realign_install_args( python: &Path, build: &VllmRocmDiscoverBuild, @@ -703,7 +718,7 @@ fn vllm_rocm10_discover_realign_install_args( args.push("--no-deps".to_owned()); args.push("--prerelease".to_owned()); args.push("allow".to_owned()); - args.push("--extra-index-url".to_owned()); + args.push("--index-url".to_owned()); args.push(build.torch_index_url.to_owned()); args } @@ -1707,6 +1722,12 @@ mod tests { assert!(args.contains(&"--prerelease".to_owned())); assert!(args.contains(&"allow".to_owned())); assert!(args.contains(&build.torch_index_url.to_owned())); + assert!( + args.contains(&"--index-url".to_owned()), + "{args:?} must use --index-url, not --extra-index-url, or a same-numbered public \ + PyPI wheel can silently outrank AMD's build" + ); + assert!(!args.contains(&"--extra-index-url".to_owned())); assert!( !args.contains(&build.vllm_index_url.to_owned()), "{args:?} must never reference vllm_index_url, or `uv` may probe torch under it \ @@ -1913,7 +1934,8 @@ exit 0 ); let torch_install = real_installs[0]; - assert!(torch_install.contains(&"--extra-index-url".to_owned())); + assert!(torch_install.contains(&"--index-url".to_owned())); + assert!(!torch_install.contains(&"--extra-index-url".to_owned())); assert!(!torch_install.contains(&"--config-file".to_owned())); let remaining_install = real_installs[1]; @@ -1938,7 +1960,8 @@ exit 0 or the full-dependency resolve can leave an ABI-incompatible build in place: \ {stack_realign:?}" ); - assert!(stack_realign.contains(&"--extra-index-url".to_owned())); + assert!(stack_realign.contains(&"--index-url".to_owned())); + assert!(!stack_realign.contains(&"--extra-index-url".to_owned())); assert!(!stack_realign.contains(&"--config-file".to_owned())); Ok(())