diff --git a/.gitignore b/.gitignore index 61435d60..5a70dc60 100755 --- a/.gitignore +++ b/.gitignore @@ -160,6 +160,7 @@ work_dirs examples/regression_test/regression_outputs/ INSTALL_HYS.md *.webp +!examples/data/qwen_image_21/*.webp AGENTS.md _version.py.mcp.json telefuser/_version.py diff --git a/examples/data/qwen_image_21/0f2d052b-6ccf-46e2-821d-d073ead679ac.webp b/examples/data/qwen_image_21/0f2d052b-6ccf-46e2-821d-d073ead679ac.webp new file mode 100644 index 00000000..f52009f2 Binary files /dev/null and b/examples/data/qwen_image_21/0f2d052b-6ccf-46e2-821d-d073ead679ac.webp differ diff --git a/examples/data/qwen_image_21/0f5d38f9-d04a-498e-aaef-9dae01da8c6d.webp b/examples/data/qwen_image_21/0f5d38f9-d04a-498e-aaef-9dae01da8c6d.webp new file mode 100644 index 00000000..264b89a2 Binary files /dev/null and b/examples/data/qwen_image_21/0f5d38f9-d04a-498e-aaef-9dae01da8c6d.webp differ diff --git a/examples/data/qwen_image_21/3acf3276-0572-490e-b8ce-334833a6ca20.webp b/examples/data/qwen_image_21/3acf3276-0572-490e-b8ce-334833a6ca20.webp new file mode 100644 index 00000000..2ef65a92 Binary files /dev/null and b/examples/data/qwen_image_21/3acf3276-0572-490e-b8ce-334833a6ca20.webp differ diff --git a/examples/data/qwen_image_21/3f4d7519-6c33-491f-bad8-661f1bf03601.webp b/examples/data/qwen_image_21/3f4d7519-6c33-491f-bad8-661f1bf03601.webp new file mode 100644 index 00000000..078dd4e4 Binary files /dev/null and b/examples/data/qwen_image_21/3f4d7519-6c33-491f-bad8-661f1bf03601.webp differ diff --git a/examples/data/qwen_image_21/447da49d-dcd0-46ec-9ca8-20127c0f07d1.webp b/examples/data/qwen_image_21/447da49d-dcd0-46ec-9ca8-20127c0f07d1.webp new file mode 100644 index 00000000..f80c3dfd Binary files /dev/null and b/examples/data/qwen_image_21/447da49d-dcd0-46ec-9ca8-20127c0f07d1.webp differ diff --git a/examples/qwen_image/README.md b/examples/qwen_image/README.md index d0bbb849..7c69a5a6 100644 --- a/examples/qwen_image/README.md +++ b/examples/qwen_image/README.md @@ -1,13 +1,14 @@ # Qwen-Image Examples These examples provide text-to-image generation, image editing, quantized inference, and feature-cache calibration -with Qwen-Image checkpoints. +with Qwen-Image checkpoints, including the unified Qwen-Image 2.1 pipeline. ## Model Source | Model | HuggingFace | ModelScope | Purpose | | --- | --- | --- | --- | | Qwen-Image | [Qwen/Qwen-Image](https://huggingface.co/Qwen/Qwen-Image) | [Qwen/Qwen-Image](https://modelscope.cn/models/Qwen/Qwen-Image) | Text-to-image base weights | +| Qwen-Image 2.1 | [Qwen/Qwen-Image-2.1](https://huggingface.co/Qwen/Qwen-Image-2.1) | [Qwen/Qwen-Image-2.1](https://modelscope.cn/models/Qwen/Qwen-Image-2.1) | Native generation, editing, and reference examples | | Qwen-Image-Lightning | [Qwen/Qwen-Image-Lightning](https://huggingface.co/Qwen/Qwen-Image-Lightning) | [Qwen/Qwen-Image-Lightning](https://modelscope.cn/models/Qwen/Qwen-Image-Lightning) | Distilled LoRA and FP8 variants | | Qwen-Image-Edit | [Qwen/Qwen-Image-Edit](https://huggingface.co/Qwen/Qwen-Image-Edit) | [Qwen/Qwen-Image-Edit](https://modelscope.cn/models/Qwen/Qwen-Image-Edit) | Image editing | @@ -17,6 +18,7 @@ with Qwen-Image checkpoints. | --- | --- | --- | | Text-to-image | Supported | BF16, Lightning LoRA, TeleFuser FP8, and NF4 examples | | Image editing | Supported | TeleFuser and Diffusers reference paths | +| Qwen-Image 2.1 reference generation | Supported | One to ten condition images in the CLI example | | Multi-GPU inference | Supported | CFG and Ulysses parallelism on scripts exposing `--gpu_num` | | LoRA | Supported | Lightning LoRA example | | Quantization | Supported | Pre-quantized FP8, online TeleFuser FP8, and NF4 | @@ -24,14 +26,22 @@ with Qwen-Image checkpoints. | Feature cache | Supported | Separate T2I and edit calibration tools | | Server API | Supported | Native examples expose standard pipeline functions | +Qwen-Image 2.1 uses a 64-channel latent VAE, Qwen3-VL prompt encoding, and a +single-stream block-causal transformer. The native TeleFuser path uses 40 steps +by default with classifier-free guidance disabled. The same stage pipeline handles +text-to-image, image editing, and reference-guided generation. + ## Requirements - GPU: one H100-class CUDA GPU for the documented configurations; use supported multi-GPU degrees as needed -- Software: the standard TeleFuser installation; Diffusers is required for the official reference scripts -- Input assets: a readable image for editing; T2I requires no input asset +- Software: the standard TeleFuser installation plus Diffusers standalone Qwen-Image 2.1 VAE classes and Transformers Qwen3-VL classes +- Input assets: T2I requires no input asset; edit uses `examples/data/edit2511input.png`, and reference generation + uses five official demo images under `examples/data/qwen_image_21/` Install TeleFuser by following the [development setup](../../CONTRIBUTING.md#development-setup). +The TeleFuser entry point implements the 2.1 DiT locally. It uses Diffusers only for the standalone VAE class and Transformers for the Qwen3-VL text encoder and processor. + ## Model Directory ```text @@ -41,6 +51,12 @@ ${TF_MODEL_ZOO_PATH}/ | |-- vae/ | |-- text_encoder/ | \-- tokenizer/ +|-- Qwen-Image-2.1/ +| |-- transformer/ +| |-- vae/ +| |-- text_encoder/ +| |-- processor/ +| \-- scheduler/ |-- Qwen-Image-2512-Lightning/ | \-- Qwen-Image-2512-Lightning-8steps-V1.0-fp32.safetensors |-- Qwen-Image-Edit-2509/ @@ -66,10 +82,45 @@ python examples/qwen_image/qwen_image_t2i_h100.py \ The command writes the generated image to `work_dirs/qwen-image.png`. +The 2.1 standard example uses the same regression prompt, negative prompt, seed, and default `16:9` aspect ratio +as the existing Qwen-Image example. These defaults are declared in the Python scripts. Both examples map `16:9` +to **1664×928** pixels. + +Select physical GPU 3 by masking it into the process (it is then visible as +`cuda:0`): + +```bash +CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_t2i_h100.py \ + --model_root /hhb-data/aigc/model_zoo/Qwen-Image-2.1 \ + --output_path work_dirs/qwen-image-2.1.png +``` + ## Examples ### Text-To-Image +#### `qwen_image_21_t2i_h100.py` + +This standard example loads Qwen-Image 2.1 through TeleFuser for text-to-image +generation. + +```bash +CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_t2i_h100.py \ + --model_root /hhb-data/aigc/model_zoo/Qwen-Image-2.1 \ + --prompt "A neon shop sign that reads QWEN IMAGE 2.1 in the rain" \ + --output_path work_dirs/qwen-image-2.1.png +``` + +Key options: + +| Option | Default | Description | +| --- | --- | --- | +| `--model_root` | `/hhb-data/aigc/model_zoo/Qwen-Image-2.1` | Diffusers model directory | +| `--gpu_num` | `1` | Qwen-Image 2.1 currently uses one GPU | +| `--aspect_ratio` | `16:9` | Uses the existing Qwen-Image size mapping; default is 1664×928 | +| `--height`, `--width` | unset | Optional overrides for the mapped output dimensions | +| `--output_path` | generated example name | PNG output path | + #### `qwen_image_t2i_h100.py` ```bash @@ -126,6 +177,33 @@ python examples/qwen_image/qwen_image_edit_plus_h100.py \ --output work_dirs/qwen-image-edit.png ``` +#### `qwen_image_21_edit_h100.py` + +The native 2.1 edit example uses the existing Qwen-Image edit test image. Without +`--height` or `--width`, output dimensions follow the input image aspect ratio +at approximately one megapixel. + +```bash +CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_edit_h100.py \ + --image_path examples/data/edit2511input.png \ + --output_path work_dirs/qwen-image-2.1-edit.png +``` + +### Reference-Guided Generation + +#### `qwen_image_21_reference_h100.py` + +The default test uses the official Qwen-Image 2.1 ["Outfit styling (5 images)" demo +case](https://huggingface.co/spaces/Qwen/Qwen-Image-2.1/blob/main/examples/cases.json). Its prompt and five input +images are included in the script and `examples/data/qwen_image_21/` in the official order: model, down jacket, +Mary Jane shoes, handbag, and fur hat. Run it without extra arguments, or pass `--image_path` once per custom +reference image. The output aspect ratio follows the last reference unless dimensions are supplied. + +```bash +CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_reference_h100.py \ + --output_path work_dirs/qwen-image-2.1-reference.png +``` + ### Cache Calibration #### `qwen_image_cache_calibrate.py` @@ -166,7 +244,39 @@ python examples/qwen_image/qwen_image_edit_plus_official.py \ This reference keeps the official fixed sampling parameters while making model, input, prompt, and output paths explicit. +## Serving + +The standard entry points expose `t2i` and `i2i` service contracts. The edit +example uses the existing `i2i` service task because `telefuser serve --task` +does not offer a separate `edit` value. +The existing service schema supplies one image through `first_image_path` for +`edit` and `i2i`; use the reference CLI for multiple images. + +```bash +CUDA_VISIBLE_DEVICES=3 telefuser serve examples/qwen_image/qwen_image_21_t2i_h100.py \ + --task t2i --port 8091 +``` + +To serve either image-conditioned example, select its script with the `i2i` +task. For example: + +```bash +CUDA_VISIBLE_DEVICES=3 telefuser serve examples/qwen_image/qwen_image_21_edit_h100.py \ + --task i2i --port 8092 +``` + +Upload the source image through `/v1/tasks/form` with `first_image_file`. + +The service loads one pipeline replica on the selected GPU and writes each +request result to the service output directory. + ## Configuration Supported aspect ratios include `1:1`, `16:9`, `9:16`, `4:3`, `3:4`, `3:2`, and `2:3`. The exact resolution mapping is defined in each entry point. + +## Troubleshooting + +If loading fails with `safetensors ... invalid JSON in header`, one or more +checkpoint shards are incomplete. Verify the four files under +`text_encoder/` and copy them again before starting the service. diff --git a/examples/qwen_image/qwen_image_21_edit_h100.py b/examples/qwen_image/qwen_image_21_edit_h100.py new file mode 100644 index 00000000..1bb38088 --- /dev/null +++ b/examples/qwen_image/qwen_image_21_edit_h100.py @@ -0,0 +1,125 @@ +"""Edit an existing image with the native Qwen-Image 2.1 pipeline. + +Usage: + CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_edit_h100.py \ + --image_path examples/data/edit2511input.png \ + --output_path work_dirs/qwen-image-2.1-edit.png +""" + +from __future__ import annotations + +import os +from pathlib import Path + +import click +import torch +from PIL import Image + +from telefuser.pipelines.qwen_image import QwenImage21Pipeline +from telefuser.service.core.contract_templates import build_pipeline_manifest, build_task_contract_template + +PPL_CONFIG = { + "name": "qwen_image_2.1_edit", + "model_root": os.path.join(os.environ.get("TF_MODEL_ZOO_PATH", "/hhb-data/aigc/model_zoo"), "Qwen-Image-2.1"), + "prompt": '这个女生看着面前的电视屏幕,屏幕上面写着"阿里巴巴"', + "image_path": "examples/data/edit2511input.png", + "seed": 42, + "num_inference_steps": 40, +} + +PIPELINE_CONTRACT = build_pipeline_manifest( + pipeline_name=PPL_CONFIG["name"], + supported_tasks=["i2i"], + task_contracts={ + "i2i": build_task_contract_template( + "i2i", + required_inputs=["first_image_path"], + parameter_overrides={ + "prompt": {"default": PPL_CONFIG["prompt"]}, + "seed": {"default": PPL_CONFIG["seed"]}, + }, + excluded_parameters=["negative_prompt", "resolution", "aspect_ratio"], + ) + }, +) + + +def get_pipeline( + parallelism: int = 1, + model_root: str = PPL_CONFIG["model_root"], + device: str | None = None, +) -> QwenImage21Pipeline: + """Load the shared native Qwen-Image 2.1 pipeline.""" + + if parallelism != 1: + raise ValueError("Qwen-Image 2.1 example currently supports one GPU") + return QwenImage21Pipeline.from_pretrained(model_root, device=device or "cuda", torch_dtype=torch.bfloat16) + + +def run( + pipeline: QwenImage21Pipeline, + prompt: str, + image: Image.Image, + seed: int = PPL_CONFIG["seed"], + height: int | None = None, + width: int | None = None, +) -> list[Image.Image]: + """Apply an editing instruction to one source image.""" + + return pipeline( + prompt=prompt, + image=image, + seed=seed, + height=height, + width=width, + num_inference_steps=PPL_CONFIG["num_inference_steps"], + ) + + +def run_with_file( + pipeline: QwenImage21Pipeline, + prompt: str, + first_image_path: str, + output_path: str, + seed: int = PPL_CONFIG["seed"], + height: int | None = None, + width: int | None = None, +) -> dict[str, str]: + """Save an edited image for the standard service entrypoint.""" + + with Image.open(first_image_path) as source: + images = run(pipeline, prompt, source.copy(), seed=seed, height=height, width=width) + destination = Path(output_path) + destination.parent.mkdir(parents=True, exist_ok=True) + images[0].save(destination) + return {"output_path": str(destination)} + + +@click.command() +@click.option("--gpu_num", default=1, type=int, show_default=True) +@click.option("--model_root", default=PPL_CONFIG["model_root"], show_default=True) +@click.option("--image_path", default=PPL_CONFIG["image_path"], type=click.Path(exists=True), show_default=True) +@click.option("--prompt", default=PPL_CONFIG["prompt"], show_default=True) +@click.option("--seed", default=PPL_CONFIG["seed"], type=int, show_default=True) +@click.option("--height", type=int, default=None) +@click.option("--width", type=int, default=None) +@click.option("--output_path", type=click.Path(path_type=Path), default=Path("work_dirs/qwen-image-2.1-edit.png")) +def main( + gpu_num: int, + model_root: str, + image_path: str, + prompt: str, + seed: int, + height: int | None, + width: int | None, + output_path: Path, +) -> None: + """Run Qwen-Image 2.1 image editing.""" + + pipeline = get_pipeline(gpu_num, model_root) + result = run_with_file(pipeline, prompt, image_path, str(output_path), seed=seed, height=height, width=width) + print(f"Image saved to: {result['output_path']}") + + +if __name__ == "__main__": + main() diff --git a/examples/qwen_image/qwen_image_21_reference_h100.py b/examples/qwen_image/qwen_image_21_reference_h100.py new file mode 100644 index 00000000..23c61b28 --- /dev/null +++ b/examples/qwen_image/qwen_image_21_reference_h100.py @@ -0,0 +1,157 @@ +"""Generate a new image from one or more references with native Qwen-Image 2.1. + +Usage: + CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_reference_h100.py \ + --output_path work_dirs/qwen-image-2.1-reference.png +""" + +from __future__ import annotations + +import os +from pathlib import Path + +import click +import torch +from PIL import Image + +from telefuser.pipelines.qwen_image import QwenImage21Pipeline +from telefuser.service.core.contract_templates import build_pipeline_manifest, build_task_contract_template + +PPL_CONFIG = { + "name": "qwen_image_2.1_reference", + "model_root": os.path.join(os.environ.get("TF_MODEL_ZOO_PATH", "/hhb-data/aigc/model_zoo"), "Qwen-Image-2.1"), + "prompt": ( + "让【图1】中的模特换上【图3】中的玛丽珍鞋,拿着【图4】中的手提包,并戴上【图5】中的绒毛帽。" + "将【图2】中的羽绒服敞开穿在外面,露出原有的内搭上衣。保持模特姿势和背景不变。" + ), + "image_paths": ( + "examples/data/qwen_image_21/447da49d-dcd0-46ec-9ca8-20127c0f07d1.webp", + "examples/data/qwen_image_21/0f2d052b-6ccf-46e2-821d-d073ead679ac.webp", + "examples/data/qwen_image_21/3f4d7519-6c33-491f-bad8-661f1bf03601.webp", + "examples/data/qwen_image_21/0f5d38f9-d04a-498e-aaef-9dae01da8c6d.webp", + "examples/data/qwen_image_21/3acf3276-0572-490e-b8ce-334833a6ca20.webp", + ), + "seed": 42, + "num_inference_steps": 40, +} + +PIPELINE_CONTRACT = build_pipeline_manifest( + pipeline_name=PPL_CONFIG["name"], + supported_tasks=["i2i"], + task_contracts={ + "i2i": build_task_contract_template( + "i2i", + parameter_overrides={ + "prompt": {"default": PPL_CONFIG["prompt"]}, + "seed": {"default": PPL_CONFIG["seed"]}, + }, + excluded_parameters=["negative_prompt", "resolution", "aspect_ratio"], + ) + }, +) + + +def get_pipeline( + parallelism: int = 1, + model_root: str = PPL_CONFIG["model_root"], + device: str | None = None, +) -> QwenImage21Pipeline: + """Load the shared native Qwen-Image 2.1 pipeline.""" + + if parallelism != 1: + raise ValueError("Qwen-Image 2.1 example currently supports one GPU") + return QwenImage21Pipeline.from_pretrained(model_root, device=device or "cuda", torch_dtype=torch.bfloat16) + + +def run( + pipeline: QwenImage21Pipeline, + prompt: str, + images: list[Image.Image], + seed: int = PPL_CONFIG["seed"], + height: int | None = None, + width: int | None = None, +) -> list[Image.Image]: + """Generate a new image from one to ten shared reference images.""" + + if not 1 <= len(images) <= 10: + raise ValueError("Pass one to ten reference images") + return pipeline( + prompt=prompt, + image=images, + seed=seed, + height=height, + width=width, + num_inference_steps=PPL_CONFIG["num_inference_steps"], + ) + + +def run_with_file( + pipeline: QwenImage21Pipeline, + prompt: str, + first_image_path: str, + output_path: str, + reference_image_paths: tuple[str, ...] = (), + seed: int = PPL_CONFIG["seed"], + height: int | None = None, + width: int | None = None, +) -> dict[str, str]: + """Save reference-guided output for CLI or the single-image service entrypoint.""" + + paths = (first_image_path, *reference_image_paths) + images = [] + for path in paths: + with Image.open(path) as source: + images.append(source.copy()) + outputs = run(pipeline, prompt, images, seed=seed, height=height, width=width) + destination = Path(output_path) + destination.parent.mkdir(parents=True, exist_ok=True) + outputs[0].save(destination) + return {"output_path": str(destination)} + + +@click.command() +@click.option("--gpu_num", default=1, type=int, show_default=True) +@click.option("--model_root", default=PPL_CONFIG["model_root"], show_default=True) +@click.option( + "--image_path", + "image_paths", + multiple=True, + type=click.Path(exists=True), + default=PPL_CONFIG["image_paths"], + show_default=True, +) +@click.option("--prompt", default=PPL_CONFIG["prompt"], show_default=True) +@click.option("--seed", default=PPL_CONFIG["seed"], type=int, show_default=True) +@click.option("--height", type=int, default=None) +@click.option("--width", type=int, default=None) +@click.option("--output_path", type=click.Path(path_type=Path), default=Path("work_dirs/qwen-image-2.1-reference.png")) +def main( + gpu_num: int, + model_root: str, + image_paths: tuple[str, ...], + prompt: str, + seed: int, + height: int | None, + width: int | None, + output_path: Path, +) -> None: + """Run Qwen-Image 2.1 reference-guided generation.""" + + if not 1 <= len(image_paths) <= 10: + raise click.BadParameter("Pass one to ten --image_path values") + pipeline = get_pipeline(gpu_num, model_root) + result = run_with_file( + pipeline, + prompt, + image_paths[0], + str(output_path), + reference_image_paths=image_paths[1:], + seed=seed, + height=height, + width=width, + ) + print(f"Image saved to: {result['output_path']}") + + +if __name__ == "__main__": + main() diff --git a/examples/qwen_image/qwen_image_21_t2i_h100.py b/examples/qwen_image/qwen_image_21_t2i_h100.py new file mode 100644 index 00000000..14acb1ac --- /dev/null +++ b/examples/qwen_image/qwen_image_21_t2i_h100.py @@ -0,0 +1,179 @@ +"""Qwen-Image 2.1 text-to-image generation. + +Usage: + CUDA_VISIBLE_DEVICES=3 python examples/qwen_image/qwen_image_21_t2i_h100.py \ + --model_root /hhb-data/aigc/model_zoo/Qwen-Image-2.1 \ + --prompt "A paper boat on a moonlit lake" \ + --output_path work_dirs/qwen-image-2.1.png +""" + +from __future__ import annotations + +import os +from pathlib import Path +from typing import Any + +import click +import torch +from PIL import Image + +from telefuser.pipelines.qwen_image import QwenImage21Pipeline +from telefuser.pipelines.qwen_image.qwen_image import ASPECT_RATIO_TO_SIZE +from telefuser.service.core.contract_templates import build_pipeline_manifest, build_task_contract_template +from telefuser.utils.utils import get_example_name + +TF_MODEL_ZOO_PATH = os.environ.get("TF_MODEL_ZOO_PATH", "/hhb-data/aigc/model_zoo") +PPL_CONFIG = { + "name": "qwen_image_2.1_t2i", + "model_root": os.path.join(TF_MODEL_ZOO_PATH, "Qwen-Image-2.1"), + "prompt": ( + "A 20-year-old East Asian girl with delicate, charming features and large, bright brown eyes—expressive and " + "lively, with a cheerful or subtly smiling expression. Her naturally wavy long hair is either loose or tied " + "in twin ponytails. She has fair skin and light makeup accentuating her youthful freshness. She wears a modern, " + "cute dress or relaxed outfit in bright, soft colors—lightweight fabric, minimalist cut. She stands indoors at " + "an anime convention, surrounded by banners, posters, or stalls. Lighting is typical indoor illumination—no " + "staged lighting—and the image resembles a casual iPhone snapshot: unpretentious composition, yet brimming " + "with vivid, fresh, youthful charm." + ), + "negative_prompt": ( + "低分辨率,低画质,肢体畸形,手指畸形,画面过饱和,蜡像感,人脸无细节,过度光滑," + "画面具有AI感。构图混乱。文字模糊,扭曲。" + ), + "seed": 42, + "aspect_ratio": "16:9", + "num_inference_steps": 40, + "cfg_scale": 1.0, + "output_resolution": 1024, +} + +PIPELINE_CONTRACT = build_pipeline_manifest( + pipeline_name=PPL_CONFIG["name"], + supported_tasks=["t2i"], + task_contracts={ + "t2i": build_task_contract_template( + "t2i", + parameter_overrides={ + "prompt": {"default": PPL_CONFIG["prompt"]}, + "negative_prompt": {"default": PPL_CONFIG["negative_prompt"]}, + "seed": {"default": PPL_CONFIG["seed"]}, + "aspect_ratio": {"default": PPL_CONFIG["aspect_ratio"]}, + }, + excluded_parameters=["resolution"], + ) + }, +) + + +def get_pipeline( + parallelism: int = 1, + model_root: str = PPL_CONFIG["model_root"], + device: str | None = None, +) -> QwenImage21Pipeline: + """Load Qwen-Image 2.1 for a single GPU.""" + + if parallelism != 1: + raise ValueError("Qwen-Image 2.1 example currently supports one GPU; use CUDA_VISIBLE_DEVICES to select it") + runtime_device = device or "cuda" + return QwenImage21Pipeline.from_pretrained( + model_root, + device=runtime_device, + torch_dtype=torch.bfloat16, + ) + + +def run( + pipeline: QwenImage21Pipeline, + prompt: str, + negative_prompt: str = PPL_CONFIG["negative_prompt"], + seed: int = PPL_CONFIG["seed"], + aspect_ratio: str = PPL_CONFIG["aspect_ratio"], + height: int | None = None, + width: int | None = None, + num_inference_steps: int = PPL_CONFIG["num_inference_steps"], + cfg_scale: float = PPL_CONFIG["cfg_scale"], +) -> list[Image.Image]: + """Generate images using the existing Qwen-Image aspect-ratio sizes.""" + + default_width, default_height = ASPECT_RATIO_TO_SIZE[aspect_ratio] + + return pipeline( + prompt=prompt, + negative_prompt=negative_prompt if cfg_scale > 1 else None, + seed=seed, + height=height if height is not None else default_height, + width=width if width is not None else default_width, + num_inference_steps=num_inference_steps, + cfg_scale=cfg_scale, + output_resolution=PPL_CONFIG["output_resolution"], + ) + + +def run_with_file( + pipeline: QwenImage21Pipeline, + prompt: str, + output_path: str, + negative_prompt: str = PPL_CONFIG["negative_prompt"], + seed: int = PPL_CONFIG["seed"], + aspect_ratio: str = PPL_CONFIG["aspect_ratio"], + height: int | None = None, + width: int | None = None, + **kwargs: Any, +) -> dict[str, str]: + """Generate an image and save it for the TeleFuser service contract.""" + + images = run( + pipeline, + prompt=prompt, + negative_prompt=negative_prompt, + seed=seed, + aspect_ratio=aspect_ratio, + height=height, + width=width, + cfg_scale=float(kwargs.get("cfg_scale", PPL_CONFIG["cfg_scale"])), + ) + destination = Path(output_path) + destination.parent.mkdir(parents=True, exist_ok=True) + images[0].save(destination) + return {"output_path": str(destination)} + + +@click.command() +@click.option("--gpu_num", default=1, type=int, show_default=True, help="Number of GPUs; Qwen-Image 2.1 uses one GPU") +@click.option("--model_root", default=PPL_CONFIG["model_root"], show_default=True, help="Model directory") +@click.option("--prompt", default=PPL_CONFIG["prompt"], show_default=True, help="Text prompt") +@click.option("--negative_prompt", default=PPL_CONFIG["negative_prompt"], show_default=True) +@click.option("--aspect_ratio", "-ar", default=PPL_CONFIG["aspect_ratio"], show_default=True) +@click.option("--height", type=int, default=None) +@click.option("--width", type=int, default=None) +@click.option("--seed", default=PPL_CONFIG["seed"], type=int, show_default=True) +@click.option("--output_path", type=click.Path(path_type=Path), default=None) +def main( + gpu_num: int, + model_root: str, + prompt: str, + negative_prompt: str, + aspect_ratio: str, + height: int | None, + width: int | None, + seed: int, + output_path: Path | None, +) -> None: + """Run Qwen-Image 2.1 inference.""" + + pipeline = get_pipeline(gpu_num, model_root) + destination = output_path or Path(get_example_name(__file__, "png")) + run_with_file( + pipeline, + prompt=prompt, + negative_prompt=negative_prompt, + aspect_ratio=aspect_ratio, + seed=seed, + height=height, + width=width, + output_path=str(destination), + ) + print(f"Image saved to: {destination}") + + +if __name__ == "__main__": + main() diff --git a/examples/qwen_image/qwen_image_t2i_h100.py b/examples/qwen_image/qwen_image_t2i_h100.py index ba4c5b78..0a0d5399 100644 --- a/examples/qwen_image/qwen_image_t2i_h100.py +++ b/examples/qwen_image/qwen_image_t2i_h100.py @@ -14,6 +14,15 @@ from telefuser.utils.utils import get_example_name TF_MODEL_ZOO_PATH = os.environ.get("TF_MODEL_ZOO_PATH", "model_zoo") +DEFAULT_PROMPT = ( + "A 20-year-old East Asian girl with delicate, charming features and large, bright brown eyes—expressive and " + "lively, with a cheerful or subtly smiling expression. Her naturally wavy long hair is either loose or tied " + "in twin ponytails. She has fair skin and light makeup accentuating her youthful freshness. She wears a modern, " + "cute dress or relaxed outfit in bright, soft colors—lightweight fabric, minimalist cut. She stands indoors at " + "an anime convention, surrounded by banners, posters, or stalls. Lighting is typical indoor illumination—no " + "staged lighting—and the image resembles a casual iPhone snapshot: unpretentious composition, yet brimming " + "with vivid, fresh, youthful charm." +) PPL_CONFIG = dict( name="qwen_image_t2i", model_root=TF_MODEL_ZOO_PATH + "/Qwen-Image-2512", @@ -38,7 +47,10 @@ "text_encoder/model-00004-of-00004.safetensors", ], tokenizer_path="tokenizer", - negative_prompt="低分辨率,低画质,肢体畸形,手指畸形,画面过饱和,蜡像感,人脸无细节,过度光滑,画面具有AI感。构图混乱。文字模糊,扭曲。", + negative_prompt=( + "低分辨率,低画质,肢体畸形,手指畸形,画面过饱和,蜡像感,人脸无细节,过度光滑," + "画面具有AI感。构图混乱。文字模糊,扭曲。" + ), attn_impl=AttnImplType.TORCH_SDPA, seed=42, sample_solver="euler", @@ -115,7 +127,7 @@ def run( @click.option("--gpu_num", default=1, help="Number of GPUs to use", type=int) @click.option( "--prompt", - default="A 20-year-old East Asian girl with delicate, charming features and large, bright brown eyes—expressive and lively, with a cheerful or subtly smiling expression. Her naturally wavy long hair is either loose or tied in twin ponytails. She has fair skin and light makeup accentuating her youthful freshness. She wears a modern, cute dress or relaxed outfit in bright, soft colors—lightweight fabric, minimalist cut. She stands indoors at an anime convention, surrounded by banners, posters, or stalls. Lighting is typical indoor illumination—no staged lighting—and the image resembles a casual iPhone snapshot: unpretentious composition, yet brimming with vivid, fresh, youthful charm.", + default=DEFAULT_PROMPT, help="Custom prompt text", ) @click.option("--negative_prompt", default=PPL_CONFIG["negative_prompt"], help="Negative prompt") diff --git a/telefuser/models/qwen_image_21_dit.py b/telefuser/models/qwen_image_21_dit.py new file mode 100644 index 00000000..acd151dc --- /dev/null +++ b/telefuser/models/qwen_image_21_dit.py @@ -0,0 +1,384 @@ +"""Native Qwen-Image 2.1 diffusion transformer. + +The 2.1 checkpoint is a single-stream transformer. Text tokens form a causal +prefix and the target image tokens form one bidirectional block. This module +keeps that computation in TeleFuser instead of delegating the model forward to +Diffusers. +""" + +from __future__ import annotations + +import math +import os +from typing import Any + +import torch +import torch.nn as nn +import torch.nn.functional as F + +from telefuser.core.base_model import BaseModel +from telefuser.core.config import AttentionConfig, AttnImplType +from telefuser.ops.attention import attention as attn_func +from telefuser.ops.normalization import RMSNorm +from telefuser.utils.model_weight import init_weights_on_device, load_state_dict + + +class QwenImage21TemporalTimesteps(nn.Module): + """Cosine/sine timestep embedding used by the 2.1 checkpoint.""" + + def __init__(self, timestep_dim: int = 256, max_period: int = 10000, time_factor: float = 1000.0): + super().__init__() + half = timestep_dim // 2 + freqs = torch.exp(-math.log(max_period) * torch.arange(half, dtype=torch.float32) / half) + self.register_buffer("freqs", freqs, persistent=False) + self.timestep_dim = timestep_dim + self.time_factor = time_factor + + def forward(self, timestep: torch.Tensor) -> torch.Tensor: + args = timestep.float()[:, None] * self.time_factor * self.freqs.to(timestep.device)[None] + embedding = torch.cat([torch.cos(args), torch.sin(args)], dim=-1) + if self.timestep_dim % 2: + embedding = F.pad(embedding, (0, 1)) + return embedding.to(timestep.dtype) + + +class QwenImage21TimestepProjEmbeddings(nn.Module): + def __init__(self, embedding_dim: int): + super().__init__() + self.time_proj = QwenImage21TemporalTimesteps() + self.timestep_embedder = QwenImage21TimestepEmbedding(embedding_dim) + + def forward(self, timestep: torch.Tensor, dtype: torch.dtype) -> torch.Tensor: + return self.timestep_embedder(self.time_proj(timestep).to(dtype=dtype)) + + +class QwenImage21TimestepEmbedding(nn.Module): + """Named linear layers matching the Diffusers checkpoint layout.""" + + def __init__(self, embedding_dim: int): + super().__init__() + self.linear_1 = nn.Linear(256, embedding_dim, bias=False) + self.act = nn.SiLU() + self.linear_2 = nn.Linear(embedding_dim, embedding_dim, bias=False) + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + return self.linear_2(self.act(self.linear_1(hidden_states))) + + +class ZeroCenterRMSNorm(nn.Module): + """RMSNorm whose stored scale is centered at zero in the checkpoint.""" + + def __init__(self, dim: int, eps: float = 1e-6): + super().__init__() + self.weight = nn.Parameter(torch.zeros(dim)) + self.eps = eps + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + dtype = hidden_states.dtype + hidden_states = hidden_states.float() + norm = torch.rsqrt(hidden_states.square().mean(dim=-1, keepdim=True) + self.eps) + return (hidden_states * norm * (self.weight.float() + 1.0)).to(dtype) + + +class QwenImage21TextProjection(nn.Module): + def __init__(self, context_in_dim: int, hidden_size: int, eps: float = 1e-6): + super().__init__() + self.text_norm = ZeroCenterRMSNorm(context_in_dim, eps) + self.in_layer = nn.Linear(context_in_dim, hidden_size, bias=False) + self.act = nn.GELU(approximate="tanh") + self.out_layer = nn.Linear(hidden_size, hidden_size, bias=False) + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + return self.out_layer(self.act(self.in_layer(self.text_norm(hidden_states)))) + + +class QwenImage21SwiGLU(nn.Module): + def __init__(self, hidden_size: int, mlp_hidden_size: int): + super().__init__() + self.proj = nn.Linear(hidden_size, mlp_hidden_size, bias=False) + self.gate_layer = nn.Linear(hidden_size, mlp_hidden_size, bias=False) + self.out = nn.Linear(mlp_hidden_size, hidden_size, bias=False) + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + return self.out(F.silu(self.gate_layer(hidden_states)) * self.proj(hidden_states)) + + +def _rope_params(index: torch.Tensor, dim: int, theta: int = 10000) -> torch.Tensor: + freqs = torch.outer( + index.float(), 1.0 / torch.pow(theta, torch.arange(0, dim, 2, device=index.device).float() / dim) + ) + return torch.polar(torch.ones_like(freqs), freqs) + + +class QwenImage21Rope(nn.Module): + """Three-axis rotary embedding for a text prefix and image token block.""" + + def __init__(self, axes_dims: tuple[int, int, int] = (16, 56, 56), theta: int = 10000): + super().__init__() + self.axes_dims = axes_dims + self.theta = theta + + def forward( + self, img_shapes: list[tuple[int, int, int]], image_pad_mask: torch.Tensor, device: torch.device + ) -> torch.Tensor: + mask = image_pad_mask.tolist() + frame_index, height_index, width_index = [], [], [] + cursor = position = 0 + for frames, height, width in img_shapes: + if frames != 1: + raise ValueError("Qwen-Image 2.1 supports one frame per condition image") + block_start = mask.index(True, cursor) + text_indices = list(range(position, position + block_start - cursor)) + frame_index.extend(text_indices) + height_index.extend(text_indices) + width_index.extend(text_indices) + position += block_start - cursor + block_length = height * width + frame_index.extend([position] * block_length) + height_index.extend(h for h in range(-(height - height // 2), height // 2) for _ in range(width)) + width_index.extend(w for _ in range(height) for w in range(-(width - width // 2), width // 2)) + cursor = block_start + block_length + position += max(height, width) + trailing = list(range(position, position + len(mask) - cursor)) + frame_index.extend(trailing) + height_index.extend(trailing) + width_index.extend(trailing) + if len(frame_index) != len(mask): + raise ValueError("Image shapes do not match the expanded image-token mask") + frame = torch.tensor(frame_index, device=device) + height_index = torch.tensor(height_index, device=device) + width_index = torch.tensor(width_index, device=device) + return torch.cat( + [ + _rope_params(frame, self.axes_dims[0], self.theta), + _rope_params(height_index, self.axes_dims[1], self.theta), + _rope_params(width_index, self.axes_dims[2], self.theta), + ], + dim=-1, + ) + + +def apply_rotary_emb(x: torch.Tensor, freqs: torch.Tensor) -> torch.Tensor: + """Apply complex rotary frequencies to ``(B, S, heads, head_dim)`` tensors.""" + + x_complex = torch.view_as_complex(x.float().reshape(*x.shape[:-1], -1, 2)) + rotated = x_complex * freqs.to(x.device)[None, :, None] + return torch.view_as_real(rotated).flatten(-2).to(x.dtype) + + +class QwenImage21Attention(nn.Module): + def __init__(self, dim: int, heads: int, head_dim: int, eps: float): + super().__init__() + self.heads = heads + self.to_q = nn.Linear(dim, heads * head_dim, bias=False) + self.to_k = nn.Linear(dim, heads * head_dim, bias=False) + self.to_v = nn.Linear(dim, heads * head_dim, bias=False) + self.norm_q = RMSNorm(head_dim, eps=eps) + self.norm_k = RMSNorm(head_dim, eps=eps) + self.to_out = nn.ModuleList([nn.Linear(heads * head_dim, dim, bias=False), nn.Dropout(0.0)]) + self.attention_config = AttentionConfig.dense_attention(AttnImplType.TORCH_SDPA) + + def set_attention_config(self, config: AttentionConfig) -> None: + self.attention_config = config + + def forward( + self, hidden_states: torch.Tensor, rotary_emb: torch.Tensor, attention_mask: torch.Tensor + ) -> torch.Tensor: + query = self.to_q(hidden_states).unflatten(-1, (self.heads, -1)) + key = self.to_k(hidden_states).unflatten(-1, (self.heads, -1)) + value = self.to_v(hidden_states).unflatten(-1, (self.heads, -1)) + query = apply_rotary_emb(self.norm_q(query), rotary_emb) + key = apply_rotary_emb(self.norm_k(key), rotary_emb) + output = attn_func( + query, + key, + value, + attention_config=self.attention_config, + attn_mask=attention_mask, + input_layout="BSND", + output_layout="BSND", + ).flatten(-2) + return self.to_out[1](self.to_out[0](output)) + + +class QwenImage21TransformerBlock(nn.Module): + def __init__(self, dim: int, heads: int, head_dim: int, mlp_ratio: int, eps: float): + super().__init__() + self.img_norm1 = nn.LayerNorm(dim, eps=eps, elementwise_affine=False) + self.attn = QwenImage21Attention(dim, heads, head_dim, eps) + self.img_norm2 = nn.LayerNorm(dim, eps=eps, elementwise_affine=False) + self.img_mlp = QwenImage21SwiGLU(dim, dim * mlp_ratio) + + @staticmethod + def _modulate( + hidden_states: torch.Tensor, params: torch.Tensor, target_mask: torch.Tensor + ) -> tuple[torch.Tensor, torch.Tensor]: + scale, gate = params.chunk(2, dim=-1) + scale = _select_modulation(scale, target_mask) + gate = _select_modulation(gate, target_mask) + return hidden_states * (1 + scale), gate + + def forward( + self, + hidden_states: torch.Tensor, + modulation: torch.Tensor, + target_mask: torch.Tensor, + rotary_emb: torch.Tensor, + attention_mask: torch.Tensor, + ) -> torch.Tensor: + mod_attn, mod_mlp = modulation.chunk(2, dim=-1) + normed, gate = self._modulate(self.img_norm1(hidden_states), mod_attn, target_mask) + hidden_states = hidden_states + gate.tanh() * self.attn(normed, rotary_emb, attention_mask) + normed, gate = self._modulate(self.img_norm2(hidden_states), mod_mlp, target_mask) + hidden_states = hidden_states + gate.tanh() * self.img_mlp(normed) + return hidden_states.clip(-65504, 65504) if hidden_states.dtype == torch.float16 else hidden_states + + +def _select_modulation(params: torch.Tensor, target_mask: torch.Tensor) -> torch.Tensor: + real = params[:-1].unsqueeze(1) + zero = params[-1:].unsqueeze(0) + return torch.where(target_mask.view(1, -1, 1), real, zero) + + +class QwenImage21AdaLayerNormContinuous(nn.Module): + def __init__(self, dim: int, eps: float): + super().__init__() + self.silu = nn.SiLU() + self.linear = nn.Linear(dim, dim, bias=False) + self.norm = nn.LayerNorm(dim, eps=eps, elementwise_affine=False) + + def forward( + self, hidden_states: torch.Tensor, conditioning: torch.Tensor, target_mask: torch.Tensor + ) -> torch.Tensor: + scale = _select_modulation(self.linear(self.silu(conditioning)), target_mask) + return self.norm(hidden_states) * (1 + scale) + + +class QwenImage21DiT(BaseModel): + """Native single-stream Qwen-Image 2.1 DiT.""" + + def __init__( + self, + patch_size: int = 1, + in_channels: int = 64, + out_channels: int = 64, + num_layers: int = 32, + attention_head_dim: int = 128, + num_attention_heads: int = 32, + context_in_dim: int = 4096, + mlp_ratio: int = 3, + axes_dims_rope: tuple[int, int, int] = (16, 56, 56), + eps: float = 1e-6, + causal_condition: bool = True, + ): + super().__init__() + del patch_size + self.in_channels = in_channels + self.out_channels = out_channels + self.inner_dim = num_attention_heads * attention_head_dim + self.causal_condition = causal_condition + self.pos_embed = QwenImage21Rope(axes_dims_rope) + self.time_text_embed = QwenImage21TimestepProjEmbeddings(self.inner_dim) + self.txt_in = QwenImage21TextProjection(context_in_dim, self.inner_dim, eps) + self.img_in = nn.Linear(in_channels, self.inner_dim, bias=False) + self.modulation = nn.Sequential(nn.SiLU(), nn.Linear(self.inner_dim, 4 * self.inner_dim, bias=False)) + self.transformer_blocks = nn.ModuleList( + [ + QwenImage21TransformerBlock(self.inner_dim, num_attention_heads, attention_head_dim, mlp_ratio, eps) + for _ in range(num_layers) + ] + ) + self.norm_out = QwenImage21AdaLayerNormContinuous(self.inner_dim, eps) + self.proj_out = nn.Linear(self.inner_dim, out_channels, bias=False) + self.attention_config = AttentionConfig.dense_attention(AttnImplType.TORCH_SDPA) + + def set_attention_config(self, config: AttentionConfig) -> None: + self.attention_config = config + for block in self.transformer_blocks: + block.attn.set_attention_config(config) + + def forward( + self, + hidden_states: torch.Tensor, + encoder_hidden_states: torch.Tensor, + timestep: torch.Tensor, + img_shapes: list[list[tuple[int, int, int]]], + encoder_hidden_states_mask: torch.Tensor | None = None, + img_mask: torch.Tensor | None = None, + return_dict: bool = False, + **_: Any, + ) -> torch.Tensor | tuple[torch.Tensor]: + batch_size, target_tokens, _ = hidden_states.shape + text_tokens = encoder_hidden_states.shape[1] + if len(img_shapes) != batch_size or any(shapes != img_shapes[0] for shapes in img_shapes): + raise ValueError("All batch items must share the same image-token layout") + target_tokens = math.prod(img_shapes[0][-1]) + if sum(math.prod(shape) for shape in img_shapes[0]) != hidden_states.shape[1]: + raise ValueError("Image shapes must match condition and target latent tokens") + if img_mask is None: + img_mask = torch.zeros(batch_size, text_tokens, dtype=torch.bool, device=hidden_states.device) + target_mask = torch.ones(batch_size, target_tokens // 4, dtype=torch.bool, device=hidden_states.device) + full_mask = torch.cat([img_mask, target_mask], dim=1) + repeats = torch.where(full_mask[0], 4, 1) + image_pad_mask = torch.repeat_interleave(full_mask[0], repeats) + embedded_text = self.txt_in(encoder_hidden_states) + joint = torch.cat( + [embedded_text, embedded_text.new_zeros(batch_size, target_tokens // 4, self.inner_dim)], dim=1 + ).repeat_interleave(repeats, dim=1) + joint[:, image_pad_mask] = self.img_in(hidden_states) + image_positions = image_pad_mask.nonzero(as_tuple=True)[0] + image_ids = torch.full((joint.shape[1],), -1, dtype=torch.long, device=hidden_states.device) + cursor = 0 + for image_id, shape in enumerate(img_shapes[0]): + length = math.prod(shape) + image_ids[image_positions[cursor : cursor + length]] = image_id + cursor += length + rotary_emb = self.pos_embed(img_shapes[0], image_pad_mask, hidden_states.device) + target_token_mask = torch.zeros(joint.shape[1], dtype=torch.bool, device=hidden_states.device) + target_token_mask[-target_tokens:] = True + valid_keys = torch.ones(batch_size, joint.shape[1], dtype=torch.bool, device=hidden_states.device) + if encoder_hidden_states_mask is not None: + text_positions = (~image_pad_mask).nonzero(as_tuple=True)[0] + valid_keys[:, text_positions] = encoder_hidden_states_mask[:, ~img_mask[0]].bool() + indices = torch.arange(joint.shape[1], device=hidden_states.device) + same_image = (image_ids[:, None] == image_ids[None, :]) & (image_ids[:, None] >= 0) + allowed = (indices[:, None] >= indices[None, :]) | same_image + allowed = allowed[None, None] & valid_keys[:, None, None, :] + timestep = timestep.to(hidden_states.dtype) + if self.causal_condition: + timestep = torch.cat([timestep, timestep.new_zeros(1)], dim=0) + conditioning = self.time_text_embed(timestep, hidden_states.dtype) + modulation = self.modulation(conditioning) + for block in self.transformer_blocks: + joint = block(joint, modulation, target_token_mask, rotary_emb, allowed) + joint = self.norm_out(joint, conditioning, target_token_mask) + output = self.proj_out(joint[:, -target_tokens:]) + return (output,) if not return_dict else output + + @staticmethod + def state_dict_converter() -> "QwenImage21DiTStateDictConverter": + return QwenImage21DiTStateDictConverter() + + @classmethod + def from_pretrained(cls, path: str, torch_dtype: torch.dtype = torch.bfloat16) -> "QwenImage21DiT": + config_path = os.path.join(path, "config.json") + import json + + with open(config_path, encoding="utf-8") as handle: + config = json.load(handle) + index_path = os.path.join(path, "diffusion_pytorch_model.safetensors.index.json") + with init_weights_on_device("meta"): + model = cls(**{key: value for key, value in config.items() if key in cls.__init__.__code__.co_varnames}) + state = load_state_dict(index_path, torch_dtype=torch_dtype) + model.load_state_dict(state, assign=True) + return model.to(dtype=torch_dtype).eval() + + +class QwenImage21DiTStateDictConverter: + @staticmethod + def from_diffusers(state_dict: dict[str, torch.Tensor]) -> tuple[dict[str, torch.Tensor], dict[str, Any]]: + return state_dict, {} + + @staticmethod + def from_official(state_dict: dict[str, torch.Tensor]) -> tuple[dict[str, torch.Tensor], dict[str, Any]]: + return state_dict, {} diff --git a/telefuser/pipelines/qwen_image/__init__.py b/telefuser/pipelines/qwen_image/__init__.py index c75d42cf..f72a0175 100644 --- a/telefuser/pipelines/qwen_image/__init__.py +++ b/telefuser/pipelines/qwen_image/__init__.py @@ -6,6 +6,14 @@ """ from .qwen_image import QwenImagePipeline, QwenImagePipelineConfig +from .qwen_image_21 import QwenImage21Pipeline, QwenImage21PipelineConfig from .qwen_image_edit import QwenImageEditPipeline, QwenImageEditPipelineConfig -__all__ = ["QwenImagePipeline", "QwenImagePipelineConfig", "QwenImageEditPipeline", "QwenImageEditPipelineConfig"] +__all__ = [ + "QwenImagePipeline", + "QwenImagePipelineConfig", + "QwenImage21Pipeline", + "QwenImage21PipelineConfig", + "QwenImageEditPipeline", + "QwenImageEditPipelineConfig", +] diff --git a/telefuser/pipelines/qwen_image/dit_denoising_21.py b/telefuser/pipelines/qwen_image/dit_denoising_21.py new file mode 100644 index 00000000..32131648 --- /dev/null +++ b/telefuser/pipelines/qwen_image/dit_denoising_21.py @@ -0,0 +1,102 @@ +"""Qwen-Image 2.1 native DiT denoising stage.""" + +from __future__ import annotations + +import torch +from tqdm import tqdm + +from telefuser.core.base_stage import BaseStage, with_model_offload +from telefuser.core.config import ModelRuntimeConfig +from telefuser.core.module_manager import ModuleManager +from telefuser.metrics import with_metrics +from telefuser.models.qwen_image_21_dit import QwenImage21DiT +from telefuser.schedulers.flow_match import FlowMatchScheduler + + +class DitDenoising21Stage(BaseStage): + """Denoise packed 64-channel latents with the native Qwen-Image 2.1 DiT.""" + + def __init__( + self, name: str, module_manager: ModuleManager, config: ModelRuntimeConfig, scheduler: FlowMatchScheduler + ): + super().__init__(name, config) + self.dit: QwenImage21DiT = module_manager.fetch_module("dit") + self.dit.set_attention_config(config.attention_config) + self.model_names = ["dit"] + self.scheduler = scheduler + + def _predict( + self, + latents: torch.Tensor, + condition_latents: torch.Tensor | None, + img_shapes: list[list[tuple[int, int, int]]], + timestep: torch.Tensor, + prompt_embeds: torch.Tensor, + prompt_mask: torch.Tensor | None, + image_mask: torch.Tensor, + negative_image_mask: torch.Tensor | None, + cfg_scale: float, + negative_embeds: torch.Tensor | None, + negative_mask: torch.Tensor | None, + ) -> torch.Tensor: + kwargs = { + "hidden_states": torch.cat([condition_latents, latents], dim=1) + if condition_latents is not None + else latents, + "img_shapes": img_shapes, + "timestep": timestep / 1000, + "encoder_hidden_states": prompt_embeds, + "encoder_hidden_states_mask": prompt_mask, + "img_mask": image_mask, + "return_dict": False, + } + positive = self.dit(**kwargs)[0][:, -latents.shape[1] :] + if cfg_scale <= 1 or negative_embeds is None: + return positive + kwargs["encoder_hidden_states"] = negative_embeds + kwargs["encoder_hidden_states_mask"] = negative_mask + kwargs["img_mask"] = negative_image_mask + negative = self.dit(**kwargs)[0][:, -latents.shape[1] :] + return negative + cfg_scale * (positive - negative) + + @with_model_offload(["dit"]) + @torch.inference_mode() + @with_metrics + def process( + self, + latents: torch.Tensor, + condition_latents: torch.Tensor | None, + img_shapes: list[list[tuple[int, int, int]]], + prompt_embeds: torch.Tensor, + prompt_mask: torch.Tensor | None, + image_mask: torch.Tensor, + negative_image_mask: torch.Tensor | None, + num_inference_steps: int, + cfg_scale: float = 1.0, + negative_embeds: torch.Tensor | None = None, + negative_mask: torch.Tensor | None = None, + ) -> torch.Tensor: + self.scheduler.set_timesteps( + num_inference_steps, + dynamic_shift_len=latents.shape[1], + shift_terminal=0.02, + max_shift=1.15, + ) + for timestep in tqdm(self.scheduler.timesteps): + timestep = timestep.expand(latents.shape[0]).to(self.device, dtype=self.torch_dtype) + with torch.autocast(device_type=self.device_type, dtype=self.torch_dtype): + noise_pred = self._predict( + latents, + condition_latents, + img_shapes, + timestep, + prompt_embeds, + prompt_mask, + image_mask, + negative_image_mask, + cfg_scale, + negative_embeds, + negative_mask, + ) + latents = self.scheduler.step(noise_pred, timestep, latents) + return latents diff --git a/telefuser/pipelines/qwen_image/qwen_image_21.py b/telefuser/pipelines/qwen_image/qwen_image_21.py new file mode 100644 index 00000000..e5d4481c --- /dev/null +++ b/telefuser/pipelines/qwen_image/qwen_image_21.py @@ -0,0 +1,215 @@ +"""Stage-composed TeleFuser pipeline for Qwen-Image 2.1.""" + +from __future__ import annotations + +import math +import os +from dataclasses import dataclass, field +from typing import Any + +import torch +from PIL import Image + +from telefuser.core.base_pipeline import BasePipeline +from telefuser.core.config import AttentionConfig, AttnImplType, ModelRuntimeConfig +from telefuser.core.module_manager import ModuleManager +from telefuser.models.qwen_image_21_dit import QwenImage21DiT +from telefuser.schedulers.flow_match import FlowMatchScheduler +from telefuser.utils.hf_model_utils import resolve_hf_path +from telefuser.utils.logging import logger + +from .dit_denoising_21 import DitDenoising21Stage +from .text_encoding_21 import TextEncoding21Stage +from .vae_21 import VAE21Stage + + +def _diffusers_components() -> tuple[type, type, type]: + """Import only the standalone VAE, text encoder, and processor classes.""" + try: + from diffusers.utils import import_utils + + if getattr(import_utils, "_xformers_available", False): + import_utils._xformers_available = False + from diffusers import AutoencoderKLQwenImage21 + from transformers import AutoProcessor, Qwen3VLForConditionalGeneration + except (ImportError, RuntimeError) as exc: + raise ImportError( + "Qwen-Image 2.1 requires Diffusers with AutoencoderKLQwenImage21 and " + "Transformers with Qwen3VLForConditionalGeneration." + ) from exc + return AutoencoderKLQwenImage21, Qwen3VLForConditionalGeneration, AutoProcessor + + +@dataclass +class QwenImage21PipelineConfig: + """Runtime configuration for the stage-composed Qwen-Image 2.1 pipeline.""" + + vae_config: ModelRuntimeConfig = field(default_factory=ModelRuntimeConfig) + dit_config: ModelRuntimeConfig = field(default_factory=ModelRuntimeConfig) + text_encoding_config: ModelRuntimeConfig = field(default_factory=ModelRuntimeConfig) + sample_solver: str = "euler" + enable_denoising_parallel: bool = False + enable_vae_parallel: bool = False + enable_text_encoding_parallel: bool = False + enable_metrics: bool = False + + +class QwenImage21Pipeline(BasePipeline): + """Qwen-Image 2.1 pipeline assembled from text, DiT, and VAE stages.""" + + def __init__(self, device: str | torch.device, torch_dtype: torch.dtype = torch.bfloat16) -> None: + super().__init__(device=device, torch_dtype=torch_dtype) + self.height_division_factor = 32 + self.width_division_factor = 32 + + def _get_stages(self) -> list[Any]: + return [self.text_encoding_stage, self.denoise_stage, self.vae_stage] + + def init(self, module_manager: ModuleManager, config: QwenImage21PipelineConfig) -> None: + self._model_info = module_manager.get_model_info() + self.config = config + if config.sample_solver != "euler": + raise NotImplementedError(f"solver {config.sample_solver} is not supported") + scheduler = FlowMatchScheduler("Qwen-Image") + self.text_encoding_stage = TextEncoding21Stage("text_encoding", module_manager, config.text_encoding_config) + self.denoise_stage = DitDenoising21Stage("denoise", module_manager, config.dit_config, scheduler) + self.vae_stage = VAE21Stage("vae", module_manager, config.vae_config) + if config.enable_metrics: + self.enable_metrics() + + @torch.no_grad() + def __call__( + self, + prompt: str | list[str], + negative_prompt: str | list[str] | None = None, + image: Image.Image | list[Image.Image] | None = None, + seed: int | None = None, + height: int | None = None, + width: int | None = None, + cfg_scale: float = 1.0, + true_cfg_scale: float | None = None, + num_inference_steps: int = 40, + num_images_per_prompt: int = 1, + output_resolution: int = 1024, + use_kv_cache: bool = False, + **_: Any, + ) -> list[Image.Image]: + """Generate an image from text and optional condition images.""" + del use_kv_cache + if true_cfg_scale is not None: + cfg_scale = true_cfg_scale + condition_images = None + if image is not None: + condition_images = image if isinstance(image, list) else [image] + if not condition_images or len(condition_images) > 10: + raise ValueError("Qwen-Image 2.1 accepts one to ten condition images") + if not all(isinstance(item, Image.Image) for item in condition_images): + raise TypeError("Condition images must be PIL images") + ratio = condition_images[-1].width / condition_images[-1].height + width_at_area = math.sqrt(output_resolution**2 * ratio) + calculated_width = round(width_at_area / 32) * 32 + calculated_height = round((width_at_area / ratio) / 32) * 32 + height = height or calculated_height + width = width or calculated_width + height = height or output_resolution + width = width or output_resolution + height, width = self.check_resize_height_width(height, width) + resized_conditions = None + if condition_images is not None: + resized_conditions = [] + for condition in condition_images: + ratio = condition.width / condition.height + width_at_area = math.sqrt(output_resolution**2 * ratio) + input_width = round(width_at_area / 32) * 32 + input_height = round((width_at_area / ratio) / 32) * 32 + resized_conditions.append( + self.vae_stage.image_processor.resize( + condition.convert("RGBA"), width=input_width, height=input_height + ) + ) + batch_size = len(prompt) if isinstance(prompt, list) else 1 + batch_size *= num_images_per_prompt + latent_height, latent_width = height // 16, width // 16 + generator = torch.Generator(device=self.device) + if seed is not None: + generator.manual_seed(seed) + latents = torch.randn( + batch_size, + latent_height * latent_width, + 64, + generator=generator, + device=self.device, + dtype=self.torch_dtype, + ) + negative = negative_prompt if cfg_scale > 1 else None + prompt_embeds, prompt_mask, image_mask, negative_embeds, negative_mask, negative_image_mask = ( + self.text_encoding_stage.process(prompt, negative, num_images_per_prompt, resized_conditions) + ) + condition_latents = None + condition_shapes = [] + if resized_conditions is not None: + condition_latents, condition_shapes = self.vae_stage.encode_conditions(resized_conditions, batch_size) + latents = self.denoise_stage.process( + latents=latents, + condition_latents=condition_latents, + img_shapes=[[*condition_shapes, (1, latent_height, latent_width)]] * batch_size, + prompt_embeds=prompt_embeds, + prompt_mask=prompt_mask, + image_mask=image_mask, + negative_image_mask=negative_image_mask, + negative_embeds=negative_embeds, + negative_mask=negative_mask, + cfg_scale=cfg_scale, + num_inference_steps=num_inference_steps, + ) + return self.vae_stage.process(latents, latent_height, latent_width) + + @classmethod + def from_pretrained( + cls, + model_id_or_path: str, + device: str = "cuda", + torch_dtype: torch.dtype = torch.bfloat16, + cache_dir: str | None = None, + attention_config: AttentionConfig | None = None, + enable_metrics: bool = False, + **kwargs: Any, + ) -> "QwenImage21Pipeline": + """Load standalone modules into ModuleManager and compose stages.""" + model_root = resolve_hf_path(model_id_or_path, cache_dir) + component_paths = { + "transformer": os.path.join(model_root, "transformer"), + "vae": os.path.join(model_root, "vae"), + "text_encoder": os.path.join(model_root, "text_encoder"), + "processor": os.path.join(model_root, "processor"), + } + for name, path in component_paths.items(): + if not os.path.isdir(path): + raise FileNotFoundError(f"Qwen-Image 2.1 component directory not found: {name} ({path})") + + vae_cls, text_encoder_cls, processor_cls = _diffusers_components() + manager = ModuleManager(torch_dtype=torch_dtype, device="cpu") + manager.load_model( + component_paths["transformer"], + device="cpu", + torch_dtype=torch_dtype, + name="dit", + model_class=QwenImage21DiT, + model_resource="diffusers", + ) + vae = vae_cls.from_pretrained(component_paths["vae"], torch_dtype=torch_dtype) + text_encoder = text_encoder_cls.from_pretrained(component_paths["text_encoder"], torch_dtype=torch_dtype) + processor = processor_cls.from_pretrained(component_paths["processor"]) + manager.add_module(vae, "vae", component_paths["vae"]) + manager.add_module(text_encoder, "text_encoder", component_paths["text_encoder"]) + manager.add_module(processor, "processor", component_paths["processor"]) + + pipeline = cls(device=device, torch_dtype=torch_dtype) + config = QwenImage21PipelineConfig(enable_metrics=enable_metrics) + config.sample_solver = kwargs.pop("sample_solver", "euler") + config.dit_config.attention_config = attention_config or AttentionConfig.dense_attention( + AttnImplType.TORCH_SDPA + ) + pipeline.init(manager, config) + logger.info("Successfully loaded native Qwen-Image 2.1 stages from %s", model_root) + return pipeline diff --git a/telefuser/pipelines/qwen_image/text_encoding_21.py b/telefuser/pipelines/qwen_image/text_encoding_21.py new file mode 100644 index 00000000..b403fc84 --- /dev/null +++ b/telefuser/pipelines/qwen_image/text_encoding_21.py @@ -0,0 +1,150 @@ +"""Qwen-Image 2.1 prompt encoding stage.""" + +from __future__ import annotations + +from typing import Any + +import torch +from PIL import Image + +from telefuser.core.base_stage import BaseStage, with_model_offload +from telefuser.core.config import ModelRuntimeConfig +from telefuser.core.module_manager import ModuleManager + + +class TextEncoding21Stage(BaseStage): + """Encode prompts with the ModuleManager-owned Qwen3-VL model.""" + + def __init__(self, name: str, module_manager: ModuleManager, model_runtime_config: ModelRuntimeConfig): + super().__init__(name, model_runtime_config) + self.text_encoder = module_manager.fetch_module("text_encoder") + self.processor = module_manager.fetch_module("processor") + self.model_names = ["text_encoder"] + self.system_prompt = "Comprehend and analyze the provided prompt." + self.prompt_template = ( + f"<|im_start|>system\n{self.system_prompt}<|im_end|>\n" + "<|im_start|>user\n{}<|im_end|>\n<|im_start|>assistant\n" + ) + sys_message = [{"role": "system", "content": [{"type": "text", "text": self.system_prompt}]}] + self.drop_idx = len(self.processor.apply_chat_template(sys_message, tokenize=True, return_dict=False)[0]) + self.image_token_id = self.processor.tokenizer.encode("<|image_pad|>")[0] + + @staticmethod + def _extract_masked_hidden(hidden_states: torch.Tensor, mask: torch.Tensor) -> list[torch.Tensor]: + bool_mask = mask.bool() + lengths = bool_mask.sum(dim=1) + selected = hidden_states[bool_mask] + return list(torch.split(selected, lengths.tolist(), dim=0)) + + @with_model_offload(["text_encoder"]) + @torch.inference_mode() + def process( + self, + prompt: str | list[str], + negative_prompt: str | list[str] | None = None, + num_images_per_prompt: int = 1, + images: list[Image.Image] | None = None, + ) -> tuple[ + torch.Tensor, + torch.Tensor | None, + torch.Tensor, + torch.Tensor | None, + torch.Tensor | None, + torch.Tensor | None, + ]: + """Return positive/negative embeddings, masks, and image-slot masks.""" + + positive = self._encode(prompt, images) + negative = self._encode(negative_prompt, images) if negative_prompt is not None else None + pos_embeds, pos_mask, pos_image_mask = positive + neg_embeds = neg_mask = neg_image_mask = None + if negative is not None: + neg_embeds, neg_mask, neg_image_mask = negative + batch_size, seq_len, _ = pos_embeds.shape + pos_embeds = pos_embeds.repeat_interleave(num_images_per_prompt, dim=0) + pos_mask = pos_mask.repeat_interleave(num_images_per_prompt, dim=0) if pos_mask is not None else None + pos_image_mask = pos_image_mask.repeat_interleave(num_images_per_prompt, dim=0) + if neg_embeds is not None: + neg_embeds = neg_embeds.repeat_interleave(num_images_per_prompt, dim=0) + neg_mask = neg_mask.repeat_interleave(num_images_per_prompt, dim=0) if neg_mask is not None else None + neg_image_mask = neg_image_mask.repeat_interleave(num_images_per_prompt, dim=0) + del batch_size, seq_len + return pos_embeds, pos_mask, pos_image_mask, neg_embeds, neg_mask, neg_image_mask + + def _encode( + self, prompt: str | list[str], images: list[Image.Image] | None = None + ) -> tuple[torch.Tensor, torch.Tensor | None, torch.Tensor]: + prompts = [prompt] if isinstance(prompt, str) else prompt + if images: + image_prefix = " ".join( + f"<|vision_start|><|image_pad|><|vision_end|>" for index in range(1, len(images) + 1) + ) + prompts = [self.prompt_template.format(f"{image_prefix}{text or ' '}") for text in prompts] + vision_images = [] + for _ in prompts: + for image in images: + if image.mode == "RGBA": + white = Image.new("RGB", image.size, (255, 255, 255)) + white.paste(image, mask=image.getchannel("A")) + vision_images.append(white) + else: + vision_images.append(image) + else: + prompts = [self.prompt_template.format(text or " ") for text in prompts] + vision_images = None + processor_kwargs = { + "text": prompts, + "padding": True, + "padding_side": "left", + "return_tensors": "pt", + } + if vision_images is not None: + processor_kwargs["images"] = vision_images + inputs = self.processor(**processor_kwargs).to(self.device) + model = getattr(self.text_encoder, "model", self.text_encoder) + language_model = getattr(model, "language_model", model) + hook = language_model.norm.register_forward_hook(lambda module, args, output: args[0]) + try: + forward_kwargs = { + "input_ids": inputs.input_ids, + "attention_mask": inputs.attention_mask, + "output_hidden_states": True, + } + if vision_images is not None and hasattr(inputs, "pixel_values"): + forward_kwargs["pixel_values"] = inputs.pixel_values + forward_kwargs["image_grid_thw"] = inputs.image_grid_thw + if hasattr(inputs, "mm_token_type_ids"): + forward_kwargs["mm_token_type_ids"] = inputs.mm_token_type_ids + outputs = self.text_encoder(**forward_kwargs) + finally: + hook.remove() + hidden = outputs.hidden_states[-1] + split_hidden = [item[self.drop_idx :] for item in self._extract_masked_hidden(hidden, inputs.attention_mask)] + image_masks = [ + (ids[mask.bool()] == self.image_token_id)[self.drop_idx :] + for ids, mask in zip(inputs.input_ids, inputs.attention_mask) + ] + max_len = max(item.shape[0] for item in split_hidden) + embeds = torch.stack( + [torch.cat([item, item.new_zeros(max_len - item.shape[0], item.shape[1])]) for item in split_hidden] + ) + valid_masks = torch.stack( + [ + torch.cat( + [ + torch.ones(item.shape[0], dtype=torch.bool, device=item.device), + torch.zeros(max_len - item.shape[0], dtype=torch.bool, device=item.device), + ] + ) + for item in split_hidden + ] + ) + image_mask = torch.stack( + [ + torch.cat([item, torch.zeros(max_len - item.shape[0], dtype=torch.bool, device=item.device)]) + for item in image_masks + ] + ) + if bool(valid_masks.all()): + valid_masks = None + return embeds.to(dtype=self.torch_dtype), valid_masks, image_mask diff --git a/telefuser/pipelines/qwen_image/vae_21.py b/telefuser/pipelines/qwen_image/vae_21.py new file mode 100644 index 00000000..78e7b7e0 --- /dev/null +++ b/telefuser/pipelines/qwen_image/vae_21.py @@ -0,0 +1,62 @@ +"""Qwen-Image 2.1 VAE stage.""" + +from __future__ import annotations + +from typing import Any + +import numpy as np +import torch +from PIL import Image +from diffusers.image_processor import VaeImageProcessor + +from telefuser.core.base_stage import BaseStage, with_model_offload +from telefuser.core.config import ModelRuntimeConfig +from telefuser.core.module_manager import ModuleManager +from telefuser.metrics import with_metrics + + +class VAE21Stage(BaseStage): + """Decode the ModuleManager-owned Qwen-Image 2.1 VAE.""" + + def __init__(self, name: str, module_manager: ModuleManager, config: ModelRuntimeConfig): + super().__init__(name, config) + self.vae = module_manager.fetch_module("vae") + self.model_names = ["vae"] + self.image_processor = VaeImageProcessor(vae_scale_factor=16, vae_latent_channels=64) + + @with_model_offload(["vae"]) + @torch.inference_mode() + @with_metrics + def encode_conditions( + self, images: list[Image.Image], batch_size: int + ) -> tuple[torch.Tensor, list[tuple[int, int, int]]]: + """Encode condition images into the DiT's normalized latent tokens.""" + + mean = torch.tensor(self.vae.config.latents_mean, device=self.device, dtype=self.torch_dtype).view( + 1, 64, 1, 1, 1 + ) + std = torch.tensor(self.vae.config.latents_std, device=self.device, dtype=self.torch_dtype).view(1, 64, 1, 1, 1) + all_latents = [] + shapes = [] + for image in images: + pixels = self.image_processor.preprocess(image, width=image.width, height=image.height).unsqueeze(2) + pixels = pixels.to(device=self.device, dtype=self.torch_dtype) + with torch.autocast(device_type=self.device_type, dtype=self.torch_dtype): + encoded = self.vae.encode(pixels).latent_dist.mode() + normalized = (encoded - mean) / std + latent_height, latent_width = normalized.shape[-2:] + shapes.append((1, latent_height, latent_width)) + packed = normalized.flatten(2).transpose(1, 2) + all_latents.append(packed.repeat(batch_size, 1, 1)) + return torch.cat(all_latents, dim=1), shapes + + @with_model_offload(["vae"]) + @torch.inference_mode() + @with_metrics + def process(self, latents: torch.Tensor, latent_height: int, latent_width: int) -> list[Image.Image]: + mean = torch.tensor(self.vae.config.latents_mean, device=self.device, dtype=latents.dtype).view(1, 64, 1, 1, 1) + std = torch.tensor(self.vae.config.latents_std, device=self.device, dtype=latents.dtype).view(1, 64, 1, 1, 1) + latents = latents.transpose(1, 2).reshape(latents.shape[0], 64, 1, latent_height, latent_width) * std + mean + images = self.vae.decode(latents, return_dict=False)[0][:, :, 0] + images = ((images.float() / 2 + 0.5).clip(0, 1) * 255).byte().permute(0, 2, 3, 1).cpu().numpy() + return [Image.fromarray(item) for item in images] diff --git a/tests/unit/service/test_example_service_parity.py b/tests/unit/service/test_example_service_parity.py index dde078f3..5892740d 100644 --- a/tests/unit/service/test_example_service_parity.py +++ b/tests/unit/service/test_example_service_parity.py @@ -27,6 +27,9 @@ from telefuser.service_types import PipelineRunStatus, TaskStatus SERVICE_EXAMPLES = { + "qwen_image_21_t2i": (Path("examples/qwen_image/qwen_image_21_t2i_h100.py"), "t2i", True), + "qwen_image_21_edit": (Path("examples/qwen_image/qwen_image_21_edit_h100.py"), "i2i", True), + "qwen_image_21_reference": (Path("examples/qwen_image/qwen_image_21_reference_h100.py"), "i2i", True), "wan21_i2v_service": (Path("examples/wan_video/wan21_14b_image_to_video_480p_service.py"), "i2v", True), "minimax_h3_fl2va": (Path("examples/minimax_h3/minimax_h3_fl2va_h100.py"), "t2v", True), "minimax_h3_ref2va": (Path("examples/minimax_h3/minimax_h3_ref2va_h100.py"), "s2v", True), @@ -85,6 +88,21 @@ def _load_example(name: str) -> ModuleType: return module +def test_qwen_image_21_reuses_baseline_output_dimensions() -> None: + module = _load_example("qwen_image_21_t2i") + calls: list[dict[str, Any]] = [] + + def capture_pipeline(**kwargs: Any) -> list[Image.Image]: + calls.append(kwargs) + return [Image.new("RGB", (kwargs["width"], kwargs["height"]))] + + images = module.run(capture_pipeline, module.PPL_CONFIG["prompt"]) + + assert images[0].size == (1664, 928) + assert (calls[0]["width"], calls[0]["height"]) == module.ASPECT_RATIO_TO_SIZE["16:9"] + assert module.PPL_CONFIG["aspect_ratio"] == "16:9" + + def _parameter_value(parameter_type: str) -> object: values = { "boolean": True, diff --git a/tests/unit/test_example_registry.py b/tests/unit/test_example_registry.py index dea5a9c7..db79191e 100644 --- a/tests/unit/test_example_registry.py +++ b/tests/unit/test_example_registry.py @@ -10,6 +10,9 @@ PROJECT_ROOT = Path(__file__).resolve().parents[2] EXAMPLES_ROOT = PROJECT_ROOT / "examples" SERVICE_PARITY_EXAMPLES = { + "qwen_image/qwen_image_21_t2i_h100.py", + "qwen_image/qwen_image_21_edit_h100.py", + "qwen_image/qwen_image_21_reference_h100.py", "lingbot_vla_v2/lingbot_vla_v2_native_service.py", "wan_video/wan21_14b_image_to_video_480p_service.py", "wan_video/wan22_14b_image_to_video_distill_h100.py",