Merge pull request #734 from ROCm/sync-develop-from-external

Sync develop from external for 7.2.2 GA
This commit is contained in:
alexxu-amd
2026-04-14 15:01:54 -04:00
committed by GitHub
9 changed files with 722 additions and 33 deletions
+2
View File
@@ -146,6 +146,7 @@ article_pages = [
{"file": "how-to/rocm-for-ai/training/benchmark-docker/previous-versions/pytorch-training-v25.4", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/training/benchmark-docker/previous-versions/pytorch-training-v25.5", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/training/benchmark-docker/previous-versions/pytorch-training-v25.6", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/inference/xdit-diffusion-inference", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/training/benchmark-docker/previous-versions/pytorch-training-v25.7", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/training/benchmark-docker/previous-versions/pytorch-training-v25.8", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/training/benchmark-docker/previous-versions/pytorch-training-v25.9", "os": ["linux"]},
@@ -204,6 +205,7 @@ article_pages = [
{"file": "how-to/rocm-for-ai/inference/benchmark-docker/previous-versions/xdit-25.13", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/inference/benchmark-docker/previous-versions/xdit-26.1", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/inference/benchmark-docker/previous-versions/xdit-26.2", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/inference/benchmark-docker/previous-versions/xdit-26.3", "os": ["linux"]},
{"file": "how-to/rocm-for-ai/inference/deploy-your-model", "os": ["linux"]},
@@ -0,0 +1,354 @@
docker:
pull_tag: rocm/pytorch-xdit:v26.3
docker_hub_url: https://hub.docker.com/layers/rocm/pytorch-xdit/v26.3/images/sha256-ac78a03d2911bf1b49c001d3be2e8bd745c1bc455cb49ae972825a7986880902
ROCm: 7.12.0
whats_new:
- "Qwen-Image support"
- "Qwen-Image-Edit support"
- "Aiter update to support Sage attention v2"
- "xDiT update to support MXFP4 GEMMs in Wan2.2, Wan2.1 and Flux.2"
components:
TheRock:
version: e40a6da
url: https://github.com/ROCm/TheRock
rocm-libraries:
version: 9e9e900
url: https://github.com/ROCm/rocm-libraries
rocm-systems:
version: ca89a1a
url: https://github.com/ROCm/rocm-systems
torch:
version: 91be249
url: https://github.com/ROCm/pytorch
torchaudio:
version: e3c6ee2
url: https://github.com/pytorch/audio
torchvision:
version: b919bd0
url: https://github.com/pytorch/vision
triton:
version: a272dfa
url: https://github.com/ROCm/triton
accelerate:
version: 46ba481
url: https://github.com/huggingface/accelerate
aiter:
version: 82d733f
url: https://github.com/ROCm/aiter
diffusers:
version: a80b192
url: https://github.com/huggingface/diffusers
xfuser:
version: 2882027
url: https://github.com/xdit-project/xDiT
yunchang:
version: 631bdfd
url: https://github.com/feifeibear/long-context-attention
supported_models:
- group: Hunyuan Video
js_tag: hunyuan
models:
- model: Hunyuan Video
model_repo: tencent/HunyuanVideo
revision: refs/pr/18
url: https://huggingface.co/tencent/HunyuanVideo
github: https://github.com/Tencent-Hunyuan/HunyuanVideo
mad_tag: pyt_xdit_hunyuanvideo
js_tag: hunyuan_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--prompt "In the large cage, two puppies were wagging their tails at each other." \'
- '--batch_size 1 \'
- '--height 720 --width 1280 \'
- '--seed 1168860793 \'
- '--num_frames 129 \'
- '--num_inference_steps 50 \'
- '--warmup_calls 1 \'
- '--num_iterations 1 \'
- '--ulysses_degree 8 \'
- '--enable_tiling --enable_slicing \'
- '--guidance_scale 6.0 \'
- '--use_torch_compile \'
- '--attention_backend aiter \'
- '--output_directory results'
- model: Hunyuan Video 1.5
model_repo: hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-720p_t2v
url: https://huggingface.co/hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-720p_t2v
github: https://github.com/Tencent-Hunyuan/HunyuanVideo-1.5
mad_tag: pyt_xdit_hunyuanvideo_1_5
js_tag: hunyuan_1_5_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--prompt "In the large cage, two puppies were wagging their tails at each other." \'
- '--task t2v \'
- '--height 720 --width 1280 \'
- '--seed 1168860793 \'
- '--num_frames 129 \'
- '--num_inference_steps 50 \'
- '--num_iterations 1 \'
- '--ulysses_degree 8 \'
- '--enable_tiling --enable_slicing \'
- '--use_torch_compile \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: Wan-AI
js_tag: wan
models:
- model: Wan2.1
model_repo: Wan-AI/Wan2.1-I2V-14B-720P-Diffusers
url: https://huggingface.co/Wan-AI/Wan2.1-I2V-14B-720P-Diffusers
github: https://github.com/Wan-Video/Wan2.1
mad_tag: pyt_xdit_wan_2_1
js_tag: wan_21_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline''s intricate details and the refreshing atmosphere of the seaside." \'
- '--height 720 \'
- '--width 1280 \'
- '--input_images /app/data/wan_input.jpg \'
- '--num_frames 81 \'
- '--ulysses_degree 8 \'
- '--seed 42 \'
- '--num_iterations 1 \'
- '--num_inference_steps 40 \'
- '--use_torch_compile \'
- '--attention_backend aiter \'
- '--output_directory results'
- model: Wan2.2
model_repo: Wan-AI/Wan2.2-I2V-A14B-Diffusers
url: https://huggingface.co/Wan-AI/Wan2.2-I2V-A14B-Diffusers
github: https://github.com/Wan-Video/Wan2.2
mad_tag: pyt_xdit_wan_2_2
js_tag: wan_22_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--prompt "Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline''s intricate details and the refreshing atmosphere of the seaside." \'
- '--height 720 \'
- '--width 1280 \'
- '--input_images /app/data/wan_input.jpg \'
- '--num_frames 81 \'
- '--ulysses_degree 8 \'
- '--seed 42 \'
- '--num_iterations 1 \'
- '--num_inference_steps 40 \'
- '--use_torch_compile \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: FLUX
js_tag: flux
models:
- model: FLUX.1
model_repo: black-forest-labs/FLUX.1-dev
url: https://huggingface.co/black-forest-labs/FLUX.1-dev
github: https://github.com/black-forest-labs/flux
mad_tag: pyt_xdit_flux
js_tag: flux_1_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "A small cat" \'
- '--height 1024 \'
- '--width 1024 \'
- '--num_inference_steps 25 \'
- '--max_sequence_length 256 \'
- '--warmup_calls 5 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--guidance_scale 0.0 \'
- '--num_iterations 50 \'
- '--attention_backend aiter \'
- '--output_directory results'
- model: FLUX.1 Kontext
model_repo: black-forest-labs/FLUX.1-Kontext-dev
url: https://huggingface.co/black-forest-labs/FLUX.1-Kontext-dev
github: https://github.com/black-forest-labs/flux
mad_tag: pyt_xdit_flux_kontext
js_tag: flux_1_kontext_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "Add a cool hat to the cat" \'
- '--height 1024 \'
- '--width 1024 \'
- '--num_inference_steps 30 \'
- '--max_sequence_length 512 \'
- '--warmup_calls 5 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--input_images /app/data/flux_cat.png \'
- '--guidance_scale 2.5 \'
- '--num_iterations 25 \'
- '--attention_backend aiter \'
- '--output_directory results'
- model: FLUX.2
model_repo: black-forest-labs/FLUX.2-dev
url: https://huggingface.co/black-forest-labs/FLUX.2-dev
github: https://github.com/black-forest-labs/flux2
mad_tag: pyt_xdit_flux_2
js_tag: flux_2_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "Add a cool hat to the cat" \'
- '--height 1024 \'
- '--width 1024 \'
- '--num_inference_steps 50 \'
- '--max_sequence_length 512 \'
- '--warmup_calls 5 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--input_images /app/data/flux_cat.png \'
- '--guidance_scale 4.0 \'
- '--num_iterations 25 \'
- '--attention_backend aiter \'
- '--output_directory results'
- model: FLUX.2 Klein
model_repo: black-forest-labs/FLUX.2-klein-9B
url: https://huggingface.co/black-forest-labs/FLUX.2-klein-9B
github: https://github.com/black-forest-labs/flux2
mad_tag: pyt_xdit_flux_2_klein
js_tag: flux_2_klein_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "A spectacular sunset over the ocean" \'
- '--height 2048 \'
- '--width 2048 \'
- '--num_inference_steps 4 \'
- '--warmup_calls 5 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--guidance_scale 1.0 \'
- '--num_iterations 25 \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: StableDiffusion
js_tag: stablediffusion
models:
- model: stable-diffusion-3.5-large
model_repo: stabilityai/stable-diffusion-3.5-large
url: https://huggingface.co/stabilityai/stable-diffusion-3.5-large
github: https://github.com/Stability-AI/sd3.5
mad_tag: pyt_xdit_sd_3_5
js_tag: stable_diffusion_3_5_large_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--prompt "A capybara holding a sign that reads Hello World" \'
- '--num_iterations 50 \'
- '--num_inference_steps 28 \'
- '--pipefusion_parallel_degree 4 \'
- '--use_cfg_parallel \'
- '--use_torch_compile \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: Z-Image
js_tag: z_image
models:
- model: Z-Image Turbo
model_repo: Tongyi-MAI/Z-Image-Turbo
url: https://huggingface.co/Tongyi-MAI/Z-Image-Turbo
github: https://github.com/Tongyi-MAI/Z-Image
mad_tag: pyt_xdit_z_image_turbo
js_tag: z_image_turbo_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "A crowded beach" \'
- '--height 1088 \'
- '--width 1920 \'
- '--num_inference_steps 9 \'
- '--ulysses_degree 2 \'
- '--use_torch_compile \'
- '--guidance_scale 0.0 \'
- '--num_iterations 50 \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: LTX
js_tag: ltx
models:
- model: LTX-2
model_repo: Lightricks/LTX-2
url: https://huggingface.co/Lightricks/LTX-2
github: https://github.com/Lightricks/LTX-2
mad_tag: pyt_xdit_ltx2
js_tag: ltx2_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "Cinematic action packed shot. The man says silently: \"We need to run.\". The camera zooms in on his mouth then immediately screams: \"NOW!\". The camera zooms back out, he turns around and bolts it." \'
- '--height 1088 \'
- '--width 1920 \'
- '--num_inference_steps 40 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--guidance_scale 4.0 \'
- '--num_iterations 1 \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: Qwen-Image
js_tag: qwen_image
models:
- model: Qwen-Image
model_repo: Qwen/Qwen-Image
url: https://huggingface.co/Qwen/Qwen-Image
github: https://github.com/QwenLM/Qwen-Image
mad_tag: pyt_xdit_qwen_image
js_tag: qwen_image_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "A cat holding a sign that says hello world" \'
- '--height 2048 \'
- '--width 2048 \'
- '--num_inference_steps 50 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--num_iterations 1 \'
- '--attention_backend aiter \'
- '--output_directory results'
- model: Qwen-Image-Edit
model_repo: Qwen/Qwen-Image-Edit
url: https://huggingface.co/Qwen/Qwen-Image-Edit
github: https://github.com/QwenLM/Qwen-Image
mad_tag: pyt_xdit_qwen_image_edit
js_tag: qwen_image_edit_tag
benchmark_command:
- mkdir results
- 'xdit \'
- '--model {model_repo} \'
- '--seed 42 \'
- '--prompt "Add a cool hat to the cat." \'
- '--negative_prompt "" \'
- '--input_images /app/data/flux_cat.png \'
- '--height 2048 \'
- '--width 2048 \'
- '--num_inference_steps 50 \'
- '--ulysses_degree 8 \'
- '--use_torch_compile \'
- '--num_iterations 1 \'
- '--attention_backend aiter \'
- '--output_directory results'
@@ -1,24 +1,24 @@
docker:
pull_tag: rocm/pytorch-xdit:v26.3
docker_hub_url: https://hub.docker.com/layers/rocm/pytorch-xdit/v26.3/images/sha256-ac78a03d2911bf1b49c001d3be2e8bd745c1bc455cb49ae972825a7986880902
pull_tag: rocm/pytorch-xdit:v26.4
docker_hub_url: https://hub.docker.com/layers/rocm/pytorch-xdit/v26.4/images/sha256-b4296a638eb8dc7ebcafc808e180b78a3c44177580c21986082ec9539496067c
ROCm: 7.12.0
whats_new:
- "Qwen-Image support"
- "Qwen-Image-Edit support"
- "Aiter update to support Sage attention v2"
- "xDiT update to support MXFP4 GEMMs in Wan2.2, Wan2.1 and Flux.2"
- "Qwen-Image-2512 support"
- "Z-Image support"
- "Parallel VAE decode support for Wan models"
- "Batch inference and data parallel support"
components:
TheRock:
version: e40a6da
version: 9b611c6
url: https://github.com/ROCm/TheRock
rocm-libraries:
version: 9e9e900
version: 7567d83
url: https://github.com/ROCm/rocm-libraries
rocm-systems:
version: ca89a1a
version: 93bc019
url: https://github.com/ROCm/rocm-systems
torch:
version: 91be249
version: ff65f5b
url: https://github.com/ROCm/pytorch
torchaudio:
version: e3c6ee2
@@ -33,13 +33,16 @@ docker:
version: 46ba481
url: https://github.com/huggingface/accelerate
aiter:
version: 82d733f
version: a169e14
url: https://github.com/ROCm/aiter
diffusers:
version: a80b192
url: https://github.com/huggingface/diffusers
distvae:
version: bf7531e
url: https://github.com/xdit-project/DistVAE
xfuser:
version: 2882027
version: 45c44e7
url: https://github.com/xdit-project/xDiT
yunchang:
version: 631bdfd
@@ -114,6 +117,7 @@ docker:
- '--input_images /app/data/wan_input.jpg \'
- '--num_frames 81 \'
- '--ulysses_degree 8 \'
- '--use_parallel_vae \'
- '--seed 42 \'
- '--num_iterations 1 \'
- '--num_inference_steps 40 \'
@@ -136,6 +140,7 @@ docker:
- '--input_images /app/data/wan_input.jpg \'
- '--num_frames 81 \'
- '--ulysses_degree 8 \'
- '--use_parallel_vae \'
- '--seed 42 \'
- '--num_iterations 1 \'
- '--num_inference_steps 40 \'
@@ -262,12 +267,12 @@ docker:
- group: Z-Image
js_tag: z_image
models:
- model: Z-Image Turbo
model_repo: Tongyi-MAI/Z-Image-Turbo
url: https://huggingface.co/Tongyi-MAI/Z-Image-Turbo
- model: Z-Image
model_repo: Tongyi-MAI/Z-Image
url: https://huggingface.co/Tongyi-MAI/Z-Image
github: https://github.com/Tongyi-MAI/Z-Image
mad_tag: pyt_xdit_z_image_turbo
js_tag: z_image_turbo_tag
mad_tag: pyt_xdit_z_image
js_tag: z_image_tag
benchmark_command:
- mkdir results
- 'xdit \'
@@ -276,11 +281,13 @@ docker:
- '--prompt "A crowded beach" \'
- '--height 1088 \'
- '--width 1920 \'
- '--num_inference_steps 9 \'
- '--num_inference_steps 50 \'
- '--ulysses_degree 2 \'
- '--ring_degree 2 \'
- '--use_cfg_parallel \'
- '--use_torch_compile \'
- '--guidance_scale 0.0 \'
- '--num_iterations 50 \'
- '--guidance_scale 4.0 \'
- '--num_iterations 25 \'
- '--attention_backend aiter \'
- '--output_directory results'
- group: LTX
@@ -311,8 +318,8 @@ docker:
js_tag: qwen_image
models:
- model: Qwen-Image
model_repo: Qwen/Qwen-Image
url: https://huggingface.co/Qwen/Qwen-Image
model_repo: Qwen/Qwen-Image-2512
url: https://huggingface.co/Qwen/Qwen-Image-2512
github: https://github.com/QwenLM/Qwen-Image
mad_tag: pyt_xdit_qwen_image
js_tag: qwen_image_tag
@@ -0,0 +1,318 @@
.. meta::
:description: Learn to validate diffusion model video generation on MI300X, MI350X and MI355X accelerators using
prebuilt and optimized docker images.
:keywords: xDiT, diffusion, video, video generation, image, image generation, validate, benchmark
************************
xDiT diffusion inference
************************
.. caution::
This documentation does not reflect the latest version of the xDiT diffusion
inference performance documentation. See
:doc:`/how-to/rocm-for-ai/inference/xdit-diffusion-inference` for the latest
version.
.. _xdit-video-diffusion-263:
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
The `rocm/pytorch-xdit <{{ docker.docker_hub_url }}>`_ Docker image offers a prebuilt, optimized environment based on `xDiT <https://github.com/xdit-project/xDiT>`_ for
benchmarking diffusion model video and image generation on gfx942 and gfx950 series (AMD Instinct™ MI300X, MI325X, MI350X, and MI355X) GPUs.
The image runs ROCm **{{docker.ROCm}}** (preview) based on `TheRock <https://github.com/ROCm/TheRock>`_
and includes the following components:
.. dropdown:: Software components - {{ docker.pull_tag.split('-')|last }}
.. list-table::
:header-rows: 1
* - Software component
- Version
{% for component_name, component_data in docker.components.items() %}
* - `{{ component_name }} <{{ component_data.url }}>`_
- {{ component_data.version }}
{% endfor %}
Follow this guide to pull the required image, spin up a container, download the model, and run a benchmark.
For preview and development releases, see `amdsiloai/pytorch-xdit <https://hub.docker.com/r/amdsiloai/pytorch-xdit>`_.
What's new
==========
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
{% for item in docker.whats_new %}
* {{ item }}
{% endfor %}
.. _xdit-video-diffusion-supported-models-263:
Supported models
================
The following models are supported for inference performance benchmarking.
Some instructions, commands, and recommendations in this documentation might
vary by model -- select one to get started.
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
.. raw:: html
<div id="vllm-benchmark-ud-params-picker" class="container-fluid">
<div class="row gx-0">
<div class="col-2 me-1 px-2 model-param-head">Model</div>
<div class="row col-10 pe-0">
{% for model_group in docker.supported_models %}
<div class="col-6 px-2 model-param" data-param-k="model-group" data-param-v="{{ model_group.js_tag }}" tabindex="0">{{ model_group.group }}</div>
{% endfor %}
</div>
</div>
<div class="row gx-0 pt-1">
<div class="col-2 me-1 px-2 model-param-head">Variant</div>
<div class="row col-10 pe-0">
{% for model_group in docker.supported_models %}
{% set models = model_group.models %}
{% for model in models %}
{% if models|length % 3 == 0 %}
<div class="col-4 px-2 model-param" data-param-k="model" data-param-v="{{ model.js_tag }}" data-param-group="{{ model_group.js_tag }}" tabindex="0">{{ model.model }}</div>
{% else %}
<div class="col-6 px-2 model-param" data-param-k="model" data-param-v="{{ model.js_tag }}" data-param-group="{{ model_group.js_tag }}" tabindex="0">{{ model.model }}</div>
{% endif %}
{% endfor %}
{% endfor %}
</div>
</div>
</div>
{% for model_group in docker.supported_models %}
{% for model in model_group.models %}
.. container:: model-doc {{ model.js_tag }}
.. note::
To learn more about your specific model see the `{{ model.model }} model card on Hugging Face <{{ model.url }}>`_
or visit the `GitHub page <{{ model.github }}>`__. Note that some models require access authorization before use via an
external license agreement through a third party.
{% endfor %}
{% endfor %}
System validation
=================
Before running AI workloads, it's important to validate that your AMD hardware is configured
correctly and performing optimally.
If you have already validated your system settings, including aspects like NUMA auto-balancing, you
can skip this step. Otherwise, complete the procedures in the :ref:`System validation and
optimization <rocm-for-ai-system-optimization>` guide to properly configure your system settings
before starting.
To test for optimal performance, consult the recommended :ref:`System health benchmarks
<rocm-for-ai-system-health-bench>`. This suite of tests will help you verify and fine-tune your
system's configuration.
Pull the Docker image
=====================
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
For this tutorial, it's recommended to use the latest ``{{ docker.pull_tag }}`` Docker image.
Pull the image using the following command:
.. code-block:: shell
docker pull {{ docker.pull_tag }}
Validate and benchmark
======================
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
Once the image has been downloaded you can follow these steps to
run benchmarks and generate outputs.
{% for model_group in docker.supported_models %}
{% for model in model_group.models %}
.. container:: model-doc {{model.js_tag}}
The following commands are written for {{ model.model }}.
See :ref:`xdit-video-diffusion-supported-models-263` to switch to another available model.
{% endfor %}
{% endfor %}
Choose your setup method
------------------------
You can either use an existing Hugging Face cache or download the model fresh inside the container.
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
{% for model_group in docker.supported_models %}
{% for model in model_group.models %}
.. container:: model-doc {{model.js_tag}}
.. tab-set::
.. tab-item:: Option 1: Use existing Hugging Face cache
If you already have models downloaded on your host system, you can mount your existing cache.
1. Set your Hugging Face cache location.
.. code-block:: shell
export HF_HOME=/your/hf_cache/location
2. Download the model (if not already cached).
.. code-block:: shell
huggingface-cli download {{ model.model_repo }} {% if model.revision %} --revision {{ model.revision }} {% endif %}
3. Launch the container with mounted cache.
.. code-block:: shell
docker run \
-it --rm \
--cap-add=SYS_PTRACE \
--security-opt seccomp=unconfined \
--user root \
--device=/dev/kfd \
--device=/dev/dri \
--group-add video \
--ipc=host \
--network host \
--privileged \
--shm-size 128G \
--name pytorch-xdit \
-e HSA_NO_SCRATCH_RECLAIM=1 \
-e OMP_NUM_THREADS=16 \
-e CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 \
-e HF_HOME=/app/huggingface_models \
-v $HF_HOME:/app/huggingface_models \
{{ docker.pull_tag }}
.. tab-item:: Option 2: Download inside container
If you prefer to keep the container self-contained or don't have an existing cache.
1. Launch the container
.. code-block:: shell
docker run \
-it --rm \
--cap-add=SYS_PTRACE \
--security-opt seccomp=unconfined \
--user root \
--device=/dev/kfd \
--device=/dev/dri \
--group-add video \
--ipc=host \
--network host \
--privileged \
--shm-size 128G \
--name pytorch-xdit \
-e HSA_NO_SCRATCH_RECLAIM=1 \
-e OMP_NUM_THREADS=16 \
-e CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 \
{{ docker.pull_tag }}
2. Inside the container, set the Hugging Face cache location and download the model.
.. code-block:: shell
export HF_HOME=/app/huggingface_models
huggingface-cli download {{ model.model_repo }} {% if model.revision %} --revision {{ model.revision }} {% endif %}
.. warning::
Models will be downloaded to the container's filesystem and will be lost when the container is removed unless you persist the data with a volume.
{% endfor %}
{% endfor %}
Run inference
=============
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/previous-versions/xdit_26.3-inference-models.yaml
{% set docker = data.docker %}
{% for model_group in docker.supported_models %}
{% for model in model_group.models %}
.. container:: model-doc {{ model.js_tag }}
.. tab-set::
.. tab-item:: MAD-integrated benchmarking
1. Clone the ROCm Model Automation and Dashboarding (`<https://github.com/ROCm/MAD>`__) repository to a local
directory and install the required packages on the host machine.
.. code-block:: shell
git clone https://github.com/ROCm/MAD
cd MAD
pip install -r requirements.txt
2. On the host machine, use this command to run the performance benchmark test on
the `{{model.model}} <{{ model.url }}>`_ model using one node.
.. code-block:: shell
export MAD_SECRETS_HFTOKEN="your personal Hugging Face token to access gated models"
madengine run \
--tags {{model.mad_tag}} \
--keep-model-dir \
--live-output
MAD launches a Docker container with the name
``container_ci-{{model.mad_tag}}``. The throughput and serving reports of the
model are collected in the following paths: ``{{ model.mad_tag }}_throughput.csv``
and ``{{ model.mad_tag }}_serving.csv``.
.. tab-item:: Standalone benchmarking
To run the benchmarks for {{ model.model }}, use the following command:
.. code-block:: shell
{{ model.benchmark_command
| map('replace', '{model_repo}', model.model_repo)
| map('trim')
| join('\n ') }}
The generated content and timing information will be stored under the results directory.
{% endfor %}
{% endfor %}
Previous versions
=================
See
:doc:`/how-to/rocm-for-ai/inference/benchmark-docker/previous-versions/xdit-history`
to find documentation for previous releases of xDiT diffusion inference
performance testing.
@@ -15,11 +15,20 @@ benchmarking, see the version-specific documentation.
- Components
- Resources
* - ``rocm/pytorch-xdit:v26.3`` (latest)
* - ``rocm/pytorch-xdit:v26.4`` (latest)
-
* TheRock e40a6da
* `ROCm 7.12.0 preview <https://rocm.docs.amd.com/en/7.12.0-preview/about/release-notes.html>`__
* TheRock 9b611c6
-
* :doc:`Documentation </how-to/rocm-for-ai/inference/xdit-diffusion-inference>`
* `Docker Hub <https://hub.docker.com/layers/rocm/pytorch-xdit/v26.4/images/sha256-b4296a638eb8dc7ebcafc808e180b78a3c44177580c21986082ec9539496067c>`__
* - ``rocm/pytorch-xdit:v26.3``
-
* `ROCm 7.12.0 preview <https://rocm.docs.amd.com/en/7.12.0-preview/about/release-notes.html>`__
* TheRock e40a6da
-
* :doc:`Documentation <xdit-26.3>`
* `Docker Hub <https://hub.docker.com/layers/rocm/pytorch-xdit/v26.3/images/sha256-ac78a03d2911bf1b49c001d3be2e8bd745c1bc455cb49ae972825a7986880902>`__
* - ``rocm/pytorch-xdit:v26.2``
@@ -35,3 +35,5 @@ training, fine-tuning, and inference. It leverages popular machine learning fram
- :doc:`xDiT diffusion inference <xdit-diffusion-inference>`
- :doc:`Deploying your model <deploy-your-model>`
- :doc:`xDiT diffusion inference <xdit-diffusion-inference>`
@@ -15,7 +15,7 @@ xDiT diffusion inference
The `rocm/pytorch-xdit <{{ docker.docker_hub_url }}>`_ Docker image offers a prebuilt, optimized environment based on `xDiT <https://github.com/xdit-project/xDiT>`_ for
benchmarking diffusion model video and image generation on gfx942 and gfx950 series (AMD Instinct™ MI300X, MI325X, MI350X, and MI355X) GPUs.
The image runs ROCm **{{docker.ROCm}}** (preview) based on `TheRock <https://github.com/ROCm/TheRock>`_
The image runs `ROCm {{docker.ROCm}} (preview) <https://rocm.docs.amd.com/en/7.12.0-preview/about/release-notes.html>`__ based on `TheRock <https://github.com/ROCm/TheRock>`_
and includes the following components:
.. dropdown:: Software components - {{ docker.pull_tag.split('-')|last }}
@@ -36,6 +36,7 @@ For preview and development releases, see `amdsiloai/pytorch-xdit <https://hub.d
What's new
==========
.. datatemplate:yaml:: /data/how-to/rocm-for-ai/inference/xdit-inference-models.yaml
{% set docker = data.docker %}
+3 -7
View File
@@ -28,7 +28,7 @@ Memory settings
===============
AMD Ryzen APUs with RDNA3.5 architecture (gfx1150, gfx1151, and gfx1152 LLVM
target) memory access is handled through GPU Virtual Memory (GPUVM), which
targets) memory access is handled through GPU Virtual Memory (GPUVM), which
provides per-process GPU virtual address spaces (VMIDs) rather than a separate,
discrete VRAM pool.
@@ -45,7 +45,7 @@ physical memory.
On systems with physically shared CPU and GPU memory, such as RDNA3.5-based
systems, this mapped system memory effectively serves as VRAM for the GPU.
GART is typically kept relatively small to limit GPU page-table size and is
mainly used for driver-internal operations.
primarily used for driver-internal operations.
* **GTT**
@@ -81,10 +81,6 @@ allocations using GTT (GTT-backed allocations), as described in the
`torvalds/linux@759e764 <https://github.com/torvalds/linux/commit/759e764f7d587283b4e0b01ff930faca64370e59>`_
GitHub commit.
Because memory is physically shared, there is no performance distinction
similar to discrete GPUs where dedicated VRAM is significantly faster than
system memory. Firmware may optionally reserve some memory exclusively for GPU
use, but this provides little benefit for most workloads while permanently
Because memory is physically shared, there's no performance distinction
like that of discrete GPUs where dedicated VRAM is significantly faster than
system memory. Firmware may optionally reserve some memory exclusively for GPU
@@ -183,7 +179,7 @@ Operating system support
The ROCm compatibility tables can be found at the following links:
- `System requirements (Linux) <https://rocm.docs.amd.com/projects/install-on-linux/en/latest/reference/system-requirements.html>`_
- `System requirements (Windows) <https://rocm.docs.amd.com/projects/install-on-windows/en/latest/reference/system-requirements.html>`_
- `System requirements (Microsoft Windows) <https://rocm.docs.amd.com/projects/install-on-windows/en/latest/reference/system-requirements.html>`_
AMD Ryzen AI Max series APUs (gfx1151) have additional kernel version
requirements, as described in the following section.
+1 -1
View File
@@ -37,7 +37,7 @@ click==8.3.1
# sphinx-external-toc
comm==0.2.3
# via ipykernel
cryptography==46.0.6
cryptography==46.0.7
# via pyjwt
debugpy==1.8.19
# via ipykernel