Compare commits
11
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b1bbc9e6d0 | ||
|
|
d692d1cc3b | ||
|
|
02a027c49f | ||
|
|
e235b6a332 | ||
|
|
8b51380466 | ||
|
|
527a3165b9 | ||
|
|
f06d824054 | ||
|
|
6ded84ee0f | ||
|
|
a4d6416c52 | ||
|
|
0cc41a22dc | ||
|
|
8444c0897a |
@@ -68,7 +68,7 @@ Copy `templates/component_parity_test.py` and fill every `TODO` marker. The
|
||||
template is distilled from:
|
||||
|
||||
- `tests/local_tests/transformers/test_ltx2.py`
|
||||
- `tests/local_tests/transformers/test_gamecraft_parity.py`
|
||||
- `tests/local_tests/gen3c/test_gen3c.py`
|
||||
- `tests/local_tests/encoders/test_ltx2_gemma_parity.py`
|
||||
- `tests/local_tests/vaes/test_oobleck_vae_parity.py`
|
||||
- `tests/local_tests/sd35/test_sd35_component_parity.py`
|
||||
|
||||
@@ -87,7 +87,7 @@ def _load_official_model(device: torch.device, dtype: torch.dtype) -> torch.nn.M
|
||||
# TODO: import official class/factory and load real weights strictly.
|
||||
# Examples in-tree:
|
||||
# - LTX2: SingleGPUModelBuilder(...).build(device=device, dtype=dtype)
|
||||
# - GameCraft: torch.load(...)["module"] -> official_model.load_state_dict(...)
|
||||
# - GEN3C: torch.load(...)["state_dict"] -> official_model.load_state_dict(...)
|
||||
# - Oobleck: create_model_from_config(config) + ckpt state_dict
|
||||
OfficialClass = _import_or_skip(OFFICIAL_MODULE, OFFICIAL_CLASS)
|
||||
model = OfficialClass() # TODO: pass official config kwargs.
|
||||
|
||||
@@ -52,8 +52,8 @@ scaling constants, dtype casts, state-dict names, and every output head.
|
||||
- Loader path: `TransformerLoader` reads `transformer/config.json`, calls
|
||||
`dit_config.update_model_arch(config)`, resolves `_class_name` through
|
||||
`ModelRegistry`, and constructs the class with `config` and `hf_config`.
|
||||
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, `longcat.py`,
|
||||
and `ltx2.py`.
|
||||
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, and
|
||||
`ltx2.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
@@ -51,7 +51,7 @@ posterior behavior, encode/decode output objects, tiling flags, and cropping.
|
||||
- Loader path: VAE loaders resolve `_class_name` through `ModelRegistry` and
|
||||
load converted component weights from the VAE subdir.
|
||||
- Reference examples: `oobleck.py`, `autoencoder_kl.py`, `wanvae.py`,
|
||||
`ltx2vae.py`, and `gamecraftvae.py`.
|
||||
`ltx2vae.py`, and `hunyuanvae.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
@@ -43,11 +43,12 @@ from `../add-model/contracts/conversion_request.md`.
|
||||
- `scripts/checkpoint_conversion/stable_audio_to_diffusers.py`: monolithic
|
||||
`model.safetensors` split into transformer/VAE/conditioner, plus copied
|
||||
passthrough subfolders. Use this shape for single-checkpoint official repos.
|
||||
- `scripts/checkpoint_conversion/convert_gamecraft_full.py`: separate official
|
||||
sources for transformer, VAE, text encoders, tokenizers, scheduler, and root
|
||||
`model_index.json`.
|
||||
- `scripts/checkpoint_conversion/longcat_to_fastvideo.py`: fused QKV/KV split,
|
||||
renamed native transformer weights, and copied existing Diffusers components.
|
||||
- `scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py`: separate
|
||||
official sources for transformer, VAE, encoders, and vocoder, assembled under
|
||||
a root `model_index.json`.
|
||||
- `scripts/checkpoint_conversion/convert_flux2_klein.py`: fused QKV split,
|
||||
renamed native transformer weights, and copied passthrough text encoder,
|
||||
tokenizer, and scheduler components.
|
||||
- `scripts/checkpoint_conversion/pt_to_safetensors.py`: simple `.pt` extraction
|
||||
helper for nested checkpoint dictionaries.
|
||||
|
||||
|
||||
@@ -247,7 +247,7 @@ setup gap, not a pass.
|
||||
- `fastvideo/configs/pipelines/stable_audio.py` and
|
||||
`fastvideo/pipelines/basic/stable_audio/presets.py` for config/preset shape.
|
||||
- `fastvideo/registry.py` for `register_configs(...)` and preset registration.
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for latent
|
||||
- `tests/local_tests/pipelines/test_lingbot_video_pipeline_parity.py` for latent
|
||||
parity structure.
|
||||
- `tests/local_tests/pipelines/test_stable_audio_pipeline_parity.py` for audio
|
||||
parity structure.
|
||||
|
||||
@@ -80,36 +80,26 @@ def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
str(model_path),
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False,
|
||||
"offload": {
|
||||
"dit": False,
|
||||
"vae": False,
|
||||
"text_encoder": False,
|
||||
},
|
||||
},
|
||||
},
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=False,
|
||||
)
|
||||
try:
|
||||
return generator.generate({
|
||||
"prompt": params["prompt"],
|
||||
"negative_prompt": params.get("negative_prompt"),
|
||||
"sampling": {
|
||||
"height": params.get("height"),
|
||||
"width": params.get("width"),
|
||||
"num_frames": params.get("num_frames"),
|
||||
"fps": params.get("fps"),
|
||||
"num_inference_steps": params["num_inference_steps"],
|
||||
"guidance_scale": params.get("guidance_scale"),
|
||||
"seed": params["seed"],
|
||||
},
|
||||
"output": {
|
||||
"output_path": f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
"save_video": False,
|
||||
},
|
||||
})
|
||||
return generator.generate_video(
|
||||
prompt=params["prompt"],
|
||||
negative_prompt=params.get("negative_prompt"),
|
||||
output_path=f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
save_video=False,
|
||||
height=params.get("height"),
|
||||
width=params.get("width"),
|
||||
num_frames=params.get("num_frames"),
|
||||
fps=params.get("fps"),
|
||||
num_inference_steps=params["num_inference_steps"],
|
||||
guidance_scale=params.get("guidance_scale"),
|
||||
seed=params["seed"],
|
||||
)
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
@@ -413,7 +413,7 @@ matching `*secret*`.
|
||||
- `fastvideo/pipelines/basic/wan/` for standard T2V/I2V/DMD/Causal variants.
|
||||
- `fastvideo/pipelines/basic/ltx2/` for non-standard stages and audio/video
|
||||
patterns.
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for pipeline
|
||||
- `tests/local_tests/pipelines/test_lingbot_video_pipeline_parity.py` for pipeline
|
||||
parity shape.
|
||||
- `tests/local_tests/transformers/test_ltx2.py`,
|
||||
`tests/local_tests/vaes/test_ltx2_vae.py`, and
|
||||
|
||||
@@ -84,9 +84,7 @@ fastvideo/configs/models/dits/__init__.py
|
||||
fastvideo/configs/models/encoders/__init__.py
|
||||
fastvideo/configs/models/vaes/__init__.py
|
||||
fastvideo/envs.py
|
||||
fastvideo/api/schema.py
|
||||
fastvideo/api/resolution.py
|
||||
fastvideo/api/inference_resolution.py
|
||||
fastvideo/fastvideo_args.py
|
||||
fastvideo/distributed/**
|
||||
fastvideo/layers/**
|
||||
fastvideo/attention/**
|
||||
|
||||
@@ -20,8 +20,7 @@ on a summary here.
|
||||
- Read `docs/contributing/env_vars.md` in full.
|
||||
- Decide whether the setting belongs in an environment variable or an argument
|
||||
(rule 5 in the policy doc). Settings that users change per deployment are
|
||||
arguments; add them as typed config fields in `fastvideo/api/schema.py`
|
||||
instead.
|
||||
arguments; add them through `fastvideo/fastvideo_args.py` instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
|
||||
@@ -105,8 +105,8 @@ Detect artefact type by inspecting the file's imports / helper call:
|
||||
- **pixel** (`.mp4`) — file imports
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
from `fastvideo.tests.ssim.inference_similarity_utils`, OR uses the
|
||||
legacy custom-inline helper pattern (see `test_gamecraft`,
|
||||
`test_longcat`, etc.). Default to pixel when both heuristics fail.
|
||||
legacy custom-inline helper pattern (see `test_gen3c`). Default to pixel
|
||||
when both heuristics fail.
|
||||
|
||||
Record `ARTEFACT_TYPE ∈ {pixel, latent}` for use in step 4. Steps 2, 3, 5,
|
||||
and 6 are artefact-type-agnostic — `_iter_reference_files`,
|
||||
|
||||
+93
-90
@@ -23,7 +23,100 @@ notify:
|
||||
# dispatcher, and every test payload executes inside the Slinky Slurm tray.
|
||||
# fastvideo/tests/modal remains available only for an explicit manual rollback;
|
||||
# no active pipeline or slash-command route invokes it.
|
||||
# Buildkite hands jobs to free agents in the order they appear here. Golden-gate comes first
|
||||
# because every later merge lane waits for it; the fastcheck lanes follow from longest to
|
||||
# shortest measured runtime, so the longest lane never starts last and stretches the build.
|
||||
steps:
|
||||
- label: ":test_tube: Golden-Gate Tests"
|
||||
key: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,golden-gate,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "golden_gate" || build.env("TEST_TYPE") == "golden_gate_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "golden_gate_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Unit Tests"
|
||||
key: "unit"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,unit,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "unit_test" || build.env("TEST_TYPE") == "unit_test_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-unit"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "unit_test_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Kernel Tests"
|
||||
key: "kernel-tests"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,kernel-tests,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "kernel_tests" || build.env("TEST_TYPE") == "kernel_tests_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "kernel_tests_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
key: "dreamverse"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,dreamverse,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "dreamverse_app" || build.env("TEST_TYPE") == "dreamverse_app_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "dreamverse_app_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Encoder Tests"
|
||||
key: "encoder"
|
||||
if: |
|
||||
@@ -93,96 +186,6 @@ steps:
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Kernel Tests"
|
||||
key: "kernel-tests"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,kernel-tests,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "kernel_tests" || build.env("TEST_TYPE") == "kernel_tests_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "kernel_tests_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Unit Tests"
|
||||
key: "unit"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,unit,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "unit_test" || build.env("TEST_TYPE") == "unit_test_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-unit"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "unit_test_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
key: "dreamverse"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,dreamverse,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "dreamverse_app" || build.env("TEST_TYPE") == "dreamverse_app_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "dreamverse_app_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Golden-Gate Tests"
|
||||
key: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,golden-gate,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "golden_gate" || build.env("TEST_TYPE") == "golden_gate_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "golden_gate_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
|
||||
@@ -15,8 +15,12 @@ exec pytest \
|
||||
./fastvideo/tests/ops/ \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/inference/test_basic_fasth3_omniref_pdd.py \
|
||||
./fastvideo/tests/attention/test_sdpa_metadata_mask_contract.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_tile_grad_safety.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_metadata.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_ref2va_regions.py \
|
||||
./fastvideo/tests/layers/test_pdd_linear.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
|
||||
@@ -12,6 +12,7 @@ from __future__ import annotations
|
||||
import argparse
|
||||
import fnmatch
|
||||
import re
|
||||
from collections.abc import Iterable
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import TextIO
|
||||
@@ -130,11 +131,6 @@ class FamilyCoverage:
|
||||
|
||||
|
||||
FAMILY_COVERAGE = (
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])dreamx(_world)?([/_.-]|$)"),
|
||||
("test_dreamx.py", ),
|
||||
("test_dreamx_world_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])flux[_-]?2([/_.-]|$)"),
|
||||
("test_flux2_klein.py", ),
|
||||
@@ -145,21 +141,11 @@ FAMILY_COVERAGE = (
|
||||
("test_flux.py", ),
|
||||
("test_flux_t2i_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])(hunyuan)?gamecraft([/_.-]|$)"),
|
||||
("test_gamecraft.py", ),
|
||||
("test_gamecraft_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])gen3c([/_.-]|$)"),
|
||||
("test_gen3c.py", ),
|
||||
("test_gen3c_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])glm[_-]?image([/_.-]|$)"),
|
||||
("test_glm_image.py", ),
|
||||
("test_glm_image_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])kandinsky[_-]?5([/_.-]|$)"),
|
||||
("test_kandinsky5.py", ),
|
||||
@@ -170,11 +156,6 @@ FAMILY_COVERAGE = (
|
||||
("test_lingbot.py", ),
|
||||
("test_lingbot_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])longcat([/_.-]|$)"),
|
||||
("test_longcat.py", ),
|
||||
("test_longcat_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])ltx[_-]?2([/_.-]|$)"),
|
||||
("test_ltx2.py", ),
|
||||
@@ -205,11 +186,6 @@ FAMILY_COVERAGE = (
|
||||
("test_stable_audio.py", ),
|
||||
("test_stable_audio_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])turbo(diffusion)?([/_.-]|$)"),
|
||||
(),
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
@@ -219,11 +195,6 @@ FAMILY_COVERAGE = (
|
||||
"test_wan_t2v_similarity.py",
|
||||
),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])z[_-]?image([/_.-]|$)"),
|
||||
("test_zimage.py", ),
|
||||
("test_zimage_similarity.py", ),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@@ -319,16 +290,22 @@ def _select_output_coverage(plan: MergePlan, path: str) -> None:
|
||||
plan.add_ssim(SSIM_SMOKE_TESTS, reason=f"shared output SSIM smoke coverage: {path}")
|
||||
|
||||
|
||||
def classify_paths(paths: list[str]) -> MergePlan:
|
||||
def _normalize_path(raw_path: str) -> str:
|
||||
path = raw_path.strip()
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
return path
|
||||
|
||||
|
||||
def classify_paths(paths: list[str], removed_paths: Iterable[str] = ()) -> MergePlan:
|
||||
"""Plan merge lanes for ``paths``.
|
||||
|
||||
``removed_paths`` lists changed paths that no longer exist at the PR head
|
||||
(deleted files and rename sources); removed golden/SSIM tests are not run.
|
||||
"""
|
||||
plan = MergePlan()
|
||||
normalized_paths: list[str] = []
|
||||
for raw_path in paths:
|
||||
path = raw_path.strip()
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
if path:
|
||||
normalized_paths.append(path)
|
||||
normalized_paths = sorted(set(normalized_paths))
|
||||
normalized_paths = sorted({path for path in map(_normalize_path, paths) if path})
|
||||
removed = {path for path in map(_normalize_path, removed_paths) if path}
|
||||
if not normalized_paths:
|
||||
plan.require_all("changed-file list was empty; failing closed")
|
||||
return plan
|
||||
@@ -368,7 +345,10 @@ def classify_paths(paths: list[str]) -> MergePlan:
|
||||
if path.startswith("fastvideo/tests/golden_gate/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
plan.add_golden((name, ), reason=f"changed golden test: {path}")
|
||||
if path in removed:
|
||||
plan.reasons.append(f"removed golden test has nothing to run: {path}")
|
||||
else:
|
||||
plan.add_golden((name, ), reason=f"changed golden test: {path}")
|
||||
elif name in {"AGENTS.md", "README.md"}:
|
||||
plan.reasons.append(f"golden documentation only: {path}")
|
||||
else:
|
||||
@@ -379,7 +359,10 @@ def classify_paths(paths: list[str]) -> MergePlan:
|
||||
if path.startswith("fastvideo/tests/ssim/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
plan.add_ssim((name, ), reason=f"changed SSIM test: {path}")
|
||||
if path in removed:
|
||||
plan.reasons.append(f"removed SSIM test has nothing to run: {path}")
|
||||
else:
|
||||
plan.add_ssim((name, ), reason=f"changed SSIM test: {path}")
|
||||
elif path.endswith((".py", ".json", ".pt", ".png", ".mp4")):
|
||||
plan.ssim_all = True
|
||||
plan.add_lanes("ssim", reason=f"shared SSIM harness/reference: {path}")
|
||||
@@ -489,6 +472,7 @@ def classify_paths(paths: list[str]) -> MergePlan:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path in {
|
||||
"fastvideo/fastvideo_args.py",
|
||||
"fastvideo/forward_context.py",
|
||||
"fastvideo/image_processor.py",
|
||||
"fastvideo/registry.py",
|
||||
@@ -554,6 +538,7 @@ def _write_summary(output: TextIO, plan: MergePlan) -> None:
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--paths-file", type=Path, required=True)
|
||||
parser.add_argument("--removed-paths-file", type=Path)
|
||||
parser.add_argument("--github-output", type=Path)
|
||||
parser.add_argument("--summary-file", type=Path)
|
||||
return parser.parse_args()
|
||||
@@ -562,7 +547,9 @@ def parse_args() -> argparse.Namespace:
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
paths = args.paths_file.read_text(encoding="utf-8").splitlines()
|
||||
plan = classify_paths(paths)
|
||||
removed_paths = (args.removed_paths_file.read_text(encoding="utf-8").splitlines()
|
||||
if args.removed_paths_file else [])
|
||||
plan = classify_paths(paths, removed_paths)
|
||||
print(f"MERGE_TEST_PLAN={plan.encoded_lanes()}")
|
||||
print(f"MERGE_GOLDEN_TESTS={plan.encoded_golden_tests()}")
|
||||
print(f"MERGE_SSIM_TESTS={plan.encoded_ssim_tests()}")
|
||||
|
||||
@@ -33,15 +33,15 @@ jobs:
|
||||
env:
|
||||
FASTVIDEO_ATTENTION_BACKEND: TORCH_SDPA
|
||||
TOKENIZERS_PARALLELISM: "false"
|
||||
MASTER_ADDR: localhost
|
||||
MASTER_ADDR: "127.0.0.1"
|
||||
MASTER_PORT: "29513"
|
||||
GLOO_SOCKET_IFNAME: lo0
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: pip
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
@@ -49,9 +49,9 @@ jobs:
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.11.0 torchvision torchaudio
|
||||
torch==2.12.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
pytest pytest-timeout numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru mlx \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
@@ -65,8 +65,8 @@ jobs:
|
||||
print("machine:", platform.machine())
|
||||
print("processor:", platform.processor())
|
||||
print("mlx default device:", mx.default_device())
|
||||
memory_size = mx.metal.device_info().get("memory_size") if mx.metal.is_available() else "metal unavailable"
|
||||
print("mlx memory_size:", memory_size)
|
||||
device_info = mx.metal.device_info() if mx.metal.is_available() else "metal unavailable"
|
||||
print("mlx device_info:", device_info)
|
||||
print("torch:", torch.__version__)
|
||||
print("torch mps available:", torch.backends.mps.is_available())
|
||||
PY
|
||||
@@ -74,6 +74,7 @@ jobs:
|
||||
- name: Run MLX smoke tests
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/mlx_runtime/tests/ \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
@@ -101,7 +102,7 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s -o faulthandler_timeout=120
|
||||
-v -s --timeout=120 -o faulthandler_timeout=120
|
||||
|
||||
# Same tests on MLX's CPU backend. Hosted macOS runners are scarce and
|
||||
# slower to schedule; this Linux job gives fast PR signal on the identical
|
||||
@@ -122,7 +123,6 @@ jobs:
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: pip
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
@@ -130,15 +130,16 @@ jobs:
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.11.0 torchvision torchaudio
|
||||
torch==2.12.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
pytest pytest-timeout numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru "mlx[cpu]" \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
- name: Run MLX smoke tests (CPU backend)
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/mlx_runtime/tests/ \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
@@ -166,4 +167,4 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s -o faulthandler_timeout=120
|
||||
-v -s --timeout=120 -o faulthandler_timeout=120
|
||||
|
||||
@@ -76,6 +76,8 @@ jobs:
|
||||
set -euo pipefail
|
||||
changed_json="$RUNNER_TEMP/merge-changed-files.json"
|
||||
changed_paths="$RUNNER_TEMP/merge-changed-paths.txt"
|
||||
removed_paths="$RUNNER_TEMP/merge-removed-paths.txt"
|
||||
: > "$removed_paths"
|
||||
if gh api --paginate --slurp \
|
||||
"repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}/files?per_page=100" \
|
||||
> "$changed_json"; then
|
||||
@@ -83,6 +85,10 @@ jobs:
|
||||
if [ "$observed" = "$EXPECTED_CHANGED_FILES" ]; then
|
||||
jq -r '.[][] | .filename, (.previous_filename // empty)' "$changed_json" \
|
||||
| sort -u > "$changed_paths"
|
||||
# Paths absent from the PR head, so the planner never selects a deleted test.
|
||||
jq -r '.[][] | if .status == "removed" then .filename
|
||||
elif .status == "renamed" then (.previous_filename // empty) else empty end' \
|
||||
"$changed_json" | sort -u > "$removed_paths"
|
||||
else
|
||||
echo "::warning::Changed-file API returned $observed of $EXPECTED_CHANGED_FILES paths; selecting all merge lanes."
|
||||
echo '__FASTVIDEO_CI_PLAN_ALL__' > "$changed_paths"
|
||||
@@ -98,6 +104,7 @@ jobs:
|
||||
run: |
|
||||
python3 .github/scripts/plan_merge_ci.py \
|
||||
--paths-file "$RUNNER_TEMP/merge-changed-paths.txt" \
|
||||
--removed-paths-file "$RUNNER_TEMP/merge-removed-paths.txt" \
|
||||
--github-output "$GITHUB_OUTPUT" \
|
||||
--summary-file "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
|
||||
+3
-1
@@ -36,7 +36,6 @@ env
|
||||
*.log
|
||||
weights/
|
||||
logs/
|
||||
/Z-Image/
|
||||
official_weights/
|
||||
converted_weights/
|
||||
|
||||
@@ -134,6 +133,7 @@ openspec/
|
||||
fastvideo/tests/ssim/reference_videos/**
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.mp4
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.png
|
||||
fastvideo/tests/ssim/.reference_videos_download.lock
|
||||
|
||||
# Local H3 MLX kernel / exactness benches (JSON, logs, frames, videos)
|
||||
.kernel_bench/
|
||||
@@ -142,3 +142,5 @@ fastvideo/tests/ssim/reference_videos/**
|
||||
*.nvimlog
|
||||
.nvimlog
|
||||
.python-version
|
||||
scripts/benchmarks/minimax_h3_pro6000/headline_results/
|
||||
fastvideo/tests/ssim/.reference_videos_download.lock
|
||||
|
||||
@@ -9,6 +9,8 @@
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
## NEWS
|
||||
- `2026/10/06`: FastH3 V2 now runs on a single consumer machine: NVIDIA RTX 5090, RTX 4090 and RTX PRO 6000 GPUs, DGX Spark and Apple Silicon. We also release [FastH3 Trim](https://huggingface.co/FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4), an experimental pruned model that is 4.2× smaller than base H3 and runs in as little as 8 GB of GPU memory. Get the [models](https://huggingface.co/collections/FastVideo/fastvideo-fasth3) and read the [Blog](https://haoailab.com/blogs/fasth3-rtx/).
|
||||
- `2026/10/06`: FastVideo now supports [Kandinsky 6](https://x.com/kandinskylab_ai/status/2107374635218055345) from Kandinsky Lab: text- and image-to-video with synchronized audio (base and 10-step distilled pi-Flow checkpoints) plus video super-resolution up to 4x. See the [Kandinsky 6 recipes](https://haoailab.com/FastVideo/cookbook/kandinsky6/).
|
||||
- `2026/09/15`: Release [FastH3 8-Step V2](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2), an eight-forward data-free DMD2 checkpoint distilled from MiniMax-H3 with 80% Video Sparse Attention. Run it with `examples/inference/basic/basic_fasth3_8step.py` or the [FastH3 8-Step V2 recipe](https://haoailab.com/FastVideo/cookbook/minimax-h3/).
|
||||
- `2026/09/01`: FastH3 now runs locally on Apple Silicon through MLX and on NVIDIA DGX Spark through CUDA 13, including two-Spark inference. Follow the [FastH3 recipes](https://haoailab.com/FastVideo/cookbook/minimax-h3/) and read the [Blog](https://haoailab.com/blogs/fasth3-local/).
|
||||
- `2026/08/27`: [FastH3 Preview v1](https://haoailab.com/blogs/fasth3-preview/) is an open-weight 4-step sparse-distilled MiniMax-H3 model for synchronized video-and-audio generation, developed in collaboration with [Nuva Lab](https://nuvalab.ai/) and the [NVIDIA FastGen team](https://github.com/NVlabs/FastGen). Download the recommended [VSA / Data-Free weights](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree), or see the [full FastH3 collection](https://huggingface.co/collections/FastVideo/fastvideo-fasth3).
|
||||
@@ -135,20 +137,18 @@ def main():
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
{"engine": {"num_gpus": 1}}, # Adjust based on your hardware
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
# Define a prompt for your video
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
|
||||
|
||||
# Generate the video
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
||||
@@ -42,7 +42,7 @@ Near-term OSS note:
|
||||
- `apps/dreamverse/dreamverse/main.py`: websocket endpoint, request handling,
|
||||
session state machine, rewrite orchestration, REST routes, and stream relay
|
||||
- `apps/dreamverse/dreamverse/gpu_pool.py`: GPU worker processes, warmup, model
|
||||
loading, and `generate()` calls through FastVideo
|
||||
loading, and `generate_video()` calls through FastVideo
|
||||
- `apps/dreamverse/dreamverse/prompt_enhancer.py`: prompt enhancement, rollout
|
||||
rewrite execution, provider selection, and timeout/fallback behavior
|
||||
- `apps/dreamverse/dreamverse/rewrite_prompt_payload.py`: canonical rewrite request payload
|
||||
|
||||
@@ -52,8 +52,7 @@ import torch # noqa: E402
|
||||
|
||||
from fastvideo import VideoGenerator # noqa: E402
|
||||
from fastvideo.api import ( # noqa: E402
|
||||
ComponentConfig, CompileConfig, EngineConfig, GenerationResult, GeneratorConfig, OffloadConfig, PipelineSelection,
|
||||
QuantizationConfig,
|
||||
ComponentConfig, CompileConfig, EngineConfig, GeneratorConfig, OffloadConfig, PipelineSelection, QuantizationConfig,
|
||||
)
|
||||
|
||||
DEFAULT_PROMPT = ("A cinematic drone shot over coastal cliffs at sunrise, golden "
|
||||
@@ -129,9 +128,9 @@ def _build_generator_config(model_path: str, enable_compile: bool, num_gpus: int
|
||||
)
|
||||
|
||||
|
||||
def _extract_stage_times(result: GenerationResult) -> OrderedDict[str, float]:
|
||||
def _extract_stage_times(result: dict) -> OrderedDict[str, float]:
|
||||
out: OrderedDict[str, float] = OrderedDict()
|
||||
info = result.logging_info if isinstance(result, GenerationResult) else None
|
||||
info = result.get("logging_info") if isinstance(result, dict) else None
|
||||
if info is None:
|
||||
return out
|
||||
stages = getattr(info, "stages", None)
|
||||
@@ -163,25 +162,19 @@ def _do_one_run(generator: VideoGenerator, prompt: str, *, height: int, width: i
|
||||
_reset_peak_gpu()
|
||||
t0 = time.perf_counter()
|
||||
try:
|
||||
result = generator.generate({
|
||||
"prompt": prompt,
|
||||
"negative_prompt": "",
|
||||
"sampling": {
|
||||
"height": height,
|
||||
"width": width,
|
||||
"num_frames": num_frames,
|
||||
"fps": 24,
|
||||
"num_inference_steps": num_inference_steps,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": seed,
|
||||
},
|
||||
"output": {
|
||||
"save_video": False
|
||||
},
|
||||
"extensions": {
|
||||
"ltx2_image_crf": 0.0
|
||||
},
|
||||
})
|
||||
result = generator.generate_video(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
save_video=False,
|
||||
height=height,
|
||||
width=width,
|
||||
num_frames=num_frames,
|
||||
fps=24,
|
||||
num_inference_steps=num_inference_steps,
|
||||
guidance_scale=1.0,
|
||||
seed=seed,
|
||||
ltx2_image_crf=0.0,
|
||||
)
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.synchronize()
|
||||
except Exception as exc:
|
||||
|
||||
@@ -14,7 +14,6 @@ import gc
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from typing import TYPE_CHECKING
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
@@ -35,9 +34,6 @@ from dreamverse.config import (
|
||||
)
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from fastvideo.api import GenerationResult
|
||||
|
||||
# Multi-frame decoded continuation defaults from
|
||||
# examples/inference/basic/basic_ltx2_distilled_video_continuation.py.
|
||||
# Overridable via environment variables.
|
||||
@@ -99,8 +95,8 @@ class ContinuationState:
|
||||
self.video_images = None
|
||||
self.audio_latents = None
|
||||
|
||||
def apply_video(self, request: dict, segment_idx: int) -> None:
|
||||
"""Seed the next-segment request with the cached tail frames."""
|
||||
def apply_video(self, request_kwargs: dict, segment_idx: int) -> None:
|
||||
"""Seed next-segment kwargs with the cached tail frames."""
|
||||
if segment_idx <= 1 or not self.video_images:
|
||||
return
|
||||
from PIL import Image
|
||||
@@ -114,21 +110,21 @@ class ContinuationState:
|
||||
arr = np.clip(arr, 0, 255).astype(np.uint8)
|
||||
noisy.append(Image.fromarray(arr))
|
||||
cond_images = noisy
|
||||
request["extensions"]["ltx2_video_conditions"] = [(
|
||||
request_kwargs["ltx2_video_conditions"] = [(
|
||||
cond_images,
|
||||
LTX2_VIDEO_CONDITIONING_FRAME_IDX,
|
||||
LTX2_VIDEO_CONDITIONING_STRENGTH,
|
||||
)]
|
||||
request["extensions"]["ltx2_images"] = None
|
||||
request["inputs"]["image_path"] = None
|
||||
request_kwargs["ltx2_images"] = None
|
||||
request_kwargs["image_path"] = None
|
||||
|
||||
def apply_audio(
|
||||
self,
|
||||
request: dict,
|
||||
request_kwargs: dict,
|
||||
segment_idx: int,
|
||||
audio_lps: float,
|
||||
) -> None:
|
||||
"""Seed the next-segment request with clean audio latents + denoise mask.
|
||||
"""Seed next-segment kwargs with clean audio latents + denoise mask.
|
||||
|
||||
When audio conditioning is longer than video, extend audio
|
||||
generation and shift video RoPE forward so the audio prefix
|
||||
@@ -145,9 +141,9 @@ class ContinuationState:
|
||||
audio_extra = max(0, AUDIO_CONDITIONING_NUM_FRAMES - LTX2_VIDEO_CONDITIONING_NUM_FRAMES)
|
||||
if audio_extra > 0:
|
||||
audio_num_frames = NUM_FRAMES + audio_extra
|
||||
request["extensions"]["audio_num_frames"] = (audio_num_frames)
|
||||
request_kwargs["audio_num_frames"] = (audio_num_frames)
|
||||
prefix_sec = float(audio_extra) / 24.0
|
||||
request["extensions"]["video_position_offset_sec"] = prefix_sec
|
||||
request_kwargs["video_position_offset_sec"] = prefix_sec
|
||||
|
||||
new_duration = float(NUM_FRAMES + audio_extra) / 24.0
|
||||
total_T = max(
|
||||
@@ -165,8 +161,8 @@ class ContinuationState:
|
||||
mask = torch.ones((B, 1, total_T, 1), dtype=torch.float32)
|
||||
mask[:, :, :audio_cond_T, :] = (1.0 - AUDIO_CONDITIONING_STRENGTH)
|
||||
|
||||
request["extensions"]["ltx2_audio_clean_latent"] = clean
|
||||
request["extensions"]["ltx2_audio_denoise_mask"] = mask
|
||||
request_kwargs["ltx2_audio_clean_latent"] = clean
|
||||
request_kwargs["ltx2_audio_denoise_mask"] = mask
|
||||
|
||||
def save_video(self, frames: list) -> None:
|
||||
"""Snapshot trailing N frames as PIL images for next-segment conditioning."""
|
||||
@@ -310,7 +306,7 @@ class LTX2GenerationBackend:
|
||||
),
|
||||
)
|
||||
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
self.generator = VideoGenerator.from_pretrained(config=generator_config)
|
||||
print(f"[GPU {self.gpu_id}] After model load: {self._gpu_mem()}")
|
||||
|
||||
lora_stack = DREAMVERSE_LORA_STACK or ([(DREAMVERSE_LORA_PATH,
|
||||
@@ -407,7 +403,7 @@ class LTX2GenerationBackend:
|
||||
return
|
||||
|
||||
loader = ComponentLoader.for_module_type("audio_encoder", "diffusers")
|
||||
enc = loader.load(audio_vae_path, self.generator.resolved_config)
|
||||
enc = loader.load(audio_vae_path, self.generator.fastvideo_args)
|
||||
target = getattr(enc, "model", enc)
|
||||
|
||||
proc = AudioProcessor(
|
||||
@@ -464,59 +460,51 @@ class LTX2GenerationBackend:
|
||||
|
||||
prompt = self._inject_style_trigger(prompt)
|
||||
|
||||
request = {
|
||||
"prompt": prompt,
|
||||
"negative_prompt": "",
|
||||
"inputs": {
|
||||
"image_path": image_path if segment_idx == 1 else None
|
||||
},
|
||||
"sampling": {
|
||||
"height": FRAME_HEIGHT,
|
||||
"width": FRAME_WIDTH,
|
||||
"num_frames": NUM_FRAMES,
|
||||
"fps": 24,
|
||||
"num_inference_steps": NUM_INFERENCE_STEPS,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": 10,
|
||||
},
|
||||
"output": {
|
||||
"save_video": False
|
||||
},
|
||||
"extensions": {
|
||||
"ltx2_image_crf": 0.0,
|
||||
"return_continuation_state": False,
|
||||
},
|
||||
}
|
||||
request_kwargs = dict(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
save_video=False,
|
||||
height=FRAME_HEIGHT,
|
||||
width=FRAME_WIDTH,
|
||||
num_frames=NUM_FRAMES,
|
||||
fps=24,
|
||||
num_inference_steps=NUM_INFERENCE_STEPS,
|
||||
guidance_scale=1.0,
|
||||
seed=10,
|
||||
ltx2_image_crf=0.0,
|
||||
image_path=image_path if segment_idx == 1 else None,
|
||||
return_continuation_state=False,
|
||||
)
|
||||
|
||||
if reset_conditioning:
|
||||
self.continuation.clear()
|
||||
|
||||
audio_lps = (DEFAULT_LTX2_AUDIO_SAMPLE_RATE / DEFAULT_LTX2_AUDIO_HOP_LENGTH / DEFAULT_LTX2_AUDIO_DOWNSAMPLE)
|
||||
|
||||
# Phase 1: seed the request with prior-segment conditioning.
|
||||
self.continuation.apply_video(request, segment_idx)
|
||||
self.continuation.apply_audio(request, segment_idx, audio_lps)
|
||||
# Phase 1: seed kwargs with prior-segment conditioning.
|
||||
self.continuation.apply_video(request_kwargs, segment_idx)
|
||||
self.continuation.apply_audio(request_kwargs, segment_idx, audio_lps)
|
||||
|
||||
# Phase 2: generate.
|
||||
t0 = time.perf_counter()
|
||||
result = self.generator.generate(request)
|
||||
result = self.generator.generate_video(**request_kwargs)
|
||||
torch.cuda.synchronize()
|
||||
timings["generation_ms"] = (time.perf_counter() - t0) * 1000
|
||||
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("Expected a single GenerationResult from generate.")
|
||||
frames = result.frames
|
||||
if not isinstance(result, dict):
|
||||
raise RuntimeError("Expected dictionary output from generate_video.")
|
||||
frames = result.get("frames")
|
||||
if not isinstance(frames, list) or len(frames) == 0:
|
||||
raise RuntimeError("Generation did not return frames.")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
audio = result.get("audio")
|
||||
audio_sample_rate = result.get("audio_sample_rate")
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
# LTX2 audio decoding stage uses 24kHz output by default.
|
||||
audio_sample_rate = 24000
|
||||
print(f"[GPU {self.gpu_id}] audio_sample_rate missing from result; "
|
||||
f"defaulting to {audio_sample_rate}Hz")
|
||||
|
||||
timings["generation_time_ms"] = (result.generation_time or 0.0) * 1000
|
||||
timings["generation_time_ms"] = result.get("generation_time", 0.0) * 1000
|
||||
|
||||
# Phase 3: snapshot continuation state for the next segment.
|
||||
t_save_start = time.perf_counter()
|
||||
@@ -554,7 +542,7 @@ class LTX2GenerationBackend:
|
||||
self,
|
||||
audio: object,
|
||||
audio_sample_rate: int | None,
|
||||
result: "GenerationResult",
|
||||
result: dict,
|
||||
segment_idx: int,
|
||||
) -> torch.Tensor | None:
|
||||
"""Pick which tensor to cache for next-segment audio conditioning."""
|
||||
@@ -569,7 +557,7 @@ class LTX2GenerationBackend:
|
||||
f"for segment {segment_idx + 1}")
|
||||
return re_encoded
|
||||
return None
|
||||
audio_latents = result.extra.get("ltx2_audio_latents")
|
||||
audio_latents = result.get("ltx2_audio_latents")
|
||||
if audio_latents is not None:
|
||||
print(f"[GPU {self.gpu_id}] Cached audio latents "
|
||||
f"shape={tuple(audio_latents.shape)} "
|
||||
|
||||
@@ -79,24 +79,32 @@ class MiniMaxH3GenerationBackend:
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
AttentionConfig,
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GeneratorConfig,
|
||||
MiniMaxH3Options,
|
||||
OffloadConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
|
||||
adapter_path = hf_hub_download(repo_id=adapter_repo, filename=adapter_filename)
|
||||
use_vsa = attention_backend == "VIDEO_SPARSE_ATTN_H3"
|
||||
experimental = {
|
||||
"attention_backend": attention_backend,
|
||||
"inference_torch_compile": attention_backend == "FLASH_ATTN",
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
}
|
||||
if attention_backend == "VIDEO_SPARSE_ATTN_H3":
|
||||
experimental.update({
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
})
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=model_path,
|
||||
pipeline=PipelineSelection(
|
||||
components=ComponentConfig(lora_path=adapter_path, lora_strength=1.0),
|
||||
model=MiniMaxH3Options(vae_parallel_decode=True, vae_parallel_decode_strategy="gather"),
|
||||
experimental=experimental,
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=DREAMVERSE_SP_SIZE,
|
||||
@@ -109,12 +117,7 @@ class MiniMaxH3GenerationBackend:
|
||||
vae=True,
|
||||
pin_cpu_memory=True,
|
||||
),
|
||||
compile=CompileConfig(enabled=False, vae_enabled=True, regional=attention_backend == "FLASH_ATTN"),
|
||||
attention=AttentionConfig(
|
||||
backend=attention_backend,
|
||||
vsa_sparsity=0.9 if use_vsa else None,
|
||||
vsa_tile_size=64 if use_vsa else None,
|
||||
),
|
||||
compile=CompileConfig(enabled=False, vae_enabled=True),
|
||||
use_fsdp_inference=False,
|
||||
),
|
||||
)
|
||||
|
||||
@@ -13,6 +13,7 @@ FORBIDDEN_PREFIXES = (
|
||||
"fastvideo.models",
|
||||
"fastvideo.layers",
|
||||
"fastvideo.worker",
|
||||
"fastvideo.fastvideo_args",
|
||||
)
|
||||
ALLOWED_INTERNAL_IMPORTS = {
|
||||
(
|
||||
|
||||
@@ -88,13 +88,14 @@ def test_initialize_builds_vsa_datafree_fasth3_generator(monkeypatch):
|
||||
assert config.model_path == "MiniMaxAI/MiniMax-H3"
|
||||
assert config.pipeline.components.lora_path.endswith("vsa-datafree/adapter_model.safetensors")
|
||||
assert config.pipeline.components.lora_strength == 1.0
|
||||
assert config.pipeline.experimental == {}
|
||||
assert config.engine.attention.backend == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert config.engine.attention.vsa_sparsity == 0.9
|
||||
assert config.engine.attention.vsa_tile_size == 64
|
||||
assert config.engine.compile.regional is False
|
||||
assert config.pipeline.model.vae_parallel_decode is True
|
||||
assert config.pipeline.model.vae_parallel_decode_strategy == "gather"
|
||||
assert config.pipeline.experimental == {
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"inference_torch_compile": False,
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
}
|
||||
assert config.engine.num_gpus == 4
|
||||
assert config.engine.parallelism.tp_size == 1
|
||||
assert config.engine.parallelism.sp_size == 4
|
||||
|
||||
@@ -207,7 +207,7 @@
|
||||
</Array>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="e_dsg" value="generator.generate()" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#9673a6;endArrow=classic;fontSize=10;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="do_step" target="generator" edge="1">
|
||||
<mxCell id="e_dsg" value="generator.generate_video()" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#9673a6;endArrow=classic;fontSize=10;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="do_step" target="generator" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_dscache" value="read / write" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#d6b656;endArrow=classic;startArrow=classic;fontSize=10;exitX=0;exitY=0.8;exitDx=0;exitDy=0;entryX=1;entryY=0.2;entryDx=0;entryDy=0;" parent="1" source="do_step" target="caches" edge="1">
|
||||
@@ -534,7 +534,7 @@
|
||||
<mxPoint x="1040" y="1610" as="targetPoint"/>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (ltx2_generation.py:380)
 → generator.generate()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (ltx2_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="955" y="1640" width="180" height="70" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="dm11" value="10b. resp_q.put(MediaInit / MediaChunk / MediaComplete / StepComplete)" style="endArrow=classic;html=1;strokeColor=#b85450;fontSize=10;labelBackgroundColor=#ffffff;" parent="1" edge="1">
|
||||
|
||||
File diff suppressed because one or more lines are too long
|
Before Width: | Height: | Size: 85 KiB After Width: | Height: | Size: 85 KiB |
@@ -64,8 +64,8 @@ generator:
|
||||
# internal: pipeline_config.dit_config.quant_config = FP4Config()
|
||||
# set in gpu_pool.py:280 (via the legacy in-place mutation). The
|
||||
# public typed surface resolves "NVFP4" to NVFP4Config() and pins
|
||||
# it on dit_config when resolution materializes the PipelineConfig.
|
||||
# Comment this block out on hosts without flashinfer / NVFP4 hardware.
|
||||
# it on dit_config in FastVideoArgs.__post_init__. Comment this
|
||||
# block out on hosts without flashinfer / NVFP4 hardware.
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ test.describe('preset prompt generation', () => {
|
||||
// on a B200 plus encode/transfer time. The "Continuation flipped
|
||||
// to Generating + Leave button rendered" pair above is the proof
|
||||
// the integration works: FE → /readyz → /curated-presets → WS
|
||||
// /ws → BE → GPU pool → VideoGenerator.generate, all green.
|
||||
// /ws → BE → GPU pool → VideoGenerator.generate_video, all green.
|
||||
const video = page.locator('video').first();
|
||||
await expect(video).toHaveCount(1);
|
||||
});
|
||||
|
||||
@@ -862,40 +862,23 @@ class JobRunner:
|
||||
sp_size,
|
||||
)
|
||||
|
||||
gen = VideoGenerator.from_config(
|
||||
{
|
||||
"model_path": model_id,
|
||||
"engine": {
|
||||
"num_gpus": num_gpus,
|
||||
"parallelism": {
|
||||
"tp_size": tp_size,
|
||||
"sp_size": sp_size,
|
||||
},
|
||||
"offload": {
|
||||
"dit": dit_cpu_offload,
|
||||
"dit_layerwise": dit_layerwise_offload,
|
||||
"text_encoder": text_encoder_cpu_offload,
|
||||
"image_encoder": image_encoder_cpu_offload,
|
||||
"vae": vae_cpu_offload,
|
||||
},
|
||||
"compile": {
|
||||
"enabled": enable_torch_compile
|
||||
},
|
||||
"attention": {
|
||||
"vsa_sparsity": vsa_sparsity
|
||||
},
|
||||
"use_fsdp_inference": use_fsdp_inference,
|
||||
},
|
||||
"pipeline": {
|
||||
"workload_type":
|
||||
workload_type,
|
||||
**({
|
||||
"components": {
|
||||
"override_pipeline_cls_name": override_pipeline_cls_name
|
||||
}
|
||||
} if override_pipeline_cls_name else {}),
|
||||
},
|
||||
},
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
model_id,
|
||||
workload_type=workload_type,
|
||||
num_gpus=num_gpus,
|
||||
dit_layerwise_offload=dit_layerwise_offload,
|
||||
**({
|
||||
"override_pipeline_cls_name": override_pipeline_cls_name
|
||||
} if override_pipeline_cls_name else {}),
|
||||
dit_cpu_offload=dit_cpu_offload,
|
||||
text_encoder_cpu_offload=text_encoder_cpu_offload,
|
||||
vae_cpu_offload=vae_cpu_offload,
|
||||
image_encoder_cpu_offload=image_encoder_cpu_offload,
|
||||
use_fsdp_inference=use_fsdp_inference,
|
||||
enable_torch_compile=enable_torch_compile,
|
||||
VSA_sparsity=vsa_sparsity,
|
||||
tp_size=tp_size,
|
||||
sp_size=sp_size,
|
||||
log_queue=log_queue,
|
||||
)
|
||||
|
||||
@@ -1120,33 +1103,30 @@ class JobRunner:
|
||||
# Without a name FastVideo derives the filename from the prompt.
|
||||
safe_name = re.sub(r'[\\/:*?"<>|]+', "", job.name).strip().strip(".")
|
||||
output_target = (os.path.join(job_output_dir, f"{safe_name[:80]}.mp4") if safe_name else job_output_dir)
|
||||
request: dict[str, Any] = {
|
||||
gen_kwargs: dict[str, Any] = {
|
||||
"prompt": job.prompt,
|
||||
"output_path": output_target,
|
||||
"save_video": True,
|
||||
"num_inference_steps": job.num_inference_steps,
|
||||
"num_frames": job.num_frames,
|
||||
"height": job.height,
|
||||
"width": job.width,
|
||||
"guidance_scale": job.guidance_scale,
|
||||
"guidance_rescale": job.guidance_rescale,
|
||||
"fps": job.fps,
|
||||
"seed": job.seed,
|
||||
"negative_prompt": job.negative_prompt or "",
|
||||
"sampling": {
|
||||
"num_inference_steps": job.num_inference_steps,
|
||||
"num_frames": job.num_frames,
|
||||
"height": job.height,
|
||||
"width": job.width,
|
||||
"guidance_scale": job.guidance_scale,
|
||||
"guidance_rescale": job.guidance_rescale,
|
||||
"fps": job.fps,
|
||||
"seed": job.seed,
|
||||
},
|
||||
"output": {
|
||||
"output_path": output_target,
|
||||
"save_video": True,
|
||||
},
|
||||
"log_queue": log_queue,
|
||||
}
|
||||
if job.image_path:
|
||||
request.setdefault("inputs", {})["image_path"] = job.image_path
|
||||
gen_kwargs["image_path"] = job.image_path
|
||||
if job.references:
|
||||
request.setdefault("inputs", {})["references"] = _build_h3_references(job.references)
|
||||
gen_kwargs["references"] = _build_h3_references(job.references)
|
||||
if job.last_image_path:
|
||||
# _prepare_fl2va requires a PIL image, not a path.
|
||||
from PIL import Image as _PILImage
|
||||
request.setdefault("inputs", {})["last_image"] = _PILImage.open(job.last_image_path)
|
||||
generator.generate(request, log_queue=log_queue)
|
||||
gen_kwargs["last_image"] = _PILImage.open(job.last_image_path)
|
||||
generator.generate_video(**gen_kwargs)
|
||||
|
||||
buf.phase = "saving"
|
||||
logger.info("Generation completed, searching for output file...")
|
||||
|
||||
@@ -131,9 +131,8 @@ export default function CreateJobModal({
|
||||
const editingJobId = editingJob?.id ?? null;
|
||||
const editingJobModelId = editingJob?.model_id ?? null;
|
||||
|
||||
// Layerwise offload and FSDP compete for the DiT weights and the device offload
|
||||
// policy (resolve_device_offload_conflicts in fastvideo/api/device_policy.py)
|
||||
// silently picks a winner; resolve it visibly here.
|
||||
// Layerwise offload and FSDP compete for the DiT weights and FastVideoArgs
|
||||
// silently picks a winner (fastvideo_args.py:859); resolve it visibly here.
|
||||
// dit_cpu_offload is deliberately not interlocked -- it is a modifier, not a
|
||||
// competing strategy.
|
||||
const handleDitLayerwiseOffloadChange = React.useCallback((next: boolean) => {
|
||||
|
||||
@@ -15,9 +15,6 @@ from fastvideo import VideoGenerator as FastVideoGenerator
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))))
|
||||
|
||||
# InferenceArgs keys that are SamplingConfig fields of a GenerationRequest.
|
||||
_SAMPLING_INFERENCE_ARGS = ("height", "width", "num_frames", "num_inference_steps", "guidance_scale", "seed", "fps")
|
||||
|
||||
|
||||
# Custom exception for interruption
|
||||
class GenerationInterruptedException(Exception):
|
||||
@@ -157,17 +154,7 @@ class VideoGenerator:
|
||||
"""Thread function to run the generation"""
|
||||
try:
|
||||
if self.generator is not None:
|
||||
# Place each InferenceArgs value in the GenerationRequest section that owns it.
|
||||
request: dict[str, Any] = {"prompt": prompt, "output": {"output_path": output_path}}
|
||||
for key, value in inference_args.items():
|
||||
if key == "image_path":
|
||||
section = "inputs"
|
||||
elif key in _SAMPLING_INFERENCE_ARGS:
|
||||
section = "sampling"
|
||||
else:
|
||||
section = "extensions"
|
||||
request.setdefault(section, {})[key] = value
|
||||
self.generator.generate(request)
|
||||
self.generator.generate_video(prompt=prompt, output_path=output_path, **inference_args)
|
||||
self._generation_result = os.path.join(output_path, f"{prompt[:100]}.mp4")
|
||||
else:
|
||||
raise RuntimeError("Generator is not initialized")
|
||||
@@ -266,24 +253,9 @@ class VideoGenerator:
|
||||
if self.generator is None:
|
||||
print('generation_args', generation_args)
|
||||
print('pipeline_config', pipeline_config)
|
||||
# Place each generation argument at its GeneratorConfig engine path.
|
||||
engine_config: dict[str, Any] = {}
|
||||
if "num_gpus" in generation_args:
|
||||
engine_config["num_gpus"] = generation_args["num_gpus"]
|
||||
for parallelism_key in ("tp_size", "sp_size"):
|
||||
if parallelism_key in generation_args:
|
||||
engine_config.setdefault("parallelism", {})[parallelism_key] = generation_args[parallelism_key]
|
||||
if "dit_cpu_offload" in generation_args:
|
||||
engine_config["offload"] = {"dit": generation_args["dit_cpu_offload"]}
|
||||
self.generator = FastVideoGenerator.from_config({
|
||||
"model_path": model_path,
|
||||
"engine": engine_config,
|
||||
"pipeline": {
|
||||
"experimental": {
|
||||
"pipeline_config": pipeline_config
|
||||
}
|
||||
},
|
||||
})
|
||||
self.generator = FastVideoGenerator.from_pretrained(model_path=model_path,
|
||||
**generation_args,
|
||||
pipeline_config=pipeline_config)
|
||||
|
||||
print('inference_args', inference_args)
|
||||
|
||||
|
||||
+167
-133
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"version": 11,
|
||||
"version": 15,
|
||||
"recipes": [
|
||||
{
|
||||
"id": "fastwan21-t2v",
|
||||
@@ -10,6 +10,11 @@
|
||||
"summary": "Generate a video in three denoising steps with the distilled FastWan2.1 1.3B checkpoint and video sparse attention.",
|
||||
"model": "FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
"source": "scripts/inference/inference_wan_VSA_DMD_1_3B.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fastwan21_1_3b.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN"
|
||||
},
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN fastvideo generate --config scripts/inference/inference_wan_VSA_DMD_1_3B.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
@@ -25,6 +30,13 @@
|
||||
"summary": "The maintained high-capacity Wan2.2 text-to-video example with CPU offload settings encoded in its checked-in Python source.",
|
||||
"model": "Wan-AI/Wan2.2-T2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_wan2_2.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_wan22_t2v_a14b.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"limitations": [
|
||||
"Generation at 1280 × 720 on two GPUs with DiT CPU offload is slow. The client examples stop polling after 30 minutes while the job keeps running on the server; retrieve it by ID or raise the deadline."
|
||||
]
|
||||
},
|
||||
"command": "python examples/inference/basic/basic_wan2_2.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
@@ -54,6 +66,14 @@
|
||||
"summary": "Use one maintained 5B checkpoint for text-to-video or add an image input to switch the same recipe to image-to-video.",
|
||||
"model": "Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_wan2_2_ti2v.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_wan22_ti2v_5b.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"task": "Text to video",
|
||||
"limitations": [
|
||||
"This server walkthrough sends a text prompt only. The server API also accepts an image reference (input_reference) for this checkpoint, but the checked-in clients do not upload one. For the cookbook's image-to-video workflow, use the Python command."
|
||||
]
|
||||
},
|
||||
"command": "python examples/inference/basic/basic_wan2_2_ti2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
@@ -141,53 +161,6 @@
|
||||
"Native MLX FastMetal T2V on 36 GB+ unified memory. Same --fast, --fast-spatial, and --refine flags as the 1.3B script."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "turbodiffusion-wan21-1-3b-t2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "TurboWan2.1 1.3B",
|
||||
"summary": "A TurboDiffusion-accelerated Wan2.1 1.3B text-to-video run from its maintained single-GPU example.",
|
||||
"model": "loayrashid/TurboWan2.1-T2V-1.3B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_turbodiffusion/",
|
||||
"related": ["turbodiffusion-wan21-14b-t2v", "turbowan22-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "turbodiffusion-wan21-14b-t2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "TurboWan2.1 14B",
|
||||
"summary": "TurboDiffusion acceleration applied to the 14B Wan2.1 text-to-video checkpoint; the checked-in source is configured for two GPUs.",
|
||||
"model": "loayrashid/TurboWan2.1-T2V-14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_14b.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_14b.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_turbodiffusion_14B/",
|
||||
"related": ["turbodiffusion-wan21-1-3b-t2v", "turbowan22-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "turbowan22-i2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "TurboWan2.2 A14B",
|
||||
"summary": "A one-to-four-step image-to-video path using TurboDiffusion and the SLA attention backend from its maintained example.",
|
||||
"model": "loayrashid/TurboWan2.2-I2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"related": ["turbodiffusion-wan21-14b-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "ltx2-distilled-t2v",
|
||||
"family": "ltx2",
|
||||
@@ -312,6 +285,70 @@
|
||||
"expected_artifact": "MP4 videos under video_samples_kandinsky5_i2v/",
|
||||
"related": ["kandinsky5-t2v-lite-sft"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-ti2va-base",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Text or image to video with audio",
|
||||
"label": "Kandinsky 6 Pro TI2VA",
|
||||
"summary": "Generate a five-second, 512x768 video with synchronized audio from text, or set IMAGE_PATH in the maintained example to condition on an image.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"command": "python examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 video with synchronized audio under video_samples_kandinsky6_ti2va/",
|
||||
"related": ["kandinsky6-ti2va-piflow", "kandinsky6-vsr"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-ti2va-piflow",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Distilled text or image to video with audio",
|
||||
"label": "Kandinsky 6 Pro pi-Flow",
|
||||
"summary": "Generate a five-second video with synchronized audio using the distilled pi-Flow checkpoint in ten inference steps and guidance 1.0.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"command": "KANDINSKY6_MODEL_PATH=kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers python examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 video with synchronized audio under video_samples_kandinsky6_ti2va/",
|
||||
"related": ["kandinsky6-ti2va-base", "kandinsky6-vsr-distilled"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-vsr",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Video super-resolution",
|
||||
"label": "Kandinsky 6 VSR",
|
||||
"summary": "Upscale an existing clip by 2.25x with the four-step flow-matching VSR checkpoint while preserving its source audio.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_sr.py",
|
||||
"command": ": \"${INPUT_VIDEO:?Set INPUT_VIDEO to an existing video}\"\npython examples/inference/basic/basic_kandinsky6_sr.py --video-path \"$INPUT_VIDEO\" --scale 2.25",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "Upscaled MP4 under outputs_video/kandinsky6_sr/",
|
||||
"related": ["kandinsky6-vsr-distilled", "kandinsky6-ti2va-base"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-vsr-distilled",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Distilled video super-resolution",
|
||||
"label": "Kandinsky 6 VSR distilled",
|
||||
"summary": "Upscale an existing clip by 2.25x with the distilled two-step VSR checkpoint while preserving its source audio.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_sr.py",
|
||||
"command": ": \"${INPUT_VIDEO:?Set INPUT_VIDEO to an existing video}\"\npython examples/inference/basic/basic_kandinsky6_sr.py --model-path kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers --video-path \"$INPUT_VIDEO\" --scale 2.25",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "Upscaled MP4 under outputs_video/kandinsky6_sr/",
|
||||
"related": ["kandinsky6-vsr", "kandinsky6-ti2va-piflow"]
|
||||
},
|
||||
{
|
||||
"id": "flux2-klein-t2i",
|
||||
"family": "flux",
|
||||
@@ -360,53 +397,6 @@
|
||||
"expected_artifact": "PNG images under outputs/flux_dev/samples/",
|
||||
"limitations": ["FLUX.1 is loadable by ID but registers no model_family in fastvideo/registry.py; it is grouped under FLUX for documentation only."]
|
||||
},
|
||||
{
|
||||
"id": "glm-image-t2i",
|
||||
"family": "glm_image",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "GLM-Image",
|
||||
"summary": "GLM-Image text-to-image generation from its maintained example.",
|
||||
"model": "zai-org/GLM-Image",
|
||||
"source": "examples/inference/basic/basic_glm_image.py",
|
||||
"command": "python examples/inference/basic/basic_glm_image.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at image_output/landscape.png",
|
||||
"related": ["glm-image-edit"]
|
||||
},
|
||||
{
|
||||
"id": "glm-image-edit",
|
||||
"family": "glm_image",
|
||||
"stage": "inference",
|
||||
"task": "Image editing",
|
||||
"label": "GLM-Image editing",
|
||||
"summary": "Edit an input image with an instruction prompt using GLM-Image, from its maintained editing example.",
|
||||
"model": "zai-org/GLM-Image",
|
||||
"source": "examples/inference/basic/edit_glm_image.py",
|
||||
"command": "python examples/inference/basic/edit_glm_image.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at image_output/edited.png (input: assets/images/couple.jpg)",
|
||||
"related": ["glm-image-t2i"]
|
||||
},
|
||||
{
|
||||
"id": "zimage-turbo-t2i",
|
||||
"family": "zimage",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "Z-Image Turbo",
|
||||
"summary": "Z-Image Turbo text-to-image on a single GPU from its maintained example.",
|
||||
"model": "Tongyi-MAI/Z-Image-Turbo",
|
||||
"source": "examples/inference/basic/basic_zimage.py",
|
||||
"command": "python examples/inference/basic/basic_zimage.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at outputs/zimage/zimage_turbo.png"
|
||||
},
|
||||
{
|
||||
"id": "sd35-medium-t2i",
|
||||
"family": "sd35",
|
||||
@@ -456,7 +446,8 @@
|
||||
"source": "examples/inference/basic/basic_fasth3.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"",
|
||||
"audio": true
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile all",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
@@ -481,7 +472,8 @@
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\""
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\"",
|
||||
"audio": true
|
||||
},
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
@@ -528,7 +520,8 @@
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_spark.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3"
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3",
|
||||
"audio": true
|
||||
},
|
||||
"command": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
@@ -544,7 +537,7 @@
|
||||
"limitations": [
|
||||
"Install from the DGX Spark guide, not the generic CUDA extra. GB10 has no FA4 / sm_100a VSA kernel; keep FASTVIDEO_FA4=0 and FASTVIDEO_VSA_SM100A=0.",
|
||||
"Legal num_frames values are 17n+5, capped at 345 (15 s). Native 16:9 sizes include 832x480 and 1344x768.",
|
||||
"Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Do not set engine.offload.lazy_module_load to false on this box.",
|
||||
"Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Do not pass --no-lazy-module-load on this box.",
|
||||
"A 345-frame request on one Spark can OOM. Prefer 124 or 243 frames, TAEH3 decode, or two Sparks over QSFP."
|
||||
]
|
||||
},
|
||||
@@ -578,6 +571,77 @@
|
||||
"Height, width, frames, and steps in the YAML are examples. Edit them or pass CLI flags. See docs/getting_started/installation/spark_pair.md."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "compacth3-rtx5090",
|
||||
"group": "compacth3-rtx5090",
|
||||
"group_label": "CompactH3 on RTX 5090",
|
||||
"group_task": "4-step 42-block text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "CompactH3 NVFP4 on RTX 5090",
|
||||
"summary": "Run the 42-block CompactH3 NVFP4 DiT with the NVFP4 Qwen3-VL encoder, Comfy int8-convrot VAE, and SageAttention3 FP4 on one 32 GB RTX 5090. Sequential load parks the encoder in pinned host RAM.",
|
||||
"model": "./CompactH3",
|
||||
"source": "examples/inference/basic/basic_compacth3_rtx5090.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_compacth3_rtx5090.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu128 uv pip install -e \".[fasth3]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree --local-dir ./CompactH3 --include model_index.json --include \"tokenizer/**\" --include \"processor/**\" --include \"scheduler/**\" --include \"audio_scheduler/**\" --include \"audio_vae/**\" --include \"vae/**\"\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-bf16 --local-dir ./CompactH3/transformer\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-nvfp4 --local-dir ./CompactH3/transformer --include nvfp4_weights.safetensors\nhf download KyleNeverGivesUp/FastH3-text-encoder-nvfp4 --local-dir ./CompactH3/text_encoder\nhf download Comfy-Org/MiniMax-H3 --local-dir ./Comfy-MiniMax-H3 --include vae/minimax_h3_video_vae_int8_convrot.safetensors\ncp ./Comfy-MiniMax-H3/vae/minimax_h3_video_vae_int8_convrot.safetensors ./CompactH3/vae/",
|
||||
"env": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a"
|
||||
},
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_compacth3_rtx5090.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/compacth3_rtx5090/",
|
||||
"modes": ["T2VA", "CompactH3 NVFP4", "RTX 5090"],
|
||||
"limitations": [
|
||||
"Assemble ./CompactH3 before running. The DiT NVFP4 export and encoder snapshot are gated; run huggingface-cli login and accept each repo license.",
|
||||
"32 GB cannot keep the NVFP4 encoder and DiT on the GPU together. Keep h3_sequential_load on and lazy_module_load off.",
|
||||
"Blackwell sm_120 needs ATTN_QAT_INFER, FLASHINFER_CUDA_ARCH_LIST=12.0a, and a CUDA 12.8 PyTorch wheel.",
|
||||
"Legal num_frames values are 17n+5, capped at 362 (15.08 s). Native 16:9 sizes include 832x480 and 1344x768; 1344x768 on 32 GB is unmeasured."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "compacth3-rtx-pro6000",
|
||||
"group": "compacth3-rtx-pro6000",
|
||||
"group_label": "CompactH3 on RTX PRO 6000",
|
||||
"group_task": "4-step 42-block text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "CompactH3 NVFP4 on RTX PRO 6000 Blackwell",
|
||||
"summary": "Run CompactH3 NVFP4 with the encoder, DiT, and int8-convrot VAE resident on one 96 GB RTX PRO 6000 Blackwell. The checked-in example is 1344x768 and 124 frames (5.17 s) with VAE torch.compile. Use 832x480 for clip-queue playground traffic.",
|
||||
"model": "./CompactH3",
|
||||
"source": "examples/inference/basic/basic_compacth3_rtx_pro6000.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_compacth3_rtx_pro6000.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu128 uv pip install -e \".[fasth3]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree --local-dir ./CompactH3 --include model_index.json --include \"tokenizer/**\" --include \"processor/**\" --include \"scheduler/**\" --include \"audio_scheduler/**\" --include \"audio_vae/**\" --include \"vae/**\"\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-bf16 --local-dir ./CompactH3/transformer\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-nvfp4 --local-dir ./CompactH3/transformer --include nvfp4_weights.safetensors\nhf download KyleNeverGivesUp/FastH3-text-encoder-nvfp4 --local-dir ./CompactH3/text_encoder\nhf download Comfy-Org/MiniMax-H3 --local-dir ./Comfy-MiniMax-H3 --include vae/minimax_h3_video_vae_int8_convrot.safetensors\ncp ./Comfy-MiniMax-H3/vae/minimax_h3_video_vae_int8_convrot.safetensors ./CompactH3/vae/",
|
||||
"env": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a"
|
||||
},
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_compacth3_rtx_pro6000.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/compacth3_rtx_pro6000/",
|
||||
"modes": ["T2VA", "CompactH3 NVFP4", "RTX PRO 6000"],
|
||||
"limitations": [
|
||||
"Assemble ./CompactH3 before running. The DiT NVFP4 export and encoder snapshot are gated; run huggingface-cli login and accept each repo license.",
|
||||
"96 GB keeps the encoder, DiT, and VAE on GPU. Do not enable h3_sequential_load or lazy_module_load on this box.",
|
||||
"Enable compile.vae_enabled. Leave inference_torch_compile off: FlashInfer and Sage3 custom ops cannot be compiled.",
|
||||
"Blackwell sm_120 needs ATTN_QAT_INFER, FLASHINFER_CUDA_ARCH_LIST=12.0a, and a CUDA 12.8 PyTorch wheel.",
|
||||
"Legal num_frames values are 17n+5, capped at 362 (15.08 s). Native 16:9 sizes include 832x480 and 1344x768. Dense CompactH3 has zero VSA gates; the pipeline raises if VIDEO_SPARSE_ATTN_H3 is loaded with all-zero to_gate_compress weights."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-cuda",
|
||||
"group": "fasth3-8step-v2",
|
||||
@@ -592,7 +656,8 @@
|
||||
"source": "examples/inference/basic/basic_fasth3_8step.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_8step.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"",
|
||||
"audio": true
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3_8step.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile strict --no-inference-torch-compile --no-compile-vae",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
@@ -621,7 +686,8 @@
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3_8step.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa"
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa",
|
||||
"audio": true
|
||||
},
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
@@ -722,38 +788,6 @@
|
||||
],
|
||||
"limitations": ["Supply a compatible FastH3 adapter. The script infers dense or VSA attention from the adapter payload unless you override it."]
|
||||
},
|
||||
{
|
||||
"id": "longcat-t2v",
|
||||
"family": "longcat",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "LongCat Video T2V",
|
||||
"summary": "LongCat Video text-to-video at 480p (50 steps), with distilled and 720p refinement passes included in the same maintained script.",
|
||||
"model": "FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
"source": "examples/inference/basic/basic_longcat_t2v.py",
|
||||
"command": "python examples/inference/basic/basic_longcat_t2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video/longcat_t2v_basic/, longcat_t2v_distill/, and longcat_t2v_refine_720p/",
|
||||
"related": ["longcat-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "longcat-i2v",
|
||||
"family": "longcat",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "LongCat Video I2V",
|
||||
"summary": "LongCat Video image-to-video with optional distilled and refinement passes, from its maintained example.",
|
||||
"model": "FastVideo/LongCat-Video-I2V-Diffusers",
|
||||
"source": "examples/inference/basic/basic_longcat_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_longcat_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video/longcat_i2v_basic/ and longcat_i2v_distill/",
|
||||
"related": ["longcat-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "stable-audio-open-t2a",
|
||||
"family": "stable_audio",
|
||||
|
||||
+31
-12
@@ -506,6 +506,10 @@
|
||||
const runtime = runtimeFor(recipe);
|
||||
const profile = servingPanel && servingProfiles[recipe.id];
|
||||
const useServer = Boolean(profile && usagePreference === "server");
|
||||
// Only some servers expose the browser playground or return audio; both come from the serving profile.
|
||||
const hasPlayground = Boolean(profile && profile.playground_url);
|
||||
const hasAudio = Boolean(profile && profile.audio);
|
||||
const playgroundAny = familyRecipes.some((item) => servingProfiles[item.id] && servingProfiles[item.id].playground_url);
|
||||
// The measured local profile and the server config have separate evidence.
|
||||
const activeRecipe = useServer ? { ...recipe, hardware: profile.hardware, evidence: "Source-backed" } : recipe;
|
||||
const knobs = knobsFor(recipe);
|
||||
@@ -515,14 +519,19 @@
|
||||
usage.querySelectorAll("[data-cookbook-mode]").forEach((option) => {
|
||||
const selected = option.dataset.cookbookMode === (useServer ? "server" : "python");
|
||||
option.disabled = option.dataset.cookbookMode === "server" && !profile;
|
||||
const hint = option.querySelector("[data-cookbook-server-hint]");
|
||||
if (hint) hint.textContent = (profile ? hasPlayground : playgroundAny) ? "Playground, cURL, or an API client" : "cURL or an API client";
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
const servedNames = [...new Set(familyRecipes.filter((item) => servingProfiles[item.id]).map((item) => item.group_label || item.label))];
|
||||
servingAvailability.textContent = profile
|
||||
? "The playground and the OpenAI Python client share one server process. Both workflows can run on your own machine."
|
||||
? `${hasPlayground ? "The playground" : "cURL"} and the OpenAI Python client share one server process. Both workflows can run on your own machine.`
|
||||
: servingLoadFailed
|
||||
? "Server examples could not be loaded. Open the H3 server guide below, or use Python directly."
|
||||
: "This recipe uses Python directly. FastH3 V1 and FastH3 V2 can also run a local server for the playground and the OpenAI Python client.";
|
||||
? "Server examples could not be loaded. Use Python directly."
|
||||
: servedNames.length
|
||||
? `This recipe uses Python directly. ${new Intl.ListFormat("en").format(servedNames)} can also run a local server for ${playgroundAny ? "the playground and " : "cURL and "}the OpenAI Python client.`
|
||||
: "This recipe uses Python directly.";
|
||||
servingPanel.hidden = !useServer;
|
||||
commandBlock.hidden = useServer;
|
||||
root.querySelector("[data-cookbook-python-note]").hidden = useServer;
|
||||
@@ -531,11 +540,12 @@
|
||||
if (useServer) {
|
||||
const isMLX = profile.runtime === "mlx";
|
||||
const isSpark = runtime.id === "spark";
|
||||
const where = hasPlayground ? "in the playground or your app" : "in your app";
|
||||
servingPanel.querySelector("[data-cookbook-server-lifetime]").textContent = isMLX
|
||||
? "Start once, then change prompts in the playground or your app. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident."
|
||||
? `Start once, then change prompts ${where}. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident.`
|
||||
: isSpark
|
||||
? "Start once, then change prompts in the playground or your app. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
? `Start once, then change prompts ${where}. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache.`
|
||||
: `Start once, then change prompts ${where}. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.`;
|
||||
servingPanel.querySelector("[data-cookbook-install-guide]").href = isMLX
|
||||
? "../../getting_started/installation/mlx/"
|
||||
: isSpark
|
||||
@@ -546,7 +556,10 @@
|
||||
servingPanel.querySelector("[data-cookbook-server-install]").textContent = profile.install;
|
||||
servingPanel.querySelector("[data-cookbook-server-command]").textContent = profile.command;
|
||||
servingPanel.querySelector("[data-cookbook-health-command]").textContent = profile.health_command;
|
||||
servingPanel.querySelector("[data-cookbook-playground]").href = profile.playground_url;
|
||||
servingPanel.querySelectorAll("[data-cookbook-playground-only]").forEach((element) => {
|
||||
element.hidden = !hasPlayground;
|
||||
});
|
||||
if (hasPlayground) servingPanel.querySelector("[data-cookbook-playground]").href = profile.playground_url;
|
||||
const client = profile.clients[selectedClient];
|
||||
const filename = client.source.split("/").pop();
|
||||
servingPanel.querySelector("[data-cookbook-client-install]").textContent = client.install;
|
||||
@@ -574,13 +587,17 @@
|
||||
});
|
||||
|
||||
description.textContent = useServer
|
||||
? `${recipe.group_label || recipe.label} generates video with audio. Start the local server, then use the playground or the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
? `${recipe.group_label || recipe.label} generates video${hasAudio ? " with audio" : ""}. Start the local server, then use ${hasPlayground ? "the playground or " : "cURL or "}the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
: recipe.summary;
|
||||
label.textContent = useServer ? `${recipe.group_label || recipe.label} · Server` : recipe.label;
|
||||
model.textContent = recipe.model;
|
||||
task.textContent = recipe.task;
|
||||
task.textContent = (useServer && recipe.serving && recipe.serving.task) || recipe.task;
|
||||
hardwareValue.textContent = runtimeSummary(activeRecipe);
|
||||
if (artifact) artifact.textContent = useServer ? "MP4 with audio" : recipe.expected_artifact || "Not yet documented for this recipe.";
|
||||
if (artifact) {
|
||||
artifact.textContent = useServer
|
||||
? (hasAudio ? "MP4 with audio" : "MP4 video")
|
||||
: recipe.expected_artifact || "Not yet documented for this recipe.";
|
||||
}
|
||||
if (evidenceCell) {
|
||||
evidenceCell.textContent = activeRecipe.evidence || "Source-backed";
|
||||
evidenceCell.classList.toggle("cookbook-badge--verified", activeRecipe.evidence === "Verified");
|
||||
@@ -602,10 +619,12 @@
|
||||
? "This MLX server config has no recorded hardware run. Measurements from the Python recipe are not server memory requirements. Only text-to-video/audio is wired; reference inputs and fast modes are not exposed here."
|
||||
: runtime.id === "spark"
|
||||
? "This Spark server config has no recorded serving benchmark. Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Compilation of the DiT is disabled."
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
: `This server config has no recorded serving benchmark.${profile.compile_enabled === false ? " Compilation is disabled, unlike the measured Python performance profile." : ""}`,
|
||||
`${profile.sampling.width} × ${profile.sampling.height} · ${profile.sampling.num_frames} frames · ${profile.sampling.fps} fps. The server supplies these defaults; the client sends the model and prompt.`,
|
||||
"Generation is serialized. Job metadata is held in memory and is lost when the server restarts.",
|
||||
] : recipe.limitations || []), ...knobCaveats];
|
||||
] : recipe.limitations || []),
|
||||
...(useServer && recipe.serving && Array.isArray(recipe.serving.limitations) ? recipe.serving.limitations : []),
|
||||
...knobCaveats];
|
||||
notes.replaceChildren();
|
||||
notes.hidden = limitations.length === 0;
|
||||
if (limitations.length) {
|
||||
|
||||
@@ -1515,7 +1515,8 @@ img {
|
||||
|
||||
.cookbook-serving[hidden],
|
||||
.cookbook-command[hidden],
|
||||
[data-cookbook-python-note][hidden] {
|
||||
[data-cookbook-python-note][hidden],
|
||||
[data-cookbook-playground-only][hidden] {
|
||||
display: none;
|
||||
}
|
||||
|
||||
|
||||
@@ -9,18 +9,15 @@ plain typographic tile instead — that tile is a UI placeholder, not a logo.
|
||||
| --- | --- | --- |
|
||||
| `wan-ai.webp` | [Official Wan-AI Hugging Face organization avatar](https://huggingface.co/Wan-AI) | Wan family card and page header |
|
||||
| `ltx.webp` | [Official Lightricks Hugging Face organization avatar](https://huggingface.co/Lightricks) | LTX family card and page header |
|
||||
| `tencent-hunyuan.webp` | [Official Tencent Hunyuan Hugging Face organization avatar](https://huggingface.co/Tencent-Hunyuan) | Hunyuan and GameCraft cards, Hunyuan page header |
|
||||
| `tencent-hunyuan.webp` | [Official Tencent Hunyuan Hugging Face organization avatar](https://huggingface.co/Tencent-Hunyuan) | Hunyuan family card and page header |
|
||||
| `nvidia.webp` | [Official NVIDIA Hugging Face organization avatar](https://huggingface.co/nvidia) | Cosmos and GEN3C cards, Cosmos page header |
|
||||
| `kandinsky.webp` | [Official Kandinsky Lab Hugging Face organization avatar](https://huggingface.co/kandinskylab) | Kandinsky 5 family card and page header |
|
||||
| `kandinsky.webp` | [Official Kandinsky Lab Hugging Face organization avatar](https://huggingface.co/kandinskylab) | Kandinsky 5 and Kandinsky 6 cards and page headers |
|
||||
| `black-forest-labs.webp` | [Official Black Forest Labs Hugging Face organization avatar](https://huggingface.co/black-forest-labs) | FLUX family card and page header |
|
||||
| `minimax.webp` | [Official MiniMax Hugging Face organization avatar](https://huggingface.co/MiniMaxAI) | MiniMax H3 family card and page header |
|
||||
| `tongyi.webp` | [Official Tongyi MAI Hugging Face organization avatar](https://huggingface.co/Tongyi-MAI) | Z-Image family card and page header |
|
||||
| `zai.webp` | [Official Z.ai Hugging Face organization avatar](https://huggingface.co/zai-org) | GLM-Image family card and page header |
|
||||
| `stabilityai.webp` | [Official Stability AI Hugging Face organization avatar](https://huggingface.co/stabilityai) | Stable Diffusion and Stable Audio cards and page headers |
|
||||
| `meituan-longcat.webp` | [Official Meituan LongCat Hugging Face organization avatar](https://huggingface.co/meituan-longcat) | LongCat family card and page header |
|
||||
| `fastvideo.webp` | [Official FastVideo Hugging Face organization avatar](https://huggingface.co/FastVideo) | Matrix Game and MMAudio cards (converted weights published by this org), DreamX card |
|
||||
| `fastvideo.webp` | [Official FastVideo Hugging Face organization avatar](https://huggingface.co/FastVideo) | Matrix Game and MMAudio cards (converted weights published by this org) |
|
||||
|
||||
Typographic tiles (no vendored image): TurboDiffusion ("Turbo"), HY-World
|
||||
("HY"), LingBot ("LB"), MMAudio page header ("MMA"). These publishers have no
|
||||
Typographic tiles (no vendored image): HY-World ("HY"), LingBot ("LB"),
|
||||
MMAudio page header ("MMA"). These publishers have no
|
||||
single official mark appropriate for reuse in the catalog; add a licensed
|
||||
asset here if one becomes available.
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 5.1 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 6.0 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 2.4 KiB |
@@ -57,15 +57,19 @@ Minimal usage example (based on `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, {"engine": {"num_gpus": 1}})
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
|
||||
video = generator.generate({
|
||||
"prompt": "A vibrant city street at sunset.",
|
||||
"sampling": {"num_frames": 45},
|
||||
"output": {"output_path": "video_samples", "save_video": True},
|
||||
})
|
||||
sampling = SamplingParam.from_pretrained(model_id)
|
||||
sampling.num_frames = 45
|
||||
video = generator.generate_video(
|
||||
"A vibrant city street at sunset.",
|
||||
sampling_param=sampling,
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
```
|
||||
|
||||
## Some questions to ask yourself before starting
|
||||
|
||||
@@ -122,9 +122,8 @@ in `examples/`, `scripts/`, `docs/`, `apps/`, and the tests in the same pull req
|
||||
minor release.
|
||||
|
||||
To remove a variable that no code reads, delete its entry and add the name to `DEPRECATED_VARIABLES` in
|
||||
`fastvideo/envs.py` with a reason. The config resolution step `warn_deprecated_environment_variables` in
|
||||
`fastvideo/api/inference_resolution.py` calls `envs.warn_deprecated_variables()`, which logs a warning for each listed
|
||||
variable that is set. Delete the entry in the next minor release.
|
||||
`fastvideo/envs.py` with a reason. `FastVideoArgs` calls `envs.warn_deprecated_variables()`, which logs a warning for
|
||||
each listed variable that is set. Delete the entry in the next minor release.
|
||||
|
||||
## What the contract test checks
|
||||
|
||||
@@ -189,17 +188,18 @@ longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
| `FASTVIDEO_LOGGING_LEVEL` | str | `INFO` | logging | Default logging level. |
|
||||
| `FASTVIDEO_LOGGING_PREFIX` | str | `""` | logging | Prefix prepended to every log message. |
|
||||
| `FASTVIDEO_STAGE_LOGGING` | bool | `0` | logging | Log the time that each pipeline stage takes. |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | str | unset | attention | Attention backend, as an AttentionBackendEnum name such as TORCH_SDPA, FLASH_ATTN, VIDEO_SPARSE_ATTN, SAGE_ATTN, or SAGE_ATTN_THREE. Config resolution uses it when engine.attention.backend is unset. An unsupported name raises an error. |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | str | unset | attention | Attention backend, as an AttentionBackendEnum name such as TORCH_SDPA, FLASH_ATTN, VIDEO_SPARSE_ATTN, SAGE_ATTN, or SAGE_ATTN_THREE. FastVideoArgs uses it when FastVideoArgs.attention_backend is unset. |
|
||||
| `FASTVIDEO_FA4` | bool | `0` | attention | The FLASH_ATTN backend uses FlashAttention-4 (flash_attn.cute) instead of FA3 or FA2. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FA4_PACKED_VARLEN` | bool | `0` | attention | MiniMax-H3 dense DiT self-attention uses the FlashAttention-4 packed-varlen entry point. This changes the floating-point reduction order, so it is an inference-only opt-in. |
|
||||
| `FASTVIDEO_VSA_SM100A` | bool | `0` | attention | VIDEO_SPARSE_ATTN_H3 sends no-grad tile-64 forwards to the data-center Blackwell (sm_100a) kernel. fastvideo-kernel reads the same variable with the same rule. |
|
||||
| `FASTVIDEO_VSA_TRITON` | bool | `0` | attention | Force the Triton MiniMax-H3 sparse attention kernel. fastvideo-kernel reads the same variable. |
|
||||
| `FASTVIDEO_NVFP4_FA4` | bool | `0` | attention | FlashAttention-4 quantizes Q and K to NVFP4. An explicit nvfp4_fa4 attention implementation argument takes precedence. |
|
||||
| `FASTVIDEO_DISABLE_ATTENTION_COMPILE` | bool | `1` | attention | Keep attention forward out of torch.compile graphs (torch.compiler.disable). Set it to 0 to let attention constructed under that setting be traced. Setting it explicitly to true also blocks regional compile. |
|
||||
| `FASTVIDEO_MLX_WINDOW` | int | `0` | attention | MLX FastWan windowed attention size in tokens. 0 uses full attention. |
|
||||
| `FASTVIDEO_MLX_WINDOW_SINK` | int | `0` | attention | Number of sink tokens that MLX windowed attention always attends to. |
|
||||
| `FASTVIDEO_INFERENCE_TORCH_COMPILE` | bool | `0` | performance | Compile each DiT transformer block with fullgraph torch.compile at inference. Same as engine.compile.regional=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE` | bool | `0` | performance | MiniMax-H3 VAE decode splits its temporal chunks across the sequence-parallel ranks instead of running serially on the output rank. Same as pipeline.model.minimax_h3.vae_parallel_decode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_ENCODE` | bool | `0` | performance | MiniMax-H3 reference-video VAE encode splits its temporal chunks across the sequence-parallel ranks. Same as pipeline.model.minimax_h3.vae_parallel_encode=True. |
|
||||
| `FASTVIDEO_INFERENCE_TORCH_COMPILE` | bool | `0` | performance | Compile each DiT transformer block with fullgraph torch.compile at inference. Same as FastVideoArgs.inference_torch_compile=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE` | bool | `0` | performance | MiniMax-H3 VAE decode splits its temporal chunks across the sequence-parallel ranks instead of running serially on the output rank. Same as FastVideoArgs.vae_parallel_decode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_ENCODE` | bool | `0` | performance | MiniMax-H3 reference-video VAE encode splits its temporal chunks across the sequence-parallel ranks. Same as FastVideoArgs.vae_parallel_encode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE_STRATEGY` | str | unset | performance | Collective that moves chunks in parallel VAE decode: gather (used when unset) or all_gather. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FUSIONS` | str | `""` | performance | MiniMax-H3 inference-only Triton fusions: all, 1, or a comma-separated subset of modulate,qknorm_rope,swiglu. Empty, 0, or none keeps the eager implementation. |
|
||||
| `FASTVIDEO_FSDP2_AUTOWRAP` | bool | `0` | performance | FSDP2 shards modules by parameter count instead of the model's shard conditions. Not supported by self-forcing distillation. |
|
||||
@@ -235,12 +235,41 @@ longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
| `FASTVIDEO_LTX2_GEMMA_LOG` | str | `""` | debug | Log file for LTX-2 Gemma text-encoder hidden states, used by parity tests. Deprecated names: `LTX2_FASTVIDEO_GEMMA_LOG`. |
|
||||
| `FASTVIDEO_COSMOS25_LOG_KNOBS` | bool | `0` | debug | Log the Cosmos 2.5 latent-preparation conditioning inputs. |
|
||||
| `FASTVIDEO_CFG_GATE_STEP` | float | `1.0` | sampling | CFG gating fraction in [0, 1]. Steps before len(timesteps) \* X run the conditional and unconditional forwards; later steps reuse the cached difference. 1.0 disables gating. |
|
||||
| `FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS` | bool | `1` | sampling | LTX-2 uses the distilled sigma schedule when pipeline.model.ltx2.use_distilled_sigmas is also true. Deprecated names: `LTX2_USE_DISTILLED_SIGMAS`. |
|
||||
| `FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS` | bool | `1` | sampling | LTX-2 uses the distilled sigma schedule when FastVideoArgs.ltx2_use_distilled_sigmas is also true. Deprecated names: `LTX2_USE_DISTILLED_SIGMAS`. |
|
||||
| `FASTVIDEO_EVAL_CACHE` | path | computed | eval | Cache directory for evaluation models and datasets. Defaults to $FASTVIDEO_CACHE_ROOT/eval. |
|
||||
| `FASTVIDEO_PHYSICS_IQ_BUCKET_URL` | str | `https://storage.googleapis.com/physics-iq-benchmark` | eval | Base URL of the Physics-IQ benchmark bucket. |
|
||||
| `FASTVIDEO_VBENCH_FULL_INFO_JSON` | str | unset | eval | Path to VBench_full_info.json, used instead of the vendored copy. Deprecated names: `VBENCH_FULL_INFO_JSON`. |
|
||||
| `FASTVIDEO_FVD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the FVD metric. |
|
||||
| `FASTVIDEO_FAD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the audio Frechet distance metric. |
|
||||
| `FASTVIDEO_H3_VSA_FP4` | bool | `0` | attention | Run MiniMax-H3 VSA attention on the block-sparse SageAttention3 FP4 kernel (sm_120, no-grad, single sequence-parallel rank). |
|
||||
| `FASTVIDEO_H3_VSA_TILE_FIRST` | bool | `0` | attention | Single-rank MiniMax-H3 VSA with one tile gather of the block input instead of separate Q/K/V/gate scatters. |
|
||||
| `FASTVIDEO_H3_VSA_SM89_KERNEL` | one of original, bf16, int8 | `original` | attention | Fine-attention kernel for MiniMax-H3 VSA on sm_89: original, bf16, or int8 (INT8 QK, BF16 PV). |
|
||||
| `FASTVIDEO_H3_SIM_SP_FP8` | bool | `0` | debug | Simulate the FP8 sequence-parallel exchange of the MiniMax-H3 FP4 VSA path on one rank. |
|
||||
| `FASTVIDEO_H3_FFN_CHUNK_TOKENS` | int | `0` | performance | Inference-only MiniMax-H3 FFN token chunk size; 0 runs the FFN unchunked. |
|
||||
| `FASTVIDEO_H3_FP8_ATTENTION` | bool | `0` | performance | With NVFP4 layer_profile h3_dit_ffn, run MiniMax-H3 attention projections in FP8. |
|
||||
| `FASTVIDEO_H3_FP8_GRANULARITY` | one of tensor, channel | `tensor` | performance | FP8 scaling granularity for FASTVIDEO_H3_FP8_ATTENTION. |
|
||||
| `FASTVIDEO_NVFP4_MM_BACKEND` | str | `auto` | performance | FlashInfer mm_fp4 backend for NVFP4 linears, e.g. auto or cutlass. |
|
||||
| `FASTVIDEO_NVFP4_ACT_AMAX` | path | unset | performance | JSON of calibrated NVFP4 input amax per linear, keyed b<block>.<sub> or full prefix; sets a static activation scale. |
|
||||
| `FASTVIDEO_NVFP4_DYNAMIC_ACT` | str | `""` | performance | NVFP4 linears that derive the activation scale per call: all, or comma-separated layer-name suffixes such as ff.fc_out. |
|
||||
| `FASTVIDEO_H3_ADALN_CACHE` | bool | `0` | performance | Cache MiniMax-H3 AdaLN modulation per timestep instead of keeping the projection weights resident. |
|
||||
| `FASTVIDEO_H3_ADALN_TABLE` | path | unset | performance | Precomputed MiniMax-H3 AdaLN modulation table; enables the cache and skips loading the AdaLN projection weights. |
|
||||
| `FASTVIDEO_H3_ADALN_DUMP` | path | unset | debug | Write the MiniMax-H3 AdaLN modulation table to this path while sampling. |
|
||||
| `FASTVIDEO_H3_SPLICE_TRANSFORMER` | path | unset | eval | Second MiniMax-H3 transformer that runs the late DMD steps (checkpoint step-splice evaluation). |
|
||||
| `FASTVIDEO_H3_SPLICE_FROM_STEP` | int | `4` | eval | First denoising step run by FASTVIDEO_H3_SPLICE_TRANSFORMER. |
|
||||
| `FASTVIDEO_H3_ENCODER_LAYERWISE` | bool | `0` | performance | Stream MiniMax-H3 text-encoder language layers through exact-size pinned host memory (text-only prompts). |
|
||||
| `FASTVIDEO_H3_ENCODER_FUSED_DEQUANT` | bool | `0` | performance | Expand the serialized NVFP4 MiniMax-H3 text encoder with one fused Triton pass on GPUs without FP4 GEMM. |
|
||||
| `FASTVIDEO_H3_VAE_TILE_BATCH` | int | `1` | performance | Spatial tiles per MiniMax-H3 video VAE decoder call; 1 decodes per tile. |
|
||||
| `FASTVIDEO_H3_VAE_INT8_SHARED_QKV` | bool | `0` | performance | Share the INT8 activation rotation and quantization across the MiniMax-H3 VAE Q/K/V projections. |
|
||||
| `FASTVIDEO_H3_VAE_INT8_TRANSPOSE_VIEW` | bool | `0` | performance | Use transposed weight views in the MiniMax-H3 VAE INT8 projections. |
|
||||
| `FASTVIDEO_H3_VAE_INT8_FUSED_DEQUANT` | bool | `0` | performance | Fused dequantization epilogue for the MiniMax-H3 VAE INT8 projections. |
|
||||
| `FASTVIDEO_H3_PINNED_SWAP` | bool | `1` | performance | Swap offloaded MiniMax-H3 modules through exact-size pinned host arenas. |
|
||||
| `FASTVIDEO_H3_PARK_MODULES` | str | unset | performance | Comma-separated MiniMax-H3 denoise modules parked on the host while the text encoder runs, e.g. vae,audio_vae. |
|
||||
| `FASTVIDEO_LAYERWISE_OFFLOAD_BUFFERS` | bool | `0` | performance | Layerwise offload also streams large buffers such as packed FP4/FP8 weights. |
|
||||
| `FASTVIDEO_LAYERWISE_RESIDENT_BLOCKS` | int | `0` | performance | Keep the first N layerwise-offloaded blocks resident on the GPU. |
|
||||
| `FASTVIDEO_H3_SP_PROFILE` | bool | `0` | profiling | CUDA-event spans per stage over one MiniMax-H3 FP4 VSA DiT forward. |
|
||||
| `FASTVIDEO_H3_CAPTURE_QKV` | path | unset | debug | Directory for captured real MiniMax-H3 Q/K/V attention inputs. |
|
||||
| `FASTVIDEO_CUDA_MEMORY_CAP_GIB` | float | `0.0` | debug | Cap this process's CUDA allocator at this many GiB to emulate a smaller GPU; 0 leaves it uncapped. |
|
||||
| `FASTVIDEO_MEMORY_REPORT` | bool | `0` | debug | Log bytes held per pipeline component by device and dtype after loading. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_DATA_DIR` | str | `data/cats` | test | Raw data directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_CAPTION_JSON` | str | `videos2caption_1_sample.json` | test | Caption file, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_CAPTION_JSON`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_VIDEO_SUBDIR` | str | `video` | test | Video subdirectory, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_VIDEO_SUBDIR`. |
|
||||
@@ -260,35 +289,43 @@ longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ALLOW_LOW_MEMORY` | bool | `0` | test | Run the MLX Wan2.2 5B real-weights parity test on hosts with little memory. Deprecated names: `FASTVIDEO_WAN22_5B_ALLOW_LOW_MEMORY`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ROOT` | str | unset | test | Local Wan2.2 5B checkpoint for the MLX real-weights parity test. Deprecated names: `FASTVIDEO_WAN22_5B_ROOT`. |
|
||||
| `FASTVIDEO_TEST_GRADNORM_UPDATE` | bool | `0` | test | Gradient-norm regression tests update their references. Deprecated names: `FASTVIDEO_GRADNORM_UPDATE`. |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_SSIM_MODEL_PATH` | str | `FastVideo/DreamX-World-5B-Cam-Diffusers` | test | Model for the DreamX-World camera SSIM test. Deprecated names: `DREAMX_WORLD_SSIM_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_AR_SSIM_MODEL_PATH` | str | `FastVideo/DreamX-World-5B-Diffusers` | test | Model for the DreamX-World autoregressive SSIM test. Deprecated names: `DREAMX_WORLD_AR_SSIM_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_FLUX_T2I_MODEL_DIR` | str | `black-forest-labs/FLUX.1-dev` | test | Model for the Flux text-to-image SSIM test. Deprecated names: `FLUX_T2I_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_FLUX_TRANSFORMER_PATH` | str | unset | test | Local Flux transformer for the Flux transformer test. Deprecated names: `FLUX_TRANSFORMER_PATH`. |
|
||||
| `FASTVIDEO_TEST_GAMECRAFT_MODEL_PATH` | str | `FastVideo/HunyuanGameCraft-Diffusers` | test | Model for the HunyuanGameCraft SSIM test. Deprecated names: `GAMECRAFT_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_MODEL_PATH` | str | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | test | Model for the GEN3C SSIM test. Deprecated names: `GEN3C_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_IMAGE_PATH` | str | unset | test | Input image for the GEN3C SSIM test. Deprecated names: `GEN3C_TEST_IMAGE_PATH`. |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_LOCAL_WEIGHTS_DIR` | str | unset | test | Local official GLM-Image weights for the GLM-Image SSIM test. Deprecated names: `GLM_IMAGE_LOCAL_WEIGHTS_DIR`. |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_MODEL_DIR` | str | unset | test | Model for the GLM-Image SSIM test. Deprecated names: `GLM_IMAGE_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_NUM_GPUS` | int | `1` | test | GPUs for the Kandinsky5 nightly end-to-end overfit test. Deprecated names: `KANDINSKY5_E2E_NUM_GPUS`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_WRITE_REFERENCE` | bool | `0` | test | The Kandinsky5 nightly end-to-end test writes a missing reference video. Deprecated names: `KANDINSKY5_E2E_WRITE_REFERENCE`. |
|
||||
| `FASTVIDEO_TEST_LONGCAT_MODEL_ROOT` | str | unset | test | Local LongCat-Video checkpoint for the golden-gate test. Deprecated names: `LONGCAT_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_GOLDEN_DIR` | str | unset | test | Local directory of MiniMax-H3 golden-gate tensors. Deprecated names: `MINIMAX_H3_GATE_GOLDEN_DIR`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_LAYER` | int | `0` | test | Transformer layer that the MiniMax-H3 golden-gate test checks. Deprecated names: `MINIMAX_H3_GATE_LAYER`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_MODEL_ROOT` | str | unset | test | Local MiniMax-H3 checkpoint for the golden-gate test. Deprecated names: `MINIMAX_H3_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_SD35_MODEL_DIR` | str | `stabilityai/stable-diffusion-3.5-medium` | test | Model for the Stable Diffusion 3.5 SSIM test. Deprecated names: `SD35_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_TAEH3_REFERENCE_DIR` | str | unset | test | Upstream taehv checkout for the MLX TAEH3 parity test. Deprecated names: `TAEH3_REFERENCE_DIR`. |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_DIR` | str | `Tongyi-MAI/Z-Image-Turbo` | test | Model for the Z-Image SSIM test. Deprecated names: `ZIMAGE_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_REVISION` | str | `f332072aa78be7aecdf3ee76d5c247082da564a6` | test | Hugging Face revision of the Z-Image model for its SSIM test. Deprecated names: `ZIMAGE_MODEL_REVISION`. |
|
||||
|
||||
Variables that FastVideo no longer reads; setting one logs a warning:
|
||||
|
||||
| Deprecated variable | Reason |
|
||||
| ----------------------------------------- | ---------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
| Deprecated variable | Reason |
|
||||
| ------------------------------------------------ | ---------------------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `DREAMX_WORLD_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_AR_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `DREAMX_WORLD_AR_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `FASTVIDEO_TEST_GAMECRAFT_MODEL_PATH` | HunyuanGameCraft was removed |
|
||||
| `GAMECRAFT_MODEL_PATH` | HunyuanGameCraft was removed |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_LOCAL_WEIGHTS_DIR` | GLM-Image was removed |
|
||||
| `GLM_IMAGE_LOCAL_WEIGHTS_DIR` | GLM-Image was removed |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_MODEL_DIR` | GLM-Image was removed |
|
||||
| `GLM_IMAGE_MODEL_DIR` | GLM-Image was removed |
|
||||
| `FASTVIDEO_TEST_LONGCAT_MODEL_ROOT` | LongCat was removed |
|
||||
| `LONGCAT_MODEL_ROOT` | LongCat was removed |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_DIR` | Z-Image was removed |
|
||||
| `ZIMAGE_MODEL_DIR` | Z-Image was removed |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_REVISION` | Z-Image was removed |
|
||||
| `ZIMAGE_MODEL_REVISION` | Z-Image was removed |
|
||||
<!-- END GENERATED ENV TABLE -->
|
||||
|
||||
@@ -112,7 +112,7 @@ per-metric policy with direction, percent threshold, absolute threshold, and a
|
||||
|
||||
`test_inference_performance.py` temporarily sets `FASTVIDEO_STAGE_LOGGING=1`
|
||||
while it runs so pipeline stage execution times are available in
|
||||
`generate(...).logging_info`. Stage logs use pipeline-unique keys such as
|
||||
`generate_video(...).logging_info`. Stage logs use pipeline-unique keys such as
|
||||
`prompt_encoding_stage` so duplicate stage classes do not collide. For
|
||||
`PipelineStage` entries, shared component stage bases emit a stable
|
||||
`component_metric`: text encoding stages map to `text_encoder_time_s`,
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Cosmos recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# FLUX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -1,140 +0,0 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# GLM-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/zai.webp" alt="Z.ai" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>GLM-Image inference recipes</h2>
|
||||
<p>GLM-Image from Z.ai supports both text-to-image generation and instruction-based image editing, each with a maintained example.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading GLM-Image recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/zai-org">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The editing example reads <code>assets/images/couple.jpg</code> from the repository root, so run it from a repo checkout rather than an arbitrary working directory.</li>
|
||||
<li>Output paths default under <code>image_output/</code>; pass <code>--output</code> to change them.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Hunyuan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
+7
-99
@@ -141,46 +141,25 @@ hide:
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./longcat/" aria-label="Open LongCat recipes">
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./kandinsky6/" aria-label="Open Kandinsky 6 recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/meituan-longcat.webp" alt="" width="132" height="132" loading="lazy">
|
||||
<img class="off-glb" src="../assets/logos/kandinsky.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>LongCat</strong><small>T2V, I2V, optional refine</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
<span><strong>Kandinsky 6</strong><small>Video with audio, and video SR</small></span>
|
||||
<span class="cookbook-count">4 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./turbodiffusion/" aria-label="Open TurboDiffusion recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">Turbo</span>
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>TurboDiffusion</strong><small>Accelerated Wan profiles</small></span>
|
||||
<span class="cookbook-count">3 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
<li>T2VA</li>
|
||||
<li>I2VA</li>
|
||||
<li>VSR</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
@@ -213,49 +192,6 @@ hide:
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./glm-image/" aria-label="Open GLM-Image recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/zai.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GLM-Image</strong><small>Generate and edit</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
<li>Edit</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./z-image/" aria-label="Open Z-Image recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tongyi.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Z-Image</strong><small>Turbo text to image</small></span>
|
||||
<span class="cookbook-count">1 recipe</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./stable-diffusion/" aria-label="Open Stable Diffusion recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
@@ -384,20 +320,6 @@ hide:
|
||||
<p>These families already have runnable examples. The cookbook page is not ready, so the cards are not links.</p>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="GameCraft cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tencent-hunyuan.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GameCraft</strong><small>Game world generation</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="GEN3C cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
@@ -426,20 +348,6 @@ hide:
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="DreamX cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/fastvideo.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>DreamX</strong><small>World generation</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="LingBot cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Kandinsky 5 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -3,19 +3,19 @@ hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# Z-Image recipes
|
||||
# Kandinsky 6 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="zimage" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky6" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/tongyi.webp" alt="Tongyi MAI" width="112" height="112">
|
||||
<img class="off-glb" src="../../assets/logos/kandinsky.webp" alt="Kandinsky Lab" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>Z-Image inference recipes</h2>
|
||||
<p>Z-Image Turbo from Tongyi MAI is a fast text-to-image model. The maintained example runs it on a single GPU.</p>
|
||||
<h2>Kandinsky 6 inference recipes</h2>
|
||||
<p>Kandinsky 6 from the Kandinsky Lab generates five-second video with synchronized audio from text or an image, with base and distilled pi-Flow checkpoints, and upscales existing clips with base or distilled video super-resolution.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
@@ -49,7 +49,7 @@ hide:
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading Z-Image recipes...</button>
|
||||
<button type="button" disabled>Loading Kandinsky 6 recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -97,7 +97,7 @@ hide:
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/Tongyi-MAI">View model card</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/kandinskylab">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
@@ -119,6 +119,7 @@ hide:
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
<p>For model-specific behavior, see <a href="../../inference/kandinsky6/">Kandinsky 6 video with audio</a> and <a href="../../inference/kandinsky6_sr/">Kandinsky 6 video super-resolution</a>.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -126,7 +127,10 @@ cd FastVideo</code></pre>
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Output defaults to <code>outputs/zimage/zimage_turbo.png</code>; pass <code>--output</code> to redirect.</li>
|
||||
<li>Text and image conditioning use the same TI2VA pipeline and example. Set <code>IMAGE_PATH</code> in the script to condition on an image; both base and distilled outputs include synchronized audio.</li>
|
||||
<li>For pi-Flow, set <code>KANDINSKY6_MODEL_PATH</code> to the distilled model ID when running <code>basic_kandinsky6_ti2va.py</code>. Its registered preset uses 10 inference steps, guidance 1.0, <code>eps=1e-6</code>, <code>final_step_size_scale=0.5</code>, and <code>num_policy_substeps=128</code>.</li>
|
||||
<li>For VSR, set <code>INPUT_VIDEO</code> in the generated command. The pipeline accepts x2, x2.25 and x4 scales, processes up to 121 frames at 24 fps, and preserves source audio.</li>
|
||||
<li>Checkpoint key-layout errors indicate that the checkpoint and FastVideo checkout target different Kandinsky 6 Diffusers revisions.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
@@ -1,140 +0,0 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# LongCat recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="longcat" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/meituan-longcat.webp" alt="Meituan LongCat" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>LongCat inference recipes</h2>
|
||||
<p>LongCat Video from Meituan covers text-to-video and image-to-video, and its maintained examples chain optional distilled and 720p refinement passes.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading LongCat recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/meituan-longcat">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Each LongCat script runs multiple passes (basic, distilled, refine); total runtime scales accordingly, and every pass prints its own output directory.</li>
|
||||
<li>Out of memory: the sources already enable VAE and text-encoder CPU offload; further options are covered in <a href="../../inference/offloading/">Offloading</a>.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# LTX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Matrix Game recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
+24
-11
@@ -10,8 +10,15 @@ launch. Some Hub repo names still say Preview. That name is historical. V1 is
|
||||
a full model, not a demo. **V2** is the eight-step checkpoint. More forwards
|
||||
is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
[FastH3 distilled checkpoint schedules](../inference/fasth3-distilled.md).
|
||||
**CompactH3** is the 42-block 20B NVFP4 H3 checkpoint for one Blackwell GPU
|
||||
(RTX 5090 or RTX PRO 6000). The FastH3 V1 and V2 recipes are unchanged.
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
The 42-block pruned checkpoint has an [MLX INT8/INT6 conversion and
|
||||
eight-forward T2VA command](../getting_started/installation/mlx.md#pruned-eight-forward-checkpoint).
|
||||
It reads `fastvideo_inference.json` for the trained schedule. The command
|
||||
uses native 832x480 resolution and all requested frames.
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -21,8 +28,8 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Primary focus · Inference</p>
|
||||
<h2>MiniMax H3 recipes</h2>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>9 maintained recipes</span>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA, one Blackwell GPU, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>11 maintained recipes</span>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
@@ -47,7 +54,7 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
<h2 id="h3-modes-heading">Supported modes</h2>
|
||||
<p>
|
||||
CUDA covers T2VA, FL2VA, and Ref2VA on the full checkpoint, plus FastH3
|
||||
V1 and FastH3 V2. FastH3 V1 also has a DGX Spark runtime with
|
||||
V1 and FastH3 V2, plus CompactH3 NVFP4 on one Blackwell GPU. FastH3 V1 also has a DGX Spark runtime with
|
||||
a 1-Spark or 2-Spark device row. MLX is T2VA only: V1 and V2.
|
||||
Temporal <code>--fast</code>, spatial <code>--fast-spatial</code>, and opt-in VSA are flags on the same
|
||||
MLX script, not extra recipes.
|
||||
@@ -64,7 +71,7 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
<tbody>
|
||||
<tr>
|
||||
<td>T2VA</td>
|
||||
<td>Full H3, FastH3 V1, FastH3 LoRA, FastH3 V2</td>
|
||||
<td>Full H3, FastH3 V1, FastH3 LoRA, FastH3 V2, CompactH3 NVFP4</td>
|
||||
<td>FastH3 V1 or FastH3 V2 after a local DiT conversion</td>
|
||||
</tr>
|
||||
<tr>
|
||||
@@ -102,6 +109,11 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
<td>FastH3 V1 on one GB10, or two Sparks with Ray sequence parallel (<code>sp_size=2</code>) over QSFP RoCE. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks.</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>CompactH3 NVFP4</td>
|
||||
<td>42-block 20B checkpoint on one RTX 5090 (32 GB, sequential encoder offload) or one RTX PRO 6000 Blackwell (96 GB, encoder+DiT+VAE resident). SageAttention3 FP4, packed NVFP4 DiT, Comfy int8-convrot VAE.</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
@@ -110,7 +122,7 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick an H3 recipe and runtime</h2>
|
||||
<p>Choose the result you want, then use a maintained CUDA, DGX Spark, or MLX path.
|
||||
<p>Choose the result you want, then use a maintained CUDA, Blackwell, DGX Spark, or MLX path.
|
||||
Device claims stay tied to checked-in sources and recorded runs.</p>
|
||||
</div>
|
||||
|
||||
@@ -154,7 +166,7 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
<span>Both can run locally</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" role="group" aria-label="How to run this recipe">
|
||||
<button type="button" data-cookbook-mode="server" aria-pressed="false"><strong>Run a server</strong><span>Playground, cURL, or an API client</span></button>
|
||||
<button type="button" data-cookbook-mode="server" aria-pressed="false"><strong>Run a server</strong><span data-cookbook-server-hint>Playground, cURL, or an API client</span></button>
|
||||
<button type="button" data-cookbook-mode="python" aria-pressed="false"><strong>Use Python directly</strong><span>Call the model in your own process</span></button>
|
||||
</div>
|
||||
</div>
|
||||
@@ -201,17 +213,17 @@ is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-start-heading">
|
||||
<h4 id="serving-start-heading"><span aria-hidden="true">2</span> Start the server</h4>
|
||||
<p>Keep this terminal running while you use the playground or API clients.</p>
|
||||
<p>Keep this terminal running while you use <span data-cookbook-playground-only>the playground or </span>API clients.</p>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>GPU machine · Terminal</span></div><pre id="cookbook-server-command"><code class="language-bash" data-cookbook-server-command></code></pre></div>
|
||||
<details class="cookbook-serving__check"><summary>Check that the server is ready</summary><p>In another terminal, this returns <code>{"status":"ok"}</code> after startup.</p><div class="cookbook-command"><pre id="cookbook-health-command"><code class="language-bash" data-cookbook-health-command></code></pre></div></details>
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-client-heading">
|
||||
<h4 id="serving-client-heading"><span aria-hidden="true">3</span> Generate and download a video</h4>
|
||||
<div class="cookbook-serving__playground">
|
||||
<div class="cookbook-serving__playground" data-cookbook-playground-only>
|
||||
<div><strong>Try prompts in your browser</strong><p>Edit a prompt, generate, and watch the result. The playground uses the same server as cURL and your app.</p></div>
|
||||
<a class="cookbook-serving__launch" data-cookbook-playground href="http://127.0.0.1:8000/playground/" target="_blank" rel="noopener">Open playground <span aria-hidden="true">↗</span></a>
|
||||
</div>
|
||||
<p class="cookbook-serving__local-hint">Open after the server is ready. On a remote GPU machine, <a href="../openai-api/#connect-your-app">forward port 8000</a> to your computer first. This opens a local page, not a hosted demo.</p>
|
||||
<p class="cookbook-serving__local-hint"><span data-cookbook-playground-only>Open after the server is ready. </span>On a remote GPU machine, <a href="../openai-api/#connect-your-app">forward port 8000</a> to your computer first.<span data-cookbook-playground-only> This opens a local page, not a hosted demo.</span></p>
|
||||
<details class="cookbook-serving__code"><summary>Use cURL or an SDK</summary>
|
||||
<p>Each example submits a job, checks its status, and saves the MP4. The Python and JavaScript examples use OpenAI-compatible clients; no OpenAI account is needed.</p>
|
||||
<div class="cookbook-serving__clients" role="group" aria-label="API client language">
|
||||
@@ -275,7 +287,8 @@ cd FastVideo</code></pre>
|
||||
<li>The MLX source runtime supports T2VA, optional temporal <code>--fast</code>, optional spatial <code>--fast-spatial</code>, and opt-in VSA on <code>--include-vsa</code> checkpoints. FastH3 V2 MLX converts with <code>--include-vsa</code> and runs eight forwards. FL2VA, Ref2VA, and two-pass refinement are not wired.</li>
|
||||
<li>GPU count and VAE decode backend are configurable in the builder above for FastH3 CUDA recipes. Only the value shown by default has a recorded run; other supported values are unmeasured here.</li>
|
||||
<li>DGX Spark is a runtime on FastH3 V1, not a separate family card. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks. The CUDA GPU-count knob does not apply to Spark.</li>
|
||||
<li>GB10 has no FA4 / sm_100a VSA kernel. Keep <code>FASTVIDEO_FA4=0</code> and <code>FASTVIDEO_VSA_SM100A=0</code>. Legal <code>num_frames</code> values are <code>17n+5</code>, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM.</li>
|
||||
<li>CompactH3 NVFP4 is one Blackwell GPU. RTX 5090 (32 GB) parks the encoder in pinned host RAM. RTX PRO 6000 Blackwell (96 GB) keeps encoder, DiT, and VAE resident. Keep <code>FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER</code>, <code>FASTVIDEO_FA4=0</code>, <code>FASTVIDEO_VSA_SM100A=0</code>, and <code>FLASHINFER_CUDA_ARCH_LIST=12.0a</code>. On PRO 6000 enable VAE compile and leave DiT <code>inference_torch_compile</code> off. CompactH3 is a dense prune; do not enable <code>VIDEO_SPARSE_ATTN_H3</code> until a VSA-trained student exists.</li>
|
||||
<li>GB10 has no FA4 / sm_100a VSA kernel. Keep <code>FASTVIDEO_FA4=0</code> and <code>FASTVIDEO_VSA_SM100A=0</code>. Legal <code>num_frames</code> values are <code>17n+5</code>, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM. Native 16:9 sizes include 832×480 and 1344×768.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# MMAudio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Audio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Diffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -1,140 +0,0 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# TurboDiffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="turbodiffusion" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">Turbo</span>
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>TurboDiffusion inference recipes</h2>
|
||||
<p>TurboDiffusion profiles accelerate Wan checkpoints with step-distilled sampling and the SLA attention backend. These recipes follow the registry's `turbodiffusion` model family.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading TurboDiffusion recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/loayrashid">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>TurboDiffusion paths load community-published <code>loayrashid/TurboWan*</code> checkpoints; availability is governed by those repos.</li>
|
||||
<li>The SLA attention backend used by the I2V recipe is selected inside the example source; do not combine it with another <code>FASTVIDEO_ATTENTION_BACKEND</code> override in the same shell.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
+51
-2
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Wan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -122,6 +122,17 @@ hide:
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<div class="cookbook-selection-row" data-cookbook-usage>
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Workflow</strong>
|
||||
<span>Both can run locally</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" role="group" aria-label="How to run this recipe">
|
||||
<button type="button" data-cookbook-mode="server" aria-pressed="false"><strong>Run a server</strong><span data-cookbook-server-hint>Playground, cURL, or an API client</span></button>
|
||||
<button type="button" data-cookbook-mode="python" aria-pressed="false"><strong>Use Python directly</strong><span>Call the model in your own process</span></button>
|
||||
</div>
|
||||
</div>
|
||||
<p class="cookbook-hardware-note" data-cookbook-serving-availability></p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
@@ -150,7 +161,45 @@ hide:
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
<pre id="cookbook-local-command"><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
<p class="cookbook-hardware-note" data-cookbook-python-note>Running this script again starts a new process and reloads the model. To iterate in Python, create the generator once and reuse it for multiple prompts.</p>
|
||||
|
||||
<div class="cookbook-serving" data-cookbook-serving hidden>
|
||||
<p class="cookbook-serving__intro" data-cookbook-server-lifetime>Start once, then change prompts in the playground or your app. You can run the server and clients on the same machine.</p>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-install-heading">
|
||||
<h4 id="serving-install-heading"><span aria-hidden="true">1</span> Prepare the machine</h4>
|
||||
<p>Run from your FastVideo clone in an activated Python environment. See <a data-cookbook-install-guide href="../../getting_started/installation/gpu/">installation requirements</a>.</p>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>GPU machine · Terminal</span></div><pre id="cookbook-server-install"><code class="language-bash" data-cookbook-server-install></code></pre></div>
|
||||
<details class="cookbook-serving__prepare" data-cookbook-prepare hidden><summary>Download and convert MLX weights once</summary><p>Skip this if the weights are already prepared. Edit the paths in the serving config to use your existing files.</p><div class="cookbook-command"><pre id="cookbook-server-prepare"><code class="language-bash" data-cookbook-server-prepare></code></pre></div></details>
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-start-heading">
|
||||
<h4 id="serving-start-heading"><span aria-hidden="true">2</span> Start the server</h4>
|
||||
<p>Keep this terminal running while you use <span data-cookbook-playground-only>the playground or </span>API clients.</p>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>GPU machine · Terminal</span></div><pre id="cookbook-server-command"><code class="language-bash" data-cookbook-server-command></code></pre></div>
|
||||
<details class="cookbook-serving__check"><summary>Check that the server is ready</summary><p>In another terminal, this returns <code>{"status":"ok"}</code> after startup.</p><div class="cookbook-command"><pre id="cookbook-health-command"><code class="language-bash" data-cookbook-health-command></code></pre></div></details>
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-client-heading">
|
||||
<h4 id="serving-client-heading"><span aria-hidden="true">3</span> Generate and download a video</h4>
|
||||
<div class="cookbook-serving__playground" data-cookbook-playground-only>
|
||||
<div><strong>Try prompts in your browser</strong><p>Edit a prompt, generate, and watch the result. The playground uses the same server as cURL and your app.</p></div>
|
||||
<a class="cookbook-serving__launch" data-cookbook-playground href="http://127.0.0.1:8000/playground/" target="_blank" rel="noopener">Open playground <span aria-hidden="true">↗</span></a>
|
||||
</div>
|
||||
<p class="cookbook-serving__local-hint"><span data-cookbook-playground-only>Open after the server is ready. </span>On a remote GPU machine, <a href="../openai-api/#connect-your-app">forward port 8000</a> to your computer first.<span data-cookbook-playground-only> This opens a local page, not a hosted demo.</span></p>
|
||||
<details class="cookbook-serving__code"><summary>Use cURL or an SDK</summary>
|
||||
<p>Each example submits a job, checks its status, and saves the MP4. The Python and JavaScript examples use OpenAI-compatible clients; no OpenAI account is needed.</p>
|
||||
<div class="cookbook-serving__clients" role="group" aria-label="API client language">
|
||||
<button type="button" data-cookbook-client="curl" aria-pressed="false">cURL</button>
|
||||
<button type="button" data-cookbook-client="python" aria-pressed="true">Python</button>
|
||||
<button type="button" data-cookbook-client="javascript" aria-pressed="false">JavaScript</button>
|
||||
</div>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>Client dependencies</span></div><pre id="cookbook-client-install"><code class="language-bash" data-cookbook-client-install></code></pre></div>
|
||||
<div class="cookbook-command cookbook-command--client"><div class="cookbook-command__bar"><span data-cookbook-client-filename>video.py</span><a data-cookbook-client-source href="https://github.com/hao-ai-lab/FastVideo/tree/main/examples/serving/clients">View source</a></div><pre id="cookbook-client-code"><code data-cookbook-client-code></code></pre></div>
|
||||
<p data-cookbook-client-run></p>
|
||||
</details>
|
||||
</section>
|
||||
<p class="cookbook-serving__boundary">This is a local development server without built-in API-key authentication. The client key <code>local</code> is a placeholder. Keep the server on loopback; use an authenticated TLS proxy before exposing it publicly. Run the JavaScript client in your webapp's backend, not in a browser with a private key.</p>
|
||||
<a href="../openai-api/">Server guide and API compatibility →</a>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
|
||||
@@ -2,158 +2,136 @@ status_definitions:
|
||||
kept: "Public field remains on a public adapter surface with the same meaning."
|
||||
moved: "Public field remains supported but normalizes into a different nested path."
|
||||
preset_owned: "Public field remains supported only through a model/preset-specific surface."
|
||||
compatibility_only: "Public field that an adapter or an open mapping accepts outside the typed fields of the canonical schema."
|
||||
compatibility_only: "Legacy public field remains adapter-only during migration and is not part of the canonical typed schema."
|
||||
private_only: "Field should only be handled by private adapters and is not a public FastVideo compatibility promise."
|
||||
internal_only: "Field is runtime/config plumbing that model code, config resolution, or the runtime fills; it is not a public input."
|
||||
unsupported: "Typed config field that no runtime code reads; resolution rejects a value."
|
||||
internal_only: "Field is runtime/config plumbing and should not be part of the new public typed inference API."
|
||||
|
||||
surfaces:
|
||||
generator_config:
|
||||
kept:
|
||||
- model_path
|
||||
- mode
|
||||
- revision
|
||||
- trust_remote_code
|
||||
- engine.num_gpus
|
||||
- engine.execution_backend
|
||||
- engine.parallelism.tp_size
|
||||
- engine.parallelism.sp_size
|
||||
- engine.parallelism.hsdp_replicate_dim
|
||||
- engine.parallelism.hsdp_shard_dim
|
||||
- engine.parallelism.dist_timeout
|
||||
- engine.offload.dit
|
||||
- engine.offload.dit_layerwise
|
||||
- engine.offload.text_encoder
|
||||
- engine.offload.image_encoder
|
||||
- engine.offload.vae
|
||||
- engine.offload.pin_cpu_memory
|
||||
- engine.compile.enabled
|
||||
- engine.compile.backend
|
||||
- engine.compile.fullgraph
|
||||
- engine.compile.mode
|
||||
- engine.compile.dynamic
|
||||
- engine.compile.extras
|
||||
- engine.enable_stage_verification
|
||||
- engine.use_fsdp_inference
|
||||
- engine.disable_autocast
|
||||
- engine.attention.nvfp4_fa4
|
||||
- pipeline.components.lora_path
|
||||
- pipeline.components.lora_strength
|
||||
- pipeline.output_type
|
||||
- engine.parallelism.master_port
|
||||
- engine.offload.lazy_module_load
|
||||
- engine.compile.text_encoder_enabled
|
||||
- engine.compile.vae_enabled
|
||||
- engine.compile.audio_vae_enabled
|
||||
- engine.compile.regional
|
||||
- engine.compile.dit_kwargs
|
||||
- engine.compile.text_encoder_kwargs
|
||||
- engine.compile.vae_kwargs
|
||||
- engine.compile.audio_vae_kwargs
|
||||
- engine.attention.backend
|
||||
- engine.attention.vsa_sparsity
|
||||
- engine.attention.vsa_tile_size
|
||||
- engine.attention.moba_config_path
|
||||
- engine.precision.dit
|
||||
- engine.precision.vae
|
||||
- engine.precision.vae_decode
|
||||
- engine.precision.image_encoder
|
||||
- engine.precision.text_encoders
|
||||
- engine.quantization.text_encoder_quant
|
||||
- engine.quantization.transformer_quant
|
||||
- pipeline.workload_type
|
||||
- pipeline.components.config_root
|
||||
- pipeline.components.pipeline_config_path
|
||||
- pipeline.components.text_encoder_weights
|
||||
- pipeline.components.transformer_weights
|
||||
- pipeline.components.transformer_2_weights
|
||||
- pipeline.components.upsampler_weights
|
||||
- pipeline.components.lora_nickname
|
||||
- pipeline.components.lora_target_modules
|
||||
- pipeline.components.override_pipeline_cls_name
|
||||
- pipeline.components.override_transformer_cls_name
|
||||
- pipeline.vae_tiling
|
||||
- pipeline.vae_sp
|
||||
- pipeline.flow_shift
|
||||
- pipeline.embedded_cfg_scale
|
||||
- pipeline.dmd_denoising_steps
|
||||
- pipeline.boundary_ratio
|
||||
- pipeline.model.generic.dit
|
||||
- pipeline.model.generic.vae
|
||||
- pipeline.model.ltx2.dit
|
||||
- pipeline.model.ltx2.vae
|
||||
- pipeline.model.minimax_h3.dit
|
||||
- pipeline.model.minimax_h3.vae
|
||||
- pipeline.model.longcat.dit
|
||||
- pipeline.model.longcat.vae
|
||||
unsupported:
|
||||
- pipeline.preset
|
||||
- pipeline.preset_version
|
||||
- pipeline.components.vae_weights
|
||||
fastvideo_args:
|
||||
moved:
|
||||
pipeline.preset_overrides:
|
||||
target: generator.pipeline.model.ltx2.refine
|
||||
note: "Only the refine mapping applies: resolution copies pipeline.preset_overrides.refine into the pipeline.model.ltx2.refine fields of the same names when the model is LTX-2. Other keys have no effect."
|
||||
model_path: generator.model_path
|
||||
workload_type: generator.pipeline.workload_type
|
||||
distributed_executor_backend: generator.engine.execution_backend
|
||||
trust_remote_code: generator.trust_remote_code
|
||||
revision: generator.revision
|
||||
num_gpus: generator.engine.num_gpus
|
||||
tp_size: generator.engine.parallelism.tp_size
|
||||
sp_size: generator.engine.parallelism.sp_size
|
||||
hsdp_replicate_dim: generator.engine.parallelism.hsdp_replicate_dim
|
||||
hsdp_shard_dim: generator.engine.parallelism.hsdp_shard_dim
|
||||
dist_timeout: generator.engine.parallelism.dist_timeout
|
||||
lora_path: generator.pipeline.components.lora_path
|
||||
lora_nickname: generator.pipeline.components.lora_nickname
|
||||
lora_strength: generator.pipeline.components.lora_strength
|
||||
dit_cpu_offload: generator.engine.offload.dit
|
||||
use_fsdp_inference: generator.engine.use_fsdp_inference
|
||||
dit_layerwise_offload: generator.engine.offload.dit_layerwise
|
||||
text_encoder_cpu_offload: generator.engine.offload.text_encoder
|
||||
image_encoder_cpu_offload: generator.engine.offload.image_encoder
|
||||
vae_cpu_offload: generator.engine.offload.vae
|
||||
pin_cpu_memory: generator.engine.offload.pin_cpu_memory
|
||||
lazy_module_load: generator.engine.offload.lazy_module_load
|
||||
enable_torch_compile: generator.engine.compile.enabled
|
||||
enable_torch_compile_text_encoder: generator.engine.compile.text_encoder_enabled
|
||||
enable_torch_compile_vae: generator.engine.compile.vae_enabled
|
||||
enable_torch_compile_audio_vae: generator.engine.compile.audio_vae_enabled
|
||||
torch_compile_kwargs: generator.engine.compile.backend,fullgraph,mode,dynamic,extras
|
||||
torch_compile_kwargs_dit: generator.engine.compile.dit_kwargs
|
||||
torch_compile_kwargs_text_encoder: generator.engine.compile.text_encoder_kwargs
|
||||
torch_compile_kwargs_vae: generator.engine.compile.vae_kwargs
|
||||
torch_compile_kwargs_audio_vae: generator.engine.compile.audio_vae_kwargs
|
||||
transformer_quant: generator.engine.quantization.transformer_quant
|
||||
disable_autocast: generator.engine.disable_autocast
|
||||
enable_stage_verification: generator.engine.enable_stage_verification
|
||||
prompt_txt: request.inputs.prompt_path
|
||||
override_text_encoder_safetensors: generator.pipeline.components.text_encoder_weights
|
||||
override_text_encoder_quant: generator.engine.quantization.text_encoder_quant
|
||||
transformer_quant: generator.engine.quantization.transformer_quant
|
||||
override_transformer_cls_name: generator.pipeline.components.override_transformer_cls_name
|
||||
init_weights_from_safetensors: generator.pipeline.components.transformer_weights
|
||||
init_weights_from_safetensors_2: generator.pipeline.components.transformer_2_weights
|
||||
override_pipeline_cls_name: generator.pipeline.components.override_pipeline_cls_name
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
ltx2_vae_tiling: generator.pipeline.vae_tiling
|
||||
refine_enabled: generator.pipeline.preset_overrides.refine.enabled
|
||||
refine_upsampler_path: generator.pipeline.components.upsampler_weights
|
||||
refine_lora_path: generator.pipeline.components.lora_path
|
||||
refine_num_inference_steps: request.stage_overrides.refine.num_inference_steps
|
||||
refine_guidance_scale: request.stage_overrides.refine.guidance_scale
|
||||
refine_add_noise: generator.pipeline.preset_overrides.refine.add_noise
|
||||
ltx2_refine_enabled: generator.pipeline.preset_overrides.refine.enabled
|
||||
ltx2_refine_upsampler_path: generator.pipeline.components.upsampler_weights
|
||||
ltx2_refine_lora_path: generator.pipeline.components.lora_path
|
||||
ltx2_refine_num_inference_steps: request.stage_overrides.refine.num_inference_steps
|
||||
ltx2_refine_guidance_scale: request.stage_overrides.refine.guidance_scale
|
||||
ltx2_refine_add_noise: generator.pipeline.preset_overrides.refine.add_noise
|
||||
preset_owned:
|
||||
- pipeline.model.ltx2.vae_spatial_tile_size_in_pixels
|
||||
- pipeline.model.ltx2.vae_spatial_tile_overlap_in_pixels
|
||||
- pipeline.model.ltx2.vae_temporal_tile_size_in_frames
|
||||
- pipeline.model.ltx2.vae_temporal_tile_overlap_in_frames
|
||||
- pipeline.model.ltx2.initial_latent_path
|
||||
- pipeline.model.ltx2.audio_latent_path
|
||||
- pipeline.model.ltx2.legacy_native_noise_order
|
||||
- pipeline.model.ltx2.use_distilled_sigmas
|
||||
- pipeline.model.ltx2.refine.enabled
|
||||
- pipeline.model.ltx2.refine.num_inference_steps
|
||||
- pipeline.model.ltx2.refine.guidance_scale
|
||||
- pipeline.model.ltx2.refine.add_noise
|
||||
- pipeline.model.ltx2.refine.image_crf
|
||||
- pipeline.model.ltx2.refine.video_position_offset_sec
|
||||
- pipeline.model.ltx2.refine.transformer_path
|
||||
- pipeline.model.ltx2.refine.lora_path
|
||||
- pipeline.model.ltx2.refine.noise_path
|
||||
- pipeline.model.ltx2.refine.audio_noise_path
|
||||
- pipeline.model.minimax_h3.sequential_load
|
||||
- pipeline.model.minimax_h3.video_decode_backend
|
||||
- pipeline.model.minimax_h3.taeh3_checkpoint
|
||||
- pipeline.model.minimax_h3.taeh3_chunk_size
|
||||
- pipeline.model.minimax_h3.vae_parallel_decode
|
||||
- pipeline.model.minimax_h3.vae_parallel_encode
|
||||
- pipeline.model.minimax_h3.vae_parallel_decode_strategy
|
||||
- pipeline.model.longcat.enable_bsa
|
||||
- pipeline.model.longcat.bsa_sparsity
|
||||
- pipeline.model.longcat.bsa_cdf_threshold
|
||||
- pipeline.model.longcat.bsa_chunk_q
|
||||
- pipeline.model.longcat.bsa_chunk_k
|
||||
ltx2_vae_spatial_tile_size_in_pixels: generator.pipeline.preset_overrides.ltx2.vae.spatial_tile_size_in_pixels
|
||||
ltx2_vae_spatial_tile_overlap_in_pixels: generator.pipeline.preset_overrides.ltx2.vae.spatial_tile_overlap_in_pixels
|
||||
ltx2_vae_temporal_tile_size_in_frames: generator.pipeline.preset_overrides.ltx2.vae.temporal_tile_size_in_frames
|
||||
ltx2_vae_temporal_tile_overlap_in_frames: generator.pipeline.preset_overrides.ltx2.vae.temporal_tile_overlap_in_frames
|
||||
ltx2_initial_latent_path: request.extensions.ltx2.initial_latent_path
|
||||
ltx2_audio_latent_path: request.extensions.ltx2.audio_latent_path
|
||||
compatibility_only:
|
||||
pipeline.experimental: "Open mapping for settings without a typed path: the pipeline_config source (a JSON path, a mapping, or a PipelineConfig), keys that runtime code reads by name (for example ray_runtime_env), and PipelineConfig attribute overrides for model-only fields (for example flow_shift_sr). Resolution rejects a key whose PipelineConfig attribute has a typed path."
|
||||
mode: "Legacy multi-mode FastVideoArgs switch; typed inference config should not expose execution mode."
|
||||
inference_mode: "Legacy boolean mirror of mode; kept only through adapters while FastVideoArgs remains."
|
||||
lora_target_modules: "Legacy LoRA configuration surface pending dedicated component API."
|
||||
output_type: "Legacy output formatting surface pending GenerationResult cleanup."
|
||||
VSA_sparsity: "Model-specific inference optimization not yet represented in the typed public schema."
|
||||
VSA_tile_size: "VSA-H3 tile geometry request; model-specific optimization not yet represented in the typed public schema."
|
||||
inference_torch_compile: "Regional inference compile opt-in currently carried through PipelineSelection.experimental rather than CompileConfig."
|
||||
vae_parallel_decode: "MiniMax-H3 sequence-parallel VAE decode opt-in; model-specific optimization not yet represented in the typed public schema."
|
||||
h3_sequential_load: "MiniMax-H3 sequential text-encoder then DiT/VAE load; model-specific optimization not yet represented in the typed public schema."
|
||||
video_decode_backend: "MiniMax-H3 video decoder selection (full VAE vs TAEH3 preview); model-specific optimization not yet represented in the typed public schema."
|
||||
taeh3_checkpoint: "Optional local TAEH3 safetensors path; model-specific optimization not yet represented in the typed public schema."
|
||||
taeh3_chunk_size: "TAEH3 temporal chunk length; model-specific optimization not yet represented in the typed public schema."
|
||||
vae_parallel_encode: "MiniMax-H3 sequence-parallel reference VAE encode opt-in; model-specific optimization not yet represented in the typed public schema."
|
||||
vae_parallel_decode_strategy: "Chunk-transport collective for vae_parallel_decode; model-specific optimization not yet represented in the typed public schema."
|
||||
attention_backend: "Process-wide default attention-backend request applied per component at load time; kernel-selection knob not yet represented in the typed public schema."
|
||||
moba_config_path: "Model-specific MoBA optimization surface not yet represented in the typed public schema."
|
||||
master_port: "Executor/bootstrap compatibility field; not part of the canonical inference schema."
|
||||
refine_transformer_path: "Generic stage-2 refine transformer override; no typed equivalent yet."
|
||||
refine_noise_path: "Generic stage-2 refine noise override; no typed equivalent yet."
|
||||
refine_audio_noise_path: "Generic stage-2 refine audio noise override; no typed equivalent yet."
|
||||
ltx2_refine_transformer_path: "LTX-2 refine transformer carrier; no typed equivalent yet."
|
||||
ltx2_refine_noise_path: "LTX-2 refine noise carrier; no typed equivalent yet."
|
||||
ltx2_refine_audio_noise_path: "LTX-2 refine audio noise carrier; no typed equivalent yet."
|
||||
ltx2_legacy_native_noise_order: "LTX-2 SSIM compatibility knob preserving legacy native latent noise ordering."
|
||||
ltx2_use_distilled_sigmas: "LTX-2 compatibility knob gating use of distilled sigma schedule."
|
||||
private_only:
|
||||
ray_placement_group: "Ray deployment-only field."
|
||||
ray_runtime_env: "Ray deployment-only field."
|
||||
internal_only:
|
||||
engine.attention.moba_config: "V-MoBA attention settings that resolution loads from engine.attention.moba_config_path."
|
||||
pipeline_config: "Legacy internal carrier object."
|
||||
preprocess_config: "Legacy preprocess carrier object."
|
||||
moba_config: "Derived runtime config loaded from moba_config_path."
|
||||
model_paths: "Runtime bookkeeping."
|
||||
model_loaded: "Runtime bookkeeping."
|
||||
|
||||
pipeline_config_base:
|
||||
moved:
|
||||
model_path: generator.model_path
|
||||
pipeline_config_path: generator.pipeline.components.pipeline_config_path
|
||||
embedded_cfg_scale: generator.pipeline.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.flow_shift
|
||||
disable_autocast: generator.engine.disable_autocast
|
||||
vae_tiling: generator.pipeline.vae_tiling
|
||||
vae_sp: generator.pipeline.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.dmd_denoising_steps
|
||||
boundary_ratio: generator.pipeline.boundary_ratio
|
||||
dit_precision: generator.engine.precision.dit
|
||||
vae_precision: generator.engine.precision.vae
|
||||
vae_decode_precision: generator.engine.precision.vae_decode
|
||||
image_encoder_precision: generator.engine.precision.image_encoder
|
||||
text_encoder_precisions: generator.engine.precision.text_encoders
|
||||
preset_owned:
|
||||
flow_shift_sr: generator.pipeline.experimental.flow_shift_sr
|
||||
is_causal: generator.pipeline.experimental.is_causal
|
||||
ti2v_task: generator.pipeline.experimental.ti2v_task
|
||||
lucy_edit_task: generator.pipeline.experimental.lucy_edit_task
|
||||
embedded_cfg_scale: generator.pipeline.preset_overrides.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.preset_overrides.flow_shift
|
||||
flow_shift_sr: generator.pipeline.preset_overrides.flow_shift_sr
|
||||
is_causal: generator.pipeline.preset_overrides.is_causal
|
||||
vae_tiling: generator.pipeline.preset_overrides.vae_tiling
|
||||
vae_sp: generator.pipeline.preset_overrides.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.preset_overrides.dmd_denoising_steps
|
||||
ti2v_task: generator.pipeline.preset_overrides.ti2v_task
|
||||
lucy_edit_task: generator.pipeline.preset_overrides.lucy_edit_task
|
||||
boundary_ratio: generator.pipeline.preset_overrides.boundary_ratio
|
||||
compatibility_only:
|
||||
model_path: "Redundant with generator.model_path."
|
||||
disable_autocast: "Duplicated by generator.engine.disable_autocast during migration."
|
||||
dit_precision: "Precision override pending dedicated typed component precision design."
|
||||
upsampler_precision: "Precision override pending dedicated typed component precision design."
|
||||
vae_precision: "Precision override pending dedicated typed component precision design."
|
||||
vae_decode_precision: "Decode-only precision override pending dedicated typed component precision design."
|
||||
image_encoder_precision: "Precision override pending dedicated typed component precision design."
|
||||
image_encoder_precisions: "Precision overrides pending dedicated typed component precision design."
|
||||
text_encoder_precisions: "Precision override pending dedicated typed component precision design."
|
||||
internal_only:
|
||||
dit_config: "Legacy internal component config object."
|
||||
upsampler_config: "Legacy internal component config object."
|
||||
@@ -166,22 +144,6 @@ surfaces:
|
||||
scheduler_step_in_fp32: "Runtime scheduler precision toggle; not part of the public typed inference API."
|
||||
|
||||
pipeline_config_extensions:
|
||||
moved:
|
||||
enable_bsa:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.enable_bsa
|
||||
bsa_sparsity:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_sparsity
|
||||
bsa_cdf_threshold:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_cdf_threshold
|
||||
bsa_chunk_q:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_chunk_q
|
||||
bsa_chunk_k:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_chunk_k
|
||||
preset_owned:
|
||||
flux2_text_encoder_type:
|
||||
sources:
|
||||
@@ -243,9 +205,6 @@ surfaces:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
color_correction_strength:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.dreamx_world.DreamXWorld5BARPipelineConfig
|
||||
default_camera_rotation:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
@@ -320,11 +279,6 @@ surfaces:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
|
||||
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.matrixgame2.MatrixGame2BaseI2V480PConfig
|
||||
@@ -343,11 +297,6 @@ surfaces:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
|
||||
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.matrixgame2.MatrixGame2BaseI2V480PConfig
|
||||
@@ -362,38 +311,21 @@ surfaces:
|
||||
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
|
||||
bsa_params:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enhance_hf:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
offload_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
t_thresh:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
use_distill:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
scheduler_arch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
text_encoder_archs:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
tokenizer_archs:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
transformer_arch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
vae_arch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
expand_timesteps:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
@@ -486,6 +418,14 @@ surfaces:
|
||||
sources: [fastvideo.pipelines.basic.magi_human.pipeline_configs.MagiHumanBaseConfig]
|
||||
video_txt_guidance_scale:
|
||||
sources: [fastvideo.pipelines.basic.magi_human.pipeline_configs.MagiHumanBaseConfig]
|
||||
piflow_eps:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
piflow_final_step_size_scale:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
piflow_num_policy_substeps:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
sample_fps:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
compatibility_only:
|
||||
batch_size: "Gen3C inference-only tuning field pending typed batching design."
|
||||
gradient_checkpointing: "Gen3C inference-only compatibility field pending typed batching design."
|
||||
@@ -498,15 +438,17 @@ surfaces:
|
||||
coords_style: "MagiHuman internal data-proxy coordinate convention."
|
||||
frame_receptive_field: "MagiHuman internal data-proxy receptive-field setting."
|
||||
image_conditioning: "MagiHuman preset variant marker for reference-image conditioning."
|
||||
pdd_step_indices: "MiniMax-H3 PDD fine-grid partition; resolve_checkpoint_settings sets it from fastvideo_inference.json."
|
||||
ref_audio_offset: "MagiHuman internal data-proxy audio alignment offset."
|
||||
scheduler_sigma_min: "Z-Image scheduler parity invariant; not part of the public typed inference API."
|
||||
scheduler_use_reference_discrete_timesteps: "Z-Image scheduler parity invariant; not part of the public typed inference API."
|
||||
sr_local_attn_layers: "MagiHuman SR internal sparse-attention layer selection."
|
||||
text_offset: "MagiHuman internal data-proxy text alignment offset."
|
||||
vae_stride: "MagiHuman internal VAE/data-proxy stride setting."
|
||||
vsa_ref_keep_rate: "MiniMax-H3 PDD reference-video VSA keep rate; defaults to the checkpoint's trained value."
|
||||
z_dim: "MagiHuman internal VAE latent channel setting."
|
||||
vocoder_config: "Legacy internal component config object."
|
||||
vocoder_precision: "Precision override pending dedicated component precision design."
|
||||
audio_sample_rate: "Kandinsky6 audio-latent length constant (audio VAE sample rate); tied to the checkpoint, not a request knob."
|
||||
audio_downsample_factor: "Kandinsky6 audio-latent length constant (audio VAE downsample factor); tied to the checkpoint, not a request knob."
|
||||
|
||||
sampling_param_base:
|
||||
moved:
|
||||
@@ -523,8 +465,6 @@ surfaces:
|
||||
pose: request.inputs.pose
|
||||
c2ws_plucker_emb: request.inputs.c2ws_plucker_emb
|
||||
action_path: request.inputs.action_path
|
||||
refine_from: request.inputs.refine_from
|
||||
stage1_video: request.inputs.stage1_video
|
||||
prompt: request.prompt
|
||||
negative_prompt: request.negative_prompt
|
||||
prompt_path: request.inputs.prompt_path
|
||||
@@ -544,11 +484,8 @@ surfaces:
|
||||
guidance_scale: request.sampling.guidance_scale
|
||||
batch_cfg: request.sampling.batch_cfg
|
||||
guidance_scale_2: request.sampling.guidance_scale_2
|
||||
cfg_normalization: request.sampling.cfg_normalization
|
||||
cfg_truncation: request.sampling.cfg_truncation
|
||||
guidance_rescale: request.sampling.guidance_rescale
|
||||
use_embedded_guidance: request.sampling.use_embedded_guidance
|
||||
embedded_cfg_scale: request.sampling.embedded_cfg_scale
|
||||
true_cfg_scale: request.sampling.true_cfg_scale
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
sigmas: request.sampling.sigmas
|
||||
@@ -561,19 +498,12 @@ surfaces:
|
||||
return_continuation_state: request.output.return_state
|
||||
preset_owned:
|
||||
t_thresh: request.stage_overrides.refine.t_thresh
|
||||
spatial_refine_only: request.stage_overrides.refine.spatial_refine_only
|
||||
num_cond_frames: request.stage_overrides.refine.num_cond_frames
|
||||
trajectory_type: request.extensions.gen3c.trajectory_type
|
||||
movement_distance: request.extensions.gen3c.movement_distance
|
||||
camera_rotation: request.extensions.gen3c.camera_rotation
|
||||
prompt_attention_mask: request.extensions.hyworld.prompt_attention_mask
|
||||
negative_attention_mask: request.extensions.hyworld.negative_attention_mask
|
||||
camera_states: request.extensions.hunyuangamecraft.camera_states
|
||||
camera_trajectory: request.extensions.hunyuangamecraft.camera_trajectory
|
||||
action_list: request.extensions.hunyuangamecraft.action_list
|
||||
action_speed_list: request.extensions.hunyuangamecraft.action_speed_list
|
||||
gt_latents: request.extensions.hunyuangamecraft.gt_latents
|
||||
conditioning_mask: request.extensions.hunyuangamecraft.conditioning_mask
|
||||
ltx2_cfg_scale_video: request.extensions.ltx2.cfg_scale_video
|
||||
ltx2_cfg_scale_audio: request.extensions.ltx2.cfg_scale_audio
|
||||
ltx2_modality_scale_video: request.extensions.ltx2.modality_scale_video
|
||||
|
||||
+12
-11
@@ -28,15 +28,19 @@ Minimal usage (from `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, {"engine": {"num_gpus": 1}})
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
|
||||
video = generator.generate({
|
||||
"prompt": "A vibrant city street at sunset.",
|
||||
"sampling": {"num_frames": 45},
|
||||
"output": {"output_path": "video_samples", "save_video": True},
|
||||
})
|
||||
sampling = SamplingParam.from_pretrained(model_id)
|
||||
sampling.num_frames = 45
|
||||
video = generator.generate_video(
|
||||
"A vibrant city street at sunset.",
|
||||
sampling_param=sampling,
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
```
|
||||
|
||||
## Configuration system
|
||||
@@ -70,11 +74,8 @@ not override checkpoint manifests, user pipeline overrides, or component
|
||||
precision settings. HF IDs, local checkpoints, and old config imports retain
|
||||
their existing resolution behavior, including first-match detector ordering.
|
||||
|
||||
`ResolvedGeneratorConfig` (in `fastvideo/api/resolution.py`) provides runtime
|
||||
settings and is passed into pipeline construction and stages as `resolved_config`.
|
||||
`resolve_inference_config` (in `fastvideo/api/inference_resolution.py`) builds it
|
||||
from the typed config in `fastvideo/api/schema.py` and attaches the model's
|
||||
frozen `PipelineConfig` as `resolved_config.pipeline_config`.
|
||||
`FastVideoArgs` (in `fastvideo/fastvideo_args.py`) provides runtime settings and
|
||||
is passed into pipeline construction and stages.
|
||||
|
||||
## Weights and Diffusers format
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ from fastvideo.api import (
|
||||
|
||||
| Surface | Availability | Notes |
|
||||
| --- | --- | --- |
|
||||
| `VideoGenerator.from_pretrained(model_path, config)` | Today | `config` is a nested `GeneratorConfig` mapping without `model_path`; no flat keywords |
|
||||
| `VideoGenerator.from_pretrained(model_path, **typed_kwargs)` | Today | `typed_kwargs` is a stable subset from `GeneratorConfig` — no flat legacy LTX-2 kwargs (guaranteed after PR 6) |
|
||||
| `VideoGenerator.generate(request: GenerationRequest) -> GenerationResult` | Today | Aggregated; Dynamo wraps in `asyncio.to_thread` under `asyncio.Lock` |
|
||||
| `VideoGenerator.generate_async(request) -> AsyncGenerator[VideoEvent, None]` | **PR 7.10** | Canonical execution substrate; sync wrapper reroutes through this |
|
||||
| `VideoGenerator.default_health_check_request() -> GenerationRequest` | **PR 7.10** | 256x256 / 8 frames / 1 step; lets Dynamo build its health payload without knowing any FastVideo internals |
|
||||
@@ -225,7 +225,7 @@ async def init_video_generation(runtime, config, shutdown_endpoints):
|
||||
from fastvideo.api import config_to_dict
|
||||
|
||||
server_args, dynamo_args = config.server_args, config.dynamo_args
|
||||
generator = VideoGenerator.from_config(build_generator_config(server_args))
|
||||
generator = VideoGenerator.from_pretrained(**config.fastvideo_kwargs())
|
||||
|
||||
dump_config(dynamo_args.dump_config_to, config)
|
||||
|
||||
@@ -262,8 +262,7 @@ this adapter can build the config purely from the public typed schema:
|
||||
def build_generator_config(args) -> "GeneratorConfig":
|
||||
from fastvideo.api import (
|
||||
CompileConfig, ComponentConfig, EngineConfig, GeneratorConfig,
|
||||
LTX2Options, LTX2RefineOptions, OffloadConfig, ParallelismConfig,
|
||||
PipelineSelection,
|
||||
OffloadConfig, ParallelismConfig, PipelineSelection,
|
||||
)
|
||||
return GeneratorConfig(
|
||||
model_path=args.model_path,
|
||||
@@ -276,8 +275,10 @@ def build_generator_config(args) -> "GeneratorConfig":
|
||||
pipeline=PipelineSelection(
|
||||
workload_type=args.workload or "t2v",
|
||||
preset=args.preset, # e.g. "ltx2_two_stage"
|
||||
components=ComponentConfig(upsampler_weights=args.refine_upsampler),
|
||||
model=LTX2Options(refine=LTX2RefineOptions(lora_path=args.refine_lora)),
|
||||
components=ComponentConfig(
|
||||
upsampler_weights=args.refine_upsampler,
|
||||
lora_path=args.refine_lora,
|
||||
),
|
||||
),
|
||||
)
|
||||
```
|
||||
@@ -309,11 +310,8 @@ re-chase FastVideo drift:
|
||||
2. `ContinuationState.payload` is JSON-serializable or references
|
||||
opaque blob ids. Dynamo can round-trip it through RPC without
|
||||
special-casing torch tensors.
|
||||
3. `VideoGenerator.from_pretrained(model_path, config)` takes a typed
|
||||
`GeneratorConfig` or its nested mapping; any flat keyword raises
|
||||
`TypeError` that points to the nested config.
|
||||
`VideoGenerator.from_config(...)` takes the same settings with
|
||||
`model_path` inside.
|
||||
3. `VideoGenerator.from_pretrained` accepts a typed `GeneratorConfig`;
|
||||
legacy flat kwargs are compatibility-only and deprecate in PR 13.
|
||||
4. `generate_async` (PR 7.10+) emits events in order
|
||||
`Progress* → Partial* → Final`; the final event always has exactly
|
||||
one occurrence per request.
|
||||
@@ -331,6 +329,7 @@ at FastVideo's CI — before the Dynamo-side integration even knows.
|
||||
* Anything under `fastvideo.pipelines.*` directly (pipelines are
|
||||
internal; presets identify them by name on
|
||||
`PipelineSelection.preset`).
|
||||
* `fastvideo.fastvideo_args.FastVideoArgs` (legacy compat type).
|
||||
* `fastvideo.api.compat.*` private helpers
|
||||
(`_validate_continuation_state` etc.) — the public boundary is
|
||||
`VideoGenerator` + `fastvideo.api`.
|
||||
|
||||
@@ -38,17 +38,14 @@ COOKBOOK_SOURCE_ROOTS = (
|
||||
# (e.g. black-forest-labs/FLUX.1-dev) and are grouped for documentation only.
|
||||
COOKBOOK_FAMILIES = {
|
||||
"wan",
|
||||
"turbodiffusion",
|
||||
"ltx2",
|
||||
"hunyuan",
|
||||
"cosmos",
|
||||
"kandinsky5",
|
||||
"kandinsky6",
|
||||
"flux",
|
||||
"glm_image",
|
||||
"zimage",
|
||||
"sd35",
|
||||
"minimax_h3",
|
||||
"longcat",
|
||||
"stable_audio",
|
||||
"mmaudio",
|
||||
"matrixgame",
|
||||
@@ -293,12 +290,16 @@ def cookbook_serving_profile(recipe: dict) -> dict:
|
||||
for language, (filename, install) in COOKBOOK_CLIENTS.items():
|
||||
path = ROOT_DIR / "examples/serving/clients" / filename
|
||||
text = path.read_text(encoding="utf-8")
|
||||
# Keep displayed snippets identical to executable sources for the
|
||||
# checked-in H3 profile, with only endpoint and alias substitutions.
|
||||
# Keep displayed snippets identical to the executable sources, with
|
||||
# only endpoint and alias substitutions.
|
||||
text = text.replace("http://127.0.0.1:8000/v1", base_url)
|
||||
text = text.replace('"fasth3"', json.dumps(model)).replace("${FASTVIDEO_MODEL:-fasth3}",
|
||||
"${FASTVIDEO_MODEL:-" + model + "}")
|
||||
clients[language] = {"source": path.relative_to(ROOT_DIR).as_posix(), "code": text, "install": install}
|
||||
# The playground router only serves H3 servers (see require_h3 in
|
||||
# fastvideo/entrypoints/openai/playground.py), so other families must not link to it.
|
||||
has_playground = recipe["family"] == "minimax_h3"
|
||||
compile_enabled = ((generator.get("engine") or {}).get("compile") or {}).get("enabled")
|
||||
return {
|
||||
"source": serving["source"],
|
||||
"install": serving["install"],
|
||||
@@ -307,7 +308,10 @@ def cookbook_serving_profile(recipe: dict) -> dict:
|
||||
"prepare": serving.get("prepare", ""),
|
||||
"model": model,
|
||||
"base_url": base_url,
|
||||
"playground_url": f"http://127.0.0.1:{port}/playground/",
|
||||
"playground_url": f"http://127.0.0.1:{port}/playground/" if has_playground else None,
|
||||
"audio": bool(serving.get("audio")),
|
||||
# None when the config leaves compilation at its default.
|
||||
"compile_enabled": compile_enabled,
|
||||
"health_command": f"curl --fail-with-body http://127.0.0.1:{port}/health",
|
||||
"hardware": hardware,
|
||||
"sampling": config["default_request"]["sampling"],
|
||||
|
||||
@@ -46,6 +46,91 @@ is the higher-quality FastH3.
|
||||
Recorded shapes and evidence live in the
|
||||
[support matrix](../../inference/support_matrix.md#apple-silicon-native-runtime).
|
||||
|
||||
## FastH3 V2 and Trim with INT6
|
||||
|
||||
The released sources are `FastVideo/FastVideo-FastH3-8-Step-V2` and
|
||||
`FastVideo/FastVideo-FastH3-Trim-8-Step`. Trim has 42 transformer blocks and
|
||||
rank-16 AdaLN. Both use eight denoising forwards, video/audio shifts of 10/3,
|
||||
VSA sparsity 0.8, a native NVFP4 text encoder, and the 26-layer light video VAE.
|
||||
Keep `fastvideo_inference.json` beside `transformer/`; conversion reads its
|
||||
schedule to build the AdaLN cache.
|
||||
|
||||
Convert the BF16 transformer to affine INT6 with its VSA gates:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-Trim-8-Step \
|
||||
--local-dir ./FastH3-Trim
|
||||
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py \
|
||||
--model-root ./FastH3-Trim/transformer \
|
||||
--out ./FastH3-Trim-MLX \
|
||||
--formats "int6" --include-vsa
|
||||
```
|
||||
|
||||
For V2, use `FastVideo/FastVideo-FastH3-8-Step-V2` and separate source/output
|
||||
directories. Preconverted release snapshots use the same names with the
|
||||
`-MLX-INT6` suffix. Each snapshot includes the encoder in MLX layout, both VAEs,
|
||||
and the trained schedule, so it does not require a second encoder download.
|
||||
|
||||
The 36 GiB M4 Max release recipe uses phased placement. It loads the encoder,
|
||||
DiT, and decoders in turn. Use reference attention and native output geometry:
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
from fastvideo.mlx_runtime.minimax_h3_pipeline import MiniMaxH3MLXPipeline
|
||||
|
||||
root = Path("./FastH3-Trim")
|
||||
pipeline = MiniMaxH3MLXPipeline(
|
||||
model_root=root,
|
||||
mlx_dit_checkpoint="./FastH3-Trim-MLX/int6",
|
||||
conditioner_mode="nvfp4",
|
||||
resident=False,
|
||||
vae_dtype="fp16",
|
||||
metal_wired_limit_gib=27,
|
||||
)
|
||||
try:
|
||||
pipeline.generate(
|
||||
"A corgi news anchor sits behind a desk and gives a cheerful bark.",
|
||||
output_path="./outputs/trim-int6-corgi.mp4",
|
||||
width=832, height=480, num_frames=124, seed=1234,
|
||||
num_steps=8, vsa=True, vsa_sparsity=0.8, vsa_tile_size=64,
|
||||
vsa_impl="reference", vae_tile_height=256, vae_tile_width=256,
|
||||
)
|
||||
finally:
|
||||
pipeline.close()
|
||||
```
|
||||
|
||||
Set `FASTVIDEO_MLX_DQ_GEMM=1` before running this Python command. It selects the
|
||||
validated affine dequantization followed by dense matrix multiplication.
|
||||
124 frames at 24 fps is roughly five seconds. The recipe preserves all frames
|
||||
and the requested resolution.
|
||||
|
||||
### Native NVFP4 encoder and cache
|
||||
|
||||
The MLX conditioner reads the released packed NVFP4 weights without
|
||||
requantization. It retains the layers H3 reads and can cache the packed weights
|
||||
in MLX layout. The cache is written in a staging directory and published by a
|
||||
single rename. A cache hit changes storage layout, not encoder arithmetic.
|
||||
|
||||
MLX uses BF16 embeddings and FP32 activations; CUDA uses quantized activations.
|
||||
Generated video and audio must be reviewed before claiming cross-runtime
|
||||
quality parity. MLX 0.32.2 supports the required operator on Apple Silicon.
|
||||
|
||||
### Metal wired memory
|
||||
|
||||
MLX's allocation limit and wired-memory limit are separate. The optional
|
||||
`metal_wired_limit_gib` calls `mx.set_wired_limit` to keep selected Metal
|
||||
allocations in physical memory. It does not add RAM. An explicit request fails
|
||||
if the installed MLX build cannot apply it. `close()` restores the previous
|
||||
wired limit. The phased pipeline also sets a 30 GiB maximum allocator guideline
|
||||
and restores its previous value on close. Resident placement keeps the existing
|
||||
allocator limit so larger Macs can hold all components.
|
||||
|
||||
The tested 36 GiB M4 Max recipe uses 27 GiB and phased placement. Resident
|
||||
placement also requires capacity for all components and peak activations;
|
||||
wiring cannot make an oversized stack fit. Inspect `mx.device_info()` before
|
||||
choosing a limit on another Mac.
|
||||
|
||||
## Hardware
|
||||
|
||||
- FastMetal 1.3B and 5B: 16 GB unified memory and up
|
||||
|
||||
@@ -144,6 +144,9 @@ for which models are practical on the GB10, what makes them faster, and what
|
||||
won't help on this hardware (and why) — so you don't spend a night tuning knobs
|
||||
that can't move here.
|
||||
|
||||
For the eight-forward FastH3 V2 NVFP4 stack with a trimmed encoder and light
|
||||
VAE, use the [one-Spark resident recipe](spark_performance.md#fasth3-v2-nvfp4-on-one-spark).
|
||||
|
||||
Two Sparks with QSFP cables: [Pair two NVIDIA DGX Sparks](spark_pair.md) for
|
||||
one FastH3 clip across both GPUs (`sp_size=2` over Ray). Copy-paste commands
|
||||
for one or two Sparks also live on the
|
||||
|
||||
@@ -17,7 +17,7 @@ xDiT vendor. Do not install xDiT for this path.
|
||||
|
||||
| Goal | How | Use two Sparks? |
|
||||
|---|---|---|
|
||||
| Two independent videos at once | One process per box, `engine.num_gpus: 1` | Throughput only. Each clip still takes the 1-GPU time for that size. |
|
||||
| Two independent videos at once | One process per box, `num_gpus=1` | Throughput only. Each clip still takes the 1-GPU time for that size. |
|
||||
| One clip, faster | Ray + `sp_size=2` + parallel VAE | **Yes.** One 768×1344×124 recipe was 292 s vs 374 s on one GB10. |
|
||||
| One clip, longer | Same, more frames | **Yes.** 345 frames (~14.4 s at 24 fps) finished in 587 s at 768×1344. |
|
||||
|
||||
@@ -35,7 +35,7 @@ over ~21 GB/s RoCE.
|
||||
QSFP; do not download 100+ GB twice over Wi-Fi.
|
||||
- Ray in the FastVideo venv (`uv pip install ray` if it is not already there).
|
||||
|
||||
Each Spark has **one** GPU. `engine.num_gpus: 2` therefore means two nodes, which is
|
||||
Each Spark has **one** GPU. `num_gpus=2` therefore means two nodes, which is
|
||||
why the executor must be Ray (`mp` only works inside one process tree).
|
||||
|
||||
## 1. Put IPv4 on the QSFP NICs
|
||||
@@ -149,6 +149,29 @@ fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yam
|
||||
|
||||
Stop the cluster when you are done: `ray stop` on both nodes.
|
||||
|
||||
## Released V2 and Trim NVFP4 stacks
|
||||
|
||||
The public eight-forward V2 and Trim stacks include the NVFP4 encoder and
|
||||
lightweight video VAE. Use the environment above plus the release kernel
|
||||
settings, then run one of these configs from the head:
|
||||
|
||||
```bash
|
||||
export FASTVIDEO_MINIMAX_H3_FUSIONS=all FASTVIDEO_NVFP4_MM_BACKEND=cutlass
|
||||
export FASTVIDEO_H3_VAE_TILE_BATCH=1 FASTVIDEO_VSA_TRITON=1
|
||||
export FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0
|
||||
export FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1
|
||||
fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair_v2_nvfp4.yaml
|
||||
# Or basic_fasth3_spark_pair_pruned_nvfp4.yaml for FastH3 Trim.
|
||||
```
|
||||
|
||||
Both configs use their released Hugging Face model paths, `h3_dit_vsa`,
|
||||
832x480, 124 frames, seed 1234, VSA 0.8 and 64-token tiles. All components
|
||||
stay resident; lazy/sequential loading and compilation are disabled for
|
||||
these compact stacks. The older BF16/Preview memory guidance below applies
|
||||
to those larger stacks. Set 1344x768 for native 768p with the same 124 frames.
|
||||
See [the resident release recipe](spark_performance.md#fasth3-v2-nvfp4-on-one-spark)
|
||||
for checkpoint contents and the timing protocol.
|
||||
|
||||
## FastH3 frame counts
|
||||
|
||||
H3 is 24 fps. Legal `num_frames` values are `17n+5`. The pipeline rejects
|
||||
@@ -197,8 +220,8 @@ sm_100a VSA kernel is not on this chip, so denoise is slower than a GB200
|
||||
| NCCL hangs or uses Wi-Fi | `source spark_pair_env.sh`. Confirm `NCCL_SOCKET_IFNAME` is the QSFP NIC. |
|
||||
| Gloo `connectFullMesh` / `remote=[127.0.0.1]` | Two 1-GPU nodes must not use loopback as the Gloo store. Source `spark_pair_env.sh` so `GLOO_SOCKET_IFNAME` is the QSFP NIC on **each** box. FastVideo no longer copies that NIC name from the driver onto workers. |
|
||||
| Second `generate()` crashes `NoneType.parameters` | Sequential load used to drop the text encoder without reloading it. This branch reloads Qwen for later requests so `--warmup --repeats N` works. |
|
||||
| OOM / `earlyoom` prefers Python | Lazy module load must stay on (do not pass `--no-lazy-module-load` to `basic_fasth3.py` or set `engine.offload.lazy_module_load: false`). Peak GPU during 345-frame denoise is ~90 GiB/node. |
|
||||
| `engine.num_gpus: 2` on one Spark | Each Spark has one GPU. Use Ray across two nodes, or `engine.num_gpus: 1` on one box. |
|
||||
| OOM / `earlyoom` prefers Python | Lazy module load must stay on (do not pass `--no-lazy-module-load`). Peak GPU during 345-frame denoise is ~90 GiB/node. |
|
||||
| `num_gpus=2` on one Spark | Each Spark has one GPU. Use Ray across two nodes, or `num_gpus=1` on one box. |
|
||||
|
||||
## What we are not claiming
|
||||
|
||||
|
||||
@@ -62,12 +62,10 @@ nothing to set. If you run a model that still defaults to an fp32 decode, set th
|
||||
decode-only override yourself:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.configs.pipelines.base import PipelineConfig
|
||||
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": model_id,
|
||||
"engine": {"precision": {"vae_decode": "bf16"}}, # decode-only; leaves encode precision alone
|
||||
})
|
||||
pipeline_config = PipelineConfig.from_pretrained(model_id)
|
||||
pipeline_config.vae_decode_precision = "bf16" # decode-only; leaves encode precision alone
|
||||
```
|
||||
|
||||
Decode is output-only, so lowering its precision is safe. (Encode seeds the
|
||||
@@ -163,15 +161,16 @@ is power-cycled. To avoid it:
|
||||
on: "CPU" offload uses the same unified RAM. Multi-GPU FSDP sharding remains
|
||||
available because it partitions weights without parking them in a separate
|
||||
host pool.
|
||||
- **MiniMax H3 / FastH3** still needs deferred loading on one GB10. The Qwen3-VL
|
||||
conditioner is tens of gigabytes of BF16. If the DiT and VAEs load while that
|
||||
encoder is still resident, the process is a typical `earlyoom` kill (Python is
|
||||
preferred). On unified memory, `lazy_module_load` auto-enables and owns that
|
||||
- **Older MiniMax H3 / FastH3 bf16 weights** need deferred loading on one GB10.
|
||||
The full Qwen3-VL conditioner is tens of gigabytes of BF16. If the DiT and
|
||||
VAEs load while that encoder is still resident, the process can be killed by
|
||||
`earlyoom`. On unified memory, `lazy_module_load` auto-enables and owns that
|
||||
split (encoder, then DiT, then VAE; DiT can drop before decode). Sequential
|
||||
load is the H3-only fallback when lazy is off; do not set
|
||||
`engine.offload.lazy_module_load` to false here (`--no-lazy-module-load` in
|
||||
`basic_fasth3.py` and `basic_minimax_h3_t2v.py`). Geometry scalars come from checkpoint
|
||||
`config.json`, not live weights. See [Offloading](../../inference/offloading.md).
|
||||
load is the H3-only fallback when lazy is off. Keep deferred loading for
|
||||
those older checkpoints. The trimmed NVFP4 encoder and light VAE in the
|
||||
[V2 resident recipe](#fasth3-v2-nvfp4-on-one-spark) are a different memory
|
||||
profile. Geometry scalars come from checkpoint `config.json`, not live
|
||||
weights. See [Offloading](../../inference/offloading.md).
|
||||
- **FastH3 TAEH3** (`--video-decode-backend taeh3`) is an opt-in preview decoder.
|
||||
T2VA never materializes the 9.7 GiB video VAE (DiT still loads after Qwen via
|
||||
sequential start). On this box, alpine 768×1344×124 decoded in **2.4 s** versus
|
||||
@@ -204,12 +203,127 @@ A few things that surprise people on this box (beyond the memory notes above):
|
||||
build recent enough to include its `transformers`-compatibility handling before
|
||||
running it.
|
||||
- **MiniMax H3 worker init can look healthy and still die on the first generate**
|
||||
if deferred loading is off (`engine.offload.lazy_module_load: false` and sequential also off)
|
||||
if deferred loading is off (`--no-lazy-module-load` and sequential also off)
|
||||
and encoder, VAE, and DiT load together. On GB10 the log should show
|
||||
`lazy_module_load owns deferral` (or, if lazy is off, sequential
|
||||
`Released MiniMax-H3 text encoder after conditioning` before
|
||||
`Loading MiniMax-H3 denoise modules`).
|
||||
|
||||
## FastH3 V2 NVFP4 on one Spark
|
||||
|
||||
This recipe uses the full V2 eight-forward transformer, the 50-layer NVFP4
|
||||
Qwen3-VL encoder, and the light H3 video VAE. Its configuration keeps all
|
||||
three resident on one GB10. This stack fits in the Spark's unified memory;
|
||||
benchmark your installed runtime and review the clips before publishing a
|
||||
speed claim. The earlier bf16 H3 memory guidance above concerns a larger checkpoint.
|
||||
|
||||
Install FastVideo following [the Spark install guide](spark.md). The released
|
||||
repositories are complete inference stacks:
|
||||
|
||||
| Model | Repository | Packed DiT profile |
|
||||
|---|---|---|
|
||||
| FastH3 V2 | `FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer` | `h3_dit_vsa` |
|
||||
| FastH3 Trim | `FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4` | `h3_dit_vsa` |
|
||||
|
||||
Each ships its own trained schedule, NVFP4 transformer and 50-layer NVFP4
|
||||
text encoder, lightweight 26-layer video VAE with the INT8-weight overlay,
|
||||
and audio VAE. Download the complete repository; the runtime selects these
|
||||
components from its model index. The released Trim transformer also packs
|
||||
attention and VSA gates, unlike the earlier FFN-only pruned export. Neither
|
||||
release requires local checkpoint conversion or a separate encoder download.
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer
|
||||
hf download FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4
|
||||
```
|
||||
|
||||
Keep `fastvideo_inference.json` with the transformer if you stage the stack
|
||||
in a local directory. It declares the trained eight-forward ladder and
|
||||
video/audio shifts of 10/3. The recipes use `num_inference_steps: 9` for
|
||||
nine sigma points and eight DiT forwards.
|
||||
|
||||
On GB10 with FlashInfer 0.6.18, FastVideo fences activation quantization before
|
||||
releasing its padded input. Without this completion fence, identical H3
|
||||
requests produced different DiT latents and occasionally corrupt video.
|
||||
The fence applies to `sm_121`; other architectures retain asynchronous execution.
|
||||
|
||||
Run `examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml` from the
|
||||
repository root. It uses 832x480, 124 frames and seed 1234, VSA sparsity 0.8 with
|
||||
64-token tiles, and the light H3 VAE through the `h3-vae` decode backend.
|
||||
It does not use frame dropping or spatial upscaling.
|
||||
|
||||
```bash
|
||||
FASTVIDEO_MINIMAX_H3_FUSIONS=all \
|
||||
FASTVIDEO_NVFP4_MM_BACKEND=cutlass \
|
||||
FASTVIDEO_H3_VAE_TILE_BATCH=1 \
|
||||
FASTVIDEO_VSA_TRITON=1 FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 \
|
||||
FASTVIDEO_STAGE_LOGGING=1 \
|
||||
nice -n 19 fastvideo generate \
|
||||
--config examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml
|
||||
```
|
||||
|
||||
The defaults produce a roughly five-second clip at 24 fps. For native 768p,
|
||||
set `--request.sampling.width 1344` and `--request.sampling.height 768`,
|
||||
keeping 124 frames. For the separate ten-second setting, use 243 frames.
|
||||
H3 permits frame counts of `17n+5`; 124 is the nearest legal count above
|
||||
five seconds and 243 is the nearest above ten seconds.
|
||||
|
||||
For Trim, use `basic_fasth3_spark_pruned_nvfp4.yaml` with the same environment.
|
||||
Both recipes keep the encoder, DiT and VAEs resident, disable compilation,
|
||||
and use `h3_dit_vsa`. On two Sparks, use the corresponding
|
||||
`basic_fasth3_spark_pair_{pruned,v2}_nvfp4.yaml` after following the
|
||||
[pair setup guide](spark_pair.md). The pair uses SP2/TP1 and parallel VAE
|
||||
gathering, with tile batch 1 on each worker.
|
||||
|
||||
For release timing, create one generator per model/resolution. Run one
|
||||
untimed ceramics warmup, then two timed ceramics calls and two timed harbor
|
||||
calls in that same process, using the exact release prompt strings, seed
|
||||
1234 and the settings above. Measure each `generate()` call through finished
|
||||
MP4 output and report the median of the two timed calls per prompt. Keep the
|
||||
warmup excluded. Record the model revision, code commit, command, environment
|
||||
and peak-memory scope with the results. Review every clip's video and audio
|
||||
before publishing a quality or speed claim.
|
||||
|
||||
The benchmark helper requires a local stack so it can validate the schedule
|
||||
before loading. For example:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer \
|
||||
--local-dir ./FastH3-V2-Consumer
|
||||
python examples/inference/basic/benchmark_fasth3_spark_nvfp4.py \
|
||||
--config examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml \
|
||||
--model-path ./FastH3-V2-Consumer --frames 124 \
|
||||
--prompts /path/to/benchmark_prompts.json \
|
||||
--output-dir outputs/fasth3_spark_v2_nvfp4/benchmark-124
|
||||
```
|
||||
|
||||
Set the environment from the generation command above before benchmarking.
|
||||
The prompt JSON must contain `latency-ceramics-005` and
|
||||
`latency-harbor-005`. Pass `--width 1344 --height 768` for the native 768p
|
||||
protocol. Use the Trim repository and config for its corresponding run.
|
||||
|
||||
### Released model measurements
|
||||
|
||||
The released Trim stack at revision `cae9ceb6feefe77d34a56640782cda3909363f19`
|
||||
completed the native 832x480, 124-frame protocol on one Spark, seed 1234.
|
||||
The tested code is `6ccdbc761e6b854003f472e39b826c06cb54de60`, using the
|
||||
resident recipe above. Medians exclude warmups and cover finished MP4 output.
|
||||
|
||||
| Released model | Resolution | One Spark, ceramics / harbor | Two Sparks |
|
||||
|---|---|---:|---|
|
||||
| FastH3 Trim NVFP4 | 832x480 | 124.090 / 127.519 s | Pending |
|
||||
| FastH3 Trim NVFP4 | 1344x768 | Review pending | Pending |
|
||||
| FastH3 V2 NVFP4-Consumer | 832x480 | Repeatability review pending | Pending |
|
||||
| FastH3 V2 NVFP4-Consumer | 1344x768 | Review pending | Pending |
|
||||
|
||||
The completed Trim 480p batch used one extra untimed harbor warmup in the
|
||||
same process, six calls total. Both warmups are excluded from the medians.
|
||||
Each prompt's three decoded videos and audio streams match exactly, and
|
||||
sampled frames are coherent. These checks do not establish BF16 parity,
|
||||
speech accuracy or lip sync. Pending cells are not measured substitutes
|
||||
from older checkpoints.
|
||||
|
||||
## Reproduce these numbers
|
||||
|
||||
Two scripts under `examples/inference/optimizations/` reproduce the claims on
|
||||
|
||||
@@ -210,7 +210,7 @@ from fastvideo.pipelines.stages import (
|
||||
InputValidationStage, CLIPTextEncodingStage, TimestepPreparationStage,
|
||||
LatentPreparationStage, DenoisingStage, DecodingStage
|
||||
)
|
||||
from fastvideo.api.resolution import ResolvedGeneratorConfig
|
||||
from fastvideo.fastvideo_args import FastVideoArgs
|
||||
from fastvideo.pipelines.pipeline_batch_info import ForwardBatch
|
||||
import torch
|
||||
|
||||
@@ -226,11 +226,11 @@ class MyCustomPipeline(ComposedPipelineBase):
|
||||
def required_config_modules(self) -> List[str]:
|
||||
return self._required_config_modules
|
||||
|
||||
def initialize_pipeline(self, resolved_config: ResolvedGeneratorConfig):
|
||||
def initialize_pipeline(self, fastvideo_args: FastVideoArgs):
|
||||
"""Initialize pipeline-specific components."""
|
||||
pass
|
||||
|
||||
def create_pipeline_stages(self, resolved_config: ResolvedGeneratorConfig):
|
||||
def create_pipeline_stages(self, fastvideo_args: FastVideoArgs):
|
||||
"""Set up pipeline stages with proper dependency injection."""
|
||||
self.add_stage(
|
||||
stage_name="input_validation_stage",
|
||||
@@ -294,7 +294,7 @@ class MyCustomStage(PipelineStage):
|
||||
self.custom_module = custom_module
|
||||
self.other_param = other_param
|
||||
|
||||
def forward(self, batch: ForwardBatch, resolved_config: ResolvedGeneratorConfig) -> ForwardBatch:
|
||||
def forward(self, batch: ForwardBatch, fastvideo_args: FastVideoArgs) -> ForwardBatch:
|
||||
# Access input data
|
||||
input_data = batch.some_attribute
|
||||
|
||||
|
||||
@@ -103,10 +103,6 @@ PipelineConfig (fastvideo/configs/pipelines/base.py)
|
||||
- Precision settings: `dit_precision`, `vae_precision`,
|
||||
`text_encoder_precisions`.
|
||||
|
||||
These generation and precision attributes hold the model defaults. Resolution
|
||||
copies them into the typed fields of the resolved config (`pipeline.flow_shift`,
|
||||
`engine.precision.dit`, ...), and runtime code reads the typed fields.
|
||||
|
||||
Model-specific subclasses override defaults. For example,
|
||||
`WanT2V480PConfig` sets `flow_shift=3.0` and uses `WanVideoConfig` as
|
||||
its DiT config.
|
||||
@@ -131,11 +127,9 @@ Concrete hierarchy: `DiTConfig` → `DiTArchConfig`, `VAEConfig` →
|
||||
|
||||
- `PipelineConfig.from_pretrained(model_path)` — resolves config class
|
||||
via `get_pipeline_config_cls_from_name()`, instantiates with defaults.
|
||||
- `PipelineConfig.from_source(model_path, source)` — resolves the registry
|
||||
class of `model_path`, then updates it from `source`: a JSON path loaded
|
||||
via `load_from_json()`, a mapping of field values applied via
|
||||
`update_pipeline_config()`, or a `PipelineConfig` that replaces the
|
||||
registry instance.
|
||||
- `PipelineConfig.from_kwargs(kwargs)` — resolves class, optionally loads
|
||||
JSON via `load_from_json()`, then applies CLI overrides via
|
||||
`update_config_from_dict()`.
|
||||
- `dump_to_json()` / `load_from_json()` — JSON persistence. Callable
|
||||
fields and `arch_config` are excluded from dumps.
|
||||
|
||||
@@ -153,7 +147,7 @@ sp = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
|
||||
### ComponentLoader (`fastvideo/models/loader/component_loader.py`)
|
||||
|
||||
Abstract base with a `load(model_path, resolved_config)` method.
|
||||
Abstract base with a `load(model_path, fastvideo_args)` method.
|
||||
`ComponentLoader.for_module_type(module_type, library)` is a factory
|
||||
that dispatches to specialized loaders via a `module_loaders` dict:
|
||||
|
||||
@@ -173,7 +167,7 @@ that dispatches to specialized loaders via a `module_loaders` dict:
|
||||
`TransformerLoader` reads `config.json` from the component directory,
|
||||
resolves the class via `ModelRegistry.resolve_model_cls()`, instantiates
|
||||
the model, and loads safetensors weights. CPU offload and layerwise
|
||||
offload are applied based on `resolved_config.engine.offload`.
|
||||
offload are applied based on `FastVideoArgs`.
|
||||
|
||||
Unknown module types fall back to `GenericComponentLoader`.
|
||||
|
||||
@@ -211,14 +205,14 @@ loading by calling `ComponentLoader.for_module_type()` then `.load()`.
|
||||
|
||||
Abstract base class using the Template Method pattern:
|
||||
|
||||
- `__call__(batch, resolved_config)` — orchestrates verification, timing,
|
||||
- `__call__(batch, fastvideo_args)` — orchestrates verification, timing,
|
||||
and error handling. Not overridden by subclasses.
|
||||
- `forward(batch, resolved_config) -> ForwardBatch` — abstract, contains
|
||||
- `forward(batch, fastvideo_args) -> ForwardBatch` — abstract, contains
|
||||
the stage logic.
|
||||
- `verify_input()` / `verify_output()` — optional hooks returning
|
||||
`VerificationResult`. Default: no checks.
|
||||
|
||||
When `resolved_config.engine.enable_stage_verification` is `True`, `__call__`
|
||||
When `fastvideo_args.enable_stage_verification` is `True`, `__call__`
|
||||
runs input verification before `forward()` and output verification after.
|
||||
When `envs.FASTVIDEO_STAGE_LOGGING` is set, execution time is measured
|
||||
with `torch.cuda.synchronize()` and logged.
|
||||
@@ -238,8 +232,7 @@ Dataclass carrying all pipeline state between stages. Key field groups:
|
||||
- **Scheduler**: `timesteps`, `num_inference_steps`, `guidance_scale`,
|
||||
`sigmas`.
|
||||
- **Task-specific**: `mouse_cond`/`keyboard_cond` (Matrix-Game 2.0), `pose`
|
||||
(HYWorld), `camera_states` (GameCraft), `c2ws_plucker_emb`
|
||||
(LingBotWorld).
|
||||
(HYWorld), `c2ws_plucker_emb` (LingBotWorld).
|
||||
- **Output**: `output: Tensor | None`.
|
||||
- **Logging**: `logging_info: PipelineLoggingInfo`.
|
||||
|
||||
@@ -262,7 +255,6 @@ Standard stages (typical execution order):
|
||||
| `DecodingStage` | `stages/decoding.py` | Decodes latents to video via VAE |
|
||||
|
||||
Specialized variants: `CausalDenoisingStage`, `LTX2DenoisingStage`,
|
||||
`LongCatDenoisingStage`, `GameCraftDenoisingStage`,
|
||||
`HYWorldDenoisingStage`, `MatrixGame2CausalDenoisingStage`,
|
||||
`SRDenoisingStage`, `LTX2AudioDecodingStage`, `SD35ConditioningStage`,
|
||||
`LTX2TextEncodingStage`, `LTX2LatentPreparationStage`.
|
||||
@@ -301,7 +293,7 @@ provides detailed error messages. Failed verification raises
|
||||
|
||||
Abstract base for all inference pipelines. Lifecycle:
|
||||
|
||||
1. **`__init__(model_path, resolved_config)`** — initializes distributed
|
||||
1. **`__init__(model_path, fastvideo_args)`** — initializes distributed
|
||||
environment via `maybe_init_distributed_environment_and_model_parallel
|
||||
(tp_size, sp_size)`, then calls `load_modules()` to populate
|
||||
`self.modules`.
|
||||
@@ -309,7 +301,7 @@ Abstract base for all inference pipelines. Lifecycle:
|
||||
setup), `create_pipeline_stages()` (abstract — subclasses wire stages),
|
||||
optionally applies `torch.compile` to transformers, and calls
|
||||
`warmup_sequence_parallel_communication()`.
|
||||
3. **`forward(batch, resolved_config)`** — iterates `self.stages` calling
|
||||
3. **`forward(batch, fastvideo_args)`** — iterates `self.stages` calling
|
||||
each stage in order. Decorated with `@torch.no_grad()`.
|
||||
|
||||
Key class attributes:
|
||||
@@ -322,9 +314,8 @@ Key methods:
|
||||
- `add_stage(name, stage)` — appends to `_stages` list and
|
||||
`_stage_name_mapping` dict, also sets attribute on `self`.
|
||||
- `get_module(name, default)` — retrieves a loaded module.
|
||||
- `from_pretrained(model_path, *, resolved_config)` — class method that
|
||||
builds the pipeline from a resolved config (from
|
||||
`resolve_inference_config({...})`) by calling `cls(...)` then `post_init()`.
|
||||
- `from_pretrained(model_path, **kwargs)` — class method constructing
|
||||
`FastVideoArgs` and calling `cls(...)` then `post_init()`.
|
||||
|
||||
### LoRAPipeline (`fastvideo/pipelines/lora_pipeline.py`)
|
||||
|
||||
@@ -359,12 +350,12 @@ Key APIs: `get_tp_rank()`, `get_tp_world_size()`, `get_sp_rank()`,
|
||||
`warmup_sequence_parallel_communication()` pre-warms NCCL communicators
|
||||
to avoid slow first forward passes.
|
||||
|
||||
Usage: `fastvideo generate --config run.yaml
|
||||
--generator.engine.parallelism.tp_size N --generator.engine.parallelism.sp_size M`.
|
||||
Usage: `torchrun --nproc-per-node=N -m fastvideo.entrypoints.cli.main
|
||||
generate --model-path ... --tp-size N --sp-size M`.
|
||||
|
||||
### torch.compile Integration
|
||||
|
||||
When `resolved_config.engine.compile.enabled` is `True`,
|
||||
When `fastvideo_args.enable_torch_compile` is `True`,
|
||||
`_maybe_compile_pipeline_module()` checks for a `_compile_conditions`
|
||||
attribute on the module. If present, only matching submodules are
|
||||
compiled. Otherwise, the entire module is compiled. FSDP-wrapped
|
||||
@@ -376,53 +367,47 @@ modules are skipped.
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
{"engine": {"num_gpus": 1, "parallelism": {"tp_size": 1, "sp_size": 1}}},
|
||||
model_path="Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1, tp_size=1, sp_size=1,
|
||||
)
|
||||
result = generator.generate_video(
|
||||
prompt="A cat dancing",
|
||||
height=720, width=1280, num_frames=81,
|
||||
)
|
||||
result = generator.generate({
|
||||
"prompt": "A cat dancing",
|
||||
"sampling": {"height": 720, "width": 1280, "num_frames": 81},
|
||||
})
|
||||
```
|
||||
|
||||
**CLI** (`fastvideo/entrypoints/cli/`):
|
||||
|
||||
```bash
|
||||
# run.yaml holds `generator: {model_path: Wan-AI/Wan2.1-T2V-14B-Diffusers}`.
|
||||
fastvideo generate \
|
||||
--config run.yaml \
|
||||
--request.prompt "A cat dancing" \
|
||||
--generator.engine.num_gpus 1
|
||||
--model-path "Wan-AI/Wan2.1-T2V-14B-Diffusers" \
|
||||
--prompt "A cat dancing" \
|
||||
--num-gpus 1
|
||||
```
|
||||
|
||||
**ResolvedGeneratorConfig** (`fastvideo/api/resolution.py`): The frozen
|
||||
runtime config that the executor, workers, pipelines, stages, and loaders
|
||||
read. Key paths: `model_path`, `mode` (`ExecutionMode`),
|
||||
`pipeline.workload_type` (`WorkloadType`), `engine.num_gpus`,
|
||||
`engine.parallelism.tp_size`, `engine.parallelism.sp_size`,
|
||||
`pipeline.components.lora_path`, `engine.offload.dit`,
|
||||
`engine.offload.dit_layerwise`, `engine.compile.enabled`,
|
||||
`engine.enable_stage_verification`, and `pipeline_config` (the frozen
|
||||
`PipelineConfig`).
|
||||
**FastVideoArgs** (`fastvideo/fastvideo_args.py`): Central args dataclass.
|
||||
Key fields: `model_path`, `mode` (`ExecutionMode`), `workload_type`
|
||||
(`WorkloadType`), `pipeline_config` (`PipelineConfig`), `num_gpus`,
|
||||
`tp_size`, `sp_size`, `lora_path`, `dit_cpu_offload`,
|
||||
`dit_layerwise_offload`, `enable_torch_compile`,
|
||||
`enable_stage_verification`.
|
||||
|
||||
Built by `resolve_inference_config(config)`
|
||||
(`fastvideo/api/inference_resolution.py`), which runs the named resolution
|
||||
steps (environment variables, model defaults, derived values, validation) in
|
||||
order, records each decision, and then builds the `PipelineConfig` from the
|
||||
registry, applies a JSON config if provided, and freezes it.
|
||||
Constructed via `FastVideoArgs.from_kwargs(**kwargs)` which resolves the
|
||||
`PipelineConfig` from the registry, applies JSON config if provided, and
|
||||
merges CLI overrides.
|
||||
|
||||
## End-to-End Inference Flow
|
||||
|
||||
```
|
||||
User: VideoGenerator.from_pretrained(model_path, config)
|
||||
User: VideoGenerator.from_pretrained(model_path, **kwargs)
|
||||
│
|
||||
├─ resolve_inference_config() → PipelineConfig resolved via registry
|
||||
├─ FastVideoArgs.from_kwargs() → PipelineConfig resolved via registry
|
||||
├─ get_model_info() → ModelInfo(pipeline_cls, sampling_param_cls, ...)
|
||||
│ ├─ model_index.json read → _class_name extracted
|
||||
│ ├─ pipeline_registry resolves pipeline_cls from _class_name
|
||||
│ └─ config_registry resolves config classes from model_path
|
||||
│
|
||||
├─ pipeline_cls.__init__(model_path, resolved_config)
|
||||
├─ pipeline_cls.__init__(model_path, fastvideo_args)
|
||||
│ ├─ maybe_init_distributed(tp_size, sp_size)
|
||||
│ └─ load_modules() → reads model_index.json, loads each component
|
||||
│ ├─ ComponentLoader.for_module_type() → specialized loader
|
||||
@@ -434,10 +419,10 @@ User: VideoGenerator.from_pretrained(model_path, config)
|
||||
├─ torch.compile (if enabled)
|
||||
└─ warmup_sequence_parallel_communication()
|
||||
|
||||
User: generator.generate(request)
|
||||
User: generator.generate_video(prompt, ...)
|
||||
│
|
||||
├─ ForwardBatch constructed from SamplingParam + user args
|
||||
└─ pipeline.forward(batch, resolved_config)
|
||||
└─ pipeline.forward(batch, fastvideo_args)
|
||||
├─ InputValidationStage → validates dims
|
||||
├─ TextEncodingStage → prompt → embeddings
|
||||
├─ ConditioningStage → prepares conditioning
|
||||
|
||||
@@ -8,7 +8,7 @@ FastVideo automatically distributes the generation process when multiple GPUs ar
|
||||
# Will use 4 GPUs in parallel for faster generation
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{"engine": {"num_gpus": 4}},
|
||||
num_gpus=4,
|
||||
)
|
||||
```
|
||||
|
||||
@@ -30,53 +30,58 @@ bring-up: [Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pai
|
||||
|
||||
## Customizing Generation
|
||||
|
||||
`VideoGenerator.from_pretrained(model_path, config)` takes the startup
|
||||
settings as a nested mapping at their typed config paths, such as
|
||||
`{"engine": {"num_gpus": 2, "offload": {"dit": False}}}`; it is
|
||||
`VideoGenerator.from_config` with `model_path` added to the mapping. Pass
|
||||
generation settings to `VideoGenerator.generate` as a request:
|
||||
- `PipelineConfig`: Initialization time parameters
|
||||
- `SamplingParam`: Generation time parameters
|
||||
|
||||
You can customize generation behavior using `PipelineConfig` and
|
||||
`SamplingParam`:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo import VideoGenerator, SamplingParam, PipelineConfig
|
||||
|
||||
def main():
|
||||
model_name = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
config = PipelineConfig.from_pretrained(model_name)
|
||||
config.vae_precision = "fp16"
|
||||
|
||||
# Create the generator
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": model_name,
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"offload": {"dit_layerwise": True},
|
||||
"precision": {"vae": "fp16"},
|
||||
},
|
||||
})
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_name,
|
||||
num_gpus=1,
|
||||
dit_layerwise_offload=True, # FastVideoArgs option
|
||||
pipeline_config=config
|
||||
)
|
||||
|
||||
# Create and customize sampling parameters
|
||||
sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
|
||||
# How many frames to generate
|
||||
sampling_param.num_frames = 45
|
||||
|
||||
# Video resolution (width, height)
|
||||
sampling_param.width = 1024
|
||||
sampling_param.height = 576
|
||||
|
||||
# How many steps we denoise the video (higher = better quality, slower generation)
|
||||
sampling_param.num_inference_steps = 30
|
||||
|
||||
# How strongly the video conforms to the prompt (higher = more faithful to prompt)
|
||||
sampling_param.guidance_scale = 7.5
|
||||
|
||||
# Random seed for reproducibility
|
||||
sampling_param.seed = 42 # Optional, leave unset for random results
|
||||
|
||||
# Generate video with custom parameters
|
||||
prompt = "A beautiful sunset over a calm ocean, with gentle waves."
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"sampling": {
|
||||
# How many frames to generate
|
||||
"num_frames": 45,
|
||||
# Video resolution (width, height)
|
||||
"width": 1024,
|
||||
"height": 576,
|
||||
# How many steps we denoise the video (higher = better quality, slower generation)
|
||||
"num_inference_steps": 30,
|
||||
# How strongly the video conforms to the prompt (higher = more faithful to prompt)
|
||||
"guidance_scale": 7.5,
|
||||
# Random seed for reproducibility
|
||||
"seed": 42, # Optional, leave unset for random results
|
||||
},
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
sampling_param=sampling_param,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
|
||||
# If return_frames=True, frames are available in video.frames
|
||||
print(f"Generated {len(video.frames)} frames")
|
||||
# If return_frames=True, frames are available in video["frames"]
|
||||
print(f"Generated {len(video['frames'])} frames")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -120,26 +125,6 @@ Override individual values from the CLI with dotted paths:
|
||||
fastvideo generate --config config.yaml --request.sampling.seed 42
|
||||
```
|
||||
|
||||
## Where a Value Came From
|
||||
|
||||
FastVideo resolves the generator config once at startup and records the source of every value: the input config
|
||||
(`input`; `explicit` tells whether you wrote the value or it is the schema default), a `FASTVIDEO_*` environment
|
||||
variable, the model's defaults, or a derived value. A worker's device policy and values read from checkpoint files
|
||||
are recorded too.
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_config(config)
|
||||
generator.resolved_config.provenance("engine.parallelism.sp_size")
|
||||
# PathProvenance(path='engine.parallelism.sp_size', value=2, source='derive_parallel_sizes', ...)
|
||||
|
||||
result = generator.generate(request)
|
||||
result.resolved_request.provenance("sampling.num_frames")
|
||||
# PathProvenance(..., value=81, source='fill_sampling_defaults[preset wan_t2v_1_3b]', explicit=False)
|
||||
```
|
||||
|
||||
`resolved_config.provenance_table()` lists every path. Every value is decided before resolution ends, including the
|
||||
device offload policy and the checkpoint defaults; after that, `resolved_config` is read-only.
|
||||
|
||||
## Performance Optimization
|
||||
|
||||
For configuring optimizations, please see our [optimizations guide](optimizations.md)
|
||||
|
||||
@@ -66,7 +66,7 @@ as above.
|
||||
|
||||
For exports without this sidecar, an explicit ladder is supported via
|
||||
`MiniMaxH3PipelineConfig.dmd_denoising_steps`, or through the typed API's
|
||||
`PipelineSelection(dmd_denoising_steps=[...])`. The shifts
|
||||
`PipelineSelection(experimental={"dmd_denoising_steps": [...]})`. The shifts
|
||||
still come from the checkpoint scheduler configs. Keep generic `flow_shift`
|
||||
unset: H3 has separate video and audio shifts, not one shared shift.
|
||||
|
||||
@@ -74,6 +74,130 @@ This documents execution support for the published checkpoint. It is not a
|
||||
quality claim: compare video/audio output against base MiniMax-H3 on your own
|
||||
prompts before adopting it.
|
||||
|
||||
## Ref2VA PDD students
|
||||
|
||||
A Parallel Decoding Distillation (PDD) student widens the transformer's two
|
||||
output projections to `pdd_steps` heads, one per interval of a fixed fine
|
||||
time grid on `[0, 0.999]`. Each transformer forward fuses a block of
|
||||
consecutive heads into their integration-weighted mean, and an ordinary Euler
|
||||
step over the block's two node sigmas applies it. The FastH3 OmniRef PDD-8
|
||||
student is a Ref2VA (`transformer_ref`) student with 32 heads, sampled in
|
||||
eight blocks of four.
|
||||
|
||||
Its export carries only what distillation changed:
|
||||
|
||||
| Path | Contents |
|
||||
| --- | --- |
|
||||
| `transformer_ref/` | Widened `proj_out` and `audio_proj_out`, trained VSA compression gates; `config.json` records `pdd_steps` |
|
||||
| `scheduler/`, `audio_scheduler/` | Video and audio shifts (12 and 3) |
|
||||
| `fastvideo_inference.json` | The sampling contract below |
|
||||
| `modular_model_index.json` | The Diffusers manifest |
|
||||
|
||||
The text encoder, tokenizer, processor, and both VAEs are base MiniMax-H3's.
|
||||
`basic_fasth3_omniref_pdd.py` composes the two into one local directory of
|
||||
symlinks and runs the recipe the contract records. `--model-path` is the
|
||||
export, as a local directory or a Hugging Face repo id. The base comes from
|
||||
the revision the contract pins in `base_model_revision`, or from
|
||||
`--base-model-path`:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_fasth3_omniref_pdd.py \
|
||||
--model-path <local export directory or Hugging Face repo id> \
|
||||
--video reference.mp4 --image character.png \
|
||||
--prompt 'The dancer from the video performs the routine in the pictured outfit.' \
|
||||
--height 480 --width 832 --num-frames 124 \
|
||||
--output outputs/fasth3-omniref-pdd
|
||||
```
|
||||
|
||||
References are ordered: pass `--image`, `--video`, and `--audio` in the order
|
||||
the prompt refers to them. At least one image or video is required.
|
||||
|
||||
A PDD export uses `fasth3-inference-contract-v1` with PDD fields in place of
|
||||
the DMD ladder:
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": "fasth3-inference-contract-v1",
|
||||
"model_type": "ref2va",
|
||||
"transformer_component": "transformer_ref",
|
||||
"pdd_steps": 32,
|
||||
"pdd_step_indices": [0, 4, 8, 12, 16, 20, 24, 28, 32],
|
||||
"num_inference_steps": 8,
|
||||
"transformer_forwards": 8,
|
||||
"grid_max_t": 0.999,
|
||||
"video_scheduler_shift": 12.0,
|
||||
"audio_scheduler_shift": 3.0,
|
||||
"guidance_scale": 1.0,
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"vsa_sparsity": 0.9,
|
||||
"vsa_tile_size": 128,
|
||||
"vsa_ref_policy": "p2_multi_region",
|
||||
"vsa_ref_keep_rate": 0.1
|
||||
}
|
||||
```
|
||||
|
||||
The export also records `schema`, `conditioning`, and `base_model_revision`,
|
||||
the base snapshot it was distilled against, as `hf://<repo id>@<revision>`
|
||||
(for example `hf://MiniMaxAI/MiniMax-H3@<commit>`); any other form is an
|
||||
error. FastVideo reads the file once, when the run's arguments are built and
|
||||
before any weights load. The file must carry exactly these fields: a missing
|
||||
or unknown field is an error. Fields that repeat a value stored elsewhere must
|
||||
equal it: `pdd_steps` equals `transformer_ref/config.json`, the shifts equal
|
||||
the scheduler configs, and `num_inference_steps` and `transformer_forwards`
|
||||
equal the block count of `pdd_step_indices`, a strictly increasing partition
|
||||
of the grid from 0 to `pdd_steps`. Only `MiniMaxH3Ref2VAModularPipeline` runs
|
||||
the export. A `transformer_ref` whose `config.json` sets `pdd_steps` without
|
||||
this file is an error.
|
||||
|
||||
Each sampling setting has one source, the file:
|
||||
|
||||
| Setting | Run leaves it unset | Run sets a different value |
|
||||
| ------------------------------- | ---------------------- | ---------------------------------------------- |
|
||||
| `pdd_step_indices` | Taken from the file | Error |
|
||||
| `num_inference_steps` (request) | Set to the block count | Error |
|
||||
| `attention_backend` | Taken from the file | Error, including `FASTVIDEO_ATTENTION_BACKEND` |
|
||||
| `VSA_tile_size` | Taken from the file | Error |
|
||||
| `VSA_sparsity` | Taken from the file | The run's value, with a warning |
|
||||
| `vsa_ref_keep_rate` | Taken from the file | The run's value, with a warning |
|
||||
|
||||
A request leaves `num_inference_steps` unset only when it is parsed from a
|
||||
mapping or a config file. A `GenerationRequest` built in Python counts every
|
||||
field as set, so it must pass the block count. The trained compression gates
|
||||
of `transformer_ref` load only under `VIDEO_SPARSE_ATTN_H3`, so the attention
|
||||
backend cannot change. `dmd_denoising_steps` must stay unset.
|
||||
|
||||
A MiniMax-H3 checkpoint without PDD fields, such as a DMD export, whose
|
||||
transformer carries VSA compression gates (`to_gate_compress` weights) also
|
||||
runs only with `VIDEO_SPARSE_ATTN_H3`. The pipeline reads the shard index (or
|
||||
the safetensors headers) and rejects any other backend, including automatic
|
||||
selection, before it loads any component.
|
||||
|
||||
### Reference-video sparsity
|
||||
|
||||
Every PDD contract sets `"vsa_ref_policy": "p2_multi_region"`, the only value FastVideo accepts, which means:
|
||||
`VIDEO_SPARSE_ATTN_H3` tiles every reference video as its own sparse region,
|
||||
in place in the packed sequence. Each video query keeps `vsa_ref_keep_rate` of
|
||||
every reference video's tiles and `1 - VSA_sparsity` of the target video's
|
||||
tiles. Text, audio, and image references stay dense. With `VSA_sparsity` 0,
|
||||
every region is dense and the reference keep rate has no effect. Checkpoints
|
||||
other than PDD students keep every conditioning row dense.
|
||||
|
||||
### Hardware
|
||||
|
||||
The contract's 128-token tiles, `(4, 4, 8)`, run only on the sm_100a/sm_103a
|
||||
CUDA block-sparse kernel (B200, B300, GB200, GB300) of a fastvideo-kernel
|
||||
build with the Blackwell VSA extension. There is no Triton fallback for tile
|
||||
128: the backend raises instead. Tile 128 runs eagerly. Regional compile
|
||||
requires 64-token tiles. With `--num-gpus` above 1 the example shards the DiT
|
||||
across the GPUs (FSDP) and splits the sequence across them.
|
||||
|
||||
On GB200, a 480x832, 124-frame request fits on one GPU: the eight forwards
|
||||
take about 54 s, and device memory in use peaks near 103 GiB. A 768x1344,
|
||||
345-frame request with `--num-gpus 4` takes about 36 s for the eight
|
||||
forwards. With the DiT sharded, peak allocated memory is about 66 GiB per GPU
|
||||
(about 95 GiB in use), against 83 GiB (103 GiB) with a full DiT copy on each
|
||||
GPU; the output is identical. Model loading and decoding come on top of this.
|
||||
|
||||
## Apple Silicon
|
||||
|
||||
`mlx_fasth3.py` stays on FastH3 V1 and its uniform AdaLN cache.
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
# FastH3 NVFP4 on RTX PRO 6000 (sm_120)
|
||||
|
||||
This page covers serving the FastH3 NVFP4 checkpoints on one RTX PRO 6000
|
||||
Blackwell (sm_120, 96 GB) with every component resident: the NVFP4 text
|
||||
encoder, the NVFP4 denoiser and an INT8 light VAE. It also records what was
|
||||
measured and tried along the way. Every switch is opt-in and defaults to the
|
||||
existing behavior.
|
||||
|
||||
## Results
|
||||
|
||||
All results are for one RTX PRO 6000 (Modal), a 10.1 s clip at 1344x768
|
||||
(243 frames, 73.6k packed tokens), the light INT8 VAE and warm runs.
|
||||
|
||||
| Checkpoint | Denoise | Video decode | End to end | Peak memory |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| [`FastH3-4-step-Preview-v1-VSA-DataFree-NVFP4`](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree-NVFP4) (4 forwards, VSA 0.9) | 30.1–31.0 s | 9.0 s | **41.2 / 42.4 s** | 79.6 GB |
|
||||
| [`FastH3-8-Step-V2-NVFP4`](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4) (8 forwards, VSA 0.8) | 75.6 s | 9.0 s | **86.5 s** | — |
|
||||
|
||||
These end-to-end runs predate the warp-skip kernel change below, which cuts
|
||||
sparse attention by a further 1.5x, so they are upper bounds. Conditioning
|
||||
takes about 0.12 s, frame post-processing plus MP4 writing about 0.9 s, and
|
||||
the first clip at a new shape about 200 s (VAE compile).
|
||||
|
||||
At 480p (124 frames, 15.1k tokens) V2 8-step denoises in 12.5 s on a cold
|
||||
run, down from 16.8 s warm before this work.
|
||||
|
||||
## Usage
|
||||
|
||||
### 1. Convert the checkpoint
|
||||
|
||||
The published NVFP4 checkpoints use ModelOpt's unified Hugging Face layout.
|
||||
`convert_minimax_h3_modelopt_nvfp4_dit.py` repacks it into FastVideo's packed
|
||||
export, `transformer/nvfp4_weights.safetensors`. The calibrated FFN weights
|
||||
and scales are carried over bit for bit and only the scale bytes are
|
||||
swizzled. Optionally it also quantizes the BF16 attention projections and the
|
||||
VSA compression gates:
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_modelopt_nvfp4_dit.py \
|
||||
--src /models/FastH3-8-Step-V2-NVFP4/transformer \
|
||||
--dst /models/fasth3-v2-fv/transformer \
|
||||
--quantize-attention --quantize-gate
|
||||
```
|
||||
|
||||
Every exported linear is probed through the loader's `mm_fp4` path against a
|
||||
BF16 matmul with the dequantized weight. Conversion refuses to write when any
|
||||
error exceeds `--max-probe-error` (default 0.3; the converted V2 and 4-step
|
||||
checkpoints probe at 0.14). The rest of the model folder (text encoder,
|
||||
VAEs, schedulers, `fastvideo_inference.json`) is used as-is; the
|
||||
NVFP4 text encoder comes from
|
||||
`convert_minimax_h3_text_encoder_nvfp4.py`.
|
||||
|
||||
The flags select the `layer_profile` to load the result with:
|
||||
|
||||
| Flags | Linears in NVFP4 | `layer_profile` |
|
||||
| --- | --- | --- |
|
||||
| *(none)* | FFN `fc_in`/`fc_out` | `h3_dit_ffn` |
|
||||
| `--quantize-attention` | + attention `to_{q,k,v,out}` | `h3_dit` |
|
||||
| `--quantize-attention --quantize-gate` | + VSA `to_gate_compress` | `h3_dit_vsa` |
|
||||
|
||||
### 2. Generate
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ.update({
|
||||
"FASTVIDEO_H3_VSA_FP4": "1", # sparse FP4 attention
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all", # Triton norm/modulate/RoPE/SwiGLU fusions
|
||||
"FASTVIDEO_NVFP4_MM_BACKEND": "cutlass", # see "FP4 GEMM backend" below
|
||||
"FASTVIDEO_H3_VAE_TILE_BATCH": "28", # one decoder call per 1344x768 tile grid
|
||||
})
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": "/models/fasth3-v2-fv",
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"quantization": {"transformer_quant": "NVFP4", "layer_profile": "h3_dit_vsa"},
|
||||
"compile": {"enabled": False, "vae_enabled": True},
|
||||
},
|
||||
"pipeline": {"experimental": {"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"VSA_sparsity": 0.8, "VSA_tile_size": 64}},
|
||||
})
|
||||
generator.generate({"prompt": "...", "sampling": {"height": 768, "width": 1344, "num_frames": 243,
|
||||
"num_inference_steps": 9}})
|
||||
```
|
||||
|
||||
Use `VSA_sparsity` 0.9 and `num_inference_steps` 5 for the 4-step checkpoint
|
||||
(see its `fastvideo_inference.json`). Frame counts must be `17n + 5`: 243
|
||||
frames is the closest to 10 s.
|
||||
|
||||
## What changed
|
||||
|
||||
### Block-sparse FP4 attention for VSA tiles (`fastvideo-kernel`)
|
||||
|
||||
SageAttention3's sm_120 FP4 kernel (`attn_qat_infer`) gains a block-sparse
|
||||
forward, `fwd_sparse`, exposed as `sageattn_blackwell_sparse` (head-major
|
||||
inputs) and `sageattn_blackwell_sparse_bshd` (sequence-major inputs, quantized
|
||||
in place without a transpose). `vsa_tile_mask_to_fp4_blocks` turns a VSA tile
|
||||
mask into the kernel's lists:
|
||||
|
||||
- **Block lists.** Query block `m` visits only the 128-token KV blocks in
|
||||
`q2k_idx[b, h, m, :q2k_num[b, h, m]]`.
|
||||
- **Quadrant masks for 64-token tiles.** The kernel computes on 128x128
|
||||
blocks, but V2 and the 4-step preview use 64-token VSA tiles. Each listed
|
||||
block carries a 4-bit `q2k_quad` (one bit per 64x64 quadrant). The kernel
|
||||
masks unselected quadrants to `-inf`, so the result is exactly VSA's tile-64
|
||||
semantics.
|
||||
- **Valid counts per 64-column half.** `kv_valid` gives the valid tokens in
|
||||
each 64-column half, so partially filled tiles can pad mid-block.
|
||||
- **Warp-level skipping.** Each MMA warp owns 16 query rows and so sits
|
||||
inside one 64-row half. A warp skips a listed block that its half did not
|
||||
select, and the P·V chunk of a key half it did not select. Masked scores
|
||||
contribute exactly zero, and the warp sharing its tensor-core partition runs
|
||||
faster meanwhile. This recovers most of the work that pairing 64-token tiles
|
||||
into 128-token blocks adds.
|
||||
- **First-visited block.** Lists run in descending block order because the
|
||||
kernel visits them last entry first. Block 0 (the first prefix tile, which
|
||||
VSA-H3's exempt mode gives every query) is therefore visited first, so every
|
||||
row starts from a finite running max. `validate=True` checks this. The model
|
||||
integration uses only exempt mode.
|
||||
|
||||
The dense and sparse entry points also stop allocating `delta_s`. With Q
|
||||
smoothing off (the default), each call used to allocate and zero a
|
||||
`[B, H, L/128, L]` fp32 tensor: 9.5 GB at 73k tokens. Its int32 batch stride
|
||||
also overflowed the TMA descriptor ("Failed to initialize the TMA descriptor
|
||||
1", then an illegal instruction), so FP4 attention could not run 10 s 1344x768
|
||||
clips at all. A cached `[B, H, 1, L]` zero row read with `per_block_mean=False`
|
||||
replaces it, and the outputs are bit-identical.
|
||||
|
||||
Correctness (`fastvideo-kernel/tests/test_attn_qat_infer_sparse.py`, RTX PRO
|
||||
6000):
|
||||
|
||||
| Layout | Error vs token-masked fp32 | Dense FP4 floor |
|
||||
| --- | ---: | ---: |
|
||||
| 64-token tiles, odd count, partial tiles | 0.190 | 0.191 |
|
||||
| 64-token tiles, even count | 0.191 | 0.193 |
|
||||
| 256-token tiles, partial tile | 0.195 | 0.196 |
|
||||
|
||||
Errors are relative L2. The sparse kernel sits exactly at the FP4 noise
|
||||
floor; random Gaussian inputs make that floor large. Full block lists
|
||||
reproduce the dense kernel bit for bit.
|
||||
|
||||
### Model integration (`FASTVIDEO_H3_VSA_FP4=1`)
|
||||
|
||||
`fastvideo/models/dits/minimax_h3_vsa_fp4.py` replaces only the attention core
|
||||
of `MiniMaxH3Attention`. VSA-H3's tile pooling, top-k mask, exempt prefix and
|
||||
gated compression branch are unchanged. Per block, it:
|
||||
|
||||
1. Gathers the attention input into tile order once (one `hidden_size`-wide
|
||||
pass). Pad rows stay zero, so the q/k/v pad rows are exactly zero through
|
||||
the bias-free projections, RMSNorm and RoPE.
|
||||
2. Quantizes that input once and shares it between `to_q`, `to_k` and `to_v`.
|
||||
NVFP4 activations use a unit global scale, so this is exact.
|
||||
3. Applies QK-norm and RoPE with tile-ordered `cos`/`sin`, computed once per
|
||||
step.
|
||||
4. Runs the sparse FP4 kernel on sequence-major tensors and gathers the output
|
||||
back to packed order before `to_out`.
|
||||
|
||||
This replaces the generic path's concat, four tile scatters and three
|
||||
transposes. The route applies only to no-grad, non-compiled, single
|
||||
sequence-parallel-rank calls in exempt mode; everything else keeps the
|
||||
existing path.
|
||||
|
||||
Two smaller pieces ship alongside it:
|
||||
|
||||
- **Packed gate check.** With `--quantize-gate`, `to_gate_compress` loses its
|
||||
BF16 weight, so the gate-activity check reads the packed E2M1 bytes instead.
|
||||
- **Compiled residual.** With `FASTVIDEO_MINIMAX_H3_FUSIONS` enabled, each
|
||||
block's final `hidden + gate[indices] * ffn_out` runs as one compiled op
|
||||
instead of materializing the gathered gate.
|
||||
|
||||
### FP4 GEMM backend (`FASTVIDEO_NVFP4_MM_BACKEND`)
|
||||
|
||||
FlashInfer's `mm_fp4(backend="auto")` picks a kernel about 2x slower than
|
||||
`cutlass` or `cudnn` on sm_120 once activations reach tens of thousands of
|
||||
rows. At 15k rows all three match.
|
||||
|
||||
| Linear | 73.6k rows: `auto` | 73.6k rows: `cutlass` | 73.6k rows: `cudnn` | 15.1k rows: `auto` |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| `to_q` (5376→7168) | 7.96 ms | 3.96 ms | 4.24 ms | 0.88 ms |
|
||||
| `to_out` (7168→5376) | 9.07 ms | 4.09 ms | 4.27 ms | 0.92 ms |
|
||||
| `fc_in` (5376→28672) | 21.54 ms | 16.59 ms | 16.17 ms | 3.01 ms |
|
||||
| `fc_out` (14336→5376) | 18.06 ms | 8.09 ms | 8.43 ms | 1.65 ms |
|
||||
|
||||
### Batched VAE tile decode (`FASTVIDEO_H3_VAE_TILE_BATCH`)
|
||||
|
||||
The H3 video VAE decodes 256-pixel spatial tiles one at a time. A 1344x768
|
||||
clip is a 4x7 grid per temporal chunk, so a 10 s clip is roughly 400 small
|
||||
decoder calls. The ViT decoder treats batch entries independently, so
|
||||
`FASTVIDEO_H3_VAE_TILE_BATCH=N` decodes up to `N` equal-shaped tiles per call;
|
||||
28 covers a full 1344x768 grid. The decoded tiles are the same as per-tile
|
||||
decoding.
|
||||
|
||||
## Per-block measurements
|
||||
|
||||
One H3 transformer block (hidden 5376, 56 heads, FFN 14336), RTX PRO 6000:
|
||||
|
||||
| Component | 480p, 124 f (15.1k tokens) | 768p, 243 f (73.6k tokens) |
|
||||
| --- | ---: | ---: |
|
||||
| VSA Triton BF16 attention (kernel + pooling/mask) | 14.8 ms | 180.1 ms |
|
||||
| Tile scatter of q/k/v/gate + gather (generic path) | 2.7 ms | 12.7 ms |
|
||||
| Dense FP4 attention (SageAttention3) | 10.5 ms | 211.9 ms |
|
||||
| Sparse FP4, quadrant masks | 7.9 ms | 123.4 ms |
|
||||
| Sparse FP4, quadrant masks + warp skip | — | **83.0 ms** (VSA 0.8) / **49.3 ms** (VSA 0.9) |
|
||||
| Dense BF16 SDPA | 18.1 ms | — |
|
||||
| Modulation: eager / fused / compiled | 4.07 / 1.42 / 0.58 ms | 20.3 / 7.0 / 2.8 ms |
|
||||
| SwiGLU: eager / fused | 1.53 / 0.88 ms | 7.37 / 4.24 ms |
|
||||
| QK-norm + RoPE: eager / fused | 4.64 / 1.12 ms | 22.4 / 5.2 ms |
|
||||
|
||||
Before this work, a 480p block cost about 40 ms: 17.7 ms of attention, 12 ms
|
||||
of linears and 10 ms of eager elementwise ops. Over 50 blocks that is 2.0 s
|
||||
per step, which matches the measured 2.1 s.
|
||||
|
||||
Block density each kernel granularity computes at 768p (fraction of the
|
||||
dense attention). "Selected" is what VSA needs; the other columns are what
|
||||
each block shape computes:
|
||||
|
||||
| VSA sparsity | Selected (64x64) | 128x128 blocks | 64-row x 128 | 128 x 64-col |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| 0.8 | 0.222 | 0.434 | 0.317 | 0.313 |
|
||||
| 0.9 | 0.125 | 0.254 | 0.180 | 0.177 |
|
||||
|
||||
## What was tried and not shipped
|
||||
|
||||
- **Dense FP4 attention for VSA-trained students.** `ATTN_QAT_INFER` does not
|
||||
build `to_gate_compress`. A VSA-distilled checkpoint such as V2 carries
|
||||
trained gates, so the strict loader refuses it ("Parameter
|
||||
...to_gate_compress.weight not found"). Dense FP4 attention is also slower
|
||||
than sparse FP4 at 768p (212 vs 83 ms per block).
|
||||
- **Multi-GPU (Ulysses) FP8 exchange.** On 8x RTX PRO 6000 (PCIe only), NCCL
|
||||
all-to-all moves about 21 GB/s per GPU, with NCCL P2P on or off. A BF16
|
||||
q/k/v/gate exchange at 73.6k tokens therefore costs 24 ms per block, and the
|
||||
attention output another 6.6 ms; with FP4 payloads q/k/v/gate drop to 7.2 ms.
|
||||
The branch `h3-sm120-experimental` keeps a sequence-parallel path that:
|
||||
- sends q/k/v as FP8 with one scale per token and head;
|
||||
- never sends the VSA gate, applying it on each rank after a small
|
||||
all-gather of the per-tile compression output;
|
||||
- returns the attention output as FP8.
|
||||
|
||||
It is estimated at about 18–20 s per 10 s clip for V2 8-step on 8 GPUs. It
|
||||
has not executed yet (8-GPU capacity was unavailable), so it is not part of
|
||||
this change. The same branch holds the Modal drivers behind every number on
|
||||
this page.
|
||||
- **64-row query blocks.** The kernel traits allow `kBlockM = 64`, which would
|
||||
remove the query-side pairing waste. Warp-level skipping recovers most of
|
||||
that waste without a second kernel instantiation, so it was not built.
|
||||
|
||||
## Known limitations
|
||||
|
||||
- **End-to-end quality.** The kernel matches a masked reference at the FP4
|
||||
noise floor. Generated videos have not yet been A/B-compared against the
|
||||
BF16 Triton VSA path on the H3 audio/video metrics.
|
||||
- **Activation scales.** The packed export drops ModelOpt's calibrated static
|
||||
`input_scale` and quantizes activations with a unit global scale and dynamic
|
||||
per-16 block scales, as FastVideo's NVFP4 linears do elsewhere.
|
||||
- **Decode cost.** Video decode (9 s at 10 s/768p with the light INT8 VAE) is
|
||||
the next largest cost after denoising.
|
||||
- **Hardware.** Everything here targets sm_120. The GeForce RTX 5090 shares
|
||||
the architecture but has 32 GB, which needs a reduced-AdaLN checkpoint and a
|
||||
non-resident text encoder at this resolution.
|
||||
@@ -38,20 +38,18 @@ def main():
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{"engine": {"num_gpus": 1}}, # Adjust based on your hardware
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
# Define a prompt for your video
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
|
||||
|
||||
# Generate the video
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -77,24 +75,23 @@ Please see the [support matrix](support_matrix.md) for the list of supported mod
|
||||
You can generate a video starting from an initial image:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo import VideoGenerator, SamplingParam
|
||||
|
||||
def main():
|
||||
# Create the generator
|
||||
model_name = "Wan-AI/Wan2.1-I2V-14B-480P-Diffusers"
|
||||
generator = VideoGenerator.from_pretrained(model_name, {"engine": {"num_gpus": 1}})
|
||||
generator = VideoGenerator.from_pretrained(model_name, num_gpus=1)
|
||||
|
||||
# Set up parameters with an initial image
|
||||
sampling_param = SamplingParam.from_pretrained(model_name)
|
||||
sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
sampling_param.num_frames = 107
|
||||
|
||||
# Generate video based on the image
|
||||
prompt = "A photograph coming to life with gentle movement"
|
||||
generator.generate({
|
||||
"prompt": prompt,
|
||||
# Set up parameters with an initial image
|
||||
"inputs": {
|
||||
"image_path": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg",
|
||||
},
|
||||
"sampling": {"num_frames": 107},
|
||||
"output": {"output_path": "my_videos/", "save_video": True},
|
||||
})
|
||||
generator.generate_video(prompt, sampling_param=sampling_param,
|
||||
output_path="my_videos/",
|
||||
save_video=True)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -109,12 +106,12 @@ Common issues and their solutions:
|
||||
If you encounter CUDA out of memory errors:
|
||||
|
||||
- Reduce `num_frames` or video resolution
|
||||
- Enable FastVideo offloading options such as `engine.offload.dit_layerwise: true`
|
||||
(single GPU) or `engine.use_fsdp_inference: true` (multi-GPU)
|
||||
- Enable FastVideo offloading options such as `dit_layerwise_offload=True`
|
||||
(single GPU) or `use_fsdp_inference=True` (multi-GPU)
|
||||
- Try a smaller model or use distilled versions
|
||||
- Use `engine.num_gpus` > 1 if multiple GPUs are available
|
||||
- Try enabling FSDP inference with `engine.use_fsdp_inference: true` (may slow down generation)
|
||||
- Try enabling DiT layerwise offload with `engine.offload.dit_layerwise: true` (now only a few models support this, but may introduce less overhead than FSDP)
|
||||
- Use `num_gpus` > 1 if multiple GPUs are available
|
||||
- Try enabling FSDP inference with `use_fsdp_inference=True` (may slow down generation)
|
||||
- Try enabling DiT layerwise offload with `dit_layerwise_offload=True` (now only a few models support this, but may introduce less overhead than FSDP)
|
||||
|
||||
### Slow Generation
|
||||
|
||||
|
||||
@@ -0,0 +1,107 @@
|
||||
# Kandinsky 6 Text/Image to Video with Audio (T2IVA)
|
||||
|
||||
`Kandinsky6TI2VAPipeline` generates a video and a synchronized audio track from a text prompt, optionally conditioned
|
||||
on an image: one pipeline serves both, so pass `image_path` to condition on an image and leave it out for text only.
|
||||
The audio is decoded by the checkpoint's audio VAE (mel decoder plus vocoder in one component) and muxed into the saved
|
||||
mp4 automatically. To upscale a generated clip, see [Kandinsky 6 Video SR](kandinsky6_sr.md).
|
||||
|
||||
## Models
|
||||
|
||||
Both variants are official Diffusers repos, loaded directly through their `model_index.json`:
|
||||
|
||||
| Variant | Hub repo | Scheduler | Steps | Guidance | Example |
|
||||
|---|---|---|---|---|---|
|
||||
| T2IVA | `kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers` | `FlowMatchEulerDiscreteScheduler` (shift 5.0) | 50 | 5.0 | [`basic_kandinsky6_ti2va.py`](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
| T2IVA distilled | `kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers` | `PiflowScheduler` (`n_grid` 10, shift 5.0) | 10 | 1.0 | [`basic_kandinsky6_ti2va.py`](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
|
||||
The steps and guidance columns are the defaults of the preset the registry selects for each repo id. Everything else is
|
||||
shared: 512x768, 121 frames (5 s at 24 fps) and the Diffusers default negative prompt (only used when
|
||||
`guidance_scale > 1`). The distilled preset is named `kandinsky6_ti2va_distilled`.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_kandinsky6_ti2va.py
|
||||
KANDINSKY6_MODEL_PATH=kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers \
|
||||
python examples/inference/basic/basic_kandinsky6_ti2va.py
|
||||
```
|
||||
|
||||
Set `IMAGE_PATH` in the script to condition on an image.
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers",
|
||||
num_gpus=1,
|
||||
dit_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
)
|
||||
generator.generate_video(
|
||||
"cinematic shot: a giant stone samurai on a stormy cliff above a neon city opens glowing golden eyes and "
|
||||
"raises a katana. Blue lightning strikes the blade, creating a massive shockwave through the clouds. The "
|
||||
"camera rapidly pulls back from a low angle. Photorealistic, epic scale, dark blue and gold lighting, rain, "
|
||||
"sparks, volumetric lightning, blockbuster quality. Audio: heavy rain, deep thunder, metallic sword hum, "
|
||||
"rising brass and choir, electrical crackle, perfectly synchronized lightning impact, sub-bass shockwave. "
|
||||
"No dialogue, text, or logos.",
|
||||
image_path=None, # or the path of a conditioning image
|
||||
output_path="video_samples_kandinsky6_ti2va",
|
||||
height=512,
|
||||
width=768,
|
||||
num_frames=121,
|
||||
)
|
||||
```
|
||||
|
||||
A local copy of a repo works the same way. A local directory is treated as the distilled variant only when its name is
|
||||
a Kandinsky-6 name containing `distill` (for example `Kandinsky-6.0-Pro-distill-5s-Diffusers`); any other directory name
|
||||
selects the base preset (see below).
|
||||
|
||||
## Distilled (pi-Flow) checkpoint
|
||||
|
||||
- The distilled repo replaces the flow-matching scheduler with `PiflowScheduler` (`n_grid` 10, `eps` 1e-6,
|
||||
`final_step_size_scale` 0.5, `num_policy_substeps` 128). Its DiT emits `n_grid` predictions per latent channel
|
||||
(`out_visual_dim` 160 = 16 x 10, `out_audio_dim` 400 = 40 x 10).
|
||||
- pi-Flow runs without classifier-free guidance: `guidance_scale` must be exactly `1.0`. The official Diffusers
|
||||
pipeline rejects any other guidance for a `PiflowScheduler` too; FastVideo raises a `ValueError` naming the
|
||||
required value. `num_inference_steps` is not constrained by the checkpoint -- `PiflowScheduler.set_timesteps`
|
||||
accepts any step count and ignores the scheduler's `nfe`; `10` is only the value the distilled checkpoint was
|
||||
trained for and the `kandinsky6_ti2va_distilled` preset's default.
|
||||
- The `kandinsky6_ti2va_distilled` preset (10 steps, guidance 1.0) is the default for the distilled repo id and for
|
||||
local directories named like it. Set `KANDINSKY6_MODEL_PATH` to either one when running the shared
|
||||
`basic_kandinsky6_ti2va.py` example. A distilled copy under another directory name selects the base preset, so
|
||||
retain `Kandinsky-6.0-Pro-distill-5s-Diffusers` as the final directory name.
|
||||
- The policy values can be overridden on `Kandinsky6TI2VAConfig` (`piflow_eps`, `piflow_final_step_size_scale`,
|
||||
`piflow_num_policy_substeps`); `None` keeps the values from `scheduler_config.json`.
|
||||
|
||||
## Differences from the Diffusers pipeline
|
||||
|
||||
A few Diffusers pipeline options are not ported, and are surfaced here instead of as a knob that would silently do
|
||||
nothing:
|
||||
|
||||
- `sample_audio=False` (video-only, no audio stream) is not exposed; FastVideo's DiT raises `NotImplementedError` for
|
||||
a partial-modality call instead of denoising video alone.
|
||||
- `expand_prompts` (the Qwen prompt-beautifier pass) is not ported; FastVideo's `PromptEnhancerConfig` is a separate,
|
||||
external (Cerebras/Groq streaming) feature, not this pipeline's built-in expansion.
|
||||
- Of the Diffusers reference's `visual_cond_scheme` values, only `tail_cond_first_frame` (append one clean reference
|
||||
frame to the end of the sequence) is implemented; `pretrain` and plain `i2v` are not.
|
||||
- Precomputed `prompt_embeds`/`negative_prompt_embeds` are not accepted; every call encodes its own prompt text.
|
||||
- MagCache is not ported: a `magcache` block in a checkpoint's `transformer/config.json` is parsed and dropped
|
||||
by `update_model_arch`, not read automatically or exposed as an opt-in cache config.
|
||||
- RNG differs: FastVideo draws video then audio noise from one per-request CPU generator seeded by `seed` (default
|
||||
1024); Diffusers seeds a device generator from a value drawn out of `generator` (audio uses `seed+1`). The same
|
||||
numeric seed produces different noise on the two stacks -- pass `latents`/`audio_latents` directly for bit-level
|
||||
comparisons.
|
||||
- Qwen prompt tokens are unpadded and carry no attention mask (Diffusers pads to a fixed length and masks the
|
||||
padding in text self/cross-attention). Mathematically equivalent for a single prompt (measured 2e-7 relative
|
||||
difference), but FastVideo has no attention-mask plumbing, so a hand-built batch of unequal-length prompts is not
|
||||
supported.
|
||||
- Attention is dense (`LocalAttention`, flash/SDPA); NABLA sparse attention is wired but unverified against a real
|
||||
NABLA-flagged checkpoint. This matches the Diffusers pipeline itself, which never enables NABLA either.
|
||||
- VAE tiling is on by default (`vae_tiling=True`); Diffusers decodes untiled.
|
||||
|
||||
## Memory
|
||||
|
||||
The Pro DiT has 30.1B parameters, about 60 GB in bf16 (`dit_precision` defaults to `bf16`), and the Qwen2.5-VL text
|
||||
encoder adds 16.6 GB. FastVideo enables `dit_cpu_offload` by default; the examples turn it off (`dit_cpu_offload=False`)
|
||||
to keep the DiT resident on the GPU and offload the text encoder instead (`text_encoder_cpu_offload=True`). See
|
||||
[Offloading](offloading.md) for the memory knobs.
|
||||
@@ -0,0 +1,83 @@
|
||||
# Kandinsky 6 Video Super-Resolution
|
||||
|
||||
`Kandinsky6SRPipeline` upscales a low-resolution video by x2, x2.25 or x4 (video-to-video, no text prompt). The clip is
|
||||
encoded once with the SR VAE (KVAE); its latent is cut into overlapping tiles, each tile is enlarged by the *latent
|
||||
upscaler*, refined by a text-free SR DiT in a few denoising steps and decoded, and the tiles are blended back together.
|
||||
The source audio is kept. The clip can come from anywhere, for example from the [Kandinsky 6 T2IVA](kandinsky6.md)
|
||||
pipeline.
|
||||
|
||||
## Models
|
||||
|
||||
| Variant | Hub repo | Scheduler | Default steps per tile |
|
||||
|---|---|---|---|
|
||||
| Flow matching | `kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers` | `FlowMatchEulerDiscreteScheduler` (shift 5.0) | 4 |
|
||||
| Distilled | `kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers` | `PiflowScheduler` (shift 3.5, `n_grid` 10) | 2 |
|
||||
|
||||
Both repos have the same layout and share `vae/` and `latent_upscaler/`:
|
||||
|
||||
```text
|
||||
model_index.json _class_name = Kandinsky6SRPipeline
|
||||
transformer/ Kandinsky6SRTransformer3DModel (sr_params: trained base resolution, RoPE scale, noise level)
|
||||
vae/ Kandinsky6SRVAE
|
||||
latent_upscaler/ Kandinsky6SRLatentUpscalerBank (x2 and x4 models)
|
||||
scheduler/ FlowMatchEulerDiscreteScheduler or PiflowScheduler
|
||||
```
|
||||
|
||||
The scheduler component drives the denoising loop, and each repo resolves to its own preset (4 or 2 steps). The
|
||||
distilled transformer's head holds `n_grid` predictions per latent channel, which `PiflowScheduler` integrates.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_kandinsky6_sr.py --video-path input.mp4 --scale 2.25
|
||||
python examples/inference/basic/basic_kandinsky6_sr.py --video-path input.mp4 \
|
||||
--model-path kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers
|
||||
```
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained("kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers", num_gpus=1)
|
||||
result = generator.generate({
|
||||
"inputs": {"video_path": "input.mp4"},
|
||||
"output": {"output_path": "outputs_video/sr", "return_frames": False},
|
||||
"extensions": {"sr_resolution_scale": 2.25, "sr_target_resolution": "fullhd"},
|
||||
})
|
||||
```
|
||||
|
||||
The output geometry and frame rate follow the input clip; the request's `height` / `width` / `num_frames` are not used.
|
||||
Clips are resampled to 24 fps by a fixed stride and only the first 121 frames (5 s, `1 + 8k`-aligned) are processed.
|
||||
The source audio (mono, 44.1 kHz) is trimmed to the processed span and muxed into the output.
|
||||
|
||||
## Request parameters
|
||||
|
||||
The `sr_*` options belong only to Kandinsky6 SR. Pass them in `request.extensions` (as above), or under
|
||||
`request.stage_overrides.sr`. The SR pipeline reads them from `ForwardBatch.extra`; they are not fields of the
|
||||
shared `SamplingParam` or `ForwardBatch`. Other model families reject these options. Existing
|
||||
`generate_video(..., sr_resolution_scale=...)` calls remain supported for Kandinsky6 SR; replace direct
|
||||
`SamplingParam(sr_...=...)` construction with request extensions or these keyword arguments.
|
||||
For the config-based CLI, use dotted overrides such as `--request.extensions.sr_resolution_scale 4`, rather than
|
||||
shared `--sr-*` flags. The model-specific example above keeps its `--scale` and `--tiles-batch-size` flags.
|
||||
|
||||
| Field | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `num_inference_steps` | 4 / 2 | Denoising steps (DiT calls) per tile. The upstream Diffusers pipeline counts grid points instead (its 5 is 4 steps here). The distilled model was trained for 2; other values run with a warning. |
|
||||
| `sr_resolution_scale` | `2.25` | Total upscale: `2`, `4` or `2.25` (x1.125 pixel pre-upscale, then x2). |
|
||||
| `sr_tiles_batch_size` | `1` | Tiles denoised per DiT call (raise it only if memory allows). |
|
||||
| `sr_tile_min_overlap` | `0.20` | Minimum overlap between neighbouring tiles, as a fraction of the tile. |
|
||||
| `sr_target_resolution` | `None` | Downscale the result to `hd`, `fullhd`, `2k` or `WxH`. |
|
||||
| `sr_target_resize_mode` | `fit` | `fit` keeps the aspect ratio, `exact` uses the bucket dimensions. |
|
||||
| `seed` | `42` | Tile group *k* (of `sr_tiles_batch_size` tiles) is seeded with `seed + index of its first tile`. |
|
||||
|
||||
Two Python-only inputs take raw tensors and are passed as keyword arguments of `generate_video()`:
|
||||
|
||||
- `sr_lr_latent`: an unscaled KVAE latent `[T, C, H, W]` of the source, instead of `video_path` (skips decoding and
|
||||
encoding the video; scale 2 or 4 only, since 2.25 needs the pixel pre-upscale).
|
||||
- `sr_audio` / `sr_audio_sample_rate`: a mono waveform in `[-1, 1]` to mux instead of the source's own audio.
|
||||
|
||||
## Limitations
|
||||
|
||||
- NABLA block-sparse attention, which both repos request for 512-pixel tiles, is not wired; the DiT runs dense attention
|
||||
and logs a warning.
|
||||
- No sequence or tensor parallelism: the DiT runs on one GPU. Tiles are processed one group after another.
|
||||
- One clip per request.
|
||||
@@ -4,17 +4,15 @@ This page describes how to use offloading techniques for inference to reduce GPU
|
||||
|
||||
## Default Behavior
|
||||
|
||||
```yaml
|
||||
engine:
|
||||
use_fsdp_inference: false
|
||||
offload:
|
||||
dit: true # dit_cpu_offload
|
||||
dit_layerwise: true # dit_layerwise_offload
|
||||
text_encoder: true # text_encoder_cpu_offload
|
||||
image_encoder: true # image_encoder_cpu_offload
|
||||
vae: true # vae_cpu_offload
|
||||
pin_cpu_memory: true
|
||||
lazy_module_load: null # auto
|
||||
```python
|
||||
dit_cpu_offload: bool = True
|
||||
use_fsdp_inference: bool = False
|
||||
dit_layerwise_offload: bool = True
|
||||
text_encoder_cpu_offload: bool = True
|
||||
image_encoder_cpu_offload: bool = True
|
||||
vae_cpu_offload: bool = True
|
||||
pin_cpu_memory: bool = True
|
||||
lazy_module_load: bool | None = None
|
||||
```
|
||||
|
||||
On unified-memory accelerators such as NVIDIA GB10 and Apple silicon, FastVideo
|
||||
@@ -37,8 +35,8 @@ channels, DiT patch size) so those stages do not materialize weights just to
|
||||
read two integers. The MLX FastH3 runtime always uses this phase order. When
|
||||
host offload is off, DiT safetensors are read onto the accelerator instead of
|
||||
CPU-then-copy. Both flags default to auto (`None`) and turn on for
|
||||
unified-memory devices such as GB10; lazy then disables sequential. Set
|
||||
`engine.offload.lazy_module_load: false` to keep every component resident (sequential may still
|
||||
unified-memory devices such as GB10; lazy then disables sequential. Pass
|
||||
`--no-lazy-module-load` to keep every component resident (sequential may still
|
||||
auto-arm). Two-node Spark
|
||||
jobs still need this split: sequence parallel replicates the DiT on each GB10
|
||||
(~66 GiB of weights plus activations). See
|
||||
@@ -47,10 +45,7 @@ jobs still need this split: sequence parallel replicates the DiT on each GB10
|
||||
## Behavior Explanation
|
||||
|
||||
!!! note
|
||||
`VideoGenerator.from_pretrained` accepts the option names below as keywords, except `lazy_module_load` and
|
||||
`h3_sequential_load`. In a YAML config or a dotted override, each option is a typed field:
|
||||
`engine.use_fsdp_inference`, `engine.offload.<field>` as listed in the defaults above, and
|
||||
`pipeline.model.minimax_h3.sequential_load` for `h3_sequential_load`.
|
||||
For CLI usage, replace underscores (`_`) with hyphens (`-`).
|
||||
|
||||
### `use_fsdp_inference`
|
||||
|
||||
@@ -109,8 +104,8 @@ because the encoder has been released.
|
||||
Leave the default on Spark / DGX Spark when `lazy_module_load` is off. When
|
||||
both would arm (the GB10 auto case), lazy owns deferral and sequential stands
|
||||
down so VAE `torch.compile` can attach to the lazy proxy. Force
|
||||
`pipeline.model.minimax_h3.sequential_load: true` only when you need the split on a discrete GPU without
|
||||
lazy load. Set `pipeline.model.minimax_h3.sequential_load: false` when you need more than one prompt per
|
||||
`--h3-sequential-load` only when you need the split on a discrete GPU without
|
||||
lazy load. Use `--no-h3-sequential-load` when you need more than one prompt per
|
||||
worker and have enough memory to keep the encoder.
|
||||
|
||||
### `text_encoder_cpu_offload`
|
||||
@@ -165,7 +160,7 @@ options above cannot help with because they act after loading. It is
|
||||
particularly relevant on unified-memory devices, where host and device draw on
|
||||
the same pool and moving weights to the host frees nothing. FastVideo
|
||||
auto-enables it there (`lazy_module_load=None`). Leave it off when the model
|
||||
already fits, or set `engine.offload.lazy_module_load: false` to keep components resident for
|
||||
already fits, or pass `--no-lazy-module-load` to keep components resident for
|
||||
later `generate()` calls.
|
||||
|
||||
This option applies to inference only. Training keeps every component resident
|
||||
@@ -173,8 +168,8 @@ and logs a warning if the flag is set.
|
||||
|
||||
Deferral is opt-in per pipeline. Releasing a component and loading it again is
|
||||
only safe when nothing outside the loader has changed it, and two common habits
|
||||
break that without raising: mutating a component after load, as LongCat does
|
||||
when it enables block-sparse attention, and reading a component's attributes
|
||||
break that without raising: mutating a component after load, such as enabling
|
||||
a sparse-attention mode in `initialize_pipeline`, and reading a component's attributes
|
||||
while stages are built, as the shared denoising stage does to pick an attention
|
||||
backend. A pipeline therefore lists the components it has checked in
|
||||
`_lazy_module_names`, which is empty in the base class. MiniMax-H3 opts in. On
|
||||
@@ -208,25 +203,19 @@ from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"offload": {
|
||||
# Recommended for single GPU
|
||||
"dit_layerwise": True,
|
||||
# Enable if OOM happens
|
||||
"vae": True,
|
||||
"image_encoder": True,
|
||||
"text_encoder": True,
|
||||
# Speeds up CPU-GPU transfer
|
||||
"pin_cpu_memory": True,
|
||||
},
|
||||
},
|
||||
},
|
||||
num_gpus=1,
|
||||
# Recommended for single GPU
|
||||
dit_layerwise_offload=True,
|
||||
# Enable if OOM happens
|
||||
vae_cpu_offload=True,
|
||||
image_encoder_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
# Speeds up CPU-GPU transfer
|
||||
pin_cpu_memory=True,
|
||||
)
|
||||
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers."
|
||||
video = generator.generate({"prompt": prompt, "output": {"output_path": "output/", "save_video": True}})
|
||||
video = generator.generate_video(prompt, output_path="output/", save_video=True)
|
||||
```
|
||||
|
||||
### Multi-GPU with FSDP
|
||||
@@ -236,24 +225,18 @@ from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 2,
|
||||
# Recommended for multi-GPU
|
||||
"use_fsdp_inference": True,
|
||||
"offload": {
|
||||
"dit_layerwise": False,
|
||||
"dit": False,
|
||||
# Enable if OOM happens
|
||||
"vae": True,
|
||||
"image_encoder": True,
|
||||
"text_encoder": True,
|
||||
"pin_cpu_memory": True,
|
||||
},
|
||||
},
|
||||
},
|
||||
num_gpus=2,
|
||||
# Recommended for multi-GPU
|
||||
use_fsdp_inference=True,
|
||||
dit_layerwise_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
# Enable if OOM happens
|
||||
vae_cpu_offload=True,
|
||||
image_encoder_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True,
|
||||
)
|
||||
|
||||
prompt = "A majestic lion strides across the golden savanna."
|
||||
video = generator.generate({"prompt": prompt, "output": {"output_path": "output/", "save_video": True}})
|
||||
video = generator.generate_video(prompt, output_path="output/", save_video=True)
|
||||
```
|
||||
|
||||
@@ -166,26 +166,22 @@ Enable FP4 attention via the `--nvfp4_fa4` flag:
|
||||
python examples/inference/optimizations/fp4_attn_wan2_1_1_3b.py --nvfp4_fa4
|
||||
```
|
||||
|
||||
Or in Python via the `engine.attention.nvfp4_fa4` field (resolution sets the env vars):
|
||||
Or in Python via the `nvfp4_fa4` kwarg (sets env vars automatically):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{
|
||||
"engine": {
|
||||
"attention": {"nvfp4_fa4": True},
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False, # FSDP is incompatible with FP4 pointer path
|
||||
},
|
||||
},
|
||||
nvfp4_fa4=True,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False, # FSDP is incompatible with FP4 pointer path
|
||||
)
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
gen.generate_video(prompt="A raccoon in sunflowers", save_video=True)
|
||||
```
|
||||
|
||||
#### Known Limitations
|
||||
|
||||
- `engine.use_fsdp_inference: true` is incompatible with the FP4 path (FSDP shards invalidate tensor pointers)
|
||||
- `use_fsdp_inference=True` is incompatible with the FP4 path (FSDP shards invalidate tensor pointers)
|
||||
- Per-call cosine similarity vs BF16: ~0.99 (slight quantization error accumulates over denoising steps)
|
||||
- Only supports `headdim >= 128`
|
||||
|
||||
@@ -202,6 +198,11 @@ The `attn_qat_infer` kernel hard-gates on **sm_120 (consumer Blackwell / RTX
|
||||
5090)**; on other GPUs the backend logs a notice and falls back to Flash
|
||||
Attention. See the [Attn-QAT paper](https://arxiv.org/abs/2603.00040).
|
||||
|
||||
For VSA-distilled MiniMax-H3 students, the same kernel has a block-sparse
|
||||
forward that runs VSA's 64-token tile selection in FP4
|
||||
(`FASTVIDEO_H3_VSA_FP4=1`). See
|
||||
[FastH3 NVFP4 on RTX PRO 6000](fasth3_rtx_pro_6000.md).
|
||||
|
||||
Enable both halves — attention via the env var, linear via `transformer_quant`:
|
||||
|
||||
```python
|
||||
@@ -209,15 +210,15 @@ import os
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
gen = VideoGenerator.from_config({
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False, # FSDP shards invalidate the FP4 tensor pointers
|
||||
# Wan-2.1 uses the nvfp4_qat config (NVFP4 is LTX2-specific).
|
||||
"quantization": {"transformer_quant": "nvfp4_qat"},
|
||||
},
|
||||
})
|
||||
from fastvideo.layers.quantization import get_quantization_config
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1,
|
||||
# Wan-2.1 uses the nvfp4_qat config (NVFP4 is LTX2-specific). Pass an
|
||||
# instance — the bare string is not resolved on the from_pretrained path.
|
||||
transformer_quant=get_quantization_config("nvfp4_qat")(),
|
||||
use_fsdp_inference=False, # FSDP shards invalidate the FP4 tensor pointers
|
||||
)
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
```
|
||||
|
||||
@@ -305,34 +306,17 @@ automatically.
|
||||
|
||||
### Usage
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
gen = VideoGenerator.from_config({
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"engine": {"quantization": {"transformer_quant": "FP8"}}, # per-tensor (default)
|
||||
})
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
```
|
||||
|
||||
`engine.quantization.transformer_quant` takes a quantization registry name and builds that config with its default
|
||||
arguments. To pass constructor arguments, such as per-channel granularity, set the config instance on the DiT config
|
||||
through `pipeline.model.generic.dit` instead:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.layers.quantization import get_quantization_config
|
||||
|
||||
gen = VideoGenerator.from_config({
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"pipeline": {
|
||||
"model": {
|
||||
"generic": {
|
||||
"dit": {"quant_config": get_quantization_config("FP8")(granularity="channel")}, # slower, higher accuracy
|
||||
},
|
||||
},
|
||||
},
|
||||
})
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
# Pass an instance — the bare string is not resolved on the from_pretrained path.
|
||||
transformer_quant=get_quantization_config("FP8")(), # per-tensor (default)
|
||||
# transformer_quant=get_quantization_config("FP8")(granularity="channel"), # slower, higher accuracy
|
||||
)
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
```
|
||||
|
||||
Or run the example script:
|
||||
@@ -362,7 +346,7 @@ end-to-end speedup. It is **off by default** and enabled per-run.
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{"engine": {"compile": {"enabled": True}}},
|
||||
enable_torch_compile=True,
|
||||
)
|
||||
```
|
||||
|
||||
@@ -397,14 +381,12 @@ device is unsupported. Legacy VSA, MiniMax-H3 tile-256 VSA, and the explicit
|
||||
eager with one warning instead of failing mid-denoise.
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"engine": {"compile": {"regional": True}}, # or FASTVIDEO_INFERENCE_TORCH_COMPILE=1
|
||||
})
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"MiniMaxAI/MiniMax-H3",
|
||||
inference_torch_compile=True, # or FASTVIDEO_INFERENCE_TORCH_COMPILE=1
|
||||
)
|
||||
```
|
||||
|
||||
In a YAML config, set `generator.engine.compile.regional: true`.
|
||||
|
||||
Do not combine it with `torch_compile_kwargs['mode']` (the loader injects
|
||||
inductor options, and torch.compile forbids mode+options); it is
|
||||
independent of `enable_torch_compile`, and when both are set the regional
|
||||
@@ -413,7 +395,7 @@ compile wins for the DiT.
|
||||
### What to expect from generic compile
|
||||
|
||||
The Wan result below measures the existing generic
|
||||
`engine.compile.enabled: true` path. It is useful evidence that compile can help,
|
||||
`enable_torch_compile=True` path. It is useful evidence that compile can help,
|
||||
but it is **not** a benchmark or numerical gate for the stricter regional
|
||||
fullgraph path above.
|
||||
|
||||
@@ -456,12 +438,12 @@ not asserted by any standing SSIM regression here — the SSIM tests in
|
||||
run with `enable_torch_compile` disabled. If you depend on compile
|
||||
output staying close to eager (or your previous compiled run), run an
|
||||
MS-SSIM gate on *your* config, especially when combining
|
||||
`engine.compile.enabled: true` with other numerics-affecting flags
|
||||
`enable_torch_compile=True` with other numerics-affecting flags
|
||||
(quantized attention backends, FP4, layerwise offload edge cases).
|
||||
|
||||
### Known interactions
|
||||
|
||||
- **Layerwise CPU offload** (`engine.offload.dit_layerwise: true`, the default):
|
||||
- **Layerwise CPU offload** (`dit_layerwise_offload=True`, the default):
|
||||
the offload hook previously caused an implicit graph break once per
|
||||
transformer layer, fragmenting the compiled region. Addressed in
|
||||
hao-ai-lab/FastVideo#1365 — keep that fix to get a clean compiled
|
||||
@@ -476,17 +458,16 @@ MS-SSIM gate on *your* config, especially when combining
|
||||
grad-enabled path remain outside it. Use the default inductor mode shown
|
||||
above unless your exact configuration has its own gate.
|
||||
|
||||
Extra `torch.compile` options live at `engine.compile.backend`,
|
||||
`fullgraph`, `mode`, and `dynamic`; any other `torch.compile` kwargs go in
|
||||
`engine.compile.extras`. Set them in the nested config, in a config file,
|
||||
or as a CLI dotted override (for example
|
||||
`--generator.engine.compile.mode reduce-overhead`). Example (currently
|
||||
Extra `torch.compile` options are passed through `torch_compile_kwargs`
|
||||
(a dict), accepted by `VideoGenerator.from_pretrained(...)` and by the
|
||||
CLI as a JSON string via `--torch-compile-kwargs`. Example (currently
|
||||
**not** recommended — see the CUDA-graphs caveat above):
|
||||
|
||||
```python
|
||||
VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
{"engine": {"compile": {"enabled": True, "mode": "reduce-overhead"}}}, # may error today
|
||||
enable_torch_compile=True,
|
||||
torch_compile_kwargs={"mode": "reduce-overhead"}, # may error today
|
||||
)
|
||||
```
|
||||
|
||||
@@ -499,7 +480,7 @@ config; **discard the first generation** (graph build):
|
||||
import time
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
gen = VideoGenerator.from_pretrained("your-model-id", {"engine": {"compile": {"enabled": True}}})
|
||||
gen = VideoGenerator.from_pretrained("your-model-id", enable_torch_compile=True)
|
||||
req = {"prompt": "Your prompt", "sampling": {"seed": 1024},
|
||||
"output": {"save_video": False}}
|
||||
gen.generate(req) # warmup: graph build, discard
|
||||
@@ -523,10 +504,10 @@ for backend in ["TORCH_SDPA", "FLASH_ATTN", "SAGE_ATTN"]:
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = backend
|
||||
generator = VideoGenerator.from_pretrained("your-model-id")
|
||||
start_time = time.perf_counter()
|
||||
generator.generate({
|
||||
"prompt": "Your prompt",
|
||||
"sampling": {"seed": 1024},
|
||||
})
|
||||
generator.generate_video(
|
||||
prompt="Your prompt",
|
||||
seed=1024,
|
||||
)
|
||||
elapsed = time.perf_counter() - start_time
|
||||
print(f"{backend}: {elapsed:.2f}s")
|
||||
```
|
||||
|
||||
@@ -31,14 +31,10 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| cosmos | `nvidia/Cosmos-Predict2-2B-Video2World` | T2V | — |
|
||||
| cosmos25 | `KyleShao/Cosmos-Predict2.5-2B-Diffusers` | T2V | [basic_cosmos2_5_t2w.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_cosmos2_5_t2w.py) |
|
||||
| cosmos25 | `nvidia/Cosmos-Predict2.5-14B` | T2V | [basic_cosmos2_5_t2w.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_cosmos2_5_t2w.py) |
|
||||
| dreamx_world | `FastVideo/DreamX-World-5B-Cam-Diffusers` | I2V | [basic_dreamx_world.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_dreamx_world.py) |
|
||||
| dreamx_world | `FastVideo/DreamX-World-5B-Diffusers` | I2V | [basic_dreamx_world.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_dreamx_world.py) |
|
||||
| flux | `black-forest-labs/FLUX.1-dev` | T2I | [basic_flux_dev.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_flux_dev.py) |
|
||||
| flux2 | `black-forest-labs/FLUX.2-klein-4B`<br>`black-forest-labs/FLUX.2-klein-9B` | T2I | [basic_flux2_klein.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_flux2_klein.py) |
|
||||
| flux2 | `black-forest-labs/FLUX.2-dev` | T2I | [basic_flux2.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_flux2.py) |
|
||||
| gamecraft | `FastVideo/HunyuanGameCraft-Diffusers` | I2V | [basic_gamecraft.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_gamecraft.py) |
|
||||
| gen3c | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | T2V | [basic_gen3c.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_gen3c.py) |
|
||||
| glm_image | `zai-org/GLM-Image` | T2I | [basic_glm_image.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_glm_image.py) |
|
||||
| hunyuan | `hunyuanvideo-community/HunyuanVideo` | T2V | — |
|
||||
| hunyuan | `FastVideo/FastHunyuan-diffusers` | T2V | — |
|
||||
| hunyuan15 | `hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v` | T2V | [basic_hy15.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_hy15.py) |
|
||||
@@ -54,13 +50,13 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| kandinsky5 | `kandinskylab/Kandinsky-5.0-I2V-Lite-5s-Diffusers` | I2V | [basic_kandinsky5_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky5_i2v.py) |
|
||||
| kandinsky5 | `kandinskylab/Kandinsky-5.0-I2V-Pro-sft-5s-Diffusers` | I2V | [basic_kandinsky5_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky5_i2v.py) |
|
||||
| kandinsky5 | `kandinskylab/Kandinsky-5.0-I2V-Pro-distilled-5s-Diffusers` | I2V | [basic_kandinsky5_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky5_i2v.py) |
|
||||
| kandinsky6 | `kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers`<br>`kandinskylab/Kandinsky-6.0-Pro-sft-5s-Diffusers` | T2V, I2V | [basic_kandinsky6_ti2va.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
| kandinsky6 | `kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers` | T2V, I2V | [basic_kandinsky6_ti2va.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
| kandinsky6_sr | `kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers`<br>`kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers` | — | [basic_kandinsky6_sr.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_sr.py) |
|
||||
| lingbot_video | `FastVideo/LingBot-Video-MoE-30B-A3B-Diffusers` | T2V | [basic_lingbot_video.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbot_video.py) |
|
||||
| lingbot_video | `FastVideo/LingBot-Video-Dense-1.3B-Diffusers` | T2V | [basic_lingbot_video.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbot_video.py) |
|
||||
| lingbotworld | `FastVideo/LingBot-World-Base-Cam-Diffusers` | I2V | [basic_lingbotworld_base_cam.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbotworld_base_cam.py) |
|
||||
| lingbotworld2 | `robbyant/lingbot-world-v2-14b-causal-fast` | I2V | [basic_lingbotworld2_causal_fast.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbotworld2_causal_fast.py) |
|
||||
| longcat | `FastVideo/LongCat-Video-T2V-Diffusers` | T2V | [basic_longcat_t2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_longcat_t2v.py) |
|
||||
| longcat | `FastVideo/LongCat-Video-I2V-Diffusers` | I2V | [basic_longcat_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_longcat_i2v.py) |
|
||||
| longcat | `FastVideo/LongCat-Video-VC-Diffusers` | — | [basic_longcat_vc.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_longcat_vc.py) |
|
||||
| ltx2 | `FastVideo/LTX2-Distilled-Diffusers`<br>`FastVideo/LTX2.3-Distilled-Diffusers`<br>`FastVideo/LTX-2.3-Distilled-Diffusers` | T2V | [basic_ltx2_distilled.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_ltx2_distilled.py) |
|
||||
| ltx2 | `Lightricks/LTX-2.3`<br>`FastVideo/LTX2.3-base`<br>`FastVideo/LTX2.3-Diffusers` | T2V | [basic_ltx2.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_ltx2.py) |
|
||||
| ltx2 | `Lightricks/LTX-2`<br>`FastVideo/LTX2-base`<br>`FastVideo/LTX2-Diffusers` | T2V | [basic_ltx2.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_ltx2.py) |
|
||||
@@ -71,9 +67,6 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| sd35 | `stabilityai/stable-diffusion-3.5-medium` | T2I | [basic_sd35_t2i.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_sd35_t2i.py) |
|
||||
| stable_audio | `FastVideo/stable-audio-open-1.0-Diffusers` | T2V | [basic_stable_audio.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_stable_audio.py) |
|
||||
| stable_audio | `FastVideo/stable-audio-open-small-Diffusers` | T2V | [basic_stable_audio_small.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_stable_audio_small.py) |
|
||||
| turbodiffusion | `loayrashid/TurboWan2.1-T2V-1.3B-Diffusers` | T2V | [basic_turbodiffusion.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_turbodiffusion.py) |
|
||||
| turbodiffusion | `loayrashid/TurboWan2.1-T2V-14B-Diffusers` | T2V | [basic_turbodiffusion_14b.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_turbodiffusion_14b.py) |
|
||||
| turbodiffusion | `loayrashid/TurboWan2.2-I2V-A14B-Diffusers` | I2V | [basic_turbodiffusion_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_turbodiffusion_i2v.py) |
|
||||
| wan | `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` | T2V | [basic.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic.py) |
|
||||
| wan | `Wan-AI/Wan2.1-T2V-14B-Diffusers`<br>`FastVideo/Wan2.1-VSA-T2V-14B-720P-Diffusers` | T2V | — |
|
||||
| wan | `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` | I2V | — |
|
||||
@@ -89,7 +82,6 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| wan | `wlsaidhi/SFWan2.1-T2V-1.3B-Diffusers` | T2V | [basic_self_forcing_causal.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal.py) |
|
||||
| wan | `rand0nmr/SFWan2.2-T2V-A14B-Diffusers` | T2V | [basic_self_forcing_causal_wan2_2_t2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_t2v.py) |
|
||||
| wan | `FastVideo/SFWan2.2-I2V-A14B-Preview-Diffusers` | I2V | [basic_self_forcing_causal_wan2_2_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_i2v.py) |
|
||||
| zimage | `Tongyi-MAI/Z-Image-Turbo` | T2I | [basic_zimage.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_zimage.py) |
|
||||
|
||||
**Note (stable_audio)**: the Stable Audio Open pipelines generate audio
|
||||
(`StableAudioT2AConfig` / `StableAudioOpenSmallConfig`); they are registered
|
||||
@@ -101,7 +93,15 @@ to convert the official weights locally and set `MMAUDIO_MODEL_PATH`.
|
||||
|
||||
**Note (MiniMax H3)**: T2VA, FL2VA, and Ref2VA all generate video with stereo
|
||||
audio. Use the Ref2VA example when passing ordered image, video, or audio
|
||||
references.
|
||||
references. Distilled Ref2VA PDD students (eight transformer forwards, with
|
||||
reference videos as sparse VSA regions) run through
|
||||
[basic_fasth3_omniref_pdd.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_fasth3_omniref_pdd.py);
|
||||
see [FastH3 distilled checkpoint schedules](fasth3-distilled.md#ref2va-pdd-students).
|
||||
|
||||
**Note (Kandinsky 6)**: the two `kandinsky6` IDs generate video with audio from
|
||||
text, optionally plus an image (see the [T2IVA guide](kandinsky6.md)); the
|
||||
`kandinsky6_sr` IDs upscale an existing video (see the
|
||||
[Video SR guide](kandinsky6_sr.md)).
|
||||
|
||||
**Note (Wan-VACE)**: not currently supported — no VACE pipeline or registered
|
||||
model ID exists on `main`
|
||||
@@ -152,31 +152,25 @@ optimizations: absence means **untested**, not incompatible.
|
||||
}
|
||||
</style>
|
||||
|
||||
| Model Name | HuggingFace Model ID | Resolutions | TeaCache | Sliding Tile Attn (Legacy Branch) | Sage Attn | VSA | BSA |
|
||||
|------------|---------------------|-------------|----------|-------------------|-----------|-----|-----|
|
||||
| FastWan2.1 T2V 1.3B | `FastVideo/FastWan2.1-T2V-1.3B-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ✅ | ⭕ |
|
||||
| FastWan2.2 TI2V 5B Full Attn | `FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers` | 720P | ⭕ | ⭕ | ⭕ | ✅ | ⭕ |
|
||||
| Wan2.2 TI2V 5B | `Wan-AI/Wan2.2-TI2V-5B-Diffusers` | 720P | ⭕ | ⭕ | ✅ | ⭕ | ⭕ |
|
||||
| DreamX-World 5B Cam | `FastVideo/DreamX-World-5B-Cam-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| DreamX-World 5B AR | `FastVideo/DreamX-World-5B-Diffusers` | 704px1280p | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Lucy Edit Dev 5B*** | `decart-ai/Lucy-Edit-Dev` | 480P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Wan2.2 T2V A14B | `Wan-AI/Wan2.2-T2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.2 I2V A14B | `Wan-AI/Wan2.2-I2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ | ⭕ |
|
||||
| HunyuanVideo | `hunyuanvideo-community/HunyuanVideo` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| FastHunyuan | `FastVideo/FastHunyuan-diffusers` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 T2V 1.3B | `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 T2V 14B | `Wan-AI/Wan2.1-T2V-14B-Diffusers` | 480P, 720P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 I2V 480P | `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 I2V 720P | `Wan-AI/Wan2.1-I2V-14B-720P-Diffusers` | 720P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| TurboWan2.1 T2V 1.3B | `loayrashid/TurboWan2.1-T2V-1.3B-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| TurboWan2.1 T2V 14B | `loayrashid/TurboWan2.1-T2V-14B-Diffusers` | 480P, 720P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| TurboWan2.2 I2V A14B | `loayrashid/TurboWan2.2-I2V-A14B-Diffusers` | 480P<br>720P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| LongCat T2V 13.6B | `FastVideo/LongCat-Video-T2V-Diffusers` | 480P<br>720P | ❌ | ❌ | ❌ | ⭕ | ✅ |
|
||||
| Matrix Game 2.0 Base Distilled | `FastVideo/Matrix-Game-2.0-Base-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 GTA Distilled | `FastVideo/Matrix-Game-2.0-GTA-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 TempleRun Distilled | `FastVideo/Matrix-Game-2.0-TempleRun-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 3.0 Base Distilled | `FastVideo/Matrix-Game-3.0-Base-Distilled-Diffusers` | 720x1280 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| GEN3C Cosmos 7B | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | 704px1280p | ❌ | ❌ | ❌ | ⭕ | ⭕ |
|
||||
| Model Name | HuggingFace Model ID | Resolutions | TeaCache | Sliding Tile Attn (Legacy Branch) | Sage Attn | VSA |
|
||||
|------------|---------------------|-------------|----------|-------------------|-----------|-----|
|
||||
| FastWan2.1 T2V 1.3B | `FastVideo/FastWan2.1-T2V-1.3B-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ✅ |
|
||||
| FastWan2.2 TI2V 5B Full Attn | `FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers` | 720P | ⭕ | ⭕ | ⭕ | ✅ |
|
||||
| Wan2.2 TI2V 5B | `Wan-AI/Wan2.2-TI2V-5B-Diffusers` | 720P | ⭕ | ⭕ | ✅ | ⭕ |
|
||||
| Lucy Edit Dev 5B*** | `decart-ai/Lucy-Edit-Dev` | 480P | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Wan2.2 T2V A14B | `Wan-AI/Wan2.2-T2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ |
|
||||
| Wan2.2 I2V A14B | `Wan-AI/Wan2.2-I2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ |
|
||||
| HunyuanVideo | `hunyuanvideo-community/HunyuanVideo` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ |
|
||||
| FastHunyuan | `FastVideo/FastHunyuan-diffusers` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 T2V 1.3B | `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 T2V 14B | `Wan-AI/Wan2.1-T2V-14B-Diffusers` | 480P, 720P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 I2V 480P | `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 I2V 720P | `Wan-AI/Wan2.1-I2V-14B-720P-Diffusers` | 720P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Matrix Game 2.0 Base Distilled | `FastVideo/Matrix-Game-2.0-Base-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 GTA Distilled | `FastVideo/Matrix-Game-2.0-GTA-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 TempleRun Distilled | `FastVideo/Matrix-Game-2.0-TempleRun-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 3.0 Base Distilled | `FastVideo/Matrix-Game-3.0-Base-Distilled-Diffusers` | 720x1280 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| GEN3C Cosmos 7B | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | 704px1280p | ❌ | ❌ | ❌ | ⭕ |
|
||||
|
||||
## Apple Silicon native runtime
|
||||
|
||||
@@ -240,11 +234,6 @@ listed under [Special requirements](#special-requirements).
|
||||
https://github.com/hao-ai-lab/FastVideo/tree/sta_do_not_delete
|
||||
- STA currently requires Hopper GPUs (H100s).
|
||||
|
||||
### TurboWan2.1 (TurboDiffusion)
|
||||
- Uses TurboDiffusionPipeline with RCM scheduler for 1-4 step generation
|
||||
- Requires SLA attention backend: `export FASTVIDEO_ATTENTION_BACKEND=SLA_ATTN`
|
||||
- Uses `guidance_scale=1.0` (no classifier-free guidance)
|
||||
|
||||
### Matrix Game 2.0
|
||||
- Image-to-video game world models with keyboard/mouse control input
|
||||
- Three variants available: Base (universal), GTA, and TempleRun
|
||||
|
||||
@@ -19,56 +19,46 @@ bash examples/training/finetune/wan_t2v_1.3B/crush_smol/preprocess_wan_data_t2v_
|
||||
|
||||
## Preprocessing Pipeline
|
||||
|
||||
The preprocessing pipeline supports multiple dataset formats and video loaders. It reads a `PreprocessRunConfig`
|
||||
YAML file (`fastvideo/api/training_schema.py`): the model and workload type at the top level, and the preprocessing
|
||||
settings in the `preprocess` section (the fields of `PreprocessConfig` in `fastvideo/configs/configs.py`):
|
||||
|
||||
```yaml
|
||||
# preprocess_t2v.yaml
|
||||
model_path: Wan-AI/Wan2.1-T2V-1.3B-Diffusers
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
preprocess:
|
||||
video_loader_type: torchvision
|
||||
dataset_type: merged
|
||||
preprocess_video_batch_size: 2
|
||||
dataloader_num_workers: 0
|
||||
max_height: 480
|
||||
max_width: 832
|
||||
num_frames: 77
|
||||
train_fps: 16
|
||||
samples_per_file: 8
|
||||
flush_frequency: 8
|
||||
video_length_tolerance_range: 5
|
||||
```
|
||||
|
||||
Pass the file with `--config`. Each dotted override after it sets one field, for example the values that come from
|
||||
shell variables:
|
||||
The new preprocessing pipeline supports multiple dataset formats and video loaders:
|
||||
|
||||
```bash
|
||||
GPU_NUM=2
|
||||
MODEL_PATH="Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
DATASET_PATH="data/crush-smol/"
|
||||
OUTPUT_DIR="data/crush-smol_processed_t2v/"
|
||||
|
||||
torchrun --nproc_per_node=$GPU_NUM \
|
||||
-m fastvideo.pipelines.preprocess.v1_preprocessing_new \
|
||||
--config preprocess_t2v.yaml \
|
||||
--preprocess.dataset_path "$DATASET_PATH" \
|
||||
--preprocess.dataset_output_dir "$OUTPUT_DIR"
|
||||
--model_path $MODEL_PATH \
|
||||
--mode preprocess \
|
||||
--workload_type t2v \
|
||||
--preprocess.video_loader_type torchvision \
|
||||
--preprocess.dataset_type merged \
|
||||
--preprocess.dataset_path $DATASET_PATH \
|
||||
--preprocess.dataset_output_dir $OUTPUT_DIR \
|
||||
--preprocess.preprocess_video_batch_size 2 \
|
||||
--preprocess.dataloader_num_workers 0 \
|
||||
--preprocess.max_height 480 \
|
||||
--preprocess.max_width 832 \
|
||||
--preprocess.num_frames 77 \
|
||||
--preprocess.train_fps 16 \
|
||||
--preprocess.samples_per_file 8 \
|
||||
--preprocess.flush_frequency 8 \
|
||||
--preprocess.video_length_tolerance_range 5
|
||||
```
|
||||
|
||||
### Key Parameters
|
||||
|
||||
| Parameter | Description |
|
||||
| ------------------------------------- | ----------------------------------------------------------- |
|
||||
| `pipeline.workload_type` | Task type: `t2v` (text-to-video) or `i2v` (image-to-video) |
|
||||
| `preprocess.dataset_type` | Input format: `hf` (HuggingFace) or `merged` (local folder) |
|
||||
| `preprocess.dataset_path` | Path to dataset (HF repo ID or local folder) |
|
||||
| `preprocess.dataset_output_dir` | Output directory for Parquet files |
|
||||
| `preprocess.video_loader_type` | Video decoder: `torchcodec` or `torchvision` |
|
||||
| `preprocess.max_height` / `max_width` | Target resolution for videos |
|
||||
| `preprocess.num_frames` | Number of frames to extract per video |
|
||||
| `preprocess.train_fps` | Target FPS for frame extraction |
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `--workload_type` | Task type: `t2v` (text-to-video) or `i2v` (image-to-video) |
|
||||
| `--preprocess.dataset_type` | Input format: `hf` (HuggingFace) or `merged` (local folder) |
|
||||
| `--preprocess.dataset_path` | Path to dataset (HF repo ID or local folder) |
|
||||
| `--preprocess.dataset_output_dir` | Output directory for Parquet files |
|
||||
| `--preprocess.video_loader_type` | Video decoder: `torchcodec` or `torchvision` |
|
||||
| `--preprocess.max_height` / `max_width` | Target resolution for videos |
|
||||
| `--preprocess.num_frames` | Number of frames to extract per video |
|
||||
| `--preprocess.train_fps` | Target FPS for frame extraction |
|
||||
|
||||
## Dataset Formats
|
||||
|
||||
|
||||
+37
-48
@@ -4,58 +4,48 @@ This guide covers finetuning video diffusion models with FastVideo, including fu
|
||||
|
||||
## Training Arguments
|
||||
|
||||
Each training launcher passes a `TrainingRunConfig` YAML file to its entry point with `--config` (the schema is in
|
||||
`fastvideo/api/training_schema.py`). A dotted override after `--config` sets one field, for example
|
||||
`--training.optimizer.learning_rate 1e-5` or `--engine.num_gpus "$NUM_GPUS"`:
|
||||
|
||||
```bash
|
||||
torchrun --nnodes 1 --nproc_per_node 4 \
|
||||
fastvideo/training/wan_training_pipeline.py \
|
||||
--config finetune_t2v.yaml \
|
||||
--engine.num_gpus 4
|
||||
```
|
||||
|
||||
The settings are grouped as follows:
|
||||
FastVideo training scripts use several argument groups:
|
||||
|
||||
### Training Arguments
|
||||
|
||||
| Config path | Description |
|
||||
| ------------------------------------------------------ | ------------------------------------------------- |
|
||||
| `training.loop.max_train_steps` | Total training steps |
|
||||
| `training.data.train_batch_size` | Batch size per GPU |
|
||||
| `training.loop.gradient_accumulation_steps` | Steps to accumulate before optimizer update |
|
||||
| `training.data.num_latent_t` | Temporal latent dimension (reduce to save memory) |
|
||||
| `training.data.num_height` / `training.data.num_width` | Video resolution |
|
||||
| `training.data.num_frames` | Number of frames per video |
|
||||
| `training.checkpoint.output_dir` | Directory for checkpoints |
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| `--max_train_steps` | Total training steps |
|
||||
| `--train_batch_size` | Batch size per GPU |
|
||||
| `--gradient_accumulation_steps` | Steps to accumulate before optimizer update |
|
||||
| `--num_latent_t` | Temporal latent dimension (reduce to save memory) |
|
||||
| `--num_height` / `--num_width` | Video resolution |
|
||||
| `--num_frames` | Number of frames per video |
|
||||
| `--output_dir` | Directory for checkpoints |
|
||||
|
||||
### Parallelism Arguments
|
||||
|
||||
| Config path | Description |
|
||||
| --------------------------------------- | ---------------------------------------------------------- |
|
||||
| `engine.num_gpus` | Total number of GPUs |
|
||||
| `engine.parallelism.sp_size` | Sequence parallel size (increase to reduce memory per GPU) |
|
||||
| `engine.parallelism.tp_size` | Tensor parallel size |
|
||||
| `engine.parallelism.hsdp_replicate_dim` | HSDP replication dimension |
|
||||
| `engine.parallelism.hsdp_shard_dim` | HSDP sharding dimension |
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| `--num_gpus` | Total number of GPUs |
|
||||
| `--sp_size` | Sequence parallel size (increase to reduce memory per GPU) |
|
||||
| `--tp_size` | Tensor parallel size |
|
||||
| `--hsdp_replicate_dim` | HSDP replication dimension |
|
||||
| `--hsdp_shard_dim` | HSDP sharding dimension |
|
||||
|
||||
### Optimizer Arguments
|
||||
|
||||
| Config path | Description |
|
||||
| ---------------------------------- | ------------------------------- |
|
||||
| `training.optimizer.learning_rate` | Base learning rate |
|
||||
| `training.optimizer.weight_decay` | Weight decay for regularization |
|
||||
| `training.optimizer.max_grad_norm` | Gradient clipping threshold |
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| `--learning_rate` | Base learning rate |
|
||||
| `--mixed_precision` | Precision mode (`bf16` recommended) |
|
||||
| `--weight_decay` | Weight decay for regularization |
|
||||
| `--max_grad_norm` | Gradient clipping threshold |
|
||||
|
||||
### Validation Arguments
|
||||
|
||||
| Config path | Description |
|
||||
| ------------------------------------ | ----------------------------------------------------------- |
|
||||
| `training.validation.enabled` | Enable validation logging |
|
||||
| `training.validation.dataset_file` | JSON file with validation prompts |
|
||||
| `training.validation.every_steps` | Run validation every N steps |
|
||||
| `training.validation.sampling_steps` | Inference steps for validation (a list, for example `[50]`) |
|
||||
| `training.validation.guidance_scale` | CFG scale for validation |
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| `--log_validation` | Enable validation logging |
|
||||
| `--validation_dataset_file` | JSON file with validation prompts |
|
||||
| `--validation_steps` | Run validation every N steps |
|
||||
| `--validation_sampling_steps` | Inference steps for validation |
|
||||
| `--validation_guidance_scale` | CFG scale for validation |
|
||||
|
||||
## Full Finetuning
|
||||
|
||||
@@ -69,8 +59,8 @@ bash examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v.sh
|
||||
**Typical settings:**
|
||||
|
||||
- Learning rate: `1e-5` to `5e-5`
|
||||
- Gradient checkpointing: `training.model.enable_gradient_checkpointing_type: full`
|
||||
- Memory scaling: Increase `engine.parallelism.sp_size` or reduce `training.data.num_latent_t` to fit in memory
|
||||
- Gradient checkpointing: `--enable_gradient_checkpointing_type "full"`
|
||||
- Memory scaling: Increase `--sp_size` or reduce `--num_latent_t` to fit in memory
|
||||
|
||||
## Attention Quantization-Aware Training
|
||||
|
||||
@@ -90,10 +80,10 @@ LoRA (Low-Rank Adaptation) trains lightweight adapters while keeping the base mo
|
||||
|
||||
### LoRA-Specific Arguments
|
||||
|
||||
| Config path | Description |
|
||||
| ----------------------------- | --------------------------------------- |
|
||||
| `training.lora.enabled: true` | Enable LoRA mode |
|
||||
| `training.lora.rank` | Rank of LoRA adapters (16, 32, 64, 128) |
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| `--lora_training True` | Enable LoRA mode |
|
||||
| `--lora_rank` | Rank of LoRA adapters (16, 32, 64, 128) |
|
||||
|
||||
### Learning Rate for LoRA
|
||||
|
||||
@@ -113,7 +103,7 @@ bash examples/training/finetune/wan_t2v_1.3B/crush_smol/finetune_t2v_lora.sh
|
||||
|
||||
Key differences from full finetune:
|
||||
|
||||
- Set `training.lora.enabled: true` and `training.lora.rank: 32`
|
||||
- Add `--lora_training True --lora_rank 32`
|
||||
- Use higher learning rate (10–20× full finetune)
|
||||
- Can run on fewer GPUs (even single GPU)
|
||||
- Outputs adapter weights instead of full model
|
||||
@@ -196,5 +186,4 @@ Each example includes:
|
||||
- `preprocess_*.sh` — run preprocessing
|
||||
- `finetune_*.sh` — full finetune launcher
|
||||
- `finetune_*_lora.sh` — LoRA finetune launcher
|
||||
- a YAML file next to each launcher — the `TrainingRunConfig` that the launcher passes with `--config`
|
||||
- `validation.json` — validation prompts
|
||||
|
||||
@@ -16,17 +16,111 @@ DATA_DIR="data/matrixgame2"
|
||||
VALIDATION_DATASET_FILE="examples/distill/MatrixGame2.0/validation.json"
|
||||
NUM_GPUS=1
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name "matrixgame2_sf"
|
||||
--output_dir "checkpoints/matrixgame2_sf_${RUN_NAME}"
|
||||
--wandb_run_name "${RUN_NAME}_test"
|
||||
--max_train_steps 5
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 352
|
||||
--num_width 640
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
--simulate_generator_forward
|
||||
--num_frames 81
|
||||
--num_frame_per_block 3 # Frame generation block size for self-forcing
|
||||
# --enable_gradient_masking
|
||||
# --gradient_mask_last_n_frames 21
|
||||
# --init_weights_from_safetensors "path/to/generator_ema.safetensors"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus $NUM_GPUS
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1
|
||||
--hsdp_shard_dim $NUM_GPUS
|
||||
)
|
||||
|
||||
model_args=(
|
||||
--model_path $GENERATOR_MODEL_PATH # TODO: check if you can remove this in this script
|
||||
--pretrained_model_name_or_path $GENERATOR_MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--log_visualization
|
||||
--visualization-steps 100
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "4"
|
||||
--validation_guidance_scale "6.0"
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 3e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 400
|
||||
--training_state_checkpointing_steps 400
|
||||
--weight_decay 0
|
||||
--betas "0.9,0.95"
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--use_ema True
|
||||
--ema_decay 0.99
|
||||
--ema_start_step 200
|
||||
)
|
||||
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,750,500,250'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--dfake_gen_update_ratio 5
|
||||
--real_score_guidance_scale 3.0
|
||||
--fake_score_learning_rate 3e-7
|
||||
--fake_score_betas "0.9,0.95"
|
||||
--warp_denoising_step
|
||||
)
|
||||
|
||||
self_forcing_args=(
|
||||
--independent_first_frame False # Whether to treat first frame independently
|
||||
--same_step_across_blocks True # Whether to use same denoising step across all blocks
|
||||
--last_step_only False # Whether to only use the last denoising step
|
||||
--context_noise 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
)
|
||||
|
||||
torchrun \
|
||||
--nnodes 1 \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
fastvideo/training/matrixgame2_self_forcing_distillation_pipeline.py \
|
||||
--config examples/distill/MatrixGame2.0/distill_dmd.yaml \
|
||||
--model_path "$GENERATOR_MODEL_PATH" \
|
||||
--engine.num_gpus "$NUM_GPUS" \
|
||||
--engine.parallelism.hsdp_shard_dim "$NUM_GPUS" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.checkpoint.output_dir "checkpoints/matrixgame2_sf_${RUN_NAME}" \
|
||||
--training.tracker.run_name "${RUN_NAME}_test" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}" \
|
||||
"${self_forcing_args[@]}"
|
||||
@@ -63,6 +63,93 @@ FAKE_SCORE_MODEL_PATH="FastVideo/Matrix-Game-2.0-Base-Diffusers"
|
||||
DATA_DIR="data/matrixgame2"
|
||||
VALIDATION_DATASET_FILE="examples/distill/MatrixGame2.0/validation.json"
|
||||
|
||||
training_args=(
|
||||
--tracker_project_name "matrixgame2_sf"
|
||||
--output_dir "checkpoints/matrixgame2_sf_${RUN_NAME}"
|
||||
--wandb_run_name "${RUN_NAME}_test"
|
||||
--max_train_steps 1200
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 2
|
||||
--num_latent_t 21
|
||||
--num_height 352
|
||||
--num_width 640
|
||||
--simulate_generator_forward
|
||||
--num_frames 81
|
||||
--num_frame_per_block 3
|
||||
# --init_weights_from_safetensors "path/to/generator_ema.safetensors"
|
||||
)
|
||||
|
||||
parallel_args=(
|
||||
--num_gpus "${TOTAL_GPUS}"
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1
|
||||
--hsdp_shard_dim "${TOTAL_GPUS}"
|
||||
)
|
||||
|
||||
model_args=(
|
||||
--model_path "${GENERATOR_MODEL_PATH}"
|
||||
--pretrained_model_name_or_path "${GENERATOR_MODEL_PATH}"
|
||||
--real_score_model_path "${REAL_SCORE_MODEL_PATH}"
|
||||
--fake_score_model_path "${FAKE_SCORE_MODEL_PATH}"
|
||||
)
|
||||
|
||||
dataset_args=(
|
||||
--data_path "${DATA_DIR}"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--log_visualization
|
||||
--visualization-steps 100
|
||||
--validation_dataset_file "${VALIDATION_DATASET_FILE}"
|
||||
--validation_steps 100
|
||||
--validation_sampling_steps "4"
|
||||
--validation_guidance_scale "6.0"
|
||||
)
|
||||
|
||||
optimizer_args=(
|
||||
--learning_rate 3e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_only_checkpointing_steps 400
|
||||
--training_state_checkpointing_steps 400
|
||||
--weight_decay 0
|
||||
--betas "0.9,0.95"
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--use_ema True
|
||||
--ema_decay 0.99
|
||||
--ema_start_step 200
|
||||
)
|
||||
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps "1000,750,500,250"
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--dfake_gen_update_ratio 5
|
||||
--real_score_guidance_scale 3.0
|
||||
--fake_score_learning_rate 3e-7
|
||||
--fake_score_betas "0.9,0.95"
|
||||
--warp_denoising_step
|
||||
)
|
||||
|
||||
self_forcing_args=(
|
||||
--independent_first_frame False
|
||||
--same_step_across_blocks True
|
||||
--last_step_only False
|
||||
--context_noise 0
|
||||
)
|
||||
|
||||
srun python -m torch.distributed.run \
|
||||
--nnodes "${SLURM_JOB_NUM_NODES}" \
|
||||
--nproc_per_node "${GPUS_PER_NODE}" \
|
||||
@@ -70,13 +157,12 @@ srun python -m torch.distributed.run \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="${MASTER_ADDR}:${MASTER_PORT}" \
|
||||
fastvideo/training/matrixgame2_self_forcing_distillation_pipeline.py \
|
||||
--config examples/distill/MatrixGame2.0/distill_dmd_slurm.yaml \
|
||||
--model_path "${GENERATOR_MODEL_PATH}" \
|
||||
--engine.num_gpus "${TOTAL_GPUS}" \
|
||||
--engine.parallelism.hsdp_shard_dim "${TOTAL_GPUS}" \
|
||||
--training.distillation.real_score_model_path "${REAL_SCORE_MODEL_PATH}" \
|
||||
--training.distillation.fake_score_model_path "${FAKE_SCORE_MODEL_PATH}" \
|
||||
--training.data.data_path "${DATA_DIR}" \
|
||||
--training.checkpoint.output_dir "checkpoints/matrixgame2_sf_${RUN_NAME}" \
|
||||
--training.tracker.run_name "${RUN_NAME}_test" \
|
||||
--training.validation.dataset_file "${VALIDATION_DATASET_FILE}"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}" \
|
||||
"${self_forcing_args[@]}"
|
||||
|
||||
@@ -1,68 +0,0 @@
|
||||
# Read by distill_dmd.sh, which passes model_path, engine.num_gpus, engine.parallelism.hsdp_shard_dim,
|
||||
# training.distillation.real_score_model_path, training.distillation.fake_score_model_path, training.data.data_path,
|
||||
# training.checkpoint.output_dir, training.tracker.run_name, and training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 1
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
# components:
|
||||
# transformer_weights: path/to/generator_ema.safetensors
|
||||
flow_shift: 5.0
|
||||
dmd_denoising_steps: [1000, 750, 500, 250]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 352
|
||||
num_width: 640
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 3.0e-06
|
||||
betas: [0.9, 0.95]
|
||||
weight_decay: 0.0
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 5
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
training_state_checkpointing_steps: 400
|
||||
weight_only_checkpointing_steps: 400
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: matrixgame2_sf
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [4]
|
||||
guidance_scale: 6.0
|
||||
every_steps: 100
|
||||
log_visualization: true
|
||||
visualization_steps: 100
|
||||
distillation:
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.0
|
||||
fake_score_learning_rate: 3.0e-07
|
||||
fake_score_betas: [0.9, 0.95]
|
||||
simulate_generator_forward: true
|
||||
warp_denoising_step: true
|
||||
ema:
|
||||
enabled: true
|
||||
decay: 0.99
|
||||
start_step: 200
|
||||
self_forcing:
|
||||
dfake_gen_update_ratio: 5
|
||||
num_frame_per_block: 3 # Frame generation block size for self-forcing
|
||||
independent_first_frame: false # Whether to treat first frame independently
|
||||
same_step_across_blocks: true # Whether to use same denoising step across all blocks
|
||||
last_step_only: false # Whether to only use the last denoising step
|
||||
context_noise: 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -1,66 +0,0 @@
|
||||
# Read by distill_dmd.slurm, which passes model_path, engine.num_gpus, engine.parallelism.hsdp_shard_dim,
|
||||
# training.distillation.real_score_model_path, training.distillation.fake_score_model_path, training.data.data_path,
|
||||
# training.checkpoint.output_dir, training.tracker.run_name, and training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 1
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
# components:
|
||||
# transformer_weights: path/to/generator_ema.safetensors
|
||||
flow_shift: 5.0
|
||||
dmd_denoising_steps: [1000, 750, 500, 250]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 352
|
||||
num_width: 640
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 3.0e-06
|
||||
betas: [0.9, 0.95]
|
||||
weight_decay: 0.0
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 1200
|
||||
gradient_accumulation_steps: 2
|
||||
checkpoint:
|
||||
training_state_checkpointing_steps: 400
|
||||
weight_only_checkpointing_steps: 400
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: matrixgame2_sf
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [4]
|
||||
guidance_scale: 6.0
|
||||
every_steps: 100
|
||||
log_visualization: true
|
||||
visualization_steps: 100
|
||||
distillation:
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.0
|
||||
fake_score_learning_rate: 3.0e-07
|
||||
fake_score_betas: [0.9, 0.95]
|
||||
simulate_generator_forward: true
|
||||
warp_denoising_step: true
|
||||
ema:
|
||||
enabled: true
|
||||
decay: 0.99
|
||||
start_step: 200
|
||||
self_forcing:
|
||||
dfake_gen_update_ratio: 5
|
||||
num_frame_per_block: 3
|
||||
independent_first_frame: false
|
||||
same_step_across_blocks: true
|
||||
last_step_only: false
|
||||
context_noise: 0
|
||||
@@ -36,16 +36,105 @@ VALIDATION_DATASET_FILE=your_validation_data_dir
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
training_args=(
|
||||
--tracker_project_name SFwan_t2v_distill_self_forcing_dmd
|
||||
--output_dir your_output_dir
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
--log_visualization
|
||||
--simulate_generator_forward
|
||||
--num_frames 81
|
||||
--num_frame_per_block 3 # Frame generation block size for self-forcing
|
||||
--enable_gradient_masking
|
||||
--gradient_mask_last_n_frames 21
|
||||
)
|
||||
|
||||
parallel_args=(
|
||||
--num_gpus $NUM_GPUS # 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1 # 64
|
||||
--hsdp_shard_dim $NUM_GPUS
|
||||
)
|
||||
|
||||
model_args=(
|
||||
--model_path $GENERATOR_MODEL_PATH # TODO: check if you can remove this in this script
|
||||
--pretrained_model_name_or_path $GENERATOR_MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 50
|
||||
--validation_sampling_steps "4"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-5
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 500
|
||||
--weight_decay 0.01
|
||||
--betas '0.0,0.999'
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--use_ema True
|
||||
--ema_decay 0.99
|
||||
--ema_start_step 100
|
||||
--init_weights_from_safetensors your_ode_init_weights_path
|
||||
)
|
||||
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,750,500,250'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--dfake_gen_update_ratio 5
|
||||
--real_score_guidance_scale 3.0
|
||||
--fake_score_learning_rate 8e-6
|
||||
--fake_score_betas '0.0,0.999'
|
||||
--warp_denoising_step
|
||||
)
|
||||
|
||||
self_forcing_args=(
|
||||
--independent_first_frame False # Whether to treat first frame independently
|
||||
--same_step_across_blocks True # Whether to use same denoising step across all blocks
|
||||
--last_step_only False # Whether to only use the last denoising step
|
||||
--context_noise 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
)
|
||||
|
||||
torchrun \
|
||||
--nnodes 1 \
|
||||
--master_port $MASTER_PORT \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
fastvideo/training/wan_self_forcing_distillation_pipeline.py \
|
||||
--config examples/distill/SFWan2.1-T2V/distill_dmd_t2v_1.3B.yaml \
|
||||
--model_path "$GENERATOR_MODEL_PATH" \
|
||||
--engine.num_gpus "$NUM_GPUS" \
|
||||
--engine.parallelism.hsdp_shard_dim "$NUM_GPUS" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}" \
|
||||
"${self_forcing_args[@]}"
|
||||
|
||||
@@ -1,68 +0,0 @@
|
||||
# Read by distill_dmd_t2v_1.3B.sh, which passes model_path, engine.num_gpus, engine.parallelism.hsdp_shard_dim,
|
||||
# training.distillation.real_score_model_path, training.distillation.fake_score_model_path, training.data.data_path, and
|
||||
# training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 1 # 64
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
components:
|
||||
transformer_weights: your_ode_init_weights_path
|
||||
flow_shift: 5.0
|
||||
dmd_denoising_steps: [1000, 750, 500, 250]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 480
|
||||
num_width: 832
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 1.0e-05
|
||||
betas: [0.0, 0.999]
|
||||
weight_decay: 0.01
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
output_dir: your_output_dir
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 500
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: SFwan_t2v_distill_self_forcing_dmd
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [4]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 50
|
||||
log_visualization: true
|
||||
distillation:
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.0
|
||||
fake_score_learning_rate: 8.0e-06
|
||||
fake_score_betas: [0.0, 0.999]
|
||||
simulate_generator_forward: true
|
||||
warp_denoising_step: true
|
||||
ema:
|
||||
enabled: true
|
||||
decay: 0.99
|
||||
start_step: 100
|
||||
self_forcing:
|
||||
dfake_gen_update_ratio: 5
|
||||
num_frame_per_block: 3 # Frame generation block size for self-forcing
|
||||
independent_first_frame: false # Whether to treat first frame independently
|
||||
same_step_across_blocks: true # Whether to use same denoising step across all blocks
|
||||
last_step_only: false # Whether to only use the last denoising step
|
||||
context_noise: 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -8,7 +8,17 @@ OUTPUT_DIR="data/crush-smol_processed_t2v/"
|
||||
|
||||
torchrun --nproc_per_node=$GPU_NUM \
|
||||
fastvideo/pipelines/preprocess/v1_preprocess.py \
|
||||
--config examples/distill/SFWan2.1-T2V/preprocess_data.yaml \
|
||||
--model_path $MODEL_PATH \
|
||||
--preprocess.data_merge_path $DATA_MERGE_PATH \
|
||||
--preprocess.dataset_output_dir=$OUTPUT_DIR
|
||||
--data_merge_path $DATA_MERGE_PATH \
|
||||
--preprocess_video_batch_size 8 \
|
||||
--seed 42 \
|
||||
--max_height 480 \
|
||||
--max_width 832 \
|
||||
--num_frames 81 \
|
||||
--dataloader_num_workers 0 \
|
||||
--output_dir=$OUTPUT_DIR \
|
||||
--train_fps 16 \
|
||||
--samples_per_file 8 \
|
||||
--flush_frequency 8 \
|
||||
--video_length_tolerance_range 5 \
|
||||
--preprocess_task "t2v"
|
||||
|
||||
@@ -1,19 +0,0 @@
|
||||
# Read by preprocess_data.sh, which passes model_path, preprocess.data_merge_path, and preprocess.dataset_output_dir.
|
||||
# Offload settings of the per-task preprocessing pipelines (v1_preprocess.py).
|
||||
engine:
|
||||
offload:
|
||||
dit_layerwise: true
|
||||
image_encoder: true
|
||||
pin_cpu_memory: true
|
||||
preprocess:
|
||||
preprocess_video_batch_size: 8
|
||||
seed: 42
|
||||
max_height: 480
|
||||
max_width: 832
|
||||
num_frames: 81
|
||||
dataloader_num_workers: 0
|
||||
train_fps: 16
|
||||
samples_per_file: 8
|
||||
flush_frequency: 8
|
||||
video_length_tolerance_range: 5
|
||||
preprocess_task: t2v
|
||||
@@ -49,6 +49,96 @@ VALIDATION_DATASET_FILE="/mnt/weka/home/hao.zhang/wl/FastVideo/examples/distill/
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
training_args=(
|
||||
--tracker_project_name SFwan2.2_t2v_distill_self_forcing_dmd # Updated for Wan2.2
|
||||
--output_dir "/mnt/sharefs/users/hao.zhang/SFwan2.2_t2v_finetune"
|
||||
--override_transformer_cls_name "CausalWanTransformer3DModel"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 448 # Updated to match Wan2.2 config
|
||||
--num_width 832 # Updated to match Wan2.2 config
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
--simulate_generator_forward
|
||||
# --log_visualization
|
||||
--num_frames 81
|
||||
--num_frame_per_block 3 # Frame generation block size for self-forcing
|
||||
--enable_gradient_masking
|
||||
--gradient_mask_last_n_frames 21
|
||||
# --init_weights_from_safetensors /mnt/sharefs/users/hao.zhang/wl/models/sf_ode_init_wan22_checkpoints/high/3k/
|
||||
# --init_weights_from_safetensors_2 /mnt/sharefs/users/hao.zhang/wl/models/sf_ode_init_wan22_checkpoints/low/3k/
|
||||
)
|
||||
|
||||
parallel_args=(
|
||||
--num_gpus 32 # 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1 # 64
|
||||
--hsdp_shard_dim 32
|
||||
)
|
||||
|
||||
model_args=(
|
||||
--model_path $GENERATOR_MODEL_PATH
|
||||
--pretrained_model_name_or_path $GENERATOR_MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 20
|
||||
--validation_sampling_steps "4"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
optimizer_args=(
|
||||
--learning_rate 1e-5
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 500
|
||||
--weight_decay 0.01
|
||||
--betas '0.0,0.999'
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
--use_ema True
|
||||
--ema_decay 0.99
|
||||
--ema_start_step 100
|
||||
)
|
||||
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,750,500,250'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--dfake_gen_update_ratio 5
|
||||
--real_score_guidance_scale 3.0
|
||||
--fake_score_learning_rate 8e-6
|
||||
--fake_score_betas '0.0,0.999'
|
||||
--warp_denoising_step
|
||||
)
|
||||
|
||||
self_forcing_args=(
|
||||
--independent_first_frame False # Whether to treat first frame independently
|
||||
--same_step_across_blocks True # Whether to use same denoising step across all blocks
|
||||
--last_step_only False # Whether to only use the last denoising step
|
||||
--context_noise 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
@@ -56,9 +146,12 @@ srun torchrun \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_self_forcing_distillation_pipeline.py \
|
||||
--config examples/distill/SFWan2.2-A14B/distill_dmd.yaml \
|
||||
--model_path "$GENERATOR_MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}" \
|
||||
"${self_forcing_args[@]}"
|
||||
@@ -1,72 +0,0 @@
|
||||
# Read by distill_dmd.sh, which passes model_path, training.distillation.real_score_model_path,
|
||||
# training.distillation.fake_score_model_path, training.data.data_path, and training.validation.dataset_file on the
|
||||
# command line.
|
||||
engine:
|
||||
num_gpus: 32 # 64
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 1 # 64
|
||||
hsdp_shard_dim: 32
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
components:
|
||||
# transformer_weights: /mnt/sharefs/users/hao.zhang/wl/models/sf_ode_init_wan22_checkpoints/high/3k/
|
||||
# transformer_2_weights: /mnt/sharefs/users/hao.zhang/wl/models/sf_ode_init_wan22_checkpoints/low/3k/
|
||||
override_transformer_cls_name: CausalWanTransformer3DModel
|
||||
flow_shift: 5.0
|
||||
dmd_denoising_steps: [1000, 750, 500, 250]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 448 # Updated to match Wan2.2 config
|
||||
num_width: 832 # Updated to match Wan2.2 config
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 1.0e-05
|
||||
betas: [0.0, 0.999]
|
||||
weight_decay: 0.01
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
output_dir: /mnt/sharefs/users/hao.zhang/SFwan2.2_t2v_finetune
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 500
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: SFwan2.2_t2v_distill_self_forcing_dmd # Updated for Wan2.2
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [4]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 20
|
||||
# log_visualization: true
|
||||
distillation:
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.0
|
||||
fake_score_learning_rate: 8.0e-06
|
||||
fake_score_betas: [0.0, 0.999]
|
||||
simulate_generator_forward: true
|
||||
warp_denoising_step: true
|
||||
ema:
|
||||
enabled: true
|
||||
decay: 0.99
|
||||
start_step: 100
|
||||
self_forcing:
|
||||
dfake_gen_update_ratio: 5
|
||||
num_frame_per_block: 3 # Frame generation block size for self-forcing
|
||||
independent_first_frame: false # Whether to treat first frame independently
|
||||
same_step_across_blocks: true # Whether to use same denoising step across all blocks
|
||||
last_step_only: false # Whether to only use the last denoising step
|
||||
context_noise: 0 # Amount of noise to add during context caching (0 = no noise)
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -47,6 +47,84 @@ OUTPUT_DIR="checkpoints/wan_t2v_finetune"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name wan_t2v_distill_dmd_VSA
|
||||
--output_dir $OUTPUT_DIR
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 81
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 2e-6
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 500
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 8
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3.5
|
||||
--VSA_sparsity 0.8
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
@@ -54,10 +132,11 @@ srun torchrun \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
--config examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/distill_dmd_VSA_t2v_1.3B.yaml \
|
||||
--model_path "$MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.checkpoint.output_dir "$OUTPUT_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
|
||||
@@ -1,55 +0,0 @@
|
||||
# Read by distill_dmd_VSA_t2v_1.3B.slurm, which passes model_path, training.distillation.real_score_model_path,
|
||||
# training.distillation.fake_score_model_path, training.data.data_path, training.checkpoint.output_dir, and
|
||||
# training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
num_gpus: 64
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 64
|
||||
hsdp_shard_dim: 1
|
||||
attention:
|
||||
vsa_sparsity: 0.8
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
flow_shift: 8.0
|
||||
dmd_denoising_steps: [1000, 757, 522]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 480
|
||||
num_width: 832
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 2.0e-06
|
||||
weight_decay: 0.01
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 500
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: wan_t2v_distill_dmd_VSA
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [3]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 200
|
||||
distillation:
|
||||
generator_update_interval: 5
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.5
|
||||
ema:
|
||||
start_step: 0
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -47,6 +47,84 @@ OUTPUT_DIR="checkpoints/wan_t2v_finetune"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name wan_t2v_distill_dmd_VSA
|
||||
--output_dir "$OUTPUT_DIR"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 81
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 4
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 8
|
||||
--hsdp_shard_dim 8
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 2e-6
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 500
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 3
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3.5
|
||||
--VSA_sparsity 0.9
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
@@ -54,10 +132,11 @@ srun torchrun \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
--config examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/distill_dmd_VSA_t2v_14B.yaml \
|
||||
--model_path "$MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.checkpoint.output_dir "$OUTPUT_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
|
||||
@@ -1,55 +0,0 @@
|
||||
# Read by distill_dmd_VSA_t2v_14B.slurm, which passes model_path, training.distillation.real_score_model_path,
|
||||
# training.distillation.fake_score_model_path, training.data.data_path, training.checkpoint.output_dir, and
|
||||
# training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
num_gpus: 64
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 4
|
||||
hsdp_replicate_dim: 8
|
||||
hsdp_shard_dim: 8
|
||||
attention:
|
||||
vsa_sparsity: 0.9
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
flow_shift: 3.0
|
||||
dmd_denoising_steps: [1000, 757, 522]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 480
|
||||
num_width: 832
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 2.0e-06
|
||||
weight_decay: 0.01
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 500
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: wan_t2v_distill_dmd_VSA
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [3]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 200
|
||||
distillation:
|
||||
generator_update_interval: 5
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.5
|
||||
ema:
|
||||
start_step: 0
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -47,6 +47,83 @@ OUTPUT_DIR="checkpoints/wan_t2v_finetune"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name wan_t2v_distill_dmd
|
||||
--output_dir "$OUTPUT_DIR"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 21
|
||||
--num_height 480
|
||||
--num_width 832
|
||||
--num_frames 81
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 2e-6
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 500
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 8
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3.5
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
@@ -54,10 +131,11 @@ srun torchrun \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
--config examples/distill/Wan2.1-T2V/Wan-Syn-Data-480P/distill_dmd_t2v_1.3B.yaml \
|
||||
--model_path "$MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.checkpoint.output_dir "$OUTPUT_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
|
||||
@@ -1,53 +0,0 @@
|
||||
# Read by distill_dmd_t2v_1.3B.slurm, which passes model_path, training.distillation.real_score_model_path,
|
||||
# training.distillation.fake_score_model_path, training.data.data_path, training.checkpoint.output_dir, and
|
||||
# training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
num_gpus: 64
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 64
|
||||
hsdp_shard_dim: 1
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
flow_shift: 8.0
|
||||
dmd_denoising_steps: [1000, 757, 522]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 480
|
||||
num_width: 832
|
||||
num_frames: 81
|
||||
train_batch_size: 1
|
||||
num_latent_t: 21
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 2.0e-06
|
||||
weight_decay: 0.01
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 500
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: wan_t2v_distill_dmd
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [3]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 200
|
||||
distillation:
|
||||
generator_update_interval: 5
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.5
|
||||
ema:
|
||||
start_step: 0
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -8,4 +8,4 @@ uv pip install vsa
|
||||
```
|
||||
|
||||
### Data-free Distillation
|
||||
When `training.distillation.simulate_generator_forward` is enabled, distillation becomes data-free by simulating intermediate steps through forward inference of the generator. This helps avoid training–inference mismatch. See Section 4.5 of [DMD2](https://arxiv.org/pdf/2405.14867) for details.
|
||||
When `--simulate_generator_forward` is enabled, distillation becomes data-free by simulating intermediate steps through forward inference of the generator. This helps avoid training–inference mismatch. See Section 4.5 of [DMD2](https://arxiv.org/pdf/2405.14867) for details.
|
||||
@@ -48,6 +48,90 @@ OUTPUT_DIR="checkpoints/wan_t2v_finetune"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name Wan_distillation
|
||||
--output_dir "$OUTPUT_DIR"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 31
|
||||
--num_height 704
|
||||
--num_width 1280
|
||||
--num_frames 121
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DIR"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 4e-6
|
||||
--lr_scheduler "cosine_with_min_lr"
|
||||
--min_lr_ratio 0.5
|
||||
--lr_warmup_steps 100
|
||||
--fake_score_learning_rate 2e-6
|
||||
--fake_score_lr_scheduler "cosine_with_min_lr"
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3
|
||||
--simulate_generator_forward
|
||||
--log_visualization # disable if oom
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
@@ -55,10 +139,11 @@ srun torchrun \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
--config examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free/distill_dmd_t2v_5B.yaml \
|
||||
--model_path "$MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.checkpoint.output_dir "$OUTPUT_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DIR"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
|
||||
@@ -1,60 +0,0 @@
|
||||
# Read by distill_dmd_t2v_5B.sh, which passes model_path, training.distillation.real_score_model_path,
|
||||
# training.distillation.fake_score_model_path, training.data.data_path, training.checkpoint.output_dir, and
|
||||
# training.validation.dataset_file on the command line.
|
||||
engine:
|
||||
num_gpus: 64
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 64
|
||||
hsdp_shard_dim: 1
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
flow_shift: 5.0
|
||||
dmd_denoising_steps: [1000, 757, 522]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 704
|
||||
num_width: 1280
|
||||
num_frames: 121
|
||||
train_batch_size: 1
|
||||
num_latent_t: 31
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 4.0e-06
|
||||
weight_decay: 0.01
|
||||
lr_scheduler: cosine_with_min_lr
|
||||
lr_warmup_steps: 100
|
||||
min_lr_ratio: 0.5
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 200
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: Wan_distillation
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [3]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 200
|
||||
log_visualization: true # disable if oom
|
||||
distillation:
|
||||
generator_update_interval: 5
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.0
|
||||
fake_score_learning_rate: 2.0e-06
|
||||
fake_score_lr_scheduler: cosine_with_min_lr
|
||||
simulate_generator_forward: true
|
||||
ema:
|
||||
start_step: 0
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -47,6 +47,91 @@ VALIDATION_DIR=your_validation_path #(example:validation_64.json)
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name Wan_distillation
|
||||
--output_dir "your_output_dir"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 31
|
||||
--num_height 704
|
||||
--num_width 1280
|
||||
--num_frames 121
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 64
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 64
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DIR"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 4e-6
|
||||
--lr_scheduler "cosine_with_min_lr"
|
||||
--min_lr_ratio 0.5
|
||||
--lr_warmup_steps 100
|
||||
--fake_score_learning_rate 2e-6
|
||||
--fake_score_lr_scheduler "cosine_with_min_lr"
|
||||
--mixed_precision "bf16"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 200
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 5
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3
|
||||
--simulate_generator_forward
|
||||
--log_visualization # disable if oom
|
||||
--VSA_sparsity 0.8
|
||||
)
|
||||
|
||||
srun torchrun \
|
||||
--nnodes $SLURM_JOB_NUM_NODES \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
@@ -54,9 +139,11 @@ srun torchrun \
|
||||
--rdzv_backend=c10d \
|
||||
--rdzv_endpoint="$MASTER_ADDR:$MASTER_PORT" \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
--config examples/distill/Wan2.2-TI2V-5B-Diffusers/Data-free/distill_dmd_t2v_5B_VSA.yaml \
|
||||
--model_path "$MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DIR"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
|
||||
@@ -1,63 +0,0 @@
|
||||
# Read by distill_dmd_t2v_5B_VSA.sh, which passes model_path, training.distillation.real_score_model_path,
|
||||
# training.distillation.fake_score_model_path, training.data.data_path, and training.validation.dataset_file on the
|
||||
# command line.
|
||||
engine:
|
||||
num_gpus: 64
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
hsdp_replicate_dim: 64
|
||||
hsdp_shard_dim: 1
|
||||
attention:
|
||||
vsa_sparsity: 0.8
|
||||
precision:
|
||||
dit: fp32
|
||||
pipeline:
|
||||
flow_shift: 5.0
|
||||
dmd_denoising_steps: [1000, 757, 522]
|
||||
training:
|
||||
data:
|
||||
dataloader_num_workers: 4
|
||||
num_height: 704
|
||||
num_width: 1280
|
||||
num_frames: 121
|
||||
train_batch_size: 1
|
||||
num_latent_t: 31
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
train_sp_batch_size: 1
|
||||
optimizer:
|
||||
learning_rate: 4.0e-06
|
||||
weight_decay: 0.01
|
||||
lr_scheduler: cosine_with_min_lr
|
||||
lr_warmup_steps: 100
|
||||
min_lr_ratio: 0.5
|
||||
max_grad_norm: 1.0
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
checkpoint:
|
||||
output_dir: your_output_dir
|
||||
training_state_checkpointing_steps: 500
|
||||
weight_only_checkpointing_steps: 200
|
||||
checkpoints_total_limit: 3
|
||||
tracker:
|
||||
project_name: Wan_distillation
|
||||
validation:
|
||||
enabled: true
|
||||
sampling_steps: [3]
|
||||
guidance_scale: 6.0 # not used for dmd inference
|
||||
every_steps: 200
|
||||
log_visualization: true # disable if oom
|
||||
distillation:
|
||||
generator_update_interval: 5
|
||||
min_timestep_ratio: 0.02
|
||||
max_timestep_ratio: 0.98
|
||||
real_score_guidance_scale: 3.0
|
||||
fake_score_learning_rate: 2.0e-06
|
||||
fake_score_lr_scheduler: cosine_with_min_lr
|
||||
simulate_generator_forward: true
|
||||
ema:
|
||||
start_step: 0
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
@@ -22,15 +22,94 @@ OUTPUT_DIR="checkpoints/wan_t2v_finetune"
|
||||
# export CUDA_VISIBLE_DEVICES=4,5
|
||||
# IP=[MASTER NODE IP]
|
||||
|
||||
# Training arguments
|
||||
training_args=(
|
||||
--tracker_project_name wan_t2v_distill_dmd_VSA
|
||||
--output_dir "$OUTPUT_DIR"
|
||||
--max_train_steps 4000
|
||||
--train_batch_size 1
|
||||
--train_sp_batch_size 1
|
||||
--gradient_accumulation_steps 1
|
||||
--num_latent_t 31
|
||||
--num_height 704
|
||||
--num_width 1280
|
||||
--num_frames 121
|
||||
--enable_gradient_checkpointing_type "full"
|
||||
--training_state_checkpointing_steps 500
|
||||
--weight_only_checkpointing_steps 500
|
||||
)
|
||||
|
||||
# Parallel arguments
|
||||
parallel_args=(
|
||||
--num_gpus 1
|
||||
--sp_size 1
|
||||
--tp_size 1
|
||||
--hsdp_replicate_dim 1
|
||||
--hsdp_shard_dim 1
|
||||
)
|
||||
|
||||
# Model arguments
|
||||
model_args=(
|
||||
--model_path $MODEL_PATH
|
||||
--pretrained_model_name_or_path $MODEL_PATH
|
||||
--real_score_model_path $REAL_SCORE_MODEL_PATH
|
||||
--fake_score_model_path $FAKE_SCORE_MODEL_PATH
|
||||
)
|
||||
|
||||
# Dataset arguments
|
||||
dataset_args=(
|
||||
--data_path "$DATA_DIR"
|
||||
--dataloader_num_workers 4
|
||||
)
|
||||
|
||||
# Validation arguments
|
||||
validation_args=(
|
||||
--log_validation
|
||||
--validation_dataset_file "$VALIDATION_DATASET_FILE"
|
||||
--validation_steps 200
|
||||
--validation_sampling_steps "3"
|
||||
--validation_guidance_scale "6.0" # not used for dmd inference
|
||||
)
|
||||
|
||||
# Optimizer arguments
|
||||
optimizer_args=(
|
||||
--learning_rate 2e-6
|
||||
--mixed_precision "bf16"
|
||||
--weight_decay 0.01
|
||||
--max_grad_norm 1.0
|
||||
)
|
||||
|
||||
# Miscellaneous arguments
|
||||
miscellaneous_args=(
|
||||
--inference_mode False
|
||||
--checkpoints_total_limit 3
|
||||
--training_cfg_rate 0.0
|
||||
--dit_precision "fp32"
|
||||
--ema_start_step 0
|
||||
--flow_shift 8
|
||||
--seed 1000
|
||||
)
|
||||
|
||||
# DMD arguments
|
||||
dmd_args=(
|
||||
--dmd_denoising_steps '1000,757,522'
|
||||
--min_timestep_ratio 0.02
|
||||
--max_timestep_ratio 0.98
|
||||
--generator_update_interval 5
|
||||
--real_score_guidance_scale 3.5
|
||||
--VSA_sparsity 0.8
|
||||
)
|
||||
|
||||
torchrun \
|
||||
--nnodes 1 \
|
||||
--nproc_per_node $NUM_GPUS \
|
||||
--master_port $MASTER_PORT \
|
||||
fastvideo/training/wan_distillation_pipeline.py \
|
||||
--config examples/distill/Wan2.2-TI2V-5B-Diffusers/crush_smol/distill_dmd_VSA_t2v_5B.yaml \
|
||||
--model_path "$MODEL_PATH" \
|
||||
--training.distillation.real_score_model_path "$REAL_SCORE_MODEL_PATH" \
|
||||
--training.distillation.fake_score_model_path "$FAKE_SCORE_MODEL_PATH" \
|
||||
--training.data.data_path "$DATA_DIR" \
|
||||
--training.checkpoint.output_dir "$OUTPUT_DIR" \
|
||||
--training.validation.dataset_file "$VALIDATION_DATASET_FILE"
|
||||
"${parallel_args[@]}" \
|
||||
"${model_args[@]}" \
|
||||
"${dataset_args[@]}" \
|
||||
"${training_args[@]}" \
|
||||
"${optimizer_args[@]}" \
|
||||
"${validation_args[@]}" \
|
||||
"${miscellaneous_args[@]}" \
|
||||
"${dmd_args[@]}"
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user