Compare commits
108
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1ea7278c00 | ||
|
|
a5aa64ab6b | ||
|
|
0bcd15d5e2 | ||
|
|
0b0c692012 | ||
|
|
6c2a250fd0 | ||
|
|
0607bd78c7 | ||
|
|
c4bb13a650 | ||
|
|
d051603677 | ||
|
|
00a8e5bcb6 | ||
|
|
4117696a4f | ||
|
|
23cb8c0c2b | ||
|
|
ca7965ed28 | ||
|
|
61dffbe904 | ||
|
|
7ec6251133 | ||
|
|
b323cb787c | ||
|
|
e263249175 | ||
|
|
6c776a219b | ||
|
|
3f6893a098 | ||
|
|
d728a0dc5f | ||
|
|
d6ba5ced94 | ||
|
|
9139c0411b | ||
|
|
f571621ae7 | ||
|
|
96b7c0d223 | ||
|
|
74a52c027b | ||
|
|
2cd2018137 | ||
|
|
6627366805 | ||
|
|
7aa1f78e10 | ||
|
|
38096b94a3 | ||
|
|
4d05e9c020 | ||
|
|
62780cf53b | ||
|
|
40ff9af3fb | ||
|
|
93d921fe33 | ||
|
|
d26aa6b2d7 | ||
|
|
b9c397dd1a | ||
|
|
867f960bf0 | ||
|
|
1d7e7e2e35 | ||
|
|
0d93b16fdd | ||
|
|
7ba7dbbbd5 | ||
|
|
c7c2d77fae | ||
|
|
991fa391ff | ||
|
|
ecb15bf9e0 | ||
|
|
b24396561a | ||
|
|
0b80dd52db | ||
|
|
04878f0584 | ||
|
|
cd1ffc09c9 | ||
|
|
db4a60c5db | ||
|
|
9491c8638a | ||
|
|
6809a751fb | ||
|
|
8322b01815 | ||
|
|
9edc8adf5f | ||
|
|
e1f3904799 | ||
|
|
02f1ce11ae | ||
|
|
7f03e03dc6 | ||
|
|
e3b88bb12a | ||
|
|
cb66acd400 | ||
|
|
442e2d2e18 | ||
|
|
dd35763ad6 | ||
|
|
e90be598e5 | ||
|
|
ba5e81083c | ||
|
|
76ce9c7fd6 | ||
|
|
08d99c089e | ||
|
|
20751a21aa | ||
|
|
9dd2a837f4 | ||
|
|
93aab45ac2 | ||
|
|
017ce6602d | ||
|
|
a575055eec | ||
|
|
81f3fec7fd | ||
|
|
d265a454bf | ||
|
|
c100c66578 | ||
|
|
8760eb7a06 | ||
|
|
361f919c88 | ||
|
|
8b5377aab2 | ||
|
|
d995516da0 | ||
|
|
f47ad3f5b7 | ||
|
|
10bdf5e076 | ||
|
|
3e26db40b0 | ||
|
|
c73dd0ab55 | ||
|
|
430e52154e | ||
|
|
384eee8aef | ||
|
|
c4824c7764 | ||
|
|
9b0e57fe4b | ||
|
|
0100218594 | ||
|
|
39718cd54d | ||
|
|
37d06a832f | ||
|
|
8839ba8d4d | ||
|
|
61b91220c0 | ||
|
|
316f3876c2 | ||
|
|
614b59543c | ||
|
|
1c14afd559 | ||
|
|
9a3c45779c | ||
|
|
bfc9c01797 | ||
|
|
3a3ad3d209 | ||
|
|
aef4e9b3b1 | ||
|
|
a943220c11 | ||
|
|
556ac7088e | ||
|
|
e7456f1b75 | ||
|
|
c993d7393e | ||
|
|
7f83164233 | ||
|
|
4e52f47d1e | ||
|
|
e19913f6e9 | ||
|
|
2413a57651 | ||
|
|
7bb76b5ec9 | ||
|
|
0bd19a976b | ||
|
|
40b93784d2 | ||
|
|
33d3478bad | ||
|
|
3d8ac9d14b | ||
|
|
aaef49bfc6 | ||
|
|
cf6a00b9be |
@@ -80,26 +80,36 @@ def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
str(model_path),
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=False,
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False,
|
||||
"offload": {
|
||||
"dit": False,
|
||||
"vae": False,
|
||||
"text_encoder": False,
|
||||
},
|
||||
},
|
||||
},
|
||||
)
|
||||
try:
|
||||
return generator.generate_video(
|
||||
prompt=params["prompt"],
|
||||
negative_prompt=params.get("negative_prompt"),
|
||||
output_path=f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
save_video=False,
|
||||
height=params.get("height"),
|
||||
width=params.get("width"),
|
||||
num_frames=params.get("num_frames"),
|
||||
fps=params.get("fps"),
|
||||
num_inference_steps=params["num_inference_steps"],
|
||||
guidance_scale=params.get("guidance_scale"),
|
||||
seed=params["seed"],
|
||||
)
|
||||
return generator.generate({
|
||||
"prompt": params["prompt"],
|
||||
"negative_prompt": params.get("negative_prompt"),
|
||||
"sampling": {
|
||||
"height": params.get("height"),
|
||||
"width": params.get("width"),
|
||||
"num_frames": params.get("num_frames"),
|
||||
"fps": params.get("fps"),
|
||||
"num_inference_steps": params["num_inference_steps"],
|
||||
"guidance_scale": params.get("guidance_scale"),
|
||||
"seed": params["seed"],
|
||||
},
|
||||
"output": {
|
||||
"output_path": f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
"save_video": False,
|
||||
},
|
||||
})
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
@@ -84,7 +84,9 @@ fastvideo/configs/models/dits/__init__.py
|
||||
fastvideo/configs/models/encoders/__init__.py
|
||||
fastvideo/configs/models/vaes/__init__.py
|
||||
fastvideo/envs.py
|
||||
fastvideo/fastvideo_args.py
|
||||
fastvideo/api/schema.py
|
||||
fastvideo/api/resolution.py
|
||||
fastvideo/api/inference_resolution.py
|
||||
fastvideo/distributed/**
|
||||
fastvideo/layers/**
|
||||
fastvideo/attention/**
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
---
|
||||
name: env-var-conventions
|
||||
description: Add, read, rename, or remove an environment variable in FastVideo, or change the environment-variable policy. Use before touching fastvideo/envs.py, os.environ, os.getenv, or monkeypatch.setenv in fastvideo/, and when fastvideo/tests/contract/test_env_policy.py fails.
|
||||
---
|
||||
|
||||
# Environment Variable Conventions
|
||||
|
||||
## Purpose
|
||||
|
||||
FastVideo registers its environment variables as typed fields in
|
||||
`fastvideo/envs.py`. The policy that governs them is
|
||||
`docs/contributing/env_vars.md`, and the contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit
|
||||
CI lane. This skill routes an environment-variable change through that policy.
|
||||
The policy doc is the single source of the rules; read it instead of relying
|
||||
on a summary here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Read `docs/contributing/env_vars.md` in full.
|
||||
- Decide whether the setting belongs in an environment variable or an argument
|
||||
(rule 5 in the policy doc). Settings that users change per deployment are
|
||||
arguments; add them as typed config fields in `fastvideo/api/schema.py`
|
||||
instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------- | -------- | -------------------------------------------------------------- |
|
||||
| `change` | Yes | Add, read, rename, or remove a variable, or change the policy. |
|
||||
| `variable` | Yes | The variable name, with the `FASTVIDEO_` prefix. |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Declare or edit the variable in `fastvideo/envs.py`.**
|
||||
- Pick the field type and category that the policy doc lists.
|
||||
- Write a description that states what the variable does and its units.
|
||||
- To rename, keep the old name in `deprecated_names`. To remove, add the
|
||||
name to `DEPRECATED_VARIABLES`. Update the uses in `examples/`,
|
||||
`scripts/`, `docs/`, `apps/`, and the tests.
|
||||
2. **Read the variable with `envs.NAME.get()` inside a function.**
|
||||
- In tests, change the value with `envs.NAME.override(value)`, and a variable
|
||||
outside the registry with `envs.override_external(name, value)`; the
|
||||
`env_overrides` fixture keeps either until the end of the test.
|
||||
- Name a variable that only tests read `FASTVIDEO_TEST_*`.
|
||||
- Do not call `os.environ`, `os.getenv`, or `monkeypatch.setenv` for a
|
||||
FastVideo variable.
|
||||
- To set a variable that another tool reads, call `envs.set_external`,
|
||||
`envs.setdefault_external`, or `envs.unset_external`.
|
||||
3. **Regenerate the table in the policy doc.**
|
||||
- Run `python fastvideo/tests/contract/test_env_policy.py`.
|
||||
4. **Run the contract test.**
|
||||
- Run `pytest fastvideo/tests/contract/test_env_policy.py`.
|
||||
- When the test reports a fixed known violation, delete or lower its entry
|
||||
in `KNOWN_VIOLATIONS`. Never add an entry to `KNOWN_VIOLATIONS`.
|
||||
5. **When the policy itself changes, update the policy doc and the contract
|
||||
test in the same pull request.**
|
||||
- The rules in `docs/contributing/env_vars.md`, the checks and allowlist in
|
||||
`fastvideo/tests/contract/test_env_policy.py`, and this skill must agree.
|
||||
|
||||
## Outputs
|
||||
|
||||
- A registry entry in `fastvideo/envs.py` and call sites that use
|
||||
`envs.NAME.get()`.
|
||||
- A regenerated table in `docs/contributing/env_vars.md`.
|
||||
- A passing `fastvideo/tests/contract/test_env_policy.py`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Add a FASTVIDEO_DEBUG_MY_STAGE switch that logs MyStage inputs.
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- `docs/contributing/env_vars.md`: the policy, the field types, and the
|
||||
violation kinds that the contract test reports.
|
||||
- `fastvideo/envs.py`: the registry.
|
||||
- `fastvideo/tests/contract/test_env_policy.py`: the contract test,
|
||||
`EXTERNAL_ALLOWLIST`, and `KNOWN_VIOLATIONS`.
|
||||
@@ -57,7 +57,7 @@ Hardcoded:
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
`FASTVIDEO_SSIM_REFERENCE_HF_REPO`).
|
||||
`FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
@@ -185,6 +185,7 @@ steps:
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
build.env("TEST_SCOPE") == "scheduled" ||
|
||||
@@ -211,6 +212,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
key: "lora-inference"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -232,6 +234,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Extraction Tests"
|
||||
key: "lora-extraction"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -253,6 +256,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Training Tests"
|
||||
key: "training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -276,6 +280,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
key: "distillation"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -297,6 +302,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
key: "self-forcing"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -318,6 +324,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
key: "lora-training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -341,6 +348,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
key: "training-vsa"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -364,6 +372,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
key: "inference-vmoba"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -385,6 +394,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Performance Tests"
|
||||
key: "performance"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -406,6 +416,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: API Server Tests"
|
||||
key: "api-server"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -427,6 +438,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
key: "train-framework"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -448,6 +460,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Eval Metrics Tests"
|
||||
key: "eval"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
|
||||
@@ -13,7 +13,7 @@ if [ -z "$selected" ]; then
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" = all ]; then
|
||||
exec pytest "$golden_root" -vs
|
||||
exec pytest "$golden_root" -xvs
|
||||
fi
|
||||
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
@@ -32,4 +32,4 @@ for golden_file in "${golden_files[@]}"; do
|
||||
golden_paths+=("$golden_path")
|
||||
done
|
||||
|
||||
exec pytest "${golden_paths[@]}" -vs
|
||||
exec pytest "${golden_paths[@]}" -xvs
|
||||
|
||||
@@ -2,4 +2,8 @@
|
||||
# Canonical Slurm CI selection for the transformer lane.
|
||||
set -euo pipefail
|
||||
|
||||
# The existing block reference records an absent FASTVIDEO_FA4 (FA2). Keep
|
||||
# that reference identity; the component lane also selects FA2 explicitly.
|
||||
env -u FASTVIDEO_FA4 pytest ./fastvideo/tests/golden_gate/test_wan_t2v.py -xvs
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_causal.py -xvs
|
||||
exec pytest ./fastvideo/tests/transformers -vs
|
||||
|
||||
@@ -2,4 +2,5 @@
|
||||
# Canonical Slurm CI selection for the VAE lane.
|
||||
set -euo pipefail
|
||||
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_vae.py -xvs
|
||||
exec pytest ./fastvideo/tests/vaes -vs
|
||||
|
||||
@@ -16,6 +16,7 @@ exec pytest \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/attention/test_sdpa_metadata_mask_contract.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_tile_grad_safety.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
|
||||
@@ -211,8 +211,8 @@ FAMILY_COVERAGE = (
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", ),
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
(
|
||||
"test_causal_similarity.py",
|
||||
"test_wan_i2v_similarity.py",
|
||||
@@ -291,6 +291,18 @@ def _family_coverage(path: str) -> tuple[set[str], set[str]]:
|
||||
if family.pattern.search(normalized):
|
||||
golden.update(family.golden_tests)
|
||||
ssim.update(family.ssim_tests)
|
||||
# Select the component actually touched, including compatibility paths.
|
||||
# Family configs/pipeline wiring can affect all four Wan gates.
|
||||
if re.search(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)", normalized):
|
||||
if (normalized.endswith(("/wan/vae.py", "/wan/vae_config.py", "/vaes/wanvae.py"))
|
||||
or normalized.endswith("/wan/stages/conditioning.py")):
|
||||
golden = {"test_wan_vae.py"}
|
||||
elif normalized.endswith(("/wan/causal_transformer.py", "/dits/causal_wanvideo.py",
|
||||
"/wan/stages/causal_denoising.py")):
|
||||
golden = {"test_wan_causal.py"}
|
||||
elif (normalized == "fastvideo/models/dits/wanvideo.py"
|
||||
or normalized.endswith(("/wan/transformer.py", "/wan/stages/denoising.py", "/wan/stages/dmd.py"))):
|
||||
golden = {"test_wan_t2v.py", "test_wan_denoising.py"}
|
||||
return golden, ssim
|
||||
|
||||
|
||||
@@ -477,7 +489,6 @@ def classify_paths(paths: list[str]) -> MergePlan:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path in {
|
||||
"fastvideo/fastvideo_args.py",
|
||||
"fastvideo/forward_context.py",
|
||||
"fastvideo/image_processor.py",
|
||||
"fastvideo/registry.py",
|
||||
|
||||
@@ -92,6 +92,7 @@ jobs:
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
@@ -156,6 +157,7 @@ jobs:
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
|
||||
@@ -63,7 +63,7 @@ jobs:
|
||||
torch-cuda-short: 'cu130'
|
||||
platform:
|
||||
# x86_64 builds the full cu126 + cu130 set. cu130 ships the
|
||||
# data-center Blackwell sm_100a VSA and consumer sm_120a FP4
|
||||
# data-center Blackwell sm_100a/sm_103a VSA and consumer sm_120a FP4
|
||||
# kernels.
|
||||
- os: ubuntu-22.04
|
||||
arch: x86_64
|
||||
@@ -169,18 +169,18 @@ jobs:
|
||||
# covers sm_120a; turbodiffusion covers sm_100a+sm_120a. The sm_100 FP4
|
||||
# forward is the FA4 CuTe DSL path in the fastvideo package (PR #1221),
|
||||
# JIT-compiled at runtime — not built into this wheel.
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a VSA
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a/sm_103a VSA
|
||||
# + consumer Blackwell sm_120a FP4.
|
||||
# * x86_64 cu126 = Hopper TK only (older drivers; CUDA < 12.8 has no FP4).
|
||||
# The per-arch split in CMakeLists pins the FP4 targets to sm_120a and builds
|
||||
# the main extension for the full arch list. CMAKE_BUILD_PARALLEL_LEVEL caps
|
||||
# Ninja so heavy CUTLASS/TK template TUs don't OOM the 16 GB runner (exit 143).
|
||||
if [ "${{ matrix.platform.arch }}" = "aarch64" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;10.3a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=OFF -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON"
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
elif [ "${{ matrix.torch-cuda.torch-cuda-short }}" = "cu130" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;10.3a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
|
||||
# A single FP4 TU (attn_qat_infer) can use ~8-12 GB on its own, so serialize.
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
|
||||
@@ -9,7 +9,7 @@ exclude: |
|
||||
tests/.*|
|
||||
scripts/.*|
|
||||
fastvideo/dataset/.*|
|
||||
fastvideo/models/.*|
|
||||
fastvideo/models/(?!wan/(config|vae_config|pipeline_config|definition|__init__)\.py$).*|
|
||||
^apps/dreamverse/web/.*|
|
||||
examples/.*|
|
||||
\.agents/.*|
|
||||
|
||||
@@ -66,14 +66,18 @@ Local guidance lives next to the code. Read the in-scope file before editing:
|
||||
| `fastvideo/AGENTS.md` | Core package map, public API, registry-driven model dispatch |
|
||||
| `fastvideo/configs/AGENTS.md` | Arch + pipeline config dataclasses, `param_names_mapping` |
|
||||
| `fastvideo/models/AGENTS.md` | DiT / VAE / encoder / scheduler / loader layout (pre-commit excluded) |
|
||||
| `fastvideo/models/wan/AGENTS.md` | Wan family-local transformers, VAE, configs, and the SP sharding invariant |
|
||||
| `fastvideo/layers/AGENTS.md` | Tensor-parallel linear/attention layer rules for ports |
|
||||
| `fastvideo/attention/AGENTS.md` | Backend registry + env-var override |
|
||||
| `fastvideo/pipelines/AGENTS.md` | Stage ABC, `basic/<model>/`, `preprocess/`, presets |
|
||||
| `fastvideo/pipelines/basic/wan/AGENTS.md` | Wan sampling stages, first-frame conditioning, DMD/causal boundaries |
|
||||
| `fastvideo/pipelines/basic/magi_human/AGENTS.md` | MagiHuman umbrella repo, lazy-loaded components, packing invariants |
|
||||
| `fastvideo/training/AGENTS.md` | Legacy monolithic pipelines (frozen for existing models) |
|
||||
| `fastvideo/train/AGENTS.md` | New modular trainer (methods × models × callbacks, YAML) |
|
||||
| `fastvideo/tests/AGENTS.md` | Test taxonomy, conftest, pre-commit-excluded path |
|
||||
| `fastvideo/tests/ssim/AGENTS.md` | GPU SSIM regression authoring + reference video sync |
|
||||
| `scripts/checkpoint_conversion/AGENTS.md` | Adding a converter for a new HF/official checkpoint |
|
||||
| `apps/dreamverse/AGENTS.md` | DreamVerse app structure and conventions |
|
||||
|
||||
## Critical: Two Training Stacks Coexist
|
||||
|
||||
|
||||
@@ -3,14 +3,16 @@
|
||||
</div>
|
||||
|
||||
<p align="center">
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://haoailab.com/FastVideo/cookbook/"><b>Cookbook</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
</p>
|
||||
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
## NEWS
|
||||
- `2026/09/15`: Release [FastH3 8-Step V2](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2), an eight-forward data-free DMD2 checkpoint distilled from MiniMax-H3 with 80% Video Sparse Attention. Run it with `examples/inference/basic/basic_fasth3_8step.py` or the [FastH3 8-Step V2 recipe](https://haoailab.com/FastVideo/cookbook/minimax-h3/).
|
||||
- `2026/09/01`: FastH3 now runs locally on Apple Silicon through MLX and on NVIDIA DGX Spark through CUDA 13, including two-Spark inference. Follow the [FastH3 recipes](https://haoailab.com/FastVideo/cookbook/minimax-h3/) and read the [Blog](https://haoailab.com/blogs/fasth3-local/).
|
||||
- `2026/08/27`: [FastH3 Preview v1](https://haoailab.com/blogs/fasth3-preview/) is an open-weight 4-step sparse-distilled MiniMax-H3 model for synchronized video-and-audio generation, developed in collaboration with [Nuva Lab](https://nuvalab.ai/) and the [NVIDIA FastGen team](https://github.com/NVlabs/FastGen). Download the recommended [VSA / Data-Free weights](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree), or see the [full FastH3 collection](https://huggingface.co/collections/FastVideo/fastvideo-fasth3).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac—follow the [Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac. Follow the [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/06/23`: Release FastWan-QAD: 5s of Video generated in 1.8s E2E. See the [FastWan-QAD models](https://huggingface.co/FastVideo/FastWan-QAD-FP8-1.3B), [Attn-QAT training guide](https://haoailab.com/FastVideo/training/attn_qat/), and [blog](https://haoailab.com/blogs/fastwan-qad/).
|
||||
- `2026/03/17`: Release demo: Into the Dreamverse: Vibe Directing in FastVideo, check out the [Blog](https://haoailab.com/blogs/dreamverse/).
|
||||
- `2026/03/13`: Release demo: Create a 5s 1080p Video in 4.5s with FastVideo on a Single GPU, check out the [Blog](https://haoailab.com/blogs/fastvideo_realtime_1080p/).
|
||||
@@ -62,13 +64,12 @@ UV_TORCH_BACKEND=cu126 uv pip install fastvideo
|
||||
```
|
||||
|
||||
Use `UV_TORCH_BACKEND=cu130` on CUDA 13. Apple silicon users should follow the
|
||||
[MPS installation guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
> **On an Apple Silicon Mac?** FastVideo runs FastMetal-QAD through an MLX
|
||||
> runtime. Install with `uv pip install -e '.[mlx]'`, download
|
||||
> [`FastVideo/FastMetal-1.3B-QAD`](https://huggingface.co/FastVideo/FastMetal-1.3B-QAD),
|
||||
> and follow the
|
||||
> [Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
> **On an Apple Silicon Mac?** Install with `uv pip install -e '.[mlx]'` from
|
||||
> a clone, then pick a recipe in the
|
||||
> [cookbook](https://haoailab.com/FastVideo/cookbook/). See the
|
||||
> [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/) for more detailed installation instructions.
|
||||
|
||||
@@ -86,7 +87,7 @@ Install FastVideo (https://github.com/hao-ai-lab/FastVideo) into a fresh uv virt
|
||||
https://hao-ai-lab.github.io/FastVideo/getting_started/installation/):
|
||||
- NVIDIA GPU, x86_64 -> docs/getting_started/installation/gpu.md
|
||||
- NVIDIA DGX Spark / GB10, aarch64, CUDA 13 -> docs/getting_started/installation/spark.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mps.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mlx.md
|
||||
3. Use uv for every step. If a command fails, debug it and tell me what you changed.
|
||||
4. Verify the result:
|
||||
python -c "import fastvideo, torch; print('cuda', torch.cuda.is_available())"
|
||||
@@ -134,18 +135,20 @@ def main():
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
{"engine": {"num_gpus": 1}}, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
# Define a prompt for your video
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
|
||||
|
||||
# Generate the video
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
||||
@@ -97,13 +97,33 @@ dreamverse-server --port 8009
|
||||
dreamverse-mock-server --port 8009
|
||||
```
|
||||
|
||||
### Run Dreamverse with FastH3
|
||||
|
||||
Select the VSA data-free FastH3 Preview profile when you start the backend:
|
||||
|
||||
```bash
|
||||
DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
The `fast-h3` profile uses four visible GPUs by default. It loads the `MiniMaxAI/MiniMax-H3` base checkpoint and the
|
||||
`vsa-datafree/adapter_model.safetensors` adapter from
|
||||
`FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA`. Each request generates a 124-frame, 768×1344 video with
|
||||
synchronized audio and five sigma-grid points. Dreamverse uses the last frame of each segment as first-frame
|
||||
conditioning for the following segment.
|
||||
|
||||
Set `CUDA_VISIBLE_DEVICES` when you need to choose the four physical GPUs:
|
||||
|
||||
```bash
|
||||
CUDA_VISIBLE_DEVICES=0,1,2,3 DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
> **Expect a slow first boot.** With `torch.compile` and startup warmup enabled
|
||||
> (the default), the backend compiles the segment 1 and segment 2 inference
|
||||
> paths before it reports ready — this can take **tens of minutes on a cold
|
||||
> cache**, regardless of how you deploy (local, server, Docker, or Modal).
|
||||
> `/healthz` responds as soon as the process is up; `/readyz` stays `503` until
|
||||
> warmup finishes. For a faster, uncompiled startup while testing, set
|
||||
> `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
> warmup finishes. To defer compilation until the first generated request while
|
||||
> testing, set `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
|
||||
## Frontend Setup
|
||||
|
||||
@@ -219,6 +239,7 @@ selection, and mock-server behavior:
|
||||
pytest apps/dreamverse/dreamverse/tests/test_config.py \
|
||||
apps/dreamverse/dreamverse/tests/test_entrypoints.py \
|
||||
apps/dreamverse/dreamverse/tests/test_gpu_pool.py \
|
||||
apps/dreamverse/dreamverse/tests/test_minimax_h3_generation.py \
|
||||
apps/dreamverse/dreamverse/tests/test_mock_server.py -q
|
||||
```
|
||||
|
||||
|
||||
+13
-2
@@ -42,7 +42,7 @@ Near-term OSS note:
|
||||
- `apps/dreamverse/dreamverse/main.py`: websocket endpoint, request handling,
|
||||
session state machine, rewrite orchestration, REST routes, and stream relay
|
||||
- `apps/dreamverse/dreamverse/gpu_pool.py`: GPU worker processes, warmup, model
|
||||
loading, and `generate_video()` calls through FastVideo
|
||||
loading, and `generate()` calls through FastVideo
|
||||
- `apps/dreamverse/dreamverse/prompt_enhancer.py`: prompt enhancement, rollout
|
||||
rewrite execution, provider selection, and timeout/fallback behavior
|
||||
- `apps/dreamverse/dreamverse/rewrite_prompt_payload.py`: canonical rewrite request payload
|
||||
@@ -139,7 +139,18 @@ session.
|
||||
- startup warmup
|
||||
- user join/leave commands
|
||||
- `USER_STEP` execution for each segment
|
||||
- continuation state between segments
|
||||
- generation-command routing and stream-result delivery
|
||||
|
||||
Model generation has a separate ownership boundary inside each GPU process:
|
||||
|
||||
- `apps/dreamverse/dreamverse/generation_worker.py` selects the backend that the active model profile declares and owns
|
||||
the backend lifecycle.
|
||||
- `apps/dreamverse/dreamverse/ltx2_generation.py` owns LTX-2 generator configuration, video and audio continuation, and
|
||||
runtime LoRA application.
|
||||
- `apps/dreamverse/dreamverse/minimax_h3_generation.py` owns the VSA data-free FastH3 adapter, FastH3 generator and
|
||||
request configuration, and last-frame continuation through MiniMax H3 first-frame conditioning.
|
||||
- `apps/dreamverse/dreamverse/generation_contracts.py` defines the decoded media and stream-trimming result that both
|
||||
model backends return to `apps/dreamverse/dreamverse/gpu_pool.py`.
|
||||
|
||||
`apps/dreamverse/dreamverse/prompt_enhancer.py` manages:
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Benchmark the LTX-2 generation pipeline driven by the dreamverse Python SDK path.
|
||||
|
||||
Mirrors how ``apps/dreamverse/dreamverse/video_generation.py`` constructs
|
||||
Mirrors how ``apps/dreamverse/dreamverse/ltx2_generation.py`` constructs
|
||||
``GeneratorConfig`` and calls ``VideoGenerator.generate()``, then
|
||||
captures per-stage timings via the ``FASTVIDEO_STAGE_LOGGING=1`` log
|
||||
hooks (same mechanism as ``FastVideo-internal/examples/inference/basic/
|
||||
@@ -52,7 +52,8 @@ import torch # noqa: E402
|
||||
|
||||
from fastvideo import VideoGenerator # noqa: E402
|
||||
from fastvideo.api import ( # noqa: E402
|
||||
ComponentConfig, CompileConfig, EngineConfig, GeneratorConfig, OffloadConfig, PipelineSelection, QuantizationConfig,
|
||||
ComponentConfig, CompileConfig, EngineConfig, GenerationResult, GeneratorConfig, OffloadConfig, PipelineSelection,
|
||||
QuantizationConfig,
|
||||
)
|
||||
|
||||
DEFAULT_PROMPT = ("A cinematic drone shot over coastal cliffs at sunrise, golden "
|
||||
@@ -128,9 +129,9 @@ def _build_generator_config(model_path: str, enable_compile: bool, num_gpus: int
|
||||
)
|
||||
|
||||
|
||||
def _extract_stage_times(result: dict) -> OrderedDict[str, float]:
|
||||
def _extract_stage_times(result: GenerationResult) -> OrderedDict[str, float]:
|
||||
out: OrderedDict[str, float] = OrderedDict()
|
||||
info = result.get("logging_info") if isinstance(result, dict) else None
|
||||
info = result.logging_info if isinstance(result, GenerationResult) else None
|
||||
if info is None:
|
||||
return out
|
||||
stages = getattr(info, "stages", None)
|
||||
@@ -162,19 +163,25 @@ def _do_one_run(generator: VideoGenerator, prompt: str, *, height: int, width: i
|
||||
_reset_peak_gpu()
|
||||
t0 = time.perf_counter()
|
||||
try:
|
||||
result = generator.generate_video(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
save_video=False,
|
||||
height=height,
|
||||
width=width,
|
||||
num_frames=num_frames,
|
||||
fps=24,
|
||||
num_inference_steps=num_inference_steps,
|
||||
guidance_scale=1.0,
|
||||
seed=seed,
|
||||
ltx2_image_crf=0.0,
|
||||
)
|
||||
result = generator.generate({
|
||||
"prompt": prompt,
|
||||
"negative_prompt": "",
|
||||
"sampling": {
|
||||
"height": height,
|
||||
"width": width,
|
||||
"num_frames": num_frames,
|
||||
"fps": 24,
|
||||
"num_inference_steps": num_inference_steps,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": seed,
|
||||
},
|
||||
"output": {
|
||||
"save_video": False
|
||||
},
|
||||
"extensions": {
|
||||
"ltx2_image_crf": 0.0
|
||||
},
|
||||
})
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.synchronize()
|
||||
except Exception as exc:
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
_SERVER_ROOT = Path(__file__).resolve().parent
|
||||
@@ -55,16 +56,34 @@ FRONTEND_STATIC_DIR_CANDIDATES = _resolve_frontend_static_dir_candidates()
|
||||
MODEL_REGISTRY = {
|
||||
"fast-ltx2": {
|
||||
"name": "FastLTX2",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX2-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-ltx23": {
|
||||
"name": "FastLTX23",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX-2.3-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-h3": {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
},
|
||||
}
|
||||
|
||||
DEFAULT_MODEL_ID = "fast-ltx2"
|
||||
@@ -171,7 +190,7 @@ def _optional_env(*names: str) -> str | None:
|
||||
DEVTOOLS_ENABLED = _env_bool("FASTVIDEO_ENABLE_DEVTOOLS", False)
|
||||
PROMPT_SAFETY_ENABLED = _env_bool("FASTVIDEO_ENABLE_PROMPT_SAFETY", False)
|
||||
DREAMVERSE_MAX_AUTOTUNE = _env_bool("DREAMVERSE_MAX_AUTOTUNE", True)
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", 1))
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", cast(int, MODEL_CONFIG["default_sp_size"])))
|
||||
|
||||
DREAMVERSE_MODEL_PATH = (os.getenv("DREAMVERSE_MODEL_PATH", "").strip() or None)
|
||||
if DREAMVERSE_MODEL_PATH:
|
||||
@@ -213,7 +232,7 @@ def _resolve_lora_spec(spec: str) -> str | None:
|
||||
if not spec:
|
||||
return None
|
||||
if spec.lower() == "omninft":
|
||||
return MODEL_CONFIG.get("lora_repo")
|
||||
return cast(str | None, MODEL_CONFIG.get("lora_repo"))
|
||||
if spec.lower() in AVAILABLE_LORAS:
|
||||
return AVAILABLE_LORAS[spec.lower()]["repo"]
|
||||
return spec
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Shared contract between DreamVerse generation backends and GPU workers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Protocol
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Decoded media and stream-trimming metadata for one DreamVerse segment."""
|
||||
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict[str, float]
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class GenerationBackend(Protocol):
|
||||
"""Model-owned generation operations used by one GPU worker process."""
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
...
|
||||
|
||||
def shutdown(self) -> None:
|
||||
...
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
...
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
...
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
...
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
...
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Select and own one model-specific generation backend per GPU process."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dreamverse.config import MODEL_CONFIG
|
||||
from dreamverse.generation_contracts import GenerationBackend, StepResult
|
||||
|
||||
|
||||
def _create_generation_backend(backend_name: str, gpu_id: int) -> GenerationBackend:
|
||||
"""Construct the backend that owns the selected model family's behavior."""
|
||||
if backend_name == "ltx2":
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
return LTX2GenerationBackend(gpu_id)
|
||||
if backend_name == "minimax_h3":
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
return MiniMaxH3GenerationBackend(gpu_id)
|
||||
raise ValueError(f"Unsupported DreamVerse generation backend: {backend_name!r}")
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
"""Delegate GPU lifecycle and generation calls to the active model backend."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.model_config: dict = dict(MODEL_CONFIG)
|
||||
self.backend_name: str | None = None
|
||||
self.backend: GenerationBackend | None = None
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Load the requested model through its generation backend.
|
||||
|
||||
Model selection belongs here so the GPU process and streaming layers
|
||||
use one stable media contract without importing model-specific code.
|
||||
"""
|
||||
requested_model_config = dict(model_config) if model_config is not None else dict(self.model_config)
|
||||
backend_name = requested_model_config.get("generation_backend")
|
||||
if not isinstance(backend_name, str) or not backend_name:
|
||||
raise ValueError("DreamVerse model configuration requires `generation_backend`.")
|
||||
|
||||
candidate_backend = self.backend
|
||||
if candidate_backend is None or self.backend_name != backend_name:
|
||||
if candidate_backend is not None:
|
||||
candidate_backend.shutdown()
|
||||
candidate_backend = _create_generation_backend(backend_name, self.gpu_id)
|
||||
|
||||
try:
|
||||
candidate_backend.initialize(requested_model_config)
|
||||
except Exception:
|
||||
try:
|
||||
candidate_backend.shutdown()
|
||||
except Exception as shutdown_error:
|
||||
print(f"[GPU {self.gpu_id}] Backend cleanup after initialization failure: {shutdown_error}")
|
||||
self.backend = None
|
||||
self.backend_name = None
|
||||
raise
|
||||
|
||||
self.model_config = requested_model_config
|
||||
self.backend = candidate_backend
|
||||
self.backend_name = backend_name
|
||||
|
||||
def _require_backend(self) -> GenerationBackend:
|
||||
"""Return the initialized backend or fail before processing a command."""
|
||||
if self.backend is None:
|
||||
raise RuntimeError("Generation backend is not initialized.")
|
||||
return self.backend
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release model resources owned by the selected backend."""
|
||||
if self.backend is not None:
|
||||
self.backend.shutdown()
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
self._require_backend().clear_conditioning()
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one segment through the selected model backend."""
|
||||
return self._require_backend().generate_step(
|
||||
prompt,
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
return self._require_backend().warmup(prompt)
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
return self._require_backend().apply_lora_stack(stack)
|
||||
@@ -12,7 +12,7 @@ from enum import Enum
|
||||
from multiprocessing import Process, Queue
|
||||
|
||||
from dreamverse.config import (
|
||||
DEFAULT_MODEL_ID,
|
||||
ACTIVE_MODEL_ID,
|
||||
DREAMVERSE_SP_SIZE,
|
||||
MODEL_REGISTRY,
|
||||
STARTUP_WARMUP_ENABLED,
|
||||
@@ -54,7 +54,7 @@ from dreamverse.worker_ipc import (
|
||||
def _parse_requested_gpu_limit() -> int | None:
|
||||
raw_value = os.getenv("FASTVIDEO_GPU_COUNT", "").strip().lower()
|
||||
if not raw_value:
|
||||
return 1
|
||||
return DREAMVERSE_SP_SIZE
|
||||
if raw_value == "all":
|
||||
return None
|
||||
try:
|
||||
@@ -164,12 +164,12 @@ def gpu_worker_process(
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = cuda_device
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
|
||||
from dreamverse.video_generation import VideoGenerationWorker
|
||||
from dreamverse.generation_worker import VideoGenerationWorker
|
||||
|
||||
worker = VideoGenerationWorker(gpu_id)
|
||||
|
||||
def event_loop(first_cmd: Command = None):
|
||||
"""Blocking event loop for LTX2; dispatches user commands."""
|
||||
"""Block on generation commands after the model is initialized."""
|
||||
print(f"[GPU {gpu_id}] Entering event loop")
|
||||
|
||||
def handle_command(cmd: Command):
|
||||
@@ -435,7 +435,7 @@ class GPUSlot:
|
||||
self._response_reader_task: asyncio.Task | None = None
|
||||
self._active: bool = False
|
||||
self._reader_lock: asyncio.Lock | None = None
|
||||
self.current_model_id: str = DEFAULT_MODEL_ID
|
||||
self.current_model_id: str | None = ACTIVE_MODEL_ID
|
||||
self.shared_stream_buffer = None
|
||||
self.shared_stream_buffer_size = SHARED_STREAM_BUFFER_BYTES
|
||||
|
||||
@@ -690,7 +690,7 @@ class GPUSlot:
|
||||
async def join_user(self, user_id: str, model_id: str = None) -> JoinAck:
|
||||
"""Add a user to this GPU."""
|
||||
if model_id is None:
|
||||
model_id = DEFAULT_MODEL_ID
|
||||
model_id = ACTIVE_MODEL_ID
|
||||
|
||||
# Reload model if a different one is requested
|
||||
if model_id != self.current_model_id and model_id in MODEL_REGISTRY:
|
||||
@@ -705,16 +705,23 @@ class GPUSlot:
|
||||
self.connected_users.clear()
|
||||
|
||||
model_config = MODEL_REGISTRY[model_id]
|
||||
reload_response = await self._send_command(Command(CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
try:
|
||||
reload_response = await self._send_command(Command(
|
||||
CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
except Exception:
|
||||
self.current_model_id = None
|
||||
raise
|
||||
match reload_response:
|
||||
case ReloadAck():
|
||||
pass
|
||||
case WorkerError(message=msg):
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Model reload failed: {msg}")
|
||||
case _:
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Unexpected reload response: "
|
||||
f"{type(reload_response).__name__}")
|
||||
|
||||
|
||||
+56
-62
@@ -1,9 +1,9 @@
|
||||
"""LTX2 model lifecycle and continuation conditioning.
|
||||
"""LTX-2 model lifecycle and continuation conditioning.
|
||||
|
||||
Runs inside a GPU worker subprocess. Owns the model, the audio
|
||||
encoder, and the per-session continuation state carried across
|
||||
segments. Callers must set ``os.environ["CUDA_VISIBLE_DEVICES"]``
|
||||
before constructing ``VideoGenerationWorker`` — all ``fastvideo.*``
|
||||
before constructing ``LTX2GenerationBackend`` — all ``fastvideo.*``
|
||||
imports are deferred to method bodies so nothing touches CUDA at
|
||||
module import time.
|
||||
"""
|
||||
@@ -14,9 +14,7 @@ import gc
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
@@ -35,6 +33,10 @@ from dreamverse.config import (
|
||||
DREAMVERSE_LORA_STACK,
|
||||
_resolve_lora_spec,
|
||||
)
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from fastvideo.api import GenerationResult
|
||||
|
||||
# Multi-frame decoded continuation defaults from
|
||||
# examples/inference/basic/basic_ltx2_distilled_video_continuation.py.
|
||||
@@ -80,22 +82,6 @@ def _reset_lora_registry(worker) -> dict:
|
||||
return {"status": "lora_registry_reset"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Output of one generation step.
|
||||
|
||||
``head_trim_frames`` / ``head_trim_audio_frames`` are derived here
|
||||
so downstream AV streaming never needs to import conditioning
|
||||
constants.
|
||||
"""
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class ContinuationState:
|
||||
"""Per-session video + audio conditioning carried across segments."""
|
||||
|
||||
@@ -113,8 +99,8 @@ class ContinuationState:
|
||||
self.video_images = None
|
||||
self.audio_latents = None
|
||||
|
||||
def apply_video(self, request_kwargs: dict, segment_idx: int) -> None:
|
||||
"""Seed next-segment kwargs with the cached tail frames."""
|
||||
def apply_video(self, request: dict, segment_idx: int) -> None:
|
||||
"""Seed the next-segment request with the cached tail frames."""
|
||||
if segment_idx <= 1 or not self.video_images:
|
||||
return
|
||||
from PIL import Image
|
||||
@@ -128,21 +114,21 @@ class ContinuationState:
|
||||
arr = np.clip(arr, 0, 255).astype(np.uint8)
|
||||
noisy.append(Image.fromarray(arr))
|
||||
cond_images = noisy
|
||||
request_kwargs["ltx2_video_conditions"] = [(
|
||||
request["extensions"]["ltx2_video_conditions"] = [(
|
||||
cond_images,
|
||||
LTX2_VIDEO_CONDITIONING_FRAME_IDX,
|
||||
LTX2_VIDEO_CONDITIONING_STRENGTH,
|
||||
)]
|
||||
request_kwargs["ltx2_images"] = None
|
||||
request_kwargs["image_path"] = None
|
||||
request["extensions"]["ltx2_images"] = None
|
||||
request["inputs"]["image_path"] = None
|
||||
|
||||
def apply_audio(
|
||||
self,
|
||||
request_kwargs: dict,
|
||||
request: dict,
|
||||
segment_idx: int,
|
||||
audio_lps: float,
|
||||
) -> None:
|
||||
"""Seed next-segment kwargs with clean audio latents + denoise mask.
|
||||
"""Seed the next-segment request with clean audio latents + denoise mask.
|
||||
|
||||
When audio conditioning is longer than video, extend audio
|
||||
generation and shift video RoPE forward so the audio prefix
|
||||
@@ -159,9 +145,9 @@ class ContinuationState:
|
||||
audio_extra = max(0, AUDIO_CONDITIONING_NUM_FRAMES - LTX2_VIDEO_CONDITIONING_NUM_FRAMES)
|
||||
if audio_extra > 0:
|
||||
audio_num_frames = NUM_FRAMES + audio_extra
|
||||
request_kwargs["audio_num_frames"] = (audio_num_frames)
|
||||
request["extensions"]["audio_num_frames"] = (audio_num_frames)
|
||||
prefix_sec = float(audio_extra) / 24.0
|
||||
request_kwargs["video_position_offset_sec"] = prefix_sec
|
||||
request["extensions"]["video_position_offset_sec"] = prefix_sec
|
||||
|
||||
new_duration = float(NUM_FRAMES + audio_extra) / 24.0
|
||||
total_T = max(
|
||||
@@ -179,8 +165,8 @@ class ContinuationState:
|
||||
mask = torch.ones((B, 1, total_T, 1), dtype=torch.float32)
|
||||
mask[:, :, :audio_cond_T, :] = (1.0 - AUDIO_CONDITIONING_STRENGTH)
|
||||
|
||||
request_kwargs["ltx2_audio_clean_latent"] = clean
|
||||
request_kwargs["ltx2_audio_denoise_mask"] = mask
|
||||
request["extensions"]["ltx2_audio_clean_latent"] = clean
|
||||
request["extensions"]["ltx2_audio_denoise_mask"] = mask
|
||||
|
||||
def save_video(self, frames: list) -> None:
|
||||
"""Snapshot trailing N frames as PIL images for next-segment conditioning."""
|
||||
@@ -202,7 +188,7 @@ class ContinuationState:
|
||||
self.audio_latents = latents.detach().clone().cpu()
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
class LTX2GenerationBackend:
|
||||
"""Single-GPU LTX2 generator with continuation state.
|
||||
|
||||
Caller must set ``os.environ["CUDA_VISIBLE_DEVICES"]`` before
|
||||
@@ -324,7 +310,7 @@ class VideoGenerationWorker:
|
||||
),
|
||||
)
|
||||
|
||||
self.generator = VideoGenerator.from_pretrained(config=generator_config)
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
print(f"[GPU {self.gpu_id}] After model load: {self._gpu_mem()}")
|
||||
|
||||
lora_stack = DREAMVERSE_LORA_STACK or ([(DREAMVERSE_LORA_PATH,
|
||||
@@ -421,7 +407,7 @@ class VideoGenerationWorker:
|
||||
return
|
||||
|
||||
loader = ComponentLoader.for_module_type("audio_encoder", "diffusers")
|
||||
enc = loader.load(audio_vae_path, self.generator.fastvideo_args)
|
||||
enc = loader.load(audio_vae_path, self.generator.resolved_config)
|
||||
target = getattr(enc, "model", enc)
|
||||
|
||||
proc = AudioProcessor(
|
||||
@@ -478,51 +464,59 @@ class VideoGenerationWorker:
|
||||
|
||||
prompt = self._inject_style_trigger(prompt)
|
||||
|
||||
request_kwargs = dict(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
save_video=False,
|
||||
height=FRAME_HEIGHT,
|
||||
width=FRAME_WIDTH,
|
||||
num_frames=NUM_FRAMES,
|
||||
fps=24,
|
||||
num_inference_steps=NUM_INFERENCE_STEPS,
|
||||
guidance_scale=1.0,
|
||||
seed=10,
|
||||
ltx2_image_crf=0.0,
|
||||
image_path=image_path if segment_idx == 1 else None,
|
||||
return_continuation_state=False,
|
||||
)
|
||||
request = {
|
||||
"prompt": prompt,
|
||||
"negative_prompt": "",
|
||||
"inputs": {
|
||||
"image_path": image_path if segment_idx == 1 else None
|
||||
},
|
||||
"sampling": {
|
||||
"height": FRAME_HEIGHT,
|
||||
"width": FRAME_WIDTH,
|
||||
"num_frames": NUM_FRAMES,
|
||||
"fps": 24,
|
||||
"num_inference_steps": NUM_INFERENCE_STEPS,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": 10,
|
||||
},
|
||||
"output": {
|
||||
"save_video": False
|
||||
},
|
||||
"extensions": {
|
||||
"ltx2_image_crf": 0.0,
|
||||
"return_continuation_state": False,
|
||||
},
|
||||
}
|
||||
|
||||
if reset_conditioning:
|
||||
self.continuation.clear()
|
||||
|
||||
audio_lps = (DEFAULT_LTX2_AUDIO_SAMPLE_RATE / DEFAULT_LTX2_AUDIO_HOP_LENGTH / DEFAULT_LTX2_AUDIO_DOWNSAMPLE)
|
||||
|
||||
# Phase 1: seed kwargs with prior-segment conditioning.
|
||||
self.continuation.apply_video(request_kwargs, segment_idx)
|
||||
self.continuation.apply_audio(request_kwargs, segment_idx, audio_lps)
|
||||
# Phase 1: seed the request with prior-segment conditioning.
|
||||
self.continuation.apply_video(request, segment_idx)
|
||||
self.continuation.apply_audio(request, segment_idx, audio_lps)
|
||||
|
||||
# Phase 2: generate.
|
||||
t0 = time.perf_counter()
|
||||
result = self.generator.generate_video(**request_kwargs)
|
||||
result = self.generator.generate(request)
|
||||
torch.cuda.synchronize()
|
||||
timings["generation_ms"] = (time.perf_counter() - t0) * 1000
|
||||
|
||||
if not isinstance(result, dict):
|
||||
raise RuntimeError("Expected dictionary output from generate_video.")
|
||||
frames = result.get("frames")
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("Expected a single GenerationResult from generate.")
|
||||
frames = result.frames
|
||||
if not isinstance(frames, list) or len(frames) == 0:
|
||||
raise RuntimeError("Generation did not return frames.")
|
||||
audio = result.get("audio")
|
||||
audio_sample_rate = result.get("audio_sample_rate")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
# LTX2 audio decoding stage uses 24kHz output by default.
|
||||
audio_sample_rate = 24000
|
||||
print(f"[GPU {self.gpu_id}] audio_sample_rate missing from result; "
|
||||
f"defaulting to {audio_sample_rate}Hz")
|
||||
|
||||
timings["generation_time_ms"] = result.get("generation_time", 0.0) * 1000
|
||||
timings["generation_time_ms"] = (result.generation_time or 0.0) * 1000
|
||||
|
||||
# Phase 3: snapshot continuation state for the next segment.
|
||||
t_save_start = time.perf_counter()
|
||||
@@ -560,7 +554,7 @@ class VideoGenerationWorker:
|
||||
self,
|
||||
audio: object,
|
||||
audio_sample_rate: int | None,
|
||||
result: dict,
|
||||
result: "GenerationResult",
|
||||
segment_idx: int,
|
||||
) -> torch.Tensor | None:
|
||||
"""Pick which tensor to cache for next-segment audio conditioning."""
|
||||
@@ -575,7 +569,7 @@ class VideoGenerationWorker:
|
||||
f"for segment {segment_idx + 1}")
|
||||
return re_encoded
|
||||
return None
|
||||
audio_latents = result.get("ltx2_audio_latents")
|
||||
audio_latents = result.extra.get("ltx2_audio_latents")
|
||||
if audio_latents is not None:
|
||||
print(f"[GPU {self.gpu_id}] Cached audio latents "
|
||||
f"shape={tuple(audio_latents.shape)} "
|
||||
@@ -0,0 +1,294 @@
|
||||
"""FastH3 model lifecycle and first-frame continuation for DreamVerse."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import os
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from dreamverse.config import DREAMVERSE_SP_SIZE
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from PIL.Image import Image
|
||||
|
||||
|
||||
def _required_config_str(model_config: dict, field_name: str) -> str:
|
||||
"""Read one required non-empty string from a DreamVerse model profile."""
|
||||
value = model_config.get(field_name)
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError(f"FastH3 model configuration requires `{field_name}`.")
|
||||
return value.strip()
|
||||
|
||||
|
||||
class MiniMaxH3GenerationBackend:
|
||||
"""Run the VSA data-free FastH3 adapter and retain one continuation frame."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.generator: Any | None = None
|
||||
self.model_config: dict = {}
|
||||
self.continuation_image: Image | None = None
|
||||
|
||||
def _gpu_mem(self) -> str:
|
||||
allocated_gib = torch.cuda.memory_allocated() / 1024**3
|
||||
reserved_gib = torch.cuda.memory_reserved() / 1024**3
|
||||
return f"alloc={allocated_gib:.2f}GiB, reserved={reserved_gib:.2f}GiB"
|
||||
|
||||
@staticmethod
|
||||
def _configure_environment(attention_backend: str) -> None:
|
||||
"""Apply the fixed boot-time switches from the FastH3 reference recipe."""
|
||||
os.environ.update({
|
||||
"FASTVIDEO_ATTENTION_BACKEND": attention_backend,
|
||||
"FASTVIDEO_FA4": "1",
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all",
|
||||
"FASTVIDEO_VSA_SM100A": "0",
|
||||
})
|
||||
os.environ.pop("FASTVIDEO_INFERENCE_TORCH_COMPILE", None)
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Download the fixed Preview adapter and load the FastH3 generator.
|
||||
|
||||
The model profile owns the base checkpoint, adapter file, attention
|
||||
backend, and generation geometry. The backend translates that profile
|
||||
into FastVideo's typed generator configuration.
|
||||
"""
|
||||
if model_config is not None:
|
||||
self.model_config = dict(model_config)
|
||||
if not self.model_config:
|
||||
raise ValueError("FastH3 initialization requires a model configuration.")
|
||||
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
self.clear_conditioning()
|
||||
model_path = _required_config_str(self.model_config, "model_path")
|
||||
adapter_repo = _required_config_str(self.model_config, "adapter_repo")
|
||||
adapter_filename = _required_config_str(self.model_config, "adapter_filename")
|
||||
attention_backend = _required_config_str(self.model_config, "attention_backend")
|
||||
self._configure_environment(attention_backend)
|
||||
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
AttentionConfig,
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GeneratorConfig,
|
||||
MiniMaxH3Options,
|
||||
OffloadConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
|
||||
adapter_path = hf_hub_download(repo_id=adapter_repo, filename=adapter_filename)
|
||||
use_vsa = attention_backend == "VIDEO_SPARSE_ATTN_H3"
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=model_path,
|
||||
pipeline=PipelineSelection(
|
||||
components=ComponentConfig(lora_path=adapter_path, lora_strength=1.0),
|
||||
model=MiniMaxH3Options(vae_parallel_decode=True, vae_parallel_decode_strategy="gather"),
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=DREAMVERSE_SP_SIZE,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=DREAMVERSE_SP_SIZE),
|
||||
offload=OffloadConfig(
|
||||
dit=False,
|
||||
dit_layerwise=False,
|
||||
text_encoder=True,
|
||||
image_encoder=True,
|
||||
vae=True,
|
||||
pin_cpu_memory=True,
|
||||
),
|
||||
compile=CompileConfig(enabled=False, vae_enabled=True, regional=attention_backend == "FLASH_ATTN"),
|
||||
attention=AttentionConfig(
|
||||
backend=attention_backend,
|
||||
vsa_sparsity=0.9 if use_vsa else None,
|
||||
vsa_tile_size=64 if use_vsa else None,
|
||||
),
|
||||
use_fsdp_inference=False,
|
||||
),
|
||||
)
|
||||
|
||||
print(f"[GPU {self.gpu_id}] Loading FastH3 model: {model_path}")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 adapter: {adapter_repo}/{adapter_filename}")
|
||||
print(f"[GPU {self.gpu_id}] Before model load: {self._gpu_mem()}")
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
print(f"[GPU {self.gpu_id}] FastH3 loaded: {self._gpu_mem()} (warmup pending)")
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release the FastVideo generator and cached continuation image."""
|
||||
self.clear_conditioning()
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
"""Release the first-frame image retained for the next segment."""
|
||||
if self.continuation_image is not None:
|
||||
self.continuation_image.close()
|
||||
self.continuation_image = None
|
||||
|
||||
@staticmethod
|
||||
def _load_rgb_image(image_path: str) -> Image:
|
||||
"""Load an image into an independent RGB buffer with no open file handle."""
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(image_path) as image:
|
||||
return image.convert("RGB").copy()
|
||||
|
||||
def _select_conditioning_image(
|
||||
self,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> tuple[Image | None, bool]:
|
||||
"""Select the initial upload or retained last frame for one segment."""
|
||||
if reset_conditioning:
|
||||
self.clear_conditioning()
|
||||
if segment_idx > 1 and self.continuation_image is not None:
|
||||
return self.continuation_image.copy(), True
|
||||
if segment_idx > 1 and not reset_conditioning:
|
||||
raise RuntimeError(f"FastH3 segment {segment_idx} requires a retained continuation frame.")
|
||||
if segment_idx == 1 and image_path:
|
||||
return self._load_rgb_image(image_path), False
|
||||
return None, False
|
||||
|
||||
def _build_request(self, prompt: str, conditioning_image: Image | None):
|
||||
"""Build the typed FastVideo request owned by the FastH3 profile."""
|
||||
from fastvideo.api import GenerationRequest, InputConfig, OutputConfig, SamplingConfig
|
||||
|
||||
return GenerationRequest(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
inputs=InputConfig(pil_image=conditioning_image),
|
||||
sampling=SamplingConfig(
|
||||
height=int(self.model_config["height"]),
|
||||
width=int(self.model_config["width"]),
|
||||
num_frames=int(self.model_config["num_frames"]),
|
||||
fps=24,
|
||||
num_inference_steps=int(self.model_config["num_inference_steps"]),
|
||||
guidance_scale=1.0,
|
||||
batch_cfg=False,
|
||||
seed=int(self.model_config["seed"]),
|
||||
),
|
||||
output=OutputConfig(save_video=False, return_frames=True),
|
||||
)
|
||||
|
||||
def _save_continuation_frame(self, frames: list) -> None:
|
||||
"""Retain the last decoded frame as first-frame conditioning."""
|
||||
from PIL import Image
|
||||
|
||||
self.clear_conditioning()
|
||||
self.continuation_image = Image.fromarray(np.ascontiguousarray(frames[-1])).convert("RGB")
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one synchronized FastH3 segment and retain its last frame.
|
||||
|
||||
Later segments use MiniMax H3's first-frame-to-video path. The first
|
||||
conditioned frame and its matching audio duration are trimmed before
|
||||
streaming so adjacent segments do not duplicate media.
|
||||
"""
|
||||
if self.generator is None:
|
||||
raise RuntimeError("FastH3 generator is not initialized.")
|
||||
conditioning_image, uses_continuation = self._select_conditioning_image(
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
request = self._build_request(prompt, conditioning_image)
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
result = self.generator.generate(request)
|
||||
finally:
|
||||
if conditioning_image is not None:
|
||||
conditioning_image.close()
|
||||
torch.cuda.synchronize()
|
||||
generation_ms = (time.perf_counter() - started) * 1000.0
|
||||
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("FastH3 returned multiple results for one DreamVerse segment.")
|
||||
frames = result.frames
|
||||
if not isinstance(frames, list) or not frames:
|
||||
raise RuntimeError("FastH3 generation did not return decoded frames.")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
raise RuntimeError("FastH3 returned audio without an audio sample rate.")
|
||||
|
||||
save_started = time.perf_counter()
|
||||
self._save_continuation_frame(frames)
|
||||
save_conditioning_ms = (time.perf_counter() - save_started) * 1000.0
|
||||
timings = {
|
||||
"generation_ms": generation_ms,
|
||||
"generation_time_ms": float(result.generation_time or 0.0) * 1000.0,
|
||||
"save_conditioning_ms": save_conditioning_ms,
|
||||
"e2e_latency_ms": (time.perf_counter() - started) * 1000.0,
|
||||
}
|
||||
trim_frames = 1 if uses_continuation else 0
|
||||
print(f"[GPU {self.gpu_id}] FastH3 segment {segment_idx}: "
|
||||
f"{len(frames)} frames, gen={generation_ms:.0f}ms, "
|
||||
f"save_conditioning={save_conditioning_ms:.0f}ms, "
|
||||
f"e2e={timings['e2e_latency_ms']:.0f}ms")
|
||||
return StepResult(
|
||||
frames=frames,
|
||||
audio=audio,
|
||||
audio_sample_rate=audio_sample_rate,
|
||||
timings=timings,
|
||||
head_trim_frames=trim_frames,
|
||||
head_trim_audio_frames=trim_frames,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
"""Compile the FastH3 text and first-frame paths before readiness."""
|
||||
warmup_prompt = (prompt or "").strip()
|
||||
if not warmup_prompt:
|
||||
raise RuntimeError("Startup warmup prompt must be non-empty.")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup starting "
|
||||
"(synthetic segments: text-to-video, first-frame-to-video)")
|
||||
started = time.perf_counter()
|
||||
text_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
first_frame_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
total_ms = (time.perf_counter() - started) * 1000.0
|
||||
self.clear_conditioning()
|
||||
text_ms = float(text_result.timings.get("e2e_latency_ms", 0.0))
|
||||
first_frame_ms = float(first_frame_result.timings.get("e2e_latency_ms", 0.0))
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup complete: "
|
||||
f"text_to_video={text_ms:.0f}ms, "
|
||||
f"first_frame_to_video={first_frame_ms:.0f}ms, "
|
||||
f"total={total_ms:.0f}ms")
|
||||
return {
|
||||
"warmup_text_to_video_ms": text_ms,
|
||||
"warmup_first_frame_to_video_ms": first_frame_ms,
|
||||
"warmup_total_ms": total_ms,
|
||||
}
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
"""Reject runtime LoRA mutation because FastH3 uses one startup adapter."""
|
||||
del stack
|
||||
raise RuntimeError("FastH3 uses its fixed startup adapter and does not support runtime LoRA changes.")
|
||||
@@ -30,7 +30,7 @@ from dreamverse.session_init_image import cleanup_session_init_image, persist_se
|
||||
from dreamverse.worker_ipc import MediaChunk, MediaComplete, MediaInit
|
||||
|
||||
from dreamverse.config import (
|
||||
DEFAULT_MODEL_ID,
|
||||
ACTIVE_MODEL_ID,
|
||||
GENERATION_SEGMENT_CAP,
|
||||
PROMPT_AUTO_SLEEP_MS,
|
||||
PROMPT_AUTO_TIMEOUT_MS,
|
||||
@@ -264,7 +264,7 @@ class SessionController:
|
||||
timeout_task = asyncio.create_task(session_timeout())
|
||||
|
||||
# Join the engine on this GPU.
|
||||
await slot.join_user(client_id, model_id=DEFAULT_MODEL_ID)
|
||||
await slot.join_user(client_id, model_id=ACTIVE_MODEL_ID)
|
||||
|
||||
# Notify client they're connected to a GPU.
|
||||
await ws_send_json({
|
||||
|
||||
@@ -2,13 +2,14 @@ from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
import pytest
|
||||
|
||||
SERVER_DIR = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
def _load_config_module():
|
||||
def _load_config_module() -> ModuleType:
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"server_config_test_module",
|
||||
SERVER_DIR / "config.py",
|
||||
@@ -20,7 +21,7 @@ def _load_config_module():
|
||||
return module
|
||||
|
||||
|
||||
def _set_required_prompt_keys(monkeypatch):
|
||||
def _set_required_prompt_keys(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CEREBRAS_API_KEY", "cerebras-key")
|
||||
monkeypatch.setenv("GROQ_API_KEY", "groq-key")
|
||||
|
||||
@@ -150,3 +151,38 @@ def test_config_rejects_invalid_prompt_provider(monkeypatch):
|
||||
|
||||
with pytest.raises(RuntimeError, match="Invalid FASTVIDEO_PROMPT_PROVIDER"):
|
||||
_load_config_module()
|
||||
|
||||
|
||||
def test_config_registers_vsa_datafree_fasth3_profile(monkeypatch):
|
||||
"""The FastH3 registry entry owns the complete fixed Preview recipe."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.MODEL_REGISTRY["fast-h3"] == {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
def test_config_uses_fasth3_sequence_parallel_default(monkeypatch):
|
||||
"""Selecting FastH3 defaults DreamVerse to its four-GPU topology."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
monkeypatch.setenv("DREAMVERSE_MODEL_ID", "fast-h3")
|
||||
monkeypatch.delenv("DREAMVERSE_SP_SIZE", raising=False)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.ACTIVE_MODEL_ID == "fast-h3"
|
||||
assert module.MODEL_CONFIG["generation_backend"] == "minimax_h3"
|
||||
assert module.DREAMVERSE_SP_SIZE == 4
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
from dreamverse.generation_worker import _create_generation_backend
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
|
||||
def test_create_generation_backend_ltx2_module_import():
|
||||
backend = _create_generation_backend("ltx2", gpu_id=3)
|
||||
|
||||
assert isinstance(backend, LTX2GenerationBackend)
|
||||
assert backend.gpu_id == 3
|
||||
@@ -63,6 +63,14 @@ def test_get_available_gpus_defaults_to_first_visible_device(monkeypatch):
|
||||
assert gpu_pool.get_available_gpus() == [3]
|
||||
|
||||
|
||||
def test_get_available_gpus_defaults_to_active_model_sequence_parallel_size(monkeypatch):
|
||||
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1,2,3,4")
|
||||
monkeypatch.delenv("FASTVIDEO_GPU_COUNT", raising=False)
|
||||
monkeypatch.setattr(gpu_pool, "DREAMVERSE_SP_SIZE", 4)
|
||||
|
||||
assert gpu_pool.get_available_gpus() == [0, 1, 2, 3]
|
||||
|
||||
|
||||
def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
|
||||
monkeypatch.setenv("FASTVIDEO_GPU_COUNT", "zero")
|
||||
@@ -71,6 +79,23 @@ def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
gpu_pool.get_available_gpus()
|
||||
|
||||
|
||||
def test_join_user_failed_reload_marks_model_uninitialized(monkeypatch):
|
||||
"""A failed model reload forces the next join to reload a model."""
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.current_model_id = "fast-ltx2"
|
||||
|
||||
async def fake_send_command(command, timeout):
|
||||
del command, timeout
|
||||
return gpu_pool.WorkerError(user_id="__reload__", message="load failed")
|
||||
|
||||
monkeypatch.setattr(slot, "_send_command", fake_send_command)
|
||||
|
||||
with pytest.raises(RuntimeError, match="Model reload failed"):
|
||||
asyncio.run(slot.join_user("client-id", model_id="fast-h3"))
|
||||
|
||||
assert slot.current_model_id is None
|
||||
|
||||
|
||||
def test_send_command_raises_on_worker_death():
|
||||
"""A worker that consumes a command and exits without replying must
|
||||
surface as RuntimeError via sentinel detection, not after the long
|
||||
@@ -92,9 +117,9 @@ def test_send_command_raises_on_worker_death():
|
||||
ready = resp_q.get(timeout=30.0)
|
||||
assert ready == "READY"
|
||||
|
||||
async def runner():
|
||||
async def runner() -> None:
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.process = proc
|
||||
slot.process = proc # type: ignore[assignment]
|
||||
slot.command_queue = cmd_q
|
||||
slot.response_queue = resp_q
|
||||
|
||||
|
||||
@@ -13,15 +13,14 @@ FORBIDDEN_PREFIXES = (
|
||||
"fastvideo.models",
|
||||
"fastvideo.layers",
|
||||
"fastvideo.worker",
|
||||
"fastvideo.fastvideo_args",
|
||||
)
|
||||
ALLOWED_INTERNAL_IMPORTS = {
|
||||
(
|
||||
"video_generation.py",
|
||||
"ltx2_generation.py",
|
||||
"fastvideo.models.audio.ltx2_audio_processing",
|
||||
),
|
||||
(
|
||||
"video_generation.py",
|
||||
"ltx2_generation.py",
|
||||
"fastvideo.models.loader.component_loader",
|
||||
),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
import dreamverse.generation_worker as generation_worker
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
|
||||
FASTH3_MODEL_CONFIG = {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
class _RecordingGenerator:
|
||||
"""Record typed requests and return small synchronized media fixtures."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.requests: list[Any] = []
|
||||
self.conditioning_pixels: list[np.ndarray | None] = []
|
||||
|
||||
def generate(self, request):
|
||||
"""Capture the request and return two tiny video frames with audio."""
|
||||
self.requests.append(request)
|
||||
conditioning_image = request.inputs.pil_image
|
||||
self.conditioning_pixels.append(
|
||||
None if conditioning_image is None else np.asarray(conditioning_image).copy())
|
||||
frames = [
|
||||
np.full((2, 3, 3), 10, dtype=np.uint8),
|
||||
np.full((2, 3, 3), 20, dtype=np.uint8),
|
||||
]
|
||||
return SimpleNamespace(
|
||||
frames=frames,
|
||||
audio=np.zeros((2, 16), dtype=np.float32),
|
||||
audio_sample_rate=44100,
|
||||
generation_time=0.25,
|
||||
)
|
||||
|
||||
|
||||
def test_initialize_builds_vsa_datafree_fasth3_generator(monkeypatch):
|
||||
"""Initialization translates the DreamVerse profile into typed FastVideo config."""
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
captured = {}
|
||||
fake_generator = SimpleNamespace(shutdown=lambda: None)
|
||||
|
||||
def fake_from_config(config):
|
||||
captured["config"] = config
|
||||
return fake_generator
|
||||
|
||||
def fake_download(**kwargs):
|
||||
captured["download"] = kwargs
|
||||
return f"/models/{kwargs['filename']}"
|
||||
|
||||
monkeypatch.setattr("huggingface_hub.hf_hub_download", fake_download)
|
||||
monkeypatch.setattr(VideoGenerator, "from_config", fake_from_config)
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.DREAMVERSE_SP_SIZE", 4)
|
||||
monkeypatch.setenv("FASTVIDEO_ATTENTION_BACKEND", "test-attention")
|
||||
monkeypatch.setenv("FASTVIDEO_FA4", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_MINIMAX_H3_FUSIONS", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_VSA_SM100A", "1")
|
||||
monkeypatch.setenv("FASTVIDEO_INFERENCE_TORCH_COMPILE", "1")
|
||||
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
monkeypatch.setattr(backend, "_gpu_mem", lambda: "alloc=0.00GiB, reserved=0.00GiB")
|
||||
backend.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
config = captured["config"]
|
||||
assert captured["download"] == {
|
||||
"repo_id": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"filename": "vsa-datafree/adapter_model.safetensors",
|
||||
}
|
||||
assert config.model_path == "MiniMaxAI/MiniMax-H3"
|
||||
assert config.pipeline.components.lora_path.endswith("vsa-datafree/adapter_model.safetensors")
|
||||
assert config.pipeline.components.lora_strength == 1.0
|
||||
assert config.pipeline.experimental == {}
|
||||
assert config.engine.attention.backend == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert config.engine.attention.vsa_sparsity == 0.9
|
||||
assert config.engine.attention.vsa_tile_size == 64
|
||||
assert config.engine.compile.regional is False
|
||||
assert config.pipeline.model.vae_parallel_decode is True
|
||||
assert config.pipeline.model.vae_parallel_decode_strategy == "gather"
|
||||
assert config.engine.num_gpus == 4
|
||||
assert config.engine.parallelism.tp_size == 1
|
||||
assert config.engine.parallelism.sp_size == 4
|
||||
assert config.engine.offload.dit is False
|
||||
assert config.engine.offload.dit_layerwise is False
|
||||
assert config.engine.offload.text_encoder is True
|
||||
assert config.engine.offload.vae is True
|
||||
assert config.engine.compile.vae_enabled is True
|
||||
assert config.engine.use_fsdp_inference is False
|
||||
assert os.environ["FASTVIDEO_ATTENTION_BACKEND"] == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert os.environ["FASTVIDEO_FA4"] == "1"
|
||||
assert os.environ["FASTVIDEO_MINIMAX_H3_FUSIONS"] == "all"
|
||||
assert os.environ["FASTVIDEO_VSA_SM100A"] == "0"
|
||||
assert "FASTVIDEO_INFERENCE_TORCH_COMPILE" not in os.environ
|
||||
|
||||
|
||||
def test_initialize_selects_declared_generation_backend(monkeypatch):
|
||||
"""The GPU worker constructs the backend that the active model profile declares."""
|
||||
from unittest.mock import Mock
|
||||
|
||||
selected_backend = Mock()
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: selected_backend,
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend_name == "minimax_h3"
|
||||
assert worker.backend is selected_backend
|
||||
selected_backend.initialize.assert_called_once_with(FASTH3_MODEL_CONFIG)
|
||||
|
||||
|
||||
def test_initialize_failure_clears_backend_ownership(monkeypatch):
|
||||
"""A failed family change leaves the GPU worker explicitly uninitialized."""
|
||||
ltx_backend = SimpleNamespace(initialize=lambda config: None, shutdown=lambda: None)
|
||||
|
||||
def fail_initialize(config):
|
||||
del config
|
||||
raise RuntimeError("load failed")
|
||||
|
||||
fasth3_backend = SimpleNamespace(
|
||||
initialize=fail_initialize,
|
||||
shutdown=lambda: None,
|
||||
)
|
||||
backends = {
|
||||
"ltx2": ltx_backend,
|
||||
"minimax_h3": fasth3_backend,
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: backends[backend_name],
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
worker.initialize({"generation_backend": "ltx2"})
|
||||
|
||||
with pytest.raises(RuntimeError, match="load failed"):
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend is None
|
||||
assert worker.backend_name is None
|
||||
assert worker.model_config == {"generation_backend": "ltx2"}
|
||||
|
||||
|
||||
def test_generate_step_uses_last_frame_for_continuation(monkeypatch):
|
||||
"""A later segment receives the prior segment's last decoded frame."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
first_result = backend.generate_step(
|
||||
"first prompt",
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
second_result = backend.generate_step(
|
||||
"second prompt",
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
|
||||
first_request = backend.generator.requests[0]
|
||||
assert first_request.inputs.pil_image is None
|
||||
assert first_request.negative_prompt == ""
|
||||
assert first_request.sampling.height == 768
|
||||
assert first_request.sampling.width == 1344
|
||||
assert first_request.sampling.num_frames == 124
|
||||
assert first_request.sampling.num_inference_steps == 5
|
||||
assert first_request.sampling.fps == 24
|
||||
assert first_request.sampling.guidance_scale == 1.0
|
||||
assert first_request.sampling.batch_cfg is False
|
||||
assert first_request.sampling.seed == 1000
|
||||
assert first_request.output.save_video is False
|
||||
assert first_request.output.return_frames is True
|
||||
assert backend.generator.conditioning_pixels[1].tolist() == np.full((2, 3, 3), 20).tolist()
|
||||
assert first_result.head_trim_frames == 0
|
||||
assert first_result.head_trim_audio_frames == 0
|
||||
assert second_result.head_trim_frames == 1
|
||||
assert second_result.head_trim_audio_frames == 1
|
||||
assert second_result.audio_sample_rate == 44100
|
||||
|
||||
|
||||
def test_generate_step_reset_uses_text_to_video_path(monkeypatch):
|
||||
"""Resetting continuation produces an unconditioned text-to-video request."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
backend.generate_step("first prompt", 1, None, True)
|
||||
reset_result = backend.generate_step("reset prompt", 2, None, True)
|
||||
|
||||
assert backend.generator.requests[-1].inputs.pil_image is None
|
||||
assert reset_result.head_trim_frames == 0
|
||||
assert reset_result.head_trim_audio_frames == 0
|
||||
|
||||
|
||||
def test_generate_step_missing_continuation_frame(monkeypatch):
|
||||
"""A later segment fails when no reset or retained frame defines its input."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
|
||||
with pytest.raises(RuntimeError, match="requires a retained continuation frame"):
|
||||
backend.generate_step("later prompt", 2, None, False)
|
||||
|
||||
assert backend.generator.requests == []
|
||||
|
||||
|
||||
def test_warmup_exercises_text_and_first_frame_paths(monkeypatch):
|
||||
"""Warmup covers both request shapes used by a DreamVerse session."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
timings = backend.warmup("warmup prompt")
|
||||
|
||||
assert backend.generator.conditioning_pixels[0] is None
|
||||
assert backend.generator.conditioning_pixels[1] is not None
|
||||
assert backend.continuation_image is None
|
||||
assert "warmup_text_to_video_ms" in timings
|
||||
assert "warmup_first_frame_to_video_ms" in timings
|
||||
@@ -331,11 +331,11 @@ def test_rewrite_prompt_sequence_accepts_numbered_prose_output():
|
||||
]
|
||||
|
||||
|
||||
def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
def test_enhance_prompt_uses_groq_when_it_returns_first():
|
||||
enhancer = _build_staged_enhancer(
|
||||
cerebras_payload=_chat_payload_with_content('{"prompt":"Cerebras prompt"}'),
|
||||
groq_payload=_chat_payload_with_content('{"prompt":"Groq prompt"}'),
|
||||
cerebras_delay_s=0.01,
|
||||
cerebras_delay_s=0.08,
|
||||
groq_delay_s=0.01,
|
||||
)
|
||||
|
||||
@@ -346,12 +346,12 @@ def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.provider == "cerebras"
|
||||
assert result.provider == "groq"
|
||||
assert result.model == "gpt-test"
|
||||
assert result.prompt == "Cerebras prompt"
|
||||
assert result.prompt == "Groq prompt"
|
||||
assert enhancer.get_provider_success_counts() == {
|
||||
"cerebras": 1,
|
||||
"groq": 0,
|
||||
"cerebras": 0,
|
||||
"groq": 1,
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@
|
||||
<mxCell id="dispatcher" value="command dispatcher

gpu_worker_process() branches on
CommandType; asserts payload type

INIT / WARMUP / RELOAD_MODEL
USER_JOIN / USER_STEP / USER_LEAVE
SHUTDOWN" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe6cc;strokeColor=#d79b00;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
video_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
ltx2_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="stream_av" value="stream_fmp4()
av_streaming.py:121

trims overlap, pipes to ffmpeg,
publishes StreamInit / StreamChunk /
StreamComplete via callback" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#b1d8d7;strokeColor=#23445d;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
@@ -79,13 +79,13 @@
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-2" value="" style="edgeStyle=none;html=1;" parent="1" source="generator" target="Ot8BU52QTIb4EhyRSe7I-1" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
video_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
ltx2_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="ffmpeg" value="ffmpeg subprocess

libx264 / *_nvenc
fragmented mp4" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="800" y="1300" width="260" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="caches" value="ContinuationState
video_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="caches" value="ContinuationState
ltx2_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_cp" value="acquire" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#6c8ebf;endArrow=classic;fontSize=11;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="client" target="pool" edge="1">
|
||||
@@ -207,7 +207,7 @@
|
||||
</Array>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="e_dsg" value="generator.generate_video()" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#9673a6;endArrow=classic;fontSize=10;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="do_step" target="generator" edge="1">
|
||||
<mxCell id="e_dsg" value="generator.generate()" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#9673a6;endArrow=classic;fontSize=10;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="do_step" target="generator" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_dscache" value="read / write" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#d6b656;endArrow=classic;startArrow=classic;fontSize=10;exitX=0;exitY=0.8;exitDx=0;exitDy=0;entryX=1;entryY=0.2;entryDx=0;entryDy=0;" parent="1" source="do_step" target="caches" edge="1">
|
||||
@@ -250,7 +250,7 @@
|
||||
<mxPoint x="690" y="880"/>
|
||||
</Array>
|
||||
</mxCell>
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender video_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender ltx2_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="39" y="-200" width="270" height="380" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-1" value="FastVideo video_generator" style="whiteSpace=wrap;html=1;fontSize=11;fillColor=#e1d5e7;strokeColor=#9673a6;rounded=1;" parent="1" vertex="1">
|
||||
@@ -389,10 +389,10 @@
|
||||
<mxCell id="cw2" value="from fastvideo.entrypoints.video_generator import VideoGenerator
from fastvideo.models.dits.ltx2 import DEFAULT_LTX2_AUDIO_*

** Dreamverse reaches into fastvideo internals here **" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="695" width="550" height="60" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (video_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (ltx2_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="765" width="550" height="95" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (video_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (ltx2_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="870" width="550" height="55" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw5" value="enter main worker loop → waits for JOIN_USER / USER_STEP / LEAVE" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#c8e6c9;strokeColor=#388e3c;fontSize=11;fontStyle=1;fontFamily=monospace;" parent="1" vertex="1">
|
||||
@@ -534,7 +534,7 @@
|
||||
<mxPoint x="1040" y="1610" as="targetPoint"/>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (video_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (ltx2_generation.py:380)
 → generator.generate()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="955" y="1640" width="180" height="70" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="dm11" value="10b. resp_q.put(MediaInit / MediaChunk / MediaComplete / StepComplete)" style="endArrow=classic;html=1;strokeColor=#b85450;fontSize=10;labelBackgroundColor=#ffffff;" parent="1" edge="1">
|
||||
|
||||
File diff suppressed because one or more lines are too long
|
Before Width: | Height: | Size: 85 KiB After Width: | Height: | Size: 85 KiB |
@@ -64,8 +64,8 @@ generator:
|
||||
# internal: pipeline_config.dit_config.quant_config = FP4Config()
|
||||
# set in gpu_pool.py:280 (via the legacy in-place mutation). The
|
||||
# public typed surface resolves "NVFP4" to NVFP4Config() and pins
|
||||
# it on dit_config in FastVideoArgs.__post_init__. Comment this
|
||||
# block out on hosts without flashinfer / NVFP4 hardware.
|
||||
# it on dit_config when resolution materializes the PipelineConfig.
|
||||
# Comment this block out on hosts without flashinfer / NVFP4 hardware.
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ test.describe('preset prompt generation', () => {
|
||||
// on a B200 plus encode/transfer time. The "Continuation flipped
|
||||
// to Generating + Leave button rendered" pair above is the proof
|
||||
// the integration works: FE → /readyz → /curated-presets → WS
|
||||
// /ws → BE → GPU pool → VideoGenerator.generate_video, all green.
|
||||
// /ws → BE → GPU pool → VideoGenerator.generate, all green.
|
||||
const video = page.locator('video').first();
|
||||
await expect(video).toHaveCount(1);
|
||||
});
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
|
||||
import { skipWithoutMock } from './helpers';
|
||||
|
||||
test.describe('create job interactions', () => {
|
||||
skipWithoutMock();
|
||||
|
||||
for (const jobType of ['inference', 'finetuning', 'distillation']) {
|
||||
test(`${jobType} remains interactive after repeated dialog dismissals`, async ({ page }) => {
|
||||
await page.goto(`/${jobType}`);
|
||||
const trigger = page.getByRole('button', { name: 'Create Job', exact: true });
|
||||
const dialog = page.getByRole('dialog');
|
||||
|
||||
// Exercise both dismissal paths and reopen without reloading the page.
|
||||
for (const closeWithEscape of [false, true]) {
|
||||
await trigger.click();
|
||||
await page.getByRole('menuitem').first().click();
|
||||
await expect(dialog).toBeVisible();
|
||||
if (closeWithEscape) {
|
||||
await page.keyboard.press('Escape');
|
||||
} else {
|
||||
await dialog.getByRole('button', { name: 'Close', exact: true }).click();
|
||||
}
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
await expect(trigger).toBeFocused();
|
||||
}
|
||||
|
||||
await page.getByRole('link', { name: 'Datasets', exact: true }).click();
|
||||
await expect(page).toHaveURL(/\/datasets$/);
|
||||
});
|
||||
}
|
||||
|
||||
test('preserves keyboard menu dismissal and dialog focus trapping', async ({ page }) => {
|
||||
await page.goto('/inference');
|
||||
const trigger = page.getByRole('button', { name: 'Create Job', exact: true });
|
||||
await trigger.focus();
|
||||
await page.keyboard.press('Enter');
|
||||
const firstItem = page.getByRole('menuitem').first();
|
||||
await expect(firstItem).toBeFocused();
|
||||
await page.keyboard.press('Escape');
|
||||
await expect(page.getByRole('menu')).toBeHidden();
|
||||
await expect(trigger).toBeFocused();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
|
||||
await page.keyboard.press('Enter');
|
||||
await expect(firstItem).toBeFocused();
|
||||
await page.keyboard.press('Enter');
|
||||
const dialog = page.getByRole('dialog');
|
||||
await expect(dialog).toBeVisible();
|
||||
await expect(dialog.getByLabel('Name (optional)')).toBeFocused();
|
||||
|
||||
// Shift+Tab from the first field wraps to Close, then Tab wraps back.
|
||||
await page.keyboard.press('Shift+Tab');
|
||||
await expect(dialog.getByRole('button', { name: 'Close', exact: true })).toBeFocused();
|
||||
await page.keyboard.press('Tab');
|
||||
await expect(dialog.getByLabel('Name (optional)')).toBeFocused();
|
||||
await page.keyboard.press('Escape');
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(trigger).toBeFocused();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
});
|
||||
});
|
||||
@@ -1,6 +1,6 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
|
||||
import { skipWithoutMock } from './helpers';
|
||||
import { API_BASE, skipWithoutMock } from './helpers';
|
||||
|
||||
/**
|
||||
* Create-job flow: open the Create Job modal on /inference, fill the prompt
|
||||
@@ -10,7 +10,8 @@ import { skipWithoutMock } from './helpers';
|
||||
test.describe('create inference job', () => {
|
||||
skipWithoutMock();
|
||||
|
||||
test('creates a T2V job and shows it in the queue', async ({ page }) => {
|
||||
test('creates a T2V job and starts it without refreshing', async ({ page, request }) => {
|
||||
await request.put(`${API_BASE}/settings`, { data: { autoStartJob: false } });
|
||||
await page.goto('/inference');
|
||||
|
||||
// The trigger opens a real menu on click, so this path works for touch,
|
||||
@@ -38,5 +39,20 @@ test.describe('create inference job', () => {
|
||||
// Modal closes and the queue refreshes with the newly created job.
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(page.getByText(prompt)).toBeVisible();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
|
||||
const card = page.getByRole('article').filter({ hasText: prompt });
|
||||
await expect(card.getByText('pending', { exact: true })).toBeVisible();
|
||||
const started = page.waitForResponse((response) =>
|
||||
response.url().startsWith(`${API_BASE}/jobs/`) &&
|
||||
response.url().endsWith('/start') &&
|
||||
response.request().method() === 'POST',
|
||||
);
|
||||
await card.getByRole('button', { name: 'Start', exact: true }).click();
|
||||
expect((await started).ok()).toBe(true);
|
||||
await expect(card.getByText('running', { exact: true })).toBeVisible();
|
||||
|
||||
await page.getByRole('link', { name: 'Datasets', exact: true }).click();
|
||||
await expect(page).toHaveURL(/\/datasets$/);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -862,23 +862,40 @@ class JobRunner:
|
||||
sp_size,
|
||||
)
|
||||
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
model_id,
|
||||
workload_type=workload_type,
|
||||
num_gpus=num_gpus,
|
||||
dit_layerwise_offload=dit_layerwise_offload,
|
||||
**({
|
||||
"override_pipeline_cls_name": override_pipeline_cls_name
|
||||
} if override_pipeline_cls_name else {}),
|
||||
dit_cpu_offload=dit_cpu_offload,
|
||||
text_encoder_cpu_offload=text_encoder_cpu_offload,
|
||||
vae_cpu_offload=vae_cpu_offload,
|
||||
image_encoder_cpu_offload=image_encoder_cpu_offload,
|
||||
use_fsdp_inference=use_fsdp_inference,
|
||||
enable_torch_compile=enable_torch_compile,
|
||||
VSA_sparsity=vsa_sparsity,
|
||||
tp_size=tp_size,
|
||||
sp_size=sp_size,
|
||||
gen = VideoGenerator.from_config(
|
||||
{
|
||||
"model_path": model_id,
|
||||
"engine": {
|
||||
"num_gpus": num_gpus,
|
||||
"parallelism": {
|
||||
"tp_size": tp_size,
|
||||
"sp_size": sp_size,
|
||||
},
|
||||
"offload": {
|
||||
"dit": dit_cpu_offload,
|
||||
"dit_layerwise": dit_layerwise_offload,
|
||||
"text_encoder": text_encoder_cpu_offload,
|
||||
"image_encoder": image_encoder_cpu_offload,
|
||||
"vae": vae_cpu_offload,
|
||||
},
|
||||
"compile": {
|
||||
"enabled": enable_torch_compile
|
||||
},
|
||||
"attention": {
|
||||
"vsa_sparsity": vsa_sparsity
|
||||
},
|
||||
"use_fsdp_inference": use_fsdp_inference,
|
||||
},
|
||||
"pipeline": {
|
||||
"workload_type":
|
||||
workload_type,
|
||||
**({
|
||||
"components": {
|
||||
"override_pipeline_cls_name": override_pipeline_cls_name
|
||||
}
|
||||
} if override_pipeline_cls_name else {}),
|
||||
},
|
||||
},
|
||||
log_queue=log_queue,
|
||||
)
|
||||
|
||||
@@ -1103,30 +1120,33 @@ class JobRunner:
|
||||
# Without a name FastVideo derives the filename from the prompt.
|
||||
safe_name = re.sub(r'[\\/:*?"<>|]+', "", job.name).strip().strip(".")
|
||||
output_target = (os.path.join(job_output_dir, f"{safe_name[:80]}.mp4") if safe_name else job_output_dir)
|
||||
gen_kwargs: dict[str, Any] = {
|
||||
request: dict[str, Any] = {
|
||||
"prompt": job.prompt,
|
||||
"output_path": output_target,
|
||||
"save_video": True,
|
||||
"num_inference_steps": job.num_inference_steps,
|
||||
"num_frames": job.num_frames,
|
||||
"height": job.height,
|
||||
"width": job.width,
|
||||
"guidance_scale": job.guidance_scale,
|
||||
"guidance_rescale": job.guidance_rescale,
|
||||
"fps": job.fps,
|
||||
"seed": job.seed,
|
||||
"negative_prompt": job.negative_prompt or "",
|
||||
"log_queue": log_queue,
|
||||
"sampling": {
|
||||
"num_inference_steps": job.num_inference_steps,
|
||||
"num_frames": job.num_frames,
|
||||
"height": job.height,
|
||||
"width": job.width,
|
||||
"guidance_scale": job.guidance_scale,
|
||||
"guidance_rescale": job.guidance_rescale,
|
||||
"fps": job.fps,
|
||||
"seed": job.seed,
|
||||
},
|
||||
"output": {
|
||||
"output_path": output_target,
|
||||
"save_video": True,
|
||||
},
|
||||
}
|
||||
if job.image_path:
|
||||
gen_kwargs["image_path"] = job.image_path
|
||||
request.setdefault("inputs", {})["image_path"] = job.image_path
|
||||
if job.references:
|
||||
gen_kwargs["references"] = _build_h3_references(job.references)
|
||||
request.setdefault("inputs", {})["references"] = _build_h3_references(job.references)
|
||||
if job.last_image_path:
|
||||
# _prepare_fl2va requires a PIL image, not a path.
|
||||
from PIL import Image as _PILImage
|
||||
gen_kwargs["last_image"] = _PILImage.open(job.last_image_path)
|
||||
generator.generate_video(**gen_kwargs)
|
||||
request.setdefault("inputs", {})["last_image"] = _PILImage.open(job.last_image_path)
|
||||
generator.generate(request, log_queue=log_queue)
|
||||
|
||||
buf.phase = "saving"
|
||||
logger.info("Generation completed, searching for output file...")
|
||||
|
||||
Generated
+868
-722
File diff suppressed because it is too large
Load Diff
@@ -17,20 +17,11 @@
|
||||
"start:all": "concurrently --kill-others-on-fail \"npm:start:api\" \"npm:start:web\""
|
||||
},
|
||||
"dependencies": {
|
||||
"@radix-ui/react-dialog": "^1.1.0",
|
||||
"@radix-ui/react-dropdown-menu": "^2.1.24",
|
||||
"@radix-ui/react-label": "^2.1.8",
|
||||
"@radix-ui/react-scroll-area": "^1.2.10",
|
||||
"@radix-ui/react-select": "^2.2.6",
|
||||
"@radix-ui/react-separator": "^1.1.8",
|
||||
"@radix-ui/react-slider": "^1.2.0",
|
||||
"@radix-ui/react-slot": "^1.2.4",
|
||||
"@radix-ui/react-switch": "^1.1.0",
|
||||
"@radix-ui/react-tabs": "^1.1.0",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"lucide-react": "^0.577.0",
|
||||
"next": "15.5.18",
|
||||
"radix-ui": "^1.6.7",
|
||||
"react": "^19.1.0",
|
||||
"react-dom": "^19.1.0",
|
||||
"sonner": "^2.0.7",
|
||||
|
||||
@@ -1,24 +1,26 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { render, screen, waitFor, within } from '@testing-library/react';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import CreateJobButton from './CreateJobButton';
|
||||
import { getDatasets, getModels } from '@/lib/api';
|
||||
|
||||
vi.mock('./CreateJobModal', () => ({
|
||||
default: ({
|
||||
isOpen,
|
||||
workloadType,
|
||||
}: {
|
||||
isOpen: boolean;
|
||||
workloadType: string;
|
||||
}) =>
|
||||
isOpen ? (
|
||||
<div role="dialog" data-workload-type={workloadType}>
|
||||
Create job form
|
||||
</div>
|
||||
) : null,
|
||||
vi.mock('@/lib/api', () => ({
|
||||
createJob: vi.fn(),
|
||||
getModels: vi.fn(),
|
||||
getDatasets: vi.fn(),
|
||||
uploadImage: vi.fn(),
|
||||
getSettings: vi.fn(),
|
||||
updateSettings: vi.fn(),
|
||||
}));
|
||||
|
||||
beforeEach(() => {
|
||||
vi.mocked(getModels).mockResolvedValue([
|
||||
{ id: 'wan/t2v-1.3b', label: 'Wan T2V' },
|
||||
]);
|
||||
vi.mocked(getDatasets).mockResolvedValue([]);
|
||||
});
|
||||
|
||||
describe('CreateJobButton', () => {
|
||||
it('opens the workload menu on click and selects an item', async () => {
|
||||
const user = userEvent.setup();
|
||||
@@ -27,10 +29,9 @@ describe('CreateJobButton', () => {
|
||||
await user.click(screen.getByRole('button', { name: 'Create Job' }));
|
||||
await user.click(screen.getByRole('menuitem', { name: /I2V/i }));
|
||||
|
||||
expect(screen.getByRole('dialog')).toHaveAttribute(
|
||||
'data-workload-type',
|
||||
'i2v',
|
||||
);
|
||||
expect(
|
||||
screen.getByRole('dialog', { name: 'New Inference Job (I2V)' }),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('opens and operates the workload menu from the keyboard', async () => {
|
||||
@@ -45,9 +46,42 @@ describe('CreateJobButton', () => {
|
||||
expect(firstItem).toHaveFocus();
|
||||
await user.keyboard('{Enter}');
|
||||
|
||||
expect(screen.getByRole('dialog')).toHaveAttribute(
|
||||
'data-workload-type',
|
||||
't2v',
|
||||
expect(
|
||||
screen.getByRole('dialog', { name: 'New Inference Job (T2V)' }),
|
||||
).toBeInTheDocument();
|
||||
await user.keyboard('{Escape}');
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument(),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(document.body.style.pointerEvents).not.toBe('none'),
|
||||
);
|
||||
expect(trigger).toHaveFocus();
|
||||
});
|
||||
|
||||
it.each(['inference', 'finetuning', 'distillation'] as const)(
|
||||
'restores page interaction after closing the real %s dialog',
|
||||
async (jobType) => {
|
||||
const user = userEvent.setup();
|
||||
render(<CreateJobButton jobType={jobType} />);
|
||||
const trigger = screen.getByRole('button', { name: 'Create Job' });
|
||||
|
||||
// Keep the real Dialog mounted: mocking it hides conflicting Radix layers.
|
||||
for (let attempt = 0; attempt < 2; attempt++) {
|
||||
await user.click(trigger);
|
||||
await user.click(screen.getAllByRole('menuitem')[0]);
|
||||
const dialog = screen.getByRole('dialog');
|
||||
await user.click(
|
||||
within(dialog).getByRole('button', { name: 'Close' }),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument(),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(document.body.style.pointerEvents).not.toBe('none'),
|
||||
);
|
||||
expect(trigger).toHaveFocus();
|
||||
}
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
import * as React from 'react';
|
||||
import { ChevronDown } from 'lucide-react';
|
||||
import * as DropdownMenu from '@radix-ui/react-dropdown-menu';
|
||||
import { DropdownMenu } from 'radix-ui';
|
||||
|
||||
import CreateJobModal from '@/components/jobs/CreateJobModal';
|
||||
import { Button } from '@/components/ui/button';
|
||||
@@ -16,6 +16,7 @@ interface CreateJobButtonProps {
|
||||
|
||||
export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
const options = WORKLOAD_OPTIONS[jobType] ?? [];
|
||||
const triggerRef = React.useRef<HTMLButtonElement>(null);
|
||||
|
||||
const [modalOpen, setModalOpen] = React.useState(false);
|
||||
const [workloadType, setWorkloadType] = React.useState(
|
||||
@@ -36,7 +37,7 @@ export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
<>
|
||||
<DropdownMenu.Root>
|
||||
<DropdownMenu.Trigger asChild>
|
||||
<Button type="button" className="gap-1.5">
|
||||
<Button ref={triggerRef} type="button" className="gap-1.5">
|
||||
Create Job
|
||||
<ChevronDown className="size-3.5 opacity-85" aria-hidden />
|
||||
</Button>
|
||||
@@ -66,6 +67,11 @@ export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
<CreateJobModal
|
||||
isOpen={modalOpen}
|
||||
onClose={() => setModalOpen(false)}
|
||||
onCloseAutoFocus={(event) => {
|
||||
// This dialog opens from a menu item, so it has no DialogTrigger.
|
||||
event.preventDefault();
|
||||
triggerRef.current?.focus();
|
||||
}}
|
||||
onSuccess={handleSuccess}
|
||||
jobType={jobType}
|
||||
workloadType={workloadType}
|
||||
|
||||
@@ -55,6 +55,9 @@ import { jobToFormFields, type JobLike } from '@/lib/jobToFields';
|
||||
export interface CreateJobModalProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
onCloseAutoFocus?: React.ComponentProps<
|
||||
typeof DialogContent
|
||||
>['onCloseAutoFocus'];
|
||||
onSuccess: () => void;
|
||||
jobType: JobType;
|
||||
workloadType: string;
|
||||
@@ -67,6 +70,7 @@ export interface CreateJobModalProps {
|
||||
export default function CreateJobModal({
|
||||
isOpen,
|
||||
onClose,
|
||||
onCloseAutoFocus,
|
||||
onSuccess,
|
||||
jobType,
|
||||
workloadType,
|
||||
@@ -127,8 +131,9 @@ export default function CreateJobModal({
|
||||
const editingJobId = editingJob?.id ?? null;
|
||||
const editingJobModelId = editingJob?.model_id ?? null;
|
||||
|
||||
// Layerwise offload and FSDP compete for the DiT weights and FastVideoArgs
|
||||
// silently picks a winner (fastvideo_args.py:859); resolve it visibly here.
|
||||
// Layerwise offload and FSDP compete for the DiT weights and the device offload
|
||||
// policy (resolve_device_offload_conflicts in fastvideo/api/device_policy.py)
|
||||
// silently picks a winner; resolve it visibly here.
|
||||
// dit_cpu_offload is deliberately not interlocked -- it is a modifier, not a
|
||||
// competing strategy.
|
||||
const handleDitLayerwiseOffloadChange = React.useCallback((next: boolean) => {
|
||||
@@ -644,6 +649,7 @@ export default function CreateJobModal({
|
||||
>
|
||||
<DialogContent
|
||||
className="max-h-[90vh] w-[90vw] max-w-[850px] overflow-y-auto"
|
||||
onCloseAutoFocus={onCloseAutoFocus}
|
||||
onEscapeKeyDown={(e) => {
|
||||
if (isSubmitting) e.preventDefault();
|
||||
}}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import * as React from 'react';
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { Button } from './button';
|
||||
import { Input } from './input';
|
||||
@@ -8,6 +10,24 @@ import { Slider } from './slider';
|
||||
import { Switch } from './switch';
|
||||
|
||||
describe('shared control accessibility', () => {
|
||||
it('forwards refs and click handlers to the asChild button', async () => {
|
||||
const user = userEvent.setup();
|
||||
const ref = React.createRef<HTMLButtonElement>();
|
||||
const onClick = vi.fn();
|
||||
render(
|
||||
<Button asChild ref={ref} onClick={onClick}>
|
||||
<button type="button">Slotted action</button>
|
||||
</Button>,
|
||||
);
|
||||
|
||||
const button = screen.getByRole('button', { name: 'Slotted action' });
|
||||
expect(screen.getAllByRole('button')).toHaveLength(1);
|
||||
expect(ref.current).toBe(button);
|
||||
await user.click(button);
|
||||
expect(onClick).toHaveBeenCalledTimes(1);
|
||||
expect(button).toHaveFocus();
|
||||
});
|
||||
|
||||
it('keeps button, input, and select targets at least 44px tall', () => {
|
||||
render(
|
||||
<>
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"use client";
|
||||
|
||||
import * as React from "react";
|
||||
import { Slot } from "@radix-ui/react-slot";
|
||||
import { Slot } from "radix-ui";
|
||||
import { cva, type VariantProps } from "class-variance-authority";
|
||||
|
||||
import { cn } from "@/lib/utils";
|
||||
@@ -37,7 +37,7 @@ export interface ButtonProps extends React.ButtonHTMLAttributes<HTMLButtonElemen
|
||||
}
|
||||
|
||||
const Button = React.forwardRef<HTMLButtonElement, ButtonProps>(({ className, variant, size, asChild = false, ...props }, ref) => {
|
||||
const Comp = asChild ? Slot : "button";
|
||||
const Comp = asChild ? Slot.Root : "button";
|
||||
return <Comp className={cn(buttonVariants({ variant, size, className }))} ref={ref} {...props} />;
|
||||
});
|
||||
Button.displayName = "Button";
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as DialogPrimitive from '@radix-ui/react-dialog';
|
||||
import { Dialog as DialogPrimitive } from 'radix-ui';
|
||||
import { X } from 'lucide-react';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as LabelPrimitive from '@radix-ui/react-label';
|
||||
import { Label as LabelPrimitive } from 'radix-ui';
|
||||
import { cva, type VariantProps } from 'class-variance-authority';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as ScrollAreaPrimitive from '@radix-ui/react-scroll-area';
|
||||
import { ScrollArea as ScrollAreaPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SelectPrimitive from '@radix-ui/react-select';
|
||||
import { Select as SelectPrimitive } from 'radix-ui';
|
||||
import { Check, ChevronDown, ChevronUp } from 'lucide-react';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SeparatorPrimitive from '@radix-ui/react-separator';
|
||||
import { Separator as SeparatorPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SliderPrimitive from '@radix-ui/react-slider';
|
||||
import { Slider as SliderPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SwitchPrimitives from '@radix-ui/react-switch';
|
||||
import { Switch as SwitchPrimitives } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as TabsPrimitive from '@radix-ui/react-tabs';
|
||||
import { Tabs as TabsPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -15,6 +15,9 @@ from fastvideo import VideoGenerator as FastVideoGenerator
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))))
|
||||
|
||||
# InferenceArgs keys that are SamplingConfig fields of a GenerationRequest.
|
||||
_SAMPLING_INFERENCE_ARGS = ("height", "width", "num_frames", "num_inference_steps", "guidance_scale", "seed", "fps")
|
||||
|
||||
|
||||
# Custom exception for interruption
|
||||
class GenerationInterruptedException(Exception):
|
||||
@@ -154,7 +157,17 @@ class VideoGenerator:
|
||||
"""Thread function to run the generation"""
|
||||
try:
|
||||
if self.generator is not None:
|
||||
self.generator.generate_video(prompt=prompt, output_path=output_path, **inference_args)
|
||||
# Place each InferenceArgs value in the GenerationRequest section that owns it.
|
||||
request: dict[str, Any] = {"prompt": prompt, "output": {"output_path": output_path}}
|
||||
for key, value in inference_args.items():
|
||||
if key == "image_path":
|
||||
section = "inputs"
|
||||
elif key in _SAMPLING_INFERENCE_ARGS:
|
||||
section = "sampling"
|
||||
else:
|
||||
section = "extensions"
|
||||
request.setdefault(section, {})[key] = value
|
||||
self.generator.generate(request)
|
||||
self._generation_result = os.path.join(output_path, f"{prompt[:100]}.mp4")
|
||||
else:
|
||||
raise RuntimeError("Generator is not initialized")
|
||||
@@ -253,9 +266,24 @@ class VideoGenerator:
|
||||
if self.generator is None:
|
||||
print('generation_args', generation_args)
|
||||
print('pipeline_config', pipeline_config)
|
||||
self.generator = FastVideoGenerator.from_pretrained(model_path=model_path,
|
||||
**generation_args,
|
||||
pipeline_config=pipeline_config)
|
||||
# Place each generation argument at its GeneratorConfig engine path.
|
||||
engine_config: dict[str, Any] = {}
|
||||
if "num_gpus" in generation_args:
|
||||
engine_config["num_gpus"] = generation_args["num_gpus"]
|
||||
for parallelism_key in ("tp_size", "sp_size"):
|
||||
if parallelism_key in generation_args:
|
||||
engine_config.setdefault("parallelism", {})[parallelism_key] = generation_args[parallelism_key]
|
||||
if "dit_cpu_offload" in generation_args:
|
||||
engine_config["offload"] = {"dit": generation_args["dit_cpu_offload"]}
|
||||
self.generator = FastVideoGenerator.from_config({
|
||||
"model_path": model_path,
|
||||
"engine": engine_config,
|
||||
"pipeline": {
|
||||
"experimental": {
|
||||
"pipeline_config": pipeline_config
|
||||
}
|
||||
},
|
||||
})
|
||||
|
||||
print('inference_args', inference_args)
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"version": 6,
|
||||
"version": 11,
|
||||
"recipes": [
|
||||
{
|
||||
"id": "fastwan21-t2v",
|
||||
@@ -437,18 +437,21 @@
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_t2v/minimax_h3_t2v.mp4",
|
||||
"modes": ["T2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["The checked-in example defaults to four-way sequence parallelism. It does not record a GPU model or memory requirement."]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-cuda",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 Preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 Preview on CUDA",
|
||||
"summary": "Run the DMD2-distilled FastH3 Preview with four DiT forwards, trained H3 sparse attention, compiled decode, and synchronized audio.",
|
||||
"label": "FastH3 V1 on CUDA",
|
||||
"summary": "Run FastH3 V1 with four DiT forwards, trained H3 sparse attention, compiled decode, and synchronized audio.",
|
||||
"model": "FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2",
|
||||
"source": "examples/inference/basic/basic_fasth3.py",
|
||||
"serving": {
|
||||
@@ -467,6 +470,10 @@
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3/",
|
||||
"modes": ["T2VA", "4-step FastH3"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": ["The default all profile is the measured GB200 performance route and can change floating-point operation order. Use --profile strict --no-inference-torch-compile for the eager strict route."]
|
||||
},
|
||||
{
|
||||
@@ -477,13 +484,13 @@
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\""
|
||||
},
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 Preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 Preview on MLX",
|
||||
"summary": "Run FastH3 Preview on Apple Silicon with a locally converted INT6 DiT, streamed Qwen3-VL conditioning, and native MLX video and audio VAEs.",
|
||||
"label": "FastH3 V1 on MLX",
|
||||
"summary": "Run FastH3 V1 on Apple Silicon with a locally converted INT6 DiT, streamed Qwen3-VL conditioning, and native MLX video and audio VAEs.",
|
||||
"model": "FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2",
|
||||
"source": "examples/inference/basic/mlx_fasth3.py",
|
||||
"command": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\"\npython examples/inference/basic/mlx_fasth3.py --model-root ./FastH3-Preview-v0.2 --mlx-checkpoint ./FastH3-MLX/int6 --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --height 480 --width 832 --num-frames 124 --seed 2026 --output-path ./outputs/fasth3_int6.mp4",
|
||||
@@ -499,10 +506,155 @@
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 with H.264 video and stereo AAC audio at outputs/fasth3_int6.mp4",
|
||||
"modes": ["T2VA", "temporal --fast", "spatial --fast-spatial", "opt-in VSA"],
|
||||
"knobs": [
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"The MLX path supports T2VA, optional temporal --fast, optional spatial --fast-spatial, and opt-in VSA on --include-vsa checkpoints. FL2VA, Ref2VA, and two-pass refinement are not wired."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-spark",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on one DGX Spark",
|
||||
"summary": "Run FastH3 V1 on one GB10 with Triton VSA, FA4 off, and lazy module load. Height, width, frames, and steps in the YAML are examples.",
|
||||
"model": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree",
|
||||
"source": "examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_spark.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3"
|
||||
},
|
||||
"command": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"device": "spark",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/fasth3_spark/",
|
||||
"modes": ["T2VA", "1-Spark"],
|
||||
"limitations": [
|
||||
"Install from the DGX Spark guide, not the generic CUDA extra. GB10 has no FA4 / sm_100a VSA kernel; keep FASTVIDEO_FA4=0 and FASTVIDEO_VSA_SM100A=0.",
|
||||
"Legal num_frames values are 17n+5, capped at 345 (15 s). Native 16:9 sizes include 832x480 and 1344x768.",
|
||||
"Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Do not set engine.offload.lazy_module_load to false on this box.",
|
||||
"A 345-frame request on one Spark can OOM. Prefer 124 or 243 frames, TAEH3 decode, or two Sparks over QSFP."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-spark-pair",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on two DGX Sparks",
|
||||
"summary": "Run one FastH3 clip across two GB10s with Ray sequence parallel over QSFP RoCE. Sequential load and lazy module load stay on because SP replicates the DiT on each node.",
|
||||
"model": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree",
|
||||
"source": "examples/inference/basic/basic_fasth3_spark_pair.yaml",
|
||||
"command": "source examples/inference/optimizations/spark_pair_env.sh && FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_VAE_PARALLEL_DECODE=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"device": "spark",
|
||||
"gpu_count": 2,
|
||||
"accelerator": "NVIDIA GB10 (DGX Spark pair)",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1803"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 under outputs/fasth3_spark_pair/",
|
||||
"modes": ["T2VA", "2-Spark SP"],
|
||||
"limitations": [
|
||||
"Requires a two-node Ray cluster on the QSFP interconnect. There is no cookbook server for this path; use Python / generate.",
|
||||
"Height, width, frames, and steps in the YAML are examples. Edit them or pass CLI flags. See docs/getting_started/installation/spark_pair.md."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-cuda",
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
"group_task": "8-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V2 on CUDA",
|
||||
"summary": "Run FastH3 V2, the eight-forward checkpoint (video/audio shifts 10/3, VSA 0.8, 64-token tiles) with the trained DMD ladder loaded from the checkpoint's fastvideo_inference.json.",
|
||||
"model": "FastVideo/FastVideo-FastH3-8-Step-V2",
|
||||
"source": "examples/inference/basic/basic_fasth3_8step.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_8step.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3_8step.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile strict --no-inference-torch-compile --no-compile-vae",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 4,
|
||||
"accelerator": "NVIDIA GB200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1852"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3_8step/",
|
||||
"modes": ["T2VA", "8-step FastH3"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"Nine sigma-grid points (eight transformer forwards) are fixed by the checkpoint's trained ladder; the example rejects any other --steps.",
|
||||
"Validated with the eager strict route (--profile strict --no-inference-torch-compile --no-compile-vae). The compiled all profile has not been measured for this checkpoint.",
|
||||
"T2AV only; no FL2VA/Ref2VA distillation and no matching LoRA. V2 uses eight forwards rather than V1's four."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-mlx",
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3_8step.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa"
|
||||
},
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
"group_task": "8-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V2 on MLX",
|
||||
"summary": "Run FastH3 V2 on Apple Silicon with a locally converted INT8 DiT, the checkpoint's DMD contract ladder, and trained VSA (sparsity 0.8, 64-token tiles).",
|
||||
"model": "FastVideo/FastVideo-FastH3-8-Step-V2",
|
||||
"source": "examples/inference/basic/mlx_fasth3_8step.py",
|
||||
"command": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa\npython examples/inference/basic/mlx_fasth3_8step.py --model-root ./FastH3-8-Step-V2 --mlx-checkpoint ./FastH3-8-Step-V2-MLX/int8 --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --height 480 --width 832 --num-frames 124 --seed 2026 --output-path ./outputs/fasth3_8step_int8.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"peak_memory": "27.39 GiB peak MLX memory during denoising",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1863"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 with H.264 video and stereo AAC audio at outputs/fasth3_8step_int8.mp4",
|
||||
"modes": ["T2VA", "8-step FastH3", "trained VSA"],
|
||||
"knobs": [
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"Convert with --include-vsa. mlx_fasth3_8step.py turns VSA on (sparsity 0.8, tile 64). A dense export fails at configure_vsa.",
|
||||
"--steps 8 or 9 both run the eight trained forwards. Reuse the preview VAE, audio VAE, text encoder, and tokenizer if those directories already exist.",
|
||||
"The MLX path supports T2VA only. FL2VA, Ref2VA, and two-pass refinement are not wired."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "minimax-h3-fl2va",
|
||||
"family": "minimax_h3",
|
||||
@@ -518,6 +670,9 @@
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_fl2va/minimax_h3_fl2va.mp4",
|
||||
"modes": ["FL2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["Pass --last-image to constrain the final frame. The checked-in source defaults to four GPUs."]
|
||||
},
|
||||
{
|
||||
@@ -535,6 +690,9 @@
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_ref2va/minimax_h3_ref2va.mp4",
|
||||
"modes": ["Ref2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["Pass --reference-audio for an additional audio reference. The checked-in source defaults to four GPUs."]
|
||||
},
|
||||
{
|
||||
@@ -558,6 +716,10 @@
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3_lora_preview/",
|
||||
"modes": ["T2VA", "FastH3 LoRA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": ["Supply a compatible FastH3 adapter. The script infers dense or VSA attention from the adapter payload unless you override it."]
|
||||
},
|
||||
{
|
||||
|
||||
+242
-30
@@ -89,11 +89,38 @@
|
||||
|
||||
const groupIdFor = (recipe) => recipe.group || recipe.id;
|
||||
|
||||
const gpuCountLabel = (hardware) => {
|
||||
const gpuCountLabel = (hardware, runtimeId) => {
|
||||
const count = hardware?.gpu_count;
|
||||
if (count == null) return "";
|
||||
if (runtimeId === "spark") return `${count} Spark${count === 1 ? "" : "s"}`;
|
||||
return `${count} GPU${count === 1 ? "" : "s"}`;
|
||||
};
|
||||
|
||||
const knobsFor = (recipe) => recipe.knobs || [];
|
||||
|
||||
const knobOptions = (knob) =>
|
||||
knob.options.map((option) =>
|
||||
option !== null && typeof option === "object" ? option : { value: option, label: String(option) },
|
||||
);
|
||||
|
||||
const knobDefaultLabel = (knob) => {
|
||||
const match = knobOptions(knob).find((option) => option.value === knob.default);
|
||||
return match ? match.label : String(knob.default);
|
||||
};
|
||||
|
||||
// Knob flags are always shown explicitly in the displayed command, even at
|
||||
// their default value, so the command stays copy-pasteable and precise
|
||||
// about what it runs -- not just "trust the script's own default".
|
||||
const appendKnobFlags = (commandText, knobs, knobValues) => {
|
||||
const flags = knobs
|
||||
.filter((knob) => knobValues[knob.key] !== undefined)
|
||||
.map((knob) => `${knob.flag} ${knobValues[knob.key]}`);
|
||||
if (!flags.length) return commandText;
|
||||
const lines = commandText.split("\n");
|
||||
lines[lines.length - 1] = `${lines[lines.length - 1]} ${flags.join(" ")}`;
|
||||
return lines.join("\n");
|
||||
};
|
||||
|
||||
const runtimeFor = (recipe) => {
|
||||
const platform = recipe.hardware?.platform || "cuda";
|
||||
if (platform === "mlx") {
|
||||
@@ -109,17 +136,24 @@
|
||||
if (platform === "mps") {
|
||||
return {
|
||||
id: "mps",
|
||||
label: "Apple Silicon · MPS",
|
||||
label: "Apple Silicon · PyTorch MPS",
|
||||
hint: recipe.hardware?.minimum_memory || recipe.hardware?.system_memory || "Memory not recorded",
|
||||
};
|
||||
}
|
||||
const hardware = recipe.hardware || {};
|
||||
if (hardware.device === "spark") {
|
||||
return {
|
||||
id: "spark",
|
||||
label: "NVIDIA DGX Spark",
|
||||
hint: "GB10 · 128 GB unified memory",
|
||||
};
|
||||
}
|
||||
return {
|
||||
id: "cuda",
|
||||
label: "NVIDIA CUDA",
|
||||
hint: hardware.accelerator
|
||||
? `${hardware.accelerator} · ${gpuCountLabel(hardware)}`
|
||||
: `${gpuCountLabel(hardware)} configured · GPU model not recorded`,
|
||||
? `${hardware.accelerator} · ${gpuCountLabel(hardware, "cuda")}`
|
||||
: `${gpuCountLabel(hardware, "cuda")} configured · GPU model not recorded`,
|
||||
};
|
||||
};
|
||||
|
||||
@@ -136,8 +170,11 @@
|
||||
.filter(Boolean)
|
||||
.join(" · ");
|
||||
}
|
||||
if (hardware.accelerator) return [hardware.accelerator, gpuCountLabel(hardware)].join(" · ");
|
||||
return `NVIDIA CUDA · ${gpuCountLabel(hardware)} configured · GPU model and VRAM not recorded`;
|
||||
if (runtime.id === "spark") {
|
||||
return [hardware.accelerator || "NVIDIA GB10", gpuCountLabel(hardware, "spark")].filter(Boolean).join(" · ");
|
||||
}
|
||||
if (hardware.accelerator) return [hardware.accelerator, gpuCountLabel(hardware, runtime.id)].join(" · ");
|
||||
return `NVIDIA CUDA · ${gpuCountLabel(hardware, "cuda")} configured · GPU model and VRAM not recorded`;
|
||||
};
|
||||
|
||||
const renderHardwareEvidence = (container, badge, recipe) => {
|
||||
@@ -157,15 +194,17 @@
|
||||
if (isValidated) {
|
||||
const recorded = [
|
||||
hardware.accelerator,
|
||||
runtime.id === "cuda" ? gpuCountLabel(hardware) : hardware.system_memory,
|
||||
runtime.id === "mlx" || runtime.id === "mps" ? hardware.system_memory : gpuCountLabel(hardware, runtime.id),
|
||||
].filter(Boolean);
|
||||
const statements = [`${recorded.join(" · ")}.`];
|
||||
if (hardware.minimum_memory) statements.push(`Documented minimum: ${hardware.minimum_memory}.`);
|
||||
if (hardware.peak_memory) statements.push(`Measured: ${hardware.peak_memory}.`);
|
||||
if (!hardware.minimum_memory) statements.push("This recorded device is not a minimum requirement.");
|
||||
details.textContent = ` ${statements.join(" ")}`;
|
||||
} else if (runtime.id === "spark") {
|
||||
details.textContent = ` NVIDIA DGX Spark · ${gpuCountLabel(hardware, "spark")}. GB10 has no FA4 / sm_100a VSA kernel; keep Triton VSA and FA4 off.`;
|
||||
} else if (runtime.id === "cuda") {
|
||||
details.textContent = ` NVIDIA CUDA · ${gpuCountLabel(hardware)}. The source does not record the GPU model or VRAM.`;
|
||||
details.textContent = ` NVIDIA CUDA · ${gpuCountLabel(hardware, "cuda")}. The source does not record the GPU model or VRAM.`;
|
||||
} else {
|
||||
details.textContent = ` ${runtime.label}. The source does not record a device or memory requirement.`;
|
||||
}
|
||||
@@ -223,6 +262,10 @@
|
||||
const servingPanel = root.querySelector("[data-cookbook-serving]");
|
||||
const usage = root.querySelector("[data-cookbook-usage]");
|
||||
const servingAvailability = root.querySelector("[data-cookbook-serving-availability]");
|
||||
const knobsContainer = root.querySelector("[data-cookbook-knobs]");
|
||||
const deviceRow = root.querySelector("[data-cookbook-device-row]");
|
||||
const deviceOptions = root.querySelector("[data-cookbook-device-options]");
|
||||
const deviceCaption = root.querySelector("[data-cookbook-device-caption]");
|
||||
|
||||
modelOptions.setAttribute("aria-label", "Recipe");
|
||||
hardwareOptions.setAttribute("aria-label", "Runtime");
|
||||
@@ -289,16 +332,96 @@
|
||||
const clientDetails = servingPanel?.querySelector(".cookbook-serving__code");
|
||||
if (clientDetails && query.has("client")) clientDetails.open = true;
|
||||
|
||||
const knobDefs = new Map();
|
||||
familyRecipes.forEach((recipe) => knobsFor(recipe).forEach((knob) => {
|
||||
if (!knobDefs.has(knob.key)) knobDefs.set(knob.key, knob);
|
||||
}));
|
||||
const knobValues = {};
|
||||
knobDefs.forEach((knob, key) => {
|
||||
const fromQuery = query.get(key);
|
||||
const validValues = knobOptions(knob).map((option) => String(option.value));
|
||||
const useQueryValue = fromQuery !== null && validValues.includes(fromQuery);
|
||||
const raw = useQueryValue ? fromQuery : knob.default;
|
||||
knobValues[key] = typeof knob.default === "number" ? Number(raw) : raw;
|
||||
});
|
||||
|
||||
const renderKnobs = (recipe, hidden) => {
|
||||
if (!knobsContainer) return;
|
||||
const knobs = knobsFor(recipe);
|
||||
const renderedKeys = [...knobsContainer.querySelectorAll("[data-knob-row]")].map((row) => row.dataset.knobRow);
|
||||
if (renderedKeys.join(",") !== knobs.map((knob) => knob.key).join(",")) {
|
||||
knobsContainer.replaceChildren();
|
||||
knobs.forEach((knob) => {
|
||||
const row = document.createElement("div");
|
||||
row.className = "cookbook-selection-row";
|
||||
row.dataset.knobRow = knob.key;
|
||||
const labelWrap = document.createElement("div");
|
||||
labelWrap.className = "cookbook-selection-row__label";
|
||||
const strongLabel = document.createElement("strong");
|
||||
strongLabel.textContent = knob.label;
|
||||
const hintLabel = document.createElement("span");
|
||||
hintLabel.textContent = knob.hint || "";
|
||||
labelWrap.append(strongLabel, hintLabel);
|
||||
const grid = document.createElement("div");
|
||||
grid.className = "cookbook-option-grid cookbook-option-grid--hardware";
|
||||
grid.setAttribute("role", "group");
|
||||
grid.setAttribute("aria-label", knob.label);
|
||||
knobOptions(knob).forEach((option) => {
|
||||
const optionButton = document.createElement("button");
|
||||
optionButton.type = "button";
|
||||
optionButton.dataset.knobKey = knob.key;
|
||||
optionButton.dataset.knobValue = String(option.value);
|
||||
optionButton.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = option.label;
|
||||
optionButton.append(optionLabel);
|
||||
grid.append(optionButton);
|
||||
});
|
||||
row.append(labelWrap, grid);
|
||||
knobsContainer.append(row);
|
||||
});
|
||||
}
|
||||
knobsContainer.hidden = hidden || knobs.length === 0;
|
||||
knobsContainer.querySelectorAll("button[data-knob-key]").forEach((optionButton) => {
|
||||
const selected = String(knobValues[optionButton.dataset.knobKey]) === optionButton.dataset.knobValue;
|
||||
optionButton.classList.toggle("cookbook-option--selected", selected);
|
||||
optionButton.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
};
|
||||
|
||||
const recipesForRuntime = (runtimeId, groupRecipes) =>
|
||||
groupRecipes.filter((item) => runtimeFor(item).id === runtimeId);
|
||||
|
||||
const uniqueRuntimeIds = (groupRecipes) => {
|
||||
const ids = [];
|
||||
groupRecipes.forEach((item) => {
|
||||
const id = runtimeFor(item).id;
|
||||
if (!ids.includes(id)) ids.push(id);
|
||||
});
|
||||
return ids;
|
||||
};
|
||||
|
||||
const pickRecipeForRuntime = (runtimeId, preferredCount, groupRecipes) => {
|
||||
const siblings = recipesForRuntime(runtimeId, groupRecipes);
|
||||
if (!siblings.length) return null;
|
||||
if (preferredCount != null) {
|
||||
const match = siblings.find((item) => item.hardware?.gpu_count === preferredCount);
|
||||
if (match) return match;
|
||||
}
|
||||
return siblings.find((item) => item.hardware?.gpu_count === 1) || siblings[0];
|
||||
};
|
||||
|
||||
const renderRuntimeOptions = () => {
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const renderedIds = [...hardwareOptions.querySelectorAll("[data-recipe-id]")].map((option) => option.dataset.recipeId);
|
||||
if (renderedIds.join(",") === groupRecipes.map((recipe) => recipe.id).join(",")) return;
|
||||
const runtimeIds = uniqueRuntimeIds(groupRecipes);
|
||||
const renderedIds = [...hardwareOptions.querySelectorAll("[data-runtime-id]")].map((option) => option.dataset.runtimeId);
|
||||
if (renderedIds.join(",") === runtimeIds.join(",")) return;
|
||||
hardwareOptions.replaceChildren();
|
||||
groupRecipes.forEach((recipe) => {
|
||||
const runtime = runtimeFor(recipe);
|
||||
runtimeIds.forEach((runtimeId) => {
|
||||
const representative = pickRecipeForRuntime(runtimeId, 1, groupRecipes);
|
||||
const runtime = runtimeFor(representative);
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.recipeId = recipe.id;
|
||||
option.dataset.runtimeId = runtime.id;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
@@ -310,6 +433,49 @@
|
||||
});
|
||||
};
|
||||
|
||||
const renderDeviceOptions = (recipe) => {
|
||||
if (!deviceRow || !deviceOptions) return;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const runtime = runtimeFor(recipe);
|
||||
const siblings = recipesForRuntime(runtime.id, groupRecipes)
|
||||
.slice()
|
||||
.sort((left, right) => (left.hardware?.gpu_count || 0) - (right.hardware?.gpu_count || 0));
|
||||
const show = siblings.length > 1;
|
||||
deviceRow.hidden = !show;
|
||||
if (deviceCaption) {
|
||||
deviceCaption.textContent = runtime.id === "spark" ? "1 Spark or a QSFP pair" : "GPU count for this runtime";
|
||||
}
|
||||
if (!show) {
|
||||
deviceOptions.replaceChildren();
|
||||
return;
|
||||
}
|
||||
const renderedIds = [...deviceOptions.querySelectorAll("[data-recipe-id]")].map((option) => option.dataset.recipeId);
|
||||
if (renderedIds.join(",") !== siblings.map((item) => item.id).join(",")) {
|
||||
deviceOptions.replaceChildren();
|
||||
siblings.forEach((candidate) => {
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.recipeId = candidate.id;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = gpuCountLabel(candidate.hardware, runtime.id);
|
||||
const optionHint = document.createElement("span");
|
||||
optionHint.textContent = runtime.id === "spark" && candidate.hardware?.gpu_count === 2
|
||||
? "Ray sequence parallel over QSFP"
|
||||
: runtime.id === "spark"
|
||||
? "One GB10, local process"
|
||||
: `${gpuCountLabel(candidate.hardware, runtime.id)} configured`;
|
||||
option.append(optionLabel, optionHint);
|
||||
deviceOptions.append(option);
|
||||
});
|
||||
}
|
||||
deviceOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.recipeId === recipe.id;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
};
|
||||
|
||||
let notes = root.querySelector("[data-cookbook-notes]");
|
||||
if (!notes) {
|
||||
notes = document.createElement("aside");
|
||||
@@ -325,8 +491,9 @@
|
||||
let recipe = byId.get(selectedRecipeId);
|
||||
if (groupChanged || groupIdFor(recipe) !== selectedGroupId) {
|
||||
const currentRuntime = runtimeFor(recipe).id;
|
||||
const currentCount = recipe.hardware?.gpu_count;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
recipe = groupRecipes.find((candidate) => runtimeFor(candidate).id === currentRuntime) || groupRecipes[0];
|
||||
recipe = pickRecipeForRuntime(currentRuntime, currentCount, groupRecipes) || groupRecipes[0];
|
||||
selectedRecipeId = recipe.id;
|
||||
}
|
||||
|
||||
@@ -335,11 +502,14 @@
|
||||
renderRuntimeOptions();
|
||||
renderedRuntimeGroup = selectedGroupId;
|
||||
}
|
||||
renderDeviceOptions(recipe);
|
||||
const runtime = runtimeFor(recipe);
|
||||
const profile = servingPanel && servingProfiles[recipe.id];
|
||||
const useServer = Boolean(profile && usagePreference === "server");
|
||||
// The measured local profile and the server config have separate evidence.
|
||||
const activeRecipe = useServer ? { ...recipe, hardware: profile.hardware, evidence: "Source-backed" } : recipe;
|
||||
const knobs = knobsFor(recipe);
|
||||
renderKnobs(recipe, useServer);
|
||||
|
||||
if (usage) {
|
||||
usage.querySelectorAll("[data-cookbook-mode]").forEach((option) => {
|
||||
@@ -349,10 +519,10 @@
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
servingAvailability.textContent = profile
|
||||
? "The playground and API clients share one server process. Both workflows can run on your own machine."
|
||||
? "The playground and the OpenAI Python client share one server process. Both workflows can run on your own machine."
|
||||
: servingLoadFailed
|
||||
? "Server examples could not be loaded. Open the H3 server guide below, or use Python directly."
|
||||
: "This recipe uses Python directly. For the playground and API clients, choose FastH3 Preview with CUDA or MLX.";
|
||||
: "This recipe uses Python directly. FastH3 V1 and FastH3 V2 can also run a local server for the playground and the OpenAI Python client.";
|
||||
servingPanel.hidden = !useServer;
|
||||
commandBlock.hidden = useServer;
|
||||
root.querySelector("[data-cookbook-python-note]").hidden = useServer;
|
||||
@@ -360,11 +530,17 @@
|
||||
|
||||
if (useServer) {
|
||||
const isMLX = profile.runtime === "mlx";
|
||||
const isSpark = runtime.id === "spark";
|
||||
servingPanel.querySelector("[data-cookbook-server-lifetime]").textContent = isMLX
|
||||
? "Start once, then change prompts in the playground or your app. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
: isSpark
|
||||
? "Start once, then change prompts in the playground or your app. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
servingPanel.querySelector("[data-cookbook-install-guide]").href = isMLX
|
||||
? "../../getting_started/installation/mps/#run-fasth3-preview" : "../../getting_started/installation/gpu/";
|
||||
? "../../getting_started/installation/mlx/"
|
||||
: isSpark
|
||||
? "../../getting_started/installation/spark/"
|
||||
: "../../getting_started/installation/gpu/";
|
||||
servingPanel.querySelector("[data-cookbook-prepare]").hidden = !profile.prepare;
|
||||
servingPanel.querySelector("[data-cookbook-server-prepare]").textContent = profile.prepare;
|
||||
servingPanel.querySelector("[data-cookbook-server-install]").textContent = profile.install;
|
||||
@@ -392,18 +568,13 @@
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
hardwareOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.recipeId === recipe.id;
|
||||
const selected = option.dataset.runtimeId === runtime.id;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
const candidate = byId.get(option.dataset.recipeId);
|
||||
const candidateProfile = servingProfiles[candidate.id];
|
||||
const candidateRuntime = runtimeFor(candidateProfile && usagePreference === "server"
|
||||
? { ...candidate, hardware: candidateProfile.hardware } : candidate);
|
||||
option.querySelector("span").textContent = candidateRuntime.hint;
|
||||
});
|
||||
|
||||
description.textContent = useServer
|
||||
? `FastH3 Preview generates video with audio. This server profile uses the checked-in ${profile.runtime.toUpperCase()} configuration.`
|
||||
? `${recipe.group_label || recipe.label} generates video with audio. Start the local server, then use the playground or the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
: recipe.summary;
|
||||
label.textContent = useServer ? `${recipe.group_label || recipe.label} · Server` : recipe.label;
|
||||
model.textContent = recipe.model;
|
||||
@@ -418,17 +589,23 @@
|
||||
source.href = `https://github.com/hao-ai-lab/FastVideo/blob/main/${useServer ? profile.source : recipe.source}`;
|
||||
source.textContent = useServer ? "View server configuration" : "Open example source";
|
||||
modelLink.href = `https://huggingface.co/${recipe.model}`;
|
||||
command.textContent = recipe.command;
|
||||
command.textContent = useServer ? recipe.command : appendKnobFlags(recipe.command, knobs, knobValues);
|
||||
|
||||
renderHardwareEvidence(hardwareState, hardwareBadge, activeRecipe);
|
||||
|
||||
const limitations = useServer ? [
|
||||
const knobCaveats = useServer ? [] : knobs
|
||||
.filter((knob) => String(knobValues[knob.key]) !== String(knob.default))
|
||||
.map((knob) => `${knob.label} is set away from its recorded default (${knobDefaultLabel(knob)}). ` +
|
||||
"The script accepts this value, but it has not been benchmarked here.");
|
||||
const limitations = [...(useServer ? [
|
||||
profile.runtime === "mlx"
|
||||
? "This MLX server config has no recorded hardware run. Measurements from the Python recipe are not server memory requirements. Only text-to-video/audio is wired; reference inputs and fast modes are not exposed here."
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
: runtime.id === "spark"
|
||||
? "This Spark server config has no recorded serving benchmark. Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Compilation of the DiT is disabled."
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
`${profile.sampling.width} × ${profile.sampling.height} · ${profile.sampling.num_frames} frames · ${profile.sampling.fps} fps. The server supplies these defaults; the client sends the model and prompt.`,
|
||||
"Generation is serialized. Job metadata is held in memory and is lost when the server restarts.",
|
||||
] : recipe.limitations || [];
|
||||
] : recipe.limitations || []), ...knobCaveats];
|
||||
notes.replaceChildren();
|
||||
notes.hidden = limitations.length === 0;
|
||||
if (limitations.length) {
|
||||
@@ -446,7 +623,16 @@
|
||||
const nextQuery = new URLSearchParams(window.location.search);
|
||||
nextQuery.set("recipe", recipe.id);
|
||||
nextQuery.set("runtime", runtime.id);
|
||||
nextQuery.delete("gpus");
|
||||
const deviceSiblings = recipesForRuntime(runtime.id, groups.get(selectedGroupId) || []);
|
||||
if (deviceSiblings.length > 1 && recipe.hardware?.gpu_count != null) {
|
||||
nextQuery.set("gpus", String(recipe.hardware.gpu_count));
|
||||
} else {
|
||||
nextQuery.delete("gpus");
|
||||
}
|
||||
knobDefs.forEach((knob, key) => {
|
||||
if (knobs.some((activeKnob) => activeKnob.key === key)) nextQuery.set(key, String(knobValues[key]));
|
||||
else nextQuery.delete(key);
|
||||
});
|
||||
if (usage) {
|
||||
nextQuery.set("use", useServer ? "server" : "python");
|
||||
if (useServer && clientDetails?.open) nextQuery.set("client", selectedClient);
|
||||
@@ -466,6 +652,17 @@
|
||||
render({ groupChanged: true, historyMode: "push" });
|
||||
});
|
||||
hardwareOptions.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-runtime-id]");
|
||||
if (!option) return;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const currentCount = byId.get(selectedRecipeId)?.hardware?.gpu_count;
|
||||
const nextRecipe = pickRecipeForRuntime(option.dataset.runtimeId, currentCount, groupRecipes);
|
||||
if (!nextRecipe) return;
|
||||
selectedRecipeId = nextRecipe.id;
|
||||
selectedGroupId = groupIdFor(nextRecipe);
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
deviceOptions?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-recipe-id]");
|
||||
if (!option) return;
|
||||
selectedRecipeId = option.dataset.recipeId;
|
||||
@@ -484,6 +681,14 @@
|
||||
selectedClient = option.dataset.cookbookClient;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
knobsContainer?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-knob-key]");
|
||||
if (!option) return;
|
||||
const knob = knobDefs.get(option.dataset.knobKey);
|
||||
knobValues[option.dataset.knobKey] = typeof knob.default === "number"
|
||||
? Number(option.dataset.knobValue) : option.dataset.knobValue;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
|
||||
render();
|
||||
|
||||
@@ -496,6 +701,13 @@
|
||||
usagePreference = workflow(nextQuery.get("use"));
|
||||
selectedClient = ["python", "javascript", "curl"].includes(nextQuery.get("client")) ? nextQuery.get("client") : "curl";
|
||||
if (clientDetails) clientDetails.open = nextQuery.has("client");
|
||||
knobDefs.forEach((knob, key) => {
|
||||
const fromQuery = nextQuery.get(key);
|
||||
const validValues = knobOptions(knob).map((option) => String(option.value));
|
||||
if (fromQuery !== null && validValues.includes(fromQuery)) {
|
||||
knobValues[key] = typeof knob.default === "number" ? Number(fromQuery) : fromQuery;
|
||||
}
|
||||
});
|
||||
render({ historyMode: "none" });
|
||||
};
|
||||
bindFamilyPopstate();
|
||||
|
||||
+58
-14
@@ -146,8 +146,7 @@ img {
|
||||
}
|
||||
|
||||
.cookbook-section,
|
||||
.cookbook-builder,
|
||||
.cookbook-roadmap {
|
||||
.cookbook-builder {
|
||||
scroll-margin-top: 4.5rem;
|
||||
}
|
||||
|
||||
@@ -180,8 +179,7 @@ img {
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-section__heading h2,
|
||||
.md-typeset .cookbook-builder__intro h2,
|
||||
.md-typeset .cookbook-roadmap h2 {
|
||||
.md-typeset .cookbook-builder__intro h2 {
|
||||
margin: 0;
|
||||
color: var(--md-default-fg-color);
|
||||
font-size: clamp(1.4rem, 2.6vw, 1.85rem);
|
||||
@@ -493,8 +491,7 @@ img {
|
||||
max-width: 45rem;
|
||||
}
|
||||
|
||||
.cookbook-builder__intro > p,
|
||||
.cookbook-roadmap > p {
|
||||
.cookbook-builder__intro > p {
|
||||
color: var(--md-default-fg-color--light);
|
||||
font-size: 0.75rem;
|
||||
line-height: 1.6;
|
||||
@@ -567,7 +564,7 @@ img {
|
||||
}
|
||||
|
||||
.cookbook-option-grid--hardware {
|
||||
grid-template-columns: repeat(2, minmax(0, 1fr));
|
||||
grid-template-columns: repeat(auto-fit, minmax(6.4rem, 1fr));
|
||||
}
|
||||
|
||||
.cookbook-option-grid button {
|
||||
@@ -769,12 +766,6 @@ img {
|
||||
font-weight: 700;
|
||||
}
|
||||
|
||||
.cookbook-roadmap {
|
||||
padding: 2.5rem 0;
|
||||
margin: 3.5rem 0 1rem;
|
||||
border-top: 1px solid var(--cookbook-border);
|
||||
}
|
||||
|
||||
/* Lifecycle roadmap on family pages: inference is live; later stages are
|
||||
explicitly labelled planned and must not look runnable. */
|
||||
.cookbook-lifecycle {
|
||||
@@ -1058,6 +1049,55 @@ img {
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
|
||||
.cookbook-jumpnav {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 1.1rem;
|
||||
max-width: 48rem;
|
||||
margin: 0 auto 1.75rem;
|
||||
padding: 0.55rem 0;
|
||||
border-top: 1px solid var(--cookbook-border);
|
||||
border-bottom: 1px solid var(--cookbook-border);
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-jumpnav a {
|
||||
font-family: var(--md-code-font-family);
|
||||
font-size: 0.7rem;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--md-default-fg-color--light);
|
||||
text-decoration: none;
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-jumpnav a:hover {
|
||||
color: var(--cookbook-accent-strong);
|
||||
}
|
||||
|
||||
.cookbook-family-page details.cookbook-collapsible {
|
||||
max-width: 48rem;
|
||||
margin: 0 auto 0.75rem;
|
||||
border-color: var(--cookbook-border);
|
||||
box-shadow: none;
|
||||
}
|
||||
|
||||
.md-typeset details.cookbook-collapsible > summary {
|
||||
background: var(--cookbook-surface-raised);
|
||||
font-size: 0.85rem;
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-collapsible__body {
|
||||
font-size: 0.8rem;
|
||||
line-height: 1.65;
|
||||
color: var(--md-default-fg-color--light);
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-collapsible__body .cookbook-eyebrow {
|
||||
margin-top: 0.9rem;
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-collapsible__body pre {
|
||||
margin: 0.5rem 0 1rem;
|
||||
}
|
||||
|
||||
.cookbook-family-page {
|
||||
max-width: 62rem;
|
||||
}
|
||||
@@ -1184,6 +1224,10 @@ img {
|
||||
margin-top: 0.55rem;
|
||||
}
|
||||
|
||||
[data-cookbook-knobs]:not(:empty) {
|
||||
margin-top: 0.55rem;
|
||||
}
|
||||
|
||||
.cookbook-selection-row__label {
|
||||
padding-inline-start: 0.15rem;
|
||||
}
|
||||
@@ -1207,7 +1251,7 @@ img {
|
||||
}
|
||||
|
||||
.cookbook-option-grid--hardware {
|
||||
grid-template-columns: repeat(2, minmax(0, 1fr));
|
||||
grid-template-columns: repeat(auto-fit, minmax(6.4rem, 1fr));
|
||||
}
|
||||
|
||||
.cookbook-option-grid button {
|
||||
|
||||
@@ -18,8 +18,8 @@ the CUDA `fastvideo-kernel` package:
|
||||
- **Dense-only checkpoints** (the default converter) drop the 50 gate
|
||||
matrices and keep fused SDPA. They remain valid for dense inference.
|
||||
- **VSA-capable checkpoints** retain those gates, quantize them on the same
|
||||
affine grid, and record `vsa.capable` in `mlx_h3_dit.json`. Runtime VSA is
|
||||
still off until you pass `--vsa`.
|
||||
affine grid, and record `vsa.capable` in `mlx_h3_dit.json`. Preview leaves
|
||||
runtime VSA off until you pass `--vsa`. `mlx_fasth3_8step.py` turns it on.
|
||||
- **Tile sizes** 64 `(4, 4, 4)` and 256 `(4, 8, 8)`. Prefix keys can be
|
||||
`exempt` or `compete`. `--vsa-dense-first-n-steps` and `--vsa-dense-layers`
|
||||
force dense SDPA on the selected steps or blocks.
|
||||
@@ -32,9 +32,12 @@ the CUDA `fastvideo-kernel` package:
|
||||
but does not yet match reference video. `--vsa-impl reference` is the same
|
||||
as `auto`.
|
||||
|
||||
See the [Apple Silicon guide](../../getting_started/installation/mps.md) for
|
||||
conversion and `mlx_fasth3.py` flags. Do not enable VSA on a dense-only
|
||||
checkpoint; reconvert with `--include-vsa` first.
|
||||
See the [MLX install guide](../../getting_started/installation/mlx.md)
|
||||
and the [MiniMax H3 cookbook](../../cookbook/minimax-h3.md) for conversion
|
||||
and `mlx_fasth3.py` / `mlx_fasth3_8step.py` flags. Do not enable
|
||||
VSA on a dense-only checkpoint; reconvert with `--include-vsa` first. V1
|
||||
VSA is opt-in. FastH3 V2 converts with `--include-vsa` and turns VSA on
|
||||
by default.
|
||||
|
||||
H3 uses fused MLX RMSNorm by default, including dense inference. This can
|
||||
change BF16 rounding relative to the older explicit normalization path.
|
||||
|
||||
@@ -30,7 +30,7 @@ FastVideo and the reference model first produce different numbers?"
|
||||
| General logging | `init_logger(__name__)` |
|
||||
| Per-stage timing | `FASTVIDEO_STAGE_LOGGING` |
|
||||
| Profiling kernel timings | `FASTVIDEO_TORCH_PROFILER_DIR` (see [Profiling](profiling.md)) |
|
||||
| Function-call tracing | `FASTVIDEO_TRACE_FUNCTION` (heavy) |
|
||||
| Function-call tracing | `fastvideo.logger.enable_trace_function_call()` (heavy) |
|
||||
|
||||
## Quickstart
|
||||
|
||||
|
||||
@@ -141,6 +141,24 @@ it), while a GitHub outage or a >25 min wait lets it run anyway (fail open).
|
||||
The complete static graph remains available through `/test full`; path
|
||||
selection never deletes or dynamically invents a Buildkite step.
|
||||
|
||||
Integration steps depend on `golden-gate`. A failed selected golden prevents
|
||||
the expensive downstream jobs from starting; `/test full` still selects all
|
||||
twenty lanes. A condition-skipped golden satisfies the dependency, so direct
|
||||
lane reruns, scheduled SSIM, and training-only merge plans keep their existing
|
||||
meaning. This follows Buildkite's
|
||||
[conditional dependency rules](https://buildkite.com/docs/pipelines/configure/depends-on).
|
||||
No dependency allows failures. The trusted uploader accepts either the old
|
||||
complete graph or the complete golden-first graph during rollout, and rejects
|
||||
partial or arbitrary dependency changes.
|
||||
|
||||
The six automatic Fastcheck lanes are unchanged. Within VAE and transformer
|
||||
lanes, small Wan goldens run before independent component parity. Wan paths
|
||||
select matching VAE, dense/trajectory, or causal-cache goldens; shared Wan
|
||||
config and pipeline wiring select all four. Shared runtime changes continue
|
||||
to select broader coverage. The existing block references retain their exact
|
||||
environment identity; new tensor gates distinguish the effective FA2/FA4
|
||||
switch and VAE gates do not depend on an unused attention backend.
|
||||
|
||||
| Lane | Public `TEST_TYPE` | GPUs | Typical merge trigger |
|
||||
|---|---|---:|---|
|
||||
| Encoder | `encoder` | 1 | Universal Fastcheck |
|
||||
|
||||
@@ -43,6 +43,8 @@ FastVideo maps a Diffusers-style repo into a pipeline like:
|
||||
- `fastvideo/models/*`: model implementations (DiT, VAE, encoders, upsamplers).
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/models/wan/`: Wan's dense transformer, VAE, and component configs
|
||||
live together. The old Wan modules remain compatibility re-exports.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipeline logic built from stages.
|
||||
@@ -55,19 +57,15 @@ Minimal usage example (based on `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
generator = VideoGenerator.from_pretrained(model_id, {"engine": {"num_gpus": 1}})
|
||||
|
||||
sampling = SamplingParam.from_pretrained(model_id)
|
||||
sampling.num_frames = 45
|
||||
video = generator.generate_video(
|
||||
"A vibrant city street at sunset.",
|
||||
sampling_param=sampling,
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": "A vibrant city street at sunset.",
|
||||
"sampling": {"num_frames": 45},
|
||||
"output": {"output_path": "video_samples", "save_video": True},
|
||||
})
|
||||
```
|
||||
|
||||
## Some questions to ask yourself before starting
|
||||
@@ -209,7 +207,7 @@ class OfficialWanTransformer(torch.nn.Module):
|
||||
def forward(self, x):
|
||||
return self.patch_embedding(x)
|
||||
|
||||
# FastVideo model (simplified) in fastvideo/models/dits/wanvideo.py
|
||||
# FastVideo model (simplified) in fastvideo/models/wan/transformer.py
|
||||
class PatchEmbed(torch.nn.Module):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
@@ -227,7 +225,7 @@ class WanTransformer3DModel(torch.nn.Module):
|
||||
return self.patch_embedding(x)
|
||||
|
||||
# Mapping defined in a config (simplified; see the real mapping in
|
||||
# fastvideo/configs/models/dits/wanvideo.py)
|
||||
# fastvideo/models/wan/config.py)
|
||||
param_names_mapping = {
|
||||
r"^patch_embedding\.(.*)$": r"patch_embedding.proj.\1",
|
||||
r"^blocks\.(\d+)\.attn1\.to_q\.(.*)$": r"blocks.\1.to_q.\2",
|
||||
@@ -275,7 +273,7 @@ Mapping steps:
|
||||
- Instantiate the FastVideo DiT (`WanTransformer3DModel`) and compare
|
||||
its `state_dict().keys()` to the official keys.
|
||||
- Update `param_names_mapping` in
|
||||
fastvideo/configs/models/dits/wanvideo.py to resolve missing/unexpected keys.
|
||||
fastvideo/models/wan/config.py to resolve missing/unexpected keys.
|
||||
- Use `load_state_dict(strict=False)` during iteration to surface mismatches.
|
||||
```
|
||||
|
||||
@@ -466,19 +464,24 @@ The Wan2.1 T2V 1.3B Diffusers pipeline is a good “standard” example for
|
||||
FastVideo integration.
|
||||
|
||||
1. Verify model config + mapping.
|
||||
- DiT mapping: `fastvideo/configs/models/dits/wanvideo.py`
|
||||
- VAE: `fastvideo/models/vaes/wanvae.py`
|
||||
- DiT: `fastvideo/models/wan/transformer.py`
|
||||
- DiT mapping: `fastvideo/models/wan/config.py`
|
||||
- VAE: `fastvideo/models/wan/vae.py`
|
||||
- VAE config: `fastvideo/models/wan/vae_config.py`
|
||||
- Text encoder: `fastvideo/models/encoders/t5.py`
|
||||
|
||||
2. Parity test the core components.
|
||||
- Start with `bash scripts/validate_wan.sh all`: contracts and tiny goldens.
|
||||
- Example tests: `fastvideo/tests/transformers/test_wanvideo.py`,
|
||||
`fastvideo/tests/vaes/test_wan_vae.py`,
|
||||
`fastvideo/tests/encoders/test_t5_encoder.py`
|
||||
|
||||
3. Pipeline wiring.
|
||||
- Pipeline: `fastvideo/pipelines/basic/wan/wan_pipeline.py`
|
||||
- Pipeline config: `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
- Denoising and first-frame preparation: `fastvideo/pipelines/basic/wan/stages/`
|
||||
- Variant definitions: `fastvideo/models/wan/definition.py`
|
||||
- Pipeline config: `fastvideo/models/wan/pipeline_config.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/presets.py`
|
||||
|
||||
4. Minimal example.
|
||||
- Script: `examples/inference/basic/basic.py`
|
||||
|
||||
@@ -0,0 +1,294 @@
|
||||
# Environment Variables
|
||||
|
||||
FastVideo reads environment variables for expert switches, debugging, profiling, and the settings that launchers such
|
||||
as `torchrun` provide. This page is the policy for those variables. The contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit CI lane, and the coding-agent skill
|
||||
`.agents/skills/env-var-conventions/SKILL.md` points here. When the policy changes, update this page and the contract
|
||||
test in the same pull request.
|
||||
|
||||
## Rules
|
||||
|
||||
1. **Register every FastVideo variable in `fastvideo/envs.py`.** Each entry declares a type, a default, a category,
|
||||
and a description. Variables that other tools own (CUDA, NCCL, PyTorch, launchers) are not registered; code reads
|
||||
them directly with `os.environ.get("NAME")`, and the name must be in the external-variable allowlist
|
||||
(`EXTERNAL_ALLOWLIST` in the contract test). When FastVideo sets such a variable for the other tool, it calls
|
||||
`envs.set_external`, `envs.setdefault_external`, or `envs.unset_external`, and the name must be in
|
||||
`EXTERNAL_WRITE_ALLOWLIST`. Variables that FastVideo's CI and CI tooling define (for example `TEST_SCOPE` and
|
||||
`PERF_RUN_SOURCE`) keep their names, and test code under `fastvideo/tests/` reads them directly; they are listed
|
||||
in `CI_ONLY_VARIABLES` in the contract test, together with the file that sets each one.
|
||||
2. **Read with `envs.NAME.get()`, write with `envs.NAME.set()`, and change a value in tests with
|
||||
`envs.NAME.override()`.** Each type has one parsing rule. A value that the rule rejects raises
|
||||
`fastvideo.envs.EnvVarError` instead of falling back to the default.
|
||||
3. **Name FastVideo variables with the `FASTVIDEO_` prefix.** The second word states the purpose where one applies:
|
||||
`ENABLE_`, `DISABLE_`, `USE_`, `FORCE_`, `DEBUG_`, `TEST_`. Variables that only tests read use
|
||||
`FASTVIDEO_TEST_`, for example `FASTVIDEO_TEST_SD35_MODEL_DIR`, and the category `test`.
|
||||
4. **Keep a renamed variable as a deprecated alias until the next minor release.** Setting the old name logs a
|
||||
warning. Delete a variable that no code reads, and list it in `DEPRECATED_VARIABLES` so that setting it logs a
|
||||
warning.
|
||||
5. **Give each setting one source: an argument or an environment variable.** Settings that users change per
|
||||
deployment are arguments (CLI or YAML). Expert switches, emergency off switches, and debugging and test switches
|
||||
are environment variables.
|
||||
6. **Read variables inside functions.** `envs.NAME.get()` runs when the function runs, so a changed value takes
|
||||
effect without re-importing a module. Module level, class bodies, decorators, and default argument values run at
|
||||
import time.
|
||||
7. **Do not write the environment to pass values between parts of FastVideo.** Pass an argument instead. Tests use
|
||||
`envs.NAME.override()`, and `envs.override_external()` for variables outside the registry; both restore the
|
||||
previous value.
|
||||
|
||||
## Field types
|
||||
|
||||
| Class | Value type | Parsing rule |
|
||||
| ------------ | ---------------- | ----------------------------------------------------------------------------------- |
|
||||
| `EnvBool` | `bool` | `1`, `true`, `yes`, `on` are true; `0`, `false`, `no`, `off`, and `""` are false. |
|
||||
| | | Case-insensitive; surrounding whitespace is ignored. |
|
||||
| `EnvInt` | `int` | `int(value)` |
|
||||
| `EnvFloat` | `float` | `float(value)` |
|
||||
| `EnvStr` | `str` or `None` | The raw string. A `None` default means that the variable has no default. |
|
||||
| `EnvPath` | `str` or `None` | The raw string with a leading `~` expanded. |
|
||||
| `EnvChoice` | `str` | Stripped and lower-cased, then checked against the declared `choices`. |
|
||||
|
||||
A default can be a zero-argument function; `get()` calls it on each read while the variable is unset. The path roots
|
||||
use this to follow `XDG_CONFIG_HOME` and `XDG_CACHE_HOME`.
|
||||
|
||||
Using a field without a method, as in `if envs.FASTVIDEO_FA4:`, raises `TypeError`.
|
||||
|
||||
## Add a variable
|
||||
|
||||
1. Declare the variable in the matching section of `fastvideo/envs.py`:
|
||||
|
||||
```python
|
||||
FASTVIDEO_DEBUG_MY_STAGE = EnvBool(False, category="debug", doc="Log the inputs of MyStage.")
|
||||
```
|
||||
|
||||
The category is one of the values in `envs.CATEGORIES`.
|
||||
|
||||
2. Read the variable inside a function:
|
||||
|
||||
```python
|
||||
import fastvideo.envs as envs
|
||||
|
||||
def forward(self, batch):
|
||||
if envs.FASTVIDEO_DEBUG_MY_STAGE.get():
|
||||
logger.info("MyStage inputs: %s", batch.keys())
|
||||
```
|
||||
|
||||
3. Regenerate the table at the end of this page:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/contract/test_env_policy.py
|
||||
```
|
||||
|
||||
4. Run the contract test:
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/contract/test_env_policy.py
|
||||
```
|
||||
|
||||
In a test, change the value with `override`, which restores the previous value on exit:
|
||||
|
||||
```python
|
||||
with envs.FASTVIDEO_DEBUG_MY_STAGE.override(True):
|
||||
run_stage()
|
||||
```
|
||||
|
||||
For a variable outside the registry, such as `MASTER_PORT` or `TEST_SCOPE`, use `envs.override_external`. To keep
|
||||
an override until the end of a test, enter it through the `env_overrides` fixture from `fastvideo/tests/conftest.py`,
|
||||
which restores every value at teardown:
|
||||
|
||||
```python
|
||||
def test_my_stage(env_overrides):
|
||||
env_overrides.enter_context(envs.FASTVIDEO_DEBUG_MY_STAGE.override(True))
|
||||
env_overrides.enter_context(envs.override_external("MASTER_PORT", "29512"))
|
||||
run_stage()
|
||||
```
|
||||
|
||||
In test code under `fastvideo/tests/`, `override_external` accepts any name that code may read directly
|
||||
(`EXTERNAL_ALLOWLIST`, `CI_ONLY_VARIABLES`) or that is in `EXTERNAL_WRITE_ALLOWLIST`. Library code may write only
|
||||
the names in `EXTERNAL_WRITE_ALLOWLIST`.
|
||||
|
||||
## Rename or remove a variable
|
||||
|
||||
To rename a variable, declare it under the new name and list the old name in `deprecated_names`:
|
||||
|
||||
```python
|
||||
FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS = EnvBool(True,
|
||||
category="sampling",
|
||||
doc="...",
|
||||
deprecated_names=("LTX2_USE_DISTILLED_SIGMAS", ))
|
||||
```
|
||||
|
||||
`get()` reads an old name only when the new name is unset, and logs a warning once. Update the uses of the old name
|
||||
in `examples/`, `scripts/`, `docs/`, `apps/`, and the tests in the same pull request. Delete the old name in the next
|
||||
minor release.
|
||||
|
||||
To remove a variable that no code reads, delete its entry and add the name to `DEPRECATED_VARIABLES` in
|
||||
`fastvideo/envs.py` with a reason. The config resolution step `warn_deprecated_environment_variables` in
|
||||
`fastvideo/api/inference_resolution.py` calls `envs.warn_deprecated_variables()`, which logs a warning for each listed
|
||||
variable that is set. Delete the entry in the next minor release.
|
||||
|
||||
## What the contract test checks
|
||||
|
||||
The test parses every Python file under `fastvideo/`, including `fastvideo/tests/`, with Python's `ast` module. It
|
||||
skips `fastvideo/third_party/`, which is copied from upstream projects, and the registry `fastvideo/envs.py`. It does
|
||||
not check `apps/`, `examples/`, `scripts/`, `fastvideo-kernel/`, or `docs/`.
|
||||
|
||||
It reports each violation as `<path>: <kind> <name>`:
|
||||
|
||||
| Kind | Code that triggers it | Fix |
|
||||
| ------------------ | ------------------------------------------------------------ | -------------------------------------------- |
|
||||
| `read` | `os.getenv`, `os.environ.get`, `os.environ[...]`, or | Register the variable and call |
|
||||
| | `"NAME" in os.environ` with a name outside the allowlist, or | `envs.NAME.get()`. For a variable that |
|
||||
| | with a name built at runtime (`<dynamic>`) | another tool owns, add it to |
|
||||
| | | `EXTERNAL_ALLOWLIST` with a reason. |
|
||||
| `write` | `os.environ[...] = ...`, `setdefault`, `pop`, `del`, | Pass an argument instead. In tests, use |
|
||||
| | `os.putenv`, `os.unsetenv`, `monkeypatch.setenv`/`delenv`, | `envs.NAME.override()`, or |
|
||||
| | or an `envs.*_external` call with a name that the | `envs.override_external()` for a variable |
|
||||
| | helper does not accept | outside the registry. For a variable that |
|
||||
| | | another tool reads, call an |
|
||||
| | | `envs.*_external` helper and add the name to |
|
||||
| | | `EXTERNAL_WRITE_ALLOWLIST` with a reason. |
|
||||
| `whole-environ` | `os.environ.copy()`, `dict(os.environ)`, iteration, | Read the specific variables that the code |
|
||||
| | `mock.patch.dict(os.environ, ...)`, `os.environ.update` | needs. |
|
||||
| `bare-field` | A registry field used without calling one of its methods, | Call `envs.NAME.get()`. |
|
||||
| | as in `envs.NAME == "auto"` or `getter = envs.NAME.get` | |
|
||||
| `import-time-read` | `envs.NAME.get()` outside a function | Move the read into the function that uses |
|
||||
| | | the value. |
|
||||
| `prefix` | A registry entry without the `FASTVIDEO_` prefix | Rename the variable and keep the old name in |
|
||||
| | | `deprecated_names`. |
|
||||
| `unread` | A registry entry that no code reads with `get()` or | Delete the variable and add it to |
|
||||
| | `is_set()` | `DEPRECATED_VARIABLES`. |
|
||||
|
||||
The test recognizes `os` imported under another name, `from os import environ, getenv`, and a name held in a
|
||||
module-level string constant. Code that reaches the environment through `importlib` or `getattr(os, "environ")` is
|
||||
left to code review.
|
||||
|
||||
The test also checks that every registry entry has a category from `envs.CATEGORIES` and a description, and that the
|
||||
table at the end of this page matches the registry.
|
||||
|
||||
**Known violations.** `KNOWN_VIOLATIONS` in the contract test lists the violations that existed when the policy was
|
||||
introduced. The list only shrinks. A violation that is not in the list fails the test. A listed violation that no
|
||||
longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
|
||||
## Registered variables
|
||||
|
||||
<!-- BEGIN GENERATED ENV TABLE: python fastvideo/tests/contract/test_env_policy.py -->
|
||||
| Variable | Type | Default | Category | Description |
|
||||
| ---------------------------------------------------------- | ------------------------------ | ----------------------------------------------------- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `FASTVIDEO_CONFIG_ROOT` | path | computed | path | Root directory for FastVideo configuration files, at runtime and at installation. Defaults to ~/.config/fastvideo, or $XDG_CONFIG_HOME/fastvideo when XDG_CONFIG_HOME is set. |
|
||||
| `FASTVIDEO_CACHE_ROOT` | path | computed | path | Root directory for FastVideo cache files. Defaults to ~/.cache/fastvideo, or $XDG_CACHE_HOME/fastvideo when XDG_CACHE_HOME is set. |
|
||||
| `FASTVIDEO_REASON1_WEIGHTS_PATH` | str | unset | path | Local path or Hugging Face id of Reason1 weights to load instead of the checkpoint's own. |
|
||||
| `FASTVIDEO_HOST_IP` | str | `""` | distributed | IP address of this node when the node has several network interfaces. Set it on each node for multi-node inference. |
|
||||
| `FASTVIDEO_LOOPBACK_IP` | str | `""` | distributed | Loopback IP address to use instead of the detected one. |
|
||||
| `FASTVIDEO_RAY_PER_WORKER_GPUS` | float | `1.0` | distributed | GPUs per Ray worker. A fraction lets Ray schedule several actors on one GPU, so other actors can share the GPUs with FastVideo. |
|
||||
| `FASTVIDEO_NCCL_SO_PATH` | str | unset | distributed | Path to the NCCL library file. Needed because the nccl>=2.19 that PyTorch ships has a bug (https://github.com/NVIDIA/nccl/issues/1234). |
|
||||
| `FASTVIDEO_HCCL_SO_PATH` | str | unset | distributed | Path to the HCCL library file on Ascend NPUs. Deprecated names: `HCCL_SO_PATH`. |
|
||||
| `FASTVIDEO_WORKER_MULTIPROC_METHOD` | one of spawn, fork, forkserver | `spawn` | distributed | Multiprocessing start method for worker processes. |
|
||||
| `FASTVIDEO_ULYSSES_A2A` | one of off, auto | `off` | distributed | Sequence-parallel all-to-all backend. off uses the NCCL path in DistributedAutograd.AllToAll4D. auto uses the fused NVLink kernel when the group is a load-store accessible mesh of 2, 4, 6, or 8 ranks in eager execution, and the NCCL path otherwise. |
|
||||
| `FASTVIDEO_CONFIGURE_LOGGING` | bool | `1` | logging | Configure logging at import. When true, FastVideo uses its default logging configuration or the file in FASTVIDEO_LOGGING_CONFIG_PATH. |
|
||||
| `FASTVIDEO_LOGGING_CONFIG_PATH` | str | unset | logging | Path to a JSON logging configuration file. |
|
||||
| `FASTVIDEO_LOGGING_LEVEL` | str | `INFO` | logging | Default logging level. |
|
||||
| `FASTVIDEO_LOGGING_PREFIX` | str | `""` | logging | Prefix prepended to every log message. |
|
||||
| `FASTVIDEO_STAGE_LOGGING` | bool | `0` | logging | Log the time that each pipeline stage takes. |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | str | unset | attention | Attention backend, as an AttentionBackendEnum name such as TORCH_SDPA, FLASH_ATTN, VIDEO_SPARSE_ATTN, SAGE_ATTN, or SAGE_ATTN_THREE. Config resolution uses it when engine.attention.backend is unset. An unsupported name raises an error. |
|
||||
| `FASTVIDEO_FA4` | bool | `0` | attention | The FLASH_ATTN backend uses FlashAttention-4 (flash_attn.cute) instead of FA3 or FA2. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FA4_PACKED_VARLEN` | bool | `0` | attention | MiniMax-H3 dense DiT self-attention uses the FlashAttention-4 packed-varlen entry point. This changes the floating-point reduction order, so it is an inference-only opt-in. |
|
||||
| `FASTVIDEO_VSA_SM100A` | bool | `0` | attention | VIDEO_SPARSE_ATTN_H3 sends no-grad tile-64 forwards to the data-center Blackwell (sm_100a) kernel. fastvideo-kernel reads the same variable with the same rule. |
|
||||
| `FASTVIDEO_NVFP4_FA4` | bool | `0` | attention | FlashAttention-4 quantizes Q and K to NVFP4. An explicit nvfp4_fa4 attention implementation argument takes precedence. |
|
||||
| `FASTVIDEO_DISABLE_ATTENTION_COMPILE` | bool | `1` | attention | Keep attention forward out of torch.compile graphs (torch.compiler.disable). Set it to 0 to let attention constructed under that setting be traced. Setting it explicitly to true also blocks regional compile. |
|
||||
| `FASTVIDEO_MLX_WINDOW` | int | `0` | attention | MLX FastWan windowed attention size in tokens. 0 uses full attention. |
|
||||
| `FASTVIDEO_MLX_WINDOW_SINK` | int | `0` | attention | Number of sink tokens that MLX windowed attention always attends to. |
|
||||
| `FASTVIDEO_INFERENCE_TORCH_COMPILE` | bool | `0` | performance | Compile each DiT transformer block with fullgraph torch.compile at inference. Same as engine.compile.regional=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE` | bool | `0` | performance | MiniMax-H3 VAE decode splits its temporal chunks across the sequence-parallel ranks instead of running serially on the output rank. Same as pipeline.model.minimax_h3.vae_parallel_decode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_ENCODE` | bool | `0` | performance | MiniMax-H3 reference-video VAE encode splits its temporal chunks across the sequence-parallel ranks. Same as pipeline.model.minimax_h3.vae_parallel_encode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE_STRATEGY` | str | unset | performance | Collective that moves chunks in parallel VAE decode: gather (used when unset) or all_gather. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FUSIONS` | str | `""` | performance | MiniMax-H3 inference-only Triton fusions: all, 1, or a comma-separated subset of modulate,qknorm_rope,swiglu. Empty, 0, or none keeps the eager implementation. |
|
||||
| `FASTVIDEO_FSDP2_AUTOWRAP` | bool | `0` | performance | FSDP2 shards modules by parameter count instead of the model's shard conditions. Not supported by self-forcing distillation. |
|
||||
| `FASTVIDEO_FSDP2_MIN_PARAMS` | int | `10000000` | performance | Minimum parameter count of a module that FASTVIDEO_FSDP2_AUTOWRAP shards. |
|
||||
| `FASTVIDEO_MLX_COMPILE` | bool | `0` | performance | Compile the MLX DiT forward with mx.compile. |
|
||||
| `FASTVIDEO_MLX_FAST_NORM` | bool | `0` | performance | Use MLX fast normalization kernels. |
|
||||
| `FASTVIDEO_MLX_DQ_GEMM` | str | `1` | performance | MLX dequantized GEMM for affine-quantized weights: 0 turns it off, 1 uses the measured minimum row count, and an integer sets the minimum row count. |
|
||||
| `FASTVIDEO_LTX2_VAE_CHANNELS_LAST_3D` | bool | `1` | performance | LTX-2 VAE uses the channels_last_3d memory format. |
|
||||
| `FASTVIDEO_LTX2_DISABLE_AUDIO_AUTOCAST` | bool | `1` | performance | LTX-2 audio decoding runs without CUDA autocast. Deprecated names: `LTX2_DISABLE_AUDIO_AUTOCAST`. |
|
||||
| `FASTVIDEO_FLUX2_DISABLE_BF16_REDUCED_PRECISION_REDUCTION` | bool | `0` | performance | Flux denoising disables reduced-precision reductions in bf16 matmuls, which tightens accumulation for the 4-step Klein model. |
|
||||
| `FASTVIDEO_FFMPEG_BIN` | str | `ffmpeg` | output | ffmpeg executable used to save video with audio. |
|
||||
| `FASTVIDEO_VIDEO_CODEC` | str | `libx264` | output | ffmpeg video codec for saved videos. |
|
||||
| `FASTVIDEO_NVENC_PRESET` | str | `p1` | output | NVENC preset when the codec is an \*_nvenc codec. |
|
||||
| `FASTVIDEO_NVENC_TUNE` | str | `ull` | output | NVENC tune option. |
|
||||
| `FASTVIDEO_NVENC_RC` | str | `constqp` | output | NVENC rate-control mode. |
|
||||
| `FASTVIDEO_NVENC_QP` | str | `28` | output | NVENC quantization parameter. |
|
||||
| `FASTVIDEO_NVENC_BF` | str | `0` | output | NVENC number of B-frames. |
|
||||
| `FASTVIDEO_X264_PRESET` | str | `ultrafast` | output | x264 preset for non-NVENC codecs. |
|
||||
| `FASTVIDEO_OUTPUT_PIX_FMT` | str | `yuv420p` | output | ffmpeg pixel format for saved videos. |
|
||||
| `FASTVIDEO_NVTX_PROFILE` | bool | `0` | profiling | Emit NVTX ranges for external profilers such as Nsight Systems. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_DIR` | path | unset | profiling | Enables the torch profiler and sets the directory for its traces. Must be an absolute path. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_RECORD_SHAPES` | bool | `0` | profiling | Torch profiler records shapes. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_PROFILE_MEMORY` | bool | `0` | profiling | Torch profiler profiles memory. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_STACK` | bool | `0` | profiling | Torch profiler captures stacks. Costs about 1.5x runtime and 1.4x trace size. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_FLOPS` | bool | `0` | profiling | Torch profiler profiles FLOPs. |
|
||||
| `FASTVIDEO_TORCH_PROFILE_REGIONS` | str | `""` | profiling | Comma-separated profiler regions to record. The torch profiler requires at least one region. |
|
||||
| `FASTVIDEO_TRACE_ACTIVATIONS` | bool | `0` | debug | Enable activation trace hooks. |
|
||||
| `FASTVIDEO_TRACE_LAYERS` | str | `""` | debug | Regex filter for traced module names. Empty means all. |
|
||||
| `FASTVIDEO_TRACE_STATS` | str | `abs_mean,sum` | debug | Comma-separated activation statistics dumped for each output tensor. |
|
||||
| `FASTVIDEO_TRACE_OUTPUT` | str | `/tmp/fv_trace_<pid>.jsonl` | debug | JSONL path for activation traces. The literal <pid> is replaced at runtime. |
|
||||
| `FASTVIDEO_TRACE_STEPS` | str | `""` | debug | Comma-separated denoising step indices. Empty means all. |
|
||||
| `FASTVIDEO_H3_VSA_PROBE` | str | unset | debug | Output directory for the VSA-H3 attention-mass probe, which writes one .pt file per step, layer, and rank. Keeps the model out of regional compile. |
|
||||
| `FASTVIDEO_LTX2_GEMMA_LOG` | str | `""` | debug | Log file for LTX-2 Gemma text-encoder hidden states, used by parity tests. Deprecated names: `LTX2_FASTVIDEO_GEMMA_LOG`. |
|
||||
| `FASTVIDEO_COSMOS25_LOG_KNOBS` | bool | `0` | debug | Log the Cosmos 2.5 latent-preparation conditioning inputs. |
|
||||
| `FASTVIDEO_CFG_GATE_STEP` | float | `1.0` | sampling | CFG gating fraction in [0, 1]. Steps before len(timesteps) \* X run the conditional and unconditional forwards; later steps reuse the cached difference. 1.0 disables gating. |
|
||||
| `FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS` | bool | `1` | sampling | LTX-2 uses the distilled sigma schedule when pipeline.model.ltx2.use_distilled_sigmas is also true. Deprecated names: `LTX2_USE_DISTILLED_SIGMAS`. |
|
||||
| `FASTVIDEO_EVAL_CACHE` | path | computed | eval | Cache directory for evaluation models and datasets. Defaults to $FASTVIDEO_CACHE_ROOT/eval. |
|
||||
| `FASTVIDEO_PHYSICS_IQ_BUCKET_URL` | str | `https://storage.googleapis.com/physics-iq-benchmark` | eval | Base URL of the Physics-IQ benchmark bucket. |
|
||||
| `FASTVIDEO_VBENCH_FULL_INFO_JSON` | str | unset | eval | Path to VBench_full_info.json, used instead of the vendored copy. Deprecated names: `VBENCH_FULL_INFO_JSON`. |
|
||||
| `FASTVIDEO_FVD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the FVD metric. |
|
||||
| `FASTVIDEO_FAD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the audio Frechet distance metric. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_DATA_DIR` | str | `data/cats` | test | Raw data directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_CAPTION_JSON` | str | `videos2caption_1_sample.json` | test | Caption file, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_CAPTION_JSON`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_VIDEO_SUBDIR` | str | `video` | test | Video subdirectory, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_VIDEO_SUBDIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_OUTPUT_DIR` | str | `data/ltx2_overfit_preprocessed` | test | Output directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_OUTPUT_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_MODEL` | str | `FastVideo/LTX2-Distilled-Diffusers` | test | Model repository whose encoders preprocess_ltx2_overfit.py uses. Deprecated names: `LTX2_OVERFIT_MODEL`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_NUM_COPIES` | int | `4` | test | Number of copies of the overfit sample in the parquet file. Deprecated names: `LTX2_OVERFIT_NUM_COPIES`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_DATA_DIR` | str | `data/kandinsky5_overfit` | test | Raw data directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_OUTPUT_DIR` | str | `data/kandinsky5_overfit_preprocessed` | test | Output directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_OUTPUT_DIR`. |
|
||||
| `FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO` | str | `FastVideo/ssim-reference-videos` | test | Hugging Face repository that holds the SSIM reference videos. Deprecated names: `FASTVIDEO_SSIM_REFERENCE_HF_REPO`. |
|
||||
| `FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO_TYPE` | str | `dataset` | test | Repository type of FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO. Deprecated names: `FASTVIDEO_SSIM_REFERENCE_HF_REPO_TYPE`. |
|
||||
| `FASTVIDEO_TEST_SSIM_SKIP_REFERENCE_DOWNLOAD` | bool | `0` | test | SSIM tests use local reference videos without downloading. Deprecated names: `FASTVIDEO_SSIM_SKIP_REFERENCE_DOWNLOAD`. |
|
||||
| `FASTVIDEO_TEST_SSIM_FULL_QUALITY` | bool | `0` | test | SSIM tests use the full-quality sampling configurations. Deprecated names: `FASTVIDEO_SSIM_FULL_QUALITY`. |
|
||||
| `FASTVIDEO_TEST_NIGHTLY` | bool | `0` | test | Run the nightly end-to-end overfit tests. Deprecated names: `FASTVIDEO_NIGHTLY`. |
|
||||
| `FASTVIDEO_TEST_ULYSSES_FAULT_RANK` | str | unset | test | Rank that fails in the Ulysses fault-injection test. The test sets it for its worker processes. Deprecated names: `FASTVIDEO_ULYSSES_FAULT_RANK`. |
|
||||
| `FASTVIDEO_TEST_ULYSSES_FAULT_STAGE` | str | unset | test | Stage that fails in the Ulysses fault-injection test. The test sets it for its worker processes. Deprecated names: `FASTVIDEO_ULYSSES_FAULT_STAGE`. |
|
||||
| `FASTVIDEO_TEST_GOLDEN_GATE_DIR` | str | unset | test | Local directory of golden-gate reference tensors. Deprecated names: `FASTVIDEO_GOLDEN_GATE_DIR`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ALLOW_LOW_MEMORY` | bool | `0` | test | Run the MLX Wan2.2 5B real-weights parity test on hosts with little memory. Deprecated names: `FASTVIDEO_WAN22_5B_ALLOW_LOW_MEMORY`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ROOT` | str | unset | test | Local Wan2.2 5B checkpoint for the MLX real-weights parity test. Deprecated names: `FASTVIDEO_WAN22_5B_ROOT`. |
|
||||
| `FASTVIDEO_TEST_GRADNORM_UPDATE` | bool | `0` | test | Gradient-norm regression tests update their references. Deprecated names: `FASTVIDEO_GRADNORM_UPDATE`. |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_SSIM_MODEL_PATH` | str | `FastVideo/DreamX-World-5B-Cam-Diffusers` | test | Model for the DreamX-World camera SSIM test. Deprecated names: `DREAMX_WORLD_SSIM_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_AR_SSIM_MODEL_PATH` | str | `FastVideo/DreamX-World-5B-Diffusers` | test | Model for the DreamX-World autoregressive SSIM test. Deprecated names: `DREAMX_WORLD_AR_SSIM_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_FLUX_T2I_MODEL_DIR` | str | `black-forest-labs/FLUX.1-dev` | test | Model for the Flux text-to-image SSIM test. Deprecated names: `FLUX_T2I_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_FLUX_TRANSFORMER_PATH` | str | unset | test | Local Flux transformer for the Flux transformer test. Deprecated names: `FLUX_TRANSFORMER_PATH`. |
|
||||
| `FASTVIDEO_TEST_GAMECRAFT_MODEL_PATH` | str | `FastVideo/HunyuanGameCraft-Diffusers` | test | Model for the HunyuanGameCraft SSIM test. Deprecated names: `GAMECRAFT_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_MODEL_PATH` | str | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | test | Model for the GEN3C SSIM test. Deprecated names: `GEN3C_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_IMAGE_PATH` | str | unset | test | Input image for the GEN3C SSIM test. Deprecated names: `GEN3C_TEST_IMAGE_PATH`. |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_LOCAL_WEIGHTS_DIR` | str | unset | test | Local official GLM-Image weights for the GLM-Image SSIM test. Deprecated names: `GLM_IMAGE_LOCAL_WEIGHTS_DIR`. |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_MODEL_DIR` | str | unset | test | Model for the GLM-Image SSIM test. Deprecated names: `GLM_IMAGE_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_NUM_GPUS` | int | `1` | test | GPUs for the Kandinsky5 nightly end-to-end overfit test. Deprecated names: `KANDINSKY5_E2E_NUM_GPUS`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_WRITE_REFERENCE` | bool | `0` | test | The Kandinsky5 nightly end-to-end test writes a missing reference video. Deprecated names: `KANDINSKY5_E2E_WRITE_REFERENCE`. |
|
||||
| `FASTVIDEO_TEST_LONGCAT_MODEL_ROOT` | str | unset | test | Local LongCat-Video checkpoint for the golden-gate test. Deprecated names: `LONGCAT_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_GOLDEN_DIR` | str | unset | test | Local directory of MiniMax-H3 golden-gate tensors. Deprecated names: `MINIMAX_H3_GATE_GOLDEN_DIR`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_LAYER` | int | `0` | test | Transformer layer that the MiniMax-H3 golden-gate test checks. Deprecated names: `MINIMAX_H3_GATE_LAYER`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_MODEL_ROOT` | str | unset | test | Local MiniMax-H3 checkpoint for the golden-gate test. Deprecated names: `MINIMAX_H3_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_SD35_MODEL_DIR` | str | `stabilityai/stable-diffusion-3.5-medium` | test | Model for the Stable Diffusion 3.5 SSIM test. Deprecated names: `SD35_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_TAEH3_REFERENCE_DIR` | str | unset | test | Upstream taehv checkout for the MLX TAEH3 parity test. Deprecated names: `TAEH3_REFERENCE_DIR`. |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_DIR` | str | `Tongyi-MAI/Z-Image-Turbo` | test | Model for the Z-Image SSIM test. Deprecated names: `ZIMAGE_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_REVISION` | str | `f332072aa78be7aecdf3ee76d5c247082da564a6` | test | Hugging Face revision of the Z-Image model for its SSIM test. Deprecated names: `ZIMAGE_MODEL_REVISION`. |
|
||||
|
||||
Variables that FastVideo no longer reads; setting one logs a warning:
|
||||
|
||||
| Deprecated variable | Reason |
|
||||
| ----------------------------------------- | ---------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
<!-- END GENERATED ENV TABLE -->
|
||||
@@ -112,7 +112,7 @@ per-metric policy with direction, percent threshold, absolute threshold, and a
|
||||
|
||||
`test_inference_performance.py` temporarily sets `FASTVIDEO_STAGE_LOGGING=1`
|
||||
while it runs so pipeline stage execution times are available in
|
||||
`generate_video(...).logging_info`. Stage logs use pipeline-unique keys such as
|
||||
`generate(...).logging_info`. Stage logs use pipeline-unique keys such as
|
||||
`prompt_encoding_stage` so duplicate stage classes do not collide. For
|
||||
`PipelineStage` entries, shared component stage bases emit a stable
|
||||
`component_metric`: text encoding stages map to `text_encoder_time_s`,
|
||||
|
||||
@@ -10,6 +10,7 @@ slash-command mappings, and workflow ownership live in
|
||||
|---|---|---|
|
||||
| Unit tests | `fastvideo/tests/api`, `fastvideo/tests/dataset`, `fastvideo/tests/entrypoints`, `fastvideo/tests/workflow`, CPU-safe `fastvideo/tests/train` subsets | Validate individual functions, APIs, contracts, and lightweight workflows. |
|
||||
| Component tests | `fastvideo/tests/encoders`, `fastvideo/tests/transformers`, `fastvideo/tests/vaes` | Validate loading and basic behavior for model components. |
|
||||
| Golden gates | `fastvideo/tests/golden_gate` | Compare small, deterministic component outputs exactly against device/runtime-matched reference tensors. |
|
||||
| Train framework tests | `fastvideo/tests/train/models`, `fastvideo/tests/train/methods` | Exercise the new `fastvideo/train/` framework on real checkpoints and tiny synthetic batches. |
|
||||
| SSIM tests | `fastvideo/tests/ssim` | Compare generated videos against references to catch visual regressions. |
|
||||
| Training tests | `fastvideo/tests/training` | Validate legacy training loops, LoRA, distillation, self-forcing, and VSA behavior. |
|
||||
@@ -20,14 +21,74 @@ slash-command mappings, and workflow ownership live in
|
||||
|
||||
## Running Tests Locally
|
||||
|
||||
Run the narrowest useful suite while iterating:
|
||||
Start with the cheapest checks that cover the changed behavior, and stop on a
|
||||
failure before starting heavier dependent checks:
|
||||
|
||||
1. Run pre-commit on changed files and focused import, config, and contract tests.
|
||||
2. Run the smallest matching component golden gate. Prefer one GPU, tiny fixed
|
||||
inputs, cached component weights, and direct tensor comparisons over renders.
|
||||
3. Run focused component parity or default SSIM when the golden does not cover
|
||||
the changed behavior, such as VAE normalization or pipeline wiring.
|
||||
4. Run full-quality renders or broad suites when required by the change, an
|
||||
explicit request, or CI policy, rather than on every edit.
|
||||
|
||||
A golden must cover the component being changed. Wan has four small gates:
|
||||
|
||||
| Gate | Boundary |
|
||||
|---|---|
|
||||
| `test_wan_t2v.py` | Dense transformer block 0 |
|
||||
| `test_wan_vae.py` | FP32 encode, BF16 decode, streaming, and cache reset |
|
||||
| `test_wan_causal.py` | Real block weights, cache append/rewrite, and sink eviction |
|
||||
| `test_wan_denoising.py` | Three real 1.3B DiT/UniPC steps with fixed prompt embeddings |
|
||||
|
||||
These use an immutable Wan checkpoint revision and only download the required
|
||||
component. The trajectory gate needs no tokenizer, text encoder, VAE, or video
|
||||
reference. Weight-free stage tests also exercise every step of a 50-step UniPC
|
||||
loop, CFG caching, expert switching, conditioning layouts, and DMD RNG order.
|
||||
These checks do not replace independent Diffusers parity or end-to-end SSIM.
|
||||
|
||||
Use the golden's matching GPU, dtype, backend, and runtime. A missing reference
|
||||
or environment mismatch is not a pass. Do not replace a reference with the
|
||||
candidate output just to clear a failure. For a relocation, the unchanged
|
||||
parent is the baseline; two imports of the same class are not numerical proof.
|
||||
|
||||
The VAE and transformer CI lanes run their Wan component goldens first.
|
||||
Selected integration lanes wait for the golden lane in merge/full builds.
|
||||
See [CI/CD Architecture](ci_architecture.md) for direct-rerun and skip semantics.
|
||||
|
||||
For Wan, one command enforces the local ordering and stops on failure:
|
||||
|
||||
The initial contracts include `tests/api/test_wan_definitions.py`: all registered
|
||||
Wan aliases, local-manifest detector precedence, sampling/precision defaults,
|
||||
config isolation, and legacy serialized-config compatibility. These require no
|
||||
weights or Hub access; the current package import still needs its prepared
|
||||
runtime environment.
|
||||
|
||||
```bash
|
||||
pytest tests/
|
||||
pytest fastvideo/tests/ -v
|
||||
pytest fastvideo/tests/encoders -vs
|
||||
pytest fastvideo/tests/transformers -vs
|
||||
pytest fastvideo/tests/vaes -vs
|
||||
bash scripts/validate_wan.sh vae # contracts, then the VAE golden
|
||||
bash scripts/validate_wan.sh dense parity # contracts, goldens, Diffusers parity
|
||||
bash scripts/validate_wan.sh all default # then focused T2V/I2V/causal SSIM
|
||||
```
|
||||
|
||||
The second argument is an upper validation tier, not a reference override.
|
||||
Supply the GPU/backend/runtime and SSIM model/tier settings that match the
|
||||
reference. The script never updates references. Causal component coverage is
|
||||
a block-cache fingerprint, not independent full causal-pipeline parity.
|
||||
|
||||
New named-tensor gates write a missing output to `*.candidate.pt` and fail;
|
||||
running candidate code again cannot turn it into an approved reference. Seed
|
||||
from unchanged, pushed source, verify two independent processes bit-for-bit,
|
||||
then review and publish only the new reference files. Preserve the baseline
|
||||
source SHA, test recipe SHA, checkpoint revision, runtime, and comparison
|
||||
receipt with the run artifacts. Never overwrite an existing video reference
|
||||
as a side effect of adding a tensor gate.
|
||||
|
||||
Examples of focused checks:
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/loader/test_wan_family_imports.py -q
|
||||
pytest fastvideo/tests/golden_gate/test_wan_t2v.py -q
|
||||
pytest fastvideo/tests/vaes/test_wan_vae.py -q
|
||||
```
|
||||
|
||||
GPU-heavy suites need the right hardware, credentials, local caches, and
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Cosmos recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>World-generation prompts work best describing a scene and camera motion; the built-in prompt in the example is a known-good starting point.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- World-generation prompts work best describing a scene and camera motion; the built-in prompt in the example is a known-good starting point.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+32
-20
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# FLUX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,24 +112,30 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>FLUX.1 dev defaults to a local <code>official_weights/FLUX.1-dev</code> directory in the example; the cookbook command passes the Hugging Face ID explicitly instead.</li>
|
||||
<li>Image outputs land under <code>outputs/</code>; adjust <code>--output</code> if that path is not writable.</li>
|
||||
<li>Gated checkpoints (FLUX.1 dev): run <code>huggingface-cli login</code> and accept the license on Hugging Face first.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- FLUX.1 dev defaults to a local `official_weights/FLUX.1-dev` directory in the example; the cookbook command passes the Hugging Face ID explicitly instead.
|
||||
- Image outputs land under `outputs/`; adjust `--output` if that path is not writable.
|
||||
- Gated checkpoints (FLUX.1 dev): run `huggingface-cli login` and accept the license on Hugging Face first.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# GLM-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The editing example reads <code>assets/images/couple.jpg</code> from the repository root, so run it from a repo checkout rather than an arbitrary working directory.</li>
|
||||
<li>Output paths default under <code>image_output/</code>; pass <code>--output</code> to change them.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The editing example reads `assets/images/couple.jpg` from the repository root, so run it from a repo checkout rather than an arbitrary working directory.
|
||||
- Output paths default under `image_output/`; pass `--output` to change them.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+32
-20
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Hunyuan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,24 +112,30 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Out of memory: <code>basic_hy15.py</code> already enables dit/VAE/text-encoder CPU offload; further tradeoffs are described in <a href="../../inference/optimizations/">Optimizations</a>.</li>
|
||||
<li><code>pin_cpu_memory</code> errors on low-RAM machines are documented inline in the example; set it to false as the source comment suggests.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Out of memory: `basic_hy15.py` already enables dit/VAE/text-encoder CPU offload; further tradeoffs are described in [Optimizations](../inference/optimizations.md).
|
||||
- `pin_cpu_memory` errors on low-RAM machines are documented inline in the example; set it to false as the source comment suggests.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+2
-11
@@ -41,13 +41,14 @@ hide:
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>MiniMax H3</strong><small>Video + stereo audio</small></span>
|
||||
<span class="cookbook-count">6 recipes</span>
|
||||
<span class="cookbook-count">9 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2VA</li>
|
||||
<li>FL2VA</li>
|
||||
<li>Ref2VA</li>
|
||||
<li>MLX T2VA</li>
|
||||
<li>DGX Spark</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
@@ -454,16 +455,6 @@ hide:
|
||||
</article>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="cookbook-roadmap" aria-labelledby="roadmap-heading">
|
||||
<h2 id="roadmap-heading">Inference first, then the full workflow</h2>
|
||||
<p>
|
||||
Inference is the first complete stage. Distillation, fine-tuning,
|
||||
training, evaluation, optimization, and deployment will reuse the same
|
||||
family-first structure as their recipes land. Each family page shows
|
||||
which stages are available and which are planned.
|
||||
</p>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<small class="cookbook-logo-credit">
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Kandinsky 5 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Alternative Lite/Pro checkpoints are listed inline in the maintained examples; swap the model string only after checking its Hugging Face card.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Alternative Lite/Pro checkpoints are listed inline in the maintained examples; swap the model string only after checking its Hugging Face card.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
Both recipes map to checked-in FastVideo examples and recorded single-GPU B200 runs. The image-to-video run also records 10,365.89 MB peak GPU memory. These measurements describe the recorded runs; they are not minimum hardware requirements.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Both recipes map to checked-in FastVideo examples and recorded single-GPU B200 runs. The image-to-video run also records 10,365.89 MB peak GPU memory. These measurements describe the recorded runs; they are not minimum hardware requirements.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# LongCat recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="longcat" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="longcat" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Each LongCat script runs multiple passes (basic, distilled, refine); total runtime scales accordingly, and every pass prints its own output directory.</li>
|
||||
<li>Out of memory: the sources already enable VAE and text-encoder CPU offload; further options are covered in <a href="../../inference/offloading/">Offloading</a>.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Each LongCat script runs multiple passes (basic, distilled, refine); total runtime scales accordingly, and every pass prints its own output directory.
|
||||
- Out of memory: the sources already enable VAE and text-encoder CPU offload; further options are covered in [Offloading](../inference/offloading.md).
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+32
-20
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# LTX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,24 +112,30 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The distilled recipe is source-configured for four GPUs; running it on fewer GPUs is unverified and may fail during distributed setup.</li>
|
||||
<li>Audio-less output usually means the base checkpoint resolved instead of the distilled LTX-2 checkpoint with audio; check the loaded model ID in the logs.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The distilled recipe is source-configured for four GPUs; running it on fewer GPUs is unverified and may fail during distributed setup.
|
||||
- Audio-less output usually means the base checkpoint resolved instead of the distilled LTX-2 checkpoint with audio; check the loaded model ID in the logs.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Matrix Game recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Matrix Game 3.0 downloads its reference input image from GitHub; offline machines should pre-download it and edit the <code>IMAGE_URL</code> constant locally.</li>
|
||||
<li>Streaming variants of Matrix Game 2.0 exist under <code>examples/inference/basic/basic_matrixgame2_streaming.py</code> but are not included as cookbook recipes.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Matrix Game 3.0 downloads its reference input image from GitHub; offline machines should pre-download it and edit the `IMAGE_URL` constant locally.
|
||||
- Streaming variants of Matrix Game 2.0 exist under `examples/inference/basic/basic_matrixgame2_streaming.py` but are not included as cookbook recipes.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+76
-39
@@ -5,7 +5,13 @@ hide:
|
||||
|
||||
# MiniMax H3 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
FastH3 is two distilled MiniMax-H3 checkpoints. **V1** is the four-step
|
||||
launch. Some Hub repo names still say Preview. That name is historical. V1 is
|
||||
a full model, not a demo. **V2** is the eight-step checkpoint. More forwards
|
||||
is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
[FastH3 distilled checkpoint schedules](../inference/fasth3-distilled.md).
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -15,8 +21,8 @@ hide:
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Primary focus · Inference</p>
|
||||
<h2>MiniMax H3 recipes</h2>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>6 maintained recipes</span>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>9 maintained recipes</span>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
@@ -29,14 +35,21 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<details class="cookbook-modes">
|
||||
<summary>Compare H3 modes and options</summary>
|
||||
<h2 id="h3-modes-heading">Supported modes</h2>
|
||||
<p>
|
||||
CUDA covers T2VA, FL2VA, and Ref2VA on the full checkpoint, plus FastH3
|
||||
Preview and FastH3 LoRA. MLX is T2VA only. Temporal <code>--fast</code>,
|
||||
spatial <code>--fast-spatial</code>, and opt-in VSA are flags on the same
|
||||
V1 and FastH3 V2. FastH3 V1 also has a DGX Spark runtime with
|
||||
a 1-Spark or 2-Spark device row. MLX is T2VA only: V1 and V2.
|
||||
Temporal <code>--fast</code>, spatial <code>--fast-spatial</code>, and opt-in VSA are flags on the same
|
||||
MLX script, not extra recipes.
|
||||
</p>
|
||||
<div class="cookbook-modes__table-wrap">
|
||||
@@ -51,8 +64,8 @@ hide:
|
||||
<tbody>
|
||||
<tr>
|
||||
<td>T2VA</td>
|
||||
<td>Full H3, FastH3 Preview, FastH3 LoRA</td>
|
||||
<td>FastH3 Preview after a local DiT conversion</td>
|
||||
<td>Full H3, FastH3 V1, FastH3 LoRA, FastH3 V2</td>
|
||||
<td>FastH3 V1 or FastH3 V2 after a local DiT conversion</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>FL2VA</td>
|
||||
@@ -84,6 +97,11 @@ hide:
|
||||
<td>No cookbook recipe</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>DGX Spark</td>
|
||||
<td>FastH3 V1 on one GB10, or two Sparks with Ray sequence parallel (<code>sp_size=2</code>) over QSFP RoCE. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks.</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
@@ -92,7 +110,7 @@ hide:
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick an H3 recipe and runtime</h2>
|
||||
<p>Choose the result you want, then use a maintained CUDA or MLX path.
|
||||
<p>Choose the result you want, then use a maintained CUDA, DGX Spark, or MLX path.
|
||||
Device claims stay tied to checked-in sources and recorded runs.</p>
|
||||
</div>
|
||||
|
||||
@@ -118,6 +136,17 @@ hide:
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row" data-cookbook-device-row hidden>
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Devices</strong>
|
||||
<span data-cookbook-device-caption>1 Spark or a QSFP pair</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-device-options role="group" aria-label="Devices">
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div data-cookbook-knobs></div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<div class="cookbook-selection-row" data-cookbook-usage>
|
||||
<div class="cookbook-selection-row__label">
|
||||
@@ -217,36 +246,44 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
<p class="cookbook-eyebrow">CUDA</p>
|
||||
<pre><code>UV_TORCH_BACKEND=cu130 uv pip install -e ".[fasth3]"</code></pre>
|
||||
<p class="cookbook-eyebrow">Apple Silicon</p>
|
||||
<pre><code>uv pip install -e ".[mlx]"</code></pre>
|
||||
<p>Follow the <a href="../../getting_started/installation/mlx/">MLX install guide</a> for the extra, <code>ffmpeg</code>, and a clone. Then pick FastH3 V1 or V2 in the builder above.</p>
|
||||
<p class="cookbook-eyebrow">NVIDIA DGX Spark</p>
|
||||
<pre><code>UV_TORCH_BACKEND=cu130 uv pip install -e .</code></pre>
|
||||
<p>Follow the <a href="../../getting_started/installation/spark/">DGX Spark install guide</a> for ARM64 CUDA 13. One Spark is a local process. Two Sparks need Ray on the QSFP link:</p>
|
||||
<pre><code>uv pip install ray</code></pre>
|
||||
<p>Bring up the cluster from <a href="../../getting_started/installation/spark_pair/">pairing two Sparks</a> before selecting 2 Sparks in the builder.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The full CUDA H3 examples request four GPUs by default. Their sources do not claim a GPU model or memory minimum.</li>
|
||||
<li>The FastH3 CUDA performance profile was measured on four GB200 GPUs. Use its strict profile when exact operation order matters more than the measured performance configuration.</li>
|
||||
<li>The MLX source runtime supports T2VA, optional temporal <code>--fast</code>, optional spatial <code>--fast-spatial</code>, and opt-in VSA on <code>--include-vsa</code> checkpoints. FastH3 V2 MLX converts with <code>--include-vsa</code> and runs eight forwards. FL2VA, Ref2VA, and two-pass refinement are not wired.</li>
|
||||
<li>GPU count and VAE decode backend are configurable in the builder above for FastH3 CUDA recipes. Only the value shown by default has a recorded run; other supported values are unmeasured here.</li>
|
||||
<li>DGX Spark is a runtime on FastH3 V1, not a separate family card. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks. The CUDA GPU-count knob does not apply to Spark.</li>
|
||||
<li>GB10 has no FA4 / sm_100a VSA kernel. Keep <code>FASTVIDEO_FA4=0</code> and <code>FASTVIDEO_VSA_SM100A=0</code>. Legal <code>num_frames</code> values are <code>17n+5</code>, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
CUDA FastH3 uses the pinned performance dependencies:
|
||||
|
||||
UV_TORCH_BACKEND=cu130 uv pip install -e ".[fasth3]"
|
||||
|
||||
Apple Silicon uses the native MLX extra and a locally converted H3 DiT:
|
||||
|
||||
uv pip install -e ".[mlx]"
|
||||
|
||||
Follow the [Apple Silicon guide](../getting_started/installation/mps.md#run-fasth3-preview)
|
||||
for the download, conversion, and storage requirements.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The full CUDA H3 examples request four GPUs by default. Their sources do not claim a GPU model or memory minimum.
|
||||
- The FastH3 CUDA performance profile was measured on four GB200 GPUs. Use its strict profile when exact operation order matters more than the measured performance configuration.
|
||||
- The MLX source runtime supports T2VA, optional temporal `--fast`, optional spatial `--fast-spatial`, and opt-in VSA on `--include-vsa` checkpoints. FL2VA, Ref2VA, and two-pass refinement are not wired.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
Every command, model ID, and flag on this page maps to a checked-in FastVideo source. Recipes marked **Verified** also have a recorded hardware path in linked FastVideo evidence. The full H3 CUDA examples remain **Source-backed** where the source records a GPU count but no GPU model or memory requirement. Unlisted hardware is unknown, not unsupported.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Every command, model ID, and flag on this page maps to a checked-in FastVideo source. Recipes marked <strong>Verified</strong> also have a recorded hardware path in linked FastVideo evidence. The full H3 CUDA examples remain <strong>Source-backed</strong> where the source records a GPU count but no GPU model or memory requirement. Unlisted hardware is unknown, not unsupported.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# MMAudio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>If the example reports a missing local <code>converted_weights/mmaudio/large_44k_v2</code> path, the <code>MMAUDIO_MODEL_PATH</code> env var from the cookbook command was not applied; export it in the same shell.</li>
|
||||
<li>Alternatively convert upstream weights yourself with <code>scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py</code> and point the env var at the result.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- If the example reports a missing local `converted_weights/mmaudio/large_44k_v2` path, the `MMAUDIO_MODEL_PATH` env var from the cookbook command was not applied; export it in the same shell.
|
||||
- Alternatively convert upstream weights yourself with `scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py` and point the env var at the result.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Run H3 with a server and playground
|
||||
|
||||
Start FastH3 on CUDA or Apple Silicon MLX, then iterate on prompts in a browser,
|
||||
Start FastH3 on CUDA, one DGX Spark, or Apple Silicon MLX, then iterate on prompts in a browser,
|
||||
with cURL, or from your app. The server and clients can run on the same machine.
|
||||
|
||||
The server supports the OpenAI-compatible video-job API. The Python and
|
||||
@@ -8,8 +8,9 @@ JavaScript client examples use that interface, but requests go to FastVideo.
|
||||
You do not need an OpenAI account or cloud key.
|
||||
|
||||
The [H3 recipe selector](minimax-h3.md) provides the same workflow with runtime
|
||||
selection. This guide covers FastH3 Preview text-to-video/audio. Other H3
|
||||
recipes keep their direct Python commands.
|
||||
selection. This guide covers FastH3 V1 (four forwards) and FastH3 V2
|
||||
(eight forwards). Start a server, then iterate in the playground or with the
|
||||
OpenAI Python client. Other H3 recipes keep their direct Python commands.
|
||||
|
||||
CUDA requests reuse one loaded `VideoGenerator`. The Python SDK can do the same
|
||||
when you reuse the generator across `generate()` calls. MLX keeps one
|
||||
@@ -36,6 +37,40 @@ advertises it as `fasth3`. It configures four CUDA GPUs but does not record
|
||||
a GPU model or VRAM requirement. This is a source-backed server profile, not
|
||||
the measured GB200 Python performance profile. Compilation is disabled.
|
||||
|
||||
For FastH3 V2, use the same install and the 8-step config (nine sigma
|
||||
points, eight DiT forwards, VSA sparsity 0.8):
|
||||
|
||||
```bash
|
||||
UV_TORCH_BACKEND=cu130 uv pip install -e ".[fasth3]"
|
||||
fastvideo serve --config examples/serving/openai_fasth3_8step.yaml --server.host 127.0.0.1
|
||||
```
|
||||
|
||||
Keep the server running. In another terminal, check readiness:
|
||||
|
||||
```bash
|
||||
curl --fail-with-body http://127.0.0.1:8000/health
|
||||
```
|
||||
|
||||
After model loading completes, the response is `{"status":"ok"}`.
|
||||
|
||||
### NVIDIA DGX Spark
|
||||
|
||||
Complete the [DGX Spark installation](../getting_started/installation/spark.md)
|
||||
before running these commands. GB10 has no FA4 / sm_100a VSA kernel:
|
||||
|
||||
```bash
|
||||
UV_TORCH_BACKEND=cu130 uv pip install -e .
|
||||
FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 \
|
||||
fastvideo serve --config examples/serving/openai_fasth3_spark.yaml --server.host 127.0.0.1
|
||||
```
|
||||
|
||||
The configuration loads `FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree`
|
||||
on one GB10 and advertises it as `fasth3`. Lazy module load still reloads
|
||||
Qwen3-VL and the DiT between phases of each request. Legal `num_frames` values
|
||||
are `17n+5`, capped at 362 (15.08 s); a 345-frame request on one Spark can OOM.
|
||||
There is no cookbook server for two Sparks; use the generate YAML after
|
||||
[pairing two Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
Keep the server running. In another terminal, check readiness:
|
||||
|
||||
```bash
|
||||
@@ -46,7 +81,7 @@ After model loading completes, the response is `{"status":"ok"}`.
|
||||
|
||||
### Apple Silicon MLX
|
||||
|
||||
Complete the [Apple Silicon installation](../getting_started/installation/mps.md#run-fasth3-preview),
|
||||
Complete the [MLX install](../getting_started/installation/mlx.md),
|
||||
including `ffmpeg`. From your FastVideo clone, install the MLX extra:
|
||||
|
||||
```bash
|
||||
@@ -77,14 +112,30 @@ LoRA selection, and alternate decoders are not exposed by this server adapter.
|
||||
Unsupported request options return HTTP 400 before a job starts.
|
||||
|
||||
The server binds to `127.0.0.1:8000` and advertises `fasth3`, so the playground
|
||||
and clients below work unchanged. Do not run CUDA and MLX servers on the same
|
||||
port. MLX readiness means the pipeline is initialized; components load during
|
||||
and clients below work unchanged. Do not run CUDA, Spark, and MLX servers on the
|
||||
same port. MLX readiness means the pipeline is initialized; components load during
|
||||
generation. The first request can take longer than a repeated cached prompt.
|
||||
|
||||
The MLX server has no recorded device or unified-memory requirement. The
|
||||
direct Python recipe's M4 Max measurements are not a server benchmark or a
|
||||
minimum-memory claim.
|
||||
|
||||
For FastH3 V2, convert that checkpoint's transformer with `--include-vsa`
|
||||
and start the 8-step config. Do not overwrite a V1 export. The
|
||||
[MiniMax H3 cookbook](minimax-h3.md) has the same download and convert
|
||||
commands as the Python recipe.
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats "int8" --include-vsa
|
||||
python -m fastvideo.entrypoints.openai.mlx_server --config examples/serving/mlx_fasth3_8step.yaml
|
||||
```
|
||||
|
||||
That adapter passes `num_steps=8` when the HTTP field is `num_inference_steps=9`,
|
||||
and it enables the trained VSA recipe (sparsity 0.8, 64-token tiles). Reuse the
|
||||
preview VAE, audio VAE, text encoder, and tokenizer if those directories already
|
||||
exist; edit the YAML paths if your files live elsewhere.
|
||||
|
||||
## Open the playground
|
||||
|
||||
Open [the local H3 playground](http://127.0.0.1:8000/playground/) after startup.
|
||||
@@ -110,9 +161,11 @@ or manage a GPU server for you.
|
||||
## Generate with cURL or an SDK
|
||||
|
||||
These examples use the server's resolution, frame count, and sampling defaults.
|
||||
Do not copy Sora-specific durations or resolutions onto H3. Both server configs
|
||||
use 124 frames, 24 fps, and the five-point distilled sigma schedule with four
|
||||
DiT forwards. CUDA uses 1344 × 768; MLX uses 832 × 480.
|
||||
Do not copy Sora-specific durations or resolutions onto H3. V1 configs use
|
||||
124 frames, 24 fps, and the five-point distilled sigma schedule with four DiT
|
||||
forwards. V2 configs use nine sigma points and eight DiT forwards. CUDA
|
||||
and one Spark use 1344 × 768; MLX uses 832 × 480. The OpenAI Python client is
|
||||
the same for every FastH3 server that advertises `fasth3`.
|
||||
|
||||
Each client submits a job, checks for completion or failure, and downloads an
|
||||
MP4 named after the job ID. Polling stops after 30 minutes; a timeout does not
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Audio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Loader errors about monolithic checkpoints mean an upstream <code>stabilityai/stable-audio-open-*</code> ID was used; use the FastVideo converted repos from the recipes.</li>
|
||||
<li>Duration and step knobs (<code>audio_end_in_s</code>, <code>num_inference_steps</code>) are documented inline in the example source.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Loader errors about monolithic checkpoints mean an upstream `stabilityai/stable-audio-open-*` ID was used; use the FastVideo converted repos from the recipes.
|
||||
- Duration and step knobs (`audio_end_in_s`, `num_inference_steps`) are documented inline in the example source.
|
||||
|
||||
## Evidence status
|
||||
|
||||
The Stable Audio Open 1.0 recipe maps to a checked-in example and a recorded single-GPU B200 run. The Stable Audio Open Small recipe remains **Source-backed** because the implementation PR did not record a full run for that gated checkpoint. Neither recipe claims a minimum VRAM requirement.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The Stable Audio Open 1.0 recipe maps to a checked-in example and a recorded single-GPU B200 run. The Stable Audio Open Small recipe remains <strong>Source-backed</strong> because the implementation PR did not record a full run for that gated checkpoint. Neither recipe claims a minimum VRAM requirement.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Diffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The example writes several PNGs under <code>outputs/sd35/samples/</code>; make sure the output directory is writable.</li>
|
||||
<li>Gated checkpoints: Stability AI models require accepting the license and <code>huggingface-cli login</code>.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The example writes several PNGs under `outputs/sd35/samples/`; make sure the output directory is writable.
|
||||
- Gated checkpoints: Stability AI models require accepting the license and `huggingface-cli login`.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# TurboDiffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="turbodiffusion" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="turbodiffusion" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>TurboDiffusion paths load community-published <code>loayrashid/TurboWan*</code> checkpoints; availability is governed by those repos.</li>
|
||||
<li>The SLA attention backend used by the I2V recipe is selected inside the example source; do not combine it with another <code>FASTVIDEO_ATTENTION_BACKEND</code> override in the same shell.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- TurboDiffusion paths load community-published `loayrashid/TurboWan*` checkpoints; availability is governed by those repos.
|
||||
- The SLA attention backend used by the I2V recipe is selected inside the example source; do not combine it with another `FASTVIDEO_ATTENTION_BACKEND` override in the same shell.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+37
-24
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Wan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -29,8 +29,15 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-modes" aria-labelledby="wan-modes-heading">
|
||||
<details class="cookbook-modes">
|
||||
<summary>Compare Wan modes and options</summary>
|
||||
<h2 id="wan-modes-heading">Supported modes</h2>
|
||||
<p>
|
||||
FastMetal MLX is T2V in the checked-in examples. Image-to-video and
|
||||
@@ -84,7 +91,7 @@ hide:
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</section>
|
||||
</details>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -163,26 +170,32 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Out of memory on the A14B recipes: the checked-in sources already enable CPU offload; see <a href="../../inference/configuration/">Configuration</a> for the offload surface before reducing resolution or frames.</li>
|
||||
<li>The FastWan2.1 recipe requires <code>VIDEO_SPARSE_ATTN</code>; confirm the environment variable in the command was set in the same shell.</li>
|
||||
<li>FastMetal MLX: install with the <a href="../../getting_started/installation/mlx/">MLX install guide</a>, then pick a FastMetal recipe in the builder. CUDA FastWan-QAD checkpoints are refused on the MLX runtime.</li>
|
||||
<li>FastMetal 5B uses <code>mlx_wan22_generate.py</code>. 1.3B and 14B use <code>mlx_wan_prompt_to_video.py</code>.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Out of memory on the A14B recipes: the checked-in sources already enable CPU offload; see [Configuration](../inference/configuration.md) for the offload surface before reducing resolution or frames.
|
||||
- The FastWan2.1 recipe requires `VIDEO_SPARSE_ATTN`; confirm the environment variable in the command was set in the same shell.
|
||||
- FastMetal MLX: install with `uv pip install -e ".[mlx]"`, then follow the [Apple Silicon guide](../getting_started/installation/mps.md). CUDA FastWan-QAD checkpoints are refused on the MLX runtime.
|
||||
- FastMetal 5B uses `mlx_wan22_generate.py`. 1.3B and 14B use `mlx_wan_prompt_to_video.py`.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
Every recipe on this page maps to a checked-in FastVideo source. The FastMetal MLX releases include the recorded M4 Max system memory, documented unified-memory floor, and measured peak MLX memory. CUDA entries remain **Source-backed** where the examples record a GPU count but no exact GPU model or VRAM. Unlisted hardware is unknown, not unsupported.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Every recipe on this page maps to a checked-in FastVideo source. The FastMetal MLX releases include the recorded M4 Max system memory, documented unified-memory floor, and measured peak MLX memory. CUDA entries remain <strong>Source-backed</strong> where the examples record a GPU count but no exact GPU model or VRAM. Unlisted hardware is unknown, not unsupported.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Z-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="zimage" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="zimage" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Output defaults to <code>outputs/zimage/zimage_turbo.png</code>; pass <code>--output</code> to redirect.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Output defaults to `outputs/zimage/zimage_turbo.png`; pass `--output` to redirect.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -2,132 +2,158 @@ status_definitions:
|
||||
kept: "Public field remains on a public adapter surface with the same meaning."
|
||||
moved: "Public field remains supported but normalizes into a different nested path."
|
||||
preset_owned: "Public field remains supported only through a model/preset-specific surface."
|
||||
compatibility_only: "Legacy public field remains adapter-only during migration and is not part of the canonical typed schema."
|
||||
compatibility_only: "Public field that an adapter or an open mapping accepts outside the typed fields of the canonical schema."
|
||||
private_only: "Field should only be handled by private adapters and is not a public FastVideo compatibility promise."
|
||||
internal_only: "Field is runtime/config plumbing and should not be part of the new public typed inference API."
|
||||
internal_only: "Field is runtime/config plumbing that model code, config resolution, or the runtime fills; it is not a public input."
|
||||
unsupported: "Typed config field that no runtime code reads; resolution rejects a value."
|
||||
|
||||
surfaces:
|
||||
fastvideo_args:
|
||||
generator_config:
|
||||
kept:
|
||||
- model_path
|
||||
- mode
|
||||
- revision
|
||||
- trust_remote_code
|
||||
- engine.num_gpus
|
||||
- engine.execution_backend
|
||||
- engine.parallelism.tp_size
|
||||
- engine.parallelism.sp_size
|
||||
- engine.parallelism.hsdp_replicate_dim
|
||||
- engine.parallelism.hsdp_shard_dim
|
||||
- engine.parallelism.dist_timeout
|
||||
- engine.offload.dit
|
||||
- engine.offload.dit_layerwise
|
||||
- engine.offload.text_encoder
|
||||
- engine.offload.image_encoder
|
||||
- engine.offload.vae
|
||||
- engine.offload.pin_cpu_memory
|
||||
- engine.compile.enabled
|
||||
- engine.compile.backend
|
||||
- engine.compile.fullgraph
|
||||
- engine.compile.mode
|
||||
- engine.compile.dynamic
|
||||
- engine.compile.extras
|
||||
- engine.enable_stage_verification
|
||||
- engine.use_fsdp_inference
|
||||
- engine.disable_autocast
|
||||
- engine.attention.nvfp4_fa4
|
||||
- pipeline.components.lora_path
|
||||
- pipeline.components.lora_strength
|
||||
- pipeline.output_type
|
||||
- engine.parallelism.master_port
|
||||
- engine.offload.lazy_module_load
|
||||
- engine.compile.text_encoder_enabled
|
||||
- engine.compile.vae_enabled
|
||||
- engine.compile.audio_vae_enabled
|
||||
- engine.compile.regional
|
||||
- engine.compile.dit_kwargs
|
||||
- engine.compile.text_encoder_kwargs
|
||||
- engine.compile.vae_kwargs
|
||||
- engine.compile.audio_vae_kwargs
|
||||
- engine.attention.backend
|
||||
- engine.attention.vsa_sparsity
|
||||
- engine.attention.vsa_tile_size
|
||||
- engine.attention.moba_config_path
|
||||
- engine.precision.dit
|
||||
- engine.precision.vae
|
||||
- engine.precision.vae_decode
|
||||
- engine.precision.image_encoder
|
||||
- engine.precision.text_encoders
|
||||
- engine.quantization.text_encoder_quant
|
||||
- engine.quantization.transformer_quant
|
||||
- pipeline.workload_type
|
||||
- pipeline.components.config_root
|
||||
- pipeline.components.pipeline_config_path
|
||||
- pipeline.components.text_encoder_weights
|
||||
- pipeline.components.transformer_weights
|
||||
- pipeline.components.transformer_2_weights
|
||||
- pipeline.components.upsampler_weights
|
||||
- pipeline.components.lora_nickname
|
||||
- pipeline.components.lora_target_modules
|
||||
- pipeline.components.override_pipeline_cls_name
|
||||
- pipeline.components.override_transformer_cls_name
|
||||
- pipeline.vae_tiling
|
||||
- pipeline.vae_sp
|
||||
- pipeline.flow_shift
|
||||
- pipeline.embedded_cfg_scale
|
||||
- pipeline.dmd_denoising_steps
|
||||
- pipeline.boundary_ratio
|
||||
- pipeline.model.generic.dit
|
||||
- pipeline.model.generic.vae
|
||||
- pipeline.model.ltx2.dit
|
||||
- pipeline.model.ltx2.vae
|
||||
- pipeline.model.minimax_h3.dit
|
||||
- pipeline.model.minimax_h3.vae
|
||||
- pipeline.model.longcat.dit
|
||||
- pipeline.model.longcat.vae
|
||||
unsupported:
|
||||
- pipeline.preset
|
||||
- pipeline.preset_version
|
||||
- pipeline.components.vae_weights
|
||||
moved:
|
||||
model_path: generator.model_path
|
||||
workload_type: generator.pipeline.workload_type
|
||||
distributed_executor_backend: generator.engine.execution_backend
|
||||
trust_remote_code: generator.trust_remote_code
|
||||
revision: generator.revision
|
||||
num_gpus: generator.engine.num_gpus
|
||||
tp_size: generator.engine.parallelism.tp_size
|
||||
sp_size: generator.engine.parallelism.sp_size
|
||||
hsdp_replicate_dim: generator.engine.parallelism.hsdp_replicate_dim
|
||||
hsdp_shard_dim: generator.engine.parallelism.hsdp_shard_dim
|
||||
dist_timeout: generator.engine.parallelism.dist_timeout
|
||||
lora_path: generator.pipeline.components.lora_path
|
||||
lora_nickname: generator.pipeline.components.lora_nickname
|
||||
lora_strength: generator.pipeline.components.lora_strength
|
||||
dit_cpu_offload: generator.engine.offload.dit
|
||||
use_fsdp_inference: generator.engine.use_fsdp_inference
|
||||
dit_layerwise_offload: generator.engine.offload.dit_layerwise
|
||||
text_encoder_cpu_offload: generator.engine.offload.text_encoder
|
||||
image_encoder_cpu_offload: generator.engine.offload.image_encoder
|
||||
vae_cpu_offload: generator.engine.offload.vae
|
||||
pin_cpu_memory: generator.engine.offload.pin_cpu_memory
|
||||
enable_torch_compile: generator.engine.compile.enabled
|
||||
enable_torch_compile_text_encoder: generator.engine.compile.text_encoder_enabled
|
||||
enable_torch_compile_vae: generator.engine.compile.vae_enabled
|
||||
enable_torch_compile_audio_vae: generator.engine.compile.audio_vae_enabled
|
||||
torch_compile_kwargs: generator.engine.compile.backend,fullgraph,mode,dynamic,extras
|
||||
torch_compile_kwargs_dit: generator.engine.compile.dit_kwargs
|
||||
torch_compile_kwargs_text_encoder: generator.engine.compile.text_encoder_kwargs
|
||||
torch_compile_kwargs_vae: generator.engine.compile.vae_kwargs
|
||||
torch_compile_kwargs_audio_vae: generator.engine.compile.audio_vae_kwargs
|
||||
transformer_quant: generator.engine.quantization.transformer_quant
|
||||
disable_autocast: generator.engine.disable_autocast
|
||||
enable_stage_verification: generator.engine.enable_stage_verification
|
||||
prompt_txt: request.inputs.prompt_path
|
||||
override_text_encoder_safetensors: generator.pipeline.components.text_encoder_weights
|
||||
override_text_encoder_quant: generator.engine.quantization.text_encoder_quant
|
||||
transformer_quant: generator.engine.quantization.transformer_quant
|
||||
override_transformer_cls_name: generator.pipeline.components.override_transformer_cls_name
|
||||
init_weights_from_safetensors: generator.pipeline.components.transformer_weights
|
||||
init_weights_from_safetensors_2: generator.pipeline.components.transformer_2_weights
|
||||
override_pipeline_cls_name: generator.pipeline.components.override_pipeline_cls_name
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
ltx2_vae_tiling: generator.pipeline.vae_tiling
|
||||
refine_enabled: generator.pipeline.preset_overrides.refine.enabled
|
||||
refine_upsampler_path: generator.pipeline.components.upsampler_weights
|
||||
refine_lora_path: generator.pipeline.components.lora_path
|
||||
refine_num_inference_steps: request.stage_overrides.refine.num_inference_steps
|
||||
refine_guidance_scale: request.stage_overrides.refine.guidance_scale
|
||||
refine_add_noise: generator.pipeline.preset_overrides.refine.add_noise
|
||||
ltx2_refine_enabled: generator.pipeline.preset_overrides.refine.enabled
|
||||
ltx2_refine_upsampler_path: generator.pipeline.components.upsampler_weights
|
||||
ltx2_refine_lora_path: generator.pipeline.components.lora_path
|
||||
ltx2_refine_num_inference_steps: request.stage_overrides.refine.num_inference_steps
|
||||
ltx2_refine_guidance_scale: request.stage_overrides.refine.guidance_scale
|
||||
ltx2_refine_add_noise: generator.pipeline.preset_overrides.refine.add_noise
|
||||
pipeline.preset_overrides:
|
||||
target: generator.pipeline.model.ltx2.refine
|
||||
note: "Only the refine mapping applies: resolution copies pipeline.preset_overrides.refine into the pipeline.model.ltx2.refine fields of the same names when the model is LTX-2. Other keys have no effect."
|
||||
preset_owned:
|
||||
ltx2_vae_spatial_tile_size_in_pixels: generator.pipeline.preset_overrides.ltx2.vae.spatial_tile_size_in_pixels
|
||||
ltx2_vae_spatial_tile_overlap_in_pixels: generator.pipeline.preset_overrides.ltx2.vae.spatial_tile_overlap_in_pixels
|
||||
ltx2_vae_temporal_tile_size_in_frames: generator.pipeline.preset_overrides.ltx2.vae.temporal_tile_size_in_frames
|
||||
ltx2_vae_temporal_tile_overlap_in_frames: generator.pipeline.preset_overrides.ltx2.vae.temporal_tile_overlap_in_frames
|
||||
ltx2_initial_latent_path: request.extensions.ltx2.initial_latent_path
|
||||
ltx2_audio_latent_path: request.extensions.ltx2.audio_latent_path
|
||||
- pipeline.model.ltx2.vae_spatial_tile_size_in_pixels
|
||||
- pipeline.model.ltx2.vae_spatial_tile_overlap_in_pixels
|
||||
- pipeline.model.ltx2.vae_temporal_tile_size_in_frames
|
||||
- pipeline.model.ltx2.vae_temporal_tile_overlap_in_frames
|
||||
- pipeline.model.ltx2.initial_latent_path
|
||||
- pipeline.model.ltx2.audio_latent_path
|
||||
- pipeline.model.ltx2.legacy_native_noise_order
|
||||
- pipeline.model.ltx2.use_distilled_sigmas
|
||||
- pipeline.model.ltx2.refine.enabled
|
||||
- pipeline.model.ltx2.refine.num_inference_steps
|
||||
- pipeline.model.ltx2.refine.guidance_scale
|
||||
- pipeline.model.ltx2.refine.add_noise
|
||||
- pipeline.model.ltx2.refine.image_crf
|
||||
- pipeline.model.ltx2.refine.video_position_offset_sec
|
||||
- pipeline.model.ltx2.refine.transformer_path
|
||||
- pipeline.model.ltx2.refine.lora_path
|
||||
- pipeline.model.ltx2.refine.noise_path
|
||||
- pipeline.model.ltx2.refine.audio_noise_path
|
||||
- pipeline.model.minimax_h3.sequential_load
|
||||
- pipeline.model.minimax_h3.video_decode_backend
|
||||
- pipeline.model.minimax_h3.taeh3_checkpoint
|
||||
- pipeline.model.minimax_h3.taeh3_chunk_size
|
||||
- pipeline.model.minimax_h3.vae_parallel_decode
|
||||
- pipeline.model.minimax_h3.vae_parallel_encode
|
||||
- pipeline.model.minimax_h3.vae_parallel_decode_strategy
|
||||
- pipeline.model.longcat.enable_bsa
|
||||
- pipeline.model.longcat.bsa_sparsity
|
||||
- pipeline.model.longcat.bsa_cdf_threshold
|
||||
- pipeline.model.longcat.bsa_chunk_q
|
||||
- pipeline.model.longcat.bsa_chunk_k
|
||||
compatibility_only:
|
||||
mode: "Legacy multi-mode FastVideoArgs switch; typed inference config should not expose execution mode."
|
||||
inference_mode: "Legacy boolean mirror of mode; kept only through adapters while FastVideoArgs remains."
|
||||
lora_target_modules: "Legacy LoRA configuration surface pending dedicated component API."
|
||||
output_type: "Legacy output formatting surface pending GenerationResult cleanup."
|
||||
VSA_sparsity: "Model-specific inference optimization not yet represented in the typed public schema."
|
||||
VSA_tile_size: "VSA-H3 tile geometry request; model-specific optimization not yet represented in the typed public schema."
|
||||
inference_torch_compile: "Regional inference compile opt-in currently carried through PipelineSelection.experimental rather than CompileConfig."
|
||||
vae_parallel_decode: "MiniMax-H3 sequence-parallel VAE decode opt-in; model-specific optimization not yet represented in the typed public schema."
|
||||
h3_sequential_load: "MiniMax-H3 sequential text-encoder then DiT/VAE load; model-specific optimization not yet represented in the typed public schema."
|
||||
vae_parallel_encode: "MiniMax-H3 sequence-parallel reference VAE encode opt-in; model-specific optimization not yet represented in the typed public schema."
|
||||
vae_parallel_decode_strategy: "Chunk-transport collective for vae_parallel_decode; model-specific optimization not yet represented in the typed public schema."
|
||||
attention_backend: "Process-wide default attention-backend request applied per component at load time; kernel-selection knob not yet represented in the typed public schema."
|
||||
moba_config_path: "Model-specific MoBA optimization surface not yet represented in the typed public schema."
|
||||
master_port: "Executor/bootstrap compatibility field; not part of the canonical inference schema."
|
||||
refine_transformer_path: "Generic stage-2 refine transformer override; no typed equivalent yet."
|
||||
refine_noise_path: "Generic stage-2 refine noise override; no typed equivalent yet."
|
||||
refine_audio_noise_path: "Generic stage-2 refine audio noise override; no typed equivalent yet."
|
||||
ltx2_refine_transformer_path: "LTX-2 refine transformer carrier; no typed equivalent yet."
|
||||
ltx2_refine_noise_path: "LTX-2 refine noise carrier; no typed equivalent yet."
|
||||
ltx2_refine_audio_noise_path: "LTX-2 refine audio noise carrier; no typed equivalent yet."
|
||||
ltx2_legacy_native_noise_order: "LTX-2 SSIM compatibility knob preserving legacy native latent noise ordering."
|
||||
ltx2_use_distilled_sigmas: "LTX-2 compatibility knob gating use of distilled sigma schedule."
|
||||
private_only:
|
||||
ray_placement_group: "Ray deployment-only field."
|
||||
ray_runtime_env: "Ray deployment-only field."
|
||||
pipeline.experimental: "Open mapping for settings without a typed path: the pipeline_config source (a JSON path, a mapping, or a PipelineConfig), keys that runtime code reads by name (for example ray_runtime_env), and PipelineConfig attribute overrides for model-only fields (for example flow_shift_sr). Resolution rejects a key whose PipelineConfig attribute has a typed path."
|
||||
internal_only:
|
||||
pipeline_config: "Legacy internal carrier object."
|
||||
preprocess_config: "Legacy preprocess carrier object."
|
||||
moba_config: "Derived runtime config loaded from moba_config_path."
|
||||
model_paths: "Runtime bookkeeping."
|
||||
model_loaded: "Runtime bookkeeping."
|
||||
engine.attention.moba_config: "V-MoBA attention settings that resolution loads from engine.attention.moba_config_path."
|
||||
|
||||
pipeline_config_base:
|
||||
moved:
|
||||
model_path: generator.model_path
|
||||
pipeline_config_path: generator.pipeline.components.pipeline_config_path
|
||||
embedded_cfg_scale: generator.pipeline.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.flow_shift
|
||||
disable_autocast: generator.engine.disable_autocast
|
||||
vae_tiling: generator.pipeline.vae_tiling
|
||||
vae_sp: generator.pipeline.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.dmd_denoising_steps
|
||||
boundary_ratio: generator.pipeline.boundary_ratio
|
||||
dit_precision: generator.engine.precision.dit
|
||||
vae_precision: generator.engine.precision.vae
|
||||
vae_decode_precision: generator.engine.precision.vae_decode
|
||||
image_encoder_precision: generator.engine.precision.image_encoder
|
||||
text_encoder_precisions: generator.engine.precision.text_encoders
|
||||
preset_owned:
|
||||
embedded_cfg_scale: generator.pipeline.preset_overrides.embedded_cfg_scale
|
||||
flow_shift: generator.pipeline.preset_overrides.flow_shift
|
||||
flow_shift_sr: generator.pipeline.preset_overrides.flow_shift_sr
|
||||
is_causal: generator.pipeline.preset_overrides.is_causal
|
||||
vae_tiling: generator.pipeline.preset_overrides.vae_tiling
|
||||
vae_sp: generator.pipeline.preset_overrides.vae_sp
|
||||
dmd_denoising_steps: generator.pipeline.preset_overrides.dmd_denoising_steps
|
||||
ti2v_task: generator.pipeline.preset_overrides.ti2v_task
|
||||
lucy_edit_task: generator.pipeline.preset_overrides.lucy_edit_task
|
||||
boundary_ratio: generator.pipeline.preset_overrides.boundary_ratio
|
||||
flow_shift_sr: generator.pipeline.experimental.flow_shift_sr
|
||||
is_causal: generator.pipeline.experimental.is_causal
|
||||
ti2v_task: generator.pipeline.experimental.ti2v_task
|
||||
lucy_edit_task: generator.pipeline.experimental.lucy_edit_task
|
||||
compatibility_only:
|
||||
model_path: "Redundant with generator.model_path."
|
||||
disable_autocast: "Duplicated by generator.engine.disable_autocast during migration."
|
||||
dit_precision: "Precision override pending dedicated typed component precision design."
|
||||
upsampler_precision: "Precision override pending dedicated typed component precision design."
|
||||
vae_precision: "Precision override pending dedicated typed component precision design."
|
||||
vae_decode_precision: "Decode-only precision override pending dedicated typed component precision design."
|
||||
image_encoder_precision: "Precision override pending dedicated typed component precision design."
|
||||
image_encoder_precisions: "Precision overrides pending dedicated typed component precision design."
|
||||
text_encoder_precisions: "Precision override pending dedicated typed component precision design."
|
||||
internal_only:
|
||||
dit_config: "Legacy internal component config object."
|
||||
upsampler_config: "Legacy internal component config object."
|
||||
@@ -140,6 +166,22 @@ surfaces:
|
||||
scheduler_step_in_fp32: "Runtime scheduler precision toggle; not part of the public typed inference API."
|
||||
|
||||
pipeline_config_extensions:
|
||||
moved:
|
||||
enable_bsa:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.enable_bsa
|
||||
bsa_sparsity:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_sparsity
|
||||
bsa_cdf_threshold:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_cdf_threshold
|
||||
bsa_chunk_q:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_chunk_q
|
||||
bsa_chunk_k:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
target: generator.pipeline.model.longcat.bsa_chunk_k
|
||||
preset_owned:
|
||||
flux2_text_encoder_type:
|
||||
sources:
|
||||
@@ -320,18 +362,8 @@ surfaces:
|
||||
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
|
||||
bsa_cdf_threshold:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_chunk_k:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_chunk_q:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_params:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_sparsity:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_bsa:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enhance_hf:
|
||||
@@ -516,6 +548,7 @@ surfaces:
|
||||
cfg_truncation: request.sampling.cfg_truncation
|
||||
guidance_rescale: request.sampling.guidance_rescale
|
||||
use_embedded_guidance: request.sampling.use_embedded_guidance
|
||||
embedded_cfg_scale: request.sampling.embedded_cfg_scale
|
||||
true_cfg_scale: request.sampling.true_cfg_scale
|
||||
boundary_ratio: request.sampling.boundary_ratio
|
||||
sigmas: request.sampling.sigmas
|
||||
|
||||
+44
-23
@@ -11,6 +11,8 @@ FastVideo maps a Diffusers-style repo into a pipeline like this:
|
||||
- `fastvideo/models/*`: model implementations (DiT, VAE, encoders, upsamplers).
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/models/wan/`: Wan's transformers, VAE, configs, and variant definitions
|
||||
together; the old Wan component/config modules remain compatibility imports.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipelines.
|
||||
@@ -26,19 +28,15 @@ Minimal usage (from `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
generator = VideoGenerator.from_pretrained(model_id, {"engine": {"num_gpus": 1}})
|
||||
|
||||
sampling = SamplingParam.from_pretrained(model_id)
|
||||
sampling.num_frames = 45
|
||||
video = generator.generate_video(
|
||||
"A vibrant city street at sunset.",
|
||||
sampling_param=sampling,
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": "A vibrant city street at sunset.",
|
||||
"sampling": {"num_frames": 45},
|
||||
"output": {"output_path": "video_samples", "save_video": True},
|
||||
})
|
||||
```
|
||||
|
||||
## Configuration system
|
||||
@@ -48,16 +46,35 @@ runtime parameters consistent:
|
||||
|
||||
- `fastvideo/configs/models/`: architecture definitions, layer shapes, and
|
||||
`param_names_mapping` rules for key renaming.
|
||||
- `fastvideo/models/wan/config.py`: Wan's transformer architecture and mapping rules,
|
||||
co-located with `transformer.py`.
|
||||
- `fastvideo/models/wan/vae_config.py`: Wan's VAE architecture and runtime
|
||||
settings, co-located with `vae.py`.
|
||||
- `fastvideo/models/wan/pipeline_config.py`: Wan component and pipeline defaults;
|
||||
`configs/pipelines/wan.py` remains a compatibility import.
|
||||
- `fastvideo/models/wan/definition.py`: data-only Wan variants linking HF aliases,
|
||||
config classes, presets, workload metadata, and default sampling algorithms.
|
||||
- `fastvideo/configs/pipelines/`: pipeline wiring and required components.
|
||||
- `fastvideo/api/sampling_param.py`: sampling parameters (steps, frames,
|
||||
guidance scale, resolution, fps). Defaults come from profiles in
|
||||
`fastvideo/pipelines/basic/<family>/profiles.py`.
|
||||
guidance scale, resolution, fps). Defaults come from presets in
|
||||
`fastvideo/pipelines/basic/<family>/presets.py`.
|
||||
- `fastvideo/registry.py`: unified registry for pipeline config + sampling
|
||||
defaults and model metadata resolution, defined via explicit
|
||||
`register_configs(...)` blocks (no separate dict registries).
|
||||
defaults and model metadata resolution. Wan registrations consume its
|
||||
family-local definitions; other families use `register_configs(...)` blocks.
|
||||
|
||||
`FastVideoArgs` (in `fastvideo/fastvideo_args.py`) provides runtime settings and
|
||||
is passed into pipeline construction and stages.
|
||||
Wan definitions reference existing defaults rather than copying them. Dense
|
||||
UniPC, dense DMD, and causal DMD remain separate sampling algorithms. Dense
|
||||
DMD uses a full training-noise scheduler with shift 8.0, separate from the
|
||||
configurable scheduler mutated during timestep preparation. The catalog does
|
||||
not override checkpoint manifests, user pipeline overrides, or component
|
||||
precision settings. HF IDs, local checkpoints, and old config imports retain
|
||||
their existing resolution behavior, including first-match detector ordering.
|
||||
|
||||
`ResolvedGeneratorConfig` (in `fastvideo/api/resolution.py`) provides runtime
|
||||
settings and is passed into pipeline construction and stages as `resolved_config`.
|
||||
`resolve_inference_config` (in `fastvideo/api/inference_resolution.py`) builds it
|
||||
from the typed config in `fastvideo/api/schema.py` and attaches the model's
|
||||
frozen `PipelineConfig` as `resolved_config.pipeline_config`.
|
||||
|
||||
## Weights and Diffusers format
|
||||
|
||||
@@ -96,7 +113,8 @@ Note on tensor names:
|
||||
|
||||
Official checkpoints often use different `state_dict` names than FastVideo's
|
||||
module layout. We translate tensor names via the DiT arch config mapping
|
||||
(`param_names_mapping` under `fastvideo/configs/models/dits/`). This is similar
|
||||
(`param_names_mapping` under `fastvideo/configs/models/dits/`, or
|
||||
`fastvideo/models/wan/config.py` for Wan). This is similar
|
||||
in spirit to name-translation layers used in systems like vLLM and SGLang.
|
||||
|
||||
Example HF repo (Wan 2.1 T2V 1.3B Diffusers):
|
||||
@@ -137,13 +155,16 @@ Example `model_index.json` from that repo:
|
||||
How this maps to FastVideo:
|
||||
|
||||
- `WanPipeline` -> `fastvideo/pipelines/basic/wan/wan_pipeline.py`
|
||||
- `WanTransformer3DModel` -> `fastvideo/models/dits/wanvideo.py`
|
||||
- `AutoencoderKLWan` -> `fastvideo/models/vaes/wanvae.py`
|
||||
- `WanTransformer3DModel` -> `fastvideo/models/wan/transformer.py`
|
||||
- `WanVideoConfig` -> `fastvideo/models/wan/config.py`
|
||||
- `AutoencoderKLWan` -> `fastvideo/models/wan/vae.py`
|
||||
- `WanVAEConfig` -> `fastvideo/models/wan/vae_config.py`
|
||||
- `UMT5EncoderModel` -> `fastvideo/models/encoders/t5.py`
|
||||
- `T5TokenizerFast` -> loaded via HF in `fastvideo/models/loader/`
|
||||
- `UniPCMultistepScheduler` -> loaded via Diffusers scheduler utilities
|
||||
- Pipeline defaults -> `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults -> `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
- Variant definitions -> `fastvideo/models/wan/definition.py`
|
||||
- Pipeline defaults -> `fastvideo/models/wan/pipeline_config.py`
|
||||
- Sampling defaults -> `fastvideo/pipelines/basic/wan/presets.py`
|
||||
|
||||
## Pipeline system
|
||||
|
||||
@@ -157,8 +178,8 @@ How this maps to FastVideo:
|
||||
|
||||
## Model components
|
||||
|
||||
- DiT models: `fastvideo/models/dits/`
|
||||
- VAEs: `fastvideo/models/vaes/`
|
||||
- DiT models: `fastvideo/models/dits/`; dense Wan: `fastvideo/models/wan/`
|
||||
- VAEs: `fastvideo/models/vaes/`; Wan: `fastvideo/models/wan/vae.py`
|
||||
- Text/image encoders: `fastvideo/models/encoders/`
|
||||
- Schedulers: `fastvideo/models/schedulers/`
|
||||
- Upsamplers: `fastvideo/models/upsamplers/`
|
||||
|
||||
@@ -32,7 +32,7 @@ from fastvideo.api import (
|
||||
|
||||
| Surface | Availability | Notes |
|
||||
| --- | --- | --- |
|
||||
| `VideoGenerator.from_pretrained(model_path, **typed_kwargs)` | Today | `typed_kwargs` is a stable subset from `GeneratorConfig` — no flat legacy LTX-2 kwargs (guaranteed after PR 6) |
|
||||
| `VideoGenerator.from_pretrained(model_path, config)` | Today | `config` is a nested `GeneratorConfig` mapping without `model_path`; no flat keywords |
|
||||
| `VideoGenerator.generate(request: GenerationRequest) -> GenerationResult` | Today | Aggregated; Dynamo wraps in `asyncio.to_thread` under `asyncio.Lock` |
|
||||
| `VideoGenerator.generate_async(request) -> AsyncGenerator[VideoEvent, None]` | **PR 7.10** | Canonical execution substrate; sync wrapper reroutes through this |
|
||||
| `VideoGenerator.default_health_check_request() -> GenerationRequest` | **PR 7.10** | 256x256 / 8 frames / 1 step; lets Dynamo build its health payload without knowing any FastVideo internals |
|
||||
@@ -225,7 +225,7 @@ async def init_video_generation(runtime, config, shutdown_endpoints):
|
||||
from fastvideo.api import config_to_dict
|
||||
|
||||
server_args, dynamo_args = config.server_args, config.dynamo_args
|
||||
generator = VideoGenerator.from_pretrained(**config.fastvideo_kwargs())
|
||||
generator = VideoGenerator.from_config(build_generator_config(server_args))
|
||||
|
||||
dump_config(dynamo_args.dump_config_to, config)
|
||||
|
||||
@@ -262,7 +262,8 @@ this adapter can build the config purely from the public typed schema:
|
||||
def build_generator_config(args) -> "GeneratorConfig":
|
||||
from fastvideo.api import (
|
||||
CompileConfig, ComponentConfig, EngineConfig, GeneratorConfig,
|
||||
OffloadConfig, ParallelismConfig, PipelineSelection,
|
||||
LTX2Options, LTX2RefineOptions, OffloadConfig, ParallelismConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
return GeneratorConfig(
|
||||
model_path=args.model_path,
|
||||
@@ -275,10 +276,8 @@ def build_generator_config(args) -> "GeneratorConfig":
|
||||
pipeline=PipelineSelection(
|
||||
workload_type=args.workload or "t2v",
|
||||
preset=args.preset, # e.g. "ltx2_two_stage"
|
||||
components=ComponentConfig(
|
||||
upsampler_weights=args.refine_upsampler,
|
||||
lora_path=args.refine_lora,
|
||||
),
|
||||
components=ComponentConfig(upsampler_weights=args.refine_upsampler),
|
||||
model=LTX2Options(refine=LTX2RefineOptions(lora_path=args.refine_lora)),
|
||||
),
|
||||
)
|
||||
```
|
||||
@@ -310,8 +309,11 @@ re-chase FastVideo drift:
|
||||
2. `ContinuationState.payload` is JSON-serializable or references
|
||||
opaque blob ids. Dynamo can round-trip it through RPC without
|
||||
special-casing torch tensors.
|
||||
3. `VideoGenerator.from_pretrained` accepts a typed `GeneratorConfig`;
|
||||
legacy flat kwargs are compatibility-only and deprecate in PR 13.
|
||||
3. `VideoGenerator.from_pretrained(model_path, config)` takes a typed
|
||||
`GeneratorConfig` or its nested mapping; any flat keyword raises
|
||||
`TypeError` that points to the nested config.
|
||||
`VideoGenerator.from_config(...)` takes the same settings with
|
||||
`model_path` inside.
|
||||
4. `generate_async` (PR 7.10+) emits events in order
|
||||
`Progress* → Partial* → Final`; the final event always has exactly
|
||||
one occurrence per request.
|
||||
@@ -329,7 +331,6 @@ at FastVideo's CI — before the Dynamo-side integration even knows.
|
||||
* Anything under `fastvideo.pipelines.*` directly (pipelines are
|
||||
internal; presets identify them by name on
|
||||
`PipelineSelection.preset`).
|
||||
* `fastvideo.fastvideo_args.FastVideoArgs` (legacy compat type).
|
||||
* `fastvideo.api.compat.*` private helpers
|
||||
(`_validate_continuation_state` etc.) — the public boundary is
|
||||
`VideoGenerator` + `fastvideo.api`.
|
||||
|
||||
@@ -25,11 +25,13 @@ requests. HTTP handling and job polling remain asynchronous.
|
||||
| `GET` | `/v1/videos/{id}/content` | Download a completed MP4 |
|
||||
| `DELETE` | `/v1/videos/{id}` | Delete a job and its completed artifact |
|
||||
| `POST` | `/v1/images` | Generate an image |
|
||||
| `POST` | `/v1/images/generations` | OpenAI-compatible alias for image generation |
|
||||
| `POST` | `/v1/images/edits` | Generate an image from image references |
|
||||
| `GET` | `/v1/images/{id}/content` | Download a generated image |
|
||||
| `GET` | `/health` | Liveness probe |
|
||||
|
||||
`POST /v1/videos/generations` remains an alias for older FastVideo clients.
|
||||
`POST /v1/videos/generations` remains an alias for older FastVideo clients, and
|
||||
`POST /v1/images/generations` is the OpenAI Python client's image generation path.
|
||||
The OpenAI Python and JavaScript clients can create, retrieve, list, download,
|
||||
and delete video jobs. Use the [H3 server cookbook](../../cookbook/openai-api.md)
|
||||
for pinned client versions and executable examples. Download variants other
|
||||
|
||||
@@ -76,6 +76,7 @@ COOKBOOK_EVIDENCE_STATES = {
|
||||
}
|
||||
COOKBOOK_HARDWARE_EVIDENCE = {"validated", "source-configured", "estimated", "unknown"}
|
||||
COOKBOOK_HARDWARE_PLATFORMS = {"cuda", "mlx", "mps"}
|
||||
COOKBOOK_HARDWARE_DEVICES = {"spark"}
|
||||
COOKBOOK_GPU_TYPES = {"NVIDIA", "Apple Silicon"}
|
||||
COOKBOOK_HARDWARE_TEXT_FIELDS = {
|
||||
"accelerator",
|
||||
@@ -128,6 +129,12 @@ def validate_cookbook() -> None:
|
||||
raise ValueError(f"CUDA cookbook recipe needs an integer gpu_count >= 1: {recipe['id']}")
|
||||
if platform != "cuda" and gpu_count is not None:
|
||||
raise ValueError(f"Non-CUDA cookbook recipe must not use gpu_count: {recipe['id']}")
|
||||
device = hardware.get("device")
|
||||
if device is not None:
|
||||
if device not in COOKBOOK_HARDWARE_DEVICES:
|
||||
raise ValueError(f"Cookbook recipe has an unknown hardware device: {recipe['id']}: {device}")
|
||||
if platform != "cuda":
|
||||
raise ValueError(f"Cookbook hardware device requires platform cuda: {recipe['id']}: {device}")
|
||||
hardware_evidence = hardware.get("evidence")
|
||||
if hardware_evidence not in COOKBOOK_HARDWARE_EVIDENCE:
|
||||
raise ValueError(f"Cookbook recipe has unknown hardware evidence: {recipe['id']}: {hardware_evidence}\n"
|
||||
@@ -163,6 +170,31 @@ def validate_cookbook() -> None:
|
||||
or any(not isinstance(item, str) or not item.strip() for item in modes)):
|
||||
raise ValueError(f"Cookbook recipe modes must be a non-empty list of strings: {recipe['id']}")
|
||||
|
||||
knobs = recipe.get("knobs", [])
|
||||
if not isinstance(knobs, list):
|
||||
raise ValueError(f"Cookbook recipe knobs must be a list: {recipe['id']}")
|
||||
knob_keys: set[str] = set()
|
||||
for knob in knobs:
|
||||
if not isinstance(knob, dict):
|
||||
raise ValueError(f"Cookbook recipe knob must be an object: {recipe['id']}")
|
||||
required_knob = ("key", "label", "flag", "options", "default")
|
||||
missing_knob = {key for key in required_knob if key not in knob}
|
||||
if missing_knob:
|
||||
raise ValueError(f"Cookbook recipe knob is missing: {recipe['id']}: {', '.join(sorted(missing_knob))}")
|
||||
if knob["key"] in knob_keys:
|
||||
raise ValueError(f"Duplicate cookbook recipe knob key: {recipe['id']}: {knob['key']}")
|
||||
knob_keys.add(knob["key"])
|
||||
if not isinstance(knob["flag"], str) or not knob["flag"].startswith("--"):
|
||||
raise ValueError(f"Cookbook recipe knob flag must start with --: {recipe['id']}: {knob['key']}")
|
||||
options = knob["options"]
|
||||
if not isinstance(options, list) or not options:
|
||||
raise ValueError(
|
||||
f"Cookbook recipe knob options must be a non-empty list: {recipe['id']}: {knob['key']}")
|
||||
option_values = [option["value"] if isinstance(option, dict) else option for option in options]
|
||||
if knob["default"] not in option_values:
|
||||
raise ValueError(
|
||||
f"Cookbook recipe knob default must be one of its options: {recipe['id']}: {knob['key']}")
|
||||
|
||||
source = (ROOT_DIR / recipe["source"]).resolve()
|
||||
if not any(source.is_relative_to(root.resolve()) for root in COOKBOOK_SOURCE_ROOTS):
|
||||
raise ValueError(f"Cookbook source is outside an approved directory: {recipe['source']}")
|
||||
@@ -170,6 +202,19 @@ def validate_cookbook() -> None:
|
||||
raise ValueError(f"Cookbook source does not exist: {recipe['source']}")
|
||||
|
||||
source_text = source.read_text(encoding="utf-8")
|
||||
if knobs:
|
||||
# A script that shares its argument parser with a sibling module
|
||||
# (e.g. `from . import basic_fasth3`) inherits that module's
|
||||
# flags without the flag text appearing in its own source.
|
||||
knob_source_text = source_text
|
||||
for match in re.finditer(r'(?:from\s+\.\s+import|^\s*import)\s+(\w+)', source_text, flags=re.MULTILINE):
|
||||
sibling = source.parent / f"{match.group(1)}.py"
|
||||
if sibling.is_file() and sibling != source:
|
||||
knob_source_text += "\n" + sibling.read_text(encoding="utf-8")
|
||||
for knob in knobs:
|
||||
if knob["flag"] not in knob_source_text:
|
||||
raise ValueError(f"Cookbook recipe knob flag is not in its source: {recipe['id']}: {knob['flag']}")
|
||||
|
||||
# The model must be traceable to the checked-in source itself, or be
|
||||
# passed explicitly on the command line (e.g. --model-path <model> or
|
||||
# MODEL_PATH=<model>) when the source reads it from arguments/env.
|
||||
@@ -223,12 +268,22 @@ def cookbook_serving_profile(recipe: dict) -> dict:
|
||||
if type(port) is not int or not 1 <= port <= 65535:
|
||||
raise ValueError(f"Serving config needs a valid port: {recipe['id']}")
|
||||
hardware = {"platform": runtime, "evidence": "source-configured"}
|
||||
device = recipe["hardware"].get("device")
|
||||
if device:
|
||||
hardware["device"] = device
|
||||
if runtime == "cuda":
|
||||
count = generator["engine"]["num_gpus"]
|
||||
if type(count) is not int or count < 1:
|
||||
raise ValueError(f"Serving config needs a valid GPU count: {recipe['id']}")
|
||||
hardware["gpu_count"] = count
|
||||
command = f"fastvideo serve --config {serving['source']} --server.host 127.0.0.1"
|
||||
env = serving.get("env")
|
||||
if env is not None:
|
||||
if not isinstance(env, str) or not env.strip() or "\n" in env:
|
||||
raise ValueError(f"Serving env must be a single-line command prefix: {recipe['id']}")
|
||||
prefix = env.strip() + " "
|
||||
else:
|
||||
prefix = ""
|
||||
command = f"{prefix}fastvideo serve --config {serving['source']} --server.host 127.0.0.1"
|
||||
else:
|
||||
if server.get("host") != "127.0.0.1":
|
||||
raise ValueError(f"MLX cookbook server must bind to loopback: {recipe['id']}")
|
||||
|
||||
@@ -4,9 +4,10 @@
|
||||
FastVideo supports the following hardware platforms:
|
||||
|
||||
- [NVIDIA CUDA](installation/gpu.md)
|
||||
- [NVIDIA DGX Spark / GB10 (ARM64 + CUDA 13)](installation/spark.md)
|
||||
([performance & tuning](installation/spark_performance.md))
|
||||
- [Apple silicon](installation/mps.md)
|
||||
- **NVIDIA DGX Spark / GB10 (ARM64 + CUDA 13)** — [install](installation/spark.md),
|
||||
[performance](installation/spark_performance.md),
|
||||
[pair two Sparks](installation/spark_pair.md)
|
||||
- [Apple silicon (MLX)](installation/mlx.md)
|
||||
|
||||
## Quick Installation
|
||||
|
||||
@@ -14,7 +15,7 @@ FastVideo supports the following hardware platforms:
|
||||
|
||||
Use uv as the default environment manager for faster and more stable installs.
|
||||
The commands below target NVIDIA CUDA 12; use `UV_TORCH_BACKEND=cu130` on
|
||||
CUDA 13. Apple silicon users should follow the [MPS guide](installation/mps.md).
|
||||
CUDA 13. Apple silicon users should follow the [MLX install guide](installation/mlx.md).
|
||||
|
||||
```bash
|
||||
# Create and activate a new uv environment
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
# Install FastVideo with MLX
|
||||
|
||||
Install FastVideo on a Mac, then generate from the cookbook. Local video uses
|
||||
the native MLX runtime, not CUDA, and not the old PyTorch MPS demo at
|
||||
`examples/inference/basic/basic_mps.py`.
|
||||
|
||||
## Requirements
|
||||
|
||||
- macOS 14 or newer
|
||||
- Python 3.12
|
||||
- `ffmpeg` (`brew install ffmpeg`)
|
||||
|
||||
## Install
|
||||
|
||||
Cookbook commands run from a clone.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git && cd FastVideo
|
||||
uv venv --python 3.12 --seed
|
||||
source .venv/bin/activate
|
||||
brew install ffmpeg
|
||||
uv pip install -e ".[mlx]"
|
||||
```
|
||||
|
||||
Conda is optional. After you activate a Conda env, still install with
|
||||
`uv pip` as above.
|
||||
|
||||
`uv pip install "fastvideo[mlx]"` from PyPI installs the extra only. It does
|
||||
not ship the example scripts the cookbook copies.
|
||||
|
||||
## Generate a video
|
||||
|
||||
Open the cookbook. Select Apple Silicon as the runtime. Each recipe has a
|
||||
Python command. FastH3 also has a server path for the playground and the
|
||||
OpenAI Python client.
|
||||
|
||||
- [Wan recipes](../../cookbook/wan.md) for FastMetal 1.3B, 5B, and 14B
|
||||
- [MiniMax H3 recipes](../../cookbook/minimax-h3.md) for FastH3 V1 and FastH3 V2
|
||||
- [H3 server guide](../../cookbook/openai-api.md) for the playground, cURL, and SDKs
|
||||
|
||||
FastH3 is two distilled MiniMax-H3 checkpoints. V1 is the four-step launch.
|
||||
Some Hub repo names still say Preview. That name is historical. V1 is a full
|
||||
model, not a demo. V2 is the eight-step checkpoint. More forwards is why V2
|
||||
is the higher-quality FastH3.
|
||||
|
||||
Recorded shapes and evidence live in the
|
||||
[support matrix](../../inference/support_matrix.md#apple-silicon-native-runtime).
|
||||
|
||||
## Hardware
|
||||
|
||||
- FastMetal 1.3B and 5B: 16 GB unified memory and up
|
||||
- FastMetal 14B: 36 GB unified memory and up
|
||||
- FastH3 V1 and V2: validated on an M4 Max with 36 GB unified memory
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **`basic_mps.py` is the wrong path.** That script is PyTorch MPS. Use an
|
||||
Apple Silicon recipe in the cookbook.
|
||||
- **Muxing fails.** Install `ffmpeg` with Homebrew.
|
||||
- **A cookbook command cannot find a script.** Run it from the FastVideo
|
||||
clone after `uv pip install -e ".[mlx]"`.
|
||||
|
||||
If that does not match what you see, open an issue on the
|
||||
[GitHub repository](https://github.com/hao-ai-lab/FastVideo) or ask in the
|
||||
[Slack community](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ).
|
||||
@@ -1,260 +0,0 @@
|
||||
# MPS (Apple Silicon)
|
||||
|
||||
Install FastVideo on Apple Silicon and run FastMetal-QAD or FastH3 Preview.
|
||||
|
||||
Apple Silicon uses the MLX runtime. FastMetal-QAD ships ready-to-run MLX
|
||||
checkpoints; FastH3 Preview currently requires a local MLX DiT conversion.
|
||||
See the [FastMetal-QAD blog](https://haoailab.com/blogs/fastmetal/) and the
|
||||
[FastMetal collection](https://huggingface.co/collections/FastVideo/fastmetal).
|
||||
|
||||
## Requirements
|
||||
|
||||
- **OS: macOS 14 or newer**
|
||||
- **Python: 3.12.4**
|
||||
|
||||
## Set up using Python
|
||||
|
||||
### Create a new Python environment
|
||||
|
||||
#### uv
|
||||
Recommended default: use [uv](https://docs.astral.sh/uv/) for faster and more stable environment setup.
|
||||
|
||||
Please follow the [documentation](https://docs.astral.sh/uv/#getting-started) to install `uv`. After installing `uv`, create a new environment using:
|
||||
|
||||
```console
|
||||
# (Recommended) Create a new uv environment. Use `--seed` to install `pip` and `setuptools`.
|
||||
uv venv --python 3.12 --seed
|
||||
source .venv/bin/activate
|
||||
```
|
||||
|
||||
#### Conda (alternative)
|
||||
|
||||
You can also create a Python environment using [Conda](https://docs.conda.io/projects/conda/en/stable/user-guide/getting-started.html).
|
||||
|
||||
##### 1. Install Miniconda (if not already installed)
|
||||
|
||||
```bash
|
||||
wget https://repo.anaconda.com/miniconda/Miniconda3-latest-MacOSX-arm64.sh
|
||||
bash Miniconda3-latest-MacOSX-arm64.sh
|
||||
source ~/.zshrc
|
||||
```
|
||||
|
||||
##### 2. Create and activate a Conda environment for FastVideo
|
||||
|
||||
```bash
|
||||
conda create -n fastvideo python=3.12.4 -y
|
||||
conda activate fastvideo
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
|
||||
```
|
||||
brew install ffmpeg
|
||||
```
|
||||
|
||||
### Installation
|
||||
|
||||
FastMetal's native Apple Silicon runtime requires the `mlx` extra.
|
||||
|
||||
#### With uv (recommended)
|
||||
|
||||
```bash
|
||||
uv pip install "fastvideo[mlx]"
|
||||
```
|
||||
|
||||
#### With Conda environment (alternative)
|
||||
|
||||
`uv` works inside an active conda env too, so prefer `uv pip` for the actual install:
|
||||
|
||||
```bash
|
||||
uv pip install "fastvideo[mlx]"
|
||||
```
|
||||
|
||||
### Installation from Source
|
||||
|
||||
#### 1. Clone the FastVideo repository
|
||||
|
||||
```bash
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git && cd FastVideo
|
||||
```
|
||||
|
||||
#### 2. Install FastVideo
|
||||
|
||||
Basic installation:
|
||||
|
||||
```bash
|
||||
uv pip install -e ".[mlx]"
|
||||
```
|
||||
|
||||
Alternative with Conda environment:
|
||||
|
||||
```bash
|
||||
uv pip install -e ".[mlx]"
|
||||
```
|
||||
|
||||
## Run FastMetal-QAD
|
||||
|
||||
Each release is self-contained. Download one checkpoint and point both
|
||||
`--model-root` and `--mlx-checkpoint` at it (the example also auto-detects
|
||||
`mlx_dit.json` under `--model-root`).
|
||||
|
||||
| Checkpoint | Script | Mac tier |
|
||||
| --- | --- | --- |
|
||||
| [`FastVideo/FastMetal-1.3B-QAD`](https://huggingface.co/FastVideo/FastMetal-1.3B-QAD) | `mlx_wan_prompt_to_video.py` | 16 GB+ |
|
||||
| [`FastVideo/FastMetal-5B-QAD`](https://huggingface.co/FastVideo/FastMetal-5B-QAD) | `mlx_wan22_generate.py` | 16 GB+ |
|
||||
| [`FastVideo/FastMetal-14B-QAD`](https://huggingface.co/FastVideo/FastMetal-14B-QAD) | `mlx_wan_prompt_to_video.py` | 36 GB+ |
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastMetal-1.3B-QAD --local-dir ./FastMetal-1.3B-QAD
|
||||
|
||||
python examples/inference/basic/mlx_wan_prompt_to_video.py \
|
||||
--model-root ./FastMetal-1.3B-QAD \
|
||||
--mlx-checkpoint ./FastMetal-1.3B-QAD \
|
||||
--height 480 --width 832 --num-frames 81 \
|
||||
--prompt "A bird's-eye view of a misty forest valley at dawn."
|
||||
```
|
||||
|
||||
14B uses the same script. Point both flags at `./FastMetal-14B-QAD`. That repo also ships an EMA variant: keep `--model-root` at the repo root and set `--mlx-checkpoint ./FastMetal-14B-QAD/ema`.
|
||||
|
||||
Wan2.2 5B uses a different latent layout, so it has its own entrypoint:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastMetal-5B-QAD --local-dir ./FastMetal-5B-QAD
|
||||
|
||||
python examples/inference/basic/mlx_wan22_generate.py \
|
||||
--mlx-checkpoint ./FastMetal-5B-QAD \
|
||||
--text-encoder-root ./FastMetal-5B-QAD \
|
||||
--vae-root ./FastMetal-5B-QAD/vae \
|
||||
--height 704 --width 1280 --num-frames 81 \
|
||||
--prompt "A cinematic portrait with soft neon lighting and smooth camera motion."
|
||||
```
|
||||
|
||||
CUDA FastWan-QAD (`FastVideo/FastWan-QAD-1.3B`, `FastVideo/FastWan-QAD-FP8-1.3B`) is a separate NVIDIA release. The MLX examples look for FastMetal packed weights (`mlx_dit.json`).
|
||||
|
||||
`basic_mps.py` is a generic PyTorch MPS demo. For local video on Mac, use the FastMetal commands above.
|
||||
|
||||
## Run FastH3 Preview
|
||||
|
||||
FastH3 Preview uses the existing MLX runtime for text-to-video-with-audio
|
||||
(T2VA). The runtime streams the Qwen3-VL text conditioner, loads one
|
||||
heavyweight component at a time, denoises synchronized video and audio
|
||||
latents with a converted INT8, INT6, or INT4 DiT, and decodes both modalities
|
||||
with native MLX VAEs.
|
||||
|
||||
Download the FastH3 snapshot, then convert one or more DiT formats:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 \
|
||||
--local-dir ./FastH3-Preview-v0.2
|
||||
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py \
|
||||
--model-root ./FastH3-Preview-v0.2/transformer \
|
||||
--out ./FastH3-MLX \
|
||||
--formats "int6"
|
||||
```
|
||||
|
||||
Dense conversion drops the trained VSA gate projections. To keep them (INT6
|
||||
weight-only, same affine grid as the other linear matrices) write a **new**
|
||||
directory:
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py \
|
||||
--model-root ./FastH3-Preview-v0.2/transformer \
|
||||
--out ./FastH3-MLX-vsa \
|
||||
--formats "int6" \
|
||||
--include-vsa
|
||||
```
|
||||
|
||||
Do not overwrite an existing dense export such as `./FastH3-MLX/int6`.
|
||||
|
||||
Run the baseline path:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX/int6 \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is amazing.</d>" \
|
||||
--height 480 --width 832 --num-frames 124 --seed 2026 \
|
||||
--output-path ./outputs/fasth3_int6.mp4
|
||||
```
|
||||
|
||||
Add `--fast` for temporal fast mode. It denoises a shorter video sequence,
|
||||
uses MLX RIFE to restore the requested frame count, and keeps the audio
|
||||
sequence at full duration:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX/int6 \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is even faster.</d>" \
|
||||
--height 720 --width 1280 --num-frames 124 --seed 2027 \
|
||||
--fast \
|
||||
--output-path ./outputs/fasth3_int6_fast_720p.mp4
|
||||
```
|
||||
|
||||
Add `--fast-spatial` for spatial fast mode, `--fast`'s spatial twin. It
|
||||
denoises and decodes on the smallest 32px-aligned canvas covering the
|
||||
requested size divided by `--fast-spatial-scale` (a 480x832 request runs on a
|
||||
256x416 canvas), then resamples the decoded frames up to the requested size
|
||||
in pixel space. It composes with `--fast`. This is a speed/quality trade-off
|
||||
and stays off by default: the output carries the reduced canvas's detail
|
||||
budget, so it reads softer than a native-resolution render, with the unsharp
|
||||
pass countering some but not all of the difference:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX/int6 \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is fastest.</d>" \
|
||||
--height 480 --width 832 --num-frames 124 --seed 2028 \
|
||||
--fast --fast-spatial \
|
||||
--output-path ./outputs/fasth3_int6_fast_spatial.mp4
|
||||
```
|
||||
|
||||
VSA is off by default. A dense-only checkpoint (no `--include-vsa`) keeps the
|
||||
existing fused-SDPA path. After converting with `--include-vsa`, enable the
|
||||
sparse path explicitly:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX-vsa/int6 \
|
||||
--vsa --vsa-sparsity 0.9 --vsa-tile-size 64 --vsa-prefix-mode exempt \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is amazing.</d>" \
|
||||
--height 720 --width 1280 --num-frames 124 --seed 2026 \
|
||||
--output-path ./outputs/fasth3_int6_vsa_720p.mp4
|
||||
```
|
||||
|
||||
`--vsa-impl auto` uses the chunked gather+SDPA **reference** path.
|
||||
`--vsa-impl simd` is an opt-in SIMD-group kernel (tile 64, head dim 128) that
|
||||
falls back to reference on unsupported shapes. It is not the default.
|
||||
`--vsa-impl reference` is the same as `auto`.
|
||||
|
||||
!!! note "Current MLX scope"
|
||||
This source runtime supports T2VA, temporal `--fast`, spatial
|
||||
`--fast-spatial`, and opt-in VSA. FL2VA, Ref2VA, two-pass refinement, and
|
||||
`VideoGenerator` registry dispatch are not wired yet. INT8/INT6/INT4
|
||||
quantization is **weight-only**; VSA attention Q/K/V stay BF16. Old dense
|
||||
MLX checkpoints remain valid for dense inference and raise a reconvert
|
||||
error if `--vsa` is set. The checkpoint uses the MiniMax H3
|
||||
Community License; review the model card before use or redistribution.
|
||||
|
||||
## Development Environment Setup
|
||||
|
||||
If you're planning to contribute to FastVideo please see the following page:
|
||||
[Contributor Guide](../../contributing/overview.md)
|
||||
|
||||
## Hardware Requirements
|
||||
|
||||
- **1.3B / 5B:** 16 GB unified memory and up (M1 and later)
|
||||
- **14B:** 36 GB unified memory and up
|
||||
- **FastH3 Preview:** validated on an M4 Max with 36 GB unified memory; use one
|
||||
converted DiT format at a time and leave substantial free disk space for the
|
||||
source snapshot plus the converted checkpoint
|
||||
- Fanless 13-inch MacBook Air can run 1.3B and 5B at the same resolutions
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
If you encounter any issues during installation, please open an issue on our [GitHub repository](https://github.com/hao-ai-lab/FastVideo).
|
||||
|
||||
You can also join our [Slack community](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ) for additional support.
|
||||
@@ -144,6 +144,12 @@ for which models are practical on the GB10, what makes them faster, and what
|
||||
won't help on this hardware (and why) — so you don't spend a night tuning knobs
|
||||
that can't move here.
|
||||
|
||||
Two Sparks with QSFP cables: [Pair two NVIDIA DGX Sparks](spark_pair.md) for
|
||||
one FastH3 clip across both GPUs (`sp_size=2` over Ray). Copy-paste commands
|
||||
for one or two Sparks also live on the
|
||||
[MiniMax H3 cookbook](../../cookbook/minimax-h3.md): pick FastH3 V1,
|
||||
then NVIDIA DGX Spark, then 1 Spark or 2 Sparks.
|
||||
|
||||
## Development Environment Setup
|
||||
|
||||
If you're planning to contribute to FastVideo please see the
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
# Pair two NVIDIA DGX Sparks
|
||||
|
||||
One GB10 is 128 GB of unified LPDDR5X. FastH3 still fits on a single Spark with
|
||||
[`lazy_module_load`](../../inference/offloading.md) (auto on GB10; sequential
|
||||
load stands down when lazy owns deferral). Two boxes connected
|
||||
by the QSFP ConnectX-7 cables can run **one clip faster** and can hold a
|
||||
**longer clip** (up to the FastH3 15 s cap).
|
||||
|
||||
Copy-paste commands for both counts are on the
|
||||
[MiniMax H3 cookbook](../../cookbook/minimax-h3.md): FastH3 V1 → NVIDIA DGX
|
||||
Spark → 1 Spark or 2 Sparks.
|
||||
|
||||
This is FastVideo sequence parallel (`sp_size=2`) over Ray, not a third-party
|
||||
xDiT vendor. Do not install xDiT for this path.
|
||||
|
||||
## What two Sparks buy you
|
||||
|
||||
| Goal | How | Use two Sparks? |
|
||||
|---|---|---|
|
||||
| Two independent videos at once | One process per box, `engine.num_gpus: 1` | Throughput only. Each clip still takes the 1-GPU time for that size. |
|
||||
| One clip, faster | Ray + `sp_size=2` + parallel VAE | **Yes.** One 768×1344×124 recipe was 292 s vs 374 s on one GB10. |
|
||||
| One clip, longer | Same, more frames | **Yes.** 345 frames (~14.4 s at 24 fps) finished in 587 s at 768×1344. |
|
||||
|
||||
Sequence parallel **replicates** the DiT (~66 GiB per node). Lazy module load
|
||||
is still required on each box. FSDP would shard weights;
|
||||
it is untested on this fabric and is likely slower because every layer gathers
|
||||
over ~21 GB/s RoCE.
|
||||
|
||||
## Requirements
|
||||
|
||||
- Two DGX Sparks with FastVideo [installed](spark.md) (CUDA 13, `aarch64`).
|
||||
- The QSFP cables that ship with a dual-Spark kit, **ACTIVE** at 200 Gb/s:
|
||||
`ibstat` should show the ConnectX-7 ports `LinkUp`.
|
||||
- The same FastH3 snapshot on **both** NVMes. Copy the Hugging Face cache over
|
||||
QSFP; do not download 100+ GB twice over Wi-Fi.
|
||||
- Ray in the FastVideo venv (`uv pip install ray` if it is not already there).
|
||||
|
||||
Each Spark has **one** GPU. `engine.num_gpus: 2` therefore means two nodes, which is
|
||||
why the executor must be Ray (`mp` only works inside one process tree).
|
||||
|
||||
## 1. Put IPv4 on the QSFP NICs
|
||||
|
||||
The RoCE links often come up with no IPv4. Wi-Fi (`192.168.1.x`) is fine for
|
||||
SSH and must stay the default route. NCCL and Ray must **not** use it.
|
||||
|
||||
Pick a /24 that does not collide with your LAN. Example:
|
||||
|
||||
| Node | QSFP IPv4 | Interface (typical) |
|
||||
|---|---|---|
|
||||
| Spark A | `192.168.23.1/24` | `enp1s0f1np1` |
|
||||
| Spark B (Ray head) | `192.168.23.2/24` | `enp1s0f1np1` |
|
||||
|
||||
Confirm names with `ibdev2netdev` and `ip -br link`. Then, as root, on each
|
||||
box (NetworkManager likes to steal the NIC; unmanaged is enough for a session):
|
||||
|
||||
```bash
|
||||
sudo nmcli device set enp1s0f1np1 managed no
|
||||
sudo ip addr replace 192.168.23.1/24 dev enp1s0f1np1 # .2 on the other box
|
||||
sudo ip link set enp1s0f1np1 mtu 9000 up
|
||||
```
|
||||
|
||||
These addresses do **not** survive reboot. Ping across the cable before
|
||||
continuing: `ping -c 3 -I enp1s0f1np1 192.168.23.2`.
|
||||
|
||||
A healthy fabric on this hardware looks like:
|
||||
|
||||
- TCP iperf (jumbo 9000): ~40 Gb/s
|
||||
- NCCL allreduce 1 GiB × 10: ~21 GB/s busbw (NVIDIA's dual-Spark figure is ~21.7)
|
||||
|
||||
## 2. Start a two-node Ray cluster on the cable
|
||||
|
||||
On **both** nodes, from the FastVideo repo, with the venv active:
|
||||
|
||||
```bash
|
||||
source examples/inference/optimizations/spark_pair_env.sh
|
||||
```
|
||||
|
||||
That script pins NCCL and Gloo to the QSFP NIC/HCA, disables NVLink-style P2P
|
||||
(there is none between boxes), and turns off Ray's memory monitor. The monitor
|
||||
treats GB10 unified RSS during a 14-shard DiT load as a runaway and SIGTERMs
|
||||
the worker around shard 11/14. Override `NCCL_SOCKET_IFNAME` /
|
||||
`GLOO_SOCKET_IFNAME` if `ibdev2netdev` shows a different name.
|
||||
|
||||
Cap Ray's object store. The default (~30% of 128 GB) leaves too little room
|
||||
for the DiT:
|
||||
|
||||
```bash
|
||||
# Spark B — head
|
||||
export FASTVIDEO_HOST_IP=192.168.23.2
|
||||
ray start --head --node-ip-address=192.168.23.2 --port=6379 --num-gpus=1 \
|
||||
--disable-usage-stats --object-store-memory=2147483648 --memory=4294967296
|
||||
|
||||
# Spark A — worker
|
||||
export FASTVIDEO_HOST_IP=192.168.23.1
|
||||
ray start --address=192.168.23.2:6379 --node-ip-address=192.168.23.1 --num-gpus=1 \
|
||||
--disable-usage-stats --object-store-memory=2147483648 --memory=4294967296
|
||||
```
|
||||
|
||||
`FASTVIDEO_HOST_IP` **must** match `--node-ip-address`. If you omit it, Ray
|
||||
advertises the Wi-Fi address, FastVideo builds a placement group for
|
||||
`node:192.168.1.x`, and the QSFP workers never match.
|
||||
|
||||
Check `ray status` on the head: `0.0/2.0 GPU` idle.
|
||||
|
||||
## 3. Generate one FastH3 clip on both GPUs
|
||||
|
||||
Run the driver on the **head**, same venv, same QSFP IP.
|
||||
|
||||
`basic_fasth3.py` defaults target a four-GPU GB200 profile: 768×1344, `sm100a`
|
||||
VSA, FA4, four GPUs. On Sparks you must override the kernel flags. Height,
|
||||
width, frames, steps, seed, and prompt are yours. Change them. Legal
|
||||
`num_frames` values are `17n+5`, capped at 362.
|
||||
|
||||
GB10 has no FA4 / sm_100a VSA kernel, so `--vsa-kernel triton --no-fa4` stays
|
||||
required on this box. `--execution-backend ray` is optional when `RAY_ADDRESS`
|
||||
is already set.
|
||||
|
||||
`--warmup --repeats 3` prints a median of three `generate()` calls after an
|
||||
excluded warmup. Sequential load reloads Qwen for each later request, so that
|
||||
protocol works. For a single cold process, pass `--no-warmup --repeats 1`.
|
||||
|
||||
The command below is one example, not a required recipe:
|
||||
|
||||
```bash
|
||||
source examples/inference/optimizations/spark_pair_env.sh
|
||||
export RAY_ADDRESS=192.168.23.2:6379
|
||||
export FASTVIDEO_HOST_IP=192.168.23.2
|
||||
|
||||
python examples/inference/basic/basic_fasth3.py \
|
||||
--model-path FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree \
|
||||
--num-gpus 2 --execution-backend ray \
|
||||
--vsa-kernel triton --no-fa4 \
|
||||
--warmup --repeats 3 --parallel-vae \
|
||||
--height 768 --width 1344 --num-frames 124 --steps 5 \
|
||||
--seed 2026 \
|
||||
--prompt "A wide cinematic shot of an alpine meadow at sunrise, pale pink mountain peaks above a blue valley filled with thin morning mist." \
|
||||
--output outputs/fasth3_spark_pair
|
||||
```
|
||||
|
||||
Config-first equivalent. Edit the YAML the same way, `request.sampling` is not
|
||||
locked:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 \
|
||||
FASTVIDEO_VAE_PARALLEL_DECODE=1 FASTVIDEO_STAGE_LOGGING=1 \
|
||||
fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yaml
|
||||
```
|
||||
|
||||
Stop the cluster when you are done: `ray stop` on both nodes.
|
||||
|
||||
## FastH3 frame counts
|
||||
|
||||
H3 is 24 fps. Legal `num_frames` values are `17n+5`. The pipeline rejects
|
||||
clips longer than **15 s**. The longest legal length is **362 frames**
|
||||
(15.083 s). 360 frames aligns to 362 and is accepted.
|
||||
|
||||
## Measured on two GB10s (2026-08-31)
|
||||
|
||||
These rows are full H3 VAE decode, Triton VSA, sequential + lazy load (auto on
|
||||
GB10), parallel VAE. They are not a required size. Denoise times include
|
||||
deferred DiT load (~35 s on the first generate).
|
||||
|
||||
Cold process, `--no-warmup --repeats 1`, alpine prompt, 768×1344, 5 sigma
|
||||
points (4 DiT forwards):
|
||||
|
||||
| Run | GPUs | Frames | E2E | Denoise | VAE decode |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| One Spark | 1 | 124 | 374–393 s | 180–188 s | 151–156 s |
|
||||
| Two Sparks, SP=2 | 2 | 124 | **292 s** | **122 s** | **102 s** |
|
||||
| Two Sparks, SP=2 | 2 | 345 | **587 s** | **351 s** | **173 s** |
|
||||
|
||||
Warmup excluded, `--warmup --repeats 3` median, 512×896, 5 sigma points, full
|
||||
VAE, same 4-step schedule:
|
||||
|
||||
| Run | GPUs | Frames | Median E2E | Median denoise |
|
||||
|---|---:|---:|---:|---:|
|
||||
| One Spark | 1 | 124 | **251.4 s** | 94.2 s |
|
||||
| Two Sparks, SP=2 | 2 | 124 | **215.2 s** | 72.4 s |
|
||||
|
||||
Those medians used `--height` / `--width` / `--num-frames` as CLI flags. Swap
|
||||
them. Native 480p on this model is 480×832, 124 frames. The 15 s cap is 362
|
||||
frames.
|
||||
|
||||
The first VAE decode still pays `torch.compile`. Later `generate()` calls in
|
||||
the same workers are cheaper. GB10 regional DiT compile stays off because the
|
||||
sm_100a VSA kernel is not on this chip, so denoise is slower than a GB200
|
||||
`sm100a` run at the same geometry.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
| Symptom | Fix |
|
||||
|---|---|
|
||||
| Placement group waits forever / `node:192.168.1.x` | Set `FASTVIDEO_HOST_IP` to the QSFP address on **every** `ray start` **and** on the driver. |
|
||||
| `RayDistributedExecutor` TypeError / abstract `set_log_queue` | Use a FastVideo build that implements those methods on the Ray executor (this page). |
|
||||
| Worker SIGTERM during DiT shard 11/14 | `RAY_memory_monitor_refresh_ms=0` **before** `ray start`. Do not leave Ray's default 30% object store. |
|
||||
| NCCL hangs or uses Wi-Fi | `source spark_pair_env.sh`. Confirm `NCCL_SOCKET_IFNAME` is the QSFP NIC. |
|
||||
| Gloo `connectFullMesh` / `remote=[127.0.0.1]` | Two 1-GPU nodes must not use loopback as the Gloo store. Source `spark_pair_env.sh` so `GLOO_SOCKET_IFNAME` is the QSFP NIC on **each** box. FastVideo no longer copies that NIC name from the driver onto workers. |
|
||||
| Second `generate()` crashes `NoneType.parameters` | Sequential load used to drop the text encoder without reloading it. This branch reloads Qwen for later requests so `--warmup --repeats N` works. |
|
||||
| OOM / `earlyoom` prefers Python | Lazy module load must stay on (do not pass `--no-lazy-module-load` to `basic_fasth3.py` or set `engine.offload.lazy_module_load: false`). Peak GPU during 345-frame denoise is ~90 GiB/node. |
|
||||
| `engine.num_gpus: 2` on one Spark | Each Spark has one GPU. Use Ray across two nodes, or `engine.num_gpus: 1` on one box. |
|
||||
|
||||
## What we are not claiming
|
||||
|
||||
- **Throughput of many clips.** Two independent 1-GPU jobs still win if you
|
||||
want two videos, not one faster video.
|
||||
- **xDiT PipeFusion / CFG-parallel.** FastH3 is 4-step and has no CFG.
|
||||
- **FSDP or tensor parallel as a speedup** on this 21 GB/s link.
|
||||
- **Persistent networking.** The example IPs are session `ip addr replace`.
|
||||
|
||||
More GPUs are legal while `num_attention_heads` (56 on FastH3) is divisible by
|
||||
`sp_size`. Four Sparks would need a four-node fabric that this bring-up did
|
||||
not exercise.
|
||||
@@ -62,10 +62,12 @@ nothing to set. If you run a model that still defaults to an fp32 decode, set th
|
||||
decode-only override yourself:
|
||||
|
||||
```python
|
||||
from fastvideo.configs.pipelines.base import PipelineConfig
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
pipeline_config = PipelineConfig.from_pretrained(model_id)
|
||||
pipeline_config.vae_decode_precision = "bf16" # decode-only; leaves encode precision alone
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": model_id,
|
||||
"engine": {"precision": {"vae_decode": "bf16"}}, # decode-only; leaves encode precision alone
|
||||
})
|
||||
```
|
||||
|
||||
Decode is output-only, so lowering its precision is safe. (Encode seeds the
|
||||
@@ -161,15 +163,30 @@ is power-cycled. To avoid it:
|
||||
on: "CPU" offload uses the same unified RAM. Multi-GPU FSDP sharding remains
|
||||
available because it partitions weights without parking them in a separate
|
||||
host pool.
|
||||
- **MiniMax H3 / FastH3** still needs sequential loading on one GB10. The Qwen3-VL
|
||||
- **MiniMax H3 / FastH3** still needs deferred loading on one GB10. The Qwen3-VL
|
||||
conditioner is tens of gigabytes of BF16. If the DiT and VAEs load while that
|
||||
encoder is still resident, the process is a typical `earlyoom` kill (Python is
|
||||
preferred). `h3_sequential_load` defaults to auto and turns this split on for
|
||||
unified-memory devices. Do not pass `--no-h3-sequential-load` here. Force
|
||||
`--h3-sequential-load` only if auto-detect misses the device. The CUDA pipeline
|
||||
encodes first, releases the encoder, then loads DiT and VAEs onto the
|
||||
accelerator (`to_cpu` follows `cpu_offload`, which is off here). See
|
||||
[Offloading](../../inference/offloading.md).
|
||||
preferred). On unified memory, `lazy_module_load` auto-enables and owns that
|
||||
split (encoder, then DiT, then VAE; DiT can drop before decode). Sequential
|
||||
load is the H3-only fallback when lazy is off; do not set
|
||||
`engine.offload.lazy_module_load` to false here (`--no-lazy-module-load` in
|
||||
`basic_fasth3.py` and `basic_minimax_h3_t2v.py`). Geometry scalars come from checkpoint
|
||||
`config.json`, not live weights. See [Offloading](../../inference/offloading.md).
|
||||
- **FastH3 TAEH3** (`--video-decode-backend taeh3`) is an opt-in preview decoder.
|
||||
T2VA never materializes the 9.7 GiB video VAE (DiT still loads after Qwen via
|
||||
sequential start). On this box, alpine 768×1344×124 decoded in **2.4 s** versus
|
||||
**68 s** for the full VAE, and one T2VA generation finished in **224 s**
|
||||
end-to-end. Reconstruction is approximate, not lossless. FL2VA/Ref2VA still
|
||||
need the full VAE to encode references.
|
||||
- **Two Sparks, one clip.** Sequence parallel (`sp_size=2`) over the QSFP RoCE
|
||||
link ran one 768×1344×124 FastH3 recipe in **292 s** vs **374–393 s** on
|
||||
one GB10, and a 345-frame (~14.4 s) clip in **587 s**. Other heights, widths,
|
||||
and frame counts are valid. Weights stay replicated, so lazy module load
|
||||
(auto on GB10) is still required on each box. Bring-up and knobs:
|
||||
[Pair two NVIDIA DGX Sparks](spark_pair.md).
|
||||
- A worker's SIGTERM log and traceback show where it was interrupted, not why it
|
||||
was selected; confirm the cause in the `earlyoom` service or system logs. A
|
||||
later SIGKILL or kernel OOM kill cannot be caught and reported by Python.
|
||||
|
||||
## Gotchas specific to the GB10
|
||||
|
||||
@@ -187,10 +204,11 @@ A few things that surprise people on this box (beyond the memory notes above):
|
||||
build recent enough to include its `transformers`-compatibility handling before
|
||||
running it.
|
||||
- **MiniMax H3 worker init can look healthy and still die on the first generate**
|
||||
if sequential load is off (`--no-h3-sequential-load`, or auto-off on a
|
||||
misclassified device) and encoder, VAE, and DiT load together. Confirm the log
|
||||
contains `Released MiniMax-H3 text encoder after conditioning` before
|
||||
`Loading MiniMax-H3 denoise modules`.
|
||||
if deferred loading is off (`engine.offload.lazy_module_load: false` and sequential also off)
|
||||
and encoder, VAE, and DiT load together. On GB10 the log should show
|
||||
`lazy_module_load owns deferral` (or, if lazy is off, sequential
|
||||
`Released MiniMax-H3 text encoder after conditioning` before
|
||||
`Loading MiniMax-H3 denoise modules`).
|
||||
|
||||
## Reproduce these numbers
|
||||
|
||||
|
||||
@@ -210,7 +210,7 @@ from fastvideo.pipelines.stages import (
|
||||
InputValidationStage, CLIPTextEncodingStage, TimestepPreparationStage,
|
||||
LatentPreparationStage, DenoisingStage, DecodingStage
|
||||
)
|
||||
from fastvideo.fastvideo_args import FastVideoArgs
|
||||
from fastvideo.api.resolution import ResolvedGeneratorConfig
|
||||
from fastvideo.pipelines.pipeline_batch_info import ForwardBatch
|
||||
import torch
|
||||
|
||||
@@ -226,11 +226,11 @@ class MyCustomPipeline(ComposedPipelineBase):
|
||||
def required_config_modules(self) -> List[str]:
|
||||
return self._required_config_modules
|
||||
|
||||
def initialize_pipeline(self, fastvideo_args: FastVideoArgs):
|
||||
def initialize_pipeline(self, resolved_config: ResolvedGeneratorConfig):
|
||||
"""Initialize pipeline-specific components."""
|
||||
pass
|
||||
|
||||
def create_pipeline_stages(self, fastvideo_args: FastVideoArgs):
|
||||
def create_pipeline_stages(self, resolved_config: ResolvedGeneratorConfig):
|
||||
"""Set up pipeline stages with proper dependency injection."""
|
||||
self.add_stage(
|
||||
stage_name="input_validation_stage",
|
||||
@@ -294,7 +294,7 @@ class MyCustomStage(PipelineStage):
|
||||
self.custom_module = custom_module
|
||||
self.other_param = other_param
|
||||
|
||||
def forward(self, batch: ForwardBatch, fastvideo_args: FastVideoArgs) -> ForwardBatch:
|
||||
def forward(self, batch: ForwardBatch, resolved_config: ResolvedGeneratorConfig) -> ForwardBatch:
|
||||
# Access input data
|
||||
input_data = batch.some_attribute
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ to FastVideo model classes. Two discovery mechanisms:
|
||||
`fastvideo/models/` and parses each `.py` file's AST looking for an
|
||||
`EntryClass` variable assignment. Discovered models take priority over
|
||||
hardcoded entries. For example,
|
||||
`fastvideo/models/dits/wanvideo.py` exports
|
||||
`fastvideo/models/wan/transformer.py` exports
|
||||
`EntryClass = WanTransformer3DModel`.
|
||||
|
||||
Both feed into a unified `_FAST_VIDEO_MODELS` dict, which populates the
|
||||
@@ -86,7 +86,7 @@ from `model_index.json`.
|
||||
|
||||
```
|
||||
PipelineConfig (fastvideo/configs/pipelines/base.py)
|
||||
├── WanT2V480PConfig (fastvideo/configs/pipelines/wan.py)
|
||||
├── WanT2V480PConfig (fastvideo/models/wan/pipeline_config.py)
|
||||
│ ├── WanT2V720PConfig
|
||||
│ └── WanI2V480PConfig
|
||||
├── HunyuanConfig (fastvideo/configs/pipelines/hunyuan.py)
|
||||
@@ -103,10 +103,19 @@ PipelineConfig (fastvideo/configs/pipelines/base.py)
|
||||
- Precision settings: `dit_precision`, `vae_precision`,
|
||||
`text_encoder_precisions`.
|
||||
|
||||
These generation and precision attributes hold the model defaults. Resolution
|
||||
copies them into the typed fields of the resolved config (`pipeline.flow_shift`,
|
||||
`engine.precision.dit`, ...), and runtime code reads the typed fields.
|
||||
|
||||
Model-specific subclasses override defaults. For example,
|
||||
`WanT2V480PConfig` sets `flow_shift=3.0` and uses `WanVideoConfig` as
|
||||
its DiT config.
|
||||
|
||||
Wan's `models/wan/definition.py` links each registered variant to its pipeline
|
||||
config and sampling preset. The shared registry consumes these definitions
|
||||
without changing detector precedence or checkpoint/override-based pipeline
|
||||
selection. `configs/pipelines/wan.py` remains a compatibility import.
|
||||
|
||||
### ModelConfig / ArchConfig (`fastvideo/configs/models/base.py`)
|
||||
|
||||
`ModelConfig` wraps an `ArchConfig` using `__getattr__` proxy — attribute
|
||||
@@ -122,9 +131,11 @@ Concrete hierarchy: `DiTConfig` → `DiTArchConfig`, `VAEConfig` →
|
||||
|
||||
- `PipelineConfig.from_pretrained(model_path)` — resolves config class
|
||||
via `get_pipeline_config_cls_from_name()`, instantiates with defaults.
|
||||
- `PipelineConfig.from_kwargs(kwargs)` — resolves class, optionally loads
|
||||
JSON via `load_from_json()`, then applies CLI overrides via
|
||||
`update_config_from_dict()`.
|
||||
- `PipelineConfig.from_source(model_path, source)` — resolves the registry
|
||||
class of `model_path`, then updates it from `source`: a JSON path loaded
|
||||
via `load_from_json()`, a mapping of field values applied via
|
||||
`update_pipeline_config()`, or a `PipelineConfig` that replaces the
|
||||
registry instance.
|
||||
- `dump_to_json()` / `load_from_json()` — JSON persistence. Callable
|
||||
fields and `arch_config` are excluded from dumps.
|
||||
|
||||
@@ -142,7 +153,7 @@ sp = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
|
||||
### ComponentLoader (`fastvideo/models/loader/component_loader.py`)
|
||||
|
||||
Abstract base with a `load(model_path, fastvideo_args)` method.
|
||||
Abstract base with a `load(model_path, resolved_config)` method.
|
||||
`ComponentLoader.for_module_type(module_type, library)` is a factory
|
||||
that dispatches to specialized loaders via a `module_loaders` dict:
|
||||
|
||||
@@ -162,7 +173,7 @@ that dispatches to specialized loaders via a `module_loaders` dict:
|
||||
`TransformerLoader` reads `config.json` from the component directory,
|
||||
resolves the class via `ModelRegistry.resolve_model_cls()`, instantiates
|
||||
the model, and loads safetensors weights. CPU offload and layerwise
|
||||
offload are applied based on `FastVideoArgs`.
|
||||
offload are applied based on `resolved_config.engine.offload`.
|
||||
|
||||
Unknown module types fall back to `GenericComponentLoader`.
|
||||
|
||||
@@ -200,14 +211,14 @@ loading by calling `ComponentLoader.for_module_type()` then `.load()`.
|
||||
|
||||
Abstract base class using the Template Method pattern:
|
||||
|
||||
- `__call__(batch, fastvideo_args)` — orchestrates verification, timing,
|
||||
- `__call__(batch, resolved_config)` — orchestrates verification, timing,
|
||||
and error handling. Not overridden by subclasses.
|
||||
- `forward(batch, fastvideo_args) -> ForwardBatch` — abstract, contains
|
||||
- `forward(batch, resolved_config) -> ForwardBatch` — abstract, contains
|
||||
the stage logic.
|
||||
- `verify_input()` / `verify_output()` — optional hooks returning
|
||||
`VerificationResult`. Default: no checks.
|
||||
|
||||
When `fastvideo_args.enable_stage_verification` is `True`, `__call__`
|
||||
When `resolved_config.engine.enable_stage_verification` is `True`, `__call__`
|
||||
runs input verification before `forward()` and output verification after.
|
||||
When `envs.FASTVIDEO_STAGE_LOGGING` is set, execution time is measured
|
||||
with `torch.cuda.synchronize()` and logged.
|
||||
@@ -256,6 +267,16 @@ Specialized variants: `CausalDenoisingStage`, `LTX2DenoisingStage`,
|
||||
`SRDenoisingStage`, `LTX2AudioDecodingStage`, `SD35ConditioningStage`,
|
||||
`LTX2TextEncodingStage`, `LTX2LatentPreparationStage`.
|
||||
|
||||
Wan owns its sampling recipes under `basic/wan/stages/`. `WanDenoisingStage`
|
||||
specializes input packing, expert selection, timesteps, and first-frame
|
||||
restoration around the shared dense loop. `WanFirstFrameEncodingStage`
|
||||
produces normalized `ForwardBatch.first_frame_latent` before sampling; the
|
||||
sampler no longer executes a VAE. Dense DMD and the two causal samplers have
|
||||
family-local implementations and explicit scheduler ownership. Standard and
|
||||
DMD causal sampling share cache allocation, not their sampling algorithm.
|
||||
Legacy imports from `stages/` remain compatibility aliases. Sampling invariants
|
||||
also live beside the code in `fastvideo/pipelines/basic/wan/AGENTS.md`.
|
||||
|
||||
### Verification System (`fastvideo/pipelines/stages/validators.py`)
|
||||
|
||||
`StageValidators` (aliased as `V`) provides static validators:
|
||||
@@ -280,7 +301,7 @@ provides detailed error messages. Failed verification raises
|
||||
|
||||
Abstract base for all inference pipelines. Lifecycle:
|
||||
|
||||
1. **`__init__(model_path, fastvideo_args)`** — initializes distributed
|
||||
1. **`__init__(model_path, resolved_config)`** — initializes distributed
|
||||
environment via `maybe_init_distributed_environment_and_model_parallel
|
||||
(tp_size, sp_size)`, then calls `load_modules()` to populate
|
||||
`self.modules`.
|
||||
@@ -288,7 +309,7 @@ Abstract base for all inference pipelines. Lifecycle:
|
||||
setup), `create_pipeline_stages()` (abstract — subclasses wire stages),
|
||||
optionally applies `torch.compile` to transformers, and calls
|
||||
`warmup_sequence_parallel_communication()`.
|
||||
3. **`forward(batch, fastvideo_args)`** — iterates `self.stages` calling
|
||||
3. **`forward(batch, resolved_config)`** — iterates `self.stages` calling
|
||||
each stage in order. Decorated with `@torch.no_grad()`.
|
||||
|
||||
Key class attributes:
|
||||
@@ -301,8 +322,9 @@ Key methods:
|
||||
- `add_stage(name, stage)` — appends to `_stages` list and
|
||||
`_stage_name_mapping` dict, also sets attribute on `self`.
|
||||
- `get_module(name, default)` — retrieves a loaded module.
|
||||
- `from_pretrained(model_path, **kwargs)` — class method constructing
|
||||
`FastVideoArgs` and calling `cls(...)` then `post_init()`.
|
||||
- `from_pretrained(model_path, *, resolved_config)` — class method that
|
||||
builds the pipeline from a resolved config (from
|
||||
`resolve_inference_config({...})`) by calling `cls(...)` then `post_init()`.
|
||||
|
||||
### LoRAPipeline (`fastvideo/pipelines/lora_pipeline.py`)
|
||||
|
||||
@@ -337,12 +359,12 @@ Key APIs: `get_tp_rank()`, `get_tp_world_size()`, `get_sp_rank()`,
|
||||
`warmup_sequence_parallel_communication()` pre-warms NCCL communicators
|
||||
to avoid slow first forward passes.
|
||||
|
||||
Usage: `torchrun --nproc-per-node=N -m fastvideo.entrypoints.cli.main
|
||||
generate --model-path ... --tp-size N --sp-size M`.
|
||||
Usage: `fastvideo generate --config run.yaml
|
||||
--generator.engine.parallelism.tp_size N --generator.engine.parallelism.sp_size M`.
|
||||
|
||||
### torch.compile Integration
|
||||
|
||||
When `fastvideo_args.enable_torch_compile` is `True`,
|
||||
When `resolved_config.engine.compile.enabled` is `True`,
|
||||
`_maybe_compile_pipeline_module()` checks for a `_compile_conditions`
|
||||
attribute on the module. If present, only matching submodules are
|
||||
compiled. Otherwise, the entire module is compiled. FSDP-wrapped
|
||||
@@ -354,47 +376,53 @@ modules are skipped.
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_path="Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
num_gpus=1, tp_size=1, sp_size=1,
|
||||
)
|
||||
result = generator.generate_video(
|
||||
prompt="A cat dancing",
|
||||
height=720, width=1280, num_frames=81,
|
||||
"Wan-AI/Wan2.1-T2V-14B-Diffusers",
|
||||
{"engine": {"num_gpus": 1, "parallelism": {"tp_size": 1, "sp_size": 1}}},
|
||||
)
|
||||
result = generator.generate({
|
||||
"prompt": "A cat dancing",
|
||||
"sampling": {"height": 720, "width": 1280, "num_frames": 81},
|
||||
})
|
||||
```
|
||||
|
||||
**CLI** (`fastvideo/entrypoints/cli/`):
|
||||
|
||||
```bash
|
||||
# run.yaml holds `generator: {model_path: Wan-AI/Wan2.1-T2V-14B-Diffusers}`.
|
||||
fastvideo generate \
|
||||
--model-path "Wan-AI/Wan2.1-T2V-14B-Diffusers" \
|
||||
--prompt "A cat dancing" \
|
||||
--num-gpus 1
|
||||
--config run.yaml \
|
||||
--request.prompt "A cat dancing" \
|
||||
--generator.engine.num_gpus 1
|
||||
```
|
||||
|
||||
**FastVideoArgs** (`fastvideo/fastvideo_args.py`): Central args dataclass.
|
||||
Key fields: `model_path`, `mode` (`ExecutionMode`), `workload_type`
|
||||
(`WorkloadType`), `pipeline_config` (`PipelineConfig`), `num_gpus`,
|
||||
`tp_size`, `sp_size`, `lora_path`, `dit_cpu_offload`,
|
||||
`dit_layerwise_offload`, `enable_torch_compile`,
|
||||
`enable_stage_verification`.
|
||||
**ResolvedGeneratorConfig** (`fastvideo/api/resolution.py`): The frozen
|
||||
runtime config that the executor, workers, pipelines, stages, and loaders
|
||||
read. Key paths: `model_path`, `mode` (`ExecutionMode`),
|
||||
`pipeline.workload_type` (`WorkloadType`), `engine.num_gpus`,
|
||||
`engine.parallelism.tp_size`, `engine.parallelism.sp_size`,
|
||||
`pipeline.components.lora_path`, `engine.offload.dit`,
|
||||
`engine.offload.dit_layerwise`, `engine.compile.enabled`,
|
||||
`engine.enable_stage_verification`, and `pipeline_config` (the frozen
|
||||
`PipelineConfig`).
|
||||
|
||||
Constructed via `FastVideoArgs.from_kwargs(**kwargs)` which resolves the
|
||||
`PipelineConfig` from the registry, applies JSON config if provided, and
|
||||
merges CLI overrides.
|
||||
Built by `resolve_inference_config(config)`
|
||||
(`fastvideo/api/inference_resolution.py`), which runs the named resolution
|
||||
steps (environment variables, model defaults, derived values, validation) in
|
||||
order, records each decision, and then builds the `PipelineConfig` from the
|
||||
registry, applies a JSON config if provided, and freezes it.
|
||||
|
||||
## End-to-End Inference Flow
|
||||
|
||||
```
|
||||
User: VideoGenerator.from_pretrained(model_path, **kwargs)
|
||||
User: VideoGenerator.from_pretrained(model_path, config)
|
||||
│
|
||||
├─ FastVideoArgs.from_kwargs() → PipelineConfig resolved via registry
|
||||
├─ resolve_inference_config() → PipelineConfig resolved via registry
|
||||
├─ get_model_info() → ModelInfo(pipeline_cls, sampling_param_cls, ...)
|
||||
│ ├─ model_index.json read → _class_name extracted
|
||||
│ ├─ pipeline_registry resolves pipeline_cls from _class_name
|
||||
│ └─ config_registry resolves config classes from model_path
|
||||
│
|
||||
├─ pipeline_cls.__init__(model_path, fastvideo_args)
|
||||
├─ pipeline_cls.__init__(model_path, resolved_config)
|
||||
│ ├─ maybe_init_distributed(tp_size, sp_size)
|
||||
│ └─ load_modules() → reads model_index.json, loads each component
|
||||
│ ├─ ComponentLoader.for_module_type() → specialized loader
|
||||
@@ -406,10 +434,10 @@ User: VideoGenerator.from_pretrained(model_path, **kwargs)
|
||||
├─ torch.compile (if enabled)
|
||||
└─ warmup_sequence_parallel_communication()
|
||||
|
||||
User: generator.generate_video(prompt, ...)
|
||||
User: generator.generate(request)
|
||||
│
|
||||
├─ ForwardBatch constructed from SamplingParam + user args
|
||||
└─ pipeline.forward(batch, fastvideo_args)
|
||||
└─ pipeline.forward(batch, resolved_config)
|
||||
├─ InputValidationStage → validates dims
|
||||
├─ TextEncodingStage → prompt → embeddings
|
||||
├─ ConditioningStage → prepares conditioning
|
||||
|
||||
@@ -8,64 +8,75 @@ FastVideo automatically distributes the generation process when multiple GPUs ar
|
||||
# Will use 4 GPUs in parallel for faster generation
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=4,
|
||||
{"engine": {"num_gpus": 4}},
|
||||
)
|
||||
```
|
||||
|
||||
One node uses the multiprocessing executor (`execution_backend: mp`, the
|
||||
default). Two machines — for example two DGX Sparks, one GPU each — need Ray:
|
||||
|
||||
```yaml
|
||||
generator:
|
||||
engine:
|
||||
num_gpus: 2
|
||||
execution_backend: ray
|
||||
parallelism:
|
||||
sp_size: 2
|
||||
```
|
||||
|
||||
Set `RAY_ADDRESS` and `FASTVIDEO_HOST_IP` to the interconnect IPs, not Wi-Fi.
|
||||
The FastH3 example selects Ray automatically when `RAY_ADDRESS` is set. Full
|
||||
bring-up: [Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
## Customizing Generation
|
||||
|
||||
- `PipelineConfig`: Initialization time parameters
|
||||
- `SamplingParam`: Generation time parameters
|
||||
|
||||
You can customize generation behavior using `PipelineConfig` and
|
||||
`SamplingParam`:
|
||||
`VideoGenerator.from_pretrained(model_path, config)` takes the startup
|
||||
settings as a nested mapping at their typed config paths, such as
|
||||
`{"engine": {"num_gpus": 2, "offload": {"dit": False}}}`; it is
|
||||
`VideoGenerator.from_config` with `model_path` added to the mapping. Pass
|
||||
generation settings to `VideoGenerator.generate` as a request:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator, SamplingParam, PipelineConfig
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
def main():
|
||||
model_name = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers"
|
||||
config = PipelineConfig.from_pretrained(model_name)
|
||||
config.vae_precision = "fp16"
|
||||
|
||||
# Create the generator
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_name,
|
||||
num_gpus=1,
|
||||
dit_layerwise_offload=True, # FastVideoArgs option
|
||||
pipeline_config=config
|
||||
)
|
||||
|
||||
# Create and customize sampling parameters
|
||||
sampling_param = SamplingParam.from_pretrained("Wan-AI/Wan2.1-T2V-1.3B-Diffusers")
|
||||
|
||||
# How many frames to generate
|
||||
sampling_param.num_frames = 45
|
||||
|
||||
# Video resolution (width, height)
|
||||
sampling_param.width = 1024
|
||||
sampling_param.height = 576
|
||||
|
||||
# How many steps we denoise the video (higher = better quality, slower generation)
|
||||
sampling_param.num_inference_steps = 30
|
||||
|
||||
# How strongly the video conforms to the prompt (higher = more faithful to prompt)
|
||||
sampling_param.guidance_scale = 7.5
|
||||
|
||||
# Random seed for reproducibility
|
||||
sampling_param.seed = 42 # Optional, leave unset for random results
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": model_name,
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"offload": {"dit_layerwise": True},
|
||||
"precision": {"vae": "fp16"},
|
||||
},
|
||||
})
|
||||
|
||||
# Generate video with custom parameters
|
||||
prompt = "A beautiful sunset over a calm ocean, with gentle waves."
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
sampling_param=sampling_param,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"sampling": {
|
||||
# How many frames to generate
|
||||
"num_frames": 45,
|
||||
# Video resolution (width, height)
|
||||
"width": 1024,
|
||||
"height": 576,
|
||||
# How many steps we denoise the video (higher = better quality, slower generation)
|
||||
"num_inference_steps": 30,
|
||||
# How strongly the video conforms to the prompt (higher = more faithful to prompt)
|
||||
"guidance_scale": 7.5,
|
||||
# Random seed for reproducibility
|
||||
"seed": 42, # Optional, leave unset for random results
|
||||
},
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
|
||||
# If return_frames=True, frames are available in video["frames"]
|
||||
print(f"Generated {len(video['frames'])} frames")
|
||||
# If return_frames=True, frames are available in video.frames
|
||||
print(f"Generated {len(video.frames)} frames")
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -109,6 +120,26 @@ Override individual values from the CLI with dotted paths:
|
||||
fastvideo generate --config config.yaml --request.sampling.seed 42
|
||||
```
|
||||
|
||||
## Where a Value Came From
|
||||
|
||||
FastVideo resolves the generator config once at startup and records the source of every value: the input config
|
||||
(`input`; `explicit` tells whether you wrote the value or it is the schema default), a `FASTVIDEO_*` environment
|
||||
variable, the model's defaults, or a derived value. A worker's device policy and values read from checkpoint files
|
||||
are recorded too.
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_config(config)
|
||||
generator.resolved_config.provenance("engine.parallelism.sp_size")
|
||||
# PathProvenance(path='engine.parallelism.sp_size', value=2, source='derive_parallel_sizes', ...)
|
||||
|
||||
result = generator.generate(request)
|
||||
result.resolved_request.provenance("sampling.num_frames")
|
||||
# PathProvenance(..., value=81, source='fill_sampling_defaults[preset wan_t2v_1_3b]', explicit=False)
|
||||
```
|
||||
|
||||
`resolved_config.provenance_table()` lists every path. Every value is decided before resolution ends, including the
|
||||
device offload policy and the checkpoint defaults; after that, `resolved_config` is read-only.
|
||||
|
||||
## Performance Optimization
|
||||
|
||||
For configuring optimizations, please see our [optimizations guide](optimizations.md)
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
# FastH3 distilled checkpoint schedules
|
||||
|
||||
Base MiniMax-H3 still uses the scheduler shifts in its checkpoint (video 12,
|
||||
audio 3), BF16 text encoding, and the existing uniform schedule. The default
|
||||
`basic_fasth3.py` example still targets the four-forward preview. Selecting a
|
||||
shift-10 eight-forward checkpoint is an explicit choice of model and recipe;
|
||||
it does not change either default or enable NVFP4.
|
||||
|
||||
## Eight-forward T2AV recipe
|
||||
|
||||
The public checkpoint is
|
||||
[`FastVideo/FastVideo-FastH3-8-Step-V2`](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2)
|
||||
(MiniMax H3 Community License), trained with video/audio shifts 10/3, VSA
|
||||
sparsity 0.8, 64-token tiles, and the DMD rungs
|
||||
`[999, 874, 749, 624, 500, 375, 250, 125]`. `basic_fasth3_8step.py` pins that
|
||||
checkpoint and recipe as defaults; it shares the preview example's CLI, so every
|
||||
other flag works unchanged:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_fasth3_8step.py \
|
||||
--prompt 'A slow cinematic drone shot glides over a coastal town; gulls call over the harbor.' \
|
||||
--num-gpus 4 --vsa-kernel sm100a \
|
||||
--profile strict --no-inference-torch-compile --no-compile-vae \
|
||||
--height 768 --width 1344 --num-frames 124 \
|
||||
--output outputs/fasth3-8step
|
||||
```
|
||||
|
||||
Pass `--model-path` to use a local snapshot of the full export (not just its
|
||||
`transformer` subdirectory). `--steps` is the number of sigma-grid points,
|
||||
including the terminal zero; nine points run exactly eight transformer
|
||||
forwards, and the script rejects any other value because the checkpoint's
|
||||
ladder has eight rungs. The rungs are unshifted noise levels on the 1000-step training
|
||||
clock; each scheduler applies its own shift once, and the transformer receives
|
||||
H3 clean-time values (`1 - sigma`). A uniform nine-point grid is not a substitute
|
||||
for those rungs.
|
||||
|
||||
Compilation and H3 fusions are disabled above to establish an eager reference;
|
||||
they can be evaluated separately. On hardware without the sm100a extension,
|
||||
use `--vsa-kernel triton`; compare outputs and performance before adopting that
|
||||
backend. This recipe is T2AV-only, not a distilled `transformer_ref` model.
|
||||
|
||||
## Export metadata and validation
|
||||
|
||||
The export's `fastvideo_inference.json` supplies the trained ladder. The
|
||||
schedule fields of `fasth3-inference-contract-v1` are:
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": "fasth3-inference-contract-v1",
|
||||
"dmd_denoising_steps": [999, 874, 749, 624, 500, 375, 250, 125],
|
||||
"num_inference_steps": 9,
|
||||
"transformer_forwards": 8,
|
||||
"video_scheduler_shift": 10.0,
|
||||
"audio_scheduler_shift": 3.0
|
||||
}
|
||||
```
|
||||
|
||||
The loader keeps this file when downloading the selected H3 components from
|
||||
Hugging Face. It checks that the two declared shifts agree with
|
||||
`scheduler/scheduler_config.json` and `audio_scheduler/scheduler_config.json`.
|
||||
Missing/invalid rungs, inconsistent counts, or an explicit conflicting ladder
|
||||
are errors. The denoiser rejects a request with the wrong number of grid points.
|
||||
The metadata does not silently change request dimensions, step count, attention
|
||||
backend, sparsity, precision, or offload/compile settings: set those explicitly
|
||||
as above.
|
||||
|
||||
For exports without this sidecar, an explicit ladder is supported via
|
||||
`MiniMaxH3PipelineConfig.dmd_denoising_steps`, or through the typed API's
|
||||
`PipelineSelection(dmd_denoising_steps=[...])`. The shifts
|
||||
still come from the checkpoint scheduler configs. Keep generic `flow_shift`
|
||||
unset: H3 has separate video and audio shifts, not one shared shift.
|
||||
|
||||
This documents execution support for the published checkpoint. It is not a
|
||||
quality claim: compare video/audio output against base MiniMax-H3 on your own
|
||||
prompts before adopting it.
|
||||
|
||||
## Apple Silicon
|
||||
|
||||
`mlx_fasth3.py` stays on FastH3 V1 and its uniform AdaLN cache.
|
||||
`mlx_fasth3_8step.py` is the eight-forward MLX recipe for FastH3 V2. It
|
||||
reads the same `fastvideo_inference.json` rungs and shifts, and it expects an
|
||||
MLX DiT whose AdaLN cache was converted from that contract. Reuse the V1
|
||||
snapshot's VAE, audio VAE, text encoder, and tokenizer; only the DiT and the
|
||||
sidecar change. Rank-reduced AdaLN checkpoints are unchanged and are not
|
||||
produced by the MLX converter.
|
||||
|
||||
Install is in the
|
||||
[MLX install guide](../getting_started/installation/mlx.md). Generation and
|
||||
serving are in the [MiniMax H3 cookbook](../cookbook/minimax-h3.md) and the
|
||||
[H3 server guide](../cookbook/openai-api.md).
|
||||
@@ -5,10 +5,10 @@ This page contains step-by-step instructions to get you quickly started with vid
|
||||
## Requirements
|
||||
|
||||
- **OS**: Linux (tested on Ubuntu 22.04+), or macOS on Apple silicon via the
|
||||
[MPS installation guide](../getting_started/installation/mps.md)
|
||||
[MLX install guide](../getting_started/installation/mlx.md)
|
||||
- **Python**: 3.10-3.12
|
||||
- **CUDA**: 12.6 or 13.0 (NVIDIA GPUs)
|
||||
- **GPU**: At least one NVIDIA GPU, or an Apple silicon chip with MPS
|
||||
- **GPU**: At least one NVIDIA GPU, or an Apple silicon chip with the MLX runtime
|
||||
|
||||
## Installation
|
||||
|
||||
@@ -38,18 +38,20 @@ def main():
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
{"engine": {"num_gpus": 1}}, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
# Define a prompt for your video
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
|
||||
|
||||
# Generate the video
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -75,23 +77,24 @@ Please see the [support matrix](support_matrix.md) for the list of supported mod
|
||||
You can generate a video starting from an initial image:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator, SamplingParam
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
def main():
|
||||
# Create the generator
|
||||
model_name = "Wan-AI/Wan2.1-I2V-14B-480P-Diffusers"
|
||||
generator = VideoGenerator.from_pretrained(model_name, num_gpus=1)
|
||||
|
||||
# Set up parameters with an initial image
|
||||
sampling_param = SamplingParam.from_pretrained(model_name)
|
||||
sampling_param.image_path = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg"
|
||||
sampling_param.num_frames = 107
|
||||
generator = VideoGenerator.from_pretrained(model_name, {"engine": {"num_gpus": 1}})
|
||||
|
||||
# Generate video based on the image
|
||||
prompt = "A photograph coming to life with gentle movement"
|
||||
generator.generate_video(prompt, sampling_param=sampling_param,
|
||||
output_path="my_videos/",
|
||||
save_video=True)
|
||||
generator.generate({
|
||||
"prompt": prompt,
|
||||
# Set up parameters with an initial image
|
||||
"inputs": {
|
||||
"image_path": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/diffusers/astronaut.jpg",
|
||||
},
|
||||
"sampling": {"num_frames": 107},
|
||||
"output": {"output_path": "my_videos/", "save_video": True},
|
||||
})
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -106,12 +109,12 @@ Common issues and their solutions:
|
||||
If you encounter CUDA out of memory errors:
|
||||
|
||||
- Reduce `num_frames` or video resolution
|
||||
- Enable FastVideo offloading options such as `dit_layerwise_offload=True`
|
||||
(single GPU) or `use_fsdp_inference=True` (multi-GPU)
|
||||
- Enable FastVideo offloading options such as `engine.offload.dit_layerwise: true`
|
||||
(single GPU) or `engine.use_fsdp_inference: true` (multi-GPU)
|
||||
- Try a smaller model or use distilled versions
|
||||
- Use `num_gpus` > 1 if multiple GPUs are available
|
||||
- Try enabling FSDP inference with `use_fsdp_inference=True` (may slow down generation)
|
||||
- Try enabling DiT layerwise offload with `dit_layerwise_offload=True` (now only a few models support this, but may introduce less overhead than FSDP)
|
||||
- Use `engine.num_gpus` > 1 if multiple GPUs are available
|
||||
- Try enabling FSDP inference with `engine.use_fsdp_inference: true` (may slow down generation)
|
||||
- Try enabling DiT layerwise offload with `engine.offload.dit_layerwise: true` (now only a few models support this, but may introduce less overhead than FSDP)
|
||||
|
||||
### Slow Generation
|
||||
|
||||
|
||||
@@ -10,7 +10,9 @@ that tradeoff. It is not a lossless acceleration of the full VAE.
|
||||
|
||||
## Generate a video
|
||||
|
||||
Use your existing MLX FastH3 environment and converted checkpoint:
|
||||
Use your existing MLX FastH3 environment and converted checkpoint. The same
|
||||
`--video-decode-backend taeh3` flag works on `mlx_fasth3.py` (V1) and
|
||||
`mlx_fasth3_8step.py` (V2):
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
@@ -68,7 +70,7 @@ Run the numerical tests against a local TAEHV checkout containing the released
|
||||
weights:
|
||||
|
||||
```bash
|
||||
TAEH3_REFERENCE_DIR=/path/to/taehv \
|
||||
FASTVIDEO_TEST_TAEH3_REFERENCE_DIR=/path/to/taehv \
|
||||
python -m pytest fastvideo/tests/mlx/test_mlx_taeh3.py -q
|
||||
```
|
||||
|
||||
|
||||
+125
-45
@@ -4,14 +4,17 @@ This page describes how to use offloading techniques for inference to reduce GPU
|
||||
|
||||
## Default Behavior
|
||||
|
||||
```python
|
||||
dit_cpu_offload: bool = True
|
||||
use_fsdp_inference: bool = False
|
||||
dit_layerwise_offload: bool = True
|
||||
text_encoder_cpu_offload: bool = True
|
||||
image_encoder_cpu_offload: bool = True
|
||||
vae_cpu_offload: bool = True
|
||||
pin_cpu_memory: bool = True
|
||||
```yaml
|
||||
engine:
|
||||
use_fsdp_inference: false
|
||||
offload:
|
||||
dit: true # dit_cpu_offload
|
||||
dit_layerwise: true # dit_layerwise_offload
|
||||
text_encoder: true # text_encoder_cpu_offload
|
||||
image_encoder: true # image_encoder_cpu_offload
|
||||
vae: true # vae_cpu_offload
|
||||
pin_cpu_memory: true
|
||||
lazy_module_load: null # auto
|
||||
```
|
||||
|
||||
On unified-memory accelerators such as NVIDIA GB10 and Apple silicon, FastVideo
|
||||
@@ -21,23 +24,33 @@ pool there, so offload adds transfers and duplicate residency instead of freeing
|
||||
memory. CUDA FSDP sharding remains enabled when requested; MPS continues to
|
||||
disable FSDP. `pin_cpu_memory` is not an offload mode and is left unchanged.
|
||||
|
||||
MiniMax H3 CUDA inference can use a second lever that does not copy weights to a
|
||||
host pool: load the Qwen3-VL text encoder, run conditioning, then release that
|
||||
encoder before loading the DiT and video/audio VAEs. `h3_sequential_load`
|
||||
defaults to auto (`None`): on for unified-memory devices such as GB10, off on
|
||||
discrete GPUs. Pass `--h3-sequential-load` to force it, or
|
||||
`--no-h3-sequential-load` to keep the encoder resident for later `generate()`
|
||||
calls on the same worker. Sequential load currently cannot re-encode a new prompt
|
||||
on that worker; start a new generator until prompt-cache reload exists. The MLX
|
||||
FastH3 runtime always uses this phase order. When host offload is off, DiT
|
||||
safetensors are read onto the accelerator instead of CPU-then-copy.
|
||||
Input-preparation geometry (spatial ratio, latent channels, audio sample rate)
|
||||
comes from the VAE arch configs until those weights load.
|
||||
MiniMax H3 CUDA inference can use two levers that do not copy weights to a host
|
||||
pool. `lazy_module_load` is the general path: each opted-in component loads on
|
||||
first use and is freed after its last stage, so a later `generate()` reloads
|
||||
from disk in-process and the DiT can drop before VAE decode. On GB10 it
|
||||
auto-enables and owns deferral. `h3_sequential_load` is the H3-only fallback
|
||||
when lazy is off: load Qwen3-VL, run conditioning, release that encoder, then
|
||||
load the DiT and VAEs. When both would arm, sequential stands down so VAE
|
||||
`torch.compile` can attach to the lazy proxy. Input preparation and unpatchify
|
||||
read geometry from checkpoint `config.json` (VAE spatial ratio / latent
|
||||
channels, DiT patch size) so those stages do not materialize weights just to
|
||||
read two integers. The MLX FastH3 runtime always uses this phase order. When
|
||||
host offload is off, DiT safetensors are read onto the accelerator instead of
|
||||
CPU-then-copy. Both flags default to auto (`None`) and turn on for
|
||||
unified-memory devices such as GB10; lazy then disables sequential. Set
|
||||
`engine.offload.lazy_module_load: false` to keep every component resident (sequential may still
|
||||
auto-arm). Two-node Spark
|
||||
jobs still need this split: sequence parallel replicates the DiT on each GB10
|
||||
(~66 GiB of weights plus activations). See
|
||||
[Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
## Behavior Explanation
|
||||
|
||||
!!! note
|
||||
For CLI usage, replace underscores (`_`) with hyphens (`-`).
|
||||
`VideoGenerator.from_pretrained` accepts the option names below as keywords, except `lazy_module_load` and
|
||||
`h3_sequential_load`. In a YAML config or a dotted override, each option is a typed field:
|
||||
`engine.use_fsdp_inference`, `engine.offload.<field>` as listed in the defaults above, and
|
||||
`pipeline.model.minimax_h3.sequential_load` for `h3_sequential_load`.
|
||||
|
||||
### `use_fsdp_inference`
|
||||
|
||||
@@ -93,9 +106,12 @@ because the encoder has been released.
|
||||
|
||||
#### Usage Recommendation
|
||||
|
||||
Leave the default on Spark / DGX Spark. Force `--h3-sequential-load` only when
|
||||
you need the split on a discrete GPU. Use `--no-h3-sequential-load` when you
|
||||
need more than one prompt per worker and have enough memory to keep the encoder.
|
||||
Leave the default on Spark / DGX Spark when `lazy_module_load` is off. When
|
||||
both would arm (the GB10 auto case), lazy owns deferral and sequential stands
|
||||
down so VAE `torch.compile` can attach to the lazy proxy. Force
|
||||
`pipeline.model.minimax_h3.sequential_load: true` only when you need the split on a discrete GPU without
|
||||
lazy load. Set `pipeline.model.minimax_h3.sequential_load: false` when you need more than one prompt per
|
||||
worker and have enough memory to keep the encoder.
|
||||
|
||||
### `text_encoder_cpu_offload`
|
||||
|
||||
@@ -121,6 +137,52 @@ These options introduce performance overhead due to PCIe data transfer.
|
||||
|
||||
We recommend enabling these options when OOM happens.
|
||||
|
||||
### `lazy_module_load`
|
||||
|
||||
Every option above moves weights between host and device. This one changes
|
||||
whether they are in memory at all.
|
||||
|
||||
By default a pipeline loads every component before the first stage runs, so
|
||||
peak memory is the sum of all of them even though no two are needed at the same
|
||||
moment. With `lazy_module_load` enabled, each heavy component loads on first use
|
||||
and is freed once the last stage that needs it has returned, so peak memory
|
||||
becomes the largest overlapping set instead of the sum. MiniMax-H3 T2VA is
|
||||
`max(text encoder, DiT, VAE)` rather than `text encoder + DiT + VAE`, because
|
||||
the DiT is not held through VAE decode.
|
||||
|
||||
#### Performance Impact
|
||||
|
||||
A freed component is read from disk again on the next generation, so a
|
||||
multi-prompt run pays one reload per component per request. For a large text
|
||||
encoder that is tens of seconds. If pipeline-level `torch.compile` is enabled,
|
||||
the compile setup is reapplied after each reload; PyTorch can reuse its graph
|
||||
and kernel caches when the component structure and input shapes are unchanged.
|
||||
|
||||
#### Usage Recommendation
|
||||
|
||||
Enable this when a model does not fit at load time, which the CPU offload
|
||||
options above cannot help with because they act after loading. It is
|
||||
particularly relevant on unified-memory devices, where host and device draw on
|
||||
the same pool and moving weights to the host frees nothing. FastVideo
|
||||
auto-enables it there (`lazy_module_load=None`). Leave it off when the model
|
||||
already fits, or set `engine.offload.lazy_module_load: false` to keep components resident for
|
||||
later `generate()` calls.
|
||||
|
||||
This option applies to inference only. Training keeps every component resident
|
||||
and logs a warning if the flag is set.
|
||||
|
||||
Deferral is opt-in per pipeline. Releasing a component and loading it again is
|
||||
only safe when nothing outside the loader has changed it, and two common habits
|
||||
break that without raising: mutating a component after load, as LongCat does
|
||||
when it enables block-sparse attention, and reading a component's attributes
|
||||
while stages are built, as the shared denoising stage does to pick an attention
|
||||
backend. A pipeline therefore lists the components it has checked in
|
||||
`_lazy_module_names`, which is empty in the base class. MiniMax-H3 opts in. On
|
||||
a pipeline that has not, the flag is a no-op: hooks are not installed and no
|
||||
warning is logged. Sequential MiniMax-H3 (`h3_sequential_load`) reloads the
|
||||
text encoder for a later `generate()` on the same worker; you do not need to
|
||||
start a new generator.
|
||||
|
||||
## General Recommendations
|
||||
|
||||
### Single GPU Inference
|
||||
@@ -131,6 +193,12 @@ We recommend enabling `dit_layerwise_offload`. If OOM happens, also enable `imag
|
||||
|
||||
We recommend enabling `use_fsdp_inference` and disabling both `dit_layerwise_offload` and `dit_cpu_offload`. If OOM happens, consider enabling `text_encoder_cpu_offload`, `image_encoder_cpu_offload`, and `vae_cpu_offload`. If OOM still happens, consider enabling `dit_cpu_offload`.
|
||||
|
||||
### When the Model Does Not Fit at Load Time
|
||||
|
||||
The offload options only help once loading has finished. If the run dies while
|
||||
components are still being placed, or if the machine has unified memory so
|
||||
there is no separate host pool to offload into, enable `lazy_module_load`.
|
||||
|
||||
## Examples
|
||||
|
||||
### Single GPU with Layerwise Offloading
|
||||
@@ -140,19 +208,25 @@ from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1,
|
||||
# Recommended for single GPU
|
||||
dit_layerwise_offload=True,
|
||||
# Enable if OOM happens
|
||||
vae_cpu_offload=True,
|
||||
image_encoder_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
# Speeds up CPU-GPU transfer
|
||||
pin_cpu_memory=True,
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"offload": {
|
||||
# Recommended for single GPU
|
||||
"dit_layerwise": True,
|
||||
# Enable if OOM happens
|
||||
"vae": True,
|
||||
"image_encoder": True,
|
||||
"text_encoder": True,
|
||||
# Speeds up CPU-GPU transfer
|
||||
"pin_cpu_memory": True,
|
||||
},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers."
|
||||
video = generator.generate_video(prompt, output_path="output/", save_video=True)
|
||||
video = generator.generate({"prompt": prompt, "output": {"output_path": "output/", "save_video": True}})
|
||||
```
|
||||
|
||||
### Multi-GPU with FSDP
|
||||
@@ -162,18 +236,24 @@ from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=2,
|
||||
# Recommended for multi-GPU
|
||||
use_fsdp_inference=True,
|
||||
dit_layerwise_offload=False,
|
||||
dit_cpu_offload=False,
|
||||
# Enable if OOM happens
|
||||
vae_cpu_offload=True,
|
||||
image_encoder_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True,
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 2,
|
||||
# Recommended for multi-GPU
|
||||
"use_fsdp_inference": True,
|
||||
"offload": {
|
||||
"dit_layerwise": False,
|
||||
"dit": False,
|
||||
# Enable if OOM happens
|
||||
"vae": True,
|
||||
"image_encoder": True,
|
||||
"text_encoder": True,
|
||||
"pin_cpu_memory": True,
|
||||
},
|
||||
},
|
||||
},
|
||||
)
|
||||
|
||||
prompt = "A majestic lion strides across the golden savanna."
|
||||
video = generator.generate_video(prompt, output_path="output/", save_video=True)
|
||||
video = generator.generate({"prompt": prompt, "output": {"output_path": "output/", "save_video": True}})
|
||||
```
|
||||
|
||||
@@ -7,7 +7,8 @@ This page describes the various options for speeding up generation times in Fast
|
||||
Several options on this page behave differently on the GB10's unified-memory
|
||||
hardware — some give little or nothing there. See
|
||||
[DGX Spark: Performance & Tuning](../getting_started/installation/spark_performance.md)
|
||||
for what actually helps on that platform and why.
|
||||
for what actually helps on that platform and why. Two Sparks, one clip:
|
||||
[Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
## Table of Contents
|
||||
|
||||
@@ -165,22 +166,26 @@ Enable FP4 attention via the `--nvfp4_fa4` flag:
|
||||
python examples/inference/optimizations/fp4_attn_wan2_1_1_3b.py --nvfp4_fa4
|
||||
```
|
||||
|
||||
Or in Python via the `nvfp4_fa4` kwarg (sets env vars automatically):
|
||||
Or in Python via the `engine.attention.nvfp4_fa4` field (resolution sets the env vars):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
nvfp4_fa4=True,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False, # FSDP is incompatible with FP4 pointer path
|
||||
{
|
||||
"engine": {
|
||||
"attention": {"nvfp4_fa4": True},
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False, # FSDP is incompatible with FP4 pointer path
|
||||
},
|
||||
},
|
||||
)
|
||||
gen.generate_video(prompt="A raccoon in sunflowers", save_video=True)
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
```
|
||||
|
||||
#### Known Limitations
|
||||
|
||||
- `use_fsdp_inference=True` is incompatible with the FP4 path (FSDP shards invalidate tensor pointers)
|
||||
- `engine.use_fsdp_inference: true` is incompatible with the FP4 path (FSDP shards invalidate tensor pointers)
|
||||
- Per-call cosine similarity vs BF16: ~0.99 (slight quantization error accumulates over denoising steps)
|
||||
- Only supports `headdim >= 128`
|
||||
|
||||
@@ -204,15 +209,15 @@ import os
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "ATTN_QAT_INFER"
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.layers.quantization import get_quantization_config
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1,
|
||||
# Wan-2.1 uses the nvfp4_qat config (NVFP4 is LTX2-specific). Pass an
|
||||
# instance — the bare string is not resolved on the from_pretrained path.
|
||||
transformer_quant=get_quantization_config("nvfp4_qat")(),
|
||||
use_fsdp_inference=False, # FSDP shards invalidate the FP4 tensor pointers
|
||||
)
|
||||
gen = VideoGenerator.from_config({
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False, # FSDP shards invalidate the FP4 tensor pointers
|
||||
# Wan-2.1 uses the nvfp4_qat config (NVFP4 is LTX2-specific).
|
||||
"quantization": {"transformer_quant": "nvfp4_qat"},
|
||||
},
|
||||
})
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
```
|
||||
|
||||
@@ -293,21 +298,41 @@ automatically.
|
||||
### Requirements
|
||||
|
||||
- **GPU**: sm89+ (H100, L40S, RTX 4090, or newer) for hardware FP8 compute
|
||||
- **ROCm**: CDNA4 (MI350X / MI355X, gfx950) runs the FP8 `_scaled_mm` path through
|
||||
hipBLASLt (OCP e4m3fn). MI300X (gfx942) only exposes the `fnuz` FP8 formats and
|
||||
takes the bf16 dequant fallback like a pre-sm89 GPU.
|
||||
- No additional packages required beyond the base FastVideo install
|
||||
|
||||
### Usage
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
gen = VideoGenerator.from_config({
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"engine": {"quantization": {"transformer_quant": "FP8"}}, # per-tensor (default)
|
||||
})
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
```
|
||||
|
||||
`engine.quantization.transformer_quant` takes a quantization registry name and builds that config with its default
|
||||
arguments. To pass constructor arguments, such as per-channel granularity, set the config instance on the DiT config
|
||||
through `pipeline.model.generic.dit` instead:
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.layers.quantization import get_quantization_config
|
||||
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
# Pass an instance — the bare string is not resolved on the from_pretrained path.
|
||||
transformer_quant=get_quantization_config("FP8")(), # per-tensor (default)
|
||||
# transformer_quant=get_quantization_config("FP8")(granularity="channel"), # slower, higher accuracy
|
||||
)
|
||||
gen.generate(request={"prompt": "A raccoon in sunflowers", "output": {"save_video": True}})
|
||||
gen = VideoGenerator.from_config({
|
||||
"model_path": "Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
"pipeline": {
|
||||
"model": {
|
||||
"generic": {
|
||||
"dit": {"quant_config": get_quantization_config("FP8")(granularity="channel")}, # slower, higher accuracy
|
||||
},
|
||||
},
|
||||
},
|
||||
})
|
||||
```
|
||||
|
||||
Or run the example script:
|
||||
@@ -337,7 +362,7 @@ end-to-end speedup. It is **off by default** and enabled per-run.
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
enable_torch_compile=True,
|
||||
{"engine": {"compile": {"enabled": True}}},
|
||||
)
|
||||
```
|
||||
|
||||
@@ -372,12 +397,14 @@ device is unsupported. Legacy VSA, MiniMax-H3 tile-256 VSA, and the explicit
|
||||
eager with one warning instead of failing mid-denoise.
|
||||
|
||||
```python
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"MiniMaxAI/MiniMax-H3",
|
||||
inference_torch_compile=True, # or FASTVIDEO_INFERENCE_TORCH_COMPILE=1
|
||||
)
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"engine": {"compile": {"regional": True}}, # or FASTVIDEO_INFERENCE_TORCH_COMPILE=1
|
||||
})
|
||||
```
|
||||
|
||||
In a YAML config, set `generator.engine.compile.regional: true`.
|
||||
|
||||
Do not combine it with `torch_compile_kwargs['mode']` (the loader injects
|
||||
inductor options, and torch.compile forbids mode+options); it is
|
||||
independent of `enable_torch_compile`, and when both are set the regional
|
||||
@@ -386,7 +413,7 @@ compile wins for the DiT.
|
||||
### What to expect from generic compile
|
||||
|
||||
The Wan result below measures the existing generic
|
||||
`enable_torch_compile=True` path. It is useful evidence that compile can help,
|
||||
`engine.compile.enabled: true` path. It is useful evidence that compile can help,
|
||||
but it is **not** a benchmark or numerical gate for the stricter regional
|
||||
fullgraph path above.
|
||||
|
||||
@@ -429,12 +456,12 @@ not asserted by any standing SSIM regression here — the SSIM tests in
|
||||
run with `enable_torch_compile` disabled. If you depend on compile
|
||||
output staying close to eager (or your previous compiled run), run an
|
||||
MS-SSIM gate on *your* config, especially when combining
|
||||
`enable_torch_compile=True` with other numerics-affecting flags
|
||||
`engine.compile.enabled: true` with other numerics-affecting flags
|
||||
(quantized attention backends, FP4, layerwise offload edge cases).
|
||||
|
||||
### Known interactions
|
||||
|
||||
- **Layerwise CPU offload** (`dit_layerwise_offload=True`, the default):
|
||||
- **Layerwise CPU offload** (`engine.offload.dit_layerwise: true`, the default):
|
||||
the offload hook previously caused an implicit graph break once per
|
||||
transformer layer, fragmenting the compiled region. Addressed in
|
||||
hao-ai-lab/FastVideo#1365 — keep that fix to get a clean compiled
|
||||
@@ -449,16 +476,17 @@ MS-SSIM gate on *your* config, especially when combining
|
||||
grad-enabled path remain outside it. Use the default inductor mode shown
|
||||
above unless your exact configuration has its own gate.
|
||||
|
||||
Extra `torch.compile` options are passed through `torch_compile_kwargs`
|
||||
(a dict), accepted by `VideoGenerator.from_pretrained(...)` and by the
|
||||
CLI as a JSON string via `--torch-compile-kwargs`. Example (currently
|
||||
Extra `torch.compile` options live at `engine.compile.backend`,
|
||||
`fullgraph`, `mode`, and `dynamic`; any other `torch.compile` kwargs go in
|
||||
`engine.compile.extras`. Set them in the nested config, in a config file,
|
||||
or as a CLI dotted override (for example
|
||||
`--generator.engine.compile.mode reduce-overhead`). Example (currently
|
||||
**not** recommended — see the CUDA-graphs caveat above):
|
||||
|
||||
```python
|
||||
VideoGenerator.from_pretrained(
|
||||
"Wan-AI/Wan2.1-T2V-1.3B-Diffusers",
|
||||
enable_torch_compile=True,
|
||||
torch_compile_kwargs={"mode": "reduce-overhead"}, # may error today
|
||||
{"engine": {"compile": {"enabled": True, "mode": "reduce-overhead"}}}, # may error today
|
||||
)
|
||||
```
|
||||
|
||||
@@ -471,7 +499,7 @@ config; **discard the first generation** (graph build):
|
||||
import time
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
gen = VideoGenerator.from_pretrained("your-model-id", enable_torch_compile=True)
|
||||
gen = VideoGenerator.from_pretrained("your-model-id", {"engine": {"compile": {"enabled": True}}})
|
||||
req = {"prompt": "Your prompt", "sampling": {"seed": 1024},
|
||||
"output": {"save_video": False}}
|
||||
gen.generate(req) # warmup: graph build, discard
|
||||
@@ -495,10 +523,10 @@ for backend in ["TORCH_SDPA", "FLASH_ATTN", "SAGE_ATTN"]:
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = backend
|
||||
generator = VideoGenerator.from_pretrained("your-model-id")
|
||||
start_time = time.perf_counter()
|
||||
generator.generate_video(
|
||||
prompt="Your prompt",
|
||||
seed=1024,
|
||||
)
|
||||
generator.generate({
|
||||
"prompt": "Your prompt",
|
||||
"sampling": {"seed": 1024},
|
||||
})
|
||||
elapsed = time.perf_counter() - start_time
|
||||
print(f"{backend}: {elapsed:.2f}s")
|
||||
```
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user