Compare commits
60
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0ef47caf86 | ||
|
|
e71d01648c | ||
|
|
e1f3904799 | ||
|
|
02f1ce11ae | ||
|
|
7f03e03dc6 | ||
|
|
e3b88bb12a | ||
|
|
cb66acd400 | ||
|
|
442e2d2e18 | ||
|
|
dd35763ad6 | ||
|
|
e90be598e5 | ||
|
|
ba5e81083c | ||
|
|
76ce9c7fd6 | ||
|
|
08d99c089e | ||
|
|
20751a21aa | ||
|
|
9dd2a837f4 | ||
|
|
93aab45ac2 | ||
|
|
017ce6602d | ||
|
|
a575055eec | ||
|
|
81f3fec7fd | ||
|
|
d265a454bf | ||
|
|
c100c66578 | ||
|
|
8760eb7a06 | ||
|
|
361f919c88 | ||
|
|
8b5377aab2 | ||
|
|
d995516da0 | ||
|
|
f47ad3f5b7 | ||
|
|
10bdf5e076 | ||
|
|
3e26db40b0 | ||
|
|
c73dd0ab55 | ||
|
|
430e52154e | ||
|
|
384eee8aef | ||
|
|
c4824c7764 | ||
|
|
9b0e57fe4b | ||
|
|
0100218594 | ||
|
|
39718cd54d | ||
|
|
37d06a832f | ||
|
|
8839ba8d4d | ||
|
|
61b91220c0 | ||
|
|
316f3876c2 | ||
|
|
614b59543c | ||
|
|
1c14afd559 | ||
|
|
9a3c45779c | ||
|
|
bfc9c01797 | ||
|
|
3a3ad3d209 | ||
|
|
aef4e9b3b1 | ||
|
|
a943220c11 | ||
|
|
556ac7088e | ||
|
|
e7456f1b75 | ||
|
|
c993d7393e | ||
|
|
7f83164233 | ||
|
|
4e52f47d1e | ||
|
|
e19913f6e9 | ||
|
|
2413a57651 | ||
|
|
7bb76b5ec9 | ||
|
|
0bd19a976b | ||
|
|
40b93784d2 | ||
|
|
33d3478bad | ||
|
|
3d8ac9d14b | ||
|
|
aaef49bfc6 | ||
|
|
cf6a00b9be |
@@ -0,0 +1,76 @@
|
||||
---
|
||||
name: env-var-conventions
|
||||
description: Add, read, rename, or remove an environment variable in FastVideo, or change the environment-variable policy. Use before touching fastvideo/envs.py, os.environ, os.getenv, or monkeypatch.setenv in fastvideo/, and when fastvideo/tests/contract/test_env_policy.py fails.
|
||||
---
|
||||
|
||||
# Environment Variable Conventions
|
||||
|
||||
## Purpose
|
||||
|
||||
FastVideo registers its environment variables as typed fields in
|
||||
`fastvideo/envs.py`. The policy that governs them is
|
||||
`docs/contributing/env_vars.md`, and the contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit
|
||||
CI lane. This skill routes an environment-variable change through that policy.
|
||||
The policy doc is the single source of the rules; read it instead of relying
|
||||
on a summary here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Read `docs/contributing/env_vars.md` in full.
|
||||
- Decide whether the setting belongs in an environment variable or an argument
|
||||
(rule 5 in the policy doc). Settings that users change per deployment are
|
||||
arguments; add them through `fastvideo/fastvideo_args.py` instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------- | -------- | -------------------------------------------------------------- |
|
||||
| `change` | Yes | Add, read, rename, or remove a variable, or change the policy. |
|
||||
| `variable` | Yes | The variable name, with the `FASTVIDEO_` prefix. |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Declare or edit the variable in `fastvideo/envs.py`.**
|
||||
- Pick the field type and category that the policy doc lists.
|
||||
- Write a description that states what the variable does and its units.
|
||||
- To rename, keep the old name in `deprecated_names`. To remove, add the
|
||||
name to `DEPRECATED_VARIABLES`. Update the uses in `examples/`,
|
||||
`scripts/`, `docs/`, `apps/`, and the tests.
|
||||
2. **Read the variable with `envs.NAME.get()` inside a function.**
|
||||
- In tests, change the value with `envs.NAME.override(value)`.
|
||||
- Do not call `os.environ`, `os.getenv`, or `monkeypatch.setenv` for a
|
||||
FastVideo variable.
|
||||
- To set a variable that another tool reads, call `envs.set_external`,
|
||||
`envs.setdefault_external`, or `envs.unset_external`.
|
||||
3. **Regenerate the table in the policy doc.**
|
||||
- Run `python fastvideo/tests/contract/test_env_policy.py`.
|
||||
4. **Run the contract test.**
|
||||
- Run `pytest fastvideo/tests/contract/test_env_policy.py`.
|
||||
- When the test reports a fixed known violation, delete or lower its entry
|
||||
in `KNOWN_VIOLATIONS`. Never add an entry to `KNOWN_VIOLATIONS`.
|
||||
5. **When the policy itself changes, update the policy doc and the contract
|
||||
test in the same pull request.**
|
||||
- The rules in `docs/contributing/env_vars.md`, the checks and allowlist in
|
||||
`fastvideo/tests/contract/test_env_policy.py`, and this skill must agree.
|
||||
|
||||
## Outputs
|
||||
|
||||
- A registry entry in `fastvideo/envs.py` and call sites that use
|
||||
`envs.NAME.get()`.
|
||||
- A regenerated table in `docs/contributing/env_vars.md`.
|
||||
- A passing `fastvideo/tests/contract/test_env_policy.py`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Add a FASTVIDEO_DEBUG_MY_STAGE switch that logs MyStage inputs.
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- `docs/contributing/env_vars.md`: the policy, the field types, and the
|
||||
violation kinds that the contract test reports.
|
||||
- `fastvideo/envs.py`: the registry.
|
||||
- `fastvideo/tests/contract/test_env_policy.py`: the contract test,
|
||||
`EXTERNAL_ALLOWLIST`, and `KNOWN_VIOLATIONS`.
|
||||
@@ -185,6 +185,7 @@ steps:
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
build.env("TEST_SCOPE") == "scheduled" ||
|
||||
@@ -211,6 +212,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
key: "lora-inference"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -232,6 +234,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Extraction Tests"
|
||||
key: "lora-extraction"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -253,6 +256,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Training Tests"
|
||||
key: "training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -276,6 +280,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
key: "distillation"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -297,6 +302,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
key: "self-forcing"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -318,6 +324,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
key: "lora-training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -341,6 +348,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
key: "training-vsa"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -364,6 +372,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
key: "inference-vmoba"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -385,6 +394,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Performance Tests"
|
||||
key: "performance"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -406,6 +416,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: API Server Tests"
|
||||
key: "api-server"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -427,6 +438,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
key: "train-framework"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -448,6 +460,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Eval Metrics Tests"
|
||||
key: "eval"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
|
||||
@@ -13,7 +13,7 @@ if [ -z "$selected" ]; then
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" = all ]; then
|
||||
exec pytest "$golden_root" -vs
|
||||
exec pytest "$golden_root" -xvs
|
||||
fi
|
||||
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
@@ -32,4 +32,4 @@ for golden_file in "${golden_files[@]}"; do
|
||||
golden_paths+=("$golden_path")
|
||||
done
|
||||
|
||||
exec pytest "${golden_paths[@]}" -vs
|
||||
exec pytest "${golden_paths[@]}" -xvs
|
||||
|
||||
@@ -22,6 +22,12 @@ else
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
# Alternate GPU backends compare against references without publishing records.
|
||||
# Their worker has read-only Hub credentials; publication is an operator task.
|
||||
if [ "${FASTVIDEO_CI_LOCAL_ONLY:-0}" = 1 ]; then
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
nvidia-smi \
|
||||
--query-gpu=index,timestamp,clocks.sm,clocks.max.sm,power.draw,power.limit,temperature.gpu \
|
||||
--format=csv -l 10 > "$PERF_REPORTS_DIR/gpu_telemetry.csv" 2>/dev/null &
|
||||
|
||||
@@ -2,4 +2,8 @@
|
||||
# Canonical Slurm CI selection for the transformer lane.
|
||||
set -euo pipefail
|
||||
|
||||
# The existing block reference records an absent FASTVIDEO_FA4 (FA2). Keep
|
||||
# that reference identity; the component lane also selects FA2 explicitly.
|
||||
env -u FASTVIDEO_FA4 pytest ./fastvideo/tests/golden_gate/test_wan_t2v.py -xvs
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_causal.py -xvs
|
||||
exec pytest ./fastvideo/tests/transformers -vs
|
||||
|
||||
@@ -2,4 +2,5 @@
|
||||
# Canonical Slurm CI selection for the VAE lane.
|
||||
set -euo pipefail
|
||||
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_vae.py -xvs
|
||||
exec pytest ./fastvideo/tests/vaes -vs
|
||||
|
||||
@@ -16,6 +16,7 @@ exec pytest \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/attention/test_sdpa_metadata_mask_contract.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_tile_grad_safety.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
|
||||
@@ -211,8 +211,8 @@ FAMILY_COVERAGE = (
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", ),
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
(
|
||||
"test_causal_similarity.py",
|
||||
"test_wan_i2v_similarity.py",
|
||||
@@ -291,6 +291,18 @@ def _family_coverage(path: str) -> tuple[set[str], set[str]]:
|
||||
if family.pattern.search(normalized):
|
||||
golden.update(family.golden_tests)
|
||||
ssim.update(family.ssim_tests)
|
||||
# Select the component actually touched, including compatibility paths.
|
||||
# Family configs/pipeline wiring can affect all four Wan gates.
|
||||
if re.search(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)", normalized):
|
||||
if (normalized.endswith(("/wan/vae.py", "/wan/vae_config.py", "/vaes/wanvae.py"))
|
||||
or normalized.endswith("/wan/stages/conditioning.py")):
|
||||
golden = {"test_wan_vae.py"}
|
||||
elif normalized.endswith(("/wan/causal_transformer.py", "/dits/causal_wanvideo.py",
|
||||
"/wan/stages/causal_denoising.py")):
|
||||
golden = {"test_wan_causal.py"}
|
||||
elif (normalized == "fastvideo/models/dits/wanvideo.py"
|
||||
or normalized.endswith(("/wan/transformer.py", "/wan/stages/denoising.py", "/wan/stages/dmd.py"))):
|
||||
golden = {"test_wan_t2v.py", "test_wan_denoising.py"}
|
||||
return golden, ssim
|
||||
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ jobs:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
&& (vars.CI_GPU_BACKEND == '' || vars.CI_GPU_BACKEND == 'slurm')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
name: Promote Selected GPU Backend Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
concurrency:
|
||||
group: gpu-ci-status-${{ github.event.sha }}-${{ vars.CI_GPU_BACKEND }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
promote:
|
||||
if: >-
|
||||
(vars.CI_GPU_BACKEND == 'modal' || vars.CI_GPU_BACKEND == 'vllm')
|
||||
&& (github.event.context == format('gpu-ci/{0}/fastcheck-passed', vars.CI_GPU_BACKEND)
|
||||
|| github.event.context == format('gpu-ci/{0}/full-suite-passed', vars.CI_GPU_BACKEND))
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
SELECTED_BACKEND: ${{ vars.CI_GPU_BACKEND }}
|
||||
steps:
|
||||
- name: Mirror the selected backend's latest suite results
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const backend = process.env.SELECTED_BACKEND;
|
||||
if (!['modal', 'vllm'].includes(backend)) {
|
||||
throw new Error('Unsupported selected GPU backend');
|
||||
}
|
||||
const sha = context.payload.sha;
|
||||
// Read current state after entering the serialized workflow. A
|
||||
// delayed event must not overwrite a newer failure with success.
|
||||
const statuses = await github.paginate(github.rest.repos.listCommitStatusesForRef, {
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
for (const suffix of ['fastcheck-passed', 'full-suite-passed']) {
|
||||
const sourceContext = `gpu-ci/${backend}/${suffix}`;
|
||||
const matches = statuses.filter(status => status.context === sourceContext);
|
||||
matches.sort((a, b) =>
|
||||
Date.parse(b.updated_at) - Date.parse(a.updated_at) || b.id - a.id
|
||||
);
|
||||
const latest = matches[0];
|
||||
const state = latest ? latest.state : 'pending';
|
||||
if (!['pending', 'success', 'failure', 'error'].includes(state)) {
|
||||
throw new Error(`Unsupported status state for ${sourceContext}`);
|
||||
}
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
context: suffix,
|
||||
state,
|
||||
description: latest
|
||||
? `${backend} ${suffix}: ${state}`
|
||||
: `Waiting for ${backend} ${suffix}`,
|
||||
...(latest && latest.target_url ? {target_url: latest.target_url} : {}),
|
||||
});
|
||||
}
|
||||
@@ -92,6 +92,7 @@ jobs:
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
@@ -156,6 +157,7 @@ jobs:
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
|
||||
@@ -203,7 +203,8 @@ jobs:
|
||||
docker buildx imagetools create "${TAG_ARGS[@]}" "${IMAGE_REFS[@]}"
|
||||
docker buildx imagetools inspect "${TAGS[0]}"
|
||||
|
||||
# The CI runner is ARM64 like DGX Spark, but targets sm_100 rather than sm_121.
|
||||
# The CI runner is ARM64 like DGX Spark, but targets sm_100a rather than sm_121.
|
||||
# The architecture-specific target includes the GB200 VSA CUDA extensions.
|
||||
# Publish a single-architecture variant so the self-hosted CI runner can reuse
|
||||
# the exact prebuilt kernel instead of compiling it in every job.
|
||||
build-ci-runner-image:
|
||||
@@ -219,7 +220,7 @@ jobs:
|
||||
PYTHON_VERSION=3.12
|
||||
CUDA_VERSION=13.0.0
|
||||
UV_TORCH_BACKEND=cu130
|
||||
TORCH_CUDA_ARCH_LIST=10.0
|
||||
TORCH_CUDA_ARCH_LIST=10.0a
|
||||
CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
FLASH_ATTN_WHEEL_TAG=cu130torch2.12
|
||||
FLASH_ATTN_WHEEL_RELEASE_ARM64=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.22
|
||||
|
||||
@@ -63,7 +63,7 @@ jobs:
|
||||
torch-cuda-short: 'cu130'
|
||||
platform:
|
||||
# x86_64 builds the full cu126 + cu130 set. cu130 ships the
|
||||
# data-center Blackwell sm_100a VSA and consumer sm_120a FP4
|
||||
# data-center Blackwell sm_100a/sm_103a VSA and consumer sm_120a FP4
|
||||
# kernels.
|
||||
- os: ubuntu-22.04
|
||||
arch: x86_64
|
||||
@@ -169,18 +169,18 @@ jobs:
|
||||
# covers sm_120a; turbodiffusion covers sm_100a+sm_120a. The sm_100 FP4
|
||||
# forward is the FA4 CuTe DSL path in the fastvideo package (PR #1221),
|
||||
# JIT-compiled at runtime — not built into this wheel.
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a VSA
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a/sm_103a VSA
|
||||
# + consumer Blackwell sm_120a FP4.
|
||||
# * x86_64 cu126 = Hopper TK only (older drivers; CUDA < 12.8 has no FP4).
|
||||
# The per-arch split in CMakeLists pins the FP4 targets to sm_120a and builds
|
||||
# the main extension for the full arch list. CMAKE_BUILD_PARALLEL_LEVEL caps
|
||||
# Ninja so heavy CUTLASS/TK template TUs don't OOM the 16 GB runner (exit 143).
|
||||
if [ "${{ matrix.platform.arch }}" = "aarch64" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;10.3a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=OFF -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON"
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
elif [ "${{ matrix.torch-cuda.torch-cuda-short }}" = "cu130" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;10.3a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
|
||||
# A single FP4 TU (attn_qat_infer) can use ~8-12 GB on its own, so serialize.
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
|
||||
@@ -9,7 +9,7 @@ exclude: |
|
||||
tests/.*|
|
||||
scripts/.*|
|
||||
fastvideo/dataset/.*|
|
||||
fastvideo/models/.*|
|
||||
fastvideo/models/(?!wan/(config|vae_config|pipeline_config|definition|__init__)\.py$).*|
|
||||
^apps/dreamverse/web/.*|
|
||||
examples/.*|
|
||||
\.agents/.*|
|
||||
|
||||
@@ -66,14 +66,18 @@ Local guidance lives next to the code. Read the in-scope file before editing:
|
||||
| `fastvideo/AGENTS.md` | Core package map, public API, registry-driven model dispatch |
|
||||
| `fastvideo/configs/AGENTS.md` | Arch + pipeline config dataclasses, `param_names_mapping` |
|
||||
| `fastvideo/models/AGENTS.md` | DiT / VAE / encoder / scheduler / loader layout (pre-commit excluded) |
|
||||
| `fastvideo/models/wan/AGENTS.md` | Wan family-local transformers, VAE, configs, and the SP sharding invariant |
|
||||
| `fastvideo/layers/AGENTS.md` | Tensor-parallel linear/attention layer rules for ports |
|
||||
| `fastvideo/attention/AGENTS.md` | Backend registry + env-var override |
|
||||
| `fastvideo/pipelines/AGENTS.md` | Stage ABC, `basic/<model>/`, `preprocess/`, presets |
|
||||
| `fastvideo/pipelines/basic/wan/AGENTS.md` | Wan sampling stages, first-frame conditioning, DMD/causal boundaries |
|
||||
| `fastvideo/pipelines/basic/magi_human/AGENTS.md` | MagiHuman umbrella repo, lazy-loaded components, packing invariants |
|
||||
| `fastvideo/training/AGENTS.md` | Legacy monolithic pipelines (frozen for existing models) |
|
||||
| `fastvideo/train/AGENTS.md` | New modular trainer (methods × models × callbacks, YAML) |
|
||||
| `fastvideo/tests/AGENTS.md` | Test taxonomy, conftest, pre-commit-excluded path |
|
||||
| `fastvideo/tests/ssim/AGENTS.md` | GPU SSIM regression authoring + reference video sync |
|
||||
| `scripts/checkpoint_conversion/AGENTS.md` | Adding a converter for a new HF/official checkpoint |
|
||||
| `apps/dreamverse/AGENTS.md` | DreamVerse app structure and conventions |
|
||||
|
||||
## Critical: Two Training Stacks Coexist
|
||||
|
||||
|
||||
@@ -3,14 +3,16 @@
|
||||
</div>
|
||||
|
||||
<p align="center">
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://haoailab.com/FastVideo/cookbook/"><b>Cookbook</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
</p>
|
||||
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
## NEWS
|
||||
- `2026/09/15`: Release [FastH3 8-Step V2](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2), an eight-forward data-free DMD2 checkpoint distilled from MiniMax-H3 with 80% Video Sparse Attention. Run it with `examples/inference/basic/basic_fasth3_8step.py` or the [FastH3 8-Step V2 recipe](https://haoailab.com/FastVideo/cookbook/minimax-h3/).
|
||||
- `2026/09/01`: FastH3 now runs locally on Apple Silicon through MLX and on NVIDIA DGX Spark through CUDA 13, including two-Spark inference. Follow the [FastH3 recipes](https://haoailab.com/FastVideo/cookbook/minimax-h3/) and read the [Blog](https://haoailab.com/blogs/fasth3-local/).
|
||||
- `2026/08/27`: [FastH3 Preview v1](https://haoailab.com/blogs/fasth3-preview/) is an open-weight 4-step sparse-distilled MiniMax-H3 model for synchronized video-and-audio generation, developed in collaboration with [Nuva Lab](https://nuvalab.ai/) and the [NVIDIA FastGen team](https://github.com/NVlabs/FastGen). Download the recommended [VSA / Data-Free weights](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree), or see the [full FastH3 collection](https://huggingface.co/collections/FastVideo/fastvideo-fasth3).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac—follow the [Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac. Follow the [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/06/23`: Release FastWan-QAD: 5s of Video generated in 1.8s E2E. See the [FastWan-QAD models](https://huggingface.co/FastVideo/FastWan-QAD-FP8-1.3B), [Attn-QAT training guide](https://haoailab.com/FastVideo/training/attn_qat/), and [blog](https://haoailab.com/blogs/fastwan-qad/).
|
||||
- `2026/03/17`: Release demo: Into the Dreamverse: Vibe Directing in FastVideo, check out the [Blog](https://haoailab.com/blogs/dreamverse/).
|
||||
- `2026/03/13`: Release demo: Create a 5s 1080p Video in 4.5s with FastVideo on a Single GPU, check out the [Blog](https://haoailab.com/blogs/fastvideo_realtime_1080p/).
|
||||
@@ -62,13 +64,12 @@ UV_TORCH_BACKEND=cu126 uv pip install fastvideo
|
||||
```
|
||||
|
||||
Use `UV_TORCH_BACKEND=cu130` on CUDA 13. Apple silicon users should follow the
|
||||
[MPS installation guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
> **On an Apple Silicon Mac?** FastVideo runs FastMetal-QAD through an MLX
|
||||
> runtime. Install with `uv pip install -e '.[mlx]'`, download
|
||||
> [`FastVideo/FastMetal-1.3B-QAD`](https://huggingface.co/FastVideo/FastMetal-1.3B-QAD),
|
||||
> and follow the
|
||||
> [Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
> **On an Apple Silicon Mac?** Install with `uv pip install -e '.[mlx]'` from
|
||||
> a clone, then pick a recipe in the
|
||||
> [cookbook](https://haoailab.com/FastVideo/cookbook/). See the
|
||||
> [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/) for more detailed installation instructions.
|
||||
|
||||
@@ -86,7 +87,7 @@ Install FastVideo (https://github.com/hao-ai-lab/FastVideo) into a fresh uv virt
|
||||
https://hao-ai-lab.github.io/FastVideo/getting_started/installation/):
|
||||
- NVIDIA GPU, x86_64 -> docs/getting_started/installation/gpu.md
|
||||
- NVIDIA DGX Spark / GB10, aarch64, CUDA 13 -> docs/getting_started/installation/spark.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mps.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mlx.md
|
||||
3. Use uv for every step. If a command fails, debug it and tell me what you changed.
|
||||
4. Verify the result:
|
||||
python -c "import fastvideo, torch; print('cuda', torch.cuda.is_available())"
|
||||
|
||||
@@ -97,13 +97,33 @@ dreamverse-server --port 8009
|
||||
dreamverse-mock-server --port 8009
|
||||
```
|
||||
|
||||
### Run Dreamverse with FastH3
|
||||
|
||||
Select the VSA data-free FastH3 Preview profile when you start the backend:
|
||||
|
||||
```bash
|
||||
DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
The `fast-h3` profile uses four visible GPUs by default. It loads the `MiniMaxAI/MiniMax-H3` base checkpoint and the
|
||||
`vsa-datafree/adapter_model.safetensors` adapter from
|
||||
`FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA`. Each request generates a 124-frame, 768×1344 video with
|
||||
synchronized audio and five sigma-grid points. Dreamverse uses the last frame of each segment as first-frame
|
||||
conditioning for the following segment.
|
||||
|
||||
Set `CUDA_VISIBLE_DEVICES` when you need to choose the four physical GPUs:
|
||||
|
||||
```bash
|
||||
CUDA_VISIBLE_DEVICES=0,1,2,3 DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
> **Expect a slow first boot.** With `torch.compile` and startup warmup enabled
|
||||
> (the default), the backend compiles the segment 1 and segment 2 inference
|
||||
> paths before it reports ready — this can take **tens of minutes on a cold
|
||||
> cache**, regardless of how you deploy (local, server, Docker, or Modal).
|
||||
> `/healthz` responds as soon as the process is up; `/readyz` stays `503` until
|
||||
> warmup finishes. For a faster, uncompiled startup while testing, set
|
||||
> `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
> warmup finishes. To defer compilation until the first generated request while
|
||||
> testing, set `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
|
||||
## Frontend Setup
|
||||
|
||||
@@ -219,6 +239,7 @@ selection, and mock-server behavior:
|
||||
pytest apps/dreamverse/dreamverse/tests/test_config.py \
|
||||
apps/dreamverse/dreamverse/tests/test_entrypoints.py \
|
||||
apps/dreamverse/dreamverse/tests/test_gpu_pool.py \
|
||||
apps/dreamverse/dreamverse/tests/test_minimax_h3_generation.py \
|
||||
apps/dreamverse/dreamverse/tests/test_mock_server.py -q
|
||||
```
|
||||
|
||||
|
||||
+12
-1
@@ -139,7 +139,18 @@ session.
|
||||
- startup warmup
|
||||
- user join/leave commands
|
||||
- `USER_STEP` execution for each segment
|
||||
- continuation state between segments
|
||||
- generation-command routing and stream-result delivery
|
||||
|
||||
Model generation has a separate ownership boundary inside each GPU process:
|
||||
|
||||
- `apps/dreamverse/dreamverse/generation_worker.py` selects the backend that the active model profile declares and owns
|
||||
the backend lifecycle.
|
||||
- `apps/dreamverse/dreamverse/ltx2_generation.py` owns LTX-2 generator configuration, video and audio continuation, and
|
||||
runtime LoRA application.
|
||||
- `apps/dreamverse/dreamverse/minimax_h3_generation.py` owns the VSA data-free FastH3 adapter, FastH3 generator and
|
||||
request configuration, and last-frame continuation through MiniMax H3 first-frame conditioning.
|
||||
- `apps/dreamverse/dreamverse/generation_contracts.py` defines the decoded media and stream-trimming result that both
|
||||
model backends return to `apps/dreamverse/dreamverse/gpu_pool.py`.
|
||||
|
||||
`apps/dreamverse/dreamverse/prompt_enhancer.py` manages:
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Benchmark the LTX-2 generation pipeline driven by the dreamverse Python SDK path.
|
||||
|
||||
Mirrors how ``apps/dreamverse/dreamverse/video_generation.py`` constructs
|
||||
Mirrors how ``apps/dreamverse/dreamverse/ltx2_generation.py`` constructs
|
||||
``GeneratorConfig`` and calls ``VideoGenerator.generate()``, then
|
||||
captures per-stage timings via the ``FASTVIDEO_STAGE_LOGGING=1`` log
|
||||
hooks (same mechanism as ``FastVideo-internal/examples/inference/basic/
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
_SERVER_ROOT = Path(__file__).resolve().parent
|
||||
@@ -55,16 +56,34 @@ FRONTEND_STATIC_DIR_CANDIDATES = _resolve_frontend_static_dir_candidates()
|
||||
MODEL_REGISTRY = {
|
||||
"fast-ltx2": {
|
||||
"name": "FastLTX2",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX2-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-ltx23": {
|
||||
"name": "FastLTX23",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX-2.3-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-h3": {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
},
|
||||
}
|
||||
|
||||
DEFAULT_MODEL_ID = "fast-ltx2"
|
||||
@@ -171,7 +190,7 @@ def _optional_env(*names: str) -> str | None:
|
||||
DEVTOOLS_ENABLED = _env_bool("FASTVIDEO_ENABLE_DEVTOOLS", False)
|
||||
PROMPT_SAFETY_ENABLED = _env_bool("FASTVIDEO_ENABLE_PROMPT_SAFETY", False)
|
||||
DREAMVERSE_MAX_AUTOTUNE = _env_bool("DREAMVERSE_MAX_AUTOTUNE", True)
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", 1))
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", cast(int, MODEL_CONFIG["default_sp_size"])))
|
||||
|
||||
DREAMVERSE_MODEL_PATH = (os.getenv("DREAMVERSE_MODEL_PATH", "").strip() or None)
|
||||
if DREAMVERSE_MODEL_PATH:
|
||||
@@ -213,7 +232,7 @@ def _resolve_lora_spec(spec: str) -> str | None:
|
||||
if not spec:
|
||||
return None
|
||||
if spec.lower() == "omninft":
|
||||
return MODEL_CONFIG.get("lora_repo")
|
||||
return cast(str | None, MODEL_CONFIG.get("lora_repo"))
|
||||
if spec.lower() in AVAILABLE_LORAS:
|
||||
return AVAILABLE_LORAS[spec.lower()]["repo"]
|
||||
return spec
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Shared contract between DreamVerse generation backends and GPU workers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Protocol
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Decoded media and stream-trimming metadata for one DreamVerse segment."""
|
||||
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict[str, float]
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class GenerationBackend(Protocol):
|
||||
"""Model-owned generation operations used by one GPU worker process."""
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
...
|
||||
|
||||
def shutdown(self) -> None:
|
||||
...
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
...
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
...
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
...
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
...
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Select and own one model-specific generation backend per GPU process."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dreamverse.config import MODEL_CONFIG
|
||||
from dreamverse.generation_contracts import GenerationBackend, StepResult
|
||||
|
||||
|
||||
def _create_generation_backend(backend_name: str, gpu_id: int) -> GenerationBackend:
|
||||
"""Construct the backend that owns the selected model family's behavior."""
|
||||
if backend_name == "ltx2":
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
return LTX2GenerationBackend(gpu_id)
|
||||
if backend_name == "minimax_h3":
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
return MiniMaxH3GenerationBackend(gpu_id)
|
||||
raise ValueError(f"Unsupported DreamVerse generation backend: {backend_name!r}")
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
"""Delegate GPU lifecycle and generation calls to the active model backend."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.model_config: dict = dict(MODEL_CONFIG)
|
||||
self.backend_name: str | None = None
|
||||
self.backend: GenerationBackend | None = None
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Load the requested model through its generation backend.
|
||||
|
||||
Model selection belongs here so the GPU process and streaming layers
|
||||
use one stable media contract without importing model-specific code.
|
||||
"""
|
||||
requested_model_config = dict(model_config) if model_config is not None else dict(self.model_config)
|
||||
backend_name = requested_model_config.get("generation_backend")
|
||||
if not isinstance(backend_name, str) or not backend_name:
|
||||
raise ValueError("DreamVerse model configuration requires `generation_backend`.")
|
||||
|
||||
candidate_backend = self.backend
|
||||
if candidate_backend is None or self.backend_name != backend_name:
|
||||
if candidate_backend is not None:
|
||||
candidate_backend.shutdown()
|
||||
candidate_backend = _create_generation_backend(backend_name, self.gpu_id)
|
||||
|
||||
try:
|
||||
candidate_backend.initialize(requested_model_config)
|
||||
except Exception:
|
||||
try:
|
||||
candidate_backend.shutdown()
|
||||
except Exception as shutdown_error:
|
||||
print(f"[GPU {self.gpu_id}] Backend cleanup after initialization failure: {shutdown_error}")
|
||||
self.backend = None
|
||||
self.backend_name = None
|
||||
raise
|
||||
|
||||
self.model_config = requested_model_config
|
||||
self.backend = candidate_backend
|
||||
self.backend_name = backend_name
|
||||
|
||||
def _require_backend(self) -> GenerationBackend:
|
||||
"""Return the initialized backend or fail before processing a command."""
|
||||
if self.backend is None:
|
||||
raise RuntimeError("Generation backend is not initialized.")
|
||||
return self.backend
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release model resources owned by the selected backend."""
|
||||
if self.backend is not None:
|
||||
self.backend.shutdown()
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
self._require_backend().clear_conditioning()
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one segment through the selected model backend."""
|
||||
return self._require_backend().generate_step(
|
||||
prompt,
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
return self._require_backend().warmup(prompt)
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
return self._require_backend().apply_lora_stack(stack)
|
||||
@@ -12,7 +12,7 @@ from enum import Enum
|
||||
from multiprocessing import Process, Queue
|
||||
|
||||
from dreamverse.config import (
|
||||
DEFAULT_MODEL_ID,
|
||||
ACTIVE_MODEL_ID,
|
||||
DREAMVERSE_SP_SIZE,
|
||||
MODEL_REGISTRY,
|
||||
STARTUP_WARMUP_ENABLED,
|
||||
@@ -54,7 +54,7 @@ from dreamverse.worker_ipc import (
|
||||
def _parse_requested_gpu_limit() -> int | None:
|
||||
raw_value = os.getenv("FASTVIDEO_GPU_COUNT", "").strip().lower()
|
||||
if not raw_value:
|
||||
return 1
|
||||
return DREAMVERSE_SP_SIZE
|
||||
if raw_value == "all":
|
||||
return None
|
||||
try:
|
||||
@@ -164,12 +164,12 @@ def gpu_worker_process(
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = cuda_device
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
|
||||
from dreamverse.video_generation import VideoGenerationWorker
|
||||
from dreamverse.generation_worker import VideoGenerationWorker
|
||||
|
||||
worker = VideoGenerationWorker(gpu_id)
|
||||
|
||||
def event_loop(first_cmd: Command = None):
|
||||
"""Blocking event loop for LTX2; dispatches user commands."""
|
||||
"""Block on generation commands after the model is initialized."""
|
||||
print(f"[GPU {gpu_id}] Entering event loop")
|
||||
|
||||
def handle_command(cmd: Command):
|
||||
@@ -435,7 +435,7 @@ class GPUSlot:
|
||||
self._response_reader_task: asyncio.Task | None = None
|
||||
self._active: bool = False
|
||||
self._reader_lock: asyncio.Lock | None = None
|
||||
self.current_model_id: str = DEFAULT_MODEL_ID
|
||||
self.current_model_id: str | None = ACTIVE_MODEL_ID
|
||||
self.shared_stream_buffer = None
|
||||
self.shared_stream_buffer_size = SHARED_STREAM_BUFFER_BYTES
|
||||
|
||||
@@ -690,7 +690,7 @@ class GPUSlot:
|
||||
async def join_user(self, user_id: str, model_id: str = None) -> JoinAck:
|
||||
"""Add a user to this GPU."""
|
||||
if model_id is None:
|
||||
model_id = DEFAULT_MODEL_ID
|
||||
model_id = ACTIVE_MODEL_ID
|
||||
|
||||
# Reload model if a different one is requested
|
||||
if model_id != self.current_model_id and model_id in MODEL_REGISTRY:
|
||||
@@ -705,16 +705,23 @@ class GPUSlot:
|
||||
self.connected_users.clear()
|
||||
|
||||
model_config = MODEL_REGISTRY[model_id]
|
||||
reload_response = await self._send_command(Command(CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
try:
|
||||
reload_response = await self._send_command(Command(
|
||||
CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
except Exception:
|
||||
self.current_model_id = None
|
||||
raise
|
||||
match reload_response:
|
||||
case ReloadAck():
|
||||
pass
|
||||
case WorkerError(message=msg):
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Model reload failed: {msg}")
|
||||
case _:
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Unexpected reload response: "
|
||||
f"{type(reload_response).__name__}")
|
||||
|
||||
|
||||
+4
-22
@@ -1,9 +1,9 @@
|
||||
"""LTX2 model lifecycle and continuation conditioning.
|
||||
"""LTX-2 model lifecycle and continuation conditioning.
|
||||
|
||||
Runs inside a GPU worker subprocess. Owns the model, the audio
|
||||
encoder, and the per-session continuation state carried across
|
||||
segments. Callers must set ``os.environ["CUDA_VISIBLE_DEVICES"]``
|
||||
before constructing ``VideoGenerationWorker`` — all ``fastvideo.*``
|
||||
before constructing ``LTX2GenerationBackend`` — all ``fastvideo.*``
|
||||
imports are deferred to method bodies so nothing touches CUDA at
|
||||
module import time.
|
||||
"""
|
||||
@@ -14,9 +14,6 @@ import gc
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
@@ -35,6 +32,7 @@ from dreamverse.config import (
|
||||
DREAMVERSE_LORA_STACK,
|
||||
_resolve_lora_spec,
|
||||
)
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
# Multi-frame decoded continuation defaults from
|
||||
# examples/inference/basic/basic_ltx2_distilled_video_continuation.py.
|
||||
@@ -80,22 +78,6 @@ def _reset_lora_registry(worker) -> dict:
|
||||
return {"status": "lora_registry_reset"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Output of one generation step.
|
||||
|
||||
``head_trim_frames`` / ``head_trim_audio_frames`` are derived here
|
||||
so downstream AV streaming never needs to import conditioning
|
||||
constants.
|
||||
"""
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class ContinuationState:
|
||||
"""Per-session video + audio conditioning carried across segments."""
|
||||
|
||||
@@ -202,7 +184,7 @@ class ContinuationState:
|
||||
self.audio_latents = latents.detach().clone().cpu()
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
class LTX2GenerationBackend:
|
||||
"""Single-GPU LTX2 generator with continuation state.
|
||||
|
||||
Caller must set ``os.environ["CUDA_VISIBLE_DEVICES"]`` before
|
||||
@@ -0,0 +1,297 @@
|
||||
"""FastH3 model lifecycle and first-frame continuation for DreamVerse."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import os
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from dreamverse.config import DREAMVERSE_SP_SIZE
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from PIL.Image import Image
|
||||
|
||||
|
||||
def _required_config_str(model_config: dict, field_name: str) -> str:
|
||||
"""Read one required non-empty string from a DreamVerse model profile."""
|
||||
value = model_config.get(field_name)
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError(f"FastH3 model configuration requires `{field_name}`.")
|
||||
return value.strip()
|
||||
|
||||
|
||||
class MiniMaxH3GenerationBackend:
|
||||
"""Run the VSA data-free FastH3 adapter and retain one continuation frame."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.generator: Any | None = None
|
||||
self.model_config: dict = {}
|
||||
self.continuation_image: Image | None = None
|
||||
|
||||
def _gpu_mem(self) -> str:
|
||||
allocated_gib = torch.cuda.memory_allocated() / 1024**3
|
||||
reserved_gib = torch.cuda.memory_reserved() / 1024**3
|
||||
return f"alloc={allocated_gib:.2f}GiB, reserved={reserved_gib:.2f}GiB"
|
||||
|
||||
@staticmethod
|
||||
def _configure_environment(attention_backend: str) -> None:
|
||||
"""Apply the fixed boot-time switches from the FastH3 reference recipe."""
|
||||
os.environ.update({
|
||||
"FASTVIDEO_ATTENTION_BACKEND": attention_backend,
|
||||
"FASTVIDEO_FA4": "1",
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all",
|
||||
"FASTVIDEO_VSA_SM100A": "0",
|
||||
})
|
||||
os.environ.pop("FASTVIDEO_INFERENCE_TORCH_COMPILE", None)
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Download the fixed Preview adapter and load the FastH3 generator.
|
||||
|
||||
The model profile owns the base checkpoint, adapter file, attention
|
||||
backend, and generation geometry. The backend translates that profile
|
||||
into FastVideo's typed generator configuration.
|
||||
"""
|
||||
if model_config is not None:
|
||||
self.model_config = dict(model_config)
|
||||
if not self.model_config:
|
||||
raise ValueError("FastH3 initialization requires a model configuration.")
|
||||
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
self.clear_conditioning()
|
||||
model_path = _required_config_str(self.model_config, "model_path")
|
||||
adapter_repo = _required_config_str(self.model_config, "adapter_repo")
|
||||
adapter_filename = _required_config_str(self.model_config, "adapter_filename")
|
||||
attention_backend = _required_config_str(self.model_config, "attention_backend")
|
||||
self._configure_environment(attention_backend)
|
||||
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GeneratorConfig,
|
||||
OffloadConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
|
||||
adapter_path = hf_hub_download(repo_id=adapter_repo, filename=adapter_filename)
|
||||
experimental = {
|
||||
"attention_backend": attention_backend,
|
||||
"inference_torch_compile": attention_backend == "FLASH_ATTN",
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
}
|
||||
if attention_backend == "VIDEO_SPARSE_ATTN_H3":
|
||||
experimental.update({
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
})
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=model_path,
|
||||
pipeline=PipelineSelection(
|
||||
components=ComponentConfig(lora_path=adapter_path, lora_strength=1.0),
|
||||
experimental=experimental,
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=DREAMVERSE_SP_SIZE,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=DREAMVERSE_SP_SIZE),
|
||||
offload=OffloadConfig(
|
||||
dit=False,
|
||||
dit_layerwise=False,
|
||||
text_encoder=True,
|
||||
image_encoder=True,
|
||||
vae=True,
|
||||
pin_cpu_memory=True,
|
||||
),
|
||||
compile=CompileConfig(enabled=False, vae_enabled=True),
|
||||
use_fsdp_inference=False,
|
||||
),
|
||||
)
|
||||
|
||||
print(f"[GPU {self.gpu_id}] Loading FastH3 model: {model_path}")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 adapter: {adapter_repo}/{adapter_filename}")
|
||||
print(f"[GPU {self.gpu_id}] Before model load: {self._gpu_mem()}")
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
print(f"[GPU {self.gpu_id}] FastH3 loaded: {self._gpu_mem()} (warmup pending)")
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release the FastVideo generator and cached continuation image."""
|
||||
self.clear_conditioning()
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
"""Release the first-frame image retained for the next segment."""
|
||||
if self.continuation_image is not None:
|
||||
self.continuation_image.close()
|
||||
self.continuation_image = None
|
||||
|
||||
@staticmethod
|
||||
def _load_rgb_image(image_path: str) -> Image:
|
||||
"""Load an image into an independent RGB buffer with no open file handle."""
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(image_path) as image:
|
||||
return image.convert("RGB").copy()
|
||||
|
||||
def _select_conditioning_image(
|
||||
self,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> tuple[Image | None, bool]:
|
||||
"""Select the initial upload or retained last frame for one segment."""
|
||||
if reset_conditioning:
|
||||
self.clear_conditioning()
|
||||
if segment_idx > 1 and self.continuation_image is not None:
|
||||
return self.continuation_image.copy(), True
|
||||
if segment_idx > 1 and not reset_conditioning:
|
||||
raise RuntimeError(f"FastH3 segment {segment_idx} requires a retained continuation frame.")
|
||||
if segment_idx == 1 and image_path:
|
||||
return self._load_rgb_image(image_path), False
|
||||
return None, False
|
||||
|
||||
def _build_request(self, prompt: str, conditioning_image: Image | None):
|
||||
"""Build the typed FastVideo request owned by the FastH3 profile."""
|
||||
from fastvideo.api import GenerationRequest, InputConfig, OutputConfig, SamplingConfig
|
||||
|
||||
return GenerationRequest(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
inputs=InputConfig(pil_image=conditioning_image),
|
||||
sampling=SamplingConfig(
|
||||
height=int(self.model_config["height"]),
|
||||
width=int(self.model_config["width"]),
|
||||
num_frames=int(self.model_config["num_frames"]),
|
||||
fps=24,
|
||||
num_inference_steps=int(self.model_config["num_inference_steps"]),
|
||||
guidance_scale=1.0,
|
||||
batch_cfg=False,
|
||||
seed=int(self.model_config["seed"]),
|
||||
),
|
||||
output=OutputConfig(save_video=False, return_frames=True),
|
||||
)
|
||||
|
||||
def _save_continuation_frame(self, frames: list) -> None:
|
||||
"""Retain the last decoded frame as first-frame conditioning."""
|
||||
from PIL import Image
|
||||
|
||||
self.clear_conditioning()
|
||||
self.continuation_image = Image.fromarray(np.ascontiguousarray(frames[-1])).convert("RGB")
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one synchronized FastH3 segment and retain its last frame.
|
||||
|
||||
Later segments use MiniMax H3's first-frame-to-video path. The first
|
||||
conditioned frame and its matching audio duration are trimmed before
|
||||
streaming so adjacent segments do not duplicate media.
|
||||
"""
|
||||
if self.generator is None:
|
||||
raise RuntimeError("FastH3 generator is not initialized.")
|
||||
conditioning_image, uses_continuation = self._select_conditioning_image(
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
request = self._build_request(prompt, conditioning_image)
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
result = self.generator.generate(request)
|
||||
finally:
|
||||
if conditioning_image is not None:
|
||||
conditioning_image.close()
|
||||
torch.cuda.synchronize()
|
||||
generation_ms = (time.perf_counter() - started) * 1000.0
|
||||
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("FastH3 returned multiple results for one DreamVerse segment.")
|
||||
frames = result.frames
|
||||
if not isinstance(frames, list) or not frames:
|
||||
raise RuntimeError("FastH3 generation did not return decoded frames.")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
raise RuntimeError("FastH3 returned audio without an audio sample rate.")
|
||||
|
||||
save_started = time.perf_counter()
|
||||
self._save_continuation_frame(frames)
|
||||
save_conditioning_ms = (time.perf_counter() - save_started) * 1000.0
|
||||
timings = {
|
||||
"generation_ms": generation_ms,
|
||||
"generation_time_ms": float(result.generation_time or 0.0) * 1000.0,
|
||||
"save_conditioning_ms": save_conditioning_ms,
|
||||
"e2e_latency_ms": (time.perf_counter() - started) * 1000.0,
|
||||
}
|
||||
trim_frames = 1 if uses_continuation else 0
|
||||
print(f"[GPU {self.gpu_id}] FastH3 segment {segment_idx}: "
|
||||
f"{len(frames)} frames, gen={generation_ms:.0f}ms, "
|
||||
f"save_conditioning={save_conditioning_ms:.0f}ms, "
|
||||
f"e2e={timings['e2e_latency_ms']:.0f}ms")
|
||||
return StepResult(
|
||||
frames=frames,
|
||||
audio=audio,
|
||||
audio_sample_rate=audio_sample_rate,
|
||||
timings=timings,
|
||||
head_trim_frames=trim_frames,
|
||||
head_trim_audio_frames=trim_frames,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
"""Compile the FastH3 text and first-frame paths before readiness."""
|
||||
warmup_prompt = (prompt or "").strip()
|
||||
if not warmup_prompt:
|
||||
raise RuntimeError("Startup warmup prompt must be non-empty.")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup starting "
|
||||
"(synthetic segments: text-to-video, first-frame-to-video)")
|
||||
started = time.perf_counter()
|
||||
text_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
first_frame_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
total_ms = (time.perf_counter() - started) * 1000.0
|
||||
self.clear_conditioning()
|
||||
text_ms = float(text_result.timings.get("e2e_latency_ms", 0.0))
|
||||
first_frame_ms = float(first_frame_result.timings.get("e2e_latency_ms", 0.0))
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup complete: "
|
||||
f"text_to_video={text_ms:.0f}ms, "
|
||||
f"first_frame_to_video={first_frame_ms:.0f}ms, "
|
||||
f"total={total_ms:.0f}ms")
|
||||
return {
|
||||
"warmup_text_to_video_ms": text_ms,
|
||||
"warmup_first_frame_to_video_ms": first_frame_ms,
|
||||
"warmup_total_ms": total_ms,
|
||||
}
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
"""Reject runtime LoRA mutation because FastH3 uses one startup adapter."""
|
||||
del stack
|
||||
raise RuntimeError("FastH3 uses its fixed startup adapter and does not support runtime LoRA changes.")
|
||||
@@ -30,7 +30,7 @@ from dreamverse.session_init_image import cleanup_session_init_image, persist_se
|
||||
from dreamverse.worker_ipc import MediaChunk, MediaComplete, MediaInit
|
||||
|
||||
from dreamverse.config import (
|
||||
DEFAULT_MODEL_ID,
|
||||
ACTIVE_MODEL_ID,
|
||||
GENERATION_SEGMENT_CAP,
|
||||
PROMPT_AUTO_SLEEP_MS,
|
||||
PROMPT_AUTO_TIMEOUT_MS,
|
||||
@@ -264,7 +264,7 @@ class SessionController:
|
||||
timeout_task = asyncio.create_task(session_timeout())
|
||||
|
||||
# Join the engine on this GPU.
|
||||
await slot.join_user(client_id, model_id=DEFAULT_MODEL_ID)
|
||||
await slot.join_user(client_id, model_id=ACTIVE_MODEL_ID)
|
||||
|
||||
# Notify client they're connected to a GPU.
|
||||
await ws_send_json({
|
||||
|
||||
@@ -2,13 +2,14 @@ from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
import pytest
|
||||
|
||||
SERVER_DIR = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
def _load_config_module():
|
||||
def _load_config_module() -> ModuleType:
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"server_config_test_module",
|
||||
SERVER_DIR / "config.py",
|
||||
@@ -20,7 +21,7 @@ def _load_config_module():
|
||||
return module
|
||||
|
||||
|
||||
def _set_required_prompt_keys(monkeypatch):
|
||||
def _set_required_prompt_keys(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CEREBRAS_API_KEY", "cerebras-key")
|
||||
monkeypatch.setenv("GROQ_API_KEY", "groq-key")
|
||||
|
||||
@@ -150,3 +151,38 @@ def test_config_rejects_invalid_prompt_provider(monkeypatch):
|
||||
|
||||
with pytest.raises(RuntimeError, match="Invalid FASTVIDEO_PROMPT_PROVIDER"):
|
||||
_load_config_module()
|
||||
|
||||
|
||||
def test_config_registers_vsa_datafree_fasth3_profile(monkeypatch):
|
||||
"""The FastH3 registry entry owns the complete fixed Preview recipe."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.MODEL_REGISTRY["fast-h3"] == {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
def test_config_uses_fasth3_sequence_parallel_default(monkeypatch):
|
||||
"""Selecting FastH3 defaults DreamVerse to its four-GPU topology."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
monkeypatch.setenv("DREAMVERSE_MODEL_ID", "fast-h3")
|
||||
monkeypatch.delenv("DREAMVERSE_SP_SIZE", raising=False)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.ACTIVE_MODEL_ID == "fast-h3"
|
||||
assert module.MODEL_CONFIG["generation_backend"] == "minimax_h3"
|
||||
assert module.DREAMVERSE_SP_SIZE == 4
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
from dreamverse.generation_worker import _create_generation_backend
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
|
||||
def test_create_generation_backend_ltx2_module_import():
|
||||
backend = _create_generation_backend("ltx2", gpu_id=3)
|
||||
|
||||
assert isinstance(backend, LTX2GenerationBackend)
|
||||
assert backend.gpu_id == 3
|
||||
@@ -63,6 +63,14 @@ def test_get_available_gpus_defaults_to_first_visible_device(monkeypatch):
|
||||
assert gpu_pool.get_available_gpus() == [3]
|
||||
|
||||
|
||||
def test_get_available_gpus_defaults_to_active_model_sequence_parallel_size(monkeypatch):
|
||||
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1,2,3,4")
|
||||
monkeypatch.delenv("FASTVIDEO_GPU_COUNT", raising=False)
|
||||
monkeypatch.setattr(gpu_pool, "DREAMVERSE_SP_SIZE", 4)
|
||||
|
||||
assert gpu_pool.get_available_gpus() == [0, 1, 2, 3]
|
||||
|
||||
|
||||
def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
|
||||
monkeypatch.setenv("FASTVIDEO_GPU_COUNT", "zero")
|
||||
@@ -71,6 +79,23 @@ def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
gpu_pool.get_available_gpus()
|
||||
|
||||
|
||||
def test_join_user_failed_reload_marks_model_uninitialized(monkeypatch):
|
||||
"""A failed model reload forces the next join to reload a model."""
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.current_model_id = "fast-ltx2"
|
||||
|
||||
async def fake_send_command(command, timeout):
|
||||
del command, timeout
|
||||
return gpu_pool.WorkerError(user_id="__reload__", message="load failed")
|
||||
|
||||
monkeypatch.setattr(slot, "_send_command", fake_send_command)
|
||||
|
||||
with pytest.raises(RuntimeError, match="Model reload failed"):
|
||||
asyncio.run(slot.join_user("client-id", model_id="fast-h3"))
|
||||
|
||||
assert slot.current_model_id is None
|
||||
|
||||
|
||||
def test_send_command_raises_on_worker_death():
|
||||
"""A worker that consumes a command and exits without replying must
|
||||
surface as RuntimeError via sentinel detection, not after the long
|
||||
@@ -92,9 +117,9 @@ def test_send_command_raises_on_worker_death():
|
||||
ready = resp_q.get(timeout=30.0)
|
||||
assert ready == "READY"
|
||||
|
||||
async def runner():
|
||||
async def runner() -> None:
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.process = proc
|
||||
slot.process = proc # type: ignore[assignment]
|
||||
slot.command_queue = cmd_q
|
||||
slot.response_queue = resp_q
|
||||
|
||||
|
||||
@@ -17,11 +17,11 @@ FORBIDDEN_PREFIXES = (
|
||||
)
|
||||
ALLOWED_INTERNAL_IMPORTS = {
|
||||
(
|
||||
"video_generation.py",
|
||||
"ltx2_generation.py",
|
||||
"fastvideo.models.audio.ltx2_audio_processing",
|
||||
),
|
||||
(
|
||||
"video_generation.py",
|
||||
"ltx2_generation.py",
|
||||
"fastvideo.models.loader.component_loader",
|
||||
),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
import dreamverse.generation_worker as generation_worker
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
|
||||
FASTH3_MODEL_CONFIG = {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
class _RecordingGenerator:
|
||||
"""Record typed requests and return small synchronized media fixtures."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.requests: list[Any] = []
|
||||
self.conditioning_pixels: list[np.ndarray | None] = []
|
||||
|
||||
def generate(self, request):
|
||||
"""Capture the request and return two tiny video frames with audio."""
|
||||
self.requests.append(request)
|
||||
conditioning_image = request.inputs.pil_image
|
||||
self.conditioning_pixels.append(
|
||||
None if conditioning_image is None else np.asarray(conditioning_image).copy())
|
||||
frames = [
|
||||
np.full((2, 3, 3), 10, dtype=np.uint8),
|
||||
np.full((2, 3, 3), 20, dtype=np.uint8),
|
||||
]
|
||||
return SimpleNamespace(
|
||||
frames=frames,
|
||||
audio=np.zeros((2, 16), dtype=np.float32),
|
||||
audio_sample_rate=44100,
|
||||
generation_time=0.25,
|
||||
)
|
||||
|
||||
|
||||
def test_initialize_builds_vsa_datafree_fasth3_generator(monkeypatch):
|
||||
"""Initialization translates the DreamVerse profile into typed FastVideo config."""
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
captured = {}
|
||||
fake_generator = SimpleNamespace(shutdown=lambda: None)
|
||||
|
||||
def fake_from_config(config):
|
||||
captured["config"] = config
|
||||
return fake_generator
|
||||
|
||||
def fake_download(**kwargs):
|
||||
captured["download"] = kwargs
|
||||
return f"/models/{kwargs['filename']}"
|
||||
|
||||
monkeypatch.setattr("huggingface_hub.hf_hub_download", fake_download)
|
||||
monkeypatch.setattr(VideoGenerator, "from_config", fake_from_config)
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.DREAMVERSE_SP_SIZE", 4)
|
||||
monkeypatch.setenv("FASTVIDEO_ATTENTION_BACKEND", "test-attention")
|
||||
monkeypatch.setenv("FASTVIDEO_FA4", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_MINIMAX_H3_FUSIONS", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_VSA_SM100A", "1")
|
||||
monkeypatch.setenv("FASTVIDEO_INFERENCE_TORCH_COMPILE", "1")
|
||||
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
monkeypatch.setattr(backend, "_gpu_mem", lambda: "alloc=0.00GiB, reserved=0.00GiB")
|
||||
backend.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
config = captured["config"]
|
||||
assert captured["download"] == {
|
||||
"repo_id": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"filename": "vsa-datafree/adapter_model.safetensors",
|
||||
}
|
||||
assert config.model_path == "MiniMaxAI/MiniMax-H3"
|
||||
assert config.pipeline.components.lora_path.endswith("vsa-datafree/adapter_model.safetensors")
|
||||
assert config.pipeline.components.lora_strength == 1.0
|
||||
assert config.pipeline.experimental == {
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"inference_torch_compile": False,
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
}
|
||||
assert config.engine.num_gpus == 4
|
||||
assert config.engine.parallelism.tp_size == 1
|
||||
assert config.engine.parallelism.sp_size == 4
|
||||
assert config.engine.offload.dit is False
|
||||
assert config.engine.offload.dit_layerwise is False
|
||||
assert config.engine.offload.text_encoder is True
|
||||
assert config.engine.offload.vae is True
|
||||
assert config.engine.compile.vae_enabled is True
|
||||
assert config.engine.use_fsdp_inference is False
|
||||
assert os.environ["FASTVIDEO_ATTENTION_BACKEND"] == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert os.environ["FASTVIDEO_FA4"] == "1"
|
||||
assert os.environ["FASTVIDEO_MINIMAX_H3_FUSIONS"] == "all"
|
||||
assert os.environ["FASTVIDEO_VSA_SM100A"] == "0"
|
||||
assert "FASTVIDEO_INFERENCE_TORCH_COMPILE" not in os.environ
|
||||
|
||||
|
||||
def test_initialize_selects_declared_generation_backend(monkeypatch):
|
||||
"""The GPU worker constructs the backend that the active model profile declares."""
|
||||
from unittest.mock import Mock
|
||||
|
||||
selected_backend = Mock()
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: selected_backend,
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend_name == "minimax_h3"
|
||||
assert worker.backend is selected_backend
|
||||
selected_backend.initialize.assert_called_once_with(FASTH3_MODEL_CONFIG)
|
||||
|
||||
|
||||
def test_initialize_failure_clears_backend_ownership(monkeypatch):
|
||||
"""A failed family change leaves the GPU worker explicitly uninitialized."""
|
||||
ltx_backend = SimpleNamespace(initialize=lambda config: None, shutdown=lambda: None)
|
||||
|
||||
def fail_initialize(config):
|
||||
del config
|
||||
raise RuntimeError("load failed")
|
||||
|
||||
fasth3_backend = SimpleNamespace(
|
||||
initialize=fail_initialize,
|
||||
shutdown=lambda: None,
|
||||
)
|
||||
backends = {
|
||||
"ltx2": ltx_backend,
|
||||
"minimax_h3": fasth3_backend,
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: backends[backend_name],
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
worker.initialize({"generation_backend": "ltx2"})
|
||||
|
||||
with pytest.raises(RuntimeError, match="load failed"):
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend is None
|
||||
assert worker.backend_name is None
|
||||
assert worker.model_config == {"generation_backend": "ltx2"}
|
||||
|
||||
|
||||
def test_generate_step_uses_last_frame_for_continuation(monkeypatch):
|
||||
"""A later segment receives the prior segment's last decoded frame."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
first_result = backend.generate_step(
|
||||
"first prompt",
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
second_result = backend.generate_step(
|
||||
"second prompt",
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
|
||||
first_request = backend.generator.requests[0]
|
||||
assert first_request.inputs.pil_image is None
|
||||
assert first_request.negative_prompt == ""
|
||||
assert first_request.sampling.height == 768
|
||||
assert first_request.sampling.width == 1344
|
||||
assert first_request.sampling.num_frames == 124
|
||||
assert first_request.sampling.num_inference_steps == 5
|
||||
assert first_request.sampling.fps == 24
|
||||
assert first_request.sampling.guidance_scale == 1.0
|
||||
assert first_request.sampling.batch_cfg is False
|
||||
assert first_request.sampling.seed == 1000
|
||||
assert first_request.output.save_video is False
|
||||
assert first_request.output.return_frames is True
|
||||
assert backend.generator.conditioning_pixels[1].tolist() == np.full((2, 3, 3), 20).tolist()
|
||||
assert first_result.head_trim_frames == 0
|
||||
assert first_result.head_trim_audio_frames == 0
|
||||
assert second_result.head_trim_frames == 1
|
||||
assert second_result.head_trim_audio_frames == 1
|
||||
assert second_result.audio_sample_rate == 44100
|
||||
|
||||
|
||||
def test_generate_step_reset_uses_text_to_video_path(monkeypatch):
|
||||
"""Resetting continuation produces an unconditioned text-to-video request."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
backend.generate_step("first prompt", 1, None, True)
|
||||
reset_result = backend.generate_step("reset prompt", 2, None, True)
|
||||
|
||||
assert backend.generator.requests[-1].inputs.pil_image is None
|
||||
assert reset_result.head_trim_frames == 0
|
||||
assert reset_result.head_trim_audio_frames == 0
|
||||
|
||||
|
||||
def test_generate_step_missing_continuation_frame(monkeypatch):
|
||||
"""A later segment fails when no reset or retained frame defines its input."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
|
||||
with pytest.raises(RuntimeError, match="requires a retained continuation frame"):
|
||||
backend.generate_step("later prompt", 2, None, False)
|
||||
|
||||
assert backend.generator.requests == []
|
||||
|
||||
|
||||
def test_warmup_exercises_text_and_first_frame_paths(monkeypatch):
|
||||
"""Warmup covers both request shapes used by a DreamVerse session."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
timings = backend.warmup("warmup prompt")
|
||||
|
||||
assert backend.generator.conditioning_pixels[0] is None
|
||||
assert backend.generator.conditioning_pixels[1] is not None
|
||||
assert backend.continuation_image is None
|
||||
assert "warmup_text_to_video_ms" in timings
|
||||
assert "warmup_first_frame_to_video_ms" in timings
|
||||
@@ -331,11 +331,11 @@ def test_rewrite_prompt_sequence_accepts_numbered_prose_output():
|
||||
]
|
||||
|
||||
|
||||
def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
def test_enhance_prompt_uses_groq_when_it_returns_first():
|
||||
enhancer = _build_staged_enhancer(
|
||||
cerebras_payload=_chat_payload_with_content('{"prompt":"Cerebras prompt"}'),
|
||||
groq_payload=_chat_payload_with_content('{"prompt":"Groq prompt"}'),
|
||||
cerebras_delay_s=0.01,
|
||||
cerebras_delay_s=0.08,
|
||||
groq_delay_s=0.01,
|
||||
)
|
||||
|
||||
@@ -346,12 +346,12 @@ def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.provider == "cerebras"
|
||||
assert result.provider == "groq"
|
||||
assert result.model == "gpt-test"
|
||||
assert result.prompt == "Cerebras prompt"
|
||||
assert result.prompt == "Groq prompt"
|
||||
assert enhancer.get_provider_success_counts() == {
|
||||
"cerebras": 1,
|
||||
"groq": 0,
|
||||
"cerebras": 0,
|
||||
"groq": 1,
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@
|
||||
<mxCell id="dispatcher" value="command dispatcher

gpu_worker_process() branches on
CommandType; asserts payload type

INIT / WARMUP / RELOAD_MODEL
USER_JOIN / USER_STEP / USER_LEAVE
SHUTDOWN" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe6cc;strokeColor=#d79b00;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
video_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
ltx2_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="stream_av" value="stream_fmp4()
av_streaming.py:121

trims overlap, pipes to ffmpeg,
publishes StreamInit / StreamChunk /
StreamComplete via callback" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#b1d8d7;strokeColor=#23445d;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
@@ -79,13 +79,13 @@
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-2" value="" style="edgeStyle=none;html=1;" parent="1" source="generator" target="Ot8BU52QTIb4EhyRSe7I-1" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
video_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
ltx2_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="ffmpeg" value="ffmpeg subprocess

libx264 / *_nvenc
fragmented mp4" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="800" y="1300" width="260" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="caches" value="ContinuationState
video_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="caches" value="ContinuationState
ltx2_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_cp" value="acquire" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#6c8ebf;endArrow=classic;fontSize=11;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="client" target="pool" edge="1">
|
||||
@@ -250,7 +250,7 @@
|
||||
<mxPoint x="690" y="880"/>
|
||||
</Array>
|
||||
</mxCell>
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender video_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender ltx2_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="39" y="-200" width="270" height="380" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-1" value="FastVideo video_generator" style="whiteSpace=wrap;html=1;fontSize=11;fillColor=#e1d5e7;strokeColor=#9673a6;rounded=1;" parent="1" vertex="1">
|
||||
@@ -389,10 +389,10 @@
|
||||
<mxCell id="cw2" value="from fastvideo.entrypoints.video_generator import VideoGenerator
from fastvideo.models.dits.ltx2 import DEFAULT_LTX2_AUDIO_*

** Dreamverse reaches into fastvideo internals here **" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="695" width="550" height="60" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (video_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (ltx2_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="765" width="550" height="95" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (video_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (ltx2_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="870" width="550" height="55" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw5" value="enter main worker loop → waits for JOIN_USER / USER_STEP / LEAVE" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#c8e6c9;strokeColor=#388e3c;fontSize=11;fontStyle=1;fontFamily=monospace;" parent="1" vertex="1">
|
||||
@@ -534,7 +534,7 @@
|
||||
<mxPoint x="1040" y="1610" as="targetPoint"/>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (video_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (ltx2_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="955" y="1640" width="180" height="70" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="dm11" value="10b. resp_q.put(MediaInit / MediaChunk / MediaComplete / StepComplete)" style="endArrow=classic;html=1;strokeColor=#b85450;fontSize=10;labelBackgroundColor=#ffffff;" parent="1" edge="1">
|
||||
|
||||
File diff suppressed because one or more lines are too long
|
Before Width: | Height: | Size: 85 KiB After Width: | Height: | Size: 85 KiB |
@@ -0,0 +1,63 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
|
||||
import { skipWithoutMock } from './helpers';
|
||||
|
||||
test.describe('create job interactions', () => {
|
||||
skipWithoutMock();
|
||||
|
||||
for (const jobType of ['inference', 'finetuning', 'distillation']) {
|
||||
test(`${jobType} remains interactive after repeated dialog dismissals`, async ({ page }) => {
|
||||
await page.goto(`/${jobType}`);
|
||||
const trigger = page.getByRole('button', { name: 'Create Job', exact: true });
|
||||
const dialog = page.getByRole('dialog');
|
||||
|
||||
// Exercise both dismissal paths and reopen without reloading the page.
|
||||
for (const closeWithEscape of [false, true]) {
|
||||
await trigger.click();
|
||||
await page.getByRole('menuitem').first().click();
|
||||
await expect(dialog).toBeVisible();
|
||||
if (closeWithEscape) {
|
||||
await page.keyboard.press('Escape');
|
||||
} else {
|
||||
await dialog.getByRole('button', { name: 'Close', exact: true }).click();
|
||||
}
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
await expect(trigger).toBeFocused();
|
||||
}
|
||||
|
||||
await page.getByRole('link', { name: 'Datasets', exact: true }).click();
|
||||
await expect(page).toHaveURL(/\/datasets$/);
|
||||
});
|
||||
}
|
||||
|
||||
test('preserves keyboard menu dismissal and dialog focus trapping', async ({ page }) => {
|
||||
await page.goto('/inference');
|
||||
const trigger = page.getByRole('button', { name: 'Create Job', exact: true });
|
||||
await trigger.focus();
|
||||
await page.keyboard.press('Enter');
|
||||
const firstItem = page.getByRole('menuitem').first();
|
||||
await expect(firstItem).toBeFocused();
|
||||
await page.keyboard.press('Escape');
|
||||
await expect(page.getByRole('menu')).toBeHidden();
|
||||
await expect(trigger).toBeFocused();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
|
||||
await page.keyboard.press('Enter');
|
||||
await expect(firstItem).toBeFocused();
|
||||
await page.keyboard.press('Enter');
|
||||
const dialog = page.getByRole('dialog');
|
||||
await expect(dialog).toBeVisible();
|
||||
await expect(dialog.getByLabel('Name (optional)')).toBeFocused();
|
||||
|
||||
// Shift+Tab from the first field wraps to Close, then Tab wraps back.
|
||||
await page.keyboard.press('Shift+Tab');
|
||||
await expect(dialog.getByRole('button', { name: 'Close', exact: true })).toBeFocused();
|
||||
await page.keyboard.press('Tab');
|
||||
await expect(dialog.getByLabel('Name (optional)')).toBeFocused();
|
||||
await page.keyboard.press('Escape');
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(trigger).toBeFocused();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
});
|
||||
});
|
||||
@@ -1,6 +1,6 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
|
||||
import { skipWithoutMock } from './helpers';
|
||||
import { API_BASE, skipWithoutMock } from './helpers';
|
||||
|
||||
/**
|
||||
* Create-job flow: open the Create Job modal on /inference, fill the prompt
|
||||
@@ -10,7 +10,8 @@ import { skipWithoutMock } from './helpers';
|
||||
test.describe('create inference job', () => {
|
||||
skipWithoutMock();
|
||||
|
||||
test('creates a T2V job and shows it in the queue', async ({ page }) => {
|
||||
test('creates a T2V job and starts it without refreshing', async ({ page, request }) => {
|
||||
await request.put(`${API_BASE}/settings`, { data: { autoStartJob: false } });
|
||||
await page.goto('/inference');
|
||||
|
||||
// The trigger opens a real menu on click, so this path works for touch,
|
||||
@@ -38,5 +39,20 @@ test.describe('create inference job', () => {
|
||||
// Modal closes and the queue refreshes with the newly created job.
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(page.getByText(prompt)).toBeVisible();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
|
||||
const card = page.getByRole('article').filter({ hasText: prompt });
|
||||
await expect(card.getByText('pending', { exact: true })).toBeVisible();
|
||||
const started = page.waitForResponse((response) =>
|
||||
response.url().startsWith(`${API_BASE}/jobs/`) &&
|
||||
response.url().endsWith('/start') &&
|
||||
response.request().method() === 'POST',
|
||||
);
|
||||
await card.getByRole('button', { name: 'Start', exact: true }).click();
|
||||
expect((await started).ok()).toBe(true);
|
||||
await expect(card.getByText('running', { exact: true })).toBeVisible();
|
||||
|
||||
await page.getByRole('link', { name: 'Datasets', exact: true }).click();
|
||||
await expect(page).toHaveURL(/\/datasets$/);
|
||||
});
|
||||
});
|
||||
|
||||
Generated
+868
-722
File diff suppressed because it is too large
Load Diff
@@ -17,20 +17,11 @@
|
||||
"start:all": "concurrently --kill-others-on-fail \"npm:start:api\" \"npm:start:web\""
|
||||
},
|
||||
"dependencies": {
|
||||
"@radix-ui/react-dialog": "^1.1.0",
|
||||
"@radix-ui/react-dropdown-menu": "^2.1.24",
|
||||
"@radix-ui/react-label": "^2.1.8",
|
||||
"@radix-ui/react-scroll-area": "^1.2.10",
|
||||
"@radix-ui/react-select": "^2.2.6",
|
||||
"@radix-ui/react-separator": "^1.1.8",
|
||||
"@radix-ui/react-slider": "^1.2.0",
|
||||
"@radix-ui/react-slot": "^1.2.4",
|
||||
"@radix-ui/react-switch": "^1.1.0",
|
||||
"@radix-ui/react-tabs": "^1.1.0",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"lucide-react": "^0.577.0",
|
||||
"next": "15.5.18",
|
||||
"radix-ui": "^1.6.7",
|
||||
"react": "^19.1.0",
|
||||
"react-dom": "^19.1.0",
|
||||
"sonner": "^2.0.7",
|
||||
|
||||
@@ -1,24 +1,26 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { render, screen, waitFor, within } from '@testing-library/react';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import CreateJobButton from './CreateJobButton';
|
||||
import { getDatasets, getModels } from '@/lib/api';
|
||||
|
||||
vi.mock('./CreateJobModal', () => ({
|
||||
default: ({
|
||||
isOpen,
|
||||
workloadType,
|
||||
}: {
|
||||
isOpen: boolean;
|
||||
workloadType: string;
|
||||
}) =>
|
||||
isOpen ? (
|
||||
<div role="dialog" data-workload-type={workloadType}>
|
||||
Create job form
|
||||
</div>
|
||||
) : null,
|
||||
vi.mock('@/lib/api', () => ({
|
||||
createJob: vi.fn(),
|
||||
getModels: vi.fn(),
|
||||
getDatasets: vi.fn(),
|
||||
uploadImage: vi.fn(),
|
||||
getSettings: vi.fn(),
|
||||
updateSettings: vi.fn(),
|
||||
}));
|
||||
|
||||
beforeEach(() => {
|
||||
vi.mocked(getModels).mockResolvedValue([
|
||||
{ id: 'wan/t2v-1.3b', label: 'Wan T2V' },
|
||||
]);
|
||||
vi.mocked(getDatasets).mockResolvedValue([]);
|
||||
});
|
||||
|
||||
describe('CreateJobButton', () => {
|
||||
it('opens the workload menu on click and selects an item', async () => {
|
||||
const user = userEvent.setup();
|
||||
@@ -27,10 +29,9 @@ describe('CreateJobButton', () => {
|
||||
await user.click(screen.getByRole('button', { name: 'Create Job' }));
|
||||
await user.click(screen.getByRole('menuitem', { name: /I2V/i }));
|
||||
|
||||
expect(screen.getByRole('dialog')).toHaveAttribute(
|
||||
'data-workload-type',
|
||||
'i2v',
|
||||
);
|
||||
expect(
|
||||
screen.getByRole('dialog', { name: 'New Inference Job (I2V)' }),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('opens and operates the workload menu from the keyboard', async () => {
|
||||
@@ -45,9 +46,42 @@ describe('CreateJobButton', () => {
|
||||
expect(firstItem).toHaveFocus();
|
||||
await user.keyboard('{Enter}');
|
||||
|
||||
expect(screen.getByRole('dialog')).toHaveAttribute(
|
||||
'data-workload-type',
|
||||
't2v',
|
||||
expect(
|
||||
screen.getByRole('dialog', { name: 'New Inference Job (T2V)' }),
|
||||
).toBeInTheDocument();
|
||||
await user.keyboard('{Escape}');
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument(),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(document.body.style.pointerEvents).not.toBe('none'),
|
||||
);
|
||||
expect(trigger).toHaveFocus();
|
||||
});
|
||||
|
||||
it.each(['inference', 'finetuning', 'distillation'] as const)(
|
||||
'restores page interaction after closing the real %s dialog',
|
||||
async (jobType) => {
|
||||
const user = userEvent.setup();
|
||||
render(<CreateJobButton jobType={jobType} />);
|
||||
const trigger = screen.getByRole('button', { name: 'Create Job' });
|
||||
|
||||
// Keep the real Dialog mounted: mocking it hides conflicting Radix layers.
|
||||
for (let attempt = 0; attempt < 2; attempt++) {
|
||||
await user.click(trigger);
|
||||
await user.click(screen.getAllByRole('menuitem')[0]);
|
||||
const dialog = screen.getByRole('dialog');
|
||||
await user.click(
|
||||
within(dialog).getByRole('button', { name: 'Close' }),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument(),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(document.body.style.pointerEvents).not.toBe('none'),
|
||||
);
|
||||
expect(trigger).toHaveFocus();
|
||||
}
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
import * as React from 'react';
|
||||
import { ChevronDown } from 'lucide-react';
|
||||
import * as DropdownMenu from '@radix-ui/react-dropdown-menu';
|
||||
import { DropdownMenu } from 'radix-ui';
|
||||
|
||||
import CreateJobModal from '@/components/jobs/CreateJobModal';
|
||||
import { Button } from '@/components/ui/button';
|
||||
@@ -16,6 +16,7 @@ interface CreateJobButtonProps {
|
||||
|
||||
export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
const options = WORKLOAD_OPTIONS[jobType] ?? [];
|
||||
const triggerRef = React.useRef<HTMLButtonElement>(null);
|
||||
|
||||
const [modalOpen, setModalOpen] = React.useState(false);
|
||||
const [workloadType, setWorkloadType] = React.useState(
|
||||
@@ -36,7 +37,7 @@ export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
<>
|
||||
<DropdownMenu.Root>
|
||||
<DropdownMenu.Trigger asChild>
|
||||
<Button type="button" className="gap-1.5">
|
||||
<Button ref={triggerRef} type="button" className="gap-1.5">
|
||||
Create Job
|
||||
<ChevronDown className="size-3.5 opacity-85" aria-hidden />
|
||||
</Button>
|
||||
@@ -66,6 +67,11 @@ export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
<CreateJobModal
|
||||
isOpen={modalOpen}
|
||||
onClose={() => setModalOpen(false)}
|
||||
onCloseAutoFocus={(event) => {
|
||||
// This dialog opens from a menu item, so it has no DialogTrigger.
|
||||
event.preventDefault();
|
||||
triggerRef.current?.focus();
|
||||
}}
|
||||
onSuccess={handleSuccess}
|
||||
jobType={jobType}
|
||||
workloadType={workloadType}
|
||||
|
||||
@@ -55,6 +55,9 @@ import { jobToFormFields, type JobLike } from '@/lib/jobToFields';
|
||||
export interface CreateJobModalProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
onCloseAutoFocus?: React.ComponentProps<
|
||||
typeof DialogContent
|
||||
>['onCloseAutoFocus'];
|
||||
onSuccess: () => void;
|
||||
jobType: JobType;
|
||||
workloadType: string;
|
||||
@@ -67,6 +70,7 @@ export interface CreateJobModalProps {
|
||||
export default function CreateJobModal({
|
||||
isOpen,
|
||||
onClose,
|
||||
onCloseAutoFocus,
|
||||
onSuccess,
|
||||
jobType,
|
||||
workloadType,
|
||||
@@ -644,6 +648,7 @@ export default function CreateJobModal({
|
||||
>
|
||||
<DialogContent
|
||||
className="max-h-[90vh] w-[90vw] max-w-[850px] overflow-y-auto"
|
||||
onCloseAutoFocus={onCloseAutoFocus}
|
||||
onEscapeKeyDown={(e) => {
|
||||
if (isSubmitting) e.preventDefault();
|
||||
}}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import * as React from 'react';
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { Button } from './button';
|
||||
import { Input } from './input';
|
||||
@@ -8,6 +10,24 @@ import { Slider } from './slider';
|
||||
import { Switch } from './switch';
|
||||
|
||||
describe('shared control accessibility', () => {
|
||||
it('forwards refs and click handlers to the asChild button', async () => {
|
||||
const user = userEvent.setup();
|
||||
const ref = React.createRef<HTMLButtonElement>();
|
||||
const onClick = vi.fn();
|
||||
render(
|
||||
<Button asChild ref={ref} onClick={onClick}>
|
||||
<button type="button">Slotted action</button>
|
||||
</Button>,
|
||||
);
|
||||
|
||||
const button = screen.getByRole('button', { name: 'Slotted action' });
|
||||
expect(screen.getAllByRole('button')).toHaveLength(1);
|
||||
expect(ref.current).toBe(button);
|
||||
await user.click(button);
|
||||
expect(onClick).toHaveBeenCalledTimes(1);
|
||||
expect(button).toHaveFocus();
|
||||
});
|
||||
|
||||
it('keeps button, input, and select targets at least 44px tall', () => {
|
||||
render(
|
||||
<>
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"use client";
|
||||
|
||||
import * as React from "react";
|
||||
import { Slot } from "@radix-ui/react-slot";
|
||||
import { Slot } from "radix-ui";
|
||||
import { cva, type VariantProps } from "class-variance-authority";
|
||||
|
||||
import { cn } from "@/lib/utils";
|
||||
@@ -37,7 +37,7 @@ export interface ButtonProps extends React.ButtonHTMLAttributes<HTMLButtonElemen
|
||||
}
|
||||
|
||||
const Button = React.forwardRef<HTMLButtonElement, ButtonProps>(({ className, variant, size, asChild = false, ...props }, ref) => {
|
||||
const Comp = asChild ? Slot : "button";
|
||||
const Comp = asChild ? Slot.Root : "button";
|
||||
return <Comp className={cn(buttonVariants({ variant, size, className }))} ref={ref} {...props} />;
|
||||
});
|
||||
Button.displayName = "Button";
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as DialogPrimitive from '@radix-ui/react-dialog';
|
||||
import { Dialog as DialogPrimitive } from 'radix-ui';
|
||||
import { X } from 'lucide-react';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as LabelPrimitive from '@radix-ui/react-label';
|
||||
import { Label as LabelPrimitive } from 'radix-ui';
|
||||
import { cva, type VariantProps } from 'class-variance-authority';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as ScrollAreaPrimitive from '@radix-ui/react-scroll-area';
|
||||
import { ScrollArea as ScrollAreaPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SelectPrimitive from '@radix-ui/react-select';
|
||||
import { Select as SelectPrimitive } from 'radix-ui';
|
||||
import { Check, ChevronDown, ChevronUp } from 'lucide-react';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SeparatorPrimitive from '@radix-ui/react-separator';
|
||||
import { Separator as SeparatorPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SliderPrimitive from '@radix-ui/react-slider';
|
||||
import { Slider as SliderPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SwitchPrimitives from '@radix-ui/react-switch';
|
||||
import { Switch as SwitchPrimitives } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as TabsPrimitive from '@radix-ui/react-tabs';
|
||||
import { Tabs as TabsPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"version": 6,
|
||||
"version": 11,
|
||||
"recipes": [
|
||||
{
|
||||
"id": "fastwan21-t2v",
|
||||
@@ -437,18 +437,21 @@
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_t2v/minimax_h3_t2v.mp4",
|
||||
"modes": ["T2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["The checked-in example defaults to four-way sequence parallelism. It does not record a GPU model or memory requirement."]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-cuda",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 Preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 Preview on CUDA",
|
||||
"summary": "Run the DMD2-distilled FastH3 Preview with four DiT forwards, trained H3 sparse attention, compiled decode, and synchronized audio.",
|
||||
"label": "FastH3 V1 on CUDA",
|
||||
"summary": "Run FastH3 V1 with four DiT forwards, trained H3 sparse attention, compiled decode, and synchronized audio.",
|
||||
"model": "FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2",
|
||||
"source": "examples/inference/basic/basic_fasth3.py",
|
||||
"serving": {
|
||||
@@ -467,6 +470,10 @@
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3/",
|
||||
"modes": ["T2VA", "4-step FastH3"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": ["The default all profile is the measured GB200 performance route and can change floating-point operation order. Use --profile strict --no-inference-torch-compile for the eager strict route."]
|
||||
},
|
||||
{
|
||||
@@ -477,13 +484,13 @@
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\""
|
||||
},
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 Preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 Preview on MLX",
|
||||
"summary": "Run FastH3 Preview on Apple Silicon with a locally converted INT6 DiT, streamed Qwen3-VL conditioning, and native MLX video and audio VAEs.",
|
||||
"label": "FastH3 V1 on MLX",
|
||||
"summary": "Run FastH3 V1 on Apple Silicon with a locally converted INT6 DiT, streamed Qwen3-VL conditioning, and native MLX video and audio VAEs.",
|
||||
"model": "FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2",
|
||||
"source": "examples/inference/basic/mlx_fasth3.py",
|
||||
"command": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\"\npython examples/inference/basic/mlx_fasth3.py --model-root ./FastH3-Preview-v0.2 --mlx-checkpoint ./FastH3-MLX/int6 --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --height 480 --width 832 --num-frames 124 --seed 2026 --output-path ./outputs/fasth3_int6.mp4",
|
||||
@@ -499,10 +506,155 @@
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 with H.264 video and stereo AAC audio at outputs/fasth3_int6.mp4",
|
||||
"modes": ["T2VA", "temporal --fast", "spatial --fast-spatial", "opt-in VSA"],
|
||||
"knobs": [
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"The MLX path supports T2VA, optional temporal --fast, optional spatial --fast-spatial, and opt-in VSA on --include-vsa checkpoints. FL2VA, Ref2VA, and two-pass refinement are not wired."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-spark",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on one DGX Spark",
|
||||
"summary": "Run FastH3 V1 on one GB10 with Triton VSA, FA4 off, and lazy module load. Height, width, frames, and steps in the YAML are examples.",
|
||||
"model": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree",
|
||||
"source": "examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_spark.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3"
|
||||
},
|
||||
"command": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"device": "spark",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/fasth3_spark/",
|
||||
"modes": ["T2VA", "1-Spark"],
|
||||
"limitations": [
|
||||
"Install from the DGX Spark guide, not the generic CUDA extra. GB10 has no FA4 / sm_100a VSA kernel; keep FASTVIDEO_FA4=0 and FASTVIDEO_VSA_SM100A=0.",
|
||||
"Legal num_frames values are 17n+5, capped at 345 (15 s). Native 16:9 sizes include 832x480 and 1344x768.",
|
||||
"Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Do not pass --no-lazy-module-load on this box.",
|
||||
"A 345-frame request on one Spark can OOM. Prefer 124 or 243 frames, TAEH3 decode, or two Sparks over QSFP."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-spark-pair",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on two DGX Sparks",
|
||||
"summary": "Run one FastH3 clip across two GB10s with Ray sequence parallel over QSFP RoCE. Sequential load and lazy module load stay on because SP replicates the DiT on each node.",
|
||||
"model": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree",
|
||||
"source": "examples/inference/basic/basic_fasth3_spark_pair.yaml",
|
||||
"command": "source examples/inference/optimizations/spark_pair_env.sh && FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_VAE_PARALLEL_DECODE=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"device": "spark",
|
||||
"gpu_count": 2,
|
||||
"accelerator": "NVIDIA GB10 (DGX Spark pair)",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1803"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 under outputs/fasth3_spark_pair/",
|
||||
"modes": ["T2VA", "2-Spark SP"],
|
||||
"limitations": [
|
||||
"Requires a two-node Ray cluster on the QSFP interconnect. There is no cookbook server for this path; use Python / generate.",
|
||||
"Height, width, frames, and steps in the YAML are examples. Edit them or pass CLI flags. See docs/getting_started/installation/spark_pair.md."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-cuda",
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
"group_task": "8-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V2 on CUDA",
|
||||
"summary": "Run FastH3 V2, the eight-forward checkpoint (video/audio shifts 10/3, VSA 0.8, 64-token tiles) with the trained DMD ladder loaded from the checkpoint's fastvideo_inference.json.",
|
||||
"model": "FastVideo/FastVideo-FastH3-8-Step-V2",
|
||||
"source": "examples/inference/basic/basic_fasth3_8step.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_8step.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3_8step.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile strict --no-inference-torch-compile --no-compile-vae",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 4,
|
||||
"accelerator": "NVIDIA GB200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1852"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3_8step/",
|
||||
"modes": ["T2VA", "8-step FastH3"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"Nine sigma-grid points (eight transformer forwards) are fixed by the checkpoint's trained ladder; the example rejects any other --steps.",
|
||||
"Validated with the eager strict route (--profile strict --no-inference-torch-compile --no-compile-vae). The compiled all profile has not been measured for this checkpoint.",
|
||||
"T2AV only; no FL2VA/Ref2VA distillation and no matching LoRA. V2 uses eight forwards rather than V1's four."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-mlx",
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3_8step.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa"
|
||||
},
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
"group_task": "8-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V2 on MLX",
|
||||
"summary": "Run FastH3 V2 on Apple Silicon with a locally converted INT8 DiT, the checkpoint's DMD contract ladder, and trained VSA (sparsity 0.8, 64-token tiles).",
|
||||
"model": "FastVideo/FastVideo-FastH3-8-Step-V2",
|
||||
"source": "examples/inference/basic/mlx_fasth3_8step.py",
|
||||
"command": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa\npython examples/inference/basic/mlx_fasth3_8step.py --model-root ./FastH3-8-Step-V2 --mlx-checkpoint ./FastH3-8-Step-V2-MLX/int8 --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --height 480 --width 832 --num-frames 124 --seed 2026 --output-path ./outputs/fasth3_8step_int8.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"peak_memory": "27.39 GiB peak MLX memory during denoising",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1863"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 with H.264 video and stereo AAC audio at outputs/fasth3_8step_int8.mp4",
|
||||
"modes": ["T2VA", "8-step FastH3", "trained VSA"],
|
||||
"knobs": [
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"Convert with --include-vsa. mlx_fasth3_8step.py turns VSA on (sparsity 0.8, tile 64). A dense export fails at configure_vsa.",
|
||||
"--steps 8 or 9 both run the eight trained forwards. Reuse the preview VAE, audio VAE, text encoder, and tokenizer if those directories already exist.",
|
||||
"The MLX path supports T2VA only. FL2VA, Ref2VA, and two-pass refinement are not wired."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "minimax-h3-fl2va",
|
||||
"family": "minimax_h3",
|
||||
@@ -518,6 +670,9 @@
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_fl2va/minimax_h3_fl2va.mp4",
|
||||
"modes": ["FL2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["Pass --last-image to constrain the final frame. The checked-in source defaults to four GPUs."]
|
||||
},
|
||||
{
|
||||
@@ -535,6 +690,9 @@
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_ref2va/minimax_h3_ref2va.mp4",
|
||||
"modes": ["Ref2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["Pass --reference-audio for an additional audio reference. The checked-in source defaults to four GPUs."]
|
||||
},
|
||||
{
|
||||
@@ -558,6 +716,10 @@
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3_lora_preview/",
|
||||
"modes": ["T2VA", "FastH3 LoRA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": ["Supply a compatible FastH3 adapter. The script infers dense or VSA attention from the adapter payload unless you override it."]
|
||||
},
|
||||
{
|
||||
|
||||
+242
-30
@@ -89,11 +89,38 @@
|
||||
|
||||
const groupIdFor = (recipe) => recipe.group || recipe.id;
|
||||
|
||||
const gpuCountLabel = (hardware) => {
|
||||
const gpuCountLabel = (hardware, runtimeId) => {
|
||||
const count = hardware?.gpu_count;
|
||||
if (count == null) return "";
|
||||
if (runtimeId === "spark") return `${count} Spark${count === 1 ? "" : "s"}`;
|
||||
return `${count} GPU${count === 1 ? "" : "s"}`;
|
||||
};
|
||||
|
||||
const knobsFor = (recipe) => recipe.knobs || [];
|
||||
|
||||
const knobOptions = (knob) =>
|
||||
knob.options.map((option) =>
|
||||
option !== null && typeof option === "object" ? option : { value: option, label: String(option) },
|
||||
);
|
||||
|
||||
const knobDefaultLabel = (knob) => {
|
||||
const match = knobOptions(knob).find((option) => option.value === knob.default);
|
||||
return match ? match.label : String(knob.default);
|
||||
};
|
||||
|
||||
// Knob flags are always shown explicitly in the displayed command, even at
|
||||
// their default value, so the command stays copy-pasteable and precise
|
||||
// about what it runs -- not just "trust the script's own default".
|
||||
const appendKnobFlags = (commandText, knobs, knobValues) => {
|
||||
const flags = knobs
|
||||
.filter((knob) => knobValues[knob.key] !== undefined)
|
||||
.map((knob) => `${knob.flag} ${knobValues[knob.key]}`);
|
||||
if (!flags.length) return commandText;
|
||||
const lines = commandText.split("\n");
|
||||
lines[lines.length - 1] = `${lines[lines.length - 1]} ${flags.join(" ")}`;
|
||||
return lines.join("\n");
|
||||
};
|
||||
|
||||
const runtimeFor = (recipe) => {
|
||||
const platform = recipe.hardware?.platform || "cuda";
|
||||
if (platform === "mlx") {
|
||||
@@ -109,17 +136,24 @@
|
||||
if (platform === "mps") {
|
||||
return {
|
||||
id: "mps",
|
||||
label: "Apple Silicon · MPS",
|
||||
label: "Apple Silicon · PyTorch MPS",
|
||||
hint: recipe.hardware?.minimum_memory || recipe.hardware?.system_memory || "Memory not recorded",
|
||||
};
|
||||
}
|
||||
const hardware = recipe.hardware || {};
|
||||
if (hardware.device === "spark") {
|
||||
return {
|
||||
id: "spark",
|
||||
label: "NVIDIA DGX Spark",
|
||||
hint: "GB10 · 128 GB unified memory",
|
||||
};
|
||||
}
|
||||
return {
|
||||
id: "cuda",
|
||||
label: "NVIDIA CUDA",
|
||||
hint: hardware.accelerator
|
||||
? `${hardware.accelerator} · ${gpuCountLabel(hardware)}`
|
||||
: `${gpuCountLabel(hardware)} configured · GPU model not recorded`,
|
||||
? `${hardware.accelerator} · ${gpuCountLabel(hardware, "cuda")}`
|
||||
: `${gpuCountLabel(hardware, "cuda")} configured · GPU model not recorded`,
|
||||
};
|
||||
};
|
||||
|
||||
@@ -136,8 +170,11 @@
|
||||
.filter(Boolean)
|
||||
.join(" · ");
|
||||
}
|
||||
if (hardware.accelerator) return [hardware.accelerator, gpuCountLabel(hardware)].join(" · ");
|
||||
return `NVIDIA CUDA · ${gpuCountLabel(hardware)} configured · GPU model and VRAM not recorded`;
|
||||
if (runtime.id === "spark") {
|
||||
return [hardware.accelerator || "NVIDIA GB10", gpuCountLabel(hardware, "spark")].filter(Boolean).join(" · ");
|
||||
}
|
||||
if (hardware.accelerator) return [hardware.accelerator, gpuCountLabel(hardware, runtime.id)].join(" · ");
|
||||
return `NVIDIA CUDA · ${gpuCountLabel(hardware, "cuda")} configured · GPU model and VRAM not recorded`;
|
||||
};
|
||||
|
||||
const renderHardwareEvidence = (container, badge, recipe) => {
|
||||
@@ -157,15 +194,17 @@
|
||||
if (isValidated) {
|
||||
const recorded = [
|
||||
hardware.accelerator,
|
||||
runtime.id === "cuda" ? gpuCountLabel(hardware) : hardware.system_memory,
|
||||
runtime.id === "mlx" || runtime.id === "mps" ? hardware.system_memory : gpuCountLabel(hardware, runtime.id),
|
||||
].filter(Boolean);
|
||||
const statements = [`${recorded.join(" · ")}.`];
|
||||
if (hardware.minimum_memory) statements.push(`Documented minimum: ${hardware.minimum_memory}.`);
|
||||
if (hardware.peak_memory) statements.push(`Measured: ${hardware.peak_memory}.`);
|
||||
if (!hardware.minimum_memory) statements.push("This recorded device is not a minimum requirement.");
|
||||
details.textContent = ` ${statements.join(" ")}`;
|
||||
} else if (runtime.id === "spark") {
|
||||
details.textContent = ` NVIDIA DGX Spark · ${gpuCountLabel(hardware, "spark")}. GB10 has no FA4 / sm_100a VSA kernel; keep Triton VSA and FA4 off.`;
|
||||
} else if (runtime.id === "cuda") {
|
||||
details.textContent = ` NVIDIA CUDA · ${gpuCountLabel(hardware)}. The source does not record the GPU model or VRAM.`;
|
||||
details.textContent = ` NVIDIA CUDA · ${gpuCountLabel(hardware, "cuda")}. The source does not record the GPU model or VRAM.`;
|
||||
} else {
|
||||
details.textContent = ` ${runtime.label}. The source does not record a device or memory requirement.`;
|
||||
}
|
||||
@@ -223,6 +262,10 @@
|
||||
const servingPanel = root.querySelector("[data-cookbook-serving]");
|
||||
const usage = root.querySelector("[data-cookbook-usage]");
|
||||
const servingAvailability = root.querySelector("[data-cookbook-serving-availability]");
|
||||
const knobsContainer = root.querySelector("[data-cookbook-knobs]");
|
||||
const deviceRow = root.querySelector("[data-cookbook-device-row]");
|
||||
const deviceOptions = root.querySelector("[data-cookbook-device-options]");
|
||||
const deviceCaption = root.querySelector("[data-cookbook-device-caption]");
|
||||
|
||||
modelOptions.setAttribute("aria-label", "Recipe");
|
||||
hardwareOptions.setAttribute("aria-label", "Runtime");
|
||||
@@ -289,16 +332,96 @@
|
||||
const clientDetails = servingPanel?.querySelector(".cookbook-serving__code");
|
||||
if (clientDetails && query.has("client")) clientDetails.open = true;
|
||||
|
||||
const knobDefs = new Map();
|
||||
familyRecipes.forEach((recipe) => knobsFor(recipe).forEach((knob) => {
|
||||
if (!knobDefs.has(knob.key)) knobDefs.set(knob.key, knob);
|
||||
}));
|
||||
const knobValues = {};
|
||||
knobDefs.forEach((knob, key) => {
|
||||
const fromQuery = query.get(key);
|
||||
const validValues = knobOptions(knob).map((option) => String(option.value));
|
||||
const useQueryValue = fromQuery !== null && validValues.includes(fromQuery);
|
||||
const raw = useQueryValue ? fromQuery : knob.default;
|
||||
knobValues[key] = typeof knob.default === "number" ? Number(raw) : raw;
|
||||
});
|
||||
|
||||
const renderKnobs = (recipe, hidden) => {
|
||||
if (!knobsContainer) return;
|
||||
const knobs = knobsFor(recipe);
|
||||
const renderedKeys = [...knobsContainer.querySelectorAll("[data-knob-row]")].map((row) => row.dataset.knobRow);
|
||||
if (renderedKeys.join(",") !== knobs.map((knob) => knob.key).join(",")) {
|
||||
knobsContainer.replaceChildren();
|
||||
knobs.forEach((knob) => {
|
||||
const row = document.createElement("div");
|
||||
row.className = "cookbook-selection-row";
|
||||
row.dataset.knobRow = knob.key;
|
||||
const labelWrap = document.createElement("div");
|
||||
labelWrap.className = "cookbook-selection-row__label";
|
||||
const strongLabel = document.createElement("strong");
|
||||
strongLabel.textContent = knob.label;
|
||||
const hintLabel = document.createElement("span");
|
||||
hintLabel.textContent = knob.hint || "";
|
||||
labelWrap.append(strongLabel, hintLabel);
|
||||
const grid = document.createElement("div");
|
||||
grid.className = "cookbook-option-grid cookbook-option-grid--hardware";
|
||||
grid.setAttribute("role", "group");
|
||||
grid.setAttribute("aria-label", knob.label);
|
||||
knobOptions(knob).forEach((option) => {
|
||||
const optionButton = document.createElement("button");
|
||||
optionButton.type = "button";
|
||||
optionButton.dataset.knobKey = knob.key;
|
||||
optionButton.dataset.knobValue = String(option.value);
|
||||
optionButton.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = option.label;
|
||||
optionButton.append(optionLabel);
|
||||
grid.append(optionButton);
|
||||
});
|
||||
row.append(labelWrap, grid);
|
||||
knobsContainer.append(row);
|
||||
});
|
||||
}
|
||||
knobsContainer.hidden = hidden || knobs.length === 0;
|
||||
knobsContainer.querySelectorAll("button[data-knob-key]").forEach((optionButton) => {
|
||||
const selected = String(knobValues[optionButton.dataset.knobKey]) === optionButton.dataset.knobValue;
|
||||
optionButton.classList.toggle("cookbook-option--selected", selected);
|
||||
optionButton.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
};
|
||||
|
||||
const recipesForRuntime = (runtimeId, groupRecipes) =>
|
||||
groupRecipes.filter((item) => runtimeFor(item).id === runtimeId);
|
||||
|
||||
const uniqueRuntimeIds = (groupRecipes) => {
|
||||
const ids = [];
|
||||
groupRecipes.forEach((item) => {
|
||||
const id = runtimeFor(item).id;
|
||||
if (!ids.includes(id)) ids.push(id);
|
||||
});
|
||||
return ids;
|
||||
};
|
||||
|
||||
const pickRecipeForRuntime = (runtimeId, preferredCount, groupRecipes) => {
|
||||
const siblings = recipesForRuntime(runtimeId, groupRecipes);
|
||||
if (!siblings.length) return null;
|
||||
if (preferredCount != null) {
|
||||
const match = siblings.find((item) => item.hardware?.gpu_count === preferredCount);
|
||||
if (match) return match;
|
||||
}
|
||||
return siblings.find((item) => item.hardware?.gpu_count === 1) || siblings[0];
|
||||
};
|
||||
|
||||
const renderRuntimeOptions = () => {
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const renderedIds = [...hardwareOptions.querySelectorAll("[data-recipe-id]")].map((option) => option.dataset.recipeId);
|
||||
if (renderedIds.join(",") === groupRecipes.map((recipe) => recipe.id).join(",")) return;
|
||||
const runtimeIds = uniqueRuntimeIds(groupRecipes);
|
||||
const renderedIds = [...hardwareOptions.querySelectorAll("[data-runtime-id]")].map((option) => option.dataset.runtimeId);
|
||||
if (renderedIds.join(",") === runtimeIds.join(",")) return;
|
||||
hardwareOptions.replaceChildren();
|
||||
groupRecipes.forEach((recipe) => {
|
||||
const runtime = runtimeFor(recipe);
|
||||
runtimeIds.forEach((runtimeId) => {
|
||||
const representative = pickRecipeForRuntime(runtimeId, 1, groupRecipes);
|
||||
const runtime = runtimeFor(representative);
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.recipeId = recipe.id;
|
||||
option.dataset.runtimeId = runtime.id;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
@@ -310,6 +433,49 @@
|
||||
});
|
||||
};
|
||||
|
||||
const renderDeviceOptions = (recipe) => {
|
||||
if (!deviceRow || !deviceOptions) return;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const runtime = runtimeFor(recipe);
|
||||
const siblings = recipesForRuntime(runtime.id, groupRecipes)
|
||||
.slice()
|
||||
.sort((left, right) => (left.hardware?.gpu_count || 0) - (right.hardware?.gpu_count || 0));
|
||||
const show = siblings.length > 1;
|
||||
deviceRow.hidden = !show;
|
||||
if (deviceCaption) {
|
||||
deviceCaption.textContent = runtime.id === "spark" ? "1 Spark or a QSFP pair" : "GPU count for this runtime";
|
||||
}
|
||||
if (!show) {
|
||||
deviceOptions.replaceChildren();
|
||||
return;
|
||||
}
|
||||
const renderedIds = [...deviceOptions.querySelectorAll("[data-recipe-id]")].map((option) => option.dataset.recipeId);
|
||||
if (renderedIds.join(",") !== siblings.map((item) => item.id).join(",")) {
|
||||
deviceOptions.replaceChildren();
|
||||
siblings.forEach((candidate) => {
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.recipeId = candidate.id;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = gpuCountLabel(candidate.hardware, runtime.id);
|
||||
const optionHint = document.createElement("span");
|
||||
optionHint.textContent = runtime.id === "spark" && candidate.hardware?.gpu_count === 2
|
||||
? "Ray sequence parallel over QSFP"
|
||||
: runtime.id === "spark"
|
||||
? "One GB10, local process"
|
||||
: `${gpuCountLabel(candidate.hardware, runtime.id)} configured`;
|
||||
option.append(optionLabel, optionHint);
|
||||
deviceOptions.append(option);
|
||||
});
|
||||
}
|
||||
deviceOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.recipeId === recipe.id;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
};
|
||||
|
||||
let notes = root.querySelector("[data-cookbook-notes]");
|
||||
if (!notes) {
|
||||
notes = document.createElement("aside");
|
||||
@@ -325,8 +491,9 @@
|
||||
let recipe = byId.get(selectedRecipeId);
|
||||
if (groupChanged || groupIdFor(recipe) !== selectedGroupId) {
|
||||
const currentRuntime = runtimeFor(recipe).id;
|
||||
const currentCount = recipe.hardware?.gpu_count;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
recipe = groupRecipes.find((candidate) => runtimeFor(candidate).id === currentRuntime) || groupRecipes[0];
|
||||
recipe = pickRecipeForRuntime(currentRuntime, currentCount, groupRecipes) || groupRecipes[0];
|
||||
selectedRecipeId = recipe.id;
|
||||
}
|
||||
|
||||
@@ -335,11 +502,14 @@
|
||||
renderRuntimeOptions();
|
||||
renderedRuntimeGroup = selectedGroupId;
|
||||
}
|
||||
renderDeviceOptions(recipe);
|
||||
const runtime = runtimeFor(recipe);
|
||||
const profile = servingPanel && servingProfiles[recipe.id];
|
||||
const useServer = Boolean(profile && usagePreference === "server");
|
||||
// The measured local profile and the server config have separate evidence.
|
||||
const activeRecipe = useServer ? { ...recipe, hardware: profile.hardware, evidence: "Source-backed" } : recipe;
|
||||
const knobs = knobsFor(recipe);
|
||||
renderKnobs(recipe, useServer);
|
||||
|
||||
if (usage) {
|
||||
usage.querySelectorAll("[data-cookbook-mode]").forEach((option) => {
|
||||
@@ -349,10 +519,10 @@
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
servingAvailability.textContent = profile
|
||||
? "The playground and API clients share one server process. Both workflows can run on your own machine."
|
||||
? "The playground and the OpenAI Python client share one server process. Both workflows can run on your own machine."
|
||||
: servingLoadFailed
|
||||
? "Server examples could not be loaded. Open the H3 server guide below, or use Python directly."
|
||||
: "This recipe uses Python directly. For the playground and API clients, choose FastH3 Preview with CUDA or MLX.";
|
||||
: "This recipe uses Python directly. FastH3 V1 and FastH3 V2 can also run a local server for the playground and the OpenAI Python client.";
|
||||
servingPanel.hidden = !useServer;
|
||||
commandBlock.hidden = useServer;
|
||||
root.querySelector("[data-cookbook-python-note]").hidden = useServer;
|
||||
@@ -360,11 +530,17 @@
|
||||
|
||||
if (useServer) {
|
||||
const isMLX = profile.runtime === "mlx";
|
||||
const isSpark = runtime.id === "spark";
|
||||
servingPanel.querySelector("[data-cookbook-server-lifetime]").textContent = isMLX
|
||||
? "Start once, then change prompts in the playground or your app. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
: isSpark
|
||||
? "Start once, then change prompts in the playground or your app. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
servingPanel.querySelector("[data-cookbook-install-guide]").href = isMLX
|
||||
? "../../getting_started/installation/mps/#run-fasth3-preview" : "../../getting_started/installation/gpu/";
|
||||
? "../../getting_started/installation/mlx/"
|
||||
: isSpark
|
||||
? "../../getting_started/installation/spark/"
|
||||
: "../../getting_started/installation/gpu/";
|
||||
servingPanel.querySelector("[data-cookbook-prepare]").hidden = !profile.prepare;
|
||||
servingPanel.querySelector("[data-cookbook-server-prepare]").textContent = profile.prepare;
|
||||
servingPanel.querySelector("[data-cookbook-server-install]").textContent = profile.install;
|
||||
@@ -392,18 +568,13 @@
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
hardwareOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.recipeId === recipe.id;
|
||||
const selected = option.dataset.runtimeId === runtime.id;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
const candidate = byId.get(option.dataset.recipeId);
|
||||
const candidateProfile = servingProfiles[candidate.id];
|
||||
const candidateRuntime = runtimeFor(candidateProfile && usagePreference === "server"
|
||||
? { ...candidate, hardware: candidateProfile.hardware } : candidate);
|
||||
option.querySelector("span").textContent = candidateRuntime.hint;
|
||||
});
|
||||
|
||||
description.textContent = useServer
|
||||
? `FastH3 Preview generates video with audio. This server profile uses the checked-in ${profile.runtime.toUpperCase()} configuration.`
|
||||
? `${recipe.group_label || recipe.label} generates video with audio. Start the local server, then use the playground or the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
: recipe.summary;
|
||||
label.textContent = useServer ? `${recipe.group_label || recipe.label} · Server` : recipe.label;
|
||||
model.textContent = recipe.model;
|
||||
@@ -418,17 +589,23 @@
|
||||
source.href = `https://github.com/hao-ai-lab/FastVideo/blob/main/${useServer ? profile.source : recipe.source}`;
|
||||
source.textContent = useServer ? "View server configuration" : "Open example source";
|
||||
modelLink.href = `https://huggingface.co/${recipe.model}`;
|
||||
command.textContent = recipe.command;
|
||||
command.textContent = useServer ? recipe.command : appendKnobFlags(recipe.command, knobs, knobValues);
|
||||
|
||||
renderHardwareEvidence(hardwareState, hardwareBadge, activeRecipe);
|
||||
|
||||
const limitations = useServer ? [
|
||||
const knobCaveats = useServer ? [] : knobs
|
||||
.filter((knob) => String(knobValues[knob.key]) !== String(knob.default))
|
||||
.map((knob) => `${knob.label} is set away from its recorded default (${knobDefaultLabel(knob)}). ` +
|
||||
"The script accepts this value, but it has not been benchmarked here.");
|
||||
const limitations = [...(useServer ? [
|
||||
profile.runtime === "mlx"
|
||||
? "This MLX server config has no recorded hardware run. Measurements from the Python recipe are not server memory requirements. Only text-to-video/audio is wired; reference inputs and fast modes are not exposed here."
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
: runtime.id === "spark"
|
||||
? "This Spark server config has no recorded serving benchmark. Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Compilation of the DiT is disabled."
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
`${profile.sampling.width} × ${profile.sampling.height} · ${profile.sampling.num_frames} frames · ${profile.sampling.fps} fps. The server supplies these defaults; the client sends the model and prompt.`,
|
||||
"Generation is serialized. Job metadata is held in memory and is lost when the server restarts.",
|
||||
] : recipe.limitations || [];
|
||||
] : recipe.limitations || []), ...knobCaveats];
|
||||
notes.replaceChildren();
|
||||
notes.hidden = limitations.length === 0;
|
||||
if (limitations.length) {
|
||||
@@ -446,7 +623,16 @@
|
||||
const nextQuery = new URLSearchParams(window.location.search);
|
||||
nextQuery.set("recipe", recipe.id);
|
||||
nextQuery.set("runtime", runtime.id);
|
||||
nextQuery.delete("gpus");
|
||||
const deviceSiblings = recipesForRuntime(runtime.id, groups.get(selectedGroupId) || []);
|
||||
if (deviceSiblings.length > 1 && recipe.hardware?.gpu_count != null) {
|
||||
nextQuery.set("gpus", String(recipe.hardware.gpu_count));
|
||||
} else {
|
||||
nextQuery.delete("gpus");
|
||||
}
|
||||
knobDefs.forEach((knob, key) => {
|
||||
if (knobs.some((activeKnob) => activeKnob.key === key)) nextQuery.set(key, String(knobValues[key]));
|
||||
else nextQuery.delete(key);
|
||||
});
|
||||
if (usage) {
|
||||
nextQuery.set("use", useServer ? "server" : "python");
|
||||
if (useServer && clientDetails?.open) nextQuery.set("client", selectedClient);
|
||||
@@ -466,6 +652,17 @@
|
||||
render({ groupChanged: true, historyMode: "push" });
|
||||
});
|
||||
hardwareOptions.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-runtime-id]");
|
||||
if (!option) return;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const currentCount = byId.get(selectedRecipeId)?.hardware?.gpu_count;
|
||||
const nextRecipe = pickRecipeForRuntime(option.dataset.runtimeId, currentCount, groupRecipes);
|
||||
if (!nextRecipe) return;
|
||||
selectedRecipeId = nextRecipe.id;
|
||||
selectedGroupId = groupIdFor(nextRecipe);
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
deviceOptions?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-recipe-id]");
|
||||
if (!option) return;
|
||||
selectedRecipeId = option.dataset.recipeId;
|
||||
@@ -484,6 +681,14 @@
|
||||
selectedClient = option.dataset.cookbookClient;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
knobsContainer?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-knob-key]");
|
||||
if (!option) return;
|
||||
const knob = knobDefs.get(option.dataset.knobKey);
|
||||
knobValues[option.dataset.knobKey] = typeof knob.default === "number"
|
||||
? Number(option.dataset.knobValue) : option.dataset.knobValue;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
|
||||
render();
|
||||
|
||||
@@ -496,6 +701,13 @@
|
||||
usagePreference = workflow(nextQuery.get("use"));
|
||||
selectedClient = ["python", "javascript", "curl"].includes(nextQuery.get("client")) ? nextQuery.get("client") : "curl";
|
||||
if (clientDetails) clientDetails.open = nextQuery.has("client");
|
||||
knobDefs.forEach((knob, key) => {
|
||||
const fromQuery = nextQuery.get(key);
|
||||
const validValues = knobOptions(knob).map((option) => String(option.value));
|
||||
if (fromQuery !== null && validValues.includes(fromQuery)) {
|
||||
knobValues[key] = typeof knob.default === "number" ? Number(fromQuery) : fromQuery;
|
||||
}
|
||||
});
|
||||
render({ historyMode: "none" });
|
||||
};
|
||||
bindFamilyPopstate();
|
||||
|
||||
+58
-14
@@ -146,8 +146,7 @@ img {
|
||||
}
|
||||
|
||||
.cookbook-section,
|
||||
.cookbook-builder,
|
||||
.cookbook-roadmap {
|
||||
.cookbook-builder {
|
||||
scroll-margin-top: 4.5rem;
|
||||
}
|
||||
|
||||
@@ -180,8 +179,7 @@ img {
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-section__heading h2,
|
||||
.md-typeset .cookbook-builder__intro h2,
|
||||
.md-typeset .cookbook-roadmap h2 {
|
||||
.md-typeset .cookbook-builder__intro h2 {
|
||||
margin: 0;
|
||||
color: var(--md-default-fg-color);
|
||||
font-size: clamp(1.4rem, 2.6vw, 1.85rem);
|
||||
@@ -493,8 +491,7 @@ img {
|
||||
max-width: 45rem;
|
||||
}
|
||||
|
||||
.cookbook-builder__intro > p,
|
||||
.cookbook-roadmap > p {
|
||||
.cookbook-builder__intro > p {
|
||||
color: var(--md-default-fg-color--light);
|
||||
font-size: 0.75rem;
|
||||
line-height: 1.6;
|
||||
@@ -567,7 +564,7 @@ img {
|
||||
}
|
||||
|
||||
.cookbook-option-grid--hardware {
|
||||
grid-template-columns: repeat(2, minmax(0, 1fr));
|
||||
grid-template-columns: repeat(auto-fit, minmax(6.4rem, 1fr));
|
||||
}
|
||||
|
||||
.cookbook-option-grid button {
|
||||
@@ -769,12 +766,6 @@ img {
|
||||
font-weight: 700;
|
||||
}
|
||||
|
||||
.cookbook-roadmap {
|
||||
padding: 2.5rem 0;
|
||||
margin: 3.5rem 0 1rem;
|
||||
border-top: 1px solid var(--cookbook-border);
|
||||
}
|
||||
|
||||
/* Lifecycle roadmap on family pages: inference is live; later stages are
|
||||
explicitly labelled planned and must not look runnable. */
|
||||
.cookbook-lifecycle {
|
||||
@@ -1058,6 +1049,55 @@ img {
|
||||
overflow-wrap: anywhere;
|
||||
}
|
||||
|
||||
.cookbook-jumpnav {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 1.1rem;
|
||||
max-width: 48rem;
|
||||
margin: 0 auto 1.75rem;
|
||||
padding: 0.55rem 0;
|
||||
border-top: 1px solid var(--cookbook-border);
|
||||
border-bottom: 1px solid var(--cookbook-border);
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-jumpnav a {
|
||||
font-family: var(--md-code-font-family);
|
||||
font-size: 0.7rem;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--md-default-fg-color--light);
|
||||
text-decoration: none;
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-jumpnav a:hover {
|
||||
color: var(--cookbook-accent-strong);
|
||||
}
|
||||
|
||||
.cookbook-family-page details.cookbook-collapsible {
|
||||
max-width: 48rem;
|
||||
margin: 0 auto 0.75rem;
|
||||
border-color: var(--cookbook-border);
|
||||
box-shadow: none;
|
||||
}
|
||||
|
||||
.md-typeset details.cookbook-collapsible > summary {
|
||||
background: var(--cookbook-surface-raised);
|
||||
font-size: 0.85rem;
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-collapsible__body {
|
||||
font-size: 0.8rem;
|
||||
line-height: 1.65;
|
||||
color: var(--md-default-fg-color--light);
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-collapsible__body .cookbook-eyebrow {
|
||||
margin-top: 0.9rem;
|
||||
}
|
||||
|
||||
.md-typeset .cookbook-collapsible__body pre {
|
||||
margin: 0.5rem 0 1rem;
|
||||
}
|
||||
|
||||
.cookbook-family-page {
|
||||
max-width: 62rem;
|
||||
}
|
||||
@@ -1184,6 +1224,10 @@ img {
|
||||
margin-top: 0.55rem;
|
||||
}
|
||||
|
||||
[data-cookbook-knobs]:not(:empty) {
|
||||
margin-top: 0.55rem;
|
||||
}
|
||||
|
||||
.cookbook-selection-row__label {
|
||||
padding-inline-start: 0.15rem;
|
||||
}
|
||||
@@ -1207,7 +1251,7 @@ img {
|
||||
}
|
||||
|
||||
.cookbook-option-grid--hardware {
|
||||
grid-template-columns: repeat(2, minmax(0, 1fr));
|
||||
grid-template-columns: repeat(auto-fit, minmax(6.4rem, 1fr));
|
||||
}
|
||||
|
||||
.cookbook-option-grid button {
|
||||
|
||||
@@ -18,8 +18,8 @@ the CUDA `fastvideo-kernel` package:
|
||||
- **Dense-only checkpoints** (the default converter) drop the 50 gate
|
||||
matrices and keep fused SDPA. They remain valid for dense inference.
|
||||
- **VSA-capable checkpoints** retain those gates, quantize them on the same
|
||||
affine grid, and record `vsa.capable` in `mlx_h3_dit.json`. Runtime VSA is
|
||||
still off until you pass `--vsa`.
|
||||
affine grid, and record `vsa.capable` in `mlx_h3_dit.json`. Preview leaves
|
||||
runtime VSA off until you pass `--vsa`. `mlx_fasth3_8step.py` turns it on.
|
||||
- **Tile sizes** 64 `(4, 4, 4)` and 256 `(4, 8, 8)`. Prefix keys can be
|
||||
`exempt` or `compete`. `--vsa-dense-first-n-steps` and `--vsa-dense-layers`
|
||||
force dense SDPA on the selected steps or blocks.
|
||||
@@ -32,9 +32,12 @@ the CUDA `fastvideo-kernel` package:
|
||||
but does not yet match reference video. `--vsa-impl reference` is the same
|
||||
as `auto`.
|
||||
|
||||
See the [Apple Silicon guide](../../getting_started/installation/mps.md) for
|
||||
conversion and `mlx_fasth3.py` flags. Do not enable VSA on a dense-only
|
||||
checkpoint; reconvert with `--include-vsa` first.
|
||||
See the [MLX install guide](../../getting_started/installation/mlx.md)
|
||||
and the [MiniMax H3 cookbook](../../cookbook/minimax-h3.md) for conversion
|
||||
and `mlx_fasth3.py` / `mlx_fasth3_8step.py` flags. Do not enable
|
||||
VSA on a dense-only checkpoint; reconvert with `--include-vsa` first. V1
|
||||
VSA is opt-in. FastH3 V2 converts with `--include-vsa` and turns VSA on
|
||||
by default.
|
||||
|
||||
H3 uses fused MLX RMSNorm by default, including dense inference. This can
|
||||
change BF16 rounding relative to the older explicit normalization path.
|
||||
|
||||
@@ -30,7 +30,7 @@ FastVideo and the reference model first produce different numbers?"
|
||||
| General logging | `init_logger(__name__)` |
|
||||
| Per-stage timing | `FASTVIDEO_STAGE_LOGGING` |
|
||||
| Profiling kernel timings | `FASTVIDEO_TORCH_PROFILER_DIR` (see [Profiling](profiling.md)) |
|
||||
| Function-call tracing | `FASTVIDEO_TRACE_FUNCTION` (heavy) |
|
||||
| Function-call tracing | `fastvideo.logger.enable_trace_function_call()` (heavy) |
|
||||
|
||||
## Quickstart
|
||||
|
||||
|
||||
@@ -4,6 +4,11 @@ This is the canonical reference for FastVideo's CI/CD system. Contributor-facing
|
||||
PR steps live in [Pull Requests](pull_requests.md), and test-authoring guidance
|
||||
lives in [Testing](testing.md).
|
||||
|
||||
The existing Slurm route below remains the default. Operators can also install
|
||||
the [selectable GPU dispatcher](gpu_ci_backends.md) to run the same lane scripts
|
||||
on Modal or Kubernetes in the `vllm` namespace. That opt-in path enforces two
|
||||
active PRs, four GPUs per PR, and eight total, with separate backend statuses.
|
||||
|
||||
## Overview
|
||||
|
||||
FastVideo splits validation across GitHub Actions, Buildkite, Slinky Slurm,
|
||||
@@ -141,6 +146,24 @@ it), while a GitHub outage or a >25 min wait lets it run anyway (fail open).
|
||||
The complete static graph remains available through `/test full`; path
|
||||
selection never deletes or dynamically invents a Buildkite step.
|
||||
|
||||
Integration steps depend on `golden-gate`. A failed selected golden prevents
|
||||
the expensive downstream jobs from starting; `/test full` still selects all
|
||||
twenty lanes. A condition-skipped golden satisfies the dependency, so direct
|
||||
lane reruns, scheduled SSIM, and training-only merge plans keep their existing
|
||||
meaning. This follows Buildkite's
|
||||
[conditional dependency rules](https://buildkite.com/docs/pipelines/configure/depends-on).
|
||||
No dependency allows failures. The trusted uploader accepts either the old
|
||||
complete graph or the complete golden-first graph during rollout, and rejects
|
||||
partial or arbitrary dependency changes.
|
||||
|
||||
The six automatic Fastcheck lanes are unchanged. Within VAE and transformer
|
||||
lanes, small Wan goldens run before independent component parity. Wan paths
|
||||
select matching VAE, dense/trajectory, or causal-cache goldens; shared Wan
|
||||
config and pipeline wiring select all four. Shared runtime changes continue
|
||||
to select broader coverage. The existing block references retain their exact
|
||||
environment identity; new tensor gates distinguish the effective FA2/FA4
|
||||
switch and VAE gates do not depend on an unused attention backend.
|
||||
|
||||
| Lane | Public `TEST_TYPE` | GPUs | Typical merge trigger |
|
||||
|---|---|---:|---|
|
||||
| Encoder | `encoder` | 1 | Universal Fastcheck |
|
||||
@@ -417,8 +440,11 @@ before later jobs consume the updated image pin.
|
||||
The same workflow publishes a single-architecture ARM64, CUDA 13, SM100 image
|
||||
for the self-hosted CI runner under the
|
||||
`py3.12-cuda13.0.0-sm100-{latest,sha-*}` tags. It carries the matching prebuilt
|
||||
kernel wheel so runner jobs can validate and install the exact source and ABI
|
||||
match instead of recompiling it in every lane.
|
||||
kernel wheel compiled with `TORCH_CUDA_ARCH_LIST=10.0a` to include the GB200
|
||||
VSA CUDA extensions. Runtime kernel detection uses the same target, so runner
|
||||
jobs can validate and install the exact source and ABI match instead of
|
||||
recompiling it in every lane. Older artifacts built for `10.0` have a different
|
||||
cache key and trigger a local rebuild when the worker detects `10.0a`.
|
||||
|
||||
The optional Dreamverse matrix builds backend and UI images for CUDA 12.6 and
|
||||
CUDA 13 on `amd64`. Dreamverse remains `amd64`-only because its FA4 dependency
|
||||
|
||||
@@ -43,6 +43,8 @@ FastVideo maps a Diffusers-style repo into a pipeline like:
|
||||
- `fastvideo/models/*`: model implementations (DiT, VAE, encoders, upsamplers).
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/models/wan/`: Wan's dense transformer, VAE, and component configs
|
||||
live together. The old Wan modules remain compatibility re-exports.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipeline logic built from stages.
|
||||
@@ -209,7 +211,7 @@ class OfficialWanTransformer(torch.nn.Module):
|
||||
def forward(self, x):
|
||||
return self.patch_embedding(x)
|
||||
|
||||
# FastVideo model (simplified) in fastvideo/models/dits/wanvideo.py
|
||||
# FastVideo model (simplified) in fastvideo/models/wan/transformer.py
|
||||
class PatchEmbed(torch.nn.Module):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
@@ -227,7 +229,7 @@ class WanTransformer3DModel(torch.nn.Module):
|
||||
return self.patch_embedding(x)
|
||||
|
||||
# Mapping defined in a config (simplified; see the real mapping in
|
||||
# fastvideo/configs/models/dits/wanvideo.py)
|
||||
# fastvideo/models/wan/config.py)
|
||||
param_names_mapping = {
|
||||
r"^patch_embedding\.(.*)$": r"patch_embedding.proj.\1",
|
||||
r"^blocks\.(\d+)\.attn1\.to_q\.(.*)$": r"blocks.\1.to_q.\2",
|
||||
@@ -275,7 +277,7 @@ Mapping steps:
|
||||
- Instantiate the FastVideo DiT (`WanTransformer3DModel`) and compare
|
||||
its `state_dict().keys()` to the official keys.
|
||||
- Update `param_names_mapping` in
|
||||
fastvideo/configs/models/dits/wanvideo.py to resolve missing/unexpected keys.
|
||||
fastvideo/models/wan/config.py to resolve missing/unexpected keys.
|
||||
- Use `load_state_dict(strict=False)` during iteration to surface mismatches.
|
||||
```
|
||||
|
||||
@@ -466,19 +468,24 @@ The Wan2.1 T2V 1.3B Diffusers pipeline is a good “standard” example for
|
||||
FastVideo integration.
|
||||
|
||||
1. Verify model config + mapping.
|
||||
- DiT mapping: `fastvideo/configs/models/dits/wanvideo.py`
|
||||
- VAE: `fastvideo/models/vaes/wanvae.py`
|
||||
- DiT: `fastvideo/models/wan/transformer.py`
|
||||
- DiT mapping: `fastvideo/models/wan/config.py`
|
||||
- VAE: `fastvideo/models/wan/vae.py`
|
||||
- VAE config: `fastvideo/models/wan/vae_config.py`
|
||||
- Text encoder: `fastvideo/models/encoders/t5.py`
|
||||
|
||||
2. Parity test the core components.
|
||||
- Start with `bash scripts/validate_wan.sh all`: contracts and tiny goldens.
|
||||
- Example tests: `fastvideo/tests/transformers/test_wanvideo.py`,
|
||||
`fastvideo/tests/vaes/test_wan_vae.py`,
|
||||
`fastvideo/tests/encoders/test_t5_encoder.py`
|
||||
|
||||
3. Pipeline wiring.
|
||||
- Pipeline: `fastvideo/pipelines/basic/wan/wan_pipeline.py`
|
||||
- Pipeline config: `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
- Denoising and first-frame preparation: `fastvideo/pipelines/basic/wan/stages/`
|
||||
- Variant definitions: `fastvideo/models/wan/definition.py`
|
||||
- Pipeline config: `fastvideo/models/wan/pipeline_config.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/presets.py`
|
||||
|
||||
4. Minimal example.
|
||||
- Script: `examples/inference/basic/basic.py`
|
||||
|
||||
@@ -0,0 +1,242 @@
|
||||
# Environment Variables
|
||||
|
||||
FastVideo reads environment variables for expert switches, debugging, profiling, and the settings that launchers such
|
||||
as `torchrun` provide. This page is the policy for those variables. The contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit CI lane, and the coding-agent skill
|
||||
`.agents/skills/env-var-conventions/SKILL.md` points here. When the policy changes, update this page and the contract
|
||||
test in the same pull request.
|
||||
|
||||
## Rules
|
||||
|
||||
1. **Register every FastVideo variable in `fastvideo/envs.py`.** Each entry declares a type, a default, a category,
|
||||
and a description. Variables that other tools own (CUDA, NCCL, PyTorch, launchers) are not registered; code reads
|
||||
them directly with `os.environ.get("NAME")`, and the name must be in the external-variable allowlist
|
||||
(`EXTERNAL_ALLOWLIST` in the contract test). When FastVideo sets such a variable for the other tool, it calls
|
||||
`envs.set_external`, `envs.setdefault_external`, or `envs.unset_external`, and the name must be in
|
||||
`EXTERNAL_WRITE_ALLOWLIST`.
|
||||
2. **Read with `envs.NAME.get()`, write with `envs.NAME.set()`, and change a value in tests with
|
||||
`envs.NAME.override()`.** Each type has one parsing rule. A value that the rule rejects raises
|
||||
`fastvideo.envs.EnvVarError` instead of falling back to the default.
|
||||
3. **Name FastVideo variables with the `FASTVIDEO_` prefix.** The second word states the purpose where one applies:
|
||||
`ENABLE_`, `DISABLE_`, `USE_`, `FORCE_`, `DEBUG_`, `TEST_`.
|
||||
4. **Keep a renamed variable as a deprecated alias until the next minor release.** Setting the old name logs a
|
||||
warning. Delete a variable that no code reads, and list it in `DEPRECATED_VARIABLES` so that setting it logs a
|
||||
warning.
|
||||
5. **Give each setting one source: an argument or an environment variable.** Settings that users change per
|
||||
deployment are arguments (CLI or YAML). Expert switches, emergency off switches, and debugging and test switches
|
||||
are environment variables.
|
||||
6. **Read variables inside functions.** `envs.NAME.get()` runs when the function runs, so a changed value takes
|
||||
effect without re-importing a module. Module level, class bodies, decorators, and default argument values run at
|
||||
import time.
|
||||
7. **Do not write the environment to pass values between parts of FastVideo.** Pass an argument instead. Tests use
|
||||
`envs.NAME.override()`.
|
||||
|
||||
## Field types
|
||||
|
||||
| Class | Value type | Parsing rule |
|
||||
| ------------ | ---------------- | ----------------------------------------------------------------------------------- |
|
||||
| `EnvBool` | `bool` | `1`, `true`, `yes`, `on` are true; `0`, `false`, `no`, `off`, and `""` are false. |
|
||||
| | | Case-insensitive; surrounding whitespace is ignored. |
|
||||
| `EnvInt` | `int` | `int(value)` |
|
||||
| `EnvFloat` | `float` | `float(value)` |
|
||||
| `EnvStr` | `str` or `None` | The raw string. A `None` default means that the variable has no default. |
|
||||
| `EnvPath` | `str` or `None` | The raw string with a leading `~` expanded. |
|
||||
| `EnvChoice` | `str` | Stripped and lower-cased, then checked against the declared `choices`. |
|
||||
|
||||
A default can be a zero-argument function; `get()` calls it on each read while the variable is unset. The path roots
|
||||
use this to follow `XDG_CONFIG_HOME` and `XDG_CACHE_HOME`.
|
||||
|
||||
Using a field without a method, as in `if envs.FASTVIDEO_FA4:`, raises `TypeError`.
|
||||
|
||||
## Add a variable
|
||||
|
||||
1. Declare the variable in the matching section of `fastvideo/envs.py`:
|
||||
|
||||
```python
|
||||
FASTVIDEO_DEBUG_MY_STAGE = EnvBool(False, category="debug", doc="Log the inputs of MyStage.")
|
||||
```
|
||||
|
||||
The category is one of the values in `envs.CATEGORIES`.
|
||||
|
||||
2. Read the variable inside a function:
|
||||
|
||||
```python
|
||||
import fastvideo.envs as envs
|
||||
|
||||
def forward(self, batch):
|
||||
if envs.FASTVIDEO_DEBUG_MY_STAGE.get():
|
||||
logger.info("MyStage inputs: %s", batch.keys())
|
||||
```
|
||||
|
||||
3. Regenerate the table at the end of this page:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/contract/test_env_policy.py
|
||||
```
|
||||
|
||||
4. Run the contract test:
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/contract/test_env_policy.py
|
||||
```
|
||||
|
||||
In a test, change the value with `override`, which restores the previous value on exit:
|
||||
|
||||
```python
|
||||
with envs.FASTVIDEO_DEBUG_MY_STAGE.override(True):
|
||||
run_stage()
|
||||
```
|
||||
|
||||
## Rename or remove a variable
|
||||
|
||||
To rename a variable, declare it under the new name and list the old name in `deprecated_names`:
|
||||
|
||||
```python
|
||||
FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS = EnvBool(True,
|
||||
category="sampling",
|
||||
doc="...",
|
||||
deprecated_names=("LTX2_USE_DISTILLED_SIGMAS", ))
|
||||
```
|
||||
|
||||
`get()` reads an old name only when the new name is unset, and logs a warning once. Update the uses of the old name
|
||||
in `examples/`, `scripts/`, `docs/`, `apps/`, and the tests in the same pull request. Delete the old name in the next
|
||||
minor release.
|
||||
|
||||
To remove a variable that no code reads, delete its entry and add the name to `DEPRECATED_VARIABLES` in
|
||||
`fastvideo/envs.py` with a reason. `FastVideoArgs` calls `envs.warn_deprecated_variables()`, which logs a warning for
|
||||
each listed variable that is set. Delete the entry in the next minor release.
|
||||
|
||||
## What the contract test checks
|
||||
|
||||
The test parses every Python file under `fastvideo/`, including `fastvideo/tests/`, with Python's `ast` module. It
|
||||
skips `fastvideo/third_party/`, which is copied from upstream projects, and the registry `fastvideo/envs.py`. It does
|
||||
not check `apps/`, `examples/`, `scripts/`, `fastvideo-kernel/`, or `docs/`.
|
||||
|
||||
It reports each violation as `<path>: <kind> <name>`:
|
||||
|
||||
| Kind | Code that triggers it | Fix |
|
||||
| ------------------ | ------------------------------------------------------------ | -------------------------------------------- |
|
||||
| `read` | `os.getenv`, `os.environ.get`, `os.environ[...]`, or | Register the variable and call |
|
||||
| | `"NAME" in os.environ` with a name outside the allowlist, or | `envs.NAME.get()`. For a variable that |
|
||||
| | with a name built at runtime (`<dynamic>`) | another tool owns, add it to |
|
||||
| | | `EXTERNAL_ALLOWLIST` with a reason. |
|
||||
| `write` | `os.environ[...] = ...`, `setdefault`, `pop`, `del`, | Pass an argument instead. In tests, use |
|
||||
| | `os.putenv`, `os.unsetenv`, `monkeypatch.setenv`/`delenv`, | `envs.NAME.override()`. For a variable that |
|
||||
| | or an `envs.*_external` call with a name outside | another tool reads, call an |
|
||||
| | `EXTERNAL_WRITE_ALLOWLIST` | `envs.*_external` helper and add the name to |
|
||||
| | | `EXTERNAL_WRITE_ALLOWLIST` with a reason. |
|
||||
| `whole-environ` | `os.environ.copy()`, `dict(os.environ)`, iteration, | Read the specific variables that the code |
|
||||
| | `mock.patch.dict(os.environ, ...)`, `os.environ.update` | needs. |
|
||||
| `bare-field` | A registry field used without calling one of its methods, | Call `envs.NAME.get()`. |
|
||||
| | as in `envs.NAME == "auto"` or `getter = envs.NAME.get` | |
|
||||
| `import-time-read` | `envs.NAME.get()` outside a function | Move the read into the function that uses |
|
||||
| | | the value. |
|
||||
| `prefix` | A registry entry without the `FASTVIDEO_` prefix | Rename the variable and keep the old name in |
|
||||
| | | `deprecated_names`. |
|
||||
| `unread` | A registry entry that no code reads with `get()` or | Delete the variable and add it to |
|
||||
| | `is_set()` | `DEPRECATED_VARIABLES`. |
|
||||
|
||||
The test recognizes `os` imported under another name, `from os import environ, getenv`, and a name held in a
|
||||
module-level string constant. Code that reaches the environment through `importlib` or `getattr(os, "environ")` is
|
||||
left to code review.
|
||||
|
||||
The test also checks that every registry entry has a category from `envs.CATEGORIES` and a description, and that the
|
||||
table at the end of this page matches the registry.
|
||||
|
||||
**Known violations.** `KNOWN_VIOLATIONS` in the contract test lists the violations that existed when the policy was
|
||||
introduced. The list only shrinks. A violation that is not in the list fails the test. A listed violation that no
|
||||
longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
|
||||
## Registered variables
|
||||
|
||||
<!-- BEGIN GENERATED ENV TABLE: python fastvideo/tests/contract/test_env_policy.py -->
|
||||
| Variable | Type | Default | Category | Description |
|
||||
| ---------------------------------------------------------- | ------------------------------ | ----------------------------------------------------- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `FASTVIDEO_CONFIG_ROOT` | path | computed | path | Root directory for FastVideo configuration files, at runtime and at installation. Defaults to ~/.config/fastvideo, or $XDG_CONFIG_HOME/fastvideo when XDG_CONFIG_HOME is set. |
|
||||
| `FASTVIDEO_CACHE_ROOT` | path | computed | path | Root directory for FastVideo cache files. Defaults to ~/.cache/fastvideo, or $XDG_CACHE_HOME/fastvideo when XDG_CACHE_HOME is set. |
|
||||
| `FASTVIDEO_REASON1_WEIGHTS_PATH` | str | unset | path | Local path or Hugging Face id of Reason1 weights to load instead of the checkpoint's own. |
|
||||
| `FASTVIDEO_HOST_IP` | str | `""` | distributed | IP address of this node when the node has several network interfaces. Set it on each node for multi-node inference. |
|
||||
| `FASTVIDEO_LOOPBACK_IP` | str | `""` | distributed | Loopback IP address to use instead of the detected one. |
|
||||
| `FASTVIDEO_RAY_PER_WORKER_GPUS` | float | `1.0` | distributed | GPUs per Ray worker. A fraction lets Ray schedule several actors on one GPU, so other actors can share the GPUs with FastVideo. |
|
||||
| `FASTVIDEO_NCCL_SO_PATH` | str | unset | distributed | Path to the NCCL library file. Needed because the nccl>=2.19 that PyTorch ships has a bug (https://github.com/NVIDIA/nccl/issues/1234). |
|
||||
| `FASTVIDEO_HCCL_SO_PATH` | str | unset | distributed | Path to the HCCL library file on Ascend NPUs. Deprecated names: `HCCL_SO_PATH`. |
|
||||
| `FASTVIDEO_WORKER_MULTIPROC_METHOD` | one of spawn, fork, forkserver | `spawn` | distributed | Multiprocessing start method for worker processes. |
|
||||
| `FASTVIDEO_ULYSSES_A2A` | one of off, auto | `off` | distributed | Sequence-parallel all-to-all backend. off uses the NCCL path in DistributedAutograd.AllToAll4D. auto uses the fused NVLink kernel when the group is a load-store accessible mesh of 2, 4, 6, or 8 ranks in eager execution, and the NCCL path otherwise. |
|
||||
| `FASTVIDEO_CONFIGURE_LOGGING` | bool | `1` | logging | Configure logging at import. When true, FastVideo uses its default logging configuration or the file in FASTVIDEO_LOGGING_CONFIG_PATH. |
|
||||
| `FASTVIDEO_LOGGING_CONFIG_PATH` | str | unset | logging | Path to a JSON logging configuration file. |
|
||||
| `FASTVIDEO_LOGGING_LEVEL` | str | `INFO` | logging | Default logging level. |
|
||||
| `FASTVIDEO_LOGGING_PREFIX` | str | `""` | logging | Prefix prepended to every log message. |
|
||||
| `FASTVIDEO_STAGE_LOGGING` | bool | `0` | logging | Log the time that each pipeline stage takes. |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | str | unset | attention | Attention backend, as an AttentionBackendEnum name such as TORCH_SDPA, FLASH_ATTN, VIDEO_SPARSE_ATTN, SAGE_ATTN, or SAGE_ATTN_THREE. FastVideoArgs uses it when FastVideoArgs.attention_backend is unset. |
|
||||
| `FASTVIDEO_FA4` | bool | `0` | attention | The FLASH_ATTN backend uses FlashAttention-4 (flash_attn.cute) instead of FA3 or FA2. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FA4_PACKED_VARLEN` | bool | `0` | attention | MiniMax-H3 dense DiT self-attention uses the FlashAttention-4 packed-varlen entry point. This changes the floating-point reduction order, so it is an inference-only opt-in. |
|
||||
| `FASTVIDEO_VSA_SM100A` | bool | `0` | attention | VIDEO_SPARSE_ATTN_H3 sends no-grad tile-64 forwards to the data-center Blackwell (sm_100a) kernel. fastvideo-kernel reads the same variable with the same rule. |
|
||||
| `FASTVIDEO_NVFP4_FA4` | bool | `0` | attention | FlashAttention-4 quantizes Q and K to NVFP4. An explicit nvfp4_fa4 attention implementation argument takes precedence. |
|
||||
| `FASTVIDEO_DISABLE_ATTENTION_COMPILE` | bool | `1` | attention | Keep attention forward out of torch.compile graphs (torch.compiler.disable). Set it to 0 to let attention constructed under that setting be traced. Setting it explicitly to true also blocks regional compile. |
|
||||
| `FASTVIDEO_MLX_WINDOW` | int | `0` | attention | MLX FastWan windowed attention size in tokens. 0 uses full attention. |
|
||||
| `FASTVIDEO_MLX_WINDOW_SINK` | int | `0` | attention | Number of sink tokens that MLX windowed attention always attends to. |
|
||||
| `FASTVIDEO_INFERENCE_TORCH_COMPILE` | bool | `0` | performance | Compile each DiT transformer block with fullgraph torch.compile at inference. Same as FastVideoArgs.inference_torch_compile=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE` | bool | `0` | performance | MiniMax-H3 VAE decode splits its temporal chunks across the sequence-parallel ranks instead of running serially on the output rank. Same as FastVideoArgs.vae_parallel_decode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_ENCODE` | bool | `0` | performance | MiniMax-H3 reference-video VAE encode splits its temporal chunks across the sequence-parallel ranks. Same as FastVideoArgs.vae_parallel_encode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE_STRATEGY` | str | unset | performance | Collective that moves chunks in parallel VAE decode: gather (used when unset) or all_gather. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FUSIONS` | str | `""` | performance | MiniMax-H3 inference-only Triton fusions: all, 1, or a comma-separated subset of modulate,qknorm_rope,swiglu. Empty, 0, or none keeps the eager implementation. |
|
||||
| `FASTVIDEO_FSDP2_AUTOWRAP` | bool | `0` | performance | FSDP2 shards modules by parameter count instead of the model's shard conditions. Not supported by self-forcing distillation. |
|
||||
| `FASTVIDEO_FSDP2_MIN_PARAMS` | int | `10000000` | performance | Minimum parameter count of a module that FASTVIDEO_FSDP2_AUTOWRAP shards. |
|
||||
| `FASTVIDEO_MLX_COMPILE` | bool | `0` | performance | Compile the MLX DiT forward with mx.compile. |
|
||||
| `FASTVIDEO_MLX_FAST_NORM` | bool | `0` | performance | Use MLX fast normalization kernels. |
|
||||
| `FASTVIDEO_MLX_DQ_GEMM` | str | `1` | performance | MLX dequantized GEMM for affine-quantized weights: 0 turns it off, 1 uses the measured minimum row count, and an integer sets the minimum row count. |
|
||||
| `FASTVIDEO_LTX2_VAE_CHANNELS_LAST_3D` | bool | `1` | performance | LTX-2 VAE uses the channels_last_3d memory format. |
|
||||
| `FASTVIDEO_LTX2_DISABLE_AUDIO_AUTOCAST` | bool | `1` | performance | LTX-2 audio decoding runs without CUDA autocast. Deprecated names: `LTX2_DISABLE_AUDIO_AUTOCAST`. |
|
||||
| `FASTVIDEO_FLUX2_DISABLE_BF16_REDUCED_PRECISION_REDUCTION` | bool | `0` | performance | Flux denoising disables reduced-precision reductions in bf16 matmuls, which tightens accumulation for the 4-step Klein model. |
|
||||
| `FASTVIDEO_FFMPEG_BIN` | str | `ffmpeg` | output | ffmpeg executable used to save video with audio. |
|
||||
| `FASTVIDEO_VIDEO_CODEC` | str | `libx264` | output | ffmpeg video codec for saved videos. |
|
||||
| `FASTVIDEO_NVENC_PRESET` | str | `p1` | output | NVENC preset when the codec is an \*_nvenc codec. |
|
||||
| `FASTVIDEO_NVENC_TUNE` | str | `ull` | output | NVENC tune option. |
|
||||
| `FASTVIDEO_NVENC_RC` | str | `constqp` | output | NVENC rate-control mode. |
|
||||
| `FASTVIDEO_NVENC_QP` | str | `28` | output | NVENC quantization parameter. |
|
||||
| `FASTVIDEO_NVENC_BF` | str | `0` | output | NVENC number of B-frames. |
|
||||
| `FASTVIDEO_X264_PRESET` | str | `ultrafast` | output | x264 preset for non-NVENC codecs. |
|
||||
| `FASTVIDEO_OUTPUT_PIX_FMT` | str | `yuv420p` | output | ffmpeg pixel format for saved videos. |
|
||||
| `FASTVIDEO_NVTX_PROFILE` | bool | `0` | profiling | Emit NVTX ranges for external profilers such as Nsight Systems. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_DIR` | path | unset | profiling | Enables the torch profiler and sets the directory for its traces. Must be an absolute path. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_RECORD_SHAPES` | bool | `0` | profiling | Torch profiler records shapes. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_PROFILE_MEMORY` | bool | `0` | profiling | Torch profiler profiles memory. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_STACK` | bool | `0` | profiling | Torch profiler captures stacks. Costs about 1.5x runtime and 1.4x trace size. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_FLOPS` | bool | `0` | profiling | Torch profiler profiles FLOPs. |
|
||||
| `FASTVIDEO_TORCH_PROFILE_REGIONS` | str | `""` | profiling | Comma-separated profiler regions to record. The torch profiler requires at least one region. |
|
||||
| `FASTVIDEO_TRACE_ACTIVATIONS` | bool | `0` | debug | Enable activation trace hooks. |
|
||||
| `FASTVIDEO_TRACE_LAYERS` | str | `""` | debug | Regex filter for traced module names. Empty means all. |
|
||||
| `FASTVIDEO_TRACE_STATS` | str | `abs_mean,sum` | debug | Comma-separated activation statistics dumped for each output tensor. |
|
||||
| `FASTVIDEO_TRACE_OUTPUT` | str | `/tmp/fv_trace_<pid>.jsonl` | debug | JSONL path for activation traces. The literal <pid> is replaced at runtime. |
|
||||
| `FASTVIDEO_TRACE_STEPS` | str | `""` | debug | Comma-separated denoising step indices. Empty means all. |
|
||||
| `FASTVIDEO_H3_VSA_PROBE` | str | unset | debug | Output directory for the VSA-H3 attention-mass probe, which writes one .pt file per step, layer, and rank. Keeps the model out of regional compile. |
|
||||
| `FASTVIDEO_LTX2_GEMMA_LOG` | str | `""` | debug | Log file for LTX-2 Gemma text-encoder hidden states, used by parity tests. Deprecated names: `LTX2_FASTVIDEO_GEMMA_LOG`. |
|
||||
| `FASTVIDEO_COSMOS25_LOG_KNOBS` | bool | `0` | debug | Log the Cosmos 2.5 latent-preparation conditioning inputs. |
|
||||
| `FASTVIDEO_CFG_GATE_STEP` | float | `1.0` | sampling | CFG gating fraction in [0, 1]. Steps before len(timesteps) \* X run the conditional and unconditional forwards; later steps reuse the cached difference. 1.0 disables gating. |
|
||||
| `FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS` | bool | `1` | sampling | LTX-2 uses the distilled sigma schedule when FastVideoArgs.ltx2_use_distilled_sigmas is also true. Deprecated names: `LTX2_USE_DISTILLED_SIGMAS`. |
|
||||
| `FASTVIDEO_EVAL_CACHE` | path | computed | eval | Cache directory for evaluation models and datasets. Defaults to $FASTVIDEO_CACHE_ROOT/eval. |
|
||||
| `FASTVIDEO_PHYSICS_IQ_BUCKET_URL` | str | `https://storage.googleapis.com/physics-iq-benchmark` | eval | Base URL of the Physics-IQ benchmark bucket. |
|
||||
| `FASTVIDEO_VBENCH_FULL_INFO_JSON` | str | unset | eval | Path to VBench_full_info.json, used instead of the vendored copy. Deprecated names: `VBENCH_FULL_INFO_JSON`. |
|
||||
| `FASTVIDEO_FVD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the FVD metric. |
|
||||
| `FASTVIDEO_FAD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the audio Frechet distance metric. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_DATA_DIR` | str | `data/cats` | test | Raw data directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_CAPTION_JSON` | str | `videos2caption_1_sample.json` | test | Caption file, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_CAPTION_JSON`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_VIDEO_SUBDIR` | str | `video` | test | Video subdirectory, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_VIDEO_SUBDIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_OUTPUT_DIR` | str | `data/ltx2_overfit_preprocessed` | test | Output directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_OUTPUT_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_MODEL` | str | `FastVideo/LTX2-Distilled-Diffusers` | test | Model repository whose encoders preprocess_ltx2_overfit.py uses. Deprecated names: `LTX2_OVERFIT_MODEL`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_NUM_COPIES` | int | `4` | test | Number of copies of the overfit sample in the parquet file. Deprecated names: `LTX2_OVERFIT_NUM_COPIES`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_DATA_DIR` | str | `data/kandinsky5_overfit` | test | Raw data directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_OUTPUT_DIR` | str | `data/kandinsky5_overfit_preprocessed` | test | Output directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_OUTPUT_DIR`. |
|
||||
|
||||
Variables that FastVideo no longer reads; setting one logs a warning:
|
||||
|
||||
| Deprecated variable | Reason |
|
||||
| ----------------------------------------- | ---------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
<!-- END GENERATED ENV TABLE -->
|
||||
@@ -0,0 +1,252 @@
|
||||
# Selectable GPU CI Backends
|
||||
|
||||
GPU CI can use the existing Slurm dispatcher, Modal, or GB200 Kubernetes Jobs
|
||||
in the `vllm` namespace. GitHub and Buildkite remain the trigger and reporting
|
||||
systems. `vllm` names the cluster namespace here; tests run FastVideo's existing
|
||||
lane scripts rather than a vLLM inference server.
|
||||
|
||||
The default remains Slurm. The existing `.buildkite/pipeline.yml`, private
|
||||
Slurm uploader, and dormant Modal launchers are preserved. Deploying the new
|
||||
trusted uploader is an operator step; merging these files does not change the
|
||||
live Buildkite pipelines or create cluster resources. See
|
||||
[CI/CD Architecture](ci_architecture.md) for the existing installation.
|
||||
|
||||
## Execution and Limits
|
||||
|
||||
```text
|
||||
GitHub PR / slash command / schedule
|
||||
-> trusted Buildkite bootstrap
|
||||
-> slurm: existing validated graph and private dispatcher
|
||||
-> modal or vllm: one trusted suite coordinator
|
||||
-> shared PR admission and GPU reservations
|
||||
-> isolated GPU worker per existing lane
|
||||
-> lane results, logs, and backend-specific suite status
|
||||
```
|
||||
|
||||
All new Modal and `vllm` coordinators share one admission database. It enforces:
|
||||
|
||||
- At most two distinct PRs with admitted GPU work.
|
||||
- At most four reserved GPUs per PR across its concurrent builds and lanes.
|
||||
- At most eight reserved GPUs across both backends combined.
|
||||
- Additional PRs wait without creating GPU workers.
|
||||
|
||||
A PR is identified by repository and PR number, so Fastcheck, merge builds,
|
||||
manual reruns, and separate attempts share its allowance. Its slot remains
|
||||
occupied between lanes until all admitted builds for that PR finish. A
|
||||
non-PR build, including a schedule on `main`, consumes its own slot and the
|
||||
same GPU budget. The preserved Slurm dispatcher has its existing independent
|
||||
limits; this new admission database does not control legacy Slurm jobs.
|
||||
|
||||
Lanes retain their existing one-, two-, or four-GPU requirements. Four-GPU
|
||||
SSIM and training lanes wait for that PR's other lanes to release capacity.
|
||||
Selected integration lanes wait for the golden gate and are skipped if it
|
||||
fails. CPU-only GitHub checks are unaffected. Pending admission is bounded by
|
||||
`queue_timeout_seconds`, initially six hours. The coordinator's Buildkite
|
||||
command timeout is eight hours, including admission and lane waits after the
|
||||
command starts; time waiting for a Buildkite agent is outside that timeout.
|
||||
|
||||
Reservations persist before a worker is created. Cancellation releases them
|
||||
only after worker termination is confirmed. A lost coordinator, uncertain
|
||||
create response, or unavailable backend retains its reservations until
|
||||
recovery confirms cleanup. There is no heartbeat expiry that silently frees
|
||||
GPUs while an old worker could still be running.
|
||||
|
||||
## Operator Installation
|
||||
|
||||
Use an operator-reviewed immutable checkout under
|
||||
`/opt/fastvideo-gpu-ci/source`. Install the supplied `scripts/gpu_ci/run` and
|
||||
`scripts/gpu_ci/upload` wrappers as `/opt/fastvideo-gpu-ci/run` and
|
||||
`/opt/fastvideo-gpu-ci/upload`. They invoke the reviewed Python entrypoint
|
||||
with isolated import mode. Install the supplied files instead of generating
|
||||
shell wrappers from build metadata:
|
||||
|
||||
```bash
|
||||
install -m 755 /opt/fastvideo-gpu-ci/source/scripts/gpu_ci/run /opt/fastvideo-gpu-ci/run
|
||||
install -m 755 /opt/fastvideo-gpu-ci/source/scripts/gpu_ci/upload /opt/fastvideo-gpu-ci/upload
|
||||
```
|
||||
|
||||
The controller needs Python 3.10 or newer, PyYAML,
|
||||
the Buildkite agent, and `kubectl`; Modal additionally needs the reviewed
|
||||
Modal SDK and a controller-side Modal credential.
|
||||
|
||||
Copy [the configuration example](../../scripts/gpu_ci/config.example.json)
|
||||
to `/etc/fastvideo-gpu-ci.json`. Keep the installation and configuration
|
||||
operator-owned and unwritable by workers. Configure the actual legacy
|
||||
`slurm_uploader` command, replace each enabled backend's image placeholder
|
||||
with a reviewed registry digest, and leave `default_backend` as `slurm`
|
||||
during canaries. The checked-in placeholders deliberately cannot start jobs.
|
||||
|
||||
The permanent dispatcher must reach the Kubernetes API independently of a
|
||||
developer laptop, SSH tunnel, or Tailscale session. The existing CPU development
|
||||
pod is useful for investigating access, but is not a production CI service.
|
||||
Use a dedicated controller identity and a separate `gpu-ci-dispatch` Buildkite
|
||||
queue. The trusted bootstrap also needs access to the existing Slurm uploader
|
||||
while that backend remains enabled.
|
||||
|
||||
Run all coordinators on **one host**, with `state_path` and `artifacts_dir` on
|
||||
its persistent local disk. The implementation uses SQLite transactions and
|
||||
local file locks. Do not place the database or locks on Lustre/NFS, run
|
||||
independent database copies, or scale controller replicas across nodes. Those
|
||||
configurations would invalidate the global limits. Multiple Buildkite agent
|
||||
processes on the same host can share the installation and ledger; enough
|
||||
agents are needed for concurrent suite coordinators and queued work.
|
||||
|
||||
Configure agent-owned hooks to skip repository checkout and reject commands
|
||||
other than the trusted uploader and coordinator. Disable repository hooks
|
||||
and plugins on this queue. Do not use a PR checkout to load `scripts/gpu_ci`,
|
||||
its configuration, or its worker entrypoint. Build metadata is validated by
|
||||
the dispatcher; arbitrary environment variables are not forwarded into GPU
|
||||
workers. Buildkite, Kubernetes, and Modal control-plane credentials stay on
|
||||
the dispatcher.
|
||||
|
||||
For Kubernetes, provision namespace-scoped permissions to create/get/delete
|
||||
Jobs and read Pods and their logs. Worker Pods disable service-account token
|
||||
mounting and request the exact GPU count on ARM64 GB200 nodes. The image must
|
||||
include the expected `/opt/venv` runtime, CUDA/SM100 support, and the reviewed
|
||||
FA4 dependencies. Use a separate AMD64 image digest for Modal. The Modal
|
||||
profile preserves H100 for encoder, custom-kernel, and VSA lanes, and L40S
|
||||
for the remaining lanes. Its reviewed image must support both SM89 and
|
||||
SM90a. Attention settings are selected per backend and lane to preserve the
|
||||
existing Modal and GB200 FA4 profiles; do not replace them with one global
|
||||
attention override. Validate the images against their respective hardware
|
||||
before enabling merge gates.
|
||||
|
||||
Optional Kubernetes configuration includes `context`, `hf_secret`,
|
||||
`hf_secret_key`, `cache_pvc`, `cache_subpath`, `artifacts_pvc`, and
|
||||
`artifacts_subpath`. Only provide a read-only Hugging Face credential when
|
||||
private/gated downloads require it. PR workers are untrusted and can access
|
||||
any credential supplied to them, so never provide a reference-publication
|
||||
or other write-capable token. Prefer a pre-populated read-only cache.
|
||||
|
||||
Use dedicated CI PVCs and relative CI subpaths. The personal
|
||||
`lustre-pvc-vllm` and `nfs-pvc-vllm` claims are rejected. The cache is mounted
|
||||
read-only; mutable references and locks stay inside the worker. The artifact
|
||||
init container prepares each Job's dedicated artifact subdirectory before
|
||||
the worker mounts it. An
|
||||
operator quota or admission policy scoped to CI can provide a second GPU
|
||||
ceiling; do not apply an eight-GPU quota to the shared `vllm` namespace if it
|
||||
would also cap other users' work.
|
||||
|
||||
## Wire Every Buildkite Entry Pipeline
|
||||
|
||||
Set each pipeline's operator-owned bootstrap command to
|
||||
`/opt/fastvideo-gpu-ci/upload`. Apply this to all three entry pipelines:
|
||||
|
||||
| Pipeline | Trigger and scope |
|
||||
|---|---|
|
||||
| `pr-fastcheck` | Automatic PR webhook; `TEST_SCOPE=fastcheck` or unset. |
|
||||
| `ci` | Existing API triggers for Fastcheck, full, merge, direct, and scheduled SSIM. Keep its incoming PR webhook disabled to avoid duplicate builds. |
|
||||
| `fastvideo-performance-lane` | Existing schedule; `TEST_SCOPE=direct`, `TEST_TYPE=performance`. |
|
||||
|
||||
The wrapper resolves `CI_GPU_BACKEND` from the build environment, falling
|
||||
back to `default_backend` in the trusted configuration. Allowed values are
|
||||
exactly `slurm`, `modal`, and `vllm`; unknown values fail. For Slurm, upload
|
||||
delegates to the unchanged private uploader. For Modal or `vllm`, it uploads
|
||||
one fixed `/opt/fastvideo-gpu-ci/run` command on the dedicated queue, with
|
||||
the selected backend and scope pinned in step environment.
|
||||
|
||||
Set `CI_GPU_BACKEND=vllm` in the Buildkite build environment for a rack canary,
|
||||
or `modal` for a Modal canary. The existing API build payload can carry the
|
||||
same string in its `env` object. Automatic PR webhook builds inherit the
|
||||
operator's configured default unless pipeline/build configuration explicitly
|
||||
overrides it. Do not put backend selection inside a PR-controlled command.
|
||||
|
||||
Keep the existing exact `BUILDKITE_COMMIT`, repository, PR identity, and
|
||||
`TEST_SCOPE` metadata. Direct runs also need an allowlisted `TEST_TYPE`.
|
||||
Merge runs need the trusted base-branch planner's `MERGE_TEST_PLAN`,
|
||||
`MERGE_GOLDEN_TESTS`, and `MERGE_SSIM_TESTS`. Full/direct/scheduled quality
|
||||
runs keep their complete matrices. The worker fetches and verifies the
|
||||
immutable commit before installing dependencies or invoking a lane script.
|
||||
|
||||
The new Modal adapter uses bounded one-to-four-GPU sandboxes and the same
|
||||
lane scripts as Kubernetes. It does not reactivate `pr_test.sh` or the old
|
||||
Modal SSIM fan-out, which cannot enforce this shared four-GPU-per-PR budget.
|
||||
The legacy files remain available for their existing manual workflows.
|
||||
|
||||
## Statuses and Default-Backend Cutover
|
||||
|
||||
The selected adapter reports separate suite contexts:
|
||||
|
||||
| Scope | GitHub context |
|
||||
|---|---|
|
||||
| Fastcheck | `gpu-ci/<backend>/fastcheck-passed` |
|
||||
| Merge or explicit full suite | `gpu-ci/<backend>/full-suite-passed` |
|
||||
| Direct lane | `gpu-ci/<backend>/direct-test-completed` |
|
||||
| Scheduled SSIM | `gpu-ci/<backend>/scheduled-ssim-passed` |
|
||||
|
||||
The repository variable `CI_GPU_BACKEND` controls which new backend can
|
||||
publish the existing required `fastcheck-passed` and `full-suite-passed`
|
||||
contexts. Empty or `slurm` preserves the existing status behavior. `modal`
|
||||
or `vllm` enables `ci-gpu-backend-status.yml`, which reads the latest statuses
|
||||
for that one backend and mirrors both required contexts. A missing result
|
||||
becomes pending; failure and error remain failures. It never combines one
|
||||
backend's Fastcheck with another backend's full-suite result. Late status
|
||||
events re-read current state instead of replaying stale event payloads.
|
||||
|
||||
Direct tests are diagnostic and do not promote a whole suite on the new
|
||||
backends. Rerun the matching Fastcheck or merge/full suite to clear its gate.
|
||||
Per-build tests on the other new backend remain separate diagnostics. The
|
||||
legacy direct-test aggregation workflow is disabled while a new backend is
|
||||
selected. These workflows retain the existing trust assumption that only
|
||||
authorized status-writing integrations can publish CI status contexts.
|
||||
|
||||
For production cutover, drain existing Slurm builds: its preserved pipeline
|
||||
still emits canonical status contexts. The new uploader rejects a Slurm
|
||||
override when `default_backend` is `modal` or `vllm`, preventing later legacy
|
||||
builds from overwriting the promoted backend's checks. Ensure no old bootstrap
|
||||
bypasses the new uploader. Synchronize the operator `default_backend` and
|
||||
the GitHub repository variable, then clear or rerun required checks for all
|
||||
open PRs. Old green contexts do not become new-backend validation merely
|
||||
because a setting changed. Run both Fastcheck and the merge/full suite on
|
||||
the promoted backend before allowing merge. Apply the same drain and rerun
|
||||
procedure when rolling back to Slurm.
|
||||
|
||||
## Validation and Recovery
|
||||
|
||||
Inspect the rendered opt-in pipeline without submitting a build:
|
||||
|
||||
```bash
|
||||
CI_GPU_BACKEND=vllm TEST_SCOPE=fastcheck \
|
||||
/opt/fastvideo-gpu-ci/venv/bin/python -I \
|
||||
/opt/fastvideo-gpu-ci/source/scripts/gpu_ci/entrypoint.py render \
|
||||
--config /etc/fastvideo-gpu-ci.json
|
||||
```
|
||||
|
||||
Validate a one-GPU lane, then a two-/four-GPU lane and the complete Fastcheck
|
||||
suite. Exercise two PRs plus a third waiter, concurrent builds of the same
|
||||
PR, cancellation, retries, and coordinator restart. Confirm observed GPU
|
||||
reservations never exceed two PRs, four per PR, or eight in total. A GPU
|
||||
reservation includes a pending worker, so unavailable nodes cannot cause
|
||||
the controller to submit more work than its allowance.
|
||||
|
||||
Run SSIM, training, and performance canaries separately. References must
|
||||
match the effective GPU/runtime/attention backend; do not silently reuse
|
||||
L40S performance results as GB200 baselines or reseed references as part of
|
||||
routine CI. Workers keep W&B offline and disable reference publication.
|
||||
|
||||
Inspect the persistent ledger and recover an abandoned attempt with:
|
||||
|
||||
```bash
|
||||
/opt/fastvideo-gpu-ci/venv/bin/python -I \
|
||||
/opt/fastvideo-gpu-ci/source/scripts/gpu_ci/entrypoint.py status \
|
||||
--config /etc/fastvideo-gpu-ci.json
|
||||
|
||||
/opt/fastvideo-gpu-ci/venv/bin/python -I \
|
||||
/opt/fastvideo-gpu-ci/source/scripts/gpu_ci/entrypoint.py recover \
|
||||
--config /etc/fastvideo-gpu-ci.json --build-id BUILD_ID.JOB_ID
|
||||
```
|
||||
|
||||
Use the exact ledger ID from `status`; each Buildkite retry has a distinct
|
||||
job ID. Recovery refuses a live coordinator, persists cancellation, stops
|
||||
owned resources, and releases reservations only after confirming termination.
|
||||
If creation or cleanup is ambiguous, investigate the recorded handle on its
|
||||
backend and retain the reservation until the outcome is known. Do not delete
|
||||
the database or manually zero counters to unblock the queue.
|
||||
|
||||
The dispatcher uploads controller logs, per-lane numeric results, request
|
||||
metadata, and the suite summary to Buildkite. These are the sources for its
|
||||
exit status. Generated videos and JUnit files stay in the worker unless a
|
||||
dedicated Kubernetes artifact PVC is configured; automatic publication of
|
||||
those worker files and Modal worker artifacts is not implemented. This is
|
||||
a deployment limitation to account for before replacing existing artifact
|
||||
review workflows.
|
||||
@@ -10,6 +10,7 @@ slash-command mappings, and workflow ownership live in
|
||||
|---|---|---|
|
||||
| Unit tests | `fastvideo/tests/api`, `fastvideo/tests/dataset`, `fastvideo/tests/entrypoints`, `fastvideo/tests/workflow`, CPU-safe `fastvideo/tests/train` subsets | Validate individual functions, APIs, contracts, and lightweight workflows. |
|
||||
| Component tests | `fastvideo/tests/encoders`, `fastvideo/tests/transformers`, `fastvideo/tests/vaes` | Validate loading and basic behavior for model components. |
|
||||
| Golden gates | `fastvideo/tests/golden_gate` | Compare small, deterministic component outputs exactly against device/runtime-matched reference tensors. |
|
||||
| Train framework tests | `fastvideo/tests/train/models`, `fastvideo/tests/train/methods` | Exercise the new `fastvideo/train/` framework on real checkpoints and tiny synthetic batches. |
|
||||
| SSIM tests | `fastvideo/tests/ssim` | Compare generated videos against references to catch visual regressions. |
|
||||
| Training tests | `fastvideo/tests/training` | Validate legacy training loops, LoRA, distillation, self-forcing, and VSA behavior. |
|
||||
@@ -20,14 +21,74 @@ slash-command mappings, and workflow ownership live in
|
||||
|
||||
## Running Tests Locally
|
||||
|
||||
Run the narrowest useful suite while iterating:
|
||||
Start with the cheapest checks that cover the changed behavior, and stop on a
|
||||
failure before starting heavier dependent checks:
|
||||
|
||||
1. Run pre-commit on changed files and focused import, config, and contract tests.
|
||||
2. Run the smallest matching component golden gate. Prefer one GPU, tiny fixed
|
||||
inputs, cached component weights, and direct tensor comparisons over renders.
|
||||
3. Run focused component parity or default SSIM when the golden does not cover
|
||||
the changed behavior, such as VAE normalization or pipeline wiring.
|
||||
4. Run full-quality renders or broad suites when required by the change, an
|
||||
explicit request, or CI policy, rather than on every edit.
|
||||
|
||||
A golden must cover the component being changed. Wan has four small gates:
|
||||
|
||||
| Gate | Boundary |
|
||||
|---|---|
|
||||
| `test_wan_t2v.py` | Dense transformer block 0 |
|
||||
| `test_wan_vae.py` | FP32 encode, BF16 decode, streaming, and cache reset |
|
||||
| `test_wan_causal.py` | Real block weights, cache append/rewrite, and sink eviction |
|
||||
| `test_wan_denoising.py` | Three real 1.3B DiT/UniPC steps with fixed prompt embeddings |
|
||||
|
||||
These use an immutable Wan checkpoint revision and only download the required
|
||||
component. The trajectory gate needs no tokenizer, text encoder, VAE, or video
|
||||
reference. Weight-free stage tests also exercise every step of a 50-step UniPC
|
||||
loop, CFG caching, expert switching, conditioning layouts, and DMD RNG order.
|
||||
These checks do not replace independent Diffusers parity or end-to-end SSIM.
|
||||
|
||||
Use the golden's matching GPU, dtype, backend, and runtime. A missing reference
|
||||
or environment mismatch is not a pass. Do not replace a reference with the
|
||||
candidate output just to clear a failure. For a relocation, the unchanged
|
||||
parent is the baseline; two imports of the same class are not numerical proof.
|
||||
|
||||
The VAE and transformer CI lanes run their Wan component goldens first.
|
||||
Selected integration lanes wait for the golden lane in merge/full builds.
|
||||
See [CI/CD Architecture](ci_architecture.md) for direct-rerun and skip semantics.
|
||||
|
||||
For Wan, one command enforces the local ordering and stops on failure:
|
||||
|
||||
The initial contracts include `tests/api/test_wan_definitions.py`: all registered
|
||||
Wan aliases, local-manifest detector precedence, sampling/precision defaults,
|
||||
config isolation, and legacy serialized-config compatibility. These require no
|
||||
weights or Hub access; the current package import still needs its prepared
|
||||
runtime environment.
|
||||
|
||||
```bash
|
||||
pytest tests/
|
||||
pytest fastvideo/tests/ -v
|
||||
pytest fastvideo/tests/encoders -vs
|
||||
pytest fastvideo/tests/transformers -vs
|
||||
pytest fastvideo/tests/vaes -vs
|
||||
bash scripts/validate_wan.sh vae # contracts, then the VAE golden
|
||||
bash scripts/validate_wan.sh dense parity # contracts, goldens, Diffusers parity
|
||||
bash scripts/validate_wan.sh all default # then focused T2V/I2V/causal SSIM
|
||||
```
|
||||
|
||||
The second argument is an upper validation tier, not a reference override.
|
||||
Supply the GPU/backend/runtime and SSIM model/tier settings that match the
|
||||
reference. The script never updates references. Causal component coverage is
|
||||
a block-cache fingerprint, not independent full causal-pipeline parity.
|
||||
|
||||
New named-tensor gates write a missing output to `*.candidate.pt` and fail;
|
||||
running candidate code again cannot turn it into an approved reference. Seed
|
||||
from unchanged, pushed source, verify two independent processes bit-for-bit,
|
||||
then review and publish only the new reference files. Preserve the baseline
|
||||
source SHA, test recipe SHA, checkpoint revision, runtime, and comparison
|
||||
receipt with the run artifacts. Never overwrite an existing video reference
|
||||
as a side effect of adding a tensor gate.
|
||||
|
||||
Examples of focused checks:
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/loader/test_wan_family_imports.py -q
|
||||
pytest fastvideo/tests/golden_gate/test_wan_t2v.py -q
|
||||
pytest fastvideo/tests/vaes/test_wan_vae.py -q
|
||||
```
|
||||
|
||||
GPU-heavy suites need the right hardware, credentials, local caches, and
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Cosmos recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>World-generation prompts work best describing a scene and camera motion; the built-in prompt in the example is a known-good starting point.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- World-generation prompts work best describing a scene and camera motion; the built-in prompt in the example is a known-good starting point.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+32
-20
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# FLUX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,24 +112,30 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>FLUX.1 dev defaults to a local <code>official_weights/FLUX.1-dev</code> directory in the example; the cookbook command passes the Hugging Face ID explicitly instead.</li>
|
||||
<li>Image outputs land under <code>outputs/</code>; adjust <code>--output</code> if that path is not writable.</li>
|
||||
<li>Gated checkpoints (FLUX.1 dev): run <code>huggingface-cli login</code> and accept the license on Hugging Face first.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- FLUX.1 dev defaults to a local `official_weights/FLUX.1-dev` directory in the example; the cookbook command passes the Hugging Face ID explicitly instead.
|
||||
- Image outputs land under `outputs/`; adjust `--output` if that path is not writable.
|
||||
- Gated checkpoints (FLUX.1 dev): run `huggingface-cli login` and accept the license on Hugging Face first.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# GLM-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The editing example reads <code>assets/images/couple.jpg</code> from the repository root, so run it from a repo checkout rather than an arbitrary working directory.</li>
|
||||
<li>Output paths default under <code>image_output/</code>; pass <code>--output</code> to change them.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The editing example reads `assets/images/couple.jpg` from the repository root, so run it from a repo checkout rather than an arbitrary working directory.
|
||||
- Output paths default under `image_output/`; pass `--output` to change them.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+32
-20
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Hunyuan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,24 +112,30 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Out of memory: <code>basic_hy15.py</code> already enables dit/VAE/text-encoder CPU offload; further tradeoffs are described in <a href="../../inference/optimizations/">Optimizations</a>.</li>
|
||||
<li><code>pin_cpu_memory</code> errors on low-RAM machines are documented inline in the example; set it to false as the source comment suggests.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Out of memory: `basic_hy15.py` already enables dit/VAE/text-encoder CPU offload; further tradeoffs are described in [Optimizations](../inference/optimizations.md).
|
||||
- `pin_cpu_memory` errors on low-RAM machines are documented inline in the example; set it to false as the source comment suggests.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+2
-11
@@ -41,13 +41,14 @@ hide:
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>MiniMax H3</strong><small>Video + stereo audio</small></span>
|
||||
<span class="cookbook-count">6 recipes</span>
|
||||
<span class="cookbook-count">9 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2VA</li>
|
||||
<li>FL2VA</li>
|
||||
<li>Ref2VA</li>
|
||||
<li>MLX T2VA</li>
|
||||
<li>DGX Spark</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
@@ -454,16 +455,6 @@ hide:
|
||||
</article>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="cookbook-roadmap" aria-labelledby="roadmap-heading">
|
||||
<h2 id="roadmap-heading">Inference first, then the full workflow</h2>
|
||||
<p>
|
||||
Inference is the first complete stage. Distillation, fine-tuning,
|
||||
training, evaluation, optimization, and deployment will reuse the same
|
||||
family-first structure as their recipes land. Each family page shows
|
||||
which stages are available and which are planned.
|
||||
</p>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<small class="cookbook-logo-credit">
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Kandinsky 5 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Alternative Lite/Pro checkpoints are listed inline in the maintained examples; swap the model string only after checking its Hugging Face card.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Alternative Lite/Pro checkpoints are listed inline in the maintained examples; swap the model string only after checking its Hugging Face card.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
Both recipes map to checked-in FastVideo examples and recorded single-GPU B200 runs. The image-to-video run also records 10,365.89 MB peak GPU memory. These measurements describe the recorded runs; they are not minimum hardware requirements.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Both recipes map to checked-in FastVideo examples and recorded single-GPU B200 runs. The image-to-video run also records 10,365.89 MB peak GPU memory. These measurements describe the recorded runs; they are not minimum hardware requirements.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# LongCat recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="longcat" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="longcat" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Each LongCat script runs multiple passes (basic, distilled, refine); total runtime scales accordingly, and every pass prints its own output directory.</li>
|
||||
<li>Out of memory: the sources already enable VAE and text-encoder CPU offload; further options are covered in <a href="../../inference/offloading/">Offloading</a>.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Each LongCat script runs multiple passes (basic, distilled, refine); total runtime scales accordingly, and every pass prints its own output directory.
|
||||
- Out of memory: the sources already enable VAE and text-encoder CPU offload; further options are covered in [Offloading](../inference/offloading.md).
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+32
-20
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# LTX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,24 +112,30 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The distilled recipe is source-configured for four GPUs; running it on fewer GPUs is unverified and may fail during distributed setup.</li>
|
||||
<li>Audio-less output usually means the base checkpoint resolved instead of the distilled LTX-2 checkpoint with audio; check the loaded model ID in the logs.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The distilled recipe is source-configured for four GPUs; running it on fewer GPUs is unverified and may fail during distributed setup.
|
||||
- Audio-less output usually means the base checkpoint resolved instead of the distilled LTX-2 checkpoint with audio; check the loaded model ID in the logs.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Matrix Game recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Matrix Game 3.0 downloads its reference input image from GitHub; offline machines should pre-download it and edit the <code>IMAGE_URL</code> constant locally.</li>
|
||||
<li>Streaming variants of Matrix Game 2.0 exist under <code>examples/inference/basic/basic_matrixgame2_streaming.py</code> but are not included as cookbook recipes.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Matrix Game 3.0 downloads its reference input image from GitHub; offline machines should pre-download it and edit the `IMAGE_URL` constant locally.
|
||||
- Streaming variants of Matrix Game 2.0 exist under `examples/inference/basic/basic_matrixgame2_streaming.py` but are not included as cookbook recipes.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+76
-39
@@ -5,7 +5,13 @@ hide:
|
||||
|
||||
# MiniMax H3 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
FastH3 is two distilled MiniMax-H3 checkpoints. **V1** is the four-step
|
||||
launch. Some Hub repo names still say Preview. That name is historical. V1 is
|
||||
a full model, not a demo. **V2** is the eight-step checkpoint. More forwards
|
||||
is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
[FastH3 distilled checkpoint schedules](../inference/fasth3-distilled.md).
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -15,8 +21,8 @@ hide:
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Primary focus · Inference</p>
|
||||
<h2>MiniMax H3 recipes</h2>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>6 maintained recipes</span>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>9 maintained recipes</span>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
@@ -29,14 +35,21 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<details class="cookbook-modes">
|
||||
<summary>Compare H3 modes and options</summary>
|
||||
<h2 id="h3-modes-heading">Supported modes</h2>
|
||||
<p>
|
||||
CUDA covers T2VA, FL2VA, and Ref2VA on the full checkpoint, plus FastH3
|
||||
Preview and FastH3 LoRA. MLX is T2VA only. Temporal <code>--fast</code>,
|
||||
spatial <code>--fast-spatial</code>, and opt-in VSA are flags on the same
|
||||
V1 and FastH3 V2. FastH3 V1 also has a DGX Spark runtime with
|
||||
a 1-Spark or 2-Spark device row. MLX is T2VA only: V1 and V2.
|
||||
Temporal <code>--fast</code>, spatial <code>--fast-spatial</code>, and opt-in VSA are flags on the same
|
||||
MLX script, not extra recipes.
|
||||
</p>
|
||||
<div class="cookbook-modes__table-wrap">
|
||||
@@ -51,8 +64,8 @@ hide:
|
||||
<tbody>
|
||||
<tr>
|
||||
<td>T2VA</td>
|
||||
<td>Full H3, FastH3 Preview, FastH3 LoRA</td>
|
||||
<td>FastH3 Preview after a local DiT conversion</td>
|
||||
<td>Full H3, FastH3 V1, FastH3 LoRA, FastH3 V2</td>
|
||||
<td>FastH3 V1 or FastH3 V2 after a local DiT conversion</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>FL2VA</td>
|
||||
@@ -84,6 +97,11 @@ hide:
|
||||
<td>No cookbook recipe</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>DGX Spark</td>
|
||||
<td>FastH3 V1 on one GB10, or two Sparks with Ray sequence parallel (<code>sp_size=2</code>) over QSFP RoCE. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks.</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
@@ -92,7 +110,7 @@ hide:
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick an H3 recipe and runtime</h2>
|
||||
<p>Choose the result you want, then use a maintained CUDA or MLX path.
|
||||
<p>Choose the result you want, then use a maintained CUDA, DGX Spark, or MLX path.
|
||||
Device claims stay tied to checked-in sources and recorded runs.</p>
|
||||
</div>
|
||||
|
||||
@@ -118,6 +136,17 @@ hide:
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row" data-cookbook-device-row hidden>
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Devices</strong>
|
||||
<span data-cookbook-device-caption>1 Spark or a QSFP pair</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-device-options role="group" aria-label="Devices">
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div data-cookbook-knobs></div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<div class="cookbook-selection-row" data-cookbook-usage>
|
||||
<div class="cookbook-selection-row__label">
|
||||
@@ -217,36 +246,44 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
<p class="cookbook-eyebrow">CUDA</p>
|
||||
<pre><code>UV_TORCH_BACKEND=cu130 uv pip install -e ".[fasth3]"</code></pre>
|
||||
<p class="cookbook-eyebrow">Apple Silicon</p>
|
||||
<pre><code>uv pip install -e ".[mlx]"</code></pre>
|
||||
<p>Follow the <a href="../../getting_started/installation/mlx/">MLX install guide</a> for the extra, <code>ffmpeg</code>, and a clone. Then pick FastH3 V1 or V2 in the builder above.</p>
|
||||
<p class="cookbook-eyebrow">NVIDIA DGX Spark</p>
|
||||
<pre><code>UV_TORCH_BACKEND=cu130 uv pip install -e .</code></pre>
|
||||
<p>Follow the <a href="../../getting_started/installation/spark/">DGX Spark install guide</a> for ARM64 CUDA 13. One Spark is a local process. Two Sparks need Ray on the QSFP link:</p>
|
||||
<pre><code>uv pip install ray</code></pre>
|
||||
<p>Bring up the cluster from <a href="../../getting_started/installation/spark_pair/">pairing two Sparks</a> before selecting 2 Sparks in the builder.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The full CUDA H3 examples request four GPUs by default. Their sources do not claim a GPU model or memory minimum.</li>
|
||||
<li>The FastH3 CUDA performance profile was measured on four GB200 GPUs. Use its strict profile when exact operation order matters more than the measured performance configuration.</li>
|
||||
<li>The MLX source runtime supports T2VA, optional temporal <code>--fast</code>, optional spatial <code>--fast-spatial</code>, and opt-in VSA on <code>--include-vsa</code> checkpoints. FastH3 V2 MLX converts with <code>--include-vsa</code> and runs eight forwards. FL2VA, Ref2VA, and two-pass refinement are not wired.</li>
|
||||
<li>GPU count and VAE decode backend are configurable in the builder above for FastH3 CUDA recipes. Only the value shown by default has a recorded run; other supported values are unmeasured here.</li>
|
||||
<li>DGX Spark is a runtime on FastH3 V1, not a separate family card. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks. The CUDA GPU-count knob does not apply to Spark.</li>
|
||||
<li>GB10 has no FA4 / sm_100a VSA kernel. Keep <code>FASTVIDEO_FA4=0</code> and <code>FASTVIDEO_VSA_SM100A=0</code>. Legal <code>num_frames</code> values are <code>17n+5</code>, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
CUDA FastH3 uses the pinned performance dependencies:
|
||||
|
||||
UV_TORCH_BACKEND=cu130 uv pip install -e ".[fasth3]"
|
||||
|
||||
Apple Silicon uses the native MLX extra and a locally converted H3 DiT:
|
||||
|
||||
uv pip install -e ".[mlx]"
|
||||
|
||||
Follow the [Apple Silicon guide](../getting_started/installation/mps.md#run-fasth3-preview)
|
||||
for the download, conversion, and storage requirements.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The full CUDA H3 examples request four GPUs by default. Their sources do not claim a GPU model or memory minimum.
|
||||
- The FastH3 CUDA performance profile was measured on four GB200 GPUs. Use its strict profile when exact operation order matters more than the measured performance configuration.
|
||||
- The MLX source runtime supports T2VA, optional temporal `--fast`, optional spatial `--fast-spatial`, and opt-in VSA on `--include-vsa` checkpoints. FL2VA, Ref2VA, and two-pass refinement are not wired.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
Every command, model ID, and flag on this page maps to a checked-in FastVideo source. Recipes marked **Verified** also have a recorded hardware path in linked FastVideo evidence. The full H3 CUDA examples remain **Source-backed** where the source records a GPU count but no GPU model or memory requirement. Unlisted hardware is unknown, not unsupported.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Every command, model ID, and flag on this page maps to a checked-in FastVideo source. Recipes marked <strong>Verified</strong> also have a recorded hardware path in linked FastVideo evidence. The full H3 CUDA examples remain <strong>Source-backed</strong> where the source records a GPU count but no GPU model or memory requirement. Unlisted hardware is unknown, not unsupported.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# MMAudio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>If the example reports a missing local <code>converted_weights/mmaudio/large_44k_v2</code> path, the <code>MMAUDIO_MODEL_PATH</code> env var from the cookbook command was not applied; export it in the same shell.</li>
|
||||
<li>Alternatively convert upstream weights yourself with <code>scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py</code> and point the env var at the result.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- If the example reports a missing local `converted_weights/mmaudio/large_44k_v2` path, the `MMAUDIO_MODEL_PATH` env var from the cookbook command was not applied; export it in the same shell.
|
||||
- Alternatively convert upstream weights yourself with `scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py` and point the env var at the result.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Run H3 with a server and playground
|
||||
|
||||
Start FastH3 on CUDA or Apple Silicon MLX, then iterate on prompts in a browser,
|
||||
Start FastH3 on CUDA, one DGX Spark, or Apple Silicon MLX, then iterate on prompts in a browser,
|
||||
with cURL, or from your app. The server and clients can run on the same machine.
|
||||
|
||||
The server supports the OpenAI-compatible video-job API. The Python and
|
||||
@@ -8,8 +8,9 @@ JavaScript client examples use that interface, but requests go to FastVideo.
|
||||
You do not need an OpenAI account or cloud key.
|
||||
|
||||
The [H3 recipe selector](minimax-h3.md) provides the same workflow with runtime
|
||||
selection. This guide covers FastH3 Preview text-to-video/audio. Other H3
|
||||
recipes keep their direct Python commands.
|
||||
selection. This guide covers FastH3 V1 (four forwards) and FastH3 V2
|
||||
(eight forwards). Start a server, then iterate in the playground or with the
|
||||
OpenAI Python client. Other H3 recipes keep their direct Python commands.
|
||||
|
||||
CUDA requests reuse one loaded `VideoGenerator`. The Python SDK can do the same
|
||||
when you reuse the generator across `generate()` calls. MLX keeps one
|
||||
@@ -36,6 +37,40 @@ advertises it as `fasth3`. It configures four CUDA GPUs but does not record
|
||||
a GPU model or VRAM requirement. This is a source-backed server profile, not
|
||||
the measured GB200 Python performance profile. Compilation is disabled.
|
||||
|
||||
For FastH3 V2, use the same install and the 8-step config (nine sigma
|
||||
points, eight DiT forwards, VSA sparsity 0.8):
|
||||
|
||||
```bash
|
||||
UV_TORCH_BACKEND=cu130 uv pip install -e ".[fasth3]"
|
||||
fastvideo serve --config examples/serving/openai_fasth3_8step.yaml --server.host 127.0.0.1
|
||||
```
|
||||
|
||||
Keep the server running. In another terminal, check readiness:
|
||||
|
||||
```bash
|
||||
curl --fail-with-body http://127.0.0.1:8000/health
|
||||
```
|
||||
|
||||
After model loading completes, the response is `{"status":"ok"}`.
|
||||
|
||||
### NVIDIA DGX Spark
|
||||
|
||||
Complete the [DGX Spark installation](../getting_started/installation/spark.md)
|
||||
before running these commands. GB10 has no FA4 / sm_100a VSA kernel:
|
||||
|
||||
```bash
|
||||
UV_TORCH_BACKEND=cu130 uv pip install -e .
|
||||
FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 \
|
||||
fastvideo serve --config examples/serving/openai_fasth3_spark.yaml --server.host 127.0.0.1
|
||||
```
|
||||
|
||||
The configuration loads `FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree`
|
||||
on one GB10 and advertises it as `fasth3`. Lazy module load still reloads
|
||||
Qwen3-VL and the DiT between phases of each request. Legal `num_frames` values
|
||||
are `17n+5`, capped at 362 (15.08 s); a 345-frame request on one Spark can OOM.
|
||||
There is no cookbook server for two Sparks; use the generate YAML after
|
||||
[pairing two Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
Keep the server running. In another terminal, check readiness:
|
||||
|
||||
```bash
|
||||
@@ -46,7 +81,7 @@ After model loading completes, the response is `{"status":"ok"}`.
|
||||
|
||||
### Apple Silicon MLX
|
||||
|
||||
Complete the [Apple Silicon installation](../getting_started/installation/mps.md#run-fasth3-preview),
|
||||
Complete the [MLX install](../getting_started/installation/mlx.md),
|
||||
including `ffmpeg`. From your FastVideo clone, install the MLX extra:
|
||||
|
||||
```bash
|
||||
@@ -77,14 +112,30 @@ LoRA selection, and alternate decoders are not exposed by this server adapter.
|
||||
Unsupported request options return HTTP 400 before a job starts.
|
||||
|
||||
The server binds to `127.0.0.1:8000` and advertises `fasth3`, so the playground
|
||||
and clients below work unchanged. Do not run CUDA and MLX servers on the same
|
||||
port. MLX readiness means the pipeline is initialized; components load during
|
||||
and clients below work unchanged. Do not run CUDA, Spark, and MLX servers on the
|
||||
same port. MLX readiness means the pipeline is initialized; components load during
|
||||
generation. The first request can take longer than a repeated cached prompt.
|
||||
|
||||
The MLX server has no recorded device or unified-memory requirement. The
|
||||
direct Python recipe's M4 Max measurements are not a server benchmark or a
|
||||
minimum-memory claim.
|
||||
|
||||
For FastH3 V2, convert that checkpoint's transformer with `--include-vsa`
|
||||
and start the 8-step config. Do not overwrite a V1 export. The
|
||||
[MiniMax H3 cookbook](minimax-h3.md) has the same download and convert
|
||||
commands as the Python recipe.
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats "int8" --include-vsa
|
||||
python -m fastvideo.entrypoints.openai.mlx_server --config examples/serving/mlx_fasth3_8step.yaml
|
||||
```
|
||||
|
||||
That adapter passes `num_steps=8` when the HTTP field is `num_inference_steps=9`,
|
||||
and it enables the trained VSA recipe (sparsity 0.8, 64-token tiles). Reuse the
|
||||
preview VAE, audio VAE, text encoder, and tokenizer if those directories already
|
||||
exist; edit the YAML paths if your files live elsewhere.
|
||||
|
||||
## Open the playground
|
||||
|
||||
Open [the local H3 playground](http://127.0.0.1:8000/playground/) after startup.
|
||||
@@ -110,9 +161,11 @@ or manage a GPU server for you.
|
||||
## Generate with cURL or an SDK
|
||||
|
||||
These examples use the server's resolution, frame count, and sampling defaults.
|
||||
Do not copy Sora-specific durations or resolutions onto H3. Both server configs
|
||||
use 124 frames, 24 fps, and the five-point distilled sigma schedule with four
|
||||
DiT forwards. CUDA uses 1344 × 768; MLX uses 832 × 480.
|
||||
Do not copy Sora-specific durations or resolutions onto H3. V1 configs use
|
||||
124 frames, 24 fps, and the five-point distilled sigma schedule with four DiT
|
||||
forwards. V2 configs use nine sigma points and eight DiT forwards. CUDA
|
||||
and one Spark use 1344 × 768; MLX uses 832 × 480. The OpenAI Python client is
|
||||
the same for every FastH3 server that advertises `fasth3`.
|
||||
|
||||
Each client submits a job, checks for completion or failure, and downloads an
|
||||
MP4 named after the job ID. Polling stops after 30 minutes; a timeout does not
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Audio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Loader errors about monolithic checkpoints mean an upstream <code>stabilityai/stable-audio-open-*</code> ID was used; use the FastVideo converted repos from the recipes.</li>
|
||||
<li>Duration and step knobs (<code>audio_end_in_s</code>, <code>num_inference_steps</code>) are documented inline in the example source.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Loader errors about monolithic checkpoints mean an upstream `stabilityai/stable-audio-open-*` ID was used; use the FastVideo converted repos from the recipes.
|
||||
- Duration and step knobs (`audio_end_in_s`, `num_inference_steps`) are documented inline in the example source.
|
||||
|
||||
## Evidence status
|
||||
|
||||
The Stable Audio Open 1.0 recipe maps to a checked-in example and a recorded single-GPU B200 run. The Stable Audio Open Small recipe remains **Source-backed** because the implementation PR did not record a full run for that gated checkpoint. Neither recipe claims a minimum VRAM requirement.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The Stable Audio Open 1.0 recipe maps to a checked-in example and a recorded single-GPU B200 run. The Stable Audio Open Small recipe remains <strong>Source-backed</strong> because the implementation PR did not record a full run for that gated checkpoint. Neither recipe claims a minimum VRAM requirement.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Diffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The example writes several PNGs under <code>outputs/sd35/samples/</code>; make sure the output directory is writable.</li>
|
||||
<li>Gated checkpoints: Stability AI models require accepting the license and <code>huggingface-cli login</code>.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- The example writes several PNGs under `outputs/sd35/samples/`; make sure the output directory is writable.
|
||||
- Gated checkpoints: Stability AI models require accepting the license and `huggingface-cli login`.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# TurboDiffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="turbodiffusion" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="turbodiffusion" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>TurboDiffusion paths load community-published <code>loayrashid/TurboWan*</code> checkpoints; availability is governed by those repos.</li>
|
||||
<li>The SLA attention backend used by the I2V recipe is selected inside the example source; do not combine it with another <code>FASTVIDEO_ATTENTION_BACKEND</code> override in the same shell.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- TurboDiffusion paths load community-published `loayrashid/TurboWan*` checkpoints; availability is governed by those repos.
|
||||
- The SLA attention backend used by the I2V recipe is selected inside the example source; do not combine it with another `FASTVIDEO_ATTENTION_BACKEND` override in the same shell.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+37
-24
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Wan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -29,8 +29,15 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-modes" aria-labelledby="wan-modes-heading">
|
||||
<details class="cookbook-modes">
|
||||
<summary>Compare Wan modes and options</summary>
|
||||
<h2 id="wan-modes-heading">Supported modes</h2>
|
||||
<p>
|
||||
FastMetal MLX is T2V in the checked-in examples. Image-to-video and
|
||||
@@ -84,7 +91,7 @@ hide:
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
</section>
|
||||
</details>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -163,26 +170,32 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Out of memory on the A14B recipes: the checked-in sources already enable CPU offload; see <a href="../../inference/configuration/">Configuration</a> for the offload surface before reducing resolution or frames.</li>
|
||||
<li>The FastWan2.1 recipe requires <code>VIDEO_SPARSE_ATTN</code>; confirm the environment variable in the command was set in the same shell.</li>
|
||||
<li>FastMetal MLX: install with the <a href="../../getting_started/installation/mlx/">MLX install guide</a>, then pick a FastMetal recipe in the builder. CUDA FastWan-QAD checkpoints are refused on the MLX runtime.</li>
|
||||
<li>FastMetal 5B uses <code>mlx_wan22_generate.py</code>. 1.3B and 14B use <code>mlx_wan_prompt_to_video.py</code>.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Out of memory on the A14B recipes: the checked-in sources already enable CPU offload; see [Configuration](../inference/configuration.md) for the offload surface before reducing resolution or frames.
|
||||
- The FastWan2.1 recipe requires `VIDEO_SPARSE_ATTN`; confirm the environment variable in the command was set in the same shell.
|
||||
- FastMetal MLX: install with `uv pip install -e ".[mlx]"`, then follow the [Apple Silicon guide](../getting_started/installation/mps.md). CUDA FastWan-QAD checkpoints are refused on the MLX runtime.
|
||||
- FastMetal 5B uses `mlx_wan22_generate.py`. 1.3B and 14B use `mlx_wan_prompt_to_video.py`.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
Every recipe on this page maps to a checked-in FastVideo source. The FastMetal MLX releases include the recorded M4 Max system memory, documented unified-memory floor, and measured peak MLX memory. CUDA entries remain **Source-backed** where the examples record a GPU count but no exact GPU model or VRAM. Unlisted hardware is unknown, not unsupported.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Every recipe on this page maps to a checked-in FastVideo source. The FastMetal MLX releases include the recorded M4 Max system memory, documented unified-memory floor, and measured peak MLX memory. CUDA entries remain <strong>Source-backed</strong> where the examples record a GPU count but no exact GPU model or VRAM. Unlisted hardware is unknown, not unsupported.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
+31
-19
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Z-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="zimage" data-recipes="../../assets/cookbook-recipes.json?v=6">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="zimage" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,6 +28,12 @@ hide:
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
@@ -106,23 +112,29 @@ hide:
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Before you run
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
The generated commands expect a local clone:
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Output defaults to <code>outputs/zimage/zimage_turbo.png</code>; pass <code>--output</code> to redirect.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
|
||||
Use [Configuration](../inference/configuration.md) for supported Python and
|
||||
CLI settings, [Optimizations](../inference/optimizations.md) for attention and
|
||||
memory tradeoffs, and the [support matrix](../inference/support_matrix.md) for
|
||||
the supported model and optimization surface.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Output defaults to `outputs/zimage/zimage_turbo.png`; pass `--output` to redirect.
|
||||
- Gated or missing checkpoints: run `huggingface-cli login` and confirm you accepted the model's license on Hugging Face.
|
||||
|
||||
## Evidence status
|
||||
|
||||
All recipes on this page are **Source-backed**: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are **Unknown** and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -30,6 +30,7 @@ surfaces:
|
||||
image_encoder_cpu_offload: generator.engine.offload.image_encoder
|
||||
vae_cpu_offload: generator.engine.offload.vae
|
||||
pin_cpu_memory: generator.engine.offload.pin_cpu_memory
|
||||
lazy_module_load: generator.engine.offload.lazy_module_load
|
||||
enable_torch_compile: generator.engine.compile.enabled
|
||||
enable_torch_compile_text_encoder: generator.engine.compile.text_encoder_enabled
|
||||
enable_torch_compile_vae: generator.engine.compile.vae_enabled
|
||||
@@ -81,6 +82,9 @@ surfaces:
|
||||
inference_torch_compile: "Regional inference compile opt-in currently carried through PipelineSelection.experimental rather than CompileConfig."
|
||||
vae_parallel_decode: "MiniMax-H3 sequence-parallel VAE decode opt-in; model-specific optimization not yet represented in the typed public schema."
|
||||
h3_sequential_load: "MiniMax-H3 sequential text-encoder then DiT/VAE load; model-specific optimization not yet represented in the typed public schema."
|
||||
video_decode_backend: "MiniMax-H3 video decoder selection (full VAE vs TAEH3 preview); model-specific optimization not yet represented in the typed public schema."
|
||||
taeh3_checkpoint: "Optional local TAEH3 safetensors path; model-specific optimization not yet represented in the typed public schema."
|
||||
taeh3_chunk_size: "TAEH3 temporal chunk length; model-specific optimization not yet represented in the typed public schema."
|
||||
vae_parallel_encode: "MiniMax-H3 sequence-parallel reference VAE encode opt-in; model-specific optimization not yet represented in the typed public schema."
|
||||
vae_parallel_decode_strategy: "Chunk-transport collective for vae_parallel_decode; model-specific optimization not yet represented in the typed public schema."
|
||||
attention_backend: "Process-wide default attention-backend request applied per component at load time; kernel-selection knob not yet represented in the typed public schema."
|
||||
|
||||
+33
-11
@@ -11,6 +11,8 @@ FastVideo maps a Diffusers-style repo into a pipeline like this:
|
||||
- `fastvideo/models/*`: model implementations (DiT, VAE, encoders, upsamplers).
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/models/wan/`: Wan's transformers, VAE, configs, and variant definitions
|
||||
together; the old Wan component/config modules remain compatibility imports.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipelines.
|
||||
@@ -48,13 +50,29 @@ runtime parameters consistent:
|
||||
|
||||
- `fastvideo/configs/models/`: architecture definitions, layer shapes, and
|
||||
`param_names_mapping` rules for key renaming.
|
||||
- `fastvideo/models/wan/config.py`: Wan's transformer architecture and mapping rules,
|
||||
co-located with `transformer.py`.
|
||||
- `fastvideo/models/wan/vae_config.py`: Wan's VAE architecture and runtime
|
||||
settings, co-located with `vae.py`.
|
||||
- `fastvideo/models/wan/pipeline_config.py`: Wan component and pipeline defaults;
|
||||
`configs/pipelines/wan.py` remains a compatibility import.
|
||||
- `fastvideo/models/wan/definition.py`: data-only Wan variants linking HF aliases,
|
||||
config classes, presets, workload metadata, and default sampling algorithms.
|
||||
- `fastvideo/configs/pipelines/`: pipeline wiring and required components.
|
||||
- `fastvideo/api/sampling_param.py`: sampling parameters (steps, frames,
|
||||
guidance scale, resolution, fps). Defaults come from profiles in
|
||||
`fastvideo/pipelines/basic/<family>/profiles.py`.
|
||||
guidance scale, resolution, fps). Defaults come from presets in
|
||||
`fastvideo/pipelines/basic/<family>/presets.py`.
|
||||
- `fastvideo/registry.py`: unified registry for pipeline config + sampling
|
||||
defaults and model metadata resolution, defined via explicit
|
||||
`register_configs(...)` blocks (no separate dict registries).
|
||||
defaults and model metadata resolution. Wan registrations consume its
|
||||
family-local definitions; other families use `register_configs(...)` blocks.
|
||||
|
||||
Wan definitions reference existing defaults rather than copying them. Dense
|
||||
UniPC, dense DMD, and causal DMD remain separate sampling algorithms. Dense
|
||||
DMD uses a full training-noise scheduler with shift 8.0, separate from the
|
||||
configurable scheduler mutated during timestep preparation. The catalog does
|
||||
not override checkpoint manifests, user pipeline overrides, or component
|
||||
precision settings. HF IDs, local checkpoints, and old config imports retain
|
||||
their existing resolution behavior, including first-match detector ordering.
|
||||
|
||||
`FastVideoArgs` (in `fastvideo/fastvideo_args.py`) provides runtime settings and
|
||||
is passed into pipeline construction and stages.
|
||||
@@ -96,7 +114,8 @@ Note on tensor names:
|
||||
|
||||
Official checkpoints often use different `state_dict` names than FastVideo's
|
||||
module layout. We translate tensor names via the DiT arch config mapping
|
||||
(`param_names_mapping` under `fastvideo/configs/models/dits/`). This is similar
|
||||
(`param_names_mapping` under `fastvideo/configs/models/dits/`, or
|
||||
`fastvideo/models/wan/config.py` for Wan). This is similar
|
||||
in spirit to name-translation layers used in systems like vLLM and SGLang.
|
||||
|
||||
Example HF repo (Wan 2.1 T2V 1.3B Diffusers):
|
||||
@@ -137,13 +156,16 @@ Example `model_index.json` from that repo:
|
||||
How this maps to FastVideo:
|
||||
|
||||
- `WanPipeline` -> `fastvideo/pipelines/basic/wan/wan_pipeline.py`
|
||||
- `WanTransformer3DModel` -> `fastvideo/models/dits/wanvideo.py`
|
||||
- `AutoencoderKLWan` -> `fastvideo/models/vaes/wanvae.py`
|
||||
- `WanTransformer3DModel` -> `fastvideo/models/wan/transformer.py`
|
||||
- `WanVideoConfig` -> `fastvideo/models/wan/config.py`
|
||||
- `AutoencoderKLWan` -> `fastvideo/models/wan/vae.py`
|
||||
- `WanVAEConfig` -> `fastvideo/models/wan/vae_config.py`
|
||||
- `UMT5EncoderModel` -> `fastvideo/models/encoders/t5.py`
|
||||
- `T5TokenizerFast` -> loaded via HF in `fastvideo/models/loader/`
|
||||
- `UniPCMultistepScheduler` -> loaded via Diffusers scheduler utilities
|
||||
- Pipeline defaults -> `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults -> `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
- Variant definitions -> `fastvideo/models/wan/definition.py`
|
||||
- Pipeline defaults -> `fastvideo/models/wan/pipeline_config.py`
|
||||
- Sampling defaults -> `fastvideo/pipelines/basic/wan/presets.py`
|
||||
|
||||
## Pipeline system
|
||||
|
||||
@@ -157,8 +179,8 @@ How this maps to FastVideo:
|
||||
|
||||
## Model components
|
||||
|
||||
- DiT models: `fastvideo/models/dits/`
|
||||
- VAEs: `fastvideo/models/vaes/`
|
||||
- DiT models: `fastvideo/models/dits/`; dense Wan: `fastvideo/models/wan/`
|
||||
- VAEs: `fastvideo/models/vaes/`; Wan: `fastvideo/models/wan/vae.py`
|
||||
- Text/image encoders: `fastvideo/models/encoders/`
|
||||
- Schedulers: `fastvideo/models/schedulers/`
|
||||
- Upsamplers: `fastvideo/models/upsamplers/`
|
||||
|
||||
@@ -25,11 +25,13 @@ requests. HTTP handling and job polling remain asynchronous.
|
||||
| `GET` | `/v1/videos/{id}/content` | Download a completed MP4 |
|
||||
| `DELETE` | `/v1/videos/{id}` | Delete a job and its completed artifact |
|
||||
| `POST` | `/v1/images` | Generate an image |
|
||||
| `POST` | `/v1/images/generations` | OpenAI-compatible alias for image generation |
|
||||
| `POST` | `/v1/images/edits` | Generate an image from image references |
|
||||
| `GET` | `/v1/images/{id}/content` | Download a generated image |
|
||||
| `GET` | `/health` | Liveness probe |
|
||||
|
||||
`POST /v1/videos/generations` remains an alias for older FastVideo clients.
|
||||
`POST /v1/videos/generations` remains an alias for older FastVideo clients, and
|
||||
`POST /v1/images/generations` is the OpenAI Python client's image generation path.
|
||||
The OpenAI Python and JavaScript clients can create, retrieve, list, download,
|
||||
and delete video jobs. Use the [H3 server cookbook](../../cookbook/openai-api.md)
|
||||
for pinned client versions and executable examples. Download variants other
|
||||
|
||||
@@ -76,6 +76,7 @@ COOKBOOK_EVIDENCE_STATES = {
|
||||
}
|
||||
COOKBOOK_HARDWARE_EVIDENCE = {"validated", "source-configured", "estimated", "unknown"}
|
||||
COOKBOOK_HARDWARE_PLATFORMS = {"cuda", "mlx", "mps"}
|
||||
COOKBOOK_HARDWARE_DEVICES = {"spark"}
|
||||
COOKBOOK_GPU_TYPES = {"NVIDIA", "Apple Silicon"}
|
||||
COOKBOOK_HARDWARE_TEXT_FIELDS = {
|
||||
"accelerator",
|
||||
@@ -128,6 +129,12 @@ def validate_cookbook() -> None:
|
||||
raise ValueError(f"CUDA cookbook recipe needs an integer gpu_count >= 1: {recipe['id']}")
|
||||
if platform != "cuda" and gpu_count is not None:
|
||||
raise ValueError(f"Non-CUDA cookbook recipe must not use gpu_count: {recipe['id']}")
|
||||
device = hardware.get("device")
|
||||
if device is not None:
|
||||
if device not in COOKBOOK_HARDWARE_DEVICES:
|
||||
raise ValueError(f"Cookbook recipe has an unknown hardware device: {recipe['id']}: {device}")
|
||||
if platform != "cuda":
|
||||
raise ValueError(f"Cookbook hardware device requires platform cuda: {recipe['id']}: {device}")
|
||||
hardware_evidence = hardware.get("evidence")
|
||||
if hardware_evidence not in COOKBOOK_HARDWARE_EVIDENCE:
|
||||
raise ValueError(f"Cookbook recipe has unknown hardware evidence: {recipe['id']}: {hardware_evidence}\n"
|
||||
@@ -163,6 +170,31 @@ def validate_cookbook() -> None:
|
||||
or any(not isinstance(item, str) or not item.strip() for item in modes)):
|
||||
raise ValueError(f"Cookbook recipe modes must be a non-empty list of strings: {recipe['id']}")
|
||||
|
||||
knobs = recipe.get("knobs", [])
|
||||
if not isinstance(knobs, list):
|
||||
raise ValueError(f"Cookbook recipe knobs must be a list: {recipe['id']}")
|
||||
knob_keys: set[str] = set()
|
||||
for knob in knobs:
|
||||
if not isinstance(knob, dict):
|
||||
raise ValueError(f"Cookbook recipe knob must be an object: {recipe['id']}")
|
||||
required_knob = ("key", "label", "flag", "options", "default")
|
||||
missing_knob = {key for key in required_knob if key not in knob}
|
||||
if missing_knob:
|
||||
raise ValueError(f"Cookbook recipe knob is missing: {recipe['id']}: {', '.join(sorted(missing_knob))}")
|
||||
if knob["key"] in knob_keys:
|
||||
raise ValueError(f"Duplicate cookbook recipe knob key: {recipe['id']}: {knob['key']}")
|
||||
knob_keys.add(knob["key"])
|
||||
if not isinstance(knob["flag"], str) or not knob["flag"].startswith("--"):
|
||||
raise ValueError(f"Cookbook recipe knob flag must start with --: {recipe['id']}: {knob['key']}")
|
||||
options = knob["options"]
|
||||
if not isinstance(options, list) or not options:
|
||||
raise ValueError(
|
||||
f"Cookbook recipe knob options must be a non-empty list: {recipe['id']}: {knob['key']}")
|
||||
option_values = [option["value"] if isinstance(option, dict) else option for option in options]
|
||||
if knob["default"] not in option_values:
|
||||
raise ValueError(
|
||||
f"Cookbook recipe knob default must be one of its options: {recipe['id']}: {knob['key']}")
|
||||
|
||||
source = (ROOT_DIR / recipe["source"]).resolve()
|
||||
if not any(source.is_relative_to(root.resolve()) for root in COOKBOOK_SOURCE_ROOTS):
|
||||
raise ValueError(f"Cookbook source is outside an approved directory: {recipe['source']}")
|
||||
@@ -170,6 +202,19 @@ def validate_cookbook() -> None:
|
||||
raise ValueError(f"Cookbook source does not exist: {recipe['source']}")
|
||||
|
||||
source_text = source.read_text(encoding="utf-8")
|
||||
if knobs:
|
||||
# A script that shares its argument parser with a sibling module
|
||||
# (e.g. `from . import basic_fasth3`) inherits that module's
|
||||
# flags without the flag text appearing in its own source.
|
||||
knob_source_text = source_text
|
||||
for match in re.finditer(r'(?:from\s+\.\s+import|^\s*import)\s+(\w+)', source_text, flags=re.MULTILINE):
|
||||
sibling = source.parent / f"{match.group(1)}.py"
|
||||
if sibling.is_file() and sibling != source:
|
||||
knob_source_text += "\n" + sibling.read_text(encoding="utf-8")
|
||||
for knob in knobs:
|
||||
if knob["flag"] not in knob_source_text:
|
||||
raise ValueError(f"Cookbook recipe knob flag is not in its source: {recipe['id']}: {knob['flag']}")
|
||||
|
||||
# The model must be traceable to the checked-in source itself, or be
|
||||
# passed explicitly on the command line (e.g. --model-path <model> or
|
||||
# MODEL_PATH=<model>) when the source reads it from arguments/env.
|
||||
@@ -223,12 +268,22 @@ def cookbook_serving_profile(recipe: dict) -> dict:
|
||||
if type(port) is not int or not 1 <= port <= 65535:
|
||||
raise ValueError(f"Serving config needs a valid port: {recipe['id']}")
|
||||
hardware = {"platform": runtime, "evidence": "source-configured"}
|
||||
device = recipe["hardware"].get("device")
|
||||
if device:
|
||||
hardware["device"] = device
|
||||
if runtime == "cuda":
|
||||
count = generator["engine"]["num_gpus"]
|
||||
if type(count) is not int or count < 1:
|
||||
raise ValueError(f"Serving config needs a valid GPU count: {recipe['id']}")
|
||||
hardware["gpu_count"] = count
|
||||
command = f"fastvideo serve --config {serving['source']} --server.host 127.0.0.1"
|
||||
env = serving.get("env")
|
||||
if env is not None:
|
||||
if not isinstance(env, str) or not env.strip() or "\n" in env:
|
||||
raise ValueError(f"Serving env must be a single-line command prefix: {recipe['id']}")
|
||||
prefix = env.strip() + " "
|
||||
else:
|
||||
prefix = ""
|
||||
command = f"{prefix}fastvideo serve --config {serving['source']} --server.host 127.0.0.1"
|
||||
else:
|
||||
if server.get("host") != "127.0.0.1":
|
||||
raise ValueError(f"MLX cookbook server must bind to loopback: {recipe['id']}")
|
||||
|
||||
@@ -4,9 +4,10 @@
|
||||
FastVideo supports the following hardware platforms:
|
||||
|
||||
- [NVIDIA CUDA](installation/gpu.md)
|
||||
- [NVIDIA DGX Spark / GB10 (ARM64 + CUDA 13)](installation/spark.md)
|
||||
([performance & tuning](installation/spark_performance.md))
|
||||
- [Apple silicon](installation/mps.md)
|
||||
- **NVIDIA DGX Spark / GB10 (ARM64 + CUDA 13)** — [install](installation/spark.md),
|
||||
[performance](installation/spark_performance.md),
|
||||
[pair two Sparks](installation/spark_pair.md)
|
||||
- [Apple silicon (MLX)](installation/mlx.md)
|
||||
|
||||
## Quick Installation
|
||||
|
||||
@@ -14,7 +15,7 @@ FastVideo supports the following hardware platforms:
|
||||
|
||||
Use uv as the default environment manager for faster and more stable installs.
|
||||
The commands below target NVIDIA CUDA 12; use `UV_TORCH_BACKEND=cu130` on
|
||||
CUDA 13. Apple silicon users should follow the [MPS guide](installation/mps.md).
|
||||
CUDA 13. Apple silicon users should follow the [MLX install guide](installation/mlx.md).
|
||||
|
||||
```bash
|
||||
# Create and activate a new uv environment
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
# Install FastVideo with MLX
|
||||
|
||||
Install FastVideo on a Mac, then generate from the cookbook. Local video uses
|
||||
the native MLX runtime, not CUDA, and not the old PyTorch MPS demo at
|
||||
`examples/inference/basic/basic_mps.py`.
|
||||
|
||||
## Requirements
|
||||
|
||||
- macOS 14 or newer
|
||||
- Python 3.12
|
||||
- `ffmpeg` (`brew install ffmpeg`)
|
||||
|
||||
## Install
|
||||
|
||||
Cookbook commands run from a clone.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git && cd FastVideo
|
||||
uv venv --python 3.12 --seed
|
||||
source .venv/bin/activate
|
||||
brew install ffmpeg
|
||||
uv pip install -e ".[mlx]"
|
||||
```
|
||||
|
||||
Conda is optional. After you activate a Conda env, still install with
|
||||
`uv pip` as above.
|
||||
|
||||
`uv pip install "fastvideo[mlx]"` from PyPI installs the extra only. It does
|
||||
not ship the example scripts the cookbook copies.
|
||||
|
||||
## Generate a video
|
||||
|
||||
Open the cookbook. Select Apple Silicon as the runtime. Each recipe has a
|
||||
Python command. FastH3 also has a server path for the playground and the
|
||||
OpenAI Python client.
|
||||
|
||||
- [Wan recipes](../../cookbook/wan.md) for FastMetal 1.3B, 5B, and 14B
|
||||
- [MiniMax H3 recipes](../../cookbook/minimax-h3.md) for FastH3 V1 and FastH3 V2
|
||||
- [H3 server guide](../../cookbook/openai-api.md) for the playground, cURL, and SDKs
|
||||
|
||||
FastH3 is two distilled MiniMax-H3 checkpoints. V1 is the four-step launch.
|
||||
Some Hub repo names still say Preview. That name is historical. V1 is a full
|
||||
model, not a demo. V2 is the eight-step checkpoint. More forwards is why V2
|
||||
is the higher-quality FastH3.
|
||||
|
||||
Recorded shapes and evidence live in the
|
||||
[support matrix](../../inference/support_matrix.md#apple-silicon-native-runtime).
|
||||
|
||||
## Hardware
|
||||
|
||||
- FastMetal 1.3B and 5B: 16 GB unified memory and up
|
||||
- FastMetal 14B: 36 GB unified memory and up
|
||||
- FastH3 V1 and V2: validated on an M4 Max with 36 GB unified memory
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **`basic_mps.py` is the wrong path.** That script is PyTorch MPS. Use an
|
||||
Apple Silicon recipe in the cookbook.
|
||||
- **Muxing fails.** Install `ffmpeg` with Homebrew.
|
||||
- **A cookbook command cannot find a script.** Run it from the FastVideo
|
||||
clone after `uv pip install -e ".[mlx]"`.
|
||||
|
||||
If that does not match what you see, open an issue on the
|
||||
[GitHub repository](https://github.com/hao-ai-lab/FastVideo) or ask in the
|
||||
[Slack community](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ).
|
||||
@@ -1,260 +0,0 @@
|
||||
# MPS (Apple Silicon)
|
||||
|
||||
Install FastVideo on Apple Silicon and run FastMetal-QAD or FastH3 Preview.
|
||||
|
||||
Apple Silicon uses the MLX runtime. FastMetal-QAD ships ready-to-run MLX
|
||||
checkpoints; FastH3 Preview currently requires a local MLX DiT conversion.
|
||||
See the [FastMetal-QAD blog](https://haoailab.com/blogs/fastmetal/) and the
|
||||
[FastMetal collection](https://huggingface.co/collections/FastVideo/fastmetal).
|
||||
|
||||
## Requirements
|
||||
|
||||
- **OS: macOS 14 or newer**
|
||||
- **Python: 3.12.4**
|
||||
|
||||
## Set up using Python
|
||||
|
||||
### Create a new Python environment
|
||||
|
||||
#### uv
|
||||
Recommended default: use [uv](https://docs.astral.sh/uv/) for faster and more stable environment setup.
|
||||
|
||||
Please follow the [documentation](https://docs.astral.sh/uv/#getting-started) to install `uv`. After installing `uv`, create a new environment using:
|
||||
|
||||
```console
|
||||
# (Recommended) Create a new uv environment. Use `--seed` to install `pip` and `setuptools`.
|
||||
uv venv --python 3.12 --seed
|
||||
source .venv/bin/activate
|
||||
```
|
||||
|
||||
#### Conda (alternative)
|
||||
|
||||
You can also create a Python environment using [Conda](https://docs.conda.io/projects/conda/en/stable/user-guide/getting-started.html).
|
||||
|
||||
##### 1. Install Miniconda (if not already installed)
|
||||
|
||||
```bash
|
||||
wget https://repo.anaconda.com/miniconda/Miniconda3-latest-MacOSX-arm64.sh
|
||||
bash Miniconda3-latest-MacOSX-arm64.sh
|
||||
source ~/.zshrc
|
||||
```
|
||||
|
||||
##### 2. Create and activate a Conda environment for FastVideo
|
||||
|
||||
```bash
|
||||
conda create -n fastvideo python=3.12.4 -y
|
||||
conda activate fastvideo
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
|
||||
```
|
||||
brew install ffmpeg
|
||||
```
|
||||
|
||||
### Installation
|
||||
|
||||
FastMetal's native Apple Silicon runtime requires the `mlx` extra.
|
||||
|
||||
#### With uv (recommended)
|
||||
|
||||
```bash
|
||||
uv pip install "fastvideo[mlx]"
|
||||
```
|
||||
|
||||
#### With Conda environment (alternative)
|
||||
|
||||
`uv` works inside an active conda env too, so prefer `uv pip` for the actual install:
|
||||
|
||||
```bash
|
||||
uv pip install "fastvideo[mlx]"
|
||||
```
|
||||
|
||||
### Installation from Source
|
||||
|
||||
#### 1. Clone the FastVideo repository
|
||||
|
||||
```bash
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git && cd FastVideo
|
||||
```
|
||||
|
||||
#### 2. Install FastVideo
|
||||
|
||||
Basic installation:
|
||||
|
||||
```bash
|
||||
uv pip install -e ".[mlx]"
|
||||
```
|
||||
|
||||
Alternative with Conda environment:
|
||||
|
||||
```bash
|
||||
uv pip install -e ".[mlx]"
|
||||
```
|
||||
|
||||
## Run FastMetal-QAD
|
||||
|
||||
Each release is self-contained. Download one checkpoint and point both
|
||||
`--model-root` and `--mlx-checkpoint` at it (the example also auto-detects
|
||||
`mlx_dit.json` under `--model-root`).
|
||||
|
||||
| Checkpoint | Script | Mac tier |
|
||||
| --- | --- | --- |
|
||||
| [`FastVideo/FastMetal-1.3B-QAD`](https://huggingface.co/FastVideo/FastMetal-1.3B-QAD) | `mlx_wan_prompt_to_video.py` | 16 GB+ |
|
||||
| [`FastVideo/FastMetal-5B-QAD`](https://huggingface.co/FastVideo/FastMetal-5B-QAD) | `mlx_wan22_generate.py` | 16 GB+ |
|
||||
| [`FastVideo/FastMetal-14B-QAD`](https://huggingface.co/FastVideo/FastMetal-14B-QAD) | `mlx_wan_prompt_to_video.py` | 36 GB+ |
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastMetal-1.3B-QAD --local-dir ./FastMetal-1.3B-QAD
|
||||
|
||||
python examples/inference/basic/mlx_wan_prompt_to_video.py \
|
||||
--model-root ./FastMetal-1.3B-QAD \
|
||||
--mlx-checkpoint ./FastMetal-1.3B-QAD \
|
||||
--height 480 --width 832 --num-frames 81 \
|
||||
--prompt "A bird's-eye view of a misty forest valley at dawn."
|
||||
```
|
||||
|
||||
14B uses the same script. Point both flags at `./FastMetal-14B-QAD`. That repo also ships an EMA variant: keep `--model-root` at the repo root and set `--mlx-checkpoint ./FastMetal-14B-QAD/ema`.
|
||||
|
||||
Wan2.2 5B uses a different latent layout, so it has its own entrypoint:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastMetal-5B-QAD --local-dir ./FastMetal-5B-QAD
|
||||
|
||||
python examples/inference/basic/mlx_wan22_generate.py \
|
||||
--mlx-checkpoint ./FastMetal-5B-QAD \
|
||||
--text-encoder-root ./FastMetal-5B-QAD \
|
||||
--vae-root ./FastMetal-5B-QAD/vae \
|
||||
--height 704 --width 1280 --num-frames 81 \
|
||||
--prompt "A cinematic portrait with soft neon lighting and smooth camera motion."
|
||||
```
|
||||
|
||||
CUDA FastWan-QAD (`FastVideo/FastWan-QAD-1.3B`, `FastVideo/FastWan-QAD-FP8-1.3B`) is a separate NVIDIA release. The MLX examples look for FastMetal packed weights (`mlx_dit.json`).
|
||||
|
||||
`basic_mps.py` is a generic PyTorch MPS demo. For local video on Mac, use the FastMetal commands above.
|
||||
|
||||
## Run FastH3 Preview
|
||||
|
||||
FastH3 Preview uses the existing MLX runtime for text-to-video-with-audio
|
||||
(T2VA). The runtime streams the Qwen3-VL text conditioner, loads one
|
||||
heavyweight component at a time, denoises synchronized video and audio
|
||||
latents with a converted INT8, INT6, or INT4 DiT, and decodes both modalities
|
||||
with native MLX VAEs.
|
||||
|
||||
Download the FastH3 snapshot, then convert one or more DiT formats:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 \
|
||||
--local-dir ./FastH3-Preview-v0.2
|
||||
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py \
|
||||
--model-root ./FastH3-Preview-v0.2/transformer \
|
||||
--out ./FastH3-MLX \
|
||||
--formats "int6"
|
||||
```
|
||||
|
||||
Dense conversion drops the trained VSA gate projections. To keep them (INT6
|
||||
weight-only, same affine grid as the other linear matrices) write a **new**
|
||||
directory:
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py \
|
||||
--model-root ./FastH3-Preview-v0.2/transformer \
|
||||
--out ./FastH3-MLX-vsa \
|
||||
--formats "int6" \
|
||||
--include-vsa
|
||||
```
|
||||
|
||||
Do not overwrite an existing dense export such as `./FastH3-MLX/int6`.
|
||||
|
||||
Run the baseline path:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX/int6 \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is amazing.</d>" \
|
||||
--height 480 --width 832 --num-frames 124 --seed 2026 \
|
||||
--output-path ./outputs/fasth3_int6.mp4
|
||||
```
|
||||
|
||||
Add `--fast` for temporal fast mode. It denoises a shorter video sequence,
|
||||
uses MLX RIFE to restore the requested frame count, and keeps the audio
|
||||
sequence at full duration:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX/int6 \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is even faster.</d>" \
|
||||
--height 720 --width 1280 --num-frames 124 --seed 2027 \
|
||||
--fast \
|
||||
--output-path ./outputs/fasth3_int6_fast_720p.mp4
|
||||
```
|
||||
|
||||
Add `--fast-spatial` for spatial fast mode, `--fast`'s spatial twin. It
|
||||
denoises and decodes on the smallest 32px-aligned canvas covering the
|
||||
requested size divided by `--fast-spatial-scale` (a 480x832 request runs on a
|
||||
256x416 canvas), then resamples the decoded frames up to the requested size
|
||||
in pixel space. It composes with `--fast`. This is a speed/quality trade-off
|
||||
and stays off by default: the output carries the reduced canvas's detail
|
||||
budget, so it reads softer than a native-resolution render, with the unsharp
|
||||
pass countering some but not all of the difference:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX/int6 \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is fastest.</d>" \
|
||||
--height 480 --width 832 --num-frames 124 --seed 2028 \
|
||||
--fast --fast-spatial \
|
||||
--output-path ./outputs/fasth3_int6_fast_spatial.mp4
|
||||
```
|
||||
|
||||
VSA is off by default. A dense-only checkpoint (no `--include-vsa`) keeps the
|
||||
existing fused-SDPA path. After converting with `--include-vsa`, enable the
|
||||
sparse path explicitly:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
--model-root ./FastH3-Preview-v0.2 \
|
||||
--mlx-checkpoint ./FastH3-MLX-vsa/int6 \
|
||||
--vsa --vsa-sparsity 0.9 --vsa-tile-size 64 --vsa-prefix-mode exempt \
|
||||
--prompt "(S1) A presenter says <d>[English] Fast H3 is amazing.</d>" \
|
||||
--height 720 --width 1280 --num-frames 124 --seed 2026 \
|
||||
--output-path ./outputs/fasth3_int6_vsa_720p.mp4
|
||||
```
|
||||
|
||||
`--vsa-impl auto` uses the chunked gather+SDPA **reference** path.
|
||||
`--vsa-impl simd` is an opt-in SIMD-group kernel (tile 64, head dim 128) that
|
||||
falls back to reference on unsupported shapes. It is not the default.
|
||||
`--vsa-impl reference` is the same as `auto`.
|
||||
|
||||
!!! note "Current MLX scope"
|
||||
This source runtime supports T2VA, temporal `--fast`, spatial
|
||||
`--fast-spatial`, and opt-in VSA. FL2VA, Ref2VA, two-pass refinement, and
|
||||
`VideoGenerator` registry dispatch are not wired yet. INT8/INT6/INT4
|
||||
quantization is **weight-only**; VSA attention Q/K/V stay BF16. Old dense
|
||||
MLX checkpoints remain valid for dense inference and raise a reconvert
|
||||
error if `--vsa` is set. The checkpoint uses the MiniMax H3
|
||||
Community License; review the model card before use or redistribution.
|
||||
|
||||
## Development Environment Setup
|
||||
|
||||
If you're planning to contribute to FastVideo please see the following page:
|
||||
[Contributor Guide](../../contributing/overview.md)
|
||||
|
||||
## Hardware Requirements
|
||||
|
||||
- **1.3B / 5B:** 16 GB unified memory and up (M1 and later)
|
||||
- **14B:** 36 GB unified memory and up
|
||||
- **FastH3 Preview:** validated on an M4 Max with 36 GB unified memory; use one
|
||||
converted DiT format at a time and leave substantial free disk space for the
|
||||
source snapshot plus the converted checkpoint
|
||||
- Fanless 13-inch MacBook Air can run 1.3B and 5B at the same resolutions
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
If you encounter any issues during installation, please open an issue on our [GitHub repository](https://github.com/hao-ai-lab/FastVideo).
|
||||
|
||||
You can also join our [Slack community](https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ) for additional support.
|
||||
@@ -144,6 +144,12 @@ for which models are practical on the GB10, what makes them faster, and what
|
||||
won't help on this hardware (and why) — so you don't spend a night tuning knobs
|
||||
that can't move here.
|
||||
|
||||
Two Sparks with QSFP cables: [Pair two NVIDIA DGX Sparks](spark_pair.md) for
|
||||
one FastH3 clip across both GPUs (`sp_size=2` over Ray). Copy-paste commands
|
||||
for one or two Sparks also live on the
|
||||
[MiniMax H3 cookbook](../../cookbook/minimax-h3.md): pick FastH3 V1,
|
||||
then NVIDIA DGX Spark, then 1 Spark or 2 Sparks.
|
||||
|
||||
## Development Environment Setup
|
||||
|
||||
If you're planning to contribute to FastVideo please see the
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
# Pair two NVIDIA DGX Sparks
|
||||
|
||||
One GB10 is 128 GB of unified LPDDR5X. FastH3 still fits on a single Spark with
|
||||
[`lazy_module_load`](../../inference/offloading.md) (auto on GB10; sequential
|
||||
load stands down when lazy owns deferral). Two boxes connected
|
||||
by the QSFP ConnectX-7 cables can run **one clip faster** and can hold a
|
||||
**longer clip** (up to the FastH3 15 s cap).
|
||||
|
||||
Copy-paste commands for both counts are on the
|
||||
[MiniMax H3 cookbook](../../cookbook/minimax-h3.md): FastH3 V1 → NVIDIA DGX
|
||||
Spark → 1 Spark or 2 Sparks.
|
||||
|
||||
This is FastVideo sequence parallel (`sp_size=2`) over Ray, not a third-party
|
||||
xDiT vendor. Do not install xDiT for this path.
|
||||
|
||||
## What two Sparks buy you
|
||||
|
||||
| Goal | How | Use two Sparks? |
|
||||
|---|---|---|
|
||||
| Two independent videos at once | One process per box, `num_gpus=1` | Throughput only. Each clip still takes the 1-GPU time for that size. |
|
||||
| One clip, faster | Ray + `sp_size=2` + parallel VAE | **Yes.** One 768×1344×124 recipe was 292 s vs 374 s on one GB10. |
|
||||
| One clip, longer | Same, more frames | **Yes.** 345 frames (~14.4 s at 24 fps) finished in 587 s at 768×1344. |
|
||||
|
||||
Sequence parallel **replicates** the DiT (~66 GiB per node). Lazy module load
|
||||
is still required on each box. FSDP would shard weights;
|
||||
it is untested on this fabric and is likely slower because every layer gathers
|
||||
over ~21 GB/s RoCE.
|
||||
|
||||
## Requirements
|
||||
|
||||
- Two DGX Sparks with FastVideo [installed](spark.md) (CUDA 13, `aarch64`).
|
||||
- The QSFP cables that ship with a dual-Spark kit, **ACTIVE** at 200 Gb/s:
|
||||
`ibstat` should show the ConnectX-7 ports `LinkUp`.
|
||||
- The same FastH3 snapshot on **both** NVMes. Copy the Hugging Face cache over
|
||||
QSFP; do not download 100+ GB twice over Wi-Fi.
|
||||
- Ray in the FastVideo venv (`uv pip install ray` if it is not already there).
|
||||
|
||||
Each Spark has **one** GPU. `num_gpus=2` therefore means two nodes, which is
|
||||
why the executor must be Ray (`mp` only works inside one process tree).
|
||||
|
||||
## 1. Put IPv4 on the QSFP NICs
|
||||
|
||||
The RoCE links often come up with no IPv4. Wi-Fi (`192.168.1.x`) is fine for
|
||||
SSH and must stay the default route. NCCL and Ray must **not** use it.
|
||||
|
||||
Pick a /24 that does not collide with your LAN. Example:
|
||||
|
||||
| Node | QSFP IPv4 | Interface (typical) |
|
||||
|---|---|---|
|
||||
| Spark A | `192.168.23.1/24` | `enp1s0f1np1` |
|
||||
| Spark B (Ray head) | `192.168.23.2/24` | `enp1s0f1np1` |
|
||||
|
||||
Confirm names with `ibdev2netdev` and `ip -br link`. Then, as root, on each
|
||||
box (NetworkManager likes to steal the NIC; unmanaged is enough for a session):
|
||||
|
||||
```bash
|
||||
sudo nmcli device set enp1s0f1np1 managed no
|
||||
sudo ip addr replace 192.168.23.1/24 dev enp1s0f1np1 # .2 on the other box
|
||||
sudo ip link set enp1s0f1np1 mtu 9000 up
|
||||
```
|
||||
|
||||
These addresses do **not** survive reboot. Ping across the cable before
|
||||
continuing: `ping -c 3 -I enp1s0f1np1 192.168.23.2`.
|
||||
|
||||
A healthy fabric on this hardware looks like:
|
||||
|
||||
- TCP iperf (jumbo 9000): ~40 Gb/s
|
||||
- NCCL allreduce 1 GiB × 10: ~21 GB/s busbw (NVIDIA's dual-Spark figure is ~21.7)
|
||||
|
||||
## 2. Start a two-node Ray cluster on the cable
|
||||
|
||||
On **both** nodes, from the FastVideo repo, with the venv active:
|
||||
|
||||
```bash
|
||||
source examples/inference/optimizations/spark_pair_env.sh
|
||||
```
|
||||
|
||||
That script pins NCCL and Gloo to the QSFP NIC/HCA, disables NVLink-style P2P
|
||||
(there is none between boxes), and turns off Ray's memory monitor. The monitor
|
||||
treats GB10 unified RSS during a 14-shard DiT load as a runaway and SIGTERMs
|
||||
the worker around shard 11/14. Override `NCCL_SOCKET_IFNAME` /
|
||||
`GLOO_SOCKET_IFNAME` if `ibdev2netdev` shows a different name.
|
||||
|
||||
Cap Ray's object store. The default (~30% of 128 GB) leaves too little room
|
||||
for the DiT:
|
||||
|
||||
```bash
|
||||
# Spark B — head
|
||||
export FASTVIDEO_HOST_IP=192.168.23.2
|
||||
ray start --head --node-ip-address=192.168.23.2 --port=6379 --num-gpus=1 \
|
||||
--disable-usage-stats --object-store-memory=2147483648 --memory=4294967296
|
||||
|
||||
# Spark A — worker
|
||||
export FASTVIDEO_HOST_IP=192.168.23.1
|
||||
ray start --address=192.168.23.2:6379 --node-ip-address=192.168.23.1 --num-gpus=1 \
|
||||
--disable-usage-stats --object-store-memory=2147483648 --memory=4294967296
|
||||
```
|
||||
|
||||
`FASTVIDEO_HOST_IP` **must** match `--node-ip-address`. If you omit it, Ray
|
||||
advertises the Wi-Fi address, FastVideo builds a placement group for
|
||||
`node:192.168.1.x`, and the QSFP workers never match.
|
||||
|
||||
Check `ray status` on the head: `0.0/2.0 GPU` idle.
|
||||
|
||||
## 3. Generate one FastH3 clip on both GPUs
|
||||
|
||||
Run the driver on the **head**, same venv, same QSFP IP.
|
||||
|
||||
`basic_fasth3.py` defaults target a four-GPU GB200 profile: 768×1344, `sm100a`
|
||||
VSA, FA4, four GPUs. On Sparks you must override the kernel flags. Height,
|
||||
width, frames, steps, seed, and prompt are yours. Change them. Legal
|
||||
`num_frames` values are `17n+5`, capped at 362.
|
||||
|
||||
GB10 has no FA4 / sm_100a VSA kernel, so `--vsa-kernel triton --no-fa4` stays
|
||||
required on this box. `--execution-backend ray` is optional when `RAY_ADDRESS`
|
||||
is already set.
|
||||
|
||||
`--warmup --repeats 3` prints a median of three `generate()` calls after an
|
||||
excluded warmup. Sequential load reloads Qwen for each later request, so that
|
||||
protocol works. For a single cold process, pass `--no-warmup --repeats 1`.
|
||||
|
||||
The command below is one example, not a required recipe:
|
||||
|
||||
```bash
|
||||
source examples/inference/optimizations/spark_pair_env.sh
|
||||
export RAY_ADDRESS=192.168.23.2:6379
|
||||
export FASTVIDEO_HOST_IP=192.168.23.2
|
||||
|
||||
python examples/inference/basic/basic_fasth3.py \
|
||||
--model-path FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree \
|
||||
--num-gpus 2 --execution-backend ray \
|
||||
--vsa-kernel triton --no-fa4 \
|
||||
--warmup --repeats 3 --parallel-vae \
|
||||
--height 768 --width 1344 --num-frames 124 --steps 5 \
|
||||
--seed 2026 \
|
||||
--prompt "A wide cinematic shot of an alpine meadow at sunrise, pale pink mountain peaks above a blue valley filled with thin morning mist." \
|
||||
--output outputs/fasth3_spark_pair
|
||||
```
|
||||
|
||||
Config-first equivalent. Edit the YAML the same way, `request.sampling` is not
|
||||
locked:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 \
|
||||
FASTVIDEO_VAE_PARALLEL_DECODE=1 FASTVIDEO_STAGE_LOGGING=1 \
|
||||
fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yaml
|
||||
```
|
||||
|
||||
Stop the cluster when you are done: `ray stop` on both nodes.
|
||||
|
||||
## FastH3 frame counts
|
||||
|
||||
H3 is 24 fps. Legal `num_frames` values are `17n+5`. The pipeline rejects
|
||||
clips longer than **15 s**. The longest legal length is **362 frames**
|
||||
(15.083 s). 360 frames aligns to 362 and is accepted.
|
||||
|
||||
## Measured on two GB10s (2026-08-31)
|
||||
|
||||
These rows are full H3 VAE decode, Triton VSA, sequential + lazy load (auto on
|
||||
GB10), parallel VAE. They are not a required size. Denoise times include
|
||||
deferred DiT load (~35 s on the first generate).
|
||||
|
||||
Cold process, `--no-warmup --repeats 1`, alpine prompt, 768×1344, 5 sigma
|
||||
points (4 DiT forwards):
|
||||
|
||||
| Run | GPUs | Frames | E2E | Denoise | VAE decode |
|
||||
|---|---:|---:|---:|---:|---:|
|
||||
| One Spark | 1 | 124 | 374–393 s | 180–188 s | 151–156 s |
|
||||
| Two Sparks, SP=2 | 2 | 124 | **292 s** | **122 s** | **102 s** |
|
||||
| Two Sparks, SP=2 | 2 | 345 | **587 s** | **351 s** | **173 s** |
|
||||
|
||||
Warmup excluded, `--warmup --repeats 3` median, 512×896, 5 sigma points, full
|
||||
VAE, same 4-step schedule:
|
||||
|
||||
| Run | GPUs | Frames | Median E2E | Median denoise |
|
||||
|---|---:|---:|---:|---:|
|
||||
| One Spark | 1 | 124 | **251.4 s** | 94.2 s |
|
||||
| Two Sparks, SP=2 | 2 | 124 | **215.2 s** | 72.4 s |
|
||||
|
||||
Those medians used `--height` / `--width` / `--num-frames` as CLI flags. Swap
|
||||
them. Native 480p on this model is 480×832, 124 frames. The 15 s cap is 362
|
||||
frames.
|
||||
|
||||
The first VAE decode still pays `torch.compile`. Later `generate()` calls in
|
||||
the same workers are cheaper. GB10 regional DiT compile stays off because the
|
||||
sm_100a VSA kernel is not on this chip, so denoise is slower than a GB200
|
||||
`sm100a` run at the same geometry.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
| Symptom | Fix |
|
||||
|---|---|
|
||||
| Placement group waits forever / `node:192.168.1.x` | Set `FASTVIDEO_HOST_IP` to the QSFP address on **every** `ray start` **and** on the driver. |
|
||||
| `RayDistributedExecutor` TypeError / abstract `set_log_queue` | Use a FastVideo build that implements those methods on the Ray executor (this page). |
|
||||
| Worker SIGTERM during DiT shard 11/14 | `RAY_memory_monitor_refresh_ms=0` **before** `ray start`. Do not leave Ray's default 30% object store. |
|
||||
| NCCL hangs or uses Wi-Fi | `source spark_pair_env.sh`. Confirm `NCCL_SOCKET_IFNAME` is the QSFP NIC. |
|
||||
| Gloo `connectFullMesh` / `remote=[127.0.0.1]` | Two 1-GPU nodes must not use loopback as the Gloo store. Source `spark_pair_env.sh` so `GLOO_SOCKET_IFNAME` is the QSFP NIC on **each** box. FastVideo no longer copies that NIC name from the driver onto workers. |
|
||||
| Second `generate()` crashes `NoneType.parameters` | Sequential load used to drop the text encoder without reloading it. This branch reloads Qwen for later requests so `--warmup --repeats N` works. |
|
||||
| OOM / `earlyoom` prefers Python | Lazy module load must stay on (do not pass `--no-lazy-module-load`). Peak GPU during 345-frame denoise is ~90 GiB/node. |
|
||||
| `num_gpus=2` on one Spark | Each Spark has one GPU. Use Ray across two nodes, or `num_gpus=1` on one box. |
|
||||
|
||||
## What we are not claiming
|
||||
|
||||
- **Throughput of many clips.** Two independent 1-GPU jobs still win if you
|
||||
want two videos, not one faster video.
|
||||
- **xDiT PipeFusion / CFG-parallel.** FastH3 is 4-step and has no CFG.
|
||||
- **FSDP or tensor parallel as a speedup** on this 21 GB/s link.
|
||||
- **Persistent networking.** The example IPs are session `ip addr replace`.
|
||||
|
||||
More GPUs are legal while `num_attention_heads` (56 on FastH3) is divisible by
|
||||
`sp_size`. Four Sparks would need a four-node fabric that this bring-up did
|
||||
not exercise.
|
||||
@@ -161,15 +161,29 @@ is power-cycled. To avoid it:
|
||||
on: "CPU" offload uses the same unified RAM. Multi-GPU FSDP sharding remains
|
||||
available because it partitions weights without parking them in a separate
|
||||
host pool.
|
||||
- **MiniMax H3 / FastH3** still needs sequential loading on one GB10. The Qwen3-VL
|
||||
- **MiniMax H3 / FastH3** still needs deferred loading on one GB10. The Qwen3-VL
|
||||
conditioner is tens of gigabytes of BF16. If the DiT and VAEs load while that
|
||||
encoder is still resident, the process is a typical `earlyoom` kill (Python is
|
||||
preferred). `h3_sequential_load` defaults to auto and turns this split on for
|
||||
unified-memory devices. Do not pass `--no-h3-sequential-load` here. Force
|
||||
`--h3-sequential-load` only if auto-detect misses the device. The CUDA pipeline
|
||||
encodes first, releases the encoder, then loads DiT and VAEs onto the
|
||||
accelerator (`to_cpu` follows `cpu_offload`, which is off here). See
|
||||
[Offloading](../../inference/offloading.md).
|
||||
preferred). On unified memory, `lazy_module_load` auto-enables and owns that
|
||||
split (encoder, then DiT, then VAE; DiT can drop before decode). Sequential
|
||||
load is the H3-only fallback when lazy is off; do not pass
|
||||
`--no-lazy-module-load` here. Geometry scalars come from checkpoint
|
||||
`config.json`, not live weights. See [Offloading](../../inference/offloading.md).
|
||||
- **FastH3 TAEH3** (`--video-decode-backend taeh3`) is an opt-in preview decoder.
|
||||
T2VA never materializes the 9.7 GiB video VAE (DiT still loads after Qwen via
|
||||
sequential start). On this box, alpine 768×1344×124 decoded in **2.4 s** versus
|
||||
**68 s** for the full VAE, and one T2VA generation finished in **224 s**
|
||||
end-to-end. Reconstruction is approximate, not lossless. FL2VA/Ref2VA still
|
||||
need the full VAE to encode references.
|
||||
- **Two Sparks, one clip.** Sequence parallel (`sp_size=2`) over the QSFP RoCE
|
||||
link ran one 768×1344×124 FastH3 recipe in **292 s** vs **374–393 s** on
|
||||
one GB10, and a 345-frame (~14.4 s) clip in **587 s**. Other heights, widths,
|
||||
and frame counts are valid. Weights stay replicated, so lazy module load
|
||||
(auto on GB10) is still required on each box. Bring-up and knobs:
|
||||
[Pair two NVIDIA DGX Sparks](spark_pair.md).
|
||||
- A worker's SIGTERM log and traceback show where it was interrupted, not why it
|
||||
was selected; confirm the cause in the `earlyoom` service or system logs. A
|
||||
later SIGKILL or kernel OOM kill cannot be caught and reported by Python.
|
||||
|
||||
## Gotchas specific to the GB10
|
||||
|
||||
@@ -187,10 +201,11 @@ A few things that surprise people on this box (beyond the memory notes above):
|
||||
build recent enough to include its `transformers`-compatibility handling before
|
||||
running it.
|
||||
- **MiniMax H3 worker init can look healthy and still die on the first generate**
|
||||
if sequential load is off (`--no-h3-sequential-load`, or auto-off on a
|
||||
misclassified device) and encoder, VAE, and DiT load together. Confirm the log
|
||||
contains `Released MiniMax-H3 text encoder after conditioning` before
|
||||
`Loading MiniMax-H3 denoise modules`.
|
||||
if deferred loading is off (`--no-lazy-module-load` and sequential also off)
|
||||
and encoder, VAE, and DiT load together. On GB10 the log should show
|
||||
`lazy_module_load owns deferral` (or, if lazy is off, sequential
|
||||
`Released MiniMax-H3 text encoder after conditioning` before
|
||||
`Loading MiniMax-H3 denoise modules`).
|
||||
|
||||
## Reproduce these numbers
|
||||
|
||||
|
||||
@@ -23,7 +23,7 @@ to FastVideo model classes. Two discovery mechanisms:
|
||||
`fastvideo/models/` and parses each `.py` file's AST looking for an
|
||||
`EntryClass` variable assignment. Discovered models take priority over
|
||||
hardcoded entries. For example,
|
||||
`fastvideo/models/dits/wanvideo.py` exports
|
||||
`fastvideo/models/wan/transformer.py` exports
|
||||
`EntryClass = WanTransformer3DModel`.
|
||||
|
||||
Both feed into a unified `_FAST_VIDEO_MODELS` dict, which populates the
|
||||
@@ -86,7 +86,7 @@ from `model_index.json`.
|
||||
|
||||
```
|
||||
PipelineConfig (fastvideo/configs/pipelines/base.py)
|
||||
├── WanT2V480PConfig (fastvideo/configs/pipelines/wan.py)
|
||||
├── WanT2V480PConfig (fastvideo/models/wan/pipeline_config.py)
|
||||
│ ├── WanT2V720PConfig
|
||||
│ └── WanI2V480PConfig
|
||||
├── HunyuanConfig (fastvideo/configs/pipelines/hunyuan.py)
|
||||
@@ -107,6 +107,11 @@ Model-specific subclasses override defaults. For example,
|
||||
`WanT2V480PConfig` sets `flow_shift=3.0` and uses `WanVideoConfig` as
|
||||
its DiT config.
|
||||
|
||||
Wan's `models/wan/definition.py` links each registered variant to its pipeline
|
||||
config and sampling preset. The shared registry consumes these definitions
|
||||
without changing detector precedence or checkpoint/override-based pipeline
|
||||
selection. `configs/pipelines/wan.py` remains a compatibility import.
|
||||
|
||||
### ModelConfig / ArchConfig (`fastvideo/configs/models/base.py`)
|
||||
|
||||
`ModelConfig` wraps an `ArchConfig` using `__getattr__` proxy — attribute
|
||||
@@ -256,6 +261,16 @@ Specialized variants: `CausalDenoisingStage`, `LTX2DenoisingStage`,
|
||||
`SRDenoisingStage`, `LTX2AudioDecodingStage`, `SD35ConditioningStage`,
|
||||
`LTX2TextEncodingStage`, `LTX2LatentPreparationStage`.
|
||||
|
||||
Wan owns its sampling recipes under `basic/wan/stages/`. `WanDenoisingStage`
|
||||
specializes input packing, expert selection, timesteps, and first-frame
|
||||
restoration around the shared dense loop. `WanFirstFrameEncodingStage`
|
||||
produces normalized `ForwardBatch.first_frame_latent` before sampling; the
|
||||
sampler no longer executes a VAE. Dense DMD and the two causal samplers have
|
||||
family-local implementations and explicit scheduler ownership. Standard and
|
||||
DMD causal sampling share cache allocation, not their sampling algorithm.
|
||||
Legacy imports from `stages/` remain compatibility aliases. Sampling invariants
|
||||
also live beside the code in `fastvideo/pipelines/basic/wan/AGENTS.md`.
|
||||
|
||||
### Verification System (`fastvideo/pipelines/stages/validators.py`)
|
||||
|
||||
`StageValidators` (aliased as `V`) provides static validators:
|
||||
|
||||
@@ -12,6 +12,22 @@ generator = VideoGenerator.from_pretrained(
|
||||
)
|
||||
```
|
||||
|
||||
One node uses the multiprocessing executor (`execution_backend: mp`, the
|
||||
default). Two machines — for example two DGX Sparks, one GPU each — need Ray:
|
||||
|
||||
```yaml
|
||||
generator:
|
||||
engine:
|
||||
num_gpus: 2
|
||||
execution_backend: ray
|
||||
parallelism:
|
||||
sp_size: 2
|
||||
```
|
||||
|
||||
Set `RAY_ADDRESS` and `FASTVIDEO_HOST_IP` to the interconnect IPs, not Wi-Fi.
|
||||
The FastH3 example selects Ray automatically when `RAY_ADDRESS` is set. Full
|
||||
bring-up: [Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
## Customizing Generation
|
||||
|
||||
- `PipelineConfig`: Initialization time parameters
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
# FastH3 distilled checkpoint schedules
|
||||
|
||||
Base MiniMax-H3 still uses the scheduler shifts in its checkpoint (video 12,
|
||||
audio 3), BF16 text encoding, and the existing uniform schedule. The default
|
||||
`basic_fasth3.py` example still targets the four-forward preview. Selecting a
|
||||
shift-10 eight-forward checkpoint is an explicit choice of model and recipe;
|
||||
it does not change either default or enable NVFP4.
|
||||
|
||||
## Eight-forward T2AV recipe
|
||||
|
||||
The public checkpoint is
|
||||
[`FastVideo/FastVideo-FastH3-8-Step-V2`](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2)
|
||||
(MiniMax H3 Community License), trained with video/audio shifts 10/3, VSA
|
||||
sparsity 0.8, 64-token tiles, and the DMD rungs
|
||||
`[999, 874, 749, 624, 500, 375, 250, 125]`. `basic_fasth3_8step.py` pins that
|
||||
checkpoint and recipe as defaults; it shares the preview example's CLI, so every
|
||||
other flag works unchanged:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_fasth3_8step.py \
|
||||
--prompt 'A slow cinematic drone shot glides over a coastal town; gulls call over the harbor.' \
|
||||
--num-gpus 4 --vsa-kernel sm100a \
|
||||
--profile strict --no-inference-torch-compile --no-compile-vae \
|
||||
--height 768 --width 1344 --num-frames 124 \
|
||||
--output outputs/fasth3-8step
|
||||
```
|
||||
|
||||
Pass `--model-path` to use a local snapshot of the full export (not just its
|
||||
`transformer` subdirectory). `--steps` is the number of sigma-grid points,
|
||||
including the terminal zero; nine points run exactly eight transformer
|
||||
forwards, and the script rejects any other value because the checkpoint's
|
||||
ladder has eight rungs. The rungs are unshifted noise levels on the 1000-step training
|
||||
clock; each scheduler applies its own shift once, and the transformer receives
|
||||
H3 clean-time values (`1 - sigma`). A uniform nine-point grid is not a substitute
|
||||
for those rungs.
|
||||
|
||||
Compilation and H3 fusions are disabled above to establish an eager reference;
|
||||
they can be evaluated separately. On hardware without the sm100a extension,
|
||||
use `--vsa-kernel triton`; compare outputs and performance before adopting that
|
||||
backend. This recipe is T2AV-only, not a distilled `transformer_ref` model.
|
||||
|
||||
## Export metadata and validation
|
||||
|
||||
The export's `fastvideo_inference.json` supplies the trained ladder. The
|
||||
schedule fields of `fasth3-inference-contract-v1` are:
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": "fasth3-inference-contract-v1",
|
||||
"dmd_denoising_steps": [999, 874, 749, 624, 500, 375, 250, 125],
|
||||
"num_inference_steps": 9,
|
||||
"transformer_forwards": 8,
|
||||
"video_scheduler_shift": 10.0,
|
||||
"audio_scheduler_shift": 3.0
|
||||
}
|
||||
```
|
||||
|
||||
The loader keeps this file when downloading the selected H3 components from
|
||||
Hugging Face. It checks that the two declared shifts agree with
|
||||
`scheduler/scheduler_config.json` and `audio_scheduler/scheduler_config.json`.
|
||||
Missing/invalid rungs, inconsistent counts, or an explicit conflicting ladder
|
||||
are errors. The denoiser rejects a request with the wrong number of grid points.
|
||||
The metadata does not silently change request dimensions, step count, attention
|
||||
backend, sparsity, precision, or offload/compile settings: set those explicitly
|
||||
as above.
|
||||
|
||||
For exports without this sidecar, an explicit ladder is supported via
|
||||
`MiniMaxH3PipelineConfig.dmd_denoising_steps`, or through the typed API's
|
||||
`PipelineSelection(experimental={"dmd_denoising_steps": [...]})`. The shifts
|
||||
still come from the checkpoint scheduler configs. Keep generic `flow_shift`
|
||||
unset: H3 has separate video and audio shifts, not one shared shift.
|
||||
|
||||
This documents execution support for the published checkpoint. It is not a
|
||||
quality claim: compare video/audio output against base MiniMax-H3 on your own
|
||||
prompts before adopting it.
|
||||
|
||||
## Apple Silicon
|
||||
|
||||
`mlx_fasth3.py` stays on FastH3 V1 and its uniform AdaLN cache.
|
||||
`mlx_fasth3_8step.py` is the eight-forward MLX recipe for FastH3 V2. It
|
||||
reads the same `fastvideo_inference.json` rungs and shifts, and it expects an
|
||||
MLX DiT whose AdaLN cache was converted from that contract. Reuse the V1
|
||||
snapshot's VAE, audio VAE, text encoder, and tokenizer; only the DiT and the
|
||||
sidecar change. Rank-reduced AdaLN checkpoints are unchanged and are not
|
||||
produced by the MLX converter.
|
||||
|
||||
Install is in the
|
||||
[MLX install guide](../getting_started/installation/mlx.md). Generation and
|
||||
serving are in the [MiniMax H3 cookbook](../cookbook/minimax-h3.md) and the
|
||||
[H3 server guide](../cookbook/openai-api.md).
|
||||
@@ -5,10 +5,10 @@ This page contains step-by-step instructions to get you quickly started with vid
|
||||
## Requirements
|
||||
|
||||
- **OS**: Linux (tested on Ubuntu 22.04+), or macOS on Apple silicon via the
|
||||
[MPS installation guide](../getting_started/installation/mps.md)
|
||||
[MLX install guide](../getting_started/installation/mlx.md)
|
||||
- **Python**: 3.10-3.12
|
||||
- **CUDA**: 12.6 or 13.0 (NVIDIA GPUs)
|
||||
- **GPU**: At least one NVIDIA GPU, or an Apple silicon chip with MPS
|
||||
- **GPU**: At least one NVIDIA GPU, or an Apple silicon chip with the MLX runtime
|
||||
|
||||
## Installation
|
||||
|
||||
|
||||
@@ -10,7 +10,9 @@ that tradeoff. It is not a lossless acceleration of the full VAE.
|
||||
|
||||
## Generate a video
|
||||
|
||||
Use your existing MLX FastH3 environment and converted checkpoint:
|
||||
Use your existing MLX FastH3 environment and converted checkpoint. The same
|
||||
`--video-decode-backend taeh3` flag works on `mlx_fasth3.py` (V1) and
|
||||
`mlx_fasth3_8step.py` (V2):
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/mlx_fasth3.py \
|
||||
|
||||
@@ -12,6 +12,7 @@ text_encoder_cpu_offload: bool = True
|
||||
image_encoder_cpu_offload: bool = True
|
||||
vae_cpu_offload: bool = True
|
||||
pin_cpu_memory: bool = True
|
||||
lazy_module_load: bool | None = None
|
||||
```
|
||||
|
||||
On unified-memory accelerators such as NVIDIA GB10 and Apple silicon, FastVideo
|
||||
@@ -21,18 +22,25 @@ pool there, so offload adds transfers and duplicate residency instead of freeing
|
||||
memory. CUDA FSDP sharding remains enabled when requested; MPS continues to
|
||||
disable FSDP. `pin_cpu_memory` is not an offload mode and is left unchanged.
|
||||
|
||||
MiniMax H3 CUDA inference can use a second lever that does not copy weights to a
|
||||
host pool: load the Qwen3-VL text encoder, run conditioning, then release that
|
||||
encoder before loading the DiT and video/audio VAEs. `h3_sequential_load`
|
||||
defaults to auto (`None`): on for unified-memory devices such as GB10, off on
|
||||
discrete GPUs. Pass `--h3-sequential-load` to force it, or
|
||||
`--no-h3-sequential-load` to keep the encoder resident for later `generate()`
|
||||
calls on the same worker. Sequential load currently cannot re-encode a new prompt
|
||||
on that worker; start a new generator until prompt-cache reload exists. The MLX
|
||||
FastH3 runtime always uses this phase order. When host offload is off, DiT
|
||||
safetensors are read onto the accelerator instead of CPU-then-copy.
|
||||
Input-preparation geometry (spatial ratio, latent channels, audio sample rate)
|
||||
comes from the VAE arch configs until those weights load.
|
||||
MiniMax H3 CUDA inference can use two levers that do not copy weights to a host
|
||||
pool. `lazy_module_load` is the general path: each opted-in component loads on
|
||||
first use and is freed after its last stage, so a later `generate()` reloads
|
||||
from disk in-process and the DiT can drop before VAE decode. On GB10 it
|
||||
auto-enables and owns deferral. `h3_sequential_load` is the H3-only fallback
|
||||
when lazy is off: load Qwen3-VL, run conditioning, release that encoder, then
|
||||
load the DiT and VAEs. When both would arm, sequential stands down so VAE
|
||||
`torch.compile` can attach to the lazy proxy. Input preparation and unpatchify
|
||||
read geometry from checkpoint `config.json` (VAE spatial ratio / latent
|
||||
channels, DiT patch size) so those stages do not materialize weights just to
|
||||
read two integers. The MLX FastH3 runtime always uses this phase order. When
|
||||
host offload is off, DiT safetensors are read onto the accelerator instead of
|
||||
CPU-then-copy. Both flags default to auto (`None`) and turn on for
|
||||
unified-memory devices such as GB10; lazy then disables sequential. Pass
|
||||
`--no-lazy-module-load` to keep every component resident (sequential may still
|
||||
auto-arm). Two-node Spark
|
||||
jobs still need this split: sequence parallel replicates the DiT on each GB10
|
||||
(~66 GiB of weights plus activations). See
|
||||
[Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
## Behavior Explanation
|
||||
|
||||
@@ -93,9 +101,12 @@ because the encoder has been released.
|
||||
|
||||
#### Usage Recommendation
|
||||
|
||||
Leave the default on Spark / DGX Spark. Force `--h3-sequential-load` only when
|
||||
you need the split on a discrete GPU. Use `--no-h3-sequential-load` when you
|
||||
need more than one prompt per worker and have enough memory to keep the encoder.
|
||||
Leave the default on Spark / DGX Spark when `lazy_module_load` is off. When
|
||||
both would arm (the GB10 auto case), lazy owns deferral and sequential stands
|
||||
down so VAE `torch.compile` can attach to the lazy proxy. Force
|
||||
`--h3-sequential-load` only when you need the split on a discrete GPU without
|
||||
lazy load. Use `--no-h3-sequential-load` when you need more than one prompt per
|
||||
worker and have enough memory to keep the encoder.
|
||||
|
||||
### `text_encoder_cpu_offload`
|
||||
|
||||
@@ -121,6 +132,52 @@ These options introduce performance overhead due to PCIe data transfer.
|
||||
|
||||
We recommend enabling these options when OOM happens.
|
||||
|
||||
### `lazy_module_load`
|
||||
|
||||
Every option above moves weights between host and device. This one changes
|
||||
whether they are in memory at all.
|
||||
|
||||
By default a pipeline loads every component before the first stage runs, so
|
||||
peak memory is the sum of all of them even though no two are needed at the same
|
||||
moment. With `lazy_module_load` enabled, each heavy component loads on first use
|
||||
and is freed once the last stage that needs it has returned, so peak memory
|
||||
becomes the largest overlapping set instead of the sum. MiniMax-H3 T2VA is
|
||||
`max(text encoder, DiT, VAE)` rather than `text encoder + DiT + VAE`, because
|
||||
the DiT is not held through VAE decode.
|
||||
|
||||
#### Performance Impact
|
||||
|
||||
A freed component is read from disk again on the next generation, so a
|
||||
multi-prompt run pays one reload per component per request. For a large text
|
||||
encoder that is tens of seconds. If pipeline-level `torch.compile` is enabled,
|
||||
the compile setup is reapplied after each reload; PyTorch can reuse its graph
|
||||
and kernel caches when the component structure and input shapes are unchanged.
|
||||
|
||||
#### Usage Recommendation
|
||||
|
||||
Enable this when a model does not fit at load time, which the CPU offload
|
||||
options above cannot help with because they act after loading. It is
|
||||
particularly relevant on unified-memory devices, where host and device draw on
|
||||
the same pool and moving weights to the host frees nothing. FastVideo
|
||||
auto-enables it there (`lazy_module_load=None`). Leave it off when the model
|
||||
already fits, or pass `--no-lazy-module-load` to keep components resident for
|
||||
later `generate()` calls.
|
||||
|
||||
This option applies to inference only. Training keeps every component resident
|
||||
and logs a warning if the flag is set.
|
||||
|
||||
Deferral is opt-in per pipeline. Releasing a component and loading it again is
|
||||
only safe when nothing outside the loader has changed it, and two common habits
|
||||
break that without raising: mutating a component after load, as LongCat does
|
||||
when it enables block-sparse attention, and reading a component's attributes
|
||||
while stages are built, as the shared denoising stage does to pick an attention
|
||||
backend. A pipeline therefore lists the components it has checked in
|
||||
`_lazy_module_names`, which is empty in the base class. MiniMax-H3 opts in. On
|
||||
a pipeline that has not, the flag is a no-op: hooks are not installed and no
|
||||
warning is logged. Sequential MiniMax-H3 (`h3_sequential_load`) reloads the
|
||||
text encoder for a later `generate()` on the same worker; you do not need to
|
||||
start a new generator.
|
||||
|
||||
## General Recommendations
|
||||
|
||||
### Single GPU Inference
|
||||
@@ -131,6 +188,12 @@ We recommend enabling `dit_layerwise_offload`. If OOM happens, also enable `imag
|
||||
|
||||
We recommend enabling `use_fsdp_inference` and disabling both `dit_layerwise_offload` and `dit_cpu_offload`. If OOM happens, consider enabling `text_encoder_cpu_offload`, `image_encoder_cpu_offload`, and `vae_cpu_offload`. If OOM still happens, consider enabling `dit_cpu_offload`.
|
||||
|
||||
### When the Model Does Not Fit at Load Time
|
||||
|
||||
The offload options only help once loading has finished. If the run dies while
|
||||
components are still being placed, or if the machine has unified memory so
|
||||
there is no separate host pool to offload into, enable `lazy_module_load`.
|
||||
|
||||
## Examples
|
||||
|
||||
### Single GPU with Layerwise Offloading
|
||||
|
||||
@@ -7,7 +7,8 @@ This page describes the various options for speeding up generation times in Fast
|
||||
Several options on this page behave differently on the GB10's unified-memory
|
||||
hardware — some give little or nothing there. See
|
||||
[DGX Spark: Performance & Tuning](../getting_started/installation/spark_performance.md)
|
||||
for what actually helps on that platform and why.
|
||||
for what actually helps on that platform and why. Two Sparks, one clip:
|
||||
[Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
|
||||
## Table of Contents
|
||||
|
||||
@@ -293,6 +294,9 @@ automatically.
|
||||
### Requirements
|
||||
|
||||
- **GPU**: sm89+ (H100, L40S, RTX 4090, or newer) for hardware FP8 compute
|
||||
- **ROCm**: CDNA4 (MI350X / MI355X, gfx950) runs the FP8 `_scaled_mm` path through
|
||||
hipBLASLt (OCP e4m3fn). MI300X (gfx942) only exposes the `fnuz` FP8 formats and
|
||||
takes the bf16 dequant fallback like a pre-sm89 GPU.
|
||||
- No additional packages required beyond the base FastVideo install
|
||||
|
||||
### Usage
|
||||
|
||||
@@ -185,13 +185,14 @@ optimizations: absence means **untested**, not incompatible.
|
||||
| MLX FastMetal T2V 1.3B | [`FastVideo/FastMetal-1.3B-QAD`](https://huggingface.co/FastVideo/FastMetal-1.3B-QAD) | 480x832, 81 frames, 3-step DMD, INT8 DiT + TAEHV decode | Apple M4 Max, 16 GB+ unified memory | Released |
|
||||
| MLX FastMetal T2V 5B | [`FastVideo/FastMetal-5B-QAD`](https://huggingface.co/FastVideo/FastMetal-5B-QAD) | 480p / 720p, 81 frames, 3-step DMD, INT8 DiT + TAEHV decode; optional `--fast`, `--fast-spatial`, `--refine`. The checked-in example is T2V. CUDA Wan2.2 TI2V 5B is the image-capable path. | Apple M4 Max, 16 GB+ unified memory | Released |
|
||||
| MLX FastMetal T2V 14B | [`FastVideo/FastMetal-14B-QAD`](https://huggingface.co/FastVideo/FastMetal-14B-QAD) | 480p / 720p, 81 frames, 3-step DMD, INT8 DiT + TAEHV decode | Apple M4 Max, 36 GB+ unified memory | Released |
|
||||
| MLX FastH3 Preview T2VA | [`FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2`](https://huggingface.co/FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2) + locally converted DiT | 480p / 720p, 124 frames, 4-step DMD2, INT8/INT6/INT4 **weight-only** DiT, native video + audio VAE; optional temporal RIFE fast mode; optional spatial fast mode; optional VSA (tile 64/256, exempt/compete) on `--include-vsa` checkpoints | Apple M4 Max, 36 GB unified memory | Source runtime; T2VA only |
|
||||
| MLX FastH3 V1 T2VA | [`FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2`](https://huggingface.co/FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2) + locally converted DiT | 480p / 720p, 124 frames, 4-step DMD2, INT8/INT6/INT4 **weight-only** DiT, native video + audio VAE; optional temporal RIFE fast mode; optional spatial fast mode; optional VSA (tile 64/256, exempt/compete) on `--include-vsa` checkpoints | Apple M4 Max, 36 GB unified memory | Source runtime; T2VA only |
|
||||
| MLX FastH3 V2 T2VA | [`FastVideo/FastVideo-FastH3-8-Step-V2`](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2) + locally converted DiT | 480p / 720p, 124 frames, 8-step DMD2, INT8 **weight-only** DiT with `--include-vsa`, trained VSA 0.8 / tile 64, native video + audio VAE; AdaLN cache from `fastvideo_inference.json` | Apple M4 Max, 36 GB unified memory | Source runtime; T2VA only |
|
||||
|
||||
Apple Silicon uses the native MLX runtime. FastMetal-QAD is the packaged Wan
|
||||
release, while FastH3 Preview currently uses a source checkout and local DiT
|
||||
conversion. CUDA FastWan-QAD (`FastVideo/FastWan-QAD-1.3B`,
|
||||
release. FastH3 V1 and FastH3 V2 use a source checkout and local DiT conversion.
|
||||
CUDA FastWan-QAD (`FastVideo/FastWan-QAD-1.3B`,
|
||||
`FastVideo/FastWan-QAD-FP8-1.3B`) is the NVIDIA release. See the
|
||||
[Apple Silicon guide](../getting_started/installation/mps.md) and the
|
||||
[MLX install guide](../getting_started/installation/mlx.md) and the
|
||||
[FastMetal-QAD blog](https://haoailab.com/blogs/fastmetal/).
|
||||
|
||||
**Note**: Wan2.2 TI2V 5B has some quality issues when performing I2V generation. We are working on fixing this issue.
|
||||
@@ -221,8 +222,11 @@ Per the installation guides:
|
||||
[GPU install guide](../getting_started/installation/gpu.md).
|
||||
- **NVIDIA DGX Spark (GB10, aarch64)** — CUDA 13, from-source kernel build; see
|
||||
the [DGX Spark install guide](../getting_started/installation/spark.md).
|
||||
- **Apple silicon** — macOS 14 or newer; FastMetal-QAD via the MLX runtime. See the
|
||||
[Apple Silicon guide](../getting_started/installation/mps.md). The older
|
||||
Two Sparks over QSFP use Ray sequence parallel; see
|
||||
[Pair two NVIDIA DGX Sparks](../getting_started/installation/spark_pair.md).
|
||||
- **Apple silicon** — macOS 14 or newer; MLX runtime for FastMetal-QAD, FastH3
|
||||
V1, and FastH3 V2. See the
|
||||
[MLX install guide](../getting_started/installation/mlx.md). The older
|
||||
[`basic_mps.py`](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_mps.py)
|
||||
demo is PyTorch MPS only.
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ python examples/inference/basic/basic.py
|
||||
### Apple Silicon (FastMetal-QAD)
|
||||
|
||||
Use the MLX runtime with FastMetal-QAD. See the
|
||||
[Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastMetal-1.3B-QAD --local-dir ./FastMetal-1.3B-QAD
|
||||
@@ -39,7 +39,7 @@ with
|
||||
|
||||
`examples/inference/basic/basic_mps.py` is the older PyTorch MPS demo.
|
||||
|
||||
FastH3 Preview T2VA also runs through the native MLX runtime. Convert the DiT
|
||||
FastH3 V1 T2VA also runs through the native MLX runtime. Convert the DiT
|
||||
to INT8, INT6, or INT4 first, then run:
|
||||
|
||||
```bash
|
||||
@@ -51,17 +51,21 @@ python examples/inference/basic/mlx_fasth3.py \
|
||||
--output-path ./outputs/fasth3_int6.mp4
|
||||
```
|
||||
|
||||
FastH3 V2 is a separate checkpoint and script. Convert with
|
||||
`--include-vsa` and run `mlx_fasth3_8step.py`. Do not reuse a V1 DiT.
|
||||
|
||||
Pass `--fast` for temporal RIFE fast mode and `--fast-spatial` for spatial
|
||||
fast mode (reduced-canvas denoise + pixel-space upsample); the two compose.
|
||||
VSA is opt-in: convert with `--include-vsa` and pass `--vsa` (see the
|
||||
[Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/)).
|
||||
This MLX entrypoint currently supports T2VA only; FL2VA, Ref2VA, and
|
||||
V1 VSA is opt-in: convert with `--include-vsa` and pass `--vsa`. V2
|
||||
turns VSA on by default. See the
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
This MLX entrypoint supports T2VA only; FL2VA, Ref2VA, and
|
||||
two-pass refinement remain follow-up work. INT6/INT8/INT4 are
|
||||
weight-only; VSA attention activations stay BF16. Dense-only checkpoints keep
|
||||
working for dense inference.
|
||||
weight-only; VSA attention activations stay BF16. Dense-only V1
|
||||
checkpoints keep working for dense inference.
|
||||
|
||||
The complete setup and conversion commands are in the
|
||||
[Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
For an example running DMD+VSA inference:
|
||||
```
|
||||
|
||||
@@ -48,6 +48,14 @@ def build_parser(description: str | None = None) -> argparse.ArgumentParser:
|
||||
# License review completes. A local snapshot can be passed here instead.
|
||||
parser.add_argument("--prompt", required=True)
|
||||
parser.add_argument("--output", default="outputs/fasth3")
|
||||
parser.add_argument("--lazy-module-load",
|
||||
action=argparse.BooleanOptionalAction,
|
||||
default=None,
|
||||
help="load each heavy component on first use and free it after the last stage that "
|
||||
"needs it, so peak memory is the largest overlapping set instead of the sum of every "
|
||||
"component. Omit for auto (on for unified-memory devices such as GB10; off on discrete "
|
||||
"GPUs). Costs a reload per generation; pass --no-lazy-module-load to keep every "
|
||||
"component resident")
|
||||
parser.add_argument("--profile",
|
||||
choices=("all", "strict"),
|
||||
default="all",
|
||||
@@ -69,6 +77,13 @@ def build_parser(description: str | None = None) -> argparse.ArgumentParser:
|
||||
default=True,
|
||||
help="run one excluded request before timing")
|
||||
parser.add_argument("--num-gpus", type=int, default=4)
|
||||
parser.add_argument(
|
||||
"--execution-backend",
|
||||
choices=("mp", "ray"),
|
||||
default=None,
|
||||
help="mp for one node; ray for a Ray cluster (two DGX Sparks). "
|
||||
"Default: ray when RAY_ADDRESS is set, otherwise mp",
|
||||
)
|
||||
parser.add_argument("--vsa-sparsity",
|
||||
type=float,
|
||||
default=0.9,
|
||||
@@ -104,6 +119,11 @@ def build_parser(description: str | None = None) -> argparse.ArgumentParser:
|
||||
default=None,
|
||||
help="encode with Qwen3-VL, release it, then load DiT/VAEs. Default auto: on for "
|
||||
"unified-memory devices (GB10), off on discrete GPUs")
|
||||
parser.add_argument("--video-decode-backend",
|
||||
choices=("h3-vae", "taeh3"),
|
||||
default="h3-vae",
|
||||
help="h3-vae is the full MiniMax VAE; taeh3 is the fast approximate preview decoder")
|
||||
parser.add_argument("--taeh3-checkpoint", default=None, help="local taeh3.safetensors; unset uses the pinned cache")
|
||||
parser.add_argument("--replicated-dit",
|
||||
action=argparse.BooleanOptionalAction,
|
||||
default=True,
|
||||
@@ -226,6 +246,12 @@ def validate_profile_dependencies(args: argparse.Namespace) -> None:
|
||||
"`cd fastvideo-kernel && ./build.sh`), or pass --vsa-kernel triton.")
|
||||
|
||||
|
||||
def _execution_backend(args: argparse.Namespace) -> str:
|
||||
if args.execution_backend is not None:
|
||||
return args.execution_backend
|
||||
return "ray" if os.environ.get("RAY_ADDRESS") else "mp"
|
||||
|
||||
|
||||
def build_generator_config(args: argparse.Namespace) -> GeneratorConfig:
|
||||
use_vsa = _uses_vsa(args)
|
||||
experimental: dict[str, object] = {
|
||||
@@ -236,6 +262,10 @@ def build_generator_config(args: argparse.Namespace) -> GeneratorConfig:
|
||||
}
|
||||
if args.h3_sequential_load is not None:
|
||||
experimental["h3_sequential_load"] = args.h3_sequential_load
|
||||
if args.video_decode_backend != "h3-vae":
|
||||
experimental["video_decode_backend"] = args.video_decode_backend
|
||||
if args.taeh3_checkpoint is not None:
|
||||
experimental["taeh3_checkpoint"] = args.taeh3_checkpoint
|
||||
if use_vsa:
|
||||
experimental.update({
|
||||
"VSA_sparsity": args.vsa_sparsity,
|
||||
@@ -252,6 +282,7 @@ def build_generator_config(args: argparse.Namespace) -> GeneratorConfig:
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=args.num_gpus,
|
||||
execution_backend=_execution_backend(args),
|
||||
use_fsdp_inference=args.num_gpus > 1 and not args.replicated_dit,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=args.num_gpus),
|
||||
offload=OffloadConfig(
|
||||
@@ -260,6 +291,7 @@ def build_generator_config(args: argparse.Namespace) -> GeneratorConfig:
|
||||
text_encoder=True,
|
||||
vae=True,
|
||||
pin_cpu_memory=args.pin_cpu_memory,
|
||||
lazy_module_load=args.lazy_module_load,
|
||||
),
|
||||
compile=CompileConfig(
|
||||
enabled=args.torch_compile,
|
||||
@@ -322,6 +354,7 @@ def run(args: argparse.Namespace) -> list[float]:
|
||||
f"Denoising contract override: {args.steps} sigma points = {args.steps - 1} DiT forwards")
|
||||
print("Profile environment: " + " ".join(f"{key}={value if value is not None else '<unset>'}"
|
||||
for key, value in environment.items()))
|
||||
print(f"Execution backend: {_execution_backend(args)}")
|
||||
|
||||
generator = VideoGenerator.from_config(build_generator_config(args))
|
||||
measured_wall_times: list[float] = []
|
||||
|
||||
@@ -0,0 +1,55 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Eight-forward video+audio generation with the FastH3 8-Step V2 checkpoint.
|
||||
|
||||
``FastVideo/FastVideo-FastH3-8-Step-V2`` is a DMD-distilled MiniMax H3 T2AV
|
||||
student trained with video/audio scheduler shifts 10/3, VSA sparsity 0.8 and
|
||||
64-token tiles, on the explicit rung ladder ``[999, 874, 749, 624, 500, 375,
|
||||
250, 125]``. Its ``fastvideo_inference.json`` sidecar carries that ladder; the
|
||||
pipeline loads it, checks the declared shifts against the checkpoint scheduler
|
||||
configs, and runs exactly eight transformer forwards from nine sigma-grid
|
||||
points. This script only sets those defaults; it does not change the
|
||||
four-forward preview example.
|
||||
|
||||
``--steps`` must stay at 9 for this checkpoint: the trained ladder has eight
|
||||
rungs, and a different grid is not a valid recipe for it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from collections.abc import Sequence
|
||||
|
||||
try:
|
||||
from . import basic_fasth3
|
||||
except ImportError:
|
||||
# Direct script execution puts this directory, rather than ``examples``, on
|
||||
# sys.path. Keep both ``python file.py`` and module/importlib use working.
|
||||
import basic_fasth3 # type: ignore[no-redef]
|
||||
|
||||
MODEL = "FastVideo/FastVideo-FastH3-8-Step-V2"
|
||||
GRID_POINTS = 9 # nine sigma-grid points = eight DiT forwards
|
||||
VSA_SPARSITY = 0.8
|
||||
|
||||
|
||||
def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace:
|
||||
parser = basic_fasth3.build_parser(description=__doc__)
|
||||
parser.set_defaults(
|
||||
model_path=MODEL,
|
||||
output="outputs/fasth3_8step",
|
||||
steps=GRID_POINTS,
|
||||
vsa_sparsity=VSA_SPARSITY,
|
||||
vsa_tile_size=64,
|
||||
)
|
||||
args = basic_fasth3.validate_args(parser, parser.parse_args(argv))
|
||||
if args.steps != GRID_POINTS:
|
||||
parser.error(f"--steps must be {GRID_POINTS} for {MODEL}: the checkpoint's trained ladder has "
|
||||
f"{GRID_POINTS - 1} rungs and the pipeline rejects any other grid")
|
||||
return args
|
||||
|
||||
|
||||
def main() -> None:
|
||||
basic_fasth3.run(parse_args())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,147 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Generate 5-second FastH3 videos with a supported Preview adapter."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
FASTVIDEO_ROOT = Path(__file__).resolve().parents[3]
|
||||
sys.path.insert(0, str(FASTVIDEO_ROOT))
|
||||
|
||||
from fastvideo import VideoGenerator # noqa: E402
|
||||
from fastvideo.api import ( # noqa: E402
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
OffloadConfig,
|
||||
OutputConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
SamplingConfig,
|
||||
)
|
||||
|
||||
# FastVideo exposes these three backend switches only through environment variables.
|
||||
os.environ.update({
|
||||
"FASTVIDEO_FA4": "1",
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all",
|
||||
"FASTVIDEO_VSA_SM100A": "0",
|
||||
})
|
||||
os.environ.pop("FASTVIDEO_INFERENCE_TORCH_COMPILE", None)
|
||||
|
||||
VARIANT_BACKENDS = {
|
||||
"dense-datafree": "FLASH_ATTN",
|
||||
"vsa-datafree": "VIDEO_SPARSE_ATTN_H3",
|
||||
"vsa-synthetic-step1300": "VIDEO_SPARSE_ATTN_H3",
|
||||
"vsa-synthetic-step1900": "VIDEO_SPARSE_ATTN_H3",
|
||||
}
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("variant", choices=VARIANT_BACKENDS)
|
||||
parser.add_argument("--prompt", required=True)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Run one compile warmup and three measured generations with one fixed recipe."""
|
||||
args = parse_args()
|
||||
|
||||
attention_backend = VARIANT_BACKENDS[args.variant]
|
||||
experimental = {
|
||||
"attention_backend": attention_backend,
|
||||
"inference_torch_compile": attention_backend == "FLASH_ATTN",
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
}
|
||||
if attention_backend == "VIDEO_SPARSE_ATTN_H3":
|
||||
experimental.update({
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
})
|
||||
|
||||
adapter_path = hf_hub_download(
|
||||
repo_id="FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
filename=f"{args.variant}/adapter_model.safetensors",
|
||||
)
|
||||
output_dir = FASTVIDEO_ROOT / "outputs/fasth3_lora_preview" / args.variant
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
generator = VideoGenerator.from_config(
|
||||
GeneratorConfig(
|
||||
model_path="MiniMaxAI/MiniMax-H3",
|
||||
pipeline=PipelineSelection(
|
||||
components=ComponentConfig(lora_path=adapter_path, lora_strength=1.0),
|
||||
experimental=experimental,
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=4,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=4),
|
||||
offload=OffloadConfig(dit=False, dit_layerwise=False),
|
||||
compile=CompileConfig(vae_enabled=True),
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
generation_count = 4
|
||||
warmup_count = 1
|
||||
measured_count = generation_count - warmup_count
|
||||
measured_seconds: list[float] = []
|
||||
|
||||
print(f"Variant: {args.variant} ({attention_backend})")
|
||||
print(f"Output directory: {output_dir}")
|
||||
try:
|
||||
for generation_index in range(generation_count):
|
||||
measured = generation_index >= warmup_count
|
||||
if measured:
|
||||
measured_index = generation_index - warmup_count + 1
|
||||
label = f"measured {measured_index}/{measured_count}"
|
||||
output_path = output_dir / f"fasth3_all_run_{measured_index:02d}.mp4"
|
||||
else:
|
||||
warmup_index = generation_index + 1
|
||||
label = f"warmup {warmup_index}/{warmup_count}"
|
||||
output_path = output_dir / f"_fasth3_warmup_{warmup_index:02d}.mp4"
|
||||
|
||||
request = GenerationRequest(
|
||||
prompt=args.prompt,
|
||||
negative_prompt="",
|
||||
sampling=SamplingConfig(
|
||||
height=768,
|
||||
width=1344,
|
||||
num_frames=345, # <-- Change video length: 5sec: 124; 10sec: 243; 15sec: 345.
|
||||
fps=24,
|
||||
num_inference_steps=5,
|
||||
guidance_scale=1.0,
|
||||
batch_cfg=False,
|
||||
seed=1000,
|
||||
),
|
||||
output=OutputConfig(
|
||||
output_path=str(output_path),
|
||||
save_video=True,
|
||||
return_frames=False,
|
||||
),
|
||||
)
|
||||
started = time.perf_counter()
|
||||
result = generator.generate(request)
|
||||
elapsed = time.perf_counter() - started
|
||||
if measured:
|
||||
measured_seconds.append(elapsed)
|
||||
suffix = "" if measured else " (excluded from median)"
|
||||
print(f"[{label}] {result.video_path or output_path}: {elapsed:.3f}s{suffix}")
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
print(f"Median E2E wall time: {statistics.median(measured_seconds):.3f}s")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user