Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1ea7278c00 | ||
|
|
a5aa64ab6b | ||
|
|
0bcd15d5e2 | ||
|
|
0b0c692012 | ||
|
|
6c2a250fd0 | ||
|
|
0607bd78c7 | ||
|
|
c4bb13a650 | ||
|
|
d051603677 | ||
|
|
00a8e5bcb6 | ||
|
|
4117696a4f | ||
|
|
23cb8c0c2b | ||
|
|
ca7965ed28 | ||
|
|
61dffbe904 | ||
|
|
7ec6251133 | ||
|
|
b323cb787c | ||
|
|
e263249175 | ||
|
|
6c776a219b | ||
|
|
3f6893a098 | ||
|
|
d728a0dc5f | ||
|
|
d6ba5ced94 | ||
|
|
9139c0411b | ||
|
|
f571621ae7 | ||
|
|
96b7c0d223 | ||
|
|
74a52c027b | ||
|
|
2cd2018137 | ||
|
|
6627366805 | ||
|
|
7aa1f78e10 | ||
|
|
38096b94a3 | ||
|
|
4d05e9c020 | ||
|
|
62780cf53b | ||
|
|
40ff9af3fb | ||
|
|
93d921fe33 | ||
|
|
d26aa6b2d7 | ||
|
|
b9c397dd1a | ||
|
|
867f960bf0 | ||
|
|
1d7e7e2e35 | ||
|
|
0d93b16fdd | ||
|
|
7ba7dbbbd5 | ||
|
|
c7c2d77fae | ||
|
|
991fa391ff | ||
|
|
ecb15bf9e0 | ||
|
|
b24396561a | ||
|
|
0b80dd52db | ||
|
|
04878f0584 | ||
|
|
cd1ffc09c9 | ||
|
|
db4a60c5db | ||
|
|
9491c8638a | ||
|
|
6809a751fb | ||
|
|
8322b01815 | ||
|
|
9edc8adf5f | ||
|
|
e1f3904799 | ||
|
|
02f1ce11ae | ||
|
|
7f03e03dc6 | ||
|
|
e3b88bb12a | ||
|
|
cb66acd400 | ||
|
|
442e2d2e18 | ||
|
|
dd35763ad6 | ||
|
|
e90be598e5 | ||
|
|
ba5e81083c | ||
|
|
76ce9c7fd6 | ||
|
|
08d99c089e | ||
|
|
20751a21aa | ||
|
|
9dd2a837f4 | ||
|
|
93aab45ac2 | ||
|
|
017ce6602d | ||
|
|
a575055eec | ||
|
|
81f3fec7fd | ||
|
|
d265a454bf | ||
|
|
c100c66578 | ||
|
|
8760eb7a06 | ||
|
|
361f919c88 | ||
|
|
8b5377aab2 | ||
|
|
d995516da0 | ||
|
|
f47ad3f5b7 | ||
|
|
10bdf5e076 | ||
|
|
3e26db40b0 | ||
|
|
c73dd0ab55 | ||
|
|
430e52154e | ||
|
|
384eee8aef | ||
|
|
c4824c7764 | ||
|
|
9b0e57fe4b | ||
|
|
0100218594 | ||
|
|
39718cd54d | ||
|
|
37d06a832f | ||
|
|
8839ba8d4d | ||
|
|
61b91220c0 | ||
|
|
316f3876c2 | ||
|
|
614b59543c | ||
|
|
1c14afd559 | ||
|
|
9a3c45779c | ||
|
|
bfc9c01797 | ||
|
|
3a3ad3d209 | ||
|
|
aef4e9b3b1 | ||
|
|
a943220c11 | ||
|
|
556ac7088e | ||
|
|
e7456f1b75 | ||
|
|
c993d7393e | ||
|
|
7f83164233 | ||
|
|
4e52f47d1e | ||
|
|
e19913f6e9 | ||
|
|
2413a57651 | ||
|
|
7bb76b5ec9 | ||
|
|
0bd19a976b | ||
|
|
40b93784d2 | ||
|
|
33d3478bad | ||
|
|
3d8ac9d14b | ||
|
|
aaef49bfc6 | ||
|
|
cf6a00b9be | ||
|
|
1ae39562dd | ||
|
|
26064193e2 | ||
|
|
8446fc003e | ||
|
|
a28f2bab4b | ||
|
|
f82d8be4bf | ||
|
|
8e1775183e | ||
|
|
620bc36dc4 | ||
|
|
29ff16ec96 | ||
|
|
8f9d76a80d | ||
|
|
a4d9a75e2c | ||
|
|
b2db0c0a13 | ||
|
|
6aa7d8a278 | ||
|
|
ccc9014430 |
@@ -80,26 +80,36 @@ def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
str(model_path),
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=False,
|
||||
{
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"use_fsdp_inference": False,
|
||||
"offload": {
|
||||
"dit": False,
|
||||
"vae": False,
|
||||
"text_encoder": False,
|
||||
},
|
||||
},
|
||||
},
|
||||
)
|
||||
try:
|
||||
return generator.generate_video(
|
||||
prompt=params["prompt"],
|
||||
negative_prompt=params.get("negative_prompt"),
|
||||
output_path=f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
save_video=False,
|
||||
height=params.get("height"),
|
||||
width=params.get("width"),
|
||||
num_frames=params.get("num_frames"),
|
||||
fps=params.get("fps"),
|
||||
num_inference_steps=params["num_inference_steps"],
|
||||
guidance_scale=params.get("guidance_scale"),
|
||||
seed=params["seed"],
|
||||
)
|
||||
return generator.generate({
|
||||
"prompt": params["prompt"],
|
||||
"negative_prompt": params.get("negative_prompt"),
|
||||
"sampling": {
|
||||
"height": params.get("height"),
|
||||
"width": params.get("width"),
|
||||
"num_frames": params.get("num_frames"),
|
||||
"fps": params.get("fps"),
|
||||
"num_inference_steps": params["num_inference_steps"],
|
||||
"guidance_scale": params.get("guidance_scale"),
|
||||
"seed": params["seed"],
|
||||
},
|
||||
"output": {
|
||||
"output_path": f"outputs_{_MODEL_FAMILY}/pipeline_parity",
|
||||
"save_video": False,
|
||||
},
|
||||
})
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
@@ -84,7 +84,9 @@ fastvideo/configs/models/dits/__init__.py
|
||||
fastvideo/configs/models/encoders/__init__.py
|
||||
fastvideo/configs/models/vaes/__init__.py
|
||||
fastvideo/envs.py
|
||||
fastvideo/fastvideo_args.py
|
||||
fastvideo/api/schema.py
|
||||
fastvideo/api/resolution.py
|
||||
fastvideo/api/inference_resolution.py
|
||||
fastvideo/distributed/**
|
||||
fastvideo/layers/**
|
||||
fastvideo/attention/**
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
---
|
||||
name: env-var-conventions
|
||||
description: Add, read, rename, or remove an environment variable in FastVideo, or change the environment-variable policy. Use before touching fastvideo/envs.py, os.environ, os.getenv, or monkeypatch.setenv in fastvideo/, and when fastvideo/tests/contract/test_env_policy.py fails.
|
||||
---
|
||||
|
||||
# Environment Variable Conventions
|
||||
|
||||
## Purpose
|
||||
|
||||
FastVideo registers its environment variables as typed fields in
|
||||
`fastvideo/envs.py`. The policy that governs them is
|
||||
`docs/contributing/env_vars.md`, and the contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit
|
||||
CI lane. This skill routes an environment-variable change through that policy.
|
||||
The policy doc is the single source of the rules; read it instead of relying
|
||||
on a summary here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Read `docs/contributing/env_vars.md` in full.
|
||||
- Decide whether the setting belongs in an environment variable or an argument
|
||||
(rule 5 in the policy doc). Settings that users change per deployment are
|
||||
arguments; add them as typed config fields in `fastvideo/api/schema.py`
|
||||
instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------- | -------- | -------------------------------------------------------------- |
|
||||
| `change` | Yes | Add, read, rename, or remove a variable, or change the policy. |
|
||||
| `variable` | Yes | The variable name, with the `FASTVIDEO_` prefix. |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Declare or edit the variable in `fastvideo/envs.py`.**
|
||||
- Pick the field type and category that the policy doc lists.
|
||||
- Write a description that states what the variable does and its units.
|
||||
- To rename, keep the old name in `deprecated_names`. To remove, add the
|
||||
name to `DEPRECATED_VARIABLES`. Update the uses in `examples/`,
|
||||
`scripts/`, `docs/`, `apps/`, and the tests.
|
||||
2. **Read the variable with `envs.NAME.get()` inside a function.**
|
||||
- In tests, change the value with `envs.NAME.override(value)`, and a variable
|
||||
outside the registry with `envs.override_external(name, value)`; the
|
||||
`env_overrides` fixture keeps either until the end of the test.
|
||||
- Name a variable that only tests read `FASTVIDEO_TEST_*`.
|
||||
- Do not call `os.environ`, `os.getenv`, or `monkeypatch.setenv` for a
|
||||
FastVideo variable.
|
||||
- To set a variable that another tool reads, call `envs.set_external`,
|
||||
`envs.setdefault_external`, or `envs.unset_external`.
|
||||
3. **Regenerate the table in the policy doc.**
|
||||
- Run `python fastvideo/tests/contract/test_env_policy.py`.
|
||||
4. **Run the contract test.**
|
||||
- Run `pytest fastvideo/tests/contract/test_env_policy.py`.
|
||||
- When the test reports a fixed known violation, delete or lower its entry
|
||||
in `KNOWN_VIOLATIONS`. Never add an entry to `KNOWN_VIOLATIONS`.
|
||||
5. **When the policy itself changes, update the policy doc and the contract
|
||||
test in the same pull request.**
|
||||
- The rules in `docs/contributing/env_vars.md`, the checks and allowlist in
|
||||
`fastvideo/tests/contract/test_env_policy.py`, and this skill must agree.
|
||||
|
||||
## Outputs
|
||||
|
||||
- A registry entry in `fastvideo/envs.py` and call sites that use
|
||||
`envs.NAME.get()`.
|
||||
- A regenerated table in `docs/contributing/env_vars.md`.
|
||||
- A passing `fastvideo/tests/contract/test_env_policy.py`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Add a FASTVIDEO_DEBUG_MY_STAGE switch that logs MyStage inputs.
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- `docs/contributing/env_vars.md`: the policy, the field types, and the
|
||||
violation kinds that the contract test reports.
|
||||
- `fastvideo/envs.py`: the registry.
|
||||
- `fastvideo/tests/contract/test_env_policy.py`: the contract test,
|
||||
`EXTERNAL_ALLOWLIST`, and `KNOWN_VIOLATIONS`.
|
||||
@@ -57,7 +57,7 @@ Hardcoded:
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
`FASTVIDEO_SSIM_REFERENCE_HF_REPO`).
|
||||
`FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
@@ -185,6 +185,7 @@ steps:
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
build.env("TEST_SCOPE") == "scheduled" ||
|
||||
@@ -211,6 +212,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
key: "lora-inference"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -232,6 +234,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Extraction Tests"
|
||||
key: "lora-extraction"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -253,6 +256,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Training Tests"
|
||||
key: "training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -276,6 +280,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
key: "distillation"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -297,6 +302,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
key: "self-forcing"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -318,6 +324,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
key: "lora-training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -341,6 +348,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
key: "training-vsa"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -364,6 +372,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
key: "inference-vmoba"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -385,6 +394,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Performance Tests"
|
||||
key: "performance"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -406,6 +416,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: API Server Tests"
|
||||
key: "api-server"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -427,6 +438,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
key: "train-framework"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
@@ -448,6 +460,7 @@ steps:
|
||||
|
||||
- label: ":test_tube: Eval Metrics Tests"
|
||||
key: "eval"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
|
||||
@@ -13,7 +13,7 @@ if [ -z "$selected" ]; then
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" = all ]; then
|
||||
exec pytest "$golden_root" -vs
|
||||
exec pytest "$golden_root" -xvs
|
||||
fi
|
||||
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
@@ -32,4 +32,4 @@ for golden_file in "${golden_files[@]}"; do
|
||||
golden_paths+=("$golden_path")
|
||||
done
|
||||
|
||||
exec pytest "${golden_paths[@]}" -vs
|
||||
exec pytest "${golden_paths[@]}" -xvs
|
||||
|
||||
@@ -2,4 +2,8 @@
|
||||
# Canonical Slurm CI selection for the transformer lane.
|
||||
set -euo pipefail
|
||||
|
||||
# The existing block reference records an absent FASTVIDEO_FA4 (FA2). Keep
|
||||
# that reference identity; the component lane also selects FA2 explicitly.
|
||||
env -u FASTVIDEO_FA4 pytest ./fastvideo/tests/golden_gate/test_wan_t2v.py -xvs
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_causal.py -xvs
|
||||
exec pytest ./fastvideo/tests/transformers -vs
|
||||
|
||||
@@ -2,4 +2,5 @@
|
||||
# Canonical Slurm CI selection for the VAE lane.
|
||||
set -euo pipefail
|
||||
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_vae.py -xvs
|
||||
exec pytest ./fastvideo/tests/vaes -vs
|
||||
|
||||
@@ -16,6 +16,7 @@ exec pytest \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/attention/test_sdpa_metadata_mask_contract.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_tile_grad_safety.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
|
||||
@@ -211,8 +211,8 @@ FAMILY_COVERAGE = (
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", ),
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
(
|
||||
"test_causal_similarity.py",
|
||||
"test_wan_i2v_similarity.py",
|
||||
@@ -291,6 +291,18 @@ def _family_coverage(path: str) -> tuple[set[str], set[str]]:
|
||||
if family.pattern.search(normalized):
|
||||
golden.update(family.golden_tests)
|
||||
ssim.update(family.ssim_tests)
|
||||
# Select the component actually touched, including compatibility paths.
|
||||
# Family configs/pipeline wiring can affect all four Wan gates.
|
||||
if re.search(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)", normalized):
|
||||
if (normalized.endswith(("/wan/vae.py", "/wan/vae_config.py", "/vaes/wanvae.py"))
|
||||
or normalized.endswith("/wan/stages/conditioning.py")):
|
||||
golden = {"test_wan_vae.py"}
|
||||
elif normalized.endswith(("/wan/causal_transformer.py", "/dits/causal_wanvideo.py",
|
||||
"/wan/stages/causal_denoising.py")):
|
||||
golden = {"test_wan_causal.py"}
|
||||
elif (normalized == "fastvideo/models/dits/wanvideo.py"
|
||||
or normalized.endswith(("/wan/transformer.py", "/wan/stages/denoising.py", "/wan/stages/dmd.py"))):
|
||||
golden = {"test_wan_t2v.py", "test_wan_denoising.py"}
|
||||
return golden, ssim
|
||||
|
||||
|
||||
@@ -477,7 +489,6 @@ def classify_paths(paths: list[str]) -> MergePlan:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path in {
|
||||
"fastvideo/fastvideo_args.py",
|
||||
"fastvideo/forward_context.py",
|
||||
"fastvideo/image_processor.py",
|
||||
"fastvideo/registry.py",
|
||||
|
||||
@@ -8,6 +8,8 @@ on:
|
||||
- "fastvideo/mlx_runtime/**"
|
||||
- "fastvideo/tests/mlx/**"
|
||||
- "fastvideo/tests/platforms/test_mps_vsa_error.py"
|
||||
- "fastvideo/tests/platforms/test_cpu_sdpa.py"
|
||||
- "fastvideo/platforms/cpu.py"
|
||||
- "fastvideo/platforms/mps.py"
|
||||
- "fastvideo/platforms/__init__.py"
|
||||
- "fastvideo/__init__.py"
|
||||
@@ -79,11 +81,18 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_compile_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint_compat.py \
|
||||
fastvideo/tests/mlx/test_mlx_affine_dq_gemm.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa_regressions.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_mode.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_fastwan_benchmark.py \
|
||||
fastvideo/tests/mlx/test_taehv_decode.py \
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
@@ -91,7 +100,8 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_download_unavailable_has_specific_error \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
-q
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s -o faulthandler_timeout=120
|
||||
|
||||
# Same tests on MLX's CPU backend. Hosted macOS runners are scarce and
|
||||
# slower to schedule; this Linux job gives fast PR signal on the identical
|
||||
@@ -136,11 +146,18 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_compile_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint_compat.py \
|
||||
fastvideo/tests/mlx/test_mlx_affine_dq_gemm.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa_regressions.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_mode.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_fastwan_benchmark.py \
|
||||
fastvideo/tests/mlx/test_taehv_decode.py \
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
@@ -148,4 +165,5 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_download_unavailable_has_specific_error \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
-q
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s -o faulthandler_timeout=120
|
||||
|
||||
@@ -63,7 +63,7 @@ jobs:
|
||||
torch-cuda-short: 'cu130'
|
||||
platform:
|
||||
# x86_64 builds the full cu126 + cu130 set. cu130 ships the
|
||||
# data-center Blackwell sm_100a VSA and consumer sm_120a FP4
|
||||
# data-center Blackwell sm_100a/sm_103a VSA and consumer sm_120a FP4
|
||||
# kernels.
|
||||
- os: ubuntu-22.04
|
||||
arch: x86_64
|
||||
@@ -169,18 +169,18 @@ jobs:
|
||||
# covers sm_120a; turbodiffusion covers sm_100a+sm_120a. The sm_100 FP4
|
||||
# forward is the FA4 CuTe DSL path in the fastvideo package (PR #1221),
|
||||
# JIT-compiled at runtime — not built into this wheel.
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a VSA
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a/sm_103a VSA
|
||||
# + consumer Blackwell sm_120a FP4.
|
||||
# * x86_64 cu126 = Hopper TK only (older drivers; CUDA < 12.8 has no FP4).
|
||||
# The per-arch split in CMakeLists pins the FP4 targets to sm_120a and builds
|
||||
# the main extension for the full arch list. CMAKE_BUILD_PARALLEL_LEVEL caps
|
||||
# Ninja so heavy CUTLASS/TK template TUs don't OOM the 16 GB runner (exit 143).
|
||||
if [ "${{ matrix.platform.arch }}" = "aarch64" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;10.3a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=OFF -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON"
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
elif [ "${{ matrix.torch-cuda.torch-cuda-short }}" = "cu130" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;10.3a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
|
||||
# A single FP4 TU (attn_qat_infer) can use ~8-12 GB on its own, so serialize.
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
|
||||
@@ -55,6 +55,8 @@ eggs/
|
||||
|
||||
# MkDocs documentation
|
||||
site/
|
||||
docs/assets/cookbook-serving.json
|
||||
examples/serving/clients/node_modules/
|
||||
docs/getting_started/examples/
|
||||
docs/examples/
|
||||
docs/inference/examples/
|
||||
@@ -133,6 +135,9 @@ fastvideo/tests/ssim/reference_videos/**
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.mp4
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.png
|
||||
|
||||
# Local H3 MLX kernel / exactness benches (JSON, logs, frames, videos)
|
||||
.kernel_bench/
|
||||
|
||||
# Editor logs and local Python version pins (accidentally committed)
|
||||
*.nvimlog
|
||||
.nvimlog
|
||||
|
||||
@@ -9,7 +9,7 @@ exclude: |
|
||||
tests/.*|
|
||||
scripts/.*|
|
||||
fastvideo/dataset/.*|
|
||||
fastvideo/models/.*|
|
||||
fastvideo/models/(?!wan/(config|vae_config|pipeline_config|definition|__init__)\.py$).*|
|
||||
^apps/dreamverse/web/.*|
|
||||
examples/.*|
|
||||
\.agents/.*|
|
||||
|
||||
@@ -66,14 +66,18 @@ Local guidance lives next to the code. Read the in-scope file before editing:
|
||||
| `fastvideo/AGENTS.md` | Core package map, public API, registry-driven model dispatch |
|
||||
| `fastvideo/configs/AGENTS.md` | Arch + pipeline config dataclasses, `param_names_mapping` |
|
||||
| `fastvideo/models/AGENTS.md` | DiT / VAE / encoder / scheduler / loader layout (pre-commit excluded) |
|
||||
| `fastvideo/models/wan/AGENTS.md` | Wan family-local transformers, VAE, configs, and the SP sharding invariant |
|
||||
| `fastvideo/layers/AGENTS.md` | Tensor-parallel linear/attention layer rules for ports |
|
||||
| `fastvideo/attention/AGENTS.md` | Backend registry + env-var override |
|
||||
| `fastvideo/pipelines/AGENTS.md` | Stage ABC, `basic/<model>/`, `preprocess/`, presets |
|
||||
| `fastvideo/pipelines/basic/wan/AGENTS.md` | Wan sampling stages, first-frame conditioning, DMD/causal boundaries |
|
||||
| `fastvideo/pipelines/basic/magi_human/AGENTS.md` | MagiHuman umbrella repo, lazy-loaded components, packing invariants |
|
||||
| `fastvideo/training/AGENTS.md` | Legacy monolithic pipelines (frozen for existing models) |
|
||||
| `fastvideo/train/AGENTS.md` | New modular trainer (methods × models × callbacks, YAML) |
|
||||
| `fastvideo/tests/AGENTS.md` | Test taxonomy, conftest, pre-commit-excluded path |
|
||||
| `fastvideo/tests/ssim/AGENTS.md` | GPU SSIM regression authoring + reference video sync |
|
||||
| `scripts/checkpoint_conversion/AGENTS.md` | Adding a converter for a new HF/official checkpoint |
|
||||
| `apps/dreamverse/AGENTS.md` | DreamVerse app structure and conventions |
|
||||
|
||||
## Critical: Two Training Stacks Coexist
|
||||
|
||||
|
||||
@@ -3,14 +3,16 @@
|
||||
</div>
|
||||
|
||||
<p align="center">
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://haoailab.com/FastVideo/cookbook/"><b>Cookbook</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
</p>
|
||||
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
## NEWS
|
||||
- `2026/08/23`: [FastH3 Preview v0.2](https://huggingface.co/FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2) is a 4-step DMD2-distilled MiniMax-H3 checkpoint that generates synchronized video and audio. Run the verified [basic FastH3 example](examples/inference/basic/basic_fasth3.py), see the [inference guide](examples/inference/basic/README.md#fasth3-preview), or have a coding agent install FastVideo with the [agent setup prompt](#install-with-an-ai-coding-agent).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac—follow the [Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/09/15`: Release [FastH3 8-Step V2](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2), an eight-forward data-free DMD2 checkpoint distilled from MiniMax-H3 with 80% Video Sparse Attention. Run it with `examples/inference/basic/basic_fasth3_8step.py` or the [FastH3 8-Step V2 recipe](https://haoailab.com/FastVideo/cookbook/minimax-h3/).
|
||||
- `2026/09/01`: FastH3 now runs locally on Apple Silicon through MLX and on NVIDIA DGX Spark through CUDA 13, including two-Spark inference. Follow the [FastH3 recipes](https://haoailab.com/FastVideo/cookbook/minimax-h3/) and read the [Blog](https://haoailab.com/blogs/fasth3-local/).
|
||||
- `2026/08/27`: [FastH3 Preview v1](https://haoailab.com/blogs/fasth3-preview/) is an open-weight 4-step sparse-distilled MiniMax-H3 model for synchronized video-and-audio generation, developed in collaboration with [Nuva Lab](https://nuvalab.ai/) and the [NVIDIA FastGen team](https://github.com/NVlabs/FastGen). Download the recommended [VSA / Data-Free weights](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree), or see the [full FastH3 collection](https://huggingface.co/collections/FastVideo/fastvideo-fasth3).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac. Follow the [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/06/23`: Release FastWan-QAD: 5s of Video generated in 1.8s E2E. See the [FastWan-QAD models](https://huggingface.co/FastVideo/FastWan-QAD-FP8-1.3B), [Attn-QAT training guide](https://haoailab.com/FastVideo/training/attn_qat/), and [blog](https://haoailab.com/blogs/fastwan-qad/).
|
||||
- `2026/03/17`: Release demo: Into the Dreamverse: Vibe Directing in FastVideo, check out the [Blog](https://haoailab.com/blogs/dreamverse/).
|
||||
- `2026/03/13`: Release demo: Create a 5s 1080p Video in 4.5s with FastVideo on a Single GPU, check out the [Blog](https://haoailab.com/blogs/fastvideo_realtime_1080p/).
|
||||
@@ -62,13 +64,12 @@ UV_TORCH_BACKEND=cu126 uv pip install fastvideo
|
||||
```
|
||||
|
||||
Use `UV_TORCH_BACKEND=cu130` on CUDA 13. Apple silicon users should follow the
|
||||
[MPS installation guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
> **On an Apple Silicon Mac?** FastVideo runs FastMetal-QAD through an MLX
|
||||
> runtime. Install with `uv pip install -e '.[mlx]'`, download
|
||||
> [`FastVideo/FastMetal-1.3B-QAD`](https://huggingface.co/FastVideo/FastMetal-1.3B-QAD),
|
||||
> and follow the
|
||||
> [Apple Silicon guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
> **On an Apple Silicon Mac?** Install with `uv pip install -e '.[mlx]'` from
|
||||
> a clone, then pick a recipe in the
|
||||
> [cookbook](https://haoailab.com/FastVideo/cookbook/). See the
|
||||
> [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/) for more detailed installation instructions.
|
||||
|
||||
@@ -86,7 +87,7 @@ Install FastVideo (https://github.com/hao-ai-lab/FastVideo) into a fresh uv virt
|
||||
https://hao-ai-lab.github.io/FastVideo/getting_started/installation/):
|
||||
- NVIDIA GPU, x86_64 -> docs/getting_started/installation/gpu.md
|
||||
- NVIDIA DGX Spark / GB10, aarch64, CUDA 13 -> docs/getting_started/installation/spark.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mps.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mlx.md
|
||||
3. Use uv for every step. If a command fails, debug it and tell me what you changed.
|
||||
4. Verify the result:
|
||||
python -c "import fastvideo, torch; print('cuda', torch.cuda.is_available())"
|
||||
@@ -134,18 +135,20 @@ def main():
|
||||
# Create a video generator with a pre-trained model
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
num_gpus=1, # Adjust based on your hardware
|
||||
{"engine": {"num_gpus": 1}}, # Adjust based on your hardware
|
||||
)
|
||||
|
||||
# Define a prompt for your video
|
||||
prompt = "A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes wide with interest."
|
||||
|
||||
# Generate the video
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path="my_videos/", # Controls where videos are saved
|
||||
save_video=True
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": prompt,
|
||||
"output": {
|
||||
"output_path": "my_videos/", # Controls where videos are saved
|
||||
"save_video": True,
|
||||
},
|
||||
})
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
||||
@@ -97,13 +97,33 @@ dreamverse-server --port 8009
|
||||
dreamverse-mock-server --port 8009
|
||||
```
|
||||
|
||||
### Run Dreamverse with FastH3
|
||||
|
||||
Select the VSA data-free FastH3 Preview profile when you start the backend:
|
||||
|
||||
```bash
|
||||
DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
The `fast-h3` profile uses four visible GPUs by default. It loads the `MiniMaxAI/MiniMax-H3` base checkpoint and the
|
||||
`vsa-datafree/adapter_model.safetensors` adapter from
|
||||
`FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA`. Each request generates a 124-frame, 768×1344 video with
|
||||
synchronized audio and five sigma-grid points. Dreamverse uses the last frame of each segment as first-frame
|
||||
conditioning for the following segment.
|
||||
|
||||
Set `CUDA_VISIBLE_DEVICES` when you need to choose the four physical GPUs:
|
||||
|
||||
```bash
|
||||
CUDA_VISIBLE_DEVICES=0,1,2,3 DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
> **Expect a slow first boot.** With `torch.compile` and startup warmup enabled
|
||||
> (the default), the backend compiles the segment 1 and segment 2 inference
|
||||
> paths before it reports ready — this can take **tens of minutes on a cold
|
||||
> cache**, regardless of how you deploy (local, server, Docker, or Modal).
|
||||
> `/healthz` responds as soon as the process is up; `/readyz` stays `503` until
|
||||
> warmup finishes. For a faster, uncompiled startup while testing, set
|
||||
> `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
> warmup finishes. To defer compilation until the first generated request while
|
||||
> testing, set `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
|
||||
## Frontend Setup
|
||||
|
||||
@@ -219,6 +239,7 @@ selection, and mock-server behavior:
|
||||
pytest apps/dreamverse/dreamverse/tests/test_config.py \
|
||||
apps/dreamverse/dreamverse/tests/test_entrypoints.py \
|
||||
apps/dreamverse/dreamverse/tests/test_gpu_pool.py \
|
||||
apps/dreamverse/dreamverse/tests/test_minimax_h3_generation.py \
|
||||
apps/dreamverse/dreamverse/tests/test_mock_server.py -q
|
||||
```
|
||||
|
||||
|
||||
@@ -42,7 +42,7 @@ Near-term OSS note:
|
||||
- `apps/dreamverse/dreamverse/main.py`: websocket endpoint, request handling,
|
||||
session state machine, rewrite orchestration, REST routes, and stream relay
|
||||
- `apps/dreamverse/dreamverse/gpu_pool.py`: GPU worker processes, warmup, model
|
||||
loading, and `generate_video()` calls through FastVideo
|
||||
loading, and `generate()` calls through FastVideo
|
||||
- `apps/dreamverse/dreamverse/prompt_enhancer.py`: prompt enhancement, rollout
|
||||
rewrite execution, provider selection, and timeout/fallback behavior
|
||||
- `apps/dreamverse/dreamverse/rewrite_prompt_payload.py`: canonical rewrite request payload
|
||||
@@ -139,7 +139,18 @@ session.
|
||||
- startup warmup
|
||||
- user join/leave commands
|
||||
- `USER_STEP` execution for each segment
|
||||
- continuation state between segments
|
||||
- generation-command routing and stream-result delivery
|
||||
|
||||
Model generation has a separate ownership boundary inside each GPU process:
|
||||
|
||||
- `apps/dreamverse/dreamverse/generation_worker.py` selects the backend that the active model profile declares and owns
|
||||
the backend lifecycle.
|
||||
- `apps/dreamverse/dreamverse/ltx2_generation.py` owns LTX-2 generator configuration, video and audio continuation, and
|
||||
runtime LoRA application.
|
||||
- `apps/dreamverse/dreamverse/minimax_h3_generation.py` owns the VSA data-free FastH3 adapter, FastH3 generator and
|
||||
request configuration, and last-frame continuation through MiniMax H3 first-frame conditioning.
|
||||
- `apps/dreamverse/dreamverse/generation_contracts.py` defines the decoded media and stream-trimming result that both
|
||||
model backends return to `apps/dreamverse/dreamverse/gpu_pool.py`.
|
||||
|
||||
`apps/dreamverse/dreamverse/prompt_enhancer.py` manages:
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Benchmark the LTX-2 generation pipeline driven by the dreamverse Python SDK path.
|
||||
|
||||
Mirrors how ``apps/dreamverse/dreamverse/video_generation.py`` constructs
|
||||
Mirrors how ``apps/dreamverse/dreamverse/ltx2_generation.py`` constructs
|
||||
``GeneratorConfig`` and calls ``VideoGenerator.generate()``, then
|
||||
captures per-stage timings via the ``FASTVIDEO_STAGE_LOGGING=1`` log
|
||||
hooks (same mechanism as ``FastVideo-internal/examples/inference/basic/
|
||||
@@ -52,7 +52,8 @@ import torch # noqa: E402
|
||||
|
||||
from fastvideo import VideoGenerator # noqa: E402
|
||||
from fastvideo.api import ( # noqa: E402
|
||||
ComponentConfig, CompileConfig, EngineConfig, GeneratorConfig, OffloadConfig, PipelineSelection, QuantizationConfig,
|
||||
ComponentConfig, CompileConfig, EngineConfig, GenerationResult, GeneratorConfig, OffloadConfig, PipelineSelection,
|
||||
QuantizationConfig,
|
||||
)
|
||||
|
||||
DEFAULT_PROMPT = ("A cinematic drone shot over coastal cliffs at sunrise, golden "
|
||||
@@ -128,9 +129,9 @@ def _build_generator_config(model_path: str, enable_compile: bool, num_gpus: int
|
||||
)
|
||||
|
||||
|
||||
def _extract_stage_times(result: dict) -> OrderedDict[str, float]:
|
||||
def _extract_stage_times(result: GenerationResult) -> OrderedDict[str, float]:
|
||||
out: OrderedDict[str, float] = OrderedDict()
|
||||
info = result.get("logging_info") if isinstance(result, dict) else None
|
||||
info = result.logging_info if isinstance(result, GenerationResult) else None
|
||||
if info is None:
|
||||
return out
|
||||
stages = getattr(info, "stages", None)
|
||||
@@ -162,19 +163,25 @@ def _do_one_run(generator: VideoGenerator, prompt: str, *, height: int, width: i
|
||||
_reset_peak_gpu()
|
||||
t0 = time.perf_counter()
|
||||
try:
|
||||
result = generator.generate_video(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
save_video=False,
|
||||
height=height,
|
||||
width=width,
|
||||
num_frames=num_frames,
|
||||
fps=24,
|
||||
num_inference_steps=num_inference_steps,
|
||||
guidance_scale=1.0,
|
||||
seed=seed,
|
||||
ltx2_image_crf=0.0,
|
||||
)
|
||||
result = generator.generate({
|
||||
"prompt": prompt,
|
||||
"negative_prompt": "",
|
||||
"sampling": {
|
||||
"height": height,
|
||||
"width": width,
|
||||
"num_frames": num_frames,
|
||||
"fps": 24,
|
||||
"num_inference_steps": num_inference_steps,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": seed,
|
||||
},
|
||||
"output": {
|
||||
"save_video": False
|
||||
},
|
||||
"extensions": {
|
||||
"ltx2_image_crf": 0.0
|
||||
},
|
||||
})
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.synchronize()
|
||||
except Exception as exc:
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
_SERVER_ROOT = Path(__file__).resolve().parent
|
||||
@@ -55,16 +56,34 @@ FRONTEND_STATIC_DIR_CANDIDATES = _resolve_frontend_static_dir_candidates()
|
||||
MODEL_REGISTRY = {
|
||||
"fast-ltx2": {
|
||||
"name": "FastLTX2",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX2-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-ltx23": {
|
||||
"name": "FastLTX23",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX-2.3-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-h3": {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
},
|
||||
}
|
||||
|
||||
DEFAULT_MODEL_ID = "fast-ltx2"
|
||||
@@ -171,7 +190,7 @@ def _optional_env(*names: str) -> str | None:
|
||||
DEVTOOLS_ENABLED = _env_bool("FASTVIDEO_ENABLE_DEVTOOLS", False)
|
||||
PROMPT_SAFETY_ENABLED = _env_bool("FASTVIDEO_ENABLE_PROMPT_SAFETY", False)
|
||||
DREAMVERSE_MAX_AUTOTUNE = _env_bool("DREAMVERSE_MAX_AUTOTUNE", True)
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", 1))
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", cast(int, MODEL_CONFIG["default_sp_size"])))
|
||||
|
||||
DREAMVERSE_MODEL_PATH = (os.getenv("DREAMVERSE_MODEL_PATH", "").strip() or None)
|
||||
if DREAMVERSE_MODEL_PATH:
|
||||
@@ -213,7 +232,7 @@ def _resolve_lora_spec(spec: str) -> str | None:
|
||||
if not spec:
|
||||
return None
|
||||
if spec.lower() == "omninft":
|
||||
return MODEL_CONFIG.get("lora_repo")
|
||||
return cast(str | None, MODEL_CONFIG.get("lora_repo"))
|
||||
if spec.lower() in AVAILABLE_LORAS:
|
||||
return AVAILABLE_LORAS[spec.lower()]["repo"]
|
||||
return spec
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
"""Shared contract between DreamVerse generation backends and GPU workers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Protocol
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Decoded media and stream-trimming metadata for one DreamVerse segment."""
|
||||
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict[str, float]
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class GenerationBackend(Protocol):
|
||||
"""Model-owned generation operations used by one GPU worker process."""
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
...
|
||||
|
||||
def shutdown(self) -> None:
|
||||
...
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
...
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
...
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
...
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
...
|
||||
@@ -0,0 +1,96 @@
|
||||
"""Select and own one model-specific generation backend per GPU process."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dreamverse.config import MODEL_CONFIG
|
||||
from dreamverse.generation_contracts import GenerationBackend, StepResult
|
||||
|
||||
|
||||
def _create_generation_backend(backend_name: str, gpu_id: int) -> GenerationBackend:
|
||||
"""Construct the backend that owns the selected model family's behavior."""
|
||||
if backend_name == "ltx2":
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
return LTX2GenerationBackend(gpu_id)
|
||||
if backend_name == "minimax_h3":
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
return MiniMaxH3GenerationBackend(gpu_id)
|
||||
raise ValueError(f"Unsupported DreamVerse generation backend: {backend_name!r}")
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
"""Delegate GPU lifecycle and generation calls to the active model backend."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.model_config: dict = dict(MODEL_CONFIG)
|
||||
self.backend_name: str | None = None
|
||||
self.backend: GenerationBackend | None = None
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Load the requested model through its generation backend.
|
||||
|
||||
Model selection belongs here so the GPU process and streaming layers
|
||||
use one stable media contract without importing model-specific code.
|
||||
"""
|
||||
requested_model_config = dict(model_config) if model_config is not None else dict(self.model_config)
|
||||
backend_name = requested_model_config.get("generation_backend")
|
||||
if not isinstance(backend_name, str) or not backend_name:
|
||||
raise ValueError("DreamVerse model configuration requires `generation_backend`.")
|
||||
|
||||
candidate_backend = self.backend
|
||||
if candidate_backend is None or self.backend_name != backend_name:
|
||||
if candidate_backend is not None:
|
||||
candidate_backend.shutdown()
|
||||
candidate_backend = _create_generation_backend(backend_name, self.gpu_id)
|
||||
|
||||
try:
|
||||
candidate_backend.initialize(requested_model_config)
|
||||
except Exception:
|
||||
try:
|
||||
candidate_backend.shutdown()
|
||||
except Exception as shutdown_error:
|
||||
print(f"[GPU {self.gpu_id}] Backend cleanup after initialization failure: {shutdown_error}")
|
||||
self.backend = None
|
||||
self.backend_name = None
|
||||
raise
|
||||
|
||||
self.model_config = requested_model_config
|
||||
self.backend = candidate_backend
|
||||
self.backend_name = backend_name
|
||||
|
||||
def _require_backend(self) -> GenerationBackend:
|
||||
"""Return the initialized backend or fail before processing a command."""
|
||||
if self.backend is None:
|
||||
raise RuntimeError("Generation backend is not initialized.")
|
||||
return self.backend
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release model resources owned by the selected backend."""
|
||||
if self.backend is not None:
|
||||
self.backend.shutdown()
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
self._require_backend().clear_conditioning()
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one segment through the selected model backend."""
|
||||
return self._require_backend().generate_step(
|
||||
prompt,
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
return self._require_backend().warmup(prompt)
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
return self._require_backend().apply_lora_stack(stack)
|
||||
@@ -12,7 +12,7 @@ from enum import Enum
|
||||
from multiprocessing import Process, Queue
|
||||
|
||||
from dreamverse.config import (
|
||||
DEFAULT_MODEL_ID,
|
||||
ACTIVE_MODEL_ID,
|
||||
DREAMVERSE_SP_SIZE,
|
||||
MODEL_REGISTRY,
|
||||
STARTUP_WARMUP_ENABLED,
|
||||
@@ -54,7 +54,7 @@ from dreamverse.worker_ipc import (
|
||||
def _parse_requested_gpu_limit() -> int | None:
|
||||
raw_value = os.getenv("FASTVIDEO_GPU_COUNT", "").strip().lower()
|
||||
if not raw_value:
|
||||
return 1
|
||||
return DREAMVERSE_SP_SIZE
|
||||
if raw_value == "all":
|
||||
return None
|
||||
try:
|
||||
@@ -164,12 +164,12 @@ def gpu_worker_process(
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = cuda_device
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
|
||||
from dreamverse.video_generation import VideoGenerationWorker
|
||||
from dreamverse.generation_worker import VideoGenerationWorker
|
||||
|
||||
worker = VideoGenerationWorker(gpu_id)
|
||||
|
||||
def event_loop(first_cmd: Command = None):
|
||||
"""Blocking event loop for LTX2; dispatches user commands."""
|
||||
"""Block on generation commands after the model is initialized."""
|
||||
print(f"[GPU {gpu_id}] Entering event loop")
|
||||
|
||||
def handle_command(cmd: Command):
|
||||
@@ -435,7 +435,7 @@ class GPUSlot:
|
||||
self._response_reader_task: asyncio.Task | None = None
|
||||
self._active: bool = False
|
||||
self._reader_lock: asyncio.Lock | None = None
|
||||
self.current_model_id: str = DEFAULT_MODEL_ID
|
||||
self.current_model_id: str | None = ACTIVE_MODEL_ID
|
||||
self.shared_stream_buffer = None
|
||||
self.shared_stream_buffer_size = SHARED_STREAM_BUFFER_BYTES
|
||||
|
||||
@@ -690,7 +690,7 @@ class GPUSlot:
|
||||
async def join_user(self, user_id: str, model_id: str = None) -> JoinAck:
|
||||
"""Add a user to this GPU."""
|
||||
if model_id is None:
|
||||
model_id = DEFAULT_MODEL_ID
|
||||
model_id = ACTIVE_MODEL_ID
|
||||
|
||||
# Reload model if a different one is requested
|
||||
if model_id != self.current_model_id and model_id in MODEL_REGISTRY:
|
||||
@@ -705,16 +705,23 @@ class GPUSlot:
|
||||
self.connected_users.clear()
|
||||
|
||||
model_config = MODEL_REGISTRY[model_id]
|
||||
reload_response = await self._send_command(Command(CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
try:
|
||||
reload_response = await self._send_command(Command(
|
||||
CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
except Exception:
|
||||
self.current_model_id = None
|
||||
raise
|
||||
match reload_response:
|
||||
case ReloadAck():
|
||||
pass
|
||||
case WorkerError(message=msg):
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Model reload failed: {msg}")
|
||||
case _:
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Unexpected reload response: "
|
||||
f"{type(reload_response).__name__}")
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
"""LTX2 model lifecycle and continuation conditioning.
|
||||
"""LTX-2 model lifecycle and continuation conditioning.
|
||||
|
||||
Runs inside a GPU worker subprocess. Owns the model, the audio
|
||||
encoder, and the per-session continuation state carried across
|
||||
segments. Callers must set ``os.environ["CUDA_VISIBLE_DEVICES"]``
|
||||
before constructing ``VideoGenerationWorker`` — all ``fastvideo.*``
|
||||
before constructing ``LTX2GenerationBackend`` — all ``fastvideo.*``
|
||||
imports are deferred to method bodies so nothing touches CUDA at
|
||||
module import time.
|
||||
"""
|
||||
@@ -14,9 +14,7 @@ import gc
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
@@ -35,6 +33,10 @@ from dreamverse.config import (
|
||||
DREAMVERSE_LORA_STACK,
|
||||
_resolve_lora_spec,
|
||||
)
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from fastvideo.api import GenerationResult
|
||||
|
||||
# Multi-frame decoded continuation defaults from
|
||||
# examples/inference/basic/basic_ltx2_distilled_video_continuation.py.
|
||||
@@ -80,22 +82,6 @@ def _reset_lora_registry(worker) -> dict:
|
||||
return {"status": "lora_registry_reset"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Output of one generation step.
|
||||
|
||||
``head_trim_frames`` / ``head_trim_audio_frames`` are derived here
|
||||
so downstream AV streaming never needs to import conditioning
|
||||
constants.
|
||||
"""
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class ContinuationState:
|
||||
"""Per-session video + audio conditioning carried across segments."""
|
||||
|
||||
@@ -113,8 +99,8 @@ class ContinuationState:
|
||||
self.video_images = None
|
||||
self.audio_latents = None
|
||||
|
||||
def apply_video(self, request_kwargs: dict, segment_idx: int) -> None:
|
||||
"""Seed next-segment kwargs with the cached tail frames."""
|
||||
def apply_video(self, request: dict, segment_idx: int) -> None:
|
||||
"""Seed the next-segment request with the cached tail frames."""
|
||||
if segment_idx <= 1 or not self.video_images:
|
||||
return
|
||||
from PIL import Image
|
||||
@@ -128,21 +114,21 @@ class ContinuationState:
|
||||
arr = np.clip(arr, 0, 255).astype(np.uint8)
|
||||
noisy.append(Image.fromarray(arr))
|
||||
cond_images = noisy
|
||||
request_kwargs["ltx2_video_conditions"] = [(
|
||||
request["extensions"]["ltx2_video_conditions"] = [(
|
||||
cond_images,
|
||||
LTX2_VIDEO_CONDITIONING_FRAME_IDX,
|
||||
LTX2_VIDEO_CONDITIONING_STRENGTH,
|
||||
)]
|
||||
request_kwargs["ltx2_images"] = None
|
||||
request_kwargs["image_path"] = None
|
||||
request["extensions"]["ltx2_images"] = None
|
||||
request["inputs"]["image_path"] = None
|
||||
|
||||
def apply_audio(
|
||||
self,
|
||||
request_kwargs: dict,
|
||||
request: dict,
|
||||
segment_idx: int,
|
||||
audio_lps: float,
|
||||
) -> None:
|
||||
"""Seed next-segment kwargs with clean audio latents + denoise mask.
|
||||
"""Seed the next-segment request with clean audio latents + denoise mask.
|
||||
|
||||
When audio conditioning is longer than video, extend audio
|
||||
generation and shift video RoPE forward so the audio prefix
|
||||
@@ -159,9 +145,9 @@ class ContinuationState:
|
||||
audio_extra = max(0, AUDIO_CONDITIONING_NUM_FRAMES - LTX2_VIDEO_CONDITIONING_NUM_FRAMES)
|
||||
if audio_extra > 0:
|
||||
audio_num_frames = NUM_FRAMES + audio_extra
|
||||
request_kwargs["audio_num_frames"] = (audio_num_frames)
|
||||
request["extensions"]["audio_num_frames"] = (audio_num_frames)
|
||||
prefix_sec = float(audio_extra) / 24.0
|
||||
request_kwargs["video_position_offset_sec"] = prefix_sec
|
||||
request["extensions"]["video_position_offset_sec"] = prefix_sec
|
||||
|
||||
new_duration = float(NUM_FRAMES + audio_extra) / 24.0
|
||||
total_T = max(
|
||||
@@ -179,8 +165,8 @@ class ContinuationState:
|
||||
mask = torch.ones((B, 1, total_T, 1), dtype=torch.float32)
|
||||
mask[:, :, :audio_cond_T, :] = (1.0 - AUDIO_CONDITIONING_STRENGTH)
|
||||
|
||||
request_kwargs["ltx2_audio_clean_latent"] = clean
|
||||
request_kwargs["ltx2_audio_denoise_mask"] = mask
|
||||
request["extensions"]["ltx2_audio_clean_latent"] = clean
|
||||
request["extensions"]["ltx2_audio_denoise_mask"] = mask
|
||||
|
||||
def save_video(self, frames: list) -> None:
|
||||
"""Snapshot trailing N frames as PIL images for next-segment conditioning."""
|
||||
@@ -202,7 +188,7 @@ class ContinuationState:
|
||||
self.audio_latents = latents.detach().clone().cpu()
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
class LTX2GenerationBackend:
|
||||
"""Single-GPU LTX2 generator with continuation state.
|
||||
|
||||
Caller must set ``os.environ["CUDA_VISIBLE_DEVICES"]`` before
|
||||
@@ -324,7 +310,7 @@ class VideoGenerationWorker:
|
||||
),
|
||||
)
|
||||
|
||||
self.generator = VideoGenerator.from_pretrained(config=generator_config)
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
print(f"[GPU {self.gpu_id}] After model load: {self._gpu_mem()}")
|
||||
|
||||
lora_stack = DREAMVERSE_LORA_STACK or ([(DREAMVERSE_LORA_PATH,
|
||||
@@ -421,7 +407,7 @@ class VideoGenerationWorker:
|
||||
return
|
||||
|
||||
loader = ComponentLoader.for_module_type("audio_encoder", "diffusers")
|
||||
enc = loader.load(audio_vae_path, self.generator.fastvideo_args)
|
||||
enc = loader.load(audio_vae_path, self.generator.resolved_config)
|
||||
target = getattr(enc, "model", enc)
|
||||
|
||||
proc = AudioProcessor(
|
||||
@@ -478,51 +464,59 @@ class VideoGenerationWorker:
|
||||
|
||||
prompt = self._inject_style_trigger(prompt)
|
||||
|
||||
request_kwargs = dict(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
save_video=False,
|
||||
height=FRAME_HEIGHT,
|
||||
width=FRAME_WIDTH,
|
||||
num_frames=NUM_FRAMES,
|
||||
fps=24,
|
||||
num_inference_steps=NUM_INFERENCE_STEPS,
|
||||
guidance_scale=1.0,
|
||||
seed=10,
|
||||
ltx2_image_crf=0.0,
|
||||
image_path=image_path if segment_idx == 1 else None,
|
||||
return_continuation_state=False,
|
||||
)
|
||||
request = {
|
||||
"prompt": prompt,
|
||||
"negative_prompt": "",
|
||||
"inputs": {
|
||||
"image_path": image_path if segment_idx == 1 else None
|
||||
},
|
||||
"sampling": {
|
||||
"height": FRAME_HEIGHT,
|
||||
"width": FRAME_WIDTH,
|
||||
"num_frames": NUM_FRAMES,
|
||||
"fps": 24,
|
||||
"num_inference_steps": NUM_INFERENCE_STEPS,
|
||||
"guidance_scale": 1.0,
|
||||
"seed": 10,
|
||||
},
|
||||
"output": {
|
||||
"save_video": False
|
||||
},
|
||||
"extensions": {
|
||||
"ltx2_image_crf": 0.0,
|
||||
"return_continuation_state": False,
|
||||
},
|
||||
}
|
||||
|
||||
if reset_conditioning:
|
||||
self.continuation.clear()
|
||||
|
||||
audio_lps = (DEFAULT_LTX2_AUDIO_SAMPLE_RATE / DEFAULT_LTX2_AUDIO_HOP_LENGTH / DEFAULT_LTX2_AUDIO_DOWNSAMPLE)
|
||||
|
||||
# Phase 1: seed kwargs with prior-segment conditioning.
|
||||
self.continuation.apply_video(request_kwargs, segment_idx)
|
||||
self.continuation.apply_audio(request_kwargs, segment_idx, audio_lps)
|
||||
# Phase 1: seed the request with prior-segment conditioning.
|
||||
self.continuation.apply_video(request, segment_idx)
|
||||
self.continuation.apply_audio(request, segment_idx, audio_lps)
|
||||
|
||||
# Phase 2: generate.
|
||||
t0 = time.perf_counter()
|
||||
result = self.generator.generate_video(**request_kwargs)
|
||||
result = self.generator.generate(request)
|
||||
torch.cuda.synchronize()
|
||||
timings["generation_ms"] = (time.perf_counter() - t0) * 1000
|
||||
|
||||
if not isinstance(result, dict):
|
||||
raise RuntimeError("Expected dictionary output from generate_video.")
|
||||
frames = result.get("frames")
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("Expected a single GenerationResult from generate.")
|
||||
frames = result.frames
|
||||
if not isinstance(frames, list) or len(frames) == 0:
|
||||
raise RuntimeError("Generation did not return frames.")
|
||||
audio = result.get("audio")
|
||||
audio_sample_rate = result.get("audio_sample_rate")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
# LTX2 audio decoding stage uses 24kHz output by default.
|
||||
audio_sample_rate = 24000
|
||||
print(f"[GPU {self.gpu_id}] audio_sample_rate missing from result; "
|
||||
f"defaulting to {audio_sample_rate}Hz")
|
||||
|
||||
timings["generation_time_ms"] = result.get("generation_time", 0.0) * 1000
|
||||
timings["generation_time_ms"] = (result.generation_time or 0.0) * 1000
|
||||
|
||||
# Phase 3: snapshot continuation state for the next segment.
|
||||
t_save_start = time.perf_counter()
|
||||
@@ -560,7 +554,7 @@ class VideoGenerationWorker:
|
||||
self,
|
||||
audio: object,
|
||||
audio_sample_rate: int | None,
|
||||
result: dict,
|
||||
result: "GenerationResult",
|
||||
segment_idx: int,
|
||||
) -> torch.Tensor | None:
|
||||
"""Pick which tensor to cache for next-segment audio conditioning."""
|
||||
@@ -575,7 +569,7 @@ class VideoGenerationWorker:
|
||||
f"for segment {segment_idx + 1}")
|
||||
return re_encoded
|
||||
return None
|
||||
audio_latents = result.get("ltx2_audio_latents")
|
||||
audio_latents = result.extra.get("ltx2_audio_latents")
|
||||
if audio_latents is not None:
|
||||
print(f"[GPU {self.gpu_id}] Cached audio latents "
|
||||
f"shape={tuple(audio_latents.shape)} "
|
||||
@@ -0,0 +1,294 @@
|
||||
"""FastH3 model lifecycle and first-frame continuation for DreamVerse."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import os
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from dreamverse.config import DREAMVERSE_SP_SIZE
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from PIL.Image import Image
|
||||
|
||||
|
||||
def _required_config_str(model_config: dict, field_name: str) -> str:
|
||||
"""Read one required non-empty string from a DreamVerse model profile."""
|
||||
value = model_config.get(field_name)
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError(f"FastH3 model configuration requires `{field_name}`.")
|
||||
return value.strip()
|
||||
|
||||
|
||||
class MiniMaxH3GenerationBackend:
|
||||
"""Run the VSA data-free FastH3 adapter and retain one continuation frame."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.generator: Any | None = None
|
||||
self.model_config: dict = {}
|
||||
self.continuation_image: Image | None = None
|
||||
|
||||
def _gpu_mem(self) -> str:
|
||||
allocated_gib = torch.cuda.memory_allocated() / 1024**3
|
||||
reserved_gib = torch.cuda.memory_reserved() / 1024**3
|
||||
return f"alloc={allocated_gib:.2f}GiB, reserved={reserved_gib:.2f}GiB"
|
||||
|
||||
@staticmethod
|
||||
def _configure_environment(attention_backend: str) -> None:
|
||||
"""Apply the fixed boot-time switches from the FastH3 reference recipe."""
|
||||
os.environ.update({
|
||||
"FASTVIDEO_ATTENTION_BACKEND": attention_backend,
|
||||
"FASTVIDEO_FA4": "1",
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all",
|
||||
"FASTVIDEO_VSA_SM100A": "0",
|
||||
})
|
||||
os.environ.pop("FASTVIDEO_INFERENCE_TORCH_COMPILE", None)
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Download the fixed Preview adapter and load the FastH3 generator.
|
||||
|
||||
The model profile owns the base checkpoint, adapter file, attention
|
||||
backend, and generation geometry. The backend translates that profile
|
||||
into FastVideo's typed generator configuration.
|
||||
"""
|
||||
if model_config is not None:
|
||||
self.model_config = dict(model_config)
|
||||
if not self.model_config:
|
||||
raise ValueError("FastH3 initialization requires a model configuration.")
|
||||
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
self.clear_conditioning()
|
||||
model_path = _required_config_str(self.model_config, "model_path")
|
||||
adapter_repo = _required_config_str(self.model_config, "adapter_repo")
|
||||
adapter_filename = _required_config_str(self.model_config, "adapter_filename")
|
||||
attention_backend = _required_config_str(self.model_config, "attention_backend")
|
||||
self._configure_environment(attention_backend)
|
||||
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
AttentionConfig,
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GeneratorConfig,
|
||||
MiniMaxH3Options,
|
||||
OffloadConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
|
||||
adapter_path = hf_hub_download(repo_id=adapter_repo, filename=adapter_filename)
|
||||
use_vsa = attention_backend == "VIDEO_SPARSE_ATTN_H3"
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=model_path,
|
||||
pipeline=PipelineSelection(
|
||||
components=ComponentConfig(lora_path=adapter_path, lora_strength=1.0),
|
||||
model=MiniMaxH3Options(vae_parallel_decode=True, vae_parallel_decode_strategy="gather"),
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=DREAMVERSE_SP_SIZE,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=DREAMVERSE_SP_SIZE),
|
||||
offload=OffloadConfig(
|
||||
dit=False,
|
||||
dit_layerwise=False,
|
||||
text_encoder=True,
|
||||
image_encoder=True,
|
||||
vae=True,
|
||||
pin_cpu_memory=True,
|
||||
),
|
||||
compile=CompileConfig(enabled=False, vae_enabled=True, regional=attention_backend == "FLASH_ATTN"),
|
||||
attention=AttentionConfig(
|
||||
backend=attention_backend,
|
||||
vsa_sparsity=0.9 if use_vsa else None,
|
||||
vsa_tile_size=64 if use_vsa else None,
|
||||
),
|
||||
use_fsdp_inference=False,
|
||||
),
|
||||
)
|
||||
|
||||
print(f"[GPU {self.gpu_id}] Loading FastH3 model: {model_path}")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 adapter: {adapter_repo}/{adapter_filename}")
|
||||
print(f"[GPU {self.gpu_id}] Before model load: {self._gpu_mem()}")
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
print(f"[GPU {self.gpu_id}] FastH3 loaded: {self._gpu_mem()} (warmup pending)")
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release the FastVideo generator and cached continuation image."""
|
||||
self.clear_conditioning()
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
"""Release the first-frame image retained for the next segment."""
|
||||
if self.continuation_image is not None:
|
||||
self.continuation_image.close()
|
||||
self.continuation_image = None
|
||||
|
||||
@staticmethod
|
||||
def _load_rgb_image(image_path: str) -> Image:
|
||||
"""Load an image into an independent RGB buffer with no open file handle."""
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(image_path) as image:
|
||||
return image.convert("RGB").copy()
|
||||
|
||||
def _select_conditioning_image(
|
||||
self,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> tuple[Image | None, bool]:
|
||||
"""Select the initial upload or retained last frame for one segment."""
|
||||
if reset_conditioning:
|
||||
self.clear_conditioning()
|
||||
if segment_idx > 1 and self.continuation_image is not None:
|
||||
return self.continuation_image.copy(), True
|
||||
if segment_idx > 1 and not reset_conditioning:
|
||||
raise RuntimeError(f"FastH3 segment {segment_idx} requires a retained continuation frame.")
|
||||
if segment_idx == 1 and image_path:
|
||||
return self._load_rgb_image(image_path), False
|
||||
return None, False
|
||||
|
||||
def _build_request(self, prompt: str, conditioning_image: Image | None):
|
||||
"""Build the typed FastVideo request owned by the FastH3 profile."""
|
||||
from fastvideo.api import GenerationRequest, InputConfig, OutputConfig, SamplingConfig
|
||||
|
||||
return GenerationRequest(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
inputs=InputConfig(pil_image=conditioning_image),
|
||||
sampling=SamplingConfig(
|
||||
height=int(self.model_config["height"]),
|
||||
width=int(self.model_config["width"]),
|
||||
num_frames=int(self.model_config["num_frames"]),
|
||||
fps=24,
|
||||
num_inference_steps=int(self.model_config["num_inference_steps"]),
|
||||
guidance_scale=1.0,
|
||||
batch_cfg=False,
|
||||
seed=int(self.model_config["seed"]),
|
||||
),
|
||||
output=OutputConfig(save_video=False, return_frames=True),
|
||||
)
|
||||
|
||||
def _save_continuation_frame(self, frames: list) -> None:
|
||||
"""Retain the last decoded frame as first-frame conditioning."""
|
||||
from PIL import Image
|
||||
|
||||
self.clear_conditioning()
|
||||
self.continuation_image = Image.fromarray(np.ascontiguousarray(frames[-1])).convert("RGB")
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one synchronized FastH3 segment and retain its last frame.
|
||||
|
||||
Later segments use MiniMax H3's first-frame-to-video path. The first
|
||||
conditioned frame and its matching audio duration are trimmed before
|
||||
streaming so adjacent segments do not duplicate media.
|
||||
"""
|
||||
if self.generator is None:
|
||||
raise RuntimeError("FastH3 generator is not initialized.")
|
||||
conditioning_image, uses_continuation = self._select_conditioning_image(
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
request = self._build_request(prompt, conditioning_image)
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
result = self.generator.generate(request)
|
||||
finally:
|
||||
if conditioning_image is not None:
|
||||
conditioning_image.close()
|
||||
torch.cuda.synchronize()
|
||||
generation_ms = (time.perf_counter() - started) * 1000.0
|
||||
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("FastH3 returned multiple results for one DreamVerse segment.")
|
||||
frames = result.frames
|
||||
if not isinstance(frames, list) or not frames:
|
||||
raise RuntimeError("FastH3 generation did not return decoded frames.")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
raise RuntimeError("FastH3 returned audio without an audio sample rate.")
|
||||
|
||||
save_started = time.perf_counter()
|
||||
self._save_continuation_frame(frames)
|
||||
save_conditioning_ms = (time.perf_counter() - save_started) * 1000.0
|
||||
timings = {
|
||||
"generation_ms": generation_ms,
|
||||
"generation_time_ms": float(result.generation_time or 0.0) * 1000.0,
|
||||
"save_conditioning_ms": save_conditioning_ms,
|
||||
"e2e_latency_ms": (time.perf_counter() - started) * 1000.0,
|
||||
}
|
||||
trim_frames = 1 if uses_continuation else 0
|
||||
print(f"[GPU {self.gpu_id}] FastH3 segment {segment_idx}: "
|
||||
f"{len(frames)} frames, gen={generation_ms:.0f}ms, "
|
||||
f"save_conditioning={save_conditioning_ms:.0f}ms, "
|
||||
f"e2e={timings['e2e_latency_ms']:.0f}ms")
|
||||
return StepResult(
|
||||
frames=frames,
|
||||
audio=audio,
|
||||
audio_sample_rate=audio_sample_rate,
|
||||
timings=timings,
|
||||
head_trim_frames=trim_frames,
|
||||
head_trim_audio_frames=trim_frames,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
"""Compile the FastH3 text and first-frame paths before readiness."""
|
||||
warmup_prompt = (prompt or "").strip()
|
||||
if not warmup_prompt:
|
||||
raise RuntimeError("Startup warmup prompt must be non-empty.")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup starting "
|
||||
"(synthetic segments: text-to-video, first-frame-to-video)")
|
||||
started = time.perf_counter()
|
||||
text_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
first_frame_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
total_ms = (time.perf_counter() - started) * 1000.0
|
||||
self.clear_conditioning()
|
||||
text_ms = float(text_result.timings.get("e2e_latency_ms", 0.0))
|
||||
first_frame_ms = float(first_frame_result.timings.get("e2e_latency_ms", 0.0))
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup complete: "
|
||||
f"text_to_video={text_ms:.0f}ms, "
|
||||
f"first_frame_to_video={first_frame_ms:.0f}ms, "
|
||||
f"total={total_ms:.0f}ms")
|
||||
return {
|
||||
"warmup_text_to_video_ms": text_ms,
|
||||
"warmup_first_frame_to_video_ms": first_frame_ms,
|
||||
"warmup_total_ms": total_ms,
|
||||
}
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
"""Reject runtime LoRA mutation because FastH3 uses one startup adapter."""
|
||||
del stack
|
||||
raise RuntimeError("FastH3 uses its fixed startup adapter and does not support runtime LoRA changes.")
|
||||
@@ -30,7 +30,7 @@ from dreamverse.session_init_image import cleanup_session_init_image, persist_se
|
||||
from dreamverse.worker_ipc import MediaChunk, MediaComplete, MediaInit
|
||||
|
||||
from dreamverse.config import (
|
||||
DEFAULT_MODEL_ID,
|
||||
ACTIVE_MODEL_ID,
|
||||
GENERATION_SEGMENT_CAP,
|
||||
PROMPT_AUTO_SLEEP_MS,
|
||||
PROMPT_AUTO_TIMEOUT_MS,
|
||||
@@ -264,7 +264,7 @@ class SessionController:
|
||||
timeout_task = asyncio.create_task(session_timeout())
|
||||
|
||||
# Join the engine on this GPU.
|
||||
await slot.join_user(client_id, model_id=DEFAULT_MODEL_ID)
|
||||
await slot.join_user(client_id, model_id=ACTIVE_MODEL_ID)
|
||||
|
||||
# Notify client they're connected to a GPU.
|
||||
await ws_send_json({
|
||||
|
||||
@@ -2,13 +2,14 @@ from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
import pytest
|
||||
|
||||
SERVER_DIR = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
def _load_config_module():
|
||||
def _load_config_module() -> ModuleType:
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"server_config_test_module",
|
||||
SERVER_DIR / "config.py",
|
||||
@@ -20,7 +21,7 @@ def _load_config_module():
|
||||
return module
|
||||
|
||||
|
||||
def _set_required_prompt_keys(monkeypatch):
|
||||
def _set_required_prompt_keys(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("CEREBRAS_API_KEY", "cerebras-key")
|
||||
monkeypatch.setenv("GROQ_API_KEY", "groq-key")
|
||||
|
||||
@@ -150,3 +151,38 @@ def test_config_rejects_invalid_prompt_provider(monkeypatch):
|
||||
|
||||
with pytest.raises(RuntimeError, match="Invalid FASTVIDEO_PROMPT_PROVIDER"):
|
||||
_load_config_module()
|
||||
|
||||
|
||||
def test_config_registers_vsa_datafree_fasth3_profile(monkeypatch):
|
||||
"""The FastH3 registry entry owns the complete fixed Preview recipe."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.MODEL_REGISTRY["fast-h3"] == {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
def test_config_uses_fasth3_sequence_parallel_default(monkeypatch):
|
||||
"""Selecting FastH3 defaults DreamVerse to its four-GPU topology."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
monkeypatch.setenv("DREAMVERSE_MODEL_ID", "fast-h3")
|
||||
monkeypatch.delenv("DREAMVERSE_SP_SIZE", raising=False)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.ACTIVE_MODEL_ID == "fast-h3"
|
||||
assert module.MODEL_CONFIG["generation_backend"] == "minimax_h3"
|
||||
assert module.DREAMVERSE_SP_SIZE == 4
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
from dreamverse.generation_worker import _create_generation_backend
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
|
||||
def test_create_generation_backend_ltx2_module_import():
|
||||
backend = _create_generation_backend("ltx2", gpu_id=3)
|
||||
|
||||
assert isinstance(backend, LTX2GenerationBackend)
|
||||
assert backend.gpu_id == 3
|
||||
@@ -63,6 +63,14 @@ def test_get_available_gpus_defaults_to_first_visible_device(monkeypatch):
|
||||
assert gpu_pool.get_available_gpus() == [3]
|
||||
|
||||
|
||||
def test_get_available_gpus_defaults_to_active_model_sequence_parallel_size(monkeypatch):
|
||||
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1,2,3,4")
|
||||
monkeypatch.delenv("FASTVIDEO_GPU_COUNT", raising=False)
|
||||
monkeypatch.setattr(gpu_pool, "DREAMVERSE_SP_SIZE", 4)
|
||||
|
||||
assert gpu_pool.get_available_gpus() == [0, 1, 2, 3]
|
||||
|
||||
|
||||
def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
|
||||
monkeypatch.setenv("FASTVIDEO_GPU_COUNT", "zero")
|
||||
@@ -71,6 +79,23 @@ def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
gpu_pool.get_available_gpus()
|
||||
|
||||
|
||||
def test_join_user_failed_reload_marks_model_uninitialized(monkeypatch):
|
||||
"""A failed model reload forces the next join to reload a model."""
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.current_model_id = "fast-ltx2"
|
||||
|
||||
async def fake_send_command(command, timeout):
|
||||
del command, timeout
|
||||
return gpu_pool.WorkerError(user_id="__reload__", message="load failed")
|
||||
|
||||
monkeypatch.setattr(slot, "_send_command", fake_send_command)
|
||||
|
||||
with pytest.raises(RuntimeError, match="Model reload failed"):
|
||||
asyncio.run(slot.join_user("client-id", model_id="fast-h3"))
|
||||
|
||||
assert slot.current_model_id is None
|
||||
|
||||
|
||||
def test_send_command_raises_on_worker_death():
|
||||
"""A worker that consumes a command and exits without replying must
|
||||
surface as RuntimeError via sentinel detection, not after the long
|
||||
@@ -92,9 +117,9 @@ def test_send_command_raises_on_worker_death():
|
||||
ready = resp_q.get(timeout=30.0)
|
||||
assert ready == "READY"
|
||||
|
||||
async def runner():
|
||||
async def runner() -> None:
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.process = proc
|
||||
slot.process = proc # type: ignore[assignment]
|
||||
slot.command_queue = cmd_q
|
||||
slot.response_queue = resp_q
|
||||
|
||||
|
||||
@@ -13,15 +13,14 @@ FORBIDDEN_PREFIXES = (
|
||||
"fastvideo.models",
|
||||
"fastvideo.layers",
|
||||
"fastvideo.worker",
|
||||
"fastvideo.fastvideo_args",
|
||||
)
|
||||
ALLOWED_INTERNAL_IMPORTS = {
|
||||
(
|
||||
"video_generation.py",
|
||||
"ltx2_generation.py",
|
||||
"fastvideo.models.audio.ltx2_audio_processing",
|
||||
),
|
||||
(
|
||||
"video_generation.py",
|
||||
"ltx2_generation.py",
|
||||
"fastvideo.models.loader.component_loader",
|
||||
),
|
||||
}
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
import dreamverse.generation_worker as generation_worker
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
|
||||
FASTH3_MODEL_CONFIG = {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
class _RecordingGenerator:
|
||||
"""Record typed requests and return small synchronized media fixtures."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.requests: list[Any] = []
|
||||
self.conditioning_pixels: list[np.ndarray | None] = []
|
||||
|
||||
def generate(self, request):
|
||||
"""Capture the request and return two tiny video frames with audio."""
|
||||
self.requests.append(request)
|
||||
conditioning_image = request.inputs.pil_image
|
||||
self.conditioning_pixels.append(
|
||||
None if conditioning_image is None else np.asarray(conditioning_image).copy())
|
||||
frames = [
|
||||
np.full((2, 3, 3), 10, dtype=np.uint8),
|
||||
np.full((2, 3, 3), 20, dtype=np.uint8),
|
||||
]
|
||||
return SimpleNamespace(
|
||||
frames=frames,
|
||||
audio=np.zeros((2, 16), dtype=np.float32),
|
||||
audio_sample_rate=44100,
|
||||
generation_time=0.25,
|
||||
)
|
||||
|
||||
|
||||
def test_initialize_builds_vsa_datafree_fasth3_generator(monkeypatch):
|
||||
"""Initialization translates the DreamVerse profile into typed FastVideo config."""
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
captured = {}
|
||||
fake_generator = SimpleNamespace(shutdown=lambda: None)
|
||||
|
||||
def fake_from_config(config):
|
||||
captured["config"] = config
|
||||
return fake_generator
|
||||
|
||||
def fake_download(**kwargs):
|
||||
captured["download"] = kwargs
|
||||
return f"/models/{kwargs['filename']}"
|
||||
|
||||
monkeypatch.setattr("huggingface_hub.hf_hub_download", fake_download)
|
||||
monkeypatch.setattr(VideoGenerator, "from_config", fake_from_config)
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.DREAMVERSE_SP_SIZE", 4)
|
||||
monkeypatch.setenv("FASTVIDEO_ATTENTION_BACKEND", "test-attention")
|
||||
monkeypatch.setenv("FASTVIDEO_FA4", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_MINIMAX_H3_FUSIONS", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_VSA_SM100A", "1")
|
||||
monkeypatch.setenv("FASTVIDEO_INFERENCE_TORCH_COMPILE", "1")
|
||||
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
monkeypatch.setattr(backend, "_gpu_mem", lambda: "alloc=0.00GiB, reserved=0.00GiB")
|
||||
backend.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
config = captured["config"]
|
||||
assert captured["download"] == {
|
||||
"repo_id": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"filename": "vsa-datafree/adapter_model.safetensors",
|
||||
}
|
||||
assert config.model_path == "MiniMaxAI/MiniMax-H3"
|
||||
assert config.pipeline.components.lora_path.endswith("vsa-datafree/adapter_model.safetensors")
|
||||
assert config.pipeline.components.lora_strength == 1.0
|
||||
assert config.pipeline.experimental == {}
|
||||
assert config.engine.attention.backend == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert config.engine.attention.vsa_sparsity == 0.9
|
||||
assert config.engine.attention.vsa_tile_size == 64
|
||||
assert config.engine.compile.regional is False
|
||||
assert config.pipeline.model.vae_parallel_decode is True
|
||||
assert config.pipeline.model.vae_parallel_decode_strategy == "gather"
|
||||
assert config.engine.num_gpus == 4
|
||||
assert config.engine.parallelism.tp_size == 1
|
||||
assert config.engine.parallelism.sp_size == 4
|
||||
assert config.engine.offload.dit is False
|
||||
assert config.engine.offload.dit_layerwise is False
|
||||
assert config.engine.offload.text_encoder is True
|
||||
assert config.engine.offload.vae is True
|
||||
assert config.engine.compile.vae_enabled is True
|
||||
assert config.engine.use_fsdp_inference is False
|
||||
assert os.environ["FASTVIDEO_ATTENTION_BACKEND"] == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert os.environ["FASTVIDEO_FA4"] == "1"
|
||||
assert os.environ["FASTVIDEO_MINIMAX_H3_FUSIONS"] == "all"
|
||||
assert os.environ["FASTVIDEO_VSA_SM100A"] == "0"
|
||||
assert "FASTVIDEO_INFERENCE_TORCH_COMPILE" not in os.environ
|
||||
|
||||
|
||||
def test_initialize_selects_declared_generation_backend(monkeypatch):
|
||||
"""The GPU worker constructs the backend that the active model profile declares."""
|
||||
from unittest.mock import Mock
|
||||
|
||||
selected_backend = Mock()
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: selected_backend,
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend_name == "minimax_h3"
|
||||
assert worker.backend is selected_backend
|
||||
selected_backend.initialize.assert_called_once_with(FASTH3_MODEL_CONFIG)
|
||||
|
||||
|
||||
def test_initialize_failure_clears_backend_ownership(monkeypatch):
|
||||
"""A failed family change leaves the GPU worker explicitly uninitialized."""
|
||||
ltx_backend = SimpleNamespace(initialize=lambda config: None, shutdown=lambda: None)
|
||||
|
||||
def fail_initialize(config):
|
||||
del config
|
||||
raise RuntimeError("load failed")
|
||||
|
||||
fasth3_backend = SimpleNamespace(
|
||||
initialize=fail_initialize,
|
||||
shutdown=lambda: None,
|
||||
)
|
||||
backends = {
|
||||
"ltx2": ltx_backend,
|
||||
"minimax_h3": fasth3_backend,
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: backends[backend_name],
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
worker.initialize({"generation_backend": "ltx2"})
|
||||
|
||||
with pytest.raises(RuntimeError, match="load failed"):
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend is None
|
||||
assert worker.backend_name is None
|
||||
assert worker.model_config == {"generation_backend": "ltx2"}
|
||||
|
||||
|
||||
def test_generate_step_uses_last_frame_for_continuation(monkeypatch):
|
||||
"""A later segment receives the prior segment's last decoded frame."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
first_result = backend.generate_step(
|
||||
"first prompt",
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
second_result = backend.generate_step(
|
||||
"second prompt",
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
|
||||
first_request = backend.generator.requests[0]
|
||||
assert first_request.inputs.pil_image is None
|
||||
assert first_request.negative_prompt == ""
|
||||
assert first_request.sampling.height == 768
|
||||
assert first_request.sampling.width == 1344
|
||||
assert first_request.sampling.num_frames == 124
|
||||
assert first_request.sampling.num_inference_steps == 5
|
||||
assert first_request.sampling.fps == 24
|
||||
assert first_request.sampling.guidance_scale == 1.0
|
||||
assert first_request.sampling.batch_cfg is False
|
||||
assert first_request.sampling.seed == 1000
|
||||
assert first_request.output.save_video is False
|
||||
assert first_request.output.return_frames is True
|
||||
assert backend.generator.conditioning_pixels[1].tolist() == np.full((2, 3, 3), 20).tolist()
|
||||
assert first_result.head_trim_frames == 0
|
||||
assert first_result.head_trim_audio_frames == 0
|
||||
assert second_result.head_trim_frames == 1
|
||||
assert second_result.head_trim_audio_frames == 1
|
||||
assert second_result.audio_sample_rate == 44100
|
||||
|
||||
|
||||
def test_generate_step_reset_uses_text_to_video_path(monkeypatch):
|
||||
"""Resetting continuation produces an unconditioned text-to-video request."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
backend.generate_step("first prompt", 1, None, True)
|
||||
reset_result = backend.generate_step("reset prompt", 2, None, True)
|
||||
|
||||
assert backend.generator.requests[-1].inputs.pil_image is None
|
||||
assert reset_result.head_trim_frames == 0
|
||||
assert reset_result.head_trim_audio_frames == 0
|
||||
|
||||
|
||||
def test_generate_step_missing_continuation_frame(monkeypatch):
|
||||
"""A later segment fails when no reset or retained frame defines its input."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
|
||||
with pytest.raises(RuntimeError, match="requires a retained continuation frame"):
|
||||
backend.generate_step("later prompt", 2, None, False)
|
||||
|
||||
assert backend.generator.requests == []
|
||||
|
||||
|
||||
def test_warmup_exercises_text_and_first_frame_paths(monkeypatch):
|
||||
"""Warmup covers both request shapes used by a DreamVerse session."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
timings = backend.warmup("warmup prompt")
|
||||
|
||||
assert backend.generator.conditioning_pixels[0] is None
|
||||
assert backend.generator.conditioning_pixels[1] is not None
|
||||
assert backend.continuation_image is None
|
||||
assert "warmup_text_to_video_ms" in timings
|
||||
assert "warmup_first_frame_to_video_ms" in timings
|
||||
@@ -331,11 +331,11 @@ def test_rewrite_prompt_sequence_accepts_numbered_prose_output():
|
||||
]
|
||||
|
||||
|
||||
def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
def test_enhance_prompt_uses_groq_when_it_returns_first():
|
||||
enhancer = _build_staged_enhancer(
|
||||
cerebras_payload=_chat_payload_with_content('{"prompt":"Cerebras prompt"}'),
|
||||
groq_payload=_chat_payload_with_content('{"prompt":"Groq prompt"}'),
|
||||
cerebras_delay_s=0.01,
|
||||
cerebras_delay_s=0.08,
|
||||
groq_delay_s=0.01,
|
||||
)
|
||||
|
||||
@@ -346,12 +346,12 @@ def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.provider == "cerebras"
|
||||
assert result.provider == "groq"
|
||||
assert result.model == "gpt-test"
|
||||
assert result.prompt == "Cerebras prompt"
|
||||
assert result.prompt == "Groq prompt"
|
||||
assert enhancer.get_provider_success_counts() == {
|
||||
"cerebras": 1,
|
||||
"groq": 0,
|
||||
"cerebras": 0,
|
||||
"groq": 1,
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -70,7 +70,7 @@
|
||||
<mxCell id="dispatcher" value="command dispatcher

gpu_worker_process() branches on
CommandType; asserts payload type

INIT / WARMUP / RELOAD_MODEL
USER_JOIN / USER_STEP / USER_LEAVE
SHUTDOWN" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe6cc;strokeColor=#d79b00;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
video_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
ltx2_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="stream_av" value="stream_fmp4()
av_streaming.py:121

trims overlap, pipes to ffmpeg,
publishes StreamInit / StreamChunk /
StreamComplete via callback" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#b1d8d7;strokeColor=#23445d;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
@@ -79,13 +79,13 @@
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-2" value="" style="edgeStyle=none;html=1;" parent="1" source="generator" target="Ot8BU52QTIb4EhyRSe7I-1" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
video_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
ltx2_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="ffmpeg" value="ffmpeg subprocess

libx264 / *_nvenc
fragmented mp4" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="800" y="1300" width="260" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="caches" value="ContinuationState
video_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="caches" value="ContinuationState
ltx2_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_cp" value="acquire" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#6c8ebf;endArrow=classic;fontSize=11;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="client" target="pool" edge="1">
|
||||
@@ -207,7 +207,7 @@
|
||||
</Array>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="e_dsg" value="generator.generate_video()" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#9673a6;endArrow=classic;fontSize=10;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="do_step" target="generator" edge="1">
|
||||
<mxCell id="e_dsg" value="generator.generate()" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#9673a6;endArrow=classic;fontSize=10;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="do_step" target="generator" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_dscache" value="read / write" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#d6b656;endArrow=classic;startArrow=classic;fontSize=10;exitX=0;exitY=0.8;exitDx=0;exitDy=0;entryX=1;entryY=0.2;entryDx=0;entryDy=0;" parent="1" source="do_step" target="caches" edge="1">
|
||||
@@ -250,7 +250,7 @@
|
||||
<mxPoint x="690" y="880"/>
|
||||
</Array>
|
||||
</mxCell>
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender video_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender ltx2_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="39" y="-200" width="270" height="380" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-1" value="FastVideo video_generator" style="whiteSpace=wrap;html=1;fontSize=11;fillColor=#e1d5e7;strokeColor=#9673a6;rounded=1;" parent="1" vertex="1">
|
||||
@@ -389,10 +389,10 @@
|
||||
<mxCell id="cw2" value="from fastvideo.entrypoints.video_generator import VideoGenerator
from fastvideo.models.dits.ltx2 import DEFAULT_LTX2_AUDIO_*

** Dreamverse reaches into fastvideo internals here **" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="695" width="550" height="60" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (video_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (ltx2_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="765" width="550" height="95" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (video_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (ltx2_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="870" width="550" height="55" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw5" value="enter main worker loop → waits for JOIN_USER / USER_STEP / LEAVE" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#c8e6c9;strokeColor=#388e3c;fontSize=11;fontStyle=1;fontFamily=monospace;" parent="1" vertex="1">
|
||||
@@ -534,7 +534,7 @@
|
||||
<mxPoint x="1040" y="1610" as="targetPoint"/>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (video_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (ltx2_generation.py:380)
 → generator.generate()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="955" y="1640" width="180" height="70" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="dm11" value="10b. resp_q.put(MediaInit / MediaChunk / MediaComplete / StepComplete)" style="endArrow=classic;html=1;strokeColor=#b85450;fontSize=10;labelBackgroundColor=#ffffff;" parent="1" edge="1">
|
||||
|
||||
|
Before Width: | Height: | Size: 85 KiB After Width: | Height: | Size: 85 KiB |
@@ -64,8 +64,8 @@ generator:
|
||||
# internal: pipeline_config.dit_config.quant_config = FP4Config()
|
||||
# set in gpu_pool.py:280 (via the legacy in-place mutation). The
|
||||
# public typed surface resolves "NVFP4" to NVFP4Config() and pins
|
||||
# it on dit_config in FastVideoArgs.__post_init__. Comment this
|
||||
# block out on hosts without flashinfer / NVFP4 hardware.
|
||||
# it on dit_config when resolution materializes the PipelineConfig.
|
||||
# Comment this block out on hosts without flashinfer / NVFP4 hardware.
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ test.describe('preset prompt generation', () => {
|
||||
// on a B200 plus encode/transfer time. The "Continuation flipped
|
||||
// to Generating + Leave button rendered" pair above is the proof
|
||||
// the integration works: FE → /readyz → /curated-presets → WS
|
||||
// /ws → BE → GPU pool → VideoGenerator.generate_video, all green.
|
||||
// /ws → BE → GPU pool → VideoGenerator.generate, all green.
|
||||
const video = page.locator('video').first();
|
||||
await expect(video).toHaveCount(1);
|
||||
});
|
||||
|
||||
@@ -9,6 +9,7 @@ from __future__ import annotations
|
||||
|
||||
import contextlib
|
||||
import logging
|
||||
import json
|
||||
import sqlite3
|
||||
import threading
|
||||
from pathlib import Path
|
||||
@@ -46,8 +47,13 @@ DEFAULT_SETTINGS: dict[str, Any] = {
|
||||
|
||||
|
||||
def _sqlite_row_get(row: sqlite3.Row, key: str, default: Any) -> Any:
|
||||
"""Like dict.get for sqlite3.Row (Row has no .get on Python 3.10)."""
|
||||
return row[key] if key in row else default # noqa: SIM401
|
||||
"""Like dict.get for sqlite3.Row (Row has no .get on Python 3.10).
|
||||
|
||||
NOTE: `key in row` tests Row *values*, not column names, so the membership
|
||||
check has to go through .keys() -- otherwise every lookup falls back to the
|
||||
default and jobs restored from the database lose their stored fields.
|
||||
"""
|
||||
return row[key] if key in row.keys() else default # noqa: SIM401, SIM118
|
||||
|
||||
|
||||
def _get_db_path(data_dir: Path) -> Path:
|
||||
@@ -83,6 +89,9 @@ def _migrate_db(conn: sqlite3.Connection) -> None:
|
||||
_add_column_if_missing(conn, "jobs", "fps", "INTEGER", "24")
|
||||
_add_column_if_missing(conn, "jobs", "workload_type", "TEXT", "'t2v'")
|
||||
_add_column_if_missing(conn, "jobs", "image_path", "TEXT", "''")
|
||||
_add_column_if_missing(conn, "jobs", "name", "TEXT", "''")
|
||||
_add_column_if_missing(conn, "jobs", "last_image_path", "TEXT", "''")
|
||||
_add_column_if_missing(conn, "jobs", "references_json", "TEXT", "''")
|
||||
_add_column_if_missing(conn, "jobs", "job_type", "TEXT", "'inference'")
|
||||
_add_column_if_missing(conn, "jobs", "data_path", "TEXT", "''")
|
||||
_add_column_if_missing(conn, "jobs", "max_train_steps", "INTEGER", "1000")
|
||||
@@ -242,7 +251,8 @@ class Database:
|
||||
self._execute(
|
||||
"""
|
||||
INSERT INTO jobs (
|
||||
id, model_id, prompt, workload_type, image_path, job_type, status,
|
||||
id, model_id, name, prompt, workload_type, image_path,
|
||||
last_image_path, references_json, job_type, status,
|
||||
created_at, started_at, finished_at, error, output_path, log_file_path,
|
||||
num_inference_steps, num_frames, height, width, guidance_scale,
|
||||
guidance_rescale, fps, seed, num_gpus, dit_cpu_offload,
|
||||
@@ -254,14 +264,17 @@ class Database:
|
||||
dmd_use_vsa, dmd_vsa_sparsity, dmd_denoising_steps,
|
||||
real_score_guidance_scale,
|
||||
generator_update_interval, real_score_model_path, fake_score_model_path
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
job["id"],
|
||||
job["model_id"],
|
||||
job.get("name", ""),
|
||||
job["prompt"],
|
||||
job.get("workload_type", "t2v"),
|
||||
job.get("image_path", ""),
|
||||
job.get("last_image_path", ""),
|
||||
json.dumps(job.get("references") or []),
|
||||
job.get("job_type", "inference"),
|
||||
job["status"],
|
||||
job["created_at"],
|
||||
@@ -540,9 +553,12 @@ def _row_to_job(row: sqlite3.Row) -> dict[str, Any]:
|
||||
result = {
|
||||
"id": row["id"],
|
||||
"model_id": row["model_id"],
|
||||
"name": _sqlite_row_get(row, "name", "") or "",
|
||||
"prompt": row["prompt"],
|
||||
"workload_type": _sqlite_row_get(row, "workload_type", "t2v"),
|
||||
"image_path": _sqlite_row_get(row, "image_path", "") or "",
|
||||
"last_image_path": _sqlite_row_get(row, "last_image_path", "") or "",
|
||||
"references": _sqlite_row_get(row, "references_json", "") or "",
|
||||
"job_type": _sqlite_row_get(row, "job_type", "inference"),
|
||||
"status": row["status"],
|
||||
"created_at": row["created_at"],
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
|
||||
import { skipWithoutMock } from './helpers';
|
||||
|
||||
test.describe('create job interactions', () => {
|
||||
skipWithoutMock();
|
||||
|
||||
for (const jobType of ['inference', 'finetuning', 'distillation']) {
|
||||
test(`${jobType} remains interactive after repeated dialog dismissals`, async ({ page }) => {
|
||||
await page.goto(`/${jobType}`);
|
||||
const trigger = page.getByRole('button', { name: 'Create Job', exact: true });
|
||||
const dialog = page.getByRole('dialog');
|
||||
|
||||
// Exercise both dismissal paths and reopen without reloading the page.
|
||||
for (const closeWithEscape of [false, true]) {
|
||||
await trigger.click();
|
||||
await page.getByRole('menuitem').first().click();
|
||||
await expect(dialog).toBeVisible();
|
||||
if (closeWithEscape) {
|
||||
await page.keyboard.press('Escape');
|
||||
} else {
|
||||
await dialog.getByRole('button', { name: 'Close', exact: true }).click();
|
||||
}
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
await expect(trigger).toBeFocused();
|
||||
}
|
||||
|
||||
await page.getByRole('link', { name: 'Datasets', exact: true }).click();
|
||||
await expect(page).toHaveURL(/\/datasets$/);
|
||||
});
|
||||
}
|
||||
|
||||
test('preserves keyboard menu dismissal and dialog focus trapping', async ({ page }) => {
|
||||
await page.goto('/inference');
|
||||
const trigger = page.getByRole('button', { name: 'Create Job', exact: true });
|
||||
await trigger.focus();
|
||||
await page.keyboard.press('Enter');
|
||||
const firstItem = page.getByRole('menuitem').first();
|
||||
await expect(firstItem).toBeFocused();
|
||||
await page.keyboard.press('Escape');
|
||||
await expect(page.getByRole('menu')).toBeHidden();
|
||||
await expect(trigger).toBeFocused();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
|
||||
await page.keyboard.press('Enter');
|
||||
await expect(firstItem).toBeFocused();
|
||||
await page.keyboard.press('Enter');
|
||||
const dialog = page.getByRole('dialog');
|
||||
await expect(dialog).toBeVisible();
|
||||
await expect(dialog.getByLabel('Name (optional)')).toBeFocused();
|
||||
|
||||
// Shift+Tab from the first field wraps to Close, then Tab wraps back.
|
||||
await page.keyboard.press('Shift+Tab');
|
||||
await expect(dialog.getByRole('button', { name: 'Close', exact: true })).toBeFocused();
|
||||
await page.keyboard.press('Tab');
|
||||
await expect(dialog.getByLabel('Name (optional)')).toBeFocused();
|
||||
await page.keyboard.press('Escape');
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(trigger).toBeFocused();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
});
|
||||
});
|
||||
@@ -1,6 +1,6 @@
|
||||
import { expect, test } from '@playwright/test';
|
||||
|
||||
import { skipWithoutMock } from './helpers';
|
||||
import { API_BASE, skipWithoutMock } from './helpers';
|
||||
|
||||
/**
|
||||
* Create-job flow: open the Create Job modal on /inference, fill the prompt
|
||||
@@ -10,7 +10,8 @@ import { skipWithoutMock } from './helpers';
|
||||
test.describe('create inference job', () => {
|
||||
skipWithoutMock();
|
||||
|
||||
test('creates a T2V job and shows it in the queue', async ({ page }) => {
|
||||
test('creates a T2V job and starts it without refreshing', async ({ page, request }) => {
|
||||
await request.put(`${API_BASE}/settings`, { data: { autoStartJob: false } });
|
||||
await page.goto('/inference');
|
||||
|
||||
// The trigger opens a real menu on click, so this path works for touch,
|
||||
@@ -38,5 +39,20 @@ test.describe('create inference job', () => {
|
||||
// Modal closes and the queue refreshes with the newly created job.
|
||||
await expect(dialog).toBeHidden();
|
||||
await expect(page.getByText(prompt)).toBeVisible();
|
||||
await expect(page.locator('body')).toHaveCSS('pointer-events', 'auto');
|
||||
|
||||
const card = page.getByRole('article').filter({ hasText: prompt });
|
||||
await expect(card.getByText('pending', { exact: true })).toBeVisible();
|
||||
const started = page.waitForResponse((response) =>
|
||||
response.url().startsWith(`${API_BASE}/jobs/`) &&
|
||||
response.url().endsWith('/start') &&
|
||||
response.request().method() === 'POST',
|
||||
);
|
||||
await card.getByRole('button', { name: 'Start', exact: true }).click();
|
||||
expect((await started).ok()).toBe(true);
|
||||
await expect(card.getByText('running', { exact: true })).toBeVisible();
|
||||
|
||||
await page.getByRole('link', { name: 'Datasets', exact: true }).click();
|
||||
await expect(page).toHaveURL(/\/datasets$/);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -10,7 +10,9 @@ from __future__ import annotations
|
||||
import atexit
|
||||
import collections
|
||||
import contextlib
|
||||
import copy
|
||||
import enum
|
||||
import json
|
||||
import logging
|
||||
import logging.handlers
|
||||
import multiprocessing as mp
|
||||
@@ -123,10 +125,13 @@ class LogBufferHandler(logging.Handler):
|
||||
class Job:
|
||||
id: str
|
||||
model_id: str
|
||||
prompt: str
|
||||
name: str = ""
|
||||
prompt: str = ""
|
||||
workload_type: str = "t2v"
|
||||
job_type: str = "inference"
|
||||
image_path: str = ""
|
||||
last_image_path: str = ""
|
||||
references: list[dict[str, Any]] = field(default_factory=list)
|
||||
status: JobStatus = JobStatus.PENDING
|
||||
created_at: float = field(default_factory=time.time)
|
||||
started_at: float | None = None
|
||||
@@ -145,6 +150,7 @@ class Job:
|
||||
negative_prompt: str = ""
|
||||
num_gpus: int = 1
|
||||
dit_cpu_offload: bool = False
|
||||
dit_layerwise_offload: bool = False
|
||||
text_encoder_cpu_offload: bool = False
|
||||
vae_cpu_offload: bool = False
|
||||
image_encoder_cpu_offload: bool = False
|
||||
@@ -180,10 +186,13 @@ class Job:
|
||||
return {
|
||||
"id": self.id,
|
||||
"model_id": self.model_id,
|
||||
"name": self.name,
|
||||
"prompt": self.prompt,
|
||||
"workload_type": self.workload_type,
|
||||
"job_type": self.job_type,
|
||||
"image_path": self.image_path,
|
||||
"last_image_path": self.last_image_path,
|
||||
"references": self.references,
|
||||
"status": self.status.value,
|
||||
"created_at": self.created_at,
|
||||
"started_at": self.started_at,
|
||||
@@ -202,6 +211,7 @@ class Job:
|
||||
"negative_prompt": self.negative_prompt,
|
||||
"num_gpus": self.num_gpus,
|
||||
"dit_cpu_offload": self.dit_cpu_offload,
|
||||
"dit_layerwise_offload": self.dit_layerwise_offload,
|
||||
"text_encoder_cpu_offload": self.text_encoder_cpu_offload,
|
||||
"vae_cpu_offload": self.vae_cpu_offload,
|
||||
"image_encoder_cpu_offload": self.image_encoder_cpu_offload,
|
||||
@@ -232,6 +242,75 @@ class Job:
|
||||
}
|
||||
|
||||
|
||||
MINIMAX_H3_REF2VA_PIPELINE = "MiniMaxH3Ref2VAModularPipeline"
|
||||
|
||||
|
||||
def _build_h3_references(raw: list[dict[str, Any]]) -> list[Any]:
|
||||
"""Turn the API's reference dicts into MiniMaxH3Reference objects.
|
||||
|
||||
Imported lazily so the API server starts without pulling in fastvideo.
|
||||
"""
|
||||
from fastvideo.pipelines.basic.minimax_h3 import MiniMaxH3Reference
|
||||
|
||||
built = []
|
||||
for i, ref in enumerate(raw):
|
||||
source = (ref or {}).get("source")
|
||||
if not source:
|
||||
raise ValueError(f"reference {i} has no source")
|
||||
if not os.path.isfile(source):
|
||||
raise ValueError(f"reference {i} source not found: {source}")
|
||||
kwargs: dict[str, Any] = {
|
||||
"source": source,
|
||||
"media_type": (ref.get("media_type") or "image"),
|
||||
}
|
||||
for opt in ("soundtrack", "fps", "sample_rate"):
|
||||
if ref.get(opt) not in (None, ""):
|
||||
kwargs[opt] = ref[opt]
|
||||
built.append(MiniMaxH3Reference(**kwargs))
|
||||
return built
|
||||
|
||||
|
||||
JOB_LOG_FILENAME = "out.log"
|
||||
|
||||
|
||||
def _job_log_path(output_dir: str, job_id: str) -> str:
|
||||
"""Each job's log lives beside its outputs: <output_dir>/<job_id>/out.log."""
|
||||
return os.path.join(output_dir, job_id, JOB_LOG_FILENAME)
|
||||
|
||||
|
||||
def _decode_references(value: Any) -> list[dict[str, Any]]:
|
||||
"""Reference lists round-trip through the DB as JSON text."""
|
||||
if not value:
|
||||
return []
|
||||
if isinstance(value, list):
|
||||
return list(value)
|
||||
try:
|
||||
decoded = json.loads(value)
|
||||
except (TypeError, ValueError):
|
||||
logger.warning("Could not decode stored references: %r", value)
|
||||
return []
|
||||
return list(decoded) if isinstance(decoded, list) else []
|
||||
|
||||
|
||||
def _generator_is_alive(generator: Any) -> bool:
|
||||
"""True if the generator's worker processes are all still running.
|
||||
|
||||
A cached VideoGenerator holds a MultiprocExecutor whose workers are separate
|
||||
processes; nothing notices when they exit. Probing `proc.is_alive()` is what
|
||||
the executor itself uses during shutdown. Anything unexpected in the object
|
||||
graph is treated as alive so a probe failure can never wedge the cache.
|
||||
"""
|
||||
executor = getattr(generator, "executor", None)
|
||||
workers = getattr(executor, "workers", None)
|
||||
if not workers:
|
||||
return True
|
||||
try:
|
||||
return all(w.proc.is_alive() for w in workers)
|
||||
except Exception:
|
||||
logger.debug("Worker liveness probe failed", exc_info=True)
|
||||
return True
|
||||
|
||||
|
||||
class JobRunner:
|
||||
"""Manages video generation jobs, their execution, and generator caching."""
|
||||
|
||||
@@ -276,7 +355,7 @@ class JobRunner:
|
||||
"""Populate job's log buffer from its log file if it exists."""
|
||||
path = job.log_file_path
|
||||
if not path:
|
||||
path = os.path.join(self.log_dir, f"{job.id}.log")
|
||||
path = _job_log_path(self.output_dir, job.id)
|
||||
if not os.path.isfile(path):
|
||||
return
|
||||
try:
|
||||
@@ -314,10 +393,13 @@ class JobRunner:
|
||||
job = Job(
|
||||
id=row["id"],
|
||||
model_id=row["model_id"],
|
||||
name=row.get("name", "") or "",
|
||||
prompt=row["prompt"],
|
||||
workload_type=row.get("workload_type", "t2v"),
|
||||
job_type=row.get("job_type", "inference"),
|
||||
image_path=row.get("image_path", "") or "",
|
||||
last_image_path=row.get("last_image_path", "") or "",
|
||||
references=_decode_references(row.get("references")),
|
||||
data_path=row.get("data_path", "") or "",
|
||||
max_train_steps=row.get("max_train_steps", 1000),
|
||||
train_batch_size=row.get("train_batch_size", 1),
|
||||
@@ -350,6 +432,7 @@ class JobRunner:
|
||||
negative_prompt=row.get("negative_prompt", "") or "",
|
||||
num_gpus=row.get("num_gpus", 1),
|
||||
dit_cpu_offload=row.get("dit_cpu_offload", False),
|
||||
dit_layerwise_offload=row.get("dit_layerwise_offload", False),
|
||||
text_encoder_cpu_offload=row.get("text_encoder_cpu_offload", False),
|
||||
vae_cpu_offload=row.get("vae_cpu_offload", False),
|
||||
image_encoder_cpu_offload=row.get("image_encoder_cpu_offload", False),
|
||||
@@ -394,9 +477,12 @@ class JobRunner:
|
||||
job_id: str,
|
||||
model_id: str,
|
||||
prompt: str,
|
||||
name: str = "",
|
||||
workload_type: str = "t2v",
|
||||
job_type: str = "inference",
|
||||
image_path: str = "",
|
||||
last_image_path: str = "",
|
||||
references: list[dict[str, Any]] | None = None,
|
||||
data_path: str = "",
|
||||
max_train_steps: int = 1000,
|
||||
train_batch_size: int = 1,
|
||||
@@ -422,6 +508,7 @@ class JobRunner:
|
||||
num_gpus: int = 1,
|
||||
negative_prompt: str = "",
|
||||
dit_cpu_offload: bool = False,
|
||||
dit_layerwise_offload: bool = False,
|
||||
text_encoder_cpu_offload: bool = False,
|
||||
vae_cpu_offload: bool = False,
|
||||
image_encoder_cpu_offload: bool = False,
|
||||
@@ -435,10 +522,13 @@ class JobRunner:
|
||||
job = Job(
|
||||
id=job_id,
|
||||
model_id=model_id,
|
||||
name=(name or "").strip(),
|
||||
prompt=prompt.strip(),
|
||||
workload_type=workload_type or "t2v",
|
||||
job_type=job_type or "inference",
|
||||
image_path=image_path or "",
|
||||
last_image_path=last_image_path or "",
|
||||
references=list(references or []),
|
||||
data_path=data_path or "",
|
||||
max_train_steps=max_train_steps,
|
||||
train_batch_size=train_batch_size,
|
||||
@@ -464,6 +554,7 @@ class JobRunner:
|
||||
negative_prompt=negative_prompt or "",
|
||||
num_gpus=num_gpus,
|
||||
dit_cpu_offload=dit_cpu_offload,
|
||||
dit_layerwise_offload=dit_layerwise_offload,
|
||||
text_encoder_cpu_offload=text_encoder_cpu_offload,
|
||||
vae_cpu_offload=vae_cpu_offload,
|
||||
image_encoder_cpu_offload=image_encoder_cpu_offload,
|
||||
@@ -521,6 +612,84 @@ class JobRunner:
|
||||
logger.info("Deleted job %s", job.id)
|
||||
return True
|
||||
|
||||
CONFIG_FIELDS: tuple[str, ...] = (
|
||||
"model_id",
|
||||
"name",
|
||||
"prompt",
|
||||
"workload_type",
|
||||
"job_type",
|
||||
"image_path",
|
||||
"last_image_path",
|
||||
"references",
|
||||
"negative_prompt",
|
||||
"num_inference_steps",
|
||||
"num_frames",
|
||||
"height",
|
||||
"width",
|
||||
"guidance_scale",
|
||||
"guidance_rescale",
|
||||
"fps",
|
||||
"seed",
|
||||
"num_gpus",
|
||||
"dit_cpu_offload",
|
||||
"dit_layerwise_offload",
|
||||
"text_encoder_cpu_offload",
|
||||
"vae_cpu_offload",
|
||||
"image_encoder_cpu_offload",
|
||||
"use_fsdp_inference",
|
||||
"enable_torch_compile",
|
||||
"vsa_sparsity",
|
||||
"tp_size",
|
||||
"sp_size",
|
||||
"data_path",
|
||||
"max_train_steps",
|
||||
"train_batch_size",
|
||||
"learning_rate",
|
||||
"num_latent_t",
|
||||
"validation_dataset_file",
|
||||
"lora_rank",
|
||||
"dmd_use_vsa",
|
||||
"dmd_vsa_sparsity",
|
||||
"dmd_denoising_steps",
|
||||
"real_score_guidance_scale",
|
||||
"generator_update_interval",
|
||||
"real_score_model_path",
|
||||
"fake_score_model_path",
|
||||
)
|
||||
|
||||
def duplicate_job(self, job_id: str, new_job_id: str) -> Job:
|
||||
"""Create a new pending job with an existing job's configuration.
|
||||
|
||||
Runtime state (status, timings, logs, outputs) is not carried over.
|
||||
"""
|
||||
with self._jobs_lock:
|
||||
source = self._jobs.get(job_id)
|
||||
if source is None:
|
||||
raise ValueError(f"Job {job_id} not found")
|
||||
config = {f: copy.deepcopy(getattr(source, f)) for f in self.CONFIG_FIELDS}
|
||||
return self.create_job(job_id=new_job_id, **config)
|
||||
|
||||
#: Editable exactly when startable: the same set start_job() accepts.
|
||||
EDITABLE_STATUSES = (JobStatus.PENDING, JobStatus.FAILED, JobStatus.STOPPED)
|
||||
|
||||
def update_job_config(self, job_id: str, updates: dict[str, Any]) -> Job:
|
||||
"""Edit the configuration of a job that has not produced a result."""
|
||||
with self._jobs_lock:
|
||||
job = self._jobs.get(job_id)
|
||||
if job is None:
|
||||
raise ValueError(f"Job {job_id} not found")
|
||||
if job.status not in self.EDITABLE_STATUSES:
|
||||
allowed = ", ".join(s.value for s in self.EDITABLE_STATUSES)
|
||||
raise ValueError(f"Job is {job.status.value}; only {allowed} jobs can be edited. "
|
||||
"Duplicate it instead.")
|
||||
unknown = set(updates) - set(self.CONFIG_FIELDS)
|
||||
if unknown:
|
||||
raise ValueError(f"Not editable: {', '.join(sorted(unknown))}")
|
||||
for field_name, value in updates.items():
|
||||
setattr(job, field_name, value)
|
||||
self._save_job(job)
|
||||
return job
|
||||
|
||||
def start_job(self, job_id: str) -> Job:
|
||||
"""Start (or restart) a pending / stopped / failed job.
|
||||
|
||||
@@ -623,6 +792,8 @@ class JobRunner:
|
||||
workload_type: str,
|
||||
num_gpus: int,
|
||||
dit_cpu_offload: bool = False,
|
||||
dit_layerwise_offload: bool = False,
|
||||
override_pipeline_cls_name: str | None = None,
|
||||
text_encoder_cpu_offload: bool = False,
|
||||
vae_cpu_offload: bool = False,
|
||||
image_encoder_cpu_offload: bool = False,
|
||||
@@ -638,6 +809,10 @@ class JobRunner:
|
||||
workload_type,
|
||||
num_gpus,
|
||||
dit_cpu_offload,
|
||||
dit_layerwise_offload,
|
||||
# Ref2VA loads different DiT weights (transformer_ref), so the
|
||||
# override must key the cache or a t2v/i2v generator gets reused.
|
||||
override_pipeline_cls_name,
|
||||
text_encoder_cpu_offload,
|
||||
vae_cpu_offload,
|
||||
image_encoder_cpu_offload,
|
||||
@@ -650,8 +825,21 @@ class JobRunner:
|
||||
|
||||
# Generators are cached by model_id and configuration parameters
|
||||
with self._generators_lock:
|
||||
if cache_key in self._generators:
|
||||
return self._generators[cache_key]
|
||||
cached = self._generators.get(cache_key)
|
||||
if cached is not None:
|
||||
if _generator_is_alive(cached):
|
||||
return cached
|
||||
# Workers can exit while a generator sits idle in the cache;
|
||||
# reusing it fails every later job with the same config.
|
||||
logger.warning(
|
||||
"Cached generator for %s has dead workers; reloading.",
|
||||
model_id,
|
||||
)
|
||||
self._generators.pop(cache_key, None)
|
||||
try:
|
||||
cached.shutdown()
|
||||
except Exception:
|
||||
logger.debug("Shutdown of the dead generator failed", exc_info=True)
|
||||
|
||||
# Import lazily so starting the server is fast even without a GPU.
|
||||
from fastvideo import VideoGenerator
|
||||
@@ -674,18 +862,40 @@ class JobRunner:
|
||||
sp_size,
|
||||
)
|
||||
|
||||
gen = VideoGenerator.from_pretrained(
|
||||
model_id,
|
||||
workload_type=workload_type,
|
||||
dit_cpu_offload=dit_cpu_offload,
|
||||
text_encoder_cpu_offload=text_encoder_cpu_offload,
|
||||
vae_cpu_offload=vae_cpu_offload,
|
||||
image_encoder_cpu_offload=image_encoder_cpu_offload,
|
||||
use_fsdp_inference=use_fsdp_inference,
|
||||
enable_torch_compile=enable_torch_compile,
|
||||
VSA_sparsity=vsa_sparsity,
|
||||
tp_size=tp_size,
|
||||
sp_size=sp_size,
|
||||
gen = VideoGenerator.from_config(
|
||||
{
|
||||
"model_path": model_id,
|
||||
"engine": {
|
||||
"num_gpus": num_gpus,
|
||||
"parallelism": {
|
||||
"tp_size": tp_size,
|
||||
"sp_size": sp_size,
|
||||
},
|
||||
"offload": {
|
||||
"dit": dit_cpu_offload,
|
||||
"dit_layerwise": dit_layerwise_offload,
|
||||
"text_encoder": text_encoder_cpu_offload,
|
||||
"image_encoder": image_encoder_cpu_offload,
|
||||
"vae": vae_cpu_offload,
|
||||
},
|
||||
"compile": {
|
||||
"enabled": enable_torch_compile
|
||||
},
|
||||
"attention": {
|
||||
"vsa_sparsity": vsa_sparsity
|
||||
},
|
||||
"use_fsdp_inference": use_fsdp_inference,
|
||||
},
|
||||
"pipeline": {
|
||||
"workload_type":
|
||||
workload_type,
|
||||
**({
|
||||
"components": {
|
||||
"override_pipeline_cls_name": override_pipeline_cls_name
|
||||
}
|
||||
} if override_pipeline_cls_name else {}),
|
||||
},
|
||||
},
|
||||
log_queue=log_queue,
|
||||
)
|
||||
|
||||
@@ -706,10 +916,9 @@ class JobRunner:
|
||||
def _run_training_job(self, job: Job):
|
||||
"""Run a finetuning, distillation, or LoRA job via subprocess."""
|
||||
buf = job._log_buf
|
||||
os.makedirs(self.log_dir, exist_ok=True)
|
||||
job.log_file_path = os.path.join(self.log_dir, f"{job.id}.log")
|
||||
job_output_dir = os.path.join(self.output_dir, job.id)
|
||||
os.makedirs(job_output_dir, exist_ok=True)
|
||||
job.log_file_path = _job_log_path(self.output_dir, job.id)
|
||||
|
||||
if not job.data_path or not os.path.isdir(job.data_path):
|
||||
job.status = JobStatus.FAILED
|
||||
@@ -827,8 +1036,8 @@ class JobRunner:
|
||||
|
||||
def _run_inference_job(self, job: Job):
|
||||
buf = job._log_buf
|
||||
os.makedirs(self.log_dir, exist_ok=True)
|
||||
job.log_file_path = os.path.join(self.log_dir, f"{job.id}.log")
|
||||
os.makedirs(os.path.join(self.output_dir, job.id), exist_ok=True)
|
||||
job.log_file_path = _job_log_path(self.output_dir, job.id)
|
||||
|
||||
# Add file handler to persist logs
|
||||
file_handler = logging.FileHandler(job.log_file_path, mode='w', encoding='utf-8')
|
||||
@@ -875,77 +1084,69 @@ class JobRunner:
|
||||
buf.phase = "loading model"
|
||||
logger.info("Loading model...")
|
||||
|
||||
# Run generator creation in a background thread so we
|
||||
# can poll _stop_event while the (potentially slow)
|
||||
# model download / load is in progress.
|
||||
_gen_result: list[Any] = []
|
||||
_gen_error: list[BaseException] = []
|
||||
# The generator MUST be created on this thread: building it spawns
|
||||
# the executor's worker processes, and they are torn down if the
|
||||
# creating thread exits. Running it in a helper thread (to poll
|
||||
# _stop_event during load) made every collective_rpc fail with
|
||||
# ConnectionResetError.
|
||||
if job._stop_event.is_set():
|
||||
job.status = JobStatus.STOPPED
|
||||
job.finished_at = time.time()
|
||||
self._save_job(job)
|
||||
logger.warning("Job %s stopped before model loading", job.id)
|
||||
buf.phase = "stopped"
|
||||
return
|
||||
|
||||
def _load_generator() -> None:
|
||||
try:
|
||||
gen = self._get_or_create_generator(
|
||||
job.model_id,
|
||||
job.workload_type,
|
||||
job.num_gpus,
|
||||
dit_cpu_offload=job.dit_cpu_offload,
|
||||
text_encoder_cpu_offload=(job.text_encoder_cpu_offload),
|
||||
vae_cpu_offload=job.vae_cpu_offload,
|
||||
image_encoder_cpu_offload=(job.image_encoder_cpu_offload),
|
||||
use_fsdp_inference=job.use_fsdp_inference,
|
||||
enable_torch_compile=(job.enable_torch_compile),
|
||||
vsa_sparsity=job.vsa_sparsity,
|
||||
tp_size=job.tp_size,
|
||||
sp_size=job.sp_size,
|
||||
log_queue=log_queue,
|
||||
)
|
||||
_gen_result.append(gen)
|
||||
except BaseException as exc:
|
||||
_gen_error.append(exc)
|
||||
|
||||
loader = threading.Thread(
|
||||
target=_load_generator,
|
||||
daemon=True,
|
||||
generator = self._get_or_create_generator(
|
||||
job.model_id,
|
||||
job.workload_type,
|
||||
job.num_gpus,
|
||||
dit_cpu_offload=job.dit_cpu_offload,
|
||||
dit_layerwise_offload=job.dit_layerwise_offload,
|
||||
override_pipeline_cls_name=(MINIMAX_H3_REF2VA_PIPELINE if job.references else None),
|
||||
text_encoder_cpu_offload=(job.text_encoder_cpu_offload),
|
||||
vae_cpu_offload=job.vae_cpu_offload,
|
||||
image_encoder_cpu_offload=(job.image_encoder_cpu_offload),
|
||||
use_fsdp_inference=job.use_fsdp_inference,
|
||||
enable_torch_compile=(job.enable_torch_compile),
|
||||
vsa_sparsity=job.vsa_sparsity,
|
||||
tp_size=job.tp_size,
|
||||
sp_size=job.sp_size,
|
||||
log_queue=log_queue,
|
||||
)
|
||||
loader.start()
|
||||
|
||||
while loader.is_alive():
|
||||
if job._stop_event.is_set():
|
||||
job.status = JobStatus.STOPPED
|
||||
job.finished_at = time.time()
|
||||
self._save_job(job)
|
||||
logger.warning(
|
||||
"Job %s stopped during model loading",
|
||||
job.id,
|
||||
)
|
||||
buf.phase = "stopped"
|
||||
return
|
||||
loader.join(timeout=0.5)
|
||||
|
||||
if _gen_error:
|
||||
raise _gen_error[0]
|
||||
|
||||
generator = _gen_result[0]
|
||||
buf.phase = "generating"
|
||||
logger.info("Starting generation for job %s (model=%s)", job.id, job.model_id)
|
||||
|
||||
gen_kwargs: dict[str, Any] = {
|
||||
# Without a name FastVideo derives the filename from the prompt.
|
||||
safe_name = re.sub(r'[\\/:*?"<>|]+', "", job.name).strip().strip(".")
|
||||
output_target = (os.path.join(job_output_dir, f"{safe_name[:80]}.mp4") if safe_name else job_output_dir)
|
||||
request: dict[str, Any] = {
|
||||
"prompt": job.prompt,
|
||||
"output_path": job_output_dir,
|
||||
"save_video": True,
|
||||
"num_inference_steps": job.num_inference_steps,
|
||||
"num_frames": job.num_frames,
|
||||
"height": job.height,
|
||||
"width": job.width,
|
||||
"guidance_scale": job.guidance_scale,
|
||||
"guidance_rescale": job.guidance_rescale,
|
||||
"fps": job.fps,
|
||||
"seed": job.seed,
|
||||
"negative_prompt": job.negative_prompt or "",
|
||||
"log_queue": log_queue,
|
||||
"sampling": {
|
||||
"num_inference_steps": job.num_inference_steps,
|
||||
"num_frames": job.num_frames,
|
||||
"height": job.height,
|
||||
"width": job.width,
|
||||
"guidance_scale": job.guidance_scale,
|
||||
"guidance_rescale": job.guidance_rescale,
|
||||
"fps": job.fps,
|
||||
"seed": job.seed,
|
||||
},
|
||||
"output": {
|
||||
"output_path": output_target,
|
||||
"save_video": True,
|
||||
},
|
||||
}
|
||||
if job.image_path:
|
||||
gen_kwargs["image_path"] = job.image_path
|
||||
generator.generate_video(**gen_kwargs)
|
||||
request.setdefault("inputs", {})["image_path"] = job.image_path
|
||||
if job.references:
|
||||
request.setdefault("inputs", {})["references"] = _build_h3_references(job.references)
|
||||
if job.last_image_path:
|
||||
# _prepare_fl2va requires a PIL image, not a path.
|
||||
from PIL import Image as _PILImage
|
||||
request.setdefault("inputs", {})["last_image"] = _PILImage.open(job.last_image_path)
|
||||
generator.generate(request, log_queue=log_queue)
|
||||
|
||||
buf.phase = "saving"
|
||||
logger.info("Generation completed, searching for output file...")
|
||||
@@ -977,7 +1178,7 @@ class JobRunner:
|
||||
|
||||
except Exception as exception:
|
||||
error_msg = str(exception)
|
||||
logger.error("Critical error in job thread: %s", error_msg)
|
||||
logger.exception("Critical error in job thread: %s", error_msg)
|
||||
job.status = JobStatus.FAILED
|
||||
job.error = f"Critical error ({type(exception).__name__}): {error_msg}"
|
||||
job.finished_at = time.time()
|
||||
|
||||
@@ -1,15 +1,20 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Request model for creating a job."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
||||
class CreateJobRequest(BaseModel):
|
||||
model_id: str
|
||||
name: str = ""
|
||||
prompt: str
|
||||
workload_type: str = "t2v"
|
||||
job_type: str = "inference"
|
||||
image_path: str = ""
|
||||
last_image_path: str = ""
|
||||
references: list[dict[str, Any]] | None = None
|
||||
data_path: str = ""
|
||||
max_train_steps: int = 1000
|
||||
train_batch_size: int = 1
|
||||
@@ -28,6 +33,7 @@ class CreateJobRequest(BaseModel):
|
||||
seed: int = 1024
|
||||
num_gpus: int = 1
|
||||
dit_cpu_offload: bool = False
|
||||
dit_layerwise_offload: bool = False
|
||||
text_encoder_cpu_offload: bool = False
|
||||
vae_cpu_offload: bool = False
|
||||
image_encoder_cpu_offload: bool = False
|
||||
|
||||
@@ -17,20 +17,11 @@
|
||||
"start:all": "concurrently --kill-others-on-fail \"npm:start:api\" \"npm:start:web\""
|
||||
},
|
||||
"dependencies": {
|
||||
"@radix-ui/react-dialog": "^1.1.0",
|
||||
"@radix-ui/react-dropdown-menu": "^2.1.24",
|
||||
"@radix-ui/react-label": "^2.1.8",
|
||||
"@radix-ui/react-scroll-area": "^1.2.10",
|
||||
"@radix-ui/react-select": "^2.2.6",
|
||||
"@radix-ui/react-separator": "^1.1.8",
|
||||
"@radix-ui/react-slider": "^1.2.0",
|
||||
"@radix-ui/react-slot": "^1.2.4",
|
||||
"@radix-ui/react-switch": "^1.1.0",
|
||||
"@radix-ui/react-tabs": "^1.1.0",
|
||||
"class-variance-authority": "^0.7.1",
|
||||
"clsx": "^2.1.1",
|
||||
"lucide-react": "^0.577.0",
|
||||
"next": "15.5.18",
|
||||
"radix-ui": "^1.6.7",
|
||||
"react": "^19.1.0",
|
||||
"react-dom": "^19.1.0",
|
||||
"sonner": "^2.0.7",
|
||||
|
||||
@@ -18,6 +18,7 @@ import argparse
|
||||
import contextlib
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import signal
|
||||
import time
|
||||
@@ -116,7 +117,30 @@ def list_models(workload_type: str | None = None) -> list[dict[str, Any]]:
|
||||
return _available_models
|
||||
|
||||
|
||||
def _safe_upload_name(filename: str | None, ext: str) -> str:
|
||||
"""A filesystem-safe version of the client's filename, keeping it readable.
|
||||
|
||||
Uploads live under a per-file uuid directory, so the basename does not have
|
||||
to be unique -- only safe. Keeping the original name means the path stays
|
||||
self-describing wherever it travels: the database, job logs, and payloads
|
||||
copied back out to the API.
|
||||
"""
|
||||
stem = os.path.basename(filename or "").rsplit(".", 1)[0]
|
||||
stem = re.sub(r"[^A-Za-z0-9._-]+", "_", stem).strip("._-")
|
||||
return f"{stem[:80] or 'upload'}{ext}"
|
||||
|
||||
|
||||
def _upload_destination(ext: str, filename: str | None) -> str:
|
||||
"""<upload_dir>/<uuid4>/<safe original name><ext>"""
|
||||
directory = os.path.join(upload_dir, uuid.uuid4().hex)
|
||||
os.makedirs(directory, exist_ok=True)
|
||||
return os.path.join(directory, _safe_upload_name(filename, ext))
|
||||
|
||||
|
||||
ALLOWED_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".webp", ".bmp"}
|
||||
ALLOWED_VIDEO_EXTENSIONS = {".mp4", ".mov", ".mkv", ".webm", ".avi"}
|
||||
ALLOWED_AUDIO_EXTENSIONS = {".wav", ".mp3", ".flac", ".m4a", ".ogg"}
|
||||
ALLOWED_MEDIA_EXTENSIONS = (ALLOWED_IMAGE_EXTENSIONS | ALLOWED_VIDEO_EXTENSIONS | ALLOWED_AUDIO_EXTENSIONS)
|
||||
|
||||
|
||||
@app.post("/api/upload-image")
|
||||
@@ -136,8 +160,7 @@ async def upload_image(file: Annotated[UploadFile, File()], ) -> dict[str, str]:
|
||||
f"{', '.join(ALLOWED_IMAGE_EXTENSIONS)}"),
|
||||
)
|
||||
os.makedirs(upload_dir, exist_ok=True)
|
||||
unique_name = f"{uuid.uuid4().hex}{ext}"
|
||||
dest_path = os.path.join(upload_dir, unique_name)
|
||||
dest_path = _upload_destination(ext, file.filename)
|
||||
try:
|
||||
contents = await file.read()
|
||||
with open(dest_path, "wb") as f:
|
||||
@@ -150,6 +173,47 @@ async def upload_image(file: Annotated[UploadFile, File()], ) -> dict[str, str]:
|
||||
return {"path": os.path.abspath(dest_path)}
|
||||
|
||||
|
||||
@app.post("/api/upload-media")
|
||||
async def upload_media(file: Annotated[UploadFile, File()], ) -> dict[str, str]:
|
||||
"""Upload an image, video or audio file for Ref2VA references.
|
||||
|
||||
Returns the absolute path plus the media_type MiniMax-H3 expects, so the
|
||||
caller does not have to re-derive it from the extension.
|
||||
"""
|
||||
global upload_dir # noqa: PLW0603
|
||||
if not upload_dir:
|
||||
raise HTTPException(
|
||||
status_code=503,
|
||||
detail="Upload directory not configured",
|
||||
)
|
||||
ext = Path(file.filename or "").suffix.lower()
|
||||
if ext not in ALLOWED_MEDIA_EXTENSIONS:
|
||||
raise HTTPException(
|
||||
status_code=400,
|
||||
detail=(f"Invalid file type. Allowed: "
|
||||
f"{', '.join(sorted(ALLOWED_MEDIA_EXTENSIONS))}"),
|
||||
)
|
||||
if ext in ALLOWED_VIDEO_EXTENSIONS:
|
||||
media_type = "video"
|
||||
elif ext in ALLOWED_AUDIO_EXTENSIONS:
|
||||
media_type = "audio"
|
||||
else:
|
||||
media_type = "image"
|
||||
|
||||
os.makedirs(upload_dir, exist_ok=True)
|
||||
dest_path = _upload_destination(ext, file.filename)
|
||||
try:
|
||||
contents = await file.read()
|
||||
with open(dest_path, "wb") as f:
|
||||
f.write(contents)
|
||||
except OSError as e:
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail=f"Failed to save upload: {e}",
|
||||
) from e
|
||||
return {"path": os.path.abspath(dest_path), "media_type": media_type}
|
||||
|
||||
|
||||
ALLOWED_VIDEO_EXTENSIONS = {".mp4", ".webm", ".avi", ".mov", ".mkv"}
|
||||
|
||||
|
||||
@@ -282,10 +346,13 @@ def create_job(req: CreateJobRequest) -> dict[str, Any]:
|
||||
job = job_runner.create_job(
|
||||
job_id=str(uuid.uuid4()),
|
||||
model_id=req.model_id,
|
||||
name=req.name or "",
|
||||
prompt=req.prompt,
|
||||
workload_type=req.workload_type or "t2v",
|
||||
job_type=job_type,
|
||||
image_path=req.image_path or "",
|
||||
last_image_path=req.last_image_path or "",
|
||||
references=req.references or [],
|
||||
data_path=data_path,
|
||||
max_train_steps=req.max_train_steps,
|
||||
train_batch_size=req.train_batch_size,
|
||||
@@ -304,6 +371,7 @@ def create_job(req: CreateJobRequest) -> dict[str, Any]:
|
||||
seed=req.seed,
|
||||
num_gpus=req.num_gpus,
|
||||
dit_cpu_offload=req.dit_cpu_offload,
|
||||
dit_layerwise_offload=req.dit_layerwise_offload,
|
||||
text_encoder_cpu_offload=req.text_encoder_cpu_offload,
|
||||
vae_cpu_offload=req.vae_cpu_offload,
|
||||
image_encoder_cpu_offload=req.image_encoder_cpu_offload,
|
||||
@@ -336,6 +404,28 @@ def create_job(req: CreateJobRequest) -> dict[str, Any]:
|
||||
return job.to_dict()
|
||||
|
||||
|
||||
@app.post("/api/jobs/{job_id}/duplicate", status_code=201)
|
||||
def duplicate_job(job_id: str) -> dict[str, Any]:
|
||||
"""Create a new pending job with the same configuration as an existing one."""
|
||||
try:
|
||||
job = job_runner.duplicate_job(job_id, str(uuid.uuid4()))
|
||||
except ValueError as e:
|
||||
raise HTTPException(status_code=404, detail=str(e)) from e
|
||||
return job.to_dict()
|
||||
|
||||
|
||||
@app.patch("/api/jobs/{job_id}")
|
||||
def update_job(job_id: str, updates: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Edit a pending job's configuration. Started jobs cannot be edited."""
|
||||
try:
|
||||
job = job_runner.update_job_config(job_id, updates)
|
||||
except ValueError as e:
|
||||
detail = str(e)
|
||||
status = 404 if "not found" in detail else 400
|
||||
raise HTTPException(status_code=status, detail=detail) from e
|
||||
return job.to_dict()
|
||||
|
||||
|
||||
@app.post("/api/jobs/{job_id}/start")
|
||||
def start_job(job_id: str) -> dict[str, Any]:
|
||||
"""Start (or restart) a pending / stopped / failed job."""
|
||||
|
||||
@@ -1,24 +1,26 @@
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { render, screen, waitFor, within } from '@testing-library/react';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import CreateJobButton from './CreateJobButton';
|
||||
import { getDatasets, getModels } from '@/lib/api';
|
||||
|
||||
vi.mock('./CreateJobModal', () => ({
|
||||
default: ({
|
||||
isOpen,
|
||||
workloadType,
|
||||
}: {
|
||||
isOpen: boolean;
|
||||
workloadType: string;
|
||||
}) =>
|
||||
isOpen ? (
|
||||
<div role="dialog" data-workload-type={workloadType}>
|
||||
Create job form
|
||||
</div>
|
||||
) : null,
|
||||
vi.mock('@/lib/api', () => ({
|
||||
createJob: vi.fn(),
|
||||
getModels: vi.fn(),
|
||||
getDatasets: vi.fn(),
|
||||
uploadImage: vi.fn(),
|
||||
getSettings: vi.fn(),
|
||||
updateSettings: vi.fn(),
|
||||
}));
|
||||
|
||||
beforeEach(() => {
|
||||
vi.mocked(getModels).mockResolvedValue([
|
||||
{ id: 'wan/t2v-1.3b', label: 'Wan T2V' },
|
||||
]);
|
||||
vi.mocked(getDatasets).mockResolvedValue([]);
|
||||
});
|
||||
|
||||
describe('CreateJobButton', () => {
|
||||
it('opens the workload menu on click and selects an item', async () => {
|
||||
const user = userEvent.setup();
|
||||
@@ -27,10 +29,9 @@ describe('CreateJobButton', () => {
|
||||
await user.click(screen.getByRole('button', { name: 'Create Job' }));
|
||||
await user.click(screen.getByRole('menuitem', { name: /I2V/i }));
|
||||
|
||||
expect(screen.getByRole('dialog')).toHaveAttribute(
|
||||
'data-workload-type',
|
||||
'i2v',
|
||||
);
|
||||
expect(
|
||||
screen.getByRole('dialog', { name: 'New Inference Job (I2V)' }),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it('opens and operates the workload menu from the keyboard', async () => {
|
||||
@@ -45,9 +46,42 @@ describe('CreateJobButton', () => {
|
||||
expect(firstItem).toHaveFocus();
|
||||
await user.keyboard('{Enter}');
|
||||
|
||||
expect(screen.getByRole('dialog')).toHaveAttribute(
|
||||
'data-workload-type',
|
||||
't2v',
|
||||
expect(
|
||||
screen.getByRole('dialog', { name: 'New Inference Job (T2V)' }),
|
||||
).toBeInTheDocument();
|
||||
await user.keyboard('{Escape}');
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument(),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(document.body.style.pointerEvents).not.toBe('none'),
|
||||
);
|
||||
expect(trigger).toHaveFocus();
|
||||
});
|
||||
|
||||
it.each(['inference', 'finetuning', 'distillation'] as const)(
|
||||
'restores page interaction after closing the real %s dialog',
|
||||
async (jobType) => {
|
||||
const user = userEvent.setup();
|
||||
render(<CreateJobButton jobType={jobType} />);
|
||||
const trigger = screen.getByRole('button', { name: 'Create Job' });
|
||||
|
||||
// Keep the real Dialog mounted: mocking it hides conflicting Radix layers.
|
||||
for (let attempt = 0; attempt < 2; attempt++) {
|
||||
await user.click(trigger);
|
||||
await user.click(screen.getAllByRole('menuitem')[0]);
|
||||
const dialog = screen.getByRole('dialog');
|
||||
await user.click(
|
||||
within(dialog).getByRole('button', { name: 'Close' }),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(screen.queryByRole('dialog')).not.toBeInTheDocument(),
|
||||
);
|
||||
await waitFor(() =>
|
||||
expect(document.body.style.pointerEvents).not.toBe('none'),
|
||||
);
|
||||
expect(trigger).toHaveFocus();
|
||||
}
|
||||
},
|
||||
);
|
||||
});
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
import * as React from 'react';
|
||||
import { ChevronDown } from 'lucide-react';
|
||||
import * as DropdownMenu from '@radix-ui/react-dropdown-menu';
|
||||
import { DropdownMenu } from 'radix-ui';
|
||||
|
||||
import CreateJobModal from '@/components/jobs/CreateJobModal';
|
||||
import { Button } from '@/components/ui/button';
|
||||
@@ -16,6 +16,7 @@ interface CreateJobButtonProps {
|
||||
|
||||
export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
const options = WORKLOAD_OPTIONS[jobType] ?? [];
|
||||
const triggerRef = React.useRef<HTMLButtonElement>(null);
|
||||
|
||||
const [modalOpen, setModalOpen] = React.useState(false);
|
||||
const [workloadType, setWorkloadType] = React.useState(
|
||||
@@ -36,7 +37,7 @@ export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
<>
|
||||
<DropdownMenu.Root>
|
||||
<DropdownMenu.Trigger asChild>
|
||||
<Button type="button" className="gap-1.5">
|
||||
<Button ref={triggerRef} type="button" className="gap-1.5">
|
||||
Create Job
|
||||
<ChevronDown className="size-3.5 opacity-85" aria-hidden />
|
||||
</Button>
|
||||
@@ -66,6 +67,11 @@ export default function CreateJobButton({ jobType }: CreateJobButtonProps) {
|
||||
<CreateJobModal
|
||||
isOpen={modalOpen}
|
||||
onClose={() => setModalOpen(false)}
|
||||
onCloseAutoFocus={(event) => {
|
||||
// This dialog opens from a menu item, so it has no DialogTrigger.
|
||||
event.preventDefault();
|
||||
triggerRef.current?.focus();
|
||||
}}
|
||||
onSuccess={handleSuccess}
|
||||
jobType={jobType}
|
||||
workloadType={workloadType}
|
||||
|
||||
@@ -22,30 +22,60 @@ import { useStore } from '@/hooks/useStore';
|
||||
import { defaultOptionsStore } from '@/stores/defaultOptions';
|
||||
import {
|
||||
createJob,
|
||||
updateJob,
|
||||
getDatasets,
|
||||
getModels,
|
||||
uploadImage,
|
||||
uploadMedia,
|
||||
type CreateJobRequest,
|
||||
type Model,
|
||||
} from '@/lib/api';
|
||||
import { getDefaultModelForWorkload } from '@/lib/defaultOptions';
|
||||
import { WORKLOAD_OPTIONS } from '@/lib/jobConfig';
|
||||
import type { JobType } from '@/lib/types';
|
||||
import {
|
||||
H3_MAX_REFERENCES,
|
||||
labelReferences,
|
||||
referencePromptSeed,
|
||||
validateReferences,
|
||||
type H3Reference,
|
||||
} from '@/lib/h3References';
|
||||
import {
|
||||
EMPTY_H3_PROMPT_FIELDS,
|
||||
H3_PROMPT_SECTIONS,
|
||||
H3_SECTION_HINTS,
|
||||
H3_SECTION_LABELS,
|
||||
isEmptyPromptFields,
|
||||
parseH3Prompt,
|
||||
serializeH3Prompt,
|
||||
type H3PromptFields,
|
||||
} from '@/lib/h3Prompt';
|
||||
import { jobToFormFields, type JobLike } from '@/lib/jobToFields';
|
||||
|
||||
export interface CreateJobModalProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
onCloseAutoFocus?: React.ComponentProps<
|
||||
typeof DialogContent
|
||||
>['onCloseAutoFocus'];
|
||||
onSuccess: () => void;
|
||||
jobType: JobType;
|
||||
workloadType: string;
|
||||
/** When set, the modal edits this pending job instead of creating a new one. */
|
||||
editingJob?: JobLike | null;
|
||||
/** Show the configuration without allowing changes (started/finished jobs). */
|
||||
readOnly?: boolean;
|
||||
}
|
||||
|
||||
export default function CreateJobModal({
|
||||
isOpen,
|
||||
onClose,
|
||||
onCloseAutoFocus,
|
||||
onSuccess,
|
||||
jobType,
|
||||
workloadType,
|
||||
editingJob,
|
||||
readOnly = false,
|
||||
}: CreateJobModalProps) {
|
||||
const { options } = useStore(defaultOptionsStore);
|
||||
|
||||
@@ -55,8 +85,22 @@ export default function CreateJobModal({
|
||||
|
||||
const [models, setModels] = React.useState<Model[]>([]);
|
||||
const [modelId, setModelId] = React.useState('');
|
||||
const [name, setName] = React.useState('');
|
||||
const [prompt, setPrompt] = React.useState('');
|
||||
const [imagePath, setImagePath] = React.useState('');
|
||||
const [lastImagePath, setLastImagePath] = React.useState('');
|
||||
const [references, setReferences] = React.useState<H3Reference[]>([]);
|
||||
const [isUploadingReference, setIsUploadingReference] = React.useState(false);
|
||||
const [referenceError, setReferenceError] = React.useState<string | null>(null);
|
||||
const [promptFields, setPromptFields] = React.useState<H3PromptFields>(
|
||||
EMPTY_H3_PROMPT_FIELDS,
|
||||
);
|
||||
const [useGuidedPrompt, setUseGuidedPrompt] = React.useState(true);
|
||||
const [lastImageFileName, setLastImageFileName] = React.useState('');
|
||||
const [isUploadingLastImage, setIsUploadingLastImage] = React.useState(false);
|
||||
const [lastImageUploadError, setLastImageUploadError] = React.useState<
|
||||
string | null
|
||||
>(null);
|
||||
const [imageFileName, setImageFileName] = React.useState('');
|
||||
const [isUploadingImage, setIsUploadingImage] = React.useState(false);
|
||||
const [negativePrompt, setNegativePrompt] = React.useState('');
|
||||
@@ -70,12 +114,51 @@ export default function CreateJobModal({
|
||||
const [seed, setSeed] = React.useState(1024);
|
||||
const [numGpus, setNumGpus] = React.useState(1);
|
||||
const [ditCpuOffload, setDitCpuOffload] = React.useState(false);
|
||||
const [ditLayerwiseOffload, setDitLayerwiseOffload] = React.useState(false);
|
||||
const [textEncoderCpuOffload, setTextEncoderCpuOffload] =
|
||||
React.useState(false);
|
||||
const [vaeCpuOffload, setVaeCpuOffload] = React.useState(false);
|
||||
const [imageEncoderCpuOffload, setImageEncoderCpuOffload] =
|
||||
React.useState(false);
|
||||
const [useFsdpInference, setUseFsdpInference] = React.useState(false);
|
||||
|
||||
// H3 is the only registered model with an end frame or references.
|
||||
const supportsLastImage = modelId.toLowerCase().includes('minimax-h3');
|
||||
const usingReferences = supportsLastImage && references.length > 0;
|
||||
|
||||
// JobCard re-renders on every job-list poll, so `editingJob` is a fresh
|
||||
// object each time. Effects must depend on these, never on the object.
|
||||
const editingJobId = editingJob?.id ?? null;
|
||||
const editingJobModelId = editingJob?.model_id ?? null;
|
||||
|
||||
// Layerwise offload and FSDP compete for the DiT weights and the device offload
|
||||
// policy (resolve_device_offload_conflicts in fastvideo/api/device_policy.py)
|
||||
// silently picks a winner; resolve it visibly here.
|
||||
// dit_cpu_offload is deliberately not interlocked -- it is a modifier, not a
|
||||
// competing strategy.
|
||||
const handleDitLayerwiseOffloadChange = React.useCallback((next: boolean) => {
|
||||
setDitLayerwiseOffload(next);
|
||||
if (next) {
|
||||
setUseFsdpInference(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
const handleUseFsdpInferenceChange = React.useCallback((next: boolean) => {
|
||||
setUseFsdpInference(next);
|
||||
if (next) {
|
||||
setDitLayerwiseOffload(false);
|
||||
}
|
||||
}, []);
|
||||
|
||||
const handleNumGpusChange = React.useCallback((next: number) => {
|
||||
setNumGpus(next);
|
||||
if (next > 1) {
|
||||
// Dropping back to one GPU leaves FSDP alone: single-GPU FSDP is a
|
||||
// valid way to reach its CPU offload (docs/inference/offloading.md).
|
||||
setUseFsdpInference(true);
|
||||
setDitLayerwiseOffload(false);
|
||||
}
|
||||
}, []);
|
||||
const [enableTorchCompile, setEnableTorchCompile] = React.useState(false);
|
||||
const [vsaSparsity, setVsaSparsity] = React.useState(0);
|
||||
const [tpSize, setTpSize] = React.useState(-1);
|
||||
@@ -126,6 +209,47 @@ export default function CreateJobModal({
|
||||
const justOpened = isOpen && !justOpenedRef.current;
|
||||
justOpenedRef.current = isOpen;
|
||||
if (!justOpened) return;
|
||||
if (editingJob) {
|
||||
// Must not fall through to the defaults below: a partially-seeded
|
||||
// form silently edits values the user never saw.
|
||||
const f = jobToFormFields(editingJob);
|
||||
setModelId(f.modelId);
|
||||
setName(f.name);
|
||||
setPrompt(f.prompt);
|
||||
setNegativePrompt(f.negativePrompt);
|
||||
setImagePath(f.imagePath);
|
||||
setImageFileName(f.imagePath.split('/').pop() ?? '');
|
||||
setLastImagePath(f.lastImagePath);
|
||||
setLastImageFileName(f.lastImagePath.split('/').pop() ?? '');
|
||||
setReferences(f.references);
|
||||
setPromptFields(f.promptFields ?? EMPTY_H3_PROMPT_FIELDS);
|
||||
setUseGuidedPrompt(f.promptFields !== null);
|
||||
setNumInferenceSteps(f.numInferenceSteps);
|
||||
setNumFrames(f.numFrames);
|
||||
setHeight(f.height);
|
||||
setWidth(f.width);
|
||||
setGuidanceScale(f.guidanceScale);
|
||||
setGuidanceRescale(f.guidanceRescale);
|
||||
setFps(f.fps);
|
||||
setSeed(f.seed);
|
||||
setNumGpus(f.numGpus);
|
||||
setDitCpuOffload(f.ditCpuOffload);
|
||||
setDitLayerwiseOffload(f.ditLayerwiseOffload);
|
||||
setTextEncoderCpuOffload(f.textEncoderCpuOffload);
|
||||
setVaeCpuOffload(f.vaeCpuOffload);
|
||||
setImageEncoderCpuOffload(f.imageEncoderCpuOffload);
|
||||
setUseFsdpInference(f.useFsdpInference);
|
||||
setEnableTorchCompile(f.enableTorchCompile);
|
||||
setVsaSparsity(f.vsaSparsity);
|
||||
setTpSize(f.tpSize);
|
||||
setSpSize(f.spSize);
|
||||
setReferenceError(null);
|
||||
setModelLoadError(null);
|
||||
setImageUploadError(null);
|
||||
setLastImageUploadError(null);
|
||||
setSubmitError(null);
|
||||
return;
|
||||
}
|
||||
const opts = options;
|
||||
setNumInferenceSteps(opts.numInferenceSteps);
|
||||
setNumFrames(workloadType === 't2i' ? 1 : opts.numFrames);
|
||||
@@ -137,6 +261,7 @@ export default function CreateJobModal({
|
||||
setSeed(opts.seed);
|
||||
setNumGpus(opts.numGpus);
|
||||
setDitCpuOffload(opts.ditCpuOffload);
|
||||
setDitLayerwiseOffload(opts.ditLayerwiseOffload ?? false);
|
||||
setTextEncoderCpuOffload(opts.textEncoderCpuOffload);
|
||||
setVaeCpuOffload(opts.vaeCpuOffload);
|
||||
setImageEncoderCpuOffload(opts.imageEncoderCpuOffload);
|
||||
@@ -151,8 +276,16 @@ export default function CreateJobModal({
|
||||
inferenceWorkload as 't2v' | 'i2v' | 't2i',
|
||||
),
|
||||
);
|
||||
setName('');
|
||||
setImagePath('');
|
||||
setImageFileName('');
|
||||
setLastImagePath('');
|
||||
setLastImageFileName('');
|
||||
setLastImageUploadError(null);
|
||||
setReferences([]);
|
||||
setReferenceError(null);
|
||||
setPromptFields(EMPTY_H3_PROMPT_FIELDS);
|
||||
setUseGuidedPrompt(true);
|
||||
setSelectedDatasetId('');
|
||||
setSelectedValidationDatasetId('');
|
||||
setModelLoadError(null);
|
||||
@@ -168,7 +301,7 @@ export default function CreateJobModal({
|
||||
setRealScoreModelPath('');
|
||||
setFakeScoreModelPath('');
|
||||
}
|
||||
}, [isOpen, workloadType, inferenceWorkload, options]);
|
||||
}, [isOpen, workloadType, inferenceWorkload, options, editingJobId]);
|
||||
|
||||
// Load the models available for this workload.
|
||||
React.useEffect(() => {
|
||||
@@ -188,7 +321,16 @@ export default function CreateJobModal({
|
||||
opts,
|
||||
inferenceWorkload as 't2v' | 'i2v' | 't2i',
|
||||
);
|
||||
const chosen = ids.includes(defaultId) ? defaultId : (list[0]?.id ?? '');
|
||||
// When editing, the job's own model wins over the workload default --
|
||||
// this resolves after the seeding effect, so choosing a default here
|
||||
// would silently swap the model out from under the user.
|
||||
const editedId = editingJobModelId;
|
||||
const chosen =
|
||||
editedId && ids.includes(editedId)
|
||||
? editedId
|
||||
: ids.includes(defaultId)
|
||||
? defaultId
|
||||
: (list[0]?.id ?? '');
|
||||
setModelId(chosen);
|
||||
if (workloadType === 'dmd_t2v') {
|
||||
setRealScoreModelPath(chosen);
|
||||
@@ -210,7 +352,7 @@ export default function CreateJobModal({
|
||||
return () => {
|
||||
stale = true;
|
||||
};
|
||||
}, [isOpen, inferenceWorkload, workloadType]);
|
||||
}, [isOpen, inferenceWorkload, workloadType, editingJobModelId]);
|
||||
|
||||
// Training jobs need a dataset; load the ready datasets when relevant.
|
||||
React.useEffect(() => {
|
||||
@@ -262,6 +404,106 @@ export default function CreateJobModal({
|
||||
}
|
||||
}
|
||||
|
||||
async function handleLastImageChange(
|
||||
e: React.ChangeEvent<HTMLInputElement>,
|
||||
) {
|
||||
const file = e.target.files?.[0];
|
||||
if (!file) {
|
||||
setLastImagePath('');
|
||||
setLastImageFileName('');
|
||||
setLastImageUploadError(null);
|
||||
return;
|
||||
}
|
||||
setIsUploadingLastImage(true);
|
||||
setLastImageFileName(file.name);
|
||||
setLastImageUploadError(null);
|
||||
try {
|
||||
const { path } = await uploadImage(file);
|
||||
setLastImagePath(path);
|
||||
} catch (error) {
|
||||
console.error('Failed to upload end image:', error);
|
||||
setLastImagePath('');
|
||||
setLastImageFileName('');
|
||||
setLastImageUploadError(
|
||||
error instanceof Error
|
||||
? `${error.message}. Choose the image again to retry.`
|
||||
: 'The image could not be uploaded. Choose it again to retry.',
|
||||
);
|
||||
} finally {
|
||||
setIsUploadingLastImage(false);
|
||||
}
|
||||
}
|
||||
|
||||
async function handleAddReference(
|
||||
e: React.ChangeEvent<HTMLInputElement>,
|
||||
) {
|
||||
const file = e.target.files?.[0];
|
||||
e.target.value = ''; // allow re-picking the same file
|
||||
if (!file) return;
|
||||
setIsUploadingReference(true);
|
||||
setReferenceError(null);
|
||||
try {
|
||||
const { path, media_type } = await uploadMedia(file);
|
||||
const next: H3Reference[] = [
|
||||
...references,
|
||||
{
|
||||
id: `${Date.now()}-${file.name}`,
|
||||
source: path,
|
||||
media_type,
|
||||
fileName: file.name,
|
||||
},
|
||||
];
|
||||
setReferences(next);
|
||||
setReferenceError(validateReferences(next));
|
||||
} catch (error) {
|
||||
console.error('Failed to upload reference:', error);
|
||||
setReferenceError(
|
||||
error instanceof Error ? error.message : 'The file could not be uploaded.',
|
||||
);
|
||||
} finally {
|
||||
setIsUploadingReference(false);
|
||||
}
|
||||
}
|
||||
|
||||
function removeReference(id: string) {
|
||||
const next = references.filter((r) => r.id !== id);
|
||||
setReferences(next);
|
||||
setReferenceError(validateReferences(next));
|
||||
}
|
||||
|
||||
function seedPromptFields() {
|
||||
setPromptFields({
|
||||
...EMPTY_H3_PROMPT_FIELDS,
|
||||
...referencePromptSeed(references),
|
||||
});
|
||||
setUseGuidedPrompt(true);
|
||||
}
|
||||
|
||||
function setPromptField(section: string, value: string) {
|
||||
setPromptFields((prev) => ({ ...prev, [section]: value }));
|
||||
}
|
||||
|
||||
// Switching between the guided fields and the raw editor keeps whatever was
|
||||
// typed: serialize on the way out, parse back on the way in.
|
||||
function toggleGuidedPrompt() {
|
||||
if (useGuidedPrompt) {
|
||||
if (!isEmptyPromptFields(promptFields)) {
|
||||
setPrompt(serializeH3Prompt(promptFields));
|
||||
}
|
||||
setUseGuidedPrompt(false);
|
||||
} else {
|
||||
const parsed = parseH3Prompt(prompt);
|
||||
if (parsed) setPromptFields(parsed);
|
||||
setUseGuidedPrompt(true);
|
||||
}
|
||||
}
|
||||
|
||||
function clearLastImage() {
|
||||
setLastImagePath('');
|
||||
setLastImageFileName('');
|
||||
setLastImageUploadError(null);
|
||||
}
|
||||
|
||||
function clearImage() {
|
||||
setImagePath('');
|
||||
setImageFileName('');
|
||||
@@ -271,7 +513,16 @@ export default function CreateJobModal({
|
||||
|
||||
async function handleSubmit(e: React.FormEvent<HTMLFormElement>) {
|
||||
e.preventDefault();
|
||||
if (isInference && workloadType === 'i2v' && !imagePath) return;
|
||||
if (isInference && workloadType === 'i2v' && !imagePath && !usingReferences)
|
||||
return;
|
||||
if (usingReferences && validateReferences(references)) return;
|
||||
if (
|
||||
usingReferences &&
|
||||
useGuidedPrompt &&
|
||||
isEmptyPromptFields(promptFields) &&
|
||||
!prompt.trim()
|
||||
)
|
||||
return;
|
||||
// Send the dataset id; the backend resolves it to the on-disk media dir.
|
||||
const effectiveDataPath = selectedDatasetId ?? '';
|
||||
if (!isInference && !selectedDatasetId) return;
|
||||
@@ -285,14 +536,35 @@ export default function CreateJobModal({
|
||||
try {
|
||||
const payload: CreateJobRequest = {
|
||||
model_id: modelId,
|
||||
prompt,
|
||||
name: name.trim(),
|
||||
prompt:
|
||||
usingReferences && useGuidedPrompt && !isEmptyPromptFields(promptFields)
|
||||
? serializeH3Prompt(promptFields)
|
||||
: prompt,
|
||||
workload_type: workloadType,
|
||||
job_type: effectiveJobType,
|
||||
...(isInference
|
||||
? {
|
||||
...(workloadType === 'i2v' && imagePath
|
||||
// Ref2VA and the FL2VA keyframes are mutually exclusive:
|
||||
// _prepare_ref2va rejects image_path/last_image_path outright
|
||||
// when references are present.
|
||||
...(workloadType === 'i2v' && !usingReferences && imagePath
|
||||
? { image_path: imagePath }
|
||||
: {}),
|
||||
...(workloadType === 'i2v' &&
|
||||
supportsLastImage &&
|
||||
!usingReferences &&
|
||||
lastImagePath
|
||||
? { last_image_path: lastImagePath }
|
||||
: {}),
|
||||
...(workloadType === 'i2v' && supportsLastImage && references.length
|
||||
? {
|
||||
references: references.map((r) => ({
|
||||
source: r.source,
|
||||
media_type: r.media_type,
|
||||
})),
|
||||
}
|
||||
: {}),
|
||||
negative_prompt: negativePrompt,
|
||||
num_inference_steps: numInferenceSteps,
|
||||
num_frames: numFrames,
|
||||
@@ -304,6 +576,7 @@ export default function CreateJobModal({
|
||||
seed,
|
||||
num_gpus: numGpus,
|
||||
dit_cpu_offload: ditCpuOffload,
|
||||
dit_layerwise_offload: ditLayerwiseOffload,
|
||||
text_encoder_cpu_offload: textEncoderCpuOffload,
|
||||
vae_cpu_offload: vaeCpuOffload,
|
||||
image_encoder_cpu_offload: imageEncoderCpuOffload,
|
||||
@@ -334,7 +607,14 @@ export default function CreateJobModal({
|
||||
: {}),
|
||||
}),
|
||||
};
|
||||
await createJob(payload);
|
||||
if (editingJob) {
|
||||
await updateJob(
|
||||
editingJob.id,
|
||||
payload as unknown as Record<string, unknown>,
|
||||
);
|
||||
} else {
|
||||
await createJob(payload);
|
||||
}
|
||||
onSuccess();
|
||||
onClose();
|
||||
} catch (err) {
|
||||
@@ -356,9 +636,9 @@ export default function CreateJobModal({
|
||||
|
||||
const workloadLabel =
|
||||
WORKLOAD_OPTIONS[jobType]?.find((o) => o.type === workloadType)?.label ?? '';
|
||||
const title = `New ${jobType.charAt(0).toUpperCase() + jobType.slice(1)} Job${
|
||||
workloadLabel ? ` (${workloadLabel})` : ''
|
||||
}`;
|
||||
const title = `${readOnly ? 'View' : editingJob ? 'Edit' : 'New'} ${
|
||||
jobType.charAt(0).toUpperCase() + jobType.slice(1)
|
||||
} Job${workloadLabel ? ` (${workloadLabel})` : ''}`;
|
||||
|
||||
return (
|
||||
<Dialog
|
||||
@@ -369,6 +649,7 @@ export default function CreateJobModal({
|
||||
>
|
||||
<DialogContent
|
||||
className="max-h-[90vh] w-[90vw] max-w-[850px] overflow-y-auto"
|
||||
onCloseAutoFocus={onCloseAutoFocus}
|
||||
onEscapeKeyDown={(e) => {
|
||||
if (isSubmitting) e.preventDefault();
|
||||
}}
|
||||
@@ -385,6 +666,23 @@ export default function CreateJobModal({
|
||||
autoComplete="off"
|
||||
className="flex flex-col gap-3.5"
|
||||
>
|
||||
{/* disabled cascades to every control inside; display:contents
|
||||
keeps the parent's flex layout. */}
|
||||
<fieldset
|
||||
disabled={readOnly}
|
||||
style={{ display: 'contents' }}
|
||||
className="contents"
|
||||
>
|
||||
<FieldRow htmlFor="modal-name" label="Name (optional)">
|
||||
<Input
|
||||
id="modal-name"
|
||||
value={name}
|
||||
onChange={(e) => setName(e.target.value)}
|
||||
placeholder="Shown on the job card and used for the output filename"
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
</FieldRow>
|
||||
|
||||
<FieldRow htmlFor="modal-modelId" label="Model">
|
||||
<NativeSelect
|
||||
id="modal-modelId"
|
||||
@@ -429,12 +727,12 @@ export default function CreateJobModal({
|
||||
type="file"
|
||||
accept=".png,.jpg,.jpeg,.webp,.bmp"
|
||||
onChange={handleImageChange}
|
||||
disabled={isSubmitting || isUploadingImage}
|
||||
disabled={isSubmitting || isUploadingImage || usingReferences}
|
||||
aria-describedby={
|
||||
imageUploadError ? 'modal-image-error' : undefined
|
||||
}
|
||||
aria-invalid={imageUploadError ? true : undefined}
|
||||
required
|
||||
required={!usingReferences}
|
||||
className="h-auto py-2 file:mr-3 file:cursor-pointer file:rounded-md file:border-0 file:bg-secondary file:px-2 file:py-1 file:text-sm file:text-secondary-foreground"
|
||||
/>
|
||||
{imageFileName && (
|
||||
@@ -462,24 +760,169 @@ export default function CreateJobModal({
|
||||
</FieldRow>
|
||||
)}
|
||||
|
||||
<FieldRow
|
||||
htmlFor="modal-prompt"
|
||||
label={isInference ? 'Prompt' : 'Description'}
|
||||
>
|
||||
<Textarea
|
||||
id="modal-prompt"
|
||||
value={prompt}
|
||||
onChange={(e) => setPrompt(e.target.value)}
|
||||
rows={isInference ? 3 : 2}
|
||||
placeholder={
|
||||
isInference
|
||||
? 'A curious raccoon peers through a vibrant field of yellow sunflowers…'
|
||||
: 'Brief description of this training job…'
|
||||
}
|
||||
required
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
</FieldRow>
|
||||
{isInference && workloadType === 'i2v' && supportsLastImage && (
|
||||
<FieldRow htmlFor="modal-last-image" label="End Frame (optional)">
|
||||
<Input
|
||||
id="modal-last-image"
|
||||
type="file"
|
||||
accept=".png,.jpg,.jpeg,.webp,.bmp"
|
||||
onChange={handleLastImageChange}
|
||||
disabled={isSubmitting || isUploadingLastImage}
|
||||
aria-describedby={
|
||||
lastImageUploadError ? 'modal-last-image-error' : undefined
|
||||
}
|
||||
aria-invalid={lastImageUploadError ? true : undefined}
|
||||
className="h-auto py-2 file:mr-3 file:cursor-pointer file:rounded-md file:border-0 file:bg-secondary file:px-2 file:py-1 file:text-sm file:text-secondary-foreground"
|
||||
/>
|
||||
{lastImageFileName && (
|
||||
<span className="mt-0.5 text-xs text-muted-foreground">
|
||||
{isUploadingLastImage ? 'Uploading…' : lastImageFileName} ·{' '}
|
||||
<button
|
||||
type="button"
|
||||
onClick={clearLastImage}
|
||||
disabled={isSubmitting || isUploadingLastImage}
|
||||
className="text-accent-blue underline-offset-2 hover:underline disabled:cursor-not-allowed disabled:opacity-50"
|
||||
>
|
||||
Clear
|
||||
</button>
|
||||
</span>
|
||||
)}
|
||||
{lastImageUploadError && (
|
||||
<p
|
||||
id="modal-last-image-error"
|
||||
role="alert"
|
||||
className="text-sm text-destructive"
|
||||
>
|
||||
{lastImageUploadError}
|
||||
</p>
|
||||
)}
|
||||
</FieldRow>
|
||||
)}
|
||||
|
||||
{isInference && workloadType === 'i2v' && supportsLastImage && (
|
||||
<FieldRow htmlFor="modal-reference" label="References (Ref2VA)">
|
||||
<Input
|
||||
id="modal-reference"
|
||||
type="file"
|
||||
accept=".png,.jpg,.jpeg,.webp,.bmp,.mp4,.mov,.mkv,.webm,.avi,.wav,.mp3,.flac,.m4a,.ogg"
|
||||
onChange={handleAddReference}
|
||||
disabled={
|
||||
isSubmitting ||
|
||||
isUploadingReference ||
|
||||
references.length >= H3_MAX_REFERENCES
|
||||
}
|
||||
className="h-auto py-2 file:mr-3 file:cursor-pointer file:rounded-md file:border-0 file:bg-secondary file:px-2 file:py-1 file:text-sm file:text-secondary-foreground"
|
||||
/>
|
||||
{isUploadingReference && (
|
||||
<span className="mt-0.5 text-xs text-muted-foreground">
|
||||
Uploading…
|
||||
</span>
|
||||
)}
|
||||
{references.length > 0 && (
|
||||
<ul className="mt-1 flex list-none flex-col gap-1 p-0">
|
||||
{references.map((reference, index) => (
|
||||
<li
|
||||
key={reference.id}
|
||||
className="flex items-center gap-2 text-xs text-muted-foreground"
|
||||
>
|
||||
<code className="font-mono text-accent-blue">
|
||||
{labelReferences(references)[index]}
|
||||
</code>
|
||||
<span className="truncate">{reference.fileName}</span>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => removeReference(reference.id)}
|
||||
disabled={isSubmitting}
|
||||
className="ml-auto text-accent-blue underline-offset-2 hover:underline disabled:cursor-not-allowed disabled:opacity-50"
|
||||
>
|
||||
Remove
|
||||
</button>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
{references.length > 0 && (
|
||||
<button
|
||||
type="button"
|
||||
onClick={seedPromptFields}
|
||||
disabled={isSubmitting}
|
||||
className="mt-1 self-start text-xs text-accent-blue underline-offset-2 hover:underline disabled:cursor-not-allowed disabled:opacity-50"
|
||||
>
|
||||
Fill prompt sections from references
|
||||
</button>
|
||||
)}
|
||||
{referenceError && (
|
||||
<p role="alert" className="mt-0.5 text-xs text-destructive">
|
||||
{referenceError}
|
||||
</p>
|
||||
)}
|
||||
<span className="mt-0.5 text-xs text-muted-foreground">
|
||||
Ref2VA replaces the keyframes: up to 9 images, 3 videos, 3 audio
|
||||
(12 total). Audio needs at least one image or video.
|
||||
</span>
|
||||
</FieldRow>
|
||||
)}
|
||||
|
||||
{usingReferences && useGuidedPrompt ? (
|
||||
/* Six-section format from the model's reference prompt guide. */
|
||||
<>
|
||||
{H3_PROMPT_SECTIONS.map((section) => (
|
||||
<FieldRow
|
||||
key={section}
|
||||
htmlFor={`modal-prompt-${section}`}
|
||||
label={H3_SECTION_LABELS[section]}
|
||||
>
|
||||
<Textarea
|
||||
id={`modal-prompt-${section}`}
|
||||
value={promptFields[section]}
|
||||
onChange={(e) => setPromptField(section, e.target.value)}
|
||||
rows={section === 'detailed_description' ? 5 : 2}
|
||||
placeholder={H3_SECTION_HINTS[section]}
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
</FieldRow>
|
||||
))}
|
||||
<button
|
||||
type="button"
|
||||
onClick={toggleGuidedPrompt}
|
||||
disabled={isSubmitting}
|
||||
className="self-start text-xs text-accent-blue underline-offset-2 hover:underline disabled:cursor-not-allowed disabled:opacity-50"
|
||||
>
|
||||
Edit as raw prompt
|
||||
</button>
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
<FieldRow
|
||||
htmlFor="modal-prompt"
|
||||
label={isInference ? 'Prompt' : 'Description'}
|
||||
>
|
||||
<Textarea
|
||||
id="modal-prompt"
|
||||
value={prompt}
|
||||
onChange={(e) => setPrompt(e.target.value)}
|
||||
rows={isInference ? 3 : 2}
|
||||
placeholder={
|
||||
isInference
|
||||
? 'A curious raccoon peers through a vibrant field of yellow sunflowers…'
|
||||
: 'Brief description of this training job…'
|
||||
}
|
||||
required={!(usingReferences && useGuidedPrompt)}
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
</FieldRow>
|
||||
{usingReferences && (
|
||||
<button
|
||||
type="button"
|
||||
onClick={toggleGuidedPrompt}
|
||||
disabled={isSubmitting}
|
||||
className="self-start text-xs text-accent-blue underline-offset-2 hover:underline disabled:cursor-not-allowed disabled:opacity-50"
|
||||
>
|
||||
Edit as prompt sections
|
||||
</button>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
|
||||
{isInference && (
|
||||
<FieldRow htmlFor="modal-negative-prompt" label="Negative Prompt">
|
||||
@@ -849,6 +1292,13 @@ export default function CreateJobModal({
|
||||
onChange={setDitCpuOffload}
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
<ToggleRow
|
||||
id="modal-dit-layerwise-offload"
|
||||
label="DiT Layerwise Offload"
|
||||
checked={ditLayerwiseOffload}
|
||||
onChange={handleDitLayerwiseOffloadChange}
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
<ToggleRow
|
||||
id="modal-text-encoder-cpu-offload"
|
||||
label="Text Encoder CPU Offload"
|
||||
@@ -860,7 +1310,7 @@ export default function CreateJobModal({
|
||||
id="modal-use-fsdp-inference"
|
||||
label="Use FSDP Inference"
|
||||
checked={useFsdpInference}
|
||||
onChange={setUseFsdpInference}
|
||||
onChange={handleUseFsdpInferenceChange}
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
<ToggleRow
|
||||
@@ -891,7 +1341,7 @@ export default function CreateJobModal({
|
||||
max={8}
|
||||
step={1}
|
||||
value={numGpus}
|
||||
onChange={setNumGpus}
|
||||
onChange={handleNumGpusChange}
|
||||
disabled={isSubmitting}
|
||||
/>
|
||||
<NumberRow
|
||||
@@ -906,6 +1356,8 @@ export default function CreateJobModal({
|
||||
</details>
|
||||
)}
|
||||
|
||||
</fieldset>
|
||||
|
||||
<div className="flex flex-col items-start gap-2">
|
||||
{submitError && (
|
||||
<p role="alert" className="text-sm text-destructive">
|
||||
@@ -914,14 +1366,22 @@ export default function CreateJobModal({
|
||||
)}
|
||||
<Button
|
||||
type="submit"
|
||||
hidden={readOnly}
|
||||
disabled={
|
||||
readOnly ||
|
||||
isSubmitting ||
|
||||
isUploadingImage ||
|
||||
!!modelLoadError ||
|
||||
!!datasetLoadError
|
||||
}
|
||||
>
|
||||
{isSubmitting ? 'Creating…' : 'Create Job'}
|
||||
{isSubmitting
|
||||
? editingJob
|
||||
? 'Saving…'
|
||||
: 'Creating…'
|
||||
: editingJob
|
||||
? 'Save Changes'
|
||||
: 'Create Job'}
|
||||
</Button>
|
||||
</div>
|
||||
</form>
|
||||
|
||||
@@ -3,11 +3,13 @@
|
||||
import * as React from 'react';
|
||||
import { Timer } from 'lucide-react';
|
||||
|
||||
import CreateJobModal from '@/components/jobs/CreateJobModal';
|
||||
import { Badge, type BadgeProps } from '@/components/ui/badge';
|
||||
import { Button } from '@/components/ui/button';
|
||||
import { useStore } from '@/hooks/useStore';
|
||||
import {
|
||||
deleteJob,
|
||||
duplicateJob,
|
||||
downloadJobVideo,
|
||||
startJob,
|
||||
stopJob,
|
||||
@@ -62,6 +64,8 @@ export default function JobCard({ job, onJobUpdated }: JobCardProps) {
|
||||
const isSelected = activeJobId === job.id;
|
||||
|
||||
const [isLoading, setIsLoading] = React.useState(false);
|
||||
const [isEditing, setIsEditing] = React.useState(false);
|
||||
const [isViewing, setIsViewing] = React.useState(false);
|
||||
const [currentTime, setCurrentTime] = React.useState(() => Date.now());
|
||||
|
||||
const elapsedTime = computeElapsed(job, currentTime);
|
||||
@@ -103,6 +107,21 @@ export default function JobCard({ job, onJobUpdated }: JobCardProps) {
|
||||
}
|
||||
}
|
||||
|
||||
async function handleDuplicate(e: React.MouseEvent) {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
if (isLoading) return;
|
||||
setIsLoading(true);
|
||||
try {
|
||||
await duplicateJob(job.id);
|
||||
onJobUpdated?.();
|
||||
} catch (err) {
|
||||
alert(err instanceof Error ? err.message : 'Failed to duplicate job');
|
||||
} finally {
|
||||
setIsLoading(false);
|
||||
}
|
||||
}
|
||||
|
||||
async function handleDelete(e: React.MouseEvent) {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
@@ -156,16 +175,23 @@ export default function JobCard({ job, onJobUpdated }: JobCardProps) {
|
||||
>
|
||||
<span className="flex flex-wrap items-center justify-between gap-2">
|
||||
<span className="text-[0.95rem] font-semibold text-foreground">
|
||||
{job.model_id}
|
||||
{job.name?.trim() || job.model_id}
|
||||
</span>
|
||||
<Badge variant={BADGE_VARIANTS[job.status] ?? 'secondary'}>
|
||||
{job.status}
|
||||
</Badge>
|
||||
</span>
|
||||
<span className="max-w-full overflow-hidden text-ellipsis whitespace-nowrap text-sm text-muted-foreground">
|
||||
{job.prompt}
|
||||
{job.name?.trim() ? `${job.model_id} · ${job.prompt}` : job.prompt}
|
||||
</span>
|
||||
<span className="flex flex-wrap items-center gap-4 text-xs text-muted-foreground">
|
||||
{/* Short job id; logs and output dirs are keyed on the full UUID. */}
|
||||
<span
|
||||
className="font-mono text-muted-foreground/80"
|
||||
title={job.id}
|
||||
>
|
||||
{job.id.slice(0, 8)}
|
||||
</span>
|
||||
{job.job_type === 'inference' ? (
|
||||
<>
|
||||
<span>{job.num_frames} frames</span>
|
||||
@@ -226,6 +252,51 @@ export default function JobCard({ job, onJobUpdated }: JobCardProps) {
|
||||
Download Video
|
||||
</Button>
|
||||
)}
|
||||
{!(
|
||||
job.status === 'pending' ||
|
||||
job.status === 'failed' ||
|
||||
job.status === 'stopped'
|
||||
) && (
|
||||
<Button
|
||||
size="sm"
|
||||
variant="outline"
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
setIsViewing(true);
|
||||
}}
|
||||
disabled={isLoading}
|
||||
title="View this job's configuration"
|
||||
>
|
||||
View
|
||||
</Button>
|
||||
)}
|
||||
{(job.status === 'pending' ||
|
||||
job.status === 'failed' ||
|
||||
job.status === 'stopped') && (
|
||||
<Button
|
||||
size="sm"
|
||||
variant="outline"
|
||||
onClick={(e) => {
|
||||
e.preventDefault();
|
||||
e.stopPropagation();
|
||||
setIsEditing(true);
|
||||
}}
|
||||
disabled={isLoading}
|
||||
title="Edit this job's configuration"
|
||||
>
|
||||
Edit
|
||||
</Button>
|
||||
)}
|
||||
<Button
|
||||
size="sm"
|
||||
variant="outline"
|
||||
onClick={handleDuplicate}
|
||||
disabled={isLoading}
|
||||
title="Create a new pending job with this configuration"
|
||||
>
|
||||
Duplicate
|
||||
</Button>
|
||||
<Button
|
||||
size="sm"
|
||||
variant="destructive"
|
||||
@@ -235,6 +306,30 @@ export default function JobCard({ job, onJobUpdated }: JobCardProps) {
|
||||
Delete
|
||||
</Button>
|
||||
</div>
|
||||
{isViewing && (
|
||||
<CreateJobModal
|
||||
isOpen
|
||||
readOnly
|
||||
editingJob={job}
|
||||
jobType={(job.job_type ?? 'inference') as never}
|
||||
workloadType={job.workload_type ?? 't2v'}
|
||||
onClose={() => setIsViewing(false)}
|
||||
onSuccess={() => setIsViewing(false)}
|
||||
/>
|
||||
)}
|
||||
{isEditing && (
|
||||
<CreateJobModal
|
||||
isOpen
|
||||
editingJob={job}
|
||||
jobType={(job.job_type ?? 'inference') as never}
|
||||
workloadType={job.workload_type ?? 't2v'}
|
||||
onClose={() => setIsEditing(false)}
|
||||
onSuccess={() => {
|
||||
setIsEditing(false);
|
||||
onJobUpdated?.();
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
</article>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -153,9 +153,16 @@ export default function JobDetailsSidebar({
|
||||
}}
|
||||
>
|
||||
<div className="flex items-center justify-between border-b border-border px-5 py-4">
|
||||
<h2 className="m-0 text-base font-semibold text-foreground">
|
||||
Job Details
|
||||
</h2>
|
||||
<div className="min-w-0">
|
||||
<h2 className="m-0 text-base font-semibold text-foreground">
|
||||
Job Details
|
||||
</h2>
|
||||
{/* Full job id: keys ~/h3_studio_logs/<id>.log, the output directory
|
||||
and every API route, so make it selectable for copy/paste. */}
|
||||
<code className="mt-0.5 block select-all truncate font-mono text-xs text-muted-foreground">
|
||||
{job.id}
|
||||
</code>
|
||||
</div>
|
||||
<div className="flex items-center gap-2">
|
||||
<Button
|
||||
type="button"
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import * as React from 'react';
|
||||
import { render, screen } from '@testing-library/react';
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import userEvent from '@testing-library/user-event';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { Button } from './button';
|
||||
import { Input } from './input';
|
||||
@@ -8,6 +10,24 @@ import { Slider } from './slider';
|
||||
import { Switch } from './switch';
|
||||
|
||||
describe('shared control accessibility', () => {
|
||||
it('forwards refs and click handlers to the asChild button', async () => {
|
||||
const user = userEvent.setup();
|
||||
const ref = React.createRef<HTMLButtonElement>();
|
||||
const onClick = vi.fn();
|
||||
render(
|
||||
<Button asChild ref={ref} onClick={onClick}>
|
||||
<button type="button">Slotted action</button>
|
||||
</Button>,
|
||||
);
|
||||
|
||||
const button = screen.getByRole('button', { name: 'Slotted action' });
|
||||
expect(screen.getAllByRole('button')).toHaveLength(1);
|
||||
expect(ref.current).toBe(button);
|
||||
await user.click(button);
|
||||
expect(onClick).toHaveBeenCalledTimes(1);
|
||||
expect(button).toHaveFocus();
|
||||
});
|
||||
|
||||
it('keeps button, input, and select targets at least 44px tall', () => {
|
||||
render(
|
||||
<>
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"use client";
|
||||
|
||||
import * as React from "react";
|
||||
import { Slot } from "@radix-ui/react-slot";
|
||||
import { Slot } from "radix-ui";
|
||||
import { cva, type VariantProps } from "class-variance-authority";
|
||||
|
||||
import { cn } from "@/lib/utils";
|
||||
@@ -37,7 +37,7 @@ export interface ButtonProps extends React.ButtonHTMLAttributes<HTMLButtonElemen
|
||||
}
|
||||
|
||||
const Button = React.forwardRef<HTMLButtonElement, ButtonProps>(({ className, variant, size, asChild = false, ...props }, ref) => {
|
||||
const Comp = asChild ? Slot : "button";
|
||||
const Comp = asChild ? Slot.Root : "button";
|
||||
return <Comp className={cn(buttonVariants({ variant, size, className }))} ref={ref} {...props} />;
|
||||
});
|
||||
Button.displayName = "Button";
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as DialogPrimitive from '@radix-ui/react-dialog';
|
||||
import { Dialog as DialogPrimitive } from 'radix-ui';
|
||||
import { X } from 'lucide-react';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as LabelPrimitive from '@radix-ui/react-label';
|
||||
import { Label as LabelPrimitive } from 'radix-ui';
|
||||
import { cva, type VariantProps } from 'class-variance-authority';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as ScrollAreaPrimitive from '@radix-ui/react-scroll-area';
|
||||
import { ScrollArea as ScrollAreaPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SelectPrimitive from '@radix-ui/react-select';
|
||||
import { Select as SelectPrimitive } from 'radix-ui';
|
||||
import { Check, ChevronDown, ChevronUp } from 'lucide-react';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SeparatorPrimitive from '@radix-ui/react-separator';
|
||||
import { Separator as SeparatorPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SliderPrimitive from '@radix-ui/react-slider';
|
||||
import { Slider as SliderPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as SwitchPrimitives from '@radix-ui/react-switch';
|
||||
import { Switch as SwitchPrimitives } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
'use client';
|
||||
|
||||
import * as React from 'react';
|
||||
import * as TabsPrimitive from '@radix-ui/react-tabs';
|
||||
import { Tabs as TabsPrimitive } from 'radix-ui';
|
||||
|
||||
import { cn } from '@/lib/utils';
|
||||
|
||||
|
||||
@@ -56,6 +56,8 @@ export function getJobVideoUrl(jobId: string): string {
|
||||
|
||||
export interface CreateJobRequest {
|
||||
model_id: string;
|
||||
/** Optional label; the card and output filename fall back to the prompt. */
|
||||
name?: string;
|
||||
prompt: string;
|
||||
workload_type?: string;
|
||||
job_type?: JobType;
|
||||
@@ -152,6 +154,65 @@ export async function updateSettings(
|
||||
return response.json();
|
||||
}
|
||||
|
||||
export type MediaType = "image" | "video" | "audio";
|
||||
|
||||
/**
|
||||
* Upload an image, video or audio file for a MiniMax-H3 Ref2VA reference.
|
||||
* The server derives media_type from the extension and returns it, so callers
|
||||
* do not have to duplicate that mapping.
|
||||
*/
|
||||
/** Create a new pending job with the same configuration as an existing one. */
|
||||
export async function duplicateJob(jobId: string): Promise<{ id: string }> {
|
||||
const baseApiUrl = getApiBaseUrl();
|
||||
const response = await fetch(`${baseApiUrl}/jobs/${jobId}/duplicate`, {
|
||||
method: "POST",
|
||||
});
|
||||
if (!response.ok) {
|
||||
const err = await response
|
||||
.json()
|
||||
.catch(() => ({ detail: "Duplicate failed" }));
|
||||
throw new Error(err.detail || "Duplicate failed");
|
||||
}
|
||||
return response.json();
|
||||
}
|
||||
|
||||
/** Edit a pending job's configuration. Started jobs are rejected by the API. */
|
||||
export async function updateJob(
|
||||
jobId: string,
|
||||
updates: Record<string, unknown>,
|
||||
): Promise<unknown> {
|
||||
const baseApiUrl = getApiBaseUrl();
|
||||
const response = await fetch(`${baseApiUrl}/jobs/${jobId}`, {
|
||||
method: "PATCH",
|
||||
headers: { "Content-Type": "application/json" },
|
||||
body: JSON.stringify(updates),
|
||||
});
|
||||
if (!response.ok) {
|
||||
const err = await response.json().catch(() => ({ detail: "Update failed" }));
|
||||
throw new Error(err.detail || "Update failed");
|
||||
}
|
||||
return response.json();
|
||||
}
|
||||
|
||||
export async function uploadMedia(
|
||||
file: File,
|
||||
): Promise<{ path: string; media_type: MediaType }> {
|
||||
const baseApiUrl = getApiBaseUrl();
|
||||
const formData = new FormData();
|
||||
formData.append("file", file);
|
||||
const response = await fetch(`${baseApiUrl}/upload-media`, {
|
||||
method: "POST",
|
||||
body: formData,
|
||||
});
|
||||
if (!response.ok) {
|
||||
const err = await response
|
||||
.json()
|
||||
.catch(() => ({ detail: "Upload failed" }));
|
||||
throw new Error(err.detail || "Upload failed");
|
||||
}
|
||||
return response.json();
|
||||
}
|
||||
|
||||
export async function uploadImage(file: File): Promise<{ path: string }> {
|
||||
const baseApiUrl = getApiBaseUrl();
|
||||
const formData = new FormData();
|
||||
|
||||
@@ -16,6 +16,7 @@ export interface DefaultOptions {
|
||||
seed: number;
|
||||
numGpus: number;
|
||||
ditCpuOffload: boolean;
|
||||
ditLayerwiseOffload: boolean;
|
||||
textEncoderCpuOffload: boolean;
|
||||
vaeCpuOffload: boolean;
|
||||
imageEncoderCpuOffload: boolean;
|
||||
@@ -44,6 +45,7 @@ export const DEFAULT_OPTIONS: DefaultOptions = {
|
||||
seed: 1024,
|
||||
numGpus: 1,
|
||||
ditCpuOffload: false,
|
||||
ditLayerwiseOffload: false,
|
||||
textEncoderCpuOffload: false,
|
||||
vaeCpuOffload: false,
|
||||
imageEncoderCpuOffload: false,
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
EMPTY_H3_PROMPT_FIELDS,
|
||||
H3_PROMPT_SECTIONS,
|
||||
isEmptyPromptFields,
|
||||
parseH3Prompt,
|
||||
serializeH3Prompt,
|
||||
type H3PromptFields,
|
||||
} from "@/lib/h3Prompt";
|
||||
|
||||
const filled: H3PromptFields = {
|
||||
subject_definitions: "<Subject 1> is the dog in <Picture 1>.",
|
||||
summary: "[reference generation] The target video shows <Subject 1>.",
|
||||
retention_analysis: "<Subject 1> (appears in [Shot 1]): fully_preserved - fur retained.",
|
||||
detailed_description: "[Shot 1] A medium shot establishes <Subject 1>.",
|
||||
overall_soundscape: "Room tone throughout.",
|
||||
non_diegetic_music: "N/A",
|
||||
};
|
||||
|
||||
describe("serializeH3Prompt", () => {
|
||||
it("emits sections in order, flush left, blank-line separated", () => {
|
||||
const out = serializeH3Prompt(filled);
|
||||
expect(out).toBe(
|
||||
[
|
||||
"subject_definitions:\n<Subject 1> is the dog in <Picture 1>.",
|
||||
"summary:\n[reference generation] The target video shows <Subject 1>.",
|
||||
"retention_analysis:\n<Subject 1> (appears in [Shot 1]): fully_preserved - fur retained.",
|
||||
"detailed_description:\n[Shot 1] A medium shot establishes <Subject 1>.",
|
||||
"overall_soundscape:\nRoom tone throughout.",
|
||||
"non_diegetic_music:\nN/A",
|
||||
].join("\n\n"),
|
||||
);
|
||||
// content is never indented
|
||||
expect(out).not.toMatch(/\n {2}\S/);
|
||||
});
|
||||
|
||||
it("fills blank sections with N/A rather than dropping them", () => {
|
||||
const out = serializeH3Prompt({ ...EMPTY_H3_PROMPT_FIELDS, summary: "x" });
|
||||
for (const s of H3_PROMPT_SECTIONS) expect(out).toContain(`${s}:`);
|
||||
expect(out).toContain("non_diegetic_music:\nN/A");
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseH3Prompt", () => {
|
||||
it("round-trips a serialized prompt", () => {
|
||||
expect(parseH3Prompt(serializeH3Prompt(filled))).toEqual(filled);
|
||||
});
|
||||
|
||||
it("keeps multi-line section bodies", () => {
|
||||
const p = parseH3Prompt("summary:\nline one\nline two\n\ndetailed_description:\nd");
|
||||
expect(p?.summary).toBe("line one\nline two");
|
||||
expect(p?.detailed_description).toBe("d");
|
||||
});
|
||||
|
||||
it("returns null for a plain prompt", () => {
|
||||
expect(parseH3Prompt("a toy car drives into a plush dog")).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe("isEmptyPromptFields", () => {
|
||||
it("detects blank vs filled", () => {
|
||||
expect(isEmptyPromptFields(EMPTY_H3_PROMPT_FIELDS)).toBe(true);
|
||||
expect(isEmptyPromptFields(filled)).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,94 @@
|
||||
/**
|
||||
* MiniMax-H3 full-reference prompt sections. Serialization follows the worked
|
||||
* example in the model's VIDEO_PROMPT_WRITING_GUIDE_ref_en.md: `section_name:`
|
||||
* on its own line, content flush left, one blank line between sections.
|
||||
*/
|
||||
|
||||
export const H3_PROMPT_SECTIONS = [
|
||||
'subject_definitions',
|
||||
'summary',
|
||||
'retention_analysis',
|
||||
'detailed_description',
|
||||
'overall_soundscape',
|
||||
'non_diegetic_music',
|
||||
] as const;
|
||||
|
||||
export type H3PromptSection = (typeof H3_PROMPT_SECTIONS)[number];
|
||||
|
||||
export type H3PromptFields = Record<H3PromptSection, string>;
|
||||
|
||||
export const EMPTY_H3_PROMPT_FIELDS: H3PromptFields = {
|
||||
subject_definitions: '',
|
||||
summary: '',
|
||||
retention_analysis: '',
|
||||
detailed_description: '',
|
||||
overall_soundscape: '',
|
||||
non_diegetic_music: '',
|
||||
};
|
||||
|
||||
export const H3_SECTION_LABELS: Record<H3PromptSection, string> = {
|
||||
subject_definitions: 'Subject definitions',
|
||||
summary: 'Summary',
|
||||
retention_analysis: 'Retention analysis',
|
||||
detailed_description: 'Detailed description',
|
||||
overall_soundscape: 'Overall soundscape',
|
||||
non_diegetic_music: 'Non-diegetic music',
|
||||
};
|
||||
|
||||
/** Per-section guidance, condensed from the guide's rules for each section. */
|
||||
export const H3_SECTION_HINTS: Record<H3PromptSection, string> = {
|
||||
subject_definitions:
|
||||
'One line per <Subject N>. A subject may draw on several references, e.g. "<Subject 2> is the dog in <Picture 2>, <Picture 3>, and <Picture 4>."',
|
||||
summary:
|
||||
'One paragraph, starting with a [task type] prefix: reference generation, keyframe completion, video editing, video continuation, audio reuse, audio reference. Combine with " + ".',
|
||||
retention_analysis:
|
||||
'Per subject: "<Subject 1> (appears in [Shot 1], [Shot 2]): fully_preserved - what is retained."',
|
||||
detailed_description:
|
||||
'Playback order, using [Shot N] markers, timestamps, (S1) speaker tags and <d>[English] dialogue</d>.',
|
||||
overall_soundscape: 'Ambience and physical sounds. N/A if none.',
|
||||
non_diegetic_music:
|
||||
'Background music audible only to the audience. N/A if none.',
|
||||
};
|
||||
|
||||
/** True when every section is blank. */
|
||||
export function isEmptyPromptFields(fields: H3PromptFields): boolean {
|
||||
return H3_PROMPT_SECTIONS.every((s) => !fields[s].trim());
|
||||
}
|
||||
|
||||
/**
|
||||
* Join the sections into the prompt string the model is given. Blank sections
|
||||
* become "N/A" rather than being dropped, matching the guide's example.
|
||||
*/
|
||||
export function serializeH3Prompt(fields: H3PromptFields): string {
|
||||
return H3_PROMPT_SECTIONS.map((section) => {
|
||||
const body = fields[section].trim() || 'N/A';
|
||||
return `${section}:\n${body}`;
|
||||
}).join('\n\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a serialized prompt back into sections, so switching between the
|
||||
* guided fields and the raw editor does not lose work. Returns null when the
|
||||
* text is not in section format (a plain prompt, say).
|
||||
*/
|
||||
export function parseH3Prompt(text: string): H3PromptFields | null {
|
||||
const fields = { ...EMPTY_H3_PROMPT_FIELDS };
|
||||
const headings = new Set<string>(H3_PROMPT_SECTIONS);
|
||||
let current: H3PromptSection | null = null;
|
||||
let found = false;
|
||||
|
||||
for (const line of text.split('\n')) {
|
||||
const heading = line.trim().replace(/:$/, '');
|
||||
if (line.trim().endsWith(':') && headings.has(heading)) {
|
||||
current = heading as H3PromptSection;
|
||||
found = true;
|
||||
continue;
|
||||
}
|
||||
if (current) fields[current] += (fields[current] ? '\n' : '') + line;
|
||||
}
|
||||
if (!found) return null;
|
||||
for (const section of H3_PROMPT_SECTIONS) {
|
||||
fields[section] = fields[section].trim();
|
||||
}
|
||||
return fields;
|
||||
}
|
||||
@@ -0,0 +1,70 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
labelReferences,
|
||||
referencePromptSeed,
|
||||
validateReferences,
|
||||
type H3Reference,
|
||||
} from "@/lib/h3References";
|
||||
|
||||
const ref = (media_type: H3Reference["media_type"], n: number): H3Reference => ({
|
||||
id: `${media_type}-${n}`,
|
||||
source: `/tmp/${media_type}${n}`,
|
||||
media_type,
|
||||
fileName: `${media_type}${n}`,
|
||||
});
|
||||
|
||||
describe("labelReferences", () => {
|
||||
it("numbers each media type independently, in list order", () => {
|
||||
expect(
|
||||
labelReferences([ref("image", 1), ref("video", 1), ref("image", 2)]),
|
||||
).toEqual(["<Picture 1>", "<Video 1>", "<Picture 2>"]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("validateReferences", () => {
|
||||
it("accepts an empty list and a normal mix", () => {
|
||||
expect(validateReferences([])).toBeNull();
|
||||
expect(validateReferences([ref("image", 1), ref("audio", 1)])).toBeNull();
|
||||
});
|
||||
|
||||
it("rejects audio-only lists", () => {
|
||||
expect(validateReferences([ref("audio", 1)])).toMatch(/paired/);
|
||||
});
|
||||
|
||||
it("enforces the per-type caps", () => {
|
||||
const videos = [1, 2, 3, 4].map((n) => ref("video", n));
|
||||
expect(validateReferences(videos)).toMatch(/At most 3 video/);
|
||||
});
|
||||
|
||||
it("enforces the overall cap", () => {
|
||||
const many = Array.from({ length: 13 }, (_, i) => ref("image", i));
|
||||
expect(validateReferences(many)).toMatch(/At most 12/);
|
||||
});
|
||||
});
|
||||
|
||||
describe("referencePromptSeed", () => {
|
||||
it("cites the real labels and never indents content", () => {
|
||||
const seed = referencePromptSeed([ref("image", 1), ref("video", 1)]);
|
||||
expect(seed.subject_definitions).toContain("<Picture 1>");
|
||||
expect(seed.subject_definitions).toContain("<Video 1>");
|
||||
for (const value of Object.values(seed)) {
|
||||
expect(value).not.toMatch(/^ {2}\S/m);
|
||||
}
|
||||
});
|
||||
|
||||
it("uses the guide's retention_analysis form", () => {
|
||||
const seed = referencePromptSeed([ref("image", 1)]);
|
||||
expect(seed.retention_analysis).toMatch(/fully_preserved - /);
|
||||
});
|
||||
|
||||
it("starts the summary with a bracketed task type", () => {
|
||||
const seed = referencePromptSeed([ref("image", 1)]);
|
||||
expect(seed.summary).toMatch(/^\[[a-z +]+\]/);
|
||||
});
|
||||
|
||||
it("describes audio references separately", () => {
|
||||
const seed = referencePromptSeed([ref("image", 1), ref("audio", 1)]);
|
||||
expect(seed.subject_definitions).toContain("<Audio 1>");
|
||||
expect(seed.retention_analysis).toContain("<Audio 1>: reference -");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,90 @@
|
||||
/**
|
||||
* MiniMax-H3 Ref2VA reference helpers. Labels mirror the per-type counters in
|
||||
* `build_ref2va_presentation`, so what the user sees is what the model is shown.
|
||||
*/
|
||||
import type { MediaType } from "@/lib/api";
|
||||
|
||||
export interface H3Reference {
|
||||
/** stable key for React lists */
|
||||
id: string;
|
||||
source: string;
|
||||
media_type: MediaType;
|
||||
fileName: string;
|
||||
}
|
||||
|
||||
/** Per-media-type caps enforced by validate_references (reference.py). */
|
||||
export const H3_REFERENCE_LIMITS: Record<MediaType, number> = {
|
||||
image: 9,
|
||||
video: 3,
|
||||
audio: 3,
|
||||
};
|
||||
export const H3_MAX_REFERENCES = 12;
|
||||
|
||||
const LABEL_FOR: Record<MediaType, string> = {
|
||||
image: "Picture",
|
||||
video: "Video",
|
||||
audio: "Audio",
|
||||
};
|
||||
|
||||
/** Label each reference the way the pipeline will, e.g. "<Picture 2>". */
|
||||
export function labelReferences(refs: H3Reference[]): string[] {
|
||||
const counts: Record<MediaType, number> = { image: 0, video: 0, audio: 0 };
|
||||
return refs.map((ref) => {
|
||||
counts[ref.media_type] += 1;
|
||||
return `<${LABEL_FOR[ref.media_type]} ${counts[ref.media_type]}>`;
|
||||
});
|
||||
}
|
||||
|
||||
/** Human-readable reason the list is invalid, or null when it is acceptable. */
|
||||
export function validateReferences(refs: H3Reference[]): string | null {
|
||||
if (refs.length === 0) return null;
|
||||
if (refs.length > H3_MAX_REFERENCES) {
|
||||
return `At most ${H3_MAX_REFERENCES} references (have ${refs.length}).`;
|
||||
}
|
||||
const counts: Record<MediaType, number> = { image: 0, video: 0, audio: 0 };
|
||||
for (const ref of refs) counts[ref.media_type] += 1;
|
||||
for (const type of Object.keys(counts) as MediaType[]) {
|
||||
if (counts[type] > H3_REFERENCE_LIMITS[type]) {
|
||||
return `At most ${H3_REFERENCE_LIMITS[type]} ${type} references (have ${counts[type]}).`;
|
||||
}
|
||||
}
|
||||
if (counts.audio === refs.length) {
|
||||
return "Audio references must be paired with at least one image or video.";
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Seed the guided prompt fields from the current reference list. */
|
||||
export function referencePromptSeed(
|
||||
refs: H3Reference[],
|
||||
): Record<string, string> {
|
||||
const labels = labelReferences(refs);
|
||||
const visual = labels.filter((l) => !l.startsWith("<Audio"));
|
||||
const audio = labels.filter((l) => l.startsWith("<Audio"));
|
||||
|
||||
const subjects = visual.map(
|
||||
(label, i) =>
|
||||
`<Subject ${i + 1}> is the subject from ${label}; describe appearance and distinguishing features.`,
|
||||
);
|
||||
for (const label of audio) {
|
||||
subjects.push(`${label} is the audio reference; describe what it provides.`);
|
||||
}
|
||||
|
||||
const retention = visual.map(
|
||||
(_, i) =>
|
||||
`<Subject ${i + 1}> (appears in [Shot 1]): fully_preserved - what is retained.`,
|
||||
);
|
||||
for (const label of audio) {
|
||||
retention.push(`${label}: reference - how it guides the audio.`);
|
||||
}
|
||||
|
||||
return {
|
||||
subject_definitions: subjects.join("\n"),
|
||||
summary: "[reference generation] Describe the target video and each reference's role.",
|
||||
retention_analysis: retention.join("\n"),
|
||||
detailed_description:
|
||||
"[Shot 1] Describe composition, subjects, environment, lighting, action and camera movement, saying where each reference takes effect.",
|
||||
overall_soundscape: "",
|
||||
non_diegetic_music: "",
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { jobToFormFields, referenceFileName, type JobLike } from "@/lib/jobToFields";
|
||||
|
||||
const job: JobLike = {
|
||||
id: "abc",
|
||||
model_id: "MiniMaxAI/MiniMax-H3",
|
||||
prompt: "subject_definitions:\n<Subject 1> is a dog.\n\nsummary:\n[reference generation] x",
|
||||
workload_type: "i2v",
|
||||
references: [
|
||||
{ source: "/uploads/9f/wukong_source.mp4", media_type: "video" },
|
||||
{ source: "/uploads/2a/MonkeyKing_0.jpg", media_type: "image" },
|
||||
],
|
||||
num_frames: 141,
|
||||
height: 768,
|
||||
width: 1344,
|
||||
guidance_scale: 1.0,
|
||||
guidance_rescale: 0.1,
|
||||
seed: 0,
|
||||
num_gpus: 4,
|
||||
use_fsdp_inference: true,
|
||||
};
|
||||
|
||||
describe("jobToFormFields", () => {
|
||||
it("carries the settings that differ from form defaults", () => {
|
||||
const f = jobToFormFields(job);
|
||||
expect(f.numFrames).toBe(141);
|
||||
expect(f.height).toBe(768);
|
||||
expect(f.width).toBe(1344);
|
||||
expect(f.guidanceScale).toBe(1.0);
|
||||
expect(f.guidanceRescale).toBe(0.1);
|
||||
expect(f.seed).toBe(0); // 0 must survive, not fall back to 1024
|
||||
expect(f.numGpus).toBe(4);
|
||||
expect(f.useFsdpInference).toBe(true);
|
||||
});
|
||||
|
||||
it("does not let falsy-but-valid values fall through to defaults", () => {
|
||||
const f = jobToFormFields({ ...job, seed: 0, vsa_sparsity: 0, guidance_rescale: 0 });
|
||||
expect(f.seed).toBe(0);
|
||||
expect(f.vsaSparsity).toBe(0);
|
||||
expect(f.guidanceRescale).toBe(0);
|
||||
});
|
||||
|
||||
it("rebuilds the reference list with readable names", () => {
|
||||
const f = jobToFormFields(job);
|
||||
expect(f.references).toHaveLength(2);
|
||||
expect(f.references[0].fileName).toBe("wukong_source.mp4");
|
||||
expect(f.references[1].media_type).toBe("image");
|
||||
expect(new Set(f.references.map((r) => r.id)).size).toBe(2);
|
||||
});
|
||||
|
||||
it("splits a six-section prompt back into fields", () => {
|
||||
const f = jobToFormFields(job);
|
||||
expect(f.promptFields?.subject_definitions).toBe("<Subject 1> is a dog.");
|
||||
expect(f.promptFields?.summary).toBe("[reference generation] x");
|
||||
});
|
||||
|
||||
it("returns null promptFields for a plain prompt", () => {
|
||||
const f = jobToFormFields({ ...job, prompt: "a toy car" });
|
||||
expect(f.promptFields).toBeNull();
|
||||
expect(f.prompt).toBe("a toy car");
|
||||
});
|
||||
|
||||
it("handles a job with no references", () => {
|
||||
const f = jobToFormFields({ ...job, references: null });
|
||||
expect(f.references).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("referenceFileName", () => {
|
||||
it("takes the basename", () => {
|
||||
expect(referenceFileName("/a/b/c.mp4")).toBe("c.mp4");
|
||||
expect(referenceFileName("c.mp4")).toBe("c.mp4");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,119 @@
|
||||
/**
|
||||
* Map a persisted job back onto the create-job form's fields. Every value the
|
||||
* form seeds from defaults must be covered here, or edit mode silently shows
|
||||
* defaults for whatever is missing.
|
||||
*/
|
||||
import type { H3Reference } from "@/lib/h3References";
|
||||
import { parseH3Prompt, type H3PromptFields } from "@/lib/h3Prompt";
|
||||
|
||||
export interface JobLike {
|
||||
id: string;
|
||||
model_id: string;
|
||||
name?: string;
|
||||
prompt: string;
|
||||
workload_type?: string;
|
||||
job_type?: string;
|
||||
image_path?: string;
|
||||
last_image_path?: string;
|
||||
references?: { source: string; media_type: string }[] | null;
|
||||
negative_prompt?: string;
|
||||
num_inference_steps?: number;
|
||||
num_frames?: number;
|
||||
height?: number;
|
||||
width?: number;
|
||||
guidance_scale?: number;
|
||||
guidance_rescale?: number;
|
||||
fps?: number;
|
||||
seed?: number;
|
||||
num_gpus?: number;
|
||||
dit_cpu_offload?: boolean;
|
||||
dit_layerwise_offload?: boolean;
|
||||
text_encoder_cpu_offload?: boolean;
|
||||
vae_cpu_offload?: boolean;
|
||||
image_encoder_cpu_offload?: boolean;
|
||||
use_fsdp_inference?: boolean;
|
||||
enable_torch_compile?: boolean;
|
||||
vsa_sparsity?: number;
|
||||
tp_size?: number;
|
||||
sp_size?: number;
|
||||
}
|
||||
|
||||
export interface JobFormFields {
|
||||
modelId: string;
|
||||
name: string;
|
||||
workloadType: string;
|
||||
jobType: string;
|
||||
prompt: string;
|
||||
negativePrompt: string;
|
||||
imagePath: string;
|
||||
lastImagePath: string;
|
||||
references: H3Reference[];
|
||||
promptFields: H3PromptFields | null;
|
||||
numInferenceSteps: number;
|
||||
numFrames: number;
|
||||
height: number;
|
||||
width: number;
|
||||
guidanceScale: number;
|
||||
guidanceRescale: number;
|
||||
fps: number;
|
||||
seed: number;
|
||||
numGpus: number;
|
||||
ditCpuOffload: boolean;
|
||||
ditLayerwiseOffload: boolean;
|
||||
textEncoderCpuOffload: boolean;
|
||||
vaeCpuOffload: boolean;
|
||||
imageEncoderCpuOffload: boolean;
|
||||
useFsdpInference: boolean;
|
||||
enableTorchCompile: boolean;
|
||||
vsaSparsity: number;
|
||||
tpSize: number;
|
||||
spSize: number;
|
||||
}
|
||||
|
||||
/** Uploads keep their original basename, so this is the display name. */
|
||||
export function referenceFileName(source: string): string {
|
||||
return source.split("/").filter(Boolean).pop() ?? source;
|
||||
}
|
||||
|
||||
export function jobToFormFields(job: JobLike): JobFormFields {
|
||||
const refs: H3Reference[] = (job.references ?? []).map((r, i) => ({
|
||||
id: `${job.id}-${i}`,
|
||||
source: r.source,
|
||||
media_type: r.media_type as H3Reference["media_type"],
|
||||
fileName: referenceFileName(r.source),
|
||||
}));
|
||||
|
||||
return {
|
||||
modelId: job.model_id,
|
||||
name: job.name ?? "",
|
||||
workloadType: job.workload_type ?? "t2v",
|
||||
jobType: job.job_type ?? "inference",
|
||||
prompt: job.prompt ?? "",
|
||||
negativePrompt: job.negative_prompt ?? "",
|
||||
imagePath: job.image_path ?? "",
|
||||
lastImagePath: job.last_image_path ?? "",
|
||||
references: refs,
|
||||
// null when the prompt is not in six-section form; the caller then keeps
|
||||
// the raw editor rather than silently dropping content into fields.
|
||||
promptFields: parseH3Prompt(job.prompt ?? ""),
|
||||
numInferenceSteps: job.num_inference_steps ?? 50,
|
||||
numFrames: job.num_frames ?? 81,
|
||||
height: job.height ?? 480,
|
||||
width: job.width ?? 832,
|
||||
guidanceScale: job.guidance_scale ?? 5.0,
|
||||
guidanceRescale: job.guidance_rescale ?? 0.0,
|
||||
fps: job.fps ?? 24,
|
||||
seed: job.seed ?? 1024,
|
||||
numGpus: job.num_gpus ?? 1,
|
||||
ditCpuOffload: job.dit_cpu_offload ?? false,
|
||||
ditLayerwiseOffload: job.dit_layerwise_offload ?? false,
|
||||
textEncoderCpuOffload: job.text_encoder_cpu_offload ?? false,
|
||||
vaeCpuOffload: job.vae_cpu_offload ?? false,
|
||||
imageEncoderCpuOffload: job.image_encoder_cpu_offload ?? false,
|
||||
useFsdpInference: job.use_fsdp_inference ?? false,
|
||||
enableTorchCompile: job.enable_torch_compile ?? false,
|
||||
vsaSparsity: job.vsa_sparsity ?? 0,
|
||||
tpSize: job.tp_size ?? -1,
|
||||
spSize: job.sp_size ?? -1,
|
||||
};
|
||||
}
|
||||
@@ -5,6 +5,7 @@ export type JobType = "inference" | "finetuning" | "distillation";
|
||||
export interface Job {
|
||||
id: string;
|
||||
model_id: string;
|
||||
name?: string;
|
||||
prompt: string;
|
||||
job_type?: JobType;
|
||||
workload_type?: string;
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Duplicating a job's config, and editing one that has not started."""
|
||||
import uuid
|
||||
|
||||
import pytest
|
||||
|
||||
from fastvideo_studio.database import Database
|
||||
from fastvideo_studio.job_runner import JobRunner, JobStatus
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def runner(tmp_path):
|
||||
return JobRunner(
|
||||
output_dir=str(tmp_path / "out"),
|
||||
log_dir=str(tmp_path / "logs"),
|
||||
database=Database(tmp_path / "t.db"),
|
||||
)
|
||||
|
||||
|
||||
def _make(runner, **over):
|
||||
kwargs = dict(
|
||||
job_id=str(uuid.uuid4()),
|
||||
model_id="MiniMaxAI/MiniMax-H3",
|
||||
prompt="p",
|
||||
workload_type="i2v",
|
||||
num_frames=141,
|
||||
guidance_scale=1.0,
|
||||
num_gpus=4,
|
||||
references=[{"source": "/x/clip.mp4", "media_type": "video"}],
|
||||
)
|
||||
kwargs.update(over)
|
||||
return runner.create_job(**kwargs)
|
||||
|
||||
|
||||
def test_duplicate_copies_config_but_not_runtime_state(runner):
|
||||
src = _make(runner)
|
||||
src.status = JobStatus.COMPLETED
|
||||
src.output_path = "/x/out.mp4"
|
||||
|
||||
dup = runner.duplicate_job(src.id, str(uuid.uuid4()))
|
||||
|
||||
assert dup.id != src.id
|
||||
assert dup.status is JobStatus.PENDING
|
||||
assert dup.output_path is None
|
||||
for field in ("model_id", "prompt", "workload_type", "num_frames",
|
||||
"guidance_scale", "num_gpus", "references"):
|
||||
assert getattr(dup, field) == getattr(src, field)
|
||||
|
||||
|
||||
def test_duplicate_deep_copies_references(runner):
|
||||
src = _make(runner)
|
||||
dup = runner.duplicate_job(src.id, str(uuid.uuid4()))
|
||||
dup.references[0]["source"] = "/changed"
|
||||
assert src.references[0]["source"] == "/x/clip.mp4"
|
||||
|
||||
|
||||
def test_duplicate_unknown_job(runner):
|
||||
with pytest.raises(ValueError, match="not found"):
|
||||
runner.duplicate_job("nope", str(uuid.uuid4()))
|
||||
|
||||
|
||||
def test_edit_pending_job(runner):
|
||||
job = _make(runner)
|
||||
updated = runner.update_job_config(job.id, {"num_frames": 192, "seed": 7})
|
||||
assert updated.num_frames == 192
|
||||
assert updated.seed == 7
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", [JobStatus.FAILED, JobStatus.STOPPED])
|
||||
def test_edit_allows_restartable_jobs(runner, status):
|
||||
"""Editable exactly when startable: neither has produced an output."""
|
||||
job = _make(runner)
|
||||
job.status = status
|
||||
assert runner.update_job_config(job.id, {"seed": 7}).seed == 7
|
||||
|
||||
|
||||
@pytest.mark.parametrize("status", [JobStatus.COMPLETED, JobStatus.RUNNING])
|
||||
def test_edit_rejects_jobs_with_or_producing_a_result(runner, status):
|
||||
job = _make(runner)
|
||||
job.status = status
|
||||
with pytest.raises(ValueError, match="can be edited"):
|
||||
runner.update_job_config(job.id, {"seed": 7})
|
||||
|
||||
|
||||
def test_edit_rejects_unknown_field(runner):
|
||||
job = _make(runner)
|
||||
with pytest.raises(ValueError, match="Not editable"):
|
||||
runner.update_job_config(job.id, {"status": "completed"})
|
||||
|
||||
|
||||
def test_name_is_carried_by_duplicate(runner):
|
||||
src = _make(runner, name="wukong swap v2")
|
||||
dup = runner.duplicate_job(src.id, str(uuid.uuid4()))
|
||||
assert dup.name == "wukong swap v2"
|
||||
|
||||
|
||||
def test_name_is_editable(runner):
|
||||
job = _make(runner, name="a")
|
||||
assert runner.update_job_config(job.id, {"name": "b"}).name == "b"
|
||||
@@ -0,0 +1,36 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Uploaded files keep a readable basename under a unique directory."""
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from fastvideo_studio.server import _safe_upload_name
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("filename", "ext", "expected"),
|
||||
[
|
||||
("wukong_source.mp4", ".mp4", "wukong_source.mp4"),
|
||||
("MonkeyKing_0.jpg", ".jpg", "MonkeyKing_0.jpg"),
|
||||
("my clip (final).mp4", ".mp4", "my_clip_final.mp4"),
|
||||
("../../etc/passwd.png", ".png", "passwd.png"),
|
||||
("/abs/path/frame.png", ".png", "frame.png"),
|
||||
("émoji✨.png", ".png", "moji.png"),
|
||||
("", ".png", "upload.png"),
|
||||
(None, ".png", "upload.png"),
|
||||
("...", ".png", "upload.png"),
|
||||
],
|
||||
)
|
||||
def test_safe_upload_name(filename, ext, expected):
|
||||
assert _safe_upload_name(filename, ext) == expected
|
||||
|
||||
|
||||
def test_long_names_are_capped():
|
||||
out = _safe_upload_name("x" * 300 + ".png", ".png")
|
||||
assert out == "x" * 80 + ".png"
|
||||
|
||||
|
||||
def test_no_path_separators_survive():
|
||||
for bad in ("a/b.png", "a\\b.png", "../x.png"):
|
||||
assert "/" not in _safe_upload_name(bad, ".png")
|
||||
assert "\\" not in _safe_upload_name(bad, ".png")
|
||||
@@ -15,6 +15,9 @@ from fastvideo import VideoGenerator as FastVideoGenerator
|
||||
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))))
|
||||
|
||||
# InferenceArgs keys that are SamplingConfig fields of a GenerationRequest.
|
||||
_SAMPLING_INFERENCE_ARGS = ("height", "width", "num_frames", "num_inference_steps", "guidance_scale", "seed", "fps")
|
||||
|
||||
|
||||
# Custom exception for interruption
|
||||
class GenerationInterruptedException(Exception):
|
||||
@@ -154,7 +157,17 @@ class VideoGenerator:
|
||||
"""Thread function to run the generation"""
|
||||
try:
|
||||
if self.generator is not None:
|
||||
self.generator.generate_video(prompt=prompt, output_path=output_path, **inference_args)
|
||||
# Place each InferenceArgs value in the GenerationRequest section that owns it.
|
||||
request: dict[str, Any] = {"prompt": prompt, "output": {"output_path": output_path}}
|
||||
for key, value in inference_args.items():
|
||||
if key == "image_path":
|
||||
section = "inputs"
|
||||
elif key in _SAMPLING_INFERENCE_ARGS:
|
||||
section = "sampling"
|
||||
else:
|
||||
section = "extensions"
|
||||
request.setdefault(section, {})[key] = value
|
||||
self.generator.generate(request)
|
||||
self._generation_result = os.path.join(output_path, f"{prompt[:100]}.mp4")
|
||||
else:
|
||||
raise RuntimeError("Generator is not initialized")
|
||||
@@ -253,9 +266,24 @@ class VideoGenerator:
|
||||
if self.generator is None:
|
||||
print('generation_args', generation_args)
|
||||
print('pipeline_config', pipeline_config)
|
||||
self.generator = FastVideoGenerator.from_pretrained(model_path=model_path,
|
||||
**generation_args,
|
||||
pipeline_config=pipeline_config)
|
||||
# Place each generation argument at its GeneratorConfig engine path.
|
||||
engine_config: dict[str, Any] = {}
|
||||
if "num_gpus" in generation_args:
|
||||
engine_config["num_gpus"] = generation_args["num_gpus"]
|
||||
for parallelism_key in ("tp_size", "sp_size"):
|
||||
if parallelism_key in generation_args:
|
||||
engine_config.setdefault("parallelism", {})[parallelism_key] = generation_args[parallelism_key]
|
||||
if "dit_cpu_offload" in generation_args:
|
||||
engine_config["offload"] = {"dit": generation_args["dit_cpu_offload"]}
|
||||
self.generator = FastVideoGenerator.from_config({
|
||||
"model_path": model_path,
|
||||
"engine": engine_config,
|
||||
"pipeline": {
|
||||
"experimental": {
|
||||
"pipeline_config": pipeline_config
|
||||
}
|
||||
},
|
||||
})
|
||||
|
||||
print('inference_args', inference_args)
|
||||
|
||||
|
||||
@@ -1,52 +1,842 @@
|
||||
{
|
||||
"version": 11,
|
||||
"recipes": [
|
||||
{
|
||||
"id": "fastwan21-t2v",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "FastWan2.1 1.3B (distilled + VSA)",
|
||||
"summary": "Generate a video in three denoising steps with the distilled FastWan2.1 1.3B checkpoint and video sparse attention.",
|
||||
"model": "FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
"source": "scripts/inference/inference_wan_VSA_DMD_1_3B.yaml",
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN fastvideo generate --config scripts/inference/inference_wan_VSA_DMD_1_3B.yaml"
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN fastvideo generate --config scripts/inference/inference_wan_VSA_DMD_1_3B.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video_dmd_1.3B/"
|
||||
},
|
||||
{
|
||||
"id": "wan22-t2v",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "Wan2.2 A14B",
|
||||
"summary": "The maintained high-capacity Wan2.2 text-to-video example with CPU offload settings encoded in its checked-in Python source.",
|
||||
"model": "Wan-AI/Wan2.2-T2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_wan2_2.py",
|
||||
"command": "python examples/inference/basic/basic_wan2_2.py"
|
||||
"command": "python examples/inference/basic/basic_wan2_2.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_wan2_2_14B_t2v/"
|
||||
},
|
||||
{
|
||||
"id": "wan21-i2v",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "Wan2.1 14B 480P",
|
||||
"summary": "Animate an input image at 480P using the maintained Wan2.1 YAML configuration and its recorded offload settings.",
|
||||
"model": "Wan-AI/Wan2.1-I2V-14B-480P-Diffusers",
|
||||
"source": "scripts/inference/inference_wan_i2v.yaml",
|
||||
"command": "fastvideo generate --config scripts/inference/inference_wan_i2v.yaml"
|
||||
},
|
||||
{
|
||||
"id": "turbowan22-i2v",
|
||||
"task": "Image to video",
|
||||
"label": "TurboWan2.2 A14B",
|
||||
"model": "loayrashid/TurboWan2.2-I2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_i2v.py"
|
||||
"command": "fastvideo generate --config scripts/inference/inference_wan_i2v.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed"
|
||||
},
|
||||
{
|
||||
"id": "wan22-ti2v",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Text or image to video",
|
||||
"label": "Wan2.2 TI2V 5B",
|
||||
"summary": "Use one maintained 5B checkpoint for text-to-video or add an image input to switch the same recipe to image-to-video.",
|
||||
"model": "Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_wan2_2_ti2v.py",
|
||||
"command": "python examples/inference/basic/basic_wan2_2_ti2v.py"
|
||||
"command": "python examples/inference/basic/basic_wan2_2_ti2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_wan2_2_5B_ti2v/"
|
||||
},
|
||||
{
|
||||
"id": "fastmetal-1-3b-mlx",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "FastMetal 1.3B",
|
||||
"summary": "Run the released FastMetal 1.3B QAD checkpoint through FastVideo's native Apple Silicon MLX path.",
|
||||
"model": "FastVideo/FastMetal-1.3B-QAD",
|
||||
"source": "examples/inference/basic/mlx_wan_prompt_to_video.py",
|
||||
"command": "hf download FastVideo/FastMetal-1.3B-QAD --local-dir ./FastMetal-1.3B-QAD\npython examples/inference/basic/mlx_wan_prompt_to_video.py --model-root ./FastMetal-1.3B-QAD --mlx-checkpoint ./FastMetal-1.3B-QAD --height 480 --width 832 --num-frames 81 --prompt \"A bird's-eye view of a misty forest valley at dawn.\" --output-path ./outputs/fastmetal_1_3b.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"minimum_memory": "16 GB+ unified memory",
|
||||
"peak_memory": "3.87 GiB peak MLX memory",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1638"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 video at outputs/fastmetal_1_3b.mp4",
|
||||
"modes": ["T2V", "temporal --fast", "spatial --fast-spatial", "two-pass --refine"],
|
||||
"limitations": [
|
||||
"Native MLX FastMetal T2V. Add --fast for temporal RIFE, --fast-spatial to denoise at half resolution, or --refine for a two-pass upsample. basic_mps.py is the older PyTorch MPS demo."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fastmetal-5b-mlx",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "FastMetal 5B",
|
||||
"summary": "Run the released Wan2.2 5B FastMetal checkpoint as MLX T2V with MLX DiT denoising and MLX TAEHV decode. The CUDA Wan2.2 TI2V 5B recipe is the image-capable path.",
|
||||
"model": "FastVideo/FastMetal-5B-QAD",
|
||||
"source": "examples/inference/basic/mlx_wan22_generate.py",
|
||||
"command": "hf download FastVideo/FastMetal-5B-QAD --local-dir ./FastMetal-5B-QAD\npython examples/inference/basic/mlx_wan22_generate.py --mlx-checkpoint ./FastMetal-5B-QAD --text-encoder-root ./FastMetal-5B-QAD --vae-root ./FastMetal-5B-QAD/vae --height 704 --width 1280 --num-frames 81 --prompt \"A cinematic portrait with soft neon lighting and smooth camera motion.\" --output-path ./outputs/fastmetal_5b.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"minimum_memory": "16 GB+ unified memory",
|
||||
"peak_memory": "9.34 GiB peak MLX memory",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1638"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 video at outputs/fastmetal_5b.mp4",
|
||||
"modes": ["T2V", "temporal --fast", "spatial --fast-spatial", "two-pass --refine"],
|
||||
"limitations": [
|
||||
"The checked-in MLX example is T2V. Image-to-video is not in mlx_wan22_generate.py. Add --fast, --fast-spatial, or --refine on the same script."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fastmetal-14b-mlx",
|
||||
"family": "wan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "FastMetal 14B",
|
||||
"summary": "Run the released 14B FastMetal QAD checkpoint through the same Apple Silicon MLX entrypoint as the 1.3B release.",
|
||||
"model": "FastVideo/FastMetal-14B-QAD",
|
||||
"source": "examples/inference/basic/mlx_wan_prompt_to_video.py",
|
||||
"command": "hf download FastVideo/FastMetal-14B-QAD --local-dir ./FastMetal-14B-QAD\npython examples/inference/basic/mlx_wan_prompt_to_video.py --model-root ./FastMetal-14B-QAD --mlx-checkpoint ./FastMetal-14B-QAD --height 480 --width 832 --num-frames 81 --prompt \"A wide cinematic landscape at sunrise.\" --output-path ./outputs/fastmetal_14b.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"minimum_memory": "36 GB+ unified memory",
|
||||
"peak_memory": "21.68 GiB peak MLX memory",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1638"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 video at outputs/fastmetal_14b.mp4",
|
||||
"modes": ["T2V", "temporal --fast", "spatial --fast-spatial", "two-pass --refine"],
|
||||
"limitations": [
|
||||
"Native MLX FastMetal T2V on 36 GB+ unified memory. Same --fast, --fast-spatial, and --refine flags as the 1.3B script."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "turbodiffusion-wan21-1-3b-t2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "TurboWan2.1 1.3B",
|
||||
"summary": "A TurboDiffusion-accelerated Wan2.1 1.3B text-to-video run from its maintained single-GPU example.",
|
||||
"model": "loayrashid/TurboWan2.1-T2V-1.3B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_turbodiffusion/",
|
||||
"related": ["turbodiffusion-wan21-14b-t2v", "turbowan22-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "turbodiffusion-wan21-14b-t2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "TurboWan2.1 14B",
|
||||
"summary": "TurboDiffusion acceleration applied to the 14B Wan2.1 text-to-video checkpoint; the checked-in source is configured for two GPUs.",
|
||||
"model": "loayrashid/TurboWan2.1-T2V-14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_14b.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_14b.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_turbodiffusion_14B/",
|
||||
"related": ["turbodiffusion-wan21-1-3b-t2v", "turbowan22-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "turbowan22-i2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "TurboWan2.2 A14B",
|
||||
"summary": "A one-to-four-step image-to-video path using TurboDiffusion and the SLA attention backend from its maintained example.",
|
||||
"model": "loayrashid/TurboWan2.2-I2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"related": ["turbodiffusion-wan21-14b-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "ltx2-distilled-t2v",
|
||||
"family": "ltx2",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "LTX-2 distilled",
|
||||
"summary": "The distilled LTX-2 text-to-video checkpoint with audio, from its maintained example. The source is configured for four GPUs.",
|
||||
"model": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"source": "examples/inference/basic/basic_ltx2_distilled.py",
|
||||
"command": "python examples/inference/basic/basic_ltx2_distilled.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 4, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 (video with audio) at outputs_video/ltx2_basic/output_ltx2_distilled_t2v.mp4",
|
||||
"related": ["ltx23-base-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "ltx23-base-t2v",
|
||||
"family": "ltx2",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "LTX-2 base (1088p)",
|
||||
"summary": "Base LTX-2 text-to-video at 1088x1920 using FastVideo default sampling for LTX2 base. The example loads a community Diffusers mirror of the base checkpoint; registered aliases include Lightricks/LTX-2 and FastVideo/LTX2-Diffusers.",
|
||||
"model": "Davids048/LTX2-Base-Diffusers",
|
||||
"source": "examples/inference/basic/basic_ltx2.py",
|
||||
"command": "python examples/inference/basic/basic_ltx2.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 (video with audio) at outputs_video/ltx2_basic/output_ltx2_base_t2v_1088_1920_1.1.mp4",
|
||||
"limitations": ["The maintained example loads the Davids048/LTX2-Base-Diffusers community mirror rather than a Lightricks upstream ID."],
|
||||
"related": ["ltx2-distilled-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "hy15-t2v-480p",
|
||||
"family": "hunyuan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "HunyuanVideo 1.5 480P",
|
||||
"summary": "HunyuanVideo 1.5 text-to-video at 480P with CPU offload enabled in the checked-in source for smaller GPUs.",
|
||||
"model": "hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v",
|
||||
"source": "examples/inference/basic/basic_hy15.py",
|
||||
"command": "python examples/inference/basic/basic_hy15.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_hy15/",
|
||||
"related": ["hy15-1080p-upscale"]
|
||||
},
|
||||
{
|
||||
"id": "hy15-1080p-upscale",
|
||||
"family": "hunyuan",
|
||||
"stage": "inference",
|
||||
"task": "Text to video (upscaled)",
|
||||
"label": "HunyuanVideo 1.5 1080P upscale",
|
||||
"summary": "Run HunyuanVideo 1.5 through the 480p to 720p to 1080p upscale chain in one maintained script.",
|
||||
"model": "weizhou03/HunyuanVideo-1.5-Diffusers-1080p-2SR",
|
||||
"source": "examples/inference/basic/basic_hy15_1080p.py",
|
||||
"command": "python examples/inference/basic/basic_hy15_1080p.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_hy15_1080p/",
|
||||
"related": ["hy15-t2v-480p"]
|
||||
},
|
||||
{
|
||||
"id": "cosmos25-t2w",
|
||||
"family": "cosmos",
|
||||
"stage": "inference",
|
||||
"task": "Text to world",
|
||||
"label": "Cosmos Predict 2.5 2B",
|
||||
"summary": "Generate a navigable world video from a text prompt with Cosmos Predict 2.5 2B on a single GPU.",
|
||||
"model": "KyleShao/Cosmos-Predict2.5-2B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_cosmos2_5_t2w.py",
|
||||
"command": "python examples/inference/basic/basic_cosmos2_5_t2w.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed"
|
||||
},
|
||||
{
|
||||
"id": "kandinsky5-t2v-lite-sft",
|
||||
"family": "kandinsky5",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "Kandinsky 5.0 T2V Lite SFT",
|
||||
"summary": "Kandinsky 5.0 text-to-video (Lite SFT variant) from the maintained example; alternative Lite/Pro checkpoints are listed in the source.",
|
||||
"model": "kandinskylab/Kandinsky-5.0-T2V-Lite-sft-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky5_t2v.py",
|
||||
"command": "python examples/inference/basic/basic_kandinsky5_t2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"accelerator": "NVIDIA B200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1471"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 videos under video_samples_kandinsky5_t2v/",
|
||||
"related": ["kandinsky5-i2v-pro-distilled"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky5-i2v-pro-distilled",
|
||||
"family": "kandinsky5",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "Kandinsky 5.0 I2V Pro distilled",
|
||||
"summary": "Animate an input image with Kandinsky 5.0 I2V Pro (distilled) on a single GPU.",
|
||||
"model": "kandinskylab/Kandinsky-5.0-I2V-Pro-distilled-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky5_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_kandinsky5_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"accelerator": "NVIDIA B200",
|
||||
"peak_memory": "10,365.89 MB peak GPU memory",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1471"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 videos under video_samples_kandinsky5_i2v/",
|
||||
"related": ["kandinsky5-t2v-lite-sft"]
|
||||
},
|
||||
{
|
||||
"id": "flux2-klein-t2i",
|
||||
"family": "flux",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "FLUX.2 Klein 4B",
|
||||
"summary": "Generate an image in four denoising steps with the distilled FLUX.2 Klein checkpoint.",
|
||||
"model": "black-forest-labs/FLUX.2-klein-4B",
|
||||
"source": "examples/inference/basic/basic_flux2_klein.py",
|
||||
"command": "python examples/inference/basic/basic_flux2_klein.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at outputs/flux2/flux2_klein.png",
|
||||
"related": ["flux2-dev-t2i"]
|
||||
},
|
||||
{
|
||||
"id": "flux2-dev-t2i",
|
||||
"family": "flux",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "FLUX.2 dev",
|
||||
"summary": "Full FLUX.2 dev text-to-image with embedded guidance and the Mistral3 text encoder, from its maintained example.",
|
||||
"model": "black-forest-labs/FLUX.2-dev",
|
||||
"source": "examples/inference/basic/basic_flux2.py",
|
||||
"command": "python examples/inference/basic/basic_flux2.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at outputs/flux2/flux2.png",
|
||||
"related": ["flux2-klein-t2i"]
|
||||
},
|
||||
{
|
||||
"id": "flux1-dev-t2i",
|
||||
"family": "flux",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "FLUX.1 dev",
|
||||
"summary": "FLUX.1 dev text-to-image through the Diffusers-backed pipeline. The example defaults to a local weights directory, so this recipe passes the Hugging Face ID explicitly.",
|
||||
"model": "black-forest-labs/FLUX.1-dev",
|
||||
"source": "examples/inference/basic/basic_flux_dev.py",
|
||||
"command": "python examples/inference/basic/basic_flux_dev.py --model-path black-forest-labs/FLUX.1-dev",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG images under outputs/flux_dev/samples/",
|
||||
"limitations": ["FLUX.1 is loadable by ID but registers no model_family in fastvideo/registry.py; it is grouped under FLUX for documentation only."]
|
||||
},
|
||||
{
|
||||
"id": "glm-image-t2i",
|
||||
"family": "glm_image",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "GLM-Image",
|
||||
"summary": "GLM-Image text-to-image generation from its maintained example.",
|
||||
"model": "zai-org/GLM-Image",
|
||||
"source": "examples/inference/basic/basic_glm_image.py",
|
||||
"command": "python examples/inference/basic/basic_glm_image.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at image_output/landscape.png",
|
||||
"related": ["glm-image-edit"]
|
||||
},
|
||||
{
|
||||
"id": "glm-image-edit",
|
||||
"family": "glm_image",
|
||||
"stage": "inference",
|
||||
"task": "Image editing",
|
||||
"label": "GLM-Image editing",
|
||||
"summary": "Edit an input image with an instruction prompt using GLM-Image, from its maintained editing example.",
|
||||
"model": "zai-org/GLM-Image",
|
||||
"source": "examples/inference/basic/edit_glm_image.py",
|
||||
"command": "python examples/inference/basic/edit_glm_image.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at image_output/edited.png (input: assets/images/couple.jpg)",
|
||||
"related": ["glm-image-t2i"]
|
||||
},
|
||||
{
|
||||
"id": "zimage-turbo-t2i",
|
||||
"family": "zimage",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "Z-Image Turbo",
|
||||
"summary": "Z-Image Turbo text-to-image on a single GPU from its maintained example.",
|
||||
"model": "Tongyi-MAI/Z-Image-Turbo",
|
||||
"source": "examples/inference/basic/basic_zimage.py",
|
||||
"command": "python examples/inference/basic/basic_zimage.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at outputs/zimage/zimage_turbo.png"
|
||||
},
|
||||
{
|
||||
"id": "sd35-medium-t2i",
|
||||
"family": "sd35",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "Stable Diffusion 3.5 Medium",
|
||||
"summary": "Stable Diffusion 3.5 Medium text-to-image over a small built-in prompt set, from its maintained example.",
|
||||
"model": "stabilityai/stable-diffusion-3.5-medium",
|
||||
"source": "examples/inference/basic/basic_sd35_t2i.py",
|
||||
"command": "python examples/inference/basic/basic_sd35_t2i.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG images under outputs/sd35/samples/"
|
||||
},
|
||||
{
|
||||
"id": "minimax-h3-t2v",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Text to video (with audio)",
|
||||
"label": "MiniMax H3 T2VA",
|
||||
"summary": "Generate synchronized video and stereo audio from a structured text prompt with the full MiniMax H3 checkpoint.",
|
||||
"model": "MiniMaxAI/MiniMax-H3",
|
||||
"source": "examples/inference/basic/basic_minimax_h3_t2v.py",
|
||||
"command": "python examples/inference/basic/basic_minimax_h3_t2v.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs MiniMax H3.</d>\"",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 4, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_t2v/minimax_h3_t2v.mp4",
|
||||
"modes": ["T2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["The checked-in example defaults to four-way sequence parallelism. It does not record a GPU model or memory requirement."]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-cuda",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on CUDA",
|
||||
"summary": "Run FastH3 V1 with four DiT forwards, trained H3 sparse attention, compiled decode, and synchronized audio.",
|
||||
"model": "FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2",
|
||||
"source": "examples/inference/basic/basic_fasth3.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile all",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 4,
|
||||
"accelerator": "NVIDIA GB200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1731"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3/",
|
||||
"modes": ["T2VA", "4-step FastH3"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": ["The default all profile is the measured GB200 performance route and can change floating-point operation order. Use --profile strict --no-inference-torch-compile for the eager strict route."]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-mlx",
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\""
|
||||
},
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on MLX",
|
||||
"summary": "Run FastH3 V1 on Apple Silicon with a locally converted INT6 DiT, streamed Qwen3-VL conditioning, and native MLX video and audio VAEs.",
|
||||
"model": "FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2",
|
||||
"source": "examples/inference/basic/mlx_fasth3.py",
|
||||
"command": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\"\npython examples/inference/basic/mlx_fasth3.py --model-root ./FastH3-Preview-v0.2 --mlx-checkpoint ./FastH3-MLX/int6 --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --height 480 --width 832 --num-frames 124 --seed 2026 --output-path ./outputs/fasth3_int6.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"peak_memory": "19.63 GiB peak MLX memory during denoising",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1770"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 with H.264 video and stereo AAC audio at outputs/fasth3_int6.mp4",
|
||||
"modes": ["T2VA", "temporal --fast", "spatial --fast-spatial", "opt-in VSA"],
|
||||
"knobs": [
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"The MLX path supports T2VA, optional temporal --fast, optional spatial --fast-spatial, and opt-in VSA on --include-vsa checkpoints. FL2VA, Ref2VA, and two-pass refinement are not wired."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-preview-spark",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on one DGX Spark",
|
||||
"summary": "Run FastH3 V1 on one GB10 with Triton VSA, FA4 off, and lazy module load. Height, width, frames, and steps in the YAML are examples.",
|
||||
"model": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree",
|
||||
"source": "examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_spark.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3"
|
||||
},
|
||||
"command": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"device": "spark",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/fasth3_spark/",
|
||||
"modes": ["T2VA", "1-Spark"],
|
||||
"limitations": [
|
||||
"Install from the DGX Spark guide, not the generic CUDA extra. GB10 has no FA4 / sm_100a VSA kernel; keep FASTVIDEO_FA4=0 and FASTVIDEO_VSA_SM100A=0.",
|
||||
"Legal num_frames values are 17n+5, capped at 345 (15 s). Native 16:9 sizes include 832x480 and 1344x768.",
|
||||
"Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Do not set engine.offload.lazy_module_load to false on this box.",
|
||||
"A 345-frame request on one Spark can OOM. Prefer 124 or 243 frames, TAEH3 decode, or two Sparks over QSFP."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-spark-pair",
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
"group_task": "4-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V1 on two DGX Sparks",
|
||||
"summary": "Run one FastH3 clip across two GB10s with Ray sequence parallel over QSFP RoCE. Sequential load and lazy module load stay on because SP replicates the DiT on each node.",
|
||||
"model": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree",
|
||||
"source": "examples/inference/basic/basic_fasth3_spark_pair.yaml",
|
||||
"command": "source examples/inference/optimizations/spark_pair_env.sh && FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_VAE_PARALLEL_DECODE=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"device": "spark",
|
||||
"gpu_count": 2,
|
||||
"accelerator": "NVIDIA GB10 (DGX Spark pair)",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1803"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 under outputs/fasth3_spark_pair/",
|
||||
"modes": ["T2VA", "2-Spark SP"],
|
||||
"limitations": [
|
||||
"Requires a two-node Ray cluster on the QSFP interconnect. There is no cookbook server for this path; use Python / generate.",
|
||||
"Height, width, frames, and steps in the YAML are examples. Edit them or pass CLI flags. See docs/getting_started/installation/spark_pair.md."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-cuda",
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
"group_task": "8-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V2 on CUDA",
|
||||
"summary": "Run FastH3 V2, the eight-forward checkpoint (video/audio shifts 10/3, VSA 0.8, 64-token tiles) with the trained DMD ladder loaded from the checkpoint's fastvideo_inference.json.",
|
||||
"model": "FastVideo/FastVideo-FastH3-8-Step-V2",
|
||||
"source": "examples/inference/basic/basic_fasth3_8step.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_8step.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3_8step.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile strict --no-inference-torch-compile --no-compile-vae",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 4,
|
||||
"accelerator": "NVIDIA GB200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1852"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3_8step/",
|
||||
"modes": ["T2VA", "8-step FastH3"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"Nine sigma-grid points (eight transformer forwards) are fixed by the checkpoint's trained ladder; the example rejects any other --steps.",
|
||||
"Validated with the eager strict route (--profile strict --no-inference-torch-compile --no-compile-vae). The compiled all profile has not been measured for this checkpoint.",
|
||||
"T2AV only; no FL2VA/Ref2VA distillation and no matching LoRA. V2 uses eight forwards rather than V1's four."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-mlx",
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3_8step.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa"
|
||||
},
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
"group_task": "8-step text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "FastH3 V2 on MLX",
|
||||
"summary": "Run FastH3 V2 on Apple Silicon with a locally converted INT8 DiT, the checkpoint's DMD contract ladder, and trained VSA (sparsity 0.8, 64-token tiles).",
|
||||
"model": "FastVideo/FastVideo-FastH3-8-Step-V2",
|
||||
"source": "examples/inference/basic/mlx_fasth3_8step.py",
|
||||
"command": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa\npython examples/inference/basic/mlx_fasth3_8step.py --model-root ./FastH3-8-Step-V2 --mlx-checkpoint ./FastH3-8-Step-V2-MLX/int8 --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --height 480 --width 832 --num-frames 124 --seed 2026 --output-path ./outputs/fasth3_8step_int8.mp4",
|
||||
"gpu_types": ["Apple Silicon"],
|
||||
"hardware": {
|
||||
"platform": "mlx",
|
||||
"accelerator": "Apple M4 Max",
|
||||
"system_memory": "36 GB unified memory",
|
||||
"peak_memory": "27.39 GiB peak MLX memory during denoising",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1863"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "MP4 with H.264 video and stereo AAC audio at outputs/fasth3_8step_int8.mp4",
|
||||
"modes": ["T2VA", "8-step FastH3", "trained VSA"],
|
||||
"knobs": [
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": [
|
||||
"Convert with --include-vsa. mlx_fasth3_8step.py turns VSA on (sparsity 0.8, tile 64). A dense export fails at configure_vsa.",
|
||||
"--steps 8 or 9 both run the eight trained forwards. Reuse the preview VAE, audio VAE, text encoder, and tokenizer if those directories already exist.",
|
||||
"The MLX path supports T2VA only. FL2VA, Ref2VA, and two-pass refinement are not wired."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "minimax-h3-fl2va",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "First/last frame to video (with audio)",
|
||||
"label": "MiniMax H3 FL2VA",
|
||||
"summary": "Animate a first frame, optionally guide the final frame, and generate synchronized audio with the full MiniMax H3 checkpoint.",
|
||||
"model": "MiniMaxAI/MiniMax-H3",
|
||||
"source": "examples/inference/basic/basic_minimax_h3_fl2va.py",
|
||||
"command": "python examples/inference/basic/basic_minimax_h3_fl2va.py --image path/to/first-frame.png --prompt \"(S1) The subject turns toward the camera and says <d>[English] Hello.</d>\"",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 4, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_fl2va/minimax_h3_fl2va.mp4",
|
||||
"modes": ["FL2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["Pass --last-image to constrain the final frame. The checked-in source defaults to four GPUs."]
|
||||
},
|
||||
{
|
||||
"id": "minimax-h3-ref2va",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Reference media to video (with audio)",
|
||||
"label": "MiniMax H3 Ref2VA",
|
||||
"summary": "Condition H3 on an ordered reference video and optional audio reference, then generate a new synchronized video and audio result.",
|
||||
"model": "MiniMaxAI/MiniMax-H3",
|
||||
"source": "examples/inference/basic/basic_minimax_h3_ref2va.py",
|
||||
"command": "python examples/inference/basic/basic_minimax_h3_ref2va.py --reference-video path/to/reference.mp4 --prompt \"Create a new scene that preserves the reference identity and motion language.\"",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 4, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 with synchronized audio at outputs/minimax_h3_ref2va/minimax_h3_ref2va.mp4",
|
||||
"modes": ["Ref2VA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4}
|
||||
],
|
||||
"limitations": ["Pass --reference-audio for an additional audio reference. The checked-in source defaults to four GPUs."]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-lora-preview",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "LoRA-adapted few-step video (with audio)",
|
||||
"label": "FastH3 LoRA Preview",
|
||||
"summary": "Apply a FastH3 preview adapter at load time while keeping the shared four-forward performance profile and synchronized audio output.",
|
||||
"model": "MiniMaxAI/MiniMax-H3",
|
||||
"source": "examples/inference/basic/basic_fasth3_lora_preview.py",
|
||||
"command": "python examples/inference/basic/basic_fasth3_lora_preview.py --lora-path path/to/adapter.safetensors --prompt \"(S1) A presenter says <d>[English] This is an adapted Fast H3 run.</d>\"",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 4,
|
||||
"accelerator": "NVIDIA B200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1771"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "Warmup and measured MP4 files under outputs/fasth3_lora_preview/",
|
||||
"modes": ["T2VA", "FastH3 LoRA"],
|
||||
"knobs": [
|
||||
{"key": "num_gpus", "label": "GPUs", "hint": "Sequence-parallel degree", "flag": "--num-gpus", "options": [1, 2, 4, 8], "default": 4},
|
||||
{"key": "video_decode_backend", "label": "VAE decode", "hint": "Fidelity vs. speed", "flag": "--video-decode-backend", "options": [{"value": "h3-vae", "label": "Full H3 VAE"}, {"value": "taeh3", "label": "TAEH3 preview"}], "default": "h3-vae"}
|
||||
],
|
||||
"limitations": ["Supply a compatible FastH3 adapter. The script infers dense or VSA attention from the adapter payload unless you override it."]
|
||||
},
|
||||
{
|
||||
"id": "longcat-t2v",
|
||||
"family": "longcat",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "LongCat Video T2V",
|
||||
"summary": "LongCat Video text-to-video at 480p (50 steps), with distilled and 720p refinement passes included in the same maintained script.",
|
||||
"model": "FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
"source": "examples/inference/basic/basic_longcat_t2v.py",
|
||||
"command": "python examples/inference/basic/basic_longcat_t2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video/longcat_t2v_basic/, longcat_t2v_distill/, and longcat_t2v_refine_720p/",
|
||||
"related": ["longcat-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "longcat-i2v",
|
||||
"family": "longcat",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "LongCat Video I2V",
|
||||
"summary": "LongCat Video image-to-video with optional distilled and refinement passes, from its maintained example.",
|
||||
"model": "FastVideo/LongCat-Video-I2V-Diffusers",
|
||||
"source": "examples/inference/basic/basic_longcat_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_longcat_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video/longcat_i2v_basic/ and longcat_i2v_distill/",
|
||||
"related": ["longcat-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "stable-audio-open-t2a",
|
||||
"family": "stable_audio",
|
||||
"stage": "inference",
|
||||
"task": "Text to audio",
|
||||
"label": "Stable Audio Open 1.0",
|
||||
"summary": "Six-second text-to-audio generation with Stable Audio Open 1.0 from its maintained example; duration and steps are documented knobs in the source.",
|
||||
"model": "FastVideo/stable-audio-open-1.0-Diffusers",
|
||||
"source": "examples/inference/basic/basic_stable_audio.py",
|
||||
"command": "python examples/inference/basic/basic_stable_audio.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"accelerator": "NVIDIA B200",
|
||||
"evidence": "validated",
|
||||
"evidence_url": "https://github.com/hao-ai-lab/FastVideo/pull/1260"
|
||||
},
|
||||
"evidence": "Verified",
|
||||
"expected_artifact": "WAV audio at outputs_audio/stable_audio_basic/output_stable_audio.wav",
|
||||
"limitations": ["Must load the FastVideo converted Diffusers repo; upstream stabilityai monolithic checkpoints are not loader-compatible (see scripts/checkpoint_conversion/stable_audio_to_diffusers.py)."],
|
||||
"related": ["stable-audio-small-t2a"]
|
||||
},
|
||||
{
|
||||
"id": "stable-audio-small-t2a",
|
||||
"family": "stable_audio",
|
||||
"stage": "inference",
|
||||
"task": "Text to audio",
|
||||
"label": "Stable Audio Open Small",
|
||||
"summary": "The smaller Stable Audio Open variant with its own shorter training window, from its maintained example.",
|
||||
"model": "FastVideo/stable-audio-open-small-Diffusers",
|
||||
"source": "examples/inference/basic/basic_stable_audio_small.py",
|
||||
"command": "python examples/inference/basic/basic_stable_audio_small.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"related": ["stable-audio-open-t2a"]
|
||||
},
|
||||
{
|
||||
"id": "mmaudio-v2a",
|
||||
"family": "mmaudio",
|
||||
"stage": "inference",
|
||||
"task": "Video/Text to audio",
|
||||
"label": "MMAudio large 44k v2",
|
||||
"summary": "Add synchronized audio to a video (or from a prompt) with MMAudio large 44k v2. The example reads the model path from MMAUDIO_MODEL_PATH; this recipe passes the converted Hugging Face repo explicitly.",
|
||||
"model": "FastVideo/MMAudio-large-44k-v2-Diffusers",
|
||||
"source": "examples/inference/basic/basic_mmaudio.py",
|
||||
"command": "MMAUDIO_MODEL_PATH=FastVideo/MMAudio-large-44k-v2-Diffusers python examples/inference/basic/basic_mmaudio.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"limitations": ["The upstream checkpoint must be converted to Diffusers layout via scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py unless loaded from the FastVideo converted repo as done here."]
|
||||
},
|
||||
{
|
||||
"id": "matrix-game-2",
|
||||
"family": "matrixgame",
|
||||
"stage": "inference",
|
||||
"task": "Interactive world",
|
||||
"label": "Matrix Game 2.0",
|
||||
"summary": "Generate an interactive-world sequence from the maintained Matrix Game 2.0 example.",
|
||||
"model": "FastVideo/Matrix-Game-2.0-Base-Distilled-Diffusers",
|
||||
"source": "examples/inference/basic/basic_matrixgame2.py",
|
||||
"command": "python examples/inference/basic/basic_matrixgame2.py"
|
||||
"command": "python examples/inference/basic/basic_matrixgame2.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"related": ["matrix-game-3-i2w"]
|
||||
},
|
||||
{
|
||||
"id": "matrix-game-3-i2w",
|
||||
"family": "matrixgame",
|
||||
"stage": "inference",
|
||||
"task": "Interactive world",
|
||||
"label": "Matrix Game 3.0",
|
||||
"summary": "Drive Matrix Game 3.0 from an input image plus prompt at 720p, three steps, from its maintained example.",
|
||||
"model": "FastVideo/Matrix-Game-3.0-Base-Distilled-Diffusers",
|
||||
"source": "examples/inference/basic/basic_matrixgame3.py",
|
||||
"command": "python examples/inference/basic/basic_matrixgame3.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_matrixgame3/",
|
||||
"related": ["matrix-game-2"]
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
@@ -1,57 +1,724 @@
|
||||
(() => {
|
||||
let recipesPromise;
|
||||
const PATTERN_CHARS = "ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789";
|
||||
const PATTERN_LENGTH = 900;
|
||||
const SCRAMBLE_MS = 140;
|
||||
|
||||
const loadRecipes = (url) => {
|
||||
recipesPromise ||= fetch(url).then((response) => {
|
||||
const loadRecipes = (url) =>
|
||||
fetch(url).then((response) => {
|
||||
if (!response.ok) throw new Error(`HTTP ${response.status}`);
|
||||
return response.json();
|
||||
});
|
||||
return recipesPromise;
|
||||
|
||||
const generatePattern = (length) => {
|
||||
const chars = new Array(length);
|
||||
for (let index = 0; index < length; index += 1) {
|
||||
chars[index] = PATTERN_CHARS.charAt((Math.random() * PATTERN_CHARS.length) | 0);
|
||||
}
|
||||
return chars.join("");
|
||||
};
|
||||
|
||||
const motionQuery = () =>
|
||||
window.matchMedia("(hover: hover) and (pointer: fine) and (prefers-reduced-motion: no-preference)");
|
||||
|
||||
const initEvervault = (root) => {
|
||||
root.querySelectorAll("[data-evervault]").forEach((visual) => {
|
||||
if (visual.dataset.evervaultReady) return;
|
||||
visual.dataset.evervaultReady = "true";
|
||||
const noise = visual.querySelector("[data-cookbook-pattern]");
|
||||
if (!noise) return;
|
||||
|
||||
const query = motionQuery();
|
||||
if (!query.matches) return;
|
||||
|
||||
let rect = null;
|
||||
let pointerX = 0;
|
||||
let pointerY = 0;
|
||||
let frame = 0;
|
||||
let lastScramble = 0;
|
||||
let patternReady = false;
|
||||
|
||||
const paint = (scramble) => {
|
||||
visual.style.setProperty("--mouse-x", `${pointerX}px`);
|
||||
visual.style.setProperty("--mouse-y", `${pointerY}px`);
|
||||
if (scramble) noise.textContent = generatePattern(PATTERN_LENGTH);
|
||||
frame = 0;
|
||||
};
|
||||
|
||||
const onEnter = () => {
|
||||
rect = visual.getBoundingClientRect();
|
||||
if (!patternReady) {
|
||||
noise.textContent = generatePattern(PATTERN_LENGTH);
|
||||
patternReady = true;
|
||||
}
|
||||
};
|
||||
const onMove = (event) => {
|
||||
if (!rect) rect = visual.getBoundingClientRect();
|
||||
pointerX = event.clientX - rect.left;
|
||||
pointerY = event.clientY - rect.top;
|
||||
if (frame) return;
|
||||
frame = window.requestAnimationFrame(() => {
|
||||
const now = performance.now();
|
||||
const scramble = now - lastScramble >= SCRAMBLE_MS;
|
||||
if (scramble) lastScramble = now;
|
||||
paint(scramble);
|
||||
});
|
||||
};
|
||||
const onLeave = () => {
|
||||
if (frame) window.cancelAnimationFrame(frame);
|
||||
frame = 0;
|
||||
rect = null;
|
||||
pointerX = 0;
|
||||
pointerY = 0;
|
||||
visual.style.setProperty("--mouse-x", "50%");
|
||||
visual.style.setProperty("--mouse-y", "50%");
|
||||
};
|
||||
visual.addEventListener("mouseenter", onEnter);
|
||||
visual.addEventListener("mousemove", onMove);
|
||||
visual.addEventListener("mouseleave", onLeave);
|
||||
});
|
||||
};
|
||||
|
||||
let familyPopstate = null;
|
||||
const bindFamilyPopstate = () => {
|
||||
if (bindFamilyPopstate.bound) return;
|
||||
bindFamilyPopstate.bound = true;
|
||||
window.addEventListener("popstate", () => {
|
||||
if (typeof familyPopstate === "function") familyPopstate();
|
||||
});
|
||||
};
|
||||
|
||||
const groupIdFor = (recipe) => recipe.group || recipe.id;
|
||||
|
||||
const gpuCountLabel = (hardware, runtimeId) => {
|
||||
const count = hardware?.gpu_count;
|
||||
if (count == null) return "";
|
||||
if (runtimeId === "spark") return `${count} Spark${count === 1 ? "" : "s"}`;
|
||||
return `${count} GPU${count === 1 ? "" : "s"}`;
|
||||
};
|
||||
|
||||
const knobsFor = (recipe) => recipe.knobs || [];
|
||||
|
||||
const knobOptions = (knob) =>
|
||||
knob.options.map((option) =>
|
||||
option !== null && typeof option === "object" ? option : { value: option, label: String(option) },
|
||||
);
|
||||
|
||||
const knobDefaultLabel = (knob) => {
|
||||
const match = knobOptions(knob).find((option) => option.value === knob.default);
|
||||
return match ? match.label : String(knob.default);
|
||||
};
|
||||
|
||||
// Knob flags are always shown explicitly in the displayed command, even at
|
||||
// their default value, so the command stays copy-pasteable and precise
|
||||
// about what it runs -- not just "trust the script's own default".
|
||||
const appendKnobFlags = (commandText, knobs, knobValues) => {
|
||||
const flags = knobs
|
||||
.filter((knob) => knobValues[knob.key] !== undefined)
|
||||
.map((knob) => `${knob.flag} ${knobValues[knob.key]}`);
|
||||
if (!flags.length) return commandText;
|
||||
const lines = commandText.split("\n");
|
||||
lines[lines.length - 1] = `${lines[lines.length - 1]} ${flags.join(" ")}`;
|
||||
return lines.join("\n");
|
||||
};
|
||||
|
||||
const runtimeFor = (recipe) => {
|
||||
const platform = recipe.hardware?.platform || "cuda";
|
||||
if (platform === "mlx") {
|
||||
return {
|
||||
id: "mlx",
|
||||
label: "Apple Silicon · MLX",
|
||||
hint:
|
||||
recipe.hardware?.minimum_memory ||
|
||||
[recipe.hardware?.accelerator, recipe.hardware?.system_memory].filter(Boolean).join(" · ") ||
|
||||
"Memory not recorded",
|
||||
};
|
||||
}
|
||||
if (platform === "mps") {
|
||||
return {
|
||||
id: "mps",
|
||||
label: "Apple Silicon · PyTorch MPS",
|
||||
hint: recipe.hardware?.minimum_memory || recipe.hardware?.system_memory || "Memory not recorded",
|
||||
};
|
||||
}
|
||||
const hardware = recipe.hardware || {};
|
||||
if (hardware.device === "spark") {
|
||||
return {
|
||||
id: "spark",
|
||||
label: "NVIDIA DGX Spark",
|
||||
hint: "GB10 · 128 GB unified memory",
|
||||
};
|
||||
}
|
||||
return {
|
||||
id: "cuda",
|
||||
label: "NVIDIA CUDA",
|
||||
hint: hardware.accelerator
|
||||
? `${hardware.accelerator} · ${gpuCountLabel(hardware, "cuda")}`
|
||||
: `${gpuCountLabel(hardware, "cuda")} configured · GPU model not recorded`,
|
||||
};
|
||||
};
|
||||
|
||||
const runtimeSummary = (recipe) => {
|
||||
const runtime = runtimeFor(recipe);
|
||||
const hardware = recipe.hardware || {};
|
||||
if (runtime.id === "mlx") {
|
||||
return [hardware.accelerator || "Apple Silicon", hardware.system_memory || hardware.minimum_memory, "MLX"]
|
||||
.filter(Boolean)
|
||||
.join(" · ");
|
||||
}
|
||||
if (runtime.id === "mps") {
|
||||
return [hardware.accelerator || "Apple Silicon", hardware.system_memory || hardware.minimum_memory, "PyTorch MPS"]
|
||||
.filter(Boolean)
|
||||
.join(" · ");
|
||||
}
|
||||
if (runtime.id === "spark") {
|
||||
return [hardware.accelerator || "NVIDIA GB10", gpuCountLabel(hardware, "spark")].filter(Boolean).join(" · ");
|
||||
}
|
||||
if (hardware.accelerator) return [hardware.accelerator, gpuCountLabel(hardware, runtime.id)].join(" · ");
|
||||
return `NVIDIA CUDA · ${gpuCountLabel(hardware, "cuda")} configured · GPU model and VRAM not recorded`;
|
||||
};
|
||||
|
||||
const renderHardwareEvidence = (container, badge, recipe) => {
|
||||
const hardware = recipe.hardware || {};
|
||||
const runtime = runtimeFor(recipe);
|
||||
const isValidated = hardware.evidence === "validated";
|
||||
container.classList.toggle("cookbook-hardware-state--verified", isValidated);
|
||||
container.classList.toggle("cookbook-hardware-state--source", !isValidated);
|
||||
badge.classList.toggle("cookbook-badge--verified", isValidated);
|
||||
badge.classList.toggle("cookbook-badge--configured", !isValidated);
|
||||
badge.textContent = isValidated ? "Recorded run" : "Source config";
|
||||
|
||||
const heading = document.createElement("strong");
|
||||
heading.textContent = isValidated ? "Recorded hardware" : "Source configuration";
|
||||
const details = document.createElement("span");
|
||||
|
||||
if (isValidated) {
|
||||
const recorded = [
|
||||
hardware.accelerator,
|
||||
runtime.id === "mlx" || runtime.id === "mps" ? hardware.system_memory : gpuCountLabel(hardware, runtime.id),
|
||||
].filter(Boolean);
|
||||
const statements = [`${recorded.join(" · ")}.`];
|
||||
if (hardware.minimum_memory) statements.push(`Documented minimum: ${hardware.minimum_memory}.`);
|
||||
if (hardware.peak_memory) statements.push(`Measured: ${hardware.peak_memory}.`);
|
||||
if (!hardware.minimum_memory) statements.push("This recorded device is not a minimum requirement.");
|
||||
details.textContent = ` ${statements.join(" ")}`;
|
||||
} else if (runtime.id === "spark") {
|
||||
details.textContent = ` NVIDIA DGX Spark · ${gpuCountLabel(hardware, "spark")}. GB10 has no FA4 / sm_100a VSA kernel; keep Triton VSA and FA4 off.`;
|
||||
} else if (runtime.id === "cuda") {
|
||||
details.textContent = ` NVIDIA CUDA · ${gpuCountLabel(hardware, "cuda")}. The source does not record the GPU model or VRAM.`;
|
||||
} else {
|
||||
details.textContent = ` ${runtime.label}. The source does not record a device or memory requirement.`;
|
||||
}
|
||||
|
||||
container.replaceChildren(heading, details);
|
||||
if (hardware.evidence_url) {
|
||||
const evidenceLink = document.createElement("a");
|
||||
evidenceLink.href = hardware.evidence_url;
|
||||
evidenceLink.textContent = "View run evidence";
|
||||
evidenceLink.setAttribute("aria-label", `View recorded hardware evidence for ${recipe.label}`);
|
||||
container.append(" ", evidenceLink);
|
||||
}
|
||||
};
|
||||
|
||||
const compactLifecycle = (root) => {
|
||||
const lifecycle = root.querySelector(".cookbook-lifecycle");
|
||||
if (!lifecycle || lifecycle.dataset.compact) return;
|
||||
lifecycle.dataset.compact = "true";
|
||||
const stages = [...lifecycle.querySelectorAll(".cookbook-lifecycle__stage")];
|
||||
const active = stages.find((stage) => stage.classList.contains("cookbook-lifecycle__stage--active"));
|
||||
const planned = stages
|
||||
.filter((stage) => stage !== active)
|
||||
.map((stage) => stage.childNodes[0]?.textContent?.trim())
|
||||
.filter(Boolean);
|
||||
const summary = document.createElement("span");
|
||||
summary.className = "cookbook-lifecycle__summary";
|
||||
summary.textContent = `Next: ${planned.join(", ")}`;
|
||||
lifecycle.replaceChildren(...(active ? [active] : []), summary);
|
||||
};
|
||||
|
||||
const initFamilyBuilder = async (root) => {
|
||||
const family = root.dataset.family;
|
||||
if (!family) return;
|
||||
|
||||
compactLifecycle(root);
|
||||
|
||||
const modelOptions = root.querySelector("[data-cookbook-model-options]");
|
||||
const hardwareOptions = root.querySelector("[data-cookbook-hardware-options]");
|
||||
const description = root.querySelector("[data-cookbook-description]");
|
||||
const label = root.querySelector("[data-cookbook-label]");
|
||||
const model = root.querySelector("[data-cookbook-model]");
|
||||
const task = root.querySelector("[data-cookbook-task]");
|
||||
const hardwareValue = root.querySelector("[data-cookbook-gpus]");
|
||||
const artifact = root.querySelector("[data-cookbook-artifact]");
|
||||
const evidenceCell = root.querySelector("[data-cookbook-evidence]");
|
||||
const source = root.querySelector("[data-cookbook-source]");
|
||||
const modelLink = root.querySelector("[data-cookbook-model-link]");
|
||||
const command = root.querySelector("[data-cookbook-command]");
|
||||
const status = root.querySelector("[data-cookbook-status]");
|
||||
const hardwareState = root.querySelector("[data-cookbook-hardware-state]");
|
||||
const hardwareBadge = root.querySelector("[data-cookbook-hardware-badge]");
|
||||
const count = root.querySelector("[data-cookbook-count]");
|
||||
const result = root.querySelector(".cookbook-result");
|
||||
const commandBlock = root.querySelector(".cookbook-command");
|
||||
const servingPanel = root.querySelector("[data-cookbook-serving]");
|
||||
const usage = root.querySelector("[data-cookbook-usage]");
|
||||
const servingAvailability = root.querySelector("[data-cookbook-serving-availability]");
|
||||
const knobsContainer = root.querySelector("[data-cookbook-knobs]");
|
||||
const deviceRow = root.querySelector("[data-cookbook-device-row]");
|
||||
const deviceOptions = root.querySelector("[data-cookbook-device-options]");
|
||||
const deviceCaption = root.querySelector("[data-cookbook-device-caption]");
|
||||
|
||||
modelOptions.setAttribute("aria-label", "Recipe");
|
||||
hardwareOptions.setAttribute("aria-label", "Runtime");
|
||||
|
||||
let recipes;
|
||||
try {
|
||||
({ recipes } = await loadRecipes(root.dataset.recipes));
|
||||
} catch (error) {
|
||||
if (status) status.textContent = "Recipes could not be loaded. Use the maintained examples link below.";
|
||||
console.error("Failed to load FastVideo cookbook recipes", error);
|
||||
return;
|
||||
}
|
||||
if (!root.isConnected) return;
|
||||
|
||||
const familyRecipes = recipes.filter((recipe) => recipe.family === family);
|
||||
if (!familyRecipes.length) return;
|
||||
|
||||
let servingProfiles = {};
|
||||
let servingLoadFailed = false;
|
||||
if (servingPanel && familyRecipes.some((recipe) => recipe.serving)) {
|
||||
try {
|
||||
const dataUrl = new URL(root.dataset.recipes, document.baseURI);
|
||||
servingProfiles = await loadRecipes(new URL("cookbook-serving.json", dataUrl));
|
||||
} catch (error) {
|
||||
servingLoadFailed = true;
|
||||
console.error("Failed to load FastVideo serving profiles", error);
|
||||
}
|
||||
}
|
||||
|
||||
const byId = new Map(familyRecipes.map((recipe) => [recipe.id, recipe]));
|
||||
const groups = new Map();
|
||||
familyRecipes.forEach((recipe) => {
|
||||
const groupId = groupIdFor(recipe);
|
||||
if (!groups.has(groupId)) groups.set(groupId, []);
|
||||
groups.get(groupId).push(recipe);
|
||||
});
|
||||
if (count) count.textContent = `${familyRecipes.length} maintained recipes`;
|
||||
|
||||
modelOptions.replaceChildren();
|
||||
groups.forEach((groupRecipes, groupId) => {
|
||||
const representative = groupRecipes[0];
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.recipeGroup = groupId;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = representative.group_label || representative.label;
|
||||
const optionTask = document.createElement("span");
|
||||
optionTask.textContent = representative.group_task || representative.task;
|
||||
option.append(optionLabel, optionTask);
|
||||
modelOptions.append(option);
|
||||
});
|
||||
|
||||
const query = new URLSearchParams(window.location.search);
|
||||
const requestedRecipe = query.get("recipe");
|
||||
const defaultRecipeId = byId.has(root.dataset.defaultRecipe) ? root.dataset.defaultRecipe : familyRecipes[0].id;
|
||||
let selectedRecipeId = requestedRecipe && byId.has(requestedRecipe) ? requestedRecipe : defaultRecipeId;
|
||||
let selectedGroupId = groupIdFor(byId.get(selectedRecipeId));
|
||||
let renderedRuntimeGroup = null;
|
||||
// Keep previously shared local/openai links working after renaming workflows.
|
||||
const workflow = (value) => ["local", "python"].includes(value) ? "python" : "server";
|
||||
let usagePreference = workflow(query.get("use"));
|
||||
let selectedClient = ["python", "javascript", "curl"].includes(query.get("client")) ? query.get("client") : "curl";
|
||||
const clientDetails = servingPanel?.querySelector(".cookbook-serving__code");
|
||||
if (clientDetails && query.has("client")) clientDetails.open = true;
|
||||
|
||||
const knobDefs = new Map();
|
||||
familyRecipes.forEach((recipe) => knobsFor(recipe).forEach((knob) => {
|
||||
if (!knobDefs.has(knob.key)) knobDefs.set(knob.key, knob);
|
||||
}));
|
||||
const knobValues = {};
|
||||
knobDefs.forEach((knob, key) => {
|
||||
const fromQuery = query.get(key);
|
||||
const validValues = knobOptions(knob).map((option) => String(option.value));
|
||||
const useQueryValue = fromQuery !== null && validValues.includes(fromQuery);
|
||||
const raw = useQueryValue ? fromQuery : knob.default;
|
||||
knobValues[key] = typeof knob.default === "number" ? Number(raw) : raw;
|
||||
});
|
||||
|
||||
const renderKnobs = (recipe, hidden) => {
|
||||
if (!knobsContainer) return;
|
||||
const knobs = knobsFor(recipe);
|
||||
const renderedKeys = [...knobsContainer.querySelectorAll("[data-knob-row]")].map((row) => row.dataset.knobRow);
|
||||
if (renderedKeys.join(",") !== knobs.map((knob) => knob.key).join(",")) {
|
||||
knobsContainer.replaceChildren();
|
||||
knobs.forEach((knob) => {
|
||||
const row = document.createElement("div");
|
||||
row.className = "cookbook-selection-row";
|
||||
row.dataset.knobRow = knob.key;
|
||||
const labelWrap = document.createElement("div");
|
||||
labelWrap.className = "cookbook-selection-row__label";
|
||||
const strongLabel = document.createElement("strong");
|
||||
strongLabel.textContent = knob.label;
|
||||
const hintLabel = document.createElement("span");
|
||||
hintLabel.textContent = knob.hint || "";
|
||||
labelWrap.append(strongLabel, hintLabel);
|
||||
const grid = document.createElement("div");
|
||||
grid.className = "cookbook-option-grid cookbook-option-grid--hardware";
|
||||
grid.setAttribute("role", "group");
|
||||
grid.setAttribute("aria-label", knob.label);
|
||||
knobOptions(knob).forEach((option) => {
|
||||
const optionButton = document.createElement("button");
|
||||
optionButton.type = "button";
|
||||
optionButton.dataset.knobKey = knob.key;
|
||||
optionButton.dataset.knobValue = String(option.value);
|
||||
optionButton.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = option.label;
|
||||
optionButton.append(optionLabel);
|
||||
grid.append(optionButton);
|
||||
});
|
||||
row.append(labelWrap, grid);
|
||||
knobsContainer.append(row);
|
||||
});
|
||||
}
|
||||
knobsContainer.hidden = hidden || knobs.length === 0;
|
||||
knobsContainer.querySelectorAll("button[data-knob-key]").forEach((optionButton) => {
|
||||
const selected = String(knobValues[optionButton.dataset.knobKey]) === optionButton.dataset.knobValue;
|
||||
optionButton.classList.toggle("cookbook-option--selected", selected);
|
||||
optionButton.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
};
|
||||
|
||||
const recipesForRuntime = (runtimeId, groupRecipes) =>
|
||||
groupRecipes.filter((item) => runtimeFor(item).id === runtimeId);
|
||||
|
||||
const uniqueRuntimeIds = (groupRecipes) => {
|
||||
const ids = [];
|
||||
groupRecipes.forEach((item) => {
|
||||
const id = runtimeFor(item).id;
|
||||
if (!ids.includes(id)) ids.push(id);
|
||||
});
|
||||
return ids;
|
||||
};
|
||||
|
||||
const pickRecipeForRuntime = (runtimeId, preferredCount, groupRecipes) => {
|
||||
const siblings = recipesForRuntime(runtimeId, groupRecipes);
|
||||
if (!siblings.length) return null;
|
||||
if (preferredCount != null) {
|
||||
const match = siblings.find((item) => item.hardware?.gpu_count === preferredCount);
|
||||
if (match) return match;
|
||||
}
|
||||
return siblings.find((item) => item.hardware?.gpu_count === 1) || siblings[0];
|
||||
};
|
||||
|
||||
const renderRuntimeOptions = () => {
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const runtimeIds = uniqueRuntimeIds(groupRecipes);
|
||||
const renderedIds = [...hardwareOptions.querySelectorAll("[data-runtime-id]")].map((option) => option.dataset.runtimeId);
|
||||
if (renderedIds.join(",") === runtimeIds.join(",")) return;
|
||||
hardwareOptions.replaceChildren();
|
||||
runtimeIds.forEach((runtimeId) => {
|
||||
const representative = pickRecipeForRuntime(runtimeId, 1, groupRecipes);
|
||||
const runtime = runtimeFor(representative);
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.runtimeId = runtime.id;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = runtime.label;
|
||||
const optionHint = document.createElement("span");
|
||||
optionHint.textContent = runtime.hint;
|
||||
option.append(optionLabel, optionHint);
|
||||
hardwareOptions.append(option);
|
||||
});
|
||||
};
|
||||
|
||||
const renderDeviceOptions = (recipe) => {
|
||||
if (!deviceRow || !deviceOptions) return;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const runtime = runtimeFor(recipe);
|
||||
const siblings = recipesForRuntime(runtime.id, groupRecipes)
|
||||
.slice()
|
||||
.sort((left, right) => (left.hardware?.gpu_count || 0) - (right.hardware?.gpu_count || 0));
|
||||
const show = siblings.length > 1;
|
||||
deviceRow.hidden = !show;
|
||||
if (deviceCaption) {
|
||||
deviceCaption.textContent = runtime.id === "spark" ? "1 Spark or a QSFP pair" : "GPU count for this runtime";
|
||||
}
|
||||
if (!show) {
|
||||
deviceOptions.replaceChildren();
|
||||
return;
|
||||
}
|
||||
const renderedIds = [...deviceOptions.querySelectorAll("[data-recipe-id]")].map((option) => option.dataset.recipeId);
|
||||
if (renderedIds.join(",") !== siblings.map((item) => item.id).join(",")) {
|
||||
deviceOptions.replaceChildren();
|
||||
siblings.forEach((candidate) => {
|
||||
const option = document.createElement("button");
|
||||
option.type = "button";
|
||||
option.dataset.recipeId = candidate.id;
|
||||
option.setAttribute("aria-pressed", "false");
|
||||
const optionLabel = document.createElement("strong");
|
||||
optionLabel.textContent = gpuCountLabel(candidate.hardware, runtime.id);
|
||||
const optionHint = document.createElement("span");
|
||||
optionHint.textContent = runtime.id === "spark" && candidate.hardware?.gpu_count === 2
|
||||
? "Ray sequence parallel over QSFP"
|
||||
: runtime.id === "spark"
|
||||
? "One GB10, local process"
|
||||
: `${gpuCountLabel(candidate.hardware, runtime.id)} configured`;
|
||||
option.append(optionLabel, optionHint);
|
||||
deviceOptions.append(option);
|
||||
});
|
||||
}
|
||||
deviceOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.recipeId === recipe.id;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
};
|
||||
|
||||
let notes = root.querySelector("[data-cookbook-notes]");
|
||||
if (!notes) {
|
||||
notes = document.createElement("aside");
|
||||
notes.className = "cookbook-recipe-notes";
|
||||
notes.dataset.cookbookNotes = "";
|
||||
notes.hidden = true;
|
||||
result.insertBefore(notes, commandBlock);
|
||||
}
|
||||
|
||||
const render = ({ groupChanged = false, historyMode = "replace" } = {}) => {
|
||||
if (!root.isConnected) return;
|
||||
if (!byId.has(selectedRecipeId)) selectedRecipeId = defaultRecipeId;
|
||||
let recipe = byId.get(selectedRecipeId);
|
||||
if (groupChanged || groupIdFor(recipe) !== selectedGroupId) {
|
||||
const currentRuntime = runtimeFor(recipe).id;
|
||||
const currentCount = recipe.hardware?.gpu_count;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
recipe = pickRecipeForRuntime(currentRuntime, currentCount, groupRecipes) || groupRecipes[0];
|
||||
selectedRecipeId = recipe.id;
|
||||
}
|
||||
|
||||
selectedGroupId = groupIdFor(recipe);
|
||||
if (renderedRuntimeGroup !== selectedGroupId) {
|
||||
renderRuntimeOptions();
|
||||
renderedRuntimeGroup = selectedGroupId;
|
||||
}
|
||||
renderDeviceOptions(recipe);
|
||||
const runtime = runtimeFor(recipe);
|
||||
const profile = servingPanel && servingProfiles[recipe.id];
|
||||
const useServer = Boolean(profile && usagePreference === "server");
|
||||
// The measured local profile and the server config have separate evidence.
|
||||
const activeRecipe = useServer ? { ...recipe, hardware: profile.hardware, evidence: "Source-backed" } : recipe;
|
||||
const knobs = knobsFor(recipe);
|
||||
renderKnobs(recipe, useServer);
|
||||
|
||||
if (usage) {
|
||||
usage.querySelectorAll("[data-cookbook-mode]").forEach((option) => {
|
||||
const selected = option.dataset.cookbookMode === (useServer ? "server" : "python");
|
||||
option.disabled = option.dataset.cookbookMode === "server" && !profile;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
servingAvailability.textContent = profile
|
||||
? "The playground and the OpenAI Python client share one server process. Both workflows can run on your own machine."
|
||||
: servingLoadFailed
|
||||
? "Server examples could not be loaded. Open the H3 server guide below, or use Python directly."
|
||||
: "This recipe uses Python directly. FastH3 V1 and FastH3 V2 can also run a local server for the playground and the OpenAI Python client.";
|
||||
servingPanel.hidden = !useServer;
|
||||
commandBlock.hidden = useServer;
|
||||
root.querySelector("[data-cookbook-python-note]").hidden = useServer;
|
||||
}
|
||||
|
||||
if (useServer) {
|
||||
const isMLX = profile.runtime === "mlx";
|
||||
const isSpark = runtime.id === "spark";
|
||||
servingPanel.querySelector("[data-cookbook-server-lifetime]").textContent = isMLX
|
||||
? "Start once, then change prompts in the playground or your app. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident."
|
||||
: isSpark
|
||||
? "Start once, then change prompts in the playground or your app. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
servingPanel.querySelector("[data-cookbook-install-guide]").href = isMLX
|
||||
? "../../getting_started/installation/mlx/"
|
||||
: isSpark
|
||||
? "../../getting_started/installation/spark/"
|
||||
: "../../getting_started/installation/gpu/";
|
||||
servingPanel.querySelector("[data-cookbook-prepare]").hidden = !profile.prepare;
|
||||
servingPanel.querySelector("[data-cookbook-server-prepare]").textContent = profile.prepare;
|
||||
servingPanel.querySelector("[data-cookbook-server-install]").textContent = profile.install;
|
||||
servingPanel.querySelector("[data-cookbook-server-command]").textContent = profile.command;
|
||||
servingPanel.querySelector("[data-cookbook-health-command]").textContent = profile.health_command;
|
||||
servingPanel.querySelector("[data-cookbook-playground]").href = profile.playground_url;
|
||||
const client = profile.clients[selectedClient];
|
||||
const filename = client.source.split("/").pop();
|
||||
servingPanel.querySelector("[data-cookbook-client-install]").textContent = client.install;
|
||||
const clientCode = servingPanel.querySelector("[data-cookbook-client-code]");
|
||||
clientCode.className = `language-${selectedClient === "curl" ? "bash" : selectedClient}`;
|
||||
clientCode.textContent = client.code;
|
||||
servingPanel.querySelector("[data-cookbook-client-filename]").textContent = filename;
|
||||
servingPanel.querySelector("[data-cookbook-client-source]").href = `https://github.com/hao-ai-lab/FastVideo/blob/main/${client.source}`;
|
||||
const runner = { python: "python", javascript: "node", curl: "bash" }[selectedClient];
|
||||
servingPanel.querySelector("[data-cookbook-client-run]").textContent = `Save as ${filename} and run ${runner} ${filename}. The MP4 is saved with the job ID as its filename.`;
|
||||
servingPanel.querySelectorAll("[data-cookbook-client]").forEach((option) => {
|
||||
option.setAttribute("aria-pressed", String(option.dataset.cookbookClient === selectedClient));
|
||||
});
|
||||
}
|
||||
|
||||
modelOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.recipeGroup === selectedGroupId;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
hardwareOptions.querySelectorAll("button").forEach((option) => {
|
||||
const selected = option.dataset.runtimeId === runtime.id;
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
|
||||
description.textContent = useServer
|
||||
? `${recipe.group_label || recipe.label} generates video with audio. Start the local server, then use the playground or the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
: recipe.summary;
|
||||
label.textContent = useServer ? `${recipe.group_label || recipe.label} · Server` : recipe.label;
|
||||
model.textContent = recipe.model;
|
||||
task.textContent = recipe.task;
|
||||
hardwareValue.textContent = runtimeSummary(activeRecipe);
|
||||
if (artifact) artifact.textContent = useServer ? "MP4 with audio" : recipe.expected_artifact || "Not yet documented for this recipe.";
|
||||
if (evidenceCell) {
|
||||
evidenceCell.textContent = activeRecipe.evidence || "Source-backed";
|
||||
evidenceCell.classList.toggle("cookbook-badge--verified", activeRecipe.evidence === "Verified");
|
||||
evidenceCell.classList.toggle("cookbook-badge--source-backed", activeRecipe.evidence !== "Verified");
|
||||
}
|
||||
source.href = `https://github.com/hao-ai-lab/FastVideo/blob/main/${useServer ? profile.source : recipe.source}`;
|
||||
source.textContent = useServer ? "View server configuration" : "Open example source";
|
||||
modelLink.href = `https://huggingface.co/${recipe.model}`;
|
||||
command.textContent = useServer ? recipe.command : appendKnobFlags(recipe.command, knobs, knobValues);
|
||||
|
||||
renderHardwareEvidence(hardwareState, hardwareBadge, activeRecipe);
|
||||
|
||||
const knobCaveats = useServer ? [] : knobs
|
||||
.filter((knob) => String(knobValues[knob.key]) !== String(knob.default))
|
||||
.map((knob) => `${knob.label} is set away from its recorded default (${knobDefaultLabel(knob)}). ` +
|
||||
"The script accepts this value, but it has not been benchmarked here.");
|
||||
const limitations = [...(useServer ? [
|
||||
profile.runtime === "mlx"
|
||||
? "This MLX server config has no recorded hardware run. Measurements from the Python recipe are not server memory requirements. Only text-to-video/audio is wired; reference inputs and fast modes are not exposed here."
|
||||
: runtime.id === "spark"
|
||||
? "This Spark server config has no recorded serving benchmark. Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Compilation of the DiT is disabled."
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
`${profile.sampling.width} × ${profile.sampling.height} · ${profile.sampling.num_frames} frames · ${profile.sampling.fps} fps. The server supplies these defaults; the client sends the model and prompt.`,
|
||||
"Generation is serialized. Job metadata is held in memory and is lost when the server restarts.",
|
||||
] : recipe.limitations || []), ...knobCaveats];
|
||||
notes.replaceChildren();
|
||||
notes.hidden = limitations.length === 0;
|
||||
if (limitations.length) {
|
||||
const notesHeading = document.createElement("strong");
|
||||
notesHeading.textContent = "Know before you run";
|
||||
const notesList = document.createElement("ul");
|
||||
limitations.forEach((item) => {
|
||||
const listItem = document.createElement("li");
|
||||
listItem.textContent = item;
|
||||
notesList.append(listItem);
|
||||
});
|
||||
notes.append(notesHeading, notesList);
|
||||
}
|
||||
|
||||
const nextQuery = new URLSearchParams(window.location.search);
|
||||
nextQuery.set("recipe", recipe.id);
|
||||
nextQuery.set("runtime", runtime.id);
|
||||
const deviceSiblings = recipesForRuntime(runtime.id, groups.get(selectedGroupId) || []);
|
||||
if (deviceSiblings.length > 1 && recipe.hardware?.gpu_count != null) {
|
||||
nextQuery.set("gpus", String(recipe.hardware.gpu_count));
|
||||
} else {
|
||||
nextQuery.delete("gpus");
|
||||
}
|
||||
knobDefs.forEach((knob, key) => {
|
||||
if (knobs.some((activeKnob) => activeKnob.key === key)) nextQuery.set(key, String(knobValues[key]));
|
||||
else nextQuery.delete(key);
|
||||
});
|
||||
if (usage) {
|
||||
nextQuery.set("use", useServer ? "server" : "python");
|
||||
if (useServer && clientDetails?.open) nextQuery.set("client", selectedClient);
|
||||
else nextQuery.delete("client");
|
||||
}
|
||||
const nextUrl = `${window.location.pathname}?${nextQuery.toString()}${window.location.hash}`;
|
||||
if (historyMode === "push") window.history.pushState({}, "", nextUrl);
|
||||
else if (historyMode === "replace") window.history.replaceState({}, "", nextUrl);
|
||||
const modeSummary = useServer ? " with a persistent server" : "";
|
||||
status.textContent = `${recipe.label} selected for ${runtime.label}${modeSummary}.`;
|
||||
};
|
||||
|
||||
modelOptions.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-recipe-group]");
|
||||
if (!option) return;
|
||||
selectedGroupId = option.dataset.recipeGroup;
|
||||
render({ groupChanged: true, historyMode: "push" });
|
||||
});
|
||||
hardwareOptions.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-runtime-id]");
|
||||
if (!option) return;
|
||||
const groupRecipes = groups.get(selectedGroupId) || [];
|
||||
const currentCount = byId.get(selectedRecipeId)?.hardware?.gpu_count;
|
||||
const nextRecipe = pickRecipeForRuntime(option.dataset.runtimeId, currentCount, groupRecipes);
|
||||
if (!nextRecipe) return;
|
||||
selectedRecipeId = nextRecipe.id;
|
||||
selectedGroupId = groupIdFor(nextRecipe);
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
deviceOptions?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-recipe-id]");
|
||||
if (!option) return;
|
||||
selectedRecipeId = option.dataset.recipeId;
|
||||
selectedGroupId = groupIdFor(byId.get(selectedRecipeId));
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
usage?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-cookbook-mode]");
|
||||
if (!option || option.disabled) return;
|
||||
usagePreference = option.dataset.cookbookMode;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
servingPanel?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-cookbook-client]");
|
||||
if (!option) return;
|
||||
selectedClient = option.dataset.cookbookClient;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
knobsContainer?.addEventListener("click", (event) => {
|
||||
const option = event.target.closest("button[data-knob-key]");
|
||||
if (!option) return;
|
||||
const knob = knobDefs.get(option.dataset.knobKey);
|
||||
knobValues[option.dataset.knobKey] = typeof knob.default === "number"
|
||||
? Number(option.dataset.knobValue) : option.dataset.knobValue;
|
||||
render({ historyMode: "push" });
|
||||
});
|
||||
|
||||
render();
|
||||
|
||||
familyPopstate = () => {
|
||||
if (!root.isConnected) return;
|
||||
const nextQuery = new URLSearchParams(window.location.search);
|
||||
const nextRecipe = nextQuery.get("recipe");
|
||||
selectedRecipeId = nextRecipe && byId.has(nextRecipe) ? nextRecipe : defaultRecipeId;
|
||||
selectedGroupId = groupIdFor(byId.get(selectedRecipeId));
|
||||
usagePreference = workflow(nextQuery.get("use"));
|
||||
selectedClient = ["python", "javascript", "curl"].includes(nextQuery.get("client")) ? nextQuery.get("client") : "curl";
|
||||
if (clientDetails) clientDetails.open = nextQuery.has("client");
|
||||
knobDefs.forEach((knob, key) => {
|
||||
const fromQuery = nextQuery.get(key);
|
||||
const validValues = knobOptions(knob).map((option) => String(option.value));
|
||||
if (fromQuery !== null && validValues.includes(fromQuery)) {
|
||||
knobValues[key] = typeof knob.default === "number" ? Number(fromQuery) : fromQuery;
|
||||
}
|
||||
});
|
||||
render({ historyMode: "none" });
|
||||
};
|
||||
bindFamilyPopstate();
|
||||
};
|
||||
|
||||
const init = () => {
|
||||
document.querySelectorAll("[data-cookbook]").forEach(async (root) => {
|
||||
initEvervault(document);
|
||||
document.querySelectorAll("[data-cookbook][data-family]").forEach((root) => {
|
||||
if (root.dataset.initialized) return;
|
||||
root.dataset.initialized = "true";
|
||||
|
||||
const select = root.querySelector("[data-cookbook-recipe]");
|
||||
const model = root.querySelector("[data-cookbook-model]");
|
||||
const source = root.querySelector("[data-cookbook-source]");
|
||||
const command = root.querySelector("[data-cookbook-command]");
|
||||
const status = root.querySelector("[data-cookbook-status]");
|
||||
|
||||
try {
|
||||
const { recipes } = await loadRecipes(root.dataset.recipes);
|
||||
const byId = new Map(recipes.map((recipe) => [recipe.id, recipe]));
|
||||
const groups = new Map();
|
||||
|
||||
select.replaceChildren();
|
||||
recipes.forEach((recipe) => {
|
||||
if (!groups.has(recipe.task)) {
|
||||
const group = document.createElement("optgroup");
|
||||
group.label = recipe.task;
|
||||
groups.set(recipe.task, group);
|
||||
select.append(group);
|
||||
}
|
||||
groups.get(recipe.task).append(new Option(recipe.label, recipe.id));
|
||||
});
|
||||
|
||||
const render = () => {
|
||||
const recipe = byId.get(select.value);
|
||||
model.textContent = recipe.model;
|
||||
source.textContent = recipe.source;
|
||||
source.href = `https://github.com/hao-ai-lab/FastVideo/blob/main/${recipe.source}`;
|
||||
command.textContent = recipe.command;
|
||||
status.textContent = `${recipe.label} selected.`;
|
||||
};
|
||||
|
||||
select.addEventListener("change", render);
|
||||
select.disabled = false;
|
||||
render();
|
||||
} catch (error) {
|
||||
status.textContent = "Recipes could not be loaded. Use the examples link below.";
|
||||
console.error("Failed to load FastVideo cookbook recipes", error);
|
||||
}
|
||||
initFamilyBuilder(root);
|
||||
});
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
# Cookbook logo sources
|
||||
|
||||
All marks are official publisher assets from Hugging Face organization pages
|
||||
(vendored byte-for-byte from each org's public avatar). No imitation or
|
||||
redrawn logos are used. Families without a publisher-appropriate mark use a
|
||||
plain typographic tile instead — that tile is a UI placeholder, not a logo.
|
||||
|
||||
| Asset | Source | Purpose |
|
||||
| --- | --- | --- |
|
||||
| `wan-ai.webp` | [Official Wan-AI Hugging Face organization avatar](https://huggingface.co/Wan-AI) | Wan family card and page header |
|
||||
| `ltx.webp` | [Official Lightricks Hugging Face organization avatar](https://huggingface.co/Lightricks) | LTX family card and page header |
|
||||
| `tencent-hunyuan.webp` | [Official Tencent Hunyuan Hugging Face organization avatar](https://huggingface.co/Tencent-Hunyuan) | Hunyuan and GameCraft cards, Hunyuan page header |
|
||||
| `nvidia.webp` | [Official NVIDIA Hugging Face organization avatar](https://huggingface.co/nvidia) | Cosmos and GEN3C cards, Cosmos page header |
|
||||
| `kandinsky.webp` | [Official Kandinsky Lab Hugging Face organization avatar](https://huggingface.co/kandinskylab) | Kandinsky 5 family card and page header |
|
||||
| `black-forest-labs.webp` | [Official Black Forest Labs Hugging Face organization avatar](https://huggingface.co/black-forest-labs) | FLUX family card and page header |
|
||||
| `minimax.webp` | [Official MiniMax Hugging Face organization avatar](https://huggingface.co/MiniMaxAI) | MiniMax H3 family card and page header |
|
||||
| `tongyi.webp` | [Official Tongyi MAI Hugging Face organization avatar](https://huggingface.co/Tongyi-MAI) | Z-Image family card and page header |
|
||||
| `zai.webp` | [Official Z.ai Hugging Face organization avatar](https://huggingface.co/zai-org) | GLM-Image family card and page header |
|
||||
| `stabilityai.webp` | [Official Stability AI Hugging Face organization avatar](https://huggingface.co/stabilityai) | Stable Diffusion and Stable Audio cards and page headers |
|
||||
| `meituan-longcat.webp` | [Official Meituan LongCat Hugging Face organization avatar](https://huggingface.co/meituan-longcat) | LongCat family card and page header |
|
||||
| `fastvideo.webp` | [Official FastVideo Hugging Face organization avatar](https://huggingface.co/FastVideo) | Matrix Game and MMAudio cards (converted weights published by this org), DreamX card |
|
||||
|
||||
Typographic tiles (no vendored image): TurboDiffusion ("Turbo"), HY-World
|
||||
("HY"), LingBot ("LB"), MMAudio page header ("MMA"). These publishers have no
|
||||
single official mark appropriate for reuse in the catalog; add a licensed
|
||||
asset here if one becomes available.
|
||||
|
After Width: | Height: | Size: 2.2 KiB |
|
After Width: | Height: | Size: 2.9 KiB |
|
After Width: | Height: | Size: 2.1 KiB |
|
After Width: | Height: | Size: 1.3 KiB |
|
After Width: | Height: | Size: 5.1 KiB |
|
After Width: | Height: | Size: 4.1 KiB |
|
After Width: | Height: | Size: 3.5 KiB |
|
After Width: | Height: | Size: 2.8 KiB |
|
After Width: | Height: | Size: 7.6 KiB |
|
After Width: | Height: | Size: 6.0 KiB |
|
After Width: | Height: | Size: 6.8 KiB |
|
After Width: | Height: | Size: 2.4 KiB |
@@ -6,6 +6,50 @@ Sparse attention mechanism selecting top-k blocks.
|
||||
|
||||
VSA is included in the `fastvideo-kernel` package. See the [main Attention page](../index.md) for build instructions.
|
||||
|
||||
## Apple Silicon (MiniMax H3 / FastH3)
|
||||
|
||||
The native MLX runtime has an inference-only H3 VSA path that is separate from
|
||||
the CUDA `fastvideo-kernel` package:
|
||||
|
||||
- **INT8 / INT6 / INT4** are **weight-only**. They cover linear matrices,
|
||||
including `attn.to_gate_compress` when you convert with `--include-vsa`.
|
||||
Attention Q/K/V stay BF16 (or the selected activation dtype). There is no
|
||||
INT6 Q/K/V attention kernel.
|
||||
- **Dense-only checkpoints** (the default converter) drop the 50 gate
|
||||
matrices and keep fused SDPA. They remain valid for dense inference.
|
||||
- **VSA-capable checkpoints** retain those gates, quantize them on the same
|
||||
affine grid, and record `vsa.capable` in `mlx_h3_dit.json`. Preview leaves
|
||||
runtime VSA off until you pass `--vsa`. `mlx_fasth3_8step.py` turns it on.
|
||||
- **Tile sizes** 64 `(4, 4, 4)` and 256 `(4, 8, 8)`. Prefix keys can be
|
||||
`exempt` or `compete`. `--vsa-dense-first-n-steps` and `--vsa-dense-layers`
|
||||
force dense SDPA on the selected steps or blocks.
|
||||
- **`--vsa-impl auto`** uses the chunked gather+SDPA **reference** path.
|
||||
`--vsa-impl simd` runs the SIMD-group Metal kernel (tile 64, head dim 128)
|
||||
and falls back to reference on unsupported shapes or kernel failure. The
|
||||
runtime executes a small kernel probe before use and remembers failures
|
||||
for the process, so later blocks do not retry a broken backend.
|
||||
It is not the default: 480p four-step generation is faster than reference
|
||||
but does not yet match reference video. `--vsa-impl reference` is the same
|
||||
as `auto`.
|
||||
|
||||
See the [MLX install guide](../../getting_started/installation/mlx.md)
|
||||
and the [MiniMax H3 cookbook](../../cookbook/minimax-h3.md) for conversion
|
||||
and `mlx_fasth3.py` / `mlx_fasth3_8step.py` flags. Do not enable
|
||||
VSA on a dense-only checkpoint; reconvert with `--include-vsa` first. V1
|
||||
VSA is opt-in. FastH3 V2 converts with `--include-vsa` and turns VSA on
|
||||
by default.
|
||||
|
||||
H3 uses fused MLX RMSNorm by default, including dense inference. This can
|
||||
change BF16 rounding relative to the older explicit normalization path.
|
||||
`FASTVIDEO_MLX_FAST_NORM` controls Wan normalization only.
|
||||
|
||||
The generation report aggregates VSA statistics across all blocks and steps.
|
||||
`impl_counts` and `fallback_reasons` show mixed execution and fallback.
|
||||
`video_keep` is the mean number of selected video-key tiles per video query
|
||||
tile and head, including dense overrides. `achieved_sparsity` is the matching
|
||||
mean video-tile sparsity; in `compete` mode it measures actual selections,
|
||||
not the requested top-k budget. These are tile counts, not token-level FLOPs.
|
||||
|
||||
## Usage
|
||||
|
||||
```python
|
||||
|
||||
@@ -30,7 +30,7 @@ FastVideo and the reference model first produce different numbers?"
|
||||
| General logging | `init_logger(__name__)` |
|
||||
| Per-stage timing | `FASTVIDEO_STAGE_LOGGING` |
|
||||
| Profiling kernel timings | `FASTVIDEO_TORCH_PROFILER_DIR` (see [Profiling](profiling.md)) |
|
||||
| Function-call tracing | `FASTVIDEO_TRACE_FUNCTION` (heavy) |
|
||||
| Function-call tracing | `fastvideo.logger.enable_trace_function_call()` (heavy) |
|
||||
|
||||
## Quickstart
|
||||
|
||||
|
||||
@@ -141,6 +141,24 @@ it), while a GitHub outage or a >25 min wait lets it run anyway (fail open).
|
||||
The complete static graph remains available through `/test full`; path
|
||||
selection never deletes or dynamically invents a Buildkite step.
|
||||
|
||||
Integration steps depend on `golden-gate`. A failed selected golden prevents
|
||||
the expensive downstream jobs from starting; `/test full` still selects all
|
||||
twenty lanes. A condition-skipped golden satisfies the dependency, so direct
|
||||
lane reruns, scheduled SSIM, and training-only merge plans keep their existing
|
||||
meaning. This follows Buildkite's
|
||||
[conditional dependency rules](https://buildkite.com/docs/pipelines/configure/depends-on).
|
||||
No dependency allows failures. The trusted uploader accepts either the old
|
||||
complete graph or the complete golden-first graph during rollout, and rejects
|
||||
partial or arbitrary dependency changes.
|
||||
|
||||
The six automatic Fastcheck lanes are unchanged. Within VAE and transformer
|
||||
lanes, small Wan goldens run before independent component parity. Wan paths
|
||||
select matching VAE, dense/trajectory, or causal-cache goldens; shared Wan
|
||||
config and pipeline wiring select all four. Shared runtime changes continue
|
||||
to select broader coverage. The existing block references retain their exact
|
||||
environment identity; new tensor gates distinguish the effective FA2/FA4
|
||||
switch and VAE gates do not depend on an unused attention backend.
|
||||
|
||||
| Lane | Public `TEST_TYPE` | GPUs | Typical merge trigger |
|
||||
|---|---|---:|---|
|
||||
| Encoder | `encoder` | 1 | Universal Fastcheck |
|
||||
|
||||
@@ -43,6 +43,8 @@ FastVideo maps a Diffusers-style repo into a pipeline like:
|
||||
- `fastvideo/models/*`: model implementations (DiT, VAE, encoders, upsamplers).
|
||||
- `fastvideo/configs/models/*`: arch configs and `param_names_mapping` for
|
||||
weight name translation.
|
||||
- `fastvideo/models/wan/`: Wan's dense transformer, VAE, and component configs
|
||||
live together. The old Wan modules remain compatibility re-exports.
|
||||
- `fastvideo/configs/pipelines/*`: pipeline wiring (component classes + names).
|
||||
- `fastvideo/api/sampling_param.py`: runtime sampling parameters.
|
||||
- `fastvideo/pipelines/basic/*`: end-to-end pipeline logic built from stages.
|
||||
@@ -55,19 +57,15 @@ Minimal usage example (based on `examples/inference/basic/basic.py`):
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api.sampling_param import SamplingParam
|
||||
|
||||
model_id = "Wan-AI/Wan2.1-T2V-1.3B-Diffusers" # or official_weights/<model_name>/
|
||||
generator = VideoGenerator.from_pretrained(model_id, num_gpus=1)
|
||||
generator = VideoGenerator.from_pretrained(model_id, {"engine": {"num_gpus": 1}})
|
||||
|
||||
sampling = SamplingParam.from_pretrained(model_id)
|
||||
sampling.num_frames = 45
|
||||
video = generator.generate_video(
|
||||
"A vibrant city street at sunset.",
|
||||
sampling_param=sampling,
|
||||
output_path="video_samples",
|
||||
save_video=True,
|
||||
)
|
||||
video = generator.generate({
|
||||
"prompt": "A vibrant city street at sunset.",
|
||||
"sampling": {"num_frames": 45},
|
||||
"output": {"output_path": "video_samples", "save_video": True},
|
||||
})
|
||||
```
|
||||
|
||||
## Some questions to ask yourself before starting
|
||||
@@ -209,7 +207,7 @@ class OfficialWanTransformer(torch.nn.Module):
|
||||
def forward(self, x):
|
||||
return self.patch_embedding(x)
|
||||
|
||||
# FastVideo model (simplified) in fastvideo/models/dits/wanvideo.py
|
||||
# FastVideo model (simplified) in fastvideo/models/wan/transformer.py
|
||||
class PatchEmbed(torch.nn.Module):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
@@ -227,7 +225,7 @@ class WanTransformer3DModel(torch.nn.Module):
|
||||
return self.patch_embedding(x)
|
||||
|
||||
# Mapping defined in a config (simplified; see the real mapping in
|
||||
# fastvideo/configs/models/dits/wanvideo.py)
|
||||
# fastvideo/models/wan/config.py)
|
||||
param_names_mapping = {
|
||||
r"^patch_embedding\.(.*)$": r"patch_embedding.proj.\1",
|
||||
r"^blocks\.(\d+)\.attn1\.to_q\.(.*)$": r"blocks.\1.to_q.\2",
|
||||
@@ -275,7 +273,7 @@ Mapping steps:
|
||||
- Instantiate the FastVideo DiT (`WanTransformer3DModel`) and compare
|
||||
its `state_dict().keys()` to the official keys.
|
||||
- Update `param_names_mapping` in
|
||||
fastvideo/configs/models/dits/wanvideo.py to resolve missing/unexpected keys.
|
||||
fastvideo/models/wan/config.py to resolve missing/unexpected keys.
|
||||
- Use `load_state_dict(strict=False)` during iteration to surface mismatches.
|
||||
```
|
||||
|
||||
@@ -466,19 +464,24 @@ The Wan2.1 T2V 1.3B Diffusers pipeline is a good “standard” example for
|
||||
FastVideo integration.
|
||||
|
||||
1. Verify model config + mapping.
|
||||
- DiT mapping: `fastvideo/configs/models/dits/wanvideo.py`
|
||||
- VAE: `fastvideo/models/vaes/wanvae.py`
|
||||
- DiT: `fastvideo/models/wan/transformer.py`
|
||||
- DiT mapping: `fastvideo/models/wan/config.py`
|
||||
- VAE: `fastvideo/models/wan/vae.py`
|
||||
- VAE config: `fastvideo/models/wan/vae_config.py`
|
||||
- Text encoder: `fastvideo/models/encoders/t5.py`
|
||||
|
||||
2. Parity test the core components.
|
||||
- Start with `bash scripts/validate_wan.sh all`: contracts and tiny goldens.
|
||||
- Example tests: `fastvideo/tests/transformers/test_wanvideo.py`,
|
||||
`fastvideo/tests/vaes/test_wan_vae.py`,
|
||||
`fastvideo/tests/encoders/test_t5_encoder.py`
|
||||
|
||||
3. Pipeline wiring.
|
||||
- Pipeline: `fastvideo/pipelines/basic/wan/wan_pipeline.py`
|
||||
- Pipeline config: `fastvideo/configs/pipelines/wan.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/profiles.py`
|
||||
- Denoising and first-frame preparation: `fastvideo/pipelines/basic/wan/stages/`
|
||||
- Variant definitions: `fastvideo/models/wan/definition.py`
|
||||
- Pipeline config: `fastvideo/models/wan/pipeline_config.py`
|
||||
- Sampling defaults: `fastvideo/pipelines/basic/wan/presets.py`
|
||||
|
||||
4. Minimal example.
|
||||
- Script: `examples/inference/basic/basic.py`
|
||||
|
||||
@@ -0,0 +1,294 @@
|
||||
# Environment Variables
|
||||
|
||||
FastVideo reads environment variables for expert switches, debugging, profiling, and the settings that launchers such
|
||||
as `torchrun` provide. This page is the policy for those variables. The contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit CI lane, and the coding-agent skill
|
||||
`.agents/skills/env-var-conventions/SKILL.md` points here. When the policy changes, update this page and the contract
|
||||
test in the same pull request.
|
||||
|
||||
## Rules
|
||||
|
||||
1. **Register every FastVideo variable in `fastvideo/envs.py`.** Each entry declares a type, a default, a category,
|
||||
and a description. Variables that other tools own (CUDA, NCCL, PyTorch, launchers) are not registered; code reads
|
||||
them directly with `os.environ.get("NAME")`, and the name must be in the external-variable allowlist
|
||||
(`EXTERNAL_ALLOWLIST` in the contract test). When FastVideo sets such a variable for the other tool, it calls
|
||||
`envs.set_external`, `envs.setdefault_external`, or `envs.unset_external`, and the name must be in
|
||||
`EXTERNAL_WRITE_ALLOWLIST`. Variables that FastVideo's CI and CI tooling define (for example `TEST_SCOPE` and
|
||||
`PERF_RUN_SOURCE`) keep their names, and test code under `fastvideo/tests/` reads them directly; they are listed
|
||||
in `CI_ONLY_VARIABLES` in the contract test, together with the file that sets each one.
|
||||
2. **Read with `envs.NAME.get()`, write with `envs.NAME.set()`, and change a value in tests with
|
||||
`envs.NAME.override()`.** Each type has one parsing rule. A value that the rule rejects raises
|
||||
`fastvideo.envs.EnvVarError` instead of falling back to the default.
|
||||
3. **Name FastVideo variables with the `FASTVIDEO_` prefix.** The second word states the purpose where one applies:
|
||||
`ENABLE_`, `DISABLE_`, `USE_`, `FORCE_`, `DEBUG_`, `TEST_`. Variables that only tests read use
|
||||
`FASTVIDEO_TEST_`, for example `FASTVIDEO_TEST_SD35_MODEL_DIR`, and the category `test`.
|
||||
4. **Keep a renamed variable as a deprecated alias until the next minor release.** Setting the old name logs a
|
||||
warning. Delete a variable that no code reads, and list it in `DEPRECATED_VARIABLES` so that setting it logs a
|
||||
warning.
|
||||
5. **Give each setting one source: an argument or an environment variable.** Settings that users change per
|
||||
deployment are arguments (CLI or YAML). Expert switches, emergency off switches, and debugging and test switches
|
||||
are environment variables.
|
||||
6. **Read variables inside functions.** `envs.NAME.get()` runs when the function runs, so a changed value takes
|
||||
effect without re-importing a module. Module level, class bodies, decorators, and default argument values run at
|
||||
import time.
|
||||
7. **Do not write the environment to pass values between parts of FastVideo.** Pass an argument instead. Tests use
|
||||
`envs.NAME.override()`, and `envs.override_external()` for variables outside the registry; both restore the
|
||||
previous value.
|
||||
|
||||
## Field types
|
||||
|
||||
| Class | Value type | Parsing rule |
|
||||
| ------------ | ---------------- | ----------------------------------------------------------------------------------- |
|
||||
| `EnvBool` | `bool` | `1`, `true`, `yes`, `on` are true; `0`, `false`, `no`, `off`, and `""` are false. |
|
||||
| | | Case-insensitive; surrounding whitespace is ignored. |
|
||||
| `EnvInt` | `int` | `int(value)` |
|
||||
| `EnvFloat` | `float` | `float(value)` |
|
||||
| `EnvStr` | `str` or `None` | The raw string. A `None` default means that the variable has no default. |
|
||||
| `EnvPath` | `str` or `None` | The raw string with a leading `~` expanded. |
|
||||
| `EnvChoice` | `str` | Stripped and lower-cased, then checked against the declared `choices`. |
|
||||
|
||||
A default can be a zero-argument function; `get()` calls it on each read while the variable is unset. The path roots
|
||||
use this to follow `XDG_CONFIG_HOME` and `XDG_CACHE_HOME`.
|
||||
|
||||
Using a field without a method, as in `if envs.FASTVIDEO_FA4:`, raises `TypeError`.
|
||||
|
||||
## Add a variable
|
||||
|
||||
1. Declare the variable in the matching section of `fastvideo/envs.py`:
|
||||
|
||||
```python
|
||||
FASTVIDEO_DEBUG_MY_STAGE = EnvBool(False, category="debug", doc="Log the inputs of MyStage.")
|
||||
```
|
||||
|
||||
The category is one of the values in `envs.CATEGORIES`.
|
||||
|
||||
2. Read the variable inside a function:
|
||||
|
||||
```python
|
||||
import fastvideo.envs as envs
|
||||
|
||||
def forward(self, batch):
|
||||
if envs.FASTVIDEO_DEBUG_MY_STAGE.get():
|
||||
logger.info("MyStage inputs: %s", batch.keys())
|
||||
```
|
||||
|
||||
3. Regenerate the table at the end of this page:
|
||||
|
||||
```bash
|
||||
python fastvideo/tests/contract/test_env_policy.py
|
||||
```
|
||||
|
||||
4. Run the contract test:
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/contract/test_env_policy.py
|
||||
```
|
||||
|
||||
In a test, change the value with `override`, which restores the previous value on exit:
|
||||
|
||||
```python
|
||||
with envs.FASTVIDEO_DEBUG_MY_STAGE.override(True):
|
||||
run_stage()
|
||||
```
|
||||
|
||||
For a variable outside the registry, such as `MASTER_PORT` or `TEST_SCOPE`, use `envs.override_external`. To keep
|
||||
an override until the end of a test, enter it through the `env_overrides` fixture from `fastvideo/tests/conftest.py`,
|
||||
which restores every value at teardown:
|
||||
|
||||
```python
|
||||
def test_my_stage(env_overrides):
|
||||
env_overrides.enter_context(envs.FASTVIDEO_DEBUG_MY_STAGE.override(True))
|
||||
env_overrides.enter_context(envs.override_external("MASTER_PORT", "29512"))
|
||||
run_stage()
|
||||
```
|
||||
|
||||
In test code under `fastvideo/tests/`, `override_external` accepts any name that code may read directly
|
||||
(`EXTERNAL_ALLOWLIST`, `CI_ONLY_VARIABLES`) or that is in `EXTERNAL_WRITE_ALLOWLIST`. Library code may write only
|
||||
the names in `EXTERNAL_WRITE_ALLOWLIST`.
|
||||
|
||||
## Rename or remove a variable
|
||||
|
||||
To rename a variable, declare it under the new name and list the old name in `deprecated_names`:
|
||||
|
||||
```python
|
||||
FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS = EnvBool(True,
|
||||
category="sampling",
|
||||
doc="...",
|
||||
deprecated_names=("LTX2_USE_DISTILLED_SIGMAS", ))
|
||||
```
|
||||
|
||||
`get()` reads an old name only when the new name is unset, and logs a warning once. Update the uses of the old name
|
||||
in `examples/`, `scripts/`, `docs/`, `apps/`, and the tests in the same pull request. Delete the old name in the next
|
||||
minor release.
|
||||
|
||||
To remove a variable that no code reads, delete its entry and add the name to `DEPRECATED_VARIABLES` in
|
||||
`fastvideo/envs.py` with a reason. The config resolution step `warn_deprecated_environment_variables` in
|
||||
`fastvideo/api/inference_resolution.py` calls `envs.warn_deprecated_variables()`, which logs a warning for each listed
|
||||
variable that is set. Delete the entry in the next minor release.
|
||||
|
||||
## What the contract test checks
|
||||
|
||||
The test parses every Python file under `fastvideo/`, including `fastvideo/tests/`, with Python's `ast` module. It
|
||||
skips `fastvideo/third_party/`, which is copied from upstream projects, and the registry `fastvideo/envs.py`. It does
|
||||
not check `apps/`, `examples/`, `scripts/`, `fastvideo-kernel/`, or `docs/`.
|
||||
|
||||
It reports each violation as `<path>: <kind> <name>`:
|
||||
|
||||
| Kind | Code that triggers it | Fix |
|
||||
| ------------------ | ------------------------------------------------------------ | -------------------------------------------- |
|
||||
| `read` | `os.getenv`, `os.environ.get`, `os.environ[...]`, or | Register the variable and call |
|
||||
| | `"NAME" in os.environ` with a name outside the allowlist, or | `envs.NAME.get()`. For a variable that |
|
||||
| | with a name built at runtime (`<dynamic>`) | another tool owns, add it to |
|
||||
| | | `EXTERNAL_ALLOWLIST` with a reason. |
|
||||
| `write` | `os.environ[...] = ...`, `setdefault`, `pop`, `del`, | Pass an argument instead. In tests, use |
|
||||
| | `os.putenv`, `os.unsetenv`, `monkeypatch.setenv`/`delenv`, | `envs.NAME.override()`, or |
|
||||
| | or an `envs.*_external` call with a name that the | `envs.override_external()` for a variable |
|
||||
| | helper does not accept | outside the registry. For a variable that |
|
||||
| | | another tool reads, call an |
|
||||
| | | `envs.*_external` helper and add the name to |
|
||||
| | | `EXTERNAL_WRITE_ALLOWLIST` with a reason. |
|
||||
| `whole-environ` | `os.environ.copy()`, `dict(os.environ)`, iteration, | Read the specific variables that the code |
|
||||
| | `mock.patch.dict(os.environ, ...)`, `os.environ.update` | needs. |
|
||||
| `bare-field` | A registry field used without calling one of its methods, | Call `envs.NAME.get()`. |
|
||||
| | as in `envs.NAME == "auto"` or `getter = envs.NAME.get` | |
|
||||
| `import-time-read` | `envs.NAME.get()` outside a function | Move the read into the function that uses |
|
||||
| | | the value. |
|
||||
| `prefix` | A registry entry without the `FASTVIDEO_` prefix | Rename the variable and keep the old name in |
|
||||
| | | `deprecated_names`. |
|
||||
| `unread` | A registry entry that no code reads with `get()` or | Delete the variable and add it to |
|
||||
| | `is_set()` | `DEPRECATED_VARIABLES`. |
|
||||
|
||||
The test recognizes `os` imported under another name, `from os import environ, getenv`, and a name held in a
|
||||
module-level string constant. Code that reaches the environment through `importlib` or `getattr(os, "environ")` is
|
||||
left to code review.
|
||||
|
||||
The test also checks that every registry entry has a category from `envs.CATEGORIES` and a description, and that the
|
||||
table at the end of this page matches the registry.
|
||||
|
||||
**Known violations.** `KNOWN_VIOLATIONS` in the contract test lists the violations that existed when the policy was
|
||||
introduced. The list only shrinks. A violation that is not in the list fails the test. A listed violation that no
|
||||
longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
|
||||
## Registered variables
|
||||
|
||||
<!-- BEGIN GENERATED ENV TABLE: python fastvideo/tests/contract/test_env_policy.py -->
|
||||
| Variable | Type | Default | Category | Description |
|
||||
| ---------------------------------------------------------- | ------------------------------ | ----------------------------------------------------- | ----------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| `FASTVIDEO_CONFIG_ROOT` | path | computed | path | Root directory for FastVideo configuration files, at runtime and at installation. Defaults to ~/.config/fastvideo, or $XDG_CONFIG_HOME/fastvideo when XDG_CONFIG_HOME is set. |
|
||||
| `FASTVIDEO_CACHE_ROOT` | path | computed | path | Root directory for FastVideo cache files. Defaults to ~/.cache/fastvideo, or $XDG_CACHE_HOME/fastvideo when XDG_CACHE_HOME is set. |
|
||||
| `FASTVIDEO_REASON1_WEIGHTS_PATH` | str | unset | path | Local path or Hugging Face id of Reason1 weights to load instead of the checkpoint's own. |
|
||||
| `FASTVIDEO_HOST_IP` | str | `""` | distributed | IP address of this node when the node has several network interfaces. Set it on each node for multi-node inference. |
|
||||
| `FASTVIDEO_LOOPBACK_IP` | str | `""` | distributed | Loopback IP address to use instead of the detected one. |
|
||||
| `FASTVIDEO_RAY_PER_WORKER_GPUS` | float | `1.0` | distributed | GPUs per Ray worker. A fraction lets Ray schedule several actors on one GPU, so other actors can share the GPUs with FastVideo. |
|
||||
| `FASTVIDEO_NCCL_SO_PATH` | str | unset | distributed | Path to the NCCL library file. Needed because the nccl>=2.19 that PyTorch ships has a bug (https://github.com/NVIDIA/nccl/issues/1234). |
|
||||
| `FASTVIDEO_HCCL_SO_PATH` | str | unset | distributed | Path to the HCCL library file on Ascend NPUs. Deprecated names: `HCCL_SO_PATH`. |
|
||||
| `FASTVIDEO_WORKER_MULTIPROC_METHOD` | one of spawn, fork, forkserver | `spawn` | distributed | Multiprocessing start method for worker processes. |
|
||||
| `FASTVIDEO_ULYSSES_A2A` | one of off, auto | `off` | distributed | Sequence-parallel all-to-all backend. off uses the NCCL path in DistributedAutograd.AllToAll4D. auto uses the fused NVLink kernel when the group is a load-store accessible mesh of 2, 4, 6, or 8 ranks in eager execution, and the NCCL path otherwise. |
|
||||
| `FASTVIDEO_CONFIGURE_LOGGING` | bool | `1` | logging | Configure logging at import. When true, FastVideo uses its default logging configuration or the file in FASTVIDEO_LOGGING_CONFIG_PATH. |
|
||||
| `FASTVIDEO_LOGGING_CONFIG_PATH` | str | unset | logging | Path to a JSON logging configuration file. |
|
||||
| `FASTVIDEO_LOGGING_LEVEL` | str | `INFO` | logging | Default logging level. |
|
||||
| `FASTVIDEO_LOGGING_PREFIX` | str | `""` | logging | Prefix prepended to every log message. |
|
||||
| `FASTVIDEO_STAGE_LOGGING` | bool | `0` | logging | Log the time that each pipeline stage takes. |
|
||||
| `FASTVIDEO_ATTENTION_BACKEND` | str | unset | attention | Attention backend, as an AttentionBackendEnum name such as TORCH_SDPA, FLASH_ATTN, VIDEO_SPARSE_ATTN, SAGE_ATTN, or SAGE_ATTN_THREE. Config resolution uses it when engine.attention.backend is unset. An unsupported name raises an error. |
|
||||
| `FASTVIDEO_FA4` | bool | `0` | attention | The FLASH_ATTN backend uses FlashAttention-4 (flash_attn.cute) instead of FA3 or FA2. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FA4_PACKED_VARLEN` | bool | `0` | attention | MiniMax-H3 dense DiT self-attention uses the FlashAttention-4 packed-varlen entry point. This changes the floating-point reduction order, so it is an inference-only opt-in. |
|
||||
| `FASTVIDEO_VSA_SM100A` | bool | `0` | attention | VIDEO_SPARSE_ATTN_H3 sends no-grad tile-64 forwards to the data-center Blackwell (sm_100a) kernel. fastvideo-kernel reads the same variable with the same rule. |
|
||||
| `FASTVIDEO_NVFP4_FA4` | bool | `0` | attention | FlashAttention-4 quantizes Q and K to NVFP4. An explicit nvfp4_fa4 attention implementation argument takes precedence. |
|
||||
| `FASTVIDEO_DISABLE_ATTENTION_COMPILE` | bool | `1` | attention | Keep attention forward out of torch.compile graphs (torch.compiler.disable). Set it to 0 to let attention constructed under that setting be traced. Setting it explicitly to true also blocks regional compile. |
|
||||
| `FASTVIDEO_MLX_WINDOW` | int | `0` | attention | MLX FastWan windowed attention size in tokens. 0 uses full attention. |
|
||||
| `FASTVIDEO_MLX_WINDOW_SINK` | int | `0` | attention | Number of sink tokens that MLX windowed attention always attends to. |
|
||||
| `FASTVIDEO_INFERENCE_TORCH_COMPILE` | bool | `0` | performance | Compile each DiT transformer block with fullgraph torch.compile at inference. Same as engine.compile.regional=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE` | bool | `0` | performance | MiniMax-H3 VAE decode splits its temporal chunks across the sequence-parallel ranks instead of running serially on the output rank. Same as pipeline.model.minimax_h3.vae_parallel_decode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_ENCODE` | bool | `0` | performance | MiniMax-H3 reference-video VAE encode splits its temporal chunks across the sequence-parallel ranks. Same as pipeline.model.minimax_h3.vae_parallel_encode=True. |
|
||||
| `FASTVIDEO_VAE_PARALLEL_DECODE_STRATEGY` | str | unset | performance | Collective that moves chunks in parallel VAE decode: gather (used when unset) or all_gather. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FUSIONS` | str | `""` | performance | MiniMax-H3 inference-only Triton fusions: all, 1, or a comma-separated subset of modulate,qknorm_rope,swiglu. Empty, 0, or none keeps the eager implementation. |
|
||||
| `FASTVIDEO_FSDP2_AUTOWRAP` | bool | `0` | performance | FSDP2 shards modules by parameter count instead of the model's shard conditions. Not supported by self-forcing distillation. |
|
||||
| `FASTVIDEO_FSDP2_MIN_PARAMS` | int | `10000000` | performance | Minimum parameter count of a module that FASTVIDEO_FSDP2_AUTOWRAP shards. |
|
||||
| `FASTVIDEO_MLX_COMPILE` | bool | `0` | performance | Compile the MLX DiT forward with mx.compile. |
|
||||
| `FASTVIDEO_MLX_FAST_NORM` | bool | `0` | performance | Use MLX fast normalization kernels. |
|
||||
| `FASTVIDEO_MLX_DQ_GEMM` | str | `1` | performance | MLX dequantized GEMM for affine-quantized weights: 0 turns it off, 1 uses the measured minimum row count, and an integer sets the minimum row count. |
|
||||
| `FASTVIDEO_LTX2_VAE_CHANNELS_LAST_3D` | bool | `1` | performance | LTX-2 VAE uses the channels_last_3d memory format. |
|
||||
| `FASTVIDEO_LTX2_DISABLE_AUDIO_AUTOCAST` | bool | `1` | performance | LTX-2 audio decoding runs without CUDA autocast. Deprecated names: `LTX2_DISABLE_AUDIO_AUTOCAST`. |
|
||||
| `FASTVIDEO_FLUX2_DISABLE_BF16_REDUCED_PRECISION_REDUCTION` | bool | `0` | performance | Flux denoising disables reduced-precision reductions in bf16 matmuls, which tightens accumulation for the 4-step Klein model. |
|
||||
| `FASTVIDEO_FFMPEG_BIN` | str | `ffmpeg` | output | ffmpeg executable used to save video with audio. |
|
||||
| `FASTVIDEO_VIDEO_CODEC` | str | `libx264` | output | ffmpeg video codec for saved videos. |
|
||||
| `FASTVIDEO_NVENC_PRESET` | str | `p1` | output | NVENC preset when the codec is an \*_nvenc codec. |
|
||||
| `FASTVIDEO_NVENC_TUNE` | str | `ull` | output | NVENC tune option. |
|
||||
| `FASTVIDEO_NVENC_RC` | str | `constqp` | output | NVENC rate-control mode. |
|
||||
| `FASTVIDEO_NVENC_QP` | str | `28` | output | NVENC quantization parameter. |
|
||||
| `FASTVIDEO_NVENC_BF` | str | `0` | output | NVENC number of B-frames. |
|
||||
| `FASTVIDEO_X264_PRESET` | str | `ultrafast` | output | x264 preset for non-NVENC codecs. |
|
||||
| `FASTVIDEO_OUTPUT_PIX_FMT` | str | `yuv420p` | output | ffmpeg pixel format for saved videos. |
|
||||
| `FASTVIDEO_NVTX_PROFILE` | bool | `0` | profiling | Emit NVTX ranges for external profilers such as Nsight Systems. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_DIR` | path | unset | profiling | Enables the torch profiler and sets the directory for its traces. Must be an absolute path. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_RECORD_SHAPES` | bool | `0` | profiling | Torch profiler records shapes. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_PROFILE_MEMORY` | bool | `0` | profiling | Torch profiler profiles memory. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_STACK` | bool | `0` | profiling | Torch profiler captures stacks. Costs about 1.5x runtime and 1.4x trace size. |
|
||||
| `FASTVIDEO_TORCH_PROFILER_WITH_FLOPS` | bool | `0` | profiling | Torch profiler profiles FLOPs. |
|
||||
| `FASTVIDEO_TORCH_PROFILE_REGIONS` | str | `""` | profiling | Comma-separated profiler regions to record. The torch profiler requires at least one region. |
|
||||
| `FASTVIDEO_TRACE_ACTIVATIONS` | bool | `0` | debug | Enable activation trace hooks. |
|
||||
| `FASTVIDEO_TRACE_LAYERS` | str | `""` | debug | Regex filter for traced module names. Empty means all. |
|
||||
| `FASTVIDEO_TRACE_STATS` | str | `abs_mean,sum` | debug | Comma-separated activation statistics dumped for each output tensor. |
|
||||
| `FASTVIDEO_TRACE_OUTPUT` | str | `/tmp/fv_trace_<pid>.jsonl` | debug | JSONL path for activation traces. The literal <pid> is replaced at runtime. |
|
||||
| `FASTVIDEO_TRACE_STEPS` | str | `""` | debug | Comma-separated denoising step indices. Empty means all. |
|
||||
| `FASTVIDEO_H3_VSA_PROBE` | str | unset | debug | Output directory for the VSA-H3 attention-mass probe, which writes one .pt file per step, layer, and rank. Keeps the model out of regional compile. |
|
||||
| `FASTVIDEO_LTX2_GEMMA_LOG` | str | `""` | debug | Log file for LTX-2 Gemma text-encoder hidden states, used by parity tests. Deprecated names: `LTX2_FASTVIDEO_GEMMA_LOG`. |
|
||||
| `FASTVIDEO_COSMOS25_LOG_KNOBS` | bool | `0` | debug | Log the Cosmos 2.5 latent-preparation conditioning inputs. |
|
||||
| `FASTVIDEO_CFG_GATE_STEP` | float | `1.0` | sampling | CFG gating fraction in [0, 1]. Steps before len(timesteps) \* X run the conditional and unconditional forwards; later steps reuse the cached difference. 1.0 disables gating. |
|
||||
| `FASTVIDEO_LTX2_USE_DISTILLED_SIGMAS` | bool | `1` | sampling | LTX-2 uses the distilled sigma schedule when pipeline.model.ltx2.use_distilled_sigmas is also true. Deprecated names: `LTX2_USE_DISTILLED_SIGMAS`. |
|
||||
| `FASTVIDEO_EVAL_CACHE` | path | computed | eval | Cache directory for evaluation models and datasets. Defaults to $FASTVIDEO_CACHE_ROOT/eval. |
|
||||
| `FASTVIDEO_PHYSICS_IQ_BUCKET_URL` | str | `https://storage.googleapis.com/physics-iq-benchmark` | eval | Base URL of the Physics-IQ benchmark bucket. |
|
||||
| `FASTVIDEO_VBENCH_FULL_INFO_JSON` | str | unset | eval | Path to VBench_full_info.json, used instead of the vendored copy. Deprecated names: `VBENCH_FULL_INFO_JSON`. |
|
||||
| `FASTVIDEO_FVD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the FVD metric. |
|
||||
| `FASTVIDEO_FAD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the audio Frechet distance metric. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_DATA_DIR` | str | `data/cats` | test | Raw data directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_CAPTION_JSON` | str | `videos2caption_1_sample.json` | test | Caption file, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_CAPTION_JSON`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_VIDEO_SUBDIR` | str | `video` | test | Video subdirectory, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_VIDEO_SUBDIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_OUTPUT_DIR` | str | `data/ltx2_overfit_preprocessed` | test | Output directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_OUTPUT_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_MODEL` | str | `FastVideo/LTX2-Distilled-Diffusers` | test | Model repository whose encoders preprocess_ltx2_overfit.py uses. Deprecated names: `LTX2_OVERFIT_MODEL`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_NUM_COPIES` | int | `4` | test | Number of copies of the overfit sample in the parquet file. Deprecated names: `LTX2_OVERFIT_NUM_COPIES`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_DATA_DIR` | str | `data/kandinsky5_overfit` | test | Raw data directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_OUTPUT_DIR` | str | `data/kandinsky5_overfit_preprocessed` | test | Output directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_OUTPUT_DIR`. |
|
||||
| `FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO` | str | `FastVideo/ssim-reference-videos` | test | Hugging Face repository that holds the SSIM reference videos. Deprecated names: `FASTVIDEO_SSIM_REFERENCE_HF_REPO`. |
|
||||
| `FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO_TYPE` | str | `dataset` | test | Repository type of FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO. Deprecated names: `FASTVIDEO_SSIM_REFERENCE_HF_REPO_TYPE`. |
|
||||
| `FASTVIDEO_TEST_SSIM_SKIP_REFERENCE_DOWNLOAD` | bool | `0` | test | SSIM tests use local reference videos without downloading. Deprecated names: `FASTVIDEO_SSIM_SKIP_REFERENCE_DOWNLOAD`. |
|
||||
| `FASTVIDEO_TEST_SSIM_FULL_QUALITY` | bool | `0` | test | SSIM tests use the full-quality sampling configurations. Deprecated names: `FASTVIDEO_SSIM_FULL_QUALITY`. |
|
||||
| `FASTVIDEO_TEST_NIGHTLY` | bool | `0` | test | Run the nightly end-to-end overfit tests. Deprecated names: `FASTVIDEO_NIGHTLY`. |
|
||||
| `FASTVIDEO_TEST_ULYSSES_FAULT_RANK` | str | unset | test | Rank that fails in the Ulysses fault-injection test. The test sets it for its worker processes. Deprecated names: `FASTVIDEO_ULYSSES_FAULT_RANK`. |
|
||||
| `FASTVIDEO_TEST_ULYSSES_FAULT_STAGE` | str | unset | test | Stage that fails in the Ulysses fault-injection test. The test sets it for its worker processes. Deprecated names: `FASTVIDEO_ULYSSES_FAULT_STAGE`. |
|
||||
| `FASTVIDEO_TEST_GOLDEN_GATE_DIR` | str | unset | test | Local directory of golden-gate reference tensors. Deprecated names: `FASTVIDEO_GOLDEN_GATE_DIR`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ALLOW_LOW_MEMORY` | bool | `0` | test | Run the MLX Wan2.2 5B real-weights parity test on hosts with little memory. Deprecated names: `FASTVIDEO_WAN22_5B_ALLOW_LOW_MEMORY`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ROOT` | str | unset | test | Local Wan2.2 5B checkpoint for the MLX real-weights parity test. Deprecated names: `FASTVIDEO_WAN22_5B_ROOT`. |
|
||||
| `FASTVIDEO_TEST_GRADNORM_UPDATE` | bool | `0` | test | Gradient-norm regression tests update their references. Deprecated names: `FASTVIDEO_GRADNORM_UPDATE`. |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_SSIM_MODEL_PATH` | str | `FastVideo/DreamX-World-5B-Cam-Diffusers` | test | Model for the DreamX-World camera SSIM test. Deprecated names: `DREAMX_WORLD_SSIM_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_AR_SSIM_MODEL_PATH` | str | `FastVideo/DreamX-World-5B-Diffusers` | test | Model for the DreamX-World autoregressive SSIM test. Deprecated names: `DREAMX_WORLD_AR_SSIM_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_FLUX_T2I_MODEL_DIR` | str | `black-forest-labs/FLUX.1-dev` | test | Model for the Flux text-to-image SSIM test. Deprecated names: `FLUX_T2I_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_FLUX_TRANSFORMER_PATH` | str | unset | test | Local Flux transformer for the Flux transformer test. Deprecated names: `FLUX_TRANSFORMER_PATH`. |
|
||||
| `FASTVIDEO_TEST_GAMECRAFT_MODEL_PATH` | str | `FastVideo/HunyuanGameCraft-Diffusers` | test | Model for the HunyuanGameCraft SSIM test. Deprecated names: `GAMECRAFT_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_MODEL_PATH` | str | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | test | Model for the GEN3C SSIM test. Deprecated names: `GEN3C_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_IMAGE_PATH` | str | unset | test | Input image for the GEN3C SSIM test. Deprecated names: `GEN3C_TEST_IMAGE_PATH`. |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_LOCAL_WEIGHTS_DIR` | str | unset | test | Local official GLM-Image weights for the GLM-Image SSIM test. Deprecated names: `GLM_IMAGE_LOCAL_WEIGHTS_DIR`. |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_MODEL_DIR` | str | unset | test | Model for the GLM-Image SSIM test. Deprecated names: `GLM_IMAGE_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_NUM_GPUS` | int | `1` | test | GPUs for the Kandinsky5 nightly end-to-end overfit test. Deprecated names: `KANDINSKY5_E2E_NUM_GPUS`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_WRITE_REFERENCE` | bool | `0` | test | The Kandinsky5 nightly end-to-end test writes a missing reference video. Deprecated names: `KANDINSKY5_E2E_WRITE_REFERENCE`. |
|
||||
| `FASTVIDEO_TEST_LONGCAT_MODEL_ROOT` | str | unset | test | Local LongCat-Video checkpoint for the golden-gate test. Deprecated names: `LONGCAT_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_GOLDEN_DIR` | str | unset | test | Local directory of MiniMax-H3 golden-gate tensors. Deprecated names: `MINIMAX_H3_GATE_GOLDEN_DIR`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_LAYER` | int | `0` | test | Transformer layer that the MiniMax-H3 golden-gate test checks. Deprecated names: `MINIMAX_H3_GATE_LAYER`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_MODEL_ROOT` | str | unset | test | Local MiniMax-H3 checkpoint for the golden-gate test. Deprecated names: `MINIMAX_H3_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_SD35_MODEL_DIR` | str | `stabilityai/stable-diffusion-3.5-medium` | test | Model for the Stable Diffusion 3.5 SSIM test. Deprecated names: `SD35_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_TAEH3_REFERENCE_DIR` | str | unset | test | Upstream taehv checkout for the MLX TAEH3 parity test. Deprecated names: `TAEH3_REFERENCE_DIR`. |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_DIR` | str | `Tongyi-MAI/Z-Image-Turbo` | test | Model for the Z-Image SSIM test. Deprecated names: `ZIMAGE_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_REVISION` | str | `f332072aa78be7aecdf3ee76d5c247082da564a6` | test | Hugging Face revision of the Z-Image model for its SSIM test. Deprecated names: `ZIMAGE_MODEL_REVISION`. |
|
||||
|
||||
Variables that FastVideo no longer reads; setting one logs a warning:
|
||||
|
||||
| Deprecated variable | Reason |
|
||||
| ----------------------------------------- | ---------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
<!-- END GENERATED ENV TABLE -->
|
||||
@@ -112,7 +112,7 @@ per-metric policy with direction, percent threshold, absolute threshold, and a
|
||||
|
||||
`test_inference_performance.py` temporarily sets `FASTVIDEO_STAGE_LOGGING=1`
|
||||
while it runs so pipeline stage execution times are available in
|
||||
`generate_video(...).logging_info`. Stage logs use pipeline-unique keys such as
|
||||
`generate(...).logging_info`. Stage logs use pipeline-unique keys such as
|
||||
`prompt_encoding_stage` so duplicate stage classes do not collide. For
|
||||
`PipelineStage` entries, shared component stage bases emit a stable
|
||||
`component_metric`: text encoding stages map to `text_encoder_time_s`,
|
||||
|
||||
@@ -10,6 +10,7 @@ slash-command mappings, and workflow ownership live in
|
||||
|---|---|---|
|
||||
| Unit tests | `fastvideo/tests/api`, `fastvideo/tests/dataset`, `fastvideo/tests/entrypoints`, `fastvideo/tests/workflow`, CPU-safe `fastvideo/tests/train` subsets | Validate individual functions, APIs, contracts, and lightweight workflows. |
|
||||
| Component tests | `fastvideo/tests/encoders`, `fastvideo/tests/transformers`, `fastvideo/tests/vaes` | Validate loading and basic behavior for model components. |
|
||||
| Golden gates | `fastvideo/tests/golden_gate` | Compare small, deterministic component outputs exactly against device/runtime-matched reference tensors. |
|
||||
| Train framework tests | `fastvideo/tests/train/models`, `fastvideo/tests/train/methods` | Exercise the new `fastvideo/train/` framework on real checkpoints and tiny synthetic batches. |
|
||||
| SSIM tests | `fastvideo/tests/ssim` | Compare generated videos against references to catch visual regressions. |
|
||||
| Training tests | `fastvideo/tests/training` | Validate legacy training loops, LoRA, distillation, self-forcing, and VSA behavior. |
|
||||
@@ -20,14 +21,74 @@ slash-command mappings, and workflow ownership live in
|
||||
|
||||
## Running Tests Locally
|
||||
|
||||
Run the narrowest useful suite while iterating:
|
||||
Start with the cheapest checks that cover the changed behavior, and stop on a
|
||||
failure before starting heavier dependent checks:
|
||||
|
||||
1. Run pre-commit on changed files and focused import, config, and contract tests.
|
||||
2. Run the smallest matching component golden gate. Prefer one GPU, tiny fixed
|
||||
inputs, cached component weights, and direct tensor comparisons over renders.
|
||||
3. Run focused component parity or default SSIM when the golden does not cover
|
||||
the changed behavior, such as VAE normalization or pipeline wiring.
|
||||
4. Run full-quality renders or broad suites when required by the change, an
|
||||
explicit request, or CI policy, rather than on every edit.
|
||||
|
||||
A golden must cover the component being changed. Wan has four small gates:
|
||||
|
||||
| Gate | Boundary |
|
||||
|---|---|
|
||||
| `test_wan_t2v.py` | Dense transformer block 0 |
|
||||
| `test_wan_vae.py` | FP32 encode, BF16 decode, streaming, and cache reset |
|
||||
| `test_wan_causal.py` | Real block weights, cache append/rewrite, and sink eviction |
|
||||
| `test_wan_denoising.py` | Three real 1.3B DiT/UniPC steps with fixed prompt embeddings |
|
||||
|
||||
These use an immutable Wan checkpoint revision and only download the required
|
||||
component. The trajectory gate needs no tokenizer, text encoder, VAE, or video
|
||||
reference. Weight-free stage tests also exercise every step of a 50-step UniPC
|
||||
loop, CFG caching, expert switching, conditioning layouts, and DMD RNG order.
|
||||
These checks do not replace independent Diffusers parity or end-to-end SSIM.
|
||||
|
||||
Use the golden's matching GPU, dtype, backend, and runtime. A missing reference
|
||||
or environment mismatch is not a pass. Do not replace a reference with the
|
||||
candidate output just to clear a failure. For a relocation, the unchanged
|
||||
parent is the baseline; two imports of the same class are not numerical proof.
|
||||
|
||||
The VAE and transformer CI lanes run their Wan component goldens first.
|
||||
Selected integration lanes wait for the golden lane in merge/full builds.
|
||||
See [CI/CD Architecture](ci_architecture.md) for direct-rerun and skip semantics.
|
||||
|
||||
For Wan, one command enforces the local ordering and stops on failure:
|
||||
|
||||
The initial contracts include `tests/api/test_wan_definitions.py`: all registered
|
||||
Wan aliases, local-manifest detector precedence, sampling/precision defaults,
|
||||
config isolation, and legacy serialized-config compatibility. These require no
|
||||
weights or Hub access; the current package import still needs its prepared
|
||||
runtime environment.
|
||||
|
||||
```bash
|
||||
pytest tests/
|
||||
pytest fastvideo/tests/ -v
|
||||
pytest fastvideo/tests/encoders -vs
|
||||
pytest fastvideo/tests/transformers -vs
|
||||
pytest fastvideo/tests/vaes -vs
|
||||
bash scripts/validate_wan.sh vae # contracts, then the VAE golden
|
||||
bash scripts/validate_wan.sh dense parity # contracts, goldens, Diffusers parity
|
||||
bash scripts/validate_wan.sh all default # then focused T2V/I2V/causal SSIM
|
||||
```
|
||||
|
||||
The second argument is an upper validation tier, not a reference override.
|
||||
Supply the GPU/backend/runtime and SSIM model/tier settings that match the
|
||||
reference. The script never updates references. Causal component coverage is
|
||||
a block-cache fingerprint, not independent full causal-pipeline parity.
|
||||
|
||||
New named-tensor gates write a missing output to `*.candidate.pt` and fail;
|
||||
running candidate code again cannot turn it into an approved reference. Seed
|
||||
from unchanged, pushed source, verify two independent processes bit-for-bit,
|
||||
then review and publish only the new reference files. Preserve the baseline
|
||||
source SHA, test recipe SHA, checkpoint revision, runtime, and comparison
|
||||
receipt with the run artifacts. Never overwrite an existing video reference
|
||||
as a side effect of adding a tensor gate.
|
||||
|
||||
Examples of focused checks:
|
||||
|
||||
```bash
|
||||
pytest fastvideo/tests/loader/test_wan_family_imports.py -q
|
||||
pytest fastvideo/tests/golden_gate/test_wan_t2v.py -q
|
||||
pytest fastvideo/tests/vaes/test_wan_vae.py -q
|
||||
```
|
||||
|
||||
GPU-heavy suites need the right hardware, credentials, local caches, and
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# Cosmos recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/nvidia.webp" alt="NVIDIA" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>Cosmos inference recipes</h2>
|
||||
<p>NVIDIA Cosmos Predict 2.5 generates navigable world videos. The maintained example runs the 2B text-to-world checkpoint on a single GPU.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading Cosmos recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/nvidia">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>World-generation prompts work best describing a scene and camera motion; the built-in prompt in the example is a known-good starting point.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -0,0 +1,141 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# FLUX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/black-forest-labs.webp" alt="Black Forest Labs" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>FLUX inference recipes</h2>
|
||||
<p>Black Forest Labs' FLUX family covers FLUX.1 dev and FLUX.2 (dev and distilled Klein) text-to-image. FLUX.1 registers no `model_family` in the registry and is grouped under FLUX for documentation only.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading FLUX recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/black-forest-labs">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>FLUX.1 dev defaults to a local <code>official_weights/FLUX.1-dev</code> directory in the example; the cookbook command passes the Hugging Face ID explicitly instead.</li>
|
||||
<li>Image outputs land under <code>outputs/</code>; adjust <code>--output</code> if that path is not writable.</li>
|
||||
<li>Gated checkpoints (FLUX.1 dev): run <code>huggingface-cli login</code> and accept the license on Hugging Face first.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -0,0 +1,140 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# GLM-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/zai.webp" alt="Z.ai" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>GLM-Image inference recipes</h2>
|
||||
<p>GLM-Image from Z.ai supports both text-to-image generation and instruction-based image editing, each with a maintained example.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading GLM-Image recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/zai-org">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The editing example reads <code>assets/images/couple.jpg</code> from the repository root, so run it from a repo checkout rather than an arbitrary working directory.</li>
|
||||
<li>Output paths default under <code>image_output/</code>; pass <code>--output</code> to change them.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -0,0 +1,141 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# Hunyuan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/tencent-hunyuan.webp" alt="Tencent Hunyuan" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>Hunyuan inference recipes</h2>
|
||||
<p>HunyuanVideo 1.5 is Tencent's video generation family. The maintained examples cover 480p text-to-video with CPU offload and a full 480p-to-1080p upscale chain.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading Hunyuan recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/Tencent-Hunyuan">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Out of memory: <code>basic_hy15.py</code> already enables dit/VAE/text-encoder CPU offload; further tradeoffs are described in <a href="../../inference/optimizations/">Optimizations</a>.</li>
|
||||
<li><code>pin_cpu_memory</code> errors on low-RAM machines are documented inline in the example; set it to false as the source comment suggests.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -1,44 +1,464 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# Inference Cookbook
|
||||
|
||||
Choose a complete recipe maintained in the FastVideo repository. Each command
|
||||
runs its checked-in source directly, so coupled model, GPU, offload, and
|
||||
attention settings do not drift into unsupported combinations.
|
||||
<div class="cookbook-shell cookbook-catalog" data-cookbook-gallery>
|
||||
<header class="cookbook-hero">
|
||||
<p class="cookbook-eyebrow">FastVideo inference cookbook</p>
|
||||
<h2>Choose by output, then by family.</h2>
|
||||
<p class="cookbook-hero__lede">
|
||||
Open a family to pick a maintained recipe and a runtime FastVideo
|
||||
actually supports. Every command runs a checked-in source, so the model,
|
||||
platform, offload, and attention settings stay tied to that example.
|
||||
Cards below group video, image, audio, and interactive world models.
|
||||
Mode chips on each card are the workloads FastVideo maintains for that
|
||||
family, not a promise that every flag works on every runtime.
|
||||
</p>
|
||||
<a class="cookbook-inline-link" href="../inference/support_matrix/">
|
||||
View the full support matrix <span aria-hidden="true">→</span>
|
||||
</a>
|
||||
<a class="cookbook-inline-link" href="./openai-api/">Run FastH3 with a playground and API <span aria-hidden="true">→</span></a>
|
||||
</header>
|
||||
|
||||
The commands expect a local clone:
|
||||
<section class="cookbook-section" id="video-models" aria-labelledby="video-models-heading">
|
||||
<div class="cookbook-section__heading">
|
||||
<h2 id="video-models-heading">Video</h2>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready cookbook-family-tile--featured" href="./minimax-h3/" aria-label="Open MiniMax H3 recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/minimax.webp" alt="" width="132" height="132" loading="eager">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>MiniMax H3</strong><small>Video + stereo audio</small></span>
|
||||
<span class="cookbook-count">9 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2VA</li>
|
||||
<li>FL2VA</li>
|
||||
<li>Ref2VA</li>
|
||||
<li>MLX T2VA</li>
|
||||
<li>DGX Spark</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
```bash
|
||||
git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo
|
||||
```
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./wan/" aria-label="Open Wan recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/wan-ai.webp" alt="" width="132" height="132" loading="eager">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Wan</strong><small>FastWan CUDA and FastMetal MLX</small></span>
|
||||
<span class="cookbook-count">7 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
<li>TI2V</li>
|
||||
<li>MLX T2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<div class="cookbook-picker" data-cookbook data-recipes="../assets/cookbook-recipes.json">
|
||||
<label for="cookbook-recipe"><strong>Recipe</strong></label>
|
||||
<select id="cookbook-recipe" data-cookbook-recipe disabled>
|
||||
<option>Loading recipes…</option>
|
||||
</select>
|
||||
<dl>
|
||||
<dt>Model</dt>
|
||||
<dd data-cookbook-model>Loading…</dd>
|
||||
<dt>Source</dt>
|
||||
<dd><a data-cookbook-source href="../inference/examples/basic/">Browse maintained examples</a></dd>
|
||||
</dl>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading…</code></pre>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
<noscript>
|
||||
JavaScript is needed for the recipe picker. Browse the
|
||||
<a href="../inference/examples/examples_inference_index/">inference examples</a>
|
||||
instead.
|
||||
</noscript>
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./ltx/" aria-label="Open LTX recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/ltx.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>LTX</strong><small>Video with synchronized audio</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./hunyuan/" aria-label="Open Hunyuan recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tencent-hunyuan.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Hunyuan</strong><small>480p and 1080p upscale</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./kandinsky5/" aria-label="Open Kandinsky 5 recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/kandinsky.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Kandinsky 5</strong><small>Text and image to video</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./longcat/" aria-label="Open LongCat recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/meituan-longcat.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>LongCat</strong><small>T2V, I2V, optional refine</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./turbodiffusion/" aria-label="Open TurboDiffusion recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">Turbo</span>
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>TurboDiffusion</strong><small>Accelerated Wan profiles</small></span>
|
||||
<span class="cookbook-count">3 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="cookbook-section" id="image-models" aria-labelledby="image-models-heading">
|
||||
<div class="cookbook-section__heading">
|
||||
<h2 id="image-models-heading">Image</h2>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./flux/" aria-label="Open FLUX recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/black-forest-labs.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>FLUX</strong><small>FLUX.1 and FLUX.2</small></span>
|
||||
<span class="cookbook-count">3 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./glm-image/" aria-label="Open GLM-Image recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/zai.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GLM-Image</strong><small>Generate and edit</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
<li>Edit</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./z-image/" aria-label="Open Z-Image recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tongyi.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Z-Image</strong><small>Turbo text to image</small></span>
|
||||
<span class="cookbook-count">1 recipe</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./stable-diffusion/" aria-label="Open Stable Diffusion recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/stabilityai.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Stable Diffusion</strong><small>SD 3.5 Medium</small></span>
|
||||
<span class="cookbook-count">1 recipe</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="cookbook-section" id="audio-models" aria-labelledby="audio-models-heading">
|
||||
<div class="cookbook-section__heading">
|
||||
<h2 id="audio-models-heading">Audio</h2>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./stable-audio/" aria-label="Open Stable Audio recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/stabilityai.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Stable Audio</strong><small>Open 1.0 and Small</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2A</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./mmaudio/" aria-label="Open MMAudio recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/fastvideo.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>MMAudio</strong><small>Video or text to audio</small></span>
|
||||
<span class="cookbook-count">1 recipe</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>V2A</li>
|
||||
<li>T2A</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="cookbook-section" id="world-models" aria-labelledby="world-models-heading">
|
||||
<div class="cookbook-section__heading">
|
||||
<h2 id="world-models-heading">World and interactive</h2>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./cosmos/" aria-label="Open Cosmos recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/nvidia.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Cosmos</strong><small>Text to world video</small></span>
|
||||
<span class="cookbook-count">1 recipe</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2W</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./matrix-game/" aria-label="Open Matrix Game recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/fastvideo.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Matrix Game</strong><small>Image-conditioned worlds</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>I2W</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
</div>
|
||||
</section>
|
||||
|
||||
<section class="cookbook-section" id="planned-families" aria-labelledby="planned-families-heading">
|
||||
<div class="cookbook-section__heading">
|
||||
<h2 id="planned-families-heading">Pages still to write</h2>
|
||||
<p>These families already have runnable examples. The cookbook page is not ready, so the cards are not links.</p>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="GameCraft cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tencent-hunyuan.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GameCraft</strong><small>Game world generation</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="GEN3C cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/nvidia.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GEN3C</strong><small>Novel-view video</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="HY-World cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">HY</span>
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>HY-World</strong><small>Interactive world play</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="DreamX cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/fastvideo.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>DreamX</strong><small>World generation</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="LingBot cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">LB</span>
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>LingBot</strong><small>Video and world models</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
## Customize a recipe
|
||||
|
||||
Start from the checked-in source, then change only the settings your model
|
||||
supports:
|
||||
|
||||
- [Configuration](../inference/configuration.md) covers the Python and CLI
|
||||
config surfaces.
|
||||
- [Optimizations](../inference/optimizations.md) covers attention backends,
|
||||
compilation, and memory tradeoffs.
|
||||
- [Support matrix](../inference/support_matrix.md) lists supported models and
|
||||
optimizations.
|
||||
<small class="cookbook-logo-credit">
|
||||
Catalog marks come from the official model publishers' Hugging Face
|
||||
organizations; typographic tiles are placeholders, never invented logos. See
|
||||
<a href="https://github.com/hao-ai-lab/FastVideo/blob/main/docs/assets/logos/SOURCES.md">docs/assets/logos/SOURCES.md</a>.
|
||||
</small>
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# Kandinsky 5 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/kandinsky.webp" alt="Kandinsky Lab" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>Kandinsky 5 inference recipes</h2>
|
||||
<p>Kandinsky 5.0 from the Kandinsky Lab covers text-to-video and image-to-video in Lite and Pro variants, including distilled checkpoints for faster sampling.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading Kandinsky 5 recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/kandinskylab">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Alternative Lite/Pro checkpoints are listed inline in the maintained examples; swap the model string only after checking its Hugging Face card.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>Both recipes map to checked-in FastVideo examples and recorded single-GPU B200 runs. The image-to-video run also records 10,365.89 MB peak GPU memory. These measurements describe the recorded runs; they are not minimum hardware requirements.</p>
|
||||
</div>
|
||||
</details>
|
||||