Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0ef47caf86 | ||
|
|
e71d01648c |
@@ -68,7 +68,7 @@ Copy `templates/component_parity_test.py` and fill every `TODO` marker. The
|
||||
template is distilled from:
|
||||
|
||||
- `tests/local_tests/transformers/test_ltx2.py`
|
||||
- `tests/local_tests/gen3c/test_gen3c.py`
|
||||
- `tests/local_tests/transformers/test_gamecraft_parity.py`
|
||||
- `tests/local_tests/encoders/test_ltx2_gemma_parity.py`
|
||||
- `tests/local_tests/vaes/test_oobleck_vae_parity.py`
|
||||
- `tests/local_tests/sd35/test_sd35_component_parity.py`
|
||||
|
||||
@@ -87,7 +87,7 @@ def _load_official_model(device: torch.device, dtype: torch.dtype) -> torch.nn.M
|
||||
# TODO: import official class/factory and load real weights strictly.
|
||||
# Examples in-tree:
|
||||
# - LTX2: SingleGPUModelBuilder(...).build(device=device, dtype=dtype)
|
||||
# - GEN3C: torch.load(...)["state_dict"] -> official_model.load_state_dict(...)
|
||||
# - GameCraft: torch.load(...)["module"] -> official_model.load_state_dict(...)
|
||||
# - Oobleck: create_model_from_config(config) + ckpt state_dict
|
||||
OfficialClass = _import_or_skip(OFFICIAL_MODULE, OFFICIAL_CLASS)
|
||||
model = OfficialClass() # TODO: pass official config kwargs.
|
||||
|
||||
@@ -52,8 +52,8 @@ scaling constants, dtype casts, state-dict names, and every output head.
|
||||
- Loader path: `TransformerLoader` reads `transformer/config.json`, calls
|
||||
`dit_config.update_model_arch(config)`, resolves `_class_name` through
|
||||
`ModelRegistry`, and constructs the class with `config` and `hf_config`.
|
||||
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, and
|
||||
`ltx2.py`.
|
||||
- Reference examples: `stable_audio.py`, `wanvideo.py`, `sd3.py`, `longcat.py`,
|
||||
and `ltx2.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
@@ -51,7 +51,7 @@ posterior behavior, encode/decode output objects, tiling flags, and cropping.
|
||||
- Loader path: VAE loaders resolve `_class_name` through `ModelRegistry` and
|
||||
load converted component weights from the VAE subdir.
|
||||
- Reference examples: `oobleck.py`, `autoencoder_kl.py`, `wanvae.py`,
|
||||
`ltx2vae.py`, and `hunyuanvae.py`.
|
||||
`ltx2vae.py`, and `gamecraftvae.py`.
|
||||
- Layer guidance: `fastvideo/layers/AGENTS.md`.
|
||||
|
||||
## Implementation Rules
|
||||
|
||||
@@ -43,12 +43,11 @@ from `../add-model/contracts/conversion_request.md`.
|
||||
- `scripts/checkpoint_conversion/stable_audio_to_diffusers.py`: monolithic
|
||||
`model.safetensors` split into transformer/VAE/conditioner, plus copied
|
||||
passthrough subfolders. Use this shape for single-checkpoint official repos.
|
||||
- `scripts/checkpoint_conversion/convert_mmaudio_to_diffusers.py`: separate
|
||||
official sources for transformer, VAE, encoders, and vocoder, assembled under
|
||||
a root `model_index.json`.
|
||||
- `scripts/checkpoint_conversion/convert_flux2_klein.py`: fused QKV split,
|
||||
renamed native transformer weights, and copied passthrough text encoder,
|
||||
tokenizer, and scheduler components.
|
||||
- `scripts/checkpoint_conversion/convert_gamecraft_full.py`: separate official
|
||||
sources for transformer, VAE, text encoders, tokenizers, scheduler, and root
|
||||
`model_index.json`.
|
||||
- `scripts/checkpoint_conversion/longcat_to_fastvideo.py`: fused QKV/KV split,
|
||||
renamed native transformer weights, and copied existing Diffusers components.
|
||||
- `scripts/checkpoint_conversion/pt_to_safetensors.py`: simple `.pt` extraction
|
||||
helper for nested checkpoint dictionaries.
|
||||
|
||||
|
||||
@@ -247,7 +247,7 @@ setup gap, not a pass.
|
||||
- `fastvideo/configs/pipelines/stable_audio.py` and
|
||||
`fastvideo/pipelines/basic/stable_audio/presets.py` for config/preset shape.
|
||||
- `fastvideo/registry.py` for `register_configs(...)` and preset registration.
|
||||
- `tests/local_tests/pipelines/test_lingbot_video_pipeline_parity.py` for latent
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for latent
|
||||
parity structure.
|
||||
- `tests/local_tests/pipelines/test_stable_audio_pipeline_parity.py` for audio
|
||||
parity structure.
|
||||
|
||||
@@ -413,7 +413,7 @@ matching `*secret*`.
|
||||
- `fastvideo/pipelines/basic/wan/` for standard T2V/I2V/DMD/Causal variants.
|
||||
- `fastvideo/pipelines/basic/ltx2/` for non-standard stages and audio/video
|
||||
patterns.
|
||||
- `tests/local_tests/pipelines/test_lingbot_video_pipeline_parity.py` for pipeline
|
||||
- `tests/local_tests/pipelines/test_gamecraft_pipeline_parity.py` for pipeline
|
||||
parity shape.
|
||||
- `tests/local_tests/transformers/test_ltx2.py`,
|
||||
`tests/local_tests/vaes/test_ltx2_vae.py`, and
|
||||
|
||||
@@ -38,10 +38,7 @@ on a summary here.
|
||||
name to `DEPRECATED_VARIABLES`. Update the uses in `examples/`,
|
||||
`scripts/`, `docs/`, `apps/`, and the tests.
|
||||
2. **Read the variable with `envs.NAME.get()` inside a function.**
|
||||
- In tests, change the value with `envs.NAME.override(value)`, and a variable
|
||||
outside the registry with `envs.override_external(name, value)`; the
|
||||
`env_overrides` fixture keeps either until the end of the test.
|
||||
- Name a variable that only tests read `FASTVIDEO_TEST_*`.
|
||||
- In tests, change the value with `envs.NAME.override(value)`.
|
||||
- Do not call `os.environ`, `os.getenv`, or `monkeypatch.setenv` for a
|
||||
FastVideo variable.
|
||||
- To set a variable that another tool reads, call `envs.set_external`,
|
||||
|
||||
@@ -57,7 +57,7 @@ Hardcoded:
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
`FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO`).
|
||||
`FASTVIDEO_SSIM_REFERENCE_HF_REPO`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
@@ -105,8 +105,8 @@ Detect artefact type by inspecting the file's imports / helper call:
|
||||
- **pixel** (`.mp4`) — file imports
|
||||
`run_text_to_video_similarity_test` / `run_image_to_video_similarity_test`
|
||||
from `fastvideo.tests.ssim.inference_similarity_utils`, OR uses the
|
||||
legacy custom-inline helper pattern (see `test_gen3c`). Default to pixel
|
||||
when both heuristics fail.
|
||||
legacy custom-inline helper pattern (see `test_gamecraft`,
|
||||
`test_longcat`, etc.). Default to pixel when both heuristics fail.
|
||||
|
||||
Record `ARTEFACT_TYPE ∈ {pixel, latent}` for use in step 4. Steps 2, 3, 5,
|
||||
and 6 are artefact-type-agnostic — `_iter_reference_files`,
|
||||
|
||||
+90
-93
@@ -23,100 +23,7 @@ notify:
|
||||
# dispatcher, and every test payload executes inside the Slinky Slurm tray.
|
||||
# fastvideo/tests/modal remains available only for an explicit manual rollback;
|
||||
# no active pipeline or slash-command route invokes it.
|
||||
# Buildkite hands jobs to free agents in the order they appear here. Golden-gate comes first
|
||||
# because every later merge lane waits for it; the fastcheck lanes follow from longest to
|
||||
# shortest measured runtime, so the longest lane never starts last and stretches the build.
|
||||
steps:
|
||||
- label: ":test_tube: Golden-Gate Tests"
|
||||
key: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,golden-gate,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "golden_gate" || build.env("TEST_TYPE") == "golden_gate_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "golden_gate_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Unit Tests"
|
||||
key: "unit"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,unit,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "unit_test" || build.env("TEST_TYPE") == "unit_test_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-unit"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "unit_test_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Kernel Tests"
|
||||
key: "kernel-tests"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,kernel-tests,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "kernel_tests" || build.env("TEST_TYPE") == "kernel_tests_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "kernel_tests_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
key: "dreamverse"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,dreamverse,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "dreamverse_app" || build.env("TEST_TYPE") == "dreamverse_app_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "dreamverse_app_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Encoder Tests"
|
||||
key: "encoder"
|
||||
if: |
|
||||
@@ -186,6 +93,96 @@ steps:
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Kernel Tests"
|
||||
key: "kernel-tests"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,kernel-tests,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "kernel_tests" || build.env("TEST_TYPE") == "kernel_tests_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "kernel_tests_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: Unit Tests"
|
||||
key: "unit"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,unit,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "unit_test" || build.env("TEST_TYPE") == "unit_test_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-unit"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "unit_test_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
key: "dreamverse"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,dreamverse,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "dreamverse_app" || build.env("TEST_TYPE") == "dreamverse_app_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "dreamverse_app_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Golden-Gate Tests"
|
||||
key: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,golden-gate,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "golden_gate" || build.env("TEST_TYPE") == "golden_gate_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "golden_gate_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
|
||||
@@ -22,6 +22,12 @@ else
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
# Alternate GPU backends compare against references without publishing records.
|
||||
# Their worker has read-only Hub credentials; publication is an operator task.
|
||||
if [ "${FASTVIDEO_CI_LOCAL_ONLY:-0}" = 1 ]; then
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
nvidia-smi \
|
||||
--query-gpu=index,timestamp,clocks.sm,clocks.max.sm,power.draw,power.limit,temperature.gpu \
|
||||
--format=csv -l 10 > "$PERF_REPORTS_DIR/gpu_telemetry.csv" 2>/dev/null &
|
||||
|
||||
@@ -15,12 +15,8 @@ exec pytest \
|
||||
./fastvideo/tests/ops/ \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/inference/test_basic_fasth3_omniref_pdd.py \
|
||||
./fastvideo/tests/attention/test_sdpa_metadata_mask_contract.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_tile_grad_safety.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_metadata.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_ref2va_regions.py \
|
||||
./fastvideo/tests/layers/test_pdd_linear.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
|
||||
@@ -12,7 +12,6 @@ from __future__ import annotations
|
||||
import argparse
|
||||
import fnmatch
|
||||
import re
|
||||
from collections.abc import Iterable
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import TextIO
|
||||
@@ -131,6 +130,11 @@ class FamilyCoverage:
|
||||
|
||||
|
||||
FAMILY_COVERAGE = (
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])dreamx(_world)?([/_.-]|$)"),
|
||||
("test_dreamx.py", ),
|
||||
("test_dreamx_world_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])flux[_-]?2([/_.-]|$)"),
|
||||
("test_flux2_klein.py", ),
|
||||
@@ -141,11 +145,21 @@ FAMILY_COVERAGE = (
|
||||
("test_flux.py", ),
|
||||
("test_flux_t2i_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])(hunyuan)?gamecraft([/_.-]|$)"),
|
||||
("test_gamecraft.py", ),
|
||||
("test_gamecraft_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])gen3c([/_.-]|$)"),
|
||||
("test_gen3c.py", ),
|
||||
("test_gen3c_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])glm[_-]?image([/_.-]|$)"),
|
||||
("test_glm_image.py", ),
|
||||
("test_glm_image_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])kandinsky[_-]?5([/_.-]|$)"),
|
||||
("test_kandinsky5.py", ),
|
||||
@@ -156,6 +170,11 @@ FAMILY_COVERAGE = (
|
||||
("test_lingbot.py", ),
|
||||
("test_lingbot_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])longcat([/_.-]|$)"),
|
||||
("test_longcat.py", ),
|
||||
("test_longcat_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])ltx[_-]?2([/_.-]|$)"),
|
||||
("test_ltx2.py", ),
|
||||
@@ -186,6 +205,11 @@ FAMILY_COVERAGE = (
|
||||
("test_stable_audio.py", ),
|
||||
("test_stable_audio_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])turbo(diffusion)?([/_.-]|$)"),
|
||||
(),
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
@@ -195,6 +219,11 @@ FAMILY_COVERAGE = (
|
||||
"test_wan_t2v_similarity.py",
|
||||
),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])z[_-]?image([/_.-]|$)"),
|
||||
("test_zimage.py", ),
|
||||
("test_zimage_similarity.py", ),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@@ -290,22 +319,16 @@ def _select_output_coverage(plan: MergePlan, path: str) -> None:
|
||||
plan.add_ssim(SSIM_SMOKE_TESTS, reason=f"shared output SSIM smoke coverage: {path}")
|
||||
|
||||
|
||||
def _normalize_path(raw_path: str) -> str:
|
||||
path = raw_path.strip()
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
return path
|
||||
|
||||
|
||||
def classify_paths(paths: list[str], removed_paths: Iterable[str] = ()) -> MergePlan:
|
||||
"""Plan merge lanes for ``paths``.
|
||||
|
||||
``removed_paths`` lists changed paths that no longer exist at the PR head
|
||||
(deleted files and rename sources); removed golden/SSIM tests are not run.
|
||||
"""
|
||||
def classify_paths(paths: list[str]) -> MergePlan:
|
||||
plan = MergePlan()
|
||||
normalized_paths = sorted({path for path in map(_normalize_path, paths) if path})
|
||||
removed = {path for path in map(_normalize_path, removed_paths) if path}
|
||||
normalized_paths: list[str] = []
|
||||
for raw_path in paths:
|
||||
path = raw_path.strip()
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
if path:
|
||||
normalized_paths.append(path)
|
||||
normalized_paths = sorted(set(normalized_paths))
|
||||
if not normalized_paths:
|
||||
plan.require_all("changed-file list was empty; failing closed")
|
||||
return plan
|
||||
@@ -345,10 +368,7 @@ def classify_paths(paths: list[str], removed_paths: Iterable[str] = ()) -> Merge
|
||||
if path.startswith("fastvideo/tests/golden_gate/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
if path in removed:
|
||||
plan.reasons.append(f"removed golden test has nothing to run: {path}")
|
||||
else:
|
||||
plan.add_golden((name, ), reason=f"changed golden test: {path}")
|
||||
plan.add_golden((name, ), reason=f"changed golden test: {path}")
|
||||
elif name in {"AGENTS.md", "README.md"}:
|
||||
plan.reasons.append(f"golden documentation only: {path}")
|
||||
else:
|
||||
@@ -359,10 +379,7 @@ def classify_paths(paths: list[str], removed_paths: Iterable[str] = ()) -> Merge
|
||||
if path.startswith("fastvideo/tests/ssim/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
if path in removed:
|
||||
plan.reasons.append(f"removed SSIM test has nothing to run: {path}")
|
||||
else:
|
||||
plan.add_ssim((name, ), reason=f"changed SSIM test: {path}")
|
||||
plan.add_ssim((name, ), reason=f"changed SSIM test: {path}")
|
||||
elif path.endswith((".py", ".json", ".pt", ".png", ".mp4")):
|
||||
plan.ssim_all = True
|
||||
plan.add_lanes("ssim", reason=f"shared SSIM harness/reference: {path}")
|
||||
@@ -538,7 +555,6 @@ def _write_summary(output: TextIO, plan: MergePlan) -> None:
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--paths-file", type=Path, required=True)
|
||||
parser.add_argument("--removed-paths-file", type=Path)
|
||||
parser.add_argument("--github-output", type=Path)
|
||||
parser.add_argument("--summary-file", type=Path)
|
||||
return parser.parse_args()
|
||||
@@ -547,9 +563,7 @@ def parse_args() -> argparse.Namespace:
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
paths = args.paths_file.read_text(encoding="utf-8").splitlines()
|
||||
removed_paths = (args.removed_paths_file.read_text(encoding="utf-8").splitlines()
|
||||
if args.removed_paths_file else [])
|
||||
plan = classify_paths(paths, removed_paths)
|
||||
plan = classify_paths(paths)
|
||||
print(f"MERGE_TEST_PLAN={plan.encoded_lanes()}")
|
||||
print(f"MERGE_GOLDEN_TESTS={plan.encoded_golden_tests()}")
|
||||
print(f"MERGE_SSIM_TESTS={plan.encoded_ssim_tests()}")
|
||||
|
||||
@@ -11,6 +11,7 @@ jobs:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
&& (vars.CI_GPU_BACKEND == '' || vars.CI_GPU_BACKEND == 'slurm')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
name: Promote Selected GPU Backend Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
concurrency:
|
||||
group: gpu-ci-status-${{ github.event.sha }}-${{ vars.CI_GPU_BACKEND }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
promote:
|
||||
if: >-
|
||||
(vars.CI_GPU_BACKEND == 'modal' || vars.CI_GPU_BACKEND == 'vllm')
|
||||
&& (github.event.context == format('gpu-ci/{0}/fastcheck-passed', vars.CI_GPU_BACKEND)
|
||||
|| github.event.context == format('gpu-ci/{0}/full-suite-passed', vars.CI_GPU_BACKEND))
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
SELECTED_BACKEND: ${{ vars.CI_GPU_BACKEND }}
|
||||
steps:
|
||||
- name: Mirror the selected backend's latest suite results
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const backend = process.env.SELECTED_BACKEND;
|
||||
if (!['modal', 'vllm'].includes(backend)) {
|
||||
throw new Error('Unsupported selected GPU backend');
|
||||
}
|
||||
const sha = context.payload.sha;
|
||||
// Read current state after entering the serialized workflow. A
|
||||
// delayed event must not overwrite a newer failure with success.
|
||||
const statuses = await github.paginate(github.rest.repos.listCommitStatusesForRef, {
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
for (const suffix of ['fastcheck-passed', 'full-suite-passed']) {
|
||||
const sourceContext = `gpu-ci/${backend}/${suffix}`;
|
||||
const matches = statuses.filter(status => status.context === sourceContext);
|
||||
matches.sort((a, b) =>
|
||||
Date.parse(b.updated_at) - Date.parse(a.updated_at) || b.id - a.id
|
||||
);
|
||||
const latest = matches[0];
|
||||
const state = latest ? latest.state : 'pending';
|
||||
if (!['pending', 'success', 'failure', 'error'].includes(state)) {
|
||||
throw new Error(`Unsupported status state for ${sourceContext}`);
|
||||
}
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
context: suffix,
|
||||
state,
|
||||
description: latest
|
||||
? `${backend} ${suffix}: ${state}`
|
||||
: `Waiting for ${backend} ${suffix}`,
|
||||
...(latest && latest.target_url ? {target_url: latest.target_url} : {}),
|
||||
});
|
||||
}
|
||||
@@ -33,15 +33,15 @@ jobs:
|
||||
env:
|
||||
FASTVIDEO_ATTENTION_BACKEND: TORCH_SDPA
|
||||
TOKENIZERS_PARALLELISM: "false"
|
||||
MASTER_ADDR: "127.0.0.1"
|
||||
MASTER_ADDR: localhost
|
||||
MASTER_PORT: "29513"
|
||||
GLOO_SOCKET_IFNAME: lo0
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: pip
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
@@ -49,9 +49,9 @@ jobs:
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.12.0 torchvision torchaudio
|
||||
torch==2.11.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest pytest-timeout numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
pytest numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru mlx \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
@@ -65,8 +65,8 @@ jobs:
|
||||
print("machine:", platform.machine())
|
||||
print("processor:", platform.processor())
|
||||
print("mlx default device:", mx.default_device())
|
||||
device_info = mx.metal.device_info() if mx.metal.is_available() else "metal unavailable"
|
||||
print("mlx device_info:", device_info)
|
||||
memory_size = mx.metal.device_info().get("memory_size") if mx.metal.is_available() else "metal unavailable"
|
||||
print("mlx memory_size:", memory_size)
|
||||
print("torch:", torch.__version__)
|
||||
print("torch mps available:", torch.backends.mps.is_available())
|
||||
PY
|
||||
@@ -74,7 +74,6 @@ jobs:
|
||||
- name: Run MLX smoke tests
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/mlx_runtime/tests/ \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
@@ -102,7 +101,7 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s --timeout=120 -o faulthandler_timeout=120
|
||||
-v -s -o faulthandler_timeout=120
|
||||
|
||||
# Same tests on MLX's CPU backend. Hosted macOS runners are scarce and
|
||||
# slower to schedule; this Linux job gives fast PR signal on the identical
|
||||
@@ -123,6 +122,7 @@ jobs:
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: pip
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
@@ -130,16 +130,15 @@ jobs:
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.12.0 torchvision torchaudio
|
||||
torch==2.11.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest pytest-timeout numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
pytest numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru "mlx[cpu]" \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
- name: Run MLX smoke tests (CPU backend)
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/mlx_runtime/tests/ \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
@@ -167,4 +166,4 @@ jobs:
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s --timeout=120 -o faulthandler_timeout=120
|
||||
-v -s -o faulthandler_timeout=120
|
||||
|
||||
@@ -76,8 +76,6 @@ jobs:
|
||||
set -euo pipefail
|
||||
changed_json="$RUNNER_TEMP/merge-changed-files.json"
|
||||
changed_paths="$RUNNER_TEMP/merge-changed-paths.txt"
|
||||
removed_paths="$RUNNER_TEMP/merge-removed-paths.txt"
|
||||
: > "$removed_paths"
|
||||
if gh api --paginate --slurp \
|
||||
"repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}/files?per_page=100" \
|
||||
> "$changed_json"; then
|
||||
@@ -85,10 +83,6 @@ jobs:
|
||||
if [ "$observed" = "$EXPECTED_CHANGED_FILES" ]; then
|
||||
jq -r '.[][] | .filename, (.previous_filename // empty)' "$changed_json" \
|
||||
| sort -u > "$changed_paths"
|
||||
# Paths absent from the PR head, so the planner never selects a deleted test.
|
||||
jq -r '.[][] | if .status == "removed" then .filename
|
||||
elif .status == "renamed" then (.previous_filename // empty) else empty end' \
|
||||
"$changed_json" | sort -u > "$removed_paths"
|
||||
else
|
||||
echo "::warning::Changed-file API returned $observed of $EXPECTED_CHANGED_FILES paths; selecting all merge lanes."
|
||||
echo '__FASTVIDEO_CI_PLAN_ALL__' > "$changed_paths"
|
||||
@@ -104,7 +98,6 @@ jobs:
|
||||
run: |
|
||||
python3 .github/scripts/plan_merge_ci.py \
|
||||
--paths-file "$RUNNER_TEMP/merge-changed-paths.txt" \
|
||||
--removed-paths-file "$RUNNER_TEMP/merge-removed-paths.txt" \
|
||||
--github-output "$GITHUB_OUTPUT" \
|
||||
--summary-file "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
|
||||
@@ -203,7 +203,8 @@ jobs:
|
||||
docker buildx imagetools create "${TAG_ARGS[@]}" "${IMAGE_REFS[@]}"
|
||||
docker buildx imagetools inspect "${TAGS[0]}"
|
||||
|
||||
# The CI runner is ARM64 like DGX Spark, but targets sm_100 rather than sm_121.
|
||||
# The CI runner is ARM64 like DGX Spark, but targets sm_100a rather than sm_121.
|
||||
# The architecture-specific target includes the GB200 VSA CUDA extensions.
|
||||
# Publish a single-architecture variant so the self-hosted CI runner can reuse
|
||||
# the exact prebuilt kernel instead of compiling it in every job.
|
||||
build-ci-runner-image:
|
||||
@@ -219,7 +220,7 @@ jobs:
|
||||
PYTHON_VERSION=3.12
|
||||
CUDA_VERSION=13.0.0
|
||||
UV_TORCH_BACKEND=cu130
|
||||
TORCH_CUDA_ARCH_LIST=10.0
|
||||
TORCH_CUDA_ARCH_LIST=10.0a
|
||||
CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
FLASH_ATTN_WHEEL_TAG=cu130torch2.12
|
||||
FLASH_ATTN_WHEEL_RELEASE_ARM64=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.22
|
||||
|
||||
+1
-3
@@ -36,6 +36,7 @@ env
|
||||
*.log
|
||||
weights/
|
||||
logs/
|
||||
/Z-Image/
|
||||
official_weights/
|
||||
converted_weights/
|
||||
|
||||
@@ -133,7 +134,6 @@ openspec/
|
||||
fastvideo/tests/ssim/reference_videos/**
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.mp4
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.png
|
||||
fastvideo/tests/ssim/.reference_videos_download.lock
|
||||
|
||||
# Local H3 MLX kernel / exactness benches (JSON, logs, frames, videos)
|
||||
.kernel_bench/
|
||||
@@ -142,5 +142,3 @@ fastvideo/tests/ssim/.reference_videos_download.lock
|
||||
*.nvimlog
|
||||
.nvimlog
|
||||
.python-version
|
||||
scripts/benchmarks/minimax_h3_pro6000/headline_results/
|
||||
fastvideo/tests/ssim/.reference_videos_download.lock
|
||||
|
||||
@@ -9,8 +9,6 @@
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
## NEWS
|
||||
- `2026/10/06`: FastH3 V2 now runs on a single consumer machine: NVIDIA RTX 5090, RTX 4090 and RTX PRO 6000 GPUs, DGX Spark and Apple Silicon. We also release [FastH3 Trim](https://huggingface.co/FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4), an experimental pruned model that is 4.2× smaller than base H3 and runs in as little as 8 GB of GPU memory. Get the [models](https://huggingface.co/collections/FastVideo/fastvideo-fasth3) and read the [Blog](https://haoailab.com/blogs/fasth3-rtx/).
|
||||
- `2026/10/06`: FastVideo now supports [Kandinsky 6](https://x.com/kandinskylab_ai/status/2107374635218055345) from Kandinsky Lab: text- and image-to-video with synchronized audio (base and 10-step distilled pi-Flow checkpoints) plus video super-resolution up to 4x. See the [Kandinsky 6 recipes](https://haoailab.com/FastVideo/cookbook/kandinsky6/).
|
||||
- `2026/09/15`: Release [FastH3 8-Step V2](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2), an eight-forward data-free DMD2 checkpoint distilled from MiniMax-H3 with 80% Video Sparse Attention. Run it with `examples/inference/basic/basic_fasth3_8step.py` or the [FastH3 8-Step V2 recipe](https://haoailab.com/FastVideo/cookbook/minimax-h3/).
|
||||
- `2026/09/01`: FastH3 now runs locally on Apple Silicon through MLX and on NVIDIA DGX Spark through CUDA 13, including two-Spark inference. Follow the [FastH3 recipes](https://haoailab.com/FastVideo/cookbook/minimax-h3/) and read the [Blog](https://haoailab.com/blogs/fasth3-local/).
|
||||
- `2026/08/27`: [FastH3 Preview v1](https://haoailab.com/blogs/fasth3-preview/) is an open-weight 4-step sparse-distilled MiniMax-H3 model for synchronized video-and-audio generation, developed in collaboration with [Nuva Lab](https://nuvalab.ai/) and the [NVIDIA FastGen team](https://github.com/NVlabs/FastGen). Download the recommended [VSA / Data-Free weights](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree), or see the [full FastH3 collection](https://huggingface.co/collections/FastVideo/fastvideo-fasth3).
|
||||
|
||||
+132
-166
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"version": 15,
|
||||
"version": 11,
|
||||
"recipes": [
|
||||
{
|
||||
"id": "fastwan21-t2v",
|
||||
@@ -10,11 +10,6 @@
|
||||
"summary": "Generate a video in three denoising steps with the distilled FastWan2.1 1.3B checkpoint and video sparse attention.",
|
||||
"model": "FastVideo/FastWan2.1-T2V-1.3B-Diffusers",
|
||||
"source": "scripts/inference/inference_wan_VSA_DMD_1_3B.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fastwan21_1_3b.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN"
|
||||
},
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN fastvideo generate --config scripts/inference/inference_wan_VSA_DMD_1_3B.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
@@ -30,13 +25,6 @@
|
||||
"summary": "The maintained high-capacity Wan2.2 text-to-video example with CPU offload settings encoded in its checked-in Python source.",
|
||||
"model": "Wan-AI/Wan2.2-T2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_wan2_2.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_wan22_t2v_a14b.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"limitations": [
|
||||
"Generation at 1280 × 720 on two GPUs with DiT CPU offload is slow. The client examples stop polling after 30 minutes while the job keeps running on the server; retrieve it by ID or raise the deadline."
|
||||
]
|
||||
},
|
||||
"command": "python examples/inference/basic/basic_wan2_2.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
@@ -66,14 +54,6 @@
|
||||
"summary": "Use one maintained 5B checkpoint for text-to-video or add an image input to switch the same recipe to image-to-video.",
|
||||
"model": "Wan-AI/Wan2.2-TI2V-5B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_wan2_2_ti2v.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_wan22_ti2v_5b.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"task": "Text to video",
|
||||
"limitations": [
|
||||
"This server walkthrough sends a text prompt only. The server API also accepts an image reference (input_reference) for this checkpoint, but the checked-in clients do not upload one. For the cookbook's image-to-video workflow, use the Python command."
|
||||
]
|
||||
},
|
||||
"command": "python examples/inference/basic/basic_wan2_2_ti2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
@@ -161,6 +141,53 @@
|
||||
"Native MLX FastMetal T2V on 36 GB+ unified memory. Same --fast, --fast-spatial, and --refine flags as the 1.3B script."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "turbodiffusion-wan21-1-3b-t2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "TurboWan2.1 1.3B",
|
||||
"summary": "A TurboDiffusion-accelerated Wan2.1 1.3B text-to-video run from its maintained single-GPU example.",
|
||||
"model": "loayrashid/TurboWan2.1-T2V-1.3B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_turbodiffusion/",
|
||||
"related": ["turbodiffusion-wan21-14b-t2v", "turbowan22-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "turbodiffusion-wan21-14b-t2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "TurboWan2.1 14B",
|
||||
"summary": "TurboDiffusion acceleration applied to the 14B Wan2.1 text-to-video checkpoint; the checked-in source is configured for two GPUs.",
|
||||
"model": "loayrashid/TurboWan2.1-T2V-14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_14b.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_14b.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under video_samples_turbodiffusion_14B/",
|
||||
"related": ["turbodiffusion-wan21-1-3b-t2v", "turbowan22-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "turbowan22-i2v",
|
||||
"family": "turbodiffusion",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "TurboWan2.2 A14B",
|
||||
"summary": "A one-to-four-step image-to-video path using TurboDiffusion and the SLA attention backend from its maintained example.",
|
||||
"model": "loayrashid/TurboWan2.2-I2V-A14B-Diffusers",
|
||||
"source": "examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_turbodiffusion_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 2, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"related": ["turbodiffusion-wan21-14b-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "ltx2-distilled-t2v",
|
||||
"family": "ltx2",
|
||||
@@ -285,70 +312,6 @@
|
||||
"expected_artifact": "MP4 videos under video_samples_kandinsky5_i2v/",
|
||||
"related": ["kandinsky5-t2v-lite-sft"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-ti2va-base",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Text or image to video with audio",
|
||||
"label": "Kandinsky 6 Pro TI2VA",
|
||||
"summary": "Generate a five-second, 512x768 video with synchronized audio from text, or set IMAGE_PATH in the maintained example to condition on an image.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"command": "python examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 video with synchronized audio under video_samples_kandinsky6_ti2va/",
|
||||
"related": ["kandinsky6-ti2va-piflow", "kandinsky6-vsr"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-ti2va-piflow",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Distilled text or image to video with audio",
|
||||
"label": "Kandinsky 6 Pro pi-Flow",
|
||||
"summary": "Generate a five-second video with synchronized audio using the distilled pi-Flow checkpoint in ten inference steps and guidance 1.0.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"command": "KANDINSKY6_MODEL_PATH=kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers python examples/inference/basic/basic_kandinsky6_ti2va.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 video with synchronized audio under video_samples_kandinsky6_ti2va/",
|
||||
"related": ["kandinsky6-ti2va-base", "kandinsky6-vsr-distilled"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-vsr",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Video super-resolution",
|
||||
"label": "Kandinsky 6 VSR",
|
||||
"summary": "Upscale an existing clip by 2.25x with the four-step flow-matching VSR checkpoint while preserving its source audio.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_sr.py",
|
||||
"command": ": \"${INPUT_VIDEO:?Set INPUT_VIDEO to an existing video}\"\npython examples/inference/basic/basic_kandinsky6_sr.py --video-path \"$INPUT_VIDEO\" --scale 2.25",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "Upscaled MP4 under outputs_video/kandinsky6_sr/",
|
||||
"related": ["kandinsky6-vsr-distilled", "kandinsky6-ti2va-base"]
|
||||
},
|
||||
{
|
||||
"id": "kandinsky6-vsr-distilled",
|
||||
"family": "kandinsky6",
|
||||
"stage": "inference",
|
||||
"task": "Distilled video super-resolution",
|
||||
"label": "Kandinsky 6 VSR distilled",
|
||||
"summary": "Upscale an existing clip by 2.25x with the distilled two-step VSR checkpoint while preserving its source audio.",
|
||||
"model": "kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers",
|
||||
"source": "examples/inference/basic/basic_kandinsky6_sr.py",
|
||||
"command": ": \"${INPUT_VIDEO:?Set INPUT_VIDEO to an existing video}\"\npython examples/inference/basic/basic_kandinsky6_sr.py --model-path kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers --video-path \"$INPUT_VIDEO\" --scale 2.25",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"platform": "cuda", "gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "Upscaled MP4 under outputs_video/kandinsky6_sr/",
|
||||
"related": ["kandinsky6-vsr", "kandinsky6-ti2va-piflow"]
|
||||
},
|
||||
{
|
||||
"id": "flux2-klein-t2i",
|
||||
"family": "flux",
|
||||
@@ -397,6 +360,53 @@
|
||||
"expected_artifact": "PNG images under outputs/flux_dev/samples/",
|
||||
"limitations": ["FLUX.1 is loadable by ID but registers no model_family in fastvideo/registry.py; it is grouped under FLUX for documentation only."]
|
||||
},
|
||||
{
|
||||
"id": "glm-image-t2i",
|
||||
"family": "glm_image",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "GLM-Image",
|
||||
"summary": "GLM-Image text-to-image generation from its maintained example.",
|
||||
"model": "zai-org/GLM-Image",
|
||||
"source": "examples/inference/basic/basic_glm_image.py",
|
||||
"command": "python examples/inference/basic/basic_glm_image.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at image_output/landscape.png",
|
||||
"related": ["glm-image-edit"]
|
||||
},
|
||||
{
|
||||
"id": "glm-image-edit",
|
||||
"family": "glm_image",
|
||||
"stage": "inference",
|
||||
"task": "Image editing",
|
||||
"label": "GLM-Image editing",
|
||||
"summary": "Edit an input image with an instruction prompt using GLM-Image, from its maintained editing example.",
|
||||
"model": "zai-org/GLM-Image",
|
||||
"source": "examples/inference/basic/edit_glm_image.py",
|
||||
"command": "python examples/inference/basic/edit_glm_image.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at image_output/edited.png (input: assets/images/couple.jpg)",
|
||||
"related": ["glm-image-t2i"]
|
||||
},
|
||||
{
|
||||
"id": "zimage-turbo-t2i",
|
||||
"family": "zimage",
|
||||
"stage": "inference",
|
||||
"task": "Text to image",
|
||||
"label": "Z-Image Turbo",
|
||||
"summary": "Z-Image Turbo text-to-image on a single GPU from its maintained example.",
|
||||
"model": "Tongyi-MAI/Z-Image-Turbo",
|
||||
"source": "examples/inference/basic/basic_zimage.py",
|
||||
"command": "python examples/inference/basic/basic_zimage.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "PNG image at outputs/zimage/zimage_turbo.png"
|
||||
},
|
||||
{
|
||||
"id": "sd35-medium-t2i",
|
||||
"family": "sd35",
|
||||
@@ -446,8 +456,7 @@
|
||||
"source": "examples/inference/basic/basic_fasth3.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"",
|
||||
"audio": true
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile all",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
@@ -472,8 +481,7 @@
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\"",
|
||||
"audio": true
|
||||
"prepare": "hf download FastVideo/FastVideo-Minimax-FastH3-Preview-v0.2 --local-dir ./FastH3-Preview-v0.2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-Preview-v0.2/transformer --out ./FastH3-MLX --formats \"int6\""
|
||||
},
|
||||
"group": "fasth3-preview",
|
||||
"group_label": "FastH3 V1",
|
||||
@@ -520,8 +528,7 @@
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_spark.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e .",
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3",
|
||||
"audio": true
|
||||
"env": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3"
|
||||
},
|
||||
"command": "FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_fasth3_spark.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
@@ -571,77 +578,6 @@
|
||||
"Height, width, frames, and steps in the YAML are examples. Edit them or pass CLI flags. See docs/getting_started/installation/spark_pair.md."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "compacth3-rtx5090",
|
||||
"group": "compacth3-rtx5090",
|
||||
"group_label": "CompactH3 on RTX 5090",
|
||||
"group_task": "4-step 42-block text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "CompactH3 NVFP4 on RTX 5090",
|
||||
"summary": "Run the 42-block CompactH3 NVFP4 DiT with the NVFP4 Qwen3-VL encoder, Comfy int8-convrot VAE, and SageAttention3 FP4 on one 32 GB RTX 5090. Sequential load parks the encoder in pinned host RAM.",
|
||||
"model": "./CompactH3",
|
||||
"source": "examples/inference/basic/basic_compacth3_rtx5090.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_compacth3_rtx5090.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu128 uv pip install -e \".[fasth3]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree --local-dir ./CompactH3 --include model_index.json --include \"tokenizer/**\" --include \"processor/**\" --include \"scheduler/**\" --include \"audio_scheduler/**\" --include \"audio_vae/**\" --include \"vae/**\"\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-bf16 --local-dir ./CompactH3/transformer\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-nvfp4 --local-dir ./CompactH3/transformer --include nvfp4_weights.safetensors\nhf download KyleNeverGivesUp/FastH3-text-encoder-nvfp4 --local-dir ./CompactH3/text_encoder\nhf download Comfy-Org/MiniMax-H3 --local-dir ./Comfy-MiniMax-H3 --include vae/minimax_h3_video_vae_int8_convrot.safetensors\ncp ./Comfy-MiniMax-H3/vae/minimax_h3_video_vae_int8_convrot.safetensors ./CompactH3/vae/",
|
||||
"env": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a"
|
||||
},
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_compacth3_rtx5090.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/compacth3_rtx5090/",
|
||||
"modes": ["T2VA", "CompactH3 NVFP4", "RTX 5090"],
|
||||
"limitations": [
|
||||
"Assemble ./CompactH3 before running. The DiT NVFP4 export and encoder snapshot are gated; run huggingface-cli login and accept each repo license.",
|
||||
"32 GB cannot keep the NVFP4 encoder and DiT on the GPU together. Keep h3_sequential_load on and lazy_module_load off.",
|
||||
"Blackwell sm_120 needs ATTN_QAT_INFER, FLASHINFER_CUDA_ARCH_LIST=12.0a, and a CUDA 12.8 PyTorch wheel.",
|
||||
"Legal num_frames values are 17n+5, capped at 362 (15.08 s). Native 16:9 sizes include 832x480 and 1344x768; 1344x768 on 32 GB is unmeasured."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "compacth3-rtx-pro6000",
|
||||
"group": "compacth3-rtx-pro6000",
|
||||
"group_label": "CompactH3 on RTX PRO 6000",
|
||||
"group_task": "4-step 42-block text to video + audio",
|
||||
"family": "minimax_h3",
|
||||
"stage": "inference",
|
||||
"task": "Few-step text to video (with audio)",
|
||||
"label": "CompactH3 NVFP4 on RTX PRO 6000 Blackwell",
|
||||
"summary": "Run CompactH3 NVFP4 with the encoder, DiT, and int8-convrot VAE resident on one 96 GB RTX PRO 6000 Blackwell. The checked-in example is 1344x768 and 124 frames (5.17 s) with VAE torch.compile. Use 832x480 for clip-queue playground traffic.",
|
||||
"model": "./CompactH3",
|
||||
"source": "examples/inference/basic/basic_compacth3_rtx_pro6000.yaml",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_compacth3_rtx_pro6000.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu128 uv pip install -e \".[fasth3]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree --local-dir ./CompactH3 --include model_index.json --include \"tokenizer/**\" --include \"processor/**\" --include \"scheduler/**\" --include \"audio_scheduler/**\" --include \"audio_vae/**\" --include \"vae/**\"\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-bf16 --local-dir ./CompactH3/transformer\nhf download aryan5v/FastH3-20B-42block-dmd2-ckpt1400-nvfp4 --local-dir ./CompactH3/transformer --include nvfp4_weights.safetensors\nhf download KyleNeverGivesUp/FastH3-text-encoder-nvfp4 --local-dir ./CompactH3/text_encoder\nhf download Comfy-Org/MiniMax-H3 --local-dir ./Comfy-MiniMax-H3 --include vae/minimax_h3_video_vae_int8_convrot.safetensors\ncp ./Comfy-MiniMax-H3/vae/minimax_h3_video_vae_int8_convrot.safetensors ./CompactH3/vae/",
|
||||
"env": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a"
|
||||
},
|
||||
"command": "FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 fastvideo generate --config examples/inference/basic/basic_compacth3_rtx_pro6000.yaml",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {
|
||||
"platform": "cuda",
|
||||
"gpu_count": 1,
|
||||
"evidence": "source-configured"
|
||||
},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 under outputs/compacth3_rtx_pro6000/",
|
||||
"modes": ["T2VA", "CompactH3 NVFP4", "RTX PRO 6000"],
|
||||
"limitations": [
|
||||
"Assemble ./CompactH3 before running. The DiT NVFP4 export and encoder snapshot are gated; run huggingface-cli login and accept each repo license.",
|
||||
"96 GB keeps the encoder, DiT, and VAE on GPU. Do not enable h3_sequential_load or lazy_module_load on this box.",
|
||||
"Enable compile.vae_enabled. Leave inference_torch_compile off: FlashInfer and Sage3 custom ops cannot be compiled.",
|
||||
"Blackwell sm_120 needs ATTN_QAT_INFER, FLASHINFER_CUDA_ARCH_LIST=12.0a, and a CUDA 12.8 PyTorch wheel.",
|
||||
"Legal num_frames values are 17n+5, capped at 362 (15.08 s). Native 16:9 sizes include 832x480 and 1344x768. Dense CompactH3 has zero VSA gates; the pipeline raises if VIDEO_SPARSE_ATTN_H3 is loaded with all-zero to_gate_compress weights."
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "fasth3-8step-v2-cuda",
|
||||
"group": "fasth3-8step-v2",
|
||||
@@ -656,8 +592,7 @@
|
||||
"source": "examples/inference/basic/basic_fasth3_8step.py",
|
||||
"serving": {
|
||||
"source": "examples/serving/openai_fasth3_8step.yaml",
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"",
|
||||
"audio": true
|
||||
"install": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\""
|
||||
},
|
||||
"command": "UV_TORCH_BACKEND=cu130 uv pip install -e \".[fasth3]\"\npython examples/inference/basic/basic_fasth3_8step.py --prompt \"(S1) A presenter says <d>[English] FastVideo runs FastH3.</d>\" --profile strict --no-inference-torch-compile --no-compile-vae",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
@@ -686,8 +621,7 @@
|
||||
"serving": {
|
||||
"source": "examples/serving/mlx_fasth3_8step.yaml",
|
||||
"install": "uv pip install -e \".[mlx]\"",
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa",
|
||||
"audio": true
|
||||
"prepare": "hf download FastVideo/FastVideo-FastH3-8-Step-V2 --local-dir ./FastH3-8-Step-V2\npython scripts/checkpoint_conversion/convert_minimax_h3_mlx.py --model-root ./FastH3-8-Step-V2/transformer --out ./FastH3-8-Step-V2-MLX --formats \"int8\" --include-vsa"
|
||||
},
|
||||
"group": "fasth3-8step-v2",
|
||||
"group_label": "FastH3 V2",
|
||||
@@ -788,6 +722,38 @@
|
||||
],
|
||||
"limitations": ["Supply a compatible FastH3 adapter. The script infers dense or VSA attention from the adapter payload unless you override it."]
|
||||
},
|
||||
{
|
||||
"id": "longcat-t2v",
|
||||
"family": "longcat",
|
||||
"stage": "inference",
|
||||
"task": "Text to video",
|
||||
"label": "LongCat Video T2V",
|
||||
"summary": "LongCat Video text-to-video at 480p (50 steps), with distilled and 720p refinement passes included in the same maintained script.",
|
||||
"model": "FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
"source": "examples/inference/basic/basic_longcat_t2v.py",
|
||||
"command": "python examples/inference/basic/basic_longcat_t2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video/longcat_t2v_basic/, longcat_t2v_distill/, and longcat_t2v_refine_720p/",
|
||||
"related": ["longcat-i2v"]
|
||||
},
|
||||
{
|
||||
"id": "longcat-i2v",
|
||||
"family": "longcat",
|
||||
"stage": "inference",
|
||||
"task": "Image to video",
|
||||
"label": "LongCat Video I2V",
|
||||
"summary": "LongCat Video image-to-video with optional distilled and refinement passes, from its maintained example.",
|
||||
"model": "FastVideo/LongCat-Video-I2V-Diffusers",
|
||||
"source": "examples/inference/basic/basic_longcat_i2v.py",
|
||||
"command": "python examples/inference/basic/basic_longcat_i2v.py",
|
||||
"gpu_types": ["NVIDIA"],
|
||||
"hardware": {"gpu_count": 1, "evidence": "source-configured"},
|
||||
"evidence": "Source-backed",
|
||||
"expected_artifact": "MP4 videos under outputs_video/longcat_i2v_basic/ and longcat_i2v_distill/",
|
||||
"related": ["longcat-t2v"]
|
||||
},
|
||||
{
|
||||
"id": "stable-audio-open-t2a",
|
||||
"family": "stable_audio",
|
||||
|
||||
+12
-31
@@ -506,10 +506,6 @@
|
||||
const runtime = runtimeFor(recipe);
|
||||
const profile = servingPanel && servingProfiles[recipe.id];
|
||||
const useServer = Boolean(profile && usagePreference === "server");
|
||||
// Only some servers expose the browser playground or return audio; both come from the serving profile.
|
||||
const hasPlayground = Boolean(profile && profile.playground_url);
|
||||
const hasAudio = Boolean(profile && profile.audio);
|
||||
const playgroundAny = familyRecipes.some((item) => servingProfiles[item.id] && servingProfiles[item.id].playground_url);
|
||||
// The measured local profile and the server config have separate evidence.
|
||||
const activeRecipe = useServer ? { ...recipe, hardware: profile.hardware, evidence: "Source-backed" } : recipe;
|
||||
const knobs = knobsFor(recipe);
|
||||
@@ -519,19 +515,14 @@
|
||||
usage.querySelectorAll("[data-cookbook-mode]").forEach((option) => {
|
||||
const selected = option.dataset.cookbookMode === (useServer ? "server" : "python");
|
||||
option.disabled = option.dataset.cookbookMode === "server" && !profile;
|
||||
const hint = option.querySelector("[data-cookbook-server-hint]");
|
||||
if (hint) hint.textContent = (profile ? hasPlayground : playgroundAny) ? "Playground, cURL, or an API client" : "cURL or an API client";
|
||||
option.classList.toggle("cookbook-option--selected", selected);
|
||||
option.setAttribute("aria-pressed", String(selected));
|
||||
});
|
||||
const servedNames = [...new Set(familyRecipes.filter((item) => servingProfiles[item.id]).map((item) => item.group_label || item.label))];
|
||||
servingAvailability.textContent = profile
|
||||
? `${hasPlayground ? "The playground" : "cURL"} and the OpenAI Python client share one server process. Both workflows can run on your own machine.`
|
||||
? "The playground and the OpenAI Python client share one server process. Both workflows can run on your own machine."
|
||||
: servingLoadFailed
|
||||
? "Server examples could not be loaded. Use Python directly."
|
||||
: servedNames.length
|
||||
? `This recipe uses Python directly. ${new Intl.ListFormat("en").format(servedNames)} can also run a local server for ${playgroundAny ? "the playground and " : "cURL and "}the OpenAI Python client.`
|
||||
: "This recipe uses Python directly.";
|
||||
? "Server examples could not be loaded. Open the H3 server guide below, or use Python directly."
|
||||
: "This recipe uses Python directly. FastH3 V1 and FastH3 V2 can also run a local server for the playground and the OpenAI Python client.";
|
||||
servingPanel.hidden = !useServer;
|
||||
commandBlock.hidden = useServer;
|
||||
root.querySelector("[data-cookbook-python-note]").hidden = useServer;
|
||||
@@ -540,12 +531,11 @@
|
||||
if (useServer) {
|
||||
const isMLX = profile.runtime === "mlx";
|
||||
const isSpark = runtime.id === "spark";
|
||||
const where = hasPlayground ? "in the playground or your app" : "in your app";
|
||||
servingPanel.querySelector("[data-cookbook-server-lifetime]").textContent = isMLX
|
||||
? `Start once, then change prompts ${where}. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident.`
|
||||
? "Start once, then change prompts in the playground or your app. MLX reuses its pipeline and prompt cache, but loads and releases model components between phases to limit unified-memory use. It does not keep all weights resident."
|
||||
: isSpark
|
||||
? `Start once, then change prompts ${where}. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache.`
|
||||
: `Start once, then change prompts ${where}. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.`;
|
||||
? "Start once, then change prompts in the playground or your app. On a DGX Spark, lazy module load still reloads Qwen3-VL and the DiT between phases of each request, so later prompts are not a free hot cache."
|
||||
: "Start once, then change prompts in the playground or your app. CUDA requests reuse the loaded model. The Python SDK can also reuse a generator within one process.";
|
||||
servingPanel.querySelector("[data-cookbook-install-guide]").href = isMLX
|
||||
? "../../getting_started/installation/mlx/"
|
||||
: isSpark
|
||||
@@ -556,10 +546,7 @@
|
||||
servingPanel.querySelector("[data-cookbook-server-install]").textContent = profile.install;
|
||||
servingPanel.querySelector("[data-cookbook-server-command]").textContent = profile.command;
|
||||
servingPanel.querySelector("[data-cookbook-health-command]").textContent = profile.health_command;
|
||||
servingPanel.querySelectorAll("[data-cookbook-playground-only]").forEach((element) => {
|
||||
element.hidden = !hasPlayground;
|
||||
});
|
||||
if (hasPlayground) servingPanel.querySelector("[data-cookbook-playground]").href = profile.playground_url;
|
||||
servingPanel.querySelector("[data-cookbook-playground]").href = profile.playground_url;
|
||||
const client = profile.clients[selectedClient];
|
||||
const filename = client.source.split("/").pop();
|
||||
servingPanel.querySelector("[data-cookbook-client-install]").textContent = client.install;
|
||||
@@ -587,17 +574,13 @@
|
||||
});
|
||||
|
||||
description.textContent = useServer
|
||||
? `${recipe.group_label || recipe.label} generates video${hasAudio ? " with audio" : ""}. Start the local server, then use ${hasPlayground ? "the playground or " : "cURL or "}the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
? `${recipe.group_label || recipe.label} generates video with audio. Start the local server, then use the playground or the OpenAI Python client. This profile uses the checked-in ${runtime.label} configuration.`
|
||||
: recipe.summary;
|
||||
label.textContent = useServer ? `${recipe.group_label || recipe.label} · Server` : recipe.label;
|
||||
model.textContent = recipe.model;
|
||||
task.textContent = (useServer && recipe.serving && recipe.serving.task) || recipe.task;
|
||||
task.textContent = recipe.task;
|
||||
hardwareValue.textContent = runtimeSummary(activeRecipe);
|
||||
if (artifact) {
|
||||
artifact.textContent = useServer
|
||||
? (hasAudio ? "MP4 with audio" : "MP4 video")
|
||||
: recipe.expected_artifact || "Not yet documented for this recipe.";
|
||||
}
|
||||
if (artifact) artifact.textContent = useServer ? "MP4 with audio" : recipe.expected_artifact || "Not yet documented for this recipe.";
|
||||
if (evidenceCell) {
|
||||
evidenceCell.textContent = activeRecipe.evidence || "Source-backed";
|
||||
evidenceCell.classList.toggle("cookbook-badge--verified", activeRecipe.evidence === "Verified");
|
||||
@@ -619,12 +602,10 @@
|
||||
? "This MLX server config has no recorded hardware run. Measurements from the Python recipe are not server memory requirements. Only text-to-video/audio is wired; reference inputs and fast modes are not exposed here."
|
||||
: runtime.id === "spark"
|
||||
? "This Spark server config has no recorded serving benchmark. Lazy module load reloads Qwen3-VL and the DiT between phases of each request. Compilation of the DiT is disabled."
|
||||
: `This server config has no recorded serving benchmark.${profile.compile_enabled === false ? " Compilation is disabled, unlike the measured Python performance profile." : ""}`,
|
||||
: "This server config has no recorded serving benchmark. Compilation is disabled, unlike the measured Python performance profile.",
|
||||
`${profile.sampling.width} × ${profile.sampling.height} · ${profile.sampling.num_frames} frames · ${profile.sampling.fps} fps. The server supplies these defaults; the client sends the model and prompt.`,
|
||||
"Generation is serialized. Job metadata is held in memory and is lost when the server restarts.",
|
||||
] : recipe.limitations || []),
|
||||
...(useServer && recipe.serving && Array.isArray(recipe.serving.limitations) ? recipe.serving.limitations : []),
|
||||
...knobCaveats];
|
||||
] : recipe.limitations || []), ...knobCaveats];
|
||||
notes.replaceChildren();
|
||||
notes.hidden = limitations.length === 0;
|
||||
if (limitations.length) {
|
||||
|
||||
@@ -1515,8 +1515,7 @@ img {
|
||||
|
||||
.cookbook-serving[hidden],
|
||||
.cookbook-command[hidden],
|
||||
[data-cookbook-python-note][hidden],
|
||||
[data-cookbook-playground-only][hidden] {
|
||||
[data-cookbook-python-note][hidden] {
|
||||
display: none;
|
||||
}
|
||||
|
||||
|
||||
@@ -9,15 +9,18 @@ plain typographic tile instead — that tile is a UI placeholder, not a logo.
|
||||
| --- | --- | --- |
|
||||
| `wan-ai.webp` | [Official Wan-AI Hugging Face organization avatar](https://huggingface.co/Wan-AI) | Wan family card and page header |
|
||||
| `ltx.webp` | [Official Lightricks Hugging Face organization avatar](https://huggingface.co/Lightricks) | LTX family card and page header |
|
||||
| `tencent-hunyuan.webp` | [Official Tencent Hunyuan Hugging Face organization avatar](https://huggingface.co/Tencent-Hunyuan) | Hunyuan family card and page header |
|
||||
| `tencent-hunyuan.webp` | [Official Tencent Hunyuan Hugging Face organization avatar](https://huggingface.co/Tencent-Hunyuan) | Hunyuan and GameCraft cards, Hunyuan page header |
|
||||
| `nvidia.webp` | [Official NVIDIA Hugging Face organization avatar](https://huggingface.co/nvidia) | Cosmos and GEN3C cards, Cosmos page header |
|
||||
| `kandinsky.webp` | [Official Kandinsky Lab Hugging Face organization avatar](https://huggingface.co/kandinskylab) | Kandinsky 5 and Kandinsky 6 cards and page headers |
|
||||
| `kandinsky.webp` | [Official Kandinsky Lab Hugging Face organization avatar](https://huggingface.co/kandinskylab) | Kandinsky 5 family card and page header |
|
||||
| `black-forest-labs.webp` | [Official Black Forest Labs Hugging Face organization avatar](https://huggingface.co/black-forest-labs) | FLUX family card and page header |
|
||||
| `minimax.webp` | [Official MiniMax Hugging Face organization avatar](https://huggingface.co/MiniMaxAI) | MiniMax H3 family card and page header |
|
||||
| `tongyi.webp` | [Official Tongyi MAI Hugging Face organization avatar](https://huggingface.co/Tongyi-MAI) | Z-Image family card and page header |
|
||||
| `zai.webp` | [Official Z.ai Hugging Face organization avatar](https://huggingface.co/zai-org) | GLM-Image family card and page header |
|
||||
| `stabilityai.webp` | [Official Stability AI Hugging Face organization avatar](https://huggingface.co/stabilityai) | Stable Diffusion and Stable Audio cards and page headers |
|
||||
| `fastvideo.webp` | [Official FastVideo Hugging Face organization avatar](https://huggingface.co/FastVideo) | Matrix Game and MMAudio cards (converted weights published by this org) |
|
||||
| `meituan-longcat.webp` | [Official Meituan LongCat Hugging Face organization avatar](https://huggingface.co/meituan-longcat) | LongCat family card and page header |
|
||||
| `fastvideo.webp` | [Official FastVideo Hugging Face organization avatar](https://huggingface.co/FastVideo) | Matrix Game and MMAudio cards (converted weights published by this org), DreamX card |
|
||||
|
||||
Typographic tiles (no vendored image): HY-World ("HY"), LingBot ("LB"),
|
||||
MMAudio page header ("MMA"). These publishers have no
|
||||
Typographic tiles (no vendored image): TurboDiffusion ("Turbo"), HY-World
|
||||
("HY"), LingBot ("LB"), MMAudio page header ("MMA"). These publishers have no
|
||||
single official mark appropriate for reuse in the catalog; add a licensed
|
||||
asset here if one becomes available.
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 5.1 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 6.0 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 2.4 KiB |
@@ -4,6 +4,11 @@ This is the canonical reference for FastVideo's CI/CD system. Contributor-facing
|
||||
PR steps live in [Pull Requests](pull_requests.md), and test-authoring guidance
|
||||
lives in [Testing](testing.md).
|
||||
|
||||
The existing Slurm route below remains the default. Operators can also install
|
||||
the [selectable GPU dispatcher](gpu_ci_backends.md) to run the same lane scripts
|
||||
on Modal or Kubernetes in the `vllm` namespace. That opt-in path enforces two
|
||||
active PRs, four GPUs per PR, and eight total, with separate backend statuses.
|
||||
|
||||
## Overview
|
||||
|
||||
FastVideo splits validation across GitHub Actions, Buildkite, Slinky Slurm,
|
||||
@@ -435,8 +440,11 @@ before later jobs consume the updated image pin.
|
||||
The same workflow publishes a single-architecture ARM64, CUDA 13, SM100 image
|
||||
for the self-hosted CI runner under the
|
||||
`py3.12-cuda13.0.0-sm100-{latest,sha-*}` tags. It carries the matching prebuilt
|
||||
kernel wheel so runner jobs can validate and install the exact source and ABI
|
||||
match instead of recompiling it in every lane.
|
||||
kernel wheel compiled with `TORCH_CUDA_ARCH_LIST=10.0a` to include the GB200
|
||||
VSA CUDA extensions. Runtime kernel detection uses the same target, so runner
|
||||
jobs can validate and install the exact source and ABI match instead of
|
||||
recompiling it in every lane. Older artifacts built for `10.0` have a different
|
||||
cache key and trigger a local rebuild when the worker detects `10.0a`.
|
||||
|
||||
The optional Dreamverse matrix builds backend and UI images for CUDA 12.6 and
|
||||
CUDA 13 on `amd64`. Dreamverse remains `amd64`-only because its FA4 dependency
|
||||
|
||||
+15
-104
@@ -13,15 +13,12 @@ test in the same pull request.
|
||||
them directly with `os.environ.get("NAME")`, and the name must be in the external-variable allowlist
|
||||
(`EXTERNAL_ALLOWLIST` in the contract test). When FastVideo sets such a variable for the other tool, it calls
|
||||
`envs.set_external`, `envs.setdefault_external`, or `envs.unset_external`, and the name must be in
|
||||
`EXTERNAL_WRITE_ALLOWLIST`. Variables that FastVideo's CI and CI tooling define (for example `TEST_SCOPE` and
|
||||
`PERF_RUN_SOURCE`) keep their names, and test code under `fastvideo/tests/` reads them directly; they are listed
|
||||
in `CI_ONLY_VARIABLES` in the contract test, together with the file that sets each one.
|
||||
`EXTERNAL_WRITE_ALLOWLIST`.
|
||||
2. **Read with `envs.NAME.get()`, write with `envs.NAME.set()`, and change a value in tests with
|
||||
`envs.NAME.override()`.** Each type has one parsing rule. A value that the rule rejects raises
|
||||
`fastvideo.envs.EnvVarError` instead of falling back to the default.
|
||||
3. **Name FastVideo variables with the `FASTVIDEO_` prefix.** The second word states the purpose where one applies:
|
||||
`ENABLE_`, `DISABLE_`, `USE_`, `FORCE_`, `DEBUG_`, `TEST_`. Variables that only tests read use
|
||||
`FASTVIDEO_TEST_`, for example `FASTVIDEO_TEST_SD35_MODEL_DIR`, and the category `test`.
|
||||
`ENABLE_`, `DISABLE_`, `USE_`, `FORCE_`, `DEBUG_`, `TEST_`.
|
||||
4. **Keep a renamed variable as a deprecated alias until the next minor release.** Setting the old name logs a
|
||||
warning. Delete a variable that no code reads, and list it in `DEPRECATED_VARIABLES` so that setting it logs a
|
||||
warning.
|
||||
@@ -32,8 +29,7 @@ test in the same pull request.
|
||||
effect without re-importing a module. Module level, class bodies, decorators, and default argument values run at
|
||||
import time.
|
||||
7. **Do not write the environment to pass values between parts of FastVideo.** Pass an argument instead. Tests use
|
||||
`envs.NAME.override()`, and `envs.override_external()` for variables outside the registry; both restore the
|
||||
previous value.
|
||||
`envs.NAME.override()`.
|
||||
|
||||
## Field types
|
||||
|
||||
@@ -91,21 +87,6 @@ with envs.FASTVIDEO_DEBUG_MY_STAGE.override(True):
|
||||
run_stage()
|
||||
```
|
||||
|
||||
For a variable outside the registry, such as `MASTER_PORT` or `TEST_SCOPE`, use `envs.override_external`. To keep
|
||||
an override until the end of a test, enter it through the `env_overrides` fixture from `fastvideo/tests/conftest.py`,
|
||||
which restores every value at teardown:
|
||||
|
||||
```python
|
||||
def test_my_stage(env_overrides):
|
||||
env_overrides.enter_context(envs.FASTVIDEO_DEBUG_MY_STAGE.override(True))
|
||||
env_overrides.enter_context(envs.override_external("MASTER_PORT", "29512"))
|
||||
run_stage()
|
||||
```
|
||||
|
||||
In test code under `fastvideo/tests/`, `override_external` accepts any name that code may read directly
|
||||
(`EXTERNAL_ALLOWLIST`, `CI_ONLY_VARIABLES`) or that is in `EXTERNAL_WRITE_ALLOWLIST`. Library code may write only
|
||||
the names in `EXTERNAL_WRITE_ALLOWLIST`.
|
||||
|
||||
## Rename or remove a variable
|
||||
|
||||
To rename a variable, declare it under the new name and list the old name in `deprecated_names`:
|
||||
@@ -140,11 +121,9 @@ It reports each violation as `<path>: <kind> <name>`:
|
||||
| | with a name built at runtime (`<dynamic>`) | another tool owns, add it to |
|
||||
| | | `EXTERNAL_ALLOWLIST` with a reason. |
|
||||
| `write` | `os.environ[...] = ...`, `setdefault`, `pop`, `del`, | Pass an argument instead. In tests, use |
|
||||
| | `os.putenv`, `os.unsetenv`, `monkeypatch.setenv`/`delenv`, | `envs.NAME.override()`, or |
|
||||
| | or an `envs.*_external` call with a name that the | `envs.override_external()` for a variable |
|
||||
| | helper does not accept | outside the registry. For a variable that |
|
||||
| | | another tool reads, call an |
|
||||
| | | `envs.*_external` helper and add the name to |
|
||||
| | `os.putenv`, `os.unsetenv`, `monkeypatch.setenv`/`delenv`, | `envs.NAME.override()`. For a variable that |
|
||||
| | or an `envs.*_external` call with a name outside | another tool reads, call an |
|
||||
| | `EXTERNAL_WRITE_ALLOWLIST` | `envs.*_external` helper and add the name to |
|
||||
| | | `EXTERNAL_WRITE_ALLOWLIST` with a reason. |
|
||||
| `whole-environ` | `os.environ.copy()`, `dict(os.environ)`, iteration, | Read the specific variables that the code |
|
||||
| | `mock.patch.dict(os.environ, ...)`, `os.environ.update` | needs. |
|
||||
@@ -192,7 +171,6 @@ longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
| `FASTVIDEO_FA4` | bool | `0` | attention | The FLASH_ATTN backend uses FlashAttention-4 (flash_attn.cute) instead of FA3 or FA2. |
|
||||
| `FASTVIDEO_MINIMAX_H3_FA4_PACKED_VARLEN` | bool | `0` | attention | MiniMax-H3 dense DiT self-attention uses the FlashAttention-4 packed-varlen entry point. This changes the floating-point reduction order, so it is an inference-only opt-in. |
|
||||
| `FASTVIDEO_VSA_SM100A` | bool | `0` | attention | VIDEO_SPARSE_ATTN_H3 sends no-grad tile-64 forwards to the data-center Blackwell (sm_100a) kernel. fastvideo-kernel reads the same variable with the same rule. |
|
||||
| `FASTVIDEO_VSA_TRITON` | bool | `0` | attention | Force the Triton MiniMax-H3 sparse attention kernel. fastvideo-kernel reads the same variable. |
|
||||
| `FASTVIDEO_NVFP4_FA4` | bool | `0` | attention | FlashAttention-4 quantizes Q and K to NVFP4. An explicit nvfp4_fa4 attention implementation argument takes precedence. |
|
||||
| `FASTVIDEO_DISABLE_ATTENTION_COMPILE` | bool | `1` | attention | Keep attention forward out of torch.compile graphs (torch.compiler.disable). Set it to 0 to let attention constructed under that setting be traced. Setting it explicitly to true also blocks regional compile. |
|
||||
| `FASTVIDEO_MLX_WINDOW` | int | `0` | attention | MLX FastWan windowed attention size in tokens. 0 uses full attention. |
|
||||
@@ -241,35 +219,6 @@ longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
| `FASTVIDEO_VBENCH_FULL_INFO_JSON` | str | unset | eval | Path to VBench_full_info.json, used instead of the vendored copy. Deprecated names: `VBENCH_FULL_INFO_JSON`. |
|
||||
| `FASTVIDEO_FVD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the FVD metric. |
|
||||
| `FASTVIDEO_FAD_REF_FEATURES` | str | unset | eval | Cached reference-feature file for the audio Frechet distance metric. |
|
||||
| `FASTVIDEO_H3_VSA_FP4` | bool | `0` | attention | Run MiniMax-H3 VSA attention on the block-sparse SageAttention3 FP4 kernel (sm_120, no-grad, single sequence-parallel rank). |
|
||||
| `FASTVIDEO_H3_VSA_TILE_FIRST` | bool | `0` | attention | Single-rank MiniMax-H3 VSA with one tile gather of the block input instead of separate Q/K/V/gate scatters. |
|
||||
| `FASTVIDEO_H3_VSA_SM89_KERNEL` | one of original, bf16, int8 | `original` | attention | Fine-attention kernel for MiniMax-H3 VSA on sm_89: original, bf16, or int8 (INT8 QK, BF16 PV). |
|
||||
| `FASTVIDEO_H3_SIM_SP_FP8` | bool | `0` | debug | Simulate the FP8 sequence-parallel exchange of the MiniMax-H3 FP4 VSA path on one rank. |
|
||||
| `FASTVIDEO_H3_FFN_CHUNK_TOKENS` | int | `0` | performance | Inference-only MiniMax-H3 FFN token chunk size; 0 runs the FFN unchunked. |
|
||||
| `FASTVIDEO_H3_FP8_ATTENTION` | bool | `0` | performance | With NVFP4 layer_profile h3_dit_ffn, run MiniMax-H3 attention projections in FP8. |
|
||||
| `FASTVIDEO_H3_FP8_GRANULARITY` | one of tensor, channel | `tensor` | performance | FP8 scaling granularity for FASTVIDEO_H3_FP8_ATTENTION. |
|
||||
| `FASTVIDEO_NVFP4_MM_BACKEND` | str | `auto` | performance | FlashInfer mm_fp4 backend for NVFP4 linears, e.g. auto or cutlass. |
|
||||
| `FASTVIDEO_NVFP4_ACT_AMAX` | path | unset | performance | JSON of calibrated NVFP4 input amax per linear, keyed b<block>.<sub> or full prefix; sets a static activation scale. |
|
||||
| `FASTVIDEO_NVFP4_DYNAMIC_ACT` | str | `""` | performance | NVFP4 linears that derive the activation scale per call: all, or comma-separated layer-name suffixes such as ff.fc_out. |
|
||||
| `FASTVIDEO_H3_ADALN_CACHE` | bool | `0` | performance | Cache MiniMax-H3 AdaLN modulation per timestep instead of keeping the projection weights resident. |
|
||||
| `FASTVIDEO_H3_ADALN_TABLE` | path | unset | performance | Precomputed MiniMax-H3 AdaLN modulation table; enables the cache and skips loading the AdaLN projection weights. |
|
||||
| `FASTVIDEO_H3_ADALN_DUMP` | path | unset | debug | Write the MiniMax-H3 AdaLN modulation table to this path while sampling. |
|
||||
| `FASTVIDEO_H3_SPLICE_TRANSFORMER` | path | unset | eval | Second MiniMax-H3 transformer that runs the late DMD steps (checkpoint step-splice evaluation). |
|
||||
| `FASTVIDEO_H3_SPLICE_FROM_STEP` | int | `4` | eval | First denoising step run by FASTVIDEO_H3_SPLICE_TRANSFORMER. |
|
||||
| `FASTVIDEO_H3_ENCODER_LAYERWISE` | bool | `0` | performance | Stream MiniMax-H3 text-encoder language layers through exact-size pinned host memory (text-only prompts). |
|
||||
| `FASTVIDEO_H3_ENCODER_FUSED_DEQUANT` | bool | `0` | performance | Expand the serialized NVFP4 MiniMax-H3 text encoder with one fused Triton pass on GPUs without FP4 GEMM. |
|
||||
| `FASTVIDEO_H3_VAE_TILE_BATCH` | int | `1` | performance | Spatial tiles per MiniMax-H3 video VAE decoder call; 1 decodes per tile. |
|
||||
| `FASTVIDEO_H3_VAE_INT8_SHARED_QKV` | bool | `0` | performance | Share the INT8 activation rotation and quantization across the MiniMax-H3 VAE Q/K/V projections. |
|
||||
| `FASTVIDEO_H3_VAE_INT8_TRANSPOSE_VIEW` | bool | `0` | performance | Use transposed weight views in the MiniMax-H3 VAE INT8 projections. |
|
||||
| `FASTVIDEO_H3_VAE_INT8_FUSED_DEQUANT` | bool | `0` | performance | Fused dequantization epilogue for the MiniMax-H3 VAE INT8 projections. |
|
||||
| `FASTVIDEO_H3_PINNED_SWAP` | bool | `1` | performance | Swap offloaded MiniMax-H3 modules through exact-size pinned host arenas. |
|
||||
| `FASTVIDEO_H3_PARK_MODULES` | str | unset | performance | Comma-separated MiniMax-H3 denoise modules parked on the host while the text encoder runs, e.g. vae,audio_vae. |
|
||||
| `FASTVIDEO_LAYERWISE_OFFLOAD_BUFFERS` | bool | `0` | performance | Layerwise offload also streams large buffers such as packed FP4/FP8 weights. |
|
||||
| `FASTVIDEO_LAYERWISE_RESIDENT_BLOCKS` | int | `0` | performance | Keep the first N layerwise-offloaded blocks resident on the GPU. |
|
||||
| `FASTVIDEO_H3_SP_PROFILE` | bool | `0` | profiling | CUDA-event spans per stage over one MiniMax-H3 FP4 VSA DiT forward. |
|
||||
| `FASTVIDEO_H3_CAPTURE_QKV` | path | unset | debug | Directory for captured real MiniMax-H3 Q/K/V attention inputs. |
|
||||
| `FASTVIDEO_CUDA_MEMORY_CAP_GIB` | float | `0.0` | debug | Cap this process's CUDA allocator at this many GiB to emulate a smaller GPU; 0 leaves it uncapped. |
|
||||
| `FASTVIDEO_MEMORY_REPORT` | bool | `0` | debug | Log bytes held per pipeline component by device and dtype after loading. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_DATA_DIR` | str | `data/cats` | test | Raw data directory for preprocess_ltx2_overfit.py. Deprecated names: `LTX2_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_CAPTION_JSON` | str | `videos2caption_1_sample.json` | test | Caption file, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_CAPTION_JSON`. |
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_VIDEO_SUBDIR` | str | `video` | test | Video subdirectory, relative to the raw data directory. Deprecated names: `LTX2_OVERFIT_VIDEO_SUBDIR`. |
|
||||
@@ -278,54 +227,16 @@ longer exists also fails the test, so the fixing pull request deletes its entry.
|
||||
| `FASTVIDEO_TEST_LTX2_OVERFIT_NUM_COPIES` | int | `4` | test | Number of copies of the overfit sample in the parquet file. Deprecated names: `LTX2_OVERFIT_NUM_COPIES`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_DATA_DIR` | str | `data/kandinsky5_overfit` | test | Raw data directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_DATA_DIR`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_OVERFIT_OUTPUT_DIR` | str | `data/kandinsky5_overfit_preprocessed` | test | Output directory for preprocess_kandinsky5_overfit.py. Deprecated names: `KANDINSKY5_OVERFIT_OUTPUT_DIR`. |
|
||||
| `FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO` | str | `FastVideo/ssim-reference-videos` | test | Hugging Face repository that holds the SSIM reference videos. Deprecated names: `FASTVIDEO_SSIM_REFERENCE_HF_REPO`. |
|
||||
| `FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO_TYPE` | str | `dataset` | test | Repository type of FASTVIDEO_TEST_SSIM_REFERENCE_HF_REPO. Deprecated names: `FASTVIDEO_SSIM_REFERENCE_HF_REPO_TYPE`. |
|
||||
| `FASTVIDEO_TEST_SSIM_SKIP_REFERENCE_DOWNLOAD` | bool | `0` | test | SSIM tests use local reference videos without downloading. Deprecated names: `FASTVIDEO_SSIM_SKIP_REFERENCE_DOWNLOAD`. |
|
||||
| `FASTVIDEO_TEST_SSIM_FULL_QUALITY` | bool | `0` | test | SSIM tests use the full-quality sampling configurations. Deprecated names: `FASTVIDEO_SSIM_FULL_QUALITY`. |
|
||||
| `FASTVIDEO_TEST_NIGHTLY` | bool | `0` | test | Run the nightly end-to-end overfit tests. Deprecated names: `FASTVIDEO_NIGHTLY`. |
|
||||
| `FASTVIDEO_TEST_ULYSSES_FAULT_RANK` | str | unset | test | Rank that fails in the Ulysses fault-injection test. The test sets it for its worker processes. Deprecated names: `FASTVIDEO_ULYSSES_FAULT_RANK`. |
|
||||
| `FASTVIDEO_TEST_ULYSSES_FAULT_STAGE` | str | unset | test | Stage that fails in the Ulysses fault-injection test. The test sets it for its worker processes. Deprecated names: `FASTVIDEO_ULYSSES_FAULT_STAGE`. |
|
||||
| `FASTVIDEO_TEST_GOLDEN_GATE_DIR` | str | unset | test | Local directory of golden-gate reference tensors. Deprecated names: `FASTVIDEO_GOLDEN_GATE_DIR`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ALLOW_LOW_MEMORY` | bool | `0` | test | Run the MLX Wan2.2 5B real-weights parity test on hosts with little memory. Deprecated names: `FASTVIDEO_WAN22_5B_ALLOW_LOW_MEMORY`. |
|
||||
| `FASTVIDEO_TEST_WAN22_5B_ROOT` | str | unset | test | Local Wan2.2 5B checkpoint for the MLX real-weights parity test. Deprecated names: `FASTVIDEO_WAN22_5B_ROOT`. |
|
||||
| `FASTVIDEO_TEST_GRADNORM_UPDATE` | bool | `0` | test | Gradient-norm regression tests update their references. Deprecated names: `FASTVIDEO_GRADNORM_UPDATE`. |
|
||||
| `FASTVIDEO_TEST_FLUX_T2I_MODEL_DIR` | str | `black-forest-labs/FLUX.1-dev` | test | Model for the Flux text-to-image SSIM test. Deprecated names: `FLUX_T2I_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_FLUX_TRANSFORMER_PATH` | str | unset | test | Local Flux transformer for the Flux transformer test. Deprecated names: `FLUX_TRANSFORMER_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_MODEL_PATH` | str | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | test | Model for the GEN3C SSIM test. Deprecated names: `GEN3C_MODEL_PATH`. |
|
||||
| `FASTVIDEO_TEST_GEN3C_IMAGE_PATH` | str | unset | test | Input image for the GEN3C SSIM test. Deprecated names: `GEN3C_TEST_IMAGE_PATH`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_NUM_GPUS` | int | `1` | test | GPUs for the Kandinsky5 nightly end-to-end overfit test. Deprecated names: `KANDINSKY5_E2E_NUM_GPUS`. |
|
||||
| `FASTVIDEO_TEST_KANDINSKY5_E2E_WRITE_REFERENCE` | bool | `0` | test | The Kandinsky5 nightly end-to-end test writes a missing reference video. Deprecated names: `KANDINSKY5_E2E_WRITE_REFERENCE`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_GOLDEN_DIR` | str | unset | test | Local directory of MiniMax-H3 golden-gate tensors. Deprecated names: `MINIMAX_H3_GATE_GOLDEN_DIR`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_GATE_LAYER` | int | `0` | test | Transformer layer that the MiniMax-H3 golden-gate test checks. Deprecated names: `MINIMAX_H3_GATE_LAYER`. |
|
||||
| `FASTVIDEO_TEST_MINIMAX_H3_MODEL_ROOT` | str | unset | test | Local MiniMax-H3 checkpoint for the golden-gate test. Deprecated names: `MINIMAX_H3_MODEL_ROOT`. |
|
||||
| `FASTVIDEO_TEST_SD35_MODEL_DIR` | str | `stabilityai/stable-diffusion-3.5-medium` | test | Model for the Stable Diffusion 3.5 SSIM test. Deprecated names: `SD35_MODEL_DIR`. |
|
||||
| `FASTVIDEO_TEST_TAEH3_REFERENCE_DIR` | str | unset | test | Upstream taehv checkout for the MLX TAEH3 parity test. Deprecated names: `TAEH3_REFERENCE_DIR`. |
|
||||
|
||||
Variables that FastVideo no longer reads; setting one logs a warning:
|
||||
|
||||
| Deprecated variable | Reason |
|
||||
| ------------------------------------------------ | ---------------------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `DREAMX_WORLD_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `FASTVIDEO_TEST_DREAMX_WORLD_AR_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `DREAMX_WORLD_AR_SSIM_MODEL_PATH` | DreamX World was removed |
|
||||
| `FASTVIDEO_TEST_GAMECRAFT_MODEL_PATH` | HunyuanGameCraft was removed |
|
||||
| `GAMECRAFT_MODEL_PATH` | HunyuanGameCraft was removed |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_LOCAL_WEIGHTS_DIR` | GLM-Image was removed |
|
||||
| `GLM_IMAGE_LOCAL_WEIGHTS_DIR` | GLM-Image was removed |
|
||||
| `FASTVIDEO_TEST_GLM_IMAGE_MODEL_DIR` | GLM-Image was removed |
|
||||
| `GLM_IMAGE_MODEL_DIR` | GLM-Image was removed |
|
||||
| `FASTVIDEO_TEST_LONGCAT_MODEL_ROOT` | LongCat was removed |
|
||||
| `LONGCAT_MODEL_ROOT` | LongCat was removed |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_DIR` | Z-Image was removed |
|
||||
| `ZIMAGE_MODEL_DIR` | Z-Image was removed |
|
||||
| `FASTVIDEO_TEST_ZIMAGE_MODEL_REVISION` | Z-Image was removed |
|
||||
| `ZIMAGE_MODEL_REVISION` | Z-Image was removed |
|
||||
| Deprecated variable | Reason |
|
||||
| ----------------------------------------- | ---------------- |
|
||||
| `FASTVIDEO_TARGET_DEVICE` | no code reads it |
|
||||
| `FASTVIDEO_USE_PRECOMPILED` | no code reads it |
|
||||
| `FASTVIDEO_RINGBUFFER_WARNING_INTERVAL` | no code reads it |
|
||||
| `FASTVIDEO_ENGINE_ITERATION_TIMEOUT_S` | no code reads it |
|
||||
| `FASTVIDEO_SERVER_DEV_MODE` | no code reads it |
|
||||
| `FASTVIDEO_TEST_DYNAMO_FULLGRAPH_CAPTURE` | no code reads it |
|
||||
| `FASTVIDEO_TRACE_FUNCTION` | no code reads it |
|
||||
<!-- END GENERATED ENV TABLE -->
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
# Selectable GPU CI Backends
|
||||
|
||||
GPU CI can use the existing Slurm dispatcher, Modal, or GB200 Kubernetes Jobs
|
||||
in the `vllm` namespace. GitHub and Buildkite remain the trigger and reporting
|
||||
systems. `vllm` names the cluster namespace here; tests run FastVideo's existing
|
||||
lane scripts rather than a vLLM inference server.
|
||||
|
||||
The default remains Slurm. The existing `.buildkite/pipeline.yml`, private
|
||||
Slurm uploader, and dormant Modal launchers are preserved. Deploying the new
|
||||
trusted uploader is an operator step; merging these files does not change the
|
||||
live Buildkite pipelines or create cluster resources. See
|
||||
[CI/CD Architecture](ci_architecture.md) for the existing installation.
|
||||
|
||||
## Execution and Limits
|
||||
|
||||
```text
|
||||
GitHub PR / slash command / schedule
|
||||
-> trusted Buildkite bootstrap
|
||||
-> slurm: existing validated graph and private dispatcher
|
||||
-> modal or vllm: one trusted suite coordinator
|
||||
-> shared PR admission and GPU reservations
|
||||
-> isolated GPU worker per existing lane
|
||||
-> lane results, logs, and backend-specific suite status
|
||||
```
|
||||
|
||||
All new Modal and `vllm` coordinators share one admission database. It enforces:
|
||||
|
||||
- At most two distinct PRs with admitted GPU work.
|
||||
- At most four reserved GPUs per PR across its concurrent builds and lanes.
|
||||
- At most eight reserved GPUs across both backends combined.
|
||||
- Additional PRs wait without creating GPU workers.
|
||||
|
||||
A PR is identified by repository and PR number, so Fastcheck, merge builds,
|
||||
manual reruns, and separate attempts share its allowance. Its slot remains
|
||||
occupied between lanes until all admitted builds for that PR finish. A
|
||||
non-PR build, including a schedule on `main`, consumes its own slot and the
|
||||
same GPU budget. The preserved Slurm dispatcher has its existing independent
|
||||
limits; this new admission database does not control legacy Slurm jobs.
|
||||
|
||||
Lanes retain their existing one-, two-, or four-GPU requirements. Four-GPU
|
||||
SSIM and training lanes wait for that PR's other lanes to release capacity.
|
||||
Selected integration lanes wait for the golden gate and are skipped if it
|
||||
fails. CPU-only GitHub checks are unaffected. Pending admission is bounded by
|
||||
`queue_timeout_seconds`, initially six hours. The coordinator's Buildkite
|
||||
command timeout is eight hours, including admission and lane waits after the
|
||||
command starts; time waiting for a Buildkite agent is outside that timeout.
|
||||
|
||||
Reservations persist before a worker is created. Cancellation releases them
|
||||
only after worker termination is confirmed. A lost coordinator, uncertain
|
||||
create response, or unavailable backend retains its reservations until
|
||||
recovery confirms cleanup. There is no heartbeat expiry that silently frees
|
||||
GPUs while an old worker could still be running.
|
||||
|
||||
## Operator Installation
|
||||
|
||||
Use an operator-reviewed immutable checkout under
|
||||
`/opt/fastvideo-gpu-ci/source`. Install the supplied `scripts/gpu_ci/run` and
|
||||
`scripts/gpu_ci/upload` wrappers as `/opt/fastvideo-gpu-ci/run` and
|
||||
`/opt/fastvideo-gpu-ci/upload`. They invoke the reviewed Python entrypoint
|
||||
with isolated import mode. Install the supplied files instead of generating
|
||||
shell wrappers from build metadata:
|
||||
|
||||
```bash
|
||||
install -m 755 /opt/fastvideo-gpu-ci/source/scripts/gpu_ci/run /opt/fastvideo-gpu-ci/run
|
||||
install -m 755 /opt/fastvideo-gpu-ci/source/scripts/gpu_ci/upload /opt/fastvideo-gpu-ci/upload
|
||||
```
|
||||
|
||||
The controller needs Python 3.10 or newer, PyYAML,
|
||||
the Buildkite agent, and `kubectl`; Modal additionally needs the reviewed
|
||||
Modal SDK and a controller-side Modal credential.
|
||||
|
||||
Copy [the configuration example](../../scripts/gpu_ci/config.example.json)
|
||||
to `/etc/fastvideo-gpu-ci.json`. Keep the installation and configuration
|
||||
operator-owned and unwritable by workers. Configure the actual legacy
|
||||
`slurm_uploader` command, replace each enabled backend's image placeholder
|
||||
with a reviewed registry digest, and leave `default_backend` as `slurm`
|
||||
during canaries. The checked-in placeholders deliberately cannot start jobs.
|
||||
|
||||
The permanent dispatcher must reach the Kubernetes API independently of a
|
||||
developer laptop, SSH tunnel, or Tailscale session. The existing CPU development
|
||||
pod is useful for investigating access, but is not a production CI service.
|
||||
Use a dedicated controller identity and a separate `gpu-ci-dispatch` Buildkite
|
||||
queue. The trusted bootstrap also needs access to the existing Slurm uploader
|
||||
while that backend remains enabled.
|
||||
|
||||
Run all coordinators on **one host**, with `state_path` and `artifacts_dir` on
|
||||
its persistent local disk. The implementation uses SQLite transactions and
|
||||
local file locks. Do not place the database or locks on Lustre/NFS, run
|
||||
independent database copies, or scale controller replicas across nodes. Those
|
||||
configurations would invalidate the global limits. Multiple Buildkite agent
|
||||
processes on the same host can share the installation and ledger; enough
|
||||
agents are needed for concurrent suite coordinators and queued work.
|
||||
|
||||
Configure agent-owned hooks to skip repository checkout and reject commands
|
||||
other than the trusted uploader and coordinator. Disable repository hooks
|
||||
and plugins on this queue. Do not use a PR checkout to load `scripts/gpu_ci`,
|
||||
its configuration, or its worker entrypoint. Build metadata is validated by
|
||||
the dispatcher; arbitrary environment variables are not forwarded into GPU
|
||||
workers. Buildkite, Kubernetes, and Modal control-plane credentials stay on
|
||||
the dispatcher.
|
||||
|
||||
For Kubernetes, provision namespace-scoped permissions to create/get/delete
|
||||
Jobs and read Pods and their logs. Worker Pods disable service-account token
|
||||
mounting and request the exact GPU count on ARM64 GB200 nodes. The image must
|
||||
include the expected `/opt/venv` runtime, CUDA/SM100 support, and the reviewed
|
||||
FA4 dependencies. Use a separate AMD64 image digest for Modal. The Modal
|
||||
profile preserves H100 for encoder, custom-kernel, and VSA lanes, and L40S
|
||||
for the remaining lanes. Its reviewed image must support both SM89 and
|
||||
SM90a. Attention settings are selected per backend and lane to preserve the
|
||||
existing Modal and GB200 FA4 profiles; do not replace them with one global
|
||||
attention override. Validate the images against their respective hardware
|
||||
before enabling merge gates.
|
||||
|
||||
Optional Kubernetes configuration includes `context`, `hf_secret`,
|
||||
`hf_secret_key`, `cache_pvc`, `cache_subpath`, `artifacts_pvc`, and
|
||||
`artifacts_subpath`. Only provide a read-only Hugging Face credential when
|
||||
private/gated downloads require it. PR workers are untrusted and can access
|
||||
any credential supplied to them, so never provide a reference-publication
|
||||
or other write-capable token. Prefer a pre-populated read-only cache.
|
||||
|
||||
Use dedicated CI PVCs and relative CI subpaths. The personal
|
||||
`lustre-pvc-vllm` and `nfs-pvc-vllm` claims are rejected. The cache is mounted
|
||||
read-only; mutable references and locks stay inside the worker. The artifact
|
||||
init container prepares each Job's dedicated artifact subdirectory before
|
||||
the worker mounts it. An
|
||||
operator quota or admission policy scoped to CI can provide a second GPU
|
||||
ceiling; do not apply an eight-GPU quota to the shared `vllm` namespace if it
|
||||
would also cap other users' work.
|
||||
|
||||
## Wire Every Buildkite Entry Pipeline
|
||||
|
||||
Set each pipeline's operator-owned bootstrap command to
|
||||
`/opt/fastvideo-gpu-ci/upload`. Apply this to all three entry pipelines:
|
||||
|
||||
| Pipeline | Trigger and scope |
|
||||
|---|---|
|
||||
| `pr-fastcheck` | Automatic PR webhook; `TEST_SCOPE=fastcheck` or unset. |
|
||||
| `ci` | Existing API triggers for Fastcheck, full, merge, direct, and scheduled SSIM. Keep its incoming PR webhook disabled to avoid duplicate builds. |
|
||||
| `fastvideo-performance-lane` | Existing schedule; `TEST_SCOPE=direct`, `TEST_TYPE=performance`. |
|
||||
|
||||
The wrapper resolves `CI_GPU_BACKEND` from the build environment, falling
|
||||
back to `default_backend` in the trusted configuration. Allowed values are
|
||||
exactly `slurm`, `modal`, and `vllm`; unknown values fail. For Slurm, upload
|
||||
delegates to the unchanged private uploader. For Modal or `vllm`, it uploads
|
||||
one fixed `/opt/fastvideo-gpu-ci/run` command on the dedicated queue, with
|
||||
the selected backend and scope pinned in step environment.
|
||||
|
||||
Set `CI_GPU_BACKEND=vllm` in the Buildkite build environment for a rack canary,
|
||||
or `modal` for a Modal canary. The existing API build payload can carry the
|
||||
same string in its `env` object. Automatic PR webhook builds inherit the
|
||||
operator's configured default unless pipeline/build configuration explicitly
|
||||
overrides it. Do not put backend selection inside a PR-controlled command.
|
||||
|
||||
Keep the existing exact `BUILDKITE_COMMIT`, repository, PR identity, and
|
||||
`TEST_SCOPE` metadata. Direct runs also need an allowlisted `TEST_TYPE`.
|
||||
Merge runs need the trusted base-branch planner's `MERGE_TEST_PLAN`,
|
||||
`MERGE_GOLDEN_TESTS`, and `MERGE_SSIM_TESTS`. Full/direct/scheduled quality
|
||||
runs keep their complete matrices. The worker fetches and verifies the
|
||||
immutable commit before installing dependencies or invoking a lane script.
|
||||
|
||||
The new Modal adapter uses bounded one-to-four-GPU sandboxes and the same
|
||||
lane scripts as Kubernetes. It does not reactivate `pr_test.sh` or the old
|
||||
Modal SSIM fan-out, which cannot enforce this shared four-GPU-per-PR budget.
|
||||
The legacy files remain available for their existing manual workflows.
|
||||
|
||||
## Statuses and Default-Backend Cutover
|
||||
|
||||
The selected adapter reports separate suite contexts:
|
||||
|
||||
| Scope | GitHub context |
|
||||
|---|---|
|
||||
| Fastcheck | `gpu-ci/<backend>/fastcheck-passed` |
|
||||
| Merge or explicit full suite | `gpu-ci/<backend>/full-suite-passed` |
|
||||
| Direct lane | `gpu-ci/<backend>/direct-test-completed` |
|
||||
| Scheduled SSIM | `gpu-ci/<backend>/scheduled-ssim-passed` |
|
||||
|
||||
The repository variable `CI_GPU_BACKEND` controls which new backend can
|
||||
publish the existing required `fastcheck-passed` and `full-suite-passed`
|
||||
contexts. Empty or `slurm` preserves the existing status behavior. `modal`
|
||||
or `vllm` enables `ci-gpu-backend-status.yml`, which reads the latest statuses
|
||||
for that one backend and mirrors both required contexts. A missing result
|
||||
becomes pending; failure and error remain failures. It never combines one
|
||||
backend's Fastcheck with another backend's full-suite result. Late status
|
||||
events re-read current state instead of replaying stale event payloads.
|
||||
|
||||
Direct tests are diagnostic and do not promote a whole suite on the new
|
||||
backends. Rerun the matching Fastcheck or merge/full suite to clear its gate.
|
||||
Per-build tests on the other new backend remain separate diagnostics. The
|
||||
legacy direct-test aggregation workflow is disabled while a new backend is
|
||||
selected. These workflows retain the existing trust assumption that only
|
||||
authorized status-writing integrations can publish CI status contexts.
|
||||
|
||||
For production cutover, drain existing Slurm builds: its preserved pipeline
|
||||
still emits canonical status contexts. The new uploader rejects a Slurm
|
||||
override when `default_backend` is `modal` or `vllm`, preventing later legacy
|
||||
builds from overwriting the promoted backend's checks. Ensure no old bootstrap
|
||||
bypasses the new uploader. Synchronize the operator `default_backend` and
|
||||
the GitHub repository variable, then clear or rerun required checks for all
|
||||
open PRs. Old green contexts do not become new-backend validation merely
|
||||
because a setting changed. Run both Fastcheck and the merge/full suite on
|
||||
the promoted backend before allowing merge. Apply the same drain and rerun
|
||||
procedure when rolling back to Slurm.
|
||||
|
||||
## Validation and Recovery
|
||||
|
||||
Inspect the rendered opt-in pipeline without submitting a build:
|
||||
|
||||
```bash
|
||||
CI_GPU_BACKEND=vllm TEST_SCOPE=fastcheck \
|
||||
/opt/fastvideo-gpu-ci/venv/bin/python -I \
|
||||
/opt/fastvideo-gpu-ci/source/scripts/gpu_ci/entrypoint.py render \
|
||||
--config /etc/fastvideo-gpu-ci.json
|
||||
```
|
||||
|
||||
Validate a one-GPU lane, then a two-/four-GPU lane and the complete Fastcheck
|
||||
suite. Exercise two PRs plus a third waiter, concurrent builds of the same
|
||||
PR, cancellation, retries, and coordinator restart. Confirm observed GPU
|
||||
reservations never exceed two PRs, four per PR, or eight in total. A GPU
|
||||
reservation includes a pending worker, so unavailable nodes cannot cause
|
||||
the controller to submit more work than its allowance.
|
||||
|
||||
Run SSIM, training, and performance canaries separately. References must
|
||||
match the effective GPU/runtime/attention backend; do not silently reuse
|
||||
L40S performance results as GB200 baselines or reseed references as part of
|
||||
routine CI. Workers keep W&B offline and disable reference publication.
|
||||
|
||||
Inspect the persistent ledger and recover an abandoned attempt with:
|
||||
|
||||
```bash
|
||||
/opt/fastvideo-gpu-ci/venv/bin/python -I \
|
||||
/opt/fastvideo-gpu-ci/source/scripts/gpu_ci/entrypoint.py status \
|
||||
--config /etc/fastvideo-gpu-ci.json
|
||||
|
||||
/opt/fastvideo-gpu-ci/venv/bin/python -I \
|
||||
/opt/fastvideo-gpu-ci/source/scripts/gpu_ci/entrypoint.py recover \
|
||||
--config /etc/fastvideo-gpu-ci.json --build-id BUILD_ID.JOB_ID
|
||||
```
|
||||
|
||||
Use the exact ledger ID from `status`; each Buildkite retry has a distinct
|
||||
job ID. Recovery refuses a live coordinator, persists cancellation, stops
|
||||
owned resources, and releases reservations only after confirming termination.
|
||||
If creation or cleanup is ambiguous, investigate the recorded handle on its
|
||||
backend and retain the reservation until the outcome is known. Do not delete
|
||||
the database or manually zero counters to unblock the queue.
|
||||
|
||||
The dispatcher uploads controller logs, per-lane numeric results, request
|
||||
metadata, and the suite summary to Buildkite. These are the sources for its
|
||||
exit status. Generated videos and JUnit files stay in the worker unless a
|
||||
dedicated Kubernetes artifact PVC is configured; automatic publication of
|
||||
those worker files and Modal worker artifacts is not implemented. This is
|
||||
a deployment limitation to account for before replacing existing artifact
|
||||
review workflows.
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Cosmos recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="cosmos" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# FLUX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="flux" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# GLM-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="glm_image" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/zai.webp" alt="Z.ai" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>GLM-Image inference recipes</h2>
|
||||
<p>GLM-Image from Z.ai supports both text-to-image generation and instruction-based image editing, each with a maintained example.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading GLM-Image recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/zai-org">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>The editing example reads <code>assets/images/couple.jpg</code> from the repository root, so run it from a repo checkout rather than an arbitrary working directory.</li>
|
||||
<li>Output paths default under <code>image_output/</code>; pass <code>--output</code> to change them.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Hunyuan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="hunyuan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
+99
-7
@@ -141,25 +141,46 @@ hide:
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./kandinsky6/" aria-label="Open Kandinsky 6 recipes">
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./longcat/" aria-label="Open LongCat recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/kandinsky.webp" alt="" width="132" height="132" loading="lazy">
|
||||
<img class="off-glb" src="../assets/logos/meituan-longcat.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Kandinsky 6</strong><small>Video with audio, and video SR</small></span>
|
||||
<span class="cookbook-count">4 recipes</span>
|
||||
<span><strong>LongCat</strong><small>T2V, I2V, optional refine</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2VA</li>
|
||||
<li>I2VA</li>
|
||||
<li>VSR</li>
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./turbodiffusion/" aria-label="Open TurboDiffusion recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">Turbo</span>
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>TurboDiffusion</strong><small>Accelerated Wan profiles</small></span>
|
||||
<span class="cookbook-count">3 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2V</li>
|
||||
<li>I2V</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
@@ -192,6 +213,49 @@ hide:
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./glm-image/" aria-label="Open GLM-Image recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/zai.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GLM-Image</strong><small>Generate and edit</small></span>
|
||||
<span class="cookbook-count">2 recipes</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
<li>Edit</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./z-image/" aria-label="Open Z-Image recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
<span class="cookbook-evervault__gradient"></span>
|
||||
<span class="cookbook-evervault__noise" data-cookbook-pattern></span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tongyi.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>Z-Image</strong><small>Turbo text to image</small></span>
|
||||
<span class="cookbook-count">1 recipe</span>
|
||||
</span>
|
||||
<ul class="cookbook-mode-row">
|
||||
<li>T2I</li>
|
||||
</ul>
|
||||
</span>
|
||||
</a>
|
||||
|
||||
<a class="cookbook-family-tile cookbook-family-tile--ready" href="./stable-diffusion/" aria-label="Open Stable Diffusion recipes">
|
||||
<span class="cookbook-family-tile__visual" data-evervault>
|
||||
<span class="cookbook-evervault" aria-hidden="true">
|
||||
@@ -320,6 +384,20 @@ hide:
|
||||
<p>These families already have runnable examples. The cookbook page is not ready, so the cards are not links.</p>
|
||||
</div>
|
||||
<div class="cookbook-family-grid">
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="GameCraft cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/tencent-hunyuan.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>GameCraft</strong><small>Game world generation</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="GEN3C cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
@@ -348,6 +426,20 @@ hide:
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="DreamX cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
<img class="off-glb" src="../assets/logos/fastvideo.webp" alt="" width="132" height="132" loading="lazy">
|
||||
</span>
|
||||
</span>
|
||||
<span class="cookbook-family-tile__footer">
|
||||
<span class="cookbook-family-tile__footer-top">
|
||||
<span><strong>DreamX</strong><small>World generation</small></span>
|
||||
<span class="cookbook-count">Page planned</span>
|
||||
</span>
|
||||
</span>
|
||||
</article>
|
||||
|
||||
<article class="cookbook-family-tile cookbook-family-tile--coming" aria-label="LingBot cookbook page planned; runnable examples exist">
|
||||
<span class="cookbook-family-tile__visual">
|
||||
<span class="cookbook-family-tile__logo-wrap">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Kandinsky 5 recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky5" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# LongCat recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="longcat" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/meituan-longcat.webp" alt="Meituan LongCat" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>LongCat inference recipes</h2>
|
||||
<p>LongCat Video from Meituan covers text-to-video and image-to-video, and its maintained examples chain optional distilled and 720p refinement passes.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading LongCat recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/meituan-longcat">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Each LongCat script runs multiple passes (basic, distilled, refine); total runtime scales accordingly, and every pass prints its own output directory.</li>
|
||||
<li>Out of memory: the sources already enable VAE and text-encoder CPU offload; further options are covered in <a href="../../inference/offloading/">Offloading</a>.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# LTX recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="ltx2" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Matrix Game recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="matrixgame" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
+11
-24
@@ -10,15 +10,8 @@ launch. Some Hub repo names still say Preview. That name is historical. V1 is
|
||||
a full model, not a demo. **V2** is the eight-step checkpoint. More forwards
|
||||
is why V2 is the higher-quality FastH3. The V2 schedule contract is in
|
||||
[FastH3 distilled checkpoint schedules](../inference/fasth3-distilled.md).
|
||||
**CompactH3** is the 42-block 20B NVFP4 H3 checkpoint for one Blackwell GPU
|
||||
(RTX 5090 or RTX PRO 6000). The FastH3 V1 and V2 recipes are unchanged.
|
||||
|
||||
The 42-block pruned checkpoint has an [MLX INT8/INT6 conversion and
|
||||
eight-forward T2VA command](../getting_started/installation/mlx.md#pruned-eight-forward-checkpoint).
|
||||
It reads `fastvideo_inference.json` for the trained schedule. The command
|
||||
uses native 832x480 resolution and all requested frames.
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="minimax_h3" data-default-recipe="fasth3-preview-cuda" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -28,8 +21,8 @@ uses native 832x480 resolution and all requested frames.
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Primary focus · Inference</p>
|
||||
<h2>MiniMax H3 recipes</h2>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA, one Blackwell GPU, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>11 maintained recipes</span>
|
||||
<p>Generate video and audio with H3. Run a server on CUDA, one DGX Spark, or Apple Silicon MLX to iterate on prompts, or call the pipeline directly from Python.</p>
|
||||
<span class="cookbook-count" data-cookbook-count>9 maintained recipes</span>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
@@ -54,7 +47,7 @@ uses native 832x480 resolution and all requested frames.
|
||||
<h2 id="h3-modes-heading">Supported modes</h2>
|
||||
<p>
|
||||
CUDA covers T2VA, FL2VA, and Ref2VA on the full checkpoint, plus FastH3
|
||||
V1 and FastH3 V2, plus CompactH3 NVFP4 on one Blackwell GPU. FastH3 V1 also has a DGX Spark runtime with
|
||||
V1 and FastH3 V2. FastH3 V1 also has a DGX Spark runtime with
|
||||
a 1-Spark or 2-Spark device row. MLX is T2VA only: V1 and V2.
|
||||
Temporal <code>--fast</code>, spatial <code>--fast-spatial</code>, and opt-in VSA are flags on the same
|
||||
MLX script, not extra recipes.
|
||||
@@ -71,7 +64,7 @@ uses native 832x480 resolution and all requested frames.
|
||||
<tbody>
|
||||
<tr>
|
||||
<td>T2VA</td>
|
||||
<td>Full H3, FastH3 V1, FastH3 LoRA, FastH3 V2, CompactH3 NVFP4</td>
|
||||
<td>Full H3, FastH3 V1, FastH3 LoRA, FastH3 V2</td>
|
||||
<td>FastH3 V1 or FastH3 V2 after a local DiT conversion</td>
|
||||
</tr>
|
||||
<tr>
|
||||
@@ -109,11 +102,6 @@ uses native 832x480 resolution and all requested frames.
|
||||
<td>FastH3 V1 on one GB10, or two Sparks with Ray sequence parallel (<code>sp_size=2</code>) over QSFP RoCE. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks.</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>CompactH3 NVFP4</td>
|
||||
<td>42-block 20B checkpoint on one RTX 5090 (32 GB, sequential encoder offload) or one RTX PRO 6000 Blackwell (96 GB, encoder+DiT+VAE resident). SageAttention3 FP4, packed NVFP4 DiT, Comfy int8-convrot VAE.</td>
|
||||
<td>Not wired</td>
|
||||
</tr>
|
||||
</tbody>
|
||||
</table>
|
||||
</div>
|
||||
@@ -122,7 +110,7 @@ uses native 832x480 resolution and all requested frames.
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick an H3 recipe and runtime</h2>
|
||||
<p>Choose the result you want, then use a maintained CUDA, Blackwell, DGX Spark, or MLX path.
|
||||
<p>Choose the result you want, then use a maintained CUDA, DGX Spark, or MLX path.
|
||||
Device claims stay tied to checked-in sources and recorded runs.</p>
|
||||
</div>
|
||||
|
||||
@@ -166,7 +154,7 @@ uses native 832x480 resolution and all requested frames.
|
||||
<span>Both can run locally</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" role="group" aria-label="How to run this recipe">
|
||||
<button type="button" data-cookbook-mode="server" aria-pressed="false"><strong>Run a server</strong><span data-cookbook-server-hint>Playground, cURL, or an API client</span></button>
|
||||
<button type="button" data-cookbook-mode="server" aria-pressed="false"><strong>Run a server</strong><span>Playground, cURL, or an API client</span></button>
|
||||
<button type="button" data-cookbook-mode="python" aria-pressed="false"><strong>Use Python directly</strong><span>Call the model in your own process</span></button>
|
||||
</div>
|
||||
</div>
|
||||
@@ -213,17 +201,17 @@ uses native 832x480 resolution and all requested frames.
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-start-heading">
|
||||
<h4 id="serving-start-heading"><span aria-hidden="true">2</span> Start the server</h4>
|
||||
<p>Keep this terminal running while you use <span data-cookbook-playground-only>the playground or </span>API clients.</p>
|
||||
<p>Keep this terminal running while you use the playground or API clients.</p>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>GPU machine · Terminal</span></div><pre id="cookbook-server-command"><code class="language-bash" data-cookbook-server-command></code></pre></div>
|
||||
<details class="cookbook-serving__check"><summary>Check that the server is ready</summary><p>In another terminal, this returns <code>{"status":"ok"}</code> after startup.</p><div class="cookbook-command"><pre id="cookbook-health-command"><code class="language-bash" data-cookbook-health-command></code></pre></div></details>
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-client-heading">
|
||||
<h4 id="serving-client-heading"><span aria-hidden="true">3</span> Generate and download a video</h4>
|
||||
<div class="cookbook-serving__playground" data-cookbook-playground-only>
|
||||
<div class="cookbook-serving__playground">
|
||||
<div><strong>Try prompts in your browser</strong><p>Edit a prompt, generate, and watch the result. The playground uses the same server as cURL and your app.</p></div>
|
||||
<a class="cookbook-serving__launch" data-cookbook-playground href="http://127.0.0.1:8000/playground/" target="_blank" rel="noopener">Open playground <span aria-hidden="true">↗</span></a>
|
||||
</div>
|
||||
<p class="cookbook-serving__local-hint"><span data-cookbook-playground-only>Open after the server is ready. </span>On a remote GPU machine, <a href="../openai-api/#connect-your-app">forward port 8000</a> to your computer first.<span data-cookbook-playground-only> This opens a local page, not a hosted demo.</span></p>
|
||||
<p class="cookbook-serving__local-hint">Open after the server is ready. On a remote GPU machine, <a href="../openai-api/#connect-your-app">forward port 8000</a> to your computer first. This opens a local page, not a hosted demo.</p>
|
||||
<details class="cookbook-serving__code"><summary>Use cURL or an SDK</summary>
|
||||
<p>Each example submits a job, checks its status, and saves the MP4. The Python and JavaScript examples use OpenAI-compatible clients; no OpenAI account is needed.</p>
|
||||
<div class="cookbook-serving__clients" role="group" aria-label="API client language">
|
||||
@@ -287,8 +275,7 @@ cd FastVideo</code></pre>
|
||||
<li>The MLX source runtime supports T2VA, optional temporal <code>--fast</code>, optional spatial <code>--fast-spatial</code>, and opt-in VSA on <code>--include-vsa</code> checkpoints. FastH3 V2 MLX converts with <code>--include-vsa</code> and runs eight forwards. FL2VA, Ref2VA, and two-pass refinement are not wired.</li>
|
||||
<li>GPU count and VAE decode backend are configurable in the builder above for FastH3 CUDA recipes. Only the value shown by default has a recorded run; other supported values are unmeasured here.</li>
|
||||
<li>DGX Spark is a runtime on FastH3 V1, not a separate family card. Select NVIDIA DGX Spark, then 1 Spark or 2 Sparks. The CUDA GPU-count knob does not apply to Spark.</li>
|
||||
<li>CompactH3 NVFP4 is one Blackwell GPU. RTX 5090 (32 GB) parks the encoder in pinned host RAM. RTX PRO 6000 Blackwell (96 GB) keeps encoder, DiT, and VAE resident. Keep <code>FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER</code>, <code>FASTVIDEO_FA4=0</code>, <code>FASTVIDEO_VSA_SM100A=0</code>, and <code>FLASHINFER_CUDA_ARCH_LIST=12.0a</code>. On PRO 6000 enable VAE compile and leave DiT <code>inference_torch_compile</code> off. CompactH3 is a dense prune; do not enable <code>VIDEO_SPARSE_ATTN_H3</code> until a VSA-trained student exists.</li>
|
||||
<li>GB10 has no FA4 / sm_100a VSA kernel. Keep <code>FASTVIDEO_FA4=0</code> and <code>FASTVIDEO_VSA_SM100A=0</code>. Legal <code>num_frames</code> values are <code>17n+5</code>, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM. Native 16:9 sizes include 832×480 and 1344×768.</li>
|
||||
<li>GB10 has no FA4 / sm_100a VSA kernel. Keep <code>FASTVIDEO_FA4=0</code> and <code>FASTVIDEO_VSA_SM100A=0</code>. Legal <code>num_frames</code> values are <code>17n+5</code>, capped at 362 (15.08 s). A 345-frame request on one Spark can OOM.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# MMAudio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="mmaudio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Audio recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="stable_audio" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Stable Diffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="sd35" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
---
|
||||
hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# TurboDiffusion recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="turbodiffusion" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<span class="cookbook-family-tile__monogram" aria-hidden="true">Turbo</span>
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>TurboDiffusion inference recipes</h2>
|
||||
<p>TurboDiffusion profiles accelerate Wan checkpoints with step-distilled sampling and the SLA attention backend. These recipes follow the registry's `turbodiffusion` model family.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
<span class="cookbook-lifecycle__stage cookbook-lifecycle__stage--active">Inference <small>live</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Distillation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Fine-tuning <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Training <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Evaluation <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Optimization <small>planned</small></span>
|
||||
<span class="cookbook-lifecycle__stage">Deployment <small>planned</small></span>
|
||||
</div>
|
||||
</header>
|
||||
<nav class="cookbook-jumpnav" aria-label="Recipe page sections">
|
||||
<a href="#recipe-builder">Builder</a>
|
||||
<a href="#cookbook-setup">Setup</a>
|
||||
<a href="#cookbook-troubleshooting">Troubleshooting</a>
|
||||
<a href="#cookbook-evidence">Evidence</a>
|
||||
</nav>
|
||||
|
||||
<section class="cookbook-builder" id="recipe-builder" aria-labelledby="builder-heading">
|
||||
<div class="cookbook-builder__intro">
|
||||
<h2 id="builder-heading">Pick a recipe and runtime</h2>
|
||||
<p>Start with the result you want, then choose one of the runtimes FastVideo actually maintains for it.</p>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-builder__layout">
|
||||
<div class="cookbook-controls">
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Recipe</strong>
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading TurboDiffusion recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-selection-row">
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Runtime</strong>
|
||||
<span>Maintained paths only</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" data-cookbook-hardware-options role="group" aria-label="Runtime">
|
||||
<button type="button" disabled>Loading runtimes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
Reading recipe evidence...
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<article class="cookbook-result">
|
||||
<div class="cookbook-result__header">
|
||||
<h3 data-cookbook-label>Loading...</h3>
|
||||
<div class="cookbook-result__badges">
|
||||
<span class="cookbook-badge">Maintained</span>
|
||||
<span class="cookbook-badge" data-cookbook-evidence>Source-backed</span>
|
||||
<span class="cookbook-badge cookbook-badge--neutral" data-cookbook-hardware-badge>Source config</span>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<dl class="cookbook-result__facts">
|
||||
<div><dt>Model</dt><dd data-cookbook-model>Loading...</dd></div>
|
||||
<div><dt>Workload</dt><dd data-cookbook-task>Loading...</dd></div>
|
||||
<div><dt>Source configuration</dt><dd data-cookbook-gpus>Loading...</dd></div>
|
||||
<div><dt>Expected output</dt><dd data-cookbook-artifact>Loading...</dd></div>
|
||||
</dl>
|
||||
|
||||
<div class="cookbook-command">
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/loayrashid">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
</div>
|
||||
|
||||
<noscript>
|
||||
<div class="cookbook-noscript">
|
||||
JavaScript is needed for the guided selector. You can still browse the
|
||||
<a href="../../inference/examples/examples_inference_index/">maintained inference examples</a>.
|
||||
</div>
|
||||
</noscript>
|
||||
</section>
|
||||
</div>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-setup">
|
||||
<summary>Setup</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>The generated commands expect a local clone:</p>
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-troubleshooting">
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>TurboDiffusion paths load community-published <code>loayrashid/TurboWan*</code> checkpoints; availability is governed by those repos.</li>
|
||||
<li>The SLA attention backend used by the I2V recipe is selected inside the example source; do not combine it with another <code>FASTVIDEO_ATTENTION_BACKEND</code> override in the same shell.</li>
|
||||
</ul>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
<details class="cookbook-collapsible" id="cookbook-evidence">
|
||||
<summary>Evidence status</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<p>All recipes on this page are <strong>Source-backed</strong>: their commands, model IDs, and flags were validated against the checked-in FastVideo sources listed above (static validation). No runtime GPU validation is recorded for these recipes, so GPU model fit, memory use, throughput, and runtime duration are <strong>Unknown</strong> and deliberately not claimed. Runtime buttons show only the GPU counts configured in checked-in sources.</p>
|
||||
</div>
|
||||
</details>
|
||||
+2
-51
@@ -5,7 +5,7 @@ hide:
|
||||
|
||||
# Wan recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="wan" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
@@ -122,17 +122,6 @@ hide:
|
||||
</div>
|
||||
|
||||
<p class="cookbook-selection-description" data-cookbook-description>Loading recipe details...</p>
|
||||
<div class="cookbook-selection-row" data-cookbook-usage>
|
||||
<div class="cookbook-selection-row__label">
|
||||
<strong>Workflow</strong>
|
||||
<span>Both can run locally</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--hardware" role="group" aria-label="How to run this recipe">
|
||||
<button type="button" data-cookbook-mode="server" aria-pressed="false"><strong>Run a server</strong><span data-cookbook-server-hint>Playground, cURL, or an API client</span></button>
|
||||
<button type="button" data-cookbook-mode="python" aria-pressed="false"><strong>Use Python directly</strong><span>Call the model in your own process</span></button>
|
||||
</div>
|
||||
</div>
|
||||
<p class="cookbook-hardware-note" data-cookbook-serving-availability></p>
|
||||
<p class="cookbook-hardware-note">Exact device and memory details appear only when a recorded run supports them.</p>
|
||||
|
||||
<div class="cookbook-hardware-state" data-cookbook-hardware-state role="status" aria-live="polite">
|
||||
@@ -161,45 +150,7 @@ hide:
|
||||
<div class="cookbook-command__bar">
|
||||
<span>Terminal</span>
|
||||
</div>
|
||||
<pre id="cookbook-local-command"><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
<p class="cookbook-hardware-note" data-cookbook-python-note>Running this script again starts a new process and reloads the model. To iterate in Python, create the generator once and reuse it for multiple prompts.</p>
|
||||
|
||||
<div class="cookbook-serving" data-cookbook-serving hidden>
|
||||
<p class="cookbook-serving__intro" data-cookbook-server-lifetime>Start once, then change prompts in the playground or your app. You can run the server and clients on the same machine.</p>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-install-heading">
|
||||
<h4 id="serving-install-heading"><span aria-hidden="true">1</span> Prepare the machine</h4>
|
||||
<p>Run from your FastVideo clone in an activated Python environment. See <a data-cookbook-install-guide href="../../getting_started/installation/gpu/">installation requirements</a>.</p>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>GPU machine · Terminal</span></div><pre id="cookbook-server-install"><code class="language-bash" data-cookbook-server-install></code></pre></div>
|
||||
<details class="cookbook-serving__prepare" data-cookbook-prepare hidden><summary>Download and convert MLX weights once</summary><p>Skip this if the weights are already prepared. Edit the paths in the serving config to use your existing files.</p><div class="cookbook-command"><pre id="cookbook-server-prepare"><code class="language-bash" data-cookbook-server-prepare></code></pre></div></details>
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-start-heading">
|
||||
<h4 id="serving-start-heading"><span aria-hidden="true">2</span> Start the server</h4>
|
||||
<p>Keep this terminal running while you use <span data-cookbook-playground-only>the playground or </span>API clients.</p>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>GPU machine · Terminal</span></div><pre id="cookbook-server-command"><code class="language-bash" data-cookbook-server-command></code></pre></div>
|
||||
<details class="cookbook-serving__check"><summary>Check that the server is ready</summary><p>In another terminal, this returns <code>{"status":"ok"}</code> after startup.</p><div class="cookbook-command"><pre id="cookbook-health-command"><code class="language-bash" data-cookbook-health-command></code></pre></div></details>
|
||||
</section>
|
||||
<section class="cookbook-serving__step" aria-labelledby="serving-client-heading">
|
||||
<h4 id="serving-client-heading"><span aria-hidden="true">3</span> Generate and download a video</h4>
|
||||
<div class="cookbook-serving__playground" data-cookbook-playground-only>
|
||||
<div><strong>Try prompts in your browser</strong><p>Edit a prompt, generate, and watch the result. The playground uses the same server as cURL and your app.</p></div>
|
||||
<a class="cookbook-serving__launch" data-cookbook-playground href="http://127.0.0.1:8000/playground/" target="_blank" rel="noopener">Open playground <span aria-hidden="true">↗</span></a>
|
||||
</div>
|
||||
<p class="cookbook-serving__local-hint"><span data-cookbook-playground-only>Open after the server is ready. </span>On a remote GPU machine, <a href="../openai-api/#connect-your-app">forward port 8000</a> to your computer first.<span data-cookbook-playground-only> This opens a local page, not a hosted demo.</span></p>
|
||||
<details class="cookbook-serving__code"><summary>Use cURL or an SDK</summary>
|
||||
<p>Each example submits a job, checks its status, and saves the MP4. The Python and JavaScript examples use OpenAI-compatible clients; no OpenAI account is needed.</p>
|
||||
<div class="cookbook-serving__clients" role="group" aria-label="API client language">
|
||||
<button type="button" data-cookbook-client="curl" aria-pressed="false">cURL</button>
|
||||
<button type="button" data-cookbook-client="python" aria-pressed="true">Python</button>
|
||||
<button type="button" data-cookbook-client="javascript" aria-pressed="false">JavaScript</button>
|
||||
</div>
|
||||
<div class="cookbook-command"><div class="cookbook-command__bar"><span>Client dependencies</span></div><pre id="cookbook-client-install"><code class="language-bash" data-cookbook-client-install></code></pre></div>
|
||||
<div class="cookbook-command cookbook-command--client"><div class="cookbook-command__bar"><span data-cookbook-client-filename>video.py</span><a data-cookbook-client-source href="https://github.com/hao-ai-lab/FastVideo/tree/main/examples/serving/clients">View source</a></div><pre id="cookbook-client-code"><code data-cookbook-client-code></code></pre></div>
|
||||
<p data-cookbook-client-run></p>
|
||||
</details>
|
||||
</section>
|
||||
<p class="cookbook-serving__boundary">This is a local development server without built-in API-key authentication. The client key <code>local</code> is a placeholder. Keep the server on loopback; use an authenticated TLS proxy before exposing it publicly. Run the JavaScript client in your webapp's backend, not in a browser with a private key.</p>
|
||||
<a href="../openai-api/">Server guide and API compatibility →</a>
|
||||
<pre><code class="language-bash" data-cookbook-command>Loading...</code></pre>
|
||||
</div>
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
|
||||
@@ -3,19 +3,19 @@ hide:
|
||||
- toc
|
||||
---
|
||||
|
||||
# Kandinsky 6 recipes
|
||||
# Z-Image recipes
|
||||
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="kandinsky6" data-recipes="../../assets/cookbook-recipes.json?v=15">
|
||||
<div class="cookbook-shell cookbook-family-page" data-cookbook data-family="zimage" data-recipes="../../assets/cookbook-recipes.json?v=11">
|
||||
<header class="cookbook-family-header">
|
||||
<a class="cookbook-back-link" href="../"><span aria-hidden="true">←</span> All model families</a>
|
||||
<div class="cookbook-family-header__body">
|
||||
<span class="cookbook-family-header__logo">
|
||||
<img class="off-glb" src="../../assets/logos/kandinsky.webp" alt="Kandinsky Lab" width="112" height="112">
|
||||
<img class="off-glb" src="../../assets/logos/tongyi.webp" alt="Tongyi MAI" width="112" height="112">
|
||||
</span>
|
||||
<div>
|
||||
<p class="cookbook-eyebrow">Maintained family · Inference</p>
|
||||
<h2>Kandinsky 6 inference recipes</h2>
|
||||
<p>Kandinsky 6 from the Kandinsky Lab generates five-second video with synchronized audio from text or an image, with base and distilled pi-Flow checkpoints, and upscales existing clips with base or distilled video super-resolution.</p>
|
||||
<h2>Z-Image inference recipes</h2>
|
||||
<p>Z-Image Turbo from Tongyi MAI is a fast text-to-image model. The maintained example runs it on a single GPU.</p>
|
||||
</div>
|
||||
</div>
|
||||
<div class="cookbook-lifecycle" aria-label="Lifecycle stages">
|
||||
@@ -49,7 +49,7 @@ hide:
|
||||
<span>Task and checkpoint</span>
|
||||
</div>
|
||||
<div class="cookbook-option-grid cookbook-option-grid--models" data-cookbook-model-options role="group" aria-label="Recipe">
|
||||
<button type="button" disabled>Loading Kandinsky 6 recipes...</button>
|
||||
<button type="button" disabled>Loading Z-Image recipes...</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -97,7 +97,7 @@ hide:
|
||||
|
||||
<div class="cookbook-result__footer">
|
||||
<a data-cookbook-source href="../../inference/examples/basic/">Open example source</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/kandinskylab">View model card</a>
|
||||
<a data-cookbook-model-link href="https://huggingface.co/Tongyi-MAI">View model card</a>
|
||||
</div>
|
||||
<p class="cookbook-picker__status" role="status" aria-live="polite" data-cookbook-status></p>
|
||||
</article>
|
||||
@@ -119,7 +119,6 @@ hide:
|
||||
<pre><code>git clone https://github.com/hao-ai-lab/FastVideo.git
|
||||
cd FastVideo</code></pre>
|
||||
<p>Use <a href="../../inference/configuration/">Configuration</a> for supported Python and CLI settings, <a href="../../inference/optimizations/">Optimizations</a> for attention and memory tradeoffs, and the <a href="../../inference/support_matrix/">support matrix</a> for the supported model and optimization surface.</p>
|
||||
<p>For model-specific behavior, see <a href="../../inference/kandinsky6/">Kandinsky 6 video with audio</a> and <a href="../../inference/kandinsky6_sr/">Kandinsky 6 video super-resolution</a>.</p>
|
||||
</div>
|
||||
</details>
|
||||
|
||||
@@ -127,10 +126,7 @@ cd FastVideo</code></pre>
|
||||
<summary>Troubleshooting</summary>
|
||||
<div class="cookbook-collapsible__body">
|
||||
<ul>
|
||||
<li>Text and image conditioning use the same TI2VA pipeline and example. Set <code>IMAGE_PATH</code> in the script to condition on an image; both base and distilled outputs include synchronized audio.</li>
|
||||
<li>For pi-Flow, set <code>KANDINSKY6_MODEL_PATH</code> to the distilled model ID when running <code>basic_kandinsky6_ti2va.py</code>. Its registered preset uses 10 inference steps, guidance 1.0, <code>eps=1e-6</code>, <code>final_step_size_scale=0.5</code>, and <code>num_policy_substeps=128</code>.</li>
|
||||
<li>For VSR, set <code>INPUT_VIDEO</code> in the generated command. The pipeline accepts x2, x2.25 and x4 scales, processes up to 121 frames at 24 fps, and preserves source audio.</li>
|
||||
<li>Checkpoint key-layout errors indicate that the checkpoint and FastVideo checkout target different Kandinsky 6 Diffusers revisions.</li>
|
||||
<li>Output defaults to <code>outputs/zimage/zimage_turbo.png</code>; pass <code>--output</code> to redirect.</li>
|
||||
<li>Gated or missing checkpoints: run <code>huggingface-cli login</code> and confirm you accepted the model's license on Hugging Face.</li>
|
||||
</ul>
|
||||
</div>
|
||||
@@ -205,6 +205,9 @@ surfaces:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CInferenceConfig
|
||||
color_correction_strength:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.dreamx_world.DreamXWorld5BARPipelineConfig
|
||||
default_camera_rotation:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.gen3c.Gen3CConfig
|
||||
@@ -279,6 +282,11 @@ surfaces:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
|
||||
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.matrixgame2.MatrixGame2BaseI2V480PConfig
|
||||
@@ -297,6 +305,11 @@ surfaces:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.lingbotworld.LingBotWorldI2V480PConfig
|
||||
- fastvideo.configs.pipelines.lingbotworld.Wan2_2_I2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionI2V_A14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2VConfig
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_14B_Config
|
||||
- fastvideo.configs.pipelines.turbodiffusion.TurboDiffusionT2V_1_3B_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_1_T2V_480P_Config
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
- fastvideo.configs.pipelines.matrixgame2.MatrixGame2BaseI2V480PConfig
|
||||
@@ -311,21 +324,48 @@ surfaces:
|
||||
- fastvideo.configs.pipelines.wan.WanI2V720PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V480PConfig
|
||||
- fastvideo.configs.pipelines.wan.WanT2V720PConfig
|
||||
bsa_cdf_threshold:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_chunk_k:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_chunk_q:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_params:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
bsa_sparsity:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_bsa:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enable_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
enhance_hf:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
offload_kv_cache:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
t_thresh:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
use_distill:
|
||||
sources: [fastvideo.configs.pipelines.longcat.LongCatT2V480PConfig, fastvideo.configs.pipelines.longcat.LongCatT2V704PConfig]
|
||||
scheduler_arch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
text_encoder_archs:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
tokenizer_archs:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
transformer_arch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
vae_arch:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.sd35.SD35Config
|
||||
- fastvideo.configs.pipelines.zimage.ZImagePipelineConfig
|
||||
expand_timesteps:
|
||||
sources:
|
||||
- fastvideo.configs.pipelines.wan.FastWan2_2_TI2V_5B_Config
|
||||
@@ -418,14 +458,6 @@ surfaces:
|
||||
sources: [fastvideo.pipelines.basic.magi_human.pipeline_configs.MagiHumanBaseConfig]
|
||||
video_txt_guidance_scale:
|
||||
sources: [fastvideo.pipelines.basic.magi_human.pipeline_configs.MagiHumanBaseConfig]
|
||||
piflow_eps:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
piflow_final_step_size_scale:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
piflow_num_policy_substeps:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
sample_fps:
|
||||
sources: [fastvideo.configs.pipelines.kandinsky6.Kandinsky6TI2VAConfig]
|
||||
compatibility_only:
|
||||
batch_size: "Gen3C inference-only tuning field pending typed batching design."
|
||||
gradient_checkpointing: "Gen3C inference-only compatibility field pending typed batching design."
|
||||
@@ -438,17 +470,15 @@ surfaces:
|
||||
coords_style: "MagiHuman internal data-proxy coordinate convention."
|
||||
frame_receptive_field: "MagiHuman internal data-proxy receptive-field setting."
|
||||
image_conditioning: "MagiHuman preset variant marker for reference-image conditioning."
|
||||
pdd_step_indices: "MiniMax-H3 PDD fine-grid partition; resolve_checkpoint_settings sets it from fastvideo_inference.json."
|
||||
ref_audio_offset: "MagiHuman internal data-proxy audio alignment offset."
|
||||
scheduler_sigma_min: "Z-Image scheduler parity invariant; not part of the public typed inference API."
|
||||
scheduler_use_reference_discrete_timesteps: "Z-Image scheduler parity invariant; not part of the public typed inference API."
|
||||
sr_local_attn_layers: "MagiHuman SR internal sparse-attention layer selection."
|
||||
text_offset: "MagiHuman internal data-proxy text alignment offset."
|
||||
vae_stride: "MagiHuman internal VAE/data-proxy stride setting."
|
||||
vsa_ref_keep_rate: "MiniMax-H3 PDD reference-video VSA keep rate; defaults to the checkpoint's trained value."
|
||||
z_dim: "MagiHuman internal VAE latent channel setting."
|
||||
vocoder_config: "Legacy internal component config object."
|
||||
vocoder_precision: "Precision override pending dedicated component precision design."
|
||||
audio_sample_rate: "Kandinsky6 audio-latent length constant (audio VAE sample rate); tied to the checkpoint, not a request knob."
|
||||
audio_downsample_factor: "Kandinsky6 audio-latent length constant (audio VAE downsample factor); tied to the checkpoint, not a request knob."
|
||||
|
||||
sampling_param_base:
|
||||
moved:
|
||||
@@ -465,6 +495,8 @@ surfaces:
|
||||
pose: request.inputs.pose
|
||||
c2ws_plucker_emb: request.inputs.c2ws_plucker_emb
|
||||
action_path: request.inputs.action_path
|
||||
refine_from: request.inputs.refine_from
|
||||
stage1_video: request.inputs.stage1_video
|
||||
prompt: request.prompt
|
||||
negative_prompt: request.negative_prompt
|
||||
prompt_path: request.inputs.prompt_path
|
||||
@@ -484,6 +516,8 @@ surfaces:
|
||||
guidance_scale: request.sampling.guidance_scale
|
||||
batch_cfg: request.sampling.batch_cfg
|
||||
guidance_scale_2: request.sampling.guidance_scale_2
|
||||
cfg_normalization: request.sampling.cfg_normalization
|
||||
cfg_truncation: request.sampling.cfg_truncation
|
||||
guidance_rescale: request.sampling.guidance_rescale
|
||||
use_embedded_guidance: request.sampling.use_embedded_guidance
|
||||
true_cfg_scale: request.sampling.true_cfg_scale
|
||||
@@ -498,12 +532,19 @@ surfaces:
|
||||
return_continuation_state: request.output.return_state
|
||||
preset_owned:
|
||||
t_thresh: request.stage_overrides.refine.t_thresh
|
||||
spatial_refine_only: request.stage_overrides.refine.spatial_refine_only
|
||||
num_cond_frames: request.stage_overrides.refine.num_cond_frames
|
||||
trajectory_type: request.extensions.gen3c.trajectory_type
|
||||
movement_distance: request.extensions.gen3c.movement_distance
|
||||
camera_rotation: request.extensions.gen3c.camera_rotation
|
||||
prompt_attention_mask: request.extensions.hyworld.prompt_attention_mask
|
||||
negative_attention_mask: request.extensions.hyworld.negative_attention_mask
|
||||
camera_states: request.extensions.hunyuangamecraft.camera_states
|
||||
camera_trajectory: request.extensions.hunyuangamecraft.camera_trajectory
|
||||
action_list: request.extensions.hunyuangamecraft.action_list
|
||||
action_speed_list: request.extensions.hunyuangamecraft.action_speed_list
|
||||
gt_latents: request.extensions.hunyuangamecraft.gt_latents
|
||||
conditioning_mask: request.extensions.hunyuangamecraft.conditioning_mask
|
||||
ltx2_cfg_scale_video: request.extensions.ltx2.cfg_scale_video
|
||||
ltx2_cfg_scale_audio: request.extensions.ltx2.cfg_scale_audio
|
||||
ltx2_modality_scale_video: request.extensions.ltx2.modality_scale_video
|
||||
|
||||
@@ -38,14 +38,17 @@ COOKBOOK_SOURCE_ROOTS = (
|
||||
# (e.g. black-forest-labs/FLUX.1-dev) and are grouped for documentation only.
|
||||
COOKBOOK_FAMILIES = {
|
||||
"wan",
|
||||
"turbodiffusion",
|
||||
"ltx2",
|
||||
"hunyuan",
|
||||
"cosmos",
|
||||
"kandinsky5",
|
||||
"kandinsky6",
|
||||
"flux",
|
||||
"glm_image",
|
||||
"zimage",
|
||||
"sd35",
|
||||
"minimax_h3",
|
||||
"longcat",
|
||||
"stable_audio",
|
||||
"mmaudio",
|
||||
"matrixgame",
|
||||
@@ -290,16 +293,12 @@ def cookbook_serving_profile(recipe: dict) -> dict:
|
||||
for language, (filename, install) in COOKBOOK_CLIENTS.items():
|
||||
path = ROOT_DIR / "examples/serving/clients" / filename
|
||||
text = path.read_text(encoding="utf-8")
|
||||
# Keep displayed snippets identical to the executable sources, with
|
||||
# only endpoint and alias substitutions.
|
||||
# Keep displayed snippets identical to executable sources for the
|
||||
# checked-in H3 profile, with only endpoint and alias substitutions.
|
||||
text = text.replace("http://127.0.0.1:8000/v1", base_url)
|
||||
text = text.replace('"fasth3"', json.dumps(model)).replace("${FASTVIDEO_MODEL:-fasth3}",
|
||||
"${FASTVIDEO_MODEL:-" + model + "}")
|
||||
clients[language] = {"source": path.relative_to(ROOT_DIR).as_posix(), "code": text, "install": install}
|
||||
# The playground router only serves H3 servers (see require_h3 in
|
||||
# fastvideo/entrypoints/openai/playground.py), so other families must not link to it.
|
||||
has_playground = recipe["family"] == "minimax_h3"
|
||||
compile_enabled = ((generator.get("engine") or {}).get("compile") or {}).get("enabled")
|
||||
return {
|
||||
"source": serving["source"],
|
||||
"install": serving["install"],
|
||||
@@ -308,10 +307,7 @@ def cookbook_serving_profile(recipe: dict) -> dict:
|
||||
"prepare": serving.get("prepare", ""),
|
||||
"model": model,
|
||||
"base_url": base_url,
|
||||
"playground_url": f"http://127.0.0.1:{port}/playground/" if has_playground else None,
|
||||
"audio": bool(serving.get("audio")),
|
||||
# None when the config leaves compilation at its default.
|
||||
"compile_enabled": compile_enabled,
|
||||
"playground_url": f"http://127.0.0.1:{port}/playground/",
|
||||
"health_command": f"curl --fail-with-body http://127.0.0.1:{port}/health",
|
||||
"hardware": hardware,
|
||||
"sampling": config["default_request"]["sampling"],
|
||||
|
||||
@@ -46,91 +46,6 @@ is the higher-quality FastH3.
|
||||
Recorded shapes and evidence live in the
|
||||
[support matrix](../../inference/support_matrix.md#apple-silicon-native-runtime).
|
||||
|
||||
## FastH3 V2 and Trim with INT6
|
||||
|
||||
The released sources are `FastVideo/FastVideo-FastH3-8-Step-V2` and
|
||||
`FastVideo/FastVideo-FastH3-Trim-8-Step`. Trim has 42 transformer blocks and
|
||||
rank-16 AdaLN. Both use eight denoising forwards, video/audio shifts of 10/3,
|
||||
VSA sparsity 0.8, a native NVFP4 text encoder, and the 26-layer light video VAE.
|
||||
Keep `fastvideo_inference.json` beside `transformer/`; conversion reads its
|
||||
schedule to build the AdaLN cache.
|
||||
|
||||
Convert the BF16 transformer to affine INT6 with its VSA gates:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-Trim-8-Step \
|
||||
--local-dir ./FastH3-Trim
|
||||
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_mlx.py \
|
||||
--model-root ./FastH3-Trim/transformer \
|
||||
--out ./FastH3-Trim-MLX \
|
||||
--formats "int6" --include-vsa
|
||||
```
|
||||
|
||||
For V2, use `FastVideo/FastVideo-FastH3-8-Step-V2` and separate source/output
|
||||
directories. Preconverted release snapshots use the same names with the
|
||||
`-MLX-INT6` suffix. Each snapshot includes the encoder in MLX layout, both VAEs,
|
||||
and the trained schedule, so it does not require a second encoder download.
|
||||
|
||||
The 36 GiB M4 Max release recipe uses phased placement. It loads the encoder,
|
||||
DiT, and decoders in turn. Use reference attention and native output geometry:
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
from fastvideo.mlx_runtime.minimax_h3_pipeline import MiniMaxH3MLXPipeline
|
||||
|
||||
root = Path("./FastH3-Trim")
|
||||
pipeline = MiniMaxH3MLXPipeline(
|
||||
model_root=root,
|
||||
mlx_dit_checkpoint="./FastH3-Trim-MLX/int6",
|
||||
conditioner_mode="nvfp4",
|
||||
resident=False,
|
||||
vae_dtype="fp16",
|
||||
metal_wired_limit_gib=27,
|
||||
)
|
||||
try:
|
||||
pipeline.generate(
|
||||
"A corgi news anchor sits behind a desk and gives a cheerful bark.",
|
||||
output_path="./outputs/trim-int6-corgi.mp4",
|
||||
width=832, height=480, num_frames=124, seed=1234,
|
||||
num_steps=8, vsa=True, vsa_sparsity=0.8, vsa_tile_size=64,
|
||||
vsa_impl="reference", vae_tile_height=256, vae_tile_width=256,
|
||||
)
|
||||
finally:
|
||||
pipeline.close()
|
||||
```
|
||||
|
||||
Set `FASTVIDEO_MLX_DQ_GEMM=1` before running this Python command. It selects the
|
||||
validated affine dequantization followed by dense matrix multiplication.
|
||||
124 frames at 24 fps is roughly five seconds. The recipe preserves all frames
|
||||
and the requested resolution.
|
||||
|
||||
### Native NVFP4 encoder and cache
|
||||
|
||||
The MLX conditioner reads the released packed NVFP4 weights without
|
||||
requantization. It retains the layers H3 reads and can cache the packed weights
|
||||
in MLX layout. The cache is written in a staging directory and published by a
|
||||
single rename. A cache hit changes storage layout, not encoder arithmetic.
|
||||
|
||||
MLX uses BF16 embeddings and FP32 activations; CUDA uses quantized activations.
|
||||
Generated video and audio must be reviewed before claiming cross-runtime
|
||||
quality parity. MLX 0.32.2 supports the required operator on Apple Silicon.
|
||||
|
||||
### Metal wired memory
|
||||
|
||||
MLX's allocation limit and wired-memory limit are separate. The optional
|
||||
`metal_wired_limit_gib` calls `mx.set_wired_limit` to keep selected Metal
|
||||
allocations in physical memory. It does not add RAM. An explicit request fails
|
||||
if the installed MLX build cannot apply it. `close()` restores the previous
|
||||
wired limit. The phased pipeline also sets a 30 GiB maximum allocator guideline
|
||||
and restores its previous value on close. Resident placement keeps the existing
|
||||
allocator limit so larger Macs can hold all components.
|
||||
|
||||
The tested 36 GiB M4 Max recipe uses 27 GiB and phased placement. Resident
|
||||
placement also requires capacity for all components and peak activations;
|
||||
wiring cannot make an oversized stack fit. Inspect `mx.device_info()` before
|
||||
choosing a limit on another Mac.
|
||||
|
||||
## Hardware
|
||||
|
||||
- FastMetal 1.3B and 5B: 16 GB unified memory and up
|
||||
|
||||
@@ -144,9 +144,6 @@ for which models are practical on the GB10, what makes them faster, and what
|
||||
won't help on this hardware (and why) — so you don't spend a night tuning knobs
|
||||
that can't move here.
|
||||
|
||||
For the eight-forward FastH3 V2 NVFP4 stack with a trimmed encoder and light
|
||||
VAE, use the [one-Spark resident recipe](spark_performance.md#fasth3-v2-nvfp4-on-one-spark).
|
||||
|
||||
Two Sparks with QSFP cables: [Pair two NVIDIA DGX Sparks](spark_pair.md) for
|
||||
one FastH3 clip across both GPUs (`sp_size=2` over Ray). Copy-paste commands
|
||||
for one or two Sparks also live on the
|
||||
|
||||
@@ -149,29 +149,6 @@ fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair.yam
|
||||
|
||||
Stop the cluster when you are done: `ray stop` on both nodes.
|
||||
|
||||
## Released V2 and Trim NVFP4 stacks
|
||||
|
||||
The public eight-forward V2 and Trim stacks include the NVFP4 encoder and
|
||||
lightweight video VAE. Use the environment above plus the release kernel
|
||||
settings, then run one of these configs from the head:
|
||||
|
||||
```bash
|
||||
export FASTVIDEO_MINIMAX_H3_FUSIONS=all FASTVIDEO_NVFP4_MM_BACKEND=cutlass
|
||||
export FASTVIDEO_H3_VAE_TILE_BATCH=1 FASTVIDEO_VSA_TRITON=1
|
||||
export FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0
|
||||
export FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1
|
||||
fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair_v2_nvfp4.yaml
|
||||
# Or basic_fasth3_spark_pair_pruned_nvfp4.yaml for FastH3 Trim.
|
||||
```
|
||||
|
||||
Both configs use their released Hugging Face model paths, `h3_dit_vsa`,
|
||||
832x480, 124 frames, seed 1234, VSA 0.8 and 64-token tiles. All components
|
||||
stay resident; lazy/sequential loading and compilation are disabled for
|
||||
these compact stacks. The older BF16/Preview memory guidance below applies
|
||||
to those larger stacks. Set 1344x768 for native 768p with the same 124 frames.
|
||||
See [the resident release recipe](spark_performance.md#fasth3-v2-nvfp4-on-one-spark)
|
||||
for checkpoint contents and the timing protocol.
|
||||
|
||||
## FastH3 frame counts
|
||||
|
||||
H3 is 24 fps. Legal `num_frames` values are `17n+5`. The pipeline rejects
|
||||
|
||||
@@ -161,16 +161,14 @@ is power-cycled. To avoid it:
|
||||
on: "CPU" offload uses the same unified RAM. Multi-GPU FSDP sharding remains
|
||||
available because it partitions weights without parking them in a separate
|
||||
host pool.
|
||||
- **Older MiniMax H3 / FastH3 bf16 weights** need deferred loading on one GB10.
|
||||
The full Qwen3-VL conditioner is tens of gigabytes of BF16. If the DiT and
|
||||
VAEs load while that encoder is still resident, the process can be killed by
|
||||
`earlyoom`. On unified memory, `lazy_module_load` auto-enables and owns that
|
||||
- **MiniMax H3 / FastH3** still needs deferred loading on one GB10. The Qwen3-VL
|
||||
conditioner is tens of gigabytes of BF16. If the DiT and VAEs load while that
|
||||
encoder is still resident, the process is a typical `earlyoom` kill (Python is
|
||||
preferred). On unified memory, `lazy_module_load` auto-enables and owns that
|
||||
split (encoder, then DiT, then VAE; DiT can drop before decode). Sequential
|
||||
load is the H3-only fallback when lazy is off. Keep deferred loading for
|
||||
those older checkpoints. The trimmed NVFP4 encoder and light VAE in the
|
||||
[V2 resident recipe](#fasth3-v2-nvfp4-on-one-spark) are a different memory
|
||||
profile. Geometry scalars come from checkpoint `config.json`, not live
|
||||
weights. See [Offloading](../../inference/offloading.md).
|
||||
load is the H3-only fallback when lazy is off; do not pass
|
||||
`--no-lazy-module-load` here. Geometry scalars come from checkpoint
|
||||
`config.json`, not live weights. See [Offloading](../../inference/offloading.md).
|
||||
- **FastH3 TAEH3** (`--video-decode-backend taeh3`) is an opt-in preview decoder.
|
||||
T2VA never materializes the 9.7 GiB video VAE (DiT still loads after Qwen via
|
||||
sequential start). On this box, alpine 768×1344×124 decoded in **2.4 s** versus
|
||||
@@ -209,121 +207,6 @@ A few things that surprise people on this box (beyond the memory notes above):
|
||||
`Released MiniMax-H3 text encoder after conditioning` before
|
||||
`Loading MiniMax-H3 denoise modules`).
|
||||
|
||||
## FastH3 V2 NVFP4 on one Spark
|
||||
|
||||
This recipe uses the full V2 eight-forward transformer, the 50-layer NVFP4
|
||||
Qwen3-VL encoder, and the light H3 video VAE. Its configuration keeps all
|
||||
three resident on one GB10. This stack fits in the Spark's unified memory;
|
||||
benchmark your installed runtime and review the clips before publishing a
|
||||
speed claim. The earlier bf16 H3 memory guidance above concerns a larger checkpoint.
|
||||
|
||||
Install FastVideo following [the Spark install guide](spark.md). The released
|
||||
repositories are complete inference stacks:
|
||||
|
||||
| Model | Repository | Packed DiT profile |
|
||||
|---|---|---|
|
||||
| FastH3 V2 | `FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer` | `h3_dit_vsa` |
|
||||
| FastH3 Trim | `FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4` | `h3_dit_vsa` |
|
||||
|
||||
Each ships its own trained schedule, NVFP4 transformer and 50-layer NVFP4
|
||||
text encoder, lightweight 26-layer video VAE with the INT8-weight overlay,
|
||||
and audio VAE. Download the complete repository; the runtime selects these
|
||||
components from its model index. The released Trim transformer also packs
|
||||
attention and VSA gates, unlike the earlier FFN-only pruned export. Neither
|
||||
release requires local checkpoint conversion or a separate encoder download.
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer
|
||||
hf download FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4
|
||||
```
|
||||
|
||||
Keep `fastvideo_inference.json` with the transformer if you stage the stack
|
||||
in a local directory. It declares the trained eight-forward ladder and
|
||||
video/audio shifts of 10/3. The recipes use `num_inference_steps: 9` for
|
||||
nine sigma points and eight DiT forwards.
|
||||
|
||||
On GB10 with FlashInfer 0.6.18, FastVideo fences activation quantization before
|
||||
releasing its padded input. Without this completion fence, identical H3
|
||||
requests produced different DiT latents and occasionally corrupt video.
|
||||
The fence applies to `sm_121`; other architectures retain asynchronous execution.
|
||||
|
||||
Run `examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml` from the
|
||||
repository root. It uses 832x480, 124 frames and seed 1234, VSA sparsity 0.8 with
|
||||
64-token tiles, and the light H3 VAE through the `h3-vae` decode backend.
|
||||
It does not use frame dropping or spatial upscaling.
|
||||
|
||||
```bash
|
||||
FASTVIDEO_MINIMAX_H3_FUSIONS=all \
|
||||
FASTVIDEO_NVFP4_MM_BACKEND=cutlass \
|
||||
FASTVIDEO_H3_VAE_TILE_BATCH=1 \
|
||||
FASTVIDEO_VSA_TRITON=1 FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 \
|
||||
FASTVIDEO_STAGE_LOGGING=1 \
|
||||
nice -n 19 fastvideo generate \
|
||||
--config examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml
|
||||
```
|
||||
|
||||
The defaults produce a roughly five-second clip at 24 fps. For native 768p,
|
||||
set `--request.sampling.width 1344` and `--request.sampling.height 768`,
|
||||
keeping 124 frames. For the separate ten-second setting, use 243 frames.
|
||||
H3 permits frame counts of `17n+5`; 124 is the nearest legal count above
|
||||
five seconds and 243 is the nearest above ten seconds.
|
||||
|
||||
For Trim, use `basic_fasth3_spark_pruned_nvfp4.yaml` with the same environment.
|
||||
Both recipes keep the encoder, DiT and VAEs resident, disable compilation,
|
||||
and use `h3_dit_vsa`. On two Sparks, use the corresponding
|
||||
`basic_fasth3_spark_pair_{pruned,v2}_nvfp4.yaml` after following the
|
||||
[pair setup guide](spark_pair.md). The pair uses SP2/TP1 and parallel VAE
|
||||
gathering, with tile batch 1 on each worker.
|
||||
|
||||
For release timing, create one generator per model/resolution. Run one
|
||||
untimed ceramics warmup, then two timed ceramics calls and two timed harbor
|
||||
calls in that same process, using the exact release prompt strings, seed
|
||||
1234 and the settings above. Measure each `generate()` call through finished
|
||||
MP4 output and report the median of the two timed calls per prompt. Keep the
|
||||
warmup excluded. Record the model revision, code commit, command, environment
|
||||
and peak-memory scope with the results. Review every clip's video and audio
|
||||
before publishing a quality or speed claim.
|
||||
|
||||
The benchmark helper requires a local stack so it can validate the schedule
|
||||
before loading. For example:
|
||||
|
||||
```bash
|
||||
hf download FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer \
|
||||
--local-dir ./FastH3-V2-Consumer
|
||||
python examples/inference/basic/benchmark_fasth3_spark_nvfp4.py \
|
||||
--config examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml \
|
||||
--model-path ./FastH3-V2-Consumer --frames 124 \
|
||||
--prompts /path/to/benchmark_prompts.json \
|
||||
--output-dir outputs/fasth3_spark_v2_nvfp4/benchmark-124
|
||||
```
|
||||
|
||||
Set the environment from the generation command above before benchmarking.
|
||||
The prompt JSON must contain `latency-ceramics-005` and
|
||||
`latency-harbor-005`. Pass `--width 1344 --height 768` for the native 768p
|
||||
protocol. Use the Trim repository and config for its corresponding run.
|
||||
|
||||
### Released model measurements
|
||||
|
||||
The released Trim stack at revision `cae9ceb6feefe77d34a56640782cda3909363f19`
|
||||
completed the native 832x480, 124-frame protocol on one Spark, seed 1234.
|
||||
The tested code is `6ccdbc761e6b854003f472e39b826c06cb54de60`, using the
|
||||
resident recipe above. Medians exclude warmups and cover finished MP4 output.
|
||||
|
||||
| Released model | Resolution | One Spark, ceramics / harbor | Two Sparks |
|
||||
|---|---|---:|---|
|
||||
| FastH3 Trim NVFP4 | 832x480 | 124.090 / 127.519 s | Pending |
|
||||
| FastH3 Trim NVFP4 | 1344x768 | Review pending | Pending |
|
||||
| FastH3 V2 NVFP4-Consumer | 832x480 | Repeatability review pending | Pending |
|
||||
| FastH3 V2 NVFP4-Consumer | 1344x768 | Review pending | Pending |
|
||||
|
||||
The completed Trim 480p batch used one extra untimed harbor warmup in the
|
||||
same process, six calls total. Both warmups are excluded from the medians.
|
||||
Each prompt's three decoded videos and audio streams match exactly, and
|
||||
sampled frames are coherent. These checks do not establish BF16 parity,
|
||||
speech accuracy or lip sync. Pending cells are not measured substitutes
|
||||
from older checkpoints.
|
||||
|
||||
## Reproduce these numbers
|
||||
|
||||
Two scripts under `examples/inference/optimizations/` reproduce the claims on
|
||||
|
||||
@@ -232,7 +232,8 @@ Dataclass carrying all pipeline state between stages. Key field groups:
|
||||
- **Scheduler**: `timesteps`, `num_inference_steps`, `guidance_scale`,
|
||||
`sigmas`.
|
||||
- **Task-specific**: `mouse_cond`/`keyboard_cond` (Matrix-Game 2.0), `pose`
|
||||
(HYWorld), `c2ws_plucker_emb` (LingBotWorld).
|
||||
(HYWorld), `camera_states` (GameCraft), `c2ws_plucker_emb`
|
||||
(LingBotWorld).
|
||||
- **Output**: `output: Tensor | None`.
|
||||
- **Logging**: `logging_info: PipelineLoggingInfo`.
|
||||
|
||||
@@ -255,6 +256,7 @@ Standard stages (typical execution order):
|
||||
| `DecodingStage` | `stages/decoding.py` | Decodes latents to video via VAE |
|
||||
|
||||
Specialized variants: `CausalDenoisingStage`, `LTX2DenoisingStage`,
|
||||
`LongCatDenoisingStage`, `GameCraftDenoisingStage`,
|
||||
`HYWorldDenoisingStage`, `MatrixGame2CausalDenoisingStage`,
|
||||
`SRDenoisingStage`, `LTX2AudioDecodingStage`, `SD35ConditioningStage`,
|
||||
`LTX2TextEncodingStage`, `LTX2LatentPreparationStage`.
|
||||
|
||||
@@ -74,130 +74,6 @@ This documents execution support for the published checkpoint. It is not a
|
||||
quality claim: compare video/audio output against base MiniMax-H3 on your own
|
||||
prompts before adopting it.
|
||||
|
||||
## Ref2VA PDD students
|
||||
|
||||
A Parallel Decoding Distillation (PDD) student widens the transformer's two
|
||||
output projections to `pdd_steps` heads, one per interval of a fixed fine
|
||||
time grid on `[0, 0.999]`. Each transformer forward fuses a block of
|
||||
consecutive heads into their integration-weighted mean, and an ordinary Euler
|
||||
step over the block's two node sigmas applies it. The FastH3 OmniRef PDD-8
|
||||
student is a Ref2VA (`transformer_ref`) student with 32 heads, sampled in
|
||||
eight blocks of four.
|
||||
|
||||
Its export carries only what distillation changed:
|
||||
|
||||
| Path | Contents |
|
||||
| --- | --- |
|
||||
| `transformer_ref/` | Widened `proj_out` and `audio_proj_out`, trained VSA compression gates; `config.json` records `pdd_steps` |
|
||||
| `scheduler/`, `audio_scheduler/` | Video and audio shifts (12 and 3) |
|
||||
| `fastvideo_inference.json` | The sampling contract below |
|
||||
| `modular_model_index.json` | The Diffusers manifest |
|
||||
|
||||
The text encoder, tokenizer, processor, and both VAEs are base MiniMax-H3's.
|
||||
`basic_fasth3_omniref_pdd.py` composes the two into one local directory of
|
||||
symlinks and runs the recipe the contract records. `--model-path` is the
|
||||
export, as a local directory or a Hugging Face repo id. The base comes from
|
||||
the revision the contract pins in `base_model_revision`, or from
|
||||
`--base-model-path`:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_fasth3_omniref_pdd.py \
|
||||
--model-path <local export directory or Hugging Face repo id> \
|
||||
--video reference.mp4 --image character.png \
|
||||
--prompt 'The dancer from the video performs the routine in the pictured outfit.' \
|
||||
--height 480 --width 832 --num-frames 124 \
|
||||
--output outputs/fasth3-omniref-pdd
|
||||
```
|
||||
|
||||
References are ordered: pass `--image`, `--video`, and `--audio` in the order
|
||||
the prompt refers to them. At least one image or video is required.
|
||||
|
||||
A PDD export uses `fasth3-inference-contract-v1` with PDD fields in place of
|
||||
the DMD ladder:
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": "fasth3-inference-contract-v1",
|
||||
"model_type": "ref2va",
|
||||
"transformer_component": "transformer_ref",
|
||||
"pdd_steps": 32,
|
||||
"pdd_step_indices": [0, 4, 8, 12, 16, 20, 24, 28, 32],
|
||||
"num_inference_steps": 8,
|
||||
"transformer_forwards": 8,
|
||||
"grid_max_t": 0.999,
|
||||
"video_scheduler_shift": 12.0,
|
||||
"audio_scheduler_shift": 3.0,
|
||||
"guidance_scale": 1.0,
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"vsa_sparsity": 0.9,
|
||||
"vsa_tile_size": 128,
|
||||
"vsa_ref_policy": "p2_multi_region",
|
||||
"vsa_ref_keep_rate": 0.1
|
||||
}
|
||||
```
|
||||
|
||||
The export also records `schema`, `conditioning`, and `base_model_revision`,
|
||||
the base snapshot it was distilled against, as `hf://<repo id>@<revision>`
|
||||
(for example `hf://MiniMaxAI/MiniMax-H3@<commit>`); any other form is an
|
||||
error. FastVideo reads the file once, when the run's arguments are built and
|
||||
before any weights load. The file must carry exactly these fields: a missing
|
||||
or unknown field is an error. Fields that repeat a value stored elsewhere must
|
||||
equal it: `pdd_steps` equals `transformer_ref/config.json`, the shifts equal
|
||||
the scheduler configs, and `num_inference_steps` and `transformer_forwards`
|
||||
equal the block count of `pdd_step_indices`, a strictly increasing partition
|
||||
of the grid from 0 to `pdd_steps`. Only `MiniMaxH3Ref2VAModularPipeline` runs
|
||||
the export. A `transformer_ref` whose `config.json` sets `pdd_steps` without
|
||||
this file is an error.
|
||||
|
||||
Each sampling setting has one source, the file:
|
||||
|
||||
| Setting | Run leaves it unset | Run sets a different value |
|
||||
| ------------------------------- | ---------------------- | ---------------------------------------------- |
|
||||
| `pdd_step_indices` | Taken from the file | Error |
|
||||
| `num_inference_steps` (request) | Set to the block count | Error |
|
||||
| `attention_backend` | Taken from the file | Error, including `FASTVIDEO_ATTENTION_BACKEND` |
|
||||
| `VSA_tile_size` | Taken from the file | Error |
|
||||
| `VSA_sparsity` | Taken from the file | The run's value, with a warning |
|
||||
| `vsa_ref_keep_rate` | Taken from the file | The run's value, with a warning |
|
||||
|
||||
A request leaves `num_inference_steps` unset only when it is parsed from a
|
||||
mapping or a config file. A `GenerationRequest` built in Python counts every
|
||||
field as set, so it must pass the block count. The trained compression gates
|
||||
of `transformer_ref` load only under `VIDEO_SPARSE_ATTN_H3`, so the attention
|
||||
backend cannot change. `dmd_denoising_steps` must stay unset.
|
||||
|
||||
A MiniMax-H3 checkpoint without PDD fields, such as a DMD export, whose
|
||||
transformer carries VSA compression gates (`to_gate_compress` weights) also
|
||||
runs only with `VIDEO_SPARSE_ATTN_H3`. The pipeline reads the shard index (or
|
||||
the safetensors headers) and rejects any other backend, including automatic
|
||||
selection, before it loads any component.
|
||||
|
||||
### Reference-video sparsity
|
||||
|
||||
Every PDD contract sets `"vsa_ref_policy": "p2_multi_region"`, the only value FastVideo accepts, which means:
|
||||
`VIDEO_SPARSE_ATTN_H3` tiles every reference video as its own sparse region,
|
||||
in place in the packed sequence. Each video query keeps `vsa_ref_keep_rate` of
|
||||
every reference video's tiles and `1 - VSA_sparsity` of the target video's
|
||||
tiles. Text, audio, and image references stay dense. With `VSA_sparsity` 0,
|
||||
every region is dense and the reference keep rate has no effect. Checkpoints
|
||||
other than PDD students keep every conditioning row dense.
|
||||
|
||||
### Hardware
|
||||
|
||||
The contract's 128-token tiles, `(4, 4, 8)`, run only on the sm_100a/sm_103a
|
||||
CUDA block-sparse kernel (B200, B300, GB200, GB300) of a fastvideo-kernel
|
||||
build with the Blackwell VSA extension. There is no Triton fallback for tile
|
||||
128: the backend raises instead. Tile 128 runs eagerly. Regional compile
|
||||
requires 64-token tiles. With `--num-gpus` above 1 the example shards the DiT
|
||||
across the GPUs (FSDP) and splits the sequence across them.
|
||||
|
||||
On GB200, a 480x832, 124-frame request fits on one GPU: the eight forwards
|
||||
take about 54 s, and device memory in use peaks near 103 GiB. A 768x1344,
|
||||
345-frame request with `--num-gpus 4` takes about 36 s for the eight
|
||||
forwards. With the DiT sharded, peak allocated memory is about 66 GiB per GPU
|
||||
(about 95 GiB in use), against 83 GiB (103 GiB) with a full DiT copy on each
|
||||
GPU; the output is identical. Model loading and decoding come on top of this.
|
||||
|
||||
## Apple Silicon
|
||||
|
||||
`mlx_fasth3.py` stays on FastH3 V1 and its uniform AdaLN cache.
|
||||
|
||||
@@ -1,260 +0,0 @@
|
||||
# FastH3 NVFP4 on RTX PRO 6000 (sm_120)
|
||||
|
||||
This page covers serving the FastH3 NVFP4 checkpoints on one RTX PRO 6000
|
||||
Blackwell (sm_120, 96 GB) with every component resident: the NVFP4 text
|
||||
encoder, the NVFP4 denoiser and an INT8 light VAE. It also records what was
|
||||
measured and tried along the way. Every switch is opt-in and defaults to the
|
||||
existing behavior.
|
||||
|
||||
## Results
|
||||
|
||||
All results are for one RTX PRO 6000 (Modal), a 10.1 s clip at 1344x768
|
||||
(243 frames, 73.6k packed tokens), the light INT8 VAE and warm runs.
|
||||
|
||||
| Checkpoint | Denoise | Video decode | End to end | Peak memory |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| [`FastH3-4-step-Preview-v1-VSA-DataFree-NVFP4`](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree-NVFP4) (4 forwards, VSA 0.9) | 30.1–31.0 s | 9.0 s | **41.2 / 42.4 s** | 79.6 GB |
|
||||
| [`FastH3-8-Step-V2-NVFP4`](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4) (8 forwards, VSA 0.8) | 75.6 s | 9.0 s | **86.5 s** | — |
|
||||
|
||||
These end-to-end runs predate the warp-skip kernel change below, which cuts
|
||||
sparse attention by a further 1.5x, so they are upper bounds. Conditioning
|
||||
takes about 0.12 s, frame post-processing plus MP4 writing about 0.9 s, and
|
||||
the first clip at a new shape about 200 s (VAE compile).
|
||||
|
||||
At 480p (124 frames, 15.1k tokens) V2 8-step denoises in 12.5 s on a cold
|
||||
run, down from 16.8 s warm before this work.
|
||||
|
||||
## Usage
|
||||
|
||||
### 1. Convert the checkpoint
|
||||
|
||||
The published NVFP4 checkpoints use ModelOpt's unified Hugging Face layout.
|
||||
`convert_minimax_h3_modelopt_nvfp4_dit.py` repacks it into FastVideo's packed
|
||||
export, `transformer/nvfp4_weights.safetensors`. The calibrated FFN weights
|
||||
and scales are carried over bit for bit and only the scale bytes are
|
||||
swizzled. Optionally it also quantizes the BF16 attention projections and the
|
||||
VSA compression gates:
|
||||
|
||||
```bash
|
||||
python scripts/checkpoint_conversion/convert_minimax_h3_modelopt_nvfp4_dit.py \
|
||||
--src /models/FastH3-8-Step-V2-NVFP4/transformer \
|
||||
--dst /models/fasth3-v2-fv/transformer \
|
||||
--quantize-attention --quantize-gate
|
||||
```
|
||||
|
||||
Every exported linear is probed through the loader's `mm_fp4` path against a
|
||||
BF16 matmul with the dequantized weight. Conversion refuses to write when any
|
||||
error exceeds `--max-probe-error` (default 0.3; the converted V2 and 4-step
|
||||
checkpoints probe at 0.14). The rest of the model folder (text encoder,
|
||||
VAEs, schedulers, `fastvideo_inference.json`) is used as-is; the
|
||||
NVFP4 text encoder comes from
|
||||
`convert_minimax_h3_text_encoder_nvfp4.py`.
|
||||
|
||||
The flags select the `layer_profile` to load the result with:
|
||||
|
||||
| Flags | Linears in NVFP4 | `layer_profile` |
|
||||
| --- | --- | --- |
|
||||
| *(none)* | FFN `fc_in`/`fc_out` | `h3_dit_ffn` |
|
||||
| `--quantize-attention` | + attention `to_{q,k,v,out}` | `h3_dit` |
|
||||
| `--quantize-attention --quantize-gate` | + VSA `to_gate_compress` | `h3_dit_vsa` |
|
||||
|
||||
### 2. Generate
|
||||
|
||||
```python
|
||||
import os
|
||||
os.environ.update({
|
||||
"FASTVIDEO_H3_VSA_FP4": "1", # sparse FP4 attention
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all", # Triton norm/modulate/RoPE/SwiGLU fusions
|
||||
"FASTVIDEO_NVFP4_MM_BACKEND": "cutlass", # see "FP4 GEMM backend" below
|
||||
"FASTVIDEO_H3_VAE_TILE_BATCH": "28", # one decoder call per 1344x768 tile grid
|
||||
})
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_config({
|
||||
"model_path": "/models/fasth3-v2-fv",
|
||||
"engine": {
|
||||
"num_gpus": 1,
|
||||
"quantization": {"transformer_quant": "NVFP4", "layer_profile": "h3_dit_vsa"},
|
||||
"compile": {"enabled": False, "vae_enabled": True},
|
||||
},
|
||||
"pipeline": {"experimental": {"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"VSA_sparsity": 0.8, "VSA_tile_size": 64}},
|
||||
})
|
||||
generator.generate({"prompt": "...", "sampling": {"height": 768, "width": 1344, "num_frames": 243,
|
||||
"num_inference_steps": 9}})
|
||||
```
|
||||
|
||||
Use `VSA_sparsity` 0.9 and `num_inference_steps` 5 for the 4-step checkpoint
|
||||
(see its `fastvideo_inference.json`). Frame counts must be `17n + 5`: 243
|
||||
frames is the closest to 10 s.
|
||||
|
||||
## What changed
|
||||
|
||||
### Block-sparse FP4 attention for VSA tiles (`fastvideo-kernel`)
|
||||
|
||||
SageAttention3's sm_120 FP4 kernel (`attn_qat_infer`) gains a block-sparse
|
||||
forward, `fwd_sparse`, exposed as `sageattn_blackwell_sparse` (head-major
|
||||
inputs) and `sageattn_blackwell_sparse_bshd` (sequence-major inputs, quantized
|
||||
in place without a transpose). `vsa_tile_mask_to_fp4_blocks` turns a VSA tile
|
||||
mask into the kernel's lists:
|
||||
|
||||
- **Block lists.** Query block `m` visits only the 128-token KV blocks in
|
||||
`q2k_idx[b, h, m, :q2k_num[b, h, m]]`.
|
||||
- **Quadrant masks for 64-token tiles.** The kernel computes on 128x128
|
||||
blocks, but V2 and the 4-step preview use 64-token VSA tiles. Each listed
|
||||
block carries a 4-bit `q2k_quad` (one bit per 64x64 quadrant). The kernel
|
||||
masks unselected quadrants to `-inf`, so the result is exactly VSA's tile-64
|
||||
semantics.
|
||||
- **Valid counts per 64-column half.** `kv_valid` gives the valid tokens in
|
||||
each 64-column half, so partially filled tiles can pad mid-block.
|
||||
- **Warp-level skipping.** Each MMA warp owns 16 query rows and so sits
|
||||
inside one 64-row half. A warp skips a listed block that its half did not
|
||||
select, and the P·V chunk of a key half it did not select. Masked scores
|
||||
contribute exactly zero, and the warp sharing its tensor-core partition runs
|
||||
faster meanwhile. This recovers most of the work that pairing 64-token tiles
|
||||
into 128-token blocks adds.
|
||||
- **First-visited block.** Lists run in descending block order because the
|
||||
kernel visits them last entry first. Block 0 (the first prefix tile, which
|
||||
VSA-H3's exempt mode gives every query) is therefore visited first, so every
|
||||
row starts from a finite running max. `validate=True` checks this. The model
|
||||
integration uses only exempt mode.
|
||||
|
||||
The dense and sparse entry points also stop allocating `delta_s`. With Q
|
||||
smoothing off (the default), each call used to allocate and zero a
|
||||
`[B, H, L/128, L]` fp32 tensor: 9.5 GB at 73k tokens. Its int32 batch stride
|
||||
also overflowed the TMA descriptor ("Failed to initialize the TMA descriptor
|
||||
1", then an illegal instruction), so FP4 attention could not run 10 s 1344x768
|
||||
clips at all. A cached `[B, H, 1, L]` zero row read with `per_block_mean=False`
|
||||
replaces it, and the outputs are bit-identical.
|
||||
|
||||
Correctness (`fastvideo-kernel/tests/test_attn_qat_infer_sparse.py`, RTX PRO
|
||||
6000):
|
||||
|
||||
| Layout | Error vs token-masked fp32 | Dense FP4 floor |
|
||||
| --- | ---: | ---: |
|
||||
| 64-token tiles, odd count, partial tiles | 0.190 | 0.191 |
|
||||
| 64-token tiles, even count | 0.191 | 0.193 |
|
||||
| 256-token tiles, partial tile | 0.195 | 0.196 |
|
||||
|
||||
Errors are relative L2. The sparse kernel sits exactly at the FP4 noise
|
||||
floor; random Gaussian inputs make that floor large. Full block lists
|
||||
reproduce the dense kernel bit for bit.
|
||||
|
||||
### Model integration (`FASTVIDEO_H3_VSA_FP4=1`)
|
||||
|
||||
`fastvideo/models/dits/minimax_h3_vsa_fp4.py` replaces only the attention core
|
||||
of `MiniMaxH3Attention`. VSA-H3's tile pooling, top-k mask, exempt prefix and
|
||||
gated compression branch are unchanged. Per block, it:
|
||||
|
||||
1. Gathers the attention input into tile order once (one `hidden_size`-wide
|
||||
pass). Pad rows stay zero, so the q/k/v pad rows are exactly zero through
|
||||
the bias-free projections, RMSNorm and RoPE.
|
||||
2. Quantizes that input once and shares it between `to_q`, `to_k` and `to_v`.
|
||||
NVFP4 activations use a unit global scale, so this is exact.
|
||||
3. Applies QK-norm and RoPE with tile-ordered `cos`/`sin`, computed once per
|
||||
step.
|
||||
4. Runs the sparse FP4 kernel on sequence-major tensors and gathers the output
|
||||
back to packed order before `to_out`.
|
||||
|
||||
This replaces the generic path's concat, four tile scatters and three
|
||||
transposes. The route applies only to no-grad, non-compiled, single
|
||||
sequence-parallel-rank calls in exempt mode; everything else keeps the
|
||||
existing path.
|
||||
|
||||
Two smaller pieces ship alongside it:
|
||||
|
||||
- **Packed gate check.** With `--quantize-gate`, `to_gate_compress` loses its
|
||||
BF16 weight, so the gate-activity check reads the packed E2M1 bytes instead.
|
||||
- **Compiled residual.** With `FASTVIDEO_MINIMAX_H3_FUSIONS` enabled, each
|
||||
block's final `hidden + gate[indices] * ffn_out` runs as one compiled op
|
||||
instead of materializing the gathered gate.
|
||||
|
||||
### FP4 GEMM backend (`FASTVIDEO_NVFP4_MM_BACKEND`)
|
||||
|
||||
FlashInfer's `mm_fp4(backend="auto")` picks a kernel about 2x slower than
|
||||
`cutlass` or `cudnn` on sm_120 once activations reach tens of thousands of
|
||||
rows. At 15k rows all three match.
|
||||
|
||||
| Linear | 73.6k rows: `auto` | 73.6k rows: `cutlass` | 73.6k rows: `cudnn` | 15.1k rows: `auto` |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| `to_q` (5376→7168) | 7.96 ms | 3.96 ms | 4.24 ms | 0.88 ms |
|
||||
| `to_out` (7168→5376) | 9.07 ms | 4.09 ms | 4.27 ms | 0.92 ms |
|
||||
| `fc_in` (5376→28672) | 21.54 ms | 16.59 ms | 16.17 ms | 3.01 ms |
|
||||
| `fc_out` (14336→5376) | 18.06 ms | 8.09 ms | 8.43 ms | 1.65 ms |
|
||||
|
||||
### Batched VAE tile decode (`FASTVIDEO_H3_VAE_TILE_BATCH`)
|
||||
|
||||
The H3 video VAE decodes 256-pixel spatial tiles one at a time. A 1344x768
|
||||
clip is a 4x7 grid per temporal chunk, so a 10 s clip is roughly 400 small
|
||||
decoder calls. The ViT decoder treats batch entries independently, so
|
||||
`FASTVIDEO_H3_VAE_TILE_BATCH=N` decodes up to `N` equal-shaped tiles per call;
|
||||
28 covers a full 1344x768 grid. The decoded tiles are the same as per-tile
|
||||
decoding.
|
||||
|
||||
## Per-block measurements
|
||||
|
||||
One H3 transformer block (hidden 5376, 56 heads, FFN 14336), RTX PRO 6000:
|
||||
|
||||
| Component | 480p, 124 f (15.1k tokens) | 768p, 243 f (73.6k tokens) |
|
||||
| --- | ---: | ---: |
|
||||
| VSA Triton BF16 attention (kernel + pooling/mask) | 14.8 ms | 180.1 ms |
|
||||
| Tile scatter of q/k/v/gate + gather (generic path) | 2.7 ms | 12.7 ms |
|
||||
| Dense FP4 attention (SageAttention3) | 10.5 ms | 211.9 ms |
|
||||
| Sparse FP4, quadrant masks | 7.9 ms | 123.4 ms |
|
||||
| Sparse FP4, quadrant masks + warp skip | — | **83.0 ms** (VSA 0.8) / **49.3 ms** (VSA 0.9) |
|
||||
| Dense BF16 SDPA | 18.1 ms | — |
|
||||
| Modulation: eager / fused / compiled | 4.07 / 1.42 / 0.58 ms | 20.3 / 7.0 / 2.8 ms |
|
||||
| SwiGLU: eager / fused | 1.53 / 0.88 ms | 7.37 / 4.24 ms |
|
||||
| QK-norm + RoPE: eager / fused | 4.64 / 1.12 ms | 22.4 / 5.2 ms |
|
||||
|
||||
Before this work, a 480p block cost about 40 ms: 17.7 ms of attention, 12 ms
|
||||
of linears and 10 ms of eager elementwise ops. Over 50 blocks that is 2.0 s
|
||||
per step, which matches the measured 2.1 s.
|
||||
|
||||
Block density each kernel granularity computes at 768p (fraction of the
|
||||
dense attention). "Selected" is what VSA needs; the other columns are what
|
||||
each block shape computes:
|
||||
|
||||
| VSA sparsity | Selected (64x64) | 128x128 blocks | 64-row x 128 | 128 x 64-col |
|
||||
| --- | ---: | ---: | ---: | ---: |
|
||||
| 0.8 | 0.222 | 0.434 | 0.317 | 0.313 |
|
||||
| 0.9 | 0.125 | 0.254 | 0.180 | 0.177 |
|
||||
|
||||
## What was tried and not shipped
|
||||
|
||||
- **Dense FP4 attention for VSA-trained students.** `ATTN_QAT_INFER` does not
|
||||
build `to_gate_compress`. A VSA-distilled checkpoint such as V2 carries
|
||||
trained gates, so the strict loader refuses it ("Parameter
|
||||
...to_gate_compress.weight not found"). Dense FP4 attention is also slower
|
||||
than sparse FP4 at 768p (212 vs 83 ms per block).
|
||||
- **Multi-GPU (Ulysses) FP8 exchange.** On 8x RTX PRO 6000 (PCIe only), NCCL
|
||||
all-to-all moves about 21 GB/s per GPU, with NCCL P2P on or off. A BF16
|
||||
q/k/v/gate exchange at 73.6k tokens therefore costs 24 ms per block, and the
|
||||
attention output another 6.6 ms; with FP4 payloads q/k/v/gate drop to 7.2 ms.
|
||||
The branch `h3-sm120-experimental` keeps a sequence-parallel path that:
|
||||
- sends q/k/v as FP8 with one scale per token and head;
|
||||
- never sends the VSA gate, applying it on each rank after a small
|
||||
all-gather of the per-tile compression output;
|
||||
- returns the attention output as FP8.
|
||||
|
||||
It is estimated at about 18–20 s per 10 s clip for V2 8-step on 8 GPUs. It
|
||||
has not executed yet (8-GPU capacity was unavailable), so it is not part of
|
||||
this change. The same branch holds the Modal drivers behind every number on
|
||||
this page.
|
||||
- **64-row query blocks.** The kernel traits allow `kBlockM = 64`, which would
|
||||
remove the query-side pairing waste. Warp-level skipping recovers most of
|
||||
that waste without a second kernel instantiation, so it was not built.
|
||||
|
||||
## Known limitations
|
||||
|
||||
- **End-to-end quality.** The kernel matches a masked reference at the FP4
|
||||
noise floor. Generated videos have not yet been A/B-compared against the
|
||||
BF16 Triton VSA path on the H3 audio/video metrics.
|
||||
- **Activation scales.** The packed export drops ModelOpt's calibrated static
|
||||
`input_scale` and quantizes activations with a unit global scale and dynamic
|
||||
per-16 block scales, as FastVideo's NVFP4 linears do elsewhere.
|
||||
- **Decode cost.** Video decode (9 s at 10 s/768p with the light INT8 VAE) is
|
||||
the next largest cost after denoising.
|
||||
- **Hardware.** Everything here targets sm_120. The GeForce RTX 5090 shares
|
||||
the architecture but has 32 GB, which needs a reduced-AdaLN checkpoint and a
|
||||
non-resident text encoder at this resolution.
|
||||
@@ -1,107 +0,0 @@
|
||||
# Kandinsky 6 Text/Image to Video with Audio (T2IVA)
|
||||
|
||||
`Kandinsky6TI2VAPipeline` generates a video and a synchronized audio track from a text prompt, optionally conditioned
|
||||
on an image: one pipeline serves both, so pass `image_path` to condition on an image and leave it out for text only.
|
||||
The audio is decoded by the checkpoint's audio VAE (mel decoder plus vocoder in one component) and muxed into the saved
|
||||
mp4 automatically. To upscale a generated clip, see [Kandinsky 6 Video SR](kandinsky6_sr.md).
|
||||
|
||||
## Models
|
||||
|
||||
Both variants are official Diffusers repos, loaded directly through their `model_index.json`:
|
||||
|
||||
| Variant | Hub repo | Scheduler | Steps | Guidance | Example |
|
||||
|---|---|---|---|---|---|
|
||||
| T2IVA | `kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers` | `FlowMatchEulerDiscreteScheduler` (shift 5.0) | 50 | 5.0 | [`basic_kandinsky6_ti2va.py`](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
| T2IVA distilled | `kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers` | `PiflowScheduler` (`n_grid` 10, shift 5.0) | 10 | 1.0 | [`basic_kandinsky6_ti2va.py`](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
|
||||
The steps and guidance columns are the defaults of the preset the registry selects for each repo id. Everything else is
|
||||
shared: 512x768, 121 frames (5 s at 24 fps) and the Diffusers default negative prompt (only used when
|
||||
`guidance_scale > 1`). The distilled preset is named `kandinsky6_ti2va_distilled`.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_kandinsky6_ti2va.py
|
||||
KANDINSKY6_MODEL_PATH=kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers \
|
||||
python examples/inference/basic/basic_kandinsky6_ti2va.py
|
||||
```
|
||||
|
||||
Set `IMAGE_PATH` in the script to condition on an image.
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers",
|
||||
num_gpus=1,
|
||||
dit_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
)
|
||||
generator.generate_video(
|
||||
"cinematic shot: a giant stone samurai on a stormy cliff above a neon city opens glowing golden eyes and "
|
||||
"raises a katana. Blue lightning strikes the blade, creating a massive shockwave through the clouds. The "
|
||||
"camera rapidly pulls back from a low angle. Photorealistic, epic scale, dark blue and gold lighting, rain, "
|
||||
"sparks, volumetric lightning, blockbuster quality. Audio: heavy rain, deep thunder, metallic sword hum, "
|
||||
"rising brass and choir, electrical crackle, perfectly synchronized lightning impact, sub-bass shockwave. "
|
||||
"No dialogue, text, or logos.",
|
||||
image_path=None, # or the path of a conditioning image
|
||||
output_path="video_samples_kandinsky6_ti2va",
|
||||
height=512,
|
||||
width=768,
|
||||
num_frames=121,
|
||||
)
|
||||
```
|
||||
|
||||
A local copy of a repo works the same way. A local directory is treated as the distilled variant only when its name is
|
||||
a Kandinsky-6 name containing `distill` (for example `Kandinsky-6.0-Pro-distill-5s-Diffusers`); any other directory name
|
||||
selects the base preset (see below).
|
||||
|
||||
## Distilled (pi-Flow) checkpoint
|
||||
|
||||
- The distilled repo replaces the flow-matching scheduler with `PiflowScheduler` (`n_grid` 10, `eps` 1e-6,
|
||||
`final_step_size_scale` 0.5, `num_policy_substeps` 128). Its DiT emits `n_grid` predictions per latent channel
|
||||
(`out_visual_dim` 160 = 16 x 10, `out_audio_dim` 400 = 40 x 10).
|
||||
- pi-Flow runs without classifier-free guidance: `guidance_scale` must be exactly `1.0`. The official Diffusers
|
||||
pipeline rejects any other guidance for a `PiflowScheduler` too; FastVideo raises a `ValueError` naming the
|
||||
required value. `num_inference_steps` is not constrained by the checkpoint -- `PiflowScheduler.set_timesteps`
|
||||
accepts any step count and ignores the scheduler's `nfe`; `10` is only the value the distilled checkpoint was
|
||||
trained for and the `kandinsky6_ti2va_distilled` preset's default.
|
||||
- The `kandinsky6_ti2va_distilled` preset (10 steps, guidance 1.0) is the default for the distilled repo id and for
|
||||
local directories named like it. Set `KANDINSKY6_MODEL_PATH` to either one when running the shared
|
||||
`basic_kandinsky6_ti2va.py` example. A distilled copy under another directory name selects the base preset, so
|
||||
retain `Kandinsky-6.0-Pro-distill-5s-Diffusers` as the final directory name.
|
||||
- The policy values can be overridden on `Kandinsky6TI2VAConfig` (`piflow_eps`, `piflow_final_step_size_scale`,
|
||||
`piflow_num_policy_substeps`); `None` keeps the values from `scheduler_config.json`.
|
||||
|
||||
## Differences from the Diffusers pipeline
|
||||
|
||||
A few Diffusers pipeline options are not ported, and are surfaced here instead of as a knob that would silently do
|
||||
nothing:
|
||||
|
||||
- `sample_audio=False` (video-only, no audio stream) is not exposed; FastVideo's DiT raises `NotImplementedError` for
|
||||
a partial-modality call instead of denoising video alone.
|
||||
- `expand_prompts` (the Qwen prompt-beautifier pass) is not ported; FastVideo's `PromptEnhancerConfig` is a separate,
|
||||
external (Cerebras/Groq streaming) feature, not this pipeline's built-in expansion.
|
||||
- Of the Diffusers reference's `visual_cond_scheme` values, only `tail_cond_first_frame` (append one clean reference
|
||||
frame to the end of the sequence) is implemented; `pretrain` and plain `i2v` are not.
|
||||
- Precomputed `prompt_embeds`/`negative_prompt_embeds` are not accepted; every call encodes its own prompt text.
|
||||
- MagCache is not ported: a `magcache` block in a checkpoint's `transformer/config.json` is parsed and dropped
|
||||
by `update_model_arch`, not read automatically or exposed as an opt-in cache config.
|
||||
- RNG differs: FastVideo draws video then audio noise from one per-request CPU generator seeded by `seed` (default
|
||||
1024); Diffusers seeds a device generator from a value drawn out of `generator` (audio uses `seed+1`). The same
|
||||
numeric seed produces different noise on the two stacks -- pass `latents`/`audio_latents` directly for bit-level
|
||||
comparisons.
|
||||
- Qwen prompt tokens are unpadded and carry no attention mask (Diffusers pads to a fixed length and masks the
|
||||
padding in text self/cross-attention). Mathematically equivalent for a single prompt (measured 2e-7 relative
|
||||
difference), but FastVideo has no attention-mask plumbing, so a hand-built batch of unequal-length prompts is not
|
||||
supported.
|
||||
- Attention is dense (`LocalAttention`, flash/SDPA); NABLA sparse attention is wired but unverified against a real
|
||||
NABLA-flagged checkpoint. This matches the Diffusers pipeline itself, which never enables NABLA either.
|
||||
- VAE tiling is on by default (`vae_tiling=True`); Diffusers decodes untiled.
|
||||
|
||||
## Memory
|
||||
|
||||
The Pro DiT has 30.1B parameters, about 60 GB in bf16 (`dit_precision` defaults to `bf16`), and the Qwen2.5-VL text
|
||||
encoder adds 16.6 GB. FastVideo enables `dit_cpu_offload` by default; the examples turn it off (`dit_cpu_offload=False`)
|
||||
to keep the DiT resident on the GPU and offload the text encoder instead (`text_encoder_cpu_offload=True`). See
|
||||
[Offloading](offloading.md) for the memory knobs.
|
||||
@@ -1,83 +0,0 @@
|
||||
# Kandinsky 6 Video Super-Resolution
|
||||
|
||||
`Kandinsky6SRPipeline` upscales a low-resolution video by x2, x2.25 or x4 (video-to-video, no text prompt). The clip is
|
||||
encoded once with the SR VAE (KVAE); its latent is cut into overlapping tiles, each tile is enlarged by the *latent
|
||||
upscaler*, refined by a text-free SR DiT in a few denoising steps and decoded, and the tiles are blended back together.
|
||||
The source audio is kept. The clip can come from anywhere, for example from the [Kandinsky 6 T2IVA](kandinsky6.md)
|
||||
pipeline.
|
||||
|
||||
## Models
|
||||
|
||||
| Variant | Hub repo | Scheduler | Default steps per tile |
|
||||
|---|---|---|---|
|
||||
| Flow matching | `kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers` | `FlowMatchEulerDiscreteScheduler` (shift 5.0) | 4 |
|
||||
| Distilled | `kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers` | `PiflowScheduler` (shift 3.5, `n_grid` 10) | 2 |
|
||||
|
||||
Both repos have the same layout and share `vae/` and `latent_upscaler/`:
|
||||
|
||||
```text
|
||||
model_index.json _class_name = Kandinsky6SRPipeline
|
||||
transformer/ Kandinsky6SRTransformer3DModel (sr_params: trained base resolution, RoPE scale, noise level)
|
||||
vae/ Kandinsky6SRVAE
|
||||
latent_upscaler/ Kandinsky6SRLatentUpscalerBank (x2 and x4 models)
|
||||
scheduler/ FlowMatchEulerDiscreteScheduler or PiflowScheduler
|
||||
```
|
||||
|
||||
The scheduler component drives the denoising loop, and each repo resolves to its own preset (4 or 2 steps). The
|
||||
distilled transformer's head holds `n_grid` predictions per latent channel, which `PiflowScheduler` integrates.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_kandinsky6_sr.py --video-path input.mp4 --scale 2.25
|
||||
python examples/inference/basic/basic_kandinsky6_sr.py --video-path input.mp4 \
|
||||
--model-path kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers
|
||||
```
|
||||
|
||||
```python
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
generator = VideoGenerator.from_pretrained("kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers", num_gpus=1)
|
||||
result = generator.generate({
|
||||
"inputs": {"video_path": "input.mp4"},
|
||||
"output": {"output_path": "outputs_video/sr", "return_frames": False},
|
||||
"extensions": {"sr_resolution_scale": 2.25, "sr_target_resolution": "fullhd"},
|
||||
})
|
||||
```
|
||||
|
||||
The output geometry and frame rate follow the input clip; the request's `height` / `width` / `num_frames` are not used.
|
||||
Clips are resampled to 24 fps by a fixed stride and only the first 121 frames (5 s, `1 + 8k`-aligned) are processed.
|
||||
The source audio (mono, 44.1 kHz) is trimmed to the processed span and muxed into the output.
|
||||
|
||||
## Request parameters
|
||||
|
||||
The `sr_*` options belong only to Kandinsky6 SR. Pass them in `request.extensions` (as above), or under
|
||||
`request.stage_overrides.sr`. The SR pipeline reads them from `ForwardBatch.extra`; they are not fields of the
|
||||
shared `SamplingParam` or `ForwardBatch`. Other model families reject these options. Existing
|
||||
`generate_video(..., sr_resolution_scale=...)` calls remain supported for Kandinsky6 SR; replace direct
|
||||
`SamplingParam(sr_...=...)` construction with request extensions or these keyword arguments.
|
||||
For the config-based CLI, use dotted overrides such as `--request.extensions.sr_resolution_scale 4`, rather than
|
||||
shared `--sr-*` flags. The model-specific example above keeps its `--scale` and `--tiles-batch-size` flags.
|
||||
|
||||
| Field | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `num_inference_steps` | 4 / 2 | Denoising steps (DiT calls) per tile. The upstream Diffusers pipeline counts grid points instead (its 5 is 4 steps here). The distilled model was trained for 2; other values run with a warning. |
|
||||
| `sr_resolution_scale` | `2.25` | Total upscale: `2`, `4` or `2.25` (x1.125 pixel pre-upscale, then x2). |
|
||||
| `sr_tiles_batch_size` | `1` | Tiles denoised per DiT call (raise it only if memory allows). |
|
||||
| `sr_tile_min_overlap` | `0.20` | Minimum overlap between neighbouring tiles, as a fraction of the tile. |
|
||||
| `sr_target_resolution` | `None` | Downscale the result to `hd`, `fullhd`, `2k` or `WxH`. |
|
||||
| `sr_target_resize_mode` | `fit` | `fit` keeps the aspect ratio, `exact` uses the bucket dimensions. |
|
||||
| `seed` | `42` | Tile group *k* (of `sr_tiles_batch_size` tiles) is seeded with `seed + index of its first tile`. |
|
||||
|
||||
Two Python-only inputs take raw tensors and are passed as keyword arguments of `generate_video()`:
|
||||
|
||||
- `sr_lr_latent`: an unscaled KVAE latent `[T, C, H, W]` of the source, instead of `video_path` (skips decoding and
|
||||
encoding the video; scale 2 or 4 only, since 2.25 needs the pixel pre-upscale).
|
||||
- `sr_audio` / `sr_audio_sample_rate`: a mono waveform in `[-1, 1]` to mux instead of the source's own audio.
|
||||
|
||||
## Limitations
|
||||
|
||||
- NABLA block-sparse attention, which both repos request for 512-pixel tiles, is not wired; the DiT runs dense attention
|
||||
and logs a warning.
|
||||
- No sequence or tensor parallelism: the DiT runs on one GPU. Tiles are processed one group after another.
|
||||
- One clip per request.
|
||||
@@ -70,7 +70,7 @@ Run the numerical tests against a local TAEHV checkout containing the released
|
||||
weights:
|
||||
|
||||
```bash
|
||||
FASTVIDEO_TEST_TAEH3_REFERENCE_DIR=/path/to/taehv \
|
||||
TAEH3_REFERENCE_DIR=/path/to/taehv \
|
||||
python -m pytest fastvideo/tests/mlx/test_mlx_taeh3.py -q
|
||||
```
|
||||
|
||||
|
||||
@@ -168,8 +168,8 @@ and logs a warning if the flag is set.
|
||||
|
||||
Deferral is opt-in per pipeline. Releasing a component and loading it again is
|
||||
only safe when nothing outside the loader has changed it, and two common habits
|
||||
break that without raising: mutating a component after load, such as enabling
|
||||
a sparse-attention mode in `initialize_pipeline`, and reading a component's attributes
|
||||
break that without raising: mutating a component after load, as LongCat does
|
||||
when it enables block-sparse attention, and reading a component's attributes
|
||||
while stages are built, as the shared denoising stage does to pick an attention
|
||||
backend. A pipeline therefore lists the components it has checked in
|
||||
`_lazy_module_names`, which is empty in the base class. MiniMax-H3 opts in. On
|
||||
|
||||
@@ -198,11 +198,6 @@ The `attn_qat_infer` kernel hard-gates on **sm_120 (consumer Blackwell / RTX
|
||||
5090)**; on other GPUs the backend logs a notice and falls back to Flash
|
||||
Attention. See the [Attn-QAT paper](https://arxiv.org/abs/2603.00040).
|
||||
|
||||
For VSA-distilled MiniMax-H3 students, the same kernel has a block-sparse
|
||||
forward that runs VSA's 64-token tile selection in FP4
|
||||
(`FASTVIDEO_H3_VSA_FP4=1`). See
|
||||
[FastH3 NVFP4 on RTX PRO 6000](fasth3_rtx_pro_6000.md).
|
||||
|
||||
Enable both halves — attention via the env var, linear via `transformer_quant`:
|
||||
|
||||
```python
|
||||
|
||||
@@ -31,10 +31,14 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| cosmos | `nvidia/Cosmos-Predict2-2B-Video2World` | T2V | — |
|
||||
| cosmos25 | `KyleShao/Cosmos-Predict2.5-2B-Diffusers` | T2V | [basic_cosmos2_5_t2w.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_cosmos2_5_t2w.py) |
|
||||
| cosmos25 | `nvidia/Cosmos-Predict2.5-14B` | T2V | [basic_cosmos2_5_t2w.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_cosmos2_5_t2w.py) |
|
||||
| dreamx_world | `FastVideo/DreamX-World-5B-Cam-Diffusers` | I2V | [basic_dreamx_world.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_dreamx_world.py) |
|
||||
| dreamx_world | `FastVideo/DreamX-World-5B-Diffusers` | I2V | [basic_dreamx_world.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_dreamx_world.py) |
|
||||
| flux | `black-forest-labs/FLUX.1-dev` | T2I | [basic_flux_dev.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_flux_dev.py) |
|
||||
| flux2 | `black-forest-labs/FLUX.2-klein-4B`<br>`black-forest-labs/FLUX.2-klein-9B` | T2I | [basic_flux2_klein.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_flux2_klein.py) |
|
||||
| flux2 | `black-forest-labs/FLUX.2-dev` | T2I | [basic_flux2.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_flux2.py) |
|
||||
| gamecraft | `FastVideo/HunyuanGameCraft-Diffusers` | I2V | [basic_gamecraft.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_gamecraft.py) |
|
||||
| gen3c | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | T2V | [basic_gen3c.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_gen3c.py) |
|
||||
| glm_image | `zai-org/GLM-Image` | T2I | [basic_glm_image.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_glm_image.py) |
|
||||
| hunyuan | `hunyuanvideo-community/HunyuanVideo` | T2V | — |
|
||||
| hunyuan | `FastVideo/FastHunyuan-diffusers` | T2V | — |
|
||||
| hunyuan15 | `hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v` | T2V | [basic_hy15.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_hy15.py) |
|
||||
@@ -50,13 +54,13 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| kandinsky5 | `kandinskylab/Kandinsky-5.0-I2V-Lite-5s-Diffusers` | I2V | [basic_kandinsky5_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky5_i2v.py) |
|
||||
| kandinsky5 | `kandinskylab/Kandinsky-5.0-I2V-Pro-sft-5s-Diffusers` | I2V | [basic_kandinsky5_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky5_i2v.py) |
|
||||
| kandinsky5 | `kandinskylab/Kandinsky-5.0-I2V-Pro-distilled-5s-Diffusers` | I2V | [basic_kandinsky5_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky5_i2v.py) |
|
||||
| kandinsky6 | `kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers`<br>`kandinskylab/Kandinsky-6.0-Pro-sft-5s-Diffusers` | T2V, I2V | [basic_kandinsky6_ti2va.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
| kandinsky6 | `kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers` | T2V, I2V | [basic_kandinsky6_ti2va.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_ti2va.py) |
|
||||
| kandinsky6_sr | `kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers`<br>`kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers` | — | [basic_kandinsky6_sr.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_kandinsky6_sr.py) |
|
||||
| lingbot_video | `FastVideo/LingBot-Video-MoE-30B-A3B-Diffusers` | T2V | [basic_lingbot_video.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbot_video.py) |
|
||||
| lingbot_video | `FastVideo/LingBot-Video-Dense-1.3B-Diffusers` | T2V | [basic_lingbot_video.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbot_video.py) |
|
||||
| lingbotworld | `FastVideo/LingBot-World-Base-Cam-Diffusers` | I2V | [basic_lingbotworld_base_cam.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbotworld_base_cam.py) |
|
||||
| lingbotworld2 | `robbyant/lingbot-world-v2-14b-causal-fast` | I2V | [basic_lingbotworld2_causal_fast.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_lingbotworld2_causal_fast.py) |
|
||||
| longcat | `FastVideo/LongCat-Video-T2V-Diffusers` | T2V | [basic_longcat_t2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_longcat_t2v.py) |
|
||||
| longcat | `FastVideo/LongCat-Video-I2V-Diffusers` | I2V | [basic_longcat_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_longcat_i2v.py) |
|
||||
| longcat | `FastVideo/LongCat-Video-VC-Diffusers` | — | [basic_longcat_vc.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_longcat_vc.py) |
|
||||
| ltx2 | `FastVideo/LTX2-Distilled-Diffusers`<br>`FastVideo/LTX2.3-Distilled-Diffusers`<br>`FastVideo/LTX-2.3-Distilled-Diffusers` | T2V | [basic_ltx2_distilled.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_ltx2_distilled.py) |
|
||||
| ltx2 | `Lightricks/LTX-2.3`<br>`FastVideo/LTX2.3-base`<br>`FastVideo/LTX2.3-Diffusers` | T2V | [basic_ltx2.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_ltx2.py) |
|
||||
| ltx2 | `Lightricks/LTX-2`<br>`FastVideo/LTX2-base`<br>`FastVideo/LTX2-Diffusers` | T2V | [basic_ltx2.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_ltx2.py) |
|
||||
@@ -67,6 +71,9 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| sd35 | `stabilityai/stable-diffusion-3.5-medium` | T2I | [basic_sd35_t2i.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_sd35_t2i.py) |
|
||||
| stable_audio | `FastVideo/stable-audio-open-1.0-Diffusers` | T2V | [basic_stable_audio.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_stable_audio.py) |
|
||||
| stable_audio | `FastVideo/stable-audio-open-small-Diffusers` | T2V | [basic_stable_audio_small.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_stable_audio_small.py) |
|
||||
| turbodiffusion | `loayrashid/TurboWan2.1-T2V-1.3B-Diffusers` | T2V | [basic_turbodiffusion.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_turbodiffusion.py) |
|
||||
| turbodiffusion | `loayrashid/TurboWan2.1-T2V-14B-Diffusers` | T2V | [basic_turbodiffusion_14b.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_turbodiffusion_14b.py) |
|
||||
| turbodiffusion | `loayrashid/TurboWan2.2-I2V-A14B-Diffusers` | I2V | [basic_turbodiffusion_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_turbodiffusion_i2v.py) |
|
||||
| wan | `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` | T2V | [basic.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic.py) |
|
||||
| wan | `Wan-AI/Wan2.1-T2V-14B-Diffusers`<br>`FastVideo/Wan2.1-VSA-T2V-14B-720P-Diffusers` | T2V | — |
|
||||
| wan | `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` | I2V | — |
|
||||
@@ -82,6 +89,7 @@ column links a runnable script in `examples/inference/basic/` where one exists.
|
||||
| wan | `wlsaidhi/SFWan2.1-T2V-1.3B-Diffusers` | T2V | [basic_self_forcing_causal.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal.py) |
|
||||
| wan | `rand0nmr/SFWan2.2-T2V-A14B-Diffusers` | T2V | [basic_self_forcing_causal_wan2_2_t2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_t2v.py) |
|
||||
| wan | `FastVideo/SFWan2.2-I2V-A14B-Preview-Diffusers` | I2V | [basic_self_forcing_causal_wan2_2_i2v.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_i2v.py) |
|
||||
| zimage | `Tongyi-MAI/Z-Image-Turbo` | T2I | [basic_zimage.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_zimage.py) |
|
||||
|
||||
**Note (stable_audio)**: the Stable Audio Open pipelines generate audio
|
||||
(`StableAudioT2AConfig` / `StableAudioOpenSmallConfig`); they are registered
|
||||
@@ -93,15 +101,7 @@ to convert the official weights locally and set `MMAUDIO_MODEL_PATH`.
|
||||
|
||||
**Note (MiniMax H3)**: T2VA, FL2VA, and Ref2VA all generate video with stereo
|
||||
audio. Use the Ref2VA example when passing ordered image, video, or audio
|
||||
references. Distilled Ref2VA PDD students (eight transformer forwards, with
|
||||
reference videos as sparse VSA regions) run through
|
||||
[basic_fasth3_omniref_pdd.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_fasth3_omniref_pdd.py);
|
||||
see [FastH3 distilled checkpoint schedules](fasth3-distilled.md#ref2va-pdd-students).
|
||||
|
||||
**Note (Kandinsky 6)**: the two `kandinsky6` IDs generate video with audio from
|
||||
text, optionally plus an image (see the [T2IVA guide](kandinsky6.md)); the
|
||||
`kandinsky6_sr` IDs upscale an existing video (see the
|
||||
[Video SR guide](kandinsky6_sr.md)).
|
||||
references.
|
||||
|
||||
**Note (Wan-VACE)**: not currently supported — no VACE pipeline or registered
|
||||
model ID exists on `main`
|
||||
@@ -152,25 +152,31 @@ optimizations: absence means **untested**, not incompatible.
|
||||
}
|
||||
</style>
|
||||
|
||||
| Model Name | HuggingFace Model ID | Resolutions | TeaCache | Sliding Tile Attn (Legacy Branch) | Sage Attn | VSA |
|
||||
|------------|---------------------|-------------|----------|-------------------|-----------|-----|
|
||||
| FastWan2.1 T2V 1.3B | `FastVideo/FastWan2.1-T2V-1.3B-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ✅ |
|
||||
| FastWan2.2 TI2V 5B Full Attn | `FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers` | 720P | ⭕ | ⭕ | ⭕ | ✅ |
|
||||
| Wan2.2 TI2V 5B | `Wan-AI/Wan2.2-TI2V-5B-Diffusers` | 720P | ⭕ | ⭕ | ✅ | ⭕ |
|
||||
| Lucy Edit Dev 5B*** | `decart-ai/Lucy-Edit-Dev` | 480P | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Wan2.2 T2V A14B | `Wan-AI/Wan2.2-T2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ |
|
||||
| Wan2.2 I2V A14B | `Wan-AI/Wan2.2-I2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ |
|
||||
| HunyuanVideo | `hunyuanvideo-community/HunyuanVideo` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ |
|
||||
| FastHunyuan | `FastVideo/FastHunyuan-diffusers` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 T2V 1.3B | `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 T2V 14B | `Wan-AI/Wan2.1-T2V-14B-Diffusers` | 480P, 720P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 I2V 480P | `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Wan2.1 I2V 720P | `Wan-AI/Wan2.1-I2V-14B-720P-Diffusers` | 720P | ✅ | ✅ | ✅ | ⭕ |
|
||||
| Matrix Game 2.0 Base Distilled | `FastVideo/Matrix-Game-2.0-Base-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 GTA Distilled | `FastVideo/Matrix-Game-2.0-GTA-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 TempleRun Distilled | `FastVideo/Matrix-Game-2.0-TempleRun-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 3.0 Base Distilled | `FastVideo/Matrix-Game-3.0-Base-Distilled-Diffusers` | 720x1280 | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| GEN3C Cosmos 7B | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | 704px1280p | ❌ | ❌ | ❌ | ⭕ |
|
||||
| Model Name | HuggingFace Model ID | Resolutions | TeaCache | Sliding Tile Attn (Legacy Branch) | Sage Attn | VSA | BSA |
|
||||
|------------|---------------------|-------------|----------|-------------------|-----------|-----|-----|
|
||||
| FastWan2.1 T2V 1.3B | `FastVideo/FastWan2.1-T2V-1.3B-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ✅ | ⭕ |
|
||||
| FastWan2.2 TI2V 5B Full Attn | `FastVideo/FastWan2.2-TI2V-5B-FullAttn-Diffusers` | 720P | ⭕ | ⭕ | ⭕ | ✅ | ⭕ |
|
||||
| Wan2.2 TI2V 5B | `Wan-AI/Wan2.2-TI2V-5B-Diffusers` | 720P | ⭕ | ⭕ | ✅ | ⭕ | ⭕ |
|
||||
| DreamX-World 5B Cam | `FastVideo/DreamX-World-5B-Cam-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| DreamX-World 5B AR | `FastVideo/DreamX-World-5B-Diffusers` | 704px1280p | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Lucy Edit Dev 5B*** | `decart-ai/Lucy-Edit-Dev` | 480P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Wan2.2 T2V A14B | `Wan-AI/Wan2.2-T2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.2 I2V A14B | `Wan-AI/Wan2.2-I2V-A14B-Diffusers` | 480P<br>720P | ❌ | ❌ | ✅ | ⭕ | ⭕ |
|
||||
| HunyuanVideo | `hunyuanvideo-community/HunyuanVideo` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| FastHunyuan | `FastVideo/FastHunyuan-diffusers` | 720px1280p<br>544px960p | ❌ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 T2V 1.3B | `Wan-AI/Wan2.1-T2V-1.3B-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 T2V 14B | `Wan-AI/Wan2.1-T2V-14B-Diffusers` | 480P, 720P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 I2V 480P | `Wan-AI/Wan2.1-I2V-14B-480P-Diffusers` | 480P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| Wan2.1 I2V 720P | `Wan-AI/Wan2.1-I2V-14B-720P-Diffusers` | 720P | ✅ | ✅ | ✅ | ⭕ | ⭕ |
|
||||
| TurboWan2.1 T2V 1.3B | `loayrashid/TurboWan2.1-T2V-1.3B-Diffusers` | 480P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| TurboWan2.1 T2V 14B | `loayrashid/TurboWan2.1-T2V-14B-Diffusers` | 480P, 720P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| TurboWan2.2 I2V A14B | `loayrashid/TurboWan2.2-I2V-A14B-Diffusers` | 480P<br>720P | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| LongCat T2V 13.6B | `FastVideo/LongCat-Video-T2V-Diffusers` | 480P<br>720P | ❌ | ❌ | ❌ | ⭕ | ✅ |
|
||||
| Matrix Game 2.0 Base Distilled | `FastVideo/Matrix-Game-2.0-Base-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 GTA Distilled | `FastVideo/Matrix-Game-2.0-GTA-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 2.0 TempleRun Distilled | `FastVideo/Matrix-Game-2.0-TempleRun-Distilled-Diffusers` | 352x640 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| Matrix Game 3.0 Base Distilled | `FastVideo/Matrix-Game-3.0-Base-Distilled-Diffusers` | 720x1280 | ⭕ | ⭕ | ⭕ | ⭕ | ⭕ |
|
||||
| GEN3C Cosmos 7B | `FastVideo/GEN3C-Cosmos-7B-Diffusers` | 704px1280p | ❌ | ❌ | ❌ | ⭕ | ⭕ |
|
||||
|
||||
## Apple Silicon native runtime
|
||||
|
||||
@@ -234,6 +240,11 @@ listed under [Special requirements](#special-requirements).
|
||||
https://github.com/hao-ai-lab/FastVideo/tree/sta_do_not_delete
|
||||
- STA currently requires Hopper GPUs (H100s).
|
||||
|
||||
### TurboWan2.1 (TurboDiffusion)
|
||||
- Uses TurboDiffusionPipeline with RCM scheduler for 1-4 step generation
|
||||
- Requires SLA attention backend: `export FASTVIDEO_ATTENTION_BACKEND=SLA_ATTN`
|
||||
- Uses `guidance_scale=1.0` (no classifier-free guidance)
|
||||
|
||||
### Matrix Game 2.0
|
||||
- Image-to-video game world models with keyboard/mouse control input
|
||||
- Three variants available: Base (universal), GTA, and TempleRun
|
||||
|
||||
@@ -167,24 +167,6 @@ launchers still use sparse attention at strength `0` and require FastVideo's
|
||||
tile-64 VSA kernel; the dense launcher selects FA4. Each launcher writes to its
|
||||
own variant directory by default so comparison outputs do not collide.
|
||||
|
||||
### FastH3 OmniRef PDD (Ref2VA)
|
||||
|
||||
[basic_fasth3_omniref_pdd.py](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_fasth3_omniref_pdd.py)
|
||||
runs a Parallel Decoding Distillation Ref2VA student in eight transformer
|
||||
forwards. The export carries only its `transformer_ref`, scheduler configs,
|
||||
and `fastvideo_inference.json`; the script links them with the base
|
||||
MiniMax-H3 components into one local model directory. Its 128-token VSA tiles
|
||||
need the sm_100a/sm_103a kernel:
|
||||
|
||||
```bash
|
||||
python examples/inference/basic/basic_fasth3_omniref_pdd.py \
|
||||
--model-path <local export directory or Hugging Face repo id> \
|
||||
--video reference.mp4 --image character.png --prompt "your prompt"
|
||||
```
|
||||
|
||||
See [FastH3 distilled checkpoint schedules](https://github.com/hao-ai-lab/FastVideo/blob/main/docs/inference/fasth3-distilled.md#ref2va-pdd-students)
|
||||
for the contract and the reference-video sparsity policy.
|
||||
|
||||
## Basic Walkthrough
|
||||
|
||||
All you need to generate videos using multi-gpus from state-of-the-art diffusion pipelines is the following few lines!
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
# FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 \
|
||||
# fastvideo generate --config examples/inference/basic/basic_compacth3_rtx5090.yaml
|
||||
generator:
|
||||
model_path: ./CompactH3
|
||||
engine:
|
||||
num_gpus: 1
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: true
|
||||
vae: false
|
||||
pin_cpu_memory: true
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: false
|
||||
pipeline:
|
||||
experimental:
|
||||
attention_backend: ATTN_QAT_INFER
|
||||
h3_sequential_load: true
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: false
|
||||
video_decode_backend: h3-vae
|
||||
request:
|
||||
prompt: >-
|
||||
A wide cinematic shot of an alpine meadow at sunrise, pale pink mountain
|
||||
peaks above a blue valley filled with thin morning mist.
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
seed: 2026
|
||||
height: 480
|
||||
width: 832
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 5
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
output:
|
||||
output_path: outputs/compacth3_rtx5090/
|
||||
save_video: true
|
||||
return_frames: false
|
||||
@@ -1,49 +0,0 @@
|
||||
# FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FLASHINFER_CUDA_ARCH_LIST=12.0a FASTVIDEO_STAGE_LOGGING=1 \
|
||||
# fastvideo generate --config examples/inference/basic/basic_compacth3_rtx_pro6000.yaml
|
||||
generator:
|
||||
model_path: ./CompactH3
|
||||
engine:
|
||||
num_gpus: 1
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: false
|
||||
vae: false
|
||||
pin_cpu_memory: true
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: true
|
||||
pipeline:
|
||||
experimental:
|
||||
attention_backend: ATTN_QAT_INFER
|
||||
h3_sequential_load: false
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: false
|
||||
video_decode_backend: h3-vae
|
||||
request:
|
||||
prompt: >-
|
||||
A wide cinematic shot of an alpine meadow at sunrise, pale pink mountain
|
||||
peaks above a blue valley filled with thin morning mist.
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
seed: 2026
|
||||
height: 768
|
||||
width: 1344
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 5
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
output:
|
||||
output_path: outputs/compacth3_rtx_pro6000/
|
||||
save_video: true
|
||||
return_frames: false
|
||||
@@ -0,0 +1,61 @@
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
OUTPUT_PATH = os.getenv("DREAMX_WORLD_OUTPUT_PATH", "video_samples_dreamx_world")
|
||||
|
||||
|
||||
def _env_int(name: str, default: int) -> int:
|
||||
return int(os.getenv(name, str(default)))
|
||||
|
||||
|
||||
def _env_float(name: str, default: float) -> float:
|
||||
return float(os.getenv(name, str(default)))
|
||||
|
||||
|
||||
def main():
|
||||
model_name = os.getenv("DREAMX_WORLD_MODEL_DIR", "FastVideo/DreamX-World-5B-Cam-Diffusers")
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
model_name,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
override_pipeline_cls_name="DreamXWorldPipeline",
|
||||
)
|
||||
|
||||
prompt = os.getenv(
|
||||
"DREAMX_WORLD_PROMPT",
|
||||
"A cinematic first-person drive through a futuristic coastal city at "
|
||||
"sunrise, reflective glass towers, clean streets, soft volumetric light.",
|
||||
)
|
||||
image_path = os.getenv(
|
||||
"DREAMX_WORLD_IMAGE_PATH",
|
||||
"https://huggingface.co/datasets/YiYiXu/testing-images/resolve/main/wan_i2v_input.JPG",
|
||||
)
|
||||
|
||||
kwargs = {
|
||||
"output_path": OUTPUT_PATH,
|
||||
"save_video": os.getenv("DREAMX_WORLD_SAVE_VIDEO", "1") != "0",
|
||||
"height": _env_int("DREAMX_WORLD_HEIGHT", 480),
|
||||
"width": _env_int("DREAMX_WORLD_WIDTH", 832),
|
||||
"num_frames": _env_int("DREAMX_WORLD_NUM_FRAMES", 161),
|
||||
"num_inference_steps": _env_int("DREAMX_WORLD_STEPS", 30),
|
||||
"guidance_scale": _env_float("DREAMX_WORLD_GUIDANCE", 5.0),
|
||||
"action_list": os.getenv("DREAMX_WORLD_ACTIONS", "w,d,w").split(","),
|
||||
"action_speed_list":
|
||||
[float(value) for value in os.getenv("DREAMX_WORLD_ACTION_SPEEDS", "4.0,2.0,4.0").split(",")],
|
||||
}
|
||||
if image_path:
|
||||
kwargs["image_path"] = image_path
|
||||
|
||||
try:
|
||||
generator.generate_video(prompt, **kwargs)
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,268 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Eight-forward Ref2VA video+audio generation with a FastH3 OmniRef PDD student.
|
||||
|
||||
A FastH3 OmniRef PDD export is a Parallel Decoding Distillation (PDD) student of
|
||||
MiniMax-H3's reference-conditioned ``transformer_ref`` partition. Its directory
|
||||
(local, or a Hugging Face repo) carries only what distillation changed:
|
||||
``transformer_ref/`` (output heads widened to the fine grid, trained VSA
|
||||
compression gates), the two scheduler configs, and ``fastvideo_inference.json``.
|
||||
The text encoder, tokenizer, processor, and both VAEs are base MiniMax-H3's.
|
||||
|
||||
This script composes the two into one local model directory of symlinks and
|
||||
runs the recipe the contract records: fused blocks of the fine grid (one
|
||||
transformer forward each), the video/audio shifts, and VIDEO_SPARSE_ATTN_H3
|
||||
with its sparsity and tile size, where every reference video is its own
|
||||
sparse region. The contract's 128-token tiles run only on the sm_100a/sm_103a
|
||||
CUDA kernel (B200/B300/GB200/GB300) of a fastvideo-kernel build with the
|
||||
Blackwell VSA extension; the VSA-H3 backend raises at the first attention call
|
||||
when either is missing.
|
||||
With ``--num-gpus`` above 1 the DiT is sharded across the GPUs (FSDP) and the
|
||||
sequence is split across them (sequence parallelism).
|
||||
|
||||
References are ordered. Pass them in order with --image / --video / --audio,
|
||||
for example ``--video dance.mp4 --image outfit.png --audio voice.wav``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
from collections.abc import Sequence
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
InputConfig,
|
||||
OffloadConfig,
|
||||
OutputConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
SamplingConfig,
|
||||
)
|
||||
from fastvideo.configs.pipelines.minimax_h3 import parse_base_model_revision
|
||||
from fastvideo.pipelines.basic.minimax_h3 import MiniMaxH3Reference
|
||||
|
||||
CONTRACT = "fastvideo_inference.json"
|
||||
# Components the distilled export replaces, and the ones it shares with base MiniMax-H3.
|
||||
EXPORT_COMPONENTS = ("transformer_ref", "scheduler", "audio_scheduler")
|
||||
BASE_COMPONENTS = ("text_encoder", "tokenizer", "processor", "vae", "audio_vae")
|
||||
MANIFESTS = ("modular_model_index.json", "model_index.json")
|
||||
# Components the composed directory's manifest must declare, besides its transformer_ref.
|
||||
MANIFEST_COMPONENTS = ("scheduler", "audio_scheduler", *BASE_COMPONENTS)
|
||||
|
||||
|
||||
class _AppendReference(argparse.Action):
|
||||
"""Collect --image/--video/--audio into one list that keeps command-line order."""
|
||||
|
||||
def __call__(self, parser, namespace, value, option_string=None):
|
||||
references = list(getattr(namespace, self.dest, None) or [])
|
||||
references.append((self.const, value))
|
||||
setattr(namespace, self.dest, references)
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--model-path",
|
||||
required=True,
|
||||
help="FastH3 OmniRef PDD export: a local directory or a Hugging Face repo id")
|
||||
parser.add_argument("--revision", default=None, help="revision of --model-path when it is a Hugging Face repo")
|
||||
parser.add_argument("--base-model-path",
|
||||
default=None,
|
||||
help="base MiniMax-H3 snapshot (local directory or repo id). Default: the repo and revision "
|
||||
"pinned by the export's base_model_revision")
|
||||
parser.add_argument("--base-revision",
|
||||
default=None,
|
||||
help="revision of --base-model-path when it is a repo id (the export's pinned revision "
|
||||
"applies only to its own base repo); without --base-model-path, overrides that pin")
|
||||
parser.add_argument("--composed-dir",
|
||||
default=None,
|
||||
help="where to write the composed model directory of symlinks (default: "
|
||||
"<output>/fasth3_omniref_model)")
|
||||
for flag in ("image", "video", "audio"):
|
||||
parser.add_argument(f"--{flag}",
|
||||
dest="references",
|
||||
action=_AppendReference,
|
||||
const=flag,
|
||||
metavar="PATH",
|
||||
help=f"an ordered {flag} reference (repeatable)")
|
||||
parser.add_argument("--prompt", required=True)
|
||||
parser.add_argument("--output", default="outputs/fasth3_omniref_pdd")
|
||||
parser.add_argument("--height", type=int, default=480)
|
||||
parser.add_argument("--width", type=int, default=832)
|
||||
parser.add_argument("--num-frames", type=int, default=124)
|
||||
parser.add_argument("--seed", type=int, default=0)
|
||||
parser.add_argument("--num-gpus", type=int, default=1, help="sequence-parallel GPUs (must divide 56 heads)")
|
||||
return parser
|
||||
|
||||
|
||||
def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace:
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(argv)
|
||||
if not args.references:
|
||||
parser.error("pass at least one ordered reference with --image, --video, or --audio")
|
||||
if not any(kind != "audio" for kind, _ in args.references):
|
||||
parser.error("Ref2VA needs at least one image or video reference")
|
||||
return args
|
||||
|
||||
|
||||
def _snapshot(path_or_repo: str, revision: str | None, allow_patterns: list[str]) -> Path:
|
||||
local = Path(path_or_repo).expanduser()
|
||||
if local.is_dir():
|
||||
return local
|
||||
from huggingface_hub import snapshot_download
|
||||
|
||||
return Path(snapshot_download(repo_id=path_or_repo, revision=revision, allow_patterns=allow_patterns))
|
||||
|
||||
|
||||
def load_contract(export_dir: Path) -> dict[str, Any]:
|
||||
path = export_dir / CONTRACT
|
||||
if not path.is_file():
|
||||
raise FileNotFoundError(f"{export_dir} has no {CONTRACT}; it is not a FastH3 distilled export.")
|
||||
contract = json.loads(path.read_text(encoding="utf-8"))
|
||||
if "pdd_steps" not in contract or contract.get("model_type") != "ref2va":
|
||||
raise ValueError(f"{path} is not a Ref2VA PDD contract; use the example matching the checkpoint.")
|
||||
return contract
|
||||
|
||||
|
||||
def base_model_source(args: argparse.Namespace, contract: dict[str, Any]) -> tuple[str, str | None]:
|
||||
"""The base snapshot and revision: --base-model-path as given, else the export's pin."""
|
||||
base_repo, base_revision = base_model_from_contract(contract)
|
||||
if args.base_model_path:
|
||||
return args.base_model_path, args.base_revision
|
||||
return base_repo, args.base_revision or base_revision
|
||||
|
||||
|
||||
def base_model_from_contract(contract: dict[str, Any]) -> tuple[str, str]:
|
||||
"""The base repo id and revision that the export pins as ``hf://<repo id>@<revision>``."""
|
||||
return parse_base_model_revision(contract.get("base_model_revision"))
|
||||
|
||||
|
||||
def _link(destination: Path, source: Path) -> None:
|
||||
if not source.exists():
|
||||
raise FileNotFoundError(f"Missing checkpoint component: {source}")
|
||||
source = source.resolve()
|
||||
if destination.is_symlink():
|
||||
if destination.resolve() == source:
|
||||
return
|
||||
destination.unlink()
|
||||
elif destination.exists():
|
||||
raise FileExistsError(f"{destination} exists and is not a symlink; choose another --composed-dir.")
|
||||
destination.symlink_to(source, target_is_directory=source.is_dir())
|
||||
|
||||
|
||||
def _declares_components(manifest: Path) -> bool:
|
||||
"""Whether a Diffusers manifest declares every component of the composed directory."""
|
||||
declared = {
|
||||
name
|
||||
for name, spec in json.loads(manifest.read_text(encoding="utf-8")).items()
|
||||
if isinstance(spec, list) and spec and spec[0] is not None
|
||||
}
|
||||
return set(MANIFEST_COMPONENTS) <= declared and bool({"transformer", "transformer_ref"} & declared)
|
||||
|
||||
|
||||
def select_manifest(export_dir: Path, base_dir: Path) -> Path:
|
||||
"""The export's manifest when it declares every composed component, else the base's."""
|
||||
for root in (export_dir, base_dir):
|
||||
for name in MANIFESTS:
|
||||
if (root / name).is_file() and _declares_components(root / name):
|
||||
return root / name
|
||||
raise FileNotFoundError("Neither the export nor the base snapshot has a Diffusers model manifest that declares "
|
||||
f"{', '.join(MANIFEST_COMPONENTS)} and transformer_ref.")
|
||||
|
||||
|
||||
def compose_model_dir(export_dir: Path, base_dir: Path, composed_dir: Path) -> Path:
|
||||
"""One MiniMax-H3 model directory: distilled components from the export, the rest from the base."""
|
||||
composed_dir.mkdir(parents=True, exist_ok=True)
|
||||
for name in (*EXPORT_COMPONENTS, CONTRACT):
|
||||
_link(composed_dir / name, export_dir / name)
|
||||
manifest_source = select_manifest(export_dir, base_dir)
|
||||
for name in MANIFESTS:
|
||||
stale = composed_dir / name
|
||||
if name != manifest_source.name and stale.is_symlink():
|
||||
stale.unlink()
|
||||
_link(composed_dir / manifest_source.name, manifest_source)
|
||||
for name in BASE_COMPONENTS:
|
||||
_link(composed_dir / name, base_dir / name)
|
||||
return composed_dir
|
||||
|
||||
|
||||
def resolve_model(args: argparse.Namespace) -> tuple[Path, dict[str, Any]]:
|
||||
export_dir = _snapshot(args.model_path, args.revision,
|
||||
[CONTRACT, *MANIFESTS, *(f"{name}/**" for name in EXPORT_COMPONENTS)])
|
||||
contract = load_contract(export_dir)
|
||||
base_source, base_revision = base_model_source(args, contract)
|
||||
base_dir = _snapshot(base_source, base_revision, [*MANIFESTS, *(f"{name}/**" for name in BASE_COMPONENTS)])
|
||||
composed_dir = Path(args.composed_dir) if args.composed_dir else Path(args.output) / "fasth3_omniref_model"
|
||||
return compose_model_dir(export_dir, base_dir, composed_dir), contract
|
||||
|
||||
|
||||
def build_generator_config(model_dir: Path, num_gpus: int) -> GeneratorConfig:
|
||||
return GeneratorConfig(
|
||||
model_path=str(model_dir),
|
||||
engine=EngineConfig(
|
||||
num_gpus=num_gpus,
|
||||
# Shard the DiT across the GPUs rather than holding a full copy on each.
|
||||
use_fsdp_inference=num_gpus > 1,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=num_gpus),
|
||||
offload=OffloadConfig(dit=False, dit_layerwise=False, text_encoder=True, vae=True, pin_cpu_memory=False),
|
||||
),
|
||||
pipeline=PipelineSelection(
|
||||
workload_type="i2v",
|
||||
# FastVideo reads the trained fused-block partition and attention
|
||||
# settings from the composed directory's fastvideo_inference.json.
|
||||
components=ComponentConfig(override_pipeline_cls_name="MiniMaxH3Ref2VAModularPipeline"),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
output_dir = Path(args.output)
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
model_dir, contract = resolve_model(args)
|
||||
print(f"Composed model directory: {model_dir}")
|
||||
print(f"Contract: {contract['num_inference_steps']} fused blocks of a {contract['pdd_steps']}-interval grid, "
|
||||
f"{contract['attention_backend']} with sparsity {contract['vsa_sparsity']}, "
|
||||
f"{contract['vsa_tile_size']}-token tiles, reference keep rate {contract['vsa_ref_keep_rate']}")
|
||||
references = [MiniMaxH3Reference(source=path, media_type=kind) for kind, path in args.references]
|
||||
|
||||
generator = VideoGenerator.from_config(build_generator_config(model_dir, args.num_gpus))
|
||||
try:
|
||||
result = generator.generate(
|
||||
GenerationRequest(
|
||||
prompt=args.prompt,
|
||||
negative_prompt="",
|
||||
inputs=InputConfig(references=references),
|
||||
sampling=SamplingConfig(
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=args.num_frames,
|
||||
fps=24,
|
||||
# PDD counts fused blocks: one transformer forward each.
|
||||
num_inference_steps=contract["num_inference_steps"],
|
||||
guidance_scale=1.0,
|
||||
batch_cfg=False,
|
||||
seed=args.seed,
|
||||
),
|
||||
output=OutputConfig(
|
||||
output_path=str(output_dir / "fasth3_omniref_pdd.mp4"),
|
||||
save_video=True,
|
||||
return_frames=False,
|
||||
),
|
||||
))
|
||||
print(f"Output written to: {result.video_path}")
|
||||
if result.generation_time is not None:
|
||||
print(f"Generation time: {result.generation_time:.1f} s")
|
||||
if result.peak_memory_mb is not None:
|
||||
print(f"Peak GPU memory allocated (rank 0): {result.peak_memory_mb / 1024:.1f} GiB")
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,64 +0,0 @@
|
||||
# Start Ray on both Sparks and source spark_pair_env.sh first; see spark_pair.md.
|
||||
# FastH3 Trim eight-forward video+audio on two DGX Sparks over QSFP RoCE.
|
||||
# Download the complete checkpoint, including its trained schedule, as described in
|
||||
# docs/getting_started/installation/spark_performance.md.
|
||||
#
|
||||
# GB10 uses Triton VSA. The release packs attention, FFN and VSA gate weights.
|
||||
# The NVFP4 encoder and lightweight VAE ship in the same repository.
|
||||
#
|
||||
# FASTVIDEO_MINIMAX_H3_FUSIONS=all FASTVIDEO_NVFP4_MM_BACKEND=cutlass \
|
||||
# FASTVIDEO_H3_VAE_TILE_BATCH=1 \
|
||||
# FASTVIDEO_VSA_TRITON=1 FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 \
|
||||
# fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair_pruned_nvfp4.yaml
|
||||
generator:
|
||||
model_path: FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4
|
||||
engine:
|
||||
num_gpus: 2
|
||||
execution_backend: ray
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit_vsa
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 2
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: false
|
||||
image_encoder: false
|
||||
vae: false
|
||||
pin_cpu_memory: false
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: false
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
vae_tiling: true
|
||||
experimental:
|
||||
attention_backend: VIDEO_SPARSE_ATTN_H3
|
||||
VSA_sparsity: 0.8
|
||||
VSA_tile_size: 64
|
||||
h3_sequential_load: false
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: true
|
||||
vae_parallel_decode_strategy: gather
|
||||
video_decode_backend: h3-vae
|
||||
request:
|
||||
prompt: A quiet pottery studio with a potter finishing a bowl at the wheel.
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
seed: 1234
|
||||
height: 480
|
||||
width: 832
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 9 # nine sigma points, eight DiT forwards
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
output:
|
||||
output_path: outputs/fasth3_spark_pair_pruned_nvfp4/
|
||||
save_video: true
|
||||
return_frames: false
|
||||
@@ -1,64 +0,0 @@
|
||||
# Start Ray on both Sparks and source spark_pair_env.sh first; see spark_pair.md.
|
||||
# FastH3 V2 eight-forward video+audio on two DGX Sparks over QSFP RoCE.
|
||||
# Download the complete release checkpoint as described in
|
||||
# docs/getting_started/installation/spark_performance.md.
|
||||
#
|
||||
# GB10 uses Triton VSA. The release packs attention, FFN and VSA gate weights.
|
||||
# The NVFP4 encoder and lightweight VAE ship in the same repository.
|
||||
#
|
||||
# FASTVIDEO_MINIMAX_H3_FUSIONS=all FASTVIDEO_NVFP4_MM_BACKEND=cutlass \
|
||||
# FASTVIDEO_H3_VAE_TILE_BATCH=1 \
|
||||
# FASTVIDEO_VSA_TRITON=1 FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 \
|
||||
# fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pair_v2_nvfp4.yaml
|
||||
generator:
|
||||
model_path: FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer
|
||||
engine:
|
||||
num_gpus: 2
|
||||
execution_backend: ray
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit_vsa
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 2
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: false
|
||||
image_encoder: false
|
||||
vae: false
|
||||
pin_cpu_memory: false
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: false
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
vae_tiling: true
|
||||
experimental:
|
||||
attention_backend: VIDEO_SPARSE_ATTN_H3
|
||||
VSA_sparsity: 0.8
|
||||
VSA_tile_size: 64
|
||||
h3_sequential_load: false
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: true
|
||||
vae_parallel_decode_strategy: gather
|
||||
video_decode_backend: h3-vae
|
||||
request:
|
||||
prompt: A quiet pottery studio with a potter finishing a bowl at the wheel.
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
seed: 1234
|
||||
height: 480
|
||||
width: 832
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 9 # nine sigma points, eight DiT forwards
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
output:
|
||||
output_path: outputs/fasth3_spark_pair_v2_nvfp4/
|
||||
save_video: true
|
||||
return_frames: false
|
||||
@@ -1,61 +0,0 @@
|
||||
# FastH3 Trim eight-forward video+audio on one DGX Spark.
|
||||
# Download the complete checkpoint, including its trained schedule, as described in
|
||||
# docs/getting_started/installation/spark_performance.md.
|
||||
#
|
||||
# GB10 uses Triton VSA. The release packs attention, FFN and VSA gate weights.
|
||||
# The NVFP4 encoder and lightweight VAE ship in the same repository.
|
||||
#
|
||||
# FASTVIDEO_MINIMAX_H3_FUSIONS=all FASTVIDEO_NVFP4_MM_BACKEND=cutlass \
|
||||
# FASTVIDEO_H3_VAE_TILE_BATCH=1 \
|
||||
# FASTVIDEO_VSA_TRITON=1 FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 \
|
||||
# fastvideo generate --config examples/inference/basic/basic_fasth3_spark_pruned_nvfp4.yaml
|
||||
generator:
|
||||
model_path: FastVideo/FastVideo-FastH3-Trim-8-Step-NVFP4
|
||||
engine:
|
||||
num_gpus: 1
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit_vsa
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: false
|
||||
image_encoder: false
|
||||
vae: false
|
||||
pin_cpu_memory: false
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: false
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
vae_tiling: true
|
||||
experimental:
|
||||
attention_backend: VIDEO_SPARSE_ATTN_H3
|
||||
VSA_sparsity: 0.8
|
||||
VSA_tile_size: 64
|
||||
h3_sequential_load: false
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: false
|
||||
video_decode_backend: h3-vae
|
||||
request:
|
||||
prompt: A quiet pottery studio with a potter finishing a bowl at the wheel.
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
seed: 1234
|
||||
height: 480
|
||||
width: 832
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 9 # nine sigma points, eight DiT forwards
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
output:
|
||||
output_path: outputs/fasth3_spark_pruned_nvfp4/
|
||||
save_video: true
|
||||
return_frames: false
|
||||
@@ -1,61 +0,0 @@
|
||||
# FastH3 V2 eight-forward video+audio on one DGX Spark.
|
||||
# Download the complete release checkpoint as described in
|
||||
# docs/getting_started/installation/spark_performance.md.
|
||||
#
|
||||
# GB10 uses Triton VSA. The release packs attention, FFN and VSA gate weights.
|
||||
# The NVFP4 encoder and lightweight VAE ship in the same repository.
|
||||
#
|
||||
# FASTVIDEO_MINIMAX_H3_FUSIONS=all FASTVIDEO_NVFP4_MM_BACKEND=cutlass \
|
||||
# FASTVIDEO_H3_VAE_TILE_BATCH=1 \
|
||||
# FASTVIDEO_VSA_TRITON=1 FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FASTVIDEO_ATTENTION_BACKEND=VIDEO_SPARSE_ATTN_H3 FASTVIDEO_STAGE_LOGGING=1 \
|
||||
# fastvideo generate --config examples/inference/basic/basic_fasth3_spark_v2_nvfp4.yaml
|
||||
generator:
|
||||
model_path: FastVideo/FastVideo-FastH3-8-Step-V2-NVFP4-Consumer
|
||||
engine:
|
||||
num_gpus: 1
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit_vsa
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: false
|
||||
image_encoder: false
|
||||
vae: false
|
||||
pin_cpu_memory: false
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: false
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
vae_tiling: true
|
||||
experimental:
|
||||
attention_backend: VIDEO_SPARSE_ATTN_H3
|
||||
VSA_sparsity: 0.8
|
||||
VSA_tile_size: 64
|
||||
h3_sequential_load: false
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: false
|
||||
video_decode_backend: h3-vae
|
||||
request:
|
||||
prompt: A quiet pottery studio with a potter finishing a bowl at the wheel.
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
seed: 1234
|
||||
height: 480
|
||||
width: 832
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 9 # nine sigma points, eight DiT forwards
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
output:
|
||||
output_path: outputs/fasth3_spark_v2_nvfp4/
|
||||
save_video: true
|
||||
return_frames: false
|
||||
@@ -0,0 +1,118 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""
|
||||
Basic inference script for HunyuanGameCraft video generation.
|
||||
|
||||
HunyuanGameCraft generates game-like videos with camera/action control.
|
||||
It takes an optional image input and generates video with camera motion
|
||||
based on simple action commands (forward, left, right, backward, rotations).
|
||||
|
||||
Available actions:
|
||||
- forward (w): Move camera forward
|
||||
- backward (s): Move camera backward
|
||||
- left (a): Move camera left (strafe)
|
||||
- right (d): Move camera right (strafe)
|
||||
- left_rot: Rotate camera left (pan)
|
||||
- right_rot: Rotate camera right (pan)
|
||||
- up_rot: Rotate camera up (tilt)
|
||||
- down_rot: Rotate camera down (tilt)
|
||||
|
||||
T2V vs I2V:
|
||||
- Default: I2V (uses a default reference image). Set GAMECRAFT_I2V_IMAGE to a
|
||||
URL or path to use a different image.
|
||||
- T2V only (no reference image): run with GAMECRAFT_I2V_IMAGE= (empty).
|
||||
"""
|
||||
import os
|
||||
|
||||
import torch
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.models.camera import create_camera_trajectory
|
||||
|
||||
# Model configuration (use GAMECRAFT_MODEL_PATH for local weights)
|
||||
MODEL_PATH = os.environ.get("GAMECRAFT_MODEL_PATH", "FastVideo/HunyuanGameCraft-Diffusers")
|
||||
|
||||
# Default prompts for demo
|
||||
DEFAULT_PROMPTS = {
|
||||
"village":
|
||||
"A charming medieval village with cobblestone streets, thatched-roof houses, and vibrant flower gardens under a bright blue sky.",
|
||||
"temple":
|
||||
"A majestic ancient temple stands under a clear blue sky, its grandeur highlighted by towering Doric columns and intricate architectural details.",
|
||||
"forest":
|
||||
"A lush green forest with tall trees, dappled sunlight filtering through the leaves, and a winding dirt path.",
|
||||
"beach": "A tropical beach with crystal clear turquoise water, white sand, and palm trees swaying in the breeze.",
|
||||
}
|
||||
|
||||
# I2V: default reference image (URL). Can override with a local path.
|
||||
DEFAULT_I2V_IMAGE_URL = ("https://huggingface.co/datasets/huggingface/documentation-images/"
|
||||
"resolve/main/diffusers/astronaut.jpg")
|
||||
DEFAULT_I2V_PROMPT = ("An astronaut hatching from an egg, on the surface of the moon, "
|
||||
"the darkness and depth of space realised in the background.")
|
||||
|
||||
OUTPUT_PATH = "video_samples_gamecraft"
|
||||
|
||||
|
||||
def main():
|
||||
# Initialize generator
|
||||
# FastVideo will automatically download weights from HuggingFace
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
MODEL_PATH,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=True,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True,
|
||||
)
|
||||
|
||||
# Video parameters
|
||||
height = 704
|
||||
width = 1280
|
||||
num_frames = 33
|
||||
action = "forward"
|
||||
action_speed = 0.2
|
||||
|
||||
# Create camera trajectory (Plücker coordinates)
|
||||
camera_states = create_camera_trajectory(
|
||||
action=action,
|
||||
height=height,
|
||||
width=width,
|
||||
num_frames=num_frames,
|
||||
action_speed=action_speed,
|
||||
dtype=torch.bfloat16,
|
||||
)
|
||||
print(f"Camera states shape: {camera_states.shape}")
|
||||
|
||||
# I2V vs T2V: unset GAMECRAFT_I2V_IMAGE -> I2V (default image). Set to "" -> T2V.
|
||||
env_image = os.environ.get("GAMECRAFT_I2V_IMAGE")
|
||||
if env_image is None:
|
||||
image_path = DEFAULT_I2V_IMAGE_URL # default: I2V
|
||||
elif env_image.strip() == "":
|
||||
image_path = None # T2V
|
||||
else:
|
||||
image_path = env_image.strip() # I2V with given URL/path
|
||||
|
||||
is_i2v = image_path is not None
|
||||
prompt = DEFAULT_I2V_PROMPT if is_i2v else DEFAULT_PROMPTS["temple"]
|
||||
print(f"Mode: {'I2V' if is_i2v else 'T2V'}, prompt: {prompt[:60]}...")
|
||||
|
||||
gen_kw = dict(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
camera_states=camera_states,
|
||||
height=height,
|
||||
width=width,
|
||||
num_frames=num_frames,
|
||||
num_inference_steps=50,
|
||||
guidance_scale=6.0,
|
||||
seed=42,
|
||||
fps=24,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
)
|
||||
if is_i2v:
|
||||
gen_kw["image_path"] = image_path
|
||||
generator.generate_video(**gen_kw)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,107 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Run GLM-Image text-to-image generation through FastVideo.
|
||||
|
||||
User story:
|
||||
"I have the HF `zai-org/GLM-Image` checkpoint and want a minimal
|
||||
text-to-image generation command, saved as a PNG."
|
||||
"""
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
from PIL import Image
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
EngineConfig,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
OutputConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
SamplingConfig,
|
||||
)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Run GLM-Image text-to-image generation.")
|
||||
parser.add_argument(
|
||||
"--model-path",
|
||||
default="zai-org/GLM-Image",
|
||||
help="HF id or local diffusers-format GLM-Image weights directory.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
default="image_output/landscape.png",
|
||||
help="Output PNG path.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--prompt",
|
||||
default=("A beautiful landscape photography with rolling hills, "
|
||||
"a winding river, and a vibrant sunset in the background. "
|
||||
"Warm golden light, photorealistic style."),
|
||||
help="Text prompt.",
|
||||
)
|
||||
parser.add_argument("--height", type=int, default=1024)
|
||||
parser.add_argument("--width", type=int, default=1024)
|
||||
parser.add_argument("--steps", type=int, default=50)
|
||||
parser.add_argument("--guidance-scale", type=float, default=1.5)
|
||||
parser.add_argument("--seed", type=int, default=1024)
|
||||
parser.add_argument("--num-gpus", type=int, default=1)
|
||||
parser.add_argument("--tp-size", type=int, default=None)
|
||||
parser.add_argument("--sp-size", type=int, default=None)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
|
||||
output = Path(args.output)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
tp_size = args.tp_size if args.tp_size is not None else (args.num_gpus if args.num_gpus > 1 else 1)
|
||||
sp_size = args.sp_size if args.sp_size is not None else (1 if args.num_gpus > 1 else args.num_gpus)
|
||||
|
||||
# GLM-Image needs trust_remote_code for its AR encoder; offload and the
|
||||
# pipeline class come from the model's registered defaults — don't override.
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=args.model_path,
|
||||
trust_remote_code=True,
|
||||
engine=EngineConfig(
|
||||
num_gpus=args.num_gpus,
|
||||
parallelism=ParallelismConfig(tp_size=tp_size, sp_size=sp_size),
|
||||
),
|
||||
pipeline=PipelineSelection(workload_type="t2i"),
|
||||
)
|
||||
|
||||
generator = VideoGenerator.from_config(generator_config)
|
||||
try:
|
||||
request = GenerationRequest(
|
||||
prompt=args.prompt,
|
||||
sampling=SamplingConfig(
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=1,
|
||||
fps=1,
|
||||
num_inference_steps=args.steps,
|
||||
guidance_scale=args.guidance_scale,
|
||||
seed=args.seed,
|
||||
),
|
||||
output=OutputConfig(
|
||||
output_path=str(output.parent),
|
||||
save_video=False,
|
||||
return_frames=True,
|
||||
),
|
||||
)
|
||||
result = generator.generate(request)
|
||||
if isinstance(result, list):
|
||||
result = result[0]
|
||||
|
||||
frames = result.frames
|
||||
if frames is not None and len(frames):
|
||||
Image.fromarray(frames[0]).save(output)
|
||||
print(f"Saved image to {output}")
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,53 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Kandinsky6 video super-resolution: upscale a low-resolution video x2, x2.25 or x4.
|
||||
|
||||
The request needs no prompt and no height / width / num_frames: the geometry and frame rate follow the input video
|
||||
(the first 121 frames, 5 s at 24 fps, are processed and the source audio is kept). Use
|
||||
``kandinskylab/Kandinsky-6.0-VSR-distilled2steps-5s-Diffusers`` for the 2-step distilled model.
|
||||
See docs/inference/kandinsky6_sr.md.
|
||||
"""
|
||||
import argparse
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("--model-path", default="kandinskylab/Kandinsky-6.0-VSR-5s-Diffusers")
|
||||
parser.add_argument("--video-path", required=True, help="Low-resolution source video")
|
||||
parser.add_argument("--output-path", default="outputs_video/kandinsky6_sr",
|
||||
help="Output directory (the file is named after the input) or .mp4 path")
|
||||
parser.add_argument("--scale", type=float, default=2.25, choices=[2.0, 4.0, 2.25], help="Total upscale factor")
|
||||
parser.add_argument("--num-steps", type=int, default=None,
|
||||
help="Denoising steps per tile (default: the checkpoint's preset, 4 or 2)")
|
||||
parser.add_argument("--tiles-batch-size", type=int, default=1, help="Tiles denoised per DiT call")
|
||||
parser.add_argument("--target-resolution", default=None, help="Optional final size: hd, fullhd, 2k or WxH")
|
||||
parser.add_argument("--seed", type=int, default=42)
|
||||
args = parser.parse_args()
|
||||
|
||||
generator = VideoGenerator.from_pretrained(args.model_path, num_gpus=1)
|
||||
sampling = {"seed": args.seed}
|
||||
if args.num_steps is not None:
|
||||
sampling["num_inference_steps"] = args.num_steps
|
||||
result = generator.generate({
|
||||
"inputs": {
|
||||
"video_path": args.video_path
|
||||
},
|
||||
"sampling": sampling,
|
||||
"output": {
|
||||
"output_path": args.output_path,
|
||||
"save_video": True,
|
||||
"return_frames": False,
|
||||
},
|
||||
"extensions": {
|
||||
"sr_resolution_scale": args.scale,
|
||||
"sr_tiles_batch_size": args.tiles_batch_size,
|
||||
"sr_target_resolution": args.target_resolution,
|
||||
},
|
||||
})
|
||||
print(f"Wrote {result.video_path}")
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,51 +0,0 @@
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
DEFAULT_PROMPT = (
|
||||
"cinematic shot: a giant stone samurai on a stormy cliff above a neon city opens glowing golden eyes and "
|
||||
"raises a katana. Blue lightning strikes the blade, creating a massive shockwave through the clouds. The "
|
||||
"camera rapidly pulls back from a low angle. Photorealistic, epic scale, dark blue and gold lighting, rain, "
|
||||
"sparks, volumetric lightning, blockbuster quality. Audio: heavy rain, deep thunder, metallic sword hum, "
|
||||
"rising brass and choir, electrical crackle, perfectly synchronized lightning impact, sub-bass shockwave. "
|
||||
"No dialogue, text, or logos."
|
||||
)
|
||||
|
||||
OUTPUT_PATH = "video_samples_kandinsky6_ti2va"
|
||||
|
||||
# The base checkpoint uses 50 steps and guidance 5.0. Set KANDINSKY6_MODEL_PATH to run the
|
||||
# distilled pi-Flow checkpoint with this same example:
|
||||
# kandinskylab/Kandinsky-6.0-Pro-distill-5s-Diffusers
|
||||
# or a local directory named ``Kandinsky-6.0-Pro-distill-5s-Diffusers``.
|
||||
# That name selects the registered distilled defaults automatically: 10 steps,
|
||||
# guidance 1.0, eps 1e-6, final_step_size_scale 0.5, and
|
||||
# num_policy_substeps 128.
|
||||
MODEL_PATH = os.environ.get("KANDINSKY6_MODEL_PATH", "kandinskylab/Kandinsky-6.0-Pro-5s-Diffusers")
|
||||
|
||||
IMAGE_PATH = None # e.g. "assets/girl.png" to condition on an image
|
||||
|
||||
|
||||
def main():
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
MODEL_PATH,
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=False,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=True,
|
||||
)
|
||||
|
||||
_ = generator.generate_video(
|
||||
DEFAULT_PROMPT,
|
||||
image_path=IMAGE_PATH,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
height=512,
|
||||
width=768,
|
||||
num_frames=121,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,204 @@
|
||||
"""
|
||||
LongCat Image-to-Video (I2V) Example Script
|
||||
|
||||
This script demonstrates LongCat I2V inference using the FastVideo Python API.
|
||||
LongCat I2V takes an input image and generates a video from it.
|
||||
|
||||
It runs both basic generation (50 steps) and distill+refine generation
|
||||
(16 steps distill + 50 steps refinement to 720p with BSA).
|
||||
|
||||
Usage:
|
||||
python examples/inference/basic/basic_longcat_i2v.py
|
||||
|
||||
Note:
|
||||
Refinement uses 768x768 dimensions where latent (48x48) is divisible by 8,
|
||||
compatible with BSA chunks [4, 4, 8].
|
||||
"""
|
||||
|
||||
import glob
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# Common prompts and settings matching the shell script examples
|
||||
PROMPT = ("A woman sits at a wooden table by the window in a cozy café. She reaches out "
|
||||
"with her right hand, picks up the white coffee cup from the saucer, and gently "
|
||||
"brings it to her lips to take a sip. After drinking, she places the cup back on "
|
||||
"the table and looks out the window, enjoying the peaceful atmosphere.")
|
||||
|
||||
NEGATIVE_PROMPT = ("Bright tones, overexposed, static, blurred details, subtitles, style, works, "
|
||||
"paintings, images, static, overall gray, worst quality, low quality, JPEG compression "
|
||||
"residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, "
|
||||
"deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, "
|
||||
"three legs, many people in the background, walking backwards")
|
||||
|
||||
# Input image path
|
||||
IMAGE_PATH = "assets/girl.png"
|
||||
|
||||
SEED = 42
|
||||
|
||||
|
||||
def basic_generation():
|
||||
"""
|
||||
Run basic LongCat I2V generation (50 steps at 480p).
|
||||
|
||||
This uses the full 50-step denoising process for highest quality.
|
||||
"""
|
||||
print("=" * 60)
|
||||
print("LongCat I2V: Basic Generation (50 steps, 480p)")
|
||||
print("=" * 60)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-I2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False, # set to True if GPU is out of memory
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=False,
|
||||
)
|
||||
|
||||
output_path = "outputs_video/longcat_i2v_basic"
|
||||
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
image_path=IMAGE_PATH,
|
||||
output_path=output_path,
|
||||
save_video=True,
|
||||
height=480,
|
||||
width=480, # Square
|
||||
num_frames=93,
|
||||
num_inference_steps=50,
|
||||
fps=15,
|
||||
guidance_scale=4.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"\nBasic generation complete! Video saved to: {output_path}")
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
def distill_refine_generation():
|
||||
"""
|
||||
Run LongCat I2V with distill+refine pipeline (16 steps + refinement to 768p).
|
||||
|
||||
This uses the distilled LoRA for fast 480p generation (16 steps),
|
||||
then refines to 768p using the refinement LoRA with BSA enabled.
|
||||
"""
|
||||
print("\n" + "=" * 60)
|
||||
print("LongCat I2V: Distill + Refine Pipeline")
|
||||
print("=" * 60)
|
||||
|
||||
# Stage 1: Distilled generation (16 steps at 480p)
|
||||
print("\n[Stage 1] Distilled generation (16 steps, 480p)")
|
||||
print("-" * 40)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-I2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=False,
|
||||
lora_path="FastVideo/LongCat-Video-T2V-Distilled-LoRA",
|
||||
lora_nickname="distilled",
|
||||
)
|
||||
|
||||
distill_output_path = "outputs_video/longcat_i2v_distill"
|
||||
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
image_path=IMAGE_PATH,
|
||||
output_path=distill_output_path,
|
||||
save_video=True,
|
||||
height=480,
|
||||
width=480, # Square
|
||||
num_frames=93,
|
||||
num_inference_steps=16,
|
||||
fps=15,
|
||||
guidance_scale=1.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"Distilled generation complete! Video saved to: {distill_output_path}")
|
||||
generator.shutdown()
|
||||
|
||||
# Stage 2: Refinement (480p -> 768p)
|
||||
print("\n[Stage 2] Refinement (480p -> 768p with BSA)")
|
||||
print("-" * 40)
|
||||
|
||||
# Find the actual saved video file from stage 1
|
||||
video_files = glob.glob(os.path.join(distill_output_path, "*.mp4"))
|
||||
if not video_files:
|
||||
raise FileNotFoundError(f"No video file found in {distill_output_path}")
|
||||
# Use the most recently created video file
|
||||
distill_video_path = max(video_files, key=os.path.getmtime)
|
||||
print(f"Using stage 1 video: {distill_video_path}")
|
||||
|
||||
# Create a new generator with refinement LoRA and BSA enabled
|
||||
# Note: Refinement uses the T2V model (not I2V) since it's upscaling the generated video
|
||||
# For BSA [4, 4, 8]: latent must be divisible by 8
|
||||
# 768x768: latent 48x48, 48%8=0 ✓
|
||||
refine_generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=True,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=True,
|
||||
bsa_sparsity=0.875,
|
||||
bsa_chunk_q=[4, 4, 4],
|
||||
bsa_chunk_k=[4, 4, 4],
|
||||
lora_path="FastVideo/LongCat-Video-T2V-Refinement-LoRA",
|
||||
lora_nickname="refinement",
|
||||
)
|
||||
|
||||
refine_output_path = "outputs_video/longcat_i2v_refine_720p"
|
||||
|
||||
refine_generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
output_path=refine_output_path,
|
||||
save_video=True,
|
||||
refine_from=distill_video_path,
|
||||
t_thresh=0.5,
|
||||
spatial_refine_only=False,
|
||||
num_cond_frames=0,
|
||||
height=720,
|
||||
width=720,
|
||||
num_inference_steps=50,
|
||||
fps=30,
|
||||
guidance_scale=1.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"Refinement complete! Video saved to: {refine_output_path}")
|
||||
refine_generator.shutdown()
|
||||
|
||||
|
||||
def main():
|
||||
"""Run both basic and distill+refine generation pipelines."""
|
||||
print("\n" + "=" * 60)
|
||||
print("LongCat Image-to-Video Example")
|
||||
print("=" * 60 + "\n")
|
||||
|
||||
# Run basic generation
|
||||
basic_generation()
|
||||
|
||||
# Run distill+refine pipeline
|
||||
distill_refine_generation()
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("All generations complete!")
|
||||
print("=" * 60)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,192 @@
|
||||
"""
|
||||
LongCat Text-to-Video (T2V) Example Script
|
||||
|
||||
This script demonstrates LongCat T2V inference using the FastVideo Python API.
|
||||
It runs both basic generation (50 steps) and distill+refine generation
|
||||
(16 steps distill + 50 steps refinement to 720p).
|
||||
|
||||
Usage:
|
||||
python examples/inference/basic/basic_longcat_t2v.py
|
||||
"""
|
||||
|
||||
import glob
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# Common prompts and settings matching the shell script examples
|
||||
PROMPT = ("In a realistic photography style, a white boy around seven or eight years old "
|
||||
"sits on a park bench, wearing a light blue T-shirt, denim shorts, and white sneakers. "
|
||||
"He holds an ice cream cone with vanilla and chocolate flavors, and beside him is a "
|
||||
"medium-sized golden Labrador. Smiling, the boy offers the ice cream to the dog, "
|
||||
"who eagerly licks it with its tongue. The sun is shining brightly, and the background "
|
||||
"features a green lawn and several tall trees, creating a warm and loving scene.")
|
||||
|
||||
NEGATIVE_PROMPT = ("Bright tones, overexposed, static, blurred details, subtitles, style, works, "
|
||||
"paintings, images, static, overall gray, worst quality, low quality, JPEG compression "
|
||||
"residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, "
|
||||
"deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, "
|
||||
"three legs, many people in the background, walking backwards")
|
||||
|
||||
SEED = 42
|
||||
|
||||
|
||||
def basic_generation():
|
||||
"""
|
||||
Run basic LongCat T2V generation (50 steps at 480p).
|
||||
|
||||
This uses the full 50-step denoising process for highest quality.
|
||||
"""
|
||||
print("=" * 60)
|
||||
print("LongCat T2V: Basic Generation (50 steps, 480p)")
|
||||
print("=" * 60)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False, # set to True if GPU is out of memory
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=False,
|
||||
)
|
||||
|
||||
output_path = "outputs_video/longcat_t2v_basic"
|
||||
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
output_path=output_path,
|
||||
save_video=True,
|
||||
height=480,
|
||||
width=832,
|
||||
num_frames=93,
|
||||
num_inference_steps=50,
|
||||
fps=15,
|
||||
guidance_scale=4.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"\nBasic generation complete! Video saved to: {output_path}")
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
def distill_refine_generation():
|
||||
"""
|
||||
Run LongCat T2V with distill+refine pipeline (16 steps + refinement to 720p).
|
||||
|
||||
This uses the distilled LoRA for fast 480p generation (16 steps),
|
||||
then refines to 720p using the refinement LoRA with BSA enabled.
|
||||
"""
|
||||
print("\n" + "=" * 60)
|
||||
print("LongCat T2V: Distill + Refine Pipeline")
|
||||
print("=" * 60)
|
||||
|
||||
# Stage 1: Distilled generation (16 steps at 480p)
|
||||
print("\n[Stage 1] Distilled generation (16 steps, 480p)")
|
||||
print("-" * 40)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=False,
|
||||
lora_path="FastVideo/LongCat-Video-T2V-Distilled-LoRA",
|
||||
lora_nickname="distilled",
|
||||
)
|
||||
|
||||
distill_output_path = "outputs_video/longcat_t2v_distill"
|
||||
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
output_path=distill_output_path,
|
||||
save_video=True,
|
||||
height=480,
|
||||
width=832,
|
||||
num_frames=93,
|
||||
num_inference_steps=16,
|
||||
fps=15,
|
||||
guidance_scale=1.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"Distilled generation complete! Video saved to: {distill_output_path}")
|
||||
generator.shutdown()
|
||||
|
||||
# Stage 2: Refinement (480p -> 720p)
|
||||
print("\n[Stage 2] Refinement (480p -> 720p with BSA)")
|
||||
print("-" * 40)
|
||||
|
||||
# Find the actual saved video file from stage 1
|
||||
video_files = glob.glob(os.path.join(distill_output_path, "*.mp4"))
|
||||
if not video_files:
|
||||
raise FileNotFoundError(f"No video file found in {distill_output_path}")
|
||||
# Use the most recently created video file
|
||||
distill_video_path = max(video_files, key=os.path.getmtime)
|
||||
print(f"Using stage 1 video: {distill_video_path}")
|
||||
|
||||
# Create a new generator with refinement LoRA and BSA enabled
|
||||
refine_generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=True,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=True,
|
||||
bsa_sparsity=0.875,
|
||||
bsa_chunk_q=[4, 4, 8],
|
||||
bsa_chunk_k=[4, 4, 8],
|
||||
lora_path="FastVideo/LongCat-Video-T2V-Refinement-LoRA",
|
||||
lora_nickname="refinement",
|
||||
)
|
||||
|
||||
refine_output_path = "outputs_video/longcat_t2v_refine_720p"
|
||||
|
||||
refine_generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
output_path=refine_output_path,
|
||||
save_video=True,
|
||||
refine_from=distill_video_path,
|
||||
t_thresh=0.5,
|
||||
spatial_refine_only=False,
|
||||
num_cond_frames=0,
|
||||
height=720,
|
||||
width=1280,
|
||||
num_inference_steps=50,
|
||||
fps=30,
|
||||
guidance_scale=1.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"Refinement complete! Video saved to: {refine_output_path}")
|
||||
refine_generator.shutdown()
|
||||
|
||||
|
||||
def main():
|
||||
"""Run both basic and distill+refine generation pipelines."""
|
||||
print("\n" + "=" * 60)
|
||||
print("LongCat Text-to-Video Example")
|
||||
print("=" * 60 + "\n")
|
||||
|
||||
# Run basic generation
|
||||
basic_generation()
|
||||
|
||||
# Run distill+refine pipeline
|
||||
distill_refine_generation()
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("All generations complete!")
|
||||
print("=" * 60)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,218 @@
|
||||
"""
|
||||
LongCat Video Continuation (VC) Example Script
|
||||
|
||||
This script demonstrates LongCat VC inference using the FastVideo Python API.
|
||||
LongCat VC takes an input video and generates a continuation of it.
|
||||
|
||||
It runs both basic generation (50 steps) and distill+refine generation
|
||||
(16 steps distill + 50 steps refinement to 720p).
|
||||
|
||||
Usage:
|
||||
python examples/inference/basic/basic_longcat_vc.py
|
||||
|
||||
Prerequisites:
|
||||
- Ensure the input video exists at assets/motorcycle.mp4
|
||||
(or provide your own video)
|
||||
"""
|
||||
|
||||
import glob
|
||||
import os
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# Common prompts and settings matching the shell script examples
|
||||
PROMPT = ("A person rides a motorcycle along a long, straight road that stretches between "
|
||||
"a body of water and a forested hillside. The rider steadily accelerates, keeping "
|
||||
"the motorcycle centered between the guardrails, while the scenery passes by on "
|
||||
"both sides. The video captures the journey from the rider's perspective, emphasizing "
|
||||
"the sense of motion and adventure.")
|
||||
|
||||
NEGATIVE_PROMPT = ("Bright tones, overexposed, static, blurred details, subtitles, style, works, "
|
||||
"paintings, images, static, overall gray, worst quality, low quality, JPEG compression "
|
||||
"residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, "
|
||||
"deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, "
|
||||
"three legs, many people in the background, walking backwards")
|
||||
|
||||
# Input video path
|
||||
VIDEO_PATH = "assets/motorcycle.mp4"
|
||||
|
||||
# Number of conditioning frames from the input video
|
||||
NUM_COND_FRAMES = 13
|
||||
|
||||
SEED = 42
|
||||
|
||||
|
||||
def basic_generation():
|
||||
"""
|
||||
Run basic LongCat VC generation (50 steps at 480p).
|
||||
|
||||
This uses the full 50-step denoising process for highest quality.
|
||||
"""
|
||||
print("=" * 60)
|
||||
print("LongCat VC: Basic Generation (50 steps, 480p)")
|
||||
print("=" * 60)
|
||||
|
||||
# Check if video exists
|
||||
if not os.path.exists(VIDEO_PATH):
|
||||
raise FileNotFoundError(f"Video not found at {VIDEO_PATH}. "
|
||||
"Please provide a valid video path.")
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-VC-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False, # set to True if GPU is out of memory
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=False,
|
||||
)
|
||||
|
||||
output_path = "outputs_video/longcat_vc_basic"
|
||||
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
video_path=VIDEO_PATH,
|
||||
num_cond_frames=NUM_COND_FRAMES,
|
||||
output_path=output_path,
|
||||
save_video=True,
|
||||
height=480,
|
||||
width=832,
|
||||
num_frames=93,
|
||||
num_inference_steps=50,
|
||||
fps=15,
|
||||
guidance_scale=4.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"\nBasic generation complete! Video saved to: {output_path}")
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
def distill_refine_generation():
|
||||
"""
|
||||
Run LongCat VC with distill+refine pipeline (16 steps + refinement to 720p).
|
||||
|
||||
This uses the distilled LoRA for fast 480p generation (16 steps),
|
||||
then refines to 720p using the refinement LoRA with BSA enabled.
|
||||
"""
|
||||
print("\n" + "=" * 60)
|
||||
print("LongCat VC: Distill + Refine Pipeline")
|
||||
print("=" * 60)
|
||||
|
||||
# Check if video exists
|
||||
if not os.path.exists(VIDEO_PATH):
|
||||
raise FileNotFoundError(f"Video not found at {VIDEO_PATH}. "
|
||||
"Please provide a valid video path.")
|
||||
|
||||
# Stage 1: Distilled generation (16 steps at 480p)
|
||||
print("\n[Stage 1] Distilled generation (16 steps, 480p)")
|
||||
print("-" * 40)
|
||||
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-VC-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=False,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=False,
|
||||
lora_path="FastVideo/LongCat-Video-T2V-Distilled-LoRA",
|
||||
lora_nickname="distilled",
|
||||
)
|
||||
|
||||
distill_output_path = "outputs_video/longcat_vc_distill"
|
||||
|
||||
generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
video_path=VIDEO_PATH,
|
||||
num_cond_frames=NUM_COND_FRAMES,
|
||||
output_path=distill_output_path,
|
||||
save_video=True,
|
||||
height=480,
|
||||
width=832,
|
||||
num_frames=93,
|
||||
num_inference_steps=16,
|
||||
fps=15,
|
||||
guidance_scale=1.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"Distilled generation complete! Video saved to: {distill_output_path}")
|
||||
generator.shutdown()
|
||||
|
||||
# Stage 2: Refinement (480p -> 720p)
|
||||
print("\n[Stage 2] Refinement (480p -> 720p with BSA)")
|
||||
print("-" * 40)
|
||||
|
||||
# Find the actual saved video file from stage 1
|
||||
video_files = glob.glob(os.path.join(distill_output_path, "*.mp4"))
|
||||
if not video_files:
|
||||
raise FileNotFoundError(f"No video file found in {distill_output_path}")
|
||||
# Use the most recently created video file
|
||||
distill_video_path = max(video_files, key=os.path.getmtime)
|
||||
print(f"Using stage 1 video: {distill_video_path}")
|
||||
|
||||
# Create a new generator with refinement LoRA and BSA enabled
|
||||
# Note: Refinement uses the T2V model (not VC) since it's upscaling the generated video
|
||||
refine_generator = VideoGenerator.from_pretrained(
|
||||
"FastVideo/LongCat-Video-T2V-Diffusers",
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=True,
|
||||
dit_cpu_offload=True,
|
||||
vae_cpu_offload=True,
|
||||
text_encoder_cpu_offload=True,
|
||||
pin_cpu_memory=False,
|
||||
enable_bsa=True,
|
||||
bsa_sparsity=0.875,
|
||||
bsa_chunk_q=[4, 4, 8],
|
||||
bsa_chunk_k=[4, 4, 8],
|
||||
lora_path="FastVideo/LongCat-Video-T2V-Refinement-LoRA",
|
||||
lora_nickname="refinement",
|
||||
)
|
||||
|
||||
refine_output_path = "outputs_video/longcat_vc_refine_720p"
|
||||
|
||||
refine_generator.generate_video(
|
||||
prompt=PROMPT,
|
||||
negative_prompt=NEGATIVE_PROMPT,
|
||||
output_path=refine_output_path,
|
||||
save_video=True,
|
||||
refine_from=distill_video_path,
|
||||
t_thresh=0.5,
|
||||
spatial_refine_only=False,
|
||||
num_cond_frames=0, # For refinement, no conditioning frames
|
||||
height=720,
|
||||
width=1280,
|
||||
num_inference_steps=50,
|
||||
fps=30,
|
||||
guidance_scale=1.0,
|
||||
seed=SEED,
|
||||
)
|
||||
|
||||
print(f"Refinement complete! Video saved to: {refine_output_path}")
|
||||
refine_generator.shutdown()
|
||||
|
||||
|
||||
def main():
|
||||
"""Run both basic and distill+refine generation pipelines."""
|
||||
print("\n" + "=" * 60)
|
||||
print("LongCat Video Continuation Example")
|
||||
print("=" * 60 + "\n")
|
||||
|
||||
# Run basic generation
|
||||
basic_generation()
|
||||
|
||||
# Run distill+refine pipeline
|
||||
distill_refine_generation()
|
||||
|
||||
print("\n" + "=" * 60)
|
||||
print("All generations complete!")
|
||||
print("=" * 60)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,51 @@
|
||||
import os
|
||||
|
||||
# Set SLA attention backend BEFORE fastvideo imports
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "SLA_ATTN"
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
OUTPUT_PATH = "video_samples_turbodiffusion"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
# TurboDiffusion: 1-4 step video generation using RCM scheduler + SLA attention
|
||||
# FastVideo will automatically use TurboDiffusionPipeline when specified
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"loayrashid/TurboWan2.1-T2V-1.3B-Diffusers",
|
||||
# FastVideo will automatically handle distributed setup
|
||||
num_gpus=1,
|
||||
use_fsdp_inference=False, # set to True if GPU is out of memory
|
||||
|
||||
# set to false if using RTX 4090
|
||||
# pin_cpu_memory=False,
|
||||
)
|
||||
|
||||
# Generate videos with the same simple API, regardless of GPU count
|
||||
# TurboDiffusion defaults: guidance_scale=1.0 and num_inference_steps=4 (from config)
|
||||
prompt = ("A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones.")
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
# Generate another video with a different prompt, without reloading the model!
|
||||
prompt2 = ("A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic.")
|
||||
video2 = generator.generate_video(
|
||||
prompt2,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,45 @@
|
||||
import os
|
||||
|
||||
# Set SLA attention backend BEFORE fastvideo imports
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "SLA_ATTN"
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
OUTPUT_PATH = "video_samples_turbodiffusion_14B"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
# TurboDiffusion 14B: 1-4 step video generation using RCM scheduler + SLA attention
|
||||
# FastVideo will automatically use TurboDiffusionPipeline when specified
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
"loayrashid/TurboWan2.1-T2V-14B-Diffusers",
|
||||
# 14B model needs more GPUs
|
||||
num_gpus=2,
|
||||
)
|
||||
|
||||
prompt = ("A curious raccoon peers through a vibrant field of yellow sunflowers, its eyes "
|
||||
"wide with interest. The playful yet serene atmosphere is complemented by soft "
|
||||
"natural light filtering through the petals. Mid-shot, warm and cheerful tones.")
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
# Generate another video with a different prompt, without reloading the model!
|
||||
prompt2 = ("A majestic lion strides across the golden savanna, its powerful frame "
|
||||
"glistening under the warm afternoon sun. The tall grass ripples gently in "
|
||||
"the breeze, enhancing the lion's commanding presence. The tone is vibrant, "
|
||||
"embodying the raw energy of the wild. Low angle, steady tracking shot, "
|
||||
"cinematic.")
|
||||
video2 = generator.generate_video(
|
||||
prompt2,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,38 @@
|
||||
import os
|
||||
|
||||
# Set SLA attention backend BEFORE fastvideo imports
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "SLA_ATTN"
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
# Use local model path
|
||||
MODEL_PATH = "loayrashid/TurboWan2.2-I2V-A14B-Diffusers"
|
||||
OUTPUT_PATH = "video_samples_turbodiffusion_i2v"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
# TurboDiffusion I2V: 1-4 step image-to-video generation
|
||||
generator = VideoGenerator.from_pretrained(
|
||||
MODEL_PATH,
|
||||
num_gpus=2,
|
||||
)
|
||||
|
||||
# Example prompt and image for I2V
|
||||
prompt = (
|
||||
"Summer beach vacation style, a white cat wearing sunglasses sits on a surfboard. The fluffy-furred feline gazes directly at the camera with a relaxed expression. Blurred beach scenery forms the background featuring crystal-clear waters, distant green hills, and a blue sky dotted with white clouds. The cat assumes a naturally relaxed posture, as if savoring the sea breeze and warm sunlight. A close-up shot highlights the feline's intricate details and the refreshing atmosphere of the seaside."
|
||||
)
|
||||
|
||||
# Use an example image path
|
||||
image_path = "https://huggingface.co/datasets/YiYiXu/testing-images/resolve/main/wan_i2v_input.JPG"
|
||||
|
||||
video = generator.generate_video(
|
||||
prompt,
|
||||
image_path=image_path,
|
||||
output_path=OUTPUT_PATH,
|
||||
save_video=True,
|
||||
seed=42,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,94 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Run Z-Image-Turbo text-to-image generation through FastVideo.
|
||||
|
||||
User story:
|
||||
"I want the official Z-Image-Turbo defaults and a deterministic PNG from
|
||||
a local or Hugging Face checkpoint."
|
||||
"""
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
EngineConfig,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
OutputConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
SamplingConfig,
|
||||
)
|
||||
|
||||
DEFAULT_PROMPT = (
|
||||
"Young Chinese woman in red Hanfu, intricate embroidery. Impeccable makeup, red floral forehead pattern. "
|
||||
"Elaborate high bun, golden phoenix headdress, red flowers, beads. Holds round folding fan with lady, trees, bird. "
|
||||
"Neon lightning-bolt lamp (⚡️), bright yellow glow, above extended left palm. Soft-lit outdoor night background, "
|
||||
"silhouetted tiered pagoda (西安大雁塔), blurred colorful distant lights.")
|
||||
DEFAULT_REVISION = "f332072aa78be7aecdf3ee76d5c247082da564a6"
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Run Z-Image-Turbo text-to-image generation.")
|
||||
parser.add_argument("--model-path", default="Tongyi-MAI/Z-Image-Turbo")
|
||||
parser.add_argument("--revision", default=DEFAULT_REVISION)
|
||||
parser.add_argument("--output", default="outputs/zimage/zimage_turbo.png")
|
||||
parser.add_argument("--prompt", default=DEFAULT_PROMPT)
|
||||
parser.add_argument("--negative-prompt", default="")
|
||||
parser.add_argument("--height", type=int, default=1024)
|
||||
parser.add_argument("--width", type=int, default=1024)
|
||||
parser.add_argument("--steps", type=int, default=8)
|
||||
parser.add_argument("--guidance-scale", type=float, default=0.0)
|
||||
parser.add_argument("--max-sequence-length", type=int, default=512)
|
||||
parser.add_argument("--cfg-normalization", action=argparse.BooleanOptionalAction, default=False)
|
||||
parser.add_argument("--cfg-truncation", type=float, default=1.0)
|
||||
parser.add_argument("--seed", type=int, default=42)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
output = Path(args.output)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
generator = VideoGenerator.from_config(
|
||||
GeneratorConfig(
|
||||
model_path=args.model_path,
|
||||
revision=args.revision,
|
||||
engine=EngineConfig(
|
||||
num_gpus=1,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=1),
|
||||
use_fsdp_inference=False,
|
||||
),
|
||||
# The model registry selects the native zimage_turbo preset.
|
||||
pipeline=PipelineSelection(workload_type="t2i"),
|
||||
))
|
||||
try:
|
||||
generator.generate(
|
||||
GenerationRequest(
|
||||
prompt=args.prompt,
|
||||
negative_prompt=args.negative_prompt,
|
||||
sampling=SamplingConfig(
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=1,
|
||||
fps=1,
|
||||
num_inference_steps=args.steps,
|
||||
guidance_scale=args.guidance_scale,
|
||||
max_sequence_length=args.max_sequence_length,
|
||||
cfg_normalization=args.cfg_normalization,
|
||||
cfg_truncation=args.cfg_truncation,
|
||||
seed=args.seed,
|
||||
),
|
||||
output=OutputConfig(
|
||||
output_path=str(output),
|
||||
save_video=True,
|
||||
return_frames=False,
|
||||
),
|
||||
))
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,149 +0,0 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Time a resident FastH3 eight-forward Spark recipe with the release prompts.
|
||||
|
||||
One process loads the model, runs one excluded ceramics warmup, then times
|
||||
each prompt at least twice. Every call writes a video. This script does not alter
|
||||
the V2 schedule, VSA sparsity, or video resolution.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import statistics
|
||||
import time
|
||||
from copy import deepcopy
|
||||
from pathlib import Path
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api.parser import load_raw_config, parse_config
|
||||
from fastvideo.api.schema import RunConfig
|
||||
|
||||
PROMPT_IDS = ("latency-ceramics-005", "latency-harbor-005")
|
||||
|
||||
|
||||
def _stage_metrics(result: object) -> dict[str, dict]:
|
||||
logging_info = getattr(result, "logging_info", None)
|
||||
stages = getattr(logging_info, "stages", None)
|
||||
if isinstance(logging_info, dict):
|
||||
stages = logging_info.get("stages", stages)
|
||||
if not isinstance(stages, dict):
|
||||
return {}
|
||||
return {name: metrics for name, metrics in stages.items() if isinstance(metrics, dict)}
|
||||
|
||||
|
||||
def _stage_seconds(metrics: dict[str, dict]) -> dict[str, float]:
|
||||
return {name: float(stage["execution_time"]) for name, stage in metrics.items()
|
||||
if stage.get("execution_time") is not None}
|
||||
|
||||
|
||||
def _peak_mb(metrics: dict[str, dict], key: str) -> float | None:
|
||||
values = [float(stage[key]) for stage in metrics.values() if stage.get(key) is not None]
|
||||
return max(values) if values else None
|
||||
|
||||
|
||||
def _stage_total(stages: dict[str, float], fragment: str, exclude: str | None = None) -> float | None:
|
||||
matches = [
|
||||
seconds for name, seconds in stages.items()
|
||||
if fragment in name.lower() and (exclude is None or exclude not in name.lower())
|
||||
]
|
||||
return sum(matches) if matches else None
|
||||
|
||||
|
||||
def _request(base: RunConfig, prompt: str, frames: int, width: int, height: int,
|
||||
output: Path):
|
||||
request = deepcopy(base.request)
|
||||
request.prompt = prompt
|
||||
request.inputs.prompt_path = None
|
||||
request.sampling.num_frames = frames
|
||||
request.sampling.width = width
|
||||
request.sampling.height = height
|
||||
request.output.output_path = str(output)
|
||||
request.output.save_video = True
|
||||
request.output.return_frames = False
|
||||
return request
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--config", type=Path, required=True)
|
||||
parser.add_argument("--prompts", type=Path, required=True)
|
||||
parser.add_argument("--output-dir", type=Path, required=True)
|
||||
parser.add_argument("--model-path", type=Path)
|
||||
parser.add_argument("--frames", type=int, default=124)
|
||||
parser.add_argument("--width", type=int, default=832)
|
||||
parser.add_argument("--height", type=int, default=480)
|
||||
parser.add_argument("--repeats", type=int, default=2)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.frames not in (124, 243):
|
||||
parser.error("use 124 frames for roughly five seconds or 243 for roughly ten seconds")
|
||||
if args.repeats < 2:
|
||||
parser.error("the release protocol requires at least two timed calls")
|
||||
|
||||
config = parse_config(RunConfig, load_raw_config(args.config))
|
||||
if args.model_path:
|
||||
config.generator.model_path = str(args.model_path)
|
||||
if config.request.sampling.num_inference_steps != 9:
|
||||
parser.error("the V2 contract requires nine sigma points for eight DiT forwards")
|
||||
if config.generator.engine.offload.lazy_module_load is not False:
|
||||
parser.error("the resident recipe requires lazy_module_load: false")
|
||||
contract = Path(config.generator.model_path) / "fastvideo_inference.json"
|
||||
if not contract.is_file():
|
||||
parser.error(f"missing trained V2 schedule: {contract}")
|
||||
inference = json.loads(contract.read_text())
|
||||
if inference.get("num_inference_steps") != 9 or inference.get("transformer_forwards") != 8:
|
||||
parser.error("the checkpoint is not the trained V2 eight-forward schedule")
|
||||
|
||||
prompts = json.loads(args.prompts.read_text())
|
||||
if any(prompt_id not in prompts for prompt_id in PROMPT_IDS):
|
||||
parser.error(f"prompt JSON must contain {', '.join(PROMPT_IDS)}")
|
||||
args.output_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
generator = VideoGenerator.from_config(config.generator)
|
||||
try:
|
||||
for prompt_index, prompt_id in enumerate(PROMPT_IDS):
|
||||
times = []
|
||||
first_index = 0 if prompt_index == 0 else 1
|
||||
for index in range(first_index, args.repeats + 1):
|
||||
warmup = index == 0
|
||||
label = "warmup" if warmup else f"run-{index:02d}"
|
||||
requested_path = args.output_dir / f"{prompt_id}-{args.width}x{args.height}-{args.frames}-{label}.mp4"
|
||||
request = _request(config, prompts[prompt_id], args.frames, args.width, args.height,
|
||||
requested_path)
|
||||
started = time.perf_counter()
|
||||
result = generator.generate(request)
|
||||
wall = time.perf_counter() - started
|
||||
output = Path(result.video_path) if result.video_path else requested_path
|
||||
if not output.is_file():
|
||||
raise RuntimeError(f"generation returned without an MP4: {output}")
|
||||
metrics = _stage_metrics(result)
|
||||
stages = _stage_seconds(metrics)
|
||||
row = {
|
||||
"prompt_id": prompt_id,
|
||||
"warmup": warmup,
|
||||
"frames": args.frames,
|
||||
"width": args.width,
|
||||
"height": args.height,
|
||||
"e2e_seconds": round(wall, 3),
|
||||
"denoise_seconds": _stage_total(stages, "denois"),
|
||||
"decode_seconds": _stage_total(stages, "decod", exclude="postdecode"),
|
||||
"postprocess_seconds": _stage_total(stages, "postdecode"),
|
||||
"peak_memory_mb": _peak_mb(metrics, "peak_allocated_mb"),
|
||||
"peak_reserved_mb": _peak_mb(metrics, "peak_reserved_mb"),
|
||||
"result_peak_memory_mb": result.peak_memory_mb,
|
||||
"stages": stages,
|
||||
"mp4": str(output),
|
||||
}
|
||||
print(json.dumps(row, sort_keys=True), flush=True)
|
||||
if not warmup:
|
||||
times.append(wall)
|
||||
print(json.dumps({"prompt_id": prompt_id, "timed_runs": len(times),
|
||||
"median_e2e_seconds": round(statistics.median(times), 3)},
|
||||
sort_keys=True), flush=True)
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,120 @@
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
"""Run GLM-Image image-to-image (edit) generation through FastVideo.
|
||||
|
||||
User story:
|
||||
"I have the HF `zai-org/GLM-Image` checkpoint and a condition image, and
|
||||
want a minimal edit command (text + image -> edited image), saved as a PNG."
|
||||
|
||||
GLM-Image is a single unified pipeline: passing a condition image switches it
|
||||
from text-to-image to the edit path (the condition enters the DiT via a KV-cache
|
||||
write pass), so the generator config is identical to `basic_glm_image.py` — the
|
||||
`inputs.pil_image` on the request is what selects the edit mode.
|
||||
"""
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
from PIL import Image
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
EngineConfig,
|
||||
GenerationRequest,
|
||||
GeneratorConfig,
|
||||
InputConfig,
|
||||
OutputConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
SamplingConfig,
|
||||
)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Run GLM-Image image-to-image (edit) generation.")
|
||||
parser.add_argument(
|
||||
"--model-path",
|
||||
default="zai-org/GLM-Image",
|
||||
help="HF id or local diffusers-format GLM-Image weights directory.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--image",
|
||||
default="assets/images/couple.jpg",
|
||||
help="Condition image to edit.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--output",
|
||||
default="image_output/edited.png",
|
||||
help="Output PNG path.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--prompt",
|
||||
default="Change the background to a snowy mountain landscape at golden hour.",
|
||||
help="Edit instruction.",
|
||||
)
|
||||
parser.add_argument("--height", type=int, default=1024)
|
||||
parser.add_argument("--width", type=int, default=1024)
|
||||
parser.add_argument("--steps", type=int, default=50)
|
||||
parser.add_argument("--guidance-scale", type=float, default=1.5)
|
||||
parser.add_argument("--seed", type=int, default=1024)
|
||||
parser.add_argument("--num-gpus", type=int, default=1)
|
||||
parser.add_argument("--tp-size", type=int, default=None)
|
||||
parser.add_argument("--sp-size", type=int, default=None)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
|
||||
output = Path(args.output)
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
condition = Image.open(args.image).convert("RGB")
|
||||
tp_size = args.tp_size if args.tp_size is not None else (args.num_gpus if args.num_gpus > 1 else 1)
|
||||
sp_size = args.sp_size if args.sp_size is not None else (1 if args.num_gpus > 1 else args.num_gpus)
|
||||
|
||||
# GLM-Image needs trust_remote_code for its AR encoder; offload and the
|
||||
# pipeline class come from the model's registered defaults — don't override.
|
||||
# The pipeline is registered as t2i; passing inputs.pil_image below switches
|
||||
# it to the edit path.
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=args.model_path,
|
||||
trust_remote_code=True,
|
||||
engine=EngineConfig(
|
||||
num_gpus=args.num_gpus,
|
||||
parallelism=ParallelismConfig(tp_size=tp_size, sp_size=sp_size),
|
||||
),
|
||||
pipeline=PipelineSelection(workload_type="t2i"),
|
||||
)
|
||||
|
||||
generator = VideoGenerator.from_config(generator_config)
|
||||
try:
|
||||
request = GenerationRequest(
|
||||
prompt=args.prompt,
|
||||
inputs=InputConfig(pil_image=condition),
|
||||
sampling=SamplingConfig(
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=1,
|
||||
fps=1,
|
||||
num_inference_steps=args.steps,
|
||||
guidance_scale=args.guidance_scale,
|
||||
seed=args.seed,
|
||||
),
|
||||
output=OutputConfig(
|
||||
output_path=str(output.parent),
|
||||
save_video=False,
|
||||
return_frames=True,
|
||||
),
|
||||
)
|
||||
result = generator.generate(request)
|
||||
if isinstance(result, list):
|
||||
result = result[0]
|
||||
|
||||
frames = result.frames
|
||||
if frames is not None and len(frames):
|
||||
Image.fromarray(frames[0]).save(output)
|
||||
print(f"Saved image to {output}")
|
||||
finally:
|
||||
generator.shutdown()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -190,37 +190,34 @@ def run(args: argparse.Namespace) -> None:
|
||||
taeh3_chunk_size=args.taeh3_chunk_size,
|
||||
prompt_cache_dir=args.prompt_cache_dir,
|
||||
)
|
||||
try:
|
||||
result = pipeline.generate(
|
||||
args.prompt,
|
||||
output_path=args.output_path,
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=args.num_frames,
|
||||
seed=args.seed,
|
||||
num_steps=args.steps,
|
||||
tiled_video_decode=args.tiled_video_decode,
|
||||
vae_tile_height=args.vae_tile_height,
|
||||
vae_tile_width=args.vae_tile_width,
|
||||
inter_step_cooldown_s=args.inter_step_cooldown_s,
|
||||
fast=args.fast,
|
||||
fast_factor=args.fast_factor,
|
||||
fast_sharpen=args.fast_sharpen,
|
||||
rife_weights_dir=args.rife_weights_dir,
|
||||
fast_spatial=args.fast_spatial,
|
||||
fast_spatial_scale=args.fast_spatial_scale,
|
||||
fast_spatial_upsample_mode=args.fast_spatial_upsample_mode,
|
||||
fast_spatial_sharpen=args.fast_spatial_sharpen,
|
||||
vsa=args.vsa,
|
||||
vsa_sparsity=args.vsa_sparsity,
|
||||
vsa_tile_size=args.vsa_tile_size,
|
||||
vsa_prefix_mode=args.vsa_prefix_mode,
|
||||
vsa_dense_first_n_steps=args.vsa_dense_first_n_steps,
|
||||
vsa_dense_layers=args.vsa_dense_layers,
|
||||
vsa_impl=args.vsa_impl,
|
||||
)
|
||||
finally:
|
||||
pipeline.close()
|
||||
result = pipeline.generate(
|
||||
args.prompt,
|
||||
output_path=args.output_path,
|
||||
height=args.height,
|
||||
width=args.width,
|
||||
num_frames=args.num_frames,
|
||||
seed=args.seed,
|
||||
num_steps=args.steps,
|
||||
tiled_video_decode=args.tiled_video_decode,
|
||||
vae_tile_height=args.vae_tile_height,
|
||||
vae_tile_width=args.vae_tile_width,
|
||||
inter_step_cooldown_s=args.inter_step_cooldown_s,
|
||||
fast=args.fast,
|
||||
fast_factor=args.fast_factor,
|
||||
fast_sharpen=args.fast_sharpen,
|
||||
rife_weights_dir=args.rife_weights_dir,
|
||||
fast_spatial=args.fast_spatial,
|
||||
fast_spatial_scale=args.fast_spatial_scale,
|
||||
fast_spatial_upsample_mode=args.fast_spatial_upsample_mode,
|
||||
fast_spatial_sharpen=args.fast_spatial_sharpen,
|
||||
vsa=args.vsa,
|
||||
vsa_sparsity=args.vsa_sparsity,
|
||||
vsa_tile_size=args.vsa_tile_size,
|
||||
vsa_prefix_mode=args.vsa_prefix_mode,
|
||||
vsa_dense_first_n_steps=args.vsa_dense_first_n_steps,
|
||||
vsa_dense_layers=args.vsa_dense_layers,
|
||||
vsa_impl=args.vsa_impl,
|
||||
)
|
||||
print(json.dumps({
|
||||
"video_path": result.video_path,
|
||||
"timings_s": {k: round(v, 2) for k, v in result.timings.items()},
|
||||
|
||||
@@ -1,53 +0,0 @@
|
||||
# FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FLASHINFER_CUDA_ARCH_LIST=12.0a \
|
||||
# fastvideo serve --config examples/serving/openai_compacth3_rtx5090.yaml
|
||||
generator:
|
||||
model_path: ./CompactH3
|
||||
engine:
|
||||
num_gpus: 1
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: true
|
||||
image_encoder: true
|
||||
vae: false
|
||||
pin_cpu_memory: true
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: false
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
experimental:
|
||||
attention_backend: ATTN_QAT_INFER
|
||||
h3_sequential_load: true
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: false
|
||||
video_decode_backend: h3-vae
|
||||
|
||||
server:
|
||||
host: 127.0.0.1
|
||||
port: 8000
|
||||
output_dir: outputs/openai_compacth3_rtx5090
|
||||
served_model_name: compacth3
|
||||
|
||||
default_request:
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
height: 480
|
||||
width: 832
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 5
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
seed: 2026
|
||||
output:
|
||||
return_frames: false
|
||||
@@ -1,53 +0,0 @@
|
||||
# FASTVIDEO_ATTENTION_BACKEND=ATTN_QAT_INFER FASTVIDEO_VSA_SM100A=0 FASTVIDEO_FA4=0 \
|
||||
# FLASHINFER_CUDA_ARCH_LIST=12.0a \
|
||||
# fastvideo serve --config examples/serving/openai_compacth3_rtx_pro6000.yaml
|
||||
generator:
|
||||
model_path: ./CompactH3
|
||||
engine:
|
||||
num_gpus: 1
|
||||
use_fsdp_inference: false
|
||||
quantization:
|
||||
transformer_quant: NVFP4
|
||||
layer_profile: h3_dit
|
||||
parallelism:
|
||||
tp_size: 1
|
||||
sp_size: 1
|
||||
offload:
|
||||
dit: false
|
||||
dit_layerwise: false
|
||||
text_encoder: false
|
||||
image_encoder: true
|
||||
vae: false
|
||||
pin_cpu_memory: true
|
||||
lazy_module_load: false
|
||||
compile:
|
||||
enabled: false
|
||||
vae_enabled: true
|
||||
pipeline:
|
||||
workload_type: t2v
|
||||
experimental:
|
||||
attention_backend: ATTN_QAT_INFER
|
||||
h3_sequential_load: false
|
||||
inference_torch_compile: false
|
||||
vae_parallel_decode: false
|
||||
video_decode_backend: h3-vae
|
||||
|
||||
server:
|
||||
host: 127.0.0.1
|
||||
port: 8000
|
||||
output_dir: outputs/openai_compacth3_rtx_pro6000
|
||||
served_model_name: compacth3
|
||||
|
||||
default_request:
|
||||
negative_prompt: ""
|
||||
sampling:
|
||||
height: 768
|
||||
width: 1344
|
||||
num_frames: 124
|
||||
fps: 24
|
||||
num_inference_steps: 5
|
||||
guidance_scale: 1.0
|
||||
batch_cfg: false
|
||||
seed: 2026
|
||||
output:
|
||||
return_frames: false
|
||||
@@ -25,6 +25,5 @@ default_request:
|
||||
num_frames: 81
|
||||
height: 720
|
||||
width: 1280
|
||||
fps: 16
|
||||
output:
|
||||
return_frames: false
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
# OpenAI-compatible server for Wan2.2 TI2V 5B. Supports both text-to-video
|
||||
# (omit the image reference) and image-to-video (set input_reference or
|
||||
# image_reference) from the same model. The checked-in cookbook clients send
|
||||
# text prompts only. The sampling block pins the values the cookbook displays;
|
||||
# they match the registered preset, and a test keeps them in sync.
|
||||
# image_reference) from the same model.
|
||||
# fastvideo serve --config examples/serving/openai_wan22_ti2v_5b.yaml
|
||||
generator:
|
||||
model_path: Wan-AI/Wan2.2-TI2V-5B-Diffusers
|
||||
@@ -25,10 +23,5 @@ server:
|
||||
served_model_name: wan22-ti2v-5b
|
||||
|
||||
default_request:
|
||||
sampling:
|
||||
height: 704
|
||||
width: 1280
|
||||
num_frames: 121
|
||||
fps: 24
|
||||
output:
|
||||
return_frames: false
|
||||
|
||||
@@ -187,7 +187,7 @@ to a supported dense backend while the FP4 linear layers still run.
|
||||
`Kandinsky5DMDConfig` override, generating deterministically (fixed
|
||||
seed), and comparing MS-SSIM against a committed reference video. The
|
||||
test fails (does not skip) if the reference is missing; record it once on
|
||||
a sanctioned GPU box with `FASTVIDEO_TEST_KANDINSKY5_E2E_WRITE_REFERENCE=1`, review the
|
||||
a sanctioned GPU box with `KANDINSKY5_E2E_WRITE_REFERENCE=1`, review the
|
||||
written video, and commit it (see the test's module docstring). Nightly
|
||||
tests are not collected per-PR (`fastvideo/tests/contract/
|
||||
test_ci_test_collection.py` allowlists the directory), so run it
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
# LongCat-Video T2V 13.6B bidirectional finetune.
|
||||
|
||||
models:
|
||||
student:
|
||||
_target_: fastvideo.train.models.longcat.LongCatModel
|
||||
init_from: FastVideo/LongCat-Video-T2V-Diffusers
|
||||
trainable: true
|
||||
|
||||
method:
|
||||
_target_: fastvideo.train.methods.fine_tuning.finetune.FineTuneMethod
|
||||
|
||||
training:
|
||||
model_path: FastVideo/LongCat-Video-T2V-Diffusers
|
||||
|
||||
distributed:
|
||||
num_gpus: 8
|
||||
sp_size: 1
|
||||
tp_size: 1
|
||||
hsdp_replicate_dim: 1
|
||||
hsdp_shard_dim: 8
|
||||
|
||||
data:
|
||||
data_path: data/LongCat-Syn
|
||||
dataloader_num_workers: 4
|
||||
train_batch_size: 1
|
||||
training_cfg_rate: 0.0
|
||||
seed: 1000
|
||||
num_latent_t: 20
|
||||
num_height: 480
|
||||
num_width: 848
|
||||
num_frames: 77
|
||||
|
||||
optimizer:
|
||||
learning_rate: 1.0e-6
|
||||
betas: [0.9, 0.999]
|
||||
weight_decay: 0.01
|
||||
lr_scheduler: constant
|
||||
lr_warmup_steps: 0
|
||||
|
||||
loop:
|
||||
max_train_steps: 4000
|
||||
gradient_accumulation_steps: 1
|
||||
|
||||
checkpoint:
|
||||
output_dir: outputs/longcat_finetune
|
||||
training_state_checkpointing_steps: 1000
|
||||
checkpoints_total_limit: 3
|
||||
|
||||
tracker:
|
||||
project_name: fastvideo
|
||||
run_name: longcat_finetune
|
||||
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
|
||||
callbacks:
|
||||
grad_clip:
|
||||
_target_: fastvideo.train.callbacks.grad_clip.GradNormClipCallback
|
||||
max_grad_norm: 1.0
|
||||
validation:
|
||||
_target_: fastvideo.train.callbacks.validation.ValidationCallback
|
||||
pipeline_target: fastvideo.pipelines.basic.longcat.longcat_pipeline.LongCatPipeline
|
||||
dataset_file: data/validation_prompts.json
|
||||
every_steps: 100
|
||||
sampling_steps: [50]
|
||||
guidance_scale: 5.0
|
||||
|
||||
pipeline:
|
||||
# Match the released LongCat scheduler config. flow_shift=0.0 collapses
|
||||
# FlowMatch training timesteps to zero in FastVideo's scheduler.
|
||||
flow_shift: 12.0
|
||||
@@ -0,0 +1,89 @@
|
||||
# TurboWan 2.1 T2V 1.3B LoRA finetune with SLA attention.
|
||||
#
|
||||
# Notes:
|
||||
# - TurboDiffusion checkpoints expect the SLA attention backend. The training
|
||||
# entrypoint now auto-selects `SLA_ATTN` for model IDs containing
|
||||
# `turbodiffusion` or `turbowan`, unless you override the environment
|
||||
# variable manually.
|
||||
# - This debug config reuses the existing crush-smol T2V parquet recipe. If you
|
||||
# preprocess data with the official TurboWan recipe, update the temporal and
|
||||
# spatial fields to match it.
|
||||
|
||||
models:
|
||||
student:
|
||||
_target_: fastvideo.train.models.wan.WanModel
|
||||
init_from: loayrashid/TurboWan2.1-T2V-1.3B-Diffusers
|
||||
trainable: true
|
||||
lora:
|
||||
enable: true
|
||||
rank: 16
|
||||
alpha: 32
|
||||
target_modules:
|
||||
- to_q
|
||||
- to_k
|
||||
- to_v
|
||||
- to_out
|
||||
|
||||
method:
|
||||
_target_: fastvideo.train.methods.fine_tuning.finetune.FineTuneMethod
|
||||
|
||||
training:
|
||||
dit_precision: bf16
|
||||
|
||||
distributed:
|
||||
num_gpus: 1
|
||||
sp_size: 1
|
||||
tp_size: 1
|
||||
hsdp_replicate_dim: 1
|
||||
hsdp_shard_dim: 1
|
||||
|
||||
data:
|
||||
data_path: data/crush-smol_processed_t2v/combined_parquet_dataset
|
||||
dataloader_num_workers: 2
|
||||
train_batch_size: 1
|
||||
training_cfg_rate: 0.0
|
||||
seed: 42
|
||||
num_latent_t: 20
|
||||
num_height: 480
|
||||
num_width: 832
|
||||
num_frames: 77
|
||||
|
||||
optimizer:
|
||||
learning_rate: 1.0e-5
|
||||
betas: [0.9, 0.999]
|
||||
weight_decay: 0.01
|
||||
lr_scheduler: constant
|
||||
lr_warmup_steps: 0
|
||||
|
||||
loop:
|
||||
max_train_steps: 10
|
||||
gradient_accumulation_steps: 5
|
||||
|
||||
checkpoint:
|
||||
output_dir: outputs/turbo_wan_t2v_lora_sla
|
||||
training_state_checkpointing_steps: 5
|
||||
checkpoints_total_limit: 2
|
||||
resume_from_checkpoint: null
|
||||
|
||||
tracker:
|
||||
project_name: fastvideo
|
||||
run_name: turbo_wan_t2v_lora_sla
|
||||
|
||||
model:
|
||||
enable_gradient_checkpointing_type: full
|
||||
|
||||
callbacks:
|
||||
grad_clip:
|
||||
_target_: fastvideo.train.callbacks.grad_clip.GradNormClipCallback
|
||||
max_grad_norm: 1.0
|
||||
|
||||
validation:
|
||||
_target_: fastvideo.train.callbacks.validation.ValidationCallback
|
||||
pipeline_target: fastvideo.pipelines.basic.turbodiffusion.turbodiffusion_pipeline.TurboDiffusionPipeline
|
||||
dataset_file: examples/training/finetune/wan_t2v_1.3B/crush_smol/validation.json
|
||||
every_steps: 5
|
||||
sampling_steps: [4]
|
||||
guidance_scale: 1.0
|
||||
|
||||
pipeline:
|
||||
flow_shift: 3
|
||||
@@ -106,32 +106,12 @@ def preprocess_qkv(q: torch.Tensor,
|
||||
if enable_smoothing_q:
|
||||
delta_s = torch.matmul(qm, k.transpose(-2, -1)).to(torch.float32).contiguous()
|
||||
else: # used to disable q smoothing
|
||||
delta_s = _zero_delta_s(q.shape[0], q.shape[1], k.shape[2], q.device)
|
||||
B, H, L, D = q.shape
|
||||
delta_s = torch.zeros((B, H, L // BLOCK_M, k.shape[2]), device=q.device, dtype=torch.float32)
|
||||
|
||||
return q, k, v, delta_s
|
||||
|
||||
|
||||
_ZERO_DELTA_S: dict = {}
|
||||
|
||||
|
||||
def _zero_delta_s(batch: int, heads: int, kv_len: int, device: torch.device) -> torch.Tensor:
|
||||
"""Cached all-zero delta_s for unsmoothed Q, read with per_block_mean=False.
|
||||
|
||||
The kernel reads delta_s as a contiguous [B, H, rows, KL] tensor with one
|
||||
row per query block when per_block_mean is set and a single row otherwise.
|
||||
With Q smoothing off every row is zero, so one shared row replaces the
|
||||
[B, H, L/128, KL] tensor that was allocated and zero-filled on every call
|
||||
(9.5 GB at 73k tokens, whose int32 batch stride also broke the TMA
|
||||
descriptor). The kernel only reads it.
|
||||
"""
|
||||
key = (batch, heads, kv_len, device)
|
||||
zeros = _ZERO_DELTA_S.get(key)
|
||||
if zeros is None:
|
||||
zeros = torch.zeros((batch, heads, 1, kv_len), device=device, dtype=torch.float32)
|
||||
_ZERO_DELTA_S[key] = zeros
|
||||
return zeros
|
||||
|
||||
|
||||
def scale_and_quant_fp4(x: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
assert x.ndim == 4
|
||||
B, H, N, D = x.shape
|
||||
@@ -174,191 +154,6 @@ def blockscaled_fp4_attn(qlist: Tuple,
|
||||
softmax_scale, is_causal, per_block_mean, is_bf16, single_level_p_quant)
|
||||
|
||||
|
||||
def blockscaled_fp4_attn_sparse(qlist: Tuple,
|
||||
klist: Tuple,
|
||||
vlist: Tuple,
|
||||
delta_s: torch.Tensor,
|
||||
KL: int,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
kv_valid: torch.Tensor | None = None,
|
||||
q2k_quad: torch.Tensor | None = None,
|
||||
per_block_mean: bool = True,
|
||||
is_bf16: bool = True,
|
||||
single_level_p_quant: bool = False,
|
||||
sm_scale: float | None = None):
|
||||
softmax_scale = sm_scale if sm_scale is not None else (qlist[0].shape[-1] * 2)**(-0.5)
|
||||
return fp4attn_cuda.fwd_sparse(qlist[0], klist[0], vlist[0], qlist[1], klist[1], vlist[1], delta_s, KL, None,
|
||||
softmax_scale, per_block_mean, is_bf16, single_level_p_quant, q2k_idx, q2k_num,
|
||||
kv_valid, q2k_quad)
|
||||
|
||||
|
||||
HALF_N = BLOCK_N // 2
|
||||
|
||||
|
||||
def vsa_tile_mask_to_fp4_blocks(
|
||||
tile_mask: torch.Tensor,
|
||||
tile_tokens: int,
|
||||
tile_valid: torch.Tensor | None = None,
|
||||
validate: bool = False,
|
||||
) -> Tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor | None]:
|
||||
"""Convert a VSA tile mask into the FP4 kernel's block lists.
|
||||
|
||||
``tile_mask`` is a ``[B, H, T, T]`` bool mask over VSA tiles of
|
||||
``tile_tokens`` tokens (query tile, key tile); ``tile_tokens`` is 64 or a
|
||||
multiple of 128. ``tile_valid`` holds each tile's valid token count (valid
|
||||
tokens first, padding last, as VSA tiling lays them out). The kernel works
|
||||
on 128x128 blocks, so 64-token tiles are paired: a block is listed when any
|
||||
of its four 64x64 quadrants is selected, and ``q2k_quad`` carries which.
|
||||
|
||||
Lists run in descending block order because the kernel visits them last
|
||||
entry first: block 0 (the first prefix tile, which VSA-H3's exempt mode
|
||||
gives every query) is then visited first, so every query row starts from a
|
||||
finite running max. ``validate=True`` checks that (host sync).
|
||||
|
||||
Returns ``(q2k_idx [B, H, M, N], q2k_num [B, H, M], kv_valid [2N],
|
||||
q2k_quad [B, H, M, N] | None)``; the queries cover ``M * 128`` padded rows.
|
||||
"""
|
||||
if tile_tokens % HALF_N or (tile_tokens > HALF_N and tile_tokens % BLOCK_N):
|
||||
raise ValueError(f"tile_tokens={tile_tokens} must be {HALF_N} or a multiple of {BLOCK_N}")
|
||||
batch, heads, n_tiles, _ = tile_mask.shape
|
||||
device = tile_mask.device
|
||||
halves_per_tile = tile_tokens // HALF_N
|
||||
half_mask = tile_mask
|
||||
if halves_per_tile > 1:
|
||||
half_mask = half_mask.repeat_interleave(halves_per_tile, dim=2).repeat_interleave(halves_per_tile, dim=3)
|
||||
n_halves = n_tiles * halves_per_tile
|
||||
if tile_valid is None:
|
||||
half_valid = torch.full((n_halves, ), HALF_N, device=device, dtype=torch.int32)
|
||||
else:
|
||||
offsets = torch.arange(halves_per_tile, device=device, dtype=torch.int32) * HALF_N
|
||||
half_valid = (tile_valid.to(torch.int32)[:, None] - offsets[None, :]).clamp(0, HALF_N).reshape(-1)
|
||||
if n_halves % 2:
|
||||
# Pad to whole 128-blocks: the extra key half is empty, the extra query
|
||||
# half (discarded output) attends block 0 only.
|
||||
half_mask = F.pad(half_mask, (0, 1, 0, 1), value=False)
|
||||
half_mask[:, :, -1, 0] = True
|
||||
half_valid = F.pad(half_valid, (0, 1), value=0)
|
||||
n_halves += 1
|
||||
half_mask = half_mask & (half_valid > 0)[None, None, None, :]
|
||||
n_blocks = n_halves // 2
|
||||
quads = half_mask.view(batch, heads, n_blocks, 2, n_blocks, 2)
|
||||
weights = torch.tensor([[1, 2], [4, 8]], device=device, dtype=torch.uint8) # [row_half, col_half]
|
||||
quad = (quads.to(torch.uint8) * weights[None, None, None, :, None, :]).sum(dim=(3, 5), dtype=torch.uint8)
|
||||
block_mask = quad != 0
|
||||
# Descending compaction without a sort: position = running count from the right.
|
||||
rev = block_mask.flip(-1)
|
||||
pos = rev.cumsum(-1, dtype=torch.int32) - 1
|
||||
q2k_num = (pos[..., -1] + 1).contiguous()
|
||||
cols = torch.arange(n_blocks - 1, -1, -1, device=device, dtype=torch.int32).expand_as(pos)
|
||||
slot = torch.where(rev, pos, torch.full_like(pos, n_blocks)).long()
|
||||
q2k_idx = torch.zeros((batch, heads, n_blocks, n_blocks + 1), device=device, dtype=torch.int32)
|
||||
q2k_idx.scatter_(-1, slot, cols)
|
||||
q2k_idx = q2k_idx[..., :n_blocks].contiguous()
|
||||
kv_valid = half_valid.contiguous()
|
||||
q2k_quad = None
|
||||
if halves_per_tile == 1:
|
||||
q2k_quad = torch.zeros((batch, heads, n_blocks, n_blocks + 1), device=device, dtype=torch.uint8)
|
||||
q2k_quad.scatter_(-1, slot, quad.flip(-1))
|
||||
q2k_quad = q2k_quad[..., :n_blocks].contiguous()
|
||||
if validate:
|
||||
if int(q2k_num.min()) < 1:
|
||||
raise ValueError("every query block must attend to at least one non-empty KV block")
|
||||
last = (q2k_num - 1).long().unsqueeze(-1)
|
||||
first_block = q2k_idx.gather(-1, last.int().long()).squeeze(-1).long()
|
||||
first_quad = (q2k_quad.gather(-1, last).squeeze(-1).int() if q2k_quad is not None else
|
||||
torch.full_like(first_block, 15, dtype=torch.int32))
|
||||
v0 = (kv_valid[2 * first_block] > 0).int()
|
||||
v1 = (kv_valid[2 * first_block + 1] > 0).int()
|
||||
row0 = ((first_quad & 1).bool() & v0.bool()) | ((first_quad & 2).bool() & v1.bool())
|
||||
row1 = ((first_quad & 4).bool() & v0.bool()) | ((first_quad & 8).bool() & v1.bool())
|
||||
if not bool((row0 & row1).all()):
|
||||
raise ValueError("the first block each query block visits must give both 64-row halves a valid key")
|
||||
return q2k_idx, q2k_num, kv_valid, q2k_quad
|
||||
|
||||
|
||||
def check_sparse_block_lists(q2k_idx: torch.Tensor, q2k_num: torch.Tensor, kv_len: int) -> None:
|
||||
"""Reject block lists the sparse kernel would read out of bounds (host sync).
|
||||
|
||||
Each row needs ``1 <= q2k_num <= q2k_idx.size(-1)`` and every listed index
|
||||
in ``[0, ceil(kv_len / BLOCK_N))``; a zero count or an out-of-range index
|
||||
makes the kernel load outside its index row or the KV tensors.
|
||||
"""
|
||||
num_kv_blocks = -(-kv_len // BLOCK_N)
|
||||
if int(q2k_num.min()) < 1 or int(q2k_num.max()) > q2k_idx.size(-1):
|
||||
raise ValueError(f"q2k_num must be in [1, {q2k_idx.size(-1)}]")
|
||||
listed = torch.arange(q2k_idx.size(-1), device=q2k_idx.device) < q2k_num.unsqueeze(-1)
|
||||
idx = q2k_idx[listed]
|
||||
if idx.numel() and (int(idx.min()) < 0 or int(idx.max()) >= num_kv_blocks):
|
||||
raise ValueError(f"q2k_idx entries must be in [0, {num_kv_blocks})")
|
||||
|
||||
|
||||
def sageattn_blackwell_sparse(q,
|
||||
k,
|
||||
v,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
kv_valid: torch.Tensor | None = None,
|
||||
q2k_quad: torch.Tensor | None = None,
|
||||
per_block_mean=True,
|
||||
single_level_p_quant=True,
|
||||
sm_scale: float | None = None,
|
||||
validate: bool = True):
|
||||
"""Block-sparse SageAttention3 FP4 forward (non-causal).
|
||||
|
||||
Query block ``m`` (``BLOCK_M`` rows) of each (batch, head) attends only to
|
||||
the ``BLOCK_N``-token KV blocks in ``q2k_idx[b, h, m, :q2k_num[b, h, m]]``,
|
||||
restricted to the quadrants in ``q2k_quad`` when given; see
|
||||
:func:`vsa_tile_mask_to_fp4_blocks`. Q/K/V are ``[B, H, L, D]``.
|
||||
Block lists are checked with :func:`check_sparse_block_lists` (a host
|
||||
sync) unless ``validate=False``; pass that only for lists built by
|
||||
:func:`vsa_tile_mask_to_fp4_blocks`, which are in range by construction.
|
||||
"""
|
||||
QL = q.size(2)
|
||||
KL = k.size(2)
|
||||
if validate:
|
||||
check_sparse_block_lists(q2k_idx, q2k_num, KL)
|
||||
is_bf16 = q.dtype == torch.bfloat16
|
||||
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean)
|
||||
per_block_mean = delta_s.shape[2] > 1
|
||||
qlist_from_cuda = scale_and_quant_fp4(q)
|
||||
klist_from_cuda = scale_and_quant_fp4_permute(k)
|
||||
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
|
||||
o_fp4 = blockscaled_fp4_attn_sparse(qlist_from_cuda, klist_from_cuda, vlist_from_cuda, delta_s, KL, q2k_idx,
|
||||
q2k_num, kv_valid, q2k_quad, per_block_mean, is_bf16, single_level_p_quant,
|
||||
sm_scale)[0][:, :, :QL, :].contiguous()
|
||||
return o_fp4
|
||||
|
||||
|
||||
def sageattn_blackwell_sparse_bshd(q,
|
||||
k,
|
||||
v,
|
||||
q2k_idx: torch.Tensor,
|
||||
q2k_num: torch.Tensor,
|
||||
kv_valid: torch.Tensor | None = None,
|
||||
q2k_quad: torch.Tensor | None = None,
|
||||
single_level_p_quant=True,
|
||||
sm_scale: float | None = None,
|
||||
validate: bool = True) -> torch.Tensor:
|
||||
""":func:`sageattn_blackwell_sparse` for ``[B, L, H, D]`` inputs, without copies.
|
||||
|
||||
The FP4 quantizers read strided input, so the sequence-major tensors a
|
||||
linear produces are quantized in place of a transpose + pad. ``L`` must be
|
||||
a multiple of ``BLOCK_M`` (callers allocate the padding) and Q is
|
||||
unsmoothed. Returns ``[B, H, L, D]``.
|
||||
"""
|
||||
batch, seq_len, heads, _ = q.shape
|
||||
if seq_len % BLOCK_M:
|
||||
raise ValueError(f"sequence length {seq_len} must be a multiple of {BLOCK_M}")
|
||||
if validate:
|
||||
check_sparse_block_lists(q2k_idx, q2k_num, seq_len)
|
||||
qh, kh, vh = (x.transpose(1, 2) for x in (q, k, v))
|
||||
delta_s = _zero_delta_s(batch, heads, seq_len, q.device)
|
||||
return blockscaled_fp4_attn_sparse(scale_and_quant_fp4(qh), scale_and_quant_fp4_permute(kh),
|
||||
scale_and_quant_fp4_transpose(vh), delta_s, seq_len, q2k_idx, q2k_num, kv_valid,
|
||||
q2k_quad, False, q.dtype == torch.bfloat16, single_level_p_quant, sm_scale)[0]
|
||||
|
||||
|
||||
def sageattn_blackwell(q,
|
||||
k,
|
||||
v,
|
||||
@@ -396,7 +191,6 @@ def sageattn_blackwell(q,
|
||||
KL = k.size(2)
|
||||
is_bf16 = q.dtype == torch.bfloat16
|
||||
q, k, v, delta_s = preprocess_qkv(q, k, v, per_block_mean)
|
||||
per_block_mean = delta_s.shape[2] > 1
|
||||
qlist_from_cuda = scale_and_quant_fp4(q)
|
||||
klist_from_cuda = scale_and_quant_fp4_permute(k)
|
||||
vlist_from_cuda = scale_and_quant_fp4_transpose(v)
|
||||
|
||||
@@ -204,17 +204,8 @@ void run_mha_fwd(Flash_fwd_params ¶ms, cudaStream_t stream, bool force_split
|
||||
}));
|
||||
}
|
||||
|
||||
struct SparseKvLists {
|
||||
int const *q2k_idx = nullptr;
|
||||
int const *q2k_num = nullptr;
|
||||
int q2k_max = 0;
|
||||
int num_m_blocks = 0;
|
||||
int const *kv_valid = nullptr;
|
||||
uint8_t const *q2k_quad = nullptr;
|
||||
};
|
||||
|
||||
static std::vector<at::Tensor>
|
||||
mha_fwd_impl(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head_size // 2)
|
||||
std::vector<at::Tensor>
|
||||
mha_fwd(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head_size // 2)
|
||||
const at::Tensor &k, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
|
||||
const at::Tensor &v, // batch_size x seqlen_k x num_heads_k x (head_size // 2)
|
||||
const at::Tensor &sfq,
|
||||
@@ -227,8 +218,7 @@ mha_fwd_impl(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head
|
||||
bool is_causal,
|
||||
bool per_block_mean,
|
||||
bool is_bf16,
|
||||
bool single_level_p_quant, // If true, use only per-row scale s_P2 (no per-block s_P1)
|
||||
SparseKvLists const &sparse
|
||||
bool single_level_p_quant=false // If true, use only per-row scale s_P2 (no per-block s_P1)
|
||||
) {
|
||||
|
||||
auto dprops = at::cuda::getCurrentDeviceProperties();
|
||||
@@ -326,12 +316,6 @@ mha_fwd_impl(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head
|
||||
// stack-local tensor whose data pointer would dangle after mha_fwd returns
|
||||
// while the async kernel may still be running.
|
||||
params.tile_count_semaphore = nullptr;
|
||||
params.q2k_idx = sparse.q2k_idx;
|
||||
params.q2k_num = sparse.q2k_num;
|
||||
params.q2k_max = sparse.q2k_max;
|
||||
params.num_m_blocks = sparse.num_m_blocks;
|
||||
params.kv_valid = sparse.kv_valid;
|
||||
params.q2k_quad = sparse.q2k_quad;
|
||||
|
||||
if (seqlen_k > 0) {
|
||||
auto stream = at::cuda::getCurrentCUDAStream().stream();
|
||||
@@ -357,72 +341,7 @@ mha_fwd_impl(at::Tensor &q, // batch_size x seqlen_q x num_heads x (head
|
||||
|
||||
|
||||
|
||||
std::vector<at::Tensor>
|
||||
mha_fwd(at::Tensor &q, const at::Tensor &k, const at::Tensor &v,
|
||||
const at::Tensor &sfq, const at::Tensor &sfk, const at::Tensor &sfv,
|
||||
const at::Tensor &delta_s, int unpadded_k, c10::optional<at::Tensor> &out_,
|
||||
const float softmax_scale, bool is_causal, bool per_block_mean, bool is_bf16,
|
||||
bool single_level_p_quant=false) {
|
||||
return mha_fwd_impl(q, k, v, sfq, sfk, sfv, delta_s, unpadded_k, out_, softmax_scale,
|
||||
is_causal, per_block_mean, is_bf16, single_level_p_quant, SparseKvLists{});
|
||||
}
|
||||
|
||||
// Block-sparse forward: query block m of (b, h) attends only to the KV blocks
|
||||
// listed in q2k_idx[b, h, m, :q2k_num[b, h, m]] (BLOCK_M x BLOCK_N granularity).
|
||||
// kv_valid, if given, holds the valid token count of each 64-column half of
|
||||
// every KV block ([2 * num_kv_blocks], valid tokens first within a half).
|
||||
// q2k_quad, if given (uint8, same shape as q2k_idx), restricts each listed
|
||||
// block to the 64x64 quadrants whose bit (2 * row_half + col_half) is set, so
|
||||
// 64-token VSA tiles run on 128x128 blocks. The block a list visits first
|
||||
// (its last entry) must leave every query row at least one valid key.
|
||||
// Non-causal only.
|
||||
std::vector<at::Tensor>
|
||||
mha_fwd_sparse(at::Tensor &q, const at::Tensor &k, const at::Tensor &v,
|
||||
const at::Tensor &sfq, const at::Tensor &sfk, const at::Tensor &sfv,
|
||||
const at::Tensor &delta_s, int unpadded_k, c10::optional<at::Tensor> &out_,
|
||||
const float softmax_scale, bool per_block_mean, bool is_bf16,
|
||||
bool single_level_p_quant,
|
||||
const at::Tensor &q2k_idx, const at::Tensor &q2k_num,
|
||||
c10::optional<at::Tensor> &kv_valid_,
|
||||
c10::optional<at::Tensor> &q2k_quad_) {
|
||||
const int batch_size = q.size(0);
|
||||
const int num_heads = q.size(1);
|
||||
const int num_m_blocks = (q.size(2) + flash::BLOCK_M - 1) / flash::BLOCK_M;
|
||||
const int num_n_blocks = (k.size(2) + flash::BLOCK_N - 1) / flash::BLOCK_N;
|
||||
for (auto const *t : {&q2k_idx, &q2k_num}) {
|
||||
TORCH_CHECK(t->scalar_type() == torch::kInt32, "q2k_idx / q2k_num must be int32");
|
||||
CHECK_DEVICE((*t)); CHECK_CONTIGUOUS((*t));
|
||||
}
|
||||
TORCH_CHECK(q2k_idx.dim() == 4, "q2k_idx must be [batch, heads, num_m_blocks, max_kv_blocks]");
|
||||
TORCH_CHECK(q2k_idx.size(0) == batch_size && q2k_idx.size(1) == num_heads && q2k_idx.size(2) == num_m_blocks,
|
||||
"q2k_idx leading dims must be [batch, heads, ceil(seqlen_q / BLOCK_M)]");
|
||||
TORCH_CHECK(q2k_idx.size(3) >= 1 && q2k_idx.size(3) <= num_n_blocks, "q2k_idx last dim must be in [1, num_kv_blocks]");
|
||||
CHECK_SHAPE(q2k_num, batch_size, num_heads, num_m_blocks);
|
||||
SparseKvLists sparse;
|
||||
sparse.q2k_idx = q2k_idx.data_ptr<int>();
|
||||
sparse.q2k_num = q2k_num.data_ptr<int>();
|
||||
sparse.q2k_max = q2k_idx.size(3);
|
||||
sparse.num_m_blocks = num_m_blocks;
|
||||
if (kv_valid_.has_value()) {
|
||||
auto const &kv_valid = kv_valid_.value();
|
||||
TORCH_CHECK(kv_valid.scalar_type() == torch::kInt32, "kv_valid must be int32");
|
||||
CHECK_DEVICE(kv_valid); CHECK_CONTIGUOUS(kv_valid);
|
||||
CHECK_SHAPE(kv_valid, 2 * num_n_blocks);
|
||||
sparse.kv_valid = kv_valid.data_ptr<int>();
|
||||
}
|
||||
if (q2k_quad_.has_value()) {
|
||||
auto const &q2k_quad = q2k_quad_.value();
|
||||
TORCH_CHECK(q2k_quad.scalar_type() == torch::kUInt8, "q2k_quad must be uint8");
|
||||
CHECK_DEVICE(q2k_quad); CHECK_CONTIGUOUS(q2k_quad);
|
||||
TORCH_CHECK(q2k_quad.sizes() == q2k_idx.sizes(), "q2k_quad must match q2k_idx's shape");
|
||||
sparse.q2k_quad = q2k_quad.data_ptr<uint8_t>();
|
||||
}
|
||||
return mha_fwd_impl(q, k, v, sfq, sfk, sfv, delta_s, unpadded_k, out_, softmax_scale,
|
||||
/*is_causal=*/false, per_block_mean, is_bf16, single_level_p_quant, sparse);
|
||||
}
|
||||
|
||||
PYBIND11_MODULE(TORCH_EXTENSION_NAME, m) {
|
||||
m.doc() = "FlashAttention";
|
||||
m.def("fwd", &mha_fwd, "Forward pass");
|
||||
m.def("fwd_sparse", &mha_fwd_sparse, "Block-sparse forward pass (non-causal)");
|
||||
}
|
||||
@@ -185,14 +185,14 @@ __global__ void __launch_bounds__(Ktraits::kNWarps * cutlass::NumThreadsPerWarp,
|
||||
auto block_coord = work_tile_info.get_block_coord(scheduler_params);
|
||||
auto [m_block, bidh, bidb] = block_coord;
|
||||
|
||||
int n_block_count = collective_mainloop.get_n_block_count(mainloop_params, m_block, bidh, bidb);
|
||||
if (Is_causal && n_block_count <= 0) { // We exit early and write 0 to gO and -inf to gLSE.
|
||||
int n_block_max = collective_mainloop.get_n_block_max(mainloop_params, m_block);
|
||||
if (Is_causal && n_block_max <= 0) { // We exit early and write 0 to gO and -inf to gLSE.
|
||||
collective_epilogue.store_zero(epilogue_params, threadIdx.x - NumCopyThreads, block_coord);
|
||||
continue;
|
||||
}
|
||||
|
||||
collective_mainloop.mma(mainloop_params, pipeline_q, pipeline_k, pipeline_v, smem_pipe_read_q, smem_pipe_read_k, smem_pipe_read_v,
|
||||
tOrO, softmax_fused, n_block_count, threadIdx.x - NumCopyThreads, work_idx, m_block, bidh, bidb, shared_storage);
|
||||
tOrO, softmax_fused, n_block_max, threadIdx.x - NumCopyThreads, work_idx, m_block, shared_storage);
|
||||
barrier_o.wait();
|
||||
collective_epilogue.mma_store(shared_storage, tiled_mma_pv, tOrO, threadIdx.x - NumCopyThreads);
|
||||
barrier_o.arrive();
|
||||
|
||||
@@ -62,9 +62,7 @@ void run_flash_fwd(Flash_fwd_params ¶ms, cudaStream_t stream) {
|
||||
static_cast<float const*>(params.delta_s_ptr),
|
||||
{params.seqlen_s, params.seqlen_k, params.h_k, params.b},
|
||||
{params.ds_row_stride, _1{}, params.ds_head_stride, params.ds_batch_stride},
|
||||
params.scale_softmax_log2,
|
||||
params.q2k_idx, params.q2k_num, params.q2k_max,
|
||||
params.num_m_blocks, params.h, params.kv_valid, params.q2k_quad
|
||||
params.scale_softmax_log2
|
||||
});
|
||||
typename CollectiveEpilogue::Params epilogue_params =
|
||||
CollectiveEpilogue::to_underlying_arguments({
|
||||
|
||||
@@ -174,13 +174,6 @@ struct CollectiveMainloopFwd {
|
||||
ShapeQKV const shape_ds;
|
||||
StrideQKV const stride_ds;
|
||||
float const softmax_scale_log2;
|
||||
int const* ptr_q2k_idx{nullptr};
|
||||
int const* ptr_q2k_num{nullptr};
|
||||
int q2k_max{0};
|
||||
int num_m_blocks{0};
|
||||
int num_heads{0};
|
||||
int const* ptr_kv_valid{nullptr};
|
||||
uint8_t const* ptr_q2k_quad{nullptr};
|
||||
};
|
||||
|
||||
// Device side kernel params
|
||||
@@ -201,13 +194,6 @@ struct CollectiveMainloopFwd {
|
||||
TMA_SFVt tma_load_SFVt;
|
||||
TMA_DS tma_load_DS;
|
||||
float const softmax_scale_log2;
|
||||
int const* ptr_q2k_idx;
|
||||
int const* ptr_q2k_num;
|
||||
int q2k_max;
|
||||
int num_m_blocks;
|
||||
int num_heads;
|
||||
int const* ptr_kv_valid;
|
||||
uint8_t const* ptr_q2k_quad;
|
||||
};
|
||||
|
||||
|
||||
@@ -275,9 +261,7 @@ struct CollectiveMainloopFwd {
|
||||
tma_load_K, tma_load_sfk,
|
||||
tma_load_Vt, tma_load_sfvt,
|
||||
tma_load_ds,
|
||||
args.softmax_scale_log2,
|
||||
args.ptr_q2k_idx, args.ptr_q2k_num, args.q2k_max,
|
||||
args.num_m_blocks, args.num_heads, args.ptr_kv_valid, args.ptr_q2k_quad};
|
||||
args.softmax_scale_log2};
|
||||
}
|
||||
|
||||
/// Issue Tma Descriptor Prefetch -- ideally from a single thread for best performance
|
||||
@@ -306,51 +290,6 @@ struct CollectiveMainloopFwd {
|
||||
return n_block_max;
|
||||
}
|
||||
|
||||
// Number of KV blocks query block m_block visits: all of them when dense,
|
||||
// its index-list length when block-sparse.
|
||||
CUTLASS_DEVICE
|
||||
int get_n_block_count(Params const& mainloop_params, int m_block, int bidh, int bidb) {
|
||||
if (mainloop_params.ptr_q2k_idx == nullptr) {
|
||||
return get_n_block_max(mainloop_params, m_block);
|
||||
}
|
||||
return mainloop_params.ptr_q2k_num[(bidb * mainloop_params.num_heads + bidh) * mainloop_params.num_m_blocks + m_block];
|
||||
}
|
||||
|
||||
// KV block visited at iteration i (iterations run from count-1 down to 0).
|
||||
CUTLASS_DEVICE
|
||||
int get_kv_block(Params const& mainloop_params, int m_block, int bidh, int bidb, int i) {
|
||||
if (mainloop_params.ptr_q2k_idx == nullptr) {
|
||||
return i;
|
||||
}
|
||||
int64_t const row = (int64_t(bidb) * mainloop_params.num_heads + bidh) * mainloop_params.num_m_blocks + m_block;
|
||||
return mainloop_params.ptr_q2k_idx[row * mainloop_params.q2k_max + i];
|
||||
}
|
||||
|
||||
// Valid key columns of each 64-column half of KV block n_block (valid tokens
|
||||
// first within a half). kv_valid, when given, stores two counts per block so
|
||||
// 64-token VSA tiles can pad mid-block; otherwise only the sequence tail is
|
||||
// masked.
|
||||
CUTLASS_DEVICE
|
||||
int get_kv_valid_half(Params const& mainloop_params, int n_block, int half, int unpadded_seqlen_k) {
|
||||
static constexpr int kBlockN = get<1>(TileShape_MNK{});
|
||||
static constexpr int kHalfN = kBlockN / 2;
|
||||
if (mainloop_params.ptr_kv_valid != nullptr) {
|
||||
return mainloop_params.ptr_kv_valid[2 * n_block + half];
|
||||
}
|
||||
return max(0, min(kHalfN, unpadded_seqlen_k - n_block * kBlockN - half * kHalfN));
|
||||
}
|
||||
|
||||
// Quadrant mask of list entry i: bit (2 * row_half + col_half) is set when the
|
||||
// 64-row query half attends the 64-column key half. 0xF when absent.
|
||||
CUTLASS_DEVICE
|
||||
int get_quad(Params const& mainloop_params, int m_block, int bidh, int bidb, int i) {
|
||||
if (mainloop_params.ptr_q2k_quad == nullptr) {
|
||||
return 0xF;
|
||||
}
|
||||
int64_t const row = (int64_t(bidb) * mainloop_params.num_heads + bidh) * mainloop_params.num_m_blocks + m_block;
|
||||
return mainloop_params.ptr_q2k_quad[row * mainloop_params.q2k_max + i];
|
||||
}
|
||||
|
||||
template <class SFATensor, class Atom, class TiledThr, class TiledPerm>
|
||||
CUTE_HOST_DEVICE constexpr
|
||||
auto
|
||||
@@ -511,7 +450,7 @@ struct CollectiveMainloopFwd {
|
||||
|
||||
auto [m_block, bidh, bidb] = work_tile_info.get_block_coord(scheduler_params);
|
||||
|
||||
int n_block_count = get_n_block_count(mainloop_params, m_block, bidh, bidb);
|
||||
int n_block_max = get_n_block_max(mainloop_params, m_block);
|
||||
|
||||
Tensor sQ = make_tensor(make_smem_ptr(shared_storage.smem_q.begin()), SmemLayoutQ{});
|
||||
Tensor sK = make_tensor(make_smem_ptr(shared_storage.smem_k.begin()), SmemLayoutK{});
|
||||
@@ -567,8 +506,7 @@ struct CollectiveMainloopFwd {
|
||||
Tensor tDSsDS = group_modes<0, 3>(block_tma_ds.partition_D(sDS));
|
||||
uint16_t mcast_mask_kv = 0;
|
||||
|
||||
int n_iter = n_block_count - 1;
|
||||
int n_block = get_kv_block(mainloop_params, m_block, bidh, bidb, n_iter);
|
||||
int n_block = n_block_max - 1;
|
||||
int lane_predicate = cute::elect_one_sync();
|
||||
if (lane_predicate) {
|
||||
pipeline_q.producer_acquire(smem_pipe_write_q);
|
||||
@@ -591,12 +529,11 @@ struct CollectiveMainloopFwd {
|
||||
++smem_pipe_write_v;
|
||||
}
|
||||
|
||||
--n_iter;
|
||||
n_block--;
|
||||
if (lane_predicate) {
|
||||
// CUTLASS_PRAGMA_NO_UNROLL
|
||||
#pragma unroll 2
|
||||
for (; n_iter >= 0; --n_iter) {
|
||||
n_block = get_kv_block(mainloop_params, m_block, bidh, bidb, n_iter);
|
||||
for (; n_block >= 0; --n_block) {
|
||||
pipeline_k.producer_acquire(smem_pipe_write_k);
|
||||
copy(mainloop_params.tma_load_K.with(*pipeline_k.producer_get_barrier(smem_pipe_write_k), mcast_mask_kv),
|
||||
tKgK(_, n_block), tKsK(_, smem_pipe_write_k.index()));
|
||||
@@ -648,8 +585,6 @@ struct CollectiveMainloopFwd {
|
||||
int thread_idx,
|
||||
int work_idx,
|
||||
int m_block,
|
||||
int bidh,
|
||||
int bidb,
|
||||
SharedStorage& shared_storage
|
||||
) {
|
||||
|
||||
@@ -732,21 +667,7 @@ struct CollectiveMainloopFwd {
|
||||
int const seqlen_q = get<0>(mainloop_params.shape_Q);
|
||||
int const seqlen_k = get<0>(mainloop_params.shape_K);
|
||||
int const unpadded_seqlen_k = get<0>(mainloop_params.unpadded_shape_K);
|
||||
int n_iter = n_block_count - 1;
|
||||
int n_block = get_kv_block(mainloop_params, m_block, bidh, bidb, n_iter);
|
||||
bool const per_block_masking = mainloop_params.ptr_q2k_idx != nullptr || mainloop_params.ptr_kv_valid != nullptr;
|
||||
static_assert(kBlockM == 128 && kBlockN == 128, "quadrant masking assumes 128x128 blocks");
|
||||
// Each MMA warp owns 16 consecutive query rows, so a warp lies in one
|
||||
// 64-row half. With quadrant lists a warp skips blocks its half did not
|
||||
// select and the P.V chunk of a key half it did not select: masked
|
||||
// scores contribute exactly zero, and the warp sharing its tensor-core
|
||||
// partition runs faster meanwhile.
|
||||
int const my_row_half = __shfl_sync(0xffffffff, [&] {
|
||||
Tensor cS0 = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS0 = thread_mma_qk.partition_C(cS0);
|
||||
return int(get<0>(tScS0(0))) >= kBlockM / 2 ? 1 : 0;
|
||||
}(), 0);
|
||||
bool const quad_skip = mainloop_params.ptr_q2k_quad != nullptr;
|
||||
int n_block = n_block_count - 1;
|
||||
|
||||
auto copy_k_block = [&](auto block_id) {
|
||||
auto tSsK_stage = tSsK(_, _, _, smem_pipe_read_k.index());
|
||||
@@ -823,26 +744,6 @@ struct CollectiveMainloopFwd {
|
||||
int local = c & 31;
|
||||
return (c & ~31) | (local & 1) | ((local & 24) >> 2) | ((local & 6) << 2);
|
||||
};
|
||||
// Sparse lists may visit a block whose 64-token halves are partially valid
|
||||
// (VSA tile tails, short prefix chunks) or that only one query half
|
||||
// selected (64-token tiles), at any iteration.
|
||||
auto apply_sparse_mask = [&](auto& acc, int n_blk, int it) {
|
||||
int const quad = get_quad(mainloop_params, m_block, bidh, bidb, it);
|
||||
int const valid0 = get_kv_valid_half(mainloop_params, n_blk, 0, unpadded_seqlen_k);
|
||||
int const valid1 = get_kv_valid_half(mainloop_params, n_blk, 1, unpadded_seqlen_k);
|
||||
if (quad == 0xF && valid0 == kBlockN / 2 && valid1 == kBlockN / 2) { return; }
|
||||
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS = thread_mma_qk.partition_C(cS);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int i = 0; i < size(acc); ++i) {
|
||||
int const col = actual_col(int(get<1>(tScS(i))));
|
||||
int const col_half = col >= kBlockN / 2;
|
||||
int const row_half = int(get<0>(tScS(i))) >= kBlockM / 2;
|
||||
bool const keep = ((quad >> (2 * row_half + col_half)) & 1)
|
||||
&& (col & (kBlockN / 2 - 1)) < (col_half ? valid1 : valid0);
|
||||
if (!keep) { acc(i) = -INFINITY; }
|
||||
}
|
||||
};
|
||||
{
|
||||
Tensor cS = cute::make_identity_tensor(select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tScS = thread_mma_qk.partition_C(cS);
|
||||
@@ -850,9 +751,7 @@ struct CollectiveMainloopFwd {
|
||||
for (int i = 0; i < size(tSrS); ++i) {
|
||||
int col = actual_col(int(get<1>(tScS(i))));
|
||||
if constexpr (!Is_causal) { // Just masking based on col
|
||||
if (!per_block_masking) {
|
||||
if (col >= int(unpadded_seqlen_k - n_block * kBlockN)) { tSrS(i) = -INFINITY; }
|
||||
}
|
||||
if (col >= int(unpadded_seqlen_k - n_block * kBlockN)) { tSrS(i) = -INFINITY; }
|
||||
} else {
|
||||
if (col >= std::min(seqlen_k - n_block * kBlockN,
|
||||
col_limit_causal(int(get<0>(tScS(i))), n_block))) {
|
||||
@@ -861,9 +760,6 @@ struct CollectiveMainloopFwd {
|
||||
}
|
||||
}
|
||||
}
|
||||
if constexpr (!Is_causal) {
|
||||
if (per_block_masking) { apply_sparse_mask(tSrS, n_block, n_iter); }
|
||||
}
|
||||
auto quantize = [&](auto mma_k, auto acc_conversion_view) {
|
||||
Tensor AbsMaxP_stagek = AbsMaxP(_, make_coord(_, _, mma_k));
|
||||
Tensor acc_conversion_stagek = acc_conversion_view(_, _, mma_k);
|
||||
@@ -931,12 +827,11 @@ struct CollectiveMainloopFwd {
|
||||
}
|
||||
}
|
||||
|
||||
--n_iter;
|
||||
n_block--;
|
||||
constexpr int n_masking_steps = !Is_causal ? 1 : cute::ceil_div(kBlockM, kBlockN) + 1;
|
||||
// // Only go through these if Is_causal, since n_masking_steps = 1 when !Is_causal
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int masking_step = 0; masking_step < n_masking_steps - 1 && n_iter >= 0; ++masking_step, --n_iter) {
|
||||
n_block = get_kv_block(mainloop_params, m_block, bidh, bidb, n_iter);
|
||||
for (int masking_step = 0; masking_step < n_masking_steps - 1 && n_block >= 0; ++masking_step, --n_block) {
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
@@ -983,22 +878,7 @@ struct CollectiveMainloopFwd {
|
||||
}
|
||||
|
||||
#pragma unroll 1
|
||||
for (; n_iter >= 0; --n_iter) {
|
||||
n_block = get_kv_block(mainloop_params, m_block, bidh, bidb, n_iter);
|
||||
int const row_bits = quad_skip
|
||||
? (get_quad(mainloop_params, m_block, bidh, bidb, n_iter) >> (2 * my_row_half)) & 3
|
||||
: 3;
|
||||
if (row_bits == 0) {
|
||||
// Neither key half is selected for this warp's rows: its softmax
|
||||
// state and output are unchanged. Keep the pipeline in step.
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
pipeline_k.consumer_release(smem_pipe_read_k);
|
||||
++smem_pipe_read_k;
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
pipeline_v.consumer_release(smem_pipe_read_v);
|
||||
++smem_pipe_read_v;
|
||||
continue;
|
||||
}
|
||||
for (; n_block >= 0; --n_block) {
|
||||
Tensor tSrS = partition_fragment_C(tiled_mma_qk, select<0, 1>(TileShape_MNK{}));
|
||||
Tensor tSrS_converion_view = make_tensor(tSrS.data(), flash::convert_to_conversion_layout(tSrS.layout()));
|
||||
consumer_wait(pipeline_k, smem_pipe_read_k);
|
||||
@@ -1016,21 +896,16 @@ struct CollectiveMainloopFwd {
|
||||
}
|
||||
}
|
||||
|
||||
if (per_block_masking) { apply_sparse_mask(tSrS, n_block, n_iter); }
|
||||
|
||||
softmax_fused.template online_softmax_with_quant</*Is_first=*/false>(tSrS, AbsMaxP, mainloop_params.softmax_scale_log2);
|
||||
Tensor tOrO = make_fragment_like(tOrO_store);
|
||||
clear(tOrO);
|
||||
consumer_wait(pipeline_v, smem_pipe_read_v);
|
||||
copy_v_block(_0{});
|
||||
quantize(_0{}, tSrS_converion_view);
|
||||
CUTLASS_PRAGMA_UNROLL
|
||||
for (int v_block = 0; v_block < size<2>(tOrP); ++v_block) {
|
||||
// v_block spans one 64-column key half (P's K mode is 2 x 64).
|
||||
if ((row_bits >> v_block) & 1) {
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
|
||||
}
|
||||
cute::gemm(tiled_mma_pv, make_zip_tensor(tOrP(_, _, v_block), tOrSFP(_, _, v_block)),
|
||||
make_zip_tensor(tOrVt(_, _, v_block), tOrSFVt(_, _, v_block)), tOrO);
|
||||
if (v_block < size<2>(tOrP) - 1) {
|
||||
copy_v_block(v_block + 1);
|
||||
quantize(v_block + 1, tSrS_converion_view);
|
||||
|
||||
@@ -114,23 +114,6 @@
|
||||
int * __restrict__ seqused_k;
|
||||
|
||||
int *__restrict__ blockmask;
|
||||
|
||||
// Block-sparse KV iteration (non-causal only). When q2k_idx is null the
|
||||
// kernel is dense. Otherwise query block m of (batch b, head h) visits
|
||||
// the q2k_num[(b*h + h)*num_m_blocks + m] KV blocks listed at
|
||||
// q2k_idx[((b*h + h)*num_m_blocks + m)*q2k_max + i], in any order.
|
||||
int const *__restrict__ q2k_idx;
|
||||
int const *__restrict__ q2k_num;
|
||||
int q2k_max;
|
||||
int num_m_blocks;
|
||||
// Optional valid token count of each 64-column half of every KV block,
|
||||
// kv_valid[2*n + half] (valid tokens first within a half); null means only
|
||||
// the sequence tail beyond the unpadded key length is masked.
|
||||
int const *__restrict__ kv_valid;
|
||||
// Optional quadrant mask per list entry (same layout as q2k_idx): bit
|
||||
// (2*row_half + col_half) set when that 64-row query half attends that
|
||||
// 64-column key half. Lets 64-token VSA tiles run on 128x128 blocks.
|
||||
uint8_t const *__restrict__ q2k_quad;
|
||||
|
||||
// The K_new and V_new matrices.
|
||||
void * __restrict__ knew_ptr;
|
||||
|
||||
@@ -2,6 +2,8 @@
|
||||
// data-center Blackwell sm_100a/sm_103a. The filename is retained for API compatibility.
|
||||
// Warp-specialized: load / MMA (tcgen05) / softmax / correction / epilogue / scheduler.
|
||||
// Writes O and, when asked, the log-sum-exp the backward consumes.
|
||||
//
|
||||
// Generated (comments stripped). Do not edit by hand.
|
||||
#ifndef BLOCK_SPARSE_VSA_KERNEL_SM100A_CUH
|
||||
#define BLOCK_SPARSE_VSA_KERNEL_SM100A_CUH
|
||||
|
||||
@@ -504,11 +506,9 @@ fmha_context_bf16_gen_kernel(const __grid_constant__ CUtensorMap tmap_q,
|
||||
const uint32_t o_tmem_addr = tmem_base + (uint32_t)(2 * S_COLS + i * O_COLS);
|
||||
|
||||
int slot = 0;
|
||||
// Declared outside the loop: with BLK128 only p == 0 initializes it and p == 1 continues
|
||||
// the p == 0 descriptor walk.
|
||||
SmemDescPair desc_bv;
|
||||
#pragma unroll
|
||||
for (int p = 0; p < V_SUBTILES; ++p) {
|
||||
SmemDescPair desc_bv;
|
||||
if (!BLK128 || p == 0) {
|
||||
slot = kv_ph.get_stage();
|
||||
mbarrier_wait_parity(smem_ptr_u32(&full_bar[slot]), kv_ph.get_phase());
|
||||
@@ -518,6 +518,8 @@ fmha_context_bf16_gen_kernel(const __grid_constant__ CUtensorMap tmap_q,
|
||||
desc_bv.u64 = desc_v0;
|
||||
desc_bv.w.x += (uint32_t)(slot * (int)KV_DESC_DELTA);
|
||||
}
|
||||
const uint64_t dbV = desc_kv0 + (uint64_t)slot * KV_DESC_DELTA
|
||||
+ (BLK128 ? (uint64_t)p * (V_BLK_BYTES >> 4) : 0u);
|
||||
#pragma unroll
|
||||
for (int ki = 0; ki < K_ATOMS_PER_TILE; ++ki) {
|
||||
const int a = p * K_ATOMS_PER_TILE + ki;
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
// primitives.cuh -- device primitives for the sm_100a/sm_103a VSA block-sparse attention
|
||||
// forward: tcgen05 (alloc / mma / ld / st / commit / wait / fence), TMA load / store /
|
||||
// tensormap, mbarrier, cluster launch control, setmaxnreg, fast math, and the FMHA helpers.
|
||||
//
|
||||
// Generated and pruned to what the kernel reaches -- do not edit by hand.
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <cstdio>
|
||||
@@ -216,8 +218,8 @@ template <int CLUSTER_SHAPE_M, int CLUSTER_SHAPE_N, ClcRasterOrder ORDER>
|
||||
__device__ __forceinline__
|
||||
ClcTileInfo clc_parse_response(uint32_t resp_smem_addr) {
|
||||
uint32_t d0, d1, d2, d3;
|
||||
clc_load_response(resp_smem_addr, d0, d1, d2, d3);
|
||||
fence_proxy_async_shared_cta();
|
||||
clc_load_response(resp_smem_addr, d0, d1, d2, d3);
|
||||
const int ctaid_x = static_cast<int>(d0);
|
||||
const int ctaid_y = static_cast<int>(d1 & 0xFFFFu);
|
||||
const bool valid = (d2 & 1u) != 0u;
|
||||
@@ -249,7 +251,6 @@ ClcTileInfo clc_fetch_next_tile(
|
||||
__cvta_generic_to_shared(&clc_response[clc_cons_stage * 4]));
|
||||
ClcTileInfo t = clc_parse_response<
|
||||
CLUSTER_SHAPE_M, CLUSTER_SHAPE_N, ORDER>(resp_addr);
|
||||
__syncwarp();
|
||||
if (do_release) {
|
||||
uint32_t empty_local = static_cast<uint32_t>(
|
||||
__cvta_generic_to_shared(&clc_empty_bar[clc_cons_stage]));
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user