Files
hao-ai-lab-FastVideo/examples/inference/basic/basic_longcat_t2v.py
T
Davids048andClaude Fable 5.1 1ea7278c00 [refactor]: group model-family settings under pipeline.model
pipeline.model is a tagged union: a mapping with exactly one key, which
names the model family (ltx2, minimax_h3, longcat, or generic) and selects
that family's options class. The former sibling sections pipeline.ltx2,
pipeline.minimax_h3, and pipeline.longcat, and the DiT/VAE architecture
override dicts pipeline.dit and pipeline.vae, live inside the family block.
The parser implements tagged unions as a general mechanism (a dataclass
with a TAGS table); the family is readable as pipeline.model.family.

Resolution validates the tag against the family that the registry chose
for model_path (validate_model_family raises and names the registry class
and the right block) and fills the family's empty block when the input has
none (fill_model_family), so readers never see None. Family-only fills and
derivations (the MiniMax-H3 parallel-VAE environment variables, the LTX-2
tile-size derivation and refine preset copy, the LongCat BSA mirrors) run
only for their family.

Against the 3f6893a0 goldens every PipelineConfig is identical and no
other field differs beyond the allowed categories; family-only settings
are no longer decided on models of another family. LingBot-Video reads
its refiner switch from pipeline.preset_overrides.refine.enabled and loads
the refiner by default when the checkpoint has one, as at 3f6893a0.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CbSuTKA9oX4YsECdcBPAsX
2026-10-05 06:04:17 +00:00

222 lines
7.1 KiB
Python

"""
LongCat Text-to-Video (T2V) Example Script
This script demonstrates LongCat T2V inference using the FastVideo Python API.
It runs both basic generation (50 steps) and distill+refine generation
(16 steps distill + 50 steps refinement to 720p).
Usage:
python examples/inference/basic/basic_longcat_t2v.py
"""
import glob
import os
from fastvideo import VideoGenerator
# Common prompts and settings matching the shell script examples
PROMPT = ("In a realistic photography style, a white boy around seven or eight years old "
"sits on a park bench, wearing a light blue T-shirt, denim shorts, and white sneakers. "
"He holds an ice cream cone with vanilla and chocolate flavors, and beside him is a "
"medium-sized golden Labrador. Smiling, the boy offers the ice cream to the dog, "
"who eagerly licks it with its tongue. The sun is shining brightly, and the background "
"features a green lawn and several tall trees, creating a warm and loving scene.")
NEGATIVE_PROMPT = ("Bright tones, overexposed, static, blurred details, subtitles, style, works, "
"paintings, images, static, overall gray, worst quality, low quality, JPEG compression "
"residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, "
"deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, "
"three legs, many people in the background, walking backwards")
SEED = 42
def basic_generation():
"""
Run basic LongCat T2V generation (50 steps at 480p).
This uses the full 50-step denoising process for highest quality.
"""
print("=" * 60)
print("LongCat T2V: Basic Generation (50 steps, 480p)")
print("=" * 60)
generator = VideoGenerator.from_config({
"model_path": "FastVideo/LongCat-Video-T2V-Diffusers",
"engine": {
"num_gpus": 1,
"use_fsdp_inference": False, # set to True if GPU is out of memory
"offload": {
"dit": False,
"vae": True,
"text_encoder": True,
"pin_cpu_memory": False,
},
},
"pipeline": {"model": {"longcat": {"enable_bsa": False}}},
})
output_path = "outputs_video/longcat_t2v_basic"
generator.generate({
"prompt": PROMPT,
"negative_prompt": NEGATIVE_PROMPT,
"sampling": {
"height": 480,
"width": 832,
"num_frames": 93,
"num_inference_steps": 50,
"fps": 15,
"guidance_scale": 4.0,
"seed": SEED,
},
"output": {"output_path": output_path, "save_video": True},
})
print(f"\nBasic generation complete! Video saved to: {output_path}")
generator.shutdown()
def distill_refine_generation():
"""
Run LongCat T2V with distill+refine pipeline (16 steps + refinement to 720p).
This uses the distilled LoRA for fast 480p generation (16 steps),
then refines to 720p using the refinement LoRA with BSA enabled.
"""
print("\n" + "=" * 60)
print("LongCat T2V: Distill + Refine Pipeline")
print("=" * 60)
# Stage 1: Distilled generation (16 steps at 480p)
print("\n[Stage 1] Distilled generation (16 steps, 480p)")
print("-" * 40)
generator = VideoGenerator.from_config({
"model_path": "FastVideo/LongCat-Video-T2V-Diffusers",
"engine": {
"num_gpus": 1,
"use_fsdp_inference": True,
"offload": {
"dit": False,
"vae": True,
"text_encoder": True,
"pin_cpu_memory": False,
},
},
"pipeline": {
"model": {"longcat": {"enable_bsa": False}},
"components": {
"lora_path": "FastVideo/LongCat-Video-T2V-Distilled-LoRA",
"lora_nickname": "distilled",
},
},
})
distill_output_path = "outputs_video/longcat_t2v_distill"
generator.generate({
"prompt": PROMPT,
"negative_prompt": NEGATIVE_PROMPT,
"sampling": {
"height": 480,
"width": 832,
"num_frames": 93,
"num_inference_steps": 16,
"fps": 15,
"guidance_scale": 1.0,
"seed": SEED,
},
"output": {"output_path": distill_output_path, "save_video": True},
})
print(f"Distilled generation complete! Video saved to: {distill_output_path}")
generator.shutdown()
# Stage 2: Refinement (480p -> 720p)
print("\n[Stage 2] Refinement (480p -> 720p with BSA)")
print("-" * 40)
# Find the actual saved video file from stage 1
video_files = glob.glob(os.path.join(distill_output_path, "*.mp4"))
if not video_files:
raise FileNotFoundError(f"No video file found in {distill_output_path}")
# Use the most recently created video file
distill_video_path = max(video_files, key=os.path.getmtime)
print(f"Using stage 1 video: {distill_video_path}")
# Create a new generator with refinement LoRA and BSA enabled
refine_generator = VideoGenerator.from_config({
"model_path": "FastVideo/LongCat-Video-T2V-Diffusers",
"engine": {
"num_gpus": 1,
"use_fsdp_inference": True,
"offload": {
"dit": True,
"vae": True,
"text_encoder": True,
"pin_cpu_memory": False,
},
},
"pipeline": {
"model": {
"longcat": {
"enable_bsa": True,
"bsa_sparsity": 0.875,
"bsa_chunk_q": [4, 4, 8],
"bsa_chunk_k": [4, 4, 8],
},
},
"components": {
"lora_path": "FastVideo/LongCat-Video-T2V-Refinement-LoRA",
"lora_nickname": "refinement",
},
},
})
refine_output_path = "outputs_video/longcat_t2v_refine_720p"
refine_generator.generate({
"prompt": PROMPT,
"negative_prompt": NEGATIVE_PROMPT,
"inputs": {"refine_from": distill_video_path},
"sampling": {
"height": 720,
"width": 1280,
"num_inference_steps": 50,
"fps": 30,
"guidance_scale": 1.0,
"seed": SEED,
},
"output": {"output_path": refine_output_path, "save_video": True},
"extensions": {
"t_thresh": 0.5,
"spatial_refine_only": False,
"num_cond_frames": 0,
},
})
print(f"Refinement complete! Video saved to: {refine_output_path}")
refine_generator.shutdown()
def main():
"""Run both basic and distill+refine generation pipelines."""
print("\n" + "=" * 60)
print("LongCat Text-to-Video Example")
print("=" * 60 + "\n")
# Run basic generation
basic_generation()
# Run distill+refine pipeline
distill_refine_generation()
print("\n" + "=" * 60)
print("All generations complete!")
print("=" * 60)
if __name__ == "__main__":
main()