Compare commits

...
1 Commits
Author SHA1 Message Date
JerryZhou54 391218d06f first pass 2026-01-29 18:31:52 +00:00
6 changed files with 97 additions and 4 deletions
+11 -1
View File
@@ -9,7 +9,7 @@ import torch
from fastvideo.configs.models import DiTConfig, EncoderConfig, VAEConfig
from fastvideo.configs.models.dits import HunyuanVideo15Config
from fastvideo.configs.models.encoders import (BaseEncoderOutput,
Qwen2_5_VLConfig, T5Config)
Qwen2_5_VLConfig, T5Config, SiglipVisionConfig)
from fastvideo.configs.models.vaes import Hunyuan15VAEConfig
from fastvideo.configs.pipelines.base import PipelineConfig
@@ -137,3 +137,13 @@ class Hunyuan15T2V720PConfig(Hunyuan15T2V480PConfig):
# HunyuanConfig-specific parameters with defaults
flow_shift: int = 9
@dataclass
class Hunyuan15I2V480PConfig(Hunyuan15T2V480PConfig):
image_encoder_config: EncoderConfig = field(
default_factory=SiglipVisionConfig)
image_encoder_precision: str = "bf16"
def __post_init__(self):
self.vae_config.load_encoder = True
self.vae_config.load_decoder = True
+3 -1
View File
@@ -8,7 +8,7 @@ from fastvideo.configs.pipelines.base import PipelineConfig
from fastvideo.configs.pipelines.cosmos import CosmosConfig
from fastvideo.configs.pipelines.cosmos2_5 import Cosmos25Config
from fastvideo.configs.pipelines.hunyuan import FastHunyuanConfig, HunyuanConfig
from fastvideo.configs.pipelines.hunyuan15 import Hunyuan15T2V480PConfig, Hunyuan15T2V720PConfig
from fastvideo.configs.pipelines.hunyuan15 import Hunyuan15T2V480PConfig, Hunyuan15T2V720PConfig, Hunyuan15I2V480PConfig
from fastvideo.configs.pipelines.hyworld import HYWorldConfig
from fastvideo.configs.pipelines.ltx2 import LTX2T2VConfig
from fastvideo.configs.pipelines.stepvideo import StepVideoT2VConfig
@@ -37,6 +37,8 @@ PIPE_NAME_TO_CONFIG: dict[str, type[PipelineConfig]] = {
"hunyuanvideo-community/HunyuanVideo": HunyuanConfig,
"hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v":
Hunyuan15T2V480PConfig,
"hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_i2v":
Hunyuan15I2V480PConfig,
"hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-720p_t2v":
Hunyuan15T2V720PConfig,
"FastVideo/HY-WorldPlay-Bidirectional-Diffusers": HYWorldConfig,
-2
View File
@@ -17,8 +17,6 @@ class Hunyuan15_480P_SamplingParam(SamplingParam):
guidance_scale: float = 6.0
prompt_attention_mask: list = field(default_factory=list)
negative_attention_mask: list = field(default_factory=list)
sigmas: list[float] | None = field(
default_factory=lambda: list(np.linspace(1.0, 0.0, 50 + 1)[:-1]))
negative_prompt: str = ""
+2
View File
@@ -46,6 +46,8 @@ SAMPLING_PARAM_REGISTRY: dict[str, Any] = {
"hunyuanvideo-community/HunyuanVideo": HunyuanSamplingParam,
"hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_t2v":
Hunyuan15_480P_SamplingParam,
"hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-480p_i2v":
Hunyuan15_480P_SamplingParam,
"hunyuanvideo-community/HunyuanVideo-1.5-Diffusers-720p_t2v":
Hunyuan15_720P_SamplingParam,
"FastVideo/HY-WorldPlay-Bidirectional-Diffusers": HYWorld_SamplingParam,
@@ -0,0 +1,80 @@
# SPDX-License-Identifier: Apache-2.0
"""
HYWorld video diffusion pipeline implementation.
This module contains an implementation of the HYWorld video diffusion pipeline
using the modular pipeline architecture with HYWorld-specific denoising stage
for chunk-based video generation with context frame selection.
"""
from fastvideo.fastvideo_args import FastVideoArgs
from fastvideo.logger import init_logger
from fastvideo.pipelines.composed_pipeline_base import ComposedPipelineBase
from fastvideo.pipelines.stages import (
ConditioningStage, DecodingStage, DenoisingStage,
InputValidationStage, LatentPreparationStage, TextEncodingStage,
TimestepPreparationStage, HYWorldImageEncodingStage)
logger = init_logger(__name__)
class HunyuanVideo15ImageToVideoPipeline(ComposedPipelineBase):
"""
HunyuanVideo15 image to video pipeline.
This pipeline implements image to video generation using the HunyuanVideo15 model.
"""
# Include image_encoder and feature_extractor for I2V support with SigLIP
# Note: guider (ClassifierFreeGuidance) is not needed - FastVideo handles CFG differently
_required_config_modules = [
"text_encoder", "tokenizer", "vae", "transformer", "scheduler",
"text_encoder_2", "tokenizer_2", "image_encoder", "feature_extractor"
]
def create_pipeline_stages(self, fastvideo_args: FastVideoArgs):
"""Set up pipeline stages with HYWorld-specific denoising stage."""
self.add_stage(stage_name="input_validation_stage",
stage=InputValidationStage())
self.add_stage(stage_name="prompt_encoding_stage_primary",
stage=TextEncodingStage(
text_encoders=[
self.get_module("text_encoder"),
self.get_module("text_encoder_2")
],
tokenizers=[
self.get_module("tokenizer"),
self.get_module("tokenizer_2")
]))
self.add_stage(stage_name="conditioning_stage",
stage=ConditioningStage())
self.add_stage(stage_name="timestep_preparation_stage",
stage=TimestepPreparationStage(
scheduler=self.get_module("scheduler")))
self.add_stage(stage_name="latent_preparation_stage",
stage=LatentPreparationStage(
scheduler=self.get_module("scheduler"),
transformer=self.get_module("transformer")))
self.add_stage(stage_name="image_encoding_stage",
stage=HYWorldImageEncodingStage(
image_encoder=self.get_module("image_encoder"),
image_processor=self.get_module("feature_extractor"),
vae=self.get_module("vae")))
self.add_stage(stage_name="denoising_stage",
stage=DenoisingStage(
transformer=self.get_module("transformer"),
scheduler=self.get_module("scheduler"),
pipeline=self))
self.add_stage(stage_name="decoding_stage",
stage=DecodingStage(vae=self.get_module("vae")))
EntryClass = HunyuanVideo15ImageToVideoPipeline
+1
View File
@@ -28,6 +28,7 @@ _PIPELINE_NAME_TO_ARCHITECTURE_NAME: dict[str, str] = {
"StepVideoPipeline": "stepvideo",
"HunyuanVideoPipeline": "hunyuan",
"HunyuanVideo15Pipeline": "hunyuan15",
"HunyuanVideo15ImageToVideoPipeline": "hunyuan15",
"HYWorldPipeline": "hyworld",
"Cosmos2VideoToWorldPipeline": "cosmos",
"Cosmos2_5Pipeline": "cosmos",