Compare commits
21
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
59cf9d6fa2 | ||
|
|
0c90c328c5 | ||
|
|
cfb54a3b2a | ||
|
|
deb51f1dcc | ||
|
|
79a5930340 | ||
|
|
3a0ce7c20d | ||
|
|
0b006a9e46 | ||
|
|
fb9acac514 | ||
|
|
2d78957219 | ||
|
|
63991d2017 | ||
|
|
b6eafbea50 | ||
|
|
7b84041642 | ||
|
|
48801c29c7 | ||
|
|
cb8be0d3f3 | ||
|
|
1bc7ff79e6 | ||
|
|
76fbe472ae | ||
|
|
eba74c43ef | ||
|
|
7744a74c13 | ||
|
|
d746b8b956 | ||
|
|
65f12605d8 | ||
|
|
1e1ac08cd0 |
@@ -10,7 +10,9 @@ from pathlib import Path
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Clone a reference repo for FastVideo parity tests.")
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Clone a reference repo for FastVideo parity tests."
|
||||
)
|
||||
parser.add_argument("repo_url", help="Official reference repository URL")
|
||||
parser.add_argument("target_dir", help="Directory to clone into")
|
||||
parser.add_argument("--branch", help="Branch or tag to clone")
|
||||
@@ -60,7 +62,9 @@ def gitignore_entry_for(target: Path) -> str:
|
||||
try:
|
||||
relative = resolved.relative_to(root)
|
||||
except ValueError as exc:
|
||||
raise ValueError("--update-gitignore requires target_dir to be under the current directory") from exc
|
||||
raise ValueError(
|
||||
"--update-gitignore requires target_dir to be under the current directory"
|
||||
) from exc
|
||||
|
||||
text = relative.as_posix().rstrip("/")
|
||||
return "/" + text + "/"
|
||||
|
||||
@@ -8,12 +8,14 @@ import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Download a HF model snapshot or selected files into a local directory.")
|
||||
description="Download a HF model snapshot or selected files into a local directory."
|
||||
)
|
||||
parser.add_argument("repo_id", help="HF repo id, for example Org/Model")
|
||||
parser.add_argument("local_dir", help="Destination directory")
|
||||
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
|
||||
|
||||
@@ -10,6 +10,7 @@ import sys
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
|
||||
HF_TOKEN_ENV_KEYS = ("HF_TOKEN", "HUGGINGFACE_HUB_TOKEN", "HF_API_KEY")
|
||||
RAW_WEIGHT_SUFFIXES = (".safetensors", ".pt", ".pth", ".ckpt", ".bin")
|
||||
KNOWN_COMPONENTS = {
|
||||
@@ -33,7 +34,8 @@ KNOWN_COMPONENTS = {
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Classify a HF repo or local directory as Diffusers, raw, custom, or unknown.")
|
||||
description="Classify a HF repo or local directory as Diffusers, raw, custom, or unknown."
|
||||
)
|
||||
parser.add_argument("source", help="HF repo id or local weights directory")
|
||||
parser.add_argument("--repo-type", default="model", help="HF repo type (default: model)")
|
||||
parser.add_argument("--revision", help="HF revision to inspect")
|
||||
@@ -92,12 +94,14 @@ def load_remote_files(
|
||||
) -> list[str]:
|
||||
from huggingface_hub import list_repo_files
|
||||
|
||||
return sorted(list_repo_files(
|
||||
repo_id,
|
||||
repo_type=repo_type,
|
||||
revision=revision,
|
||||
token=token,
|
||||
))
|
||||
return sorted(
|
||||
list_repo_files(
|
||||
repo_id,
|
||||
repo_type=repo_type,
|
||||
revision=revision,
|
||||
token=token,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def load_remote_model_index(
|
||||
@@ -211,24 +215,24 @@ def build_result(args: argparse.Namespace) -> dict[str, Any]:
|
||||
"components_seen": components,
|
||||
"file_count": len(files),
|
||||
"file_scan_truncated": truncated,
|
||||
"files_sample": files[:args.sample_limit],
|
||||
"files_sample": files[: args.sample_limit],
|
||||
}
|
||||
|
||||
|
||||
def print_human(result: dict[str, Any]) -> None:
|
||||
for key in (
|
||||
"source",
|
||||
"source_kind",
|
||||
"repo_type",
|
||||
"revision",
|
||||
"token_env",
|
||||
"source_layout",
|
||||
"needs_conversion",
|
||||
"model_index_class",
|
||||
"model_index_diffusers_version",
|
||||
"model_index_error",
|
||||
"file_count",
|
||||
"file_scan_truncated",
|
||||
"source",
|
||||
"source_kind",
|
||||
"repo_type",
|
||||
"revision",
|
||||
"token_env",
|
||||
"source_layout",
|
||||
"needs_conversion",
|
||||
"model_index_class",
|
||||
"model_index_diffusers_version",
|
||||
"model_index_error",
|
||||
"file_count",
|
||||
"file_scan_truncated",
|
||||
):
|
||||
value = result.get(key)
|
||||
if value is not None:
|
||||
|
||||
@@ -18,6 +18,7 @@ import pytest
|
||||
import torch
|
||||
from torch.testing import assert_close
|
||||
|
||||
|
||||
os.environ.setdefault("MASTER_ADDR", "localhost")
|
||||
os.environ.setdefault("MASTER_PORT", "29519")
|
||||
os.environ.setdefault("DISABLE_SP", "1")
|
||||
@@ -34,10 +35,15 @@ FASTVIDEO_CONFIG_CLASS = "<FastVideoConfig>" # TODO.
|
||||
FASTVIDEO_MODEL_MODULE = "fastvideo.models.<bucket>.<module>" # TODO.
|
||||
FASTVIDEO_MODEL_CLASS = "<FastVideoModel>" # TODO.
|
||||
|
||||
OFFICIAL_REF_DIR = Path(os.getenv("<FAMILY_UPPER>_OFFICIAL_REF_DIR", REPO_ROOT / "<ReferenceDir>"))
|
||||
LOCAL_WEIGHTS_DIR = Path(os.getenv("<FAMILY_UPPER>_LOCAL_WEIGHTS_DIR", REPO_ROOT / "official_weights" / FAMILY))
|
||||
CONVERTED_WEIGHTS_DIR = Path(os.getenv("<FAMILY_UPPER>_CONVERTED_WEIGHTS_DIR",
|
||||
REPO_ROOT / "converted_weights" / FAMILY))
|
||||
OFFICIAL_REF_DIR = Path(
|
||||
os.getenv("<FAMILY_UPPER>_OFFICIAL_REF_DIR", REPO_ROOT / "<ReferenceDir>")
|
||||
)
|
||||
LOCAL_WEIGHTS_DIR = Path(
|
||||
os.getenv("<FAMILY_UPPER>_LOCAL_WEIGHTS_DIR", REPO_ROOT / "official_weights" / FAMILY)
|
||||
)
|
||||
CONVERTED_WEIGHTS_DIR = Path(
|
||||
os.getenv("<FAMILY_UPPER>_CONVERTED_WEIGHTS_DIR", REPO_ROOT / "converted_weights" / FAMILY)
|
||||
)
|
||||
|
||||
|
||||
def _resolve_hf_token() -> str | None:
|
||||
@@ -93,14 +99,18 @@ def _load_official_model(device: torch.device, dtype: torch.dtype) -> torch.nn.M
|
||||
model = OfficialClass() # TODO: pass official config kwargs.
|
||||
state_dict = {} # TODO: load official state dict from LOCAL_WEIGHTS_DIR.
|
||||
missing, unexpected = model.load_state_dict(state_dict, strict=True)
|
||||
assert not missing and not unexpected, (f"official load mismatch missing={missing[:5]} unexpected={unexpected[:5]}")
|
||||
assert not missing and not unexpected, (
|
||||
f"official load mismatch missing={missing[:5]} unexpected={unexpected[:5]}"
|
||||
)
|
||||
return model.to(device=device, dtype=dtype).eval()
|
||||
|
||||
|
||||
def _load_fastvideo_model(device: torch.device, dtype: torch.dtype) -> torch.nn.Module:
|
||||
"""Load the FastVideo component with the same tensor content."""
|
||||
if not CONVERTED_WEIGHTS_DIR.exists() and not LOCAL_WEIGHTS_DIR.exists():
|
||||
pytest.skip(f"No FastVideo loadable weights: {CONVERTED_WEIGHTS_DIR} or {LOCAL_WEIGHTS_DIR}")
|
||||
pytest.skip(
|
||||
f"No FastVideo loadable weights: {CONVERTED_WEIGHTS_DIR} or {LOCAL_WEIGHTS_DIR}"
|
||||
)
|
||||
|
||||
# TODO: replace with the bucket-specific FastVideo config/class/loader.
|
||||
# DiT examples:
|
||||
@@ -117,7 +127,8 @@ def _load_fastvideo_model(device: torch.device, dtype: torch.dtype) -> torch.nn.
|
||||
state_dict = {} # TODO: load converted or directly mapped state dict.
|
||||
missing, unexpected = model.load_state_dict(state_dict, strict=True)
|
||||
assert not missing and not unexpected, (
|
||||
f"FastVideo load mismatch missing={missing[:5]} unexpected={unexpected[:5]}")
|
||||
f"FastVideo load mismatch missing={missing[:5]} unexpected={unexpected[:5]}"
|
||||
)
|
||||
return model.to(device=device, dtype=dtype).eval()
|
||||
|
||||
|
||||
@@ -176,9 +187,11 @@ def test_component_parity():
|
||||
|
||||
assert official_out.shape == fastvideo_out.shape
|
||||
diff = (official_out - fastvideo_out).abs()
|
||||
print(f"official abs_mean={official_out.abs().mean().item():.6f} "
|
||||
f"fastvideo abs_mean={fastvideo_out.abs().mean().item():.6f} "
|
||||
f"diff_max={diff.max().item():.6f} diff_mean={diff.mean().item():.6f}")
|
||||
print(
|
||||
f"official abs_mean={official_out.abs().mean().item():.6f} "
|
||||
f"fastvideo abs_mean={fastvideo_out.abs().mean().item():.6f} "
|
||||
f"diff_max={diff.max().item():.6f} diff_mean={diff.mean().item():.6f}"
|
||||
)
|
||||
|
||||
# TODO: pick tolerance by scope:
|
||||
# - single block / same kernel: 1e-4
|
||||
|
||||
@@ -27,6 +27,7 @@ try:
|
||||
except ImportError: # pragma: no cover - optional local conversion dependency
|
||||
snapshot_download = None
|
||||
|
||||
|
||||
# TODO: fill with authoritative component prefixes for monolithic checkpoints.
|
||||
# Example: {"model.model.": "transformer", "pretransform.model.": "vae"}
|
||||
COMPONENT_PREFIXES: dict[str, str] = {}
|
||||
@@ -46,7 +47,10 @@ SKIP_PATTERNS: tuple[str, ...] = ()
|
||||
|
||||
|
||||
def _hf_token() -> str | None:
|
||||
return (os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_HUB_TOKEN") or os.environ.get("HF_API_KEY"))
|
||||
return (
|
||||
os.environ.get("HF_TOKEN") or os.environ.get("HUGGINGFACE_HUB_TOKEN")
|
||||
or os.environ.get("HF_API_KEY")
|
||||
)
|
||||
|
||||
|
||||
def resolve_src(src: str, revision: str | None) -> Path:
|
||||
@@ -91,10 +95,11 @@ def apply_mapping(key: str) -> str | None:
|
||||
return key
|
||||
|
||||
|
||||
def split_monolithic(state: dict[str, torch.Tensor], ) -> dict[str, OrderedDict[str, torch.Tensor]]:
|
||||
def split_monolithic(
|
||||
state: dict[str, torch.Tensor],
|
||||
) -> dict[str, OrderedDict[str, torch.Tensor]]:
|
||||
components: dict[str, OrderedDict[str, torch.Tensor]] = {
|
||||
name: OrderedDict()
|
||||
for name in set(COMPONENT_PREFIXES.values())
|
||||
name: OrderedDict() for name in set(COMPONENT_PREFIXES.values())
|
||||
}
|
||||
intentionally_skipped: list[str] = []
|
||||
unowned: list[str] = []
|
||||
@@ -112,8 +117,10 @@ def split_monolithic(state: dict[str, torch.Tensor], ) -> dict[str, OrderedDict[
|
||||
unowned.append(key)
|
||||
if unowned:
|
||||
sample = ", ".join(unowned[:10])
|
||||
raise ValueError(f"Unowned monolithic keys: {len(unowned)}. "
|
||||
f"Add COMPONENT_PREFIXES or SKIP_PATTERNS entries. Sample: {sample}")
|
||||
raise ValueError(
|
||||
f"Unowned monolithic keys: {len(unowned)}. "
|
||||
f"Add COMPONENT_PREFIXES or SKIP_PATTERNS entries. Sample: {sample}"
|
||||
)
|
||||
if intentionally_skipped:
|
||||
print(f"Intentionally skipped {len(intentionally_skipped)} keys")
|
||||
return {name: weights for name, weights in components.items() if weights}
|
||||
@@ -136,12 +143,8 @@ def build_component_configs(_src_dir: Path) -> dict[str, dict[str, Any]]:
|
||||
# TODO: emit config content accepted by FastVideo loaders. Most components use
|
||||
# config.json; schedulers use scheduler_config.json.
|
||||
return {
|
||||
"transformer": {
|
||||
"_class_name": "<FastVideoTransformerClass>"
|
||||
},
|
||||
"vae": {
|
||||
"_class_name": "<FastVideoVAEClass>"
|
||||
},
|
||||
"transformer": {"_class_name": "<FastVideoTransformerClass>"},
|
||||
"vae": {"_class_name": "<FastVideoVAEClass>"},
|
||||
}
|
||||
|
||||
|
||||
@@ -174,13 +177,19 @@ def build_model_index(
|
||||
}
|
||||
if revision:
|
||||
index["_fastvideo_converted_revision"] = revision
|
||||
return {key: value for key, value in index.items() if key.startswith("_") or key in available_components}
|
||||
return {
|
||||
key: value
|
||||
for key, value in index.items()
|
||||
if key.startswith("_") or key in available_components
|
||||
}
|
||||
|
||||
|
||||
def validate_component_configs(configs: dict[str, dict[str, Any]]) -> None:
|
||||
# TODO: instantiate each FastVideo config and call update_model_arch(...) or
|
||||
# update_model_config(...) with this JSON so unknown emitted keys fail here.
|
||||
placeholder_configs = [name for name, config in configs.items() if "<" in json.dumps(config)]
|
||||
placeholder_configs = [
|
||||
name for name, config in configs.items() if "<" in json.dumps(config)
|
||||
]
|
||||
if placeholder_configs:
|
||||
raise ValueError(f"Replace config placeholders for: {placeholder_configs}")
|
||||
|
||||
@@ -192,7 +201,9 @@ def verify_conversion(
|
||||
del dst_dir, components
|
||||
# TODO: load each emitted stateful component through its production loader and
|
||||
# assert strict load, or document exact allowed missing/unexpected keys.
|
||||
raise NotImplementedError("Implement production config validation and strict-load checks")
|
||||
raise NotImplementedError(
|
||||
"Implement production config validation and strict-load checks"
|
||||
)
|
||||
|
||||
|
||||
def write_component(
|
||||
@@ -205,7 +216,9 @@ def write_component(
|
||||
if component_dir.exists() and any(component_dir.iterdir()):
|
||||
shutil.rmtree(component_dir)
|
||||
component_dir.mkdir(parents=True, exist_ok=True)
|
||||
save_file(dict(state), str(component_dir / "diffusion_pytorch_model.safetensors"))
|
||||
save_file(
|
||||
dict(state), str(component_dir / "diffusion_pytorch_model.safetensors")
|
||||
)
|
||||
if config is not None:
|
||||
config_path = component_dir / config_filename(name)
|
||||
with config_path.open("w", encoding="utf-8") as f:
|
||||
@@ -248,7 +261,9 @@ def convert(
|
||||
|
||||
if layout in {"monolithic", "raw_official"}:
|
||||
# TODO: replace model.safetensors with the official monolithic file name.
|
||||
components = split_monolithic(load_checkpoint(default_monolithic_checkpoint(src_path)))
|
||||
components = split_monolithic(
|
||||
load_checkpoint(default_monolithic_checkpoint(src_path))
|
||||
)
|
||||
elif layout in {"separate_components", "mixed"}:
|
||||
if not src_path.is_dir():
|
||||
raise ValueError(f"{layout} layout requires a source directory: {src_path}")
|
||||
@@ -256,7 +271,9 @@ def convert(
|
||||
else:
|
||||
raise ValueError(f"Unsupported template layout: {layout}")
|
||||
|
||||
copied = (copy_passthrough(src_path, dst_dir) if src_path.is_dir() else [])
|
||||
copied = (
|
||||
copy_passthrough(src_path, dst_dir) if src_path.is_dir() else []
|
||||
)
|
||||
configs = build_component_configs(src_path if src_path.is_dir() else src_path.parent)
|
||||
validate_component_configs(configs)
|
||||
for name, state in components.items():
|
||||
@@ -272,7 +289,9 @@ def convert(
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--src", required=True, help="HF repo id, local dir, or checkpoint path")
|
||||
parser.add_argument(
|
||||
"--src", required=True, help="HF repo id, local dir, or checkpoint path"
|
||||
)
|
||||
parser.add_argument("--revision", help="HF branch, tag, or commit for repo sources")
|
||||
parser.add_argument(
|
||||
"--dst",
|
||||
|
||||
@@ -24,8 +24,8 @@ from typing import Any
|
||||
|
||||
import torch
|
||||
|
||||
FAMILY: str = "<family>" # e.g. "magi_human", "ltx2", "wan"
|
||||
COMPONENT: str = "<component>" # e.g. "dit", "vae", "encoder"
|
||||
FAMILY: str = "<family>" # e.g. "magi_human", "ltx2", "wan"
|
||||
COMPONENT: str = "<component>" # e.g. "dit", "vae", "encoder"
|
||||
DRILL_LAYER_ENV: str = "<FAMILY>_DEBUG_DRILL_LAYER"
|
||||
HYPOTHESIS_ENV: str = "<FAMILY>_DEBUG_PATCH_<HYPOTHESIS>"
|
||||
REL_THRESHOLD: float = 0.005 # 0.5% abs_mean drift flags a block as divergent
|
||||
@@ -94,7 +94,6 @@ def _attach_block_hooks(
|
||||
handles: list[Any] = []
|
||||
|
||||
def _hook(name: str):
|
||||
|
||||
def fn(_module, _inputs, outputs):
|
||||
t = outputs[0] if isinstance(outputs, tuple) else outputs
|
||||
if not torch.is_tensor(t):
|
||||
@@ -102,7 +101,6 @@ def _attach_block_hooks(
|
||||
log.append({"side": label, **_stat(name, t)})
|
||||
if tensors is not None:
|
||||
tensors[name] = t.detach().float().cpu()
|
||||
|
||||
return fn
|
||||
|
||||
def _pre_hook(name: str):
|
||||
@@ -116,7 +114,6 @@ def _attach_block_hooks(
|
||||
log.append({"side": label, **_stat(key, t)})
|
||||
if tensors is not None:
|
||||
tensors[key] = t.detach().float().cpu()
|
||||
|
||||
return fn
|
||||
|
||||
# TODO: adapt attribute paths to your model. Remove adapter block if absent.
|
||||
@@ -134,21 +131,43 @@ def _attach_block_hooks(
|
||||
# magi-human uses: attention, mlp.pre_norm, mlp.up_gate_proj,
|
||||
# mlp.down_proj (pre+post), mlp, attn_post_norm, mlp_post_norm.
|
||||
if hasattr(layer, "attention"):
|
||||
handles.append(layer.attention.register_forward_hook(_hook(f"{tag}.attention")))
|
||||
handles.append(
|
||||
layer.attention.register_forward_hook(_hook(f"{tag}.attention"))
|
||||
)
|
||||
if hasattr(layer, "mlp"):
|
||||
mlp = layer.mlp
|
||||
if hasattr(mlp, "pre_norm"):
|
||||
handles.append(mlp.pre_norm.register_forward_hook(_hook(f"{tag}.mlp.pre_norm")))
|
||||
handles.append(
|
||||
mlp.pre_norm.register_forward_hook(_hook(f"{tag}.mlp.pre_norm"))
|
||||
)
|
||||
if hasattr(mlp, "up_gate_proj"):
|
||||
handles.append(mlp.up_gate_proj.register_forward_hook(_hook(f"{tag}.mlp.up_gate_proj")))
|
||||
handles.append(
|
||||
mlp.up_gate_proj.register_forward_hook(
|
||||
_hook(f"{tag}.mlp.up_gate_proj")
|
||||
)
|
||||
)
|
||||
if hasattr(mlp, "down_proj"):
|
||||
handles.append(mlp.down_proj.register_forward_pre_hook(_pre_hook(f"{tag}.mlp.down_proj")))
|
||||
handles.append(mlp.down_proj.register_forward_hook(_hook(f"{tag}.mlp.down_proj")))
|
||||
handles.append(
|
||||
mlp.down_proj.register_forward_pre_hook(
|
||||
_pre_hook(f"{tag}.mlp.down_proj")
|
||||
)
|
||||
)
|
||||
handles.append(
|
||||
mlp.down_proj.register_forward_hook(_hook(f"{tag}.mlp.down_proj"))
|
||||
)
|
||||
handles.append(mlp.register_forward_hook(_hook(f"{tag}.mlp")))
|
||||
if hasattr(layer, "attn_post_norm"):
|
||||
handles.append(layer.attn_post_norm.register_forward_hook(_hook(f"{tag}.attn_post_norm")))
|
||||
handles.append(
|
||||
layer.attn_post_norm.register_forward_hook(
|
||||
_hook(f"{tag}.attn_post_norm")
|
||||
)
|
||||
)
|
||||
if hasattr(layer, "mlp_post_norm"):
|
||||
handles.append(layer.mlp_post_norm.register_forward_hook(_hook(f"{tag}.mlp_post_norm")))
|
||||
handles.append(
|
||||
layer.mlp_post_norm.register_forward_hook(
|
||||
_hook(f"{tag}.mlp_post_norm")
|
||||
)
|
||||
)
|
||||
return handles
|
||||
|
||||
|
||||
@@ -174,9 +193,11 @@ def _write_log(entries: list[dict], path: Path) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(path, "w") as f:
|
||||
for e in entries:
|
||||
f.write(f"{e['name']} {e['shape']} "
|
||||
f"{e['abs_mean']:.8f} {e['sum']:.4f} "
|
||||
f"{e['min']:.6f} {e['max']:.6f}\n")
|
||||
f.write(
|
||||
f"{e['name']} {e['shape']} "
|
||||
f"{e['abs_mean']:.8f} {e['sum']:.4f} "
|
||||
f"{e['min']:.6f} {e['max']:.6f}\n"
|
||||
)
|
||||
|
||||
|
||||
def _sort_key(name: str, drill_layer: int) -> tuple:
|
||||
@@ -184,14 +205,9 @@ def _sort_key(name: str, drill_layer: int) -> tuple:
|
||||
return (0, "")
|
||||
if name.startswith(f"L{drill_layer:02d}."):
|
||||
sub_order = {
|
||||
"attention": 0,
|
||||
"attn_post_norm": 1,
|
||||
"mlp.pre_norm": 2,
|
||||
"mlp.up_gate_proj": 3,
|
||||
"mlp.down_proj<in>": 4,
|
||||
"mlp.down_proj": 5,
|
||||
"mlp": 6,
|
||||
"mlp_post_norm": 7,
|
||||
"attention": 0, "attn_post_norm": 1, "mlp.pre_norm": 2,
|
||||
"mlp.up_gate_proj": 3, "mlp.down_proj<in>": 4,
|
||||
"mlp.down_proj": 5, "mlp": 6, "mlp_post_norm": 7,
|
||||
}.get(name.split(".", 1)[1], 9)
|
||||
return (1, f"block[{drill_layer:02d}]", sub_order)
|
||||
if name.startswith("block["):
|
||||
@@ -200,8 +216,10 @@ def _sort_key(name: str, drill_layer: int) -> tuple:
|
||||
|
||||
|
||||
def _print_table(by_name: dict[str, dict], drill_layer: int) -> int | None:
|
||||
hdr = (f"{'name':<18} {'up_shape':<22} {'up_absmean':>12} {'fv_absmean':>12} "
|
||||
f"{'absmean_diff':>14} {'rel%':>8} {'up_sum':>14} {'fv_sum':>14} {'sum_diff':>12}")
|
||||
hdr = (
|
||||
f"{'name':<18} {'up_shape':<22} {'up_absmean':>12} {'fv_absmean':>12} "
|
||||
f"{'absmean_diff':>14} {'rel%':>8} {'up_sum':>14} {'fv_sum':>14} {'sum_diff':>12}"
|
||||
)
|
||||
print(f"\n{hdr}\n{'-' * len(hdr)}")
|
||||
first_div: int | None = None
|
||||
for name in sorted(by_name.keys(), key=lambda n: _sort_key(n, drill_layer)):
|
||||
@@ -217,9 +235,11 @@ def _print_table(by_name: dict[str, dict], drill_layer: int) -> int | None:
|
||||
flag = " <<< DIVERGE"
|
||||
if first_div is None:
|
||||
first_div = int(name[len("block["):-1])
|
||||
print(f"{name:<18} {str(up['shape']):<22} {up['abs_mean']:>12.6f} "
|
||||
f"{fv['abs_mean']:>12.6f} {am_diff:>14.6f} {am_rel * 100:>7.3f}% "
|
||||
f"{up['sum']:>14.4f} {fv['sum']:>14.4f} {sum_diff:>12.4f}{flag}")
|
||||
print(
|
||||
f"{name:<18} {str(up['shape']):<22} {up['abs_mean']:>12.6f} "
|
||||
f"{fv['abs_mean']:>12.6f} {am_diff:>14.6f} {am_rel * 100:>7.3f}% "
|
||||
f"{up['sum']:>14.4f} {fv['sum']:>14.4f} {sum_diff:>12.4f}{flag}"
|
||||
)
|
||||
return first_div
|
||||
|
||||
|
||||
@@ -235,8 +255,10 @@ def _print_elementwise(up_t: dict[str, torch.Tensor], fv_t: dict[str, torch.Tens
|
||||
continue
|
||||
diff = (a - b).abs()
|
||||
rel = (diff.mean().item() / max(a.abs().mean().item(), 1e-9)) * 100
|
||||
print(f"{name:<30} {str(tuple(a.shape)):<22} "
|
||||
f"{diff.max().item():>12.6f} {diff.mean().item():>12.6f} {rel:>9.4f}%")
|
||||
print(
|
||||
f"{name:<30} {str(tuple(a.shape)):<22} "
|
||||
f"{diff.max().item():>12.6f} {diff.mean().item():>12.6f} {rel:>9.4f}%"
|
||||
)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
|
||||
@@ -43,10 +43,12 @@ def _add_official_to_path() -> Path:
|
||||
|
||||
def _log_tensor_stats(label: str, tensor: torch.Tensor) -> None:
|
||||
value = tensor.detach().float()
|
||||
print(f"[{_MODEL_FAMILY} PIPELINE] {label}: shape={tuple(tensor.shape)} "
|
||||
f"dtype={tensor.dtype} device={tensor.device} "
|
||||
f"min={value.min().item():.6f} max={value.max().item():.6f} "
|
||||
f"mean={value.mean().item():.6f} std={value.std().item():.6f}")
|
||||
print(
|
||||
f"[{_MODEL_FAMILY} PIPELINE] {label}: shape={tuple(tensor.shape)} "
|
||||
f"dtype={tensor.dtype} device={tensor.device} "
|
||||
f"min={value.min().item():.6f} max={value.max().item():.6f} "
|
||||
f"mean={value.mean().item():.6f} std={value.std().item():.6f}"
|
||||
)
|
||||
|
||||
|
||||
def _extract_tensor(output: Any, key: str) -> torch.Tensor:
|
||||
@@ -71,8 +73,10 @@ def _run_official_pipeline(
|
||||
device: torch.device,
|
||||
) -> Any:
|
||||
del official_path, params, device
|
||||
pytest.skip("TODO: import the official pipeline/factory, load official weights, "
|
||||
"run with params, and return the comparison target.")
|
||||
pytest.skip(
|
||||
"TODO: import the official pipeline/factory, load official weights, "
|
||||
"run with params, and return the comparison target."
|
||||
)
|
||||
|
||||
|
||||
def _run_fastvideo_pipeline(model_path: Path, params: dict[str, Any]) -> Any:
|
||||
@@ -142,6 +146,8 @@ def test_todo_model_family_pipeline_official_parity() -> None:
|
||||
assert official_tensor.shape == fastvideo_tensor.shape
|
||||
|
||||
diff = (official_tensor - fastvideo_tensor).abs()
|
||||
print(f"diff max={diff.max().item():.6f} "
|
||||
f"mean={diff.mean().item():.6f} median={diff.median().item():.6f}")
|
||||
print(
|
||||
f"diff max={diff.max().item():.6f} "
|
||||
f"mean={diff.mean().item():.6f} median={diff.median().item():.6f}"
|
||||
)
|
||||
assert_close(fastvideo_tensor, official_tensor, atol=1e-2, rtol=1e-2)
|
||||
|
||||
@@ -1,99 +0,0 @@
|
||||
---
|
||||
name: ci-runner
|
||||
description: Work on FastVideo's Slurm-only, change-aware GPU CI lanes, static Buildkite graph, trusted ci-runner policy, lane scripts, and GB200 validation.
|
||||
---
|
||||
|
||||
# Slinky Slurm CI lanes
|
||||
|
||||
FastVideo's `ci-runner` Buildkite queue is the control plane for all active
|
||||
GPU CI. A private host-owned dispatcher leases GPUs from the Slinky Slurm tray
|
||||
and runs the immutable PR SHA inside an isolated Enroot container. Buildkite
|
||||
pipeline upload and Slurm submission occur on the login plane; every test
|
||||
payload executes on Slurm compute.
|
||||
|
||||
The files under `fastvideo/tests/modal/` and `.buildkite/scripts/pr_test.sh`
|
||||
are dormant rollback code. Never add an active Buildkite or slash-command
|
||||
route to them. `pr_test.sh` must continue to reject Buildkite invocations.
|
||||
|
||||
The private operator bundle is deliberately outside this repository because
|
||||
it contains site paths and credentials. See
|
||||
`docs/contributing/ci_architecture.md`; this skill covers the repository half
|
||||
and the coordination contract with that bundle.
|
||||
|
||||
## Invariants
|
||||
|
||||
- `.buildkite/pipeline.yml` contains exactly one static step for every active
|
||||
GPU lane. Each step pins a unique key and label, a 90-minute timeout, the
|
||||
trusted `/opt/fastvideo-ci-runner/run-ci` command (`run-unit` is the one
|
||||
compatibility wrapper), step-level internal `TEST_TYPE`, and
|
||||
`queue: "ci-runner"`.
|
||||
- Active CI contains no `pr_test.sh` command, Modal invocation, default queue,
|
||||
Buildkite plugin, `soft_fail`, or job-controlled artifact glob.
|
||||
- The six Fastcheck lanes use `:microscope:` labels. Full-Suite-only lanes use
|
||||
`:test_tube:` or `:bar_chart:` so direct reruns update the right aggregate.
|
||||
- SSIM and vanilla training request all four GPUs. Keep both in the
|
||||
`fastvideo/slinky/whole-tray` Buildkite concurrency group with a limit of one
|
||||
so the second job does not consume an agent or command timeout while waiting
|
||||
for the same tray.
|
||||
- `/test full` schedules all twenty lanes. `/merge`, `ready`, and new pushes to
|
||||
ready PRs use the trusted base-branch planner in
|
||||
`.github/scripts/plan_merge_ci.py`: automatic Fastcheck remains the universal
|
||||
six-lane baseline, and the merge build adds only path-relevant integration
|
||||
lanes. Unknown source/build paths fail closed to all fourteen additive lanes.
|
||||
The trusted uploader still normalizes and validates the complete static graph
|
||||
before Buildkite evaluates its plan conditions.
|
||||
- Focused merge builds may pass allowlisted golden-gate and SSIM test basenames.
|
||||
The private host validates the lane plan and basenames before staging them,
|
||||
and the in-container scripts validate them again. Direct `/test ssim`,
|
||||
explicit `/test full`, and the weekly main-branch schedule run the complete
|
||||
SSIM matrix.
|
||||
- The trusted uploader serves exactly three entry pipelines:
|
||||
`pr-fastcheck` for automatic PR builds, `ci` for slash-command/ready-label
|
||||
API builds, and `fastvideo-performance-lane` for the weekly schedule. Keep
|
||||
incoming GitHub webhook processing disabled on `ci` so it cannot duplicate
|
||||
`pr-fastcheck` on every PR update.
|
||||
- Test payloads live in `.buildkite/scripts/unit_test.sh` or executable
|
||||
`.buildkite/scripts/lanes/<lane>.sh`. Backend policy (GPU count, extras,
|
||||
secrets, kernel build, artifacts) stays in the agent-owned lane table.
|
||||
- Tests must preserve an inherited `MASTER_PORT`. Packed containers share the
|
||||
tray network namespace, so the private runner assigns a distinct port range
|
||||
per GPU lease and the SSIM scheduler assigns task offsets within its range.
|
||||
- The ARM64 runner image includes the pinned FA4 CuTe overlay validated on
|
||||
GB200. Keep SSIM at `FASTVIDEO_FA4=1` because its references were seeded with
|
||||
FA4; keep lanes with FA2 baselines at `FASTVIDEO_FA4=0`. A runner image change
|
||||
must revalidate both the FA4 import and an actual GB200 forward kernel.
|
||||
- `fastvideo/tests/ssim/ci_runner.py` is the active four-GPU SSIM scheduler.
|
||||
New SSIM files are discovered through `REQUIRED_GPUS` and
|
||||
`*_MODEL_TO_PARAMS`; do not wire them through the dormant Modal scheduler.
|
||||
- The host policy fail-closes unknown tuples. A repository-side lane change is
|
||||
inert until the operator updates the private lane table and uploader policy
|
||||
in the same rollout.
|
||||
|
||||
## Adding or changing a lane
|
||||
|
||||
1. Read the closest `AGENTS.md` and the domain-specific testing guide.
|
||||
2. Add or update the executable lane payload under `.buildkite/scripts/`.
|
||||
Keep it deterministic and free of host-specific paths or credential fetches.
|
||||
3. Add the static pipeline step and canonical `/test <name>` mapping. Keep the
|
||||
`<name>-ci` alias only when compatibility requires it.
|
||||
4. Add its source/test path ownership to `.github/scripts/plan_merge_ci.py`.
|
||||
Prefer the narrowest correctness-preserving lane set; leave unknown paths
|
||||
fail-closed. Extend `fastvideo/tests/contract/test_ci_test_collection.py`,
|
||||
`test_merge_ci_plan.py`, and focused CPU-only scheduler/policy tests.
|
||||
5. Coordinate the private lane row: GPU count (1-4), wall time, script, scope
|
||||
pairs, step key, command, HF cache/token, tracking mode, extras, attention
|
||||
backend policy, kernel policy, and artifact relay. Active training lanes
|
||||
keep W&B offline and do not stage a W&B credential.
|
||||
6. Update the trusted pipeline-uploader schema. A mismatch must reject the
|
||||
pipeline rather than silently skip a lane.
|
||||
7. Run `pre-commit run --files <changed paths>`, the planner's representative
|
||||
diff matrix, contract tests, private driver tests, and a real GB200 canary.
|
||||
Multi-GPU, hardware-reference, training, performance, and SSIM changes need
|
||||
their own target-hardware evidence.
|
||||
|
||||
## Rollback
|
||||
|
||||
Rollback the Slurm routing/configuration change or pause the `ci-runner` queue.
|
||||
Do not silently reactivate Modal. A manual Modal experiment requires the
|
||||
explicit local opt-in documented in `ci_architecture.md`; returning it to
|
||||
production CI needs a separate reviewed decision.
|
||||
@@ -1,76 +0,0 @@
|
||||
---
|
||||
name: env-var-conventions
|
||||
description: Add, read, rename, or remove an environment variable in FastVideo, or change the environment-variable policy. Use before touching fastvideo/envs.py, os.environ, os.getenv, or monkeypatch.setenv in fastvideo/, and when fastvideo/tests/contract/test_env_policy.py fails.
|
||||
---
|
||||
|
||||
# Environment Variable Conventions
|
||||
|
||||
## Purpose
|
||||
|
||||
FastVideo registers its environment variables as typed fields in
|
||||
`fastvideo/envs.py`. The policy that governs them is
|
||||
`docs/contributing/env_vars.md`, and the contract test
|
||||
`fastvideo/tests/contract/test_env_policy.py` enforces the policy in the unit
|
||||
CI lane. This skill routes an environment-variable change through that policy.
|
||||
The policy doc is the single source of the rules; read it instead of relying
|
||||
on a summary here.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- Read `docs/contributing/env_vars.md` in full.
|
||||
- Decide whether the setting belongs in an environment variable or an argument
|
||||
(rule 5 in the policy doc). Settings that users change per deployment are
|
||||
arguments; add them through `fastvideo/fastvideo_args.py` instead.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Required | Description |
|
||||
| ---------- | -------- | -------------------------------------------------------------- |
|
||||
| `change` | Yes | Add, read, rename, or remove a variable, or change the policy. |
|
||||
| `variable` | Yes | The variable name, with the `FASTVIDEO_` prefix. |
|
||||
|
||||
## Steps
|
||||
|
||||
1. **Declare or edit the variable in `fastvideo/envs.py`.**
|
||||
- Pick the field type and category that the policy doc lists.
|
||||
- Write a description that states what the variable does and its units.
|
||||
- To rename, keep the old name in `deprecated_names`. To remove, add the
|
||||
name to `DEPRECATED_VARIABLES`. Update the uses in `examples/`,
|
||||
`scripts/`, `docs/`, `apps/`, and the tests.
|
||||
2. **Read the variable with `envs.NAME.get()` inside a function.**
|
||||
- In tests, change the value with `envs.NAME.override(value)`.
|
||||
- Do not call `os.environ`, `os.getenv`, or `monkeypatch.setenv` for a
|
||||
FastVideo variable.
|
||||
- To set a variable that another tool reads, call `envs.set_external`,
|
||||
`envs.setdefault_external`, or `envs.unset_external`.
|
||||
3. **Regenerate the table in the policy doc.**
|
||||
- Run `python fastvideo/tests/contract/test_env_policy.py`.
|
||||
4. **Run the contract test.**
|
||||
- Run `pytest fastvideo/tests/contract/test_env_policy.py`.
|
||||
- When the test reports a fixed known violation, delete or lower its entry
|
||||
in `KNOWN_VIOLATIONS`. Never add an entry to `KNOWN_VIOLATIONS`.
|
||||
5. **When the policy itself changes, update the policy doc and the contract
|
||||
test in the same pull request.**
|
||||
- The rules in `docs/contributing/env_vars.md`, the checks and allowlist in
|
||||
`fastvideo/tests/contract/test_env_policy.py`, and this skill must agree.
|
||||
|
||||
## Outputs
|
||||
|
||||
- A registry entry in `fastvideo/envs.py` and call sites that use
|
||||
`envs.NAME.get()`.
|
||||
- A regenerated table in `docs/contributing/env_vars.md`.
|
||||
- A passing `fastvideo/tests/contract/test_env_policy.py`.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```
|
||||
Add a FASTVIDEO_DEBUG_MY_STAGE switch that logs MyStage inputs.
|
||||
```
|
||||
|
||||
## References
|
||||
|
||||
- `docs/contributing/env_vars.md`: the policy, the field types, and the
|
||||
violation kinds that the contract test reports.
|
||||
- `fastvideo/envs.py`: the registry.
|
||||
- `fastvideo/tests/contract/test_env_policy.py`: the contract test,
|
||||
`EXTERNAL_ALLOWLIST`, and `KNOWN_VIOLATIONS`.
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: reseed-ssim-references
|
||||
description: Re-seed HF reference videos for a single existing SSIM test on Modal L40S. Always backs up current refs locally first, regenerates on Modal, pauses for the user to eyeball before-vs-after quality, then overwrites the targeted model subtree on `FastVideo/ssim-reference-videos` with `--force`. Use when an intentional code change (model port fix, attention backend swap, kernel upgrade, hyperparameter change) has invalidated existing refs and they need to be regenerated. Pairs with `seed-ssim-references`, which is for first-time seeding only.
|
||||
description: Re-seed HF reference videos for a single existing SSIM test on Modal L40S. Always backs up current refs locally first, regenerates on Modal, pauses for the user to eyeball before-vs-after quality, then overwrites the targeted `<model_id>` subtree on `FastVideo/ssim-reference-videos` with `--force`. Use when an intentional code change (model port fix, attention backend swap, kernel upgrade, hyperparameter change) has invalidated existing refs and they need to be regenerated. Pairs with `seed-ssim-references`, which is for first-time seeding only.
|
||||
---
|
||||
|
||||
# Re-seed SSIM Reference Videos
|
||||
@@ -13,7 +13,7 @@ on HF — the old refs are overwritten — so the skill always:
|
||||
|
||||
1. Confirms intent with a one-liner the user has to type.
|
||||
2. Downloads the existing refs as a local, timestamped backup.
|
||||
3. Regenerates through the manual legacy Modal L40S maintenance path.
|
||||
3. Regenerates on Modal L40S (same code path that CI uses).
|
||||
4. Pauses for a side-by-side eyeball of backup vs new mp4s.
|
||||
5. Uploads with `--force`, scoped to the single `--model-id`.
|
||||
6. Reminds the user to keep the backup until the PR lands.
|
||||
@@ -51,9 +51,8 @@ harder to recover from than failing closed.
|
||||
|
||||
Hardcoded:
|
||||
|
||||
- Modal GPU: **L40S**. This is a manual reference-maintenance target, not the
|
||||
active Slurm CI compute path; changing the SKU also changes the historical
|
||||
`L40S_reference_videos` contract.
|
||||
- Modal GPU: **L40S** (matches CI; re-seeding from another SKU produces refs
|
||||
that L40S CI cannot match).
|
||||
- Quality tier: **`default`**. `full_quality` is a separate, deliberate
|
||||
operation.
|
||||
- HF repo: `FastVideo/ssim-reference-videos` (override via
|
||||
|
||||
@@ -35,8 +35,7 @@ The skill is run **manually**, once per new test. Before invoking it, the user
|
||||
has already sanity-tested the new test locally — it launches `VideoGenerator`
|
||||
and writes an artefact without crashing (the missing-reference assertion at
|
||||
the end is expected). The skill does not re-test locally; it goes straight
|
||||
to the manual legacy Modal L40S reference-maintenance target. Active CI runs
|
||||
on the Slinky Slurm cluster and only consumes the resulting references.
|
||||
to Modal L40S (which is what CI uses).
|
||||
|
||||
## When to use
|
||||
|
||||
@@ -62,8 +61,7 @@ Prompt the user for it if they didn't supply it.
|
||||
|
||||
Everything else is fixed:
|
||||
|
||||
- Modal maintenance GPU: **L40S** (hardcoded in
|
||||
`fastvideo/tests/modal/ssim_test.py`; this is not the active CI compute path).
|
||||
- Modal runner GPU: **L40S** (hardcoded in `fastvideo/tests/modal/ssim_test.py`).
|
||||
- Device folder: `L40S_reference_videos`.
|
||||
- Quality tier: `default` (the tier CI runs). The `full_quality` tier is not
|
||||
seeded by this skill.
|
||||
|
||||
+516
-462
@@ -1,9 +1,6 @@
|
||||
env:
|
||||
IMAGE_VERSION: "py3.12-latest"
|
||||
BUILDKITE_CLEAN_CHECKOUT: true
|
||||
# Slurm workers clone the immutable commit and initialize submodules inside
|
||||
# their isolated container. The Buildkite login-plane checkout is a no-op.
|
||||
BUILDKITE_GIT_SUBMODULES: false
|
||||
|
||||
notify:
|
||||
- github_commit_status:
|
||||
@@ -11,471 +8,528 @@ notify:
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
- github_commit_status:
|
||||
context: "full-suite-passed"
|
||||
if: build.env("TEST_SCOPE") == "full" || build.env("TEST_SCOPE") == "merge"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
- github_commit_status:
|
||||
context: "direct-test-completed"
|
||||
if: build.env("TEST_SCOPE") == "direct"
|
||||
- github_commit_status:
|
||||
context: "scheduled-ssim-passed"
|
||||
if: build.env("TEST_SCOPE") == "scheduled"
|
||||
|
||||
# This is the complete active GPU CI surface. Every command is a trusted host
|
||||
# dispatcher, and every test payload executes inside the Slinky Slurm tray.
|
||||
# fastvideo/tests/modal remains available only for an explicit manual rollback;
|
||||
# no active pipeline or slash-command route invokes it.
|
||||
steps:
|
||||
- label: ":microscope: Encoder Tests"
|
||||
key: "encoder"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,encoder,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "encoder" || build.env("TEST_TYPE") == "encoder_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "encoder_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
# ============================================================
|
||||
# Direct test: triggered by /test <name> slash command.
|
||||
# Labels match fastcheck/full-suite counterparts so the GitHub
|
||||
# check status overwrites the original failed check.
|
||||
# Only ONE step executes per build (gated by TEST_TYPE).
|
||||
# ============================================================
|
||||
|
||||
- label: ":microscope: VAE Tests"
|
||||
key: "vae"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,vae,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "vae" || build.env("TEST_TYPE") == "vae_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "vae_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
# --- Fastcheck-scope direct tests ---
|
||||
- label: ":microscope: Encoder Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "encoder"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: VAE Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "vae"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Transformer Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "transformer"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Kernel Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "kernel_tests"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: Unit Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "unit_test"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "dreamverse_app"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
- label: ":microscope: Transformer Tests"
|
||||
key: "transformer"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,transformer,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "transformer" || build.env("TEST_TYPE") == "transformer_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "transformer_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
# --- Full-suite-scope direct tests ---
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "ssim"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Extraction Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "lora_extraction"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "distillation_dmd"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "self_forcing"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_lora"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "training_vsa"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "inference_vmoba"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Performance Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "performance"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: API Server Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "api_server"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "train_framework"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- label: ":test_tube: Eval Metrics Tests"
|
||||
if: build.env("TEST_SCOPE") == "direct" && build.env("TEST_TYPE") == "eval"
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
- label: ":microscope: Kernel Tests"
|
||||
key: "kernel-tests"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,kernel-tests,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "kernel_tests" || build.env("TEST_TYPE") == "kernel_tests_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "kernel_tests_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
# ============================================================
|
||||
# Fastcheck: Runs on every PR (~10-15 min parallel)
|
||||
# Core component validation: encoders, VAEs, transformers,
|
||||
# CUDA kernels, and unit tests.
|
||||
# ============================================================
|
||||
- label: "Trigger Fastcheck"
|
||||
if: build.env("TEST_SCOPE") == "fastcheck" || build.env("TEST_SCOPE") == null
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/models/encoders/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/encoders/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Encoder Tests"
|
||||
env:
|
||||
- TEST_TYPE=encoder
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/vaes/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/vaes/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: VAE Tests"
|
||||
env:
|
||||
- TEST_TYPE=vae
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/transformers/**"
|
||||
- "fastvideo/layers/**"
|
||||
- "fastvideo/attention/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Transformer Tests"
|
||||
env:
|
||||
- TEST_TYPE=transformer
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo-kernel/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Kernel Tests"
|
||||
env:
|
||||
- TEST_TYPE=kernel_tests
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- ".buildkite/**"
|
||||
- ".github/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: Unit Tests"
|
||||
env:
|
||||
- TEST_TYPE=unit_test
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "apps/dreamverse/**"
|
||||
- "pyproject.toml"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":microscope: DreamVerse App Tests"
|
||||
env:
|
||||
- TEST_TYPE=dreamverse_app
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
- label: ":microscope: Unit Tests"
|
||||
key: "unit"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,unit,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "unit_test" || build.env("TEST_TYPE") == "unit_test_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-unit"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "unit_test_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":microscope: DreamVerse App Tests"
|
||||
key: "dreamverse"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,dreamverse,/) ||
|
||||
build.env("TEST_SCOPE") == "fastcheck" ||
|
||||
build.env("TEST_SCOPE") == null ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "dreamverse_app" || build.env("TEST_TYPE") == "dreamverse_app_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "dreamverse_app_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Golden-Gate Tests"
|
||||
key: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,golden-gate,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "golden_gate" || build.env("TEST_TYPE") == "golden_gate_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "golden_gate_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":bar_chart: SSIM Tests"
|
||||
key: "ssim"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
build.env("TEST_SCOPE") == "scheduled" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,ssim,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "ssim" || build.env("TEST_TYPE") == "ssim_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
concurrency: 1
|
||||
concurrency_group: "fastvideo/slinky/whole-tray"
|
||||
env:
|
||||
TEST_TYPE: "ssim_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: LoRA Inference Tests"
|
||||
key: "lora-inference"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,lora-inference,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "inference_lora" || build.env("TEST_TYPE") == "inference_lora_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "inference_lora_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: LoRA Extraction Tests"
|
||||
key: "lora-extraction"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,lora-extraction,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "lora_extraction" || build.env("TEST_TYPE") == "lora_extraction_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "lora_extraction_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Training Tests"
|
||||
key: "training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,training,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "training" || build.env("TEST_TYPE") == "training_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
concurrency: 1
|
||||
concurrency_group: "fastvideo/slinky/whole-tray"
|
||||
env:
|
||||
TEST_TYPE: "training_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Distillation DMD Tests"
|
||||
key: "distillation"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,distillation,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "distillation_dmd" || build.env("TEST_TYPE") == "distillation_dmd_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "distillation_dmd_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Self-Forcing Tests"
|
||||
key: "self-forcing"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,self-forcing,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "self_forcing" || build.env("TEST_TYPE") == "self_forcing_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "self_forcing_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: LoRA Training Tests"
|
||||
key: "lora-training"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,lora-training,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "training_lora" || build.env("TEST_TYPE") == "training_lora_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "training_lora_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Training Tests VSA"
|
||||
key: "training-vsa"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,training-vsa,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "training_vsa" || build.env("TEST_TYPE") == "training_vsa_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "training_vsa_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Inference Tests VMoBA"
|
||||
key: "inference-vmoba"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,inference-vmoba,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "inference_vmoba" || build.env("TEST_TYPE") == "inference_vmoba_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "inference_vmoba_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Performance Tests"
|
||||
key: "performance"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,performance,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "performance" || build.env("TEST_TYPE") == "performance_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "performance_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: API Server Tests"
|
||||
key: "api-server"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,api-server,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "api_server" || build.env("TEST_TYPE") == "api_server_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "api_server_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Train Framework Tests"
|
||||
key: "train-framework"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,train-framework,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "train_framework" || build.env("TEST_TYPE") == "train_framework_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "train_framework_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
|
||||
- label: ":test_tube: Eval Metrics Tests"
|
||||
key: "eval"
|
||||
depends_on: "golden-gate"
|
||||
if: |
|
||||
build.env("TEST_SCOPE") == "full" ||
|
||||
(build.env("TEST_SCOPE") == "merge" &&
|
||||
build.env("MERGE_TEST_PLAN") =~ /,eval,/) ||
|
||||
(build.env("TEST_SCOPE") == "direct" &&
|
||||
(build.env("TEST_TYPE") == "eval" || build.env("TEST_TYPE") == "eval_ci"))
|
||||
command: "/opt/fastvideo-ci-runner/run-ci"
|
||||
timeout_in_minutes: 90
|
||||
env:
|
||||
TEST_TYPE: "eval_ci"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "ci-runner"
|
||||
# ============================================================
|
||||
# Full Suite: Runs when TEST_SCOPE=full
|
||||
# Triggered by adding the 'ready' label (via ci-trigger-full-suite.yml)
|
||||
# or on-demand via /test full slash command.
|
||||
# Includes integration tests, SSIM regression, training pipelines,
|
||||
# and performance benchmarks.
|
||||
# ============================================================
|
||||
- label: "Trigger Full Suite"
|
||||
if: build.env("TEST_SCOPE") == "full"
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 128
|
||||
limit: 3
|
||||
- exit_status: -1
|
||||
limit: 2
|
||||
plugins:
|
||||
- monorepo-diff#v1.4.0:
|
||||
diff: 'git fetch origin "${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}" && git diff --name-only "origin/${BUILDKITE_PULL_REQUEST_BASE_BRANCH:-main}...HEAD"'
|
||||
watch:
|
||||
- path:
|
||||
- "fastvideo/**/*.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
label: ":bar_chart: SSIM Tests"
|
||||
env:
|
||||
- TEST_TYPE=ssim
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/tests/lora/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/tests/transformers/**"
|
||||
- "fastvideo/pipelines/**"
|
||||
- "fastvideo/layers/lora/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 20m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Inference Tests"
|
||||
env:
|
||||
- TEST_TYPE=inference_lora
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "scripts/lora_extraction/**"
|
||||
- "fastvideo/tests/lora_extraction/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "fastvideo/training/training_utils.py"
|
||||
- "fastvideo/layers/lora/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Extraction Tests"
|
||||
env:
|
||||
- TEST_TYPE=lora_extraction
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/training/*distillation_pipeline.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Distillation DMD Tests"
|
||||
env:
|
||||
- TEST_TYPE=distillation_dmd
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/training/*self_forcing_distillation_pipeline.py"
|
||||
- "fastvideo/tests/training/self-forcing/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Self-Forcing Tests"
|
||||
env:
|
||||
- TEST_TYPE=self_forcing
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 25m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: LoRA Training Tests"
|
||||
env:
|
||||
- TEST_TYPE=training_lora
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/**"
|
||||
- "fastvideo-kernel/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Training Tests VSA"
|
||||
env:
|
||||
- TEST_TYPE=training_vsa
|
||||
retry:
|
||||
automatic:
|
||||
- exit_status: 1
|
||||
limit: 2
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo-kernel/**"
|
||||
- "fastvideo/attention/backends/vmoba.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 15m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Inference Tests VMoBA"
|
||||
env:
|
||||
- TEST_TYPE=inference_vmoba
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/pipelines/**"
|
||||
- "fastvideo/attention/**"
|
||||
- "fastvideo/layers/**"
|
||||
- "fastvideo/worker/**"
|
||||
- "fastvideo/entrypoints/**"
|
||||
- "fastvideo/performance/**"
|
||||
- "fastvideo/tests/performance/**"
|
||||
- ".buildkite/performance-benchmarks/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Performance Tests"
|
||||
env:
|
||||
- TEST_TYPE=performance
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/entrypoints/openai/**"
|
||||
- "fastvideo/entrypoints/cli/serve.py"
|
||||
- "fastvideo/tests/entrypoints/test_openai_api_integration.py"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: API Server Tests"
|
||||
env:
|
||||
- TEST_TYPE=api_server
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/train/**"
|
||||
- "fastvideo/tests/train/models/**"
|
||||
- "fastvideo/tests/train/fixtures/**"
|
||||
- "fastvideo/models/dits/**"
|
||||
- "fastvideo/models/loader/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 30m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Train Framework Tests"
|
||||
env:
|
||||
- TEST_TYPE=train_framework
|
||||
agents:
|
||||
queue: "default"
|
||||
- path:
|
||||
- "fastvideo/eval/**"
|
||||
- "fastvideo/tests/eval/**"
|
||||
- "pyproject.toml"
|
||||
- "docker/Dockerfile"
|
||||
config:
|
||||
command: "timeout 90m .buildkite/scripts/pr_test.sh"
|
||||
label: ":test_tube: Eval Metrics Tests"
|
||||
env:
|
||||
- TEST_TYPE=eval
|
||||
agents:
|
||||
queue: "default"
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the OpenAI-compatible API lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/entrypoints/test_openai_api_integration.py -vs
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the distillation-DMD lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/training/distill/test_distill_dmd.py -vs
|
||||
@@ -1,87 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# DreamVerse needs a GPU for import-time device resolution, but it does not
|
||||
# build or exercise fastvideo-kernel. A checksummed Node archive is installed
|
||||
# in the disposable Slurm container because the shared CI image is
|
||||
# Python/CUDA focused.
|
||||
set -euo pipefail
|
||||
|
||||
node_version=v22.23.2
|
||||
case $(uname -m) in
|
||||
aarch64 | arm64)
|
||||
node_arch=arm64
|
||||
node_archive_sha256=013b59cfd2819703a6f4a14ab891fc46fc2a4e3f5bcd92de3fb4929b43e35b30
|
||||
;;
|
||||
x86_64 | amd64)
|
||||
node_arch=x64
|
||||
node_archive_sha256=b294a556e639d64338823920e5866c21c02741742d2e1529ee1a225c1ec9252a
|
||||
;;
|
||||
*)
|
||||
echo "Unsupported architecture for DreamVerse Node runtime: $(uname -m)" >&2
|
||||
exit 2
|
||||
;;
|
||||
esac
|
||||
node_archive="node-${node_version}-linux-${node_arch}.tar.gz"
|
||||
node_runtime_root=$(mktemp -d -t fastvideo-node.XXXXXX)
|
||||
node_archive_path="${node_runtime_root}/${node_archive}"
|
||||
node_install_dir="${node_runtime_root}/${node_archive%.tar.gz}"
|
||||
curl --proto '=https' --tlsv1.2 --retry 5 --retry-all-errors \
|
||||
--location --fail --silent --show-error \
|
||||
"https://nodejs.org/dist/${node_version}/${node_archive}" \
|
||||
--output "$node_archive_path"
|
||||
printf '%s %s\n' "$node_archive_sha256" "$node_archive_path" | sha256sum --check --status
|
||||
tar -xzf "$node_archive_path" -C "$node_runtime_root"
|
||||
export PATH="${node_install_dir}/bin:${PATH}"
|
||||
node --version
|
||||
npm --version
|
||||
|
||||
export PYTHONPATH="$(pwd)/apps/dreamverse${PYTHONPATH:+:$PYTHONPATH}"
|
||||
pytest apps/dreamverse/dreamverse/tests -q
|
||||
|
||||
cd apps/dreamverse/web
|
||||
npm ci
|
||||
npm run typecheck
|
||||
npm test
|
||||
machine_arch=$(uname -m)
|
||||
if [[ $machine_arch =~ ^(aarch64|arm64)$ ]]; then
|
||||
npx playwright install --with-deps chromium firefox
|
||||
else
|
||||
npx playwright install --with-deps chromium webkit firefox
|
||||
fi
|
||||
|
||||
master_port=${MASTER_PORT:-7959}
|
||||
BACKEND_PORT=${BACKEND_PORT:-$((master_port + 50))}
|
||||
python -m uvicorn dreamverse.mock_server:app --host 127.0.0.1 --port "$BACKEND_PORT" &
|
||||
mock_server_pid=$!
|
||||
cleanup() {
|
||||
kill "$mock_server_pid" 2>/dev/null || true
|
||||
wait "$mock_server_pid" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
for _ in {1..30}; do
|
||||
curl -fsS "http://127.0.0.1:$BACKEND_PORT/healthz" && break
|
||||
sleep 1
|
||||
done
|
||||
curl -fsS "http://127.0.0.1:$BACKEND_PORT/healthz"
|
||||
|
||||
if [[ $machine_arch =~ ^(aarch64|arm64)$ ]]; then
|
||||
# Playwright WebKit traps before opening a page on Linux ARM64, and its
|
||||
# bundled Chromium lacks the H.264/AAC codecs used by the fMP4 assertions.
|
||||
# Firefox covers every flow, including streaming. Chromium and its mobile
|
||||
# profile still cover all codec-independent UI behavior on GB200.
|
||||
BACKEND_HOST=127.0.0.1 BACKEND_PORT="$BACKEND_PORT" CI=1 \
|
||||
npm run e2e -- --project=firefox
|
||||
BACKEND_HOST=127.0.0.1 BACKEND_PORT="$BACKEND_PORT" CI=1 \
|
||||
npm run e2e -- \
|
||||
--project=chromium \
|
||||
--project=mobile-chromium \
|
||||
--grep-invert='streams, plays, and surfaces a downloadable clip|starts a new project and switches back to the prior session|saved projects persist across a page reload'
|
||||
else
|
||||
BACKEND_HOST=127.0.0.1 BACKEND_PORT="$BACKEND_PORT" CI=1 \
|
||||
npm run e2e -- \
|
||||
--project=chromium \
|
||||
--project=webkit \
|
||||
--project=firefox \
|
||||
--project=mobile-safari \
|
||||
--project=mobile-chromium
|
||||
fi
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the encoder lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/encoders -vs
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the evaluation lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/eval -vs
|
||||
@@ -1,35 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the golden-gate lane. Environment (HF_HOME
|
||||
# and authentication) is the runner's responsibility.
|
||||
set -euo pipefail
|
||||
|
||||
golden_root=./fastvideo/tests/golden_gate
|
||||
selected=${FASTVIDEO_GOLDEN_TEST_FILES-}
|
||||
if [ -z "$selected" ]; then
|
||||
if [ "${TEST_SCOPE:-}" = merge ]; then
|
||||
echo "Missing FASTVIDEO_GOLDEN_TEST_FILES for merge scope" >&2
|
||||
exit 2
|
||||
fi
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" = all ]; then
|
||||
exec pytest "$golden_root" -xvs
|
||||
fi
|
||||
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
echo "Invalid FASTVIDEO_GOLDEN_TEST_FILES selection" >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
IFS=, read -r -a golden_files <<< "$selected"
|
||||
golden_paths=()
|
||||
for golden_file in "${golden_files[@]}"; do
|
||||
golden_path="$golden_root/$golden_file"
|
||||
[ -f "$golden_path" ] || {
|
||||
echo "Selected golden test does not exist: $golden_file" >&2
|
||||
exit 2
|
||||
}
|
||||
golden_paths+=("$golden_path")
|
||||
done
|
||||
|
||||
exec pytest "${golden_paths[@]}" -xvs
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the LoRA-inference lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/inference/lora/test_lora_inference_similarity.py -vs
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the VMoBA-inference lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec python fastvideo/tests/inference/vmoba/test_vmoba_inference.py
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the custom-kernel lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest fastvideo-kernel/tests/ -vs
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the LoRA-extraction lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/lora_extraction/test_lora_extraction.py -vs
|
||||
@@ -1,58 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm performance lane. Reports are written outside the checkout
|
||||
# so the trusted host driver can upload them after untrusted code exits.
|
||||
set -uo pipefail
|
||||
|
||||
export PERFORMANCE_TRACKING_ROOT=/tmp/perf-tracking
|
||||
export PERF_REPORTS_DIR=/workspace/artifacts/performance
|
||||
mkdir -p "$PERF_REPORTS_DIR"
|
||||
|
||||
if [[ ${BUILDKITE_PULL_REQUEST:-false} =~ ^[1-9][0-9]*$ ]]; then
|
||||
export PERF_RUN_SOURCE=pr
|
||||
export PERF_UPLOAD_POLICY=pass
|
||||
elif [ "${BUILDKITE_BRANCH:-}" = main ] \
|
||||
&& { [ "${BUILDKITE_SOURCE:-}" = schedule ] || [ "${TEST_SCOPE:-}" = full ]; }; then
|
||||
export PERF_RUN_SOURCE=scheduled_main
|
||||
export PERF_UPLOAD_POLICY=always
|
||||
elif [ "${TEST_SCOPE:-}" = direct ]; then
|
||||
export PERF_RUN_SOURCE=unknown
|
||||
export PERF_UPLOAD_POLICY=pass
|
||||
else
|
||||
export PERF_RUN_SOURCE=unknown
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
# Alternate GPU backends compare against references without publishing records.
|
||||
# Their worker has read-only Hub credentials; publication is an operator task.
|
||||
if [ "${FASTVIDEO_CI_LOCAL_ONLY:-0}" = 1 ]; then
|
||||
export PERF_UPLOAD_POLICY=never
|
||||
fi
|
||||
|
||||
nvidia-smi \
|
||||
--query-gpu=index,timestamp,clocks.sm,clocks.max.sm,power.draw,power.limit,temperature.gpu \
|
||||
--format=csv -l 10 > "$PERF_REPORTS_DIR/gpu_telemetry.csv" 2>/dev/null &
|
||||
telemetry_pid=$!
|
||||
cleanup() {
|
||||
kill "$telemetry_pid" 2>/dev/null || true
|
||||
wait "$telemetry_pid" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
pytest ./fastvideo/tests/performance -vs
|
||||
pytest_rc=$?
|
||||
compare_rc=0
|
||||
if [ "$pytest_rc" -eq 0 ] || [ "$PERF_UPLOAD_POLICY" = always ]; then
|
||||
PERF_PYTEST_RC=$pytest_rc python ./fastvideo/tests/performance/compare_baseline.py
|
||||
compare_rc=$?
|
||||
fi
|
||||
python ./fastvideo/tests/performance/dashboard.py || true
|
||||
cp -f fastvideo/tests/performance/results/*.json "$PERF_REPORTS_DIR/" 2>/dev/null || true
|
||||
|
||||
echo "--- GPU telemetry (clocks.sm vs clocks.max.sm reveals capped hosts) ---"
|
||||
cat "$PERF_REPORTS_DIR/gpu_telemetry.csv" || true
|
||||
|
||||
final_rc=$pytest_rc
|
||||
if [ "$final_rc" -eq 0 ]; then
|
||||
final_rc=$compare_rc
|
||||
fi
|
||||
exit "$final_rc"
|
||||
@@ -1,6 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the self-forcing lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/self-forcing/test_self_forcing.py -vs
|
||||
@@ -1,40 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical four-GPU SSIM lane for the Slinky Slurm worker.
|
||||
set -euo pipefail
|
||||
|
||||
args=()
|
||||
if [ "${FASTVIDEO_SSIM_BOOTSTRAP_MODE:-0}" = 1 ]; then
|
||||
args+=(--bootstrap-mode)
|
||||
fi
|
||||
selected=${FASTVIDEO_SSIM_TEST_FILES-}
|
||||
if [ -z "$selected" ]; then
|
||||
if [ "${TEST_SCOPE:-}" = merge ]; then
|
||||
echo "Missing FASTVIDEO_SSIM_TEST_FILES for merge scope" >&2
|
||||
exit 2
|
||||
fi
|
||||
selected=all
|
||||
fi
|
||||
if [ "$selected" != all ]; then
|
||||
[[ $selected =~ ^test_[a-z0-9_]+\.py(,test_[a-z0-9_]+\.py)*$ ]] || {
|
||||
echo "Invalid FASTVIDEO_SSIM_TEST_FILES selection" >&2
|
||||
exit 2
|
||||
}
|
||||
IFS=, read -r -a ssim_files <<< "$selected"
|
||||
for ssim_file in "${ssim_files[@]}"; do
|
||||
args+=(--test-file "$ssim_file")
|
||||
done
|
||||
fi
|
||||
|
||||
# MoGe's utils3d dependency builds glcontext from source on ARM64. The current
|
||||
# runner image predates the baked-in X11 headers below, so keep this guarded
|
||||
# bootstrap until every deployed image digest contains libx11-dev.
|
||||
if [ ! -f /usr/include/X11/Xlib.h ]; then
|
||||
apt-get -o Acquire::Retries=5 update
|
||||
apt-get -o Acquire::Retries=5 install -y --no-install-recommends libx11-dev
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
fi
|
||||
|
||||
uv pip install git+https://github.com/microsoft/MoGe.git
|
||||
uv pip install k_diffusion einops_exts alias_free_torch torchsde
|
||||
|
||||
exec python fastvideo/tests/ssim/ci_runner.py "${args[@]}"
|
||||
@@ -1,5 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the modular training-framework lane.
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest ./fastvideo/tests/train/models ./fastvideo/tests/train/methods -vs
|
||||
@@ -1,6 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the legacy vanilla-training lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/Vanilla -srP
|
||||
@@ -1,6 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the legacy LoRA-training lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/lora/test_lora_training.py -srP
|
||||
@@ -1,6 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the legacy VSA-training lane.
|
||||
set -euo pipefail
|
||||
|
||||
export WANDB_MODE=offline
|
||||
exec pytest ./fastvideo/tests/training/VSA -srP
|
||||
@@ -1,9 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the transformer lane.
|
||||
set -euo pipefail
|
||||
|
||||
# The existing block reference records an absent FASTVIDEO_FA4 (FA2). Keep
|
||||
# that reference identity; the component lane also selects FA2 explicitly.
|
||||
env -u FASTVIDEO_FA4 pytest ./fastvideo/tests/golden_gate/test_wan_t2v.py -xvs
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_causal.py -xvs
|
||||
exec pytest ./fastvideo/tests/transformers -vs
|
||||
@@ -1,6 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
# Canonical Slurm CI selection for the VAE lane.
|
||||
set -euo pipefail
|
||||
|
||||
pytest ./fastvideo/tests/golden_gate/test_wan_vae.py -xvs
|
||||
exec pytest ./fastvideo/tests/vaes -vs
|
||||
@@ -1,19 +1,6 @@
|
||||
#!/bin/bash
|
||||
set -uo pipefail
|
||||
|
||||
# DORMANT ROLLBACK ONLY. Active CI is Slurm-only and pipeline.yml never calls
|
||||
# this launcher. Refuse every Buildkite invocation even if a stale step or
|
||||
# operator typo reaches this file; local rollback experiments require an
|
||||
# explicit opt-in.
|
||||
if [ -n "${BUILDKITE:-}" ]; then
|
||||
echo "Legacy Modal CI is disabled; use the Slinky Slurm runner." >&2
|
||||
exit 2
|
||||
fi
|
||||
if [ "${FASTVIDEO_ENABLE_LEGACY_MODAL_CI:-0}" != 1 ]; then
|
||||
echo "Legacy Modal CI is dormant. Set FASTVIDEO_ENABLE_LEGACY_MODAL_CI=1 only for a manual rollback test." >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
log() {
|
||||
echo "[$(date '+%Y-%m-%d %H:%M:%S')] $1"
|
||||
}
|
||||
@@ -200,10 +187,6 @@ case "$TEST_TYPE" in
|
||||
log "Running transformer tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_transformer_tests"
|
||||
;;
|
||||
"golden_gate")
|
||||
log "Running golden-gate tests..."
|
||||
MODAL_COMMAND="$MODAL_ENV HF_API_KEY=$HF_API_KEY python3 -m modal run $MODAL_TEST_FILE::run_golden_gate_tests"
|
||||
;;
|
||||
"ssim")
|
||||
log "Running SSIM tests..."
|
||||
SSIM_BOOTSTRAP_ARGS=$(ssim_bootstrap_args)
|
||||
|
||||
@@ -1,26 +0,0 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
|
||||
exec pytest \
|
||||
./fastvideo/tests/api/ \
|
||||
./fastvideo/tests/contract/ \
|
||||
./fastvideo/tests/dataset/ \
|
||||
./fastvideo/tests/workflow/ \
|
||||
./fastvideo/tests/entrypoints/ \
|
||||
./fastvideo/tests/loader/ \
|
||||
./fastvideo/tests/pipelines/ \
|
||||
./fastvideo/tests/platforms/ \
|
||||
./fastvideo/tests/train/ \
|
||||
./fastvideo/tests/stages/ \
|
||||
./fastvideo/tests/ops/ \
|
||||
./fastvideo/tests/worker/ \
|
||||
./fastvideo/tests/training/test_trackers.py \
|
||||
./fastvideo/tests/attention/test_sdpa_metadata_mask_contract.py \
|
||||
./fastvideo/tests/attention/test_vsa_h3_tile_grad_safety.py \
|
||||
./fastvideo/tests/modal/test_kernel_build_cache.py \
|
||||
./fastvideo/tests/modal/test_pr_test.py \
|
||||
./fastvideo/tests/modal/test_ssim_test.py \
|
||||
--ignore=./fastvideo/tests/entrypoints/test_openai_api_integration.py \
|
||||
--ignore=./fastvideo/tests/train/models \
|
||||
--ignore=./fastvideo/tests/train/methods \
|
||||
-vs
|
||||
@@ -8,10 +8,10 @@ PR TITLE: Must start with a type tag, e.g.:
|
||||
MERGE WORKFLOW:
|
||||
1. Ensure pre-commit passes and you have at least 1 approval
|
||||
2. Comment /merge (or add the "ready" label) to enter the Merge Queue
|
||||
3. A path-aware merge gate runs only relevant integration tests → auto-merge on success
|
||||
3. Full Test Suite runs automatically on a staging branch → auto-merge on success
|
||||
|
||||
ON-DEMAND TESTING (write access required):
|
||||
/test full — Explicit all-lane run /test ssim — Full SSIM regression
|
||||
/test full — Full Test Suite /test ssim — SSIM regression
|
||||
/test training — Training pipeline /test encoder — Encoder tests
|
||||
/test transformer — Transformer tests /test vae — VAE tests
|
||||
/test kernel — CUDA kernel tests /test unit — Unit tests
|
||||
|
||||
@@ -1,14 +1,14 @@
|
||||
#!/usr/bin/env bash
|
||||
# Gate the path-aware Buildkite merge plan on the cheap GitHub checks.
|
||||
# Gate the expensive Buildkite full suite on the cheap GitHub checks.
|
||||
#
|
||||
# Polls the workflow runs for the PR head commit and only exits 0 once the
|
||||
# watched cheap workflows (pre-commit, docs build) have succeeded, so the
|
||||
# 'ready' label cannot burn path-selected GPU lanes on a head that a cheap
|
||||
# check has already doomed.
|
||||
# 'ready' label cannot burn ~20 GPU lanes on a head that a cheap check has
|
||||
# already doomed.
|
||||
#
|
||||
# Semantics:
|
||||
# - watched run completed with a bad conclusion -> exit 1 (fail CLOSED:
|
||||
# no merge gate; the next push re-arms via the 'synchronize' trigger)
|
||||
# no full suite; the next push re-arms via the 'synchronize' trigger)
|
||||
# - watched run cancelled -> still pending: the docs
|
||||
# workflow's repo-global 'pages' concurrency group cancels runs superseded
|
||||
# by unrelated pushes, so 'cancelled' is not a verdict on this PR
|
||||
@@ -29,7 +29,7 @@ set -euo pipefail
|
||||
: "${PR_NUMBER:?PR_NUMBER (pull request number) is required}"
|
||||
: "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY is required}"
|
||||
|
||||
# Workflow-level `name:` values that must be green before the merge gate
|
||||
# Workflow-level `name:` values that must be green before the full suite
|
||||
# may start. "Deploy Documentation" is path-filtered on PRs, so its run may
|
||||
# legitimately never exist; pre-commit always runs, so it must appear.
|
||||
WATCHED_NAMES='["pre-commit", "Deploy Documentation"]'
|
||||
@@ -56,7 +56,7 @@ recheck_ready_label() {
|
||||
if pr_json=$(gh_api "repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}" 2>/dev/null); then
|
||||
if ! jq -e '[.labels[]?.name] | index("ready")' <<<"$pr_json" >/dev/null 2>&1; then
|
||||
echo "::error::PR #${PR_NUMBER} no longer has the 'ready' label —" \
|
||||
"NOT triggering the Buildkite merge gate. Re-add the label to re-arm."
|
||||
"NOT triggering the Buildkite full suite. Re-add the label to re-arm."
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
@@ -84,7 +84,7 @@ while true; do
|
||||
| map(.name) | join(", ")' <<<"$state")
|
||||
if [ -n "$failed" ]; then
|
||||
echo "::error::Cheap check(s) failed on ${PR_SHA}: ${failed}." \
|
||||
"NOT triggering the Buildkite merge gate. Push a fix (the 'ready'" \
|
||||
"NOT triggering the Buildkite full suite. Push a fix (the 'ready'" \
|
||||
"label re-arms on every push), or re-run the failed check and then" \
|
||||
"re-run this workflow."
|
||||
exit 1
|
||||
@@ -97,7 +97,7 @@ while true; do
|
||||
if [ "$pending" -eq 0 ]; then
|
||||
if [ -z "$missing" ]; then
|
||||
recheck_ready_label
|
||||
echo "All watched cheap checks are green — merge gate may proceed."
|
||||
echo "All watched cheap checks are green — full suite may proceed."
|
||||
exit 0
|
||||
fi
|
||||
case "$missing" in
|
||||
@@ -119,14 +119,14 @@ while true; do
|
||||
echo "::warning::GitHub API error querying workflow runs for ${PR_SHA} (attempt ${api_fails}/3)."
|
||||
if [ "$api_fails" -ge 3 ]; then
|
||||
recheck_ready_label
|
||||
echo "::warning::FAILING OPEN: cannot query GitHub check status — triggering the merge gate WITHOUT the cheap-check gate."
|
||||
echo "::warning::FAILING OPEN: cannot query GitHub check status — triggering the full suite WITHOUT the cheap-check gate."
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ "$elapsed" -ge "$MAX_WAIT_SECS" ]; then
|
||||
recheck_ready_label
|
||||
echo "::warning::FAILING OPEN: watched checks still pending after $(( MAX_WAIT_SECS / 60 )) min${missing:+ (never appeared: ${missing})} — triggering the merge gate anyway."
|
||||
echo "::warning::FAILING OPEN: watched checks still pending after $(( MAX_WAIT_SECS / 60 )) min${missing:+ (never appeared: ${missing})} — triggering the full suite anyway."
|
||||
exit 0
|
||||
fi
|
||||
sleep "$POLL_SECS"
|
||||
|
||||
@@ -1,582 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Select the additive GPU integration lanes needed by a PR diff.
|
||||
|
||||
Fastcheck is the universal six-lane baseline and is intentionally not repeated
|
||||
here. This planner selects only the more expensive merge-gate lanes. Unknown
|
||||
source/build paths fail closed to the complete integration set, while explicit
|
||||
documentation and repository-metadata paths require no additional GPU work.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import fnmatch
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import TextIO
|
||||
|
||||
MERGE_LANES = (
|
||||
"golden-gate",
|
||||
"ssim",
|
||||
"lora-inference",
|
||||
"lora-extraction",
|
||||
"training",
|
||||
"distillation",
|
||||
"self-forcing",
|
||||
"lora-training",
|
||||
"training-vsa",
|
||||
"inference-vmoba",
|
||||
"performance",
|
||||
"api-server",
|
||||
"train-framework",
|
||||
"eval",
|
||||
)
|
||||
|
||||
LANE_SCRIPT_TO_KEY = {
|
||||
"api_server.sh": "api-server",
|
||||
"distillation_dmd.sh": "distillation",
|
||||
"eval.sh": "eval",
|
||||
"golden_gate.sh": "golden-gate",
|
||||
"inference_lora.sh": "lora-inference",
|
||||
"inference_vmoba.sh": "inference-vmoba",
|
||||
"lora_extraction.sh": "lora-extraction",
|
||||
"performance.sh": "performance",
|
||||
"self_forcing.sh": "self-forcing",
|
||||
"ssim.sh": "ssim",
|
||||
"train_framework.sh": "train-framework",
|
||||
"training.sh": "training",
|
||||
"training_lora.sh": "lora-training",
|
||||
"training_vsa.sh": "training-vsa",
|
||||
}
|
||||
|
||||
FASTCHECK_LANE_SCRIPTS = {
|
||||
"dreamverse.sh",
|
||||
"encoder.sh",
|
||||
"kernel_tests.sh",
|
||||
"transformer.sh",
|
||||
"vae.sh",
|
||||
}
|
||||
|
||||
LEGACY_TRAINING_LANES = (
|
||||
"training",
|
||||
"distillation",
|
||||
"self-forcing",
|
||||
"lora-training",
|
||||
"training-vsa",
|
||||
)
|
||||
|
||||
ALL_TRAINING_LANES = (*LEGACY_TRAINING_LANES, "train-framework")
|
||||
|
||||
SSIM_SMOKE_TESTS = (
|
||||
"test_flux_t2i_similarity.py",
|
||||
"test_wan_t2v_similarity.py",
|
||||
)
|
||||
|
||||
SAFE_PATTERNS = (
|
||||
"*.md",
|
||||
"*.rst",
|
||||
".agents/**",
|
||||
".claude/**",
|
||||
".codex/**",
|
||||
".github/ISSUE_TEMPLATE/**",
|
||||
".github/PULL_REQUEST_TEMPLATE.md",
|
||||
".github/dependabot.yml",
|
||||
".github/mergify.yml",
|
||||
".github/scripts/**",
|
||||
".github/workflows/**",
|
||||
".buildkite/scripts/pre_commit.sh",
|
||||
".git-blame-ignore-revs",
|
||||
".gitattributes",
|
||||
".gitignore",
|
||||
".pre-commit-config.yaml",
|
||||
"AGENTS.md",
|
||||
"CITATION.cff",
|
||||
"CODE_OF_CONDUCT.md",
|
||||
"CONTRIBUTING.md",
|
||||
"LICENSE",
|
||||
"NOTICE",
|
||||
"__init__.py",
|
||||
"collect_env.py",
|
||||
"SECURITY.md",
|
||||
"assets/**",
|
||||
"comfyui/**",
|
||||
"docs/**",
|
||||
"examples/**",
|
||||
"mkdocs.yml",
|
||||
"requirements-mkdocs.in",
|
||||
"requirements-mkdocs.txt",
|
||||
"scripts/**",
|
||||
"tests/__init__.py",
|
||||
"tests/local_tests/**",
|
||||
)
|
||||
|
||||
ALL_IMPACT_PATTERNS = (
|
||||
".buildkite/pipeline.yml",
|
||||
"docker/**",
|
||||
"pyproject.toml",
|
||||
"requirements*.txt",
|
||||
"setup.cfg",
|
||||
"setup.py",
|
||||
"uv.lock",
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class FamilyCoverage:
|
||||
pattern: re.Pattern[str]
|
||||
golden_tests: tuple[str, ...]
|
||||
ssim_tests: tuple[str, ...]
|
||||
|
||||
|
||||
FAMILY_COVERAGE = (
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])dreamx(_world)?([/_.-]|$)"),
|
||||
("test_dreamx.py", ),
|
||||
("test_dreamx_world_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])flux[_-]?2([/_.-]|$)"),
|
||||
("test_flux2_klein.py", ),
|
||||
("test_flux2_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])flux(?![_-]?2)([/_.-]|$)"),
|
||||
("test_flux.py", ),
|
||||
("test_flux_t2i_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])(hunyuan)?gamecraft([/_.-]|$)"),
|
||||
("test_gamecraft.py", ),
|
||||
("test_gamecraft_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])gen3c([/_.-]|$)"),
|
||||
("test_gen3c.py", ),
|
||||
("test_gen3c_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])glm[_-]?image([/_.-]|$)"),
|
||||
("test_glm_image.py", ),
|
||||
("test_glm_image_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])kandinsky[_-]?5([/_.-]|$)"),
|
||||
("test_kandinsky5.py", ),
|
||||
("test_kandinsky5_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])lingbot([a-z0-9_-]*)([/_.-]|$)"),
|
||||
("test_lingbot.py", ),
|
||||
("test_lingbot_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])longcat([/_.-]|$)"),
|
||||
("test_longcat.py", ),
|
||||
("test_longcat_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])ltx[_-]?2([/_.-]|$)"),
|
||||
("test_ltx2.py", ),
|
||||
("test_ltx2_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])matrixgame[_-]?2([/_.-]|$)"),
|
||||
("test_matrixgame.py", ),
|
||||
("test_matrixgame2_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])matrixgame[_-]?3([/_.-]|$)"),
|
||||
("test_matrixgame.py", ),
|
||||
("test_matrixgame3_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])minimax[_-]?h3([/_.-]|$)"),
|
||||
("test_minimax_h3_t2v.py", ),
|
||||
("test_minimax_h3_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])sd[_-]?3([._-]?5)?([/_.-]|$)"),
|
||||
("test_sd35.py", ),
|
||||
("test_sd35_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])stable[_-]?audio([/_.-]|$)"),
|
||||
("test_stable_audio.py", ),
|
||||
("test_stable_audio_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])turbo(diffusion)?([/_.-]|$)"),
|
||||
(),
|
||||
("test_turbodiffusion_similarity.py", ),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)"),
|
||||
("test_wan_t2v.py", "test_wan_vae.py", "test_wan_causal.py", "test_wan_denoising.py"),
|
||||
(
|
||||
"test_causal_similarity.py",
|
||||
"test_wan_i2v_similarity.py",
|
||||
"test_wan_t2v_similarity.py",
|
||||
),
|
||||
),
|
||||
FamilyCoverage(
|
||||
re.compile(r"(^|[/_.-])z[_-]?image([/_.-]|$)"),
|
||||
("test_zimage.py", ),
|
||||
("test_zimage_similarity.py", ),
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class MergePlan:
|
||||
lanes: set[str] = field(default_factory=set)
|
||||
golden_tests: set[str] = field(default_factory=set)
|
||||
ssim_tests: set[str] = field(default_factory=set)
|
||||
golden_all: bool = False
|
||||
ssim_all: bool = False
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
|
||||
def add_lanes(self, *lanes: str, reason: str) -> None:
|
||||
unknown = set(lanes) - set(MERGE_LANES)
|
||||
if unknown:
|
||||
raise ValueError(f"Unknown merge lanes: {sorted(unknown)}")
|
||||
self.lanes.update(lanes)
|
||||
self.reasons.append(reason)
|
||||
|
||||
def add_golden(self, tests: tuple[str, ...], reason: str) -> None:
|
||||
self.add_lanes("golden-gate", reason=reason)
|
||||
self.golden_tests.update(tests)
|
||||
|
||||
def add_ssim(self, tests: tuple[str, ...], reason: str) -> None:
|
||||
self.add_lanes("ssim", reason=reason)
|
||||
self.ssim_tests.update(tests)
|
||||
|
||||
def require_all(self, reason: str) -> None:
|
||||
self.lanes.update(MERGE_LANES)
|
||||
self.golden_all = True
|
||||
self.ssim_all = True
|
||||
self.reasons.append(reason)
|
||||
|
||||
def ordered_lanes(self) -> tuple[str, ...]:
|
||||
return tuple(lane for lane in MERGE_LANES if lane in self.lanes)
|
||||
|
||||
def encoded_lanes(self) -> str:
|
||||
lanes = self.ordered_lanes()
|
||||
return "," + ",".join(lanes or ("none", )) + ","
|
||||
|
||||
def encoded_golden_tests(self) -> str:
|
||||
if "golden-gate" not in self.lanes:
|
||||
return "none"
|
||||
if self.golden_all or not self.golden_tests:
|
||||
return "all"
|
||||
return ",".join(sorted(self.golden_tests))
|
||||
|
||||
def encoded_ssim_tests(self) -> str:
|
||||
if "ssim" not in self.lanes:
|
||||
return "none"
|
||||
if self.ssim_all or not self.ssim_tests:
|
||||
return "all"
|
||||
return ",".join(sorted(self.ssim_tests))
|
||||
|
||||
|
||||
def _matches_any(path: str, patterns: tuple[str, ...]) -> bool:
|
||||
return any(fnmatch.fnmatchcase(path, pattern) for pattern in patterns)
|
||||
|
||||
|
||||
def _family_coverage(path: str) -> tuple[set[str], set[str]]:
|
||||
normalized = path.lower()
|
||||
golden: set[str] = set()
|
||||
ssim: set[str] = set()
|
||||
for family in FAMILY_COVERAGE:
|
||||
if family.pattern.search(normalized):
|
||||
golden.update(family.golden_tests)
|
||||
ssim.update(family.ssim_tests)
|
||||
# Select the component actually touched, including compatibility paths.
|
||||
# Family configs/pipeline wiring can affect all four Wan gates.
|
||||
if re.search(r"(^|[/_.-])wan(video|vae)?([/_.-]|$)", normalized):
|
||||
if (normalized.endswith(("/wan/vae.py", "/wan/vae_config.py", "/vaes/wanvae.py"))
|
||||
or normalized.endswith("/wan/stages/conditioning.py")):
|
||||
golden = {"test_wan_vae.py"}
|
||||
elif normalized.endswith(("/wan/causal_transformer.py", "/dits/causal_wanvideo.py",
|
||||
"/wan/stages/causal_denoising.py")):
|
||||
golden = {"test_wan_causal.py"}
|
||||
elif (normalized == "fastvideo/models/dits/wanvideo.py"
|
||||
or normalized.endswith(("/wan/transformer.py", "/wan/stages/denoising.py", "/wan/stages/dmd.py"))):
|
||||
golden = {"test_wan_t2v.py", "test_wan_denoising.py"}
|
||||
return golden, ssim
|
||||
|
||||
|
||||
def _select_output_coverage(plan: MergePlan, path: str) -> None:
|
||||
golden, ssim = _family_coverage(path)
|
||||
if golden:
|
||||
plan.add_golden(tuple(sorted(golden)), reason=f"model-family golden coverage: {path}")
|
||||
else:
|
||||
plan.golden_all = True
|
||||
plan.add_lanes("golden-gate", reason=f"shared output golden coverage: {path}")
|
||||
if ssim:
|
||||
plan.add_ssim(tuple(sorted(ssim)), reason=f"model-family SSIM coverage: {path}")
|
||||
else:
|
||||
plan.add_ssim(SSIM_SMOKE_TESTS, reason=f"shared output SSIM smoke coverage: {path}")
|
||||
|
||||
|
||||
def classify_paths(paths: list[str]) -> MergePlan:
|
||||
plan = MergePlan()
|
||||
normalized_paths: list[str] = []
|
||||
for raw_path in paths:
|
||||
path = raw_path.strip()
|
||||
while path.startswith("./"):
|
||||
path = path[2:]
|
||||
if path:
|
||||
normalized_paths.append(path)
|
||||
normalized_paths = sorted(set(normalized_paths))
|
||||
if not normalized_paths:
|
||||
plan.require_all("changed-file list was empty; failing closed")
|
||||
return plan
|
||||
|
||||
for path in normalized_paths:
|
||||
if path == "__FASTVIDEO_CI_PLAN_ALL__":
|
||||
plan.require_all("changed-file API failed; failing closed")
|
||||
continue
|
||||
|
||||
if path in {"requirements-mkdocs.in", "requirements-mkdocs.txt"}:
|
||||
plan.reasons.append(f"documentation dependencies need no GPU integration: {path}")
|
||||
continue
|
||||
|
||||
if _matches_any(path, ALL_IMPACT_PATTERNS):
|
||||
plan.require_all(f"cross-cutting build/runtime surface: {path}")
|
||||
continue
|
||||
|
||||
lane_script_prefix = ".buildkite/scripts/lanes/"
|
||||
if path.startswith(lane_script_prefix):
|
||||
script_name = Path(path).name
|
||||
lane = LANE_SCRIPT_TO_KEY.get(script_name)
|
||||
if lane is None:
|
||||
if script_name in FASTCHECK_LANE_SCRIPTS:
|
||||
plan.reasons.append(f"covered by automatic Fastcheck lane: {path}")
|
||||
else:
|
||||
plan.require_all(f"unknown lane script: {path}")
|
||||
elif lane == "golden-gate":
|
||||
plan.golden_all = True
|
||||
plan.add_lanes(lane, reason=f"golden lane implementation: {path}")
|
||||
elif lane == "ssim":
|
||||
plan.ssim_all = True
|
||||
plan.add_lanes(lane, reason=f"SSIM lane implementation: {path}")
|
||||
else:
|
||||
plan.add_lanes(lane, reason=f"lane implementation: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/golden_gate/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
plan.add_golden((name, ), reason=f"changed golden test: {path}")
|
||||
elif name in {"AGENTS.md", "README.md"}:
|
||||
plan.reasons.append(f"golden documentation only: {path}")
|
||||
else:
|
||||
plan.golden_all = True
|
||||
plan.add_lanes("golden-gate", reason=f"shared golden harness/reference: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/ssim/"):
|
||||
name = Path(path).name
|
||||
if name.startswith("test_") and name.endswith(".py"):
|
||||
plan.add_ssim((name, ), reason=f"changed SSIM test: {path}")
|
||||
elif path.endswith((".py", ".json", ".pt", ".png", ".mp4")):
|
||||
plan.ssim_all = True
|
||||
plan.add_lanes("ssim", reason=f"shared SSIM harness/reference: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/performance/") or path.startswith(".buildkite/performance-benchmarks/"):
|
||||
plan.add_lanes("performance", reason=f"performance coverage: {path}")
|
||||
continue
|
||||
if path.startswith(("fastvideo/performance/", "fastvideo/performance_dashboard/",
|
||||
"apps/performance_dashboard/")):
|
||||
plan.add_lanes("performance", reason=f"performance implementation: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/benchmarks/"):
|
||||
if "/mlx_" in path or Path(path).name.startswith("mlx_"):
|
||||
plan.reasons.append(f"covered by the path-filtered macOS MLX workflow: {path}")
|
||||
else:
|
||||
plan.add_lanes("performance", reason=f"benchmark implementation: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/eval/") or path.startswith("fastvideo/eval/"):
|
||||
plan.add_lanes("eval", reason=f"evaluation coverage: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/third_party/eval/"):
|
||||
plan.add_lanes("eval", reason=f"vendored evaluation implementation: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/lora_extraction/") or path.startswith("scripts/lora_extraction/"):
|
||||
plan.add_lanes("lora-extraction", reason=f"LoRA extraction coverage: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/inference/lora/"):
|
||||
plan.add_lanes("lora-inference", reason=f"LoRA inference coverage: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/inference/vmoba/"):
|
||||
plan.add_lanes("inference-vmoba", reason=f"VMoBA inference coverage: {path}")
|
||||
continue
|
||||
if path.startswith(("fastvideo/dataset/", "fastvideo/workflow/", "fastvideo/pipelines/preprocess/",
|
||||
"fastvideo/pipelines/training/")):
|
||||
plan.add_lanes(*ALL_TRAINING_LANES, reason=f"shared data/training input surface: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/train/") or path.startswith("fastvideo/train/"):
|
||||
plan.add_lanes("train-framework", reason=f"modular training coverage: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/tests/training/"):
|
||||
lowered = path.lower()
|
||||
if "/vanilla/" in lowered:
|
||||
plan.add_lanes("training", reason=f"vanilla training coverage: {path}")
|
||||
elif "/distill/" in lowered:
|
||||
plan.add_lanes("distillation", reason=f"distillation coverage: {path}")
|
||||
elif "/self-forcing/" in lowered:
|
||||
plan.add_lanes("self-forcing", reason=f"self-forcing coverage: {path}")
|
||||
elif "/lora/" in lowered:
|
||||
plan.add_lanes("lora-training", reason=f"LoRA training coverage: {path}")
|
||||
elif "/vsa/" in lowered:
|
||||
plan.add_lanes("training-vsa", reason=f"VSA training coverage: {path}")
|
||||
else:
|
||||
plan.add_lanes(*LEGACY_TRAINING_LANES, reason=f"shared legacy training coverage: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/training/"):
|
||||
lowered = path.lower()
|
||||
if "self_forcing" in lowered:
|
||||
plan.add_lanes("self-forcing", reason=f"self-forcing implementation: {path}")
|
||||
elif "distill" in lowered:
|
||||
plan.add_lanes("distillation", reason=f"distillation implementation: {path}")
|
||||
elif "lora" in lowered:
|
||||
plan.add_lanes("lora-training", reason=f"LoRA training implementation: {path}")
|
||||
else:
|
||||
plan.add_lanes(*LEGACY_TRAINING_LANES, reason=f"shared legacy training implementation: {path}")
|
||||
continue
|
||||
|
||||
lowered = path.lower()
|
||||
if "vmoba" in lowered and path.startswith(("fastvideo/", ".buildkite/")):
|
||||
plan.add_lanes("inference-vmoba", reason=f"VMoBA implementation: {path}")
|
||||
plan.add_golden(("test_wan_t2v.py", ), reason=f"VMoBA end-to-end coverage: {path}")
|
||||
continue
|
||||
if "lora" in lowered and path.startswith("fastvideo/"):
|
||||
plan.add_lanes(
|
||||
"lora-inference",
|
||||
"lora-extraction",
|
||||
"lora-training",
|
||||
reason=f"shared LoRA implementation: {path}",
|
||||
)
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
|
||||
if path.startswith("fastvideo/entrypoints/") or path.startswith("fastvideo/api/"):
|
||||
plan.add_lanes("api-server", reason=f"API/entrypoint integration: {path}")
|
||||
if "openai" not in lowered and "/cli/" not in lowered:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith("fastvideo/worker/"):
|
||||
plan.add_lanes("api-server", reason=f"worker/API integration: {path}")
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith("fastvideo/distributed/"):
|
||||
plan.add_lanes(
|
||||
"training",
|
||||
"train-framework",
|
||||
reason=f"distributed runtime integration: {path}",
|
||||
)
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith(("fastvideo/hooks/", "fastvideo/platforms/", "fastvideo/third_party/")):
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith(("fastvideo/models/", "fastvideo/pipelines/", "fastvideo/configs/",
|
||||
"fastvideo/layers/", "fastvideo/attention/")):
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path in {
|
||||
"fastvideo/fastvideo_args.py",
|
||||
"fastvideo/forward_context.py",
|
||||
"fastvideo/image_processor.py",
|
||||
"fastvideo/registry.py",
|
||||
"fastvideo/utils.py",
|
||||
}:
|
||||
_select_output_coverage(plan, path)
|
||||
continue
|
||||
if path.startswith("fastvideo/mlx_runtime/"):
|
||||
plan.reasons.append(f"covered by the path-filtered macOS MLX workflow: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/logging_utils/") or path in {
|
||||
"fastvideo/__init__.py",
|
||||
"fastvideo/envs.py",
|
||||
"fastvideo/logger.py",
|
||||
"fastvideo/profiler.py",
|
||||
"fastvideo/version.py",
|
||||
}:
|
||||
plan.reasons.append(f"covered by automatic Fastcheck: {path}")
|
||||
continue
|
||||
if path.startswith(("fastvideo-kernel/", "csrc/")):
|
||||
plan.add_golden(("test_wan_t2v.py", ), reason=f"kernel integration smoke: {path}")
|
||||
plan.add_ssim(("test_wan_t2v_similarity.py", ), reason=f"kernel numerical smoke: {path}")
|
||||
continue
|
||||
|
||||
if path.startswith("apps/dreamverse/"):
|
||||
# DreamVerse is already one of the six automatic Fastcheck lanes.
|
||||
plan.reasons.append(f"covered by automatic DreamVerse Fastcheck: {path}")
|
||||
continue
|
||||
if path.startswith("fastvideo/tests/"):
|
||||
# The automatic unit/component Fastcheck lanes own the remaining
|
||||
# package tests. Domain-specific expensive test roots were handled
|
||||
# above.
|
||||
plan.reasons.append(f"covered by automatic Fastcheck: {path}")
|
||||
continue
|
||||
if path in {".buildkite/scripts/unit_test.sh", ".buildkite/scripts/pr_test.sh"}:
|
||||
plan.reasons.append(f"covered by automatic unit Fastcheck: {path}")
|
||||
continue
|
||||
if _matches_any(path, SAFE_PATTERNS):
|
||||
plan.reasons.append(f"no additional GPU integration needed: {path}")
|
||||
continue
|
||||
|
||||
plan.require_all(f"unclassified path; failing closed: {path}")
|
||||
|
||||
return plan
|
||||
|
||||
|
||||
def _write_github_output(output: TextIO, plan: MergePlan) -> None:
|
||||
output.write(f"merge_test_plan={plan.encoded_lanes()}\n")
|
||||
output.write(f"merge_golden_tests={plan.encoded_golden_tests()}\n")
|
||||
output.write(f"merge_ssim_tests={plan.encoded_ssim_tests()}\n")
|
||||
output.write(f"merge_plan_label={','.join(plan.ordered_lanes()) or 'none'}\n")
|
||||
|
||||
|
||||
def _write_summary(output: TextIO, plan: MergePlan) -> None:
|
||||
output.write("## Change-aware merge test plan\n\n")
|
||||
output.write("| Selection | Value |\n|---|---|\n")
|
||||
output.write(f"| Additional Slurm lanes | `{','.join(plan.ordered_lanes()) or 'none'}` |\n")
|
||||
output.write(f"| Golden tests | `{plan.encoded_golden_tests()}` |\n")
|
||||
output.write(f"| SSIM tests | `{plan.encoded_ssim_tests()}` |\n\n")
|
||||
output.write("Fastcheck remains the universal six-lane baseline.\n")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--paths-file", type=Path, required=True)
|
||||
parser.add_argument("--github-output", type=Path)
|
||||
parser.add_argument("--summary-file", type=Path)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
paths = args.paths_file.read_text(encoding="utf-8").splitlines()
|
||||
plan = classify_paths(paths)
|
||||
print(f"MERGE_TEST_PLAN={plan.encoded_lanes()}")
|
||||
print(f"MERGE_GOLDEN_TESTS={plan.encoded_golden_tests()}")
|
||||
print(f"MERGE_SSIM_TESTS={plan.encoded_ssim_tests()}")
|
||||
for reason in plan.reasons:
|
||||
print(f"- {reason}")
|
||||
if args.github_output:
|
||||
with args.github_output.open("a", encoding="utf-8") as output:
|
||||
_write_github_output(output, plan)
|
||||
if args.summary_file:
|
||||
with args.summary_file.open("a", encoding="utf-8") as output:
|
||||
_write_summary(output, plan)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -53,7 +53,7 @@ PC_PENDING='{"name": "pre-commit", "id": 1, "status": "in_progress", "conclusion
|
||||
DOCS_OK='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "success"}'
|
||||
DOCS_BAD='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "failure"}'
|
||||
DOCS_CANCELLED='{"name": "Deploy Documentation", "id": 2, "status": "completed", "conclusion": "cancelled"}'
|
||||
OTHER='{"name": "Trigger Merge Gate", "id": 3, "status": "in_progress", "conclusion": null}'
|
||||
OTHER='{"name": "Trigger Full Suite", "id": 3, "status": "in_progress", "conclusion": null}'
|
||||
NULL_NAME='{"name": null, "id": 4, "status": "completed", "conclusion": "failure"}'
|
||||
PC_OK_RERUN='{"name": "pre-commit", "id": 5, "status": "completed", "conclusion": "success"}'
|
||||
|
||||
|
||||
@@ -190,7 +190,6 @@ jobs:
|
||||
if: ${{ !inputs.push_by_digest }}
|
||||
run: |
|
||||
echo "✅ Python ${{ inputs.python_version }} image successfully built and pushed to ${{ steps.image.outputs.name }}:${{ inputs.tag_suffix }}-sha-${GITHUB_SHA::7}"
|
||||
echo "Digest: ${{ steps.build-push.outputs.digest }}"
|
||||
echo "To run tests with this image, manually trigger the 'Run Tests' workflow."
|
||||
|
||||
- name: Digest success message
|
||||
|
||||
@@ -11,7 +11,6 @@ jobs:
|
||||
if: >-
|
||||
github.event.context == 'direct-test-completed'
|
||||
&& github.event.state == 'success'
|
||||
&& (vars.CI_GPU_BACKEND == '' || vars.CI_GPU_BACKEND == 'slurm')
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Check and update aggregate status
|
||||
@@ -27,48 +26,29 @@ jobs:
|
||||
per_page: 100,
|
||||
});
|
||||
|
||||
// Buildkite derives the GitHub context prefix from the label emoji.
|
||||
// Keep hard Full Suite lanes in test-tube/bar-chart namespaces and
|
||||
// Fastcheck lanes in microscope so targeted reruns cannot clear the
|
||||
// wrong aggregate status. Automatic PR jobs use pr-fastcheck while
|
||||
// slash-command and Full Suite jobs use ci; normalize the suffix
|
||||
// and keep the newest status for each logical lane.
|
||||
const FASTCHECK_PREFIXES = [
|
||||
'buildkite/pr-fastcheck/microscope-',
|
||||
'buildkite/ci/microscope-',
|
||||
];
|
||||
const bkStatuses = data.statuses.filter(
|
||||
s => s.context.startsWith('buildkite/ci/')
|
||||
);
|
||||
|
||||
const FASTCHECK_PREFIX = 'buildkite/ci/microscope-';
|
||||
const FULL_SUITE_PREFIXES = [
|
||||
'buildkite/ci/test-tube-',
|
||||
'buildkite/ci/bar-chart-',
|
||||
];
|
||||
|
||||
function newestByLane(prefixes) {
|
||||
const statuses = new Map();
|
||||
for (const status of data.statuses) {
|
||||
const prefix = prefixes.find(p => status.context.startsWith(p));
|
||||
if (!prefix) continue;
|
||||
const lane = status.context.slice(prefix.length);
|
||||
const previous = statuses.get(lane);
|
||||
if (!previous || Date.parse(status.updated_at) > Date.parse(previous.updated_at)) {
|
||||
statuses.set(lane, status);
|
||||
}
|
||||
}
|
||||
return statuses;
|
||||
}
|
||||
const fastcheck = bkStatuses.filter(
|
||||
s => s.context.startsWith(FASTCHECK_PREFIX)
|
||||
);
|
||||
const fullSuite = bkStatuses.filter(
|
||||
s => FULL_SUITE_PREFIXES.some(p => s.context.startsWith(p))
|
||||
);
|
||||
|
||||
const fastcheck = newestByLane(FASTCHECK_PREFIXES);
|
||||
const fullSuiteOnly = newestByLane(FULL_SUITE_PREFIXES);
|
||||
const fastcheckPassed =
|
||||
fastcheck.size === 6
|
||||
&& [...fastcheck.values()].every(s => s.state === 'success');
|
||||
const fullSuitePassed =
|
||||
fastcheckPassed
|
||||
&& fullSuiteOnly.size === 14
|
||||
&& [...fullSuiteOnly.values()].every(s => s.state === 'success');
|
||||
|
||||
if (fastcheckPassed) {
|
||||
if (
|
||||
fastcheck.length > 0
|
||||
&& fastcheck.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
`All ${fastcheck.size} fastcheck tests passed — updating fastcheck-passed`
|
||||
`All ${fastcheck.length} fastcheck tests passed — updating fastcheck-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
@@ -76,13 +56,17 @@ jobs:
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'fastcheck-passed',
|
||||
description: `All ${fastcheck.size} fastcheck tests passed`,
|
||||
description:
|
||||
`All ${fastcheck.length} fastcheck tests passed`,
|
||||
});
|
||||
}
|
||||
|
||||
if (fullSuitePassed) {
|
||||
if (
|
||||
fullSuite.length > 0
|
||||
&& fullSuite.every(s => s.state === 'success')
|
||||
) {
|
||||
core.info(
|
||||
'All 20 full suite tests passed — updating full-suite-passed'
|
||||
`All ${fullSuite.length} full suite tests passed — updating full-suite-passed`
|
||||
);
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
@@ -90,6 +74,7 @@ jobs:
|
||||
sha,
|
||||
state: 'success',
|
||||
context: 'full-suite-passed',
|
||||
description: 'All 20 full suite tests passed',
|
||||
description:
|
||||
`All ${fullSuite.length} full suite tests passed`,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,62 +0,0 @@
|
||||
name: Promote Selected GPU Backend Status
|
||||
|
||||
on:
|
||||
status:
|
||||
|
||||
permissions:
|
||||
statuses: write
|
||||
|
||||
concurrency:
|
||||
group: gpu-ci-status-${{ github.event.sha }}-${{ vars.CI_GPU_BACKEND }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
promote:
|
||||
if: >-
|
||||
(vars.CI_GPU_BACKEND == 'modal' || vars.CI_GPU_BACKEND == 'vllm')
|
||||
&& (github.event.context == format('gpu-ci/{0}/fastcheck-passed', vars.CI_GPU_BACKEND)
|
||||
|| github.event.context == format('gpu-ci/{0}/full-suite-passed', vars.CI_GPU_BACKEND))
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
SELECTED_BACKEND: ${{ vars.CI_GPU_BACKEND }}
|
||||
steps:
|
||||
- name: Mirror the selected backend's latest suite results
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
script: |
|
||||
const backend = process.env.SELECTED_BACKEND;
|
||||
if (!['modal', 'vllm'].includes(backend)) {
|
||||
throw new Error('Unsupported selected GPU backend');
|
||||
}
|
||||
const sha = context.payload.sha;
|
||||
// Read current state after entering the serialized workflow. A
|
||||
// delayed event must not overwrite a newer failure with success.
|
||||
const statuses = await github.paginate(github.rest.repos.listCommitStatusesForRef, {
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
ref: sha,
|
||||
per_page: 100,
|
||||
});
|
||||
for (const suffix of ['fastcheck-passed', 'full-suite-passed']) {
|
||||
const sourceContext = `gpu-ci/${backend}/${suffix}`;
|
||||
const matches = statuses.filter(status => status.context === sourceContext);
|
||||
matches.sort((a, b) =>
|
||||
Date.parse(b.updated_at) - Date.parse(a.updated_at) || b.id - a.id
|
||||
);
|
||||
const latest = matches[0];
|
||||
const state = latest ? latest.state : 'pending';
|
||||
if (!['pending', 'success', 'failure', 'error'].includes(state)) {
|
||||
throw new Error(`Unsupported status state for ${sourceContext}`);
|
||||
}
|
||||
await github.rest.repos.createCommitStatus({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
sha,
|
||||
context: suffix,
|
||||
state,
|
||||
description: latest
|
||||
? `${backend} ${suffix}: ${state}`
|
||||
: `Waiting for ${backend} ${suffix}`,
|
||||
...(latest && latest.target_url ? {target_url: latest.target_url} : {}),
|
||||
});
|
||||
}
|
||||
@@ -1,169 +0,0 @@
|
||||
name: macOS MLX Smoke
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths:
|
||||
- ".github/workflows/ci-macos-mlx.yml"
|
||||
- "fastvideo/mlx_runtime/**"
|
||||
- "fastvideo/tests/mlx/**"
|
||||
- "fastvideo/tests/platforms/test_mps_vsa_error.py"
|
||||
- "fastvideo/tests/platforms/test_cpu_sdpa.py"
|
||||
- "fastvideo/platforms/cpu.py"
|
||||
- "fastvideo/platforms/mps.py"
|
||||
- "fastvideo/platforms/__init__.py"
|
||||
- "fastvideo/__init__.py"
|
||||
- "examples/inference/basic/mlx_*.py"
|
||||
- "fastvideo/benchmarks/mlx_*.py"
|
||||
- "pyproject.toml"
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: macos-mlx-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
mlx-smoke:
|
||||
if: github.event_name == 'workflow_dispatch' || github.event.pull_request.draft != true
|
||||
runs-on: macos-15
|
||||
timeout-minutes: 25
|
||||
env:
|
||||
FASTVIDEO_ATTENTION_BACKEND: TORCH_SDPA
|
||||
TOKENIZERS_PARALLELISM: "false"
|
||||
MASTER_ADDR: localhost
|
||||
MASTER_PORT: "29513"
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: pip
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install lightweight MLX smoke dependencies
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.11.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru mlx \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
- name: Show Apple runtime
|
||||
run: |
|
||||
python - <<'PY'
|
||||
import platform
|
||||
import mlx.core as mx
|
||||
import torch
|
||||
|
||||
print("machine:", platform.machine())
|
||||
print("processor:", platform.processor())
|
||||
print("mlx default device:", mx.default_device())
|
||||
memory_size = mx.metal.device_info().get("memory_size") if mx.metal.is_available() else "metal unavailable"
|
||||
print("mlx memory_size:", memory_size)
|
||||
print("torch:", torch.__version__)
|
||||
print("torch mps available:", torch.backends.mps.is_available())
|
||||
PY
|
||||
|
||||
- name: Run MLX smoke tests
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
fastvideo/tests/mlx/test_mlx_dit_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_compile_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint_compat.py \
|
||||
fastvideo/tests/mlx/test_mlx_affine_dq_gemm.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa_regressions.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_mode.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_fastwan_benchmark.py \
|
||||
fastvideo/tests/mlx/test_taehv_decode.py \
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
fastvideo/tests/mlx/test_windowed_attention.py \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_download_unavailable_has_specific_error \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s -o faulthandler_timeout=120
|
||||
|
||||
# Same tests on MLX's CPU backend. Hosted macOS runners are scarce and
|
||||
# slower to schedule; this Linux job gives fast PR signal on the identical
|
||||
# graph (the parity tests were designed to be backend-agnostic), while the
|
||||
# macOS job above stays the source of truth for Metal behavior.
|
||||
mlx-smoke-linux-cpu:
|
||||
if: github.event_name == 'workflow_dispatch' || github.event.pull_request.draft != true
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
env:
|
||||
FASTVIDEO_ATTENTION_BACKEND: TORCH_SDPA
|
||||
TOKENIZERS_PARALLELISM: "false"
|
||||
MASTER_ADDR: localhost
|
||||
MASTER_PORT: "29513"
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
cache: pip
|
||||
|
||||
- uses: astral-sh/setup-uv@v3
|
||||
|
||||
- name: Install lightweight MLX smoke dependencies (CPU backend)
|
||||
run: |
|
||||
uv pip install --system \
|
||||
--index-url https://download.pytorch.org/whl/cpu \
|
||||
torch==2.11.0 torchvision torchaudio
|
||||
uv pip install --system \
|
||||
pytest numpy scipy pillow imageio einops cloudpickle filelock \
|
||||
PyYAML diffusers huggingface_hub remote-pdb safetensors loguru "mlx[cpu]" \
|
||||
"ftfy>=6.3.1" "opencv-python>=4.10.0.84" psutil "transformers>=5.0.0"
|
||||
|
||||
- name: Run MLX smoke tests (CPU backend)
|
||||
run: |
|
||||
python -m pytest \
|
||||
fastvideo/tests/mlx/test_dmd_sampling.py \
|
||||
fastvideo/tests/mlx/test_memory_limits.py \
|
||||
fastvideo/tests/mlx/test_quant_capability.py \
|
||||
fastvideo/tests/mlx/test_mlx_dit_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_compile_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint.py \
|
||||
fastvideo/tests/mlx/test_mlx_checkpoint_compat.py \
|
||||
fastvideo/tests/mlx/test_mlx_affine_dq_gemm.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_parity.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_vsa_regressions.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_mode.py \
|
||||
fastvideo/tests/mlx/test_mlx_minimax_h3_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_fastwan_benchmark.py \
|
||||
fastvideo/tests/mlx/test_taehv_decode.py \
|
||||
fastvideo/tests/mlx/test_frame_upsample.py \
|
||||
fastvideo/tests/mlx/test_mlx_fast_spatial.py \
|
||||
fastvideo/tests/mlx/test_mlx_refine.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_enhance.py \
|
||||
fastvideo/tests/mlx/test_mlx_prompt_to_video_decode.py \
|
||||
fastvideo/tests/mlx/test_mlx_wan22_prompt_cache_fingerprint.py \
|
||||
fastvideo/tests/mlx/test_wan22_sample.py \
|
||||
fastvideo/tests/mlx/test_windowed_attention.py \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_download_unavailable_has_specific_error \
|
||||
fastvideo/tests/mlx/test_mlx_rife_interpolation.py::test_rife_backend_regression_is_not_skip_eligible \
|
||||
fastvideo/tests/platforms/test_mps_vsa_error.py \
|
||||
fastvideo/tests/platforms/test_cpu_sdpa.py \
|
||||
-v -s -o faulthandler_timeout=120
|
||||
@@ -27,25 +27,14 @@ jobs:
|
||||
ref: ${{ inputs.ref || '' }}
|
||||
# For PR events, lint the PR head — but keep the hook definitions from
|
||||
# the base branch so an untrusted PR cannot alter what gets executed.
|
||||
# The gate scripts are saved too: the self-test step below executes them,
|
||||
# so it must run the base-branch copies, not the PR head's.
|
||||
- name: Save trusted hook config and gate scripts
|
||||
- name: Save trusted hook config
|
||||
if: github.event_name == 'pull_request_target'
|
||||
run: |
|
||||
cp .pre-commit-config.yaml "$RUNNER_TEMP/trusted-pre-commit-config.yaml"
|
||||
cp -a .github/scripts "$RUNNER_TEMP/trusted-scripts"
|
||||
echo "GATE_SCRIPTS_DIR=$RUNNER_TEMP/trusted-scripts" >> "$GITHUB_ENV"
|
||||
# allow-unsafe-pr-checkout acknowledges checkout's pull_request_target
|
||||
# guard: the head is data for the trusted hooks to lint; nothing from it
|
||||
# is executed (config and gate scripts are pinned to the base branch
|
||||
# above) and credentials are not persisted. SHA-pinned to v4.4.0 because
|
||||
# actionlint's action schema does not know the new input yet.
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
run: cp .pre-commit-config.yaml "$RUNNER_TEMP/trusted-pre-commit-config.yaml"
|
||||
- uses: actions/checkout@v4
|
||||
if: github.event_name == 'pull_request_target'
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
persist-credentials: false
|
||||
allow-unsafe-pr-checkout: true
|
||||
- name: Restore trusted hook config
|
||||
if: github.event_name == 'pull_request_target'
|
||||
run: cp "$RUNNER_TEMP/trusted-pre-commit-config.yaml" .pre-commit-config.yaml
|
||||
@@ -59,7 +48,5 @@ jobs:
|
||||
with:
|
||||
extra_args: --all-files --hook-stage manual
|
||||
# After pre-commit so a self-test failure cannot mask lint failures.
|
||||
# GATE_SCRIPTS_DIR points at the base-branch copy on fork PRs (set above);
|
||||
# push / workflow_call runs use the checked-out tree directly.
|
||||
- name: Full-suite gate self-test
|
||||
run: bash "${GATE_SCRIPTS_DIR:-.github/scripts}/test_gate_full_suite.sh"
|
||||
run: bash .github/scripts/test_gate_full_suite.sh
|
||||
|
||||
@@ -1,44 +0,0 @@
|
||||
name: Scheduled Full SSIM
|
||||
|
||||
on:
|
||||
schedule:
|
||||
- cron: "0 5 * * 0"
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
trigger:
|
||||
if: github.repository == 'hao-ai-lab/FastVideo'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Trigger weekly full SSIM on Slinky Slurm
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
SOURCE_SHA: ${{ github.sha }}
|
||||
SOURCE_BRANCH: ${{ github.event.repository.default_branch }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$SOURCE_SHA" \
|
||||
--arg branch "$SOURCE_BRANCH" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: "Weekly full SSIM on Slinky Slurm",
|
||||
ignore_pipeline_branch_filters: true,
|
||||
env: {
|
||||
TEST_SCOPE: "scheduled",
|
||||
FULL_SUITE: "false",
|
||||
TEST_TYPE: "ssim",
|
||||
PR_NUMBER: "false",
|
||||
PR_TITLE: "Scheduled full SSIM"
|
||||
}
|
||||
}')"
|
||||
@@ -33,6 +33,7 @@ jobs:
|
||||
core.setOutput('has_write', String(hasWrite));
|
||||
|
||||
- name: Add ready label and react
|
||||
id: label
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7.0.1
|
||||
with:
|
||||
@@ -47,6 +48,47 @@ jobs:
|
||||
comment_id: context.payload.comment.id,
|
||||
content: 'rocket',
|
||||
});
|
||||
const { data: pr } = await github.rest.pulls.get({ owner, repo, pull_number: prNumber });
|
||||
core.setOutput('pr_sha', pr.head.sha);
|
||||
core.setOutput('pr_branch', pr.head.ref);
|
||||
core.setOutput('pr_number', String(prNumber));
|
||||
core.setOutput('pr_title', pr.title);
|
||||
|
||||
- name: Trigger Full Suite
|
||||
if: steps.perm.outputs.has_write == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_SHA: ${{ steps.label.outputs.pr_sha }}
|
||||
PR_BRANCH: ${{ steps.label.outputs.pr_branch }}
|
||||
PR_NUMBER: ${{ steps.label.outputs.pr_number }}
|
||||
PR_TITLE: ${{ steps.label.outputs.pr_title }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
-H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
-H "Content-Type: application/json" \
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Full Suite for PR #${PR_NUMBER} (via /merge)" \
|
||||
--arg pr_title "$PR_TITLE" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
branch: $branch,
|
||||
message: $message,
|
||||
ignore_pipeline_branch_filters: true,
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "full",
|
||||
FULL_SUITE: "true",
|
||||
PR_NUMBER: ($pr_id | tostring),
|
||||
PR_TITLE: $pr_title
|
||||
}
|
||||
}')"
|
||||
|
||||
parse-command:
|
||||
if: >-
|
||||
@@ -87,7 +129,7 @@ jobs:
|
||||
set -euo pipefail
|
||||
TEST_NAME=$(echo "$COMMENT" | grep -oP '(?<=/test\s)\S+' | head -1 || true)
|
||||
|
||||
VALID="encoder vae transformer kernel unit dreamverse ssim golden-gate training lora-inference lora-training lora-extraction distillation self-forcing vsa vmoba performance api train-framework eval unit-ci kernel-ci dreamverse-ci ssim-ci golden-gate-ci encoder-ci vae-ci transformer-ci lora-inference-ci lora-training-ci lora-extraction-ci training-ci distillation-ci self-forcing-ci vsa-ci vmoba-ci performance-ci api-ci train-framework-ci eval-ci full fastcheck pre-commit"
|
||||
VALID="encoder vae transformer kernel unit dreamverse ssim training lora-inference lora-training lora-extraction distillation self-forcing vsa vmoba performance api train-framework eval full fastcheck pre-commit"
|
||||
if [ -z "$TEST_NAME" ] || ! echo "$VALID" | grep -qw "$TEST_NAME"; then
|
||||
echo "Unknown test: '$TEST_NAME'. Valid: $VALID"
|
||||
exit 1
|
||||
@@ -95,18 +137,8 @@ jobs:
|
||||
|
||||
declare -A MAP=(
|
||||
[encoder]=encoder [vae]=vae [transformer]=transformer
|
||||
[kernel]=kernel_tests [unit]=unit_test [unit-ci]=unit_test_ci
|
||||
[kernel-ci]=kernel_tests_ci [dreamverse-ci]=dreamverse_app_ci
|
||||
[ssim-ci]=ssim_ci [vmoba-ci]=inference_vmoba_ci
|
||||
[golden-gate-ci]=golden_gate_ci [training-ci]=training_ci
|
||||
[encoder-ci]=encoder_ci [vae-ci]=vae_ci [transformer-ci]=transformer_ci
|
||||
[lora-inference-ci]=inference_lora_ci [lora-training-ci]=training_lora_ci
|
||||
[lora-extraction-ci]=lora_extraction_ci [distillation-ci]=distillation_dmd_ci
|
||||
[self-forcing-ci]=self_forcing_ci [vsa-ci]=training_vsa_ci
|
||||
[performance-ci]=performance_ci [api-ci]=api_server_ci
|
||||
[train-framework-ci]=train_framework_ci [eval-ci]=eval_ci
|
||||
[dreamverse]=dreamverse_app
|
||||
[ssim]=ssim [golden-gate]=golden_gate [training]=training
|
||||
[kernel]=kernel_tests [unit]=unit_test [dreamverse]=dreamverse_app
|
||||
[ssim]=ssim [training]=training
|
||||
[lora-inference]=inference_lora [lora-training]=training_lora
|
||||
[lora-extraction]=lora_extraction
|
||||
[distillation]=distillation_dmd [self-forcing]=self_forcing
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
name: Trigger Merge Gate
|
||||
name: Trigger Full Suite
|
||||
|
||||
on:
|
||||
pull_request_target:
|
||||
@@ -10,7 +10,7 @@ permissions:
|
||||
actions: read
|
||||
|
||||
concurrency:
|
||||
group: merge-gate-${{ github.event.pull_request.number }}
|
||||
group: full-suite-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
@@ -34,72 +34,29 @@ jobs:
|
||||
});
|
||||
const hasReady = pr.labels.some(l => l.name === 'ready');
|
||||
core.setOutput('has_ready', String(hasReady));
|
||||
core.setOutput('changed_files', String(pr.changed_files));
|
||||
if (!hasReady) core.info('No ready label — skipping merge-gate trigger.');
|
||||
if (!hasReady) core.info('No ready label — skipping Full Suite trigger.');
|
||||
|
||||
- name: Cancel previous Buildkite builds
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
PR_BRANCH: ${{ github.event.pull_request.head.ref }}
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
run: |
|
||||
# Match both branch and PR number: forks can reuse the same branch name.
|
||||
builds=$(curl -sS --get -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
--data-urlencode "branch=$PR_BRANCH" \
|
||||
--data-urlencode "state=running,scheduled" \
|
||||
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds" \
|
||||
| jq -r --arg pr_number "$PR_NUMBER" \
|
||||
'.[] | select((.env.TEST_SCOPE? == "merge") and (.env.PR_NUMBER? == $pr_number)) | .number')
|
||||
# Find running builds for this branch with TEST_SCOPE=full and cancel them
|
||||
builds=$(curl -sS -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds?branch=${PR_BRANCH}&state=running,scheduled" \
|
||||
| jq -r '.[] | select(try (.env.TEST_SCOPE == "full") catch false) | .number')
|
||||
for build_num in $builds; do
|
||||
echo "Cancelling Buildkite build #$build_num"
|
||||
curl -sS -X PUT -H "Authorization: Bearer $BUILDKITE_API_TOKEN" \
|
||||
"https://api.buildkite.com/v2/organizations/${{ vars.BUILDKITE_ORG_SLUG }}/pipelines/${{ vars.BUILDKITE_PIPELINE_SLUG }}/builds/${build_num}/cancel"
|
||||
done
|
||||
|
||||
# Check out the immutable BASE SHA: pull_request_target must never run a
|
||||
# planner or gate script from the untrusted PR head.
|
||||
- name: Checkout trusted merge planner
|
||||
# Checks out the BASE branch (default for pull_request_target), so PR
|
||||
# authors cannot tamper with the gate script.
|
||||
- name: Checkout gate script
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.base.sha }}
|
||||
persist-credentials: false
|
||||
|
||||
- name: Collect changed paths
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
EXPECTED_CHANGED_FILES: ${{ steps.check.outputs.changed_files }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
changed_json="$RUNNER_TEMP/merge-changed-files.json"
|
||||
changed_paths="$RUNNER_TEMP/merge-changed-paths.txt"
|
||||
if gh api --paginate --slurp \
|
||||
"repos/${GITHUB_REPOSITORY}/pulls/${PR_NUMBER}/files?per_page=100" \
|
||||
> "$changed_json"; then
|
||||
observed=$(jq '[.[][] | .filename] | unique | length' "$changed_json")
|
||||
if [ "$observed" = "$EXPECTED_CHANGED_FILES" ]; then
|
||||
jq -r '.[][] | .filename, (.previous_filename // empty)' "$changed_json" \
|
||||
| sort -u > "$changed_paths"
|
||||
else
|
||||
echo "::warning::Changed-file API returned $observed of $EXPECTED_CHANGED_FILES paths; selecting all merge lanes."
|
||||
echo '__FASTVIDEO_CI_PLAN_ALL__' > "$changed_paths"
|
||||
fi
|
||||
else
|
||||
echo "::warning::Changed-file API failed; selecting all merge lanes."
|
||||
echo '__FASTVIDEO_CI_PLAN_ALL__' > "$changed_paths"
|
||||
fi
|
||||
|
||||
- name: Select minimal merge tests
|
||||
id: plan
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
run: |
|
||||
python3 .github/scripts/plan_merge_ci.py \
|
||||
--paths-file "$RUNNER_TEMP/merge-changed-paths.txt" \
|
||||
--github-output "$GITHUB_OUTPUT" \
|
||||
--summary-file "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Wait for pre-commit and docs build
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
@@ -109,7 +66,7 @@ jobs:
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
run: bash .github/scripts/gate_full_suite.sh
|
||||
|
||||
- name: Trigger Buildkite merge gate
|
||||
- name: Trigger Buildkite Full Suite
|
||||
if: steps.check.outputs.has_ready == 'true'
|
||||
env:
|
||||
BUILDKITE_API_TOKEN: ${{ secrets.BUILDKITE_API_TOKEN }}
|
||||
@@ -119,10 +76,6 @@ jobs:
|
||||
PR_TITLE: ${{ github.event.pull_request.title }}
|
||||
BK_ORG: ${{ vars.BUILDKITE_ORG_SLUG }}
|
||||
BK_PIPELINE: ${{ vars.BUILDKITE_PIPELINE_SLUG }}
|
||||
MERGE_TEST_PLAN: ${{ steps.plan.outputs.merge_test_plan }}
|
||||
MERGE_GOLDEN_TESTS: ${{ steps.plan.outputs.merge_golden_tests }}
|
||||
MERGE_SSIM_TESTS: ${{ steps.plan.outputs.merge_ssim_tests }}
|
||||
MERGE_PLAN_LABEL: ${{ steps.plan.outputs.merge_plan_label }}
|
||||
run: |
|
||||
curl -sS --fail-with-body -X POST \
|
||||
"https://api.buildkite.com/v2/organizations/${BK_ORG}/pipelines/${BK_PIPELINE}/builds" \
|
||||
@@ -131,11 +84,8 @@ jobs:
|
||||
--data-raw "$(jq -n \
|
||||
--arg commit "$PR_SHA" \
|
||||
--arg branch "$PR_BRANCH" \
|
||||
--arg message "Merge gate [${MERGE_PLAN_LABEL}] for PR #${PR_NUMBER}" \
|
||||
--arg message "Full Suite for PR #${PR_NUMBER}" \
|
||||
--arg pr_title "$PR_TITLE" \
|
||||
--arg merge_test_plan "$MERGE_TEST_PLAN" \
|
||||
--arg merge_golden_tests "$MERGE_GOLDEN_TESTS" \
|
||||
--arg merge_ssim_tests "$MERGE_SSIM_TESTS" \
|
||||
--argjson pr_id "$PR_NUMBER" \
|
||||
'{
|
||||
commit: $commit,
|
||||
@@ -145,11 +95,8 @@ jobs:
|
||||
pull_request_id: $pr_id,
|
||||
pull_request_base_branch: "main",
|
||||
env: {
|
||||
TEST_SCOPE: "merge",
|
||||
TEST_SCOPE: "full",
|
||||
FULL_SUITE: "true",
|
||||
MERGE_TEST_PLAN: $merge_test_plan,
|
||||
MERGE_GOLDEN_TESTS: $merge_golden_tests,
|
||||
MERGE_SSIM_TESTS: $merge_ssim_tests,
|
||||
PR_NUMBER: ($pr_id | tostring),
|
||||
PR_TITLE: $pr_title
|
||||
}
|
||||
|
||||
@@ -38,17 +38,17 @@ jobs:
|
||||
|
||||
**How our CI works:**
|
||||
|
||||
PRs run a three-tier CI system:
|
||||
PRs run a two-tier CI system:
|
||||
1. **Pre-commit** — formatting (yapf), linting (ruff), type checking (mypy). Runs immediately on every PR.
|
||||
2. **Fastcheck** — six core GPU lanes run automatically via Buildkite (~10-15 min).
|
||||
3. **Merge gate** — a reviewer adds `ready`; changed paths select only the relevant integration, training, golden, or SSIM coverage.
|
||||
2. **Fastcheck** — core GPU tests (encoders, VAEs, transformers, kernels, unit tests). Runs automatically via Buildkite on relevant file changes (~10-15 min).
|
||||
3. **Full Suite** — integration tests, training pipelines, SSIM regression. Runs only when a reviewer adds the `ready` label.
|
||||
|
||||
**Before your PR is reviewed:**
|
||||
- [ ] `pre-commit run --all-files` passes locally
|
||||
- [ ] You've added or updated tests for your changes
|
||||
- [ ] The PR description explains what and why
|
||||
|
||||
If pre-commit fails, a bot comment will explain how to fix it. Fastcheck and merge-gate results appear in the Checks section below.
|
||||
If pre-commit fails, a bot comment will explain how to fix it. Fastcheck and Full Suite results appear in the Checks section below.
|
||||
|
||||
**Useful links:**
|
||||
- [Contributing Guide](https://hao-ai-lab.github.io/FastVideo/contributing/overview/)
|
||||
|
||||
@@ -13,28 +13,16 @@ on:
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
build_ci_runner_image:
|
||||
description: 'Build the ARM64 CUDA 13 CI runner image (sm_100)'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
# Auto-rebuild the CUDA images when a repository-controlled image input
|
||||
# changes on main. This includes the trusted SM89 kernel artifact's source,
|
||||
# metadata/key helper, ABI dependency metadata, and build orchestration.
|
||||
# Dreamverse (apps/dreamverse/docker/Dockerfile) and the ROCm Dockerfile stay
|
||||
# manual-dispatch only.
|
||||
# Auto-rebuild the CUDA images when their Dockerfile changes on main. The CUDA
|
||||
# matrix is the only lane that builds from docker/Dockerfile, so a path-scoped
|
||||
# push trigger is a sufficient change detector on its own -- no separate
|
||||
# detect-changes/paths-filter job is needed now that there is a single
|
||||
# in-scope Dockerfile. Dreamverse (apps/dreamverse/docker/Dockerfile) and the
|
||||
# rocm Dockerfile stay manual-dispatch only.
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- '.dockerignore'
|
||||
- '.github/workflows/_template-build-image.yml'
|
||||
- '.github/workflows/infra-build-image.yml'
|
||||
- '.gitmodules'
|
||||
- 'docker/Dockerfile'
|
||||
- 'docker/uv-excludes'
|
||||
- 'fastvideo-kernel/**'
|
||||
- 'fastvideo/tests/modal/kernel_build_cache.py'
|
||||
- 'pyproject.toml'
|
||||
|
||||
|
||||
permissions:
|
||||
@@ -62,7 +50,7 @@ jobs:
|
||||
# 2.8.3 comes from the architecture-specific prebuilt releases.
|
||||
build-cuda-images:
|
||||
# Runs on a manual dispatch when build_cuda_matrix is set, or automatically
|
||||
# on an in-scope main push (inputs are null on push). The
|
||||
# on a push that changed docker/Dockerfile (inputs are null on push). The
|
||||
# repository guard keeps fork syncs from auto-building; manual dispatch
|
||||
# still works in forks.
|
||||
if: ${{ (github.event_name == 'push' && github.repository == 'hao-ai-lab/FastVideo') || github.event.inputs.build_cuda_matrix == 'true' }}
|
||||
@@ -203,29 +191,6 @@ jobs:
|
||||
docker buildx imagetools create "${TAG_ARGS[@]}" "${IMAGE_REFS[@]}"
|
||||
docker buildx imagetools inspect "${TAGS[0]}"
|
||||
|
||||
# The CI runner is ARM64 like DGX Spark, but targets sm_100a rather than sm_121.
|
||||
# The architecture-specific target includes the GB200 VSA CUDA extensions.
|
||||
# Publish a single-architecture variant so the self-hosted CI runner can reuse
|
||||
# the exact prebuilt kernel instead of compiling it in every job.
|
||||
build-ci-runner-image:
|
||||
if: ${{ (github.event_name == 'push' && github.repository == 'hao-ai-lab/FastVideo') || github.event.inputs.build_ci_runner_image == 'true' }}
|
||||
uses: ./.github/workflows/_template-build-image.yml
|
||||
with:
|
||||
python_version: '3.12'
|
||||
dockerfile_path: docker/Dockerfile
|
||||
tag_suffix: py3.12-cuda13.0.0-sm100
|
||||
runner: ubuntu-24.04-arm
|
||||
architecture: arm64
|
||||
build_args: |
|
||||
PYTHON_VERSION=3.12
|
||||
CUDA_VERSION=13.0.0
|
||||
UV_TORCH_BACKEND=cu130
|
||||
TORCH_CUDA_ARCH_LIST=10.0a
|
||||
CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
FLASH_ATTN_WHEEL_TAG=cu130torch2.12
|
||||
FLASH_ATTN_WHEEL_RELEASE_ARM64=https://github.com/mjun0812/flash-attention-prebuild-wheels/releases/download/v0.9.22
|
||||
secrets: inherit
|
||||
|
||||
# Dreamverse matrix: {backend, UI} x {12.6.3, 13.0.0}, Python 3.12. Torch backend
|
||||
# matches the base CUDA (cu126 / cu130). Keep these images amd64-only until the
|
||||
# required FA4 dependency stack is available and validated on arm64.
|
||||
|
||||
@@ -6,7 +6,6 @@ on:
|
||||
paths:
|
||||
- 'docs/**'
|
||||
- 'examples/**'
|
||||
- 'scripts/inference/**'
|
||||
- 'mkdocs.yml'
|
||||
- 'requirements-mkdocs.in'
|
||||
- 'requirements-mkdocs.txt'
|
||||
@@ -17,7 +16,6 @@ on:
|
||||
paths:
|
||||
- 'docs/**'
|
||||
- 'examples/**'
|
||||
- 'scripts/inference/**'
|
||||
- 'mkdocs.yml'
|
||||
- 'requirements-mkdocs.in'
|
||||
- 'requirements-mkdocs.txt'
|
||||
|
||||
@@ -62,9 +62,8 @@ jobs:
|
||||
cuda-version: '13.0.0'
|
||||
torch-cuda-short: 'cu130'
|
||||
platform:
|
||||
# x86_64 builds the full cu126 + cu130 set. cu130 ships the
|
||||
# data-center Blackwell sm_100a/sm_103a VSA and consumer sm_120a FP4
|
||||
# kernels.
|
||||
# x86_64 builds the full cu126 + cu130 set (cu130 ships the consumer
|
||||
# Blackwell sm_120a FP4 kernels).
|
||||
- os: ubuntu-22.04
|
||||
arch: x86_64
|
||||
wheel-plat: manylinux_2_35_x86_64
|
||||
@@ -125,7 +124,7 @@ jobs:
|
||||
- name: Install dependencies (GCC, Clang, CUDA Paths, Git)
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y git gcc-11 g++-11 clang-11
|
||||
sudo apt install -y git patchelf gcc-11 g++-11 clang-11
|
||||
sudo update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-11 100 --slave /usr/bin/g++ g++ /usr/bin/g++-11
|
||||
|
||||
# Allow Git to Access Safe Directory
|
||||
@@ -169,18 +168,17 @@ jobs:
|
||||
# covers sm_120a; turbodiffusion covers sm_100a+sm_120a. The sm_100 FP4
|
||||
# forward is the FA4 CuTe DSL path in the fastvideo package (PR #1221),
|
||||
# JIT-compiled at runtime — not built into this wheel.
|
||||
# * x86_64 cu130 = Hopper TK + data-center Blackwell sm_100a/sm_103a VSA
|
||||
# + consumer Blackwell sm_120a FP4.
|
||||
# * x86_64 cu130 = Hopper TK + consumer Blackwell sm_120a FP4.
|
||||
# * x86_64 cu126 = Hopper TK only (older drivers; CUDA < 12.8 has no FP4).
|
||||
# The per-arch split in CMakeLists pins the FP4 targets to sm_120a and builds
|
||||
# the main extension for the full arch list. CMAKE_BUILD_PARALLEL_LEVEL caps
|
||||
# Ninja so heavy CUTLASS/TK template TUs don't OOM the 16 GB runner (exit 143).
|
||||
if [ "${{ matrix.platform.arch }}" = "aarch64" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;10.3a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=OFF -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON"
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
elif [ "${{ matrix.torch-cuda.torch-cuda-short }}" = "cu130" ]; then
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;10.0a;10.3a;12.0a"
|
||||
export TORCH_CUDA_ARCH_LIST="9.0a;12.0a"
|
||||
export CMAKE_ARGS="${CMAKE_ARGS:-} -DFASTVIDEO_KERNEL_BUILD_TK=ON -DFASTVIDEO_KERNEL_BUILD_ATTN_QAT_INFER=ON -DCMAKE_CUDA_ARCHITECTURES=90a"
|
||||
# A single FP4 TU (attn_qat_infer) can use ~8-12 GB on its own, so serialize.
|
||||
export CMAKE_BUILD_PARALLEL_LEVEL=1
|
||||
@@ -196,11 +194,7 @@ jobs:
|
||||
python -m build --wheel --outdir dist
|
||||
|
||||
# Fix the wheel to be manylinux compliant
|
||||
# Ubuntu 22.04 ships patchelf 0.14.3, while current auditwheel
|
||||
# requires at least 0.14.5. Use the stable PyPI binary on both
|
||||
# x86_64 and aarch64 release runners.
|
||||
uv pip install --system auditwheel patchelf==0.17.2.4
|
||||
patchelf --version
|
||||
uv pip install --system auditwheel
|
||||
# Point auditwheel at torch libs, but do not vendor them into the wheel.
|
||||
TORCH_LIB_DIR=$(python - <<'PY'
|
||||
import os
|
||||
@@ -217,8 +211,7 @@ jobs:
|
||||
--exclude libtorch.so \
|
||||
--exclude libc10.so \
|
||||
--exclude libc10_cuda.so \
|
||||
--exclude libtorch_python.so \
|
||||
--exclude libnccl.so.2
|
||||
--exclude libtorch_python.so
|
||||
# Move fixed wheels back to dist for upload consistency
|
||||
rm dist/*.whl
|
||||
mv fixed_dist/*.whl dist/
|
||||
|
||||
@@ -6,7 +6,6 @@ results/
|
||||
wandb/
|
||||
*.ipynb
|
||||
*.jpg
|
||||
!examples/datasets/lingbotworld2/image.jpg
|
||||
*.safetensors
|
||||
*.mp4
|
||||
*.png
|
||||
@@ -23,7 +22,6 @@ Miniconda3-latest-Linux-x86_64.sh
|
||||
*validation/
|
||||
data/
|
||||
outputs/
|
||||
outputs_audio/
|
||||
outputs_video
|
||||
checkpoints/
|
||||
sbatch.sh
|
||||
@@ -36,7 +34,6 @@ env
|
||||
*.log
|
||||
weights/
|
||||
logs/
|
||||
/Z-Image/
|
||||
official_weights/
|
||||
converted_weights/
|
||||
|
||||
@@ -55,8 +52,6 @@ eggs/
|
||||
|
||||
# MkDocs documentation
|
||||
site/
|
||||
docs/assets/cookbook-serving.json
|
||||
examples/serving/clients/node_modules/
|
||||
docs/getting_started/examples/
|
||||
docs/examples/
|
||||
docs/inference/examples/
|
||||
@@ -78,7 +73,6 @@ docs/distillation/examples/
|
||||
*.pkl
|
||||
|
||||
# Reference videos (negations must come after the catch-all on line below)
|
||||
!fastvideo/tests/nightly/reference_video_*.mp4
|
||||
|
||||
# Static images
|
||||
!docs/assets/images/**/*.png
|
||||
@@ -135,9 +129,6 @@ fastvideo/tests/ssim/reference_videos/**
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.mp4
|
||||
!fastvideo/tests/ssim/reference_videos/**/*.png
|
||||
|
||||
# Local H3 MLX kernel / exactness benches (JSON, logs, frames, videos)
|
||||
.kernel_bench/
|
||||
|
||||
# Editor logs and local Python version pins (accidentally committed)
|
||||
*.nvimlog
|
||||
.nvimlog
|
||||
|
||||
@@ -9,7 +9,7 @@ exclude: |
|
||||
tests/.*|
|
||||
scripts/.*|
|
||||
fastvideo/dataset/.*|
|
||||
fastvideo/models/(?!wan/(config|vae_config|pipeline_config|definition|__init__)\.py$).*|
|
||||
fastvideo/models/.*|
|
||||
^apps/dreamverse/web/.*|
|
||||
examples/.*|
|
||||
\.agents/.*|
|
||||
@@ -22,7 +22,6 @@ repos:
|
||||
hooks:
|
||||
- id: yapf
|
||||
args: [--in-place, --verbose]
|
||||
language_version: python3.12
|
||||
additional_dependencies: [toml] # TODO: Remove when yapf is upgraded
|
||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||
rev: v0.11.12
|
||||
|
||||
@@ -66,18 +66,14 @@ Local guidance lives next to the code. Read the in-scope file before editing:
|
||||
| `fastvideo/AGENTS.md` | Core package map, public API, registry-driven model dispatch |
|
||||
| `fastvideo/configs/AGENTS.md` | Arch + pipeline config dataclasses, `param_names_mapping` |
|
||||
| `fastvideo/models/AGENTS.md` | DiT / VAE / encoder / scheduler / loader layout (pre-commit excluded) |
|
||||
| `fastvideo/models/wan/AGENTS.md` | Wan family-local transformers, VAE, configs, and the SP sharding invariant |
|
||||
| `fastvideo/layers/AGENTS.md` | Tensor-parallel linear/attention layer rules for ports |
|
||||
| `fastvideo/attention/AGENTS.md` | Backend registry + env-var override |
|
||||
| `fastvideo/pipelines/AGENTS.md` | Stage ABC, `basic/<model>/`, `preprocess/`, presets |
|
||||
| `fastvideo/pipelines/basic/wan/AGENTS.md` | Wan sampling stages, first-frame conditioning, DMD/causal boundaries |
|
||||
| `fastvideo/pipelines/basic/magi_human/AGENTS.md` | MagiHuman umbrella repo, lazy-loaded components, packing invariants |
|
||||
| `fastvideo/training/AGENTS.md` | Legacy monolithic pipelines (frozen for existing models) |
|
||||
| `fastvideo/train/AGENTS.md` | New modular trainer (methods × models × callbacks, YAML) |
|
||||
| `fastvideo/tests/AGENTS.md` | Test taxonomy, conftest, pre-commit-excluded path |
|
||||
| `fastvideo/tests/ssim/AGENTS.md` | GPU SSIM regression authoring + reference video sync |
|
||||
| `scripts/checkpoint_conversion/AGENTS.md` | Adding a converter for a new HF/official checkpoint |
|
||||
| `apps/dreamverse/AGENTS.md` | DreamVerse app structure and conventions |
|
||||
|
||||
## Critical: Two Training Stacks Coexist
|
||||
|
||||
|
||||
@@ -3,17 +3,13 @@
|
||||
</div>
|
||||
|
||||
<p align="center">
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://haoailab.com/FastVideo/cookbook/"><b>Cookbook</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
| <a href="https://hao-ai-lab.github.io/FastVideo"><b>Documentation</b></a> | <a href="https://hao-ai-lab.github.io/FastVideo/inference/inference_quick_start/"><b> Quick Start</b></a> | <a href="https://github.com/hao-ai-lab/FastVideo/discussions/982" target="_blank"><b>Weekly Dev Meeting</b></a> | 🟣💬 <a href="https://join.slack.com/t/fastvideo/shared_invite/zt-3f4lao1uq-u~Ipx6Lt4J27AlD2y~IdLQ" target="_blank"> <b>Slack</b> </a> | 🟣💬 <a href="https://github.com/hao-ai-lab/FastVideo/discussions/1097" target="_blank"> <b> WeChat </b> </a> |
|
||||
</p>
|
||||
|
||||
**FastVideo is a unified post-training and real-time inference framework for accelerated video generation.**
|
||||
|
||||
## NEWS
|
||||
- `2026/09/15`: Release [FastH3 8-Step V2](https://huggingface.co/FastVideo/FastVideo-FastH3-8-Step-V2), an eight-forward data-free DMD2 checkpoint distilled from MiniMax-H3 with 80% Video Sparse Attention. Run it with `examples/inference/basic/basic_fasth3_8step.py` or the [FastH3 8-Step V2 recipe](https://haoailab.com/FastVideo/cookbook/minimax-h3/).
|
||||
- `2026/09/01`: FastH3 now runs locally on Apple Silicon through MLX and on NVIDIA DGX Spark through CUDA 13, including two-Spark inference. Follow the [FastH3 recipes](https://haoailab.com/FastVideo/cookbook/minimax-h3/) and read the [Blog](https://haoailab.com/blogs/fasth3-local/).
|
||||
- `2026/08/27`: [FastH3 Preview v1](https://haoailab.com/blogs/fasth3-preview/) is an open-weight 4-step sparse-distilled MiniMax-H3 model for synchronized video-and-audio generation, developed in collaboration with [Nuva Lab](https://nuvalab.ai/) and the [NVIDIA FastGen team](https://github.com/NVlabs/FastGen). Download the recommended [VSA / Data-Free weights](https://huggingface.co/FastVideo/FastVideo-FastH3-4-step-Preview-v1-VSA-DataFree), or see the [full FastH3 collection](https://huggingface.co/collections/FastVideo/fastvideo-fasth3).
|
||||
- `2026/08/19`: FastVideo now supports MLX on Apple Silicon with [FastMetal-QAD](https://huggingface.co/collections/FastVideo/fastmetal), a family of 1.3B, 5B, and 14B models optimized for Mac. Follow the [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/) and read the [Blog](https://haoailab.com/blogs/fastmetal/).
|
||||
- `2026/06/23`: Release FastWan-QAD: 5s of Video generated in 1.8s E2E. See the [FastWan-QAD models](https://huggingface.co/FastVideo/FastWan-QAD-FP8-1.3B), [Attn-QAT training guide](https://haoailab.com/FastVideo/training/attn_qat/), and [blog](https://haoailab.com/blogs/fastwan-qad/).
|
||||
- `2026/06/23`: Release FastWan-QAD: 5s of Video generated in 1.8s E2E. [FastWan-QAD models](https://huggingface.co/FastVideo/FastWan-QAD-FP8-1.3B), check out the [Blog](https://haoailab.com/blogs/fastwan-qad/).
|
||||
- `2026/03/17`: Release demo: Into the Dreamverse: Vibe Directing in FastVideo, check out the [Blog](https://haoailab.com/blogs/dreamverse/).
|
||||
- `2026/03/13`: Release demo: Create a 5s 1080p Video in 4.5s with FastVideo on a Single GPU, check out the [Blog](https://haoailab.com/blogs/fastvideo_realtime_1080p/).
|
||||
- `2025/11/19`: Release [CausalWan2.2 I2V A14B Preview](https://huggingface.co/FastVideo/CausalWan2.2-I2V-A14B-Preview-Diffusers) models, [Blog](https://hao-ai-lab.github.io/blogs/fastvideo_causalwan_preview/) and [Inference Code!](https://github.com/hao-ai-lab/FastVideo/blob/main/examples/inference/basic/basic_self_forcing_causal_wan2_2_i2v.py).
|
||||
@@ -37,7 +33,7 @@ FastVideo has the following features:
|
||||
- [Sparse distillation](https://hao-ai-lab.github.io/blogs/fastvideo_post_training/) to achieve >50x denoising speedup
|
||||
- Scalable training with FSDP2, sequence parallelism, and selective activation checkpointing.
|
||||
- Causal distillation through Self-Forcing
|
||||
- See this [page](https://hao-ai-lab.github.io/FastVideo/training/overview/) for the supported training workflows, and the [support matrix](https://hao-ai-lab.github.io/FastVideo/inference/support_matrix/) for supported models.
|
||||
- See this [page](https://hao-ai-lab.github.io/FastVideo/training/overview/) for full list of supported models and recipes.
|
||||
- State-of-the-art performance optimizations for inference
|
||||
- Sequence Parallelism for distributed inference
|
||||
- Multiple state-of-the-art attention backends
|
||||
@@ -64,12 +60,7 @@ UV_TORCH_BACKEND=cu126 uv pip install fastvideo
|
||||
```
|
||||
|
||||
Use `UV_TORCH_BACKEND=cu130` on CUDA 13. Apple silicon users should follow the
|
||||
[MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
|
||||
> **On an Apple Silicon Mac?** Install with `uv pip install -e '.[mlx]'` from
|
||||
> a clone, then pick a recipe in the
|
||||
> [cookbook](https://haoailab.com/FastVideo/cookbook/). See the
|
||||
> [MLX install guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mlx/).
|
||||
[MPS installation guide](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/mps/).
|
||||
|
||||
Please see our [docs](https://hao-ai-lab.github.io/FastVideo/getting_started/installation/) for more detailed installation instructions.
|
||||
|
||||
@@ -87,7 +78,7 @@ Install FastVideo (https://github.com/hao-ai-lab/FastVideo) into a fresh uv virt
|
||||
https://hao-ai-lab.github.io/FastVideo/getting_started/installation/):
|
||||
- NVIDIA GPU, x86_64 -> docs/getting_started/installation/gpu.md
|
||||
- NVIDIA DGX Spark / GB10, aarch64, CUDA 13 -> docs/getting_started/installation/spark.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mlx.md
|
||||
- Apple Silicon, macOS -> docs/getting_started/installation/mps.md
|
||||
3. Use uv for every step. If a command fails, debug it and tell me what you changed.
|
||||
4. Verify the result:
|
||||
python -c "import fastvideo, torch; print('cuda', torch.cuda.is_available())"
|
||||
|
||||
@@ -97,33 +97,13 @@ dreamverse-server --port 8009
|
||||
dreamverse-mock-server --port 8009
|
||||
```
|
||||
|
||||
### Run Dreamverse with FastH3
|
||||
|
||||
Select the VSA data-free FastH3 Preview profile when you start the backend:
|
||||
|
||||
```bash
|
||||
DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
The `fast-h3` profile uses four visible GPUs by default. It loads the `MiniMaxAI/MiniMax-H3` base checkpoint and the
|
||||
`vsa-datafree/adapter_model.safetensors` adapter from
|
||||
`FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA`. Each request generates a 124-frame, 768×1344 video with
|
||||
synchronized audio and five sigma-grid points. Dreamverse uses the last frame of each segment as first-frame
|
||||
conditioning for the following segment.
|
||||
|
||||
Set `CUDA_VISIBLE_DEVICES` when you need to choose the four physical GPUs:
|
||||
|
||||
```bash
|
||||
CUDA_VISIBLE_DEVICES=0,1,2,3 DREAMVERSE_MODEL_ID=fast-h3 dreamverse-server --port 8009
|
||||
```
|
||||
|
||||
> **Expect a slow first boot.** With `torch.compile` and startup warmup enabled
|
||||
> (the default), the backend compiles the segment 1 and segment 2 inference
|
||||
> paths before it reports ready — this can take **tens of minutes on a cold
|
||||
> cache**, regardless of how you deploy (local, server, Docker, or Modal).
|
||||
> `/healthz` responds as soon as the process is up; `/readyz` stays `503` until
|
||||
> warmup finishes. To defer compilation until the first generated request while
|
||||
> testing, set `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
> warmup finishes. For a faster, uncompiled startup while testing, set
|
||||
> `FASTVIDEO_ENABLE_STARTUP_WARMUP=0` before starting the backend.
|
||||
|
||||
## Frontend Setup
|
||||
|
||||
@@ -239,7 +219,6 @@ selection, and mock-server behavior:
|
||||
pytest apps/dreamverse/dreamverse/tests/test_config.py \
|
||||
apps/dreamverse/dreamverse/tests/test_entrypoints.py \
|
||||
apps/dreamverse/dreamverse/tests/test_gpu_pool.py \
|
||||
apps/dreamverse/dreamverse/tests/test_minimax_h3_generation.py \
|
||||
apps/dreamverse/dreamverse/tests/test_mock_server.py -q
|
||||
```
|
||||
|
||||
|
||||
+1
-12
@@ -139,18 +139,7 @@ session.
|
||||
- startup warmup
|
||||
- user join/leave commands
|
||||
- `USER_STEP` execution for each segment
|
||||
- generation-command routing and stream-result delivery
|
||||
|
||||
Model generation has a separate ownership boundary inside each GPU process:
|
||||
|
||||
- `apps/dreamverse/dreamverse/generation_worker.py` selects the backend that the active model profile declares and owns
|
||||
the backend lifecycle.
|
||||
- `apps/dreamverse/dreamverse/ltx2_generation.py` owns LTX-2 generator configuration, video and audio continuation, and
|
||||
runtime LoRA application.
|
||||
- `apps/dreamverse/dreamverse/minimax_h3_generation.py` owns the VSA data-free FastH3 adapter, FastH3 generator and
|
||||
request configuration, and last-frame continuation through MiniMax H3 first-frame conditioning.
|
||||
- `apps/dreamverse/dreamverse/generation_contracts.py` defines the decoded media and stream-trimming result that both
|
||||
model backends return to `apps/dreamverse/dreamverse/gpu_pool.py`.
|
||||
- continuation state between segments
|
||||
|
||||
`apps/dreamverse/dreamverse/prompt_enhancer.py` manages:
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Benchmark the LTX-2 generation pipeline driven by the dreamverse Python SDK path.
|
||||
|
||||
Mirrors how ``apps/dreamverse/dreamverse/ltx2_generation.py`` constructs
|
||||
Mirrors how ``apps/dreamverse/dreamverse/video_generation.py`` constructs
|
||||
``GeneratorConfig`` and calls ``VideoGenerator.generate()``, then
|
||||
captures per-stage timings via the ``FASTVIDEO_STAGE_LOGGING=1`` log
|
||||
hooks (same mechanism as ``FastVideo-internal/examples/inference/basic/
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
_REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
_SERVER_ROOT = Path(__file__).resolve().parent
|
||||
@@ -56,37 +55,19 @@ FRONTEND_STATIC_DIR_CANDIDATES = _resolve_frontend_static_dir_candidates()
|
||||
MODEL_REGISTRY = {
|
||||
"fast-ltx2": {
|
||||
"name": "FastLTX2",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX2-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX2-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-ltx23": {
|
||||
"name": "FastLTX23",
|
||||
"generation_backend": "ltx2",
|
||||
"default_sp_size": 1,
|
||||
"model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"config_model_path": "FastVideo/LTX-2.3-Distilled-Diffusers",
|
||||
"lora_repo": "FastVideo/LTX-2.3-OmniNFT-LoRA",
|
||||
},
|
||||
"fast-h3": {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
},
|
||||
}
|
||||
|
||||
DEFAULT_MODEL_ID = "fast-ltx2"
|
||||
DEFAULT_MODEL_ID = "fast-ltx23"
|
||||
|
||||
ACTIVE_MODEL_ID = (os.getenv("DREAMVERSE_MODEL_ID", "").strip() or DEFAULT_MODEL_ID)
|
||||
if ACTIVE_MODEL_ID not in MODEL_REGISTRY:
|
||||
@@ -95,14 +76,11 @@ if ACTIVE_MODEL_ID not in MODEL_REGISTRY:
|
||||
# Active model configuration
|
||||
MODEL_CONFIG = MODEL_REGISTRY[ACTIVE_MODEL_ID]
|
||||
|
||||
# Generation limits
|
||||
SESSION_TIMEOUT_SECONDS = 300
|
||||
|
||||
# Frame settings
|
||||
NUM_FRAMES = 121
|
||||
FRAME_HEIGHT = 1088
|
||||
FRAME_WIDTH = 1920
|
||||
NUM_INFERENCE_STEPS = 5
|
||||
NUM_INFERENCE_STEPS = 6
|
||||
JPEG_QUALITY = 100
|
||||
BATCH_SIZE = 3
|
||||
|
||||
@@ -187,10 +165,13 @@ def _optional_env(*names: str) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
# Generation limits
|
||||
SESSION_TIMEOUT_SECONDS = _env_int("DREAMVERSE_SESSION_TIMEOUT_SECONDS", 300)
|
||||
|
||||
DEVTOOLS_ENABLED = _env_bool("FASTVIDEO_ENABLE_DEVTOOLS", False)
|
||||
PROMPT_SAFETY_ENABLED = _env_bool("FASTVIDEO_ENABLE_PROMPT_SAFETY", False)
|
||||
DREAMVERSE_MAX_AUTOTUNE = _env_bool("DREAMVERSE_MAX_AUTOTUNE", True)
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", cast(int, MODEL_CONFIG["default_sp_size"])))
|
||||
DREAMVERSE_SP_SIZE = max(1, _env_int("DREAMVERSE_SP_SIZE", 1))
|
||||
|
||||
DREAMVERSE_MODEL_PATH = (os.getenv("DREAMVERSE_MODEL_PATH", "").strip() or None)
|
||||
if DREAMVERSE_MODEL_PATH:
|
||||
@@ -232,7 +213,7 @@ def _resolve_lora_spec(spec: str) -> str | None:
|
||||
if not spec:
|
||||
return None
|
||||
if spec.lower() == "omninft":
|
||||
return cast(str | None, MODEL_CONFIG.get("lora_repo"))
|
||||
return MODEL_CONFIG.get("lora_repo")
|
||||
if spec.lower() in AVAILABLE_LORAS:
|
||||
return AVAILABLE_LORAS[spec.lower()]["repo"]
|
||||
return spec
|
||||
|
||||
@@ -1,46 +0,0 @@
|
||||
"""Shared contract between DreamVerse generation backends and GPU workers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Protocol
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Decoded media and stream-trimming metadata for one DreamVerse segment."""
|
||||
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict[str, float]
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class GenerationBackend(Protocol):
|
||||
"""Model-owned generation operations used by one GPU worker process."""
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
...
|
||||
|
||||
def shutdown(self) -> None:
|
||||
...
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
...
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
...
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
...
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
...
|
||||
@@ -1,96 +0,0 @@
|
||||
"""Select and own one model-specific generation backend per GPU process."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dreamverse.config import MODEL_CONFIG
|
||||
from dreamverse.generation_contracts import GenerationBackend, StepResult
|
||||
|
||||
|
||||
def _create_generation_backend(backend_name: str, gpu_id: int) -> GenerationBackend:
|
||||
"""Construct the backend that owns the selected model family's behavior."""
|
||||
if backend_name == "ltx2":
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
return LTX2GenerationBackend(gpu_id)
|
||||
if backend_name == "minimax_h3":
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
return MiniMaxH3GenerationBackend(gpu_id)
|
||||
raise ValueError(f"Unsupported DreamVerse generation backend: {backend_name!r}")
|
||||
|
||||
|
||||
class VideoGenerationWorker:
|
||||
"""Delegate GPU lifecycle and generation calls to the active model backend."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.model_config: dict = dict(MODEL_CONFIG)
|
||||
self.backend_name: str | None = None
|
||||
self.backend: GenerationBackend | None = None
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Load the requested model through its generation backend.
|
||||
|
||||
Model selection belongs here so the GPU process and streaming layers
|
||||
use one stable media contract without importing model-specific code.
|
||||
"""
|
||||
requested_model_config = dict(model_config) if model_config is not None else dict(self.model_config)
|
||||
backend_name = requested_model_config.get("generation_backend")
|
||||
if not isinstance(backend_name, str) or not backend_name:
|
||||
raise ValueError("DreamVerse model configuration requires `generation_backend`.")
|
||||
|
||||
candidate_backend = self.backend
|
||||
if candidate_backend is None or self.backend_name != backend_name:
|
||||
if candidate_backend is not None:
|
||||
candidate_backend.shutdown()
|
||||
candidate_backend = _create_generation_backend(backend_name, self.gpu_id)
|
||||
|
||||
try:
|
||||
candidate_backend.initialize(requested_model_config)
|
||||
except Exception:
|
||||
try:
|
||||
candidate_backend.shutdown()
|
||||
except Exception as shutdown_error:
|
||||
print(f"[GPU {self.gpu_id}] Backend cleanup after initialization failure: {shutdown_error}")
|
||||
self.backend = None
|
||||
self.backend_name = None
|
||||
raise
|
||||
|
||||
self.model_config = requested_model_config
|
||||
self.backend = candidate_backend
|
||||
self.backend_name = backend_name
|
||||
|
||||
def _require_backend(self) -> GenerationBackend:
|
||||
"""Return the initialized backend or fail before processing a command."""
|
||||
if self.backend is None:
|
||||
raise RuntimeError("Generation backend is not initialized.")
|
||||
return self.backend
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release model resources owned by the selected backend."""
|
||||
if self.backend is not None:
|
||||
self.backend.shutdown()
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
self._require_backend().clear_conditioning()
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one segment through the selected model backend."""
|
||||
return self._require_backend().generate_step(
|
||||
prompt,
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
return self._require_backend().warmup(prompt)
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
return self._require_backend().apply_lora_stack(stack)
|
||||
@@ -12,7 +12,7 @@ from enum import Enum
|
||||
from multiprocessing import Process, Queue
|
||||
|
||||
from dreamverse.config import (
|
||||
ACTIVE_MODEL_ID,
|
||||
DEFAULT_MODEL_ID,
|
||||
DREAMVERSE_SP_SIZE,
|
||||
MODEL_REGISTRY,
|
||||
STARTUP_WARMUP_ENABLED,
|
||||
@@ -54,7 +54,7 @@ from dreamverse.worker_ipc import (
|
||||
def _parse_requested_gpu_limit() -> int | None:
|
||||
raw_value = os.getenv("FASTVIDEO_GPU_COUNT", "").strip().lower()
|
||||
if not raw_value:
|
||||
return DREAMVERSE_SP_SIZE
|
||||
return 1
|
||||
if raw_value == "all":
|
||||
return None
|
||||
try:
|
||||
@@ -164,12 +164,12 @@ def gpu_worker_process(
|
||||
os.environ["CUDA_VISIBLE_DEVICES"] = cuda_device
|
||||
os.environ["FASTVIDEO_ATTENTION_BACKEND"] = "FLASH_ATTN"
|
||||
|
||||
from dreamverse.generation_worker import VideoGenerationWorker
|
||||
from dreamverse.video_generation import VideoGenerationWorker
|
||||
|
||||
worker = VideoGenerationWorker(gpu_id)
|
||||
|
||||
def event_loop(first_cmd: Command = None):
|
||||
"""Block on generation commands after the model is initialized."""
|
||||
"""Blocking event loop for LTX2; dispatches user commands."""
|
||||
print(f"[GPU {gpu_id}] Entering event loop")
|
||||
|
||||
def handle_command(cmd: Command):
|
||||
@@ -435,7 +435,7 @@ class GPUSlot:
|
||||
self._response_reader_task: asyncio.Task | None = None
|
||||
self._active: bool = False
|
||||
self._reader_lock: asyncio.Lock | None = None
|
||||
self.current_model_id: str | None = ACTIVE_MODEL_ID
|
||||
self.current_model_id: str = DEFAULT_MODEL_ID
|
||||
self.shared_stream_buffer = None
|
||||
self.shared_stream_buffer_size = SHARED_STREAM_BUFFER_BYTES
|
||||
|
||||
@@ -690,7 +690,7 @@ class GPUSlot:
|
||||
async def join_user(self, user_id: str, model_id: str = None) -> JoinAck:
|
||||
"""Add a user to this GPU."""
|
||||
if model_id is None:
|
||||
model_id = ACTIVE_MODEL_ID
|
||||
model_id = DEFAULT_MODEL_ID
|
||||
|
||||
# Reload model if a different one is requested
|
||||
if model_id != self.current_model_id and model_id in MODEL_REGISTRY:
|
||||
@@ -705,23 +705,16 @@ class GPUSlot:
|
||||
self.connected_users.clear()
|
||||
|
||||
model_config = MODEL_REGISTRY[model_id]
|
||||
try:
|
||||
reload_response = await self._send_command(Command(
|
||||
CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
except Exception:
|
||||
self.current_model_id = None
|
||||
raise
|
||||
reload_response = await self._send_command(Command(CommandType.RELOAD_MODEL,
|
||||
payload=ReloadModelPayload(model_config=model_config),
|
||||
user_id="__reload__"),
|
||||
timeout=600.0)
|
||||
match reload_response:
|
||||
case ReloadAck():
|
||||
pass
|
||||
case WorkerError(message=msg):
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Model reload failed: {msg}")
|
||||
case _:
|
||||
self.current_model_id = None
|
||||
raise RuntimeError(f"Unexpected reload response: "
|
||||
f"{type(reload_response).__name__}")
|
||||
|
||||
@@ -1030,7 +1023,11 @@ def get_available_gpus() -> list[int]:
|
||||
"""Get list of available GPU IDs from environment or auto-detect."""
|
||||
cuda_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "")
|
||||
if cuda_visible:
|
||||
visible_gpu_ids = [int(x.strip()) for x in cuda_visible.split(",") if x.strip()]
|
||||
try:
|
||||
visible_gpu_ids = [int(x.strip()) for x in cuda_visible.split(",") if x.strip()]
|
||||
except ValueError as exc:
|
||||
raise RuntimeError("CUDA_VISIBLE_DEVICES must be a comma-separated list of integer GPU "
|
||||
f"indices (got {cuda_visible!r}); GPU UUIDs are not supported.") from exc
|
||||
return _limit_gpu_ids(visible_gpu_ids)
|
||||
|
||||
# Auto-detect available GPUs
|
||||
|
||||
@@ -206,7 +206,9 @@ def cli() -> None:
|
||||
args = parser.parse_args()
|
||||
|
||||
_install_heartbeat_log_filter()
|
||||
uvicorn.run(app, host=args.host, port=args.port)
|
||||
# A 15MB init image (session_init_image.MAX_SESSION_INIT_IMAGE_BYTES) is ~20MB
|
||||
# as a base64 ws message, above uvicorn's default 16MiB frame cap.
|
||||
uvicorn.run(app, host=args.host, port=args.port, ws_max_size=32 * 1024 * 1024)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,297 +0,0 @@
|
||||
"""FastH3 model lifecycle and first-frame continuation for DreamVerse."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import gc
|
||||
import os
|
||||
import time
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
from dreamverse.config import DREAMVERSE_SP_SIZE
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from PIL.Image import Image
|
||||
|
||||
|
||||
def _required_config_str(model_config: dict, field_name: str) -> str:
|
||||
"""Read one required non-empty string from a DreamVerse model profile."""
|
||||
value = model_config.get(field_name)
|
||||
if not isinstance(value, str) or not value.strip():
|
||||
raise ValueError(f"FastH3 model configuration requires `{field_name}`.")
|
||||
return value.strip()
|
||||
|
||||
|
||||
class MiniMaxH3GenerationBackend:
|
||||
"""Run the VSA data-free FastH3 adapter and retain one continuation frame."""
|
||||
|
||||
def __init__(self, gpu_id: int):
|
||||
self.gpu_id = gpu_id
|
||||
self.generator: Any | None = None
|
||||
self.model_config: dict = {}
|
||||
self.continuation_image: Image | None = None
|
||||
|
||||
def _gpu_mem(self) -> str:
|
||||
allocated_gib = torch.cuda.memory_allocated() / 1024**3
|
||||
reserved_gib = torch.cuda.memory_reserved() / 1024**3
|
||||
return f"alloc={allocated_gib:.2f}GiB, reserved={reserved_gib:.2f}GiB"
|
||||
|
||||
@staticmethod
|
||||
def _configure_environment(attention_backend: str) -> None:
|
||||
"""Apply the fixed boot-time switches from the FastH3 reference recipe."""
|
||||
os.environ.update({
|
||||
"FASTVIDEO_ATTENTION_BACKEND": attention_backend,
|
||||
"FASTVIDEO_FA4": "1",
|
||||
"FASTVIDEO_MINIMAX_H3_FUSIONS": "all",
|
||||
"FASTVIDEO_VSA_SM100A": "0",
|
||||
})
|
||||
os.environ.pop("FASTVIDEO_INFERENCE_TORCH_COMPILE", None)
|
||||
|
||||
def initialize(self, model_config: dict | None = None) -> None:
|
||||
"""Download the fixed Preview adapter and load the FastH3 generator.
|
||||
|
||||
The model profile owns the base checkpoint, adapter file, attention
|
||||
backend, and generation geometry. The backend translates that profile
|
||||
into FastVideo's typed generator configuration.
|
||||
"""
|
||||
if model_config is not None:
|
||||
self.model_config = dict(model_config)
|
||||
if not self.model_config:
|
||||
raise ValueError("FastH3 initialization requires a model configuration.")
|
||||
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
self.clear_conditioning()
|
||||
model_path = _required_config_str(self.model_config, "model_path")
|
||||
adapter_repo = _required_config_str(self.model_config, "adapter_repo")
|
||||
adapter_filename = _required_config_str(self.model_config, "adapter_filename")
|
||||
attention_backend = _required_config_str(self.model_config, "attention_backend")
|
||||
self._configure_environment(attention_backend)
|
||||
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
from fastvideo import VideoGenerator
|
||||
from fastvideo.api import (
|
||||
CompileConfig,
|
||||
ComponentConfig,
|
||||
EngineConfig,
|
||||
GeneratorConfig,
|
||||
OffloadConfig,
|
||||
ParallelismConfig,
|
||||
PipelineSelection,
|
||||
)
|
||||
|
||||
adapter_path = hf_hub_download(repo_id=adapter_repo, filename=adapter_filename)
|
||||
experimental = {
|
||||
"attention_backend": attention_backend,
|
||||
"inference_torch_compile": attention_backend == "FLASH_ATTN",
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
}
|
||||
if attention_backend == "VIDEO_SPARSE_ATTN_H3":
|
||||
experimental.update({
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
})
|
||||
generator_config = GeneratorConfig(
|
||||
model_path=model_path,
|
||||
pipeline=PipelineSelection(
|
||||
components=ComponentConfig(lora_path=adapter_path, lora_strength=1.0),
|
||||
experimental=experimental,
|
||||
),
|
||||
engine=EngineConfig(
|
||||
num_gpus=DREAMVERSE_SP_SIZE,
|
||||
parallelism=ParallelismConfig(tp_size=1, sp_size=DREAMVERSE_SP_SIZE),
|
||||
offload=OffloadConfig(
|
||||
dit=False,
|
||||
dit_layerwise=False,
|
||||
text_encoder=True,
|
||||
image_encoder=True,
|
||||
vae=True,
|
||||
pin_cpu_memory=True,
|
||||
),
|
||||
compile=CompileConfig(enabled=False, vae_enabled=True),
|
||||
use_fsdp_inference=False,
|
||||
),
|
||||
)
|
||||
|
||||
print(f"[GPU {self.gpu_id}] Loading FastH3 model: {model_path}")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 adapter: {adapter_repo}/{adapter_filename}")
|
||||
print(f"[GPU {self.gpu_id}] Before model load: {self._gpu_mem()}")
|
||||
self.generator = VideoGenerator.from_config(generator_config)
|
||||
print(f"[GPU {self.gpu_id}] FastH3 loaded: {self._gpu_mem()} (warmup pending)")
|
||||
|
||||
def shutdown(self) -> None:
|
||||
"""Release the FastVideo generator and cached continuation image."""
|
||||
self.clear_conditioning()
|
||||
if self.generator is not None:
|
||||
self.generator.shutdown()
|
||||
self.generator = None
|
||||
|
||||
def clear_conditioning(self) -> None:
|
||||
"""Release the first-frame image retained for the next segment."""
|
||||
if self.continuation_image is not None:
|
||||
self.continuation_image.close()
|
||||
self.continuation_image = None
|
||||
|
||||
@staticmethod
|
||||
def _load_rgb_image(image_path: str) -> Image:
|
||||
"""Load an image into an independent RGB buffer with no open file handle."""
|
||||
from PIL import Image
|
||||
|
||||
with Image.open(image_path) as image:
|
||||
return image.convert("RGB").copy()
|
||||
|
||||
def _select_conditioning_image(
|
||||
self,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> tuple[Image | None, bool]:
|
||||
"""Select the initial upload or retained last frame for one segment."""
|
||||
if reset_conditioning:
|
||||
self.clear_conditioning()
|
||||
if segment_idx > 1 and self.continuation_image is not None:
|
||||
return self.continuation_image.copy(), True
|
||||
if segment_idx > 1 and not reset_conditioning:
|
||||
raise RuntimeError(f"FastH3 segment {segment_idx} requires a retained continuation frame.")
|
||||
if segment_idx == 1 and image_path:
|
||||
return self._load_rgb_image(image_path), False
|
||||
return None, False
|
||||
|
||||
def _build_request(self, prompt: str, conditioning_image: Image | None):
|
||||
"""Build the typed FastVideo request owned by the FastH3 profile."""
|
||||
from fastvideo.api import GenerationRequest, InputConfig, OutputConfig, SamplingConfig
|
||||
|
||||
return GenerationRequest(
|
||||
prompt=prompt,
|
||||
negative_prompt="",
|
||||
inputs=InputConfig(pil_image=conditioning_image),
|
||||
sampling=SamplingConfig(
|
||||
height=int(self.model_config["height"]),
|
||||
width=int(self.model_config["width"]),
|
||||
num_frames=int(self.model_config["num_frames"]),
|
||||
fps=24,
|
||||
num_inference_steps=int(self.model_config["num_inference_steps"]),
|
||||
guidance_scale=1.0,
|
||||
batch_cfg=False,
|
||||
seed=int(self.model_config["seed"]),
|
||||
),
|
||||
output=OutputConfig(save_video=False, return_frames=True),
|
||||
)
|
||||
|
||||
def _save_continuation_frame(self, frames: list) -> None:
|
||||
"""Retain the last decoded frame as first-frame conditioning."""
|
||||
from PIL import Image
|
||||
|
||||
self.clear_conditioning()
|
||||
self.continuation_image = Image.fromarray(np.ascontiguousarray(frames[-1])).convert("RGB")
|
||||
|
||||
def generate_step(
|
||||
self,
|
||||
prompt: str,
|
||||
segment_idx: int,
|
||||
image_path: str | None,
|
||||
reset_conditioning: bool,
|
||||
) -> StepResult:
|
||||
"""Generate one synchronized FastH3 segment and retain its last frame.
|
||||
|
||||
Later segments use MiniMax H3's first-frame-to-video path. The first
|
||||
conditioned frame and its matching audio duration are trimmed before
|
||||
streaming so adjacent segments do not duplicate media.
|
||||
"""
|
||||
if self.generator is None:
|
||||
raise RuntimeError("FastH3 generator is not initialized.")
|
||||
conditioning_image, uses_continuation = self._select_conditioning_image(
|
||||
segment_idx,
|
||||
image_path,
|
||||
reset_conditioning,
|
||||
)
|
||||
request = self._build_request(prompt, conditioning_image)
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
result = self.generator.generate(request)
|
||||
finally:
|
||||
if conditioning_image is not None:
|
||||
conditioning_image.close()
|
||||
torch.cuda.synchronize()
|
||||
generation_ms = (time.perf_counter() - started) * 1000.0
|
||||
|
||||
if isinstance(result, list):
|
||||
raise RuntimeError("FastH3 returned multiple results for one DreamVerse segment.")
|
||||
frames = result.frames
|
||||
if not isinstance(frames, list) or not frames:
|
||||
raise RuntimeError("FastH3 generation did not return decoded frames.")
|
||||
audio = result.audio
|
||||
audio_sample_rate = result.audio_sample_rate
|
||||
if audio is not None and audio_sample_rate is None:
|
||||
raise RuntimeError("FastH3 returned audio without an audio sample rate.")
|
||||
|
||||
save_started = time.perf_counter()
|
||||
self._save_continuation_frame(frames)
|
||||
save_conditioning_ms = (time.perf_counter() - save_started) * 1000.0
|
||||
timings = {
|
||||
"generation_ms": generation_ms,
|
||||
"generation_time_ms": float(result.generation_time or 0.0) * 1000.0,
|
||||
"save_conditioning_ms": save_conditioning_ms,
|
||||
"e2e_latency_ms": (time.perf_counter() - started) * 1000.0,
|
||||
}
|
||||
trim_frames = 1 if uses_continuation else 0
|
||||
print(f"[GPU {self.gpu_id}] FastH3 segment {segment_idx}: "
|
||||
f"{len(frames)} frames, gen={generation_ms:.0f}ms, "
|
||||
f"save_conditioning={save_conditioning_ms:.0f}ms, "
|
||||
f"e2e={timings['e2e_latency_ms']:.0f}ms")
|
||||
return StepResult(
|
||||
frames=frames,
|
||||
audio=audio,
|
||||
audio_sample_rate=audio_sample_rate,
|
||||
timings=timings,
|
||||
head_trim_frames=trim_frames,
|
||||
head_trim_audio_frames=trim_frames,
|
||||
)
|
||||
|
||||
def warmup(self, prompt: str) -> dict[str, float]:
|
||||
"""Compile the FastH3 text and first-frame paths before readiness."""
|
||||
warmup_prompt = (prompt or "").strip()
|
||||
if not warmup_prompt:
|
||||
raise RuntimeError("Startup warmup prompt must be non-empty.")
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup starting "
|
||||
"(synthetic segments: text-to-video, first-frame-to-video)")
|
||||
started = time.perf_counter()
|
||||
text_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
first_frame_result = self.generate_step(
|
||||
warmup_prompt,
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
total_ms = (time.perf_counter() - started) * 1000.0
|
||||
self.clear_conditioning()
|
||||
text_ms = float(text_result.timings.get("e2e_latency_ms", 0.0))
|
||||
first_frame_ms = float(first_frame_result.timings.get("e2e_latency_ms", 0.0))
|
||||
print(f"[GPU {self.gpu_id}] FastH3 startup warmup complete: "
|
||||
f"text_to_video={text_ms:.0f}ms, "
|
||||
f"first_frame_to_video={first_frame_ms:.0f}ms, "
|
||||
f"total={total_ms:.0f}ms")
|
||||
return {
|
||||
"warmup_text_to_video_ms": text_ms,
|
||||
"warmup_first_frame_to_video_ms": first_frame_ms,
|
||||
"warmup_total_ms": total_ms,
|
||||
}
|
||||
|
||||
def apply_lora_stack(self, stack: list[tuple[str, float]]) -> tuple[str | None, str | None]:
|
||||
"""Reject runtime LoRA mutation because FastH3 uses one startup adapter."""
|
||||
del stack
|
||||
raise RuntimeError("FastH3 uses its fixed startup adapter and does not support runtime LoRA changes.")
|
||||
@@ -30,11 +30,11 @@ from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
|
||||
from dreamverse._deps import require_dreamverse_runtime_deps
|
||||
from dreamverse.config import FRONTEND_STATIC_DIR_CANDIDATES, GENERATION_SEGMENT_CAP
|
||||
from dreamverse.config import FRONTEND_STATIC_DIR_CANDIDATES, GENERATION_SEGMENT_CAP, SESSION_TIMEOUT_SECONDS
|
||||
from dreamverse.session_init_image import cleanup_session_init_image, persist_session_init_image
|
||||
from dreamverse.utils import _resolve_generation_segment_cap
|
||||
|
||||
LATENCY_MS = 200
|
||||
SESSION_TIMEOUT_SECONDS = 300
|
||||
MOCK_FRAME_WIDTH = 640
|
||||
MOCK_FRAME_HEIGHT = 352
|
||||
MOCK_FPS = 24
|
||||
@@ -334,6 +334,7 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
auto_extension_enabled = bool(init_data.get("auto_extension_enabled", False))
|
||||
loop_generation_enabled = bool(init_data.get("loop_generation_enabled", False))
|
||||
single_clip_mode = bool(init_data.get("single_clip_mode", False))
|
||||
manual_continuation_mode = bool(init_data.get("manual_continuation_mode", False))
|
||||
generation_paused = False
|
||||
|
||||
if init_type == "session_init_v2":
|
||||
@@ -344,7 +345,8 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
incoming_prompts = []
|
||||
|
||||
curated_prompts = [prompt.strip() for prompt in incoming_prompts if isinstance(prompt, str) and prompt.strip()]
|
||||
generation_paused = bool(initial_rollout_prompt and not single_clip_mode and len(curated_prompts) == 0)
|
||||
generation_paused = bool(not manual_continuation_mode and initial_rollout_prompt and not single_clip_mode
|
||||
and len(curated_prompts) == 0)
|
||||
|
||||
try:
|
||||
session_init_image = persist_session_init_image(init_data.get("initial_image"))
|
||||
@@ -373,7 +375,10 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
prompt_sources_blocked = False
|
||||
pending_seed_reset = False
|
||||
pending_seed_reset_reason = ""
|
||||
pending_simple_submission: PromptSubmission | None = None
|
||||
pending_simple_submission: PromptSubmission | None = (PromptSubmission(
|
||||
prompt_id=str(init_data.get("initial_rollout_prompt_id") or uuid.uuid4()),
|
||||
raw_prompt=initial_rollout_prompt,
|
||||
) if manual_continuation_mode and initial_rollout_prompt else None)
|
||||
single_clip_waiting_for_request = False
|
||||
rollout_waiting_for_rewrite = False
|
||||
initial_rollout_waiting_for_rewrite = generation_paused
|
||||
@@ -393,14 +398,26 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
|
||||
async def send_stream_start(seed_reason: str) -> None:
|
||||
await ws_send_json({
|
||||
"type": "ltx2_stream_start",
|
||||
"total_segments": len(curated_prompts),
|
||||
"preset_id": preset_id,
|
||||
"stream_mode": "av_fmp4",
|
||||
"live_mode": True,
|
||||
"loop_generation_enabled": loop_generation_enabled,
|
||||
"loop_iteration": loop_iteration,
|
||||
"generation_segment_cap": 0,
|
||||
"type":
|
||||
"ltx2_stream_start",
|
||||
"total_segments":
|
||||
len(curated_prompts),
|
||||
"preset_id":
|
||||
preset_id,
|
||||
"stream_mode":
|
||||
"av_fmp4",
|
||||
"live_mode":
|
||||
True,
|
||||
"loop_generation_enabled":
|
||||
loop_generation_enabled,
|
||||
"loop_iteration":
|
||||
loop_iteration,
|
||||
"generation_segment_cap":
|
||||
_resolve_generation_segment_cap(
|
||||
single_clip_mode=single_clip_mode,
|
||||
cap=GENERATION_SEGMENT_CAP,
|
||||
manual_continuation_mode=manual_continuation_mode,
|
||||
),
|
||||
})
|
||||
if seed_reason == "init":
|
||||
await ws_send_json({
|
||||
@@ -493,6 +510,7 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
nonlocal auto_extension_enabled
|
||||
nonlocal loop_generation_enabled
|
||||
nonlocal single_clip_mode
|
||||
nonlocal manual_continuation_mode
|
||||
nonlocal generation_paused
|
||||
nonlocal seed_prompt_memory
|
||||
nonlocal curated_prompts
|
||||
@@ -537,6 +555,7 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
auto_extension_enabled = bool(payload.get("auto_extension_enabled", False))
|
||||
loop_generation_enabled = bool(payload.get("loop_generation_enabled", False))
|
||||
single_clip_mode = bool(payload.get("single_clip_mode", False))
|
||||
manual_continuation_mode = bool(payload.get("manual_continuation_mode", False))
|
||||
|
||||
seed_prompt_memory = list(next_curated_prompts)
|
||||
curated_prompts = list(seed_prompt_memory)
|
||||
@@ -544,10 +563,14 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
segment_idx = 0
|
||||
pending_seed_reset = False
|
||||
pending_seed_reset_reason = ""
|
||||
pending_simple_submission = None
|
||||
pending_simple_submission = (PromptSubmission(
|
||||
prompt_id=str(payload.get("initial_rollout_prompt_id") or uuid.uuid4()),
|
||||
raw_prompt=initial_rollout_prompt,
|
||||
) if manual_continuation_mode and initial_rollout_prompt else None)
|
||||
single_clip_waiting_for_request = False
|
||||
rollout_waiting_for_rewrite = False
|
||||
generation_paused = bool(initial_rollout_prompt and not single_clip_mode and len(curated_prompts) == 0)
|
||||
generation_paused = bool(not manual_continuation_mode and initial_rollout_prompt and not single_clip_mode
|
||||
and len(curated_prompts) == 0)
|
||||
initial_rollout_waiting_for_rewrite = generation_paused
|
||||
rewrite_restart_pending = False
|
||||
loop_iteration = 0
|
||||
@@ -968,10 +991,11 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
project_stream_started = True
|
||||
await send_stream_start(pending_seed_reset_reason)
|
||||
pending_seed_reset_reason = ""
|
||||
if pending_simple_submission is not None:
|
||||
submission = pending_simple_submission
|
||||
pending_simple_submission = None
|
||||
await promote_submission_to_ready(submission)
|
||||
|
||||
if pending_simple_submission is not None:
|
||||
submission = pending_simple_submission
|
||||
pending_simple_submission = None
|
||||
await promote_submission_to_ready(submission)
|
||||
|
||||
if generation_paused:
|
||||
await asyncio.sleep(0.05)
|
||||
@@ -981,8 +1005,8 @@ async def websocket_endpoint(websocket: WebSocket):
|
||||
await asyncio.sleep(0.05)
|
||||
continue
|
||||
|
||||
if (not single_clip_mode and not rollout_waiting_for_rewrite and GENERATION_SEGMENT_CAP > 0
|
||||
and segment_idx >= GENERATION_SEGMENT_CAP):
|
||||
if (not single_clip_mode and not manual_continuation_mode and not rollout_waiting_for_rewrite
|
||||
and GENERATION_SEGMENT_CAP > 0 and segment_idx >= GENERATION_SEGMENT_CAP):
|
||||
rollout_waiting_for_rewrite = True
|
||||
loop_generation_enabled = False
|
||||
project_stream_started = False
|
||||
@@ -1217,7 +1241,9 @@ def cli() -> None:
|
||||
print(f"Starting mock server with {LATENCY_MS}ms latency on port {args.port}")
|
||||
|
||||
_install_heartbeat_log_filter()
|
||||
uvicorn.run(app, host="0.0.0.0", port=args.port)
|
||||
# A 15MB init image (session_init_image.MAX_SESSION_INIT_IMAGE_BYTES) is ~20MB
|
||||
# as a base64 ws message, above uvicorn's default 16MiB frame cap.
|
||||
uvicorn.run(app, host="0.0.0.0", port=args.port, ws_max_size=32 * 1024 * 1024)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -313,6 +313,33 @@ def _extract_content_or_empty(response_json: dict[str, Any]) -> str:
|
||||
return ""
|
||||
|
||||
|
||||
def _find_balanced_object_end(text: str, start: int) -> int:
|
||||
"""Return the index just past the brace-balanced span opening at
|
||||
``text[start] == '{'``, honoring JSON string literals and escapes, or -1
|
||||
if the braces never balance (i.e. the object was truncated)."""
|
||||
depth = 0
|
||||
in_string = False
|
||||
escaped = False
|
||||
for i in range(start, len(text)):
|
||||
ch = text[i]
|
||||
if in_string:
|
||||
if escaped:
|
||||
escaped = False
|
||||
elif ch == "\\":
|
||||
escaped = True
|
||||
elif ch == '"':
|
||||
in_string = False
|
||||
elif ch == '"':
|
||||
in_string = True
|
||||
elif ch == "{":
|
||||
depth += 1
|
||||
elif ch == "}":
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
return i + 1
|
||||
return -1
|
||||
|
||||
|
||||
def _parse_json_response(content: str) -> dict[str, Any]:
|
||||
text = content.strip()
|
||||
if not text:
|
||||
@@ -330,6 +357,7 @@ def _parse_json_response(content: str) -> dict[str, Any]:
|
||||
r"```(?:json)?\s*([\s\S]*?)```",
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
last_fenced: dict[str, Any] | None = None
|
||||
for match in fence_pattern.finditer(text):
|
||||
block = match.group(1).strip()
|
||||
if not block:
|
||||
@@ -337,21 +365,37 @@ def _parse_json_response(content: str) -> dict[str, Any]:
|
||||
try:
|
||||
parsed = json.loads(block)
|
||||
if isinstance(parsed, dict):
|
||||
return parsed
|
||||
last_fenced = parsed
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if last_fenced is not None:
|
||||
return last_fenced
|
||||
|
||||
# Fall back to scanning for the first decodable JSON object in free-form text.
|
||||
# Scan for all decodable JSON objects and return the last — chain-of-thought
|
||||
# models emit draft JSON mid-reasoning; the final answer is always last.
|
||||
decoder = json.JSONDecoder()
|
||||
for idx, char in enumerate(text):
|
||||
if char != "{":
|
||||
continue
|
||||
last_parsed: dict[str, Any] | None = None
|
||||
pos = 0
|
||||
while (idx := text.find("{", pos)) != -1:
|
||||
try:
|
||||
parsed, _ = decoder.raw_decode(text[idx:])
|
||||
parsed, consumed = decoder.raw_decode(text[idx:])
|
||||
except json.JSONDecodeError:
|
||||
# Skip the whole failed object rather than rescanning inside it:
|
||||
# fragments nested in a malformed or truncated (finish_reason=
|
||||
# length) object must not override an earlier complete object.
|
||||
span_end = _find_balanced_object_end(text, idx)
|
||||
if span_end == -1:
|
||||
break
|
||||
pos = span_end
|
||||
continue
|
||||
# Skip past the consumed span so nested braces inside a decoded
|
||||
# object are not re-parsed as standalone objects.
|
||||
pos = idx + consumed
|
||||
if isinstance(parsed, dict):
|
||||
return parsed
|
||||
last_parsed = parsed
|
||||
|
||||
if last_parsed is not None:
|
||||
return last_parsed
|
||||
|
||||
raise ValueError("No JSON object found in assistant response.")
|
||||
|
||||
@@ -381,7 +425,7 @@ def _format_locked_segments(locked_segments: list[str]) -> str:
|
||||
|
||||
class PromptEnhancer:
|
||||
|
||||
def __init__(self):
|
||||
def __init__(self) -> None:
|
||||
self.provider = PROMPT_PROVIDER
|
||||
self.provider_label = _resolve_provider_label(PROMPT_PROVIDER)
|
||||
self.api_key = PROMPT_API_KEY
|
||||
@@ -1417,15 +1461,12 @@ class PromptEnhancer:
|
||||
locked_text = _format_locked_segments(locked_segments_clean)
|
||||
request_system_prompt = self.enhance_system_prompt
|
||||
user_payload = {
|
||||
"request": (
|
||||
"<locked_segments>\n"
|
||||
f"{locked_text}\n"
|
||||
"</locked_segments>\n\n"
|
||||
f"<conditioning_prompt>{cleaned}</conditioning_prompt>\n\n"
|
||||
f"Write exactly one new segment ({next_segment_key}) "
|
||||
"continuing from the locked segments. "
|
||||
'Respond with valid JSON only as {"next_prompt": "..."}.' # noqa: E501
|
||||
),
|
||||
"request": ("<locked_segments>\n"
|
||||
f"{locked_text}\n"
|
||||
"</locked_segments>\n\n"
|
||||
f"<conditioning_prompt>{cleaned}</conditioning_prompt>\n\n"
|
||||
f"Write exactly one new segment ({next_segment_key}) "
|
||||
"continuing from the locked segments."),
|
||||
}
|
||||
|
||||
t0 = time.perf_counter()
|
||||
@@ -1457,6 +1498,7 @@ class PromptEnhancer:
|
||||
body=request_body,
|
||||
timeout_seconds=timeout_seconds,
|
||||
)
|
||||
_enhance_print("INFO", f"raw_response: {response_content}")
|
||||
if is_single_clip_mode:
|
||||
prompt = self._extract_single_clip_prompt(response_content)
|
||||
else:
|
||||
@@ -1539,7 +1581,7 @@ class PromptEnhancer:
|
||||
f"Write exactly one new segment ({next_segment_key}) "
|
||||
"that continues linearly from the locked segments. "
|
||||
"Infer the next narrative beat from this history. "
|
||||
'Respond with valid JSON only as {"next_prompt": "..."}.' # noqa: E501
|
||||
'Respond with valid JSON only: {"next_prompt": "<your segment description here>"}.' # noqa: E501
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@@ -20,6 +20,7 @@ from __future__ import annotations
|
||||
# mypy: ignore-errors
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import time
|
||||
import uuid
|
||||
from typing import TYPE_CHECKING
|
||||
@@ -30,7 +31,7 @@ from dreamverse.session_init_image import cleanup_session_init_image, persist_se
|
||||
from dreamverse.worker_ipc import MediaChunk, MediaComplete, MediaInit
|
||||
|
||||
from dreamverse.config import (
|
||||
ACTIVE_MODEL_ID,
|
||||
DEFAULT_MODEL_ID,
|
||||
GENERATION_SEGMENT_CAP,
|
||||
PROMPT_AUTO_SLEEP_MS,
|
||||
PROMPT_AUTO_TIMEOUT_MS,
|
||||
@@ -52,6 +53,14 @@ if TYPE_CHECKING:
|
||||
from dreamverse.prompt_enhancer import PromptEnhancer
|
||||
from dreamverse.prompt_safety import PromptSafetyFilter
|
||||
|
||||
# Optional append-only log of every generated segment prompt; unset disables it.
|
||||
SEGMENT_PROMPT_LOG_PATH = os.environ.get("DREAMVERSE_SEGMENT_PROMPT_LOG", "")
|
||||
|
||||
|
||||
def _append_segment_prompt_log(path: str, text: str) -> None:
|
||||
with open(path, "a") as f:
|
||||
f.write(text)
|
||||
|
||||
|
||||
class SessionController:
|
||||
"""Runs one WebSocket session from accept() through disconnect."""
|
||||
@@ -198,6 +207,7 @@ class SessionController:
|
||||
auto_extension_enabled = bool(init_data.get("auto_extension_enabled", False))
|
||||
loop_generation_enabled = bool(init_data.get("loop_generation_enabled", False))
|
||||
single_clip_mode = bool(init_data.get("single_clip_mode", False))
|
||||
manual_continuation_mode = bool(init_data.get("manual_continuation_mode", False))
|
||||
rewrite_model = self.prompt_enhancer.resolve_rewrite_model(init_data.get("rewrite_model"))
|
||||
rewrite_system_prompt_override = str(init_data.get("rewrite_window_system_prompt") or "").strip()
|
||||
rewrite_user_system_prompt_override = str(init_data.get("rewrite_user_system_prompt") or "").strip()
|
||||
@@ -264,7 +274,7 @@ class SessionController:
|
||||
timeout_task = asyncio.create_task(session_timeout())
|
||||
|
||||
# Join the engine on this GPU.
|
||||
await slot.join_user(client_id, model_id=ACTIVE_MODEL_ID)
|
||||
await slot.join_user(client_id, model_id=DEFAULT_MODEL_ID)
|
||||
|
||||
# Notify client they're connected to a GPU.
|
||||
await ws_send_json({
|
||||
@@ -282,6 +292,9 @@ class SessionController:
|
||||
# Session queues and mutable state.
|
||||
raw_prompt_queue: asyncio.Queue[PromptSubmission] = asyncio.Queue()
|
||||
ready_prompt_queue: asyncio.Queue[ReadyPrompt] = asyncio.Queue()
|
||||
# Submissions dequeued by prompt_worker_loop but not yet resolved; while
|
||||
# non-zero the prompt sources are busy, not drained.
|
||||
prompt_enhancement_inflight = 0
|
||||
|
||||
curated_idx = 0
|
||||
segment_idx = 0
|
||||
@@ -291,13 +304,20 @@ class SessionController:
|
||||
generation_cap_blocked = False
|
||||
auto_extension_blocked_segment_idx: int | None = None
|
||||
prompt_sources_drained_logged = False
|
||||
generation_paused = bool(initial_rollout_prompt and not single_clip_mode and len(curated_prompts) == 0)
|
||||
generation_paused = bool(not manual_continuation_mode and initial_rollout_prompt and not single_clip_mode
|
||||
and len(curated_prompts) == 0)
|
||||
pending_seed_reset = False
|
||||
pending_seed_reset_reason = ""
|
||||
pending_reset_conditioning = False
|
||||
loop_iteration = 0 if generation_paused else 1
|
||||
force_curated_restart_segment = False
|
||||
pending_simple_prompt_submission: PromptSubmission | None = None
|
||||
# The frontend records the opening scene under this id; reuse it so
|
||||
# prompt lifecycle events for the opening prompt reach that record.
|
||||
pending_simple_prompt_submission: PromptSubmission | None = (PromptSubmission(
|
||||
prompt_id=str(init_data.get("initial_rollout_prompt_id") or uuid.uuid4()),
|
||||
raw_prompt=initial_rollout_prompt,
|
||||
created_at_s=time.time(),
|
||||
) if manual_continuation_mode and initial_rollout_prompt else None)
|
||||
single_clip_waiting_for_request = False
|
||||
rollout_waiting_for_rewrite = False
|
||||
initial_rollout_waiting_for_rewrite = generation_paused
|
||||
@@ -306,6 +326,7 @@ class SessionController:
|
||||
project_active = True
|
||||
project_stream_started = False
|
||||
pending_project_end = False
|
||||
segment_prompt_log_warned = False
|
||||
|
||||
def replace_session_init_image(initial_image_payload: object) -> None:
|
||||
nonlocal session_init_image
|
||||
@@ -424,6 +445,7 @@ class SessionController:
|
||||
nonlocal auto_extension_enabled
|
||||
nonlocal loop_generation_enabled
|
||||
nonlocal single_clip_mode
|
||||
nonlocal manual_continuation_mode
|
||||
nonlocal generation_paused
|
||||
nonlocal curated_prompts
|
||||
nonlocal seed_prompt_memory
|
||||
@@ -458,6 +480,7 @@ class SessionController:
|
||||
next_auto_extension_enabled = bool(payload.get("auto_extension_enabled", False))
|
||||
next_loop_generation_enabled = bool(payload.get("loop_generation_enabled", False))
|
||||
next_single_clip_mode = bool(payload.get("single_clip_mode", False))
|
||||
next_manual_continuation_mode = bool(payload.get("manual_continuation_mode", False))
|
||||
|
||||
next_preset_id = str(payload.get("preset_id") or "").strip()
|
||||
if next_preset_id:
|
||||
@@ -510,6 +533,7 @@ class SessionController:
|
||||
auto_extension_enabled = next_auto_extension_enabled
|
||||
loop_generation_enabled = next_loop_generation_enabled
|
||||
single_clip_mode = next_single_clip_mode
|
||||
manual_continuation_mode = next_manual_continuation_mode
|
||||
rewrite_model = next_rewrite_model
|
||||
rewrite_system_prompt_override = (next_rewrite_system_prompt_override)
|
||||
rewrite_user_system_prompt_override = (next_rewrite_user_system_prompt_override)
|
||||
@@ -530,10 +554,15 @@ class SessionController:
|
||||
generation_cap_blocked = False
|
||||
auto_extension_blocked_segment_idx = None
|
||||
prompt_sources_drained_logged = False
|
||||
pending_simple_prompt_submission = None
|
||||
pending_simple_prompt_submission = (PromptSubmission(
|
||||
prompt_id=str(payload.get("initial_rollout_prompt_id") or uuid.uuid4()),
|
||||
raw_prompt=initial_rollout_prompt,
|
||||
created_at_s=time.time(),
|
||||
) if manual_continuation_mode and initial_rollout_prompt else None)
|
||||
single_clip_waiting_for_request = False
|
||||
rollout_waiting_for_rewrite = False
|
||||
generation_paused = bool(initial_rollout_prompt and not single_clip_mode and len(curated_prompts) == 0)
|
||||
generation_paused = bool(not manual_continuation_mode and initial_rollout_prompt
|
||||
and not single_clip_mode and len(curated_prompts) == 0)
|
||||
initial_rollout_waiting_for_rewrite = generation_paused
|
||||
rewrite_restart_pending = False
|
||||
loop_iteration = 0
|
||||
@@ -942,6 +971,7 @@ class SessionController:
|
||||
_resolve_generation_segment_cap(
|
||||
single_clip_mode=single_clip_mode,
|
||||
cap=GENERATION_SEGMENT_CAP,
|
||||
manual_continuation_mode=manual_continuation_mode,
|
||||
),
|
||||
})
|
||||
continue
|
||||
@@ -1005,11 +1035,9 @@ class SessionController:
|
||||
continue
|
||||
|
||||
async def prompt_worker_loop():
|
||||
while not stop_event.is_set():
|
||||
try:
|
||||
submission = await asyncio.wait_for(raw_prompt_queue.get(), timeout=0.1)
|
||||
except asyncio.TimeoutError:
|
||||
continue
|
||||
nonlocal prompt_enhancement_inflight
|
||||
|
||||
async def process_submission(submission: PromptSubmission) -> None:
|
||||
_main_print("INFO", f"Received user prompt for enhancement: {submission.raw_prompt}")
|
||||
prompt_id = submission.prompt_id
|
||||
raw_prompt = submission.raw_prompt
|
||||
@@ -1032,8 +1060,9 @@ class SessionController:
|
||||
await ws_send_json({
|
||||
"type": "error",
|
||||
"message": blocked_raw_prompt_error,
|
||||
"prompt_id": prompt_id,
|
||||
})
|
||||
continue
|
||||
return
|
||||
await log_event(
|
||||
"enhance_request",
|
||||
{
|
||||
@@ -1093,8 +1122,9 @@ class SessionController:
|
||||
await ws_send_json({
|
||||
"type": "error",
|
||||
"message": blocked_final_prompt_error,
|
||||
"prompt_id": prompt_id,
|
||||
})
|
||||
continue
|
||||
return
|
||||
if result.fallback_used or not final_prompt:
|
||||
source = "user_enhancement_failed"
|
||||
_main_print(
|
||||
@@ -1113,7 +1143,7 @@ class SessionController:
|
||||
})
|
||||
# Enhancement is strict JSON-only; do not enqueue raw
|
||||
# prompt when enhancement fails.
|
||||
continue
|
||||
return
|
||||
else:
|
||||
source = "user_enhanced"
|
||||
await ws_send_json({
|
||||
@@ -1130,6 +1160,7 @@ class SessionController:
|
||||
source=source,
|
||||
fallback_used=result.fallback_used,
|
||||
loop_iteration=loop_iteration,
|
||||
raw_prompt=raw_prompt,
|
||||
))
|
||||
else:
|
||||
await ready_prompt_queue.put(
|
||||
@@ -1139,6 +1170,7 @@ class SessionController:
|
||||
source="user_raw",
|
||||
fallback_used=False,
|
||||
loop_iteration=loop_iteration,
|
||||
raw_prompt=raw_prompt,
|
||||
))
|
||||
await ws_send_json({
|
||||
"type": "prompt_ready",
|
||||
@@ -1148,6 +1180,21 @@ class SessionController:
|
||||
"latency_ms": 0.0,
|
||||
})
|
||||
|
||||
while not stop_event.is_set():
|
||||
try:
|
||||
submission = raw_prompt_queue.get_nowait()
|
||||
except asyncio.QueueEmpty:
|
||||
await asyncio.sleep(0.1)
|
||||
continue
|
||||
# Dequeue and increment without an await in between so the
|
||||
# generation loop never sees an empty queue with zero in flight
|
||||
# while this submission is still being enhanced.
|
||||
prompt_enhancement_inflight += 1
|
||||
try:
|
||||
await process_submission(submission)
|
||||
finally:
|
||||
prompt_enhancement_inflight -= 1
|
||||
|
||||
def queue_snapshot() -> dict[str, object]:
|
||||
return {
|
||||
"user_ready": ready_prompt_queue.qsize(),
|
||||
@@ -1300,6 +1347,7 @@ class SessionController:
|
||||
_resolve_generation_segment_cap(
|
||||
single_clip_mode=single_clip_mode,
|
||||
cap=GENERATION_SEGMENT_CAP,
|
||||
manual_continuation_mode=manual_continuation_mode,
|
||||
),
|
||||
})
|
||||
await ws_send_json({
|
||||
@@ -1359,6 +1407,7 @@ class SessionController:
|
||||
_resolve_generation_segment_cap(
|
||||
single_clip_mode=single_clip_mode,
|
||||
cap=GENERATION_SEGMENT_CAP,
|
||||
manual_continuation_mode=manual_continuation_mode,
|
||||
),
|
||||
})
|
||||
if nonlocal_reason == "loop_restart":
|
||||
@@ -1388,8 +1437,9 @@ class SessionController:
|
||||
await raw_prompt_queue.put(pending_simple_prompt_submission)
|
||||
pending_simple_prompt_submission = None
|
||||
|
||||
if (not single_clip_mode and not generation_cap_blocked and not rollout_waiting_for_rewrite
|
||||
and GENERATION_SEGMENT_CAP > 0 and generated_segment_count >= GENERATION_SEGMENT_CAP):
|
||||
if (not single_clip_mode and not manual_continuation_mode and not generation_cap_blocked
|
||||
and not rollout_waiting_for_rewrite and GENERATION_SEGMENT_CAP > 0
|
||||
and generated_segment_count >= GENERATION_SEGMENT_CAP):
|
||||
loop_generation_enabled = False
|
||||
rollout_waiting_for_rewrite = True
|
||||
_main_print(
|
||||
@@ -1542,7 +1592,10 @@ class SessionController:
|
||||
if single_clip_mode:
|
||||
await asyncio.sleep(PROMPT_AUTO_SLEEP_MS / 1000.0)
|
||||
continue
|
||||
if not prompt_sources_drained_logged:
|
||||
# A raw submission still queued or being enhanced will produce a
|
||||
# ready prompt shortly; that is not a drained/blocked state.
|
||||
enhancement_pending = (raw_prompt_queue.qsize() > 0 or prompt_enhancement_inflight > 0)
|
||||
if not prompt_sources_drained_logged and not enhancement_pending:
|
||||
snapshot = queue_snapshot()
|
||||
_main_print(
|
||||
"WARN",
|
||||
@@ -1581,6 +1634,24 @@ class SessionController:
|
||||
total_segments_hint = max(segment_idx, len(curated_prompts))
|
||||
prompt = selected.prompt
|
||||
locked_segment_prompts.append(prompt)
|
||||
if SEGMENT_PROMPT_LOG_PATH:
|
||||
_ts = time.strftime("%Y-%m-%d %H:%M:%S")
|
||||
_lines = [
|
||||
f"\n=== Segment {segment_idx} [{_ts}] source={selected.source} client={client_id[:8]} ===",
|
||||
]
|
||||
if selected.raw_prompt and selected.raw_prompt != prompt:
|
||||
_lines.append(f"User: {selected.raw_prompt}")
|
||||
_lines.append(f"Rewritten: {prompt}")
|
||||
try:
|
||||
await asyncio.to_thread(_append_segment_prompt_log, SEGMENT_PROMPT_LOG_PATH,
|
||||
"\n".join(_lines) + "\n")
|
||||
except Exception as exc:
|
||||
if not segment_prompt_log_warned:
|
||||
segment_prompt_log_warned = True
|
||||
_main_print(
|
||||
"WARN",
|
||||
f"Failed to write segment prompt log {SEGMENT_PROMPT_LOG_PATH}: {exc}",
|
||||
)
|
||||
if (auto_extension_blocked_segment_idx is not None
|
||||
and auto_extension_blocked_segment_idx <= segment_idx):
|
||||
auto_extension_blocked_segment_idx = None
|
||||
|
||||
@@ -19,3 +19,4 @@ class ReadyPrompt:
|
||||
fallback_used: bool = False
|
||||
seed_prompt_index: int | None = None
|
||||
loop_iteration: int | None = None
|
||||
raw_prompt: str | None = None
|
||||
|
||||
@@ -3,6 +3,7 @@ from __future__ import annotations
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
TESTS_DIR = Path(__file__).resolve().parent
|
||||
DREAMVERSE_PACKAGE_DIR = TESTS_DIR.parent
|
||||
DREAMVERSE_APP_DIR = DREAMVERSE_PACKAGE_DIR.parent
|
||||
|
||||
@@ -2,14 +2,14 @@ from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
SERVER_DIR = Path(__file__).resolve().parents[1]
|
||||
|
||||
|
||||
def _load_config_module() -> ModuleType:
|
||||
def _load_config_module():
|
||||
spec = importlib.util.spec_from_file_location(
|
||||
"server_config_test_module",
|
||||
SERVER_DIR / "config.py",
|
||||
@@ -21,7 +21,7 @@ def _load_config_module() -> ModuleType:
|
||||
return module
|
||||
|
||||
|
||||
def _set_required_prompt_keys(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
def _set_required_prompt_keys(monkeypatch):
|
||||
monkeypatch.setenv("CEREBRAS_API_KEY", "cerebras-key")
|
||||
monkeypatch.setenv("GROQ_API_KEY", "groq-key")
|
||||
|
||||
@@ -53,7 +53,9 @@ def test_config_defaults_to_cerebras_with_parallel_groq_fallback_stage(monkeypat
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.PROMPT_PROVIDER == "cerebras"
|
||||
assert module.PROMPT_PROVIDER_RUNTIME_STAGES == (("cerebras", "groq"), )
|
||||
assert module.PROMPT_PROVIDER_RUNTIME_STAGES == (
|
||||
("cerebras", "groq"),
|
||||
)
|
||||
assert module.PROMPT_PROVIDER_PRIORITY == (
|
||||
"cerebras",
|
||||
"groq",
|
||||
@@ -84,7 +86,9 @@ def test_config_ignores_legacy_groq_primary_override(monkeypatch):
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.PROMPT_PROVIDER == "cerebras"
|
||||
assert module.PROMPT_PROVIDER_RUNTIME_STAGES == (("cerebras", "groq"), )
|
||||
assert module.PROMPT_PROVIDER_RUNTIME_STAGES == (
|
||||
("cerebras", "groq"),
|
||||
)
|
||||
assert module.PROMPT_PROVIDER_PRIORITY == (
|
||||
"cerebras",
|
||||
"groq",
|
||||
@@ -102,17 +106,24 @@ def test_config_uses_local_overlay_paths_when_devtools_enabled(monkeypatch, tmp_
|
||||
|
||||
assert module.DEVTOOLS_ENABLED is True
|
||||
assert module.FRONTEND_ROOT.as_posix().endswith("apps/dreamverse/web")
|
||||
assert module.PROMPT_ENHANCE_SYSTEM_PROMPT_PATH.endswith("dreamverse/prompts.local/next_segment_system_prompt.md")
|
||||
assert module.PROMPT_ENHANCE_SYSTEM_PROMPT_PATH.endswith(
|
||||
"dreamverse/prompts.local/next_segment_system_prompt.md"
|
||||
)
|
||||
assert module.PROMPT_ENHANCE_SYSTEM_PROMPT_FALLBACK_PATH.endswith(
|
||||
"dreamverse/prompts/next_segment_system_prompt.md")
|
||||
"dreamverse/prompts/next_segment_system_prompt.md"
|
||||
)
|
||||
assert module.PROMPT_REWRITE_USER_SYSTEM_PROMPT_PATH.endswith(
|
||||
"dreamverse/prompts.local/rewrite_user_system_prompt.md")
|
||||
"dreamverse/prompts.local/rewrite_user_system_prompt.md"
|
||||
)
|
||||
assert module.PROMPT_REWRITE_USER_SYSTEM_PROMPT_FALLBACK_PATH.endswith(
|
||||
"dreamverse/prompts/rewrite_user_system_prompt.md")
|
||||
"dreamverse/prompts/rewrite_user_system_prompt.md"
|
||||
)
|
||||
assert module.CURATED_PRESETS_FILE_PATH.endswith(
|
||||
"apps/dreamverse/web/prompts.local/selected_ltx2_continuation_story_presets.json")
|
||||
"apps/dreamverse/web/prompts.local/selected_ltx2_continuation_story_presets.json"
|
||||
)
|
||||
assert module.CURATED_PRESETS_FALLBACK_FILE_PATH.endswith(
|
||||
"apps/dreamverse/web/prompts/selected_ltx2_continuation_story_presets.json")
|
||||
"apps/dreamverse/web/prompts/selected_ltx2_continuation_story_presets.json"
|
||||
)
|
||||
assert module.FRONTEND_STATIC_DIR_CANDIDATES[:2] == (
|
||||
str(module.FRONTEND_ROOT / "out"),
|
||||
str(module.FRONTEND_ROOT / "dist"),
|
||||
@@ -139,50 +150,25 @@ def test_config_enables_prompt_safety_when_requested(monkeypatch):
|
||||
|
||||
def test_config_uses_five_minute_session_timeout(monkeypatch):
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
monkeypatch.delenv("DREAMVERSE_SESSION_TIMEOUT_SECONDS", raising=False)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.SESSION_TIMEOUT_SECONDS == 300
|
||||
|
||||
|
||||
def test_config_session_timeout_env_override(monkeypatch):
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
monkeypatch.setenv("DREAMVERSE_SESSION_TIMEOUT_SECONDS", "1800")
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.SESSION_TIMEOUT_SECONDS == 1800
|
||||
|
||||
|
||||
def test_config_rejects_invalid_prompt_provider(monkeypatch):
|
||||
monkeypatch.setenv("FASTVIDEO_PROMPT_PROVIDER", "unsupported")
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
|
||||
with pytest.raises(RuntimeError, match="Invalid FASTVIDEO_PROMPT_PROVIDER"):
|
||||
_load_config_module()
|
||||
|
||||
|
||||
def test_config_registers_vsa_datafree_fasth3_profile(monkeypatch):
|
||||
"""The FastH3 registry entry owns the complete fixed Preview recipe."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.MODEL_REGISTRY["fast-h3"] == {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
def test_config_uses_fasth3_sequence_parallel_default(monkeypatch):
|
||||
"""Selecting FastH3 defaults DreamVerse to its four-GPU topology."""
|
||||
_set_required_prompt_keys(monkeypatch)
|
||||
monkeypatch.setenv("DREAMVERSE_MODEL_ID", "fast-h3")
|
||||
monkeypatch.delenv("DREAMVERSE_SP_SIZE", raising=False)
|
||||
|
||||
module = _load_config_module()
|
||||
|
||||
assert module.ACTIVE_MODEL_ID == "fast-h3"
|
||||
assert module.MODEL_CONFIG["generation_backend"] == "minimax_h3"
|
||||
assert module.DREAMVERSE_SP_SIZE == 4
|
||||
|
||||
@@ -9,7 +9,6 @@ from fastapi.testclient import TestClient
|
||||
import fastvideo.entrypoints.streaming as streaming_entrypoints
|
||||
import pytest
|
||||
|
||||
|
||||
def _install_stack03_import_stubs(monkeypatch):
|
||||
"""Keep entrypoint tests focused while later-stack runtime modules are absent."""
|
||||
if not hasattr(streaming_entrypoints, "build_health_router"):
|
||||
@@ -18,7 +17,6 @@ def _install_stack03_import_stubs(monkeypatch):
|
||||
gpu_pool_stub = types.ModuleType("dreamverse.gpu_pool")
|
||||
|
||||
class GPUPool:
|
||||
|
||||
def __init__(self, _gpu_ids):
|
||||
pass
|
||||
|
||||
@@ -51,7 +49,6 @@ def _install_stack03_import_stubs(monkeypatch):
|
||||
controller_stub = types.ModuleType("dreamverse.session.controller")
|
||||
|
||||
class SessionController:
|
||||
|
||||
def __init__(self, **_kwargs):
|
||||
pass
|
||||
|
||||
@@ -78,12 +75,15 @@ def _run_cli(module, monkeypatch, argv: list[str]) -> list[dict[str, object]]:
|
||||
calls: list[dict[str, object]] = []
|
||||
uvicorn_stub = types.ModuleType("uvicorn")
|
||||
|
||||
def run(app, host: str, port: int) -> None:
|
||||
calls.append({
|
||||
"app": app,
|
||||
"host": host,
|
||||
"port": port,
|
||||
})
|
||||
def run(app, host: str, port: int, **kwargs) -> None:
|
||||
calls.append(
|
||||
{
|
||||
"app": app,
|
||||
"host": host,
|
||||
"port": port,
|
||||
**kwargs,
|
||||
}
|
||||
)
|
||||
|
||||
uvicorn_stub.run = run
|
||||
monkeypatch.setitem(sys.modules, "uvicorn", uvicorn_stub)
|
||||
@@ -100,11 +100,14 @@ def test_server_cli_defaults_to_local_web_port(monkeypatch):
|
||||
server_main = _import_server_main(monkeypatch)
|
||||
calls = _run_cli(server_main, monkeypatch, ["dreamverse-server"])
|
||||
|
||||
assert calls == [{
|
||||
"app": server_main.app,
|
||||
"host": "0.0.0.0",
|
||||
"port": 8009,
|
||||
}]
|
||||
assert calls == [
|
||||
{
|
||||
"app": server_main.app,
|
||||
"host": "0.0.0.0",
|
||||
"port": 8009,
|
||||
"ws_max_size": 32 * 1024 * 1024,
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def test_server_cli_allows_explicit_host_and_port(monkeypatch):
|
||||
@@ -115,11 +118,14 @@ def test_server_cli_allows_explicit_host_and_port(monkeypatch):
|
||||
["dreamverse-server", "--host", "127.0.0.1", "--port", "8123"],
|
||||
)
|
||||
|
||||
assert calls == [{
|
||||
"app": server_main.app,
|
||||
"host": "127.0.0.1",
|
||||
"port": 8123,
|
||||
}]
|
||||
assert calls == [
|
||||
{
|
||||
"app": server_main.app,
|
||||
"host": "127.0.0.1",
|
||||
"port": 8123,
|
||||
"ws_max_size": 32 * 1024 * 1024,
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def test_server_does_not_expose_backend_source_as_static_assets(monkeypatch):
|
||||
@@ -139,11 +145,14 @@ def test_mock_server_cli_defaults_to_local_web_port(monkeypatch):
|
||||
["dreamverse-mock-server"],
|
||||
)
|
||||
|
||||
assert calls == [{
|
||||
"app": mock_server.app,
|
||||
"host": "0.0.0.0",
|
||||
"port": 8009,
|
||||
}]
|
||||
assert calls == [
|
||||
{
|
||||
"app": mock_server.app,
|
||||
"host": "0.0.0.0",
|
||||
"port": 8009,
|
||||
"ws_max_size": 32 * 1024 * 1024,
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def test_mock_server_cli_updates_latency(monkeypatch):
|
||||
@@ -156,11 +165,14 @@ def test_mock_server_cli_updates_latency(monkeypatch):
|
||||
["dreamverse-mock-server", "--latency", "321", "--port", "8111"],
|
||||
)
|
||||
|
||||
assert calls == [{
|
||||
"app": mock_server.app,
|
||||
"host": "0.0.0.0",
|
||||
"port": 8111,
|
||||
}]
|
||||
assert calls == [
|
||||
{
|
||||
"app": mock_server.app,
|
||||
"host": "0.0.0.0",
|
||||
"port": 8111,
|
||||
"ws_max_size": 32 * 1024 * 1024,
|
||||
}
|
||||
]
|
||||
assert mock_server.LATENCY_MS == 321
|
||||
finally:
|
||||
mock_server.LATENCY_MS = old_latency_ms
|
||||
|
||||
@@ -1,9 +0,0 @@
|
||||
from dreamverse.generation_worker import _create_generation_backend
|
||||
from dreamverse.ltx2_generation import LTX2GenerationBackend
|
||||
|
||||
|
||||
def test_create_generation_backend_ltx2_module_import():
|
||||
backend = _create_generation_backend("ltx2", gpu_id=3)
|
||||
|
||||
assert isinstance(backend, LTX2GenerationBackend)
|
||||
assert backend.gpu_id == 3
|
||||
@@ -7,6 +7,7 @@ from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
import dreamverse.gpu_pool as gpu_pool
|
||||
|
||||
|
||||
@@ -63,14 +64,6 @@ def test_get_available_gpus_defaults_to_first_visible_device(monkeypatch):
|
||||
assert gpu_pool.get_available_gpus() == [3]
|
||||
|
||||
|
||||
def test_get_available_gpus_defaults_to_active_model_sequence_parallel_size(monkeypatch):
|
||||
monkeypatch.setenv("CUDA_VISIBLE_DEVICES", "0,1,2,3,4")
|
||||
monkeypatch.delenv("FASTVIDEO_GPU_COUNT", raising=False)
|
||||
monkeypatch.setattr(gpu_pool, "DREAMVERSE_SP_SIZE", 4)
|
||||
|
||||
assert gpu_pool.get_available_gpus() == [0, 1, 2, 3]
|
||||
|
||||
|
||||
def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
monkeypatch.delenv("CUDA_VISIBLE_DEVICES", raising=False)
|
||||
monkeypatch.setenv("FASTVIDEO_GPU_COUNT", "zero")
|
||||
@@ -79,23 +72,6 @@ def test_get_available_gpus_rejects_invalid_gpu_count(monkeypatch):
|
||||
gpu_pool.get_available_gpus()
|
||||
|
||||
|
||||
def test_join_user_failed_reload_marks_model_uninitialized(monkeypatch):
|
||||
"""A failed model reload forces the next join to reload a model."""
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.current_model_id = "fast-ltx2"
|
||||
|
||||
async def fake_send_command(command, timeout):
|
||||
del command, timeout
|
||||
return gpu_pool.WorkerError(user_id="__reload__", message="load failed")
|
||||
|
||||
monkeypatch.setattr(slot, "_send_command", fake_send_command)
|
||||
|
||||
with pytest.raises(RuntimeError, match="Model reload failed"):
|
||||
asyncio.run(slot.join_user("client-id", model_id="fast-h3"))
|
||||
|
||||
assert slot.current_model_id is None
|
||||
|
||||
|
||||
def test_send_command_raises_on_worker_death():
|
||||
"""A worker that consumes a command and exits without replying must
|
||||
surface as RuntimeError via sentinel detection, not after the long
|
||||
@@ -109,7 +85,9 @@ def test_send_command_raises_on_worker_death():
|
||||
cmd_q = ctx.Queue()
|
||||
resp_q = ctx.Queue()
|
||||
|
||||
proc = ctx.Process(target=_child_consume_and_exit, args=(cmd_q, resp_q))
|
||||
proc = ctx.Process(
|
||||
target=_child_consume_and_exit, args=(cmd_q, resp_q)
|
||||
)
|
||||
proc.start()
|
||||
|
||||
# Wait for the spawn child to fully boot. Allow generous time —
|
||||
@@ -117,9 +95,9 @@ def test_send_command_raises_on_worker_death():
|
||||
ready = resp_q.get(timeout=30.0)
|
||||
assert ready == "READY"
|
||||
|
||||
async def runner() -> None:
|
||||
async def runner():
|
||||
slot = gpu_pool.GPUSlot(gpu_id=0, cuda_device="0")
|
||||
slot.process = proc # type: ignore[assignment]
|
||||
slot.process = proc
|
||||
slot.command_queue = cmd_q
|
||||
slot.response_queue = resp_q
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@ ALLOWED_PREFIXES = (
|
||||
"fastvideo.entrypoints.video_generator",
|
||||
"fastvideo.configs",
|
||||
)
|
||||
ALLOWED_EXACT = ("fastvideo", )
|
||||
ALLOWED_EXACT = ("fastvideo",)
|
||||
FORBIDDEN_PREFIXES = (
|
||||
"fastvideo.pipelines",
|
||||
"fastvideo.models",
|
||||
@@ -17,11 +17,11 @@ FORBIDDEN_PREFIXES = (
|
||||
)
|
||||
ALLOWED_INTERNAL_IMPORTS = {
|
||||
(
|
||||
"ltx2_generation.py",
|
||||
"video_generation.py",
|
||||
"fastvideo.models.audio.ltx2_audio_processing",
|
||||
),
|
||||
(
|
||||
"ltx2_generation.py",
|
||||
"video_generation.py",
|
||||
"fastvideo.models.loader.component_loader",
|
||||
),
|
||||
}
|
||||
@@ -38,13 +38,19 @@ def test_dreamverse_server_imports_only_public_fastvideo_surfaces() -> None:
|
||||
except SyntaxError as task_exc:
|
||||
raise AssertionError(f"Failed to parse {path}") from task_exc
|
||||
for node in ast.walk(tree):
|
||||
names = ([a.name for a in node.names] if isinstance(node, ast.Import) else
|
||||
[node.module] if isinstance(node, ast.ImportFrom) and node.module else [])
|
||||
names = (
|
||||
[a.name for a in node.names] if isinstance(node, ast.Import)
|
||||
else [node.module] if isinstance(node, ast.ImportFrom) and node.module
|
||||
else []
|
||||
)
|
||||
for name in names:
|
||||
if not name:
|
||||
continue
|
||||
rel_path = str(path.relative_to(root))
|
||||
if (name.startswith(FORBIDDEN_PREFIXES) and (rel_path, name) not in ALLOWED_INTERNAL_IMPORTS):
|
||||
if (
|
||||
name.startswith(FORBIDDEN_PREFIXES)
|
||||
and (rel_path, name) not in ALLOWED_INTERNAL_IMPORTS
|
||||
):
|
||||
bad.append((str(path.relative_to(root)), getattr(node, "lineno", 0), name))
|
||||
|
||||
assert bad == [], f"Forbidden internal imports: {bad}"
|
||||
|
||||
@@ -1,247 +0,0 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from types import SimpleNamespace
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
import dreamverse.generation_worker as generation_worker
|
||||
from dreamverse.minimax_h3_generation import MiniMaxH3GenerationBackend
|
||||
|
||||
|
||||
FASTH3_MODEL_CONFIG = {
|
||||
"name": "FastH3",
|
||||
"generation_backend": "minimax_h3",
|
||||
"default_sp_size": 4,
|
||||
"model_path": "MiniMaxAI/MiniMax-H3",
|
||||
"adapter_repo": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"adapter_filename": "vsa-datafree/adapter_model.safetensors",
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"height": 768,
|
||||
"width": 1344,
|
||||
"num_frames": 124,
|
||||
"num_inference_steps": 5,
|
||||
"seed": 1000,
|
||||
}
|
||||
|
||||
|
||||
class _RecordingGenerator:
|
||||
"""Record typed requests and return small synchronized media fixtures."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.requests: list[Any] = []
|
||||
self.conditioning_pixels: list[np.ndarray | None] = []
|
||||
|
||||
def generate(self, request):
|
||||
"""Capture the request and return two tiny video frames with audio."""
|
||||
self.requests.append(request)
|
||||
conditioning_image = request.inputs.pil_image
|
||||
self.conditioning_pixels.append(
|
||||
None if conditioning_image is None else np.asarray(conditioning_image).copy())
|
||||
frames = [
|
||||
np.full((2, 3, 3), 10, dtype=np.uint8),
|
||||
np.full((2, 3, 3), 20, dtype=np.uint8),
|
||||
]
|
||||
return SimpleNamespace(
|
||||
frames=frames,
|
||||
audio=np.zeros((2, 16), dtype=np.float32),
|
||||
audio_sample_rate=44100,
|
||||
generation_time=0.25,
|
||||
)
|
||||
|
||||
|
||||
def test_initialize_builds_vsa_datafree_fasth3_generator(monkeypatch):
|
||||
"""Initialization translates the DreamVerse profile into typed FastVideo config."""
|
||||
from fastvideo import VideoGenerator
|
||||
|
||||
captured = {}
|
||||
fake_generator = SimpleNamespace(shutdown=lambda: None)
|
||||
|
||||
def fake_from_config(config):
|
||||
captured["config"] = config
|
||||
return fake_generator
|
||||
|
||||
def fake_download(**kwargs):
|
||||
captured["download"] = kwargs
|
||||
return f"/models/{kwargs['filename']}"
|
||||
|
||||
monkeypatch.setattr("huggingface_hub.hf_hub_download", fake_download)
|
||||
monkeypatch.setattr(VideoGenerator, "from_config", fake_from_config)
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.DREAMVERSE_SP_SIZE", 4)
|
||||
monkeypatch.setenv("FASTVIDEO_ATTENTION_BACKEND", "test-attention")
|
||||
monkeypatch.setenv("FASTVIDEO_FA4", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_MINIMAX_H3_FUSIONS", "0")
|
||||
monkeypatch.setenv("FASTVIDEO_VSA_SM100A", "1")
|
||||
monkeypatch.setenv("FASTVIDEO_INFERENCE_TORCH_COMPILE", "1")
|
||||
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
monkeypatch.setattr(backend, "_gpu_mem", lambda: "alloc=0.00GiB, reserved=0.00GiB")
|
||||
backend.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
config = captured["config"]
|
||||
assert captured["download"] == {
|
||||
"repo_id": "FastVideo/FastVideo-FastH3-4-step-Preview-v1-LoRA",
|
||||
"filename": "vsa-datafree/adapter_model.safetensors",
|
||||
}
|
||||
assert config.model_path == "MiniMaxAI/MiniMax-H3"
|
||||
assert config.pipeline.components.lora_path.endswith("vsa-datafree/adapter_model.safetensors")
|
||||
assert config.pipeline.components.lora_strength == 1.0
|
||||
assert config.pipeline.experimental == {
|
||||
"attention_backend": "VIDEO_SPARSE_ATTN_H3",
|
||||
"inference_torch_compile": False,
|
||||
"vae_parallel_decode": True,
|
||||
"vae_parallel_decode_strategy": "gather",
|
||||
"VSA_sparsity": 0.9,
|
||||
"VSA_tile_size": 64,
|
||||
}
|
||||
assert config.engine.num_gpus == 4
|
||||
assert config.engine.parallelism.tp_size == 1
|
||||
assert config.engine.parallelism.sp_size == 4
|
||||
assert config.engine.offload.dit is False
|
||||
assert config.engine.offload.dit_layerwise is False
|
||||
assert config.engine.offload.text_encoder is True
|
||||
assert config.engine.offload.vae is True
|
||||
assert config.engine.compile.vae_enabled is True
|
||||
assert config.engine.use_fsdp_inference is False
|
||||
assert os.environ["FASTVIDEO_ATTENTION_BACKEND"] == "VIDEO_SPARSE_ATTN_H3"
|
||||
assert os.environ["FASTVIDEO_FA4"] == "1"
|
||||
assert os.environ["FASTVIDEO_MINIMAX_H3_FUSIONS"] == "all"
|
||||
assert os.environ["FASTVIDEO_VSA_SM100A"] == "0"
|
||||
assert "FASTVIDEO_INFERENCE_TORCH_COMPILE" not in os.environ
|
||||
|
||||
|
||||
def test_initialize_selects_declared_generation_backend(monkeypatch):
|
||||
"""The GPU worker constructs the backend that the active model profile declares."""
|
||||
from unittest.mock import Mock
|
||||
|
||||
selected_backend = Mock()
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: selected_backend,
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend_name == "minimax_h3"
|
||||
assert worker.backend is selected_backend
|
||||
selected_backend.initialize.assert_called_once_with(FASTH3_MODEL_CONFIG)
|
||||
|
||||
|
||||
def test_initialize_failure_clears_backend_ownership(monkeypatch):
|
||||
"""A failed family change leaves the GPU worker explicitly uninitialized."""
|
||||
ltx_backend = SimpleNamespace(initialize=lambda config: None, shutdown=lambda: None)
|
||||
|
||||
def fail_initialize(config):
|
||||
del config
|
||||
raise RuntimeError("load failed")
|
||||
|
||||
fasth3_backend = SimpleNamespace(
|
||||
initialize=fail_initialize,
|
||||
shutdown=lambda: None,
|
||||
)
|
||||
backends = {
|
||||
"ltx2": ltx_backend,
|
||||
"minimax_h3": fasth3_backend,
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
generation_worker,
|
||||
"_create_generation_backend",
|
||||
lambda backend_name, gpu_id: backends[backend_name],
|
||||
)
|
||||
worker = generation_worker.VideoGenerationWorker(gpu_id=3)
|
||||
worker.initialize({"generation_backend": "ltx2"})
|
||||
|
||||
with pytest.raises(RuntimeError, match="load failed"):
|
||||
worker.initialize(FASTH3_MODEL_CONFIG)
|
||||
|
||||
assert worker.backend is None
|
||||
assert worker.backend_name is None
|
||||
assert worker.model_config == {"generation_backend": "ltx2"}
|
||||
|
||||
|
||||
def test_generate_step_uses_last_frame_for_continuation(monkeypatch):
|
||||
"""A later segment receives the prior segment's last decoded frame."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
first_result = backend.generate_step(
|
||||
"first prompt",
|
||||
segment_idx=1,
|
||||
image_path=None,
|
||||
reset_conditioning=True,
|
||||
)
|
||||
second_result = backend.generate_step(
|
||||
"second prompt",
|
||||
segment_idx=2,
|
||||
image_path=None,
|
||||
reset_conditioning=False,
|
||||
)
|
||||
|
||||
first_request = backend.generator.requests[0]
|
||||
assert first_request.inputs.pil_image is None
|
||||
assert first_request.negative_prompt == ""
|
||||
assert first_request.sampling.height == 768
|
||||
assert first_request.sampling.width == 1344
|
||||
assert first_request.sampling.num_frames == 124
|
||||
assert first_request.sampling.num_inference_steps == 5
|
||||
assert first_request.sampling.fps == 24
|
||||
assert first_request.sampling.guidance_scale == 1.0
|
||||
assert first_request.sampling.batch_cfg is False
|
||||
assert first_request.sampling.seed == 1000
|
||||
assert first_request.output.save_video is False
|
||||
assert first_request.output.return_frames is True
|
||||
assert backend.generator.conditioning_pixels[1].tolist() == np.full((2, 3, 3), 20).tolist()
|
||||
assert first_result.head_trim_frames == 0
|
||||
assert first_result.head_trim_audio_frames == 0
|
||||
assert second_result.head_trim_frames == 1
|
||||
assert second_result.head_trim_audio_frames == 1
|
||||
assert second_result.audio_sample_rate == 44100
|
||||
|
||||
|
||||
def test_generate_step_reset_uses_text_to_video_path(monkeypatch):
|
||||
"""Resetting continuation produces an unconditioned text-to-video request."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
backend.generate_step("first prompt", 1, None, True)
|
||||
reset_result = backend.generate_step("reset prompt", 2, None, True)
|
||||
|
||||
assert backend.generator.requests[-1].inputs.pil_image is None
|
||||
assert reset_result.head_trim_frames == 0
|
||||
assert reset_result.head_trim_audio_frames == 0
|
||||
|
||||
|
||||
def test_generate_step_missing_continuation_frame(monkeypatch):
|
||||
"""A later segment fails when no reset or retained frame defines its input."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
|
||||
with pytest.raises(RuntimeError, match="requires a retained continuation frame"):
|
||||
backend.generate_step("later prompt", 2, None, False)
|
||||
|
||||
assert backend.generator.requests == []
|
||||
|
||||
|
||||
def test_warmup_exercises_text_and_first_frame_paths(monkeypatch):
|
||||
"""Warmup covers both request shapes used by a DreamVerse session."""
|
||||
backend = MiniMaxH3GenerationBackend(gpu_id=0)
|
||||
backend.model_config = dict(FASTH3_MODEL_CONFIG)
|
||||
backend.generator = _RecordingGenerator()
|
||||
monkeypatch.setattr("dreamverse.minimax_h3_generation.torch.cuda.synchronize", lambda: None)
|
||||
|
||||
timings = backend.warmup("warmup prompt")
|
||||
|
||||
assert backend.generator.conditioning_pixels[0] is None
|
||||
assert backend.generator.conditioning_pixels[1] is not None
|
||||
assert backend.continuation_image is None
|
||||
assert "warmup_text_to_video_ms" in timings
|
||||
assert "warmup_first_frame_to_video_ms" in timings
|
||||
@@ -6,6 +6,7 @@ import os
|
||||
|
||||
from fastapi import WebSocketDisconnect
|
||||
|
||||
|
||||
os.environ.setdefault("CEREBRAS_API_KEY", "dummy")
|
||||
os.environ.setdefault("GROQ_API_KEY", "dummy")
|
||||
|
||||
@@ -13,7 +14,6 @@ import dreamverse.mock_server as mock_server
|
||||
|
||||
|
||||
class _FakeWebSocket:
|
||||
|
||||
def __init__(self, messages: list[tuple[float, dict[str, object]]]):
|
||||
self._messages = messages
|
||||
self._index = 0
|
||||
@@ -49,34 +49,34 @@ def test_mock_server_matches_current_single5s_protocol():
|
||||
mock_server.MOCK_SEGMENT_BYTES = b"mock-fmp4-bytes"
|
||||
mock_server.LATENCY_MS = 1
|
||||
|
||||
ws = _FakeWebSocket([
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "simple_prompt_1",
|
||||
"curated_prompts": ["selected prompt"],
|
||||
"single_clip_mode": True,
|
||||
"enhancement_enabled": False,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.01,
|
||||
{
|
||||
"type": "simple_generate",
|
||||
"preset_id": "simple_custom_prompt",
|
||||
"prompt_id": "simple_custom_prompt",
|
||||
"prompt": "custom prompt",
|
||||
"enhancement_enabled": True,
|
||||
"initial_image": None,
|
||||
},
|
||||
),
|
||||
(0.20, {
|
||||
"type": "leave"
|
||||
}),
|
||||
])
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "simple_prompt_1",
|
||||
"curated_prompts": ["selected prompt"],
|
||||
"single_clip_mode": True,
|
||||
"enhancement_enabled": False,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.01,
|
||||
{
|
||||
"type": "simple_generate",
|
||||
"preset_id": "simple_custom_prompt",
|
||||
"prompt_id": "simple_custom_prompt",
|
||||
"prompt": "custom prompt",
|
||||
"enhancement_enabled": True,
|
||||
"initial_image": None,
|
||||
},
|
||||
),
|
||||
(0.20, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
@@ -92,14 +92,24 @@ def test_mock_server_matches_current_single5s_protocol():
|
||||
assert message_types.count("ltx2_stream_complete") == 2
|
||||
assert "prompt_sources_blocked" not in message_types
|
||||
|
||||
segment_start_events = [payload for payload in ws.sent_json if payload["type"] == "ltx2_segment_start"]
|
||||
gpu_assigned_event = next(payload for payload in ws.sent_json if payload["type"] == "gpu_assigned")
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
gpu_assigned_event = next(
|
||||
payload for payload in ws.sent_json if payload["type"] == "gpu_assigned"
|
||||
)
|
||||
assert gpu_assigned_event["session_timeout"] == mock_server.SESSION_TIMEOUT_SECONDS
|
||||
assert [payload["segment_idx"] for payload in segment_start_events] == [1, 1]
|
||||
assert segment_start_events[0]["prompt"] == "selected prompt"
|
||||
assert segment_start_events[1]["prompt"] == "custom prompt"
|
||||
|
||||
step_complete_events = [payload for payload in ws.sent_json if payload["type"] == "step_complete"]
|
||||
step_complete_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "step_complete"
|
||||
]
|
||||
assert len(step_complete_events) == 2
|
||||
assert step_complete_events[0]["latency_ms"] == {
|
||||
"total": 121.0,
|
||||
@@ -124,29 +134,29 @@ def test_mock_server_regular_cap_waits_for_rewrite_rollout():
|
||||
mock_server.LATENCY_MS = 1
|
||||
mock_server.GENERATION_SEGMENT_CAP = 1
|
||||
|
||||
ws = _FakeWebSocket([
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.02,
|
||||
{
|
||||
"type": "rewrite_seed_prompts",
|
||||
"rewrite_instruction": "start a new rollout",
|
||||
},
|
||||
),
|
||||
(0.20, {
|
||||
"type": "leave"
|
||||
}),
|
||||
])
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.02,
|
||||
{
|
||||
"type": "rewrite_seed_prompts",
|
||||
"rewrite_instruction": "start a new rollout",
|
||||
},
|
||||
),
|
||||
(0.20, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
@@ -156,7 +166,11 @@ def test_mock_server_regular_cap_waits_for_rewrite_rollout():
|
||||
assert "generation_cap_reached" not in message_types
|
||||
assert "prompt_sources_blocked" not in message_types
|
||||
|
||||
segment_start_events = [payload for payload in ws.sent_json if payload["type"] == "ltx2_segment_start"]
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
assert [payload["segment_idx"] for payload in segment_start_events] == [1, 1]
|
||||
assert segment_start_events[0]["prompt"] == "segment one"
|
||||
assert segment_start_events[1]["prompt"] == "segment one [start a new rollout]"
|
||||
@@ -173,40 +187,54 @@ def test_mock_server_rewrite_during_active_segment_restarts_from_first_rewritten
|
||||
mock_server.MOCK_SEGMENT_BYTES = b"mock-fmp4-bytes"
|
||||
mock_server.LATENCY_MS = 100
|
||||
|
||||
ws = _FakeWebSocket([
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one", "segment two"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.02,
|
||||
{
|
||||
"type": "rewrite_seed_prompts",
|
||||
"rewrite_instruction": "restart from rewrite",
|
||||
},
|
||||
),
|
||||
(0.40, {
|
||||
"type": "leave"
|
||||
}),
|
||||
])
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one", "segment two"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.02,
|
||||
{
|
||||
"type": "rewrite_seed_prompts",
|
||||
"rewrite_instruction": "restart from rewrite",
|
||||
},
|
||||
),
|
||||
(0.40, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
segment_start_events = [payload for payload in ws.sent_json if payload["type"] == "ltx2_segment_start"]
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
assert [payload["prompt"] for payload in segment_start_events[:2]] == [
|
||||
"segment one",
|
||||
"segment one [restart from rewrite]",
|
||||
]
|
||||
assert all(payload["prompt"] != "segment two" for payload in segment_start_events[1:])
|
||||
reset_events = [payload for payload in ws.sent_json if payload.get("type") == "seed_prompts_reset_applied"]
|
||||
assert any(payload.get("reason") == "rewrite_during_generation" for payload in reset_events)
|
||||
assert all(
|
||||
payload["prompt"] != "segment two"
|
||||
for payload in segment_start_events[1:]
|
||||
)
|
||||
reset_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload.get("type") == "seed_prompts_reset_applied"
|
||||
]
|
||||
assert any(
|
||||
payload.get("reason") == "rewrite_during_generation"
|
||||
for payload in reset_events
|
||||
)
|
||||
finally:
|
||||
mock_server.MOCK_SEGMENT_BYTES = old_segment_bytes
|
||||
mock_server.LATENCY_MS = old_latency_ms
|
||||
@@ -219,24 +247,24 @@ def test_mock_server_supports_initial_custom_rollout_prompt():
|
||||
mock_server.MOCK_SEGMENT_BYTES = b"mock-fmp4-bytes"
|
||||
mock_server.LATENCY_MS = 1
|
||||
|
||||
ws = _FakeWebSocket([
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "custom_editable",
|
||||
"preset_label": "Custom rollout",
|
||||
"curated_prompts": [],
|
||||
"initial_rollout_prompt": "A moonbase corridor thriller with flooding",
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.20, {
|
||||
"type": "leave"
|
||||
}),
|
||||
])
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "custom_editable",
|
||||
"preset_label": "Custom rollout",
|
||||
"curated_prompts": [],
|
||||
"initial_rollout_prompt": "A moonbase corridor thriller with flooding",
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.20, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
@@ -248,9 +276,161 @@ def test_mock_server_supports_initial_custom_rollout_prompt():
|
||||
assert "ltx2_stream_start" in message_types
|
||||
assert "prompt_sources_blocked" not in message_types
|
||||
|
||||
segment_start_events = [payload for payload in ws.sent_json if payload["type"] == "ltx2_segment_start"]
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
assert segment_start_events
|
||||
assert segment_start_events[0]["prompt"] == ("A moonbase corridor thriller with flooding [segment 1]")
|
||||
assert segment_start_events[0]["prompt"] == (
|
||||
"A moonbase corridor thriller with flooding [segment 1]"
|
||||
)
|
||||
finally:
|
||||
mock_server.MOCK_SEGMENT_BYTES = old_segment_bytes
|
||||
mock_server.LATENCY_MS = old_latency_ms
|
||||
|
||||
|
||||
def test_mock_server_manual_mode_streams_initial_prompt_without_rewrite_or_cap():
|
||||
old_segment_bytes = mock_server.MOCK_SEGMENT_BYTES
|
||||
old_latency_ms = mock_server.LATENCY_MS
|
||||
old_generation_segment_cap = mock_server.GENERATION_SEGMENT_CAP
|
||||
try:
|
||||
mock_server.MOCK_SEGMENT_BYTES = b"mock-fmp4-bytes"
|
||||
mock_server.LATENCY_MS = 1
|
||||
mock_server.GENERATION_SEGMENT_CAP = 1
|
||||
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "custom_editable",
|
||||
"preset_label": "Custom rollout",
|
||||
"curated_prompts": [],
|
||||
"initial_rollout_prompt": "A drone skims a neon canyon",
|
||||
"initial_rollout_prompt_id": "steer-prompt-1",
|
||||
"manual_continuation_mode": True,
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(
|
||||
0.08,
|
||||
{
|
||||
"type": "append_prompt",
|
||||
"prompt": "The drone dives toward the river",
|
||||
"prompt_id": "steer-prompt-2",
|
||||
},
|
||||
),
|
||||
(0.30, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
message_types = [payload["type"] for payload in ws.sent_json]
|
||||
assert "rewrite_seed_prompts_started" not in message_types
|
||||
assert "rewrite_seed_prompts_complete" not in message_types
|
||||
assert "ltx2_stream_start" in message_types
|
||||
# cap=1 must not stop a manual-mode session after the first segment
|
||||
assert "ltx2_stream_complete" not in message_types
|
||||
|
||||
prompt_ready_events = [
|
||||
payload for payload in ws.sent_json if payload["type"] == "prompt_ready"
|
||||
]
|
||||
assert [payload["prompt_id"] for payload in prompt_ready_events] == [
|
||||
"steer-prompt-1",
|
||||
"steer-prompt-2",
|
||||
]
|
||||
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
assert [payload["prompt"] for payload in segment_start_events] == [
|
||||
"A drone skims a neon canyon",
|
||||
"The drone dives toward the river",
|
||||
]
|
||||
segment_source_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "segment_prompt_source"
|
||||
]
|
||||
assert [payload["prompt_id"] for payload in segment_source_events] == [
|
||||
"steer-prompt-1",
|
||||
"steer-prompt-2",
|
||||
]
|
||||
finally:
|
||||
mock_server.MOCK_SEGMENT_BYTES = old_segment_bytes
|
||||
mock_server.LATENCY_MS = old_latency_ms
|
||||
mock_server.GENERATION_SEGMENT_CAP = old_generation_segment_cap
|
||||
|
||||
|
||||
def test_mock_server_project_init_manual_mode_streams_initial_prompt():
|
||||
old_segment_bytes = mock_server.MOCK_SEGMENT_BYTES
|
||||
old_latency_ms = mock_server.LATENCY_MS
|
||||
try:
|
||||
mock_server.MOCK_SEGMENT_BYTES = b"mock-fmp4-bytes"
|
||||
mock_server.LATENCY_MS = 1
|
||||
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.05, {"type": "end_project_keep_session"}),
|
||||
(
|
||||
0.15,
|
||||
{
|
||||
"type": "project_init_v1",
|
||||
"preset_id": "custom_editable",
|
||||
"preset_label": "Custom rollout",
|
||||
"curated_prompts": [],
|
||||
"initial_rollout_prompt": "A drone skims a neon canyon",
|
||||
"initial_rollout_prompt_id": "steer-prompt-1",
|
||||
"manual_continuation_mode": True,
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.45, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
message_types = [payload["type"] for payload in ws.sent_json]
|
||||
assert "project_idle" in message_types
|
||||
project_idle_index = message_types.index("project_idle")
|
||||
# manual-mode restart must not run the rewrite rollout
|
||||
assert "rewrite_seed_prompts_started" not in message_types[project_idle_index:]
|
||||
assert "ltx2_stream_start" in message_types[project_idle_index:]
|
||||
|
||||
prompt_ready_events = [
|
||||
payload for payload in ws.sent_json if payload["type"] == "prompt_ready"
|
||||
]
|
||||
assert [payload["prompt_id"] for payload in prompt_ready_events] == [
|
||||
"steer-prompt-1",
|
||||
]
|
||||
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
assert segment_start_events[-1]["prompt"] == "A drone skims a neon canyon"
|
||||
finally:
|
||||
mock_server.MOCK_SEGMENT_BYTES = old_segment_bytes
|
||||
mock_server.LATENCY_MS = old_latency_ms
|
||||
@@ -263,37 +443,35 @@ def test_mock_server_can_start_new_project_without_reconnecting():
|
||||
mock_server.MOCK_SEGMENT_BYTES = b"mock-fmp4-bytes"
|
||||
mock_server.LATENCY_MS = 40
|
||||
|
||||
ws = _FakeWebSocket([
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.02, {
|
||||
"type": "end_project_keep_session"
|
||||
}),
|
||||
(
|
||||
0.20,
|
||||
{
|
||||
"type": "project_init_v1",
|
||||
"preset_id": "test_preset_2",
|
||||
"preset_label": "Test Preset 2",
|
||||
"curated_prompts": ["segment two"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.40, {
|
||||
"type": "leave"
|
||||
}),
|
||||
])
|
||||
ws = _FakeWebSocket(
|
||||
[
|
||||
(
|
||||
0.0,
|
||||
{
|
||||
"type": "session_init_v2",
|
||||
"preset_id": "test_preset",
|
||||
"curated_prompts": ["segment one"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.02, {"type": "end_project_keep_session"}),
|
||||
(
|
||||
0.20,
|
||||
{
|
||||
"type": "project_init_v1",
|
||||
"preset_id": "test_preset_2",
|
||||
"preset_label": "Test Preset 2",
|
||||
"curated_prompts": ["segment two"],
|
||||
"enhancement_enabled": True,
|
||||
"auto_extension_enabled": False,
|
||||
"loop_generation_enabled": False,
|
||||
},
|
||||
),
|
||||
(0.40, {"type": "leave"}),
|
||||
]
|
||||
)
|
||||
|
||||
asyncio.run(mock_server.websocket_endpoint(ws))
|
||||
|
||||
@@ -304,11 +482,16 @@ def test_mock_server_can_start_new_project_without_reconnecting():
|
||||
|
||||
project_idle_index = message_types.index("project_idle")
|
||||
stream_start_indexes = [
|
||||
index for index, message_type in enumerate(message_types) if message_type == "ltx2_stream_start"
|
||||
index for index, message_type in enumerate(message_types)
|
||||
if message_type == "ltx2_stream_start"
|
||||
]
|
||||
assert stream_start_indexes[0] < project_idle_index < stream_start_indexes[1]
|
||||
|
||||
segment_start_events = [payload for payload in ws.sent_json if payload["type"] == "ltx2_segment_start"]
|
||||
segment_start_events = [
|
||||
payload
|
||||
for payload in ws.sent_json
|
||||
if payload["type"] == "ltx2_segment_start"
|
||||
]
|
||||
assert [payload["prompt"] for payload in segment_start_events[:2]] == [
|
||||
"segment one",
|
||||
"segment two",
|
||||
|
||||
@@ -6,6 +6,8 @@ import os
|
||||
import re
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("CEREBRAS_API_KEY", "dummy")
|
||||
os.environ.setdefault("GROQ_API_KEY", "dummy")
|
||||
|
||||
@@ -21,7 +23,6 @@ from dreamverse.prompt_enhancer import (
|
||||
|
||||
|
||||
class _FakeResponse:
|
||||
|
||||
def __init__(self, payload: dict):
|
||||
self._payload = payload
|
||||
|
||||
@@ -30,7 +31,6 @@ class _FakeResponse:
|
||||
|
||||
|
||||
class _FakeSyncCompletions:
|
||||
|
||||
def __init__(self, payload: dict):
|
||||
self._payload = payload
|
||||
|
||||
@@ -39,7 +39,6 @@ class _FakeSyncCompletions:
|
||||
|
||||
|
||||
class _FakeSyncClient:
|
||||
|
||||
def __init__(self, payload: dict):
|
||||
self.chat = type(
|
||||
"_FakeChat",
|
||||
@@ -49,7 +48,6 @@ class _FakeSyncClient:
|
||||
|
||||
|
||||
class _DelayedSyncCompletions:
|
||||
|
||||
def __init__(self, payload: dict, delay_s: float = 0.0, exc: Exception | None = None):
|
||||
self._payload = payload
|
||||
self._delay_s = delay_s
|
||||
@@ -64,26 +62,29 @@ class _DelayedSyncCompletions:
|
||||
|
||||
|
||||
class _DelayedSyncClient:
|
||||
|
||||
def __init__(self, payload: dict, delay_s: float = 0.0, exc: Exception | None = None):
|
||||
self.chat = type(
|
||||
"_FakeChat",
|
||||
(),
|
||||
{"completions": _DelayedSyncCompletions(
|
||||
payload,
|
||||
delay_s=delay_s,
|
||||
exc=exc,
|
||||
)},
|
||||
{
|
||||
"completions": _DelayedSyncCompletions(
|
||||
payload,
|
||||
delay_s=delay_s,
|
||||
exc=exc,
|
||||
)
|
||||
},
|
||||
)()
|
||||
|
||||
|
||||
def _chat_payload_with_content(content: str) -> dict:
|
||||
return {
|
||||
"choices": [{
|
||||
"message": {
|
||||
"content": content,
|
||||
"choices": [
|
||||
{
|
||||
"message": {
|
||||
"content": content,
|
||||
}
|
||||
}
|
||||
}]
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
@@ -172,7 +173,6 @@ def _build_staged_enhancer(
|
||||
|
||||
|
||||
class _FakeOpenAIClient:
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
self.kwargs = kwargs
|
||||
self.chat = type(
|
||||
@@ -183,7 +183,6 @@ class _FakeOpenAIClient:
|
||||
|
||||
|
||||
class _FakeCerebrasClient:
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
self.kwargs = kwargs
|
||||
self.chat = type(
|
||||
@@ -194,15 +193,72 @@ class _FakeCerebrasClient:
|
||||
|
||||
|
||||
def test_parse_json_response_accepts_fenced_json_with_prose():
|
||||
parsed = _parse_json_response("Here is the rewrite:\n```json\n{\"segment_prompts\":[\"A\",\"B\"]}\n```\nThanks.")
|
||||
parsed = _parse_json_response(
|
||||
"Here is the rewrite:\n```json\n{\"segment_prompts\":[\"A\",\"B\"]}\n```\nThanks."
|
||||
)
|
||||
assert parsed == {"segment_prompts": ["A", "B"]}
|
||||
|
||||
|
||||
def test_parse_json_response_extracts_first_embedded_object():
|
||||
parsed = _parse_json_response("Model output:\n{\"segment_prompts\":[\"A\",\"B\"]}\n(complete)")
|
||||
parsed = _parse_json_response(
|
||||
"Model output:\n{\"segment_prompts\":[\"A\",\"B\"]}\n(complete)"
|
||||
)
|
||||
assert parsed == {"segment_prompts": ["A", "B"]}
|
||||
|
||||
|
||||
def test_parse_json_response_returns_outer_object_not_nested_value():
|
||||
parsed = _parse_json_response(
|
||||
"Final answer: {\"next_prompt\": \"a scene\", \"style\": {\"mood\": \"noir\"}} done."
|
||||
)
|
||||
assert parsed == {"next_prompt": "a scene", "style": {"mood": "noir"}}
|
||||
|
||||
|
||||
def test_parse_json_response_returns_last_of_multiple_objects():
|
||||
parsed = _parse_json_response(
|
||||
"Draft: {\"next_prompt\": \"draft\"}\nRefined: {\"next_prompt\": \"final\"}"
|
||||
)
|
||||
assert parsed == {"next_prompt": "final"}
|
||||
|
||||
|
||||
def test_parse_json_response_ignores_fragments_of_truncated_trailing_object():
|
||||
# finish_reason=length cut the refined object short; the complete draft must
|
||||
# win over a nested fragment of the truncated object.
|
||||
parsed = _parse_json_response(
|
||||
'{"next_prompt": "draft"} refined: {"next_prompt": "final", "style": {"mood": "noir"}'
|
||||
)
|
||||
assert parsed == {"next_prompt": "draft"}
|
||||
|
||||
|
||||
def test_parse_json_response_ignores_fragments_of_mid_string_truncated_object():
|
||||
# Unterminated-string truncation reports the error at the opening quote,
|
||||
# not end-of-text; nested fragments still must not win over the draft.
|
||||
parsed = _parse_json_response(
|
||||
'{"next_prompt": "draft"} refined: {"style": {"mood": "noir"}, "next_prompt": "cut off'
|
||||
)
|
||||
assert parsed == {"next_prompt": "draft"}
|
||||
|
||||
|
||||
def test_parse_json_response_ignores_fragments_of_malformed_object_with_trailing_prose():
|
||||
parsed = _parse_json_response(
|
||||
'{"next_prompt": "draft"} {"final": {"mood": "noir"}, "x": 1 and then some prose'
|
||||
)
|
||||
assert parsed == {"next_prompt": "draft"}
|
||||
|
||||
|
||||
def test_parse_json_response_raises_when_only_object_is_truncated():
|
||||
with pytest.raises(ValueError):
|
||||
_parse_json_response('{"style": {"mood": "noir"}, "next_prompt": "cut off')
|
||||
|
||||
|
||||
def test_parse_json_response_returns_outer_rollout_dict():
|
||||
parsed = _parse_json_response(
|
||||
"{\"rollout\": {\"segment_prompts\": [{\"prompt\": \"a\"}, {\"prompt\": \"b\"}]}}"
|
||||
)
|
||||
assert parsed == {
|
||||
"rollout": {"segment_prompts": [{"prompt": "a"}, {"prompt": "b"}]}
|
||||
}
|
||||
|
||||
|
||||
def test_load_prompt_required_falls_back_to_default_path(tmp_path):
|
||||
fallback_path = tmp_path / "next_segment_system_prompt.md"
|
||||
fallback_path.write_text("fallback prompt\n", encoding="utf-8")
|
||||
@@ -266,12 +322,16 @@ def test_build_client_supports_groq_provider(monkeypatch):
|
||||
|
||||
def test_rewrite_prompt_sequence_accepts_segment_prompts_output():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
result = asyncio.run(
|
||||
enhancer.rewrite_prompt_sequence(
|
||||
["prompt one", "prompt two"],
|
||||
rewrite_instruction="make it cinematic",
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.rollout_id == "preset_a"
|
||||
@@ -280,12 +340,15 @@ def test_rewrite_prompt_sequence_accepts_segment_prompts_output():
|
||||
|
||||
|
||||
def test_rewrite_prompt_sequence_accepts_legacy_rewritten_prompts_output():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content('{"rewritten_prompts":["A","B"]}'))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"rewritten_prompts":["A","B"]}')
|
||||
)
|
||||
result = asyncio.run(
|
||||
enhancer.rewrite_prompt_sequence(
|
||||
["prompt one", "prompt two"],
|
||||
rewrite_instruction="make it cinematic",
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.rollout_id == "current_rollout"
|
||||
@@ -294,14 +357,19 @@ def test_rewrite_prompt_sequence_accepts_legacy_rewritten_prompts_output():
|
||||
|
||||
|
||||
def test_rewrite_prompt_sequence_accepts_segment_dicts_without_top_level_rollout_metadata():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content('{"segments":[{"prompt":"A"},{"text":"B"}]}'))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content(
|
||||
'{"segments":[{"prompt":"A"},{"text":"B"}]}'
|
||||
)
|
||||
)
|
||||
result = asyncio.run(
|
||||
enhancer.rewrite_prompt_sequence(
|
||||
["prompt one", "prompt two"],
|
||||
preset_id="preset_a",
|
||||
preset_label="Preset A",
|
||||
rewrite_instruction="make it cinematic",
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.rollout_id == "preset_a"
|
||||
@@ -315,12 +383,14 @@ def test_rewrite_prompt_sequence_accepts_numbered_prose_output():
|
||||
"The user is asking for a cinematic rewrite.\n\n"
|
||||
"1. A dog bounds across the moon's dusty surface, kicking up silver regolith as it chases a rabbit beneath the black sky.\n"
|
||||
"2. The rabbit darts around a crater rim while the dog lunges after it, Earth glowing blue in the distance.\n"
|
||||
))
|
||||
)
|
||||
)
|
||||
result = asyncio.run(
|
||||
enhancer.rewrite_prompt_sequence(
|
||||
["prompt one", "prompt two"],
|
||||
rewrite_instruction="make it cinematic",
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.rollout_id == "current_rollout"
|
||||
@@ -331,27 +401,29 @@ def test_rewrite_prompt_sequence_accepts_numbered_prose_output():
|
||||
]
|
||||
|
||||
|
||||
def test_enhance_prompt_uses_groq_when_it_returns_first():
|
||||
def test_enhance_prompt_prefers_cerebras_before_groq_fallback():
|
||||
enhancer = _build_staged_enhancer(
|
||||
cerebras_payload=_chat_payload_with_content('{"prompt":"Cerebras prompt"}'),
|
||||
groq_payload=_chat_payload_with_content('{"prompt":"Groq prompt"}'),
|
||||
cerebras_delay_s=0.08,
|
||||
cerebras_delay_s=0.01,
|
||||
groq_delay_s=0.01,
|
||||
)
|
||||
|
||||
result = asyncio.run(enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
))
|
||||
result = asyncio.run(
|
||||
enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.provider == "groq"
|
||||
assert result.provider == "cerebras"
|
||||
assert result.model == "gpt-test"
|
||||
assert result.prompt == "Groq prompt"
|
||||
assert result.prompt == "Cerebras prompt"
|
||||
assert enhancer.get_provider_success_counts() == {
|
||||
"cerebras": 0,
|
||||
"groq": 1,
|
||||
"cerebras": 1,
|
||||
"groq": 0,
|
||||
}
|
||||
|
||||
|
||||
@@ -363,10 +435,12 @@ def test_enhance_prompt_uses_groq_when_cerebras_fails():
|
||||
groq_delay_s=0.01,
|
||||
)
|
||||
|
||||
result = asyncio.run(enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
))
|
||||
result = asyncio.run(
|
||||
enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
@@ -388,11 +462,13 @@ def test_enhance_prompt_can_use_groq_when_cerebras_times_out():
|
||||
enhancer.http_timeout_ms = 50
|
||||
enhancer.default_timeout_ms = 50
|
||||
|
||||
result = asyncio.run(enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
timeout_ms=50,
|
||||
))
|
||||
result = asyncio.run(
|
||||
enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
timeout_ms=50,
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
@@ -412,10 +488,12 @@ def test_enhance_prompt_can_use_cerebras_when_it_returns_first():
|
||||
groq_delay_s=0.08,
|
||||
)
|
||||
|
||||
result = asyncio.run(enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
))
|
||||
result = asyncio.run(
|
||||
enhancer.enhance_prompt(
|
||||
"A rainy alley at night",
|
||||
mode="single_clip",
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
@@ -429,12 +507,15 @@ def test_enhance_prompt_can_use_cerebras_when_it_returns_first():
|
||||
|
||||
|
||||
def test_rewrite_prompt_sequence_keeps_raw_output_on_parse_error():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content("I cannot comply with JSON right now."))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content("I cannot comply with JSON right now.")
|
||||
)
|
||||
result = asyncio.run(
|
||||
enhancer.rewrite_prompt_sequence(
|
||||
["prompt one", "prompt two"],
|
||||
rewrite_instruction="make it cinematic",
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is True
|
||||
assert "No JSON object found in assistant response." in (result.error or "")
|
||||
assert result.raw_response_text == "I cannot comply with JSON right now."
|
||||
@@ -446,7 +527,9 @@ def test_rewrite_prompt_sequence_keeps_raw_output_on_parse_error():
|
||||
def test_rewrite_prompt_sequence_uses_current_rollout_payload_shape():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content(
|
||||
'{"id":"rewritten_rollout","label":"Rewritten Rollout","segment_prompts":["A","B"]}'))
|
||||
'{"id":"rewritten_rollout","label":"Rewritten Rollout","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
captured = {
|
||||
"body": None,
|
||||
"timeout_seconds": None,
|
||||
@@ -457,7 +540,8 @@ def test_rewrite_prompt_sequence_uses_current_rollout_payload_shape():
|
||||
captured["timeout_seconds"] = timeout_seconds
|
||||
return (
|
||||
_chat_payload_with_content(
|
||||
'{"id":"rewritten_rollout","label":"Rewritten Rollout","segment_prompts":["A","B"]}'),
|
||||
'{"id":"rewritten_rollout","label":"Rewritten Rollout","segment_prompts":["A","B"]}'
|
||||
),
|
||||
'{"id":"rewritten_rollout","label":"Rewritten Rollout","segment_prompts":["A","B"]}',
|
||||
)
|
||||
|
||||
@@ -472,7 +556,8 @@ def test_rewrite_prompt_sequence_uses_current_rollout_payload_shape():
|
||||
rewrite_model="gpt-test",
|
||||
rewrite_temperature=0.2,
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert captured["body"]["messages"][0] == {
|
||||
@@ -481,12 +566,12 @@ def test_rewrite_prompt_sequence_uses_current_rollout_payload_shape():
|
||||
}
|
||||
assert captured["body"]["messages"][1]["role"] == "user"
|
||||
assert prompt_enhancer_module.json.loads(captured["body"]["messages"][1]["content"]) == {
|
||||
"mode":
|
||||
"edit_existing_rollout",
|
||||
"request": ("Rewrite all segment prompts with improved continuity and cinematic detail. "
|
||||
"Keep count and ordering identical."),
|
||||
"user_instruction":
|
||||
"make it cinematic",
|
||||
"mode": "edit_existing_rollout",
|
||||
"request": (
|
||||
"Rewrite all segment prompts with improved continuity and cinematic detail. "
|
||||
"Keep count and ordering identical."
|
||||
),
|
||||
"user_instruction": "make it cinematic",
|
||||
"current_rollout": {
|
||||
"id": "preset_a",
|
||||
"label": "Preset A",
|
||||
@@ -497,8 +582,11 @@ def test_rewrite_prompt_sequence_uses_current_rollout_payload_shape():
|
||||
|
||||
def test_rewrite_prompt_sequence_supports_new_rollout_mode():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"custom_editable","label":"Custom rollout","segment_prompts":['
|
||||
'"A","B","C","D","E","F"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"custom_editable","label":"Custom rollout","segment_prompts":['
|
||||
'"A","B","C","D","E","F"]}'
|
||||
)
|
||||
)
|
||||
captured = {
|
||||
"body": None,
|
||||
}
|
||||
@@ -507,8 +595,10 @@ def test_rewrite_prompt_sequence_supports_new_rollout_mode():
|
||||
del timeout_seconds
|
||||
captured["body"] = body
|
||||
return (
|
||||
_chat_payload_with_content('{"id":"custom_editable","label":"Custom rollout","segment_prompts":['
|
||||
'"A","B","C","D","E","F"]}'),
|
||||
_chat_payload_with_content(
|
||||
'{"id":"custom_editable","label":"Custom rollout","segment_prompts":['
|
||||
'"A","B","C","D","E","F"]}'
|
||||
),
|
||||
'{"id":"custom_editable","label":"Custom rollout","segment_prompts":['
|
||||
'"A","B","C","D","E","F"]}',
|
||||
)
|
||||
@@ -524,29 +614,30 @@ def test_rewrite_prompt_sequence_supports_new_rollout_mode():
|
||||
rewrite_model="gpt-test",
|
||||
rewrite_temperature=0.2,
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert result.prompts == ["A", "B", "C", "D", "E", "F"]
|
||||
assert prompt_enhancer_module.json.loads(captured["body"]["messages"][1]["content"]) == {
|
||||
"mode":
|
||||
"new_rollout",
|
||||
"request": ("Rewrite all segment prompts with improved continuity and cinematic detail. "
|
||||
"Keep count and ordering identical."),
|
||||
"user_instruction":
|
||||
"A moonbase corridor thriller with flooding and red alarms",
|
||||
"desired_segment_count":
|
||||
6,
|
||||
"rollout_id_hint":
|
||||
"custom_editable",
|
||||
"rollout_label_hint":
|
||||
"Custom rollout",
|
||||
"mode": "new_rollout",
|
||||
"request": (
|
||||
"Rewrite all segment prompts with improved continuity and cinematic detail. "
|
||||
"Keep count and ordering identical."
|
||||
),
|
||||
"user_instruction": "A moonbase corridor thriller with flooding and red alarms",
|
||||
"desired_segment_count": 6,
|
||||
"rollout_id_hint": "custom_editable",
|
||||
"rollout_label_hint": "Custom rollout",
|
||||
}
|
||||
|
||||
|
||||
def test_rewrite_prompt_sequence_uses_session_override_system_prompt():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
enhancer.rewrite_all_system_prompt = "shared system prompt"
|
||||
captured = {
|
||||
"body": None,
|
||||
@@ -556,7 +647,9 @@ def test_rewrite_prompt_sequence_uses_session_override_system_prompt():
|
||||
del timeout_seconds
|
||||
captured["body"] = body
|
||||
return (
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'),
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
),
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}',
|
||||
)
|
||||
|
||||
@@ -570,7 +663,8 @@ def test_rewrite_prompt_sequence_uses_session_override_system_prompt():
|
||||
rewrite_instruction="make it cinematic",
|
||||
rewrite_model="gpt-test",
|
||||
system_prompt_override="session specific system prompt",
|
||||
))
|
||||
)
|
||||
)
|
||||
|
||||
assert result.fallback_used is False
|
||||
assert captured["body"]["messages"][0] == {
|
||||
@@ -581,7 +675,10 @@ def test_rewrite_prompt_sequence_uses_session_override_system_prompt():
|
||||
|
||||
def test_resolve_rewrite_new_rollout_system_prompt_uses_dedicated_prompt():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
enhancer.rewrite_all_system_prompt = "shared rewrite system prompt"
|
||||
enhancer.rewrite_user_system_prompt = "new rollout rewrite system prompt"
|
||||
|
||||
@@ -592,17 +689,24 @@ def test_resolve_rewrite_new_rollout_system_prompt_uses_dedicated_prompt():
|
||||
|
||||
def test_resolve_rewrite_new_rollout_system_prompt_prefers_override():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
enhancer.rewrite_all_system_prompt = "shared rewrite system prompt"
|
||||
enhancer.rewrite_user_system_prompt = "new rollout rewrite system prompt"
|
||||
|
||||
resolved = enhancer.resolve_rewrite_new_rollout_system_prompt("session specific system prompt")
|
||||
resolved = enhancer.resolve_rewrite_new_rollout_system_prompt(
|
||||
"session specific system prompt"
|
||||
)
|
||||
|
||||
assert resolved == "session specific system prompt"
|
||||
|
||||
|
||||
def test_generate_auto_prompt_uses_selected_model():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content('{"next_prompt":"Auto next"}'))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"next_prompt":"Auto next"}')
|
||||
)
|
||||
enhancer.auto_system_prompt = "auto system prompt"
|
||||
enhancer.rewrite_model_options = ["gpt-test", "gpt-alt"]
|
||||
enhancer.rewrite_default_model = "gpt-test"
|
||||
@@ -630,7 +734,8 @@ def test_generate_auto_prompt_uses_selected_model():
|
||||
next_segment_idx=2,
|
||||
model="gpt-alt",
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.prompt == "Auto next"
|
||||
@@ -639,7 +744,9 @@ def test_generate_auto_prompt_uses_selected_model():
|
||||
|
||||
|
||||
def test_enhance_prompt_uses_selected_model():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content('{"next_prompt":"Enhanced next"}'))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"next_prompt":"Enhanced next"}')
|
||||
)
|
||||
enhancer.enhance_system_prompt = "enhance system prompt"
|
||||
enhancer.auto_system_prompt = "auto system prompt"
|
||||
enhancer.rewrite_model_options = ["gpt-test", "gpt-alt"]
|
||||
@@ -669,7 +776,8 @@ def test_enhance_prompt_uses_selected_model():
|
||||
next_segment_idx=2,
|
||||
model="gpt-alt",
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.prompt == "Enhanced next"
|
||||
@@ -678,7 +786,9 @@ def test_enhance_prompt_uses_selected_model():
|
||||
|
||||
|
||||
def test_enhance_prompt_single_clip_uses_auto_extension_prompt_and_prompt_field():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content('{"prompt":"Extended single clip"}'))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"prompt":"Extended single clip"}')
|
||||
)
|
||||
enhancer.enhance_system_prompt = "enhance system prompt"
|
||||
enhancer.auto_system_prompt = "auto system prompt"
|
||||
enhancer.rewrite_model_options = ["gpt-test", "gpt-alt"]
|
||||
@@ -708,12 +818,14 @@ def test_enhance_prompt_single_clip_uses_auto_extension_prompt_and_prompt_field(
|
||||
|
||||
enhancer._request_content = _fake_request_content # type: ignore[attr-defined]
|
||||
|
||||
result = asyncio.run(enhancer.enhance_prompt(
|
||||
"short 5s idea",
|
||||
mode="single_clip",
|
||||
model="gpt-alt",
|
||||
timeout_ms=800,
|
||||
))
|
||||
result = asyncio.run(
|
||||
enhancer.enhance_prompt(
|
||||
"short 5s idea",
|
||||
mode="single_clip",
|
||||
model="gpt-alt",
|
||||
timeout_ms=800,
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is False
|
||||
assert result.error is None
|
||||
assert result.prompt == "Extended single clip"
|
||||
@@ -726,15 +838,17 @@ def test_enhance_prompt_single_clip_uses_auto_extension_prompt_and_prompt_field(
|
||||
"single 5-second LTX-2.3 video clip. Respond with "
|
||||
'valid JSON only as {"prompt": "..."}.' # noqa: E501
|
||||
),
|
||||
"user_prompt":
|
||||
"short 5s idea",
|
||||
"user_prompt": "short 5s idea",
|
||||
}
|
||||
|
||||
|
||||
def test_enhance_prompt_single_clip_rejects_plain_text_response():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content("Medium shot of a woman by a rainy cafe window as she lifts her "
|
||||
"phone, exhales softly, and the camera makes a slow push in."))
|
||||
_chat_payload_with_content(
|
||||
"Medium shot of a woman by a rainy cafe window as she lifts her "
|
||||
"phone, exhales softly, and the camera makes a slow push in."
|
||||
)
|
||||
)
|
||||
enhancer.auto_system_prompt = "auto system prompt"
|
||||
|
||||
result = asyncio.run(
|
||||
@@ -743,14 +857,17 @@ def test_enhance_prompt_single_clip_rejects_plain_text_response():
|
||||
mode="single_clip",
|
||||
model="gpt-test",
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is True
|
||||
assert "No JSON object found in assistant response." in result.error
|
||||
assert result.prompt == ""
|
||||
|
||||
|
||||
def test_enhance_prompt_single_clip_rejects_segment_prompts_json():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content('{"segment_prompts":["A","B"]}'))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"segment_prompts":["A","B"]}')
|
||||
)
|
||||
enhancer.auto_system_prompt = "auto system prompt"
|
||||
enhancer.rewrite_model_options = ["gpt-test"]
|
||||
enhancer.rewrite_default_model = "gpt-test"
|
||||
@@ -761,14 +878,17 @@ def test_enhance_prompt_single_clip_rejects_segment_prompts_json():
|
||||
mode="single_clip",
|
||||
model="gpt-test",
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is True
|
||||
assert result.prompt == ""
|
||||
assert "Missing prompt string." in (result.error or "")
|
||||
|
||||
|
||||
def test_enhance_prompt_requires_json_and_does_not_fallback_to_raw_text():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content("A cinematic continuation with slow dolly movement."))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content("A cinematic continuation with slow dolly movement.")
|
||||
)
|
||||
enhancer.enhance_system_prompt = "enhance system prompt"
|
||||
enhancer.rewrite_model_options = ["gpt-test"]
|
||||
enhancer.rewrite_default_model = "gpt-test"
|
||||
@@ -780,14 +900,17 @@ def test_enhance_prompt_requires_json_and_does_not_fallback_to_raw_text():
|
||||
next_segment_idx=2,
|
||||
model="gpt-test",
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is True
|
||||
assert result.prompt == ""
|
||||
assert "No JSON object found in assistant response." in (result.error or "")
|
||||
|
||||
|
||||
def test_generate_auto_prompt_requires_json_and_does_not_fallback_to_raw_text():
|
||||
enhancer = _build_test_enhancer(_chat_payload_with_content("A calm, grounded continuation with subtle motion."))
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content("A calm, grounded continuation with subtle motion.")
|
||||
)
|
||||
enhancer.auto_system_prompt = "auto system prompt"
|
||||
enhancer.rewrite_model_options = ["gpt-test"]
|
||||
enhancer.rewrite_default_model = "gpt-test"
|
||||
@@ -798,30 +921,34 @@ def test_generate_auto_prompt_requires_json_and_does_not_fallback_to_raw_text():
|
||||
next_segment_idx=2,
|
||||
model="gpt-test",
|
||||
timeout_ms=800,
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is True
|
||||
assert result.prompt == ""
|
||||
assert "No JSON object found in assistant response." in (result.error or "")
|
||||
|
||||
|
||||
def test_rewrite_prompt_sequence_includes_raw_json_when_content_empty():
|
||||
enhancer = _build_test_enhancer({
|
||||
"choices": [{
|
||||
"finish_reason": "length",
|
||||
"message": {
|
||||
"content": [],
|
||||
"refusal": None,
|
||||
},
|
||||
}],
|
||||
"usage": {
|
||||
"completion_tokens": 0
|
||||
},
|
||||
})
|
||||
enhancer = _build_test_enhancer(
|
||||
{
|
||||
"choices": [
|
||||
{
|
||||
"finish_reason": "length",
|
||||
"message": {
|
||||
"content": [],
|
||||
"refusal": None,
|
||||
},
|
||||
}
|
||||
],
|
||||
"usage": {"completion_tokens": 0},
|
||||
}
|
||||
)
|
||||
result = asyncio.run(
|
||||
enhancer.rewrite_prompt_sequence(
|
||||
["prompt one", "prompt two"],
|
||||
rewrite_instruction="make it cinematic",
|
||||
))
|
||||
)
|
||||
)
|
||||
assert result.fallback_used is True
|
||||
assert "No rewrite segment prompts found in assistant response." in (result.error or "")
|
||||
assert isinstance(result.raw_response_text, str)
|
||||
@@ -830,7 +957,10 @@ def test_rewrite_prompt_sequence_includes_raw_json_when_content_empty():
|
||||
|
||||
def test_get_rewrite_model_config_returns_fixed_defaults():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
enhancer.rewrite_default_model = "gpt-oss-120b"
|
||||
enhancer.rewrite_model_options = ["gpt-oss-120b"]
|
||||
|
||||
@@ -842,7 +972,10 @@ def test_get_rewrite_model_config_returns_fixed_defaults():
|
||||
|
||||
def test_get_prompt_config_includes_auto_extension_prompt():
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
enhancer.enhance_system_prompt_path = "/tmp/next.md"
|
||||
enhancer.auto_system_prompt_path = "/tmp/auto.md"
|
||||
enhancer.rewrite_all_system_prompt_path = "/tmp/rewrite.md"
|
||||
@@ -869,14 +1002,19 @@ def test_get_prompt_config_includes_auto_extension_prompt():
|
||||
|
||||
def test_get_prompt_config_reports_loaded_fallback_prompt_path(tmp_path):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
rewrite_fallback_path = tmp_path / "rewrite_window_system_prompt.md"
|
||||
rewrite_fallback_path.write_text("rewrite prompt\n", encoding="utf-8")
|
||||
next_path = tmp_path / "next.md"
|
||||
next_path.write_text("next prompt\n", encoding="utf-8")
|
||||
auto_path = tmp_path / "auto.md"
|
||||
auto_path.write_text("auto prompt\n", encoding="utf-8")
|
||||
enhancer.rewrite_all_system_prompt_path = str(tmp_path / "prompts.local" / "rewrite_window_system_prompt.md")
|
||||
enhancer.rewrite_all_system_prompt_path = str(
|
||||
tmp_path / "prompts.local" / "rewrite_window_system_prompt.md"
|
||||
)
|
||||
enhancer.rewrite_all_system_prompt_fallback_path = str(rewrite_fallback_path)
|
||||
enhancer.enhance_system_prompt_path = str(next_path)
|
||||
enhancer.auto_system_prompt_path = str(auto_path)
|
||||
@@ -889,9 +1027,14 @@ def test_get_prompt_config_reports_loaded_fallback_prompt_path(tmp_path):
|
||||
assert config["rewrite_window_system_prompt_path"] == str(rewrite_fallback_path)
|
||||
|
||||
|
||||
def test_reload_system_prompts_falls_back_to_rewrite_window_when_user_prompt_empty(tmp_path, ):
|
||||
def test_reload_system_prompts_falls_back_to_rewrite_window_when_user_prompt_empty(
|
||||
tmp_path,
|
||||
):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
next_path = tmp_path / "next.md"
|
||||
auto_path = tmp_path / "auto.md"
|
||||
rewrite_path = tmp_path / "rewrite_window_system_prompt.md"
|
||||
@@ -918,7 +1061,10 @@ def test_reload_system_prompts_falls_back_to_rewrite_window_when_user_prompt_emp
|
||||
|
||||
def test_save_prompt_config_updates_auto_extension_prompt(tmp_path):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
next_path = tmp_path / "next.md"
|
||||
auto_path = tmp_path / "auto.md"
|
||||
rewrite_path = tmp_path / "rewrite.md"
|
||||
@@ -929,7 +1075,9 @@ def test_save_prompt_config_updates_auto_extension_prompt(tmp_path):
|
||||
enhancer.auto_system_prompt_path = str(auto_path)
|
||||
enhancer.rewrite_all_system_prompt_path = str(rewrite_path)
|
||||
|
||||
config = enhancer.save_prompt_config(auto_extension_system_prompt="auto updated", )
|
||||
config = enhancer.save_prompt_config(
|
||||
auto_extension_system_prompt="auto updated",
|
||||
)
|
||||
|
||||
assert auto_path.read_text(encoding="utf-8").strip() == "auto updated"
|
||||
assert config["auto_extension_system_prompt"] == "auto updated"
|
||||
@@ -937,7 +1085,10 @@ def test_save_prompt_config_updates_auto_extension_prompt(tmp_path):
|
||||
|
||||
def test_save_prompt_config_updates_rewrite_user_prompt(tmp_path):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
next_path = tmp_path / "next.md"
|
||||
auto_path = tmp_path / "auto.md"
|
||||
rewrite_path = tmp_path / "rewrite.md"
|
||||
@@ -955,7 +1106,9 @@ def test_save_prompt_config_updates_rewrite_user_prompt(tmp_path):
|
||||
enhancer.rewrite_all_system_prompt_fallback_path = None
|
||||
enhancer.rewrite_user_system_prompt_fallback_path = None
|
||||
|
||||
config = enhancer.save_prompt_config(rewrite_user_system_prompt="rewrite user updated", )
|
||||
config = enhancer.save_prompt_config(
|
||||
rewrite_user_system_prompt="rewrite user updated",
|
||||
)
|
||||
|
||||
assert rewrite_user_path.read_text(encoding="utf-8").strip() == "rewrite user updated"
|
||||
assert config["rewrite_user_system_prompt"] == "rewrite user updated"
|
||||
@@ -963,7 +1116,10 @@ def test_save_prompt_config_updates_rewrite_user_prompt(tmp_path):
|
||||
|
||||
def test_save_prompt_config_updates_rewrite_model(tmp_path):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
next_path = tmp_path / "next.md"
|
||||
auto_path = tmp_path / "auto.md"
|
||||
rewrite_path = tmp_path / "rewrite.md"
|
||||
@@ -979,7 +1135,9 @@ def test_save_prompt_config_updates_rewrite_model(tmp_path):
|
||||
enhancer.rewrite_default_model = "gpt-test"
|
||||
enhancer.rewrite_model_options = ["gpt-test", "gpt-alt"]
|
||||
|
||||
config = enhancer.save_prompt_config(rewrite_model="gpt-alt", )
|
||||
config = enhancer.save_prompt_config(
|
||||
rewrite_model="gpt-alt",
|
||||
)
|
||||
|
||||
assert enhancer.rewrite_default_model == "gpt-alt"
|
||||
assert config["rewrite_model"] == "gpt-alt"
|
||||
@@ -988,7 +1146,10 @@ def test_save_prompt_config_updates_rewrite_model(tmp_path):
|
||||
|
||||
def test_save_prompt_config_updates_rewrite_temperature(tmp_path):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
next_path = tmp_path / "next.md"
|
||||
auto_path = tmp_path / "auto.md"
|
||||
rewrite_path = tmp_path / "rewrite.md"
|
||||
@@ -1002,7 +1163,9 @@ def test_save_prompt_config_updates_rewrite_temperature(tmp_path):
|
||||
enhancer.auto_system_prompt_fallback_path = None
|
||||
enhancer.rewrite_all_system_prompt_fallback_path = None
|
||||
|
||||
config = enhancer.save_prompt_config(rewrite_temperature=1.3, )
|
||||
config = enhancer.save_prompt_config(
|
||||
rewrite_temperature=1.3,
|
||||
)
|
||||
|
||||
assert enhancer.rewrite_default_temperature == 1.3
|
||||
assert config["rewrite_temperature"] == 1.3
|
||||
@@ -1010,7 +1173,10 @@ def test_save_prompt_config_updates_rewrite_temperature(tmp_path):
|
||||
|
||||
def test_save_prompt_config_creates_versioned_backup_for_existing_prompt(tmp_path):
|
||||
enhancer = _build_test_enhancer(
|
||||
_chat_payload_with_content('{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'))
|
||||
_chat_payload_with_content(
|
||||
'{"id":"preset_a","label":"Preset A","segment_prompts":["A","B"]}'
|
||||
)
|
||||
)
|
||||
next_path = tmp_path / "next.md"
|
||||
auto_path = tmp_path / "auto.md"
|
||||
rewrite_path = tmp_path / "rewrite_window_system_prompt.md"
|
||||
@@ -1024,9 +1190,13 @@ def test_save_prompt_config_creates_versioned_backup_for_existing_prompt(tmp_pat
|
||||
enhancer.auto_system_prompt_fallback_path = None
|
||||
enhancer.rewrite_all_system_prompt_fallback_path = None
|
||||
|
||||
enhancer.save_prompt_config(rewrite_window_system_prompt="rewrite updated", )
|
||||
enhancer.save_prompt_config(
|
||||
rewrite_window_system_prompt="rewrite updated",
|
||||
)
|
||||
|
||||
backup_paths = sorted(tmp_path.glob("rewrite_window_system_prompt.*.bak.md"))
|
||||
backup_paths = sorted(
|
||||
tmp_path.glob("rewrite_window_system_prompt.*.bak.md")
|
||||
)
|
||||
|
||||
assert rewrite_path.read_text(encoding="utf-8").strip() == "rewrite updated"
|
||||
assert len(backup_paths) == 1
|
||||
|
||||
@@ -27,8 +27,13 @@ try:
|
||||
except ModuleNotFoundError:
|
||||
websockets = None # type: ignore[assignment]
|
||||
|
||||
DEFAULT_PRESET_FILE = (Path(__file__).resolve().parents[2] / "web" / "prompts" /
|
||||
"selected_ltx2_continuation_story_presets.json")
|
||||
|
||||
DEFAULT_PRESET_FILE = (
|
||||
Path(__file__).resolve().parents[2]
|
||||
/ "web"
|
||||
/ "prompts"
|
||||
/ "selected_ltx2_continuation_story_presets.json"
|
||||
)
|
||||
|
||||
|
||||
def utc_now_iso() -> str:
|
||||
@@ -60,7 +65,10 @@ def safe_percentile(values: list[float], percentile: float) -> float | None:
|
||||
if lower == upper:
|
||||
return sorted_values[lower]
|
||||
fraction = rank - lower
|
||||
return (sorted_values[lower] + (sorted_values[upper] - sorted_values[lower]) * fraction)
|
||||
return (
|
||||
sorted_values[lower]
|
||||
+ (sorted_values[upper] - sorted_values[lower]) * fraction
|
||||
)
|
||||
|
||||
|
||||
def summarize_series(values: list[float]) -> dict[str, float | int | None]:
|
||||
@@ -137,16 +145,24 @@ def load_curated_prompts(
|
||||
selected_id = str(selected.get("id", "")).strip() or "unknown_preset"
|
||||
raw_prompts = selected.get("segment_prompts", [])
|
||||
if not isinstance(raw_prompts, list):
|
||||
raise ValueError(f"Preset {selected_id} has invalid segment_prompts (must be list).")
|
||||
raise ValueError(
|
||||
f"Preset {selected_id} has invalid segment_prompts (must be list)."
|
||||
)
|
||||
|
||||
prompts = [str(prompt).strip() for prompt in raw_prompts if isinstance(prompt, str) and str(prompt).strip()]
|
||||
prompts = [
|
||||
str(prompt).strip()
|
||||
for prompt in raw_prompts
|
||||
if isinstance(prompt, str) and str(prompt).strip()
|
||||
]
|
||||
if not prompts:
|
||||
raise ValueError(f"Preset {selected_id} has no non-empty prompts.")
|
||||
|
||||
limited = prompts[:curated_limit]
|
||||
if not limited:
|
||||
raise ValueError(f"curated_limit={curated_limit} produced no prompts for preset "
|
||||
f"{selected_id}.")
|
||||
raise ValueError(
|
||||
f"curated_limit={curated_limit} produced no prompts for preset "
|
||||
f"{selected_id}."
|
||||
)
|
||||
return selected_id, limited, len(prompts)
|
||||
|
||||
|
||||
@@ -208,11 +224,11 @@ async def run_single_session(
|
||||
|
||||
try:
|
||||
async with websockets.connect(
|
||||
url,
|
||||
max_size=None,
|
||||
ping_interval=None,
|
||||
open_timeout=connect_timeout_s,
|
||||
close_timeout=2.0,
|
||||
url,
|
||||
max_size=None,
|
||||
ping_interval=None,
|
||||
open_timeout=connect_timeout_s,
|
||||
close_timeout=2.0,
|
||||
) as ws:
|
||||
connect_finish_monotonic = time.monotonic()
|
||||
session_data["connect_finish_ts_utc"] = utc_now_iso()
|
||||
@@ -233,7 +249,9 @@ async def run_single_session(
|
||||
timeout_remaining = session_timeout_s - elapsed_s
|
||||
if timeout_remaining <= 0:
|
||||
session_data["status"] = "timeout"
|
||||
session_data["error"] = (f"Session timed out after {session_timeout_s:.1f}s.")
|
||||
session_data["error"] = (
|
||||
f"Session timed out after {session_timeout_s:.1f}s."
|
||||
)
|
||||
break
|
||||
|
||||
recv_start_epoch = time.time()
|
||||
@@ -247,7 +265,9 @@ async def run_single_session(
|
||||
)
|
||||
except asyncio.TimeoutError:
|
||||
session_data["status"] = "timeout"
|
||||
session_data["error"] = ("Timed out waiting for websocket message.")
|
||||
session_data["error"] = (
|
||||
"Timed out waiting for websocket message."
|
||||
)
|
||||
break
|
||||
except Exception as exc:
|
||||
session_data["status"] = "failed"
|
||||
@@ -268,16 +288,20 @@ async def run_single_session(
|
||||
|
||||
chunk_gap_ms: float | None = None
|
||||
if last_chunk_finish_monotonic is not None:
|
||||
chunk_gap_ms = (recv_finish_monotonic - last_chunk_finish_monotonic) * 1000.0
|
||||
chunk_gap_ms = (
|
||||
recv_finish_monotonic - last_chunk_finish_monotonic
|
||||
) * 1000.0
|
||||
|
||||
session_data["chunks"].append({
|
||||
"segment_idx": current_segment_idx,
|
||||
"chunk_idx": session_data["total_chunks"],
|
||||
"size_bytes": len(message),
|
||||
"chunk_start_ts_utc": recv_start_iso,
|
||||
"chunk_finish_ts_utc": recv_finish_iso,
|
||||
"chunk_gap_ms": chunk_gap_ms,
|
||||
})
|
||||
session_data["chunks"].append(
|
||||
{
|
||||
"segment_idx": current_segment_idx,
|
||||
"chunk_idx": session_data["total_chunks"],
|
||||
"size_bytes": len(message),
|
||||
"chunk_start_ts_utc": recv_start_iso,
|
||||
"chunk_finish_ts_utc": recv_finish_iso,
|
||||
"chunk_gap_ms": chunk_gap_ms,
|
||||
}
|
||||
)
|
||||
last_chunk_finish_monotonic = recv_finish_monotonic
|
||||
last_chunk_finish_epoch = recv_finish_epoch
|
||||
session_data["last_chunk_finish_ts_utc"] = recv_finish_iso
|
||||
@@ -297,7 +321,9 @@ async def run_single_session(
|
||||
if msg_type == "gpu_assigned":
|
||||
session_data["gpu_assigned_ts_utc"] = recv_finish_iso
|
||||
if connect_finish_monotonic is not None:
|
||||
session_data["queue_wait_ms"] = (recv_finish_monotonic - connect_finish_monotonic) * 1000.0
|
||||
session_data["queue_wait_ms"] = (
|
||||
recv_finish_monotonic - connect_finish_monotonic
|
||||
) * 1000.0
|
||||
elif msg_type == "ltx2_stream_start":
|
||||
if initial_total_segments is None:
|
||||
parsed_total = parse_int(data.get("total_segments"))
|
||||
@@ -312,13 +338,20 @@ async def run_single_session(
|
||||
session_data["media_segments_completed"] += 1
|
||||
if first_media_segment_complete_epoch is None:
|
||||
first_media_segment_complete_epoch = recv_finish_epoch
|
||||
session_data["first_media_segment_complete_ts_utc"] = recv_finish_iso
|
||||
session_data[
|
||||
"first_media_segment_complete_ts_utc"
|
||||
] = recv_finish_iso
|
||||
elif msg_type == "ltx2_segment_complete":
|
||||
session_data["segments_completed"] += 1
|
||||
seg_idx = parse_int(data.get("segment_idx"))
|
||||
if (initial_total_segments is not None and seg_idx is not None
|
||||
and seg_idx >= initial_total_segments):
|
||||
session_data["target_segment_complete_ts_utc"] = recv_finish_iso
|
||||
if (
|
||||
initial_total_segments is not None
|
||||
and seg_idx is not None
|
||||
and seg_idx >= initial_total_segments
|
||||
):
|
||||
session_data[
|
||||
"target_segment_complete_ts_utc"
|
||||
] = recv_finish_iso
|
||||
await asyncio.sleep(post_complete_wait_s)
|
||||
session_data["leave_sent_ts_utc"] = utc_now_iso()
|
||||
try:
|
||||
@@ -329,11 +362,15 @@ async def run_single_session(
|
||||
break
|
||||
elif msg_type == "session_timeout":
|
||||
session_data["status"] = "timeout"
|
||||
session_data["error"] = str(data.get("message") or "Backend session timeout")
|
||||
session_data["error"] = str(
|
||||
data.get("message") or "Backend session timeout"
|
||||
)
|
||||
break
|
||||
elif msg_type == "error":
|
||||
session_data["status"] = "failed"
|
||||
session_data["error"] = str(data.get("message") or "Backend error message")
|
||||
session_data["error"] = str(
|
||||
data.get("message") or "Backend error message"
|
||||
)
|
||||
break
|
||||
|
||||
if session_data["status"] == "failed" and session_data["error"] is None:
|
||||
@@ -342,18 +379,29 @@ async def run_single_session(
|
||||
session_data["status"] = "failed"
|
||||
session_data["error"] = f"WebSocket connect/run failed: {exc}"
|
||||
|
||||
if (first_chunk_finish_epoch is not None and last_chunk_finish_epoch is not None
|
||||
and session_data["total_chunk_bytes"] > 0):
|
||||
if (
|
||||
first_chunk_finish_epoch is not None
|
||||
and last_chunk_finish_epoch is not None
|
||||
and session_data["total_chunk_bytes"] > 0
|
||||
):
|
||||
duration_s = last_chunk_finish_epoch - first_chunk_finish_epoch
|
||||
if duration_s > 0:
|
||||
session_data["session_goodput_mbps"] = (session_data["total_chunk_bytes"] * 8.0 / duration_s / 1_000_000.0)
|
||||
session_data["session_goodput_mbps"] = (
|
||||
session_data["total_chunk_bytes"] * 8.0 / duration_s / 1_000_000.0
|
||||
)
|
||||
|
||||
if (first_chunk_finish_epoch is not None and first_media_segment_complete_epoch is not None):
|
||||
session_data["first_chunk_before_first_media_complete"] = (first_chunk_finish_epoch
|
||||
< first_media_segment_complete_epoch)
|
||||
if (
|
||||
first_chunk_finish_epoch is not None
|
||||
and first_media_segment_complete_epoch is not None
|
||||
):
|
||||
session_data["first_chunk_before_first_media_complete"] = (
|
||||
first_chunk_finish_epoch < first_media_segment_complete_epoch
|
||||
)
|
||||
|
||||
session_data["close_ts_utc"] = utc_now_iso()
|
||||
session_data["duration_ms"] = (time.monotonic() - session_start_monotonic) * 1000.0
|
||||
session_data["duration_ms"] = (
|
||||
time.monotonic() - session_start_monotonic
|
||||
) * 1000.0
|
||||
return session_data
|
||||
|
||||
|
||||
@@ -364,11 +412,14 @@ async def run_worker_sessions(
|
||||
config: dict[str, Any],
|
||||
) -> list[dict[str, Any]]:
|
||||
tasks = [
|
||||
asyncio.create_task(run_single_session(
|
||||
worker_id=worker_id,
|
||||
worker_session_idx=idx,
|
||||
config=config,
|
||||
)) for idx in range(session_count)
|
||||
asyncio.create_task(
|
||||
run_single_session(
|
||||
worker_id=worker_id,
|
||||
worker_session_idx=idx,
|
||||
config=config,
|
||||
)
|
||||
)
|
||||
for idx in range(session_count)
|
||||
]
|
||||
if not tasks:
|
||||
return []
|
||||
@@ -386,23 +437,29 @@ def worker_entry(
|
||||
try:
|
||||
ready_queue.put({"worker_id": worker_id, "status": "ready"})
|
||||
start_event.wait()
|
||||
sessions = asyncio.run(run_worker_sessions(
|
||||
worker_id=worker_id,
|
||||
session_count=session_count,
|
||||
config=config,
|
||||
))
|
||||
result_queue.put({
|
||||
"worker_id": worker_id,
|
||||
"status": "ok",
|
||||
"sessions": sessions,
|
||||
})
|
||||
sessions = asyncio.run(
|
||||
run_worker_sessions(
|
||||
worker_id=worker_id,
|
||||
session_count=session_count,
|
||||
config=config,
|
||||
)
|
||||
)
|
||||
result_queue.put(
|
||||
{
|
||||
"worker_id": worker_id,
|
||||
"status": "ok",
|
||||
"sessions": sessions,
|
||||
}
|
||||
)
|
||||
except Exception as exc:
|
||||
result_queue.put({
|
||||
"worker_id": worker_id,
|
||||
"status": "error",
|
||||
"error": str(exc),
|
||||
"traceback": traceback.format_exc(),
|
||||
})
|
||||
result_queue.put(
|
||||
{
|
||||
"worker_id": worker_id,
|
||||
"status": "error",
|
||||
"error": str(exc),
|
||||
"traceback": traceback.format_exc(),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def build_summary(
|
||||
@@ -460,22 +517,33 @@ def build_summary(
|
||||
if len(all_chunk_finish_epochs) >= 2 and total_chunk_bytes > 0:
|
||||
duration_s = max(all_chunk_finish_epochs) - min(all_chunk_finish_epochs)
|
||||
if duration_s > 0:
|
||||
global_goodput_mbps = (total_chunk_bytes * 8.0 / duration_s / 1_000_000.0)
|
||||
global_goodput_mbps = (
|
||||
total_chunk_bytes * 8.0 / duration_s / 1_000_000.0
|
||||
)
|
||||
|
||||
bucket_throughputs_mbps = [(bytes_count * 8.0) / 1_000_000.0 for _, bytes_count in sorted(bucket_bytes.items())]
|
||||
bucket_throughputs_mbps = [
|
||||
(bytes_count * 8.0) / 1_000_000.0
|
||||
for _, bytes_count in sorted(bucket_bytes.items())
|
||||
]
|
||||
bucket_stats = summarize_series(bucket_throughputs_mbps)
|
||||
|
||||
chunk_gap_threshold_breaches = [value for value in chunk_gaps if value >= chunk_gap_threshold_ms]
|
||||
chunk_gap_threshold_breaches = [
|
||||
value for value in chunk_gaps if value >= chunk_gap_threshold_ms
|
||||
]
|
||||
non_success = len(sessions) - status_counts.get("success", 0)
|
||||
|
||||
fail_reasons: list[str] = []
|
||||
if non_success > 0:
|
||||
fail_reasons.append(f"{non_success} session(s) did not complete successfully.")
|
||||
fail_reasons.append(
|
||||
f"{non_success} session(s) did not complete successfully."
|
||||
)
|
||||
if not chunk_gaps:
|
||||
fail_reasons.append("No chunk gap data collected.")
|
||||
if chunk_gap_threshold_breaches:
|
||||
fail_reasons.append(f"{len(chunk_gap_threshold_breaches)} chunk gap(s) were >= "
|
||||
f"{chunk_gap_threshold_ms:.0f}ms.")
|
||||
fail_reasons.append(
|
||||
f"{len(chunk_gap_threshold_breaches)} chunk gap(s) were >= "
|
||||
f"{chunk_gap_threshold_ms:.0f}ms."
|
||||
)
|
||||
|
||||
passed = len(fail_reasons) == 0
|
||||
progressive_ratio = None
|
||||
@@ -486,18 +554,20 @@ def build_summary(
|
||||
"passed": passed,
|
||||
"fail_reasons": fail_reasons,
|
||||
"sessions": {
|
||||
"total":
|
||||
len(sessions),
|
||||
"success":
|
||||
status_counts.get("success", 0),
|
||||
"failed":
|
||||
status_counts.get("failed", 0),
|
||||
"timeout":
|
||||
status_counts.get("timeout", 0),
|
||||
"protocol_error":
|
||||
status_counts.get("protocol_error", 0),
|
||||
"other": (len(sessions) - (status_counts.get("success", 0) + status_counts.get("failed", 0) +
|
||||
status_counts.get("timeout", 0) + status_counts.get("protocol_error", 0))),
|
||||
"total": len(sessions),
|
||||
"success": status_counts.get("success", 0),
|
||||
"failed": status_counts.get("failed", 0),
|
||||
"timeout": status_counts.get("timeout", 0),
|
||||
"protocol_error": status_counts.get("protocol_error", 0),
|
||||
"other": (
|
||||
len(sessions)
|
||||
- (
|
||||
status_counts.get("success", 0)
|
||||
+ status_counts.get("failed", 0)
|
||||
+ status_counts.get("timeout", 0)
|
||||
+ status_counts.get("protocol_error", 0)
|
||||
)
|
||||
),
|
||||
},
|
||||
"chunk_gap_ms": {
|
||||
**chunk_gap_stats,
|
||||
@@ -536,39 +606,51 @@ def print_summary(
|
||||
bucket_bw = bandwidth["bucketed_1s"]
|
||||
|
||||
print("=== LTX2 Realtime Stress Test Summary ===")
|
||||
print("Run: "
|
||||
f"url={run_info['url']} clients={run_info['clients']} "
|
||||
f"processes={run_info['processes']} "
|
||||
f"preset={run_info['preset_id']} "
|
||||
f"curated_limit={run_info['curated_limit']}")
|
||||
print("Sessions: "
|
||||
f"total={sessions['total']} success={sessions['success']} "
|
||||
f"failed={sessions['failed']} timeout={sessions['timeout']} "
|
||||
f"protocol_error={sessions['protocol_error']}")
|
||||
print("Chunk gap ms: "
|
||||
f"min={format_num(chunk_gap['min'])} "
|
||||
f"p50={format_num(chunk_gap['p50'])} "
|
||||
f"p95={format_num(chunk_gap['p95'])} "
|
||||
f"p99={format_num(chunk_gap['p99'])} "
|
||||
f"max={format_num(chunk_gap['max'])} "
|
||||
f"threshold={format_num(chunk_gap['threshold_ms'])} "
|
||||
f"breaches={chunk_gap['breach_count']}")
|
||||
print("Queue wait ms: "
|
||||
f"min={format_num(queue_wait['min'])} "
|
||||
f"p50={format_num(queue_wait['p50'])} "
|
||||
f"p95={format_num(queue_wait['p95'])} "
|
||||
f"max={format_num(queue_wait['max'])}")
|
||||
print(
|
||||
"Run: "
|
||||
f"url={run_info['url']} clients={run_info['clients']} "
|
||||
f"processes={run_info['processes']} "
|
||||
f"preset={run_info['preset_id']} "
|
||||
f"curated_limit={run_info['curated_limit']}"
|
||||
)
|
||||
print(
|
||||
"Sessions: "
|
||||
f"total={sessions['total']} success={sessions['success']} "
|
||||
f"failed={sessions['failed']} timeout={sessions['timeout']} "
|
||||
f"protocol_error={sessions['protocol_error']}"
|
||||
)
|
||||
print(
|
||||
"Chunk gap ms: "
|
||||
f"min={format_num(chunk_gap['min'])} "
|
||||
f"p50={format_num(chunk_gap['p50'])} "
|
||||
f"p95={format_num(chunk_gap['p95'])} "
|
||||
f"p99={format_num(chunk_gap['p99'])} "
|
||||
f"max={format_num(chunk_gap['max'])} "
|
||||
f"threshold={format_num(chunk_gap['threshold_ms'])} "
|
||||
f"breaches={chunk_gap['breach_count']}"
|
||||
)
|
||||
print(
|
||||
"Queue wait ms: "
|
||||
f"min={format_num(queue_wait['min'])} "
|
||||
f"p50={format_num(queue_wait['p50'])} "
|
||||
f"p95={format_num(queue_wait['p95'])} "
|
||||
f"max={format_num(queue_wait['max'])}"
|
||||
)
|
||||
ratio = progressive["ratio"]
|
||||
ratio_text = "n/a" if ratio is None else f"{ratio * 100:.2f}%"
|
||||
print("Progressive streaming: "
|
||||
f"{progressive['success_sessions']}/"
|
||||
f"{progressive['eligible_sessions']} ({ratio_text})")
|
||||
print("Bandwidth Mbps: "
|
||||
f"per_session_avg={format_num(per_session_bw['avg'])} "
|
||||
f"per_session_p95={format_num(per_session_bw['p95'])} "
|
||||
f"global={format_num(bandwidth['global_goodput_mbps'])} "
|
||||
f"bucket_avg={format_num(bucket_bw['avg_mbps'])} "
|
||||
f"bucket_peak={format_num(bucket_bw['peak_mbps'])}")
|
||||
print(
|
||||
"Progressive streaming: "
|
||||
f"{progressive['success_sessions']}/"
|
||||
f"{progressive['eligible_sessions']} ({ratio_text})"
|
||||
)
|
||||
print(
|
||||
"Bandwidth Mbps: "
|
||||
f"per_session_avg={format_num(per_session_bw['avg'])} "
|
||||
f"per_session_p95={format_num(per_session_bw['p95'])} "
|
||||
f"global={format_num(bandwidth['global_goodput_mbps'])} "
|
||||
f"bucket_avg={format_num(bucket_bw['avg_mbps'])} "
|
||||
f"bucket_peak={format_num(bucket_bw['peak_mbps'])}"
|
||||
)
|
||||
print(f"VERDICT: {'PASS' if summary['passed'] else 'FAIL'}")
|
||||
if summary["fail_reasons"]:
|
||||
print("Fail reasons:")
|
||||
@@ -588,8 +670,10 @@ def distribute_sessions(total_clients: int, process_count: int) -> list[int]:
|
||||
|
||||
def run_stress(args: argparse.Namespace) -> tuple[dict[str, Any], int]:
|
||||
if websockets is None:
|
||||
raise RuntimeError("Missing dependency: websockets. Install it before running this "
|
||||
"stress test.")
|
||||
raise RuntimeError(
|
||||
"Missing dependency: websockets. Install it before running this "
|
||||
"stress test."
|
||||
)
|
||||
|
||||
preset_file = Path(args.preset_file).expanduser().resolve()
|
||||
selected_preset_id, curated_prompts, total_prompt_count = load_curated_prompts(
|
||||
@@ -651,8 +735,13 @@ def run_stress(args: argparse.Namespace) -> tuple[dict[str, Any], int]:
|
||||
|
||||
start_event.set()
|
||||
|
||||
result_deadline = (time.monotonic() + args.connect_timeout_s + args.session_timeout_s +
|
||||
args.post_complete_wait_s + 180.0)
|
||||
result_deadline = (
|
||||
time.monotonic()
|
||||
+ args.connect_timeout_s
|
||||
+ args.session_timeout_s
|
||||
+ args.post_complete_wait_s
|
||||
+ 180.0
|
||||
)
|
||||
worker_results: list[dict[str, Any]] = []
|
||||
while len(worker_results) < len(processes):
|
||||
timeout_s = max(0.1, result_deadline - time.monotonic())
|
||||
@@ -676,20 +765,24 @@ def run_stress(args: argparse.Namespace) -> tuple[dict[str, Any], int]:
|
||||
if result.get("status") == "ok":
|
||||
sessions.extend(result.get("sessions", []))
|
||||
else:
|
||||
worker_errors.append({
|
||||
"worker_id": result.get("worker_id"),
|
||||
"error": result.get("error"),
|
||||
"traceback": result.get("traceback"),
|
||||
})
|
||||
worker_errors.append(
|
||||
{
|
||||
"worker_id": result.get("worker_id"),
|
||||
"error": result.get("error"),
|
||||
"traceback": result.get("traceback"),
|
||||
}
|
||||
)
|
||||
|
||||
received_workers = {result.get("worker_id") for result in worker_results}
|
||||
expected_workers = set(range(len(processes)))
|
||||
missing_workers = sorted(expected_workers - received_workers)
|
||||
for worker_id in missing_workers:
|
||||
worker_errors.append({
|
||||
"worker_id": worker_id,
|
||||
"error": "No worker result received.",
|
||||
})
|
||||
worker_errors.append(
|
||||
{
|
||||
"worker_id": worker_id,
|
||||
"error": "No worker result received.",
|
||||
}
|
||||
)
|
||||
|
||||
run_end_epoch = time.time()
|
||||
run_end_iso = iso_from_epoch(run_end_epoch)
|
||||
@@ -702,8 +795,9 @@ def run_stress(args: argparse.Namespace) -> tuple[dict[str, Any], int]:
|
||||
|
||||
if worker_errors:
|
||||
summary["passed"] = False
|
||||
summary["fail_reasons"] = list(
|
||||
summary["fail_reasons"]) + [f"{len(worker_errors)} worker error(s) occurred."]
|
||||
summary["fail_reasons"] = list(summary["fail_reasons"]) + [
|
||||
f"{len(worker_errors)} worker error(s) occurred."
|
||||
]
|
||||
|
||||
output_payload = {
|
||||
"run_info": {
|
||||
@@ -739,7 +833,9 @@ def run_stress(args: argparse.Namespace) -> tuple[dict[str, Any], int]:
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(description="Multiprocess realtime stress test for LTX2 streaming.", )
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Multiprocess realtime stress test for LTX2 streaming.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"-u",
|
||||
"--url",
|
||||
|
||||
@@ -47,11 +47,13 @@ def test_persist_session_init_image_returns_none_when_missing_data():
|
||||
|
||||
def test_persist_session_init_image_rejects_unsupported_mime():
|
||||
with pytest.raises(ValueError, match="PNG, JPEG, or WebP"):
|
||||
persist_session_init_image({
|
||||
"name": "frame.gif",
|
||||
"mime_type": "image/gif",
|
||||
"data_url": "data:image/gif;base64,R0lGODlhAQABAAAAACw=",
|
||||
})
|
||||
persist_session_init_image(
|
||||
{
|
||||
"name": "frame.gif",
|
||||
"mime_type": "image/gif",
|
||||
"data_url": "data:image/gif;base64,R0lGODlhAQABAAAAACw=",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_persist_session_init_image_rejects_large_payload(monkeypatch):
|
||||
@@ -64,8 +66,10 @@ def test_persist_session_init_image_rejects_large_payload(monkeypatch):
|
||||
monkeypatch.setattr(base64, "b64decode", fake_b64decode)
|
||||
|
||||
with pytest.raises(ValueError, match="15 MB or smaller"):
|
||||
persist_session_init_image({
|
||||
"name": "frame.png",
|
||||
"mime_type": "image/png",
|
||||
"data_url": data_url,
|
||||
})
|
||||
persist_session_init_image(
|
||||
{
|
||||
"name": "frame.png",
|
||||
"mime_type": "image/png",
|
||||
"data_url": data_url,
|
||||
}
|
||||
)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -15,5 +15,6 @@ def _utc_now_iso() -> str:
|
||||
PROMPT_EXTENSION_FAILURE_USER_MESSAGE = ("Prompt extension failed for this request.")
|
||||
|
||||
|
||||
def _resolve_generation_segment_cap(*, single_clip_mode: bool, cap: int) -> int:
|
||||
return 0 if single_clip_mode else cap
|
||||
def _resolve_generation_segment_cap(*, single_clip_mode: bool, cap: int, manual_continuation_mode: bool = False) -> int:
|
||||
# Steering (manual continuation) lets the user keep going indefinitely, like single-clip mode.
|
||||
return 0 if (single_clip_mode or manual_continuation_mode) else cap
|
||||
|
||||
+23
-5
@@ -1,9 +1,9 @@
|
||||
"""LTX-2 model lifecycle and continuation conditioning.
|
||||
"""LTX2 model lifecycle and continuation conditioning.
|
||||
|
||||
Runs inside a GPU worker subprocess. Owns the model, the audio
|
||||
encoder, and the per-session continuation state carried across
|
||||
segments. Callers must set ``os.environ["CUDA_VISIBLE_DEVICES"]``
|
||||
before constructing ``LTX2GenerationBackend`` — all ``fastvideo.*``
|
||||
before constructing ``VideoGenerationWorker`` — all ``fastvideo.*``
|
||||
imports are deferred to method bodies so nothing touches CUDA at
|
||||
module import time.
|
||||
"""
|
||||
@@ -14,6 +14,9 @@ import gc
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
@@ -32,7 +35,6 @@ from dreamverse.config import (
|
||||
DREAMVERSE_LORA_STACK,
|
||||
_resolve_lora_spec,
|
||||
)
|
||||
from dreamverse.generation_contracts import StepResult
|
||||
|
||||
# Multi-frame decoded continuation defaults from
|
||||
# examples/inference/basic/basic_ltx2_distilled_video_continuation.py.
|
||||
@@ -78,6 +80,22 @@ def _reset_lora_registry(worker) -> dict:
|
||||
return {"status": "lora_registry_reset"}
|
||||
|
||||
|
||||
@dataclass
|
||||
class StepResult:
|
||||
"""Output of one generation step.
|
||||
|
||||
``head_trim_frames`` / ``head_trim_audio_frames`` are derived here
|
||||
so downstream AV streaming never needs to import conditioning
|
||||
constants.
|
||||
"""
|
||||
frames: list
|
||||
audio: Any
|
||||
audio_sample_rate: int | None
|
||||
timings: dict
|
||||
head_trim_frames: int
|
||||
head_trim_audio_frames: int
|
||||
|
||||
|
||||
class ContinuationState:
|
||||
"""Per-session video + audio conditioning carried across segments."""
|
||||
|
||||
@@ -184,7 +202,7 @@ class ContinuationState:
|
||||
self.audio_latents = latents.detach().clone().cpu()
|
||||
|
||||
|
||||
class LTX2GenerationBackend:
|
||||
class VideoGenerationWorker:
|
||||
"""Single-GPU LTX2 generator with continuation state.
|
||||
|
||||
Caller must set ``os.environ["CUDA_VISIBLE_DEVICES"]`` before
|
||||
@@ -471,7 +489,7 @@ class LTX2GenerationBackend:
|
||||
num_inference_steps=NUM_INFERENCE_STEPS,
|
||||
guidance_scale=1.0,
|
||||
seed=10,
|
||||
ltx2_image_crf=0.0,
|
||||
ltx2_image_crf=(33.0 if image_path and segment_idx == 1 else 0.0),
|
||||
image_path=image_path if segment_idx == 1 else None,
|
||||
return_continuation_state=False,
|
||||
)
|
||||
@@ -70,7 +70,7 @@
|
||||
<mxCell id="dispatcher" value="command dispatcher

gpu_worker_process() branches on
CommandType; asserts payload type

INIT / WARMUP / RELOAD_MODEL
USER_JOIN / USER_STEP / USER_LEAVE
SHUTDOWN" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe6cc;strokeColor=#d79b00;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
ltx2_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxCell id="do_step" value="VideoGenerationWorker.generate_step()
video_generation.py:380

reads + updates ContinuationState,
calls generator" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1120" width="240" height="120" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="stream_av" value="stream_fmp4()
av_streaming.py:121

trims overlap, pipes to ffmpeg,
publishes StreamInit / StreamChunk /
StreamComplete via callback" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#b1d8d7;strokeColor=#23445d;fontSize=11;align=left;spacingLeft=10;spacingTop=8;fontStyle=1;" parent="1" vertex="1">
|
||||
@@ -79,13 +79,13 @@
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-2" value="" style="edgeStyle=none;html=1;" parent="1" source="generator" target="Ot8BU52QTIb4EhyRSe7I-1" edge="1">
|
||||
<mxGeometry relative="1" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
ltx2_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxCell id="generator" value="VideoGenerator (fastvideo)

LTX2 DiT + refine upsampler
FP4 quant, torch.compile

owned by VideoGenerationWorker
video_generation.py:211" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="460" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="ffmpeg" value="ffmpeg subprocess

libx264 / *_nvenc
fragmented mp4" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=11;" parent="1" vertex="1">
|
||||
<mxGeometry x="800" y="1300" width="260" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="caches" value="ContinuationState
ltx2_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="caches" value="ContinuationState
video_generation.py:89

• video_images: list[PIL.Image]
• audio_latents: torch.Tensor (CPU)

carried across segments" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#e1d5e7;strokeColor=#9673a6;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="120" y="1300" width="240" height="100" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="e_cp" value="acquire" style="edgeStyle=orthogonalEdgeStyle;rounded=0;html=1;strokeColor=#6c8ebf;endArrow=classic;fontSize=11;exitX=0.5;exitY=1;exitDx=0;exitDy=0;entryX=0.5;entryY=0;entryDx=0;entryDy=0;" parent="1" source="client" target="pool" edge="1">
|
||||
@@ -250,7 +250,7 @@
|
||||
<mxPoint x="690" y="880"/>
|
||||
</Array>
|
||||
</mxCell>
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender ltx2_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxCell id="legend" value="Legend

■ blue client / external
■ green main-process pool/slot
 (methods — italic label)
■ yellow containers (routing state)
■ red IPC primitives (mp.Queue, mp.RawArray)

Worker subprocess modules:
■ orange gpu_pool.py (dispatcher)
■ lavender video_generation.py
■ teal av_streaming.py
■ gray worker_ipc.py (shared types)

Flow:
 client → pool → slot
 → _send_command(_tagged) → command_queue
 → dispatcher → generate_step()
 → stream_fmp4() → ffmpeg
 → shared_buf + response_queue
 → _response_reader → futures / stream_queues
 → client awaits (via main.py AV loop)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#f5f5f5;strokeColor=#999999;fontSize=11;align=left;spacingLeft=10;spacingTop=8;" parent="1" vertex="1">
|
||||
<mxGeometry x="39" y="-200" width="270" height="380" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="Ot8BU52QTIb4EhyRSe7I-1" value="FastVideo video_generator" style="whiteSpace=wrap;html=1;fontSize=11;fillColor=#e1d5e7;strokeColor=#9673a6;rounded=1;" parent="1" vertex="1">
|
||||
@@ -389,10 +389,10 @@
|
||||
<mxCell id="cw2" value="from fastvideo.entrypoints.video_generator import VideoGenerator
from fastvideo.models.dits.ltx2 import DEFAULT_LTX2_AUDIO_*

** Dreamverse reaches into fastvideo internals here **" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="695" width="550" height="60" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (ltx2_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw3" value="on Command(INIT):
 VideoGenerationWorker.initialize() (video_generation.py:247)
 maybe_download_model(model_id)
 VideoGenerator.from_pretrained(path, FP4Config, PipelineConfig)
 load audio VAE, resolve refine upsampler
 resp_q.put(InitAck(success=True))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="765" width="550" height="95" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (ltx2_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="cw4" value="on Command(WARMUP) with WarmupPayload:
 VideoGenerationWorker.warmup(payload.prompt) (video_generation.py:518)
 two synthetic segments prime caches + torch.compile
 resp_q.put(WarmupComplete(timings=...))" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffffff;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="675" y="870" width="550" height="55" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="cw5" value="enter main worker loop → waits for JOIN_USER / USER_STEP / LEAVE" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#c8e6c9;strokeColor=#388e3c;fontSize=11;fontStyle=1;fontFamily=monospace;" parent="1" vertex="1">
|
||||
@@ -534,7 +534,7 @@
|
||||
<mxPoint x="1040" y="1610" as="targetPoint"/>
|
||||
</mxGeometry>
|
||||
</mxCell>
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (ltx2_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxCell id="dm11a" value="10a. worker runs:
VideoGenerationWorker.generate_step()
 (video_generation.py:380)
 → generator.generate_video()
 → updates ContinuationState
then stream_fmp4() (av_streaming.py:121)
 → ffmpeg (rawvideo+wav → fmp4)" style="rounded=1;whiteSpace=wrap;html=1;fillColor=#ffe0b2;strokeColor=#d79b00;fontSize=10;align=left;spacingLeft=8;fontFamily=monospace;" parent="1" vertex="1">
|
||||
<mxGeometry x="955" y="1640" width="180" height="70" as="geometry"/>
|
||||
</mxCell>
|
||||
<mxCell id="dm11" value="10b. resp_q.put(MediaInit / MediaChunk / MediaComplete / StepComplete)" style="endArrow=classic;html=1;strokeColor=#b85450;fontSize=10;labelBackgroundColor=#ffffff;" parent="1" edge="1">
|
||||
|
||||
File diff suppressed because one or more lines are too long
|
Before Width: | Height: | Size: 85 KiB After Width: | Height: | Size: 85 KiB |
@@ -29,6 +29,9 @@ test = [
|
||||
dreamverse-server = "dreamverse.server_entry:cli"
|
||||
dreamverse-mock-server = "dreamverse.mock_server:cli"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["dreamverse*"]
|
||||
|
||||
[tool.uv]
|
||||
package = false
|
||||
|
||||
|
||||
Executable
+67
@@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env bash
|
||||
# launch-dreamverse.sh — launch dreamverse-server on a compute node.
|
||||
#
|
||||
# Usage (from repo root):
|
||||
# bash apps/dreamverse/scripts/launch-dreamverse.sh # GPUs 0-3, SP_SIZE=4
|
||||
# CUDA_VISIBLE_DEVICES=0 bash apps/dreamverse/scripts/launch-dreamverse.sh # single GPU, SP_SIZE=1
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
CONDA_PREFIX="$HOME/miniconda3/envs/dreamverse"
|
||||
CUDA_RT_DIR="$CONDA_PREFIX/lib/python3.11/site-packages/nvidia/cuda_runtime/lib"
|
||||
GXX="$CONDA_PREFIX/bin/aarch64-conda-linux-gnu-g++"
|
||||
|
||||
|
||||
export CUDA_HOME="$CONDA_PREFIX"
|
||||
export FASTVIDEO_ENABLE_STARTUP_WARMUP=true
|
||||
export FASTVIDEO_ENABLE_PROMPT_SAFETY=false
|
||||
export DREAMVERSE_MAX_AUTOTUNE=true
|
||||
export LTX2_USE_DISTILLED_SIGMAS=0
|
||||
export LTX2_VIDEO_CONDITIONING_NUM_FRAMES=1
|
||||
export AUDIO_CONDITIONING_NUM_FRAMES=41
|
||||
export DREAMVERSE_SESSION_TIMEOUT_SECONDS="${DREAMVERSE_SESSION_TIMEOUT_SECONDS:-1800}"
|
||||
# GB200 max-autotune warmup compiles can run for hours; keep the watchdog generous here
|
||||
export FASTVIDEO_STARTUP_WARMUP_TIMEOUT_SECONDS="${FASTVIDEO_STARTUP_WARMUP_TIMEOUT_SECONDS:-24000}"
|
||||
export CEREBRAS_API_KEY="${CEREBRAS_API_KEY:-}" # set this in your env or ~/.env
|
||||
export FASTVIDEO_PROMPT_CEREBRAS_MODEL="gpt-oss-120b"
|
||||
export TORCHINDUCTOR_CACHE_DIR="$HOME/.cache/torchinductor"
|
||||
export TRITON_CACHE_DIR="$HOME/.triton/cache"
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a"
|
||||
|
||||
# Compiler env (needed for flashinfer JIT compilation at server startup)
|
||||
export CXX="$CONDA_PREFIX/compiler_compat/g++"
|
||||
export CC="$CONDA_PREFIX/compiler_compat/gcc"
|
||||
export CUDAHOSTCXX="$GXX"
|
||||
export NVCC_PREPEND_FLAGS="-ccbin $GXX -allow-unsupported-compiler"
|
||||
|
||||
# Link against libcudart.so.12 at JIT compile time; stubs for libcuda.so
|
||||
# cuda-compat has libcudart.so -> libcudart.so.12 (linker needs unversioned name)
|
||||
export LIBRARY_PATH="$CONDA_PREFIX/lib/cuda-compat:$CONDA_PREFIX/lib/stubs"
|
||||
|
||||
# Only libcudart.so.12 at runtime — prevents cuDNN from seeing .so.13
|
||||
export LD_LIBRARY_PATH="$CUDA_RT_DIR"
|
||||
|
||||
CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-0,1,2,3}"
|
||||
export FASTVIDEO_GPU_COUNT="${FASTVIDEO_GPU_COUNT:-all}"
|
||||
# Default SP size to the usable GPU count so single-GPU invocations work.
|
||||
# A numeric FASTVIDEO_GPU_COUNT caps the pool below the visible count, and an
|
||||
# SP size above the pool size fails GPUPool startup with "Not enough GPUs".
|
||||
IFS=',' read -ra _VISIBLE_GPUS <<< "$CUDA_VISIBLE_DEVICES"
|
||||
# Count only non-empty tokens, matching gpu_pool.get_available_gpus (e.g. ",0,1" is 2 GPUs).
|
||||
_USABLE_GPU_COUNT=0
|
||||
for _gpu in "${_VISIBLE_GPUS[@]}"; do
|
||||
[[ -n "${_gpu//[[:space:]]/}" ]] && _USABLE_GPU_COUNT=$((_USABLE_GPU_COUNT + 1))
|
||||
done
|
||||
if [[ "$FASTVIDEO_GPU_COUNT" =~ ^[0-9]+$ ]] && (( FASTVIDEO_GPU_COUNT < _USABLE_GPU_COUNT )); then
|
||||
_USABLE_GPU_COUNT="$FASTVIDEO_GPU_COUNT"
|
||||
fi
|
||||
export DREAMVERSE_SP_SIZE="${DREAMVERSE_SP_SIZE:-$_USABLE_GPU_COUNT}"
|
||||
PORT="${DREAMVERSE_PORT:-8009}"
|
||||
|
||||
FFMPEG_ENV="$(dirname "$0")/ffmpeg-env.sh"
|
||||
# shellcheck source=ffmpeg-env.sh
|
||||
[[ -f "$FFMPEG_ENV" ]] && source "$FFMPEG_ENV"
|
||||
|
||||
echo "==> Launching dreamverse-server on GPU $CUDA_VISIBLE_DEVICES port $PORT"
|
||||
CUDA_VISIBLE_DEVICES="$CUDA_VISIBLE_DEVICES" \
|
||||
"$CONDA_PREFIX/bin/dreamverse-server" --host 0.0.0.0 --port "$PORT"
|
||||
Executable
+30
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env bash
|
||||
# launch-frontend.sh — start the Dreamverse Next.js dev server and ngrok tunnel.
|
||||
#
|
||||
# Usage (from repo root):
|
||||
# bash apps/dreamverse/scripts/launch-frontend.sh
|
||||
#
|
||||
# Override backend or ngrok URL via env:
|
||||
# BACKEND_HOST=1.2.3.4 bash apps/dreamverse/scripts/launch-frontend.sh
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
CONDA_PREFIX="$HOME/miniconda3/envs/dreamverse"
|
||||
BACKEND_HOST="${BACKEND_HOST:-10.244.18.228}"
|
||||
BACKEND_PORT="${BACKEND_PORT:-8009}"
|
||||
NGROK_URL="${NGROK_URL:-ltx23.ngrok.app}"
|
||||
WEB_DIR="$(git rev-parse --show-toplevel)/apps/dreamverse/web"
|
||||
|
||||
cleanup() {
|
||||
echo "==> Shutting down..."
|
||||
kill "$FRONTEND_PID" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
echo "==> Starting frontend (backend: $BACKEND_HOST:$BACKEND_PORT)"
|
||||
BACKEND_HOST="$BACKEND_HOST" BACKEND_PORT="$BACKEND_PORT" \
|
||||
npm run --prefix "$WEB_DIR" dev &
|
||||
FRONTEND_PID=$!
|
||||
|
||||
echo "==> Starting ngrok tunnel -> $NGROK_URL"
|
||||
"$CONDA_PREFIX/bin/ngrok" http --url="$NGROK_URL" 5299
|
||||
@@ -39,17 +39,6 @@ export FASTVIDEO_GENERATION_SEGMENT_CAP="${FASTVIDEO_GENERATION_SEGMENT_CAP:-6}"
|
||||
export FASTVIDEO_PROMPT_AUTO_SLEEP_MS="${FASTVIDEO_PROMPT_AUTO_SLEEP_MS:-120}"
|
||||
export FASTVIDEO_PROMPT_AUTO_TIMEOUT_MS="${FASTVIDEO_PROMPT_AUTO_TIMEOUT_MS:-1800}"
|
||||
|
||||
if [[ "${ENABLE_TORCH_COMPILE}" == "1" ]]; then
|
||||
# Persist Inductor, AOTAutograd, and Triton artifacts across launches.
|
||||
export DREAMVERSE_TORCH_COMPILE_CACHE_ROOT="${DREAMVERSE_TORCH_COMPILE_CACHE_ROOT:-${HOME}/.cache/dreamverse/torch_compile}"
|
||||
export TORCHINDUCTOR_CACHE_DIR="${TORCHINDUCTOR_CACHE_DIR:-${DREAMVERSE_TORCH_COMPILE_CACHE_ROOT}/inductor}"
|
||||
export TRITON_CACHE_DIR="${TRITON_CACHE_DIR:-${DREAMVERSE_TORCH_COMPILE_CACHE_ROOT}/triton}"
|
||||
export TORCHINDUCTOR_FX_GRAPH_CACHE="${TORCHINDUCTOR_FX_GRAPH_CACHE:-1}"
|
||||
export TORCHINDUCTOR_AUTOGRAD_CACHE="${TORCHINDUCTOR_AUTOGRAD_CACHE:-1}"
|
||||
mkdir -p "${TORCHINDUCTOR_CACHE_DIR}" "${TRITON_CACHE_DIR}"
|
||||
echo "[launch-demo] torch.compile cache: ${DREAMVERSE_TORCH_COMPILE_CACHE_ROOT}"
|
||||
fi
|
||||
|
||||
cd "${DREAMVERSE_ROOT}"
|
||||
|
||||
if ! command -v dreamverse-server >/dev/null 2>&1; then
|
||||
|
||||
@@ -8,10 +8,12 @@ import modal
|
||||
|
||||
IMAGE = os.environ.get("DREAMVERSE_IMAGE")
|
||||
if not IMAGE:
|
||||
raise RuntimeError("DREAMVERSE_IMAGE is required. Set it to a published SHA-specific Dreamverse image, "
|
||||
"for example a dreamverse-backend-cuda13.0.0-sha-* tag or a "
|
||||
"dreamverse-ui-cuda13.0.0-sha-* tag if serving the static UI. "
|
||||
"CUDA 12 / cu126 images use the corresponding cuda12.6.3 tag.")
|
||||
raise RuntimeError(
|
||||
"DREAMVERSE_IMAGE is required. Set it to a published SHA-specific Dreamverse image, "
|
||||
"for example a dreamverse-backend-cuda13.0.0-sha-* tag or a "
|
||||
"dreamverse-ui-cuda13.0.0-sha-* tag if serving the static UI. "
|
||||
"CUDA 12 / cu126 images use the corresponding cuda12.6.3 tag."
|
||||
)
|
||||
|
||||
# ``@modal.web_server`` invokes ``serve()`` directly and bypasses the image
|
||||
# ENTRYPOINT (``docker/docker_entrypoint.sh``). That entrypoint normally
|
||||
@@ -63,10 +65,14 @@ def serve():
|
||||
# ``or ""`` collapses ``None`` (unset) into an empty string, ``.strip()``
|
||||
# collapses whitespace-only values (e.g. ``" "``) — both should be
|
||||
# treated as missing.
|
||||
missing = [k for k in _REQUIRED_SECRET_KEYS if not (os.environ.get(k) or "").strip()]
|
||||
missing = [
|
||||
k for k in _REQUIRED_SECRET_KEYS
|
||||
if not (os.environ.get(k) or "").strip()
|
||||
]
|
||||
if missing:
|
||||
raise RuntimeError("dreamverse-api-keys secret is missing required entries: "
|
||||
f"{', '.join(missing)}. Add them with `modal secret create "
|
||||
"dreamverse-api-keys ... --force` and redeploy "
|
||||
"(see apps/dreamverse/scripts/modal/README.md).")
|
||||
raise RuntimeError(
|
||||
"dreamverse-api-keys secret is missing required entries: "
|
||||
f"{', '.join(missing)}. Add them with `modal secret create "
|
||||
"dreamverse-api-keys ... --force` and redeploy "
|
||||
"(see apps/dreamverse/scripts/modal/README.md).")
|
||||
subprocess.Popen(["dreamverse-server", "--host", "0.0.0.0", "--port", "8009"])
|
||||
|
||||
+105
@@ -0,0 +1,105 @@
|
||||
#!/usr/bin/env bash
|
||||
# setup-dreamverse-env.sh — create and configure the dreamverse conda env
|
||||
# from scratch on this aarch64 NFS Slurm cluster.
|
||||
#
|
||||
# Run from the login node (from the repo root):
|
||||
# bash apps/dreamverse/scripts/setup-dreamverse-env.sh
|
||||
#
|
||||
# After this script completes, use launch-dreamverse.sh on a compute node.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
REPO_ROOT="$(git rev-parse --show-toplevel)"
|
||||
ENV_NAME="dreamverse"
|
||||
LOCAL_DIR="/mnt/local/hal-kevin" # cache/pkgs — keep on local disk
|
||||
CONDA_PREFIX="$HOME/miniconda3/envs/$ENV_NAME" # env — on shared NFS so it survives node changes
|
||||
|
||||
echo "==> Removing existing env if present"
|
||||
conda env remove -p "$CONDA_PREFIX" -y 2>/dev/null || true
|
||||
rm -rf "$CONDA_PREFIX" 2>/dev/null || true
|
||||
|
||||
echo "==> Creating conda env at $CONDA_PREFIX"
|
||||
CONDA_PKGS_DIRS="$LOCAL_DIR/conda/pkgs" conda create -p "$CONDA_PREFIX" python=3.11 -y
|
||||
|
||||
GXX="$CONDA_PREFIX/bin/aarch64-conda-linux-gnu-g++"
|
||||
GCC="$CONDA_PREFIX/bin/aarch64-conda-linux-gnu-gcc"
|
||||
|
||||
echo "==> Installing compiler"
|
||||
CONDA_PKGS_DIRS="$LOCAL_DIR/conda/pkgs" conda install -p "$CONDA_PREFIX" gxx_linux-aarch64 -y
|
||||
|
||||
echo "==> Installing CUDA toolkit (nvcc + headers)"
|
||||
CONDA_PKGS_DIRS="$LOCAL_DIR/conda/pkgs" conda install -p "$CONDA_PREFIX" -c nvidia cuda-toolkit -y
|
||||
|
||||
echo "==> Hiding conflicting libcudart.so.13 immediately"
|
||||
mkdir -p "$CONDA_PREFIX/lib/hidden"
|
||||
mv "$CONDA_PREFIX"/lib/libcudart.so* "$CONDA_PREFIX/lib/hidden/" 2>/dev/null || true
|
||||
|
||||
echo "==> Fixing compiler_compat symlinks"
|
||||
mkdir -p "$CONDA_PREFIX/compiler_compat"
|
||||
ln -sf "$GXX" "$CONDA_PREFIX/compiler_compat/g++"
|
||||
ln -sf "$GCC" "$CONDA_PREFIX/compiler_compat/gcc"
|
||||
|
||||
echo "==> Symlinking CUDA headers to standard location"
|
||||
for f in "$CONDA_PREFIX/targets/sbsa-linux/include/"*; do
|
||||
ln -sf "$f" "$CONDA_PREFIX/include/$(basename "$f")" 2>/dev/null || true
|
||||
done
|
||||
|
||||
echo "==> Installing ffmpeg (native build with x264 + NVENC)"
|
||||
CUDA_PREFIX="$CONDA_PREFIX" bash "$REPO_ROOT/apps/dreamverse/scripts/install_native_ffmpeg.sh"
|
||||
|
||||
echo "==> Installing pip and uv"
|
||||
CONDA_PKGS_DIRS="$LOCAL_DIR/conda/pkgs" conda install -p "$CONDA_PREFIX" pip -y
|
||||
"$CONDA_PREFIX/bin/pip" install uv
|
||||
|
||||
echo "==> Setting compiler env vars"
|
||||
export UV_CACHE_DIR="$LOCAL_DIR/cache"
|
||||
export UV_LINK_MODE=copy
|
||||
export CXX="$CONDA_PREFIX/compiler_compat/g++"
|
||||
export CC="$CONDA_PREFIX/compiler_compat/gcc"
|
||||
export CUDAHOSTCXX="$GXX"
|
||||
export NVCC_PREPEND_FLAGS="-ccbin $GXX -allow-unsupported-compiler"
|
||||
export CUDA_HOME="$CONDA_PREFIX"
|
||||
# Only build for GB200 (sm_100a); CUDA 13 dropped support for older archs
|
||||
export TORCH_CUDA_ARCH_LIST="10.0a"
|
||||
|
||||
echo "==> Installing torch with CUDA 12.8"
|
||||
UV_CACHE_DIR="$LOCAL_DIR/cache" "$CONDA_PREFIX/bin/uv" pip install torch==2.11.0 torchvision \
|
||||
--index-url https://download.pytorch.org/whl/cu128
|
||||
|
||||
echo "==> Hiding any newly introduced libcudart.so.13"
|
||||
mv "$CONDA_PREFIX"/lib/libcudart.so* "$CONDA_PREFIX/lib/hidden/" 2>/dev/null || true
|
||||
|
||||
# Set paths now that torch (and its nvidia packages) are installed
|
||||
CUDA_RT_DIR="$CONDA_PREFIX/lib/python3.11/site-packages/nvidia/cuda_runtime/lib"
|
||||
CUDA_RT_SO="$(ls "$CUDA_RT_DIR"/libcudart.so.* 2>/dev/null | head -1)"
|
||||
|
||||
# The pip nvidia package only has libcudart.so.12 (versioned), not libcudart.so.
|
||||
# The linker needs the unversioned name to satisfy -lcudart. Create a compat dir.
|
||||
mkdir -p "$CONDA_PREFIX/lib/cuda-compat"
|
||||
ln -sf "$CUDA_RT_SO" "$CONDA_PREFIX/lib/cuda-compat/libcudart.so"
|
||||
|
||||
export LIBRARY_PATH="$CONDA_PREFIX/lib/cuda-compat:$CONDA_PREFIX/lib/stubs"
|
||||
export CMAKE_ARGS="-DCUDA_CUDART_LIBRARY=$CUDA_RT_SO -DCUDA_INCLUDE_DIRS=$CONDA_PREFIX/targets/sbsa-linux/include"
|
||||
|
||||
echo "==> Installing build tools"
|
||||
"$CONDA_PREFIX/bin/pip" install scikit-build-core cmake ninja
|
||||
|
||||
echo "==> Initializing git submodules"
|
||||
cd "$REPO_ROOT"
|
||||
git submodule update --init fastvideo-kernel/include/cutlass fastvideo-kernel/include/tk
|
||||
|
||||
echo "==> Building fastvideo-kernel from local source"
|
||||
UV_CACHE_DIR="$LOCAL_DIR/cache" "$CONDA_PREFIX/bin/uv" pip install \
|
||||
-e "./fastvideo-kernel" --no-build-isolation
|
||||
|
||||
echo "==> Installing fastvideo + dreamverse extras"
|
||||
UV_CACHE_DIR="$LOCAL_DIR/cache" "$CONDA_PREFIX/bin/uv" pip install \
|
||||
-e ".[dreamverse]" --no-build-isolation
|
||||
|
||||
echo "==> Installing flashinfer-python (pinned, must be last)"
|
||||
UV_CACHE_DIR="$LOCAL_DIR/cache" "$CONDA_PREFIX/bin/uv" pip install \
|
||||
https://github.com/flashinfer-ai/flashinfer/releases/download/v0.6.11.post3/flashinfer_python-0.6.11.post3-py3-none-any.whl
|
||||
|
||||
echo ""
|
||||
echo "Done. On a compute node run (GPUs 0-3 by default; set CUDA_VISIBLE_DEVICES to restrict):"
|
||||
echo " bash apps/dreamverse/scripts/launch-dreamverse.sh"
|
||||
@@ -110,7 +110,7 @@ default_request:
|
||||
fps: 24 # internal: gpu_pool.py:85 TARGET_FPS
|
||||
|
||||
streaming:
|
||||
# internal: config.py:33 SESSION_TIMEOUT_SECONDS = 300
|
||||
# internal: config.py SESSION_TIMEOUT_SECONDS (env DREAMVERSE_SESSION_TIMEOUT_SECONDS, default 300)
|
||||
session_timeout_seconds: 300
|
||||
# internal: config.py:282-284 GENERATION_SEGMENT_CAP default 6
|
||||
generation_segment_cap: 6
|
||||
|
||||
@@ -9,7 +9,7 @@ import SessionTimeoutModal from "@/components/SessionTimeoutModal";
|
||||
import Sidebar from "@/components/Sidebar";
|
||||
import Header from "@/components/Header";
|
||||
import VideoPlayer from "@/components/VideoPlayer";
|
||||
import Workspace from "@/components/Workspace";
|
||||
import Workspace, { SceneHistoryList } from "@/components/Workspace";
|
||||
import { saveProject, saveProjectMetadata, listProjects, loadProjectClips, deleteProject, pruneOldProjects, type StoredProject, type StoredClip } from "@/lib/projectStorage";
|
||||
import { isInfrastructureError } from "@/lib/ws/reducer";
|
||||
import { useStore } from "@/hooks/useStore";
|
||||
@@ -31,7 +31,7 @@ import { applyNormalizedSocketEvent } from "@/lib/ws/reducer";
|
||||
import { createPromptWindowStore } from "@/stores/promptWindow";
|
||||
import { createRewriteStore } from "@/stores/rewrite";
|
||||
import { createSessionStore } from "@/stores/session";
|
||||
import { createStreamStore } from "@/stores/stream";
|
||||
import { createStreamStore, USER_PROMPT_SOURCES } from "@/stores/stream";
|
||||
import { createUiStore } from "@/stores/ui";
|
||||
import { Button } from "@/components/ui/button";
|
||||
|
||||
@@ -80,7 +80,7 @@ function yieldToEventLoop(): Promise<void> {
|
||||
|
||||
const HERO_WAVE_LIGHT = ["#2A4A98", "#4878E5", "#6FA0F2", "#B0BCC8", "#E8D99E", "#D8C844", "#C2A620"];
|
||||
const HERO_WAVE_DARK = ["#143468", "#1E58B8", "#3892F0", "#80B8E8", "#B8D0EA", "#E2D498", "#DABB50"];
|
||||
const HERO_TEXT = "Direct scenes in seconds";
|
||||
const HERO_TEXT = "Direct scenes in seconds with";
|
||||
|
||||
function HeroTagline() {
|
||||
const ref = useRef<HTMLHeadingElement>(null);
|
||||
@@ -166,6 +166,8 @@ function HeroTagline() {
|
||||
</span>
|
||||
</Fragment>
|
||||
))}
|
||||
<span data-char className="transition-[color,filter] duration-150">{" "}</span>
|
||||
<img src="/logo.svg" alt="FastVideo" className="inline-block h-[1.1em] w-auto align-middle" />
|
||||
</h1>
|
||||
);
|
||||
}
|
||||
@@ -237,6 +239,7 @@ export default function Page() {
|
||||
enhancementEnabled,
|
||||
promptExtensionError,
|
||||
autoExtensionEnabled,
|
||||
manualContinuationMode,
|
||||
autoExtensionTimeoutHint,
|
||||
loopGenerationEnabled,
|
||||
generationPaused,
|
||||
@@ -249,6 +252,8 @@ export default function Page() {
|
||||
livePromptRewriteMode,
|
||||
sessionExpired,
|
||||
projectResetPending,
|
||||
waitingForSegmentPrompt,
|
||||
generatingNextScene,
|
||||
} = sessionState;
|
||||
|
||||
const {
|
||||
@@ -323,6 +328,11 @@ export default function Page() {
|
||||
const [ttffValueMs, setTtffValueMs] = useState<number | null>(null);
|
||||
const ttffIntervalRef = useRef<ReturnType<typeof setInterval> | null>(null);
|
||||
const pendingInitialPromptRef = useRef("");
|
||||
// Prompt id the opening scene is recorded under; sent as initial_rollout_prompt_id
|
||||
// so the backend's pre-seeded opening PromptSubmission emits status updates
|
||||
// (prompt_enhancing/prompt_ready/prompt_fallback_used) against the same id.
|
||||
const pendingInitialPromptIdRef = useRef("");
|
||||
const [initialImageDataUrl, setInitialImageDataUrl] = useState("");
|
||||
const lastArchivedReplayKeyRef = useRef("");
|
||||
const [sidebarOpen, setSidebarOpen] = useState(false);
|
||||
const [currentThumbnail, setCurrentThumbnail] = useState<string | null>(null);
|
||||
@@ -422,6 +432,11 @@ export default function Page() {
|
||||
if (String(e?.source || "") === "user_rewrite" && typeof e?.text === "string" && e.text.trim()) {
|
||||
return e.text.trim();
|
||||
}
|
||||
// Steering opening: the backend overwrites text/source with the enhanced
|
||||
// prompt once ready, so fall back to the stable rawText record.
|
||||
if (e?.steeringUserPrompt && typeof e?.rawText === "string" && e.rawText.trim()) {
|
||||
return e.rawText.trim();
|
||||
}
|
||||
}
|
||||
return "Untitled project";
|
||||
}, [selectedPreset, promptEvents]);
|
||||
@@ -432,11 +447,39 @@ export default function Page() {
|
||||
const canDownloadVideo = useMemo(() => {
|
||||
const currentActiveClip = activeClip as Record<string, any> | null;
|
||||
if (currentActiveClip?.blob instanceof Blob) return true;
|
||||
return (completedClips as Record<string, any>[]).some((clip) => clip?.blob instanceof Blob);
|
||||
}, [activeClip, completedClips]);
|
||||
if ((completedClips as Record<string, any>[]).some((clip) => clip?.blob instanceof Blob)) return true;
|
||||
// Steering only: once playback has started the live AV pipeline holds playable segments, so
|
||||
// the user can download the in-progress video at any time (handleDownloadVideo remuxes live
|
||||
// segments). Auto mode keeps its original blob-gated behavior.
|
||||
return Boolean(manualContinuationMode) && Boolean(avPlaybackStarted);
|
||||
}, [activeClip, completedClips, avPlaybackStarted, manualContinuationMode]);
|
||||
|
||||
// Steering mode scene list (oldest first). Primary source is the user's own words, captured
|
||||
// stably at submit time as `rawText` (the backend later overwrites text/source with the
|
||||
// enhanced prompt, so we never read those). A segment with no user prompt — e.g. a preset's
|
||||
// opening scene — falls back to promptHistory (the actual prompt that drove that segment).
|
||||
const steeringScenes = useMemo(() => {
|
||||
if (!manualContinuationMode) return [] as Record<string, any>[];
|
||||
const userScenes = (promptEvents as Record<string, any>[])
|
||||
.filter((e) => e?.steeringUserPrompt && !e?.steeringFailed && typeof e?.rawText === "string" && e.rawText.trim())
|
||||
.slice()
|
||||
.reverse() // oldest -> newest
|
||||
.map((e) => ({ id: e.promptId, prompt: e.rawText as string }));
|
||||
const scenes: Record<string, any>[] = [];
|
||||
// Preset opening segments: curated seeds with no user prompt of their own.
|
||||
const curatedHists = (promptHistory as Record<string, any>[])
|
||||
.slice()
|
||||
.reverse() // oldest first
|
||||
.filter((h) => !USER_PROMPT_SOURCES.has(String(h?.source || "")) && typeof h?.prompt === "string" && (h.prompt as string).trim());
|
||||
scenes.push(...curatedHists.map((h) => ({ id: h.id || "scene_open", prompt: h.prompt })));
|
||||
scenes.push(...userScenes);
|
||||
return scenes;
|
||||
}, [manualContinuationMode, promptEvents, promptHistory]);
|
||||
|
||||
const hasEdits = useMemo(
|
||||
() => Boolean(sessionStarted) && (promptEvents as Record<string, any>[]).some((e) => typeof e?.text === "string" && e.text.trim() && String(e?.source || "").trim() === "user_rewrite"),
|
||||
() => Boolean(sessionStarted) && (
|
||||
(promptEvents as Record<string, any>[]).some((e) => typeof e?.text === "string" && e.text.trim() && String(e?.source || "").trim() === "user_rewrite")
|
||||
),
|
||||
[sessionStarted, promptEvents],
|
||||
);
|
||||
|
||||
@@ -1421,7 +1464,8 @@ export default function Page() {
|
||||
if (!prompt) return;
|
||||
lastSubmitTimeRef.current = now;
|
||||
const rCAR = !uiStore.get().devtoolsMode && !uiStore.get().demoMode;
|
||||
if (rCAR || (sessionStore.get().livePromptRewriteMode && !uiStore.get().demoMode)) {
|
||||
const inManualMode = sessionStore.get().manualContinuationMode;
|
||||
if (!inManualMode && (rCAR || (sessionStore.get().livePromptRewriteMode && !uiStore.get().demoMode))) {
|
||||
if (rewriteStore.get().rewritingSeedPrompts) return;
|
||||
const rewriteSourcePromptWindowPrompts = getActivePromptWindowPrompts();
|
||||
const nextPendingClip = {
|
||||
@@ -1462,6 +1506,11 @@ export default function Page() {
|
||||
status: "submitted",
|
||||
source: "user_raw",
|
||||
text: prompt,
|
||||
// Stable record of the user's own words for the steering scene list. The backend
|
||||
// later overwrites `text`/`source` with the enhanced prompt via prompt/ready, but
|
||||
// these two fields are never touched by trackPromptEvent.
|
||||
steeringUserPrompt: true,
|
||||
rawText: prompt,
|
||||
});
|
||||
ws.send(
|
||||
JSON.stringify({
|
||||
@@ -1475,7 +1524,14 @@ export default function Page() {
|
||||
activeClipId: shouldUseArchivedPlaybackFallback() ? streamStore.get().activeClipId : "",
|
||||
activePlaybackStartTime: shouldUseArchivedPlaybackFallback() ? streamStore.get().activePlaybackStartTime : 0,
|
||||
});
|
||||
sessionStore.patch({ livePromptDraft: "" });
|
||||
sessionStore.patch({
|
||||
livePromptDraft: "",
|
||||
waitingForSegmentPrompt: false,
|
||||
sessionNotice: "",
|
||||
// Light the "Generating next scene" overlay immediately on a real submit;
|
||||
// stream/media_init (or a fallback/error) clears it.
|
||||
...(inManualMode ? { generatingNextScene: true } : {}),
|
||||
});
|
||||
}
|
||||
|
||||
function setLivePromptRewriteMode(enabled: boolean) {
|
||||
@@ -1550,6 +1606,14 @@ export default function Page() {
|
||||
);
|
||||
}
|
||||
|
||||
// Steering (manual continuation) vs the automatic 6-segment rollout — a pre-session
|
||||
// preference honored when the session starts.
|
||||
function handleManualContinuationToggle(event: any) {
|
||||
sessionStore.patch({
|
||||
manualContinuationMode: Boolean(event.currentTarget.checked),
|
||||
});
|
||||
}
|
||||
|
||||
function handleLoopGenerationToggle(event: any) {
|
||||
sessionStore.patch({
|
||||
loopGenerationEnabled: Boolean(event.currentTarget.checked),
|
||||
@@ -1725,6 +1789,9 @@ export default function Page() {
|
||||
sessionNotice: preserveSessionNotice ? sessionStore.get().sessionNotice : "",
|
||||
sessionExpired: preserveSessionNotice ? sessionStore.get().sessionExpired : false,
|
||||
projectResetPending: false,
|
||||
manualContinuationMode: true,
|
||||
waitingForSegmentPrompt: false,
|
||||
generatingNextScene: false,
|
||||
});
|
||||
rewriteStore.resetSessionState();
|
||||
streamStore.resetSessionState();
|
||||
@@ -1737,6 +1804,7 @@ export default function Page() {
|
||||
function resetToProjectLobbyState() {
|
||||
setVideoMuted(true);
|
||||
pendingInitialPromptRef.current = "";
|
||||
pendingInitialPromptIdRef.current = "";
|
||||
sessionStore.patch({
|
||||
sessionStarted: false,
|
||||
livePromptDraft: "",
|
||||
@@ -1750,6 +1818,9 @@ export default function Page() {
|
||||
sessionNotice: "",
|
||||
sessionExpired: false,
|
||||
projectResetPending: false,
|
||||
manualContinuationMode: true,
|
||||
waitingForSegmentPrompt: false,
|
||||
generatingNextScene: false,
|
||||
});
|
||||
rewriteStore.resetSessionState();
|
||||
streamStore.resetSessionState();
|
||||
@@ -1760,7 +1831,14 @@ export default function Page() {
|
||||
}
|
||||
|
||||
function buildProjectInitPayload(type: "session_init_v2" | "project_init_v1") {
|
||||
const segmentPrompts = getSessionInitPrompts();
|
||||
const manualMode = Boolean(sessionStore.get().manualContinuationMode);
|
||||
let segmentPrompts = getSessionInitPrompts();
|
||||
// Steering mode: seed the first 2 segments from the preset so there's no
|
||||
// gap between segment 1 and 2; the user drives every subsequent segment.
|
||||
// Force auto/loop off so the backend waits after the seeded prompts run out.
|
||||
if (manualMode) {
|
||||
segmentPrompts = segmentPrompts.slice(0, 2);
|
||||
}
|
||||
setSeedPrompts(segmentPrompts);
|
||||
return {
|
||||
type,
|
||||
@@ -1768,11 +1846,17 @@ export default function Page() {
|
||||
preset_label: getInitialPresetLabel(),
|
||||
curated_prompts: segmentPrompts,
|
||||
initial_rollout_prompt: normalizeInitialPrompt(pendingInitialPromptRef.current),
|
||||
initial_image: null,
|
||||
// Ties the backend's pre-seeded opening PromptSubmission to the prompt event
|
||||
// recorded in beginProjectLocally so its status updates land on that record.
|
||||
initial_rollout_prompt_id: pendingInitialPromptIdRef.current,
|
||||
initial_image: initialImageDataUrl
|
||||
? { data_url: initialImageDataUrl, mime_type: initialImageDataUrl.split(";")[0].split(":")[1] || "image/png", name: "upload.png" }
|
||||
: null,
|
||||
single_clip_mode: false,
|
||||
enhancement_enabled: sessionStore.get().enhancementEnabled,
|
||||
auto_extension_enabled: sessionStore.get().autoExtensionEnabled,
|
||||
loop_generation_enabled: sessionStore.get().loopGenerationEnabled,
|
||||
auto_extension_enabled: manualMode ? false : sessionStore.get().autoExtensionEnabled,
|
||||
loop_generation_enabled: manualMode ? false : sessionStore.get().loopGenerationEnabled,
|
||||
manual_continuation_mode: manualMode,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -1951,7 +2035,11 @@ export default function Page() {
|
||||
setCurrentThumbnail(null);
|
||||
const initialPrompt = normalizeInitialPrompt(sessionStore.get().livePromptDraft as string);
|
||||
pendingInitialPromptRef.current = initialPrompt;
|
||||
setInitialImageDataUrl("");
|
||||
const rCAR = !uiStore.get().devtoolsMode && !uiStore.get().demoMode;
|
||||
// The "Steering mode" toggle is authoritative: checked = manual per-segment steering,
|
||||
// unchecked = automatic 6-segment rollout (default).
|
||||
const nextManualContinuationMode = Boolean(sessionStore.get().manualContinuationMode);
|
||||
sessionStore.patch({
|
||||
sessionNotice: "",
|
||||
sessionExpired: false,
|
||||
@@ -1966,6 +2054,9 @@ export default function Page() {
|
||||
autoExtensionTimeoutHint: "",
|
||||
generationPaused: false,
|
||||
projectResetPending: false,
|
||||
manualContinuationMode: nextManualContinuationMode,
|
||||
waitingForSegmentPrompt: false,
|
||||
generatingNextScene: false,
|
||||
});
|
||||
resetPlaybackState();
|
||||
streamStore.patch({
|
||||
@@ -1979,12 +2070,15 @@ export default function Page() {
|
||||
selectedHistoryId: "",
|
||||
});
|
||||
rewriteStore.resetSessionState();
|
||||
pendingInitialPromptIdRef.current = initialPrompt ? makePromptId() : "";
|
||||
if (initialPrompt) {
|
||||
addPromptEvent({
|
||||
promptId: makePromptId(),
|
||||
promptId: pendingInitialPromptIdRef.current,
|
||||
status: "rewrite_requested",
|
||||
source: "user_rewrite",
|
||||
text: initialPrompt,
|
||||
// In steering mode the typed opening is the user's Scene 1 — record it stably.
|
||||
...(nextManualContinuationMode ? { steeringUserPrompt: true, rawText: initialPrompt } : {}),
|
||||
});
|
||||
}
|
||||
setSeedPrompts(getSessionInitPrompts());
|
||||
@@ -2500,6 +2594,7 @@ export default function Page() {
|
||||
selectedPresetId={selectedPresetId as string}
|
||||
enhancementEnabled={enhancementEnabled as boolean}
|
||||
autoExtensionEnabled={autoExtensionEnabled as boolean}
|
||||
manualContinuationEnabled={manualContinuationMode as boolean}
|
||||
loopGenerationEnabled={loopGenerationEnabled as boolean}
|
||||
canJoinSession={canJoinSession as boolean}
|
||||
canSubmitContinuation={canSubmitContinuation}
|
||||
@@ -2512,6 +2607,7 @@ export default function Page() {
|
||||
onEnhancementToggle={handleEnhancementToggle}
|
||||
onCuratedPromptLimitChange={handleCuratedPromptLimitChange}
|
||||
onAutoExtensionToggle={handleAutoExtensionToggle}
|
||||
onManualContinuationToggle={handleManualContinuationToggle}
|
||||
onLoopToggle={handleLoopGenerationToggle}
|
||||
onJoin={joinSession}
|
||||
onLeave={leaveSession}
|
||||
@@ -2641,6 +2737,7 @@ export default function Page() {
|
||||
/>
|
||||
<Header timeLeft={headerTimeLeft} formatTime={formatTime} onToggleSidebar={() => setSidebarOpen((prev) => !prev)} />
|
||||
|
||||
<div className={cn("flex flex-1 min-h-0 flex-col", sessionStarted && "pb-16 sm:pb-28")}>
|
||||
<div className="relative flex flex-1 min-h-0 flex-col justify-center px-4 pb-2 sm:px-6 sm:pb-12">
|
||||
{isViewingMode && (
|
||||
<>
|
||||
@@ -2724,6 +2821,8 @@ export default function Page() {
|
||||
showLivePlayback={showLivePlayback}
|
||||
defaultMuted={videoMuted}
|
||||
canDownload={canDownloadVideo}
|
||||
waitingForSegmentPrompt={waitingForSegmentPrompt as boolean}
|
||||
generatingNextScene={generatingNextScene as boolean}
|
||||
onPlaying={markFirstFrameRendered}
|
||||
onDownload={handleDownloadVideo}
|
||||
/>
|
||||
@@ -2734,6 +2833,7 @@ export default function Page() {
|
||||
<section className={cn("mx-auto w-full max-w-2xl", hasEdits && "flex-1 min-h-0 overflow-y-auto")}>
|
||||
<Workspace
|
||||
promptEvents={promptEvents as any[]}
|
||||
manualMode={manualContinuationMode as boolean}
|
||||
currentThumbnail={currentThumbnail}
|
||||
originalLabel={pendingInitialPromptRef.current || (selectedPreset as Record<string, any>)?.label || ""}
|
||||
sessionStarted={sessionStarted as boolean}
|
||||
@@ -2780,11 +2880,17 @@ export default function Page() {
|
||||
isGenerating={loadingAnimation as boolean}
|
||||
storyPresets={storyPresets as any[]}
|
||||
continuationDraft={livePromptDraft as string}
|
||||
manualContinuationEnabled={manualContinuationMode as boolean}
|
||||
onModeChange={(manual) => sessionStore.patch({ manualContinuationMode: manual })}
|
||||
initialImageDataUrl={initialImageDataUrl}
|
||||
onImageUpload={(dataUrl) => setInitialImageDataUrl(dataUrl)}
|
||||
onImageClear={() => setInitialImageDataUrl("")}
|
||||
canJoinSession={canStartSession}
|
||||
canSubmitContinuation={canSubmitContinuation}
|
||||
sessionExpired={sessionExpired as boolean}
|
||||
sessionNotice={sessionNotice as string}
|
||||
projectResetPending={projectResetPending as boolean}
|
||||
waitingForSegmentPrompt={waitingForSegmentPrompt as boolean}
|
||||
onPresetGenerate={handlePresetGenerate}
|
||||
onContinuationInput={handleLivePromptInput}
|
||||
onContinuationKeydown={handleLivePromptKeydown}
|
||||
@@ -2798,6 +2904,12 @@ export default function Page() {
|
||||
</motion.div>
|
||||
</div>
|
||||
</div>
|
||||
{manualContinuationMode && (
|
||||
<div className="px-4 sm:px-6">
|
||||
<SceneHistoryList sceneHistory={steeringScenes as any[]} />
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</main>
|
||||
);
|
||||
}
|
||||
|
||||
@@ -2,13 +2,16 @@
|
||||
|
||||
import React, { useRef, useState, useCallback, useEffect } from "react";
|
||||
import Image from "next/image";
|
||||
import { Film, ArrowUp, X, Loader2, ArrowLeft } from "lucide-react";
|
||||
import { Film, ArrowUp, X, Loader2, ArrowLeft, ImagePlus } from "lucide-react";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import LeaveSessionModal, { shouldShowLeaveWarning } from "@/components/LeaveSessionModal";
|
||||
import SpeechToTextButton from "@/components/SpeechToTextButton";
|
||||
import { cn } from "@/lib/utils";
|
||||
|
||||
const PROMPT_MAX_LENGTH = 500;
|
||||
// Must match backend session_init_image.py: MAX_SESSION_INIT_IMAGE_BYTES / SUPPORTED_SESSION_INIT_IMAGE_MIME_TYPES.
|
||||
const IMAGE_MAX_BYTES = 15 * 1024 * 1024;
|
||||
const IMAGE_ALLOWED_TYPES = ["image/png", "image/jpeg", "image/webp"];
|
||||
|
||||
interface Props {
|
||||
sessionStarted?: boolean;
|
||||
@@ -22,6 +25,12 @@ interface Props {
|
||||
sessionNotice?: string;
|
||||
projectResetPending?: boolean;
|
||||
viewingReadOnly?: boolean;
|
||||
waitingForSegmentPrompt?: boolean;
|
||||
manualContinuationEnabled?: boolean;
|
||||
onModeChange?: (manual: boolean) => void;
|
||||
initialImageDataUrl?: string;
|
||||
onImageUpload?: (dataUrl: string, mimeType: string, name: string) => void;
|
||||
onImageClear?: () => void;
|
||||
onPresetGenerate?: (presetId: string) => void;
|
||||
onContinuationInput?: (e: React.ChangeEvent<HTMLTextAreaElement>) => void;
|
||||
onContinuationKeydown?: (e: React.KeyboardEvent<HTMLTextAreaElement>) => void;
|
||||
@@ -46,6 +55,12 @@ export default function ChatBar({
|
||||
sessionNotice = "",
|
||||
projectResetPending = false,
|
||||
viewingReadOnly = false,
|
||||
waitingForSegmentPrompt = false,
|
||||
manualContinuationEnabled = false,
|
||||
onModeChange = () => {},
|
||||
initialImageDataUrl = "",
|
||||
onImageUpload = () => {},
|
||||
onImageClear = () => {},
|
||||
onPresetGenerate = () => {},
|
||||
onContinuationInput = () => {},
|
||||
onContinuationKeydown = () => {},
|
||||
@@ -59,15 +74,51 @@ export default function ChatBar({
|
||||
}: Props) {
|
||||
const [sttBusy, setSttBusy] = useState(false);
|
||||
const [leaveModalOpen, setLeaveModalOpen] = useState(false);
|
||||
const [imageError, setImageError] = useState("");
|
||||
const fileInputRef = useRef<HTMLInputElement>(null);
|
||||
|
||||
const processImageFile = useCallback((file: File) => {
|
||||
if (!IMAGE_ALLOWED_TYPES.includes(file.type)) {
|
||||
setImageError("Unsupported image type. Use a PNG, JPEG, or WebP image.");
|
||||
return;
|
||||
}
|
||||
if (file.size > IMAGE_MAX_BYTES) {
|
||||
setImageError("Image is too large. The maximum size is 15MB.");
|
||||
return;
|
||||
}
|
||||
const reader = new FileReader();
|
||||
reader.onload = (e) => {
|
||||
const dataUrl = e.target?.result as string;
|
||||
if (dataUrl) {
|
||||
setImageError("");
|
||||
onImageUpload(dataUrl, file.type, file.name);
|
||||
}
|
||||
};
|
||||
reader.onerror = () => {
|
||||
setImageError("Could not read the image file. Please try again.");
|
||||
};
|
||||
reader.readAsDataURL(file);
|
||||
}, [onImageUpload]);
|
||||
|
||||
const handleImagePaste = useCallback((e: React.ClipboardEvent) => {
|
||||
if (sessionStarted) return;
|
||||
const items = Array.from(e.clipboardData?.items ?? []);
|
||||
const imageItem = items.find((item) => item.type.startsWith("image/"));
|
||||
if (!imageItem) return;
|
||||
const file = imageItem.getAsFile();
|
||||
if (file) processImageFile(file);
|
||||
}, [sessionStarted, processImageFile]);
|
||||
const showSpinner = isGenerating || rewritingSeedPrompts;
|
||||
const isBusy = isGenerating || rewritingSeedPrompts || projectResetPending;
|
||||
const messagePlaceholder = projectResetPending
|
||||
? "Starting new project\u2026"
|
||||
: isBusy
|
||||
? "Generating video\u2026"
|
||||
: !sessionStarted
|
||||
? "What video are you imagining?"
|
||||
: "What do you want to edit?";
|
||||
: waitingForSegmentPrompt
|
||||
? "Describe the next scene\u2026"
|
||||
: !sessionStarted
|
||||
? "What video are you imagining?"
|
||||
: "What do you want to edit?";
|
||||
const actionLabel = !sessionStarted ? "Generate" : "Rewrite rollout";
|
||||
|
||||
const inputRef = useRef<HTMLTextAreaElement>(null);
|
||||
@@ -267,9 +318,9 @@ export default function ChatBar({
|
||||
<Button onClick={onStartNewProject} size="sm" className="rounded-full px-5">
|
||||
New Project
|
||||
</Button>
|
||||
<a href="https://docs.google.com/forms/d/e/1FAIpQLSe5zpO1iD8Ds-Ih-fOLm64qd7YZVvuvAyHuJaAfw1hkRHTe_A/viewform?usp=publish-editor" target="_blank" rel="noopener noreferrer">
|
||||
<a href="https://haoailab.com/blogs/dreamverse/" target="_blank" rel="noopener noreferrer">
|
||||
<Button variant="outline" size="sm" className="rounded-full px-5">
|
||||
Join Waitlist
|
||||
Blog
|
||||
</Button>
|
||||
</a>
|
||||
</div>
|
||||
@@ -346,6 +397,31 @@ export default function ChatBar({
|
||||
</div>
|
||||
)}
|
||||
|
||||
|
||||
{!sessionStarted && imageError && (
|
||||
<div className="rounded-xl border border-rose-500/20 bg-rose-500/10 px-4 py-2.5 text-center text-xs text-rose-700 dark:text-rose-300">
|
||||
{imageError}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{!sessionStarted && initialImageDataUrl && (
|
||||
<div className="flex items-center gap-2 rounded-2xl border border-input bg-card/65 px-3 py-2">
|
||||
<img src={initialImageDataUrl} alt="Initial frame" className="h-12 w-12 rounded-lg object-cover" />
|
||||
<span className="flex-1 truncate text-xs text-muted-foreground">Starting image set</span>
|
||||
<button type="button" onClick={() => { setImageError(""); onImageClear(); }} className="text-muted-foreground hover:text-foreground transition-colors">
|
||||
<X className="size-4" />
|
||||
</button>
|
||||
</div>
|
||||
)}
|
||||
|
||||
<input
|
||||
ref={fileInputRef}
|
||||
type="file"
|
||||
accept={IMAGE_ALLOWED_TYPES.join(",")}
|
||||
className="hidden"
|
||||
onChange={(e) => { const f = e.target.files?.[0]; if (f) processImageFile(f); e.target.value = ""; }}
|
||||
/>
|
||||
|
||||
<div
|
||||
className={cn(
|
||||
"flex min-w-0 items-center gap-1.5 rounded-4xl border py-2.5 pl-5 pr-2.5 shadow-md backdrop-blur-sm transition-all duration-200",
|
||||
@@ -359,6 +435,7 @@ export default function ChatBar({
|
||||
value={continuationDraft}
|
||||
onChange={onContinuationInput}
|
||||
onKeyDown={handleKeyDown}
|
||||
onPaste={handleImagePaste}
|
||||
placeholder={sttBusy ? "Listening\u2026" : messagePlaceholder}
|
||||
maxLength={PROMPT_MAX_LENGTH}
|
||||
disabled={isBusy || sttBusy}
|
||||
@@ -369,6 +446,19 @@ export default function ChatBar({
|
||||
)}
|
||||
/>
|
||||
{onSpeechTranscript && <SpeechToTextButton disabled={isBusy} onTranscript={onSpeechTranscript} onInterimChange={onSpeechInterimChange} onBusyChange={setSttBusy} />}
|
||||
{!sessionStarted && (
|
||||
<Button
|
||||
type="button"
|
||||
variant="ghost"
|
||||
size="icon-sm"
|
||||
title="Add starting image"
|
||||
onClick={() => fileInputRef.current?.click()}
|
||||
disabled={isBusy}
|
||||
className="shrink-0 rounded-full text-muted-foreground hover:text-foreground"
|
||||
>
|
||||
<ImagePlus className="size-4" />
|
||||
</Button>
|
||||
)}
|
||||
{!sessionStarted ? (
|
||||
<Button
|
||||
aria-label={actionLabel}
|
||||
|
||||
@@ -8,7 +8,6 @@ import { Badge } from "@/components/ui/badge";
|
||||
import { Button } from "@/components/ui/button";
|
||||
import { ThemeToggle } from "@/components/ui/theme-toggle";
|
||||
|
||||
const FASTVIDEO_REPO_URL = "https://haoailab.com/blogs/dreamverse/";
|
||||
const FASTVIDEO_BLOG_URL = "https://haoailab.com/blogs/dreamverse/";
|
||||
|
||||
interface Props {
|
||||
@@ -29,13 +28,13 @@ export default function Header({ timeLeft = null, formatTime = (seconds) => `${s
|
||||
<SidePanelOpenFilled size={20} />
|
||||
</Button>
|
||||
)}
|
||||
<a href={FASTVIDEO_REPO_URL} target="_blank" rel="noopener noreferrer" title="FastVideo on GitHub">
|
||||
<a href="/" title="FastVideo home">
|
||||
<Image src="/logo.svg" alt="FastVideo" width={32} height={32} className="h-8 w-auto sm:h-9 transition-opacity hover:opacity-70" />
|
||||
</a>
|
||||
<div className="hidden sm:flex items-center gap-3">
|
||||
<a href="https://docs.google.com/forms/d/e/1FAIpQLSe5zpO1iD8Ds-Ih-fOLm64qd7YZVvuvAyHuJaAfw1hkRHTe_A/viewform?usp=publish-editor" target="_blank" rel="noopener noreferrer">
|
||||
<a href={FASTVIDEO_BLOG_URL} target="_blank" rel="noopener noreferrer">
|
||||
<Button variant="outline" size="sm" className="gap-1.5 rounded-full px-3 text-xs">
|
||||
Join Waitlist
|
||||
Blog
|
||||
<ExternalLink className="size-3 opacity-60" />
|
||||
</Button>
|
||||
</a>
|
||||
@@ -53,9 +52,9 @@ export default function Header({ timeLeft = null, formatTime = (seconds) => `${s
|
||||
</div>
|
||||
|
||||
<div className="flex sm:hidden items-center gap-2 px-4 pb-3">
|
||||
<a href="https://docs.google.com/forms/d/e/1FAIpQLSe5zpO1iD8Ds-Ih-fOLm64qd7YZVvuvAyHuJaAfw1hkRHTe_A/viewform?usp=publish-editor" target="_blank" rel="noopener noreferrer">
|
||||
<a href={FASTVIDEO_BLOG_URL} target="_blank" rel="noopener noreferrer">
|
||||
<Button variant="outline" size="sm" className="gap-1.5 rounded-full px-3 text-xs">
|
||||
Join Waitlist
|
||||
Blog
|
||||
<ExternalLink className="size-3 opacity-60" />
|
||||
</Button>
|
||||
</a>
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
import React, { useState, useEffect, useRef, useCallback } from "react";
|
||||
import { cn } from "@/lib/utils";
|
||||
import { PlayFilledAlt } from "@carbon/icons-react";
|
||||
import { Download, Loader2, Share } from "lucide-react";
|
||||
import { Check, ChevronDown, Download, Loader2, Share } from "lucide-react";
|
||||
import { Button } from "@/components/ui/button";
|
||||
interface VideoPlayerProps {
|
||||
videoRef?: React.RefCallback<HTMLVideoElement>;
|
||||
@@ -21,6 +21,8 @@ interface VideoPlayerProps {
|
||||
showLivePlayback?: boolean;
|
||||
defaultMuted?: boolean;
|
||||
rewritePending?: boolean;
|
||||
waitingForSegmentPrompt?: boolean;
|
||||
generatingNextScene?: boolean;
|
||||
onPlaying?: () => void;
|
||||
onDownload?: () => void;
|
||||
}
|
||||
@@ -57,6 +59,8 @@ export default function VideoPlayer({
|
||||
showLivePlayback = true,
|
||||
defaultMuted = true,
|
||||
rewritePending = false,
|
||||
waitingForSegmentPrompt = false,
|
||||
generatingNextScene = false,
|
||||
onPlaying = () => {},
|
||||
onDownload,
|
||||
}: VideoPlayerProps) {
|
||||
@@ -88,6 +92,137 @@ export default function VideoPlayer({
|
||||
setCanShare(typeof navigator.canShare === "function" && window.matchMedia("(pointer: coarse)").matches);
|
||||
}, []);
|
||||
|
||||
// Steering mode: the backend may signal "waiting for next prompt" while the current
|
||||
// segment is still PLAYING (it generates ahead). Only surface the "Segment complete"
|
||||
// overlay once the playhead actually reaches the end of the buffered segment.
|
||||
const [playbackReachedEnd, setPlaybackReachedEnd] = useState(false);
|
||||
// Steering mode: when the user submits the next scene, the segment is generated
|
||||
// (a few seconds of latency) before frames stream. Show a "Generating next scene…"
|
||||
// indicator across that gap so the frozen frame isn't silent. Driven by the explicit
|
||||
// generatingNextScene state (set on scene submit / prompt selection), never inferred
|
||||
// from waitingForSegmentPrompt edges.
|
||||
const [generatingNext, setGeneratingNext] = useState(false);
|
||||
// End of the buffered timeline captured the moment generation starts; the freshly
|
||||
// generated segment extends the buffer past this, which is how we know it landed.
|
||||
const genBoundaryRef = useRef(0);
|
||||
|
||||
// Steering mode: the backend may signal "waiting for next prompt" while the current
|
||||
// segment is still PLAYING (it generates ahead). Track whether the playhead has reached
|
||||
// the end of the buffered segment so end-overlays only show there. Keep tracking through
|
||||
// the generating phase too, so scrubbing back and replaying to the end re-shows them.
|
||||
useEffect(() => {
|
||||
if (!waitingForSegmentPrompt && !generatingNext) {
|
||||
setPlaybackReachedEnd(false);
|
||||
return;
|
||||
}
|
||||
const el = liveVideoEl.current;
|
||||
if (!el) return;
|
||||
const check = () => {
|
||||
try {
|
||||
const buffered = el.buffered;
|
||||
if (buffered.length === 0) return;
|
||||
const end = buffered.end(buffered.length - 1);
|
||||
// Track proximity both ways: scrubbing back off the end hides the overlay,
|
||||
// playing forward to the end re-shows it.
|
||||
setPlaybackReachedEnd(el.ended || end - el.currentTime <= 0.2);
|
||||
} catch {
|
||||
/* buffered access can throw mid-append */
|
||||
}
|
||||
};
|
||||
check();
|
||||
el.addEventListener("timeupdate", check);
|
||||
el.addEventListener("ended", check);
|
||||
el.addEventListener("waiting", check);
|
||||
el.addEventListener("stalled", check);
|
||||
el.addEventListener("pause", check);
|
||||
el.addEventListener("seeking", check);
|
||||
el.addEventListener("seeked", check);
|
||||
el.addEventListener("playing", check);
|
||||
el.addEventListener("progress", check);
|
||||
return () => {
|
||||
el.removeEventListener("timeupdate", check);
|
||||
el.removeEventListener("ended", check);
|
||||
el.removeEventListener("waiting", check);
|
||||
el.removeEventListener("stalled", check);
|
||||
el.removeEventListener("pause", check);
|
||||
el.removeEventListener("seeking", check);
|
||||
el.removeEventListener("seeked", check);
|
||||
el.removeEventListener("playing", check);
|
||||
el.removeEventListener("progress", check);
|
||||
};
|
||||
}, [waitingForSegmentPrompt, generatingNext]);
|
||||
|
||||
useEffect(() => {
|
||||
if (waitingForSegmentPrompt || !sessionStarted) {
|
||||
// Back to waiting (or session over): nothing is generating.
|
||||
setGeneratingNext(false);
|
||||
return;
|
||||
}
|
||||
if (!generatingNextScene) return;
|
||||
// Snapshot the current end of the buffered timeline; the generated segment will
|
||||
// extend the buffer past this boundary.
|
||||
const el = liveVideoEl.current;
|
||||
let boundary = el?.currentTime ?? 0;
|
||||
try {
|
||||
const b = el?.buffered;
|
||||
if (b && b.length) boundary = Math.max(boundary, b.end(b.length - 1));
|
||||
} catch {
|
||||
/* buffered access can throw mid-append */
|
||||
}
|
||||
genBoundaryRef.current = boundary;
|
||||
setGeneratingNext(true);
|
||||
}, [generatingNextScene, waitingForSegmentPrompt, sessionStarted]);
|
||||
|
||||
useEffect(() => {
|
||||
if (!generatingNext) return;
|
||||
if (!sessionStarted) {
|
||||
setGeneratingNext(false);
|
||||
return;
|
||||
}
|
||||
const el = liveVideoEl.current;
|
||||
if (!el) return;
|
||||
// Clear the instant the freshly generated segment lands: the buffer grows past the
|
||||
// boundary captured at generation start (or the playhead advances into the new
|
||||
// frames). Deliberately NOT a bare "playing" handler — scrubbing back and replaying
|
||||
// the EXISTING segment must keep "Generating" up until the new frames actually arrive.
|
||||
const check = () => {
|
||||
try {
|
||||
const b = el.buffered;
|
||||
const end = b.length ? b.end(b.length - 1) : 0;
|
||||
if (end > genBoundaryRef.current + 0.25 || el.currentTime > genBoundaryRef.current + 0.1) {
|
||||
setGeneratingNext(false);
|
||||
}
|
||||
} catch {
|
||||
/* buffered access can throw mid-append */
|
||||
}
|
||||
};
|
||||
check();
|
||||
el.addEventListener("progress", check);
|
||||
el.addEventListener("timeupdate", check);
|
||||
el.addEventListener("durationchange", check);
|
||||
return () => {
|
||||
el.removeEventListener("progress", check);
|
||||
el.removeEventListener("timeupdate", check);
|
||||
el.removeEventListener("durationchange", check);
|
||||
};
|
||||
}, [generatingNext, sessionStarted]);
|
||||
|
||||
// Drive a ~10.5s progress bar during generation so the wait has a visible ETA.
|
||||
const GEN_DURATION_MS = 10500;
|
||||
const [genProgress, setGenProgress] = useState(0);
|
||||
useEffect(() => {
|
||||
if (!generatingNext) {
|
||||
setGenProgress(0);
|
||||
return;
|
||||
}
|
||||
const start = performance.now();
|
||||
setGenProgress(0);
|
||||
const id = setInterval(() => {
|
||||
setGenProgress(Math.min((performance.now() - start) / GEN_DURATION_MS, 1));
|
||||
}, 50);
|
||||
return () => clearInterval(id);
|
||||
}, [generatingNext]);
|
||||
|
||||
return (
|
||||
<div className="mx-auto w-full max-w-3xl mb-2 sm:mb-6">
|
||||
<div className="rounded-2xl border border-border bg-card/50 p-2 shadow-lg backdrop-blur-md">
|
||||
@@ -104,7 +239,7 @@ export default function VideoPlayer({
|
||||
<PlayFilledAlt className="size-10 text-white/25" />
|
||||
<p className="text-sm text-white/50">Your video will appear here</p>
|
||||
</div>
|
||||
) : !avPlaybackStarted && !mediaAppendError && !inQueue && loadingAnimation ? (
|
||||
) : !avPlaybackStarted && !mediaAppendError && !inQueue && !waitingForSegmentPrompt && loadingAnimation ? (
|
||||
<div className="absolute inset-0 flex flex-col items-center justify-center gap-4 bg-slate-900/60 p-4 backdrop-blur-[2px]">
|
||||
<div className="pointer-events-none absolute inset-0 overflow-hidden">
|
||||
<div className="absolute inset-0 -translate-x-full animate-[shimmer_3s_ease-in-out_infinite] bg-gradient-to-r from-transparent via-white/[0.04] to-transparent" />
|
||||
@@ -114,6 +249,35 @@ export default function VideoPlayer({
|
||||
</div>
|
||||
) : null}
|
||||
|
||||
{/* Steering mode: this segment finished — wait gracefully for the user's next scene
|
||||
instead of spinning. The last frame stays visible behind a soft bottom gradient. */}
|
||||
{sessionStarted && waitingForSegmentPrompt && playbackReachedEnd && !mediaAppendError && !inQueue && (
|
||||
<div className="pointer-events-none absolute inset-0 flex flex-col items-center justify-end gap-2 bg-gradient-to-t from-slate-950/85 via-slate-950/15 to-transparent p-5 pb-6 text-center">
|
||||
<div className="flex size-9 items-center justify-center rounded-full border border-white/25 bg-white/10 shadow-lg backdrop-blur-md">
|
||||
<Check className="size-4 text-white/90" />
|
||||
</div>
|
||||
<div className="space-y-0.5">
|
||||
<p className="text-sm font-medium text-white/95">Segment complete</p>
|
||||
<p className="text-xs text-white/65">Describe the next scene below to keep going</p>
|
||||
</div>
|
||||
<ChevronDown className="size-4 animate-bounce text-white/45" />
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Steering mode: generating the next segment — show a ~10.5s progress bar so the wait has an ETA.
|
||||
Gated on playbackReachedEnd like "Segment complete": scrubbing back hides it, playing to the end re-shows it. */}
|
||||
{sessionStarted && generatingNext && playbackReachedEnd && !mediaAppendError && !inQueue && (
|
||||
<div className="pointer-events-none absolute inset-0 flex flex-col items-center justify-end gap-3 bg-gradient-to-t from-slate-950/85 via-slate-950/15 to-transparent p-5 pb-7 text-center">
|
||||
<p className="text-sm font-medium text-white/95">Generating next scene…</p>
|
||||
<div className="h-1.5 w-48 overflow-hidden rounded-full bg-white/15 shadow-sm">
|
||||
<div
|
||||
className="h-full rounded-full bg-white/85 transition-[width] duration-100 ease-linear"
|
||||
style={{ width: `${Math.round(genProgress * 100)}%` }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{rewritePending && avPlaybackStarted && (
|
||||
<div className="absolute inset-x-0 bottom-0 z-10 flex items-center justify-center gap-2 bg-gradient-to-t from-black/60 to-transparent px-4 pb-12 pt-8 pointer-events-none">
|
||||
<Loader2 className="size-4 animate-spin text-white/90" />
|
||||
|
||||
@@ -3,11 +3,90 @@ import React, { useRef, useMemo, useEffect, useCallback, useState } from "react"
|
||||
import { motion, useAnimationControls } from "framer-motion";
|
||||
import { Badge } from "@/components/ui/badge";
|
||||
import { cn } from "@/lib/utils";
|
||||
import { Check, Lightbulb, Pencil } from "lucide-react";
|
||||
import { Check, Clapperboard, Lightbulb, Pencil } from "lucide-react";
|
||||
|
||||
export const WORKSPACE_ORIGINAL_SELECTION_KEY = "original";
|
||||
export const WORKSPACE_CURRENT_SELECTION_KEY = "current";
|
||||
|
||||
export function SceneHistoryList({ sceneHistory = [] }: { sceneHistory?: Record<string, any>[] }) {
|
||||
const bottomSentinelRef = useRef<HTMLDivElement>(null);
|
||||
const topSentinelRef = useRef<HTMLDivElement>(null);
|
||||
const [showTopFade, setShowTopFade] = useState(false);
|
||||
|
||||
const scenes = useMemo(
|
||||
() => (sceneHistory || []).filter((s) => normalizeText(s?.prompt)),
|
||||
[sceneHistory],
|
||||
);
|
||||
|
||||
const scrollToBottom = useCallback(() => {
|
||||
setTimeout(() => {
|
||||
bottomSentinelRef.current?.scrollIntoView({ block: "end", behavior: "smooth" });
|
||||
}, 60);
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
if (scenes.length > 0) scrollToBottom();
|
||||
}, [scenes.length, scrollToBottom]);
|
||||
|
||||
useEffect(() => {
|
||||
const sentinel = bottomSentinelRef.current;
|
||||
if (!sentinel || typeof ResizeObserver === "undefined") return;
|
||||
let container: HTMLElement | null = sentinel.parentElement;
|
||||
while (container) {
|
||||
const oy = getComputedStyle(container).overflowY;
|
||||
if (oy === "auto" || oy === "scroll") break;
|
||||
container = container.parentElement;
|
||||
}
|
||||
if (!container) return;
|
||||
const ro = new ResizeObserver(() => {
|
||||
const nearBottom = container!.scrollHeight - container!.scrollTop - container!.clientHeight < 96;
|
||||
if (nearBottom) scrollToBottom();
|
||||
});
|
||||
ro.observe(container);
|
||||
return () => ro.disconnect();
|
||||
}, [scenes.length, scrollToBottom]);
|
||||
|
||||
useEffect(() => {
|
||||
const el = topSentinelRef.current;
|
||||
if (!el) return;
|
||||
const observer = new IntersectionObserver(([entry]) => setShowTopFade(!entry.isIntersecting), { threshold: 0.1 });
|
||||
observer.observe(el);
|
||||
return () => observer.disconnect();
|
||||
}, [scenes.length]);
|
||||
|
||||
if (scenes.length === 0) return null;
|
||||
|
||||
return (
|
||||
<section className="relative z-10 flex flex-col mx-auto w-full max-w-2xl max-h-32 overflow-y-auto">
|
||||
<div
|
||||
className={cn(
|
||||
"pointer-events-none sticky top-0 z-20 -mb-12 h-12 bg-linear-to-b from-background to-transparent transition-opacity duration-200",
|
||||
showTopFade ? "opacity-100" : "opacity-0",
|
||||
)}
|
||||
aria-hidden="true"
|
||||
/>
|
||||
<div ref={topSentinelRef} className="h-0 w-0" aria-hidden="true" />
|
||||
<div className="flex flex-col gap-2 pt-12 pb-4">
|
||||
{scenes.map((scene, index) => (
|
||||
<div
|
||||
key={scene.id || index}
|
||||
className="flex items-start gap-3 rounded-xl p-3 transition-colors duration-200 hover:bg-slate-200/50 hover:dark:bg-slate-800/30"
|
||||
>
|
||||
<div className="flex min-w-0 flex-1 flex-col gap-2">
|
||||
<Badge variant="secondary" className="horizontal gap-2 items-center w-fit">
|
||||
<Clapperboard className="size-3 opacity-70" />
|
||||
{`Scene ${index + 1}`}
|
||||
</Badge>
|
||||
<p className="line-clamp-2 text-sm leading-5 text-muted-foreground">{scene.prompt}</p>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
<div ref={bottomSentinelRef} className="h-0 w-0" aria-hidden="true" />
|
||||
</section>
|
||||
);
|
||||
}
|
||||
|
||||
interface WorkspaceProps {
|
||||
promptEvents?: Record<string, any>[];
|
||||
currentThumbnail?: string | null;
|
||||
@@ -19,6 +98,7 @@ interface WorkspaceProps {
|
||||
selectedClipId?: string;
|
||||
selectedEntryKey?: string;
|
||||
originalClipId?: string;
|
||||
manualMode?: boolean;
|
||||
}
|
||||
|
||||
function normalizeText(value: any): string {
|
||||
@@ -218,7 +298,7 @@ function ChromaGradient({ sessionStarted = false }: { sessionStarted?: boolean }
|
||||
);
|
||||
}
|
||||
|
||||
export default function Workspace({ promptEvents = [], currentThumbnail = null, originalLabel = "", sessionStarted = false, onSelectOriginal, onSelectEvent, onSelectCurrent, selectedClipId, selectedEntryKey: selectedEntryKeyProp, originalClipId = "" }: WorkspaceProps) {
|
||||
export default function Workspace({ promptEvents = [], currentThumbnail = null, originalLabel = "", sessionStarted = false, onSelectOriginal, onSelectEvent, onSelectCurrent, selectedClipId, selectedEntryKey: selectedEntryKeyProp, originalClipId = "", manualMode = false }: WorkspaceProps) {
|
||||
const bottomSentinelRef = useRef<HTMLDivElement>(null);
|
||||
const topSentinelRef = useRef<HTMLDivElement>(null);
|
||||
const [showTopFade, setShowTopFade] = useState(false);
|
||||
@@ -259,6 +339,14 @@ export default function Workspace({ promptEvents = [], currentThumbnail = null,
|
||||
return () => observer.disconnect();
|
||||
}, [conversationEvents.length]);
|
||||
|
||||
if (manualMode) {
|
||||
return (
|
||||
<div className="mt-auto flex flex-col">
|
||||
<ChromaGradient sessionStarted={sessionStarted} />
|
||||
</div>
|
||||
);
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="mt-auto flex flex-col">
|
||||
<ChromaGradient sessionStarted={sessionStarted} />
|
||||
|
||||
@@ -34,6 +34,7 @@ interface DevtoolsComposerProps {
|
||||
demoMode?: boolean;
|
||||
enhancementEnabled?: boolean;
|
||||
autoExtensionEnabled?: boolean;
|
||||
manualContinuationEnabled?: boolean;
|
||||
loopGenerationEnabled?: boolean;
|
||||
curatedPromptLimit?: number;
|
||||
maxCuratedPromptCount?: number;
|
||||
@@ -49,6 +50,7 @@ interface DevtoolsComposerProps {
|
||||
onEnhancementToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onCuratedPromptLimitChange?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onAutoExtensionToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onManualContinuationToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onLoopToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onLivePromptModeToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onSpeechTranscript?: (text: string) => void;
|
||||
@@ -71,6 +73,7 @@ export default function DevtoolsComposer({
|
||||
demoMode = false,
|
||||
enhancementEnabled = true,
|
||||
autoExtensionEnabled = false,
|
||||
manualContinuationEnabled = false,
|
||||
loopGenerationEnabled = false,
|
||||
curatedPromptLimit = 0,
|
||||
maxCuratedPromptCount = 0,
|
||||
@@ -86,6 +89,7 @@ export default function DevtoolsComposer({
|
||||
onEnhancementToggle = () => {},
|
||||
onCuratedPromptLimitChange = () => {},
|
||||
onAutoExtensionToggle = () => {},
|
||||
onManualContinuationToggle = () => {},
|
||||
onLoopToggle = () => {},
|
||||
onLivePromptModeToggle = () => {},
|
||||
onSpeechTranscript,
|
||||
@@ -328,6 +332,28 @@ export default function DevtoolsComposer({
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="flex items-start gap-3">
|
||||
<Checkbox
|
||||
id="devtools-steering-mode"
|
||||
checked={manualContinuationEnabled}
|
||||
onCheckedChange={(checked) =>
|
||||
onManualContinuationToggle({
|
||||
target: { checked: Boolean(checked) },
|
||||
currentTarget: { checked: Boolean(checked) },
|
||||
} as React.ChangeEvent<HTMLInputElement>)
|
||||
}
|
||||
/>
|
||||
<div className="space-y-1">
|
||||
<Label htmlFor="devtools-steering-mode">
|
||||
Steering mode
|
||||
</Label>
|
||||
<p className="text-sm text-muted-foreground">
|
||||
Drive each segment manually — type the next scene to
|
||||
continue (vs the automatic 6-segment rollout).
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="flex items-start gap-3">
|
||||
<Checkbox
|
||||
id="devtools-loop-generation"
|
||||
|
||||
@@ -21,6 +21,7 @@ interface DevtoolsShellProps {
|
||||
selectedPresetId?: string;
|
||||
enhancementEnabled?: boolean;
|
||||
autoExtensionEnabled?: boolean;
|
||||
manualContinuationEnabled?: boolean;
|
||||
loopGenerationEnabled?: boolean;
|
||||
canJoinSession?: boolean;
|
||||
canSubmitContinuation?: boolean;
|
||||
@@ -34,6 +35,7 @@ interface DevtoolsShellProps {
|
||||
onEnhancementToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onCuratedPromptLimitChange?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onAutoExtensionToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onManualContinuationToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onLoopToggle?: (e: React.ChangeEvent<HTMLInputElement>) => void;
|
||||
onJoin?: () => void;
|
||||
onLeave?: () => void;
|
||||
@@ -128,6 +130,7 @@ export default function DevtoolsShell({
|
||||
selectedPresetId = '',
|
||||
enhancementEnabled = true,
|
||||
autoExtensionEnabled = false,
|
||||
manualContinuationEnabled = false,
|
||||
loopGenerationEnabled = false,
|
||||
canJoinSession = false,
|
||||
canSubmitContinuation = false,
|
||||
@@ -141,6 +144,7 @@ export default function DevtoolsShell({
|
||||
onEnhancementToggle = () => {},
|
||||
onCuratedPromptLimitChange = () => {},
|
||||
onAutoExtensionToggle = () => {},
|
||||
onManualContinuationToggle = () => {},
|
||||
onLoopToggle = () => {},
|
||||
onJoin = () => {},
|
||||
onLeave = () => {},
|
||||
@@ -284,6 +288,7 @@ export default function DevtoolsShell({
|
||||
demoMode={demoMode}
|
||||
enhancementEnabled={enhancementEnabled}
|
||||
autoExtensionEnabled={autoExtensionEnabled}
|
||||
manualContinuationEnabled={manualContinuationEnabled}
|
||||
loopGenerationEnabled={loopGenerationEnabled}
|
||||
curatedPromptLimit={curatedPromptLimit}
|
||||
maxCuratedPromptCount={maxCuratedPromptCount}
|
||||
@@ -299,6 +304,7 @@ export default function DevtoolsShell({
|
||||
onEnhancementToggle={onEnhancementToggle}
|
||||
onCuratedPromptLimitChange={onCuratedPromptLimitChange}
|
||||
onAutoExtensionToggle={onAutoExtensionToggle}
|
||||
onManualContinuationToggle={onManualContinuationToggle}
|
||||
onLoopToggle={onLoopToggle}
|
||||
onLivePromptModeToggle={onLivePromptModeToggle}
|
||||
onSpeechTranscript={onSpeechTranscript}
|
||||
|
||||
@@ -57,4 +57,32 @@ describe('prependPromptEvent', () => {
|
||||
expect(next[0].promptId).toBe('new');
|
||||
expect(next.some((item: any) => item.promptId === 'p-23')).toBe(false);
|
||||
});
|
||||
|
||||
it('never drops steering scene events when capping', () => {
|
||||
// 30 scenes interleaved with 30 other events — well past the cap.
|
||||
let events: Record<string, any>[] = [];
|
||||
for (let i = 0; i < 30; i += 1) {
|
||||
events = prependPromptEvent(events, {
|
||||
promptId: `scene-${i}`,
|
||||
status: 'submitted',
|
||||
steeringUserPrompt: true,
|
||||
rawText: `scene ${i}`,
|
||||
});
|
||||
events = prependPromptEvent(events, {
|
||||
promptId: `other-${i}`,
|
||||
status: 'submitted',
|
||||
});
|
||||
}
|
||||
|
||||
const scenes = events.filter((e) => e.steeringUserPrompt);
|
||||
expect(scenes).toHaveLength(30);
|
||||
// Oldest-first scene order (and therefore numbering) is stable and complete.
|
||||
expect(scenes[scenes.length - 1].promptId).toBe('scene-0');
|
||||
expect(scenes[0].promptId).toBe('scene-29');
|
||||
// Non-scene events are still capped, oldest dropped first.
|
||||
const others = events.filter((e) => !e.steeringUserPrompt);
|
||||
expect(others.length).toBeLessThanOrEqual(24);
|
||||
expect(others.some((e) => e.promptId === 'other-0')).toBe(false);
|
||||
expect(others[0].promptId).toBe('other-29');
|
||||
});
|
||||
});
|
||||
|
||||
@@ -16,5 +16,19 @@ export function prependPromptEvent(
|
||||
events: Record<string, any>[],
|
||||
event: Record<string, any>,
|
||||
): Record<string, any>[] {
|
||||
return [event, ...events].slice(0, MAX_PROMPT_EVENTS);
|
||||
const next = [event, ...events];
|
||||
if (next.length <= MAX_PROMPT_EVENTS) {
|
||||
return next;
|
||||
}
|
||||
// Steering scene events (steeringUserPrompt) are exempt from the cap: the
|
||||
// scene list is derived from them and must stay complete and stably numbered
|
||||
// for long sessions. Only the oldest non-scene events are dropped.
|
||||
let nonSceneKept = 0;
|
||||
return next.filter((e) => {
|
||||
if (e?.steeringUserPrompt) {
|
||||
return true;
|
||||
}
|
||||
nonSceneKept += 1;
|
||||
return nonSceneKept <= MAX_PROMPT_EVENTS;
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,6 +1,11 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
|
||||
import { resolveSessionErrorMessage } from './reducer';
|
||||
import { applyNormalizedSocketEvent, resolveSessionErrorMessage } from './reducer';
|
||||
import { createSessionStore } from '../../stores/session';
|
||||
import { createRewriteStore } from '../../stores/rewrite';
|
||||
import { createStreamStore } from '../../stores/stream';
|
||||
import { createUiStore } from '../../stores/ui';
|
||||
import { createPromptWindowStore } from '../../stores/promptWindow';
|
||||
|
||||
describe('resolveSessionErrorMessage', () => {
|
||||
it('returns a dedicated message for IP session limit errors', () => {
|
||||
@@ -19,3 +24,162 @@ describe('resolveSessionErrorMessage', () => {
|
||||
})).toBe('Backend replica unavailable. Rejoin session.');
|
||||
});
|
||||
});
|
||||
|
||||
function buildContext(overrides: Record<string, unknown> = {}) {
|
||||
const sessionStore = createSessionStore();
|
||||
const rewriteStore = createRewriteStore();
|
||||
const streamStore = createStreamStore();
|
||||
const uiStore = createUiStore();
|
||||
const promptWindowStore = createPromptWindowStore();
|
||||
const avPipeline = {
|
||||
reset: () => {},
|
||||
setStreamCompleted: () => {},
|
||||
noteSegmentInit: () => {},
|
||||
noteSegmentComplete: () => {},
|
||||
maybeStartPlayback: () => {},
|
||||
ensurePipeline: async () => {},
|
||||
};
|
||||
return {
|
||||
sessionStore,
|
||||
promptWindowStore,
|
||||
rewriteStore,
|
||||
streamStore,
|
||||
uiStore,
|
||||
avPipeline,
|
||||
tick: async () => {},
|
||||
defaultAvMime: 'video/mp4',
|
||||
fixedRewriteModel: 'model',
|
||||
parseLatencyMs: () => null,
|
||||
formatPromptWindowEventText: () => '',
|
||||
makePromptId: () => 'generated-id',
|
||||
buildClipLabel: () => 'clip',
|
||||
startSessionCountdown: () => {},
|
||||
clearCountdownInterval: () => {},
|
||||
resetTtffTimer: () => {},
|
||||
startTtffTimer: () => {},
|
||||
preserveArchivedPlaybackSelection: false,
|
||||
finalizeStreamCompletion: async () => {},
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
describe('steering generatingNextScene flow', () => {
|
||||
it('sets generatingNextScene on prompt/sources_resumed in manual mode', async () => {
|
||||
const context = buildContext();
|
||||
await applyNormalizedSocketEvent(
|
||||
{ type: 'prompt/sources_resumed', payload: { segment_idx: 2 } },
|
||||
context,
|
||||
);
|
||||
expect(context.sessionStore.get().generatingNextScene).toBe(true);
|
||||
expect(context.sessionStore.get().waitingForSegmentPrompt).toBe(false);
|
||||
});
|
||||
|
||||
it('does NOT set generatingNextScene on session/auto_extension_updated', async () => {
|
||||
const context = buildContext();
|
||||
await applyNormalizedSocketEvent(
|
||||
{ type: 'session/auto_extension_updated', payload: { enabled: true } },
|
||||
context,
|
||||
);
|
||||
expect(context.sessionStore.get().generatingNextScene).toBe(false);
|
||||
expect(context.sessionStore.get().waitingForSegmentPrompt).toBe(false);
|
||||
});
|
||||
|
||||
it('clears generatingNextScene when segment media arrives', async () => {
|
||||
const context = buildContext();
|
||||
context.sessionStore.patch({ generatingNextScene: true });
|
||||
await applyNormalizedSocketEvent(
|
||||
{
|
||||
type: 'stream/media_init',
|
||||
payload: { segment_idx: 2, stream_id: 's', mime: 'video/mp4' },
|
||||
},
|
||||
context,
|
||||
);
|
||||
expect(context.sessionStore.get().generatingNextScene).toBe(false);
|
||||
});
|
||||
|
||||
it('clears generatingNextScene and returns to waiting on prompt/sources_blocked', async () => {
|
||||
const context = buildContext();
|
||||
context.sessionStore.patch({ generatingNextScene: true });
|
||||
await applyNormalizedSocketEvent(
|
||||
{ type: 'prompt/sources_blocked', payload: { segment_idx: 3 } },
|
||||
context,
|
||||
);
|
||||
expect(context.sessionStore.get().generatingNextScene).toBe(false);
|
||||
expect(context.sessionStore.get().waitingForSegmentPrompt).toBe(true);
|
||||
});
|
||||
|
||||
it('clears generatingNextScene when the opening prompt falls back', async () => {
|
||||
const context = buildContext();
|
||||
context.sessionStore.patch({ generatingNextScene: true });
|
||||
await applyNormalizedSocketEvent(
|
||||
{
|
||||
type: 'prompt/fallback_used',
|
||||
payload: { prompt_id: 'p1', prompt: '', source: 'user_enhancement_failed' },
|
||||
},
|
||||
context,
|
||||
);
|
||||
expect(context.sessionStore.get().generatingNextScene).toBe(false);
|
||||
expect(context.sessionStore.get().waitingForSegmentPrompt).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe('opening prompt id tracking', () => {
|
||||
it('routes prompt lifecycle updates to the frontend-recorded opening event', async () => {
|
||||
// The frontend records the opening scene under its own prompt id and sends it
|
||||
// as initial_rollout_prompt_id; the backend echoes it in status updates.
|
||||
const context = buildContext();
|
||||
context.rewriteStore.addPromptEvent({
|
||||
promptId: 'opening-id',
|
||||
status: 'rewrite_requested',
|
||||
source: 'user_rewrite',
|
||||
text: 'a castle at dawn',
|
||||
steeringUserPrompt: true,
|
||||
rawText: 'a castle at dawn',
|
||||
});
|
||||
|
||||
await applyNormalizedSocketEvent(
|
||||
{ type: 'prompt/enhancing', payload: { prompt_id: 'opening-id' } },
|
||||
context,
|
||||
);
|
||||
let opening = (context.rewriteStore.get().promptEvents as Record<string, any>[])
|
||||
.find((e) => e.promptId === 'opening-id');
|
||||
expect(opening?.status).toBe('enhancing');
|
||||
|
||||
await applyNormalizedSocketEvent(
|
||||
{
|
||||
type: 'prompt/fallback_used',
|
||||
payload: { prompt_id: 'opening-id', prompt: '', source: 'user_enhancement_failed' },
|
||||
},
|
||||
context,
|
||||
);
|
||||
opening = (context.rewriteStore.get().promptEvents as Record<string, any>[])
|
||||
.find((e) => e.promptId === 'opening-id');
|
||||
expect(opening?.status).toBe('ready_fallback');
|
||||
// A failed opening is dropped from the steering scene list instead of
|
||||
// lingering as a ghost "Scene 1".
|
||||
expect(opening?.steeringFailed).toBe(true);
|
||||
});
|
||||
|
||||
it('marks a prompt-scoped session/error (e.g. safety block) as steeringFailed', async () => {
|
||||
const context = buildContext();
|
||||
context.rewriteStore.addPromptEvent({
|
||||
promptId: 'blocked-id',
|
||||
status: 'queued',
|
||||
source: 'user_raw',
|
||||
text: 'a blocked prompt',
|
||||
steeringUserPrompt: true,
|
||||
rawText: 'a blocked prompt',
|
||||
});
|
||||
|
||||
await applyNormalizedSocketEvent(
|
||||
{
|
||||
type: 'session/error',
|
||||
payload: { message: 'Prompt blocked by safety filter.', prompt_id: 'blocked-id' },
|
||||
},
|
||||
context,
|
||||
);
|
||||
const blocked = (context.rewriteStore.get().promptEvents as Record<string, any>[])
|
||||
.find((e) => e.promptId === 'blocked-id');
|
||||
expect(blocked?.steeringFailed).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -99,7 +99,19 @@ export async function applyNormalizedSocketEvent(event: any, context: any): Prom
|
||||
status: "ready_fallback",
|
||||
source: payload.source || "user_raw",
|
||||
text: payload.prompt,
|
||||
// Steering: this prompt produced no segment — drop it from the scene list.
|
||||
steeringFailed: true,
|
||||
});
|
||||
// Steering recovery: enhancement failed so the backend enqueued nothing AND won't
|
||||
// re-emit prompt_sources_blocked (its drained flag is still set). Put the user back to
|
||||
// "describe the next scene" ourselves so the generating overlay clears and they can retry.
|
||||
if (!uiStore.get().simpleMode && sessionStore.get().manualContinuationMode) {
|
||||
sessionStore.patch({
|
||||
waitingForSegmentPrompt: true,
|
||||
generatingNextScene: false,
|
||||
sessionNotice: "Couldn't continue from that prompt — try rephrasing the next scene.",
|
||||
});
|
||||
}
|
||||
console.warn("[PromptEnhanceFallback] Prompt extension failed for this request.");
|
||||
return;
|
||||
|
||||
@@ -243,19 +255,32 @@ export async function applyNormalizedSocketEvent(event: any, context: any): Prom
|
||||
}
|
||||
|
||||
case "prompt/sources_blocked":
|
||||
sessionStore.patch({
|
||||
autoExtensionTimeoutHint: uiStore.get().simpleMode ? "" : "blocked on user input, increase prompt count for smoother experience",
|
||||
});
|
||||
if (sessionStore.get().manualContinuationMode) {
|
||||
sessionStore.patch({ waitingForSegmentPrompt: true, generatingNextScene: false, autoExtensionTimeoutHint: "" });
|
||||
} else {
|
||||
sessionStore.patch({
|
||||
autoExtensionTimeoutHint: uiStore.get().simpleMode ? "" : "blocked on user input, increase prompt count for smoother experience",
|
||||
});
|
||||
}
|
||||
return;
|
||||
|
||||
case "prompt/sources_resumed":
|
||||
sessionStore.patch({
|
||||
autoExtensionTimeoutHint: "",
|
||||
waitingForSegmentPrompt: false,
|
||||
// A real prompt was just selected for the next segment; media arriving
|
||||
// (stream/media_init) clears this again.
|
||||
...(sessionStore.get().manualContinuationMode ? { generatingNextScene: true } : {}),
|
||||
});
|
||||
return;
|
||||
|
||||
case "session/auto_extension_updated":
|
||||
sessionStore.patch({ autoExtensionTimeoutHint: "" });
|
||||
if (event.type === "session/auto_extension_updated") {
|
||||
console.log("[AutoExtensionUpdated]", {
|
||||
enabled: sessionStore.get().autoExtensionEnabled,
|
||||
});
|
||||
}
|
||||
// Deliberately does NOT touch generatingNextScene: toggling auto extension
|
||||
// starts no generation.
|
||||
sessionStore.patch({ autoExtensionTimeoutHint: "", waitingForSegmentPrompt: false });
|
||||
console.log("[AutoExtensionUpdated]", {
|
||||
enabled: sessionStore.get().autoExtensionEnabled,
|
||||
});
|
||||
return;
|
||||
|
||||
case "segment/step_complete":
|
||||
@@ -277,6 +302,7 @@ export async function applyNormalizedSocketEvent(event: any, context: any): Prom
|
||||
projectResetPending: false,
|
||||
sessionExpired: true,
|
||||
sessionNotice: "",
|
||||
generatingNextScene: false,
|
||||
});
|
||||
console.log("Session timed out");
|
||||
clearCountdownInterval();
|
||||
@@ -354,6 +380,8 @@ export async function applyNormalizedSocketEvent(event: any, context: any): Prom
|
||||
return;
|
||||
|
||||
case "stream/media_init":
|
||||
// Segment media is arriving — the "Generating next scene" phase is over.
|
||||
sessionStore.patch({ generatingNextScene: false });
|
||||
streamStore.patch({
|
||||
mediaAppendError: null,
|
||||
loadingAnimation: streamStore.get().avPlaybackStarted ? streamStore.get().loadingAnimation : true,
|
||||
@@ -471,17 +499,30 @@ export async function applyNormalizedSocketEvent(event: any, context: any): Prom
|
||||
sessionStore.patch({
|
||||
generationCapReached: false,
|
||||
sessionNotice: "",
|
||||
generatingNextScene: false,
|
||||
});
|
||||
await finalizeStreamCompletion();
|
||||
return;
|
||||
|
||||
case "session/error": {
|
||||
const errorMessage = resolveSessionErrorMessage(payload);
|
||||
if (payload?.prompt_id) {
|
||||
// Prompt-scoped error (e.g. safety-blocked): the prompt produced no
|
||||
// segment, so drop it from the steering scene list.
|
||||
rewriteStore.trackPromptEvent(payload.prompt_id, {
|
||||
steeringFailed: true,
|
||||
});
|
||||
}
|
||||
sessionStore.patch({
|
||||
generationCapReached: false,
|
||||
preservePlaybackOnClose: false,
|
||||
promptExtensionError: "",
|
||||
sessionNotice: errorMessage,
|
||||
// Steering: a blocked/failed prompt produced no segment and the backend won't re-emit
|
||||
// prompt_sources_blocked, so recover the "describe the next scene" state ourselves.
|
||||
...(!uiStore.get().simpleMode && sessionStore.get().manualContinuationMode
|
||||
? { waitingForSegmentPrompt: true, generatingNextScene: false }
|
||||
: {}),
|
||||
});
|
||||
rewriteStore.patch({
|
||||
rewritingSeedPrompts: false,
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user