[bugfix]: derive result size from the decoded output when the samples mirror is skipped

PR #1362 commit 3 leaves `samples` as an empty placeholder when
`return_frames=False`, but main picked up #1595's refiner size
reporting in the meantime, and `_resolve_output_size(samples, ...)`
silently fell back to the requested geometry in exactly the common
save flow the PR optimizes. Read the geometry from
`output_batch.output` instead (shape-only access, no D->H copy),
gated on `needs_frame_output` so metadata-only and audio-only calls
still never inspect the (possibly dropped) worker output.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
Will Lin
2026-08-21 02:34:16 +00:00
co-authored by Claude Fable 5
parent aaaa7a14a3
commit 15a164a052
2 changed files with 37 additions and 1 deletions
+6 -1
View File
@@ -877,8 +877,13 @@ class VideoGenerator:
# `GenerationResult.size` describes the produced media, not only the
# base-stage request. Refiner pipelines can change the final pixel
# dimensions, so derive this result metadata from the decoded output.
# Read the geometry from `output_batch.output` (a shape-only access,
# no D->H copy): when `return_frames=False` the `samples` mirror
# stays an empty placeholder and no longer carries the decoded
# shape. Metadata-only calls keep the request fallback and never
# inspect the (possibly dropped) worker output.
output_size = _resolve_output_size(
samples,
output_batch.output if needs_frame_output else samples,
(target_height, target_width, batch.num_frames),
pixel_output=not is_latent_output and not audio_only,
)
@@ -305,6 +305,37 @@ def test_generate_single_video_save_video_still_builds_frames(monkeypatch, tmp_p
}
def test_generate_single_video_save_only_reports_refined_output_size(monkeypatch, tmp_path):
"""`GenerationResult.size` must describe the decoded media even when the
fp32 `samples` mirror is skipped (`return_frames=False`, the CLI save
flow). Refiner pipelines can change the final pixel geometry, so the size
has to come from `output_batch.output`, not the base request. CPU-only."""
# Refiner-style output: request asks for 2 frames of 16x16, pipeline
# produces 5 frames of 32x48.
output = torch.full((1, 3, 5, 32, 48), 0.5, dtype=torch.float32)
output_batch = _single_video_output_batch(output)
fastvideo_args = _single_video_args()
generator = _single_video_generator(output_batch, fastvideo_args)
saved = {}
def fake_mimsave(path, frames, *, fps, format):
saved["frame_count"] = len(frames)
monkeypatch.setattr(video_generator_module.imageio, "mimsave", fake_mimsave)
result = generator._generate_single_video(
prompt="refined save",
sampling_param=_small_sampling_param(save_video=True, return_frames=False),
fastvideo_args=fastvideo_args,
output_path=str(tmp_path / "refined.mp4"),
)
assert result["samples"] is None
assert result["frames"] is None
assert result["size"] == (32, 48, 5)
assert saved["frame_count"] == 5
def test_generate_single_video_audio_only_metadata_returns_audio_without_frames(tmp_path):
audio = torch.zeros((16, ), dtype=torch.float32)
output_batch = _single_video_output_batch(