From f9a408a5139c890f5b7d0c28f405c6923597f138 Mon Sep 17 00:00:00 2001 From: UniversePeak <113168673+UniversePeak@users.noreply.github.com> Date: Mon, 5 Oct 2026 03:42:37 +0800 Subject: [PATCH] Stop passing fps to the transformer in Cosmos2VideoToWorldPipeline The released Cosmos-Predict2 checkpoints ship with rope_enable_fps_modulation=False, so natively they use raw integer temporal RoPE positions and ignore FPS entirely. The diffusers port of Cosmos2VideoToWorldPipeline always passed fps (default 16) to CosmosTransformer3DModel, whose CosmosRotaryPosEmbed scales temporal positions by base_fps / fps = 24 / 16 = 1.5x for every default run, diverging from the reference implementation and making the playback fps change the generated content. Cosmos2TextToImagePipeline and the Cosmos 2.5 pipelines already pass no fps, and the transformer class is shared with Cosmos-Predict1, whose pipelines still pass it, so the pipeline is the right place to encode this difference. fps stays a playback-only argument of the pipeline. --- .../cosmos/pipeline_cosmos2_video2world.py | 7 ++++--- tests/pipelines/cosmos/test_cosmos2_video2world.py | 14 ++++++++++++++ 2 files changed, 18 insertions(+), 3 deletions(-) diff --git a/src/diffusers/pipelines/cosmos/pipeline_cosmos2_video2world.py b/src/diffusers/pipelines/cosmos/pipeline_cosmos2_video2world.py index 01b8d7b73680..446343258bbf 100644 --- a/src/diffusers/pipelines/cosmos/pipeline_cosmos2_video2world.py +++ b/src/diffusers/pipelines/cosmos/pipeline_cosmos2_video2world.py @@ -538,7 +538,10 @@ def __call__( of [Imagen Paper](https://huggingface.co/papers/2205.11487). Guidance scale is enabled by setting `guidance_scale > 1`. fps (`int`, defaults to `16`): - The frames per second of the generated video. + The frames per second of the generated video. This is only a playback property used when saving the + output with e.g. [`export_to_video`]; it does not affect the denoising process. The released + Cosmos-Predict2 checkpoints were trained without FPS-conditioned RoPE, so `fps` is not passed to the + transformer. num_videos_per_prompt (`int`, *optional*, defaults to 1): The number of images to generate per prompt. generator (`torch.Generator` or `list[torch.Generator]`, *optional*): @@ -710,7 +713,6 @@ def __call__( hidden_states=cond_latent, timestep=cond_timestep, encoder_hidden_states=prompt_embeds, - fps=fps, condition_mask=cond_mask, padding_mask=padding_mask, return_dict=False, @@ -729,7 +731,6 @@ def __call__( hidden_states=uncond_latent, timestep=uncond_timestep, encoder_hidden_states=negative_prompt_embeds, - fps=fps, condition_mask=uncond_mask, padding_mask=padding_mask, return_dict=False, diff --git a/tests/pipelines/cosmos/test_cosmos2_video2world.py b/tests/pipelines/cosmos/test_cosmos2_video2world.py index 09a4c407272b..99e006880a46 100644 --- a/tests/pipelines/cosmos/test_cosmos2_video2world.py +++ b/tests/pipelines/cosmos/test_cosmos2_video2world.py @@ -135,6 +135,20 @@ def test_inference(self): generated_slice = torch.cat([generated_slice[:8], generated_slice[-8:]]) assert_tensors_close(generated_slice, expected_slice, atol=1e-3) + def test_fps_does_not_change_generation(self): + # The released Cosmos-Predict2 checkpoints are trained without FPS RoPE modulation + # (`rope_enable_fps_modulation=False`), so the playback `fps` must not influence generation. + pipe = self.get_pipeline() + + videos = [] + for fps in (16, 24, 30): + inputs = self.get_dummy_inputs() + inputs["fps"] = fps + videos.append(pipe(**inputs).frames[0]) + + for video in videos[1:]: + assert (videos[0] - video).abs().max().item() == 0 + def test_inference_batch_single_identical(self, batch_size=3, expected_max_diff=1e-2): super().test_inference_batch_single_identical(batch_size=batch_size, expected_max_diff=expected_max_diff)