Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions src/diffusers/models/unets/unet_3d_condition.py
Original file line number Diff line number Diff line change
Expand Up @@ -90,6 +90,9 @@ class UNet3DConditionModel(ModelMixin, AttentionMixin, ConfigMixin, UNet2DCondit
num_attention_heads (`int`, *optional*): The number of attention heads.
time_cond_proj_dim (`int`, *optional*, defaults to `None`):
The dimension of `cond_proj` layer in the timestep embedding.
num_class_embeds (`int`, *optional*, defaults to `None`):
Input dimension of the learnable embedding matrix to be projected to `time_embed_dim`, when performing
class conditioning.
"""

_supports_gradient_checkpointing = False
Expand Down Expand Up @@ -124,6 +127,7 @@ def __init__(
attention_head_dim: int | tuple[int] = 64,
num_attention_heads: int | tuple[int] | None = None,
time_cond_proj_dim: int | None = None,
num_class_embeds: int | None = None,
):
super().__init__()

Expand Down Expand Up @@ -177,6 +181,7 @@ def __init__(
act_fn=act_fn,
cond_proj_dim=time_cond_proj_dim,
)
self.class_embedding = nn.Embedding(num_class_embeds, time_embed_dim) if num_class_embeds is not None else None

self.transformer_in = TransformerTemporalModel(
num_attention_heads=8,
Expand Down Expand Up @@ -567,6 +572,11 @@ def forward(
t_emb = t_emb.to(dtype=self.dtype)

emb = self.time_embedding(t_emb, timestep_cond)
if self.class_embedding is not None:
if class_labels is None:
raise ValueError("class_labels should be provided when num_class_embeds > 0")
emb = emb + self.class_embedding(class_labels).to(dtype=emb.dtype)

emb = emb.repeat_interleave(num_frames, dim=0, output_size=emb.shape[0] * num_frames)
encoder_hidden_states = encoder_hidden_states.repeat_interleave(
num_frames, dim=0, output_size=encoder_hidden_states.shape[0] * num_frames
Expand Down
23 changes: 23 additions & 0 deletions tests/models/unets/test_models_unet_3d_condition.py
Original file line number Diff line number Diff line change
Expand Up @@ -91,6 +91,29 @@ def test_forward_with_norm_groups(self):

assert output.shape == self.get_dummy_inputs()["sample"].shape, "Input and output shapes do not match"

def test_class_conditioning(self):
init_dict = self.get_init_dict()
init_dict["num_class_embeds"] = 2
model = self.model_class(**init_dict).to(torch_device).eval()

inputs = self.get_dummy_inputs()
batch_size = inputs["sample"].shape[0]
labels_0 = torch.zeros(batch_size, dtype=torch.long, device=torch_device)
labels_1 = torch.ones(batch_size, dtype=torch.long, device=torch_device)

with torch.no_grad():
output_0 = model(**inputs, class_labels=labels_0).sample
output_1 = model(**inputs, class_labels=labels_1).sample

assert not torch.equal(output_0, output_1)

try:
model(**inputs)
except ValueError as error:
assert "class_labels should be provided" in str(error)
else:
raise AssertionError("Expected class-conditioned UNet3D to require class_labels")

def test_feed_forward_chunking(self):
init_dict = self.get_init_dict()
init_dict["block_out_channels"] = (32, 64)
Expand Down
Loading