diff --git a/models/unet_2d_condition.py b/models/unet_2d_condition.py index e18f69f..804a164 100644 --- a/models/unet_2d_condition.py +++ b/models/unet_2d_condition.py @@ -38,7 +38,7 @@ from diffusers.models.embeddings import ( ImageHintTimeEmbedding, ImageProjection, ImageTimeEmbedding, - #PositionNet, + GLIGENTextBoundingboxProjection, TextImageProjection, TextImageTimeEmbedding, TextTimeEmbedding, @@ -619,7 +619,7 @@ class UNet2DConditionModel(ModelMixin, ConfigMixin, UNet2DConditionLoadersMixin) positive_len = cross_attention_dim[0] feature_type = "text-only" if attention_type == "gated" else "text-image" - self.position_net = PositionNet( + self.position_net = GLIGENTextBoundingboxProjection( positive_len=positive_len, out_dim=cross_attention_dim, feature_type=feature_type )