fix decoder needing separate conditional dropping probabilities for image embeddings and text encodings, thanks to @xiankgx !

Merge pull request #37 from ProGamerGov/patch-1
Fix spelling and grammatical errors
2026-02-12 19:44:26 +01:00 · 2022-04-30 08:48:05 -07:00 · 2022-04-30 08:19:07 -07:00 · 2022-04-30 09:18:13 -06:00 · 2022-04-30 07:22:57 -07:00 · 2022-04-30 06:40:54 -07:00
3 changed files with 33 additions and 28 deletions
--- a/README.md
+++ b/README.md
@@ -47,7 +47,7 @@ clip = CLIP(
    use_all_token_embeds = True,            # whether to use fine-grained contrastive learning (FILIP)
    decoupled_contrastive_learning = True,  # use decoupled contrastive learning (DCL) objective function, removing positive pairs from the denominator of the InfoNCE loss (CLOOB + DCL)
    extra_latent_projection = True,         # whether to use separate projections for text-to-image vs image-to-text comparisons (CLOOB)
-    use_visual_ssl = True,                  # whether to do self supervised learning on iages
+    use_visual_ssl = True,                  # whether to do self supervised learning on images
    visual_ssl_type = 'simclr',             # can be either 'simclr' or 'simsiam', depending on using DeCLIP or SLIP
    use_mlm = False,                        # use masked language learning (MLM) on text (DeCLIP)
    text_ssl_loss_weight = 0.05,            # weight for text MLM loss
@@ -110,7 +110,8 @@ decoder = Decoder(
    unet = unet,
    clip = clip,
    timesteps = 100,
-    cond_drop_prob = 0.2
+    image_cond_drop_prob = 0.1,
+    text_cond_drop_prob = 0.5
 ).cuda()

 # mock images (get a lot of this)
@@ -229,7 +230,8 @@ decoder = Decoder(
    unet = (unet1, unet2),            # insert both unets in order of low resolution to highest resolution (you can have as many stages as you want here)
    image_sizes = (256, 512),         # resolutions, 256 for first unet, 512 for second. these must be unique and in ascending order (matches with the unets passed in)
    timesteps = 1000,
-    cond_drop_prob = 0.2
+    image_cond_drop_prob = 0.1,
+    text_cond_drop_prob = 0.5
 ).cuda()

 # mock images (get a lot of this)
@@ -348,7 +350,8 @@ decoder = Decoder(
    image_sizes = (128, 256),
    clip = clip,
    timesteps = 100,
-    cond_drop_prob = 0.2,
+    image_cond_drop_prob = 0.1,
+    text_cond_drop_prob = 0.5,
    condition_on_text_encodings = False  # set this to True if you wish to condition on text during training and sampling
 ).cuda()

@@ -499,9 +502,7 @@ loss.backward()

 Although there is the possibility they are using an unreleased, more powerful CLIP, you can use one of the released ones, if you do not wish to train your own CLIP from scratch. This will also allow the community to more quickly validate the conclusions of the paper.

-First you'll need to install <a href="https://github.com/openai/CLIP#usage">the prerequisites</a>
-
-Then to use a pretrained OpenAI CLIP, simply import `OpenAIClipAdapter` and pass it into the `DiffusionPrior` or `Decoder` like so
+To use a pretrained OpenAI CLIP, simply import `OpenAIClipAdapter` and pass it into the `DiffusionPrior` or `Decoder` like so

 ```python
 import torch
@@ -560,7 +561,8 @@ decoder = Decoder(
    image_sizes = (128, 256),
    clip = clip,
    timesteps = 100,
-    cond_drop_prob = 0.2,
+    image_cond_drop_prob = 0.1,
+    text_cond_drop_prob = 0.5,
    condition_on_text_encodings = False  # set this to True if you wish to condition on text during training and sampling
 ).cuda()

@@ -618,7 +620,7 @@ clip = CLIP(
 # 3 unets for the decoder (a la cascading DDPM)

 # first two unets are doing latent diffusion
-# vqgan-vae must be trained before hand
+# vqgan-vae must be trained beforehand

 vae1 = VQGanVAE(
    dim = 32,
@@ -671,7 +673,8 @@ decoder = Decoder(
    unet = (unet1, unet2, unet3),      # insert unets in order of low resolution to highest resolution (you can have as many stages as you want here)
    image_sizes = (256, 512, 1024),    # resolutions, 256 for first unet, 512 for second, 1024 for third
    timesteps = 100,
-    cond_drop_prob = 0.2
+    image_cond_drop_prob = 0.1,
+    text_cond_drop_prob = 0.5
 ).cuda()

 # mock images (get a lot of this)
--- a/dalle2_pytorch/dalle2_pytorch.py
+++ b/dalle2_pytorch/dalle2_pytorch.py
@@ -172,17 +172,13 @@ class OpenAIClipAdapter(BaseClipAdapter):
        self,
        name = 'ViT-B/32'
    ):
-        try:
-            import clip
-        except ImportError:
-            print('you must install openai clip in order to use this adapter - `pip install git+https://github.com/openai/CLIP.git` - more instructions at https://github.com/openai/CLIP#usage')
-
-        openai_clip, _ = clip.load(name)
+        import clip
+        openai_clip, preprocess = clip.load(name)
        super().__init__(openai_clip)

        text_attention_final = self.find_layer('ln_final')
        self.handle = text_attention_final.register_forward_hook(self._hook)
-        self.clip_normalize = T.Normalize((0.48145466, 0.4578275, 0.40821073), (0.26862954, 0.26130258, 0.27577711))
+        self.clip_normalize = preprocess.transforms[-1]
        self.cleared = False

    def find_layer(self,  layer):
@@ -1178,7 +1174,7 @@ class Unet(nn.Module):
        if cond_scale == 1:
            return logits

-        null_logits = self.forward(*args, cond_drop_prob = 1., **kwargs)
+        null_logits = self.forward(*args, text_cond_drop_prob = 1., image_cond_drop_prob = 1., **kwargs)
        return null_logits + (logits - null_logits) * cond_scale

    def forward(
@@ -1189,7 +1185,8 @@ class Unet(nn.Module):
        image_embed,
        lowres_cond_img = None,
        text_encodings = None,
-        cond_drop_prob = 0.,
+        image_cond_drop_prob = 0.,
+        text_cond_drop_prob = 0.,
        blur_sigma = None,
        blur_kernel_size = None
    ):
@@ -1208,8 +1205,10 @@ class Unet(nn.Module):

        # conditional dropout

-        keep_mask = prob_mask_like((batch_size,), 1 - cond_drop_prob, device = device)
-        keep_mask = rearrange(keep_mask, 'b -> b 1 1')
+        image_keep_mask = prob_mask_like((batch_size,), 1 - image_cond_drop_prob, device = device)
+        text_keep_mask = prob_mask_like((batch_size,), 1 - text_cond_drop_prob, device = device)
+
+        image_keep_mask, text_keep_mask = rearrange_many((image_keep_mask, text_keep_mask), 'b -> b 1 1')

        # mask out image embedding depending on condition dropout
        # for classifier free guidance
@@ -1220,7 +1219,7 @@ class Unet(nn.Module):
            image_tokens = self.image_to_cond(image_embed)

            image_tokens = torch.where(
-                keep_mask,
+                image_keep_mask,
                image_tokens,
                self.null_image_embed
            )
@@ -1232,7 +1231,7 @@ class Unet(nn.Module):
        if exists(text_encodings) and self.cond_on_text_encodings:
            text_tokens = self.text_to_cond(text_encodings)
            text_tokens = torch.where(
-                keep_mask,
+                text_keep_mask,
                text_tokens,
                self.null_text_embed[:, :text_tokens.shape[1]]
            )
@@ -1322,7 +1321,8 @@ class Decoder(BaseGaussianDiffusion):
        clip,
        vae = tuple(),
        timesteps = 1000,
-        cond_drop_prob = 0.2,
+        image_cond_drop_prob = 0.1,
+        text_cond_drop_prob = 0.5,
        loss_type = 'l1',
        beta_schedule = 'cosine',
        predict_x_start = False,
@@ -1406,7 +1406,8 @@ class Decoder(BaseGaussianDiffusion):

        # classifier free guidance

-        self.cond_drop_prob = cond_drop_prob
+        self.image_cond_drop_prob = image_cond_drop_prob
+        self.text_cond_drop_prob = text_cond_drop_prob

    def get_unet(self, unet_number):
        assert 0 < unet_number <= len(self.unets)
@@ -1488,7 +1489,8 @@ class Decoder(BaseGaussianDiffusion):
            image_embed = image_embed,
            text_encodings = text_encodings,
            lowres_cond_img = lowres_cond_img,
-            cond_drop_prob = self.cond_drop_prob
+            image_cond_drop_prob = self.image_cond_drop_prob,
+            text_cond_drop_prob = self.text_cond_drop_prob,
        )

        target = noise if not predict_x_start else x_start
@@ -1636,4 +1638,3 @@ class DALLE2(nn.Module):
            return images[0]

        return images
-
--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ setup(
      'dream = dalle2_pytorch.cli:dream'
    ],
  },
-  version = '0.0.72',
+  version = '0.0.74',
  license='MIT',
  description = 'DALL-E 2',
  author = 'Phil Wang',
@@ -23,6 +23,7 @@ setup(
  ],
  install_requires=[
    'click',
+    'clip-anytorch',
    'einops>=0.4',
    'einops-exts>=0.0.3',
    'kornia>=0.5.4',
Author	SHA1	Message	Date
Phil Wang	f19c99ecb0	fix decoder needing separate conditional dropping probabilities for image embeddings and text encodings, thanks to @xiankgx !	2022-04-30 08:48:05 -07:00
Phil Wang	721a444686	Merge pull request #37 from ProGamerGov/patch-1 Fix spelling and grammatical errors	2022-04-30 08:19:07 -07:00
ProGamerGov	63450b466d	Fix spelling and grammatical errors	2022-04-30 09:18:13 -06:00
Phil Wang	20e7eb5a9b	cleanup	2022-04-30 07:22:57 -07:00
Phil Wang	e2f9615afa	use @clip-anytorch , thanks to @rom1504	2022-04-30 06:40:54 -07:00