fix everything and make sure it runs end to end, document everything in readme for public

2026-02-12 19:44:26 +01:00 · 2022-04-13 18:04:00 -07:00
3 changed files with 5 additions and 5 deletions
--- a/README.md
+++ b/README.md
@@ -212,7 +212,8 @@ Let's see the whole script below

 ```python
 import torch
-from dalle2_pytorch import DALLE2, DiffusionPriorNetwork, DiffusionPrior, Unet, Decoder, CLIP
+from dalle2_pytorch.dalle2_pytorch import DALLE2
+from dalle2_pytorch import DiffusionPriorNetwork, DiffusionPrior, Unet, Decoder, CLIP

 import torch

--- a/dalle2_pytorch/dalle2_pytorch.py
+++ b/dalle2_pytorch/dalle2_pytorch.py
@@ -374,13 +374,12 @@ class DiffusionPrior(nn.Module):
        image_encoding = self.clip.visual_transformer(image)
        image_cls = image_encoding[:, 0]
        image_embed = self.clip.to_visual_latent(image_cls)
-        return l2norm(image_embed)
+        return image_embed

    def get_text_cond(self, text):
        text_encodings = self.clip.text_transformer(text)
        text_cls, text_encodings = text_encodings[:, 0], text_encodings[:, 1:]
        text_embed = self.clip.to_text_latent(text_cls)
-        text_embed = l2norm(text_embed)
        return dict(text_encodings = text_encodings, text_embed = text_embed, mask = text != 0)

    def q_mean_variance(self, x_start, t):
@@ -751,7 +750,7 @@ class Decoder(nn.Module):
        image_encoding = self.clip.visual_transformer(image)
        image_cls = image_encoding[:, 0]
        image_embed = self.clip.to_visual_latent(image_cls)
-        return l2norm(image_embed)
+        return image_embed

    def q_mean_variance(self, x_start, t):
        mean = extract(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ setup(
      'dream = dalle2_pytorch.cli:dream'
    ],
  },
-  version = '0.0.5',
+  version = '0.0.4',
  license='MIT',
  description = 'DALL-E 2',
  author = 'Phil Wang',