change up epsilon in layernorm the case of using fp16, thanks to @Veldrovive for figuring out this stabilizes training

2026-02-12 11:34:29 +01:00 · 2022-07-29 12:39:56 -07:00
4 changed files with 45 additions and 236 deletions
--- a/README.md
+++ b/README.md
@@ -627,18 +627,6 @@ images = dalle2(
 # save your image (in this example, of size 256x256)
 ```

-Alternatively, you can also use <a href="https://github.com/mlfoundations/open_clip">Open Clip</a>
-
-```bash
-$ pip install open-clip-torch
-```
-
-```python
-from dalle2_pytorch import OpenClipAdapter
-
-clip = OpenClipAdapter()
-```
-
 Now you'll just have to worry about training the Prior and the Decoder!

 ## Inpainting
@@ -1253,15 +1241,4 @@ For detailed information on training the diffusion prior, please refer to the [d
 }
 ```

-```bibtex
-@misc{chen2022analog,
-    title   = {Analog Bits: Generating Discrete Data using Diffusion Models with Self-Conditioning},
-    author  = {Ting Chen and Ruixiang Zhang and Geoffrey Hinton},
-    year    = {2022},
-    eprint  = {2208.04202},
-    archivePrefix = {arXiv},
-    primaryClass = {cs.CV}
-}
-```
-
 *Creating noise from data is easy; creating data from noise is generative modeling.* - <a href="https://arxiv.org/abs/2011.13456">Yang Song's paper</a>
--- a/dalle2_pytorch/dalle2_pytorch.py
+++ b/dalle2_pytorch/dalle2_pytorch.py
@@ -8,7 +8,6 @@ from pathlib import Path

 import torch
 import torch.nn.functional as F
-from torch.utils.checkpoint import checkpoint
 from torch import nn, einsum
 import torchvision.transforms as T

@@ -109,28 +108,6 @@ def pad_tuple_to_length(t, length, fillvalue = None):
        return t
    return (*t, *((fillvalue,) * remain_length))

-# checkpointing helper function
-
-def make_checkpointable(fn, **kwargs):
-    if isinstance(fn, nn.ModuleList):
-        return [maybe(make_checkpointable)(el, **kwargs) for el in fn]
-
-    condition = kwargs.pop('condition', None)
-
-    if exists(condition) and not condition(fn):
-        return fn
-
-    @wraps(fn)
-    def inner(*args):
-        input_needs_grad = any([isinstance(el, torch.Tensor) and el.requires_grad for el in args])
-
-        if not input_needs_grad:
-            return fn(*args)
-
-        return checkpoint(fn, *args)
-
-    return inner
-
 # for controlling freezing of CLIP

 def set_module_requires_grad_(module, requires_grad):
@@ -362,75 +339,6 @@ class OpenAIClipAdapter(BaseClipAdapter):
        image_embed = self.clip.encode_image(image)
        return EmbeddedImage(l2norm(image_embed.float()), None)

-class OpenClipAdapter(BaseClipAdapter):
-    def __init__(
-        self,
-        name = 'ViT-B/32',
-        pretrained = 'laion400m_e32'
-    ):
-        import open_clip
-        clip, _, preprocess = open_clip.create_model_and_transforms(name, pretrained = pretrained)
-
-        super().__init__(clip)
-        self.eos_id = 49407
-
-        text_attention_final = self.find_layer('ln_final')
-        self.handle = text_attention_final.register_forward_hook(self._hook)
-        self.clip_normalize = preprocess.transforms[-1]
-        self.cleared = False
-
-    def find_layer(self,  layer):
-        modules = dict([*self.clip.named_modules()])
-        return modules.get(layer, None)
-
-    def clear(self):
-        if self.cleared:
-            return
-
-        self.handle()
-
-    def _hook(self, _, inputs, outputs):
-        self.text_encodings = outputs
-
-    @property
-    def dim_latent(self):
-        return 512
-
-    @property
-    def image_size(self):
-        return self.clip.visual.image_size
-
-    @property
-    def image_channels(self):
-        return 3
-
-    @property
-    def max_text_len(self):
-        return self.clip.context_length
-
-    @torch.no_grad()
-    def embed_text(self, text):
-        text = text[..., :self.max_text_len]
-
-        is_eos_id = (text == self.eos_id)
-        text_mask_excluding_eos = is_eos_id.cumsum(dim = -1) == 0
-        text_mask = F.pad(text_mask_excluding_eos, (1, -1), value = True)
-        assert not self.cleared
-
-        text_embed = self.clip.encode_text(text)
-        text_encodings = self.text_encodings
-        text_encodings = text_encodings.masked_fill(~text_mask[..., None], 0.)
-        del self.text_encodings
-        return EmbeddedText(l2norm(text_embed.float()), text_encodings.float())
-
-    @torch.no_grad()
-    def embed_image(self, image):
-        assert not self.cleared
-        image = self.validate_and_resize_image(image)
-        image = self.clip_normalize(image)
-        image_embed = self.clip.encode_image(image)
-        return EmbeddedImage(l2norm(image_embed.float()), None)
-
 # classifier free guidance functions

 def prob_mask_like(shape, prob, device):
@@ -672,7 +580,7 @@ class ChanLayerNorm(nn.Module):

        var = torch.var(x, dim = 1, unbiased = False, keepdim = True)
        mean = torch.mean(x, dim = 1, keepdim = True)
-        return (x - mean) * (var + eps).rsqrt() * self.g
+        return (x - mean) * (var + self.eps).rsqrt() * self.g

 class Residual(nn.Module):
    def __init__(self, fn):
@@ -793,12 +701,11 @@ class Attention(nn.Module):
        dropout = 0.,
        causal = False,
        rotary_emb = None,
-        cosine_sim = True,
-        cosine_sim_scale = 16
+        pb_relax_alpha = 128
    ):
        super().__init__()
-        self.scale = cosine_sim_scale if cosine_sim else (dim_head ** -0.5)
-        self.cosine_sim = cosine_sim
+        self.pb_relax_alpha = pb_relax_alpha
+        self.scale = dim_head ** -0.5 * (pb_relax_alpha ** -1)

        self.heads = heads
        inner_dim = dim_head * heads
@@ -838,13 +745,6 @@ class Attention(nn.Module):
        k = torch.cat((nk, k), dim = -2)
        v = torch.cat((nv, v), dim = -2)

-        # whether to use cosine sim
-
-        if self.cosine_sim:
-            q, k = map(l2norm, (q, k))
-
-        q, k = map(lambda t: t * math.sqrt(self.scale), (q, k))
-
        # calculate query / key similarities

        sim = einsum('b h i d, b j d -> b h i j', q, k)
@@ -870,7 +770,10 @@ class Attention(nn.Module):

        # attention

-        attn = sim.softmax(dim = -1, dtype = torch.float32)
+        sim = sim - sim.amax(dim = -1, keepdim = True).detach()
+        sim = sim * self.pb_relax_alpha
+
+        attn = sim.softmax(dim = -1)
        attn = self.dropout(attn)

        # aggregate values
@@ -1157,17 +1060,17 @@ class DiffusionPrior(nn.Module):
        pred = self.net.forward_with_cond_scale(x, t, cond_scale = cond_scale, **text_cond)

        if self.predict_x_start:
-            x_start = pred
+            x_recon = pred
        else:
-            x_start = self.noise_scheduler.predict_start_from_noise(x, t = t, noise = pred)
+            x_recon = self.noise_scheduler.predict_start_from_noise(x, t = t, noise = pred)

        if clip_denoised and not self.predict_x_start:
-            x_start.clamp_(-1., 1.)
+            x_recon.clamp_(-1., 1.)

        if self.predict_x_start and self.sampling_clamp_l2norm:
-            x_start = l2norm(x_start) * self.image_embed_scale
+            x_recon = l2norm(x_recon) * self.image_embed_scale

-        model_mean, posterior_variance, posterior_log_variance = self.noise_scheduler.q_posterior(x_start=x_start, x_t=x, t=t)
+        model_mean, posterior_variance, posterior_log_variance = self.noise_scheduler.q_posterior(x_start=x_recon, x_t=x, t=t)
        return model_mean, posterior_variance, posterior_log_variance

    @torch.no_grad()
@@ -1571,7 +1474,7 @@ class CrossAttention(nn.Module):
            mask = rearrange(mask, 'b j -> b 1 1 j')
            sim = sim.masked_fill(~mask, max_neg_value)

-        attn = sim.softmax(dim = -1, dtype = torch.float32)
+        attn = sim.softmax(dim = -1)

        out = einsum('b h i j, b h j d -> b h i d', attn, v)
        out = rearrange(out, 'b h n d -> b n (h d)')
@@ -1582,8 +1485,7 @@ class LinearAttention(nn.Module):
        self,
        dim,
        dim_head = 32,
-        heads = 8,
-        **kwargs
+        heads = 8
    ):
        super().__init__()
        self.scale = dim_head ** -0.5
@@ -1700,10 +1602,8 @@ class Unet(nn.Module):
        attn_heads = 16,
        lowres_cond = False,             # for cascading diffusion - https://cascaded-diffusion.github.io/
        lowres_noise_cond = False,       # for conditioning on low resolution noising, based on Imagen
-        self_cond = False,
        sparse_attn = False,
        cosine_sim_cross_attn = False,
-        cosine_sim_self_attn = False,
        attend_at_middle = True,         # whether to have a layer of attention at the bottleneck (can turn off for higher resolution in cascading DDPM, before bringing in efficient attention)
        cond_on_text_encodings = False,
        max_text_len = 256,
@@ -1722,7 +1622,6 @@ class Unet(nn.Module):
        pixel_shuffle_upsample = True,
        final_conv_kernel_size = 1,
        combine_upsample_fmaps = False, # whether to combine the outputs of all upsample blocks, as in unet squared paper
-        checkpoint_during_training = False,
        **kwargs
    ):
        super().__init__()
@@ -1736,21 +1635,12 @@ class Unet(nn.Module):

        self.lowres_cond = lowres_cond

-        # whether to do self conditioning
-
-        self.self_cond = self_cond
-
        # determine dimensions

        self.channels = channels
        self.channels_out = default(channels_out, channels)

-         # initial number of channels depends on
-         # (1) low resolution conditioning from cascading ddpm paper, conditioned on previous unet output in the cascade
-         # (2) self conditioning (bit diffusion paper)
-
-        init_channels = channels * (1 + int(lowres_cond) + int(self_cond))
-
+        init_channels = channels if not lowres_cond else channels * 2 # in cascading diffusion, one concats the low resolution image, blurred, for conditioning the higher resolution synthesis
        init_dim = default(init_dim, dim)

        self.init_conv = CrossEmbedLayer(init_channels, dim_out = init_dim, kernel_sizes = init_cross_embed_kernel_sizes, stride = 1) if init_cross_embed else nn.Conv2d(init_channels, init_dim, init_conv_kernel_size, padding = init_conv_kernel_size // 2)
@@ -1834,7 +1724,7 @@ class Unet(nn.Module):

        # attention related params

-        attn_kwargs = dict(heads = attn_heads, dim_head = attn_dim_head, cosine_sim = cosine_sim_self_attn)
+        attn_kwargs = dict(heads = attn_heads, dim_head = attn_dim_head)

        self_attn = cast_tuple(self_attn, num_stages)

@@ -1942,10 +1832,6 @@ class Unet(nn.Module):

        zero_init_(self.to_out) # since both OpenAI and @crowsonkb are doing it

-        # whether to checkpoint during training
-
-        self.checkpoint_during_training = checkpoint_during_training
-
    # if the current settings for the unet are not correct
    # for cascading DDPM, then reinit the unet with the right settings
    def cast_model_parameters(
@@ -2003,9 +1889,7 @@ class Unet(nn.Module):
        image_cond_drop_prob = 0.,
        text_cond_drop_prob = 0.,
        blur_sigma = None,
-        blur_kernel_size = None,
-        disable_checkpoint = False,
-        self_cond = None
+        blur_kernel_size = None
    ):
        batch_size, device = x.shape[0], x.device

@@ -2013,14 +1897,6 @@ class Unet(nn.Module):

        assert not (self.lowres_cond and not exists(lowres_cond_img)), 'low resolution conditioning image must be present'

-        # concat self conditioning, if needed
-
-        if self.self_cond:
-            self_cond = default(self_cond, lambda: torch.zeros_like(x))
-            x = torch.cat((x, self_cond), dim = 1)
-
-        # concat low resolution conditioning
-
        if exists(lowres_cond_img):
            x = torch.cat((x, lowres_cond_img), dim = 1)

@@ -2135,29 +2011,17 @@ class Unet(nn.Module):
        c = self.norm_cond(c)
        mid_c = self.norm_mid_cond(mid_c)

-        # gradient checkpointing
-
-        can_checkpoint = self.training and self.checkpoint_during_training and not disable_checkpoint
-        apply_checkpoint_fn = make_checkpointable if can_checkpoint else identity
-
-        # make checkpointable modules
-
-        init_resnet_block, mid_block1, mid_attn, mid_block2, final_resnet_block = [maybe(apply_checkpoint_fn)(module) for module in (self.init_resnet_block, self.mid_block1, self.mid_attn, self.mid_block2, self.final_resnet_block)]
-
-        can_checkpoint_cond = lambda m: isinstance(m, ResnetBlock)
-        downs, ups = [maybe(apply_checkpoint_fn)(m, condition = can_checkpoint_cond) for m in (self.downs, self.ups)]
-
        # initial resnet block

-        if exists(init_resnet_block):
-            x = init_resnet_block(x, t)
+        if exists(self.init_resnet_block):
+            x = self.init_resnet_block(x, t)

        # go through the layers of the unet, down and up

        down_hiddens = []
        up_hiddens = []

-        for pre_downsample, init_block, resnet_blocks, attn, post_downsample in downs:
+        for pre_downsample, init_block, resnet_blocks, attn, post_downsample in self.downs:
            if exists(pre_downsample):
                x = pre_downsample(x)

@@ -2173,16 +2037,16 @@ class Unet(nn.Module):
            if exists(post_downsample):
                x = post_downsample(x)

-        x = mid_block1(x, t, mid_c)
+        x = self.mid_block1(x, t, mid_c)

-        if exists(mid_attn):
-            x = mid_attn(x)
+        if exists(self.mid_attn):
+            x = self.mid_attn(x)

-        x = mid_block2(x, t, mid_c)
+        x = self.mid_block2(x, t, mid_c)

        connect_skip = lambda fmap: torch.cat((fmap, down_hiddens.pop() * self.skip_connect_scale), dim = 1)

-        for init_block, resnet_blocks, attn, upsample in ups:
+        for init_block, resnet_blocks, attn, upsample in self.ups:
            x = connect_skip(x)
            x = init_block(x, t, c)

@@ -2199,7 +2063,7 @@ class Unet(nn.Module):

        x = torch.cat((x, r), dim = 1)

-        x = final_resnet_block(x, t)
+        x = self.final_resnet_block(x, t)

        if exists(lowres_cond_img):
            x = torch.cat((x, lowres_cond_img), dim = 1)
@@ -2590,23 +2454,23 @@ class Decoder(nn.Module):
        x = x.clamp(-s, s) / s
        return x

-    def p_mean_variance(self, unet, x, t, image_embed, noise_scheduler, text_encodings = None, lowres_cond_img = None, self_cond = None, clip_denoised = True, predict_x_start = False, learned_variance = False, cond_scale = 1., model_output = None, lowres_noise_level = None):
+    def p_mean_variance(self, unet, x, t, image_embed, noise_scheduler, text_encodings = None, lowres_cond_img = None, clip_denoised = True, predict_x_start = False, learned_variance = False, cond_scale = 1., model_output = None, lowres_noise_level = None):
        assert not (cond_scale != 1. and not self.can_classifier_guidance), 'the decoder was not trained with conditional dropout, and thus one cannot use classifier free guidance (cond_scale anything other than 1)'

-        pred = default(model_output, lambda: unet.forward_with_cond_scale(x, t, image_embed = image_embed, text_encodings = text_encodings, cond_scale = cond_scale, lowres_cond_img = lowres_cond_img, self_cond = self_cond, lowres_noise_level = lowres_noise_level))
+        pred = default(model_output, lambda: unet.forward_with_cond_scale(x, t, image_embed = image_embed, text_encodings = text_encodings, cond_scale = cond_scale, lowres_cond_img = lowres_cond_img, lowres_noise_level = lowres_noise_level))

        if learned_variance:
            pred, var_interp_frac_unnormalized = pred.chunk(2, dim = 1)

        if predict_x_start:
-            x_start = pred
+            x_recon = pred
        else:
-            x_start = noise_scheduler.predict_start_from_noise(x, t = t, noise = pred)
+            x_recon = noise_scheduler.predict_start_from_noise(x, t = t, noise = pred)

        if clip_denoised:
-            x_start = self.dynamic_threshold(x_start)
+            x_recon = self.dynamic_threshold(x_recon)

-        model_mean, posterior_variance, posterior_log_variance = noise_scheduler.q_posterior(x_start=x_start, x_t=x, t=t)
+        model_mean, posterior_variance, posterior_log_variance = noise_scheduler.q_posterior(x_start=x_recon, x_t=x, t=t)

        if learned_variance:
            # if learned variance, posterio variance and posterior log variance are predicted by the network
@@ -2622,17 +2486,16 @@ class Decoder(nn.Module):
            posterior_log_variance = var_interp_frac * max_log + (1 - var_interp_frac) * min_log
            posterior_variance = posterior_log_variance.exp()

-        return model_mean, posterior_variance, posterior_log_variance, x_start
+        return model_mean, posterior_variance, posterior_log_variance

    @torch.no_grad()
-    def p_sample(self, unet, x, t, image_embed, noise_scheduler, text_encodings = None, cond_scale = 1., lowres_cond_img = None, self_cond = None, predict_x_start = False, learned_variance = False, clip_denoised = True, lowres_noise_level = None):
+    def p_sample(self, unet, x, t, image_embed, noise_scheduler, text_encodings = None, cond_scale = 1., lowres_cond_img = None, predict_x_start = False, learned_variance = False, clip_denoised = True, lowres_noise_level = None):
        b, *_, device = *x.shape, x.device
-        model_mean, _, model_log_variance, x_start = self.p_mean_variance(unet, x = x, t = t, image_embed = image_embed, text_encodings = text_encodings, cond_scale = cond_scale, lowres_cond_img = lowres_cond_img, self_cond = self_cond, clip_denoised = clip_denoised, predict_x_start = predict_x_start, noise_scheduler = noise_scheduler, learned_variance = learned_variance, lowres_noise_level = lowres_noise_level)
+        model_mean, _, model_log_variance = self.p_mean_variance(unet, x = x, t = t, image_embed = image_embed, text_encodings = text_encodings, cond_scale = cond_scale, lowres_cond_img = lowres_cond_img, clip_denoised = clip_denoised, predict_x_start = predict_x_start, noise_scheduler = noise_scheduler, learned_variance = learned_variance, lowres_noise_level = lowres_noise_level)
        noise = torch.randn_like(x)
        # no noise when t == 0
        nonzero_mask = (1 - (t == 0).float()).reshape(b, *((1,) * (len(x.shape) - 1)))
-        pred = model_mean + nonzero_mask * (0.5 * model_log_variance).exp() * noise
-        return pred, x_start
+        return model_mean + nonzero_mask * (0.5 * model_log_variance).exp() * noise

    @torch.no_grad()
    def p_sample_loop_ddpm(
@@ -2658,8 +2521,6 @@ class Decoder(nn.Module):
        b = shape[0]
        img = torch.randn(shape, device = device)

-        x_start = None # for self-conditioning
-
        is_inpaint = exists(inpaint_image)
        resample_times = inpaint_resample_times if is_inpaint else 1

@@ -2687,16 +2548,13 @@ class Decoder(nn.Module):
                    noised_inpaint_image = noise_scheduler.q_sample(inpaint_image, t = times)
                    img = (img * ~inpaint_mask) + (noised_inpaint_image * inpaint_mask)

-                self_cond = x_start if unet.self_cond else None
-
-                img, x_start = self.p_sample(
+                img = self.p_sample(
                    unet,
                    img,
                    times,
                    image_embed = image_embed,
                    text_encodings = text_encodings,
                    cond_scale = cond_scale,
-                    self_cond = self_cond,
                    lowres_cond_img = lowres_cond_img,
                    lowres_noise_level = lowres_noise_level,
                    predict_x_start = predict_x_start,
@@ -2755,8 +2613,6 @@ class Decoder(nn.Module):

        img = torch.randn(shape, device = device)

-        x_start = None # for self-conditioning
-
        if not is_latent_diffusion:
            lowres_cond_img = maybe(self.normalize_img)(lowres_cond_img)

@@ -2777,9 +2633,7 @@ class Decoder(nn.Module):
                    noised_inpaint_image = noise_scheduler.q_sample(inpaint_image, t = time_cond)
                    img = (img * ~inpaint_mask) + (noised_inpaint_image * inpaint_mask)

-                self_cond = x_start if unet.self_cond else None
-
-                pred = unet.forward_with_cond_scale(img, time_cond, image_embed = image_embed, text_encodings = text_encodings, cond_scale = cond_scale, self_cond = self_cond, lowres_cond_img = lowres_cond_img, lowres_noise_level = lowres_noise_level)
+                pred = unet.forward_with_cond_scale(img, time_cond, image_embed = image_embed, text_encodings = text_encodings, cond_scale = cond_scale, lowres_cond_img = lowres_cond_img, lowres_noise_level = lowres_noise_level)

                if learned_variance:
                    pred, _ = pred.chunk(2, dim = 1)
@@ -2839,35 +2693,13 @@ class Decoder(nn.Module):

        x_noisy = noise_scheduler.q_sample(x_start = x_start, t = times, noise = noise)

-        # unet kwargs
-
-        unet_kwargs = dict(
+        model_output = unet(
+            x_noisy,
+            times,
            image_embed = image_embed,
            text_encodings = text_encodings,
            lowres_cond_img = lowres_cond_img,
            lowres_noise_level = lowres_noise_level,
-        )
-
-        # self conditioning
-
-        self_cond = None
-
-        if unet.self_cond and random.random() < 0.5:
-            with torch.no_grad():
-                self_cond = unet(x_noisy, times, **unet_kwargs)
-
-                if learned_variance:
-                    self_cond, _ = self_cond.chunk(2, dim = 1)
-
-                self_cond = self_cond.detach()
-
-        # forward to get model prediction
-
-        model_output = unet(
-            x_noisy,
-            times,
-            **unet_kwargs,
-            self_cond = self_cond,
            image_cond_drop_prob = self.image_cond_drop_prob,
            text_cond_drop_prob = self.text_cond_drop_prob,
        )
@@ -2898,7 +2730,7 @@ class Decoder(nn.Module):
        # if learning the variance, also include the extra weight kl loss

        true_mean, _, true_log_variance_clipped = noise_scheduler.q_posterior(x_start = x_start, x_t = x_noisy, t = times)
-        model_mean, _, model_log_variance, _ = self.p_mean_variance(unet, x = x_noisy, t = times, image_embed = image_embed, noise_scheduler = noise_scheduler, clip_denoised = clip_denoised, learned_variance = True, model_output = model_output)
+        model_mean, _, model_log_variance = self.p_mean_variance(unet, x = x_noisy, t = times, image_embed = image_embed, noise_scheduler = noise_scheduler, clip_denoised = clip_denoised, learned_variance = True, model_output = model_output)

        # kl loss with detached model predicted mean, for stability reasons as in paper

--- a/dalle2_pytorch/version.py
+++ b/dalle2_pytorch/version.py
@@ -1 +1 @@
-__version__ = '1.6.0'
+__version__ = '1.4.1'
--- a/setup.py
+++ b/setup.py
@@ -26,7 +26,7 @@ setup(
  install_requires=[
    'accelerate',
    'click',
-    'clip-anytorch>=2.4.0',
+    'clip-anytorch',
    'coca-pytorch>=0.0.5',
    'ema-pytorch>=0.0.7',
    'einops>=0.4',