normalize conditioning tokens outside of cross attention blocks

lineup with paper
be able to customize adam eps
2026-02-12 11:34:29 +01:00 · 2022-05-14 14:23:52 -07:00 · 2022-05-14 13:57:00 -07:00 · 2022-05-14 13:55:04 -07:00 · 2022-05-14 13:22:43 -07:00
5 changed files with 24 additions and 9 deletions
--- a/README.md
+++ b/README.md
@@ -1017,6 +1017,7 @@ Once built, images will be saved to the same directory the command is invoked
 - [ ] offer save / load methods on the trainer classes to automatically take care of state dicts for scalers / optimizers / saving versions and checking for breaking changes
 - [ ] bring in skip-layer excitatons (from lightweight gan paper) to see if it helps for either decoder of unet or vqgan-vae training
 - [ ] decoder needs one day worth of refactor for tech debt
+- [ ] allow for unet to be able to condition non-cross attention style as well

 ## Citations

--- a/dalle2_pytorch/dalle2_pytorch.py
+++ b/dalle2_pytorch/dalle2_pytorch.py
@@ -1163,6 +1163,7 @@ class CrossAttention(nn.Module):
        dim_head = 64,
        heads = 8,
        dropout = 0.,
+        norm_context = False
    ):
        super().__init__()
        self.scale = dim_head ** -0.5
@@ -1172,7 +1173,7 @@ class CrossAttention(nn.Module):
        context_dim = default(context_dim, dim)

        self.norm = LayerNorm(dim)
-        self.norm_context = LayerNorm(context_dim)
+        self.norm_context = LayerNorm(context_dim) if norm_context else nn.Identity()
        self.dropout = nn.Dropout(dropout)

        self.null_kv = nn.Parameter(torch.randn(2, dim_head))
@@ -1378,6 +1379,9 @@ class Unet(nn.Module):
            Rearrange('b (n d) -> b n d', n = num_image_tokens)
        ) if image_embed_dim != cond_dim else nn.Identity()

+        self.norm_cond = nn.LayerNorm(cond_dim)
+        self.norm_mid_cond = nn.LayerNorm(cond_dim)
+
        # text encoding conditioning (optional)

        self.text_to_cond = None
@@ -1593,6 +1597,11 @@ class Unet(nn.Module):

        mid_c = c if not exists(text_tokens) else torch.cat((c, text_tokens), dim = -2)

+        # normalize conditioning tokens
+
+        c = self.norm_cond(c)
+        mid_c = self.norm_mid_cond(mid_c)
+
        # go through the layers of the unet, down and up

        hiddens = []
--- a/dalle2_pytorch/optimizer.py
+++ b/dalle2_pytorch/optimizer.py
@@ -7,16 +7,17 @@ def separate_weight_decayable_params(params):

 def get_optimizer(
    params,
-    lr = 3e-4,
+    lr = 2e-5,
    wd = 1e-2,
    betas = (0.9, 0.999),
+    eps = 1e-8,
    filter_by_requires_grad = False
 ):
    if filter_by_requires_grad:
        params = list(filter(lambda t: t.requires_grad, params))

    if wd == 0:
-        return Adam(params, lr = lr, betas = betas)
+        return Adam(params, lr = lr, betas = betas, eps = eps)

    params = set(params)
    wd_params, no_wd_params = separate_weight_decayable_params(params)
@@ -26,4 +27,4 @@ def get_optimizer(
        {'params': list(no_wd_params), 'weight_decay': 0},
    ]

-    return AdamW(param_groups, lr = lr, weight_decay = wd, betas = betas)
+    return AdamW(param_groups, lr = lr, weight_decay = wd, betas = betas, eps = eps)
--- a/dalle2_pytorch/train.py
+++ b/dalle2_pytorch/train.py
@@ -90,7 +90,7 @@ class EMA(nn.Module):
    def __init__(
        self,
        model,
-        beta = 0.99,
+        beta = 0.9999,
        update_after_step = 1000,
        update_every = 10,
    ):
@@ -147,6 +147,7 @@ class DiffusionPriorTrainer(nn.Module):
        use_ema = True,
        lr = 3e-4,
        wd = 1e-2,
+        eps = 1e-6,
        max_grad_norm = None,
        amp = False,
        **kwargs
@@ -173,6 +174,7 @@ class DiffusionPriorTrainer(nn.Module):
            diffusion_prior.parameters(),
            lr = lr,
            wd = wd,
+            eps = eps,
            **kwargs
        )

@@ -221,8 +223,9 @@ class DecoderTrainer(nn.Module):
        self,
        decoder,
        use_ema = True,
-        lr = 3e-4,
+        lr = 2e-5,
        wd = 1e-2,
+        eps = 1e-8,
        max_grad_norm = None,
        amp = False,
        **kwargs
@@ -247,13 +250,14 @@ class DecoderTrainer(nn.Module):
        # be able to finely customize learning rate, weight decay
        # per unet

-        lr, wd = map(partial(cast_tuple, length = self.num_unets), (lr, wd))
+        lr, wd, eps = map(partial(cast_tuple, length = self.num_unets), (lr, wd, eps))

-        for ind, (unet, unet_lr, unet_wd) in enumerate(zip(self.decoder.unets, lr, wd)):
+        for ind, (unet, unet_lr, unet_wd, unet_eps) in enumerate(zip(self.decoder.unets, lr, wd, eps)):
            optimizer = get_optimizer(
                unet.parameters(),
                lr = unet_lr,
                wd = unet_wd,
+                eps = unet_eps,
                **kwargs
            )

--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ setup(
      'dream = dalle2_pytorch.cli:dream'
    ],
  },
-  version = '0.2.19',
+  version = '0.2.23',
  license='MIT',
  description = 'DALL-E 2',
  author = 'Phil Wang',
Author	SHA1	Message	Date
Phil Wang	ff3474f05c	normalize conditioning tokens outside of cross attention blocks	2022-05-14 14:23:52 -07:00
Phil Wang	d5293f19f1	lineup with paper	2022-05-14 13:57:00 -07:00
Phil Wang	e697183849	be able to customize adam eps	2022-05-14 13:55:04 -07:00
Phil Wang	591d37e266	lower default initial learning rate to what Jonathan Ho had in his original repo	2022-05-14 13:22:43 -07:00