Compare commits

..
5 Commits
3 changed files with 96 additions and 73 deletions
+14 -1
View File
@@ -2,7 +2,9 @@
## Denoising Diffusion Probabilistic Model, in Pytorch ## Denoising Diffusion Probabilistic Model, in Pytorch
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution. This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>. Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution.
This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a> and then modified to use <a href="https://arxiv.org/abs/2201.03545">ConvNext</a> blocks instead of Resnets.
<img src="./sample.png" width="500px"><img> <img src="./sample.png" width="500px"><img>
@@ -97,3 +99,14 @@ Samples and model checkpoints will be logged to `./results` periodically
note = {under review} note = {under review}
} }
``` ```
```bibtex
@misc{liu2022convnet,
title = {A ConvNet for the 2020s},
author = {Zhuang Liu and Hanzi Mao and Chao-Yuan Wu and Christoph Feichtenhofer and Trevor Darrell and Saining Xie},
year = {2022},
eprint = {2201.03545},
archivePrefix = {arXiv},
primaryClass = {cs.CV}
}
```
@@ -91,69 +91,72 @@ class SinusoidalPosEmb(nn.Module):
emb = torch.cat((emb.sin(), emb.cos()), dim=-1) emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
return emb return emb
class Mish(nn.Module): def Upsample(dim):
def forward(self, x): return nn.ConvTranspose2d(dim, dim, 4, 2, 1)
return x * torch.tanh(F.softplus(x))
class Upsample(nn.Module): def Downsample(dim):
def __init__(self, dim): return nn.Conv2d(dim, dim, 4, 2, 1)
class LayerNorm(nn.Module):
def __init__(self, dim, eps = 1e-5):
super().__init__() super().__init__()
self.conv = nn.ConvTranspose2d(dim, dim, 4, 2, 1) self.eps = eps
self.g = nn.Parameter(torch.ones(1, dim, 1, 1))
self.b = nn.Parameter(torch.zeros(1, dim, 1, 1))
def forward(self, x): def forward(self, x):
return self.conv(x) var = torch.var(x, dim = 1, unbiased = False, keepdim = True)
mean = torch.mean(x, dim = 1, keepdim = True)
return (x - mean) / (var + self.eps).sqrt() * self.g + self.b
class Downsample(nn.Module): class PreNorm(nn.Module):
def __init__(self, dim): def __init__(self, dim, fn):
super().__init__()
self.conv = nn.Conv2d(dim, dim, 3, 2, 1)
def forward(self, x):
return self.conv(x)
class Rezero(nn.Module):
def __init__(self, fn):
super().__init__() super().__init__()
self.fn = fn self.fn = fn
self.g = nn.Parameter(torch.zeros(1)) self.norm = LayerNorm(dim)
def forward(self, x): def forward(self, x):
return self.fn(x) * self.g x = self.norm(x)
return self.fn(x)
# building block modules # building block modules
class Block(nn.Module): class ConvNextBlock(nn.Module):
def __init__(self, dim, dim_out, groups = 8): """ https://arxiv.org/abs/2201.03545 """
super().__init__()
self.block = nn.Sequential(
nn.Conv2d(dim, dim_out, 3, padding=1),
nn.GroupNorm(groups, dim_out),
Mish()
)
def forward(self, x):
return self.block(x)
class ResnetBlock(nn.Module): def __init__(self, dim, dim_out, *, time_emb_dim = None, mult = 2, norm = True):
def __init__(self, dim, dim_out, *, time_emb_dim, groups = 8):
super().__init__() super().__init__()
self.mlp = nn.Sequential( self.mlp = nn.Sequential(
Mish(), nn.GELU(),
nn.Linear(time_emb_dim, dim_out) nn.Linear(time_emb_dim, dim)
) if exists(time_emb_dim) else None
self.ds_conv = nn.Conv2d(dim, dim, 7, padding = 3, groups = dim)
self.net = nn.Sequential(
LayerNorm(dim) if norm else nn.Identity(),
nn.Conv2d(dim, dim_out * mult, 3, padding = 1),
nn.GELU(),
nn.Conv2d(dim_out * mult, dim_out, 3, padding = 1)
) )
self.block1 = Block(dim, dim_out)
self.block2 = Block(dim_out, dim_out)
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity() self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
def forward(self, x, time_emb): def forward(self, x, time_emb = None):
h = self.block1(x) h = self.ds_conv(x)
h += self.mlp(time_emb)[:, :, None, None]
h = self.block2(h) if exists(self.mlp):
assert exists(time_emb), 'time emb must be passed in'
condition = self.mlp(time_emb)
h = h + rearrange(condition, 'b c -> b c 1 1')
h = self.net(h)
return h + self.res_conv(x) return h + self.res_conv(x)
class LinearAttention(nn.Module): class LinearAttention(nn.Module):
def __init__(self, dim, heads = 4, dim_head = 32): def __init__(self, dim, heads = 4, dim_head = 32):
super().__init__() super().__init__()
self.scale = dim_head ** -0.5
self.heads = heads self.heads = heads
hidden_dim = dim_head * heads hidden_dim = dim_head * heads
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False) self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
@@ -161,12 +164,15 @@ class LinearAttention(nn.Module):
def forward(self, x): def forward(self, x):
b, c, h, w = x.shape b, c, h, w = x.shape
qkv = self.to_qkv(x) qkv = self.to_qkv(x).chunk(3, dim = 1)
q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads, qkv=3) q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
k = k.softmax(dim=-1) q = q * self.scale
context = torch.einsum('bhdn,bhen->bhde', k, v)
out = torch.einsum('bhde,bhdn->bhen', context, q) k = k.softmax(dim = -1)
out = rearrange(out, 'b heads c (h w) -> b (heads c) h w', heads=self.heads, h=h, w=w) context = torch.einsum('b h d n, b h e n -> b h d e', k, v)
out = torch.einsum('b h d e, b h d n -> b h e n', context, q)
out = rearrange(out, 'b h c (x y) -> b (h c) x y', h = self.heads, x = h, y = w)
return self.to_out(out) return self.to_out(out)
# model # model
@@ -177,8 +183,8 @@ class Unet(nn.Module):
dim, dim,
out_dim = None, out_dim = None,
dim_mults=(1, 2, 4, 8), dim_mults=(1, 2, 4, 8),
groups = 8, channels = 3,
channels = 3 with_time_emb = True
): ):
super().__init__() super().__init__()
self.channels = channels self.channels = channels
@@ -186,12 +192,17 @@ class Unet(nn.Module):
dims = [channels, *map(lambda m: dim * m, dim_mults)] dims = [channels, *map(lambda m: dim * m, dim_mults)]
in_out = list(zip(dims[:-1], dims[1:])) in_out = list(zip(dims[:-1], dims[1:]))
self.time_pos_emb = SinusoidalPosEmb(dim) if with_time_emb:
self.mlp = nn.Sequential( time_dim = dim
nn.Linear(dim, dim * 4), self.time_mlp = nn.Sequential(
Mish(), SinusoidalPosEmb(dim),
nn.Linear(dim * 4, dim) nn.Linear(dim, dim * 4),
) nn.GELU(),
nn.Linear(dim * 4, dim)
)
else:
time_dim = None
self.time_mlp = None
self.downs = nn.ModuleList([]) self.downs = nn.ModuleList([])
self.ups = nn.ModuleList([]) self.ups = nn.ModuleList([])
@@ -201,42 +212,41 @@ class Unet(nn.Module):
is_last = ind >= (num_resolutions - 1) is_last = ind >= (num_resolutions - 1)
self.downs.append(nn.ModuleList([ self.downs.append(nn.ModuleList([
ResnetBlock(dim_in, dim_out, time_emb_dim = dim), ConvNextBlock(dim_in, dim_out, time_emb_dim = time_dim, norm = ind != 0),
ResnetBlock(dim_out, dim_out, time_emb_dim = dim), ConvNextBlock(dim_out, dim_out, time_emb_dim = time_dim),
Residual(Rezero(LinearAttention(dim_out))), Residual(PreNorm(dim_out, LinearAttention(dim_out))),
Downsample(dim_out) if not is_last else nn.Identity() Downsample(dim_out) if not is_last else nn.Identity()
])) ]))
mid_dim = dims[-1] mid_dim = dims[-1]
self.mid_block1 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim) self.mid_block1 = ConvNextBlock(mid_dim, mid_dim, time_emb_dim = time_dim)
self.mid_attn = Residual(Rezero(LinearAttention(mid_dim))) self.mid_attn = Residual(PreNorm(mid_dim, LinearAttention(mid_dim)))
self.mid_block2 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim) self.mid_block2 = ConvNextBlock(mid_dim, mid_dim, time_emb_dim = time_dim)
for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])): for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])):
is_last = ind >= (num_resolutions - 1) is_last = ind >= (num_resolutions - 1)
self.ups.append(nn.ModuleList([ self.ups.append(nn.ModuleList([
ResnetBlock(dim_out * 2, dim_in, time_emb_dim = dim), ConvNextBlock(dim_out * 2, dim_in, time_emb_dim = time_dim),
ResnetBlock(dim_in, dim_in, time_emb_dim = dim), ConvNextBlock(dim_in, dim_in, time_emb_dim = time_dim),
Residual(Rezero(LinearAttention(dim_in))), Residual(PreNorm(dim_in, LinearAttention(dim_in))),
Upsample(dim_in) if not is_last else nn.Identity() Upsample(dim_in) if not is_last else nn.Identity()
])) ]))
out_dim = default(out_dim, channels) out_dim = default(out_dim, channels)
self.final_conv = nn.Sequential( self.final_conv = nn.Sequential(
Block(dim, dim), ConvNextBlock(dim, dim),
nn.Conv2d(dim, out_dim, 1) nn.Conv2d(dim, out_dim, 1)
) )
def forward(self, x, time): def forward(self, x, time):
t = self.time_pos_emb(time) t = self.time_mlp(time) if exists(self.time_mlp) else None
t = self.mlp(t)
h = [] h = []
for resnet, resnet2, attn, downsample in self.downs: for convnext, convnext2, attn, downsample in self.downs:
x = resnet(x, t) x = convnext(x, t)
x = resnet2(x, t) x = convnext2(x, t)
x = attn(x) x = attn(x)
h.append(x) h.append(x)
x = downsample(x) x = downsample(x)
@@ -245,10 +255,10 @@ class Unet(nn.Module):
x = self.mid_attn(x) x = self.mid_attn(x)
x = self.mid_block2(x, t) x = self.mid_block2(x, t)
for resnet, resnet2, attn, upsample in self.ups: for convnext, convnext2, attn, upsample in self.ups:
x = torch.cat((x, h.pop()), dim=1) x = torch.cat((x, h.pop()), dim=1)
x = resnet(x, t) x = convnext(x, t)
x = resnet2(x, t) x = convnext2(x, t)
x = attn(x) x = attn(x)
x = upsample(x) x = upsample(x)
+1 -1
View File
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
setup( setup(
name = 'denoising-diffusion-pytorch', name = 'denoising-diffusion-pytorch',
packages = find_packages(), packages = find_packages(),
version = '0.6.6', version = '0.7.1',
license='MIT', license='MIT',
description = 'Denoising Diffusion Probabilistic Models - Pytorch', description = 'Denoising Diffusion Probabilistic Models - Pytorch',
author = 'Phil Wang', author = 'Phil Wang',