mirror of
https://github.com/wassname/denoising-diffusion-pytorch.git
synced 2026-09-10 12:01:08 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
402b7c26df | ||
|
|
09613a40f3 | ||
|
|
c6966ae95a | ||
|
|
73591cf1ad | ||
|
|
989f0fcb8e | ||
|
|
84731bb03d | ||
|
|
c6ecca555b | ||
|
|
1f5c233072 | ||
|
|
de378158e5 | ||
|
|
e274fb305a | ||
|
|
f39b3b1d3f | ||
|
|
782c904d3b | ||
|
|
71953ebd22 | ||
|
|
0b8cdb4c8b | ||
|
|
e504e0e554 | ||
|
|
bd1e3b676e | ||
|
|
f4615599bc | ||
|
|
eb6e1b508e | ||
|
|
91cff45939 | ||
|
|
7b51e30da7 | ||
|
|
dadbf20154 | ||
|
|
7706bdfc6f | ||
|
|
183e5f3cc5 | ||
|
|
16c9ae7bb3 | ||
|
|
f5916111f8 | ||
|
|
ad9e303ff3 | ||
|
|
ae42f48f6a | ||
|
|
5989f4c77e | ||
|
|
2082046888 | ||
|
|
3c5b7e2d56 | ||
|
|
d4ce9f6c38 | ||
|
|
ff451f697e | ||
|
|
3d96532c60 | ||
|
|
ef2ca0b625 | ||
|
|
9f95a03c07 | ||
|
|
a4c68d3569 | ||
|
|
b33a48e342 | ||
|
|
8e5fb17063 | ||
|
|
4bf28914bc | ||
|
|
88f83d0ff2 | ||
|
|
26b5cab6c8 | ||
|
|
1307b3115d | ||
|
|
81fb2a0386 | ||
|
|
698227ae13 | ||
|
|
c479adf960 | ||
|
|
d70fb08f8a | ||
|
|
e700a7c6de | ||
|
|
9c758662a3 | ||
|
|
1f1e42e9f9 | ||
|
|
e1800c1a8d |
@@ -1,3 +1,6 @@
|
|||||||
|
# Generation results
|
||||||
|
results/
|
||||||
|
|
||||||
# Byte-compiled / optimized / DLL files
|
# Byte-compiled / optimized / DLL files
|
||||||
__pycache__/
|
__pycache__/
|
||||||
*.py[cod]
|
*.py[cod]
|
||||||
|
|||||||
@@ -1,8 +1,14 @@
|
|||||||
<img src="./denoising-diffusion.png" width="500px"></img>
|
<img src="./denoising-diffusion.png" width="500px"></img>
|
||||||
|
|
||||||
## Denoising Diffusion Probabilistic Model, in Pytorch (wip)
|
## Denoising Diffusion Probabilistic Model, in Pytorch
|
||||||
|
|
||||||
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution. This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>.
|
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution.
|
||||||
|
|
||||||
|
This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>
|
||||||
|
|
||||||
|
<img src="./sample.png" width="500px"><img>
|
||||||
|
|
||||||
|
[](https://badge.fury.io/py/denoising-diffusion-pytorch)
|
||||||
|
|
||||||
## Install
|
## Install
|
||||||
|
|
||||||
@@ -23,19 +29,18 @@ model = Unet(
|
|||||||
|
|
||||||
diffusion = GaussianDiffusion(
|
diffusion = GaussianDiffusion(
|
||||||
model,
|
model,
|
||||||
beta_start = 0.0001,
|
image_size = 128,
|
||||||
beta_end = 0.02,
|
timesteps = 1000, # number of steps
|
||||||
num_diffusion_timesteps = 1000, # number of steps
|
loss_type = 'l1' # L1 or L2
|
||||||
loss_type = 'l1' # L1 or L2 (wavegrad paper claims l1 is better?)
|
|
||||||
)
|
)
|
||||||
|
|
||||||
training_images = torch.randn(8, 3, 128, 128)
|
training_images = torch.randn(8, 3, 128, 128) # your images need to be normalized from a range of -1 to +1
|
||||||
loss = diffusion(training_images)
|
loss = diffusion(training_images)
|
||||||
loss.backward()
|
loss.backward()
|
||||||
# after a lot of training
|
# after a lot of training
|
||||||
|
|
||||||
sampled_images = diffusion.p_sample_loop((1, 3, 128, 128))
|
sampled_images = diffusion.sample(batch_size = 4)
|
||||||
sampled_images.shape # (1, 3, 128, 128)
|
sampled_images.shape # (4, 3, 128, 128)
|
||||||
```
|
```
|
||||||
|
|
||||||
Or, if you simply want to pass in a folder name and the desired image dimensions, you can use the `Trainer` class to easily train a model.
|
Or, if you simply want to pass in a folder name and the desired image dimensions, you can use the `Trainer` class to easily train a model.
|
||||||
@@ -50,36 +55,56 @@ model = Unet(
|
|||||||
|
|
||||||
diffusion = GaussianDiffusion(
|
diffusion = GaussianDiffusion(
|
||||||
model,
|
model,
|
||||||
beta_start = 0.0001,
|
image_size = 128,
|
||||||
beta_end = 0.02,
|
timesteps = 1000, # number of steps
|
||||||
num_diffusion_timesteps = 1000, # number of steps
|
loss_type = 'l1' # L1 or L2
|
||||||
loss_type = 'l1' # L1 or L2
|
|
||||||
).cuda()
|
).cuda()
|
||||||
|
|
||||||
trainer = Trainer(
|
trainer = Trainer(
|
||||||
diffusion,
|
diffusion,
|
||||||
'path/to/your/images',
|
'path/to/your/images',
|
||||||
image_size = 128,
|
|
||||||
train_batch_size = 32,
|
train_batch_size = 32,
|
||||||
train_lr = 3e-4,
|
train_lr = 2e-5,
|
||||||
train_num_steps = 100000,
|
train_num_steps = 700000, # total training steps
|
||||||
gradient_accumulate_every = 1
|
gradient_accumulate_every = 2, # gradient accumulation steps
|
||||||
|
ema_decay = 0.995, # exponential moving average decay
|
||||||
|
amp = True # turn on mixed precision
|
||||||
)
|
)
|
||||||
|
|
||||||
trainer.train()
|
trainer.train()
|
||||||
```
|
```
|
||||||
|
|
||||||
Todo: Command line tool for one-line training
|
Samples and model checkpoints will be logged to `./results` periodically
|
||||||
|
|
||||||
## Citations
|
## Citations
|
||||||
|
|
||||||
```bibtex
|
```bibtex
|
||||||
@misc{ho2020denoising,
|
@inproceedings{NEURIPS2020_4c5bcfec,
|
||||||
title={Denoising Diffusion Probabilistic Models},
|
author = {Ho, Jonathan and Jain, Ajay and Abbeel, Pieter},
|
||||||
author={Jonathan Ho and Ajay Jain and Pieter Abbeel},
|
booktitle = {Advances in Neural Information Processing Systems},
|
||||||
year={2020},
|
editor = {H. Larochelle and M. Ranzato and R. Hadsell and M.F. Balcan and H. Lin},
|
||||||
eprint={2006.11239},
|
pages = {6840--6851},
|
||||||
archivePrefix={arXiv},
|
publisher = {Curran Associates, Inc.},
|
||||||
primaryClass={cs.LG}
|
title = {Denoising Diffusion Probabilistic Models},
|
||||||
|
url = {https://proceedings.neurips.cc/paper/2020/file/4c5bcfec8584af0d967f1ab10179ca4b-Paper.pdf},
|
||||||
|
volume = {33},
|
||||||
|
year = {2020}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
```bibtex
|
||||||
|
@InProceedings{pmlr-v139-nichol21a,
|
||||||
|
title = {Improved Denoising Diffusion Probabilistic Models},
|
||||||
|
author = {Nichol, Alexander Quinn and Dhariwal, Prafulla},
|
||||||
|
booktitle = {Proceedings of the 38th International Conference on Machine Learning},
|
||||||
|
pages = {8162--8171},
|
||||||
|
year = {2021},
|
||||||
|
editor = {Meila, Marina and Zhang, Tong},
|
||||||
|
volume = {139},
|
||||||
|
series = {Proceedings of Machine Learning Research},
|
||||||
|
month = {18--24 Jul},
|
||||||
|
publisher = {PMLR},
|
||||||
|
pdf = {http://proceedings.mlr.press/v139/nichol21a/nichol21a.pdf},
|
||||||
|
url = {https://proceedings.mlr.press/v139/nichol21a.html},
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import math
|
import math
|
||||||
|
import copy
|
||||||
import torch
|
import torch
|
||||||
from torch import nn, einsum
|
from torch import nn, einsum
|
||||||
import torch.nn.functional as F
|
import torch.nn.functional as F
|
||||||
@@ -6,20 +7,16 @@ from inspect import isfunction
|
|||||||
from functools import partial
|
from functools import partial
|
||||||
|
|
||||||
from torch.utils import data
|
from torch.utils import data
|
||||||
|
from torch.cuda.amp import autocast, GradScaler
|
||||||
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from torch.optim import Adam
|
from torch.optim import Adam
|
||||||
from torchvision import transforms, utils
|
from torchvision import transforms, utils
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
from einops import rearrange
|
from einops import rearrange
|
||||||
|
|
||||||
# constants
|
|
||||||
|
|
||||||
SAVE_AND_SAMPLE_EVERY = 1000
|
|
||||||
EXTS = ['jpg', 'png']
|
|
||||||
|
|
||||||
# helpers functions
|
# helpers functions
|
||||||
|
|
||||||
def exists(x):
|
def exists(x):
|
||||||
@@ -35,8 +32,31 @@ def cycle(dl):
|
|||||||
for data in dl:
|
for data in dl:
|
||||||
yield data
|
yield data
|
||||||
|
|
||||||
|
def num_to_groups(num, divisor):
|
||||||
|
groups = num // divisor
|
||||||
|
remainder = num % divisor
|
||||||
|
arr = [divisor] * groups
|
||||||
|
if remainder > 0:
|
||||||
|
arr.append(remainder)
|
||||||
|
return arr
|
||||||
|
|
||||||
# small helper modules
|
# small helper modules
|
||||||
|
|
||||||
|
class EMA():
|
||||||
|
def __init__(self, beta):
|
||||||
|
super().__init__()
|
||||||
|
self.beta = beta
|
||||||
|
|
||||||
|
def update_model_average(self, ma_model, current_model):
|
||||||
|
for current_params, ma_params in zip(current_model.parameters(), ma_model.parameters()):
|
||||||
|
old_weight, up_weight = ma_params.data, current_params.data
|
||||||
|
ma_params.data = self.update_average(old_weight, up_weight)
|
||||||
|
|
||||||
|
def update_average(self, old, new):
|
||||||
|
if old is None:
|
||||||
|
return new
|
||||||
|
return old * self.beta + (1 - self.beta) * new
|
||||||
|
|
||||||
class Residual(nn.Module):
|
class Residual(nn.Module):
|
||||||
def __init__(self, fn):
|
def __init__(self, fn):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
@@ -59,98 +79,162 @@ class SinusoidalPosEmb(nn.Module):
|
|||||||
emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
|
emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
|
||||||
return emb
|
return emb
|
||||||
|
|
||||||
class Mish(nn.Module):
|
def Upsample(dim):
|
||||||
def forward(self, x):
|
return nn.ConvTranspose2d(dim, dim, 4, 2, 1)
|
||||||
return x * torch.tanh(F.softplus(x))
|
|
||||||
|
|
||||||
class Upsample(nn.Module):
|
def Downsample(dim):
|
||||||
def __init__(self, dim):
|
return nn.Conv2d(dim, dim, 4, 2, 1)
|
||||||
|
|
||||||
|
class LayerNorm(nn.Module):
|
||||||
|
def __init__(self, dim, eps = 1e-5):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.conv = nn.ConvTranspose2d(dim, dim, 4, 2, 1)
|
self.eps = eps
|
||||||
|
self.g = nn.Parameter(torch.ones(1, dim, 1, 1))
|
||||||
|
self.b = nn.Parameter(torch.zeros(1, dim, 1, 1))
|
||||||
|
|
||||||
def forward(self, x):
|
def forward(self, x):
|
||||||
return self.conv(x)
|
var = torch.var(x, dim = 1, unbiased = False, keepdim = True)
|
||||||
|
mean = torch.mean(x, dim = 1, keepdim = True)
|
||||||
|
return (x - mean) / (var + self.eps).sqrt() * self.g + self.b
|
||||||
|
|
||||||
class Downsample(nn.Module):
|
class PreNorm(nn.Module):
|
||||||
def __init__(self, dim):
|
def __init__(self, dim, fn):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.conv = nn.Conv2d(dim, dim, 3, 2, 1)
|
self.fn = fn
|
||||||
|
self.norm = LayerNorm(dim)
|
||||||
|
|
||||||
def forward(self, x):
|
def forward(self, x):
|
||||||
return self.conv(x)
|
x = self.norm(x)
|
||||||
|
return self.fn(x)
|
||||||
class Rezero(nn.Module):
|
|
||||||
def __init__(self, dim):
|
|
||||||
super().__init__()
|
|
||||||
self.g = nn.Parameter(torch.zeros(1))
|
|
||||||
|
|
||||||
def forward(self, x):
|
|
||||||
return x * self.g
|
|
||||||
|
|
||||||
# building block modules
|
# building block modules
|
||||||
|
|
||||||
class Block(nn.Module):
|
class Block(nn.Module):
|
||||||
def __init__(self, dim, dim_out, groups = 32):
|
def __init__(self, dim, dim_out, groups = 8):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.block = nn.Sequential(
|
self.block = nn.Sequential(
|
||||||
nn.Conv2d(dim, dim_out, 3, padding=1),
|
nn.Conv2d(dim, dim_out, 3, padding = 1),
|
||||||
nn.GroupNorm(groups, dim_out),
|
nn.GroupNorm(groups, dim_out),
|
||||||
Mish()
|
nn.SiLU()
|
||||||
)
|
)
|
||||||
def forward(self, x):
|
def forward(self, x):
|
||||||
return self.block(x)
|
return self.block(x)
|
||||||
|
|
||||||
class ResnetBlock(nn.Module):
|
class ResnetBlock(nn.Module):
|
||||||
def __init__(self, dim, dim_out, *, time_emb_dim, groups = 32):
|
def __init__(self, dim, dim_out, *, time_emb_dim = None, groups = 8):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.mlp = nn.Sequential(
|
self.mlp = nn.Sequential(
|
||||||
Mish(),
|
nn.SiLU(),
|
||||||
nn.Linear(time_emb_dim, dim_out)
|
nn.Linear(time_emb_dim, dim_out)
|
||||||
)
|
) if exists(time_emb_dim) else None
|
||||||
|
|
||||||
self.block1 = Block(dim, dim_out)
|
self.block1 = Block(dim, dim_out, groups = groups)
|
||||||
self.block2 = Block(dim_out, dim_out)
|
self.block2 = Block(dim_out, dim_out, groups = groups)
|
||||||
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
|
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
|
||||||
|
|
||||||
def forward(self, x, time_emb):
|
def forward(self, x, time_emb = None):
|
||||||
h = self.block1(x)
|
h = self.block1(x)
|
||||||
h += self.mlp(time_emb)[:, :, None, None]
|
|
||||||
|
if exists(self.mlp) and exists(time_emb):
|
||||||
|
time_emb = self.mlp(time_emb)
|
||||||
|
h = rearrange(time_emb, 'b c -> b c 1 1') + h
|
||||||
|
|
||||||
h = self.block2(h)
|
h = self.block2(h)
|
||||||
return h + self.res_conv(x)
|
return h + self.res_conv(x)
|
||||||
|
|
||||||
class LinearAttention(nn.Module):
|
class LinearAttention(nn.Module):
|
||||||
def __init__(self, dim, heads = 8, dim_head = 32):
|
def __init__(self, dim, heads = 4, dim_head = 32):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.scale = dim_head ** -0.5
|
||||||
self.heads = heads
|
self.heads = heads
|
||||||
hidden_dim = dim_head * heads
|
hidden_dim = dim_head * heads
|
||||||
self.to_qkv = nn.Conv2d(dim, hidden_dim, 1, bias = False)
|
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
|
||||||
|
|
||||||
|
self.to_out = nn.Sequential(
|
||||||
|
nn.Conv2d(hidden_dim, dim, 1),
|
||||||
|
LayerNorm(dim)
|
||||||
|
)
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
b, c, h, w = x.shape
|
||||||
|
qkv = self.to_qkv(x).chunk(3, dim = 1)
|
||||||
|
q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
|
||||||
|
|
||||||
|
q = q.softmax(dim = -2)
|
||||||
|
k = k.softmax(dim = -1)
|
||||||
|
|
||||||
|
q = q * self.scale
|
||||||
|
context = torch.einsum('b h d n, b h e n -> b h d e', k, v)
|
||||||
|
|
||||||
|
out = torch.einsum('b h d e, b h d n -> b h e n', context, q)
|
||||||
|
out = rearrange(out, 'b h c (x y) -> b (h c) x y', h = self.heads, x = h, y = w)
|
||||||
|
return self.to_out(out)
|
||||||
|
|
||||||
|
class Attention(nn.Module):
|
||||||
|
def __init__(self, dim, heads = 4, dim_head = 32):
|
||||||
|
super().__init__()
|
||||||
|
self.scale = dim_head ** -0.5
|
||||||
|
self.heads = heads
|
||||||
|
hidden_dim = dim_head * heads
|
||||||
|
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
|
||||||
self.to_out = nn.Conv2d(hidden_dim, dim, 1)
|
self.to_out = nn.Conv2d(hidden_dim, dim, 1)
|
||||||
|
|
||||||
def forward(self, x):
|
def forward(self, x):
|
||||||
b, c, h, w = x.shape
|
b, c, h, w = x.shape
|
||||||
qkv = self.to_qkv(x)
|
qkv = self.to_qkv(x).chunk(3, dim = 1)
|
||||||
q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads)
|
q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
|
||||||
q = q.softmax(dim=-2)
|
q = q * self.scale
|
||||||
k = k.softmax(dim=-1)
|
|
||||||
context = torch.einsum('bhdn,bhen->bhde', k, v)
|
sim = einsum('b h d i, b h d j -> b h i j', q, k)
|
||||||
out = torch.einsum('bhde,bhdn->bhen', context, q)
|
sim = sim - sim.amax(dim = -1, keepdim = True).detach()
|
||||||
out = rearrange(out, 'b heads c (h w) -> b (heads c) h w', heads=self.heads, h=h, w=w)
|
attn = sim.softmax(dim = -1)
|
||||||
|
|
||||||
|
out = einsum('b h i j, b h d j -> b h i d', attn, v)
|
||||||
|
out = rearrange(out, 'b h (x y) d -> b (h d) x y', x = h, y = w)
|
||||||
return self.to_out(out)
|
return self.to_out(out)
|
||||||
|
|
||||||
# model
|
# model
|
||||||
|
|
||||||
class Unet(nn.Module):
|
class Unet(nn.Module):
|
||||||
def __init__(self, dim, out_dim = None, dim_mults=(1, 2, 4, 8), groups = 32):
|
def __init__(
|
||||||
|
self,
|
||||||
|
dim,
|
||||||
|
init_dim = None,
|
||||||
|
out_dim = None,
|
||||||
|
dim_mults=(1, 2, 4, 8),
|
||||||
|
channels = 3,
|
||||||
|
with_time_emb = True,
|
||||||
|
resnet_block_groups = 8
|
||||||
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
dims = [3, *map(lambda m: dim * m, dim_mults)]
|
|
||||||
|
# determine dimensions
|
||||||
|
|
||||||
|
self.channels = channels
|
||||||
|
|
||||||
|
init_dim = default(init_dim, dim // 3 * 2)
|
||||||
|
self.init_conv = nn.Conv2d(channels, init_dim, 7, padding = 3)
|
||||||
|
|
||||||
|
dims = [init_dim, *map(lambda m: dim * m, dim_mults)]
|
||||||
in_out = list(zip(dims[:-1], dims[1:]))
|
in_out = list(zip(dims[:-1], dims[1:]))
|
||||||
|
|
||||||
self.time_pos_emb = SinusoidalPosEmb(dim)
|
block_klass = partial(ResnetBlock, groups = resnet_block_groups)
|
||||||
self.mlp = nn.Sequential(
|
|
||||||
nn.Linear(dim, dim * 4),
|
# time embeddings
|
||||||
Mish(),
|
|
||||||
nn.Linear(dim * 4, dim)
|
if with_time_emb:
|
||||||
)
|
time_dim = dim * 4
|
||||||
|
self.time_mlp = nn.Sequential(
|
||||||
|
SinusoidalPosEmb(dim),
|
||||||
|
nn.Linear(dim, time_dim),
|
||||||
|
nn.GELU(),
|
||||||
|
nn.Linear(time_dim, time_dim)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
time_dim = None
|
||||||
|
self.time_mlp = None
|
||||||
|
|
||||||
|
# layers
|
||||||
|
|
||||||
self.downs = nn.ModuleList([])
|
self.downs = nn.ModuleList([])
|
||||||
self.ups = nn.ModuleList([])
|
self.ups = nn.ModuleList([])
|
||||||
@@ -160,39 +244,43 @@ class Unet(nn.Module):
|
|||||||
is_last = ind >= (num_resolutions - 1)
|
is_last = ind >= (num_resolutions - 1)
|
||||||
|
|
||||||
self.downs.append(nn.ModuleList([
|
self.downs.append(nn.ModuleList([
|
||||||
ResnetBlock(dim_in, dim_out, time_emb_dim = dim),
|
block_klass(dim_in, dim_out, time_emb_dim = time_dim),
|
||||||
Residual(Rezero(LinearAttention(dim_out))),
|
block_klass(dim_out, dim_out, time_emb_dim = time_dim),
|
||||||
|
Residual(PreNorm(dim_out, LinearAttention(dim_out))),
|
||||||
Downsample(dim_out) if not is_last else nn.Identity()
|
Downsample(dim_out) if not is_last else nn.Identity()
|
||||||
]))
|
]))
|
||||||
|
|
||||||
mid_dim = dims[-1]
|
mid_dim = dims[-1]
|
||||||
self.mid_block1 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim)
|
self.mid_block1 = block_klass(mid_dim, mid_dim, time_emb_dim = time_dim)
|
||||||
self.mid_attn = Residual(Rezero(LinearAttention(mid_dim)))
|
self.mid_attn = Residual(PreNorm(mid_dim, Attention(mid_dim)))
|
||||||
self.mid_block2 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim)
|
self.mid_block2 = block_klass(mid_dim, mid_dim, time_emb_dim = time_dim)
|
||||||
|
|
||||||
for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])):
|
for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])):
|
||||||
is_last = ind >= (num_resolutions - 1)
|
is_last = ind >= (num_resolutions - 1)
|
||||||
|
|
||||||
self.ups.append(nn.ModuleList([
|
self.ups.append(nn.ModuleList([
|
||||||
ResnetBlock(dim_out * 2, dim_in, time_emb_dim = dim),
|
block_klass(dim_out * 2, dim_in, time_emb_dim = time_dim),
|
||||||
Residual(Rezero(LinearAttention(dim_in))),
|
block_klass(dim_in, dim_in, time_emb_dim = time_dim),
|
||||||
|
Residual(PreNorm(dim_in, LinearAttention(dim_in))),
|
||||||
Upsample(dim_in) if not is_last else nn.Identity()
|
Upsample(dim_in) if not is_last else nn.Identity()
|
||||||
]))
|
]))
|
||||||
|
|
||||||
out_dim = default(out_dim, 3)
|
out_dim = default(out_dim, channels)
|
||||||
self.final_conv = nn.Sequential(
|
self.final_conv = nn.Sequential(
|
||||||
Block(dim, dim),
|
block_klass(dim, dim),
|
||||||
nn.Conv2d(dim, out_dim, 1)
|
nn.Conv2d(dim, out_dim, 1)
|
||||||
)
|
)
|
||||||
|
|
||||||
def forward(self, x, time):
|
def forward(self, x, time):
|
||||||
t = self.time_pos_emb(time)
|
x = self.init_conv(x)
|
||||||
t = self.mlp(t)
|
|
||||||
|
t = self.time_mlp(time) if exists(self.time_mlp) else None
|
||||||
|
|
||||||
h = []
|
h = []
|
||||||
|
|
||||||
for resnet, attn, downsample in self.downs:
|
for block1, block2, attn, downsample in self.downs:
|
||||||
x = resnet(x, t)
|
x = block1(x, t)
|
||||||
|
x = block2(x, t)
|
||||||
x = attn(x)
|
x = attn(x)
|
||||||
h.append(x)
|
h.append(x)
|
||||||
x = downsample(x)
|
x = downsample(x)
|
||||||
@@ -201,9 +289,10 @@ class Unet(nn.Module):
|
|||||||
x = self.mid_attn(x)
|
x = self.mid_attn(x)
|
||||||
x = self.mid_block2(x, t)
|
x = self.mid_block2(x, t)
|
||||||
|
|
||||||
for resnet, attn, upsample in self.ups:
|
for block1, block2, attn, upsample in self.ups:
|
||||||
x = torch.cat((x, h.pop()), dim=1)
|
x = torch.cat((x, h.pop()), dim=1)
|
||||||
x = resnet(x, t)
|
x = block1(x, t)
|
||||||
|
x = block2(x, t)
|
||||||
x = attn(x)
|
x = attn(x)
|
||||||
x = upsample(x)
|
x = upsample(x)
|
||||||
|
|
||||||
@@ -221,43 +310,72 @@ def noise_like(shape, device, repeat=False):
|
|||||||
noise = lambda: torch.randn(shape, device=device)
|
noise = lambda: torch.randn(shape, device=device)
|
||||||
return repeat_noise() if repeat else noise()
|
return repeat_noise() if repeat else noise()
|
||||||
|
|
||||||
|
def cosine_beta_schedule(timesteps, s = 0.008):
|
||||||
|
"""
|
||||||
|
cosine schedule
|
||||||
|
as proposed in https://openreview.net/forum?id=-NEXDKk8gZ
|
||||||
|
"""
|
||||||
|
steps = timesteps + 1
|
||||||
|
x = torch.linspace(0, timesteps, steps, dtype = torch.float64)
|
||||||
|
alphas_cumprod = torch.cos(((x / timesteps) + s) / (1 + s) * torch.pi * 0.5) ** 2
|
||||||
|
alphas_cumprod = alphas_cumprod / alphas_cumprod[0]
|
||||||
|
betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1])
|
||||||
|
return torch.clip(betas, 0, 0.9999)
|
||||||
|
|
||||||
class GaussianDiffusion(nn.Module):
|
class GaussianDiffusion(nn.Module):
|
||||||
def __init__(self, denoise_fn, beta_start=0.0001, beta_end=0.02, num_diffusion_timesteps=1000, loss_type='l1'):
|
def __init__(
|
||||||
|
self,
|
||||||
|
denoise_fn,
|
||||||
|
*,
|
||||||
|
image_size,
|
||||||
|
channels = 3,
|
||||||
|
timesteps = 1000,
|
||||||
|
loss_type = 'l1'
|
||||||
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.channels = channels
|
||||||
|
self.image_size = image_size
|
||||||
self.denoise_fn = denoise_fn
|
self.denoise_fn = denoise_fn
|
||||||
|
|
||||||
self.np_betas = betas = np.linspace(beta_start, beta_end, num_diffusion_timesteps).astype(np.float64)
|
betas = cosine_beta_schedule(timesteps)
|
||||||
|
|
||||||
|
alphas = 1. - betas
|
||||||
|
alphas_cumprod = torch.cumprod(alphas, axis=0)
|
||||||
|
alphas_cumprod_prev = F.pad(alphas_cumprod[:-1], (1, 0), value = 1.)
|
||||||
|
|
||||||
timesteps, = betas.shape
|
timesteps, = betas.shape
|
||||||
self.num_timesteps = int(timesteps)
|
self.num_timesteps = int(timesteps)
|
||||||
self.loss_type = loss_type
|
self.loss_type = loss_type
|
||||||
|
|
||||||
alphas = 1. - betas
|
# helper function to register buffer from float64 to float32
|
||||||
alphas_cumprod = np.cumprod(alphas, axis=0)
|
|
||||||
alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1])
|
|
||||||
|
|
||||||
to_torch = partial(torch.tensor, dtype=torch.float32)
|
register_buffer = lambda name, val: self.register_buffer(name, val.to(torch.float32))
|
||||||
|
|
||||||
self.register_buffer('betas', to_torch(betas))
|
register_buffer('betas', betas)
|
||||||
self.register_buffer('alphas_cumprod', to_torch(alphas_cumprod))
|
register_buffer('alphas_cumprod', alphas_cumprod)
|
||||||
self.register_buffer('alphas_cumprod_prev', to_torch(alphas_cumprod_prev))
|
register_buffer('alphas_cumprod_prev', alphas_cumprod_prev)
|
||||||
|
|
||||||
# calculations for diffusion q(x_t | x_{t-1}) and others
|
# calculations for diffusion q(x_t | x_{t-1}) and others
|
||||||
self.register_buffer('sqrt_alphas_cumprod', to_torch(np.sqrt(alphas_cumprod)))
|
|
||||||
self.register_buffer('sqrt_one_minus_alphas_cumprod', to_torch(np.sqrt(1. - alphas_cumprod)))
|
register_buffer('sqrt_alphas_cumprod', torch.sqrt(alphas_cumprod))
|
||||||
self.register_buffer('log_one_minus_alphas_cumprod', to_torch(np.log(1. - alphas_cumprod)))
|
register_buffer('sqrt_one_minus_alphas_cumprod', torch.sqrt(1. - alphas_cumprod))
|
||||||
self.register_buffer('sqrt_recip_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod)))
|
register_buffer('log_one_minus_alphas_cumprod', torch.log(1. - alphas_cumprod))
|
||||||
self.register_buffer('sqrt_recipm1_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod - 1)))
|
register_buffer('sqrt_recip_alphas_cumprod', torch.sqrt(1. / alphas_cumprod))
|
||||||
|
register_buffer('sqrt_recipm1_alphas_cumprod', torch.sqrt(1. / alphas_cumprod - 1))
|
||||||
|
|
||||||
# calculations for posterior q(x_{t-1} | x_t, x_0)
|
# calculations for posterior q(x_{t-1} | x_t, x_0)
|
||||||
|
|
||||||
posterior_variance = betas * (1. - alphas_cumprod_prev) / (1. - alphas_cumprod)
|
posterior_variance = betas * (1. - alphas_cumprod_prev) / (1. - alphas_cumprod)
|
||||||
|
|
||||||
# above: equal to 1. / (1. / (1. - alpha_cumprod_tm1) + alpha_t / beta_t)
|
# above: equal to 1. / (1. / (1. - alpha_cumprod_tm1) + alpha_t / beta_t)
|
||||||
self.register_buffer('posterior_variance', to_torch(posterior_variance))
|
|
||||||
|
register_buffer('posterior_variance', posterior_variance)
|
||||||
|
|
||||||
# below: log calculation clipped because the posterior variance is 0 at the beginning of the diffusion chain
|
# below: log calculation clipped because the posterior variance is 0 at the beginning of the diffusion chain
|
||||||
self.register_buffer('posterior_log_variance_clipped', to_torch(np.log(np.maximum(posterior_variance, 1e-20))))
|
|
||||||
self.register_buffer('posterior_mean_coef1', to_torch(
|
register_buffer('posterior_log_variance_clipped', torch.log(posterior_variance.clamp(min =1e-20)))
|
||||||
betas * np.sqrt(alphas_cumprod_prev) / (1. - alphas_cumprod)))
|
register_buffer('posterior_mean_coef1', betas * torch.sqrt(alphas_cumprod_prev) / (1. - alphas_cumprod))
|
||||||
self.register_buffer('posterior_mean_coef2', to_torch(
|
register_buffer('posterior_mean_coef2', (1. - alphas_cumprod_prev) * torch.sqrt(alphas) / (1. - alphas_cumprod))
|
||||||
(1. - alphas_cumprod_prev) * np.sqrt(alphas) / (1. - alphas_cumprod)))
|
|
||||||
|
|
||||||
def q_mean_variance(self, x_start, t):
|
def q_mean_variance(self, x_start, t):
|
||||||
mean = extract(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
|
mean = extract(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
|
||||||
@@ -309,6 +427,28 @@ class GaussianDiffusion(nn.Module):
|
|||||||
img = self.p_sample(img, torch.full((b,), i, device=device, dtype=torch.long))
|
img = self.p_sample(img, torch.full((b,), i, device=device, dtype=torch.long))
|
||||||
return img
|
return img
|
||||||
|
|
||||||
|
@torch.no_grad()
|
||||||
|
def sample(self, batch_size = 16):
|
||||||
|
image_size = self.image_size
|
||||||
|
channels = self.channels
|
||||||
|
return self.p_sample_loop((batch_size, channels, image_size, image_size))
|
||||||
|
|
||||||
|
@torch.no_grad()
|
||||||
|
def interpolate(self, x1, x2, t = None, lam = 0.5):
|
||||||
|
b, *_, device = *x1.shape, x1.device
|
||||||
|
t = default(t, self.num_timesteps - 1)
|
||||||
|
|
||||||
|
assert x1.shape == x2.shape
|
||||||
|
|
||||||
|
t_batched = torch.stack([torch.tensor(t, device=device)] * b)
|
||||||
|
xt1, xt2 = map(lambda x: self.q_sample(x, t=t_batched), (x1, x2))
|
||||||
|
|
||||||
|
img = (1 - lam) * xt1 + lam * xt2
|
||||||
|
for i in tqdm(reversed(range(0, t)), desc='interpolation sample time step', total=t):
|
||||||
|
img = self.p_sample(img, torch.full((b,), i, device=device, dtype=torch.long))
|
||||||
|
|
||||||
|
return img
|
||||||
|
|
||||||
def q_sample(self, x_start, t, noise=None):
|
def q_sample(self, x_start, t, noise=None):
|
||||||
noise = default(noise, lambda: torch.randn_like(x_start))
|
noise = default(noise, lambda: torch.randn_like(x_start))
|
||||||
|
|
||||||
@@ -334,24 +474,26 @@ class GaussianDiffusion(nn.Module):
|
|||||||
return loss
|
return loss
|
||||||
|
|
||||||
def forward(self, x, *args, **kwargs):
|
def forward(self, x, *args, **kwargs):
|
||||||
b, *_, device = *x.shape, x.device
|
b, c, h, w, device, img_size, = *x.shape, x.device, self.image_size
|
||||||
|
assert h == img_size and w == img_size, f'height and width of image must be {img_size}'
|
||||||
t = torch.randint(0, self.num_timesteps, (b,), device=device).long()
|
t = torch.randint(0, self.num_timesteps, (b,), device=device).long()
|
||||||
return self.p_losses(x, t, *args, **kwargs)
|
return self.p_losses(x, t, *args, **kwargs)
|
||||||
|
|
||||||
# dataset classes
|
# dataset classes
|
||||||
|
|
||||||
class Dataset(data.Dataset):
|
class Dataset(data.Dataset):
|
||||||
def __init__(self, folder, image_size):
|
def __init__(self, folder, image_size, exts = ['jpg', 'jpeg', 'png']):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.folder = folder
|
self.folder = folder
|
||||||
self.image_size = image_size
|
self.image_size = image_size
|
||||||
self.paths = [p for ext in EXTS for p in Path(f'{folder}').glob(f'**/*.{ext}')]
|
self.paths = [p for ext in exts for p in Path(f'{folder}').glob(f'**/*.{ext}')]
|
||||||
|
|
||||||
self.transform = transforms.Compose([
|
self.transform = transforms.Compose([
|
||||||
transforms.Resize(image_size),
|
transforms.Resize(image_size),
|
||||||
transforms.RandomHorizontalFlip(),
|
transforms.RandomHorizontalFlip(),
|
||||||
transforms.CenterCrop(image_size),
|
transforms.CenterCrop(image_size),
|
||||||
transforms.ToTensor()
|
transforms.ToTensor(),
|
||||||
|
transforms.Lambda(lambda t: (t * 2) - 1)
|
||||||
])
|
])
|
||||||
|
|
||||||
def __len__(self):
|
def __len__(self):
|
||||||
@@ -370,15 +512,29 @@ class Trainer(object):
|
|||||||
diffusion_model,
|
diffusion_model,
|
||||||
folder,
|
folder,
|
||||||
*,
|
*,
|
||||||
|
ema_decay = 0.995,
|
||||||
image_size = 128,
|
image_size = 128,
|
||||||
train_batch_size = 32,
|
train_batch_size = 32,
|
||||||
train_lr = 3e-4,
|
train_lr = 2e-5,
|
||||||
train_num_steps = 100000,
|
train_num_steps = 100000,
|
||||||
gradient_accumulate_every = 1
|
gradient_accumulate_every = 2,
|
||||||
|
amp = False,
|
||||||
|
step_start_ema = 2000,
|
||||||
|
update_ema_every = 10,
|
||||||
|
save_and_sample_every = 1000,
|
||||||
|
results_folder = './results'
|
||||||
):
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.model = diffusion_model
|
self.model = diffusion_model
|
||||||
self.image_size = image_size
|
self.ema = EMA(ema_decay)
|
||||||
|
self.ema_model = copy.deepcopy(self.model)
|
||||||
|
self.update_ema_every = update_ema_every
|
||||||
|
|
||||||
|
self.step_start_ema = step_start_ema
|
||||||
|
self.save_and_sample_every = save_and_sample_every
|
||||||
|
|
||||||
|
self.batch_size = train_batch_size
|
||||||
|
self.image_size = diffusion_model.image_size
|
||||||
self.gradient_accumulate_every = gradient_accumulate_every
|
self.gradient_accumulate_every = gradient_accumulate_every
|
||||||
self.train_num_steps = train_num_steps
|
self.train_num_steps = train_num_steps
|
||||||
|
|
||||||
@@ -386,25 +542,69 @@ class Trainer(object):
|
|||||||
self.dl = cycle(data.DataLoader(self.ds, batch_size = train_batch_size, shuffle=True, pin_memory=True))
|
self.dl = cycle(data.DataLoader(self.ds, batch_size = train_batch_size, shuffle=True, pin_memory=True))
|
||||||
self.opt = Adam(diffusion_model.parameters(), lr=train_lr)
|
self.opt = Adam(diffusion_model.parameters(), lr=train_lr)
|
||||||
|
|
||||||
def train(self):
|
self.step = 0
|
||||||
ind = 0
|
|
||||||
|
|
||||||
while ind < self.train_num_steps:
|
self.amp = amp
|
||||||
|
self.scaler = GradScaler(enabled = amp)
|
||||||
|
|
||||||
|
self.results_folder = Path(results_folder)
|
||||||
|
self.results_folder.mkdir(exist_ok = True)
|
||||||
|
|
||||||
|
self.reset_parameters()
|
||||||
|
|
||||||
|
def reset_parameters(self):
|
||||||
|
self.ema_model.load_state_dict(self.model.state_dict())
|
||||||
|
|
||||||
|
def step_ema(self):
|
||||||
|
if self.step < self.step_start_ema:
|
||||||
|
self.reset_parameters()
|
||||||
|
return
|
||||||
|
self.ema.update_model_average(self.ema_model, self.model)
|
||||||
|
|
||||||
|
def save(self, milestone):
|
||||||
|
data = {
|
||||||
|
'step': self.step,
|
||||||
|
'model': self.model.state_dict(),
|
||||||
|
'ema': self.ema_model.state_dict(),
|
||||||
|
'scaler': self.scaler.state_dict()
|
||||||
|
}
|
||||||
|
torch.save(data, str(self.results_folder / f'model-{milestone}.pt'))
|
||||||
|
|
||||||
|
def load(self, milestone):
|
||||||
|
data = torch.load(str(self.results_folder / f'model-{milestone}.pt'))
|
||||||
|
|
||||||
|
self.step = data['step']
|
||||||
|
self.model.load_state_dict(data['model'])
|
||||||
|
self.ema_model.load_state_dict(data['ema'])
|
||||||
|
self.scaler.load_state_dict(data['scaler'])
|
||||||
|
|
||||||
|
def train(self):
|
||||||
|
while self.step < self.train_num_steps:
|
||||||
for i in range(self.gradient_accumulate_every):
|
for i in range(self.gradient_accumulate_every):
|
||||||
data = next(self.dl).cuda()
|
data = next(self.dl).cuda()
|
||||||
loss = self.model(data)
|
|
||||||
print(f'{ind}: {loss.item()}')
|
|
||||||
loss.backward()
|
|
||||||
|
|
||||||
self.opt.step()
|
with autocast(enabled = self.amp):
|
||||||
|
loss = self.model(data)
|
||||||
|
self.scaler.scale(loss / self.gradient_accumulate_every).backward()
|
||||||
|
|
||||||
|
print(f'{self.step}: {loss.item()}')
|
||||||
|
|
||||||
|
self.scaler.step(self.opt)
|
||||||
|
self.scaler.update()
|
||||||
self.opt.zero_grad()
|
self.opt.zero_grad()
|
||||||
|
|
||||||
if ind % SAVE_AND_SAMPLE_EVERY == 0:
|
if self.step % self.update_ema_every == 0:
|
||||||
milestone = ind // SAVE_AND_SAMPLE_EVERY
|
self.step_ema()
|
||||||
all_images = self.model.p_sample_loop((64, 3, self.image_size, self.image_size))
|
|
||||||
utils.save_image(all_images, f'./sample-{milestone}.png', nrow=8)
|
|
||||||
torch.save(model.state_dict(), f'./model-{milestone}.pt')
|
|
||||||
|
|
||||||
ind += 1
|
if self.step != 0 and self.step % self.save_and_sample_every == 0:
|
||||||
|
milestone = self.step // self.save_and_sample_every
|
||||||
|
batches = num_to_groups(36, self.batch_size)
|
||||||
|
all_images_list = list(map(lambda n: self.ema_model.sample(batch_size=n), batches))
|
||||||
|
all_images = torch.cat(all_images_list, dim=0)
|
||||||
|
all_images = (all_images + 1) * 0.5
|
||||||
|
utils.save_image(all_images, str(self.results_folder / f'sample-{milestone}.png'), nrow = 6)
|
||||||
|
self.save(milestone)
|
||||||
|
|
||||||
|
self.step += 1
|
||||||
|
|
||||||
print('training completed')
|
print('training completed')
|
||||||
|
|||||||
BIN
Binary file not shown.
|
After Width: | Height: | Size: 842 KiB |
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
|
|||||||
setup(
|
setup(
|
||||||
name = 'denoising-diffusion-pytorch',
|
name = 'denoising-diffusion-pytorch',
|
||||||
packages = find_packages(),
|
packages = find_packages(),
|
||||||
version = '0.1.0',
|
version = '0.12.1',
|
||||||
license='MIT',
|
license='MIT',
|
||||||
description = 'Denoising Diffusion Probabilistic Models - Pytorch',
|
description = 'Denoising Diffusion Probabilistic Models - Pytorch',
|
||||||
author = 'Phil Wang',
|
author = 'Phil Wang',
|
||||||
@@ -15,7 +15,6 @@ setup(
|
|||||||
],
|
],
|
||||||
install_requires=[
|
install_requires=[
|
||||||
'einops',
|
'einops',
|
||||||
'numpy',
|
|
||||||
'pillow',
|
'pillow',
|
||||||
'torch',
|
'torch',
|
||||||
'torchvision',
|
'torchvision',
|
||||||
|
|||||||
Reference in New Issue
Block a user