mirror of
https://github.com/wassname/denoising-diffusion-pytorch.git
synced 2026-09-10 12:01:08 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d26acbcae6 | ||
|
|
9939a48139 | ||
|
|
75ea49a7ef | ||
|
|
8c3609a6e3 | ||
|
|
1586d1a8a0 | ||
|
|
b4fb8804d2 | ||
|
|
9fd05f1b1f | ||
|
|
ec2397f0ba | ||
|
|
844e557dfb | ||
|
|
8b30be8042 | ||
|
|
f2f3994b92 | ||
|
|
8ec4ea56a5 | ||
|
|
99cf9b5b96 | ||
|
|
ecc6f30901 | ||
|
|
f900f40f14 | ||
|
|
479f60c178 | ||
|
|
96bb2ff310 | ||
|
|
582bfe275b | ||
|
|
4284c8840d | ||
|
|
d4ffa3fced | ||
|
|
c44d3ea01d | ||
|
|
c4991f576f | ||
|
|
a19331aa59 | ||
|
|
94eabaca1a | ||
|
|
eaf9d9fdc4 | ||
|
|
3bf5e768c2 | ||
|
|
532178a6a3 | ||
|
|
3bbb6ebf16 | ||
|
|
6b93fa48f6 | ||
|
|
a291da5098 | ||
|
|
e5a18bb25c | ||
|
|
fc8e4547aa | ||
|
|
cae9f4a71f | ||
|
|
91f03fb88b | ||
|
|
60128257c5 | ||
|
|
cf6db71985 | ||
|
|
84ebb9ad13 | ||
|
|
caa5af170d | ||
|
|
55c658b967 | ||
|
|
e0f26677d6 | ||
|
|
e147839d74 | ||
|
|
62e8490385 | ||
|
|
d412d8816b | ||
|
|
402b7c26df | ||
|
|
09613a40f3 | ||
|
|
c6966ae95a | ||
|
|
73591cf1ad | ||
|
|
989f0fcb8e | ||
|
|
84731bb03d | ||
|
|
c6ecca555b | ||
|
|
1f5c233072 | ||
|
|
de378158e5 | ||
|
|
e274fb305a | ||
|
|
f39b3b1d3f | ||
|
|
782c904d3b | ||
|
|
71953ebd22 | ||
|
|
0b8cdb4c8b | ||
|
|
e504e0e554 | ||
|
|
bd1e3b676e | ||
|
|
f4615599bc | ||
|
|
eb6e1b508e | ||
|
|
91cff45939 | ||
|
|
7b51e30da7 | ||
|
|
dadbf20154 | ||
|
|
7706bdfc6f | ||
|
|
183e5f3cc5 | ||
|
|
16c9ae7bb3 |
@@ -2,7 +2,13 @@
|
|||||||
|
|
||||||
## Denoising Diffusion Probabilistic Model, in Pytorch
|
## Denoising Diffusion Probabilistic Model, in Pytorch
|
||||||
|
|
||||||
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution. This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>.
|
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution.
|
||||||
|
|
||||||
|
This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>
|
||||||
|
|
||||||
|
Youtube AI Educators - <a href="https://www.youtube.com/watch?v=W-O7AZNzbzQ">Yannic Kilcher</a> | <a href="https://www.youtube.com/watch?v=344w5h24-h8">AI Coffeebreak with Letitia</a> | <a href="https://www.youtube.com/watch?v=HoKDTa5jHvg">Outlier</a>
|
||||||
|
|
||||||
|
<a href="https://huggingface.co/blog/annotated-diffusion">Annotated code</a> by Research Scientists / Engineers from <a href="https://huggingface.co/">🤗 Huggingface</a>
|
||||||
|
|
||||||
<img src="./sample.png" width="500px"><img>
|
<img src="./sample.png" width="500px"><img>
|
||||||
|
|
||||||
@@ -32,7 +38,7 @@ diffusion = GaussianDiffusion(
|
|||||||
loss_type = 'l1' # L1 or L2
|
loss_type = 'l1' # L1 or L2
|
||||||
)
|
)
|
||||||
|
|
||||||
training_images = torch.randn(8, 3, 128, 128)
|
training_images = torch.randn(8, 3, 128, 128) # images are normalized from 0 to 1
|
||||||
loss = diffusion(training_images)
|
loss = diffusion(training_images)
|
||||||
loss.backward()
|
loss.backward()
|
||||||
# after a lot of training
|
# after a lot of training
|
||||||
@@ -62,11 +68,11 @@ trainer = Trainer(
|
|||||||
diffusion,
|
diffusion,
|
||||||
'path/to/your/images',
|
'path/to/your/images',
|
||||||
train_batch_size = 32,
|
train_batch_size = 32,
|
||||||
train_lr = 2e-5,
|
train_lr = 1e-4,
|
||||||
train_num_steps = 700000, # total training steps
|
train_num_steps = 700000, # total training steps
|
||||||
gradient_accumulate_every = 2, # gradient accumulation steps
|
gradient_accumulate_every = 2, # gradient accumulation steps
|
||||||
ema_decay = 0.995, # exponential moving average decay
|
ema_decay = 0.995, # exponential moving average decay
|
||||||
fp16 = True # turn on mixed precision training with apex
|
amp = True # turn on mixed precision
|
||||||
)
|
)
|
||||||
|
|
||||||
trainer.train()
|
trainer.train()
|
||||||
@@ -77,23 +83,53 @@ Samples and model checkpoints will be logged to `./results` periodically
|
|||||||
## Citations
|
## Citations
|
||||||
|
|
||||||
```bibtex
|
```bibtex
|
||||||
@misc{ho2020denoising,
|
@inproceedings{NEURIPS2020_4c5bcfec,
|
||||||
title = {Denoising Diffusion Probabilistic Models},
|
author = {Ho, Jonathan and Jain, Ajay and Abbeel, Pieter},
|
||||||
author = {Jonathan Ho and Ajay Jain and Pieter Abbeel},
|
booktitle = {Advances in Neural Information Processing Systems},
|
||||||
year = {2020},
|
editor = {H. Larochelle and M. Ranzato and R. Hadsell and M.F. Balcan and H. Lin},
|
||||||
eprint = {2006.11239},
|
pages = {6840--6851},
|
||||||
archivePrefix = {arXiv},
|
publisher = {Curran Associates, Inc.},
|
||||||
primaryClass = {cs.LG}
|
title = {Denoising Diffusion Probabilistic Models},
|
||||||
|
url = {https://proceedings.neurips.cc/paper/2020/file/4c5bcfec8584af0d967f1ab10179ca4b-Paper.pdf},
|
||||||
|
volume = {33},
|
||||||
|
year = {2020}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|
||||||
```bibtex
|
```bibtex
|
||||||
@inproceedings{anonymous2021improved,
|
@InProceedings{pmlr-v139-nichol21a,
|
||||||
title = {Improved Denoising Diffusion Probabilistic Models},
|
title = {Improved Denoising Diffusion Probabilistic Models},
|
||||||
author = {Anonymous},
|
author = {Nichol, Alexander Quinn and Dhariwal, Prafulla},
|
||||||
booktitle = {Submitted to International Conference on Learning Representations},
|
booktitle = {Proceedings of the 38th International Conference on Machine Learning},
|
||||||
year = {2021},
|
pages = {8162--8171},
|
||||||
url = {https://openreview.net/forum?id=-NEXDKk8gZ},
|
year = {2021},
|
||||||
note = {under review}
|
editor = {Meila, Marina and Zhang, Tong},
|
||||||
|
volume = {139},
|
||||||
|
series = {Proceedings of Machine Learning Research},
|
||||||
|
month = {18--24 Jul},
|
||||||
|
publisher = {PMLR},
|
||||||
|
pdf = {http://proceedings.mlr.press/v139/nichol21a/nichol21a.pdf},
|
||||||
|
url = {https://proceedings.mlr.press/v139/nichol21a.html},
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
```bibtex
|
||||||
|
@inproceedings{kingma2021on,
|
||||||
|
title = {On Density Estimation with Diffusion Models},
|
||||||
|
author = {Diederik P Kingma and Tim Salimans and Ben Poole and Jonathan Ho},
|
||||||
|
booktitle = {Advances in Neural Information Processing Systems},
|
||||||
|
editor = {A. Beygelzimer and Y. Dauphin and P. Liang and J. Wortman Vaughan},
|
||||||
|
year = {2021},
|
||||||
|
url = {https://openreview.net/forum?id=2LdBqxc1Yv}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
```bibtex
|
||||||
|
@article{Choi2022PerceptionPT,
|
||||||
|
title = {Perception Prioritized Training of Diffusion Models},
|
||||||
|
author = {Jooyoung Choi and Jungbeom Lee and Chaehun Shin and Sungwon Kim and Hyunwoo J. Kim and Sung-Hoon Yoon},
|
||||||
|
journal = {ArXiv},
|
||||||
|
year = {2022},
|
||||||
|
volume = {abs/2204.00227}
|
||||||
}
|
}
|
||||||
```
|
```
|
||||||
|
|||||||
@@ -1 +1,5 @@
|
|||||||
from denoising_diffusion_pytorch.denoising_diffusion_pytorch import GaussianDiffusion, Unet, Trainer
|
from denoising_diffusion_pytorch.denoising_diffusion_pytorch import GaussianDiffusion, Unet, Trainer
|
||||||
|
|
||||||
|
from denoising_diffusion_pytorch.learned_gaussian_diffusion import LearnedGaussianDiffusion
|
||||||
|
from denoising_diffusion_pytorch.continuous_time_gaussian_diffusion import ContinuousTimeGaussianDiffusion
|
||||||
|
from denoising_diffusion_pytorch.weighted_objective_gaussian_diffusion import WeightedObjectiveGaussianDiffusion
|
||||||
|
|||||||
@@ -0,0 +1,287 @@
|
|||||||
|
import math
|
||||||
|
import torch
|
||||||
|
from torch import sqrt
|
||||||
|
from torch import nn, einsum
|
||||||
|
import torch.nn.functional as F
|
||||||
|
from torch.special import expm1
|
||||||
|
|
||||||
|
from tqdm import tqdm
|
||||||
|
from einops import rearrange, repeat, reduce
|
||||||
|
from einops.layers.torch import Rearrange
|
||||||
|
|
||||||
|
# helpers
|
||||||
|
|
||||||
|
def exists(val):
|
||||||
|
return val is not None
|
||||||
|
|
||||||
|
def default(val, d):
|
||||||
|
if exists(val):
|
||||||
|
return val
|
||||||
|
return d() if callable(d) else d
|
||||||
|
|
||||||
|
# normalization functions
|
||||||
|
|
||||||
|
def normalize_to_neg_one_to_one(img):
|
||||||
|
return img * 2 - 1
|
||||||
|
|
||||||
|
def unnormalize_to_zero_to_one(t):
|
||||||
|
return (t + 1) * 0.5
|
||||||
|
|
||||||
|
# diffusion helpers
|
||||||
|
|
||||||
|
def right_pad_dims_to(x, t):
|
||||||
|
padding_dims = x.ndim - t.ndim
|
||||||
|
if padding_dims <= 0:
|
||||||
|
return t
|
||||||
|
return t.view(*t.shape, *((1,) * padding_dims))
|
||||||
|
|
||||||
|
# neural net helpers
|
||||||
|
|
||||||
|
class Residual(nn.Module):
|
||||||
|
def __init__(self, fn):
|
||||||
|
super().__init__()
|
||||||
|
self.fn = fn
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
return x + self.fn(x)
|
||||||
|
|
||||||
|
class MonotonicLinear(nn.Module):
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
super().__init__()
|
||||||
|
self.net = nn.Linear(*args, **kwargs)
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
return F.linear(x, self.net.weight.abs(), self.net.bias.abs())
|
||||||
|
|
||||||
|
# continuous schedules
|
||||||
|
|
||||||
|
# equations are taken from https://openreview.net/attachment?id=2LdBqxc1Yv&name=supplementary_material
|
||||||
|
# @crowsonkb Katherine's repository also helped here https://github.com/crowsonkb/v-diffusion-jax/blob/master/diffusion/utils.py
|
||||||
|
|
||||||
|
# log(snr) that approximates the original linear schedule
|
||||||
|
|
||||||
|
def log(t, eps = 1e-20):
|
||||||
|
return torch.log(t.clamp(min = eps))
|
||||||
|
|
||||||
|
def beta_linear_log_snr(t):
|
||||||
|
return -log(expm1(1e-4 + 10 * (t ** 2)))
|
||||||
|
|
||||||
|
def alpha_cosine_log_snr(t, s = 0.008):
|
||||||
|
return -log((torch.cos((t + s) / (1 + s) * math.pi * 0.5) ** -2) - 1, eps = 1e-5)
|
||||||
|
|
||||||
|
class learned_noise_schedule(nn.Module):
|
||||||
|
""" described in section H and then I.2 of the supplementary material for variational ddpm paper """
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
log_snr_max,
|
||||||
|
log_snr_min,
|
||||||
|
hidden_dim = 1024,
|
||||||
|
frac_gradient = 1.
|
||||||
|
):
|
||||||
|
super().__init__()
|
||||||
|
self.slope = log_snr_min - log_snr_max
|
||||||
|
self.intercept = log_snr_max
|
||||||
|
|
||||||
|
self.net = nn.Sequential(
|
||||||
|
Rearrange('... -> ... 1'),
|
||||||
|
MonotonicLinear(1, 1),
|
||||||
|
Residual(nn.Sequential(
|
||||||
|
MonotonicLinear(1, hidden_dim),
|
||||||
|
nn.Sigmoid(),
|
||||||
|
MonotonicLinear(hidden_dim, 1)
|
||||||
|
)),
|
||||||
|
Rearrange('... 1 -> ...'),
|
||||||
|
)
|
||||||
|
|
||||||
|
self.frac_gradient = frac_gradient
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
frac_gradient = self.frac_gradient
|
||||||
|
device = x.device
|
||||||
|
|
||||||
|
out_zero = self.net(torch.zeros_like(x))
|
||||||
|
out_one = self.net(torch.ones_like(x))
|
||||||
|
|
||||||
|
x = self.net(x)
|
||||||
|
|
||||||
|
normed = self.slope * ((x - out_zero) / (out_one - out_zero)) + self.intercept
|
||||||
|
return normed * frac_gradient + normed.detach() * (1 - frac_gradient)
|
||||||
|
|
||||||
|
class ContinuousTimeGaussianDiffusion(nn.Module):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
denoise_fn,
|
||||||
|
*,
|
||||||
|
image_size,
|
||||||
|
channels = 3,
|
||||||
|
loss_type = 'l1',
|
||||||
|
noise_schedule = 'linear',
|
||||||
|
num_sample_steps = 500,
|
||||||
|
clip_sample_denoised = True,
|
||||||
|
learned_schedule_net_hidden_dim = 1024,
|
||||||
|
learned_noise_schedule_frac_gradient = 1., # between 0 and 1, determines what percentage of gradients go back, so one can update the learned noise schedule more slowly
|
||||||
|
p2_loss_weight_gamma = 0., # p2 loss weight, from https://arxiv.org/abs/2204.00227 - 0 is equivalent to weight of 1 across time
|
||||||
|
p2_loss_weight_k = 1
|
||||||
|
):
|
||||||
|
super().__init__()
|
||||||
|
assert denoise_fn.learned_sinusoidal_cond
|
||||||
|
|
||||||
|
self.denoise_fn = denoise_fn
|
||||||
|
|
||||||
|
# image dimensions
|
||||||
|
|
||||||
|
self.channels = channels
|
||||||
|
self.image_size = image_size
|
||||||
|
|
||||||
|
# continuous noise schedule related stuff
|
||||||
|
|
||||||
|
self.loss_type = loss_type
|
||||||
|
|
||||||
|
if noise_schedule == 'linear':
|
||||||
|
self.log_snr = beta_linear_log_snr
|
||||||
|
elif noise_schedule == 'cosine':
|
||||||
|
self.log_snr = alpha_cosine_log_snr
|
||||||
|
elif noise_schedule == 'learned':
|
||||||
|
log_snr_max, log_snr_min = [beta_linear_log_snr(torch.tensor([time])).item() for time in (0., 1.)]
|
||||||
|
|
||||||
|
self.log_snr = learned_noise_schedule(
|
||||||
|
log_snr_max = log_snr_max,
|
||||||
|
log_snr_min = log_snr_min,
|
||||||
|
hidden_dim = learned_schedule_net_hidden_dim,
|
||||||
|
frac_gradient = learned_noise_schedule_frac_gradient
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
raise ValueError(f'unknown noise schedule {noise_schedule}')
|
||||||
|
|
||||||
|
# sampling
|
||||||
|
|
||||||
|
self.num_sample_steps = num_sample_steps
|
||||||
|
self.clip_sample_denoised = clip_sample_denoised
|
||||||
|
|
||||||
|
# p2 loss weight
|
||||||
|
# proposed https://arxiv.org/abs/2204.00227
|
||||||
|
|
||||||
|
assert p2_loss_weight_gamma <= 2, 'in paper, they noticed any gamma greater than 2 is harmful'
|
||||||
|
|
||||||
|
self.p2_loss_weight_gamma = p2_loss_weight_gamma # recommended to be 0.5 or 1
|
||||||
|
self.p2_loss_weight_k = p2_loss_weight_k
|
||||||
|
|
||||||
|
@property
|
||||||
|
def device(self):
|
||||||
|
return next(self.denoise_fn.parameters()).device
|
||||||
|
|
||||||
|
@property
|
||||||
|
def loss_fn(self):
|
||||||
|
if self.loss_type == 'l1':
|
||||||
|
return F.l1_loss
|
||||||
|
elif self.loss_type == 'l2':
|
||||||
|
return F.mse_loss
|
||||||
|
else:
|
||||||
|
raise ValueError(f'invalid loss type {self.loss_type}')
|
||||||
|
|
||||||
|
def p_mean_variance(self, x, time, time_next):
|
||||||
|
# reviewer found an error in the equation in the paper (missing sigma)
|
||||||
|
# following - https://openreview.net/forum?id=2LdBqxc1Yv¬eId=rIQgH0zKsRt
|
||||||
|
|
||||||
|
log_snr = self.log_snr(time)
|
||||||
|
log_snr_next = self.log_snr(time_next)
|
||||||
|
c = -expm1(log_snr - log_snr_next)
|
||||||
|
|
||||||
|
squared_alpha, squared_alpha_next = log_snr.sigmoid(), log_snr_next.sigmoid()
|
||||||
|
squared_sigma, squared_sigma_next = (-log_snr).sigmoid(), (-log_snr_next).sigmoid()
|
||||||
|
|
||||||
|
alpha, sigma, alpha_next = map(sqrt, (squared_alpha, squared_sigma, squared_alpha_next))
|
||||||
|
|
||||||
|
batch_log_snr = repeat(log_snr, ' -> b', b = x.shape[0])
|
||||||
|
pred_noise = self.denoise_fn(x, batch_log_snr)
|
||||||
|
|
||||||
|
if self.clip_sample_denoised:
|
||||||
|
x_start = (x - sigma * pred_noise) / alpha
|
||||||
|
|
||||||
|
# in Imagen, this was changed to dynamic thresholding
|
||||||
|
x_start.clamp_(-1., 1.)
|
||||||
|
|
||||||
|
model_mean = alpha_next * (x * (1 - c) / alpha + c * x_start)
|
||||||
|
else:
|
||||||
|
model_mean = alpha_next / alpha * (x - c * sigma * pred_noise)
|
||||||
|
|
||||||
|
posterior_variance = squared_sigma_next * c
|
||||||
|
|
||||||
|
return model_mean, posterior_variance
|
||||||
|
|
||||||
|
# sampling related functions
|
||||||
|
|
||||||
|
@torch.no_grad()
|
||||||
|
def p_sample(self, x, time, time_next):
|
||||||
|
batch, *_, device = *x.shape, x.device
|
||||||
|
|
||||||
|
model_mean, model_variance = self.p_mean_variance(x = x, time = time, time_next = time_next)
|
||||||
|
|
||||||
|
if time_next == 0:
|
||||||
|
return model_mean
|
||||||
|
|
||||||
|
noise = torch.randn_like(x)
|
||||||
|
return model_mean + sqrt(model_variance) * noise
|
||||||
|
|
||||||
|
@torch.no_grad()
|
||||||
|
def p_sample_loop(self, shape):
|
||||||
|
batch = shape[0]
|
||||||
|
|
||||||
|
img = torch.randn(shape, device = self.device)
|
||||||
|
steps = torch.linspace(1., 0., self.num_sample_steps + 1, device = self.device)
|
||||||
|
|
||||||
|
for i in tqdm(range(self.num_sample_steps), desc = 'sampling loop time step', total = self.num_sample_steps):
|
||||||
|
times = steps[i]
|
||||||
|
times_next = steps[i + 1]
|
||||||
|
img = self.p_sample(img, times, times_next)
|
||||||
|
|
||||||
|
img.clamp_(-1., 1.)
|
||||||
|
img = unnormalize_to_zero_to_one(img)
|
||||||
|
return img
|
||||||
|
|
||||||
|
@torch.no_grad()
|
||||||
|
def sample(self, batch_size = 16):
|
||||||
|
return self.p_sample_loop((batch_size, self.channels, self.image_size, self.image_size))
|
||||||
|
|
||||||
|
# training related functions - noise prediction
|
||||||
|
|
||||||
|
def q_sample(self, x_start, times, noise = None):
|
||||||
|
noise = default(noise, lambda: torch.randn_like(x_start))
|
||||||
|
|
||||||
|
log_snr = self.log_snr(times)
|
||||||
|
|
||||||
|
log_snr_padded = right_pad_dims_to(x_start, log_snr)
|
||||||
|
alpha, sigma = sqrt(log_snr_padded.sigmoid()), sqrt((-log_snr_padded).sigmoid())
|
||||||
|
x_noised = x_start * alpha + noise * sigma
|
||||||
|
|
||||||
|
return x_noised, log_snr
|
||||||
|
|
||||||
|
def random_times(self, batch_size):
|
||||||
|
# times are now uniform from 0 to 1
|
||||||
|
return torch.zeros((batch_size,), device = self.device).float().uniform_(0, 1)
|
||||||
|
|
||||||
|
def p_losses(self, x_start, times, noise = None):
|
||||||
|
noise = default(noise, lambda: torch.randn_like(x_start))
|
||||||
|
|
||||||
|
x, log_snr = self.q_sample(x_start = x_start, times = times, noise = noise)
|
||||||
|
model_out = self.denoise_fn(x, log_snr)
|
||||||
|
|
||||||
|
losses = self.loss_fn(model_out, noise, reduction = 'none')
|
||||||
|
losses = reduce(losses, 'b ... -> b', 'mean')
|
||||||
|
|
||||||
|
if self.p2_loss_weight_gamma >= 0:
|
||||||
|
# following eq 8. in https://arxiv.org/abs/2204.00227
|
||||||
|
loss_weight = (self.p2_loss_weight_k + log_snr.exp()) ** -self.p2_loss_weight_gamma
|
||||||
|
losses = losses * loss_weight
|
||||||
|
|
||||||
|
return losses.mean()
|
||||||
|
|
||||||
|
def forward(self, img, *args, **kwargs):
|
||||||
|
b, c, h, w, device, img_size, = *img.shape, img.device, self.image_size
|
||||||
|
assert h == img_size and w == img_size, f'height and width of image must be {img_size}'
|
||||||
|
|
||||||
|
times = self.random_times(b)
|
||||||
|
img = normalize_to_neg_one_to_one(img)
|
||||||
|
return self.p_losses(img, times, *args, **kwargs)
|
||||||
@@ -7,29 +7,19 @@ from inspect import isfunction
|
|||||||
from functools import partial
|
from functools import partial
|
||||||
|
|
||||||
from torch.utils import data
|
from torch.utils import data
|
||||||
|
from multiprocessing import cpu_count
|
||||||
|
from torch.cuda.amp import autocast, GradScaler
|
||||||
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from torch.optim import Adam
|
from torch.optim import Adam
|
||||||
from torchvision import transforms, utils
|
from torchvision import transforms, utils
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
from einops import rearrange
|
from einops import rearrange, reduce
|
||||||
|
from einops.layers.torch import Rearrange
|
||||||
|
|
||||||
try:
|
from ema_pytorch import EMA
|
||||||
from apex import amp
|
|
||||||
APEX_AVAILABLE = True
|
|
||||||
except:
|
|
||||||
APEX_AVAILABLE = False
|
|
||||||
|
|
||||||
# constants
|
|
||||||
|
|
||||||
SAVE_AND_SAMPLE_EVERY = 1000
|
|
||||||
UPDATE_EMA_EVERY = 10
|
|
||||||
EXTS = ['jpg', 'jpeg', 'png']
|
|
||||||
|
|
||||||
RESULTS_FOLDER = Path('./results')
|
|
||||||
RESULTS_FOLDER.mkdir(exist_ok = True)
|
|
||||||
|
|
||||||
# helpers functions
|
# helpers functions
|
||||||
|
|
||||||
@@ -54,30 +44,14 @@ def num_to_groups(num, divisor):
|
|||||||
arr.append(remainder)
|
arr.append(remainder)
|
||||||
return arr
|
return arr
|
||||||
|
|
||||||
def loss_backwards(fp16, loss, optimizer, **kwargs):
|
def normalize_to_neg_one_to_one(img):
|
||||||
if fp16:
|
return img * 2 - 1
|
||||||
with amp.scale_loss(loss, optimizer) as scaled_loss:
|
|
||||||
scaled_loss.backward(**kwargs)
|
def unnormalize_to_zero_to_one(t):
|
||||||
else:
|
return (t + 1) * 0.5
|
||||||
loss.backward(**kwargs)
|
|
||||||
|
|
||||||
# small helper modules
|
# small helper modules
|
||||||
|
|
||||||
class EMA():
|
|
||||||
def __init__(self, beta):
|
|
||||||
super().__init__()
|
|
||||||
self.beta = beta
|
|
||||||
|
|
||||||
def update_model_average(self, ma_model, current_model):
|
|
||||||
for current_params, ma_params in zip(current_model.parameters(), ma_model.parameters()):
|
|
||||||
old_weight, up_weight = ma_params.data, current_params.data
|
|
||||||
ma_params.data = self.update_average(old_weight, up_weight)
|
|
||||||
|
|
||||||
def update_average(self, old, new):
|
|
||||||
if old is None:
|
|
||||||
return new
|
|
||||||
return old * self.beta + (1 - self.beta) * new
|
|
||||||
|
|
||||||
class Residual(nn.Module):
|
class Residual(nn.Module):
|
||||||
def __init__(self, fn):
|
def __init__(self, fn):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
@@ -86,6 +60,39 @@ class Residual(nn.Module):
|
|||||||
def forward(self, x, *args, **kwargs):
|
def forward(self, x, *args, **kwargs):
|
||||||
return self.fn(x, *args, **kwargs) + x
|
return self.fn(x, *args, **kwargs) + x
|
||||||
|
|
||||||
|
def Upsample(dim, dim_out = None):
|
||||||
|
return nn.Sequential(
|
||||||
|
nn.Upsample(scale_factor = 2, mode = 'nearest'),
|
||||||
|
nn.Conv2d(dim, default(dim_out, dim), 3, padding = 1)
|
||||||
|
)
|
||||||
|
|
||||||
|
def Downsample(dim, dim_out = None):
|
||||||
|
return nn.Conv2d(dim, default(dim_out, dim), 4, 2, 1)
|
||||||
|
|
||||||
|
class LayerNorm(nn.Module):
|
||||||
|
def __init__(self, dim, eps = 1e-5):
|
||||||
|
super().__init__()
|
||||||
|
self.eps = eps
|
||||||
|
self.g = nn.Parameter(torch.ones(1, dim, 1, 1))
|
||||||
|
self.b = nn.Parameter(torch.zeros(1, dim, 1, 1))
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
var = torch.var(x, dim = 1, unbiased = False, keepdim = True)
|
||||||
|
mean = torch.mean(x, dim = 1, keepdim = True)
|
||||||
|
return (x - mean) / (var + self.eps).sqrt() * self.g + self.b
|
||||||
|
|
||||||
|
class PreNorm(nn.Module):
|
||||||
|
def __init__(self, dim, fn):
|
||||||
|
super().__init__()
|
||||||
|
self.fn = fn
|
||||||
|
self.norm = LayerNorm(dim)
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
x = self.norm(x)
|
||||||
|
return self.fn(x)
|
||||||
|
|
||||||
|
# sinusoidal positional embeds
|
||||||
|
|
||||||
class SinusoidalPosEmb(nn.Module):
|
class SinusoidalPosEmb(nn.Module):
|
||||||
def __init__(self, dim):
|
def __init__(self, dim):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
@@ -100,69 +107,101 @@ class SinusoidalPosEmb(nn.Module):
|
|||||||
emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
|
emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
|
||||||
return emb
|
return emb
|
||||||
|
|
||||||
class Mish(nn.Module):
|
class LearnedSinusoidalPosEmb(nn.Module):
|
||||||
def forward(self, x):
|
""" following @crowsonkb 's lead with learned sinusoidal pos emb """
|
||||||
return x * torch.tanh(F.softplus(x))
|
""" https://github.com/crowsonkb/v-diffusion-jax/blob/master/diffusion/models/danbooru_128.py#L8 """
|
||||||
|
|
||||||
class Upsample(nn.Module):
|
|
||||||
def __init__(self, dim):
|
def __init__(self, dim):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.conv = nn.ConvTranspose2d(dim, dim, 4, 2, 1)
|
assert (dim % 2) == 0
|
||||||
|
half_dim = dim // 2
|
||||||
|
self.weights = nn.Parameter(torch.randn(half_dim))
|
||||||
|
|
||||||
def forward(self, x):
|
def forward(self, x):
|
||||||
return self.conv(x)
|
x = rearrange(x, 'b -> b 1')
|
||||||
|
freqs = x * rearrange(self.weights, 'd -> 1 d') * 2 * math.pi
|
||||||
class Downsample(nn.Module):
|
fouriered = torch.cat((freqs.sin(), freqs.cos()), dim = -1)
|
||||||
def __init__(self, dim):
|
fouriered = torch.cat((x, fouriered), dim = -1)
|
||||||
super().__init__()
|
return fouriered
|
||||||
self.conv = nn.Conv2d(dim, dim, 3, 2, 1)
|
|
||||||
|
|
||||||
def forward(self, x):
|
|
||||||
return self.conv(x)
|
|
||||||
|
|
||||||
class Rezero(nn.Module):
|
|
||||||
def __init__(self, fn):
|
|
||||||
super().__init__()
|
|
||||||
self.fn = fn
|
|
||||||
self.g = nn.Parameter(torch.zeros(1))
|
|
||||||
|
|
||||||
def forward(self, x):
|
|
||||||
return self.fn(x) * self.g
|
|
||||||
|
|
||||||
# building block modules
|
# building block modules
|
||||||
|
|
||||||
class Block(nn.Module):
|
class Block(nn.Module):
|
||||||
def __init__(self, dim, dim_out, groups = 8):
|
def __init__(self, dim, dim_out, groups = 8):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.block = nn.Sequential(
|
self.proj = nn.Conv2d(dim, dim_out, 3, padding = 1)
|
||||||
nn.Conv2d(dim, dim_out, 3, padding=1),
|
self.norm = nn.GroupNorm(groups, dim_out)
|
||||||
nn.GroupNorm(groups, dim_out),
|
self.act = nn.SiLU()
|
||||||
Mish()
|
|
||||||
)
|
def forward(self, x, scale_shift = None):
|
||||||
def forward(self, x):
|
x = self.proj(x)
|
||||||
return self.block(x)
|
x = self.norm(x)
|
||||||
|
|
||||||
|
if exists(scale_shift):
|
||||||
|
scale, shift = scale_shift
|
||||||
|
x = x * (scale + 1) + shift
|
||||||
|
|
||||||
|
x = self.act(x)
|
||||||
|
return x
|
||||||
|
|
||||||
class ResnetBlock(nn.Module):
|
class ResnetBlock(nn.Module):
|
||||||
def __init__(self, dim, dim_out, *, time_emb_dim, groups = 8):
|
def __init__(self, dim, dim_out, *, time_emb_dim = None, groups = 8):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.mlp = nn.Sequential(
|
self.mlp = nn.Sequential(
|
||||||
Mish(),
|
nn.SiLU(),
|
||||||
nn.Linear(time_emb_dim, dim_out)
|
nn.Linear(time_emb_dim, dim_out * 2)
|
||||||
)
|
) if exists(time_emb_dim) else None
|
||||||
|
|
||||||
self.block1 = Block(dim, dim_out)
|
self.block1 = Block(dim, dim_out, groups = groups)
|
||||||
self.block2 = Block(dim_out, dim_out)
|
self.block2 = Block(dim_out, dim_out, groups = groups)
|
||||||
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
|
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
|
||||||
|
|
||||||
def forward(self, x, time_emb):
|
def forward(self, x, time_emb = None):
|
||||||
h = self.block1(x)
|
|
||||||
h += self.mlp(time_emb)[:, :, None, None]
|
scale_shift = None
|
||||||
|
if exists(self.mlp) and exists(time_emb):
|
||||||
|
time_emb = self.mlp(time_emb)
|
||||||
|
time_emb = rearrange(time_emb, 'b c -> b c 1 1')
|
||||||
|
scale_shift = time_emb.chunk(2, dim = 1)
|
||||||
|
|
||||||
|
h = self.block1(x, scale_shift = scale_shift)
|
||||||
|
|
||||||
h = self.block2(h)
|
h = self.block2(h)
|
||||||
|
|
||||||
return h + self.res_conv(x)
|
return h + self.res_conv(x)
|
||||||
|
|
||||||
class LinearAttention(nn.Module):
|
class LinearAttention(nn.Module):
|
||||||
def __init__(self, dim, heads = 4, dim_head = 32):
|
def __init__(self, dim, heads = 4, dim_head = 32):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.scale = dim_head ** -0.5
|
||||||
|
self.heads = heads
|
||||||
|
hidden_dim = dim_head * heads
|
||||||
|
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
|
||||||
|
|
||||||
|
self.to_out = nn.Sequential(
|
||||||
|
nn.Conv2d(hidden_dim, dim, 1),
|
||||||
|
LayerNorm(dim)
|
||||||
|
)
|
||||||
|
|
||||||
|
def forward(self, x):
|
||||||
|
b, c, h, w = x.shape
|
||||||
|
qkv = self.to_qkv(x).chunk(3, dim = 1)
|
||||||
|
q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
|
||||||
|
|
||||||
|
q = q.softmax(dim = -2)
|
||||||
|
k = k.softmax(dim = -1)
|
||||||
|
|
||||||
|
q = q * self.scale
|
||||||
|
context = torch.einsum('b h d n, b h e n -> b h d e', k, v)
|
||||||
|
|
||||||
|
out = torch.einsum('b h d e, b h d n -> b h e n', context, q)
|
||||||
|
out = rearrange(out, 'b h c (x y) -> b (h c) x y', h = self.heads, x = h, y = w)
|
||||||
|
return self.to_out(out)
|
||||||
|
|
||||||
|
class Attention(nn.Module):
|
||||||
|
def __init__(self, dim, heads = 4, dim_head = 32):
|
||||||
|
super().__init__()
|
||||||
|
self.scale = dim_head ** -0.5
|
||||||
self.heads = heads
|
self.heads = heads
|
||||||
hidden_dim = dim_head * heads
|
hidden_dim = dim_head * heads
|
||||||
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
|
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
|
||||||
@@ -170,12 +209,16 @@ class LinearAttention(nn.Module):
|
|||||||
|
|
||||||
def forward(self, x):
|
def forward(self, x):
|
||||||
b, c, h, w = x.shape
|
b, c, h, w = x.shape
|
||||||
qkv = self.to_qkv(x)
|
qkv = self.to_qkv(x).chunk(3, dim = 1)
|
||||||
q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads, qkv=3)
|
q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
|
||||||
k = k.softmax(dim=-1)
|
q = q * self.scale
|
||||||
context = torch.einsum('bhdn,bhen->bhde', k, v)
|
|
||||||
out = torch.einsum('bhde,bhdn->bhen', context, q)
|
sim = einsum('b h d i, b h d j -> b h i j', q, k)
|
||||||
out = rearrange(out, 'b heads c (h w) -> b (heads c) h w', heads=self.heads, h=h, w=w)
|
sim = sim - sim.amax(dim = -1, keepdim = True).detach()
|
||||||
|
attn = sim.softmax(dim = -1)
|
||||||
|
|
||||||
|
out = einsum('b h i j, b h d j -> b h i d', attn, v)
|
||||||
|
out = rearrange(out, 'b h (x y) d -> b (h d) x y', x = h, y = w)
|
||||||
return self.to_out(out)
|
return self.to_out(out)
|
||||||
|
|
||||||
# model
|
# model
|
||||||
@@ -184,24 +227,51 @@ class Unet(nn.Module):
|
|||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
dim,
|
dim,
|
||||||
|
init_dim = None,
|
||||||
out_dim = None,
|
out_dim = None,
|
||||||
dim_mults=(1, 2, 4, 8),
|
dim_mults=(1, 2, 4, 8),
|
||||||
groups = 8,
|
channels = 3,
|
||||||
channels = 3
|
resnet_block_groups = 8,
|
||||||
|
learned_variance = False,
|
||||||
|
learned_sinusoidal_cond = False,
|
||||||
|
learned_sinusoidal_dim = 16
|
||||||
):
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
|
||||||
|
# determine dimensions
|
||||||
|
|
||||||
self.channels = channels
|
self.channels = channels
|
||||||
|
|
||||||
dims = [channels, *map(lambda m: dim * m, dim_mults)]
|
init_dim = default(init_dim, dim)
|
||||||
|
self.init_conv = nn.Conv2d(channels, init_dim, 7, padding = 3)
|
||||||
|
|
||||||
|
dims = [init_dim, *map(lambda m: dim * m, dim_mults)]
|
||||||
in_out = list(zip(dims[:-1], dims[1:]))
|
in_out = list(zip(dims[:-1], dims[1:]))
|
||||||
|
|
||||||
self.time_pos_emb = SinusoidalPosEmb(dim)
|
block_klass = partial(ResnetBlock, groups = resnet_block_groups)
|
||||||
self.mlp = nn.Sequential(
|
|
||||||
nn.Linear(dim, dim * 4),
|
# time embeddings
|
||||||
Mish(),
|
|
||||||
nn.Linear(dim * 4, dim)
|
time_dim = dim * 4
|
||||||
|
|
||||||
|
self.learned_sinusoidal_cond = learned_sinusoidal_cond
|
||||||
|
|
||||||
|
if learned_sinusoidal_cond:
|
||||||
|
sinu_pos_emb = LearnedSinusoidalPosEmb(learned_sinusoidal_dim)
|
||||||
|
fourier_dim = learned_sinusoidal_dim + 1
|
||||||
|
else:
|
||||||
|
sinu_pos_emb = SinusoidalPosEmb(dim)
|
||||||
|
fourier_dim = dim
|
||||||
|
|
||||||
|
self.time_mlp = nn.Sequential(
|
||||||
|
sinu_pos_emb,
|
||||||
|
nn.Linear(fourier_dim, time_dim),
|
||||||
|
nn.GELU(),
|
||||||
|
nn.Linear(time_dim, time_dim)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# layers
|
||||||
|
|
||||||
self.downs = nn.ModuleList([])
|
self.downs = nn.ModuleList([])
|
||||||
self.ups = nn.ModuleList([])
|
self.ups = nn.ModuleList([])
|
||||||
num_resolutions = len(in_out)
|
num_resolutions = len(in_out)
|
||||||
@@ -210,57 +280,68 @@ class Unet(nn.Module):
|
|||||||
is_last = ind >= (num_resolutions - 1)
|
is_last = ind >= (num_resolutions - 1)
|
||||||
|
|
||||||
self.downs.append(nn.ModuleList([
|
self.downs.append(nn.ModuleList([
|
||||||
ResnetBlock(dim_in, dim_out, time_emb_dim = dim),
|
block_klass(dim_in, dim_in, time_emb_dim = time_dim),
|
||||||
ResnetBlock(dim_out, dim_out, time_emb_dim = dim),
|
block_klass(dim_in, dim_in, time_emb_dim = time_dim),
|
||||||
Residual(Rezero(LinearAttention(dim_out))),
|
Residual(PreNorm(dim_in, LinearAttention(dim_in))),
|
||||||
Downsample(dim_out) if not is_last else nn.Identity()
|
Downsample(dim_in, dim_out) if not is_last else nn.Conv2d(dim_in, dim_out, 3, padding = 1)
|
||||||
]))
|
]))
|
||||||
|
|
||||||
mid_dim = dims[-1]
|
mid_dim = dims[-1]
|
||||||
self.mid_block1 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim)
|
self.mid_block1 = block_klass(mid_dim, mid_dim, time_emb_dim = time_dim)
|
||||||
self.mid_attn = Residual(Rezero(LinearAttention(mid_dim)))
|
self.mid_attn = Residual(PreNorm(mid_dim, Attention(mid_dim)))
|
||||||
self.mid_block2 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim)
|
self.mid_block2 = block_klass(mid_dim, mid_dim, time_emb_dim = time_dim)
|
||||||
|
|
||||||
for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])):
|
for ind, (dim_in, dim_out) in enumerate(reversed(in_out)):
|
||||||
is_last = ind >= (num_resolutions - 1)
|
is_last = ind == (len(in_out) - 1)
|
||||||
|
|
||||||
self.ups.append(nn.ModuleList([
|
self.ups.append(nn.ModuleList([
|
||||||
ResnetBlock(dim_out * 2, dim_in, time_emb_dim = dim),
|
block_klass(dim_out + dim_in, dim_out, time_emb_dim = time_dim),
|
||||||
ResnetBlock(dim_in, dim_in, time_emb_dim = dim),
|
block_klass(dim_out + dim_in, dim_out, time_emb_dim = time_dim),
|
||||||
Residual(Rezero(LinearAttention(dim_in))),
|
Residual(PreNorm(dim_out, LinearAttention(dim_out))),
|
||||||
Upsample(dim_in) if not is_last else nn.Identity()
|
Upsample(dim_out, dim_in) if not is_last else nn.Conv2d(dim_out, dim_in, 3, padding = 1)
|
||||||
]))
|
]))
|
||||||
|
|
||||||
out_dim = default(out_dim, channels)
|
default_out_dim = channels * (1 if not learned_variance else 2)
|
||||||
self.final_conv = nn.Sequential(
|
self.out_dim = default(out_dim, default_out_dim)
|
||||||
Block(dim, dim),
|
|
||||||
nn.Conv2d(dim, out_dim, 1)
|
self.final_res_block = block_klass(dim * 2, dim, time_emb_dim = time_dim)
|
||||||
)
|
self.final_conv = nn.Conv2d(dim, self.out_dim, 1)
|
||||||
|
|
||||||
def forward(self, x, time):
|
def forward(self, x, time):
|
||||||
t = self.time_pos_emb(time)
|
x = self.init_conv(x)
|
||||||
t = self.mlp(t)
|
r = x.clone()
|
||||||
|
|
||||||
|
t = self.time_mlp(time)
|
||||||
|
|
||||||
h = []
|
h = []
|
||||||
|
|
||||||
for resnet, resnet2, attn, downsample in self.downs:
|
for block1, block2, attn, downsample in self.downs:
|
||||||
x = resnet(x, t)
|
x = block1(x, t)
|
||||||
x = resnet2(x, t)
|
h.append(x)
|
||||||
|
|
||||||
|
x = block2(x, t)
|
||||||
x = attn(x)
|
x = attn(x)
|
||||||
h.append(x)
|
h.append(x)
|
||||||
|
|
||||||
x = downsample(x)
|
x = downsample(x)
|
||||||
|
|
||||||
x = self.mid_block1(x, t)
|
x = self.mid_block1(x, t)
|
||||||
x = self.mid_attn(x)
|
x = self.mid_attn(x)
|
||||||
x = self.mid_block2(x, t)
|
x = self.mid_block2(x, t)
|
||||||
|
|
||||||
for resnet, resnet2, attn, upsample in self.ups:
|
for block1, block2, attn, upsample in self.ups:
|
||||||
x = torch.cat((x, h.pop()), dim=1)
|
x = torch.cat((x, h.pop()), dim = 1)
|
||||||
x = resnet(x, t)
|
x = block1(x, t)
|
||||||
x = resnet2(x, t)
|
|
||||||
|
x = torch.cat((x, h.pop()), dim = 1)
|
||||||
|
x = block2(x, t)
|
||||||
x = attn(x)
|
x = attn(x)
|
||||||
|
|
||||||
x = upsample(x)
|
x = upsample(x)
|
||||||
|
|
||||||
|
x = torch.cat((x, r), dim = 1)
|
||||||
|
|
||||||
|
x = self.final_res_block(x, t)
|
||||||
return self.final_conv(x)
|
return self.final_conv(x)
|
||||||
|
|
||||||
# gaussian diffusion trainer class
|
# gaussian diffusion trainer class
|
||||||
@@ -270,10 +351,11 @@ def extract(a, t, x_shape):
|
|||||||
out = a.gather(-1, t)
|
out = a.gather(-1, t)
|
||||||
return out.reshape(b, *((1,) * (len(x_shape) - 1)))
|
return out.reshape(b, *((1,) * (len(x_shape) - 1)))
|
||||||
|
|
||||||
def noise_like(shape, device, repeat=False):
|
def linear_beta_schedule(timesteps):
|
||||||
repeat_noise = lambda: torch.randn((1, *shape[1:]), device=device).repeat(shape[0], *((1,) * (len(shape) - 1)))
|
scale = 1000 / timesteps
|
||||||
noise = lambda: torch.randn(shape, device=device)
|
beta_start = scale * 0.0001
|
||||||
return repeat_noise() if repeat else noise()
|
beta_end = scale * 0.02
|
||||||
|
return torch.linspace(beta_start, beta_end, timesteps, dtype = torch.float64)
|
||||||
|
|
||||||
def cosine_beta_schedule(timesteps, s = 0.008):
|
def cosine_beta_schedule(timesteps, s = 0.008):
|
||||||
"""
|
"""
|
||||||
@@ -281,11 +363,11 @@ def cosine_beta_schedule(timesteps, s = 0.008):
|
|||||||
as proposed in https://openreview.net/forum?id=-NEXDKk8gZ
|
as proposed in https://openreview.net/forum?id=-NEXDKk8gZ
|
||||||
"""
|
"""
|
||||||
steps = timesteps + 1
|
steps = timesteps + 1
|
||||||
x = np.linspace(0, steps, steps)
|
x = torch.linspace(0, timesteps, steps, dtype = torch.float64)
|
||||||
alphas_cumprod = np.cos(((x / steps) + s) / (1 + s) * np.pi * 0.5) ** 2
|
alphas_cumprod = torch.cos(((x / timesteps) + s) / (1 + s) * math.pi * 0.5) ** 2
|
||||||
alphas_cumprod = alphas_cumprod / alphas_cumprod[0]
|
alphas_cumprod = alphas_cumprod / alphas_cumprod[0]
|
||||||
betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1])
|
betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1])
|
||||||
return np.clip(betas, a_min = 0, a_max = 0.999)
|
return torch.clip(betas, 0, 0.999)
|
||||||
|
|
||||||
class GaussianDiffusion(nn.Module):
|
class GaussianDiffusion(nn.Module):
|
||||||
def __init__(
|
def __init__(
|
||||||
@@ -296,55 +378,67 @@ class GaussianDiffusion(nn.Module):
|
|||||||
channels = 3,
|
channels = 3,
|
||||||
timesteps = 1000,
|
timesteps = 1000,
|
||||||
loss_type = 'l1',
|
loss_type = 'l1',
|
||||||
betas = None
|
objective = 'pred_noise',
|
||||||
|
beta_schedule = 'cosine',
|
||||||
|
p2_loss_weight_gamma = 0., # p2 loss weight, from https://arxiv.org/abs/2204.00227 - 0 is equivalent to weight of 1 across time - 1. is recommended
|
||||||
|
p2_loss_weight_k = 1
|
||||||
):
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
assert not (type(self) == GaussianDiffusion and denoise_fn.channels != denoise_fn.out_dim)
|
||||||
|
|
||||||
self.channels = channels
|
self.channels = channels
|
||||||
self.image_size = image_size
|
self.image_size = image_size
|
||||||
self.denoise_fn = denoise_fn
|
self.denoise_fn = denoise_fn
|
||||||
|
self.objective = objective
|
||||||
|
|
||||||
if exists(betas):
|
if beta_schedule == 'linear':
|
||||||
betas = betas.detach().cpu().numpy() if isinstance(betas, torch.Tensor) else betas
|
betas = linear_beta_schedule(timesteps)
|
||||||
else:
|
elif beta_schedule == 'cosine':
|
||||||
betas = cosine_beta_schedule(timesteps)
|
betas = cosine_beta_schedule(timesteps)
|
||||||
|
else:
|
||||||
|
raise ValueError(f'unknown beta schedule {beta_schedule}')
|
||||||
|
|
||||||
alphas = 1. - betas
|
alphas = 1. - betas
|
||||||
alphas_cumprod = np.cumprod(alphas, axis=0)
|
alphas_cumprod = torch.cumprod(alphas, axis=0)
|
||||||
alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1])
|
alphas_cumprod_prev = F.pad(alphas_cumprod[:-1], (1, 0), value = 1.)
|
||||||
|
|
||||||
timesteps, = betas.shape
|
timesteps, = betas.shape
|
||||||
self.num_timesteps = int(timesteps)
|
self.num_timesteps = int(timesteps)
|
||||||
self.loss_type = loss_type
|
self.loss_type = loss_type
|
||||||
|
|
||||||
to_torch = partial(torch.tensor, dtype=torch.float32)
|
# helper function to register buffer from float64 to float32
|
||||||
|
|
||||||
self.register_buffer('betas', to_torch(betas))
|
register_buffer = lambda name, val: self.register_buffer(name, val.to(torch.float32))
|
||||||
self.register_buffer('alphas_cumprod', to_torch(alphas_cumprod))
|
|
||||||
self.register_buffer('alphas_cumprod_prev', to_torch(alphas_cumprod_prev))
|
register_buffer('betas', betas)
|
||||||
|
register_buffer('alphas_cumprod', alphas_cumprod)
|
||||||
|
register_buffer('alphas_cumprod_prev', alphas_cumprod_prev)
|
||||||
|
|
||||||
# calculations for diffusion q(x_t | x_{t-1}) and others
|
# calculations for diffusion q(x_t | x_{t-1}) and others
|
||||||
self.register_buffer('sqrt_alphas_cumprod', to_torch(np.sqrt(alphas_cumprod)))
|
|
||||||
self.register_buffer('sqrt_one_minus_alphas_cumprod', to_torch(np.sqrt(1. - alphas_cumprod)))
|
register_buffer('sqrt_alphas_cumprod', torch.sqrt(alphas_cumprod))
|
||||||
self.register_buffer('log_one_minus_alphas_cumprod', to_torch(np.log(1. - alphas_cumprod)))
|
register_buffer('sqrt_one_minus_alphas_cumprod', torch.sqrt(1. - alphas_cumprod))
|
||||||
self.register_buffer('sqrt_recip_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod)))
|
register_buffer('log_one_minus_alphas_cumprod', torch.log(1. - alphas_cumprod))
|
||||||
self.register_buffer('sqrt_recipm1_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod - 1)))
|
register_buffer('sqrt_recip_alphas_cumprod', torch.sqrt(1. / alphas_cumprod))
|
||||||
|
register_buffer('sqrt_recipm1_alphas_cumprod', torch.sqrt(1. / alphas_cumprod - 1))
|
||||||
|
|
||||||
# calculations for posterior q(x_{t-1} | x_t, x_0)
|
# calculations for posterior q(x_{t-1} | x_t, x_0)
|
||||||
posterior_variance = betas * (1. - alphas_cumprod_prev) / (1. - alphas_cumprod)
|
|
||||||
# above: equal to 1. / (1. / (1. - alpha_cumprod_tm1) + alpha_t / beta_t)
|
|
||||||
self.register_buffer('posterior_variance', to_torch(posterior_variance))
|
|
||||||
# below: log calculation clipped because the posterior variance is 0 at the beginning of the diffusion chain
|
|
||||||
self.register_buffer('posterior_log_variance_clipped', to_torch(np.log(np.maximum(posterior_variance, 1e-20))))
|
|
||||||
self.register_buffer('posterior_mean_coef1', to_torch(
|
|
||||||
betas * np.sqrt(alphas_cumprod_prev) / (1. - alphas_cumprod)))
|
|
||||||
self.register_buffer('posterior_mean_coef2', to_torch(
|
|
||||||
(1. - alphas_cumprod_prev) * np.sqrt(alphas) / (1. - alphas_cumprod)))
|
|
||||||
|
|
||||||
def q_mean_variance(self, x_start, t):
|
posterior_variance = betas * (1. - alphas_cumprod_prev) / (1. - alphas_cumprod)
|
||||||
mean = extract(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
|
|
||||||
variance = extract(1. - self.alphas_cumprod, t, x_start.shape)
|
# above: equal to 1. / (1. / (1. - alpha_cumprod_tm1) + alpha_t / beta_t)
|
||||||
log_variance = extract(self.log_one_minus_alphas_cumprod, t, x_start.shape)
|
|
||||||
return mean, variance, log_variance
|
register_buffer('posterior_variance', posterior_variance)
|
||||||
|
|
||||||
|
# below: log calculation clipped because the posterior variance is 0 at the beginning of the diffusion chain
|
||||||
|
|
||||||
|
register_buffer('posterior_log_variance_clipped', torch.log(posterior_variance.clamp(min =1e-20)))
|
||||||
|
register_buffer('posterior_mean_coef1', betas * torch.sqrt(alphas_cumprod_prev) / (1. - alphas_cumprod))
|
||||||
|
register_buffer('posterior_mean_coef2', (1. - alphas_cumprod_prev) * torch.sqrt(alphas) / (1. - alphas_cumprod))
|
||||||
|
|
||||||
|
# calculate p2 reweighting
|
||||||
|
|
||||||
|
register_buffer('p2_loss_weight', (p2_loss_weight_k + alphas_cumprod / (1 - alphas_cumprod)) ** -p2_loss_weight_gamma)
|
||||||
|
|
||||||
def predict_start_from_noise(self, x_t, t, noise):
|
def predict_start_from_noise(self, x_t, t, noise):
|
||||||
return (
|
return (
|
||||||
@@ -362,19 +456,26 @@ class GaussianDiffusion(nn.Module):
|
|||||||
return posterior_mean, posterior_variance, posterior_log_variance_clipped
|
return posterior_mean, posterior_variance, posterior_log_variance_clipped
|
||||||
|
|
||||||
def p_mean_variance(self, x, t, clip_denoised: bool):
|
def p_mean_variance(self, x, t, clip_denoised: bool):
|
||||||
x_recon = self.predict_start_from_noise(x, t=t, noise=self.denoise_fn(x, t))
|
model_output = self.denoise_fn(x, t)
|
||||||
|
|
||||||
|
if self.objective == 'pred_noise':
|
||||||
|
x_start = self.predict_start_from_noise(x, t = t, noise = model_output)
|
||||||
|
elif self.objective == 'pred_x0':
|
||||||
|
x_start = model_output
|
||||||
|
else:
|
||||||
|
raise ValueError(f'unknown objective {self.objective}')
|
||||||
|
|
||||||
if clip_denoised:
|
if clip_denoised:
|
||||||
x_recon.clamp_(-1., 1.)
|
x_start.clamp_(-1., 1.)
|
||||||
|
|
||||||
model_mean, posterior_variance, posterior_log_variance = self.q_posterior(x_start=x_recon, x_t=x, t=t)
|
model_mean, posterior_variance, posterior_log_variance = self.q_posterior(x_start = x_start, x_t = x, t = t)
|
||||||
return model_mean, posterior_variance, posterior_log_variance
|
return model_mean, posterior_variance, posterior_log_variance
|
||||||
|
|
||||||
@torch.no_grad()
|
@torch.no_grad()
|
||||||
def p_sample(self, x, t, clip_denoised=True, repeat_noise=False):
|
def p_sample(self, x, t, clip_denoised=True):
|
||||||
b, *_, device = *x.shape, x.device
|
b, *_, device = *x.shape, x.device
|
||||||
model_mean, _, model_log_variance = self.p_mean_variance(x=x, t=t, clip_denoised=clip_denoised)
|
model_mean, _, model_log_variance = self.p_mean_variance(x=x, t=t, clip_denoised=clip_denoised)
|
||||||
noise = noise_like(x.shape, device, repeat_noise)
|
noise = torch.randn_like(x)
|
||||||
# no noise when t == 0
|
# no noise when t == 0
|
||||||
nonzero_mask = (1 - (t == 0).float()).reshape(b, *((1,) * (len(x.shape) - 1)))
|
nonzero_mask = (1 - (t == 0).float()).reshape(b, *((1,) * (len(x.shape) - 1)))
|
||||||
return model_mean + nonzero_mask * (0.5 * model_log_variance).exp() * noise
|
return model_mean + nonzero_mask * (0.5 * model_log_variance).exp() * noise
|
||||||
@@ -388,6 +489,8 @@ class GaussianDiffusion(nn.Module):
|
|||||||
|
|
||||||
for i in tqdm(reversed(range(0, self.num_timesteps)), desc='sampling loop time step', total=self.num_timesteps):
|
for i in tqdm(reversed(range(0, self.num_timesteps)), desc='sampling loop time step', total=self.num_timesteps):
|
||||||
img = self.p_sample(img, torch.full((b,), i, device=device, dtype=torch.long))
|
img = self.p_sample(img, torch.full((b,), i, device=device, dtype=torch.long))
|
||||||
|
|
||||||
|
img = unnormalize_to_zero_to_one(img)
|
||||||
return img
|
return img
|
||||||
|
|
||||||
@torch.no_grad()
|
@torch.no_grad()
|
||||||
@@ -420,40 +523,55 @@ class GaussianDiffusion(nn.Module):
|
|||||||
extract(self.sqrt_one_minus_alphas_cumprod, t, x_start.shape) * noise
|
extract(self.sqrt_one_minus_alphas_cumprod, t, x_start.shape) * noise
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def loss_fn(self):
|
||||||
|
if self.loss_type == 'l1':
|
||||||
|
return F.l1_loss
|
||||||
|
elif self.loss_type == 'l2':
|
||||||
|
return F.mse_loss
|
||||||
|
else:
|
||||||
|
raise ValueError(f'invalid loss type {self.loss_type}')
|
||||||
|
|
||||||
def p_losses(self, x_start, t, noise = None):
|
def p_losses(self, x_start, t, noise = None):
|
||||||
b, c, h, w = x_start.shape
|
b, c, h, w = x_start.shape
|
||||||
noise = default(noise, lambda: torch.randn_like(x_start))
|
noise = default(noise, lambda: torch.randn_like(x_start))
|
||||||
|
|
||||||
x_noisy = self.q_sample(x_start=x_start, t=t, noise=noise)
|
x = self.q_sample(x_start=x_start, t=t, noise=noise)
|
||||||
x_recon = self.denoise_fn(x_noisy, t)
|
model_out = self.denoise_fn(x, t)
|
||||||
|
|
||||||
if self.loss_type == 'l1':
|
if self.objective == 'pred_noise':
|
||||||
loss = (noise - x_recon).abs().mean()
|
target = noise
|
||||||
elif self.loss_type == 'l2':
|
elif self.objective == 'pred_x0':
|
||||||
loss = F.mse_loss(noise, x_recon)
|
target = x_start
|
||||||
else:
|
else:
|
||||||
raise NotImplementedError()
|
raise ValueError(f'unknown objective {self.objective}')
|
||||||
|
|
||||||
return loss
|
loss = self.loss_fn(model_out, target, reduction = 'none')
|
||||||
|
loss = reduce(loss, 'b ... -> b (...)', 'mean')
|
||||||
|
|
||||||
def forward(self, x, *args, **kwargs):
|
loss = loss * extract(self.p2_loss_weight, t, loss.shape)
|
||||||
b, c, h, w, device, img_size, = *x.shape, x.device, self.image_size
|
return loss.mean()
|
||||||
|
|
||||||
|
def forward(self, img, *args, **kwargs):
|
||||||
|
b, c, h, w, device, img_size, = *img.shape, img.device, self.image_size
|
||||||
assert h == img_size and w == img_size, f'height and width of image must be {img_size}'
|
assert h == img_size and w == img_size, f'height and width of image must be {img_size}'
|
||||||
t = torch.randint(0, self.num_timesteps, (b,), device=device).long()
|
t = torch.randint(0, self.num_timesteps, (b,), device=device).long()
|
||||||
return self.p_losses(x, t, *args, **kwargs)
|
|
||||||
|
img = normalize_to_neg_one_to_one(img)
|
||||||
|
return self.p_losses(img, t, *args, **kwargs)
|
||||||
|
|
||||||
# dataset classes
|
# dataset classes
|
||||||
|
|
||||||
class Dataset(data.Dataset):
|
class Dataset(data.Dataset):
|
||||||
def __init__(self, folder, image_size):
|
def __init__(self, folder, image_size, exts = ['jpg', 'jpeg', 'png'], augment_horizontal_flip = False):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.folder = folder
|
self.folder = folder
|
||||||
self.image_size = image_size
|
self.image_size = image_size
|
||||||
self.paths = [p for ext in EXTS for p in Path(f'{folder}').glob(f'**/*.{ext}')]
|
self.paths = [p for ext in exts for p in Path(f'{folder}').glob(f'**/*.{ext}')]
|
||||||
|
|
||||||
self.transform = transforms.Compose([
|
self.transform = transforms.Compose([
|
||||||
transforms.Resize(image_size),
|
transforms.Resize(image_size),
|
||||||
transforms.RandomHorizontalFlip(),
|
transforms.RandomHorizontalFlip() if augment_horizontal_flip else nn.Identity(),
|
||||||
transforms.CenterCrop(image_size),
|
transforms.CenterCrop(image_size),
|
||||||
transforms.ToTensor()
|
transforms.ToTensor()
|
||||||
])
|
])
|
||||||
@@ -475,87 +593,91 @@ class Trainer(object):
|
|||||||
folder,
|
folder,
|
||||||
*,
|
*,
|
||||||
ema_decay = 0.995,
|
ema_decay = 0.995,
|
||||||
image_size = 128,
|
|
||||||
train_batch_size = 32,
|
train_batch_size = 32,
|
||||||
train_lr = 2e-5,
|
train_lr = 1e-4,
|
||||||
train_num_steps = 100000,
|
train_num_steps = 100000,
|
||||||
gradient_accumulate_every = 2,
|
gradient_accumulate_every = 2,
|
||||||
fp16 = False,
|
amp = False,
|
||||||
step_start_ema = 2000
|
step_start_ema = 2000,
|
||||||
|
ema_update_every = 10,
|
||||||
|
save_and_sample_every = 1000,
|
||||||
|
results_folder = './results',
|
||||||
|
augment_horizontal_flip = True
|
||||||
):
|
):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.image_size = diffusion_model.image_size
|
||||||
|
|
||||||
self.model = diffusion_model
|
self.model = diffusion_model
|
||||||
self.ema = EMA(ema_decay)
|
self.ema = EMA(diffusion_model, beta = ema_decay, update_every = ema_update_every)
|
||||||
self.ema_model = copy.deepcopy(self.model)
|
|
||||||
self.step_start_ema = step_start_ema
|
self.step_start_ema = step_start_ema
|
||||||
|
self.save_and_sample_every = save_and_sample_every
|
||||||
|
|
||||||
self.batch_size = train_batch_size
|
self.batch_size = train_batch_size
|
||||||
self.image_size = diffusion_model.image_size
|
self.image_size = diffusion_model.image_size
|
||||||
self.gradient_accumulate_every = gradient_accumulate_every
|
self.gradient_accumulate_every = gradient_accumulate_every
|
||||||
self.train_num_steps = train_num_steps
|
self.train_num_steps = train_num_steps
|
||||||
|
|
||||||
self.ds = Dataset(folder, image_size)
|
self.ds = Dataset(folder, self.image_size, augment_horizontal_flip = augment_horizontal_flip)
|
||||||
self.dl = cycle(data.DataLoader(self.ds, batch_size = train_batch_size, shuffle=True, pin_memory=True))
|
self.dl = cycle(data.DataLoader(self.ds, batch_size = train_batch_size, shuffle = True, pin_memory = True, num_workers = cpu_count()))
|
||||||
self.opt = Adam(diffusion_model.parameters(), lr=train_lr)
|
self.opt = Adam(diffusion_model.parameters(), lr = train_lr)
|
||||||
|
|
||||||
self.step = 0
|
self.step = 0
|
||||||
|
|
||||||
assert not fp16 or fp16 and APEX_AVAILABLE, 'Apex must be installed in order for mixed precision training to be turned on'
|
self.amp = amp
|
||||||
|
self.scaler = GradScaler(enabled = amp)
|
||||||
|
|
||||||
self.fp16 = fp16
|
self.results_folder = Path(results_folder)
|
||||||
if fp16:
|
self.results_folder.mkdir(exist_ok = True)
|
||||||
(self.model, self.ema_model), self.opt = amp.initialize([self.model, self.ema_model], self.opt, opt_level='O1')
|
|
||||||
|
|
||||||
self.reset_parameters()
|
|
||||||
|
|
||||||
def reset_parameters(self):
|
|
||||||
self.ema_model.load_state_dict(self.model.state_dict())
|
|
||||||
|
|
||||||
def step_ema(self):
|
|
||||||
if self.step < self.step_start_ema:
|
|
||||||
self.reset_parameters()
|
|
||||||
return
|
|
||||||
self.ema.update_model_average(self.ema_model, self.model)
|
|
||||||
|
|
||||||
def save(self, milestone):
|
def save(self, milestone):
|
||||||
data = {
|
data = {
|
||||||
'step': self.step,
|
'step': self.step,
|
||||||
'model': self.model.state_dict(),
|
'model': self.model.state_dict(),
|
||||||
'ema': self.ema_model.state_dict()
|
'ema': self.ema.state_dict(),
|
||||||
|
'scaler': self.scaler.state_dict()
|
||||||
}
|
}
|
||||||
torch.save(data, str(RESULTS_FOLDER / f'model-{milestone}.pt'))
|
torch.save(data, str(self.results_folder / f'model-{milestone}.pt'))
|
||||||
|
|
||||||
def load(self, milestone):
|
def load(self, milestone):
|
||||||
data = torch.load(str(RESULTS_FOLDER / f'model-{milestone}.pt'))
|
data = torch.load(str(self.results_folder / f'model-{milestone}.pt'))
|
||||||
|
|
||||||
self.step = data['step']
|
self.step = data['step']
|
||||||
self.model.load_state_dict(data['model'])
|
self.model.load_state_dict(data['model'])
|
||||||
self.ema_model.load_state_dict(data['ema'])
|
self.ema.load_state_dict(data['ema'])
|
||||||
|
self.scaler.load_state_dict(data['scaler'])
|
||||||
|
|
||||||
def train(self):
|
def train(self):
|
||||||
backwards = partial(loss_backwards, self.fp16)
|
with tqdm(initial = self.step, total = self.train_num_steps) as pbar:
|
||||||
|
|
||||||
while self.step < self.train_num_steps:
|
while self.step < self.train_num_steps:
|
||||||
for i in range(self.gradient_accumulate_every):
|
for i in range(self.gradient_accumulate_every):
|
||||||
data = next(self.dl).cuda()
|
data = next(self.dl).cuda()
|
||||||
loss = self.model(data)
|
|
||||||
print(f'{self.step}: {loss.item()}')
|
|
||||||
backwards(loss / self.gradient_accumulate_every, self.opt)
|
|
||||||
|
|
||||||
self.opt.step()
|
with autocast(enabled = self.amp):
|
||||||
self.opt.zero_grad()
|
loss = self.model(data)
|
||||||
|
self.scaler.scale(loss / self.gradient_accumulate_every).backward()
|
||||||
|
|
||||||
if self.step % UPDATE_EMA_EVERY == 0:
|
pbar.set_description(f'loss: {loss.item():.4f}')
|
||||||
self.step_ema()
|
|
||||||
|
|
||||||
if self.step != 0 and self.step % SAVE_AND_SAMPLE_EVERY == 0:
|
self.scaler.step(self.opt)
|
||||||
milestone = self.step // SAVE_AND_SAMPLE_EVERY
|
self.scaler.update()
|
||||||
batches = num_to_groups(36, self.batch_size)
|
self.opt.zero_grad()
|
||||||
all_images_list = list(map(lambda n: self.ema_model.sample(batch_size=n), batches))
|
|
||||||
all_images = torch.cat(all_images_list, dim=0)
|
|
||||||
utils.save_image(all_images, str(RESULTS_FOLDER / f'sample-{milestone}.png'), nrow=6)
|
|
||||||
self.save(milestone)
|
|
||||||
|
|
||||||
self.step += 1
|
self.ema.update()
|
||||||
|
|
||||||
print('training completed')
|
if self.step != 0 and self.step % self.save_and_sample_every == 0:
|
||||||
|
self.ema.ema_model.eval()
|
||||||
|
with torch.no_grad():
|
||||||
|
milestone = self.step // self.save_and_sample_every
|
||||||
|
batches = num_to_groups(36, self.batch_size)
|
||||||
|
all_images_list = list(map(lambda n: self.ema.ema_model.sample(batch_size=n), batches))
|
||||||
|
|
||||||
|
all_images = torch.cat(all_images_list, dim=0)
|
||||||
|
utils.save_image(all_images, str(self.results_folder / f'sample-{milestone}.png'), nrow = 6)
|
||||||
|
self.save(milestone)
|
||||||
|
|
||||||
|
self.step += 1
|
||||||
|
pbar.update(1)
|
||||||
|
|
||||||
|
print('training complete')
|
||||||
|
|||||||
@@ -0,0 +1,132 @@
|
|||||||
|
import torch
|
||||||
|
from math import pi, sqrt, log as ln
|
||||||
|
from inspect import isfunction
|
||||||
|
from torch import nn, einsum
|
||||||
|
from einops import rearrange
|
||||||
|
|
||||||
|
from denoising_diffusion_pytorch.denoising_diffusion_pytorch import GaussianDiffusion, extract, unnormalize_to_zero_to_one
|
||||||
|
|
||||||
|
# constants
|
||||||
|
|
||||||
|
NAT = 1. / ln(2)
|
||||||
|
|
||||||
|
# helper functions
|
||||||
|
|
||||||
|
def exists(x):
|
||||||
|
return x is not None
|
||||||
|
|
||||||
|
def default(val, d):
|
||||||
|
if exists(val):
|
||||||
|
return val
|
||||||
|
return d() if isfunction(d) else d
|
||||||
|
|
||||||
|
# tensor helpers
|
||||||
|
|
||||||
|
def log(t, eps = 1e-12):
|
||||||
|
return torch.log(t.clamp(min = eps))
|
||||||
|
|
||||||
|
def meanflat(x):
|
||||||
|
return x.mean(dim = tuple(range(1, len(x.shape))))
|
||||||
|
|
||||||
|
def normal_kl(mean1, logvar1, mean2, logvar2):
|
||||||
|
"""
|
||||||
|
KL divergence between normal distributions parameterized by mean and log-variance.
|
||||||
|
"""
|
||||||
|
return 0.5 * (-1.0 + logvar2 - logvar1 + torch.exp(logvar1 - logvar2) + ((mean1 - mean2) ** 2) * torch.exp(-logvar2))
|
||||||
|
|
||||||
|
def approx_standard_normal_cdf(x):
|
||||||
|
return 0.5 * (1.0 + torch.tanh(sqrt(2.0 / pi) * (x + 0.044715 * (x ** 3))))
|
||||||
|
|
||||||
|
def discretized_gaussian_log_likelihood(x, *, means, log_scales, thres = 0.999):
|
||||||
|
assert x.shape == means.shape == log_scales.shape
|
||||||
|
|
||||||
|
centered_x = x - means
|
||||||
|
inv_stdv = torch.exp(-log_scales)
|
||||||
|
plus_in = inv_stdv * (centered_x + 1. / 255.)
|
||||||
|
cdf_plus = approx_standard_normal_cdf(plus_in)
|
||||||
|
min_in = inv_stdv * (centered_x - 1. / 255.)
|
||||||
|
cdf_min = approx_standard_normal_cdf(min_in)
|
||||||
|
log_cdf_plus = log(cdf_plus)
|
||||||
|
log_one_minus_cdf_min = log(1. - cdf_min)
|
||||||
|
cdf_delta = cdf_plus - cdf_min
|
||||||
|
|
||||||
|
log_probs = torch.where(x < -thres,
|
||||||
|
log_cdf_plus,
|
||||||
|
torch.where(x > thres,
|
||||||
|
log_one_minus_cdf_min,
|
||||||
|
log(cdf_delta)))
|
||||||
|
|
||||||
|
return log_probs
|
||||||
|
|
||||||
|
# https://arxiv.org/abs/2102.09672
|
||||||
|
|
||||||
|
# i thought the results were questionable, if one were to focus only on FID
|
||||||
|
# but may as well get this in here for others to try, as GLIDE is using it (and DALL-E2 first stage of cascade)
|
||||||
|
# gaussian diffusion for learned variance + hybrid eps simple + vb loss
|
||||||
|
|
||||||
|
class LearnedGaussianDiffusion(GaussianDiffusion):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
denoise_fn,
|
||||||
|
vb_loss_weight = 0.001, # lambda was 0.001 in the paper
|
||||||
|
*args,
|
||||||
|
**kwargs
|
||||||
|
):
|
||||||
|
super().__init__(denoise_fn, *args, **kwargs)
|
||||||
|
assert denoise_fn.out_dim == (denoise_fn.channels * 2), 'dimension out of unet must be twice the number of channels for learned variance - you can also set the `learned_variance` keyword argument on the Unet to be `True`'
|
||||||
|
self.vb_loss_weight = vb_loss_weight
|
||||||
|
|
||||||
|
def p_mean_variance(self, *, x, t, clip_denoised, model_output = None):
|
||||||
|
model_output = default(model_output, lambda: self.denoise_fn(x, t))
|
||||||
|
pred_noise, var_interp_frac_unnormalized = model_output.chunk(2, dim = 1)
|
||||||
|
|
||||||
|
min_log = extract(self.posterior_log_variance_clipped, t, x.shape)
|
||||||
|
max_log = extract(torch.log(self.betas), t, x.shape)
|
||||||
|
var_interp_frac = unnormalize_to_zero_to_one(var_interp_frac_unnormalized)
|
||||||
|
|
||||||
|
model_log_variance = var_interp_frac * max_log + (1 - var_interp_frac) * min_log
|
||||||
|
model_variance = model_log_variance.exp()
|
||||||
|
|
||||||
|
x_start = self.predict_start_from_noise(x, t, pred_noise)
|
||||||
|
|
||||||
|
if clip_denoised:
|
||||||
|
x_start.clamp_(-1., 1.)
|
||||||
|
|
||||||
|
model_mean, _, _ = self.q_posterior(x_start, x, t)
|
||||||
|
|
||||||
|
return model_mean, model_variance, model_log_variance
|
||||||
|
|
||||||
|
def p_losses(self, x_start, t, noise = None, clip_denoised = False):
|
||||||
|
noise = default(noise, lambda: torch.randn_like(x_start))
|
||||||
|
x_t = self.q_sample(x_start = x_start, t = t, noise = noise)
|
||||||
|
|
||||||
|
# model output
|
||||||
|
|
||||||
|
model_output = self.denoise_fn(x_t, t)
|
||||||
|
|
||||||
|
# calculating kl loss for learned variance (interpolation)
|
||||||
|
|
||||||
|
true_mean, _, true_log_variance_clipped = self.q_posterior(x_start = x_start, x_t = x_t, t = t)
|
||||||
|
model_mean, _, model_log_variance = self.p_mean_variance(x = x_t, t = t, clip_denoised = clip_denoised, model_output = model_output)
|
||||||
|
|
||||||
|
# kl loss with detached model predicted mean, for stability reasons as in paper
|
||||||
|
|
||||||
|
detached_model_mean = model_mean.detach()
|
||||||
|
|
||||||
|
kl = normal_kl(true_mean, true_log_variance_clipped, detached_model_mean, model_log_variance)
|
||||||
|
kl = meanflat(kl) * NAT
|
||||||
|
|
||||||
|
decoder_nll = -discretized_gaussian_log_likelihood(x_start, means = detached_model_mean, log_scales = 0.5 * model_log_variance)
|
||||||
|
decoder_nll = meanflat(decoder_nll) * NAT
|
||||||
|
|
||||||
|
# at the first timestep return the decoder NLL, otherwise return KL(q(x_{t-1}|x_t,x_0) || p(x_{t-1}|x_t))
|
||||||
|
|
||||||
|
vb_losses = torch.where(t == 0, decoder_nll, kl)
|
||||||
|
|
||||||
|
# simple loss - predicting noise, x0, or x_prev
|
||||||
|
|
||||||
|
pred_noise, _ = model_output.chunk(2, dim = 1)
|
||||||
|
|
||||||
|
simple_losses = self.loss_fn(pred_noise, noise)
|
||||||
|
|
||||||
|
return simple_losses + vb_losses.mean() * self.vb_loss_weight
|
||||||
@@ -0,0 +1,80 @@
|
|||||||
|
import torch
|
||||||
|
from inspect import isfunction
|
||||||
|
from torch import nn, einsum
|
||||||
|
from einops import rearrange
|
||||||
|
|
||||||
|
from denoising_diffusion_pytorch.denoising_diffusion_pytorch import GaussianDiffusion
|
||||||
|
|
||||||
|
# helper functions
|
||||||
|
|
||||||
|
def exists(x):
|
||||||
|
return x is not None
|
||||||
|
|
||||||
|
def default(val, d):
|
||||||
|
if exists(val):
|
||||||
|
return val
|
||||||
|
return d() if isfunction(d) else d
|
||||||
|
|
||||||
|
# some improvisation on my end
|
||||||
|
# where i have the model learn to both predict noise and x0
|
||||||
|
# and learn the weighted sum for each depending on time step
|
||||||
|
|
||||||
|
class WeightedObjectiveGaussianDiffusion(GaussianDiffusion):
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
denoise_fn,
|
||||||
|
*args,
|
||||||
|
pred_noise_loss_weight = 0.1,
|
||||||
|
pred_x_start_loss_weight = 0.1,
|
||||||
|
**kwargs
|
||||||
|
):
|
||||||
|
super().__init__(denoise_fn, *args, **kwargs)
|
||||||
|
channels = denoise_fn.channels
|
||||||
|
assert denoise_fn.out_dim == (channels * 2 + 2), 'dimension out (out_dim) of unet must be twice the number of channels + 2 (for the softmax weighted sum) - for channels of 3, this should be (3 * 2) + 2 = 8'
|
||||||
|
|
||||||
|
self.split_dims = (channels, channels, 2)
|
||||||
|
self.pred_noise_loss_weight = pred_noise_loss_weight
|
||||||
|
self.pred_x_start_loss_weight = pred_x_start_loss_weight
|
||||||
|
|
||||||
|
def p_mean_variance(self, *, x, t, clip_denoised, model_output = None):
|
||||||
|
model_output = self.denoise_fn(x, t)
|
||||||
|
|
||||||
|
pred_noise, pred_x_start, weights = model_output.split(self.split_dims, dim = 1)
|
||||||
|
normalized_weights = weights.softmax(dim = 1)
|
||||||
|
|
||||||
|
x_start_from_noise = self.predict_start_from_noise(x, t = t, noise = pred_noise)
|
||||||
|
|
||||||
|
x_starts = torch.stack((x_start_from_noise, pred_x_start), dim = 1)
|
||||||
|
weighted_x_start = einsum('b j h w, b j c h w -> b c h w', normalized_weights, x_starts)
|
||||||
|
|
||||||
|
if clip_denoised:
|
||||||
|
weighted_x_start.clamp_(-1., 1.)
|
||||||
|
|
||||||
|
model_mean, model_variance, model_log_variance = self.q_posterior(weighted_x_start, x, t)
|
||||||
|
|
||||||
|
return model_mean, model_variance, model_log_variance
|
||||||
|
|
||||||
|
def p_losses(self, x_start, t, noise = None, clip_denoised = False):
|
||||||
|
noise = default(noise, lambda: torch.randn_like(x_start))
|
||||||
|
x_t = self.q_sample(x_start = x_start, t = t, noise = noise)
|
||||||
|
|
||||||
|
model_output = self.denoise_fn(x_t, t)
|
||||||
|
pred_noise, pred_x_start, weights = model_output.split(self.split_dims, dim = 1)
|
||||||
|
|
||||||
|
# get loss for predicted noise and x_start
|
||||||
|
# with the loss weight given at initialization
|
||||||
|
|
||||||
|
noise_loss = self.loss_fn(noise, pred_noise) * self.pred_noise_loss_weight
|
||||||
|
x_start_loss = self.loss_fn(x_start, pred_x_start) * self.pred_x_start_loss_weight
|
||||||
|
|
||||||
|
# calculate x_start from predicted noise
|
||||||
|
# then do a weighted sum of the x_start prediction, weights also predicted by the model (softmax normalized)
|
||||||
|
|
||||||
|
x_start_from_pred_noise = self.predict_start_from_noise(x_t, t, pred_noise)
|
||||||
|
x_start_from_pred_noise = x_start_from_pred_noise.clamp(-2., 2.)
|
||||||
|
weighted_x_start = einsum('b j h w, b j c h w -> b c h w', weights.softmax(dim = 1), torch.stack((x_start_from_pred_noise, pred_x_start), dim = 1))
|
||||||
|
|
||||||
|
# main loss to x_start with the weighted one
|
||||||
|
|
||||||
|
weighted_x_start_loss = self.loss_fn(x_start, weighted_x_start)
|
||||||
|
return weighted_x_start_loss + x_start_loss + noise_loss
|
||||||
@@ -3,19 +3,20 @@ from setuptools import setup, find_packages
|
|||||||
setup(
|
setup(
|
||||||
name = 'denoising-diffusion-pytorch',
|
name = 'denoising-diffusion-pytorch',
|
||||||
packages = find_packages(),
|
packages = find_packages(),
|
||||||
version = '0.6.3',
|
version = '0.22.0',
|
||||||
license='MIT',
|
license='MIT',
|
||||||
description = 'Denoising Diffusion Probabilistic Models - Pytorch',
|
description = 'Denoising Diffusion Probabilistic Models - Pytorch',
|
||||||
author = 'Phil Wang',
|
author = 'Phil Wang',
|
||||||
author_email = 'lucidrains@gmail.com',
|
author_email = 'lucidrains@gmail.com',
|
||||||
url = 'https://github.com/lucidrains/denoising-diffusion-pytorch',
|
url = 'https://github.com/lucidrains/denoising-diffusion-pytorch',
|
||||||
|
long_description_content_type = 'text/markdown',
|
||||||
keywords = [
|
keywords = [
|
||||||
'artificial intelligence',
|
'artificial intelligence',
|
||||||
'generative models'
|
'generative models'
|
||||||
],
|
],
|
||||||
install_requires=[
|
install_requires=[
|
||||||
'einops',
|
'einops',
|
||||||
'numpy',
|
'ema-pytorch',
|
||||||
'pillow',
|
'pillow',
|
||||||
'torch',
|
'torch',
|
||||||
'torchvision',
|
'torchvision',
|
||||||
|
|||||||
Reference in New Issue
Block a user