Compare commits

..
13 Commits
4 changed files with 163 additions and 43 deletions
+24 -12
View File
@@ -4,6 +4,10 @@
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution. This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>. Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution. This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>.
<img src="./sample.png" width="500px"><img>
[![PyPI version](https://badge.fury.io/py/denoising-diffusion-pytorch.svg)](https://badge.fury.io/py/denoising-diffusion-pytorch)
## Install ## Install
```bash ```bash
@@ -23,10 +27,8 @@ model = Unet(
diffusion = GaussianDiffusion( diffusion = GaussianDiffusion(
model, model,
beta_start = 0.0001, timesteps = 1000, # number of steps
beta_end = 0.02, loss_type = 'l1' # L1 or L2
num_diffusion_timesteps = 1000, # number of steps
loss_type = 'l1' # L1 or L2 (wavegrad paper claims l1 is better?)
) )
training_images = torch.randn(8, 3, 128, 128) training_images = torch.randn(8, 3, 128, 128)
@@ -50,10 +52,8 @@ model = Unet(
diffusion = GaussianDiffusion( diffusion = GaussianDiffusion(
model, model,
beta_start = 0.0001, timesteps = 1000, # number of steps
beta_end = 0.02, loss_type = 'l1' # L1 or L2
num_diffusion_timesteps = 1000, # number of steps
loss_type = 'l1' # L1 or L2
).cuda() ).cuda()
trainer = Trainer( trainer = Trainer(
@@ -62,15 +62,15 @@ trainer = Trainer(
image_size = 128, image_size = 128,
train_batch_size = 32, train_batch_size = 32,
train_lr = 2e-5, train_lr = 2e-5,
train_num_steps = 100000, train_num_steps = 100000, # total training steps
gradient_accumulate_every = 2 gradient_accumulate_every = 2, # gradient accumulation steps
ema_decay = 0.995, # exponential moving average decay
fp16 = True # turn on mixed precision training with apex
) )
trainer.train() trainer.train()
``` ```
Todo: Command line tool for one-line training
## Citations ## Citations
```bibtex ```bibtex
@@ -83,3 +83,15 @@ Todo: Command line tool for one-line training
primaryClass={cs.LG} primaryClass={cs.LG}
} }
``` ```
```bibtex
@inproceedings{
anonymous2021improved,
title={Improved Denoising Diffusion Probabilistic Models},
author={Anonymous},
booktitle={Submitted to International Conference on Learning Representations},
year={2021},
url={https://openreview.net/forum?id=-NEXDKk8gZ},
note={under review}
}
```
@@ -1,4 +1,5 @@
import math import math
import copy
import torch import torch
from torch import nn, einsum from torch import nn, einsum
import torch.nn.functional as F import torch.nn.functional as F
@@ -15,10 +16,20 @@ import numpy as np
from tqdm import tqdm from tqdm import tqdm
from einops import rearrange from einops import rearrange
try:
from apex import amp
APEX_AVAILABLE = True
except:
APEX_AVAILABLE = False
# constants # constants
SAVE_AND_SAMPLE_EVERY = 1000 SAVE_AND_SAMPLE_EVERY = 1000
EXTS = ['jpg', 'png'] UPDATE_EMA_EVERY = 10
EXTS = ['jpg', 'jpeg', 'png']
RESULTS_FOLDER = Path('./results')
RESULTS_FOLDER.mkdir(exist_ok = True)
# helpers functions # helpers functions
@@ -35,8 +46,38 @@ def cycle(dl):
for data in dl: for data in dl:
yield data yield data
def num_to_groups(num, divisor):
groups = num // divisor
remainder = num % divisor
arr = [divisor] * groups
if remainder > 0:
arr.append(remainder)
return arr
def loss_backwards(fp16, loss, optimizer, **kwargs):
if fp16:
with amp.scale_loss(loss, optimizer) as scaled_loss:
scaled_loss.backward(**kwargs)
else:
loss.backward(**kwargs)
# small helper modules # small helper modules
class EMA():
def __init__(self, beta):
super().__init__()
self.beta = beta
def update_model_average(self, ma_model, current_model):
for current_params, ma_params in zip(current_model.parameters(), ma_model.parameters()):
old_weight, up_weight = ma_params.data, current_params.data
ma_params.data = self.update_average(old_weight, up_weight)
def update_average(self, old, new):
if old is None:
return new
return old * self.beta + (1 - self.beta) * new
class Residual(nn.Module): class Residual(nn.Module):
def __init__(self, fn): def __init__(self, fn):
super().__init__() super().__init__()
@@ -80,17 +121,18 @@ class Downsample(nn.Module):
return self.conv(x) return self.conv(x)
class Rezero(nn.Module): class Rezero(nn.Module):
def __init__(self, dim): def __init__(self, fn):
super().__init__() super().__init__()
self.fn = fn
self.g = nn.Parameter(torch.zeros(1)) self.g = nn.Parameter(torch.zeros(1))
def forward(self, x): def forward(self, x):
return x * self.g return self.fn(x) * self.g
# building block modules # building block modules
class Block(nn.Module): class Block(nn.Module):
def __init__(self, dim, dim_out, groups = 32): def __init__(self, dim, dim_out, groups = 8):
super().__init__() super().__init__()
self.block = nn.Sequential( self.block = nn.Sequential(
nn.Conv2d(dim, dim_out, 3, padding=1), nn.Conv2d(dim, dim_out, 3, padding=1),
@@ -101,7 +143,7 @@ class Block(nn.Module):
return self.block(x) return self.block(x)
class ResnetBlock(nn.Module): class ResnetBlock(nn.Module):
def __init__(self, dim, dim_out, *, time_emb_dim, groups = 32): def __init__(self, dim, dim_out, *, time_emb_dim, groups = 8):
super().__init__() super().__init__()
self.mlp = nn.Sequential( self.mlp = nn.Sequential(
Mish(), Mish(),
@@ -119,18 +161,17 @@ class ResnetBlock(nn.Module):
return h + self.res_conv(x) return h + self.res_conv(x)
class LinearAttention(nn.Module): class LinearAttention(nn.Module):
def __init__(self, dim, heads = 8, dim_head = 32): def __init__(self, dim, heads = 4, dim_head = 32):
super().__init__() super().__init__()
self.heads = heads self.heads = heads
hidden_dim = dim_head * heads hidden_dim = dim_head * heads
self.to_qkv = nn.Conv2d(dim, hidden_dim, 1, bias = False) self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
self.to_out = nn.Conv2d(hidden_dim, dim, 1) self.to_out = nn.Conv2d(hidden_dim, dim, 1)
def forward(self, x): def forward(self, x):
b, c, h, w = x.shape b, c, h, w = x.shape
qkv = self.to_qkv(x) qkv = self.to_qkv(x)
q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads) q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads, qkv=3)
q = q.softmax(dim=-2)
k = k.softmax(dim=-1) k = k.softmax(dim=-1)
context = torch.einsum('bhdn,bhen->bhde', k, v) context = torch.einsum('bhdn,bhen->bhde', k, v)
out = torch.einsum('bhde,bhdn->bhen', context, q) out = torch.einsum('bhde,bhdn->bhen', context, q)
@@ -140,7 +181,7 @@ class LinearAttention(nn.Module):
# model # model
class Unet(nn.Module): class Unet(nn.Module):
def __init__(self, dim, out_dim = None, dim_mults=(1, 2, 4, 8), groups = 32): def __init__(self, dim, out_dim = None, dim_mults=(1, 2, 4, 8), groups = 8):
super().__init__() super().__init__()
dims = [3, *map(lambda m: dim * m, dim_mults)] dims = [3, *map(lambda m: dim * m, dim_mults)]
in_out = list(zip(dims[:-1], dims[1:])) in_out = list(zip(dims[:-1], dims[1:]))
@@ -161,6 +202,7 @@ class Unet(nn.Module):
self.downs.append(nn.ModuleList([ self.downs.append(nn.ModuleList([
ResnetBlock(dim_in, dim_out, time_emb_dim = dim), ResnetBlock(dim_in, dim_out, time_emb_dim = dim),
ResnetBlock(dim_out, dim_out, time_emb_dim = dim),
Residual(Rezero(LinearAttention(dim_out))), Residual(Rezero(LinearAttention(dim_out))),
Downsample(dim_out) if not is_last else nn.Identity() Downsample(dim_out) if not is_last else nn.Identity()
])) ]))
@@ -175,6 +217,7 @@ class Unet(nn.Module):
self.ups.append(nn.ModuleList([ self.ups.append(nn.ModuleList([
ResnetBlock(dim_out * 2, dim_in, time_emb_dim = dim), ResnetBlock(dim_out * 2, dim_in, time_emb_dim = dim),
ResnetBlock(dim_in, dim_in, time_emb_dim = dim),
Residual(Rezero(LinearAttention(dim_in))), Residual(Rezero(LinearAttention(dim_in))),
Upsample(dim_in) if not is_last else nn.Identity() Upsample(dim_in) if not is_last else nn.Identity()
])) ]))
@@ -191,8 +234,9 @@ class Unet(nn.Module):
h = [] h = []
for resnet, attn, downsample in self.downs: for resnet, resnet2, attn, downsample in self.downs:
x = resnet(x, t) x = resnet(x, t)
x = resnet2(x, t)
x = attn(x) x = attn(x)
h.append(x) h.append(x)
x = downsample(x) x = downsample(x)
@@ -201,9 +245,10 @@ class Unet(nn.Module):
x = self.mid_attn(x) x = self.mid_attn(x)
x = self.mid_block2(x, t) x = self.mid_block2(x, t)
for resnet, attn, upsample in self.ups: for resnet, resnet2, attn, upsample in self.ups:
x = torch.cat((x, h.pop()), dim=1) x = torch.cat((x, h.pop()), dim=1)
x = resnet(x, t) x = resnet(x, t)
x = resnet2(x, t)
x = attn(x) x = attn(x)
x = upsample(x) x = upsample(x)
@@ -221,20 +266,36 @@ def noise_like(shape, device, repeat=False):
noise = lambda: torch.randn(shape, device=device) noise = lambda: torch.randn(shape, device=device)
return repeat_noise() if repeat else noise() return repeat_noise() if repeat else noise()
def cosine_beta_schedule(timesteps, s = 0.008):
"""
cosine schedule
as proposed in https://openreview.net/forum?id=-NEXDKk8gZ
"""
steps = timesteps + 1
x = np.linspace(0, steps, steps)
alphas_cumprod = np.cos(((x / steps) + s) / (1 + s) * np.pi * 0.5) ** 2
alphas_cumprod = alphas_cumprod / alphas_cumprod[0]
betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1])
return np.clip(betas, a_min = 0, a_max = 0.999)
class GaussianDiffusion(nn.Module): class GaussianDiffusion(nn.Module):
def __init__(self, denoise_fn, beta_start=0.0001, beta_end=0.02, num_diffusion_timesteps=1000, loss_type='l1'): def __init__(self, denoise_fn, timesteps=1000, loss_type='l1', betas = None):
super().__init__() super().__init__()
self.denoise_fn = denoise_fn self.denoise_fn = denoise_fn
self.np_betas = betas = np.linspace(beta_start, beta_end, num_diffusion_timesteps).astype(np.float64) if exists(betas):
timesteps, = betas.shape betas = betas.detach().cpu().numpy() if isinstance(betas, torch.Tensor) else betas
self.num_timesteps = int(timesteps) else:
self.loss_type = loss_type betas = cosine_beta_schedule(timesteps)
alphas = 1. - betas alphas = 1. - betas
alphas_cumprod = np.cumprod(alphas, axis=0) alphas_cumprod = np.cumprod(alphas, axis=0)
alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1]) alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1])
timesteps, = betas.shape
self.num_timesteps = int(timesteps)
self.loss_type = loss_type
to_torch = partial(torch.tensor, dtype=torch.float32) to_torch = partial(torch.tensor, dtype=torch.float32)
self.register_buffer('betas', to_torch(betas)) self.register_buffer('betas', to_torch(betas))
@@ -311,7 +372,7 @@ class GaussianDiffusion(nn.Module):
@torch.no_grad() @torch.no_grad()
def sample(self, image_size, batch_size = 16): def sample(self, image_size, batch_size = 16):
return self.p_sample_loop((16, 3, image_size, image_size)) return self.p_sample_loop((batch_size, 3, image_size, image_size))
@torch.no_grad() @torch.no_grad()
def interpolate(self, x1, x2, t = None, lam = 0.5): def interpolate(self, x1, x2, t = None, lam = 0.5):
@@ -390,14 +451,22 @@ class Trainer(object):
diffusion_model, diffusion_model,
folder, folder,
*, *,
ema_decay = 0.995,
image_size = 128, image_size = 128,
train_batch_size = 32, train_batch_size = 32,
train_lr = 2e-5, train_lr = 2e-5,
train_num_steps = 100000, train_num_steps = 100000,
gradient_accumulate_every = 2 gradient_accumulate_every = 2,
fp16 = False,
step_start_ema = 2000
): ):
super().__init__() super().__init__()
self.model = diffusion_model self.model = diffusion_model
self.ema = EMA(ema_decay)
self.ema_model = copy.deepcopy(self.model)
self.step_start_ema = step_start_ema
self.batch_size = train_batch_size
self.image_size = image_size self.image_size = image_size
self.gradient_accumulate_every = gradient_accumulate_every self.gradient_accumulate_every = gradient_accumulate_every
self.train_num_steps = train_num_steps self.train_num_steps = train_num_steps
@@ -406,25 +475,64 @@ class Trainer(object):
self.dl = cycle(data.DataLoader(self.ds, batch_size = train_batch_size, shuffle=True, pin_memory=True)) self.dl = cycle(data.DataLoader(self.ds, batch_size = train_batch_size, shuffle=True, pin_memory=True))
self.opt = Adam(diffusion_model.parameters(), lr=train_lr) self.opt = Adam(diffusion_model.parameters(), lr=train_lr)
def train(self): self.step = 0
ind = 0
while ind < self.train_num_steps: assert not fp16 or fp16 and APEX_AVAILABLE, 'Apex must be installed in order for mixed precision training to be turned on'
self.fp16 = fp16
if fp16:
(self.model, self.ema_model), self.opt = amp.initialize([self.model, self.ema_model], self.opt, opt_level='O1')
self.reset_parameters()
def reset_parameters(self):
self.ema_model.load_state_dict(self.model.state_dict())
def step_ema(self):
if self.step < self.step_start_ema:
self.reset_parameters()
return
self.ema.update_model_average(self.ema_model, self.model)
def save(self, milestone):
data = {
'step': self.step,
'model': self.model.state_dict(),
'ema': self.ema_model.state_dict()
}
torch.save(data, str(RESULTS_FOLDER / f'model-{milestone}.pt'))
def load(self, milestone):
data = torch.load(str(RESULTS_FOLDER / 'model-{milestone}.pt'))
self.step = data['step']
self.model.load_state_dict(data['model'])
self.ema_model.load_state_dict(data['ema'])
def train(self):
backwards = partial(loss_backwards, self.fp16)
while self.step < self.train_num_steps:
for i in range(self.gradient_accumulate_every): for i in range(self.gradient_accumulate_every):
data = next(self.dl).cuda() data = next(self.dl).cuda()
loss = self.model(data) loss = self.model(data)
print(f'{ind}: {loss.item()}') print(f'{self.step}: {loss.item()}')
(loss / self.gradient_accumulate_every).backward() backwards(loss / self.gradient_accumulate_every, self.opt)
self.opt.step() self.opt.step()
self.opt.zero_grad() self.opt.zero_grad()
if ind % SAVE_AND_SAMPLE_EVERY == 0: if self.step % UPDATE_EMA_EVERY == 0:
milestone = ind // SAVE_AND_SAMPLE_EVERY self.step_ema()
all_images = self.model.p_sample_loop((64, 3, self.image_size, self.image_size))
utils.save_image(all_images, f'./sample-{milestone}.png', nrow=8)
torch.save(self.model.state_dict(), f'./model-{milestone}.pt')
ind += 1 if self.step != 0 and self.step % SAVE_AND_SAMPLE_EVERY == 0:
milestone = self.step // SAVE_AND_SAMPLE_EVERY
batches = num_to_groups(36, self.batch_size)
all_images_list = list(map(lambda n: self.ema_model.sample(self.image_size, batch_size=n), batches))
all_images = torch.cat(all_images_list, dim=0)
utils.save_image(all_images, str(RESULTS_FOLDER / 'sample-{milestone}.png'), nrow=6)
self.save(milestone)
self.step += 1
print('training completed') print('training completed')
BIN
View File
Binary file not shown.

After

Width:  |  Height:  |  Size: 1.3 MiB

+1 -1
View File
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
setup( setup(
name = 'denoising-diffusion-pytorch', name = 'denoising-diffusion-pytorch',
packages = find_packages(), packages = find_packages(),
version = '0.1.4', version = '0.5.1',
license='MIT', license='MIT',
description = 'Denoising Diffusion Probabilistic Models - Pytorch', description = 'Denoising Diffusion Probabilistic Models - Pytorch',
author = 'Phil Wang', author = 'Phil Wang',