Compare commits

..
11 Commits
3 changed files with 154 additions and 82 deletions
+2 -2
View File
@@ -34,7 +34,7 @@ diffusion = GaussianDiffusion(
loss_type = 'l1' # L1 or L2 loss_type = 'l1' # L1 or L2
) )
training_images = torch.randn(8, 3, 128, 128) training_images = torch.randn(8, 3, 128, 128) # your images need to be normalized from a range of -1 to +1
loss = diffusion(training_images) loss = diffusion(training_images)
loss.backward() loss.backward()
# after a lot of training # after a lot of training
@@ -68,7 +68,7 @@ trainer = Trainer(
train_num_steps = 700000, # total training steps train_num_steps = 700000, # total training steps
gradient_accumulate_every = 2, # gradient accumulation steps gradient_accumulate_every = 2, # gradient accumulation steps
ema_decay = 0.995, # exponential moving average decay ema_decay = 0.995, # exponential moving average decay
fp16 = True # turn on mixed precision training with apex amp = True # turn on mixed precision
) )
trainer.train() trainer.train()
@@ -7,21 +7,16 @@ from inspect import isfunction
from functools import partial from functools import partial
from torch.utils import data from torch.utils import data
from torch.cuda.amp import autocast, GradScaler
from pathlib import Path from pathlib import Path
from torch.optim import Adam from torch.optim import Adam
from torchvision import transforms, utils from torchvision import transforms, utils
from PIL import Image from PIL import Image
import numpy as np
from tqdm import tqdm from tqdm import tqdm
from einops import rearrange from einops import rearrange
try:
from apex import amp
APEX_AVAILABLE = True
except:
APEX_AVAILABLE = False
# helpers functions # helpers functions
def exists(x): def exists(x):
@@ -45,13 +40,6 @@ def num_to_groups(num, divisor):
arr.append(remainder) arr.append(remainder)
return arr return arr
def loss_backwards(fp16, loss, optimizer, **kwargs):
if fp16:
with amp.scale_loss(loss, optimizer) as scaled_loss:
scaled_loss.backward(**kwargs)
else:
loss.backward(**kwargs)
# small helper modules # small helper modules
class EMA(): class EMA():
@@ -121,6 +109,39 @@ class PreNorm(nn.Module):
# building block modules # building block modules
class Block(nn.Module):
def __init__(self, dim, dim_out, groups = 8):
super().__init__()
self.block = nn.Sequential(
nn.Conv2d(dim, dim_out, 3, padding = 1),
nn.GroupNorm(groups, dim_out),
nn.SiLU()
)
def forward(self, x):
return self.block(x)
class ResnetBlock(nn.Module):
def __init__(self, dim, dim_out, *, time_emb_dim = None, groups = 8):
super().__init__()
self.mlp = nn.Sequential(
nn.SiLU(),
nn.Linear(time_emb_dim, dim_out)
) if exists(time_emb_dim) else None
self.block1 = Block(dim, dim_out)
self.block2 = Block(dim_out, dim_out)
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
def forward(self, x, time_emb = None):
h = self.block1(x)
if exists(self.mlp) and exists(time_emb):
time_emb = self.mlp(time_emb)
h = rearrange(time_emb, 'b c -> b c 1 1') + h
h = self.block2(h)
return h + self.res_conv(x)
class ConvNextBlock(nn.Module): class ConvNextBlock(nn.Module):
""" https://arxiv.org/abs/2201.03545 """ """ https://arxiv.org/abs/2201.03545 """
@@ -137,6 +158,7 @@ class ConvNextBlock(nn.Module):
LayerNorm(dim) if norm else nn.Identity(), LayerNorm(dim) if norm else nn.Identity(),
nn.Conv2d(dim, dim_out * mult, 3, padding = 1), nn.Conv2d(dim, dim_out * mult, 3, padding = 1),
nn.GELU(), nn.GELU(),
LayerNorm(dim_out * mult),
nn.Conv2d(dim_out * mult, dim_out, 3, padding = 1) nn.Conv2d(dim_out * mult, dim_out, 3, padding = 1)
) )
@@ -145,7 +167,7 @@ class ConvNextBlock(nn.Module):
def forward(self, x, time_emb = None): def forward(self, x, time_emb = None):
h = self.ds_conv(x) h = self.ds_conv(x)
if exists(self.mlp): if exists(self.mlp) and exists(time_emb):
assert exists(time_emb), 'time emb must be passed in' assert exists(time_emb), 'time emb must be passed in'
condition = self.mlp(time_emb) condition = self.mlp(time_emb)
h = h + rearrange(condition, 'b c -> b c 1 1') h = h + rearrange(condition, 'b c -> b c 1 1')
@@ -154,6 +176,34 @@ class ConvNextBlock(nn.Module):
return h + self.res_conv(x) return h + self.res_conv(x)
class LinearAttention(nn.Module): class LinearAttention(nn.Module):
def __init__(self, dim, heads = 4, dim_head = 32):
super().__init__()
self.scale = dim_head ** -0.5
self.heads = heads
hidden_dim = dim_head * heads
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
self.to_out = nn.Sequential(
nn.Conv2d(hidden_dim, dim, 1),
LayerNorm(dim)
)
def forward(self, x):
b, c, h, w = x.shape
qkv = self.to_qkv(x).chunk(3, dim = 1)
q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
q = q.softmax(dim = -2)
k = k.softmax(dim = -1)
q = q * self.scale
context = torch.einsum('b h d n, b h e n -> b h d e', k, v)
out = torch.einsum('b h d e, b h d n -> b h e n', context, q)
out = rearrange(out, 'b h c (x y) -> b (h c) x y', h = self.heads, x = h, y = w)
return self.to_out(out)
class Attention(nn.Module):
def __init__(self, dim, heads = 4, dim_head = 32): def __init__(self, dim, heads = 4, dim_head = 32):
super().__init__() super().__init__()
self.scale = dim_head ** -0.5 self.scale = dim_head ** -0.5
@@ -168,11 +218,12 @@ class LinearAttention(nn.Module):
q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv) q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
q = q * self.scale q = q * self.scale
k = k.softmax(dim = -1) sim = einsum('b h d i, b h d j -> b h i j', q, k)
context = torch.einsum('b h d n, b h e n -> b h d e', k, v) sim = sim - sim.amax(dim = -1, keepdim = True).detach()
attn = sim.softmax(dim = -1)
out = torch.einsum('b h d e, b h d n -> b h e n', context, q) out = einsum('b h i j, b h d j -> b h i d', attn, v)
out = rearrange(out, 'b h c (x y) -> b (h c) x y', h = self.heads, x = h, y = w) out = rearrange(out, 'b h (x y) d -> b (h d) x y', x = h, y = w)
return self.to_out(out) return self.to_out(out)
# model # model
@@ -181,29 +232,50 @@ class Unet(nn.Module):
def __init__( def __init__(
self, self,
dim, dim,
init_dim = None,
out_dim = None, out_dim = None,
dim_mults=(1, 2, 4, 8), dim_mults=(1, 2, 4, 8),
channels = 3, channels = 3,
with_time_emb = True with_time_emb = True,
use_convnext = False,
resnet_block_groups = 8,
convnext_mult = 2
): ):
super().__init__() super().__init__()
# determine dimensions
self.channels = channels self.channels = channels
dims = [channels, *map(lambda m: dim * m, dim_mults)] init_dim = default(init_dim, dim // 3 * 2)
self.init_conv = nn.Conv2d(channels, init_dim, 7, padding = 3)
dims = [init_dim, *map(lambda m: dim * m, dim_mults)]
in_out = list(zip(dims[:-1], dims[1:])) in_out = list(zip(dims[:-1], dims[1:]))
# resnet or convnext
if use_convnext:
block_klass = partial(ConvNextBlock, mult = convnext_mult)
else:
block_klass = partial(ResnetBlock, groups = resnet_block_groups)
# time embeddings
if with_time_emb: if with_time_emb:
time_dim = dim time_dim = dim * 4
self.time_mlp = nn.Sequential( self.time_mlp = nn.Sequential(
SinusoidalPosEmb(dim), SinusoidalPosEmb(dim),
nn.Linear(dim, dim * 4), nn.Linear(dim, time_dim),
nn.GELU(), nn.GELU(),
nn.Linear(dim * 4, dim) nn.Linear(time_dim, time_dim)
) )
else: else:
time_dim = None time_dim = None
self.time_mlp = None self.time_mlp = None
# layers
self.downs = nn.ModuleList([]) self.downs = nn.ModuleList([])
self.ups = nn.ModuleList([]) self.ups = nn.ModuleList([])
num_resolutions = len(in_out) num_resolutions = len(in_out)
@@ -212,41 +284,43 @@ class Unet(nn.Module):
is_last = ind >= (num_resolutions - 1) is_last = ind >= (num_resolutions - 1)
self.downs.append(nn.ModuleList([ self.downs.append(nn.ModuleList([
ConvNextBlock(dim_in, dim_out, time_emb_dim = time_dim, norm = ind != 0), block_klass(dim_in, dim_out, time_emb_dim = time_dim),
ConvNextBlock(dim_out, dim_out, time_emb_dim = time_dim), block_klass(dim_out, dim_out, time_emb_dim = time_dim),
Residual(PreNorm(dim_out, LinearAttention(dim_out))), Residual(PreNorm(dim_out, LinearAttention(dim_out))),
Downsample(dim_out) if not is_last else nn.Identity() Downsample(dim_out) if not is_last else nn.Identity()
])) ]))
mid_dim = dims[-1] mid_dim = dims[-1]
self.mid_block1 = ConvNextBlock(mid_dim, mid_dim, time_emb_dim = time_dim) self.mid_block1 = block_klass(mid_dim, mid_dim, time_emb_dim = time_dim)
self.mid_attn = Residual(PreNorm(mid_dim, LinearAttention(mid_dim))) self.mid_attn = Residual(PreNorm(mid_dim, Attention(mid_dim)))
self.mid_block2 = ConvNextBlock(mid_dim, mid_dim, time_emb_dim = time_dim) self.mid_block2 = block_klass(mid_dim, mid_dim, time_emb_dim = time_dim)
for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])): for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])):
is_last = ind >= (num_resolutions - 1) is_last = ind >= (num_resolutions - 1)
self.ups.append(nn.ModuleList([ self.ups.append(nn.ModuleList([
ConvNextBlock(dim_out * 2, dim_in, time_emb_dim = time_dim), block_klass(dim_out * 2, dim_in, time_emb_dim = time_dim),
ConvNextBlock(dim_in, dim_in, time_emb_dim = time_dim), block_klass(dim_in, dim_in, time_emb_dim = time_dim),
Residual(PreNorm(dim_in, LinearAttention(dim_in))), Residual(PreNorm(dim_in, LinearAttention(dim_in))),
Upsample(dim_in) if not is_last else nn.Identity() Upsample(dim_in) if not is_last else nn.Identity()
])) ]))
out_dim = default(out_dim, channels) out_dim = default(out_dim, channels)
self.final_conv = nn.Sequential( self.final_conv = nn.Sequential(
ConvNextBlock(dim, dim), block_klass(dim, dim),
nn.Conv2d(dim, out_dim, 1) nn.Conv2d(dim, out_dim, 1)
) )
def forward(self, x, time): def forward(self, x, time):
x = self.init_conv(x)
t = self.time_mlp(time) if exists(self.time_mlp) else None t = self.time_mlp(time) if exists(self.time_mlp) else None
h = [] h = []
for convnext, convnext2, attn, downsample in self.downs: for block1, block2, attn, downsample in self.downs:
x = convnext(x, t) x = block1(x, t)
x = convnext2(x, t) x = block2(x, t)
x = attn(x) x = attn(x)
h.append(x) h.append(x)
x = downsample(x) x = downsample(x)
@@ -255,10 +329,10 @@ class Unet(nn.Module):
x = self.mid_attn(x) x = self.mid_attn(x)
x = self.mid_block2(x, t) x = self.mid_block2(x, t)
for convnext, convnext2, attn, upsample in self.ups: for block1, block2, attn, upsample in self.ups:
x = torch.cat((x, h.pop()), dim=1) x = torch.cat((x, h.pop()), dim=1)
x = convnext(x, t) x = block1(x, t)
x = convnext2(x, t) x = block2(x, t)
x = attn(x) x = attn(x)
x = upsample(x) x = upsample(x)
@@ -282,11 +356,11 @@ def cosine_beta_schedule(timesteps, s = 0.008):
as proposed in https://openreview.net/forum?id=-NEXDKk8gZ as proposed in https://openreview.net/forum?id=-NEXDKk8gZ
""" """
steps = timesteps + 1 steps = timesteps + 1
x = np.linspace(0, steps, steps) x = torch.linspace(0, timesteps, steps)
alphas_cumprod = np.cos(((x / steps) + s) / (1 + s) * np.pi * 0.5) ** 2 alphas_cumprod = torch.cos(((x / timesteps) + s) / (1 + s) * torch.pi * 0.5) ** 2
alphas_cumprod = alphas_cumprod / alphas_cumprod[0] alphas_cumprod = alphas_cumprod / alphas_cumprod[0]
betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1]) betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1])
return np.clip(betas, a_min = 0, a_max = 0.999) return torch.clip(betas, 0, 0.999)
class GaussianDiffusion(nn.Module): class GaussianDiffusion(nn.Module):
def __init__( def __init__(
@@ -296,50 +370,48 @@ class GaussianDiffusion(nn.Module):
image_size, image_size,
channels = 3, channels = 3,
timesteps = 1000, timesteps = 1000,
loss_type = 'l1', loss_type = 'l1'
betas = None
): ):
super().__init__() super().__init__()
self.channels = channels self.channels = channels
self.image_size = image_size self.image_size = image_size
self.denoise_fn = denoise_fn self.denoise_fn = denoise_fn
if exists(betas): betas = cosine_beta_schedule(timesteps)
betas = betas.detach().cpu().numpy() if isinstance(betas, torch.Tensor) else betas
else:
betas = cosine_beta_schedule(timesteps)
alphas = 1. - betas alphas = 1. - betas
alphas_cumprod = np.cumprod(alphas, axis=0) alphas_cumprod = torch.cumprod(alphas, axis=0)
alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1]) alphas_cumprod_prev = F.pad(alphas_cumprod[:-1], (1, 0), value = 1.)
timesteps, = betas.shape timesteps, = betas.shape
self.num_timesteps = int(timesteps) self.num_timesteps = int(timesteps)
self.loss_type = loss_type self.loss_type = loss_type
to_torch = partial(torch.tensor, dtype=torch.float32) self.register_buffer('betas', betas)
self.register_buffer('alphas_cumprod', alphas_cumprod)
self.register_buffer('betas', to_torch(betas)) self.register_buffer('alphas_cumprod_prev', alphas_cumprod_prev)
self.register_buffer('alphas_cumprod', to_torch(alphas_cumprod))
self.register_buffer('alphas_cumprod_prev', to_torch(alphas_cumprod_prev))
# calculations for diffusion q(x_t | x_{t-1}) and others # calculations for diffusion q(x_t | x_{t-1}) and others
self.register_buffer('sqrt_alphas_cumprod', to_torch(np.sqrt(alphas_cumprod)))
self.register_buffer('sqrt_one_minus_alphas_cumprod', to_torch(np.sqrt(1. - alphas_cumprod))) self.register_buffer('sqrt_alphas_cumprod', torch.sqrt(alphas_cumprod))
self.register_buffer('log_one_minus_alphas_cumprod', to_torch(np.log(1. - alphas_cumprod))) self.register_buffer('sqrt_one_minus_alphas_cumprod', torch.sqrt(1. - alphas_cumprod))
self.register_buffer('sqrt_recip_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod))) self.register_buffer('log_one_minus_alphas_cumprod', torch.log(1. - alphas_cumprod))
self.register_buffer('sqrt_recipm1_alphas_cumprod', to_torch(np.sqrt(1. / alphas_cumprod - 1))) self.register_buffer('sqrt_recip_alphas_cumprod', torch.sqrt(1. / alphas_cumprod))
self.register_buffer('sqrt_recipm1_alphas_cumprod', torch.sqrt(1. / alphas_cumprod - 1))
# calculations for posterior q(x_{t-1} | x_t, x_0) # calculations for posterior q(x_{t-1} | x_t, x_0)
posterior_variance = betas * (1. - alphas_cumprod_prev) / (1. - alphas_cumprod) posterior_variance = betas * (1. - alphas_cumprod_prev) / (1. - alphas_cumprod)
# above: equal to 1. / (1. / (1. - alpha_cumprod_tm1) + alpha_t / beta_t) # above: equal to 1. / (1. / (1. - alpha_cumprod_tm1) + alpha_t / beta_t)
self.register_buffer('posterior_variance', to_torch(posterior_variance))
self.register_buffer('posterior_variance', posterior_variance)
# below: log calculation clipped because the posterior variance is 0 at the beginning of the diffusion chain # below: log calculation clipped because the posterior variance is 0 at the beginning of the diffusion chain
self.register_buffer('posterior_log_variance_clipped', to_torch(np.log(np.maximum(posterior_variance, 1e-20))))
self.register_buffer('posterior_mean_coef1', to_torch( self.register_buffer('posterior_log_variance_clipped', torch.log(posterior_variance.clamp(min =1e-20)))
betas * np.sqrt(alphas_cumprod_prev) / (1. - alphas_cumprod))) self.register_buffer('posterior_mean_coef1', betas * torch.sqrt(alphas_cumprod_prev) / (1. - alphas_cumprod))
self.register_buffer('posterior_mean_coef2', to_torch( self.register_buffer('posterior_mean_coef2', (1. - alphas_cumprod_prev) * torch.sqrt(alphas) / (1. - alphas_cumprod))
(1. - alphas_cumprod_prev) * np.sqrt(alphas) / (1. - alphas_cumprod)))
def q_mean_variance(self, x_start, t): def q_mean_variance(self, x_start, t):
mean = extract(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start mean = extract(self.sqrt_alphas_cumprod, t, x_start.shape) * x_start
@@ -482,7 +554,7 @@ class Trainer(object):
train_lr = 2e-5, train_lr = 2e-5,
train_num_steps = 100000, train_num_steps = 100000,
gradient_accumulate_every = 2, gradient_accumulate_every = 2,
fp16 = False, amp = False,
step_start_ema = 2000, step_start_ema = 2000,
update_ema_every = 10, update_ema_every = 10,
save_and_sample_every = 1000, save_and_sample_every = 1000,
@@ -508,11 +580,8 @@ class Trainer(object):
self.step = 0 self.step = 0
assert not fp16 or fp16 and APEX_AVAILABLE, 'Apex must be installed in order for mixed precision training to be turned on' self.amp = amp
self.scaler = GradScaler(enabled = amp)
self.fp16 = fp16
if fp16:
(self.model, self.ema_model), self.opt = amp.initialize([self.model, self.ema_model], self.opt, opt_level='O1')
self.results_folder = Path(results_folder) self.results_folder = Path(results_folder)
self.results_folder.mkdir(exist_ok = True) self.results_folder.mkdir(exist_ok = True)
@@ -532,7 +601,8 @@ class Trainer(object):
data = { data = {
'step': self.step, 'step': self.step,
'model': self.model.state_dict(), 'model': self.model.state_dict(),
'ema': self.ema_model.state_dict() 'ema': self.ema_model.state_dict(),
'scaler': self.scaler.state_dict()
} }
torch.save(data, str(self.results_folder / f'model-{milestone}.pt')) torch.save(data, str(self.results_folder / f'model-{milestone}.pt'))
@@ -542,18 +612,21 @@ class Trainer(object):
self.step = data['step'] self.step = data['step']
self.model.load_state_dict(data['model']) self.model.load_state_dict(data['model'])
self.ema_model.load_state_dict(data['ema']) self.ema_model.load_state_dict(data['ema'])
self.scaler.load_state_dict(data['scaler'])
def train(self): def train(self):
backwards = partial(loss_backwards, self.fp16)
while self.step < self.train_num_steps: while self.step < self.train_num_steps:
for i in range(self.gradient_accumulate_every): for i in range(self.gradient_accumulate_every):
data = next(self.dl).cuda() data = next(self.dl).cuda()
loss = self.model(data)
print(f'{self.step}: {loss.item()}')
backwards(loss / self.gradient_accumulate_every, self.opt)
self.opt.step() with autocast(enabled = self.amp):
loss = self.model(data)
self.scaler.scale(loss / self.gradient_accumulate_every).backward()
print(f'{self.step}: {loss.item()}')
self.scaler.step(self.opt)
self.scaler.update()
self.opt.zero_grad() self.opt.zero_grad()
if self.step % self.update_ema_every == 0: if self.step % self.update_ema_every == 0:
+1 -2
View File
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
setup( setup(
name = 'denoising-diffusion-pytorch', name = 'denoising-diffusion-pytorch',
packages = find_packages(), packages = find_packages(),
version = '0.7.1', version = '0.11.1',
license='MIT', license='MIT',
description = 'Denoising Diffusion Probabilistic Models - Pytorch', description = 'Denoising Diffusion Probabilistic Models - Pytorch',
author = 'Phil Wang', author = 'Phil Wang',
@@ -15,7 +15,6 @@ setup(
], ],
install_requires=[ install_requires=[
'einops', 'einops',
'numpy',
'pillow', 'pillow',
'torch', 'torch',
'torchvision', 'torchvision',