Compare commits

..
16 Commits
Author SHA1 Message Date
Phil Wang 91cff45939 replace resnets with convnext blocks 2022-01-25 09:02:45 -08:00
Phil Wang 7b51e30da7 fix layernorm 2021-08-24 14:28:15 -07:00
Phil Wang dadbf20154 remove stray print 2021-07-16 15:18:20 -07:00
Phil Wang 7706bdfc6f use pre-layernorm with linear attention, and also allow for turning off time embedding 2021-06-25 10:48:11 -07:00
Phil Wang 183e5f3cc5 move all constants into configurable class init parameters 2021-06-25 10:37:36 -07:00
Phil Wang 16c9ae7bb3 fix data not being normalized to range of -1 to 1 2021-06-21 18:51:36 -07:00
Phil Wang f5916111f8 0.6.3 2021-06-21 17:41:16 -07:00
Phil Wang ad9e303ff3 fix channels 2021-06-11 15:28:36 -07:00
Phil Wang ae42f48f6a prepare so that unet can work with a channel of one, and also make it so image size is hard coded in diffusion class. preparing for training on protein distograms 2021-06-11 14:06:39 -07:00
Phil Wang 5989f4c77e recommit sample 2020-10-13 09:32:56 -07:00
Phil Wang 2082046888 set higher num train steps, so non-practitioners do not think it is completed 2020-10-11 13:49:49 -07:00
Phil Wang 3c5b7e2d56 update readme 2020-10-10 10:37:44 -07:00
Phil Wang d4ce9f6c38 save samples and models to ./results path 2020-10-09 21:50:23 -07:00
Phil Wang ff451f697e update with new and improved cosine noise scheduler 2020-10-09 21:21:02 -07:00
Phil Wang 3d96532c60 update citations in preparation to add improvements from a iclr 2021 paper 2020-10-05 17:15:28 -07:00
Phil Wang ef2ca0b625 new paper suggests image linear attention is more effective without query normalization 2020-10-04 21:53:53 -07:00
5 changed files with 196 additions and 120 deletions
+3
View File
@@ -1,3 +1,6 @@
# Generation results
results/
# Byte-compiled / optimized / DLL files # Byte-compiled / optimized / DLL files
__pycache__/ __pycache__/
*.py[cod] *.py[cod]
+41 -18
View File
@@ -2,7 +2,9 @@
## Denoising Diffusion Probabilistic Model, in Pytorch ## Denoising Diffusion Probabilistic Model, in Pytorch
Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution. This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a>. Implementation of <a href="https://arxiv.org/abs/2006.11239">Denoising Diffusion Probabilistic Model</a> in Pytorch. It is a new approach to generative modeling that may <a href="https://ajolicoeur.wordpress.com/the-new-contender-to-gans-score-matching-with-langevin-sampling/">have the potential</a> to rival GANs. It uses denoising score matching to estimate the gradient of the data distribution, followed by Langevin sampling to sample from the true distribution.
This implementation was transcribed from the official Tensorflow version <a href="https://github.com/hojonathanho/diffusion">here</a> and then modified to use <a href="https://arxiv.org/abs/2201.03545">ConvNext</a> blocks instead of Resnets.
<img src="./sample.png" width="500px"><img> <img src="./sample.png" width="500px"><img>
@@ -27,10 +29,9 @@ model = Unet(
diffusion = GaussianDiffusion( diffusion = GaussianDiffusion(
model, model,
beta_start = 0.0001, image_size = 128,
beta_end = 0.02, timesteps = 1000, # number of steps
num_diffusion_timesteps = 1000, # number of steps loss_type = 'l1' # L1 or L2
loss_type = 'l1' # L1 or L2 (wavegrad paper claims l1 is better?)
) )
training_images = torch.randn(8, 3, 128, 128) training_images = torch.randn(8, 3, 128, 128)
@@ -38,7 +39,7 @@ loss = diffusion(training_images)
loss.backward() loss.backward()
# after a lot of training # after a lot of training
sampled_images = diffusion.sample(128, batch_size = 4) sampled_images = diffusion.sample(batch_size = 4)
sampled_images.shape # (4, 3, 128, 128) sampled_images.shape # (4, 3, 128, 128)
``` ```
@@ -54,19 +55,17 @@ model = Unet(
diffusion = GaussianDiffusion( diffusion = GaussianDiffusion(
model, model,
beta_start = 0.0001, image_size = 128,
beta_end = 0.02, timesteps = 1000, # number of steps
num_diffusion_timesteps = 1000, # number of steps loss_type = 'l1' # L1 or L2
loss_type = 'l1' # L1 or L2
).cuda() ).cuda()
trainer = Trainer( trainer = Trainer(
diffusion, diffusion,
'path/to/your/images', 'path/to/your/images',
image_size = 128,
train_batch_size = 32, train_batch_size = 32,
train_lr = 2e-5, train_lr = 2e-5,
train_num_steps = 100000, # total training steps train_num_steps = 700000, # total training steps
gradient_accumulate_every = 2, # gradient accumulation steps gradient_accumulate_every = 2, # gradient accumulation steps
ema_decay = 0.995, # exponential moving average decay ema_decay = 0.995, # exponential moving average decay
fp16 = True # turn on mixed precision training with apex fp16 = True # turn on mixed precision training with apex
@@ -75,15 +74,39 @@ trainer = Trainer(
trainer.train() trainer.train()
``` ```
Samples and model checkpoints will be logged to `./results` periodically
## Citations ## Citations
```bibtex ```bibtex
@misc{ho2020denoising, @misc{ho2020denoising,
title={Denoising Diffusion Probabilistic Models}, title = {Denoising Diffusion Probabilistic Models},
author={Jonathan Ho and Ajay Jain and Pieter Abbeel}, author = {Jonathan Ho and Ajay Jain and Pieter Abbeel},
year={2020}, year = {2020},
eprint={2006.11239}, eprint = {2006.11239},
archivePrefix={arXiv}, archivePrefix = {arXiv},
primaryClass={cs.LG} primaryClass = {cs.LG}
}
```
```bibtex
@inproceedings{anonymous2021improved,
title = {Improved Denoising Diffusion Probabilistic Models},
author = {Anonymous},
booktitle = {Submitted to International Conference on Learning Representations},
year = {2021},
url = {https://openreview.net/forum?id=-NEXDKk8gZ},
note = {under review}
}
```
```bibtex
@misc{liu2022convnet,
title = {A ConvNet for the 2020s},
author = {Zhuang Liu and Hanzi Mao and Chao-Yuan Wu and Christoph Feichtenhofer and Trevor Darrell and Saining Xie},
year = {2022},
eprint = {2201.03545},
archivePrefix = {arXiv},
primaryClass = {cs.CV}
} }
``` ```
@@ -22,12 +22,6 @@ try:
except: except:
APEX_AVAILABLE = False APEX_AVAILABLE = False
# constants
SAVE_AND_SAMPLE_EVERY = 1000
UPDATE_EMA_EVERY = 10
EXTS = ['jpg', 'png']
# helpers functions # helpers functions
def exists(x): def exists(x):
@@ -97,69 +91,73 @@ class SinusoidalPosEmb(nn.Module):
emb = torch.cat((emb.sin(), emb.cos()), dim=-1) emb = torch.cat((emb.sin(), emb.cos()), dim=-1)
return emb return emb
class Mish(nn.Module): def Upsample(dim):
def forward(self, x): return nn.ConvTranspose2d(dim, dim, 4, 2, 1)
return x * torch.tanh(F.softplus(x))
class Upsample(nn.Module): def Downsample(dim):
def __init__(self, dim): return nn.Conv2d(dim, dim, 3, 2, 1)
class LayerNorm(nn.Module):
def __init__(self, dim, eps = 1e-5):
super().__init__() super().__init__()
self.conv = nn.ConvTranspose2d(dim, dim, 4, 2, 1) self.eps = eps
self.g = nn.Parameter(torch.ones(1, dim, 1, 1))
self.b = nn.Parameter(torch.zeros(1, dim, 1, 1))
def forward(self, x): def forward(self, x):
return self.conv(x) var = torch.var(x, dim = 1, unbiased = False, keepdim = True)
mean = torch.mean(x, dim = 1, keepdim = True)
return (x - mean) / (var + self.eps).sqrt() * self.g + self.b
class Downsample(nn.Module): class PreNorm(nn.Module):
def __init__(self, dim): def __init__(self, dim, fn):
super().__init__()
self.conv = nn.Conv2d(dim, dim, 3, 2, 1)
def forward(self, x):
return self.conv(x)
class Rezero(nn.Module):
def __init__(self, fn):
super().__init__() super().__init__()
self.fn = fn self.fn = fn
self.g = nn.Parameter(torch.zeros(1)) self.norm = LayerNorm(dim)
def forward(self, x): def forward(self, x):
return self.fn(x) * self.g x = self.norm(x)
return self.fn(x)
# building block modules # building block modules
class Block(nn.Module): class ConvNextBlock(nn.Module):
def __init__(self, dim, dim_out, groups = 8): """ https://arxiv.org/abs/2201.03545 """
super().__init__()
self.block = nn.Sequential(
nn.Conv2d(dim, dim_out, 3, padding=1),
nn.GroupNorm(groups, dim_out),
Mish()
)
def forward(self, x):
return self.block(x)
class ResnetBlock(nn.Module): def __init__(self, dim, dim_out, *, time_emb_dim = None, mult = 2, norm = True):
def __init__(self, dim, dim_out, *, time_emb_dim, groups = 8):
super().__init__() super().__init__()
self.mlp = nn.Sequential( self.mlp = nn.Sequential(
Mish(), nn.GELU(),
nn.Linear(time_emb_dim, dim_out) nn.Linear(time_emb_dim, dim)
) if exists(time_emb_dim) else None
self.ds_conv = nn.Conv2d(dim, dim, 7, padding = 3, groups = dim)
self.net = nn.Sequential(
LayerNorm(dim) if norm else nn.Identity(),
nn.Conv2d(dim, dim_out * mult, 1),
nn.GELU(),
LayerNorm(dim_out * mult),
nn.Conv2d(dim_out * mult, dim_out, 1)
) )
self.block1 = Block(dim, dim_out)
self.block2 = Block(dim_out, dim_out)
self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity() self.res_conv = nn.Conv2d(dim, dim_out, 1) if dim != dim_out else nn.Identity()
def forward(self, x, time_emb): def forward(self, x, time_emb = None):
h = self.block1(x) h = self.ds_conv(x)
h += self.mlp(time_emb)[:, :, None, None]
h = self.block2(h) if exists(self.mlp):
assert exists(time_emb), 'time emb must be passed in'
condition = self.mlp(time_emb)
h = h + rearrange(condition, 'b c -> b c 1 1')
h = self.net(h)
return h + self.res_conv(x) return h + self.res_conv(x)
class LinearAttention(nn.Module): class LinearAttention(nn.Module):
def __init__(self, dim, heads = 4, dim_head = 32): def __init__(self, dim, heads = 4, dim_head = 32):
super().__init__() super().__init__()
self.scale = dim_head ** -0.5
self.heads = heads self.heads = heads
hidden_dim = dim_head * heads hidden_dim = dim_head * heads
self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False) self.to_qkv = nn.Conv2d(dim, hidden_dim * 3, 1, bias = False)
@@ -167,29 +165,45 @@ class LinearAttention(nn.Module):
def forward(self, x): def forward(self, x):
b, c, h, w = x.shape b, c, h, w = x.shape
qkv = self.to_qkv(x) qkv = self.to_qkv(x).chunk(3, dim = 1)
q, k, v = rearrange(qkv, 'b (qkv heads c) h w -> qkv b heads c (h w)', heads = self.heads, qkv=3) q, k, v = map(lambda t: rearrange(t, 'b (h c) x y -> b h c (x y)', h = self.heads), qkv)
q = q.softmax(dim=-2) q = q * self.scale
k = k.softmax(dim=-1)
context = torch.einsum('bhdn,bhen->bhde', k, v) k = k.softmax(dim = -1)
out = torch.einsum('bhde,bhdn->bhen', context, q) context = torch.einsum('b h d n, b h e n -> b h d e', k, v)
out = rearrange(out, 'b heads c (h w) -> b (heads c) h w', heads=self.heads, h=h, w=w)
out = torch.einsum('b h d e, b h d n -> b h e n', context, q)
out = rearrange(out, 'b h c (x y) -> b (h c) x y', h = self.heads, x = h, y = w)
return self.to_out(out) return self.to_out(out)
# model # model
class Unet(nn.Module): class Unet(nn.Module):
def __init__(self, dim, out_dim = None, dim_mults=(1, 2, 4, 8), groups = 8): def __init__(
self,
dim,
out_dim = None,
dim_mults=(1, 2, 4, 8),
channels = 3,
with_time_emb = True
):
super().__init__() super().__init__()
dims = [3, *map(lambda m: dim * m, dim_mults)] self.channels = channels
dims = [channels, *map(lambda m: dim * m, dim_mults)]
in_out = list(zip(dims[:-1], dims[1:])) in_out = list(zip(dims[:-1], dims[1:]))
self.time_pos_emb = SinusoidalPosEmb(dim) if with_time_emb:
self.mlp = nn.Sequential( time_dim = dim
nn.Linear(dim, dim * 4), self.time_mlp = nn.Sequential(
Mish(), SinusoidalPosEmb(dim),
nn.Linear(dim * 4, dim) nn.Linear(dim, dim * 4),
) nn.GELU(),
nn.Linear(dim * 4, dim)
)
else:
time_dim = None
self.time_mlp = None
self.downs = nn.ModuleList([]) self.downs = nn.ModuleList([])
self.ups = nn.ModuleList([]) self.ups = nn.ModuleList([])
@@ -199,42 +213,41 @@ class Unet(nn.Module):
is_last = ind >= (num_resolutions - 1) is_last = ind >= (num_resolutions - 1)
self.downs.append(nn.ModuleList([ self.downs.append(nn.ModuleList([
ResnetBlock(dim_in, dim_out, time_emb_dim = dim), ConvNextBlock(dim_in, dim_out, time_emb_dim = time_dim, norm = ind != 0),
ResnetBlock(dim_out, dim_out, time_emb_dim = dim), ConvNextBlock(dim_out, dim_out, time_emb_dim = time_dim),
Residual(Rezero(LinearAttention(dim_out))), Residual(PreNorm(dim_out, LinearAttention(dim_out))),
Downsample(dim_out) if not is_last else nn.Identity() Downsample(dim_out) if not is_last else nn.Identity()
])) ]))
mid_dim = dims[-1] mid_dim = dims[-1]
self.mid_block1 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim) self.mid_block1 = ConvNextBlock(mid_dim, mid_dim, time_emb_dim = time_dim)
self.mid_attn = Residual(Rezero(LinearAttention(mid_dim))) self.mid_attn = Residual(PreNorm(mid_dim, LinearAttention(mid_dim)))
self.mid_block2 = ResnetBlock(mid_dim, mid_dim, time_emb_dim = dim) self.mid_block2 = ConvNextBlock(mid_dim, mid_dim, time_emb_dim = time_dim)
for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])): for ind, (dim_in, dim_out) in enumerate(reversed(in_out[1:])):
is_last = ind >= (num_resolutions - 1) is_last = ind >= (num_resolutions - 1)
self.ups.append(nn.ModuleList([ self.ups.append(nn.ModuleList([
ResnetBlock(dim_out * 2, dim_in, time_emb_dim = dim), ConvNextBlock(dim_out * 2, dim_in, time_emb_dim = time_dim),
ResnetBlock(dim_in, dim_in, time_emb_dim = dim), ConvNextBlock(dim_in, dim_in, time_emb_dim = time_dim),
Residual(Rezero(LinearAttention(dim_in))), Residual(PreNorm(dim_in, LinearAttention(dim_in))),
Upsample(dim_in) if not is_last else nn.Identity() Upsample(dim_in) if not is_last else nn.Identity()
])) ]))
out_dim = default(out_dim, 3) out_dim = default(out_dim, channels)
self.final_conv = nn.Sequential( self.final_conv = nn.Sequential(
Block(dim, dim), ConvNextBlock(dim, dim),
nn.Conv2d(dim, out_dim, 1) nn.Conv2d(dim, out_dim, 1)
) )
def forward(self, x, time): def forward(self, x, time):
t = self.time_pos_emb(time) t = self.time_mlp(time) if exists(self.time_mlp) else None
t = self.mlp(t)
h = [] h = []
for resnet, resnet2, attn, downsample in self.downs: for convnext, convnext2, attn, downsample in self.downs:
x = resnet(x, t) x = convnext(x, t)
x = resnet2(x, t) x = convnext2(x, t)
x = attn(x) x = attn(x)
h.append(x) h.append(x)
x = downsample(x) x = downsample(x)
@@ -243,10 +256,10 @@ class Unet(nn.Module):
x = self.mid_attn(x) x = self.mid_attn(x)
x = self.mid_block2(x, t) x = self.mid_block2(x, t)
for resnet, resnet2, attn, upsample in self.ups: for convnext, convnext2, attn, upsample in self.ups:
x = torch.cat((x, h.pop()), dim=1) x = torch.cat((x, h.pop()), dim=1)
x = resnet(x, t) x = convnext(x, t)
x = resnet2(x, t) x = convnext2(x, t)
x = attn(x) x = attn(x)
x = upsample(x) x = upsample(x)
@@ -264,24 +277,47 @@ def noise_like(shape, device, repeat=False):
noise = lambda: torch.randn(shape, device=device) noise = lambda: torch.randn(shape, device=device)
return repeat_noise() if repeat else noise() return repeat_noise() if repeat else noise()
def cosine_beta_schedule(timesteps, s = 0.008):
"""
cosine schedule
as proposed in https://openreview.net/forum?id=-NEXDKk8gZ
"""
steps = timesteps + 1
x = np.linspace(0, steps, steps)
alphas_cumprod = np.cos(((x / steps) + s) / (1 + s) * np.pi * 0.5) ** 2
alphas_cumprod = alphas_cumprod / alphas_cumprod[0]
betas = 1 - (alphas_cumprod[1:] / alphas_cumprod[:-1])
return np.clip(betas, a_min = 0, a_max = 0.999)
class GaussianDiffusion(nn.Module): class GaussianDiffusion(nn.Module):
def __init__(self, denoise_fn, beta_start=0.0001, beta_end=0.02, num_diffusion_timesteps=1000, loss_type='l1', betas = None): def __init__(
self,
denoise_fn,
*,
image_size,
channels = 3,
timesteps = 1000,
loss_type = 'l1',
betas = None
):
super().__init__() super().__init__()
self.channels = channels
self.image_size = image_size
self.denoise_fn = denoise_fn self.denoise_fn = denoise_fn
if exists(betas): if exists(betas):
self.np_betas = betas.detach().cpu().numpy() if isinstance(betas, torch.Tensor) else betas betas = betas.detach().cpu().numpy() if isinstance(betas, torch.Tensor) else betas
else: else:
self.np_betas = betas = np.linspace(beta_start, beta_end, num_diffusion_timesteps).astype(np.float64) betas = cosine_beta_schedule(timesteps)
timesteps, = betas.shape
self.num_timesteps = int(timesteps)
self.loss_type = loss_type
alphas = 1. - betas alphas = 1. - betas
alphas_cumprod = np.cumprod(alphas, axis=0) alphas_cumprod = np.cumprod(alphas, axis=0)
alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1]) alphas_cumprod_prev = np.append(1., alphas_cumprod[:-1])
timesteps, = betas.shape
self.num_timesteps = int(timesteps)
self.loss_type = loss_type
to_torch = partial(torch.tensor, dtype=torch.float32) to_torch = partial(torch.tensor, dtype=torch.float32)
self.register_buffer('betas', to_torch(betas)) self.register_buffer('betas', to_torch(betas))
@@ -357,8 +393,10 @@ class GaussianDiffusion(nn.Module):
return img return img
@torch.no_grad() @torch.no_grad()
def sample(self, image_size, batch_size = 16): def sample(self, batch_size = 16):
return self.p_sample_loop((batch_size, 3, image_size, image_size)) image_size = self.image_size
channels = self.channels
return self.p_sample_loop((batch_size, channels, image_size, image_size))
@torch.no_grad() @torch.no_grad()
def interpolate(self, x1, x2, t = None, lam = 0.5): def interpolate(self, x1, x2, t = None, lam = 0.5):
@@ -401,24 +439,26 @@ class GaussianDiffusion(nn.Module):
return loss return loss
def forward(self, x, *args, **kwargs): def forward(self, x, *args, **kwargs):
b, *_, device = *x.shape, x.device b, c, h, w, device, img_size, = *x.shape, x.device, self.image_size
assert h == img_size and w == img_size, f'height and width of image must be {img_size}'
t = torch.randint(0, self.num_timesteps, (b,), device=device).long() t = torch.randint(0, self.num_timesteps, (b,), device=device).long()
return self.p_losses(x, t, *args, **kwargs) return self.p_losses(x, t, *args, **kwargs)
# dataset classes # dataset classes
class Dataset(data.Dataset): class Dataset(data.Dataset):
def __init__(self, folder, image_size): def __init__(self, folder, image_size, exts = ['jpg', 'jpeg', 'png']):
super().__init__() super().__init__()
self.folder = folder self.folder = folder
self.image_size = image_size self.image_size = image_size
self.paths = [p for ext in EXTS for p in Path(f'{folder}').glob(f'**/*.{ext}')] self.paths = [p for ext in exts for p in Path(f'{folder}').glob(f'**/*.{ext}')]
self.transform = transforms.Compose([ self.transform = transforms.Compose([
transforms.Resize(image_size), transforms.Resize(image_size),
transforms.RandomHorizontalFlip(), transforms.RandomHorizontalFlip(),
transforms.CenterCrop(image_size), transforms.CenterCrop(image_size),
transforms.ToTensor() transforms.ToTensor(),
transforms.Lambda(lambda t: (t * 2) - 1)
]) ])
def __len__(self): def __len__(self):
@@ -444,16 +484,22 @@ class Trainer(object):
train_num_steps = 100000, train_num_steps = 100000,
gradient_accumulate_every = 2, gradient_accumulate_every = 2,
fp16 = False, fp16 = False,
step_start_ema = 2000 step_start_ema = 2000,
update_ema_every = 10,
save_and_sample_every = 1000,
results_folder = './results'
): ):
super().__init__() super().__init__()
self.model = diffusion_model self.model = diffusion_model
self.ema = EMA(ema_decay) self.ema = EMA(ema_decay)
self.ema_model = copy.deepcopy(self.model) self.ema_model = copy.deepcopy(self.model)
self.update_ema_every = update_ema_every
self.step_start_ema = step_start_ema self.step_start_ema = step_start_ema
self.save_and_sample_every = save_and_sample_every
self.batch_size = train_batch_size self.batch_size = train_batch_size
self.image_size = image_size self.image_size = diffusion_model.image_size
self.gradient_accumulate_every = gradient_accumulate_every self.gradient_accumulate_every = gradient_accumulate_every
self.train_num_steps = train_num_steps self.train_num_steps = train_num_steps
@@ -469,6 +515,9 @@ class Trainer(object):
if fp16: if fp16:
(self.model, self.ema_model), self.opt = amp.initialize([self.model, self.ema_model], self.opt, opt_level='O1') (self.model, self.ema_model), self.opt = amp.initialize([self.model, self.ema_model], self.opt, opt_level='O1')
self.results_folder = Path(results_folder)
self.results_folder.mkdir(exist_ok = True)
self.reset_parameters() self.reset_parameters()
def reset_parameters(self): def reset_parameters(self):
@@ -486,10 +535,10 @@ class Trainer(object):
'model': self.model.state_dict(), 'model': self.model.state_dict(),
'ema': self.ema_model.state_dict() 'ema': self.ema_model.state_dict()
} }
torch.save(data, f'./model-{milestone}.pt') torch.save(data, str(self.results_folder / f'model-{milestone}.pt'))
def load(self, milestone): def load(self, milestone):
data = torch.load(f'./model-{milestone}.pt') data = torch.load(str(self.results_folder / f'model-{milestone}.pt'))
self.step = data['step'] self.step = data['step']
self.model.load_state_dict(data['model']) self.model.load_state_dict(data['model'])
@@ -508,15 +557,16 @@ class Trainer(object):
self.opt.step() self.opt.step()
self.opt.zero_grad() self.opt.zero_grad()
if self.step % UPDATE_EMA_EVERY == 0: if self.step % self.update_ema_every == 0:
self.step_ema() self.step_ema()
if self.step != 0 and self.step % SAVE_AND_SAMPLE_EVERY == 0: if self.step != 0 and self.step % self.save_and_sample_every == 0:
milestone = self.step // SAVE_AND_SAMPLE_EVERY milestone = self.step // self.save_and_sample_every
batches = num_to_groups(36, self.batch_size) batches = num_to_groups(36, self.batch_size)
all_images_list = list(map(lambda n: self.ema_model.sample(self.image_size, batch_size=n), batches)) all_images_list = list(map(lambda n: self.ema_model.sample(batch_size=n), batches))
all_images = torch.cat(all_images_list, dim=0) all_images = torch.cat(all_images_list, dim=0)
utils.save_image(all_images, f'./sample-{milestone}.png', nrow=6) all_images = (all_images + 1) * 0.5
utils.save_image(all_images, str(self.results_folder / f'sample-{milestone}.png'), nrow = 6)
self.save(milestone) self.save(milestone)
self.step += 1 self.step += 1
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 1.3 MiB

After

Width:  |  Height:  |  Size: 842 KiB

+1 -1
View File
@@ -3,7 +3,7 @@ from setuptools import setup, find_packages
setup( setup(
name = 'denoising-diffusion-pytorch', name = 'denoising-diffusion-pytorch',
packages = find_packages(), packages = find_packages(),
version = '0.3.2', version = '0.7.0',
license='MIT', license='MIT',
description = 'Denoising Diffusion Probabilistic Models - Pytorch', description = 'Denoising Diffusion Probabilistic Models - Pytorch',
author = 'Phil Wang', author = 'Phil Wang',