diff --git a/download_model.py b/download_model.py deleted file mode 100644 index c5a8b89..0000000 --- a/download_model.py +++ /dev/null @@ -1,28 +0,0 @@ -import os -import sys -import requests -from tqdm import tqdm - -if len(sys.argv) != 2: - print('You must enter the model name as a parameter, e.g.: download_model.py 117M') - sys.exit(1) - -model = sys.argv[1] - -subdir = os.path.join('gpt2','models', model) -if not os.path.exists(subdir): - os.makedirs(subdir) -subdir = subdir.replace('\\','/') # needed for Windows - -for filename in ['checkpoint','encoder.json','hparams.json','model.ckpt.data-00000-of-00001', 'model.ckpt.index', 'model.ckpt.meta', 'vocab.bpe']: - - r = requests.get("https://storage.googleapis.com/gpt-2/" + subdir + "/" + filename, stream=True) - - with open(os.path.join(subdir, filename), 'wb') as f: - file_size = int(r.headers["content-length"]) - chunk_size = 1000 - with tqdm(ncols=100, desc="Fetching " + filename, total=file_size, unit_scale=True) as pbar: - # 1k for chunk_size, since Ethernet packet size is around 1500 bytes - for chunk in r.iter_content(chunk_size=chunk_size): - f.write(chunk) - pbar.update(chunk_size) diff --git a/generator.py b/generator.py deleted file mode 100644 index 0e6d80d..0000000 --- a/generator.py +++ /dev/null @@ -1,78 +0,0 @@ -import json -import os -import numpy as np -import tensorflow as tf - -import gpt2.src.model as model -import gpt2.src.sample as sample -import gpt2.src.encoder as encoder -from utils import * - -pos_action_starts = ["You attack", "You tell", "You use", "You go"] - - -class StoryGenerator(): - - def __init__(self, sess, length=75, temperature=0.9, top_k=40): - - seed = None - batch_size=1 - model_path='gpt2/models/117M' - self.sess = sess - - self.enc = encoder.get_encoder(model_path) - hparams = model.default_hparams() - with open(os.path.join(model_path, 'hparams.json')) as f: - hparams.override_from_dict(json.load(f)) - - self.context = tf.placeholder(tf.int32, [batch_size, None]) - np.random.seed(seed) - tf.set_random_seed(seed) - self.output = sample.sample_sequence( - hparams=hparams, length=length, - context=self.context, - batch_size=batch_size, - ) - - saver = tf.train.Saver() - ckpt = tf.train.latest_checkpoint(model_path) - saver.restore(self.sess, ckpt) - - - def generate(self, prompt): - context_tokens = self.enc.encode(prompt) - out = self.sess.run(self.output, feed_dict={ - self.context: [context_tokens for _ in range(1)] - })[:, len(context_tokens):] - - text = self.enc.decode(out[0]) - return text - - def generate_story_block(self, prompt): - block = self.generate(prompt) - block = cut_trailing_sentence(block) - block = story_replace(block) - - return block - - def generate_action_options(self, prompt, action_starts=pos_action_starts): - - possible_actions = [] - for phrase in action_starts: - action = phrase + self.generate(prompt + phrase) - action = first_sentence(action) - possible_actions.append(action) - - return possible_actions - - def generate_action_result(self, prompt, phrase): - action = phrase + self.generate(prompt + phrase) - action_result = cut_trailing_sentence(action) - action_result = story_replace(action_result) - - action = first_sentence(action) - - - return action, action_result - - diff --git a/gpt2/CONTRIBUTORS.md b/gpt2/CONTRIBUTORS.md deleted file mode 100644 index eab7132..0000000 --- a/gpt2/CONTRIBUTORS.md +++ /dev/null @@ -1,17 +0,0 @@ -# Contributors (alphabetically) - -* **[madisonmay](https://github.com/madisonmay)** - - Added Dockerfiles - -* **[Margaret Mitchell et al](https://arxiv.org/abs/1810.03993)** - - Our [usage](./README.md#usage) writeup was loosely inspired by the paper - [Model Cards for Model Reporting](https://arxiv.org/abs/1810.03993) - and related conversations with some of the authors. - -* **[webproduktion01](https://github.com/webproduktion01)** - - Ported download script to python. - -**[Full code contributors list](https://github.com/openai/gpt-2/contributors).** diff --git a/gpt2/DEVELOPERS.md b/gpt2/DEVELOPERS.md deleted file mode 100644 index 078999b..0000000 --- a/gpt2/DEVELOPERS.md +++ /dev/null @@ -1,85 +0,0 @@ -# Installation - -Git clone this repository, and `cd` into directory for remaining commands -``` -git clone https://github.com/openai/gpt-2.git && cd gpt-2 -``` - -Then, follow instructions for either native or Docker installation. - -## Native Installation - -All steps can optionally be done in a virtual environment using tools such as `virtualenv` or `conda`. - -Install tensorflow 1.12 (with GPU support, if you have a GPU and want everything to run faster) -``` -pip3 install tensorflow==1.12.0 -``` -or -``` -pip3 install tensorflow-gpu==1.12.0 -``` - -Install other python packages: -``` -pip3 install -r requirements.txt -``` - -Download the model data -``` -python3 download_model.py 117M -``` - -## Docker Installation - -Build the Dockerfile and tag the created image as `gpt-2`: -``` -docker build --tag gpt-2 -f Dockerfile.gpu . # or Dockerfile.cpu -``` - -Start an interactive bash session from the `gpt-2` docker image. - -You can opt to use the `--runtime=nvidia` flag if you have access to a NVIDIA GPU -and a valid install of [nvidia-docker 2.0](https://github.com/nvidia/nvidia-docker/wiki/Installation-(version-2.0)). -``` -docker run --runtime=nvidia -it gpt-2 bash -``` - -# Running - -| WARNING: Samples are unfiltered and may contain offensive content. | -| --- | - -Some of the examples below may include Unicode text characters. Set the environment variable: -``` -export PYTHONIOENCODING=UTF-8 -``` -to override the standard stream settings in UTF-8 mode. - -## Unconditional sample generation - -To generate unconditional samples from the small model: -``` -python3 src/generate_unconditional_samples.py | tee /tmp/samples -``` -There are various flags for controlling the samples: -``` -python3 src/generate_unconditional_samples.py --top_k 40 --temperature 0.7 | tee /tmp/samples -``` - -To check flag descriptions, use: -``` -python3 src/generate_unconditional_samples.py -- --help -``` - -## Conditional sample generation - -To give the model custom prompts, you can use: -``` -python3 src/interactive_conditional_samples.py --top_k 40 -``` - -To check flag descriptions, use: -``` -python3 src/interactive_conditional_samples.py -- --help -``` diff --git a/gpt2/LICENSE b/gpt2/LICENSE deleted file mode 100644 index cb36e12..0000000 --- a/gpt2/LICENSE +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2019 OpenAI - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/gpt2/README.md b/gpt2/README.md deleted file mode 100644 index c1be039..0000000 --- a/gpt2/README.md +++ /dev/null @@ -1,61 +0,0 @@ -# gpt-2 - -Code and samples from the paper ["Language Models are Unsupervised Multitask Learners"](https://d4mucfpksywv.cloudfront.net/better-language-models/language-models.pdf). - -For now, we have only released a smaller (117M parameter) version of GPT-2. - -See more details in our [blog post](https://blog.openai.com/better-language-models/). - -## Usage - -This repository is meant to be a starting point for researchers and engineers to experiment with GPT-2-117M. While GPT-2-117M is less proficient than GPT-2-1.5B, it is useful for a wide range of research and applications which could also apply to larger models. - -### Some caveats - -- GPT-2-117M robustness and worst case behaviors are not well-understood. As with any machine-learned model, carefully evaluate GPT-2-117M for your use case, especially if used without fine-tuning or in safety-critical applications where reliability is important. -- The dataset our GPT-2-117M was trained on contains many texts with [biases](https://twitter.com/TomerUllman/status/1101485289720242177) and factual inaccuracies, and thus GPT-2-117M is likely to be biased and inaccurate as well. -- To avoid having samples mistaken as human-written, we recommend clearly labeling samples as synthetic before wide dissemination. Our models are often incoherent or inaccurate in subtle ways, which takes more than a quick read for a human to notice. - -### Work with us - -Please [let us know](mailto:languagequestions@openai.com) if you’re doing interesting research with or working on applications of GPT-2-117M! We’re especially interested in hearing from and potentially working with those who are studying -- Potential malicious use cases and defenses against them (e.g. the detectability of synthetic text) -- The extent of problematic content (e.g. bias) being baked into the models and effective mitigations - -## Development - -See [DEVELOPERS.md](./DEVELOPERS.md) - -## Contributors - -See [CONTRIBUTORS.md](./CONTRIBUTORS.md) - -## GPT-2 samples - -| WARNING: Samples are unfiltered and may contain offensive content. | -| --- | - -While we have not yet released GPT-2 itself, you can see some samples from it in the `gpt-2-samples` folder. -We show unconditional samples with default settings (temperature 1 and no truncation), with temperature 0.7, and with truncation with top_k 40. -We show conditional samples, with contexts drawn from `WebText`'s test set, with default settings (temperature 1 and no truncation), with temperature 0.7, and with truncation with top_k 40. - -## Citation - -Please use the following bibtex entry: -``` -@article{radford2019language, - title={Language Models are Unsupervised Multitask Learners}, - author={Radford, Alec and Wu, Jeff and Child, Rewon and Luan, David and Amodei, Dario and Sutskever, Ilya}, - year={2019} -} -``` - -## Future work - -We may release code for evaluating the models on various benchmarks. - -We are still considering release of the larger models. - -## License - -[MIT](./LICENSE) diff --git a/gpt2/__init__.py b/gpt2/__init__.py deleted file mode 100644 index e69de29..0000000 diff --git a/gpt2/__pycache__/__init__.cpython-37.pyc b/gpt2/__pycache__/__init__.cpython-37.pyc deleted file mode 100644 index b59cf9e..0000000 Binary files a/gpt2/__pycache__/__init__.cpython-37.pyc and /dev/null differ diff --git a/gpt2/download_model.py b/gpt2/download_model.py deleted file mode 100644 index 30ba84a..0000000 --- a/gpt2/download_model.py +++ /dev/null @@ -1,28 +0,0 @@ -import os -import sys -import requests -from tqdm import tqdm - -if len(sys.argv) != 2: - print('You must enter the model name as a parameter, e.g.: download_model.py 117M') - sys.exit(1) - -model = sys.argv[1] - -subdir = os.path.join('models', model) -if not os.path.exists(subdir): - os.makedirs(subdir) -subdir = subdir.replace('\\','/') # needed for Windows - -for filename in ['checkpoint','encoder.json','hparams.json','model.ckpt.data-00000-of-00001', 'model.ckpt.index', 'model.ckpt.meta', 'vocab.bpe']: - - r = requests.get("https://storage.googleapis.com/gpt-2/" + subdir + "/" + filename, stream=True) - - with open(os.path.join(subdir, filename), 'wb') as f: - file_size = int(r.headers["content-length"]) - chunk_size = 1000 - with tqdm(ncols=100, desc="Fetching " + filename, total=file_size, unit_scale=True) as pbar: - # 1k for chunk_size, since Ethernet packet size is around 1500 bytes - for chunk in r.iter_content(chunk_size=chunk_size): - f.write(chunk) - pbar.update(chunk_size) diff --git a/gpt2/src/__pycache__/encoder.cpython-37.pyc b/gpt2/src/__pycache__/encoder.cpython-37.pyc deleted file mode 100644 index c23a36b..0000000 Binary files a/gpt2/src/__pycache__/encoder.cpython-37.pyc and /dev/null differ diff --git a/gpt2/src/__pycache__/model.cpython-37.pyc b/gpt2/src/__pycache__/model.cpython-37.pyc deleted file mode 100644 index 65b91cf..0000000 Binary files a/gpt2/src/__pycache__/model.cpython-37.pyc and /dev/null differ diff --git a/gpt2/src/__pycache__/sample.cpython-37.pyc b/gpt2/src/__pycache__/sample.cpython-37.pyc deleted file mode 100644 index d243bb1..0000000 Binary files a/gpt2/src/__pycache__/sample.cpython-37.pyc and /dev/null differ diff --git a/gpt2/src/encoder.py b/gpt2/src/encoder.py deleted file mode 100644 index e9fed43..0000000 --- a/gpt2/src/encoder.py +++ /dev/null @@ -1,117 +0,0 @@ -"""Byte pair encoding utilities""" - -import os -import json -import regex as re -from functools import lru_cache - -@lru_cache() -def bytes_to_unicode(): - """ - Returns list of utf-8 byte and a corresponding list of unicode strings. - The reversible bpe codes work on unicode strings. - This means you need a large # of unicode characters in your vocab if you want to avoid UNKs. - When you're at something like a 10B token dataset you end up needing around 5K for decent coverage. - This is a signficant percentage of your normal, say, 32K bpe vocab. - To avoid that, we want lookup tables between utf-8 bytes and unicode strings. - And avoids mapping to whitespace/control characters the bpe code barfs on. - """ - bs = list(range(ord("!"), ord("~")+1))+list(range(ord("¡"), ord("¬")+1))+list(range(ord("®"), ord("ÿ")+1)) - cs = bs[:] - n = 0 - for b in range(2**8): - if b not in bs: - bs.append(b) - cs.append(2**8+n) - n += 1 - cs = [chr(n) for n in cs] - return dict(zip(bs, cs)) - -def get_pairs(word): - """Return set of symbol pairs in a word. - - Word is represented as tuple of symbols (symbols being variable-length strings). - """ - pairs = set() - prev_char = word[0] - for char in word[1:]: - pairs.add((prev_char, char)) - prev_char = char - return pairs - -class Encoder: - def __init__(self, encoder, bpe_merges, errors='replace'): - self.encoder = encoder - self.decoder = {v:k for k,v in self.encoder.items()} - self.errors = errors # how to handle errors in decoding - self.byte_encoder = bytes_to_unicode() - self.byte_decoder = {v:k for k, v in self.byte_encoder.items()} - self.bpe_ranks = dict(zip(bpe_merges, range(len(bpe_merges)))) - self.cache = {} - - # Should haved added re.IGNORECASE so BPE merges can happen for capitalized versions of contractions - self.pat = re.compile(r"""'s|'t|'re|'ve|'m|'ll|'d| ?\p{L}+| ?\p{N}+| ?[^\s\p{L}\p{N}]+|\s+(?!\S)|\s+""") - - def bpe(self, token): - if token in self.cache: - return self.cache[token] - word = tuple(token) - pairs = get_pairs(word) - - if not pairs: - return token - - while True: - bigram = min(pairs, key = lambda pair: self.bpe_ranks.get(pair, float('inf'))) - if bigram not in self.bpe_ranks: - break - first, second = bigram - new_word = [] - i = 0 - while i < len(word): - try: - j = word.index(first, i) - new_word.extend(word[i:j]) - i = j - except: - new_word.extend(word[i:]) - break - - if word[i] == first and i < len(word)-1 and word[i+1] == second: - new_word.append(first+second) - i += 2 - else: - new_word.append(word[i]) - i += 1 - new_word = tuple(new_word) - word = new_word - if len(word) == 1: - break - else: - pairs = get_pairs(word) - word = ' '.join(word) - self.cache[token] = word - return word - - def encode(self, text): - bpe_tokens = [] - for token in re.findall(self.pat, text): - token = ''.join(self.byte_encoder[b] for b in token.encode('utf-8')) - bpe_tokens.extend(self.encoder[bpe_token] for bpe_token in self.bpe(token).split(' ')) - return bpe_tokens - - def decode(self, tokens): - text = ''.join([self.decoder[token] for token in tokens]) - text = bytearray([self.byte_decoder[c] for c in text]).decode('utf-8', errors=self.errors) - return text - -def get_encoder(model_path): - with open(os.path.join(model_path, 'encoder.json'), 'r') as f: - encoder = json.load(f) - with open(os.path.join(model_path, 'vocab.bpe'), 'r', encoding="utf-8") as f: - bpe_data = f.read() - bpe_merges = [tuple(merge_str.split()) for merge_str in bpe_data.split('\n')[1:-1]] - return Encoder( - encoder=encoder, - bpe_merges=bpe_merges, - ) diff --git a/gpt2/src/model.py b/gpt2/src/model.py deleted file mode 100644 index 230b83c..0000000 --- a/gpt2/src/model.py +++ /dev/null @@ -1,174 +0,0 @@ -import numpy as np -import tensorflow as tf -from tensorflow.contrib.training import HParams - -def default_hparams(): - return HParams( - n_vocab=0, - n_ctx=1024, - n_embd=768, - n_head=12, - n_layer=12, - ) - -def shape_list(x): - """Deal with dynamic shape in tensorflow cleanly.""" - static = x.shape.as_list() - dynamic = tf.shape(x) - return [dynamic[i] if s is None else s for i, s in enumerate(static)] - -def softmax(x, axis=-1): - x = x - tf.reduce_max(x, axis=axis, keepdims=True) - ex = tf.exp(x) - return ex / tf.reduce_sum(ex, axis=axis, keepdims=True) - -def gelu(x): - return 0.5*x*(1+tf.tanh(np.sqrt(2/np.pi)*(x+0.044715*tf.pow(x, 3)))) - -def norm(x, scope, *, axis=-1, epsilon=1e-5): - """Normalize to mean = 0, std = 1, then do a diagonal affine transform.""" - with tf.variable_scope(scope): - n_state = x.shape[-1].value - g = tf.get_variable('g', [n_state], initializer=tf.constant_initializer(1)) - b = tf.get_variable('b', [n_state], initializer=tf.constant_initializer(0)) - u = tf.reduce_mean(x, axis=axis, keepdims=True) - s = tf.reduce_mean(tf.square(x-u), axis=axis, keepdims=True) - x = (x - u) * tf.rsqrt(s + epsilon) - x = x*g + b - return x - -def split_states(x, n): - """Reshape the last dimension of x into [n, x.shape[-1]/n].""" - *start, m = shape_list(x) - return tf.reshape(x, start + [n, m//n]) - -def merge_states(x): - """Smash the last two dimensions of x into a single dimension.""" - *start, a, b = shape_list(x) - return tf.reshape(x, start + [a*b]) - -def conv1d(x, scope, nf, *, w_init_stdev=0.02): - with tf.variable_scope(scope): - *start, nx = shape_list(x) - w = tf.get_variable('w', [1, nx, nf], initializer=tf.random_normal_initializer(stddev=w_init_stdev)) - b = tf.get_variable('b', [nf], initializer=tf.constant_initializer(0)) - c = tf.reshape(tf.matmul(tf.reshape(x, [-1, nx]), tf.reshape(w, [-1, nf]))+b, start+[nf]) - return c - -def attention_mask(nd, ns, *, dtype): - """1's in the lower triangle, counting from the lower right corner. - - Same as tf.matrix_band_part(tf.ones([nd, ns]), -1, ns-nd), but doesn't produce garbage on TPUs. - """ - i = tf.range(nd)[:,None] - j = tf.range(ns) - m = i >= j - ns + nd - return tf.cast(m, dtype) - - -def attn(x, scope, n_state, *, past, hparams): - assert x.shape.ndims == 3 # Should be [batch, sequence, features] - assert n_state % hparams.n_head == 0 - if past is not None: - assert past.shape.ndims == 5 # Should be [batch, 2, heads, sequence, features], where 2 is [k, v] - - def split_heads(x): - # From [batch, sequence, features] to [batch, heads, sequence, features] - return tf.transpose(split_states(x, hparams.n_head), [0, 2, 1, 3]) - - def merge_heads(x): - # Reverse of split_heads - return merge_states(tf.transpose(x, [0, 2, 1, 3])) - - def mask_attn_weights(w): - # w has shape [batch, heads, dst_sequence, src_sequence], where information flows from src to dst. - _, _, nd, ns = shape_list(w) - b = attention_mask(nd, ns, dtype=w.dtype) - b = tf.reshape(b, [1, 1, nd, ns]) - w = w*b - tf.cast(1e10, w.dtype)*(1-b) - return w - - def multihead_attn(q, k, v): - # q, k, v have shape [batch, heads, sequence, features] - w = tf.matmul(q, k, transpose_b=True) - w = w * tf.rsqrt(tf.cast(v.shape[-1].value, w.dtype)) - - w = mask_attn_weights(w) - w = softmax(w) - a = tf.matmul(w, v) - return a - - with tf.variable_scope(scope): - c = conv1d(x, 'c_attn', n_state*3) - q, k, v = map(split_heads, tf.split(c, 3, axis=2)) - present = tf.stack([k, v], axis=1) - if past is not None: - pk, pv = tf.unstack(past, axis=1) - k = tf.concat([pk, k], axis=-2) - v = tf.concat([pv, v], axis=-2) - a = multihead_attn(q, k, v) - a = merge_heads(a) - a = conv1d(a, 'c_proj', n_state) - return a, present - - -def mlp(x, scope, n_state, *, hparams): - with tf.variable_scope(scope): - nx = x.shape[-1].value - h = gelu(conv1d(x, 'c_fc', n_state)) - h2 = conv1d(h, 'c_proj', nx) - return h2 - - -def block(x, scope, *, past, hparams): - with tf.variable_scope(scope): - nx = x.shape[-1].value - a, present = attn(norm(x, 'ln_1'), 'attn', nx, past=past, hparams=hparams) - x = x + a - m = mlp(norm(x, 'ln_2'), 'mlp', nx*4, hparams=hparams) - x = x + m - return x, present - -def past_shape(*, hparams, batch_size=None, sequence=None): - return [batch_size, hparams.n_layer, 2, hparams.n_head, sequence, hparams.n_embd // hparams.n_head] - -def expand_tile(value, size): - """Add a new axis of given size.""" - value = tf.convert_to_tensor(value, name='value') - ndims = value.shape.ndims - return tf.tile(tf.expand_dims(value, axis=0), [size] + [1]*ndims) - -def positions_for(tokens, past_length): - batch_size = tf.shape(tokens)[0] - nsteps = tf.shape(tokens)[1] - return expand_tile(past_length + tf.range(nsteps), batch_size) - - -def model(hparams, X, past=None, scope='model', reuse=False): - with tf.variable_scope(scope, reuse=reuse): - results = {} - batch, sequence = shape_list(X) - - wpe = tf.get_variable('wpe', [hparams.n_ctx, hparams.n_embd], - initializer=tf.random_normal_initializer(stddev=0.01)) - wte = tf.get_variable('wte', [hparams.n_vocab, hparams.n_embd], - initializer=tf.random_normal_initializer(stddev=0.02)) - past_length = 0 if past is None else tf.shape(past)[-2] - h = tf.gather(wte, X) + tf.gather(wpe, positions_for(X, past_length)) - - # Transformer - presents = [] - pasts = tf.unstack(past, axis=1) if past is not None else [None] * hparams.n_layer - assert len(pasts) == hparams.n_layer - for layer, past in enumerate(pasts): - h, present = block(h, 'h%d' % layer, past=past, hparams=hparams) - presents.append(present) - results['present'] = tf.stack(presents, axis=1) - h = norm(h, 'ln_f') - - # Language model loss. Do tokens