reorganized src

This commit is contained in:
Thomas Lemoine committed 2023-03-29 00:53:04 -04:00
1 parent 3ce0b572d3
commit 9a2890ef89
13 files changed
+283 -75

No files matched your search

+1
View File
@@ -138,3 +138,4 @@ src/tmp.py
*.DS_Store
.vercel/
+6
View File
@@ -0,0 +1,6 @@
{
"name": "AlignmentSearch",
"lockfileVersion": 3,
"requires": true,
"packages": {}
}
+2 -1
View File
@@ -5,4 +5,5 @@ langchain
requests
tiktoken
tqdm
nltk
nltk
dateutil
@@ -10,7 +10,7 @@ from tenacity import (
import tiktoken
import config
from dataset import Dataset
from dataset.create_dataset import Dataset
from settings import PATH_TO_DATASET, EMBEDDING_MODEL, COMPLETIONS_MODEL
openai.api_key = config.OPENAI_API_KEY
@@ -120,6 +120,7 @@ class AlignmentSearch:
return explanation_prompt
def get_user_prompt(self, user_query: str, mode: str) -> str:
pass #TODO: fix junk
def create_messages(self, system_prompt: str, user_prompt: str, context: str, question: str):
return [
View File
Whitespace-only changes.
@@ -8,6 +8,8 @@ import pickle
import os
import concurrent.futures
from pathlib import Path
from tqdm.auto import tqdm
from dateutil.parser import parse, ParserError
from tenacity import (
retry,
@@ -15,10 +17,6 @@ from tenacity import (
wait_random_exponential,
) # for exponential backoff
from text_splitter import TokenSplitter, split_into_sentences
from settings import PATH_TO_DATA, PATH_TO_EMBEDDINGS, PATH_TO_DATASET, EMBEDDING_MODEL, LEN_EMBEDDINGS
import os
from tqdm.auto import tqdm
import openai
try:
@@ -28,6 +26,12 @@ except ImportError:
openai.api_key = os.environ.get('OPENAI_API_KEY')
from .settings import PATH_TO_RAW_DATA, PATH_TO_DATASET, EMBEDDING_MODEL, LEN_EMBEDDINGS
from .text_splitter import TokenSplitter, split_into_sentences
error_count_dict = {
"Entry has no source.": 0,
"Entry has no title.": 0,
@@ -43,11 +47,11 @@ class MissingDataException(Exception):
class Dataset:
def __init__(self,
jsonl_data_path: str, # Path to the dataset .jsonl file.
jsonl_data_path: str = PATH_TO_RAW_DATA, # Path to the dataset .jsonl file.
custom_sources: List[str] = None, # List of sources to include, like "alignment forum", "lesswrong", "arxiv",etc.
rate_limit_per_minute: int = 3_500, # Rate limit for the OpenAI API.
min_tokens_per_block: int = 400, # Minimum number of tokens per block.
max_tokens_per_block: int = 600, # Maximum number of tokens per block.
min_tokens_per_block: int = 300, # Minimum number of tokens per block.
max_tokens_per_block: int = 400, # Maximum number of tokens per block.
fraction_of_articles_to_use: float = 1.0, # Fraction of articles to use. If 1.0, use all articles.
):
self.jsonl_data_path = jsonl_data_path
@@ -115,6 +119,8 @@ class Dataset:
if 'date_published' in article and article['date_published'] and len(article['date_published']) >= 10: date_published = article['date_published'][:10]
elif 'published' in article and article['published'] and len(article['published']) >= 16: date_published = article['published'][:16]
else: date_published = None
if date_published is not None:
date_published = standardize_date(date_published)
# Get URL
if 'link' in article and article['link']: url = article['link']
@@ -163,12 +169,17 @@ class Dataset:
if (self.custom_sources is not None) and (entry['source'] not in self.custom_sources):
continue
self.articles_count[entry['source']] += 1
self.total_articles_count += 1
# Get title, author, date, URL, tags, and text
title, author, date_published, url, tags, text = self.extract_info_from_article(entry)
#if the text is too short, ignore this text
if len(text) < 500:
continue
#we're keeping the text so we inc the aticle count
self.articles_count[entry['source']] += 1
self.total_articles_count += 1
# Get signature
signature = ""
if title: signature += f"Title: {title}, "
@@ -185,7 +196,7 @@ class Dataset:
self.metadata.append((title, author, date_published, url, tags))
blocks = text_splitter.split(text, signature)
self.embedding_strings.extend(blocks)
self.embeddings_metadata_index.extend([self.total_articles_count] * len(blocks))
self.embeddings_metadata_index.extend([self.total_articles_count-1] * len(blocks))
# Update counts
self.total_char_count += len(text)
@@ -199,50 +210,41 @@ class Dataset:
error_count_dict[str(e)] += 1
def get_embeddings(self):
# Get an embedding for each text, with retries if necessary
#TODO: check batch size stuff at https://github.com/openai/openai-cookbook/blob/main/examples/vector_databases/pinecone/Gen_QA.ipynb
# to speed up the process
# @retry(wait=wait_random_exponential(min=1, max=20), stop=stop_after_attempt(5))
def get_embedding_at_index(text: str, i: int, delay_in_seconds: float = 0) -> np.ndarray:
time.sleep(delay_in_seconds)
embedding = openai.Embedding.create(
def get_embeddings_at_index(texts: str, batch_idx: int, batch_size: int = 200): # int, np.ndarray
embeddings = np.zeros((batch_size, 1536))
openai_output = openai.Embedding.create(
model=EMBEDDING_MODEL,
input=text
)
return i, embedding["data"][0]["embedding"]
input=texts
)['data']
for i, embedding in enumerate(openai_output):
embeddings[i] = embedding['embedding']
return batch_idx, embeddings
batch_size = 200
rate_limit = 3500 / 60 # Maximum embeddings per second
start = time.time()
self.embeddings = np.zeros((len(self.embedding_strings), LEN_EMBEDDINGS))
with concurrent.futures.ThreadPoolExecutor() as executor:
futures = [executor.submit(get_embedding_at_index, text, i) for i, text in enumerate(self.embedding_strings)]
futures = [executor.submit(
get_embeddings_at_index,
self.embedding_strings[batch_idx:batch_idx+batch_size],
batch_idx,
len(self.embedding_strings[batch_idx:batch_idx+batch_size])
) for batch_idx in range(0, len(self.embedding_strings), batch_size)]
num_completed = 0
for future in concurrent.futures.as_completed(futures):
i, embedding = future.result()
self.embeddings[i] = embedding
num_completed += 1
if num_completed % 50 == 0:
print(f"Completed {num_completed}/{len(self.embedding_strings)} embeddings in {time.time() - start:.2f} seconds.")
print(f"Completed {num_completed}/{len(self.embedding_strings)} embeddings in {time.time() - start:.2f} seconds.")
batch_idx, embeddings = future.result()
num_completed += embeddings.shape[0]
self.embeddings[batch_idx:batch_idx+embeddings.shape[0]] = embeddings
#TODO: complete this to speed up embeddings
""" def get_embeddings_in_batches(self):
# Get an embedding for each text, with retries if necessary
@retry(wait=wait_random_exponential(min=1, max=20), stop=stop_after_attempt(5))
def get_embedding_in_batches(batch: List[str], i: int, delay_in_seconds: float = 0) -> np.ndarray:
try:
res = openai.Embedding.create(input=batch, engine=EMBEDDING_MODEL)
except:
done = False
while not done:
time.sleep(5)
try:
res = openai.Embedding.create(input=batch, engine=EMBEDDING_MODEL)
done = True
except:
pass
"""
elapsed_time = time.time() - start
expected_time = num_completed / rate_limit
sleep_time = max(expected_time - elapsed_time, 0)
time.sleep(sleep_time)
print(f"Completed {num_completed}/{len(self.embedding_strings)} embeddings in {elapsed_time:.2f} seconds.")
def save_embeddings(self, path: str):
np.save(path, self.embeddings)
@@ -250,7 +252,7 @@ class Dataset:
def load_embeddings(self, path: str):
self.embeddings = np.load(path)
def save_class(self, path: str):
def save_class(self, path: str = PATH_TO_DATASET):
# Save the class to a pickle file
print(f"Saving class to {path}...")
with open(path, 'wb') as f:
@@ -272,7 +274,17 @@ def get_authors_list(authors_string: str) -> List[str]:
authors = [authors_string.strip()]
return authors
def standardize_date(date_string, default_date='n/a'):
try:
dt = parse(date_string)
return dt.strftime('%Y-%m-%d')
except (ParserError, ValueError):
return default_date
"""
if __name__ == "__main__":
# List of possible sources:
all_sources = ["https://aipulse.org", "ebook", "https://qualiacomputing.com", "alignment forum", "lesswrong", "manual", "arxiv", "https://deepmindsafetyresearch.medium.com", "waitbutwhy.com", "GitHub", "https://aiimpacts.org", "arbital.com", "carado.moe", "nonarxiv_papers", "https://vkrakovna.wordpress.com", "https://jsteinhardt.wordpress.com", "audio-transcripts", "https://intelligence.org", "youtube", "reports", "https://aisafety.camp", "curriculum", "https://www.yudkowsky.net", "distill", "Cold Takes", "printouts", "gwern.net", "generative.ink", "greaterwrong.com"] # These sources do not have a source field in the .jsonl file
@@ -311,7 +323,7 @@ if __name__ == "__main__":
]
dataset = Dataset(
jsonl_data_path=PATH_TO_DATA.resolve(),
jsonl_data_path=PATH_TO_RAW_DATA.resolve(),
custom_sources=custom_sources,
rate_limit_per_minute=3500,
min_tokens_per_block=200, max_tokens_per_block=300,
@@ -323,5 +335,5 @@ if __name__ == "__main__":
dataset.save_class(PATH_TO_DATASET.resolve())
# # dataset = pickle.load(open("dataset.pkl", "rb"))
"""
Binary file not shown.
+24
View File
@@ -0,0 +1,24 @@
from pathlib import Path
EMBEDDING_MODEL = "text-embedding-ada-002"
COMPLETIONS_MODEL = "gpt-3.5-turbo"
LEN_EMBEDDINGS = 1536
MAX_LEN_PROMPT = 4095 # This may be 8191, unsure.
def get_rawdata_file_path():
current_file_path = Path(__file__).resolve()
data_file_path = current_file_path.parent / 'data' / 'alignment_texts.jsonl'
return str(data_file_path)
def get_dataset_file_path():
current_file_path = Path(__file__).resolve()
data_file_path = current_file_path.parent / 'data' / 'dataset.pkl'
return str(data_file_path)
PATH_TO_RAW_DATA = get_rawdata_file_path()
PATH_TO_DATASET = get_dataset_file_path()
@@ -7,9 +7,13 @@ from typing import List
import nltk
# Download the Punkt tokenizer if you haven't already
# Download the Punkt tokenizer if you haven't already.
# If you want to save a second everytime you run this file you can comment
# it out after the first time it was downloaded.
nltk.download("punkt")
def split_into_sentences(text: str) -> List[str]:
"""
Splits the input text into sentences.
@@ -87,7 +91,7 @@ class TokenSplitter:
last_block = dec(enc(latest_plus_current)[-max_tokens:])
blocks.append(last_block)
return blocks
return [block.strip() for block in blocks]
def split(self, text: str, signature: str = None) -> List[str]:
if signature is None:
+146 -9
View File
@@ -1,21 +1,158 @@
import numpy as np
import pickle
import openai
"""
import config
from semantic_search import AlignmentSearch
from settings import PATH_TO_DATASET
from assistant.semantic_search import AlignmentSearch
from dataset.create_dataset import Dataset
openai.api_key = config.OPENAI_API_KEY
from settings import PATH_TO_RAW_DATA, PATH_TO_DATASET, EMBEDDING_MODEL, LEN_EMBEDDINGS
"""
from tenacity import (
retry,
stop_after_attempt,
wait_random_exponential,
)
def main():
with open(PATH_TO_DATASET, 'rb') as f:
dataset = pickle.load(f)
import numpy as np
import sys
import pickle
from pathlib import Path
import random
src_path = Path(__file__).resolve().parent
if str(src_path) not in sys.path:
sys.path.append(str(src_path))
from dataset import create_dataset
#from assistant import semantic_search
from settings import PATH_TO_DATASET, EMBEDDING_MODEL
import numpy as np
import matplotlib.pyplot as plt
def load_rawdata_into_pkl():
"""with open(PATH_TO_DATASET, 'rb') as f:
dataset = pickle.load(f)
AS = AlignmentSearch(dataset=dataset)
prompt = "What would be an idea to solve the Alignment Problem? Name the Lesswrong post by Quintin Pope that discusses this idea."
answer = AS.search_and_answer(prompt, 3, HyDE=False)
print(answer)
"""
# List of possible sources:
all_sources = ["https://aipulse.org", "ebook", "https://qualiacomputing.com", "alignment forum", "lesswrong", "manual", "arxiv", "https://deepmindsafetyresearch.medium.com", "waitbutwhy.com", "GitHub", "https://aiimpacts.org", "arbital.com", "carado.moe", "nonarxiv_papers", "https://vkrakovna.wordpress.com", "https://jsteinhardt.wordpress.com", "audio-transcripts", "https://intelligence.org", "youtube", "reports", "https://aisafety.camp", "curriculum", "https://www.yudkowsky.net", "distill", "Cold Takes", "printouts", "gwern.net", "generative.ink", "greaterwrong.com"] # These sources do not have a source field in the .jsonl file
# List of sources we are using for the test run:
custom_sources = [
# "https://aipulse.org",
# "ebook",
# "https://qualiacomputing.com",
# "alignment forum",
# "lesswrong",
"manual",
# "arxiv",
# "https://deepmindsafetyresearch.medium.com",
"waitbutwhy.com",
# "GitHub",
# "https://aiimpacts.org",
# "arbital.com",
# "carado.moe",
# "nonarxiv_papers",
# "https://vkrakovna.wordpress.com",
"https://jsteinhardt.wordpress.com",
# "audio-transcripts",
# "https://intelligence.org",
# "youtube",
# "reports",
"https://aisafety.camp",
"curriculum",
"https://www.yudkowsky.net",
# "distill",
# "Cold Takes",
# "printouts",
# "gwern.net",
# "generative.ink",
# "greaterwrong.com"
]
dataset = create_dataset.Dataset(
custom_sources=custom_sources,
rate_limit_per_minute=3500,
min_tokens_per_block=200, max_tokens_per_block=300,
# fraction_of_articles_to_use=1/2000
)
dataset.get_alignment_texts()
print(len(dataset.embedding_strings))
dataset.get_embeddings()
# dataset.save_embeddings("data/embeddings.npy")
dataset.save_class()
# # dataset = pickle.load(open("dataset.pkl", "rb"))
@retry(wait=wait_random_exponential(min=1, max=20), stop=stop_after_attempt(4))
def get_embedding(text: str) -> np.ndarray:
result = openai.Embedding.create(model=EMBEDDING_MODEL, input=text)
return np.array(result["data"][0]["embedding"])
def print_out_dataset_stuff():
with open(PATH_TO_DATASET, 'rb') as f:
dataset = pickle.load(f)
embeddings_len = len(dataset.embedding_strings)
i1 = random.randint(0,embeddings_len-1)
#i2 = random.randint(0,embeddings_len-1)
#print(len(dataset.embeddings))
#print(len(dataset.embedding_strings))
#embedding_test = get_embedding(dataset.embedding_strings[i])
#print(np.dot(embedding_test,dataset.embeddings[i]))
metadata_i1 = dataset.embeddings_metadata_index[i1]
print("metadata:",dataset.metadata[metadata_i1])
print("embedding_string:",dataset.embedding_strings[i1])
#print("embedding_vector:",dataset.embeddings[i1])
embedding_of_string1 = get_embedding(dataset.embedding_strings[i1])
#metadata_i2 = dataset.embeddings_metadata_index[i2]
#print("metadata:",dataset.metadata[metadata_i2])
#print("embedding_string:",dataset.embedding_strings[i2])
#print("embedding_vector:",dataset.embeddings[i1])
#embedding_of_string2 = get_embedding(dataset.embedding_strings[i2])
#embedding_of_string2 = get_embedding("000000000000000000000000000000000000000000000000000000000000000000000000000000000")
#print(len(embedding_of_string1))
vector = dataset.embeddings[i1]
plot_likelihood(vector)
#plot_likelihood(get_embedding("tst"))
print(max(vector), min(vector))
print(sum([x**2 for x in vector]))
#print(np.dot(embedding_of_string1, embedding_of_string2))
def plot_likelihood(embeddings, num_buckets=200):
# Calculate the histogram
histogram, bin_edges = np.histogram(embeddings.flatten(), bins=num_buckets, range=(embeddings.min(), embeddings.max()))
# Normalize the histogram to get likelihoods
likelihoods = histogram / embeddings.flatten().size
# Plot the likelihoods
plt.bar(bin_edges[:-1], likelihoods, width=(bin_edges[1] - bin_edges[0]), edgecolor="k", alpha=0.7)
plt.xlabel("Value")
plt.ylabel("Likelihood")
plt.title("Likelihood of Floats in the Vector Embedding")
plt.savefig("bla.png")
if __name__ == "__main__":
main()
#load_rawdata_into_pkl()
print_out_dataset_stuff()
+17 -5
View File
@@ -1,12 +1,24 @@
from pathlib import Path
EMBEDDING_MODEL = "text-embedding-ada-002"
COMPLETIONS_MODEL = "text-davinci-003"
COMPLETIONS_MODEL = "gpt-3.5-turbo"
LEN_EMBEDDINGS = 1536
MAX_LEN_PROMPT = 4095 # This may be 8191, unsure.
project_path = Path(__file__).parent.parent#.parent
PATH_TO_DATA = project_path / "src" / "data" / "alignment_texts.jsonl" # Path to the dataset .jsonl file.
PATH_TO_EMBEDDINGS = project_path / "src" / "data" / "embeddings.npy" # Path to the saved embeddings (.npy) file.
PATH_TO_DATASET = project_path / "src" / "data" / "dataset.pkl" # Path to the saved dataset (.pkl) file, containing the dataset class object.
def get_rawdata_file_path():
current_file_path = Path(__file__).resolve()
data_file_path = current_file_path.parent / 'dataset' / 'data' / 'alignment_texts.jsonl'
return str(data_file_path)
def get_dataset_file_path():
current_file_path = Path(__file__).resolve()
data_file_path = current_file_path.parent / 'dataset' / 'data' / 'dataset.pkl'
return str(data_file_path)
PATH_TO_RAW_DATA = get_rawdata_file_path()
PATH_TO_DATASET = get_dataset_file_path()
+1 -1
View File
@@ -35,7 +35,7 @@ import tiktoken
import asyncio
import config
from semantic_search import get_top_k_blocks
from assistant.semantic_search import get_top_k_blocks
# OpenAI API key
+14 -4
View File
@@ -53,10 +53,20 @@ MAX_LEN_PROMPT = 4095 # This may be 8191, unsure.
# Paths
from pathlib import Path
project_path = Path(__file__).parent.parent.parent
PATH_TO_DATA = project_path / "web" / "api" / "data" / "alignment_texts.jsonl" # Path to the dataset .jsonl file.
PATH_TO_EMBEDDINGS = project_path / "web" / "api" / "data" / "embeddings.npy" # Path to the saved embeddings (.npy) file.
PATH_TO_DATASET = project_path / "web" / "api" / "data" / "dataset.pkl" # Path to the saved dataset (.pkl) file, containing the dataset class object.
#current_file = Path(__file__).parent.parent.parent
#PATH_TO_DATA = project_path / "web" / "api" / "data" / "alignment_texts.jsonl" # Path to the dataset .jsonl file.
#PATH_TO_EMBEDDINGS = project_path / "web" / "api" / "data" / "embeddings.npy" # Path to the saved embeddings (.npy) file.
#PATH_TO_DATASET = project_path / "web" / "api" / "data" / "dataset.pkl" # Path to the saved dataset (.pkl) file, containing the dataset class object.
# Get the path from the environment variable
PATH_TO_DATASET = os.environ.get("PATH_TO_DATASET")
# Fallback to the local path if the environment variable is not set
if PATH_TO_DATASET is None:
PATH_TO_DATASET = Path(__file__).parent / "data" / "dataset.pkl"
else:
PATH_TO_DATASET = Path(PATH_TO_DATASET)
class Dataset: