mirror of
https://github.com/wassname/stampy-chat.git
synced 2026-09-11 12:50:34 +08:00
Added helper functions for sentence-splitting and article splitting
This commit is contained in:
@@ -0,0 +1,61 @@
|
||||
import re
|
||||
from typing import List
|
||||
|
||||
# FROM https://stackoverflow.com/a/31505798/16185542
|
||||
# -*- coding: utf-8 -*-
|
||||
alphabets= "([A-Za-z])"
|
||||
prefixes = "(Mr|St|Mrs|Ms|Dr)[.]"
|
||||
suffixes = "(Inc|Ltd|Jr|Sr|Co)"
|
||||
starters = "(Mr|Mrs|Ms|Dr|Prof|Capt|Cpt|Lt|He\s|She\s|It\s|They\s|Their\s|Our\s|We\s|But\s|However\s|That\s|This\s|Wherever)"
|
||||
acronyms = "([A-Z][.][A-Z][.](?:[A-Z][.])?)"
|
||||
websites = "[.](com|net|org|io|gov|edu|me)"
|
||||
digits = "([0-9])"
|
||||
|
||||
def split_into_sentences(text):
|
||||
text = " " + text + " "
|
||||
text = text.replace("\n"," ")
|
||||
text = text.replace("?!", "?")
|
||||
text = re.sub(prefixes,"\\1<prd>",text)
|
||||
text = re.sub(websites,"<prd>\\1",text)
|
||||
text = re.sub(digits + "[.]" + digits,"\\1<prd>\\2",text)
|
||||
if "..." in text: text = text.replace("...","<prd><prd><prd>")
|
||||
if "Ph.D" in text: text = text.replace("Ph.D.","Ph<prd>D<prd>")
|
||||
text = re.sub("\s" + alphabets + "[.] "," \\1<prd> ",text)
|
||||
text = re.sub(acronyms+" "+starters,"\\1<stop> \\2",text)
|
||||
text = re.sub(alphabets + "[.]" + alphabets + "[.]" + alphabets + "[.]","\\1<prd>\\2<prd>\\3<prd>",text)
|
||||
text = re.sub(alphabets + "[.]" + alphabets + "[.]","\\1<prd>\\2<prd>",text)
|
||||
text = re.sub(" "+suffixes+"[.] "+starters," \\1<stop> \\2",text)
|
||||
text = re.sub(" "+suffixes+"[.]"," \\1<prd>",text)
|
||||
text = re.sub(" " + alphabets + "[.]"," \\1<prd>",text)
|
||||
if "”" in text: text = text.replace(".”","”.")
|
||||
if "\"" in text: text = text.replace(".\"","\".")
|
||||
if "!" in text: text = text.replace("!\"","\"!")
|
||||
if "?" in text: text = text.replace("?\"","\"?")
|
||||
text = text.replace(".",".<stop>")
|
||||
text = text.replace("?","?<stop>")
|
||||
text = text.replace("!","!<stop>")
|
||||
text = text.replace("<prd>",".")
|
||||
|
||||
sentences = text.split("<stop>")
|
||||
sentences = sentences[:-1]
|
||||
sentences = [s.strip() for s in sentences]
|
||||
|
||||
if sentences == []:
|
||||
sentences = [text.strip()]
|
||||
return sentences
|
||||
|
||||
|
||||
def split_article(text: str) -> List[str]: # THIS IS COMPLETELY BROKEN AND WRONG. TODO: FIX IT.
|
||||
# Receives one text (str) and returns a list of sections (List[str]), each section being a few appended paragraphs that do not exceed 1000 words.
|
||||
# This is done to avoid the 8000 token limit of OpenAI embeddings.
|
||||
sections = []
|
||||
section = ""
|
||||
paragraphs = text.split('\n')
|
||||
for paragraph in paragraphs:
|
||||
if paragraph == "": continue
|
||||
if len(section.split()) + len(paragraph.split()) > 1000 or len(section) + len(paragraph) > 7000:
|
||||
sections.append(section)
|
||||
section = ""
|
||||
section += f"{paragraph}\n"
|
||||
sections.append(section)
|
||||
return sections
|
||||
Reference in New Issue
Block a user