You signed in with another tab or window. Reload to refresh your session.You signed out in another tab or window. Reload to refresh your session.You switched accounts on another tab or window. Reload to refresh your session.Dismiss alert
'''Removes HTML tags: replaces anything between opening and closing <> with empty space'''
return TAG_RE.sub('', text)
class CustomPreprocess():
'''Cleans text data up, leaving only 2 or more char long non-stepwords composed of A-Z & a-z only
in lowercase'''
def __init__(self):
pass
def preprocess_text(self,sen):
sen = sen.lower()
# Remove html tags
sentence = remove_tags(sen)
# Remove punctuations and numbers
sentence = re.sub('[^a-zA-Z]', ' ', sentence)
# Single character removal
sentence = re.sub(r"\s+[a-zA-Z]\s+", ' ', sentence) # When we remove apostrophe from the word "Mark's", the apostrophe is replaced by an empty space. Hence, we are left with single character "s" that we are removing here.
# Remove multiple spaces
sentence = re.sub(r'\s+', ' ', sentence) # Next, we remove all the single characters and replace it by a space which creates multiple spaces in our text. Finally, we remove the multiple spaces from our text as well.