It is not recognizing 'text' and I don't know why. I think it has to do with this (r'(\w*)(\w)\2(\w*)'). Cause in PyCharm is appears yellow, not green. enter image description here import nltk from nltk.corpus import wordnet import re import pandas as pd nltk.download('wordnet') nltk.download('omw-1.4')
def remove_repeated_characters(text):
repeat_pattern = re.compile(r'(\w*)(\w)\2(\w*)')
match_substitution = r'\1\2\3'
def replace(old_word):
if wordnet.synsets(old_word):
return old_word
new_word = repeat_pattern.sub(match_substitution, old_word)
return replace(new_word) if new_word != old_word else new_word
print(text)
correct_text = text.apply(lambda x:''.join([replace(word) for word in x.split(' ')]))
return correct_text
file = 'bbc_news_homework_3.csv'
df = pd.read_csv(file)
df_2 = remove_repeated_characters(df['text'].str.lower())
print(df_2)