Here is my solution.
You can partition your main string using a sub-string (for e.g, "artificial-human", Used regex for this) into first, the sub_string and last.
And I tokenize the first and recursively tokenize the last and return everything.
import re
import nltk
def regex_partition(string, regex):
first = re.split(regex, string, 1)[0]
try:
last = re.split(regex, string, 1)[1]
except IndexError:
last = ''
regp = re.compile(regex)
result = regp.search(string)
try:
match = result.group()
except AttributeError:
match = ''
return first, match, last
def my_own_tokenizer(string):
first, sub_string, last = regex_partition(string, "[a-zA-Z]+-[a-zA-Z]+")
if sub_string:
tokens = my_own_tokenizer(last)
return nltk.word_tokenize(first) + [sub_string] + tokens
else:
return nltk.word_tokenize(first)
In [2]: my_own_tokenizer("hello I am an artificial-human")
Out[2]: ['hello', 'I', 'am', 'an', 'artificial-human']
In [3]: my_own_tokenizer("hello I am an artificial-human, how are you?")
Out[3]: ['hello', 'I', 'am', 'an', 'artificial-human', ',', 'how', 'are', 'you', '?']
In [4]: my_own_tokenizer("artifical-human here how are you?")
Out[4]: ['artifical-human', 'here', 'how', 'are', 'you', '?']
In [5]: my_own_tokenizer("hello-word I am an artifical-human!")
Out[5]: ['hello-word', 'I', 'am', 'an', 'artifical-human', '!']