I'm trying to find a way in python to identify action verbs, cognition verbs, stative verbs in a text.
Below is my code. I found a rule of action verbs. Is it correct? I don't have any idea to find the cognition and stative verbs (rule of cognition and stative verbs). Anyone can help me to find these rules?
from __future__ import unicode_literals
import nltk
from nltk import pos_tag
from nltk.corpus import wordnet
# word tokenizeing and part-of-speech tagger
document = 'For the on-board control, the rover navigation controller should limit the speed to 5 km/h with a target speed of 10 km/h.'
tokens = [nltk.word_tokenize(sent) for sent in [document]]
postag = [nltk.pos_tag(sent) for sent in tokens][0]
# Rule for action verbs
grammar = r"""
NBAR:
{<NN.*>?<VB.*><RB.*>?} #Action verbs
AV:
{<NBAR>}
{<NBAR><IN><NBAR>} # Above, connected with in/of/etc...
"""
# Chunking
cp = nltk.RegexpParser(grammar)
# the result is a tree
tree = cp.parse(postag)
print(tree)
def leaves(tree):
for subtree in tree.subtrees(filter=lambda t: t.label() == ‘AV’):
yield subtree.leaves()
def get_word_postag(word):
if pos_tag([word])[0][1].startswith('J'):
return wordnet.ADJ
if pos_tag([word])[0][1].startswith('V'):
return wordnet.VERB
if pos_tag([word])[0][1].startswith('N'):
return wordnet.NOUN
else:
return wordnet.NOUN
def normalise(word):
"""Normalises words to lowercase."""
word = word.lower()
return word
def get_terms(tree):
for leaf in leaves(tree):
terms = [normalise(w) for w, t in leaf]
yield terms
terms = get_terms(tree)
features = []
for term in terms:
_term = ''
for word in term:
_term += ' ' + word
features.append(_term.strip())
print(features)
Output: limit #limit is an action verb in document
Also, how can I find trigger words at the first of the sentence(i.e. Start with a trigger word)