I am trying to write a pattern in spaCy that matches against "black" but not "black beans."
I tried the code below, but it seems to match the token that is next to "black" so long as it is not "bean." How do I modify to match against only "black"?
nlp = spacy.load("en_core_web_sm")
matcher = Matcher(nlp.vocab)
#pattern = [{"LOWER": "black"}, {"LEMMA": {"NOT_IN": ["bean", "beans"]}}]
pattern = [{"LOWER": "black"}, {"LEMMA": "bean", "OP": "!"}]
matcher.add("blackbeans", [pattern])
doc = nlp("I liked the black beans, but the avocado was black making the whole meal blackish-looking and not good.")
matches = matcher(doc)
for match_id, start, end in matches:
string_id = nlp.vocab.strings[match_id] # Get string representation
span = doc[start:end] # The matched span
print(match_id, string_id, start, end, span.text)