I'm using the code below for evaluate the pos tagger that I trained in spacy 2.3.0 (built from source- last commit : 9860b8399ed2a3d1d680e1c1cd31d85926422709):
def evaluate(nlp, examples):
scorer = Scorer()
nlp.tokenizer = Tokenizer(nlp.vocab)
for input_, annot in examples:
doc_gold_text = nlp.make_doc(input_)
gold = GoldParse(doc_gold_text, tags=annot['tags'])
pred_value = nlp(input_)
scorer.score(pred_value, gold)
return scorer.scores
def main(model='model'):
test_data = train_data_getter()[80000:]
nlp = spacy.load(model)
# nlp = None
pp = pprint.PrettyPrinter()
pp.pprint(evaluate(nlp, test_data))
and the result is:
{'ents_f': 0.0,
'ents_p': 0.0,
'ents_per_type': {},
'ents_r': 0.0,
'las': 0.0,
'las_per_type': {'': {'f': 0.0, 'p': 0.0, 'r': 0.0}},
'tags_acc': 88.9152111081426,
'textcat_score': 0.0,
'textcats_per_cat': {},
'token_acc': 100.0,
'uas': 0.0}
I wonder if there is a way to get accuracy per every part-of-speech tag?