I am using Fasttext (from Gensim). I have two issues I don't know how to solve:
- I would like to set a threshold for the vocabulary to the 100,000 most frequent words. 2. I would like to ensure that a list of words (from a text file) are part of the vocabulary as well. Say this list of words is in a text file called
list.txt.
How would I do this?
Here is my model:
from gensim.models import FastText
class paragraph_generator(object):
def __init__(self,test=True,itersize=10000,year=None,state=None):
self.test=test
self.itersize=itersize
self.sql = f"""
SELECT
text_id,
lccn_sn,
date,
ed,
chroniclingamerica_meta.statefp,
chroniclingamerica_meta.countyfp,
text_ocr
FROM
chroniclingamerica natural join chroniclingamerica_meta
WHERE date_part('year',date) BETWEEN 1870 AND 1920
AND seq = 1 """
if self.test:
self.sql = self.sql+' limit 10000' # limit 1000 means it only goes through 1000 lines of the database
else:
pass
print(self.sql)
def __iter__(self):
con, cur = database_connection.connect(cursor_type='server')
cur.itersize = self.itersize
cur.execute(self.sql)
for p in cur.fetchall():
tokens = p[-1].translate(str.maketrans('', '', punct)).replace('\n',' ').lower().split(' ')
yield tokens
con.close()
model = FastText(vector_size=256, window=8, min_count=10, epochs=1, workers=workers)
vocab = model.build_vocab(paragraph_generator(test=False, itersize=10000, year=None, state=None))
model.train(paragraph_generator(test=False, itersize=10000, year=None, state=None),
total_examples=model.corpus_count, epochs=1)
I'm thinking of a mix between the parameters total_words and sorted_vocab, but I would not know how to do this.
Many thanks in advance for your answers!