Solving memory issues when using Gensim LDA Multicore

Viewed 207

For my project I am trying to use unsupervised learning to identify different topics from application descriptions, but I am running into a strange problem. Firstly, I have 3 different datasets, one with 15k documents another with 50k documents and last with 2m documents. I am trying to test models with different number of topics (k) ranging from 5 to 100 with a step size of 5. This is in order to check which k results in the best model assessed with initially with the highest coherence score. For each k, I also build 3 different models with chunksize 10, 100 and 1000.

So now moving onto the problem I am having. Obviously my own machine is too slow and does not have enough cores for this kind of computation hence I am using my university's server. The problem here is my program seems to be consuming too much memory and I am unsure of the reason. I already made some adjustments such that the corpus is not loaded entirely to memory (or atleast I think I did). The dataset with 50k entries already at iteration k=50 (so halfway) seems to have consumed the alloted 100GB of memory, which seems very huge.

I would appreciate any help in the right direction and thanks for taking the time to look at this. Below is the code from my topic_modelling.py file. Comments on the file are a bit outdated, sorry about that.

class MyCorpus:
    texts: list
    dictionary: dict

    def __init__(self, descriptions, dictionary):
        self.texts = descriptions
        self.dictionary = dictionary

    def __iter__(self):
        for line in self.texts:
            try:
            # assume there's one document per line, tokens separated by whitespace
                yield self.dictionary.doc2bow(line)
            except StopIteration:
                pass

# Function given a dataframe creates a dictionary and corupus
# These are used to create an LDA model. Here we automatically use the Descriptionb column
# from each dataframe
def create_dict_and_corpus(df):
    text_descriptions = remove_characters_and_create_list(df, 'Description')
    # print(text_descriptions)
    dictionary = gensim.corpora.Dictionary(text_descriptions)
    corpus = MyCorpus(text_descriptions, dictionary)
    return text_descriptions, dictionary, corpus

# Given a dataframe remove and a column name in the data frame, extract all words and return a list
# Also to remove all chracters that are not alphanumeric or spaces
def remove_characters_and_create_list(df, column_name, split=True):
    df[column_name] = df[column_name].astype(str)
    texts = []
    for x in range(df[column_name].size):
        current_string = df[column_name][x]
        filtered_string = re.sub(r'[^A-Za-z0-9 ]+', '', current_string)
        if split:
            texts.append(filtered_string.split())
        else:
            texts.append(filtered_string)
    return texts


# This function given the parameters creates an LDA model for each number between
# the start limit and the end limit. After this the coherence and perplexity is calulated
# for each of those models and saved in a csv file to analyze later.
def test_lda_models(text, corpus, dictionary, start_limit, end_limit, path):
    results = []
    print("============Starting topic modelling============")
    for k in range(start_limit, end_limit+1, 5):
        for p in range(1, 4):
            chunk = pow(10, p)
            t0 = time.time()
            lda_model = gensim.models.ldamulticore.LdaMulticore(corpus,
                                            num_topics=k, 
                                            id2word=dictionary, 
                                            passes=p,
                                            chunksize=chunk)
            # To calculate the goodness of the model
            perplexity = lda_model.bound(corpus)
            coherence_model_lda = CoherenceModel(model=lda_model, texts=text, dictionary=dictionary, coherence='c_v')
            coherence_lda = coherence_model_lda.get_coherence()
            t1 = time.time()
            print(f"=====Done K={k} model with passes={p} and chunksize={chunk}, took {t1-t0} seconds=====")
            results.append((k, chunk, coherence_lda, perplexity))
        
    # Storing teh results in a csv file except the actual lda model (this would not make sense)
    path = make_dir_if_not_exists(path)
    list_tuples_to_csv(results, ['#OfTopics', 'ChunkSize', 'CoherenceScore', 'Perplexity'], f"{path}/K={start_limit}to{end_limit}.csv")
    return results


# Function plot the visualization of an LDA model. This visualization is then
# saved as an html file inside the given path
def single_lda_model_visualization(k, c, corpus, dictionary, lda_model, path):
    vis = gensimvis.prepare(lda_model, corpus, dictionary)
    pyLDAvis.save_html(vis, f"{path}/visualization.html")


# Given the results produced by test_lda_models, loop though the models and save the
# topic words of each model and the visualization of the topics in the given path
def save_lda_result(k, c, lda_model, corpus, dictionary, path):
    list_tuples_to_csv(lda_model.print_topics(num_topics=k), ['Topic#', 'Associated Words'], f"{path}/associated_words.csv")
    single_lda_model_visualization(k, c, corpus, dictionary, lda_model, path)

# This is the entire pipeline that needs to be performed for a single dataset,
# which includes computing the LDA models from start to end limit and calculating 
# and saving the topic words and visual graphs for the top n topics with the highest
# coherence score.
def perform_topic_modelling_single_df(df, start_limit, end_limit, path):

    # Extracting the necessary data required for LDA model computation
    text_descriptions,dictionary, corpus = create_dict_and_corpus(df)

    results_lda = test_lda_models(text_descriptions, corpus, dictionary, start_limit, end_limit, path)
    # Sorting the results based on the 2nd tuple value returned which is 'coherence'
    results_lda.sort(key=lambda x:x[2],reverse=True)

    # Getting the top 5 results to save pass to save_lda_results function
    results = results_lda[:5]
    corpus_for_saving = [dictionary.doc2bow(text) for text in text_descriptions]
    texts = remove_characters_and_create_list(df, 'Description', split=False)

    # Perfrom application to topic modelling for the best lda model based on the
    # coherence score (TODO maybe test with other lda models?)
    print("getting descriptions for csv")
    for k, c, _, _ in results:
        dir_path = make_dir_if_not_exists(f"{path}/k={k}_chunk={c}")
        p = int(math.log10(c))
        lda_model = gensim.models.ldamulticore.LdaMulticore(corpus,
                                            num_topics=k, 
                                            id2word=dictionary, 
                                            passes=p,
                                            chunksize=c)
        print(f"=====REDOING K={k} model with passes={p} and chunksize={c}=====")
        save_lda_result(k,c, lda_model, corpus_for_saving, dictionary, dir_path)
        application_to_topic_modelling(df, k, c, lda_model, corpus_for_saving, texts, dir_path)


# Performs the whole topic modelling pipeline taking different genre data sets 
# and the entire dataset as a whole
def perform_topic_modelling_pipeline(path_ex):
    # entire_df = pd.read_csv("../data/preprocessed_data/preprocessed_10000_trial.csv")
    entire_df = pd.read_csv(os.path.join(ROOT_DIR, f"data/preprocessed_data/preprocessedData_{path_ex}.csv"))
    print("size of df")
    print(entire_df.shape)
    # For entire df go from start limit to ngenres to find best LDA model
    nGenres = row_counter(os.path.join(ROOT_DIR, f"data/genre_wise_data/data{path_ex}/genre_frequency.csv"))
    nGenres_rounded = math.ceil(nGenres / 5) * 5
    print(f"Original number of genres should be {nGenres}, but we are rounding to {nGenres_rounded}")
    path = make_dir_if_not_exists(os.path.join(ROOT_DIR, f"results/data{path_ex}/aall_data"))
    perform_topic_modelling_single_df(entire_df, 5, 100, path)
0 Answers
Related