want to add left out string in matched string

Viewed 274

Below is my example code:

from fuzzywuzzy import fuzz
import json
from itertools import zip_longest

synonyms = open("synonyms.json","r")
synonyms = json.loads(synonyms.read())

vendor_data = ["i7 processor","solid state","Corei5 :1135G7 (11th 
                       Generation)","hard 
                      drive","ddr 8gb","something1", "something2",
                      "something3","HT (100W) DDR4-2400"]

buyer_data = ["i7 processor 12 generation","corei7:latest technology"]
vendor = []
buyer = []
for item,value in synonyms.items():
    for k,k2 in zip_longest(vendor_data,buyer_data):
        for v in value:
            if fuzz.token_set_ratio(k,v) > 70:
                if item in k:
                    vendor.append(k)
                else:
                    vendor.append(item+" "+k)
            else:
                #didnt get only "something" strings here !

            if fuzz.token_set_ratio(k2,v) > 70:
                if item in k2:
                    buyer.append(k2)
                else:
                    buyer.append(item+" "+k2)

vendor = list(set(vendor))
buyer = list(set(buyer))
vendor,buyer

Note: "something" string can be anything like "battery" or "display"etc

synonyms json

{
"processor":["corei5","core","corei7","i5","i7","ryzen5","i5 processor","i7 
           processor","processor i5","processor i7","core generation","core gen"],

"ram":["DDR4","memory","DDR3","DDR","DDR 8gb","DDR 8 gb","DDR 16gb","DDR 16 gb","DDR 
                                                          32gb","DDR 32 gb","DDR4-"],

"ssd":["solid state drive","solid drive"],

"hdd":["Hard Drive"]

 }

what do i need ?

I want to add all "something" string inside vendor list dynamically.

! NOTE -- "something" string can be anything in future.

I want to add "something" string in vendor array which is not a matched value in fuzz>70! I want to basically add left out data also.

for example like below:

current output

['processor Corei5 :1135G7 (11th Generation)',
 'i7 processor',
 'ram HT (100W) DDR4-2400',
 'ram ddr 8gb',
 'hdd hard drive',
 'ssd solid state']

expected output below

 ['processor Corei5 :1135G7 (11th Generation)',
 'i7 processor',
 'ram HT (100W) DDR4-2400',
 'ram ddr 8gb',
 'hdd hard drive',
 'ssd solid state',
 'something1',
 'something2'
 'something3']  #something string need to be added in vendor list dynamically.

what silly mistake am I doing ? Thank you.

3 Answers

I have tried to come up with a decent answer (certainly not the cleanest one)

import json
from itertools import zip_longest

from fuzzywuzzy import fuzz

synonyms = open("synonyms.json", "r")
synonyms = json.loads(synonyms.read())

vendor_data = ["i7 processor", "solid state", "Corei5 :1135G7 (11thGeneration)", "hard drive", "ddr 8gb", "something1",
               "something2",
               "something3", "HT (100W) DDR4-2400"]

buyer_data = ["i7 processor 12 generation", "corei7:latest technology"]
vendor = []
buyer = []

for k, k2 in zip_longest(vendor_data, buyer_data):
    has_matched = False
    for item, value in synonyms.items():
        for v in value:
            if fuzz.token_set_ratio(k, v) > 70:
                if item in k:
                    vendor.append(k)
                else:
                    vendor.append(item + " " + k)
                if has_matched or k2 is None:
                    break
                else:
                    has_matched = True

            if fuzz.token_set_ratio(k2, v) > 70:
                if item in k2:
                    buyer.append(k2)
                else:
                    buyer.append(item + " " + k2)
                if has_matched or k is None:
                    break
                else:
                    has_matched = True
        else:
            continue  # match not found
        break  # match is found
    else:  # only evaluates on normal loop end
        # Only something strings
        # do something with the new input values
        continue  


vendor = list(set(vendor))
buyer = list(set(buyer))

I hope you can achieve what you want with this code. Check the docs if you don't know what a for else loop does. TLDR: the else clause executes when the loop terminates normally (not with a break). Note that I put the synonyms loop inside the data loop. This is because we can't certainly know in which synonym group the data belongs, also somethimes the vendor data entry is a processor while the buyer data is memory. Also note that I have assumed an item can't match more than 1 time. If this could be the case you would need to make a more advanced check (just make a counter and break when the counter equals 2 for example).

EDIT: I took another look at the question and came up with maybe a better answer:

v_dict = dict()
for spec in vendor_data[:]:
    for item, choices in synonyms.items():
        if process.extractOne(spec, choices)[1] > 70:  # don't forget to import process from fuzzywuzzy
            v_dict[spec] = item
            break
    else:
        v_dict[spec] = "Something new"

This code matches the strings to the correct type. for example {'i7 processor': 'processor', 'solid state': 'ssd', 'Corei5 :1135G7 (11thGeneration)': 'processor', 'hard drive': 'ssd', 'ddr 8gb': 'ram', 'something1': 'Something new', 'something2': 'Something new', 'something3': 'Something new', 'HT (100W) DDR4-2400': 'ram'}. You can change the "Something new" with watherver you like. You could also do: v_dict[spec] = 0 (on a match) and v_dict[spec] = 1 (on no match). You could then sort the dict ->

it = iter(v_dict.values())
print(sorted(v_dict.keys(), key=lambda x: next(it)))

Which would give the wanted results (more or less), all the recognised items will be first, and then all the unrecognised items. You could do some more advanced sorting on this dict if you want. I think this code gives you enough flexibility to reach your goal.

Here's my attempt:

from fuzzywuzzy import process, fuzz

synonyms = {'processor': ['corei5', 'core', 'corei7', 'i5', 'i7', 'ryzen5', 'i5 processor', 'i7 processor', 'processor i5', 'processor i7', 'core generation', 'core gen'], 'ram': ['DDR4', 'memory', 'DDR3', 'DDR', 'DDR 8gb', 'DDR 8 gb', 'DDR 16gb', 'DDR 16 gb', 'DDR 32gb', 'DDR 32 gb', 'DDR4-'], 'ssd': ['solid state drive', 'solid drive'], 'hdd': ['Hard Drive']}
vendor_data = ['i7 processor', 'solid state', 'Corei5 :1135G7 (11th Generation)', 'hard drive', 'ddr 8gb', 'something1', 'something2', 'something3', 'HT (100W) DDR4-2400']
buyer_data = ['i7 processor 12 generation', 'corei7:latest technology']

def find_synonym(s: str, min_score: int = 60):
    results = process.extractBests(s, choices=synonyms, score_cutoff=min_score)
    if not results:
        return None
    return results[0][-1]

def process_data(l: list, min_score: int = 60):
    matches = []
    no_matches = []
    for item in l:
        syn = find_synonym(item, min_score=min_score)
        if syn is not None:
            new_item = f'{syn} {item}' if syn not in item else item
            matches.append(new_item)
        elif any(fuzz.partial_ratio(s, item) >= min_score for s in synonyms.keys()):
            # one of the synonyms is already in the item string
            matches.append(item)
        else:
            no_matches.append(item)
    return matches, no_matches

For process_data(vendor_data) we get:

(['i7 processor',
  'ssd solid state',
  'processor Corei5 :1135G7 (11th Generation)',
  'hdd hard drive',
  'ram ddr 8gb',
  'ram HT (100W) DDR4-2400'],
 ['something1', 'something2', 'something3'])

And for process_data(buyer_data):

(['i7 processor 12 generation', 'processor corei7:latest technology'], [])

I had to lower the cut-off score to 60 to also get results for ddr 8gb. The process_data function returns 2 lists: One with matches with words from the synonyms dict and one with items without matches. If you want exactly the output you listed in your question, just concatenate the two lists like this:

matches, no_matches = process_data(vendor_data)
matches + no_matches  # ['i7 processor', 'ssd solid state', 'processor Corei5 :1135G7 (11th Generation)', 'hdd hard drive', 'ram ddr 8gb', 'ram HT (100W) DDR4-2400', 'something1', 'something2', 'something3']

If I understand correctly, what you are trying to do is match keywords specified by a customer and/or vendor against a predefined database of keywords you have.

First, I would highly recommend using a reversed mapping of the synonyms, so it's faster to lookup, especially when the dataset will grow.

Second, considering the fuzzywuzzy API, it looks like you simply want the best match, so extractOne is a solid choice for that.

Now, extractOne returns the best match and a score:

>>> process.extractOne("cowboys", choices)
    ("Dallas Cowboys", 90)

I would split the algorithm into two:

  • A generic part that simply gets the best match, which should always exist (even if it's not a great one)
  • A filter, where you could adjust the sensitivity of the algorithm, based on different criteria of your application. This sensitivity threshold should set the minimal match quality. If you're below this threshold, just use "untagged" for the category for example.

Here is the final code, which I think is very simple and easy to understand and expand:

import json
from fuzzywuzzy import process

def load_synonyms():
    with open('synonyms.json') as fin:
        synonyms = json.load(fin)

    # Reversing the map makes it much easier to lookup
    reversed_synonyms = {}
    for key, values in synonyms.items():
        for value in values:
            reversed_synonyms[value] = key                                                                                                                                                                              
    return reversed_synonyms

def load_vendor_data():
    return [
        "i7 processor",
        "solid state",
        "Corei5 :1135G7 (11thGeneration)",
        "hard drive",
        "ddr 8gb",
        "something1",
        "something2",
        "something3",
        "HT (100W) DDR4-2400"
    ]

def load_customer_data():
    return [
        "i7 processor 12 generation",
        "corei7:latest technology"
    ]

def get_tag(keyword, synonyms):
    THRESHOLD = 80                                                                                                                                                                                                        
    DEFAULT = 'general'

    tag, score = process.extractOne(keyword, synonyms.keys())                                                                                                                                                             
    return synonyms[tag] if score > THRESHOLD else DEFAULT

def main():                                                                                                                                                                                                               
    synonyms = load_synonyms()

    customer_data = load_customer_data()
    vendor_data = load_vendor_data()
    data = customer_data + vendor_data

    tags_dict = { keyword: get_tag(keyword, synonyms) for keyword in data }
    print(json.dumps(tags_dict, indent=4))
                                                                                                                                                                                                                      if __name__ == '__main__':
    main()

When running with the specified inputs, the output is:

{
    "i7 processor 12 generation": "processor",
    "corei7:latest technology": "processor",
    "i7 processor": "processor",
    "solid state": "ssd",
    "Corei5 :1135G7 (11thGeneration)": "processor",
    "hard drive": "hdd",
    "ddr 8gb": "ram",
    "something1": "general",
    "something2": "general",
    "something3": "general",
    "HT (100W) DDR4-2400": "ram"
}
Related