I am trying to make parallelize calls to an API. The API has a limit of 1,200 calls per minute before it stops. What is the most efficient way to async this while being below the limit?
def remove_html_tags(text):
"""Remove html tags from a string"""
import re
clean = re.compile('<.*?>')
return re.sub(clean, ' ', text)
async def getRez(df, url):
async with aiohttp.ClientSession() as session:
auth = aiohttp.BasicAuth('username',pwd)
r = await session.get(url, auth=auth)
if r.status == 200:
content = await r.text()
text = remove_html_tags(str(content))
else:
text = '500 Server Error'
df.loc[df['url'] == url, ['RezText']] = [[text]]
df['wordCount'] = df['RezText'].apply(lambda x: len(str(x).split(" ")))
data = df[df["RezText"] != "500 Server Error"]
async def main(df):
df['RezText'] = None
await asyncio.gather(*[getRez(df, url) for url in df['url']])
loop = asyncio.get_event_loop()
loop.run_until_complete(main(data))