I have a crawler that parse wikipedia page (Rwanda Genocide). As you can see it works perfect for one URL, and it finds occurences number of 'genocide ' word. But what I am trying to do is parsing more than one url (let's say I wanna crawl total 8 genocide links.) All wikipedia links have the same HTML pattern so paragraph and body_text variables will work for all URL's. How can I read and save other URL in the same python file ?
import requests
import urllib.request
import time
from bs4 import BeautifulSoup
import numpy as np
import pandas as pd
from urllib.request import urlopen
from collections import Counter
import re
def my_url_function(url):
URL = 'https://en.wikipedia.org/wiki/Rwandan_genocide'
response = requests.get(urls)
soup = BeautifulSoup(response.text, 'html.parser')
paragraph = soup.find('div',{'id':'mw-content-text'}, {'class':'mw-content-ltr'}).tbody
#print(paragraph)
body_text = soup.findAll("div", class_="mw-parser-output")[0].findAll('p')
body_text_big = ""
for i in body_text:
body_text_big = body_text_big +i.text
print(body_text_big)
text_file = open("Output.txt", "w")
text_file.write(body_text_big)
text_file.close()
occurrences = body_text_big.count("genocide")
print('Number of occurrences of the word :', occurrences)
# read URLs into list `urls`
with open("urls.txt", "r") as urlsFile:
urls = urlsFile.read().splitlines()
print(urls)