Why does my webcrawler not follow into the next link containing keywords

Viewed 441

I have written a simple webcrawler that will eventually follow only news link to scrape the article text into a database. I am having problems actually following the link from the source url. This is the code so far:

import urlparse
import mechanize

url ="https://news.google.co.uk"

def spider(root, steps):
    urls = [root]
    visited =[root]
    counter = 0
    while counter < steps:
        step_url = scrape(urls)
        urls = []
        for u in step_url:
            if u not in visited:
                urls.append(u)
                visited.append(u)
        counter+=1
    return visited

def scrape(root):
    result_urls = []
    br = Browser()
    br.set_handle_robots(False)
    br.addheaders = [('User-agent', 'Chrome')]
    for url in root:
        try:
            br.open(url)
            keyWords = ['news','article','business', 'world']
            for link in br.links():
                newurl = urlparse.urljoin(link.base_url,link.url)
                result_urls.append(newurl)
                [newslinks for newslinks in result_urls if newslinks in keyWords]
                print newslinks
        except:
            print "scrape error"
    return result_urls

print spider(url, 2)

Edit:NLTK

 `for text in (parse_links_text(get_links(url), d)):
    tokenized = nltk.word_tokenize(text)
    tagged = nltk.pos_tag(tokenized)
    namedEnt = nltk.ne_chunk(tagged, binary=True)
    entities = re.findall(r'NE\s(.*?)/',str(namedEnt))
    descriptives = re.findall(r'\(\'(\w*)\',\s\'JJ\w?\'', str(tagged))`

then add to database after this.

1 Answers
Related