TheVoxcraft/web_crawler.py

## web_crawler.py
import requests
from bs4 import BeautifulSoup #NEEDS bs4 installed!


def trade_spider(max_page):
    page = 1
    while page <= max_page:
        url = "http://www.finn.no/finn/realestate/homes/result?location=0%2F20061&page=" + str(page)
        source_code = requests.get(url)
        plain_text = source_code.text
        soup = BeautifulSoup(plain_text)
        for link in soup.findAll('a', {'data-fth-event': 'searchclickthrough'}):
            href = link.get('href')
            title = link.string  # just the text
            if title != "None": #Problem
                print(" ")
                print(title)
                print(href)


        page += 1


trade_spider(30)
	import requests
	from bs4 import BeautifulSoup #NEEDS bs4 installed!


	def trade_spider(max_page):
	page = 1
	while page <= max_page:
	url = "http://www.finn.no/finn/realestate/homes/result?location=0%2F20061&page=" + str(page)
	source_code = requests.get(url)
	plain_text = source_code.text
	soup = BeautifulSoup(plain_text)
	for link in soup.findAll('a', {'data-fth-event': 'searchclickthrough'}):
	href = link.get('href')
	title = link.string # just the text
	if title != "None": #Problem
	print(" ")
	print(title)
	print(href)


	page += 1


	trade_spider(30)