Move Your Bloomin' Parse

Web Scraping in Python

Thomas Laetsch

Data Scientist, NYU

Once Again

class DCspider( scrapy.Spider ):
    name = "dcspider"

    def start_requests( self ):
        urls = [ 'https://www.datacamp.com/courses/all' ]
        for url in urls:
            yield scrapy.Request( url = url, callback = self.parse )

    def parse( self, response ):
        # simple example: write out the html
        html_file = 'DC_courses.html'
        with open( html_file, 'wb' ) as fout:
            fout.write( response.body )

You Already Know!

def parse( self, response ):

    # input parsing code with response that you already know!

    # output to a file, or...

    # crawl the web!

DataCamp Course Links: Save to File

class DCspider( scrapy.Spider ):
    name = "dcspider"

    def start_requests( self ):
        urls = [ 'https://www.datacamp.com/courses/all' ]
        for url in urls:
            yield scrapy.Request( url = url, callback = self.parse )

    def parse( self, response ):

        links = response.css('div.course-block > a::attr(href)').extract()

        filepath = 'DC_links.csv'
        with open( filepath, 'w' ) as f:
            f.writelines( [link + '/n' for link in links] )

DataCamp Course Links: Parse Again

class DCspider( scrapy.Spider ):
    name = "dcspider"

    def start_requests( self ):
        urls = [ 'https://www.datacamp.com/courses/all' ]
        for url in urls:
            yield scrapy.Request( url = url, callback = self.parse )

    def parse( self, response ):

        links = response.css('div.course-block > a::attr(href)').extract()

        for link in links:
            yield response.follow( url = link, callback = self.parse2 )

    def parse2( self, response ):
        # parse the course sites here!

A spider following links through the DataCamp website.

Johnny Parsin'

Web Scraping in Python