Date: 2014-03-06

from scrapy.contrib.linkextractors.sgml import SgmlLinkExtractor
from scrapy.contrib.spiders import CrawlSpider, Rule        #这个是预定义的蜘蛛,使用它可以自定义爬取链接的规则rule
from scrapy.selector import HtmlXPathSelector               #导入HtmlXPathSelector进行解析
from firstScrapy.items import FirstscrapyItem

class firstScrapy(CrawlSpider):
    name = "firstScrapy"                                    #爬虫的名字要唯一
    allowed_domains = [""]                   #运行爬取的网页
    start_urls = [""]   #第一个爬取的网页
    rules = [Rule(SgmlLinkExtractor(allow=(‘/ebook/[^/]+fr=booklist‘)), callback=‘myparse‘),
             Rule(SgmlLinkExtractor(allow=(‘/book/list/[^/]+pn=[^/]+‘, )), follow=True)]

    def myparse(self, response):
        x = HtmlXPathSelector(response)
        item = FirstscrapyItem()

        # get item
        item[‘link‘] = response.url
        item[‘title‘] = ""
        strlist ="//h1/@title").extract()
        if len(strlist) > 0:
            item[‘title‘] = strlist[0]
        # return the item
        return item


from scrapy.item import Item, Field

class FirstscrapyItem(Item):
    title = Field(serializer=str)
    link = Field(serializer=str)

(3) 第三个文件是,由于要连接数据库,这边用到了twisted连接mysql的方法

# Define your item pipelines here
# Don‘t forget to add your pipeline to the ITEM_PIPELINES setting
# See:

from twisted.enterprise import adbapi              #导入twisted的包
import MySQLdb
import MySQLdb.cursors

class FirstscrapyPipeline(object):
    def __init__(self):                            #初始化连接mysql的数据库相关信息
        self.dbpool = adbapi.ConnectionPool(‘MySQLdb‘,
                db = ‘bookInfo‘,
                user = ‘root‘,
                passwd = ‘123456‘,
                cursorclass = MySQLdb.cursors.DictCursor,
                charset = ‘utf8‘,
                use_unicode = False

    # pipeline dafault function                    #这个函数是pipeline默认调用的函数
    def process_item(self, item, spider):
        query = self.dbpool.runInteraction(self._conditional_insert, item)
        return item

    # insert the data to databases                 #把数据插入到数据库中
    def _conditional_insert(self, tx, item):
        sql = "insert into book values (%s, %s)"
        tx.execute(sql, (item["title"], item["link"]))





