python爬取亚马逊书籍信息代码分享_Python

我有个需求就是抓取一些简单的书籍信息存储到mysql数据库，例如，封面图片，书名，类型，作者，简历，出版社，语种。

我比较之后，决定在亚马逊来实现我的需求。

我分析网站后发现，亚马逊有个高级搜索的功能，我就通过该搜索结果来获取书籍的详情URL。

由于亚马逊的高级搜索是用get方法的，所以通过分析，搜索结果的URL，可得到node参数是代表书籍类型的。field-binding_browse-bin是代表书籍装饰。

所以我固定了书籍装饰为平装，而书籍的类型，只能每次运行的时候，爬取一种类型的书籍难过

之后就是根据书籍详情页面利用正则表达式来匹配需要的信息了。

以下源代码，命名不是很规范。。。

				?

									import requests

									import sys

									import re

									import pymysql

									class product:

									  type="历史"

									  name=""

									  author=""

									  desciption=""

									  pic1=""

									  languages=""

									  press=""

									def getProUrl():

									  urlList = []

									  headers = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.102 Safari/537.36"}

									  session = requests.Session()

									  furl="https://www.amazon.cn/gp/search/ref=sr_adv_b/?search-alias=stripbooks&field-binding_browse-bin=2038564051&sort=relevancerank&page="

									  for i in range(1,7):

									    html=""

									    print(furl+str(i)) 

									    html = session.post(furl+str(i)+'&node=658418051',headers = headers)

									    html.encoding = 'utf-8'

									    s=html.text.encode('gb2312','ignore').decode('gb2312')

									    url=r'</li><li id=".*?" data-asin="(.+?)" class="s-result-item celwidget">'

									    reg=re.compile(url,re.M)

									    items = reg.findall(html.text)

									    for i in range(0,len(items)):

									      urlList.append(items[i])

									  urlList=set(urlList)

									  return urlList

									def getProData(url):

									  pro = product()

									  headers = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/50.0.2661.102 Safari/537.36"}

									  session = requests.Session()

									  zurl="https://www.amazon.cn/dp/"

									  html = session.get(zurl+url,headers = headers)

									  html.encoding = 'utf-8'

									  s=html.text.encode('gb2312','ignore').decode('gb2312')

									  pro.pic1=getProPic(html)

									  pro.name=getProName(html)

									  pro.author=getProAuthor(html)

									  pro.desciption=getProDescrip(html)

									  pro.press=getProPress(html)

									  pro.languages=getProLanguages(html)

									  return pro

									def getProPic(html):

									  pic=r'id="imgBlkFront" data-a-dynamic-image="{&quot;(.+?)&quot;.*?}"'

									  reg=re.compile(pic,re.M)

									  items = reg.findall(html.text)

									  if len(items)==0:

									    return ""

									  else:

									    return items[0]

									def getProName(html):

									  name=r'<div class="ma-title"><p class="wraptext goto-top">(.+?)<span'

									  reg=re.compile(name,re.M)

									  items = reg.findall(html.text)

									  if len(items)==0:

									    return ""

									  else:

									    return items[0]

									def getProAuthor(html):

									  author=r'<span class="author.{0,20}" data-width="".{0,30}>.*?<a class="a-link-normal" href=".*?books" rel="external nofollow" >(.+?)</a>.*?<span class="a-color-secondary">(.+?)</span>'

									  reg=re.compile(author,re.S)

									  items = reg.findall(html.text)

									  au=""

									  for i in range(0,len(items)):

									    au=au+items[i][0]+items[i][1]

									  return au

									def getProDescrip(html):

									  Descrip=r'<noscript>.{0,30}<div>(.+?)</div>.{0,30}<em></em>.{0,30}</noscript>.{0,30}<div id="outer_postBodyPS"'

									  reg=re.compile(Descrip,re.S)

									  items = reg.findall(html.text)

									  if len(items)==0:

									    return ""

									  else:

									    position = items[0].find('海报：')

									    descrip=items[0]

									    if position != -1:

									      descrip=items[0][0:position]

									    return descrip.strip()

									def getProPress(html):

									  press=r'<li><b>出版社:</b>(.+?)</li>'

									  reg=re.compile(press,re.M)

									  items = reg.findall(html.text)

									  if len(items)==0:

									    return ""

									  else:

									    return items[0].strip()

									def getProLanguages(html):

									  languages=r'<li><b>语种：</b>(.+?)</li>'

									  reg=re.compile(languages,re.M)

									  items = reg.findall(html.text)

									  if len(items)==0:

									    return ""

									  else:

									    return items[0].strip()

									def getConnection():

									  config = {

									     'host':'121.**.**.**',

									     'port':3306,

									     'user':'root',

									     'password':'******',

									     'db':'home_work',

									     'charset':'utf8',

									     'cursorclass':pymysql.cursors.DictCursor,

									     }

									  connection = pymysql.connect(**config)

									  return connection

									urlList = getProUrl()

									i = 0

									for d in urlList:

									  i = i + 1

									  print (i)

									  connection = getConnection()

									  pro = getProData(d)

									  try:

									    with connection.cursor() as cursor:

									      sql='INSERT INTO books (type,name,author,desciption,pic1,languages,press) VALUES (%s,%s,%s,%s,%s,%s,%s)'

									      cursor.execute(sql,(pro.type,pro.name,pro.author,pro.desciption,pro.pic1,pro.languages,pro.press))

									    connection.commit()

									  finally:

									    connection.close();