爬取淘宝商品数据并保存在excel中

１.re实现

 import requests

 from requests.exceptions import RequestException

 import re,json

 import xlwt,xlrd

 # 数据

 DATA = []

 KEYWORD = 'python'

 HEADERS = {'user-agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome'\

                         '/63.0.3239.132 Safari/537.36'}

 MAX_PAGE = 10

 def get_target(data_list):

     for item in data_list:

          temp = {

         'title': item['title'],

         'price': item['view_price'],

         'sales': item['view_sales'],

         'isTmall': '否' if float(item['view_fee']) else '是',

         'area': item['item_loc'],

         'name': item['nick'],

         'url': item['detail_url']

          }

          DATA.append(temp)

     return True

 # 发送http请求，获取网页源码

 def get_html(url,*args):

     try:

         if not args:

             response = requests.get(url,headers=HEADERS)

             global COOKIES

             COOKIES = response.cookies  # 获取cookie

         else:

             response = requests.get(url,headers=HEADERS,cookies=COOKIES)

         response.encoding = response.apparent_encoding

         return response.text

     except RequestException:

         print('请求源码出错！')

 # 解析源码，得到目标信息

 def parse_html(html,*args):

     if not args:

         pattern = re.compile(r'g_page_config = (.*?)g_srp_loadCss',re.S)

         # 去掉末尾的';'

         result = re.findall(pattern, html)[0].strip()[:-1]

         # 格式化json，可以用json在线解析工具查看结构

         content = json.loads(result)

         data_list = content['mods']['itemlist']['data']['auctions']

     else:

         pattern = re.compile(r'{.*}',re.S)

         result = re.findall(pattern,html)[0]

         content = json.loads(result)

         data_list = content['API.CustomizedApi']['itemlist']['auctions']

     get_target(data_list)

 def save_to_excel():

     f_name = '淘宝%s数据'%KEYWORD

     book = xlwt.Workbook(encoding='utf-8',style_compression=0)

     sheet = book.add_sheet(f_name)

     sheet.write(0, 0, 'title')

     sheet.write(0, 1, 'price')

     sheet.write(0, 2, 'sales')

     sheet.write(0, 3, 'isTmall')

     sheet.write(0, 4, 'area')

     sheet.write(0, 5, 'name')

     sheet.write(0, 6, 'url')

     for i in range(len(DATA)):

         sheet.write(i+1, 0, DATA[i]['title'])

         sheet.write(i+1, 1, DATA[i]['price'])

         sheet.write(i+1, 2, DATA[i]['sales'])

         sheet.write(i+1, 3, DATA[i]['isTmall'])

         sheet.write(i+1, 4, DATA[i]['area'])

         sheet.write(i+1, 5, DATA[i]['name'])

         sheet.write(i+1, 6, DATA[i]['url'])

     book.save('淘宝%s数据.xls'%KEYWORD)

 def main():

     for offset in range(MAX_PAGE):

         #  首页有12条异步加载的数据　api?

         if offset == 0:

             url1 = 'https://s.taobao.com/search?q={}&s={}'.format(KEYWORD,offset*44)

             html = get_html(url1)

             contents = parse_html(html)

             url2 = 'https://s.taobao.com/api?_ksTS=1532524504679_226&callback=jsonp227&ajax=true&m=customized&' \

                    'stats_click=search_radio_all:1&q={}'.format(KEYWORD)

             html = get_html(url2,2)

             contents = parse_html(html,2)

         else:

             url = 'https://s.taobao.com/search?q={}&s={}'.format(KEYWORD,offset*44)

             html = get_html(url)

             contents = parse_html(html)

     save_to_excel()

     print(len(DATA))

 if __name__ == '__main__':

     main()

爬取淘宝商品数据并保存在excel中的更多相关文章

scrapy+selenium　爬取淘宝商城商品数据存入到mongo中
1．配置信息 # 设置mongo参数 MONGO_URI = 'localhost' MONGO_DB = 'taobao' # 设置搜索关键字 KEYWORDS=['小米手机','华为手机'] # ...
使用requests、BeautifulSoup、线程池爬取艺龙酒店信息并保存到Excel中
import requests import time, random, csv from fake_useragent import UserAgent from bs4 import Beauti ...
使用requests、re、BeautifulSoup、线程池爬取携程酒店信息并保存到Excel中
import requests import json import re import csv import threadpool import time, random from bs4 impo ...
Python爬取猫眼电影100榜并保存到excel表格
首先我们前期要导入的第三方类库有; 通过猫眼电影100榜的源码可以看到很有规律如: 亦或者是: 根据规律我们可以得到非贪婪的正则表达式 """<div class ...
爬取拉勾网所有python职位并保存到excel表格对象方式
# 1.把之间案例,使用bs4,正则,xpath,进行数据提取. # 2.爬取拉钩网上的所有python职位. from urllib import request,parse import json ...
Python 爬取淘宝商品数据挖掘分析实战
Python 爬取淘宝商品数据挖掘分析实战项目内容本案例选择>> 商品类目:沙发: 数量:共100页 4400个商品: 筛选条件:天猫.销量从高到低.价格500元以上. 爬取淘宝商品 ...
Selenium+Chrome/phantomJS模拟浏览器爬取淘宝商品信息
#使用selenium+Carome/phantomJS模拟浏览器爬取淘宝商品信息 # 思路: # 第一步:利用selenium驱动浏览器,搜索商品信息,得到商品列表 # 第二步:分析商品页数,驱动浏 ...
python3编写网络爬虫16-使用selenium 爬取淘宝商品信息
一.使用selenium 模拟浏览器操作爬取淘宝商品信息之前我们已经成功尝试分析Ajax来抓取相关数据,但是并不是所有页面都可以通过分析Ajax来完成抓取.比如,淘宝,它的整个页面数据确实也是通过A ...
Python爬虫，抓取淘宝商品评论内容!
作为一个资深吃货,网购各种零食是很频繁的,但是能否在浩瀚的商品库中找到合适的东西,就只能参考评论了!今天给大家分享用python做个抓取淘宝商品评论的小爬虫! 思路我们就拿"德州扒鸡&qu ...

随机推荐

java源码 -- AbstractList
AbstractList AbstractList是实现List接口的抽象类,AbstractList抽象类与List接口的关系类似于AbstractCollection抽象类与Collection接 ...
Destination高级特性
一.组合队列 Composite Destinations 组合队列允许用一个虚拟的destination代表多个destinations.这样就可以通过composite destinations在 ...
zookeeper安装和使用 windows环境
简介 ZooKeeper是一个分布式的,开放源码的分布式应用程序协调服务,是Google的Chubby一个开源的实现,是Hadoop和Hbase的重要组件.它是一个为分布式应用提供一致性服务的软件,提 ...
cnn健康增胖和调理好身体
吃不胖,其实大部分情况是消化系统不好,大部分食物都没有被身体吸收就被排掉了. 1,改善肠胃消化功能: 每天早上一杯全脂鲜牛奶(或者羊奶), 每天晚上一杯酸奶 ps:白天和鲜牛奶可以激发肠胃的消化能力. ...
springboot启动流程（一）构造SpringApplication实例对象
所有文章 https://www.cnblogs.com/lay2017/p/11478237.html 启动入口本文是springboot启动流程的第一篇,涉及的内容是SpringApplicat ...
基于【 Docker】三 || Docker的基本操作
一.Docker常用命令 1.搜索镜像:docker search 镜像名称 2.下载镜像:docker pull 镜像名称:版本号 3.查看镜像:docker images 4.删除镜像:docke ...
Java 面向对象（七）多态
一.多态概述(Polymorphism) 1.引入多态是继封装.继承之后,面向对象的第三大特性. 通过不同的事物,体现出来的不同的形态.多态,描述的就是这样的状态.如跑的动作,每个动物的跑的动作就是 ...
SuperMemo method
原文:https://www.supermemo.com/en/archives1990-2015/english/ol/sm2 别人的研究:http://wdxtub.lofter.com/post ...
C和指针--命令行参数
1.命令行参数 C程序的main函数具有两个形参,第1个通常称为argc,它表示命令行参数的数目.第2个称为argv,它指向一组参数值.由于参数的数目并没有内在的限制,所以argv指向这组参数值(本质 ...
利用setuptools发布Python程序到PyPI，为Python添砖加瓦
pip install的东西从哪里来的? 从PyPI (Python Package Index)来的,官网是: https://pypi.python.org/pypi/执行pip install ...

爬取淘宝商品数据并保存在excel中

爬取淘宝商品数据并保存在excel中的更多相关文章

随机推荐

热门专题