1.豆瓣

爬取单个页面数据

import requests
from lxml import etree
#import os url = "https://movie.douban.com/cinema/nowplaying/yongzhou/"
headers = {
'User-Agent':'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/68.0.3440.106 Safari/537.36'
}
req = requests.get(url=url,headers=headers)
text = req.text
dics = []
#将抓取下来的数据根据一定的规则进行提取
html = etree.HTML(text)
ul = html.xpath("//ul[@class='lists']")[0]
#print(etree.tostring(ul,encoding='utf-8').decode('utf-8'))
lis = ul.xpath("./li")
for li in lis:
title = li.xpath("@data-title")[0]
score = li.xpath("@data-actors")[0]
adress = li.xpath("@data-region")[0]
img_hai = li.xpath(".//img/@src")[0]
dic = {
'title':title,
'score':score,
'adress':adress,
'img':img_hai
}
dics.append(dic)
print(dics)

2.电影天堂

爬取多个页面数据

import requests
import json
from lxml import etree
url = "http://www.dytt8.net"
HEADERS = {
'User-Agent':'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/68.0.3440.106 Safari/537.36',
'Referer':'http://www.dytt8.net/html/gndy/dyzz/list_23_1.html'
} def get_url(urls):
response = requests.get(urls,headers=HEADERS)
text = response.text #请求页面
html = etree.HTML(text) #解析为HTML文档 html为Element对象 (可以执行xpath语法)
detail_urls = html.xpath("//table[@class='tbspan']//a/@href") #获取页面下的href
detail_urls = map(lambda urls:url+urls,detail_urls) #将detail_urls这个列表中每个url都扔给lambda这个函数合并 再将整个修改后的赋给detail_urls
return detail_urls def parse_detail_url(de_ur):
movie = {}
response = requests.get(de_ur,headers=HEADERS)
text = response.content.decode('gbk')
html = etree.HTML(text)
title = html.xpath("//div[@class='title_all']//font[@color='#07519a']/text()")[0] #获取标题
movie['title'] = title #放入字典
zoomE = html.xpath("//div[@id='Zoom']")[0]
img_hb = zoomE.xpath(".//img/@src")
cover = img_hb[0] #海报
#sst = img_hb[1] #电影截图
movie['cover'] = cover
#movie['sst'] = sst def parse_info(info,rule):
return info.replace(rule,"").strip() #.strip()把前后空格删掉
infos = zoomE.xpath(".//text()")
for index,info in enumerate(infos): #enumerate 索引序列(0 str 1 str 2 str)
if info.startswith("◎片  名"): #判断 以。。开始
info = parse_info(info,"◎片  名") #调用parse_info将"◎片  名"替换为无(没有)
movie['pian'] = info
elif info.startswith("◎年  代"):
info = parse_info(info, "◎年  代")
movie['year'] = info
elif info.startswith("◎产  地"):
info = parse_info(info, "◎产  地")
movie['adress'] = info
elif info.startswith("◎导  演"):
info = parse_info(info, "◎导  演")
movie['actor'] = info
elif info.startswith("◎类  别"):
info = parse_info(info, "◎类  别")
movie['lb'] = info
elif info.startswith("◎豆瓣评分"):
info = parse_info(info, "◎豆瓣评分")
movie['db'] = info
elif info.startswith("◎主  演"):
info = parse_info(info, "◎主  演")
actors = []
for x in range(index+1,len(infos)):
actor = infos[x]
if actor.startswith("◎"): #过滤简介部分
break
actors.append(actor)
movie['actors'] = actors
elif info.startswith("◎简  介"):
info = parse_info(info,"◎简  介")
for x in range(index+1,len(infos)):
profile = infos[x].strip()
if profile.startswith("【"): #过滤下载地址部分
break
movie['profile'] = profile
download_url = html.xpath("//td[@bgcolor='#fdfddf']/a/@href")[0] #下载地址
movie['download_url'] = download_url
return movie def write_to_file(content):
with open('result.txt','a',encoding='utf-8') as f:
f.write(json.dumps(content,ensure_ascii=False)+'\n') #ensure_ascii=False 输出为中文
f.close() def dianying():
urld = "http://www.dytt8.net/html/gndy/dyzz/list_23_{}.html" #这里用到了{} .format()的用法
movies = [] #定义一个列表
for x in range(1,8):
#第一个for循环用来控制7个页面
print(x)
urls = urld.format(x)
if x==5: #这里因为第5个页面出现报错信息 可能是编码问题 解决不了 所以我就过滤了第5页
continue
detail_ur = get_url(urls) #解析每页的详细信息
write_to_file("第%s页" % x)
for detail_url in detail_ur:
#第二个for循环用来遍历每个页
movie = parse_detail_url(detail_url)
movies.append(movie)
write_to_file(movie) if __name__ == '__main__':
dianying()

3.腾讯招聘

跟上一个电影天堂的代码差不多

import requests
import json
from lxml import etree
url = "https://hr.tencent.com/"
HEADERS = {
'User-Agent':'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/68.0.3440.106 Safari/537.36'
}
def get_url(urld):
response = requests.get(urld,headers=HEADERS)
text = response.text
html = etree.HTML(text)
detail_url = html.xpath("//tr[@class='even' or @class='odd']//a/@href")
detail_url = map(lambda x:url+x,detail_url) return detail_url def prease_url(detail_url):
dic = {}
#print(detail_url)
response = requests.get(detail_url,headers=HEADERS)
text =response.text
html = etree.HTML(text)
title = html.xpath("//tr[@class='h']//td[@class='l2 bold size16']//text()")[0]
dic['title'] = title #方法一 (死板)
adress = html.xpath("//tr[@class='c bottomline']//td//text()")[1]
dic['adress'] = adress
# 方法二 (简洁)
str = html.xpath("//tr[@class='c bottomline']//td")
leibie = str[1].xpath(".//text()")[1]
dic['leibie'] = leibie
nums = str[2].xpath(".//text()")[1]
dic['nums'] = nums
gz = html.xpath("//ul[@class='squareli']")
gzzz = gz[0].xpath(".//text()")
gzyq = gz[1].xpath(".//text()")
dic['工作职责'] = gzzz
dic['工作要求'] = gzyq
#print(dic)
return dic def write_to_file(content):
with open('tengxun.txt','a',encoding='utf-8') as f:
f.write(json.dumps(content,ensure_ascii=False)+'\n') #ensure_ascii=False 输出为中文
f.close()
def tengxun():
movies = []
urls = "https://hr.tencent.com/position.php?keywords=python&lid=0&tid=87&start={}#a"
for x in range(0,501,10): #步长为10
print(x)
urld = urls.format(x)
detail_urls = get_url(urld)
for detail_url in detail_urls:
movie = prease_url(detail_url)
movies.append(movie)
write_to_file(x)
write_to_file(movies) if __name__ == '__main__':
tengxun()

lxml爬取实验的更多相关文章

  1. 爬虫---lxml爬取博客文章

    上一篇大概写了下lxml的用法,今天我们通过案例来实践,爬取我的博客博客并保存在本地 爬取博客园博客 爬取思路: 1.首先找到需要爬取的博客园地址 2.解析博客园地址 # coding:utf-8 i ...

  2. lxml的使用(节点与xpath爬取数据)

    lxml安装 lxml是python下功能很丰富的XML和HTML解析库,性能非常的好,是对libxml3和libxlst的封装.在Windows下载这个库直接使用 pip install lxml ...

  3. 爬虫入门(四)——Scrapy框架入门:使用Scrapy框架爬取全书网小说数据

    为了入门scrapy框架,昨天写了一个爬取静态小说网站的小程序 下面我们尝试爬取全书网中网游动漫类小说的书籍信息. 一.准备阶段 明确一下爬虫页面分析的思路: 对于书籍列表页:我们需要知道打开单本书籍 ...

  4. 用Python爬取了考研吧1000条帖子,原来他们都在讨论这些!

    写在前面 考研在即,想多了解考研er的想法,就是去找学长学姐或者去网上搜索,贴吧就是一个好地方.而借助强大的工具可以快速从网络鱼龙混杂的信息中得到有价值的信息.虽然网上有很多爬取百度贴吧的教程和例子, ...

  5. Python3爬虫系列:理论+实验+爬取妹子图实战

    Github: https://github.com/wangy8961/python3-concurrency-pics-02 ,欢迎star 爬虫系列: (1) 理论 Python3爬虫系列01 ...

  6. Python爬虫使用lxml模块爬取豆瓣读书排行榜并分析

    上次使用了BeautifulSoup库爬取电影排行榜,爬取相对来说有点麻烦,爬取的速度也较慢.本次使用的lxml库,我个人是最喜欢的,爬取的语法很简单,爬取速度也快. 本次爬取的豆瓣书籍排行榜的首页地 ...

  7. lxml xpath 爬取并正常显示中文内容

    在使用python爬虫提取中文网页的内容,为了能正确显示中文的内容,在转为字符串时一定要声明编码为utf-8,否则无法正常显示中文,而是显示原编码的字符,并没有正确转换.比如下面这个简单的爬取百度页面 ...

  8. Python爬虫——使用 lxml 解析器爬取汽车之家二手车信息

    本次爬虫的目标是汽车之家的二手车销售信息,范围是全国,不过很可惜,汽车之家只显示100页信息,每页48条,也就是说最多只能够爬取4800条信息. 由于这次爬虫的主要目的是使用lxml解析器,所以在信息 ...

  9. Python爬虫爬取豆瓣电影之数据提取值xpath和lxml模块

    工具:Python 3.6.5.PyCharm开发工具.Windows 10 操作系统.谷歌浏览器 目的:爬取豆瓣电影排行榜中电影的title.链接地址.图片.评价人数.评分等 网址:https:// ...

随机推荐

  1. feed.snapdo.com 病毒

    过程:安装破解office2013 使用破解工具 Microsoft toolkit 2.7 beta  1 关闭防火墙 浏览器被木马篡改:搜索引擎被篡改: 相关进程 bittorrent.exe 无 ...

  2. bzoj4490 随机数生成器Ⅱ加强版

    题目链接 题意 给出参数\(C_1,C_2,P\)按如下方式生成一个长度为\(n \times m\)的序列\(x\): \(x_0 = C_1,x_1=C2\) \(x_i=(x_{i-1}+x_{ ...

  3. jemter+ant+jenkins进行集成测试

    一下为我学习的一些笔记: 一.安装配置ant 安装地址:http://ant.apache.org/ 1.下载ant一路傻瓜式安装 2.配置ant环境变量:path下配置ant的bin路径 3.将jm ...

  4. Python的安装与小程序的编写

    Python的安装 在此之前,我完全不了解Python,为了完成任务,在慌忙之中了解了一下Python,通过百度,一步步安装好Python 过程 1.从官网中找到下载菜单并下载最新版本 2.双击pyt ...

  5. Centos6安装Percona-tools工具

    Centos6安装Percona-tools工具 环境:centos6.x yum -y install perl-DBI yum -y install perl-DBD-MySQL yum -y i ...

  6. NFV-Based Scalable Guaranteed-Bandwidth Multicast Service for Software Defined ISP Networks

    文章名称:NFV-Based Scalable Guaranteed-Bandwidth Multicast Service for Software Defined ISP Networks 发表时 ...

  7. freetypeLCD显示

    目录 freetypeLCD显示 安装交叉编译环境 配置 头文件和库的位置 编译安装 复制到PC编译工具链 复制到文件系统 运行测试 LCD显示 编码转换问题 简单显示 角度旋转 换行显示 居中显示 ...

  8. CentOS6和CentOS7

    http://mirrors.aliyun.com/centos/7/isos/x86_64/CentOS-7-x86_64-DVD-1708.iso net.ifnames= biosdevname ...

  9. JGUI源码:开发中遇到的问题(11)

    1.IE8下浏览器下css body边缘要留一个像素,如果不留的话,很有可能看不到最边缘的像素. 2.同一种颜色在深色背景和浅色背景下给人的感觉不一样,在深色背景下,给人感觉特别亮,所以深色背景下的颜 ...

  10. 使用容器编排工具docker swarm安装clickhouse多机集群

    1.首先需要安装docker最新版,docker 目前自带swarm容器编排工具 2.选中一台机器作为master,执行命令sudo docker  swarm init [options] 3,再需 ...