BeautifulSoup

BeautifulSoup是一个模块，该模块用于接收一个HTML或XML字符串，然后将其进行格式化，之后遍可以使用他提供的方法进行快速查找指定元素，从而使得在HTML或XML中查找指定元素变得简单。

举个简单的例子对其进行运用

from django.test import TestCase

# Create your tests here.

from bs4 import BeautifulSoup

html_doc = """

<html><head><title>The Dormouse's story</title></head>

<body>

asdf

<div class="title">

<b>The Dormouse's story总共</b>

<h1>f</h1>

</div>

<div class="story">Once upon a time there were three little sisters; and their names were

<a class="sister0" id="link1">Els<span>f</span>ie</a>,

<a href="http://example.com/lacie" class="sister" id="link2">Lacie</a> and

<a href="http://example.com/tillie" class="sister" id="link3">Tillie</a>;

and they lived at the bottom of a well.</div>

ad<br/>sf

<p class="story">...</p>

</body>

</html>

"""

soup = BeautifulSoup(html_doc, features="lxml")

# 找到第一个a标签

# tag1 = soup.find(name='a')

# 循环所有标签

from bs4.element import Tag

from bs4.element import NavigableString

for tag in soup.body.descendants:

if isinstance(tag,Tag):

# print(tag.name,tag.attrs)

pass

# 内容

# tag1 = soup.find(name='a')

# tag1.clear()

# print(soup)

# 属性

tag1 = soup.find(name='a')

del tag1.attrs['class']

print(soup)

常用参数介绍　　

name，标签名称

 tag = soup.find('a')

 name = tag.name # 获取该标签的名字

 print(name)

 tag.name = 'span' # 设置

 print(soup)

attr，标签属性

 tag = soup.find('a')

 attrs = tag.attrs # 获取

 print(attrs)

 tag.attrs = {'k':1}

 tag.attrs['id'] = ''

 print(soup)

children,所有子标签

 body = soup.find('body')

 v = body.children

children,所有子子孙孙标签

 body = soup.find('body')

 v = body.descendants

clear,将标签的所有子标签全部清空（保留标签名）

 tag = soup.find('body')

 tag.clear()

 print(soup)

decompose,递归的删除所有的标签

 body = soup.find('body')

 body.decompose()

 print(soup)

extract,递归的删除所有的标签，并获取删除的标签

 body = soup.find('body')

 v = body.extract()

 print(soup)

decode,转换为字符串（含当前标签）；decode_contents（不含当前标签）

 body = soup.find('body')

 v = body.decode()

 v = body.decode_contents()

 print(v)

encode,转换为字节（含当前标签）；encode_contents（不含当前标签）

 body = soup.find('body')

 v = body.encode()

 v = body.encode_contents()

 print(v)

find,获取匹配的第一个标签

 tag = soup.find('a')

 print(tag)

 tag = soup.find(name='a', attrs={'class': 'sister'}, recursive=True, text='Lacie')

 tag = soup.find(name='a', class_='sister', recursive=True, text='Lacie')

 print(tag)

find_all,获取匹配的所有标签

 tags = soup.find_all('a')

 print(tags)

 tags = soup.find_all('a',limit=1)

 print(tags)

 tags = soup.find_all(name='a', attrs={'class': 'sister'}, recursive=True, text='Lacie')

 # tags = soup.find(name='a', class_='sister', recursive=True, text='Lacie')

 print(tags)

 ####### 列表 #######

 v = soup.find_all(name=['a','div'])

 print(v)

 v = soup.find_all(class_=['sister0', 'sister'])

 print(v)

 v = soup.find_all(text=['Tillie'])

 print(v, type(v[0]))

 v = soup.find_all(id=['link1','link2'])

 print(v)

 v = soup.find_all(href=['link1','link2'])

 print(v)

 ####### 正则 #######

 importre

 rep = re.compile('p')

 rep = re.compile('^p')

 v = soup.find_all(name=rep)

 print(v)

 rep = re.compile('sister.*')

 v = soup.find_all(class_=rep)

 print(v)

 rep = re.compile('http://www.oldboy.com/static/.*')

 v = soup.find_all(href=rep)

 print(v)

 ####### 方法筛选 #######

 def func(tag):

 return tag.has_attr('class') and tag.has_attr('id')

 v = soup.find_all(name=func)

 print(v)

 ## get,获取标签属性

 tag = soup.find('a')

 v = tag.get('id')

 print(v)

has_attr,检查标签是否具有该属性

 tag = soup.find('a')

 v = tag.has_attr('id')

 print(v)

get_text,获取标签内部文本内容

 tag = soup.find('a')

 v = tag.get_text('id')

 print(v)

index,检查标签在某标签中的索引位置

 tag = soup.find('body')

 v = tag.index(tag.find('div'))

 print(v)

 tag = soup.find('body')

 for i,v in enumerate(tag):

 print(i,v)

is_empty_element,是否是空标签(是否可以是空)或者自闭合标签，

判断是否是如下标签：'br' , 'hr', 'input', 'img', 'meta','spacer', 'link', 'frame', 'base'

 tag = soup.find('br')

 v = tag.is_empty_element

 print(v)

当前的关联标签

 # soup.next

 # soup.next_element

 # soup.next_elements

 # soup.next_sibling

 # soup.next_siblings

 #

 # tag.previous

 # tag.previous_element

 # tag.previous_elements

 # tag.previous_sibling

 # tag.previous_siblings

 #

 # tag.parent

 # tag.parents

查找某标签的关联标签

 # tag.find_next(...)

 # tag.find_all_next(...)

 # tag.find_next_sibling(...)

 # tag.find_next_siblings(...)

 # tag.find_previous(...)

 # tag.find_all_previous(...)

 # tag.find_previous_sibling(...)

 # tag.find_previous_siblings(...)

 # tag.find_parent(...)

 # tag.find_parents(...)

 # 参数同find_all

select,select_one, CSS选择器

 soup.select("title")

 soup.select("p nth-of-type(3)")

 soup.select("body a")

 soup.select("html head title")

 tag = soup.select("span,a")

 soup.select("head > title")

 soup.select("p > a")

 soup.select("p > a:nth-of-type(2)")

 soup.select("p > #link1")

 soup.select("body > a")

 soup.select("#link1 ~ .sister")

 soup.select("#link1 + .sister")

 soup.select(".sister")

 soup.select("[class~=sister]")

 soup.select("#link1")

 soup.select("a#link2")

 soup.select('a[href]')

 soup.select('a[href="http://example.com/elsie"]')

 soup.select('a[href^="http://example.com/"]')

 soup.select('a[href$="tillie"]')

 soup.select('a[href*=".com/el"]')

 from bs4.element import Tag

 def default_candidate_generator(tag):

     for child in tag.descendants:

         if not isinstance(child, Tag):

             continue

         if not child.has_attr('href'):

             continue

         yield child

 tags = soup.find('body').select("a", _candidate_generator=default_candidate_generator)

 print(type(tags), tags)

 from bs4.element import Tag

 def default_candidate_generator(tag):

     for child in tag.descendants:

         if not isinstance(child, Tag):

             continue

         if not child.has_attr('href'):

             continue

         yield child

 tags = soup.find('body').select("a", _candidate_generator=default_candidate_generator, limit=1)

 print(type(tags), tags)

标签的内容

 # tag = soup.find('span')

 # print(tag.string)          # 获取

 # tag.string = 'new content' # 设置

 # print(soup)

 # tag = soup.find('body')

 # print(tag.string)

 # tag.string = 'xxx'

 # print(soup)

 # tag = soup.find('body')

 # v = tag.stripped_strings  # 递归内部获取所有标签的文本

 # print(v)

append在当前标签内部追加一个标签

 # tag = soup.find('body')

 # tag.append(soup.find('a'))

 # print(soup)

 #

 # from bs4.element import Tag

 # obj = Tag(name='i',attrs={'id': 'it'})

 # obj.string = '我是一个新来的'

 # tag = soup.find('body')

 # tag.append(obj)

 # print(soup)

insert在当前标签内部指定位置插入一个标签

 from bs4.element import Tag

 obj = Tag(name='i', attrs={'id': 'it'})

 obj.string = '我是一个新来的'

 tag = soup.find('body')

 tag.insert(2, obj)

 print(soup)

insert_after,insert_before 在当前标签后面或前面插入

 # from bs4.element import Tag

 # obj = Tag(name='i', attrs={'id': 'it'})

 # obj.string = '我是一个新来的'

 # tag = soup.find('body')

 # # tag.insert_before(obj)

 # tag.insert_after(obj)

 # print(soup)

replace_with 在当前标签替换为指定标签

 from bs4.element import Tag

 obj = Tag(name='i', attrs={'id': 'it'})

 obj.string = '我是一个新来的'

 tag = soup.find('div')

 tag.replace_with(obj)

 print(soup)

创建标签之间的关系

 tag = soup.find('div')

 a = soup.find('a')

 tag.setup(previous_sibling=a)

 print(tag.previous_sibling)

wrap，将指定标签把当前标签包裹起来

 # from bs4.element import Tag

 # obj1 = Tag(name='div', attrs={'id': 'it'})

 # obj1.string = '我是一个新来的'

 #

 # tag = soup.find('a')

 # v = tag.wrap(obj1)

 # print(soup)

 # tag = soup.find('a')

 # v = tag.wrap(soup.find('p'))

 # print(soup)

unwrap，去掉当前标签，将保留其包裹的标签

更多详情见官方网站：http://beautifulsoup.readthedocs.io/zh_CN/v4.4.0/

BeautifulSoup详解的更多相关文章

requests+BeautifulSoup详解
简介 Python标准库中提供了:urllib.urllib2.httplib等模块以供Http请求,但是,它的 API 太渣了.它是为另一个时代.另一个互联网所创建的.它需要巨量的工作,甚至包括各种 ...
Python爬虫系列-BeautifulSoup详解
安装 pip3 install beautifulsoup4 解析库解析器使用方法优势劣势 Python标准库 BeautifulSoup(markup,'html,parser') Pyth ...
BeautifulSoup 模块详解
BeautifulSoup 模块详解 BeautifulSoup是一个模块,该模块用于接收一个HTML或XML字符串,然后将其进行格式化,之后遍可以使用他提供的方法进行快速查找指定元素,从而使得在HT ...
BuautifulSoup4库详解
1.BeautifulSoup4库简介 What is beautifulsoup ? 答:一个可以用来从HTML 和 XML中提取数据的网页解析库,支持多种解析器(代替正则的复杂用法) 2.安装 p ...
史上最全python面试题详解（一）（附带详细答案（关注、持续更新））
python基础题(53道题详解) 1.简述解释型和编译型编程语言? 概念: 编译型语言:把做好的源程序全部编译成二进制代码的可运行程序.然后,可直接运行这个程序. 解释型语言:把做好的源程序翻译一句 ...
利用wxpy进行微信信息发送详解（一）
利用wxpy进行微信信息自动发送,简直是骚扰神器,除非拉黑你. 那我们就来设置一个场景吧,五秒发送一次,一次发送10条首先我们来整理一下思路: ♦1.首先我们要从网上爬去我们想要发送的内容 ♦2.登 ...
python爬虫知识点详解
python爬虫知识点总结(一)库的安装 python爬虫知识点总结(二)爬虫的基本原理 python爬虫知识点总结(三)urllib库详解 python爬虫知识点总结(四)Requests库的基本使 ...
Scrapy笔记03- Spider详解
Scrapy笔记03- Spider详解 Spider是爬虫框架的核心,爬取流程如下: 先初始化请求URL列表,并指定下载后处理response的回调函数.初次请求URL通过start_urls指定, ...
Scrapy笔记04- Selector详解
Scrapy笔记04- Selector详解在你爬取网页的时候,最普遍的事情就是在页面源码中提取需要的数据,我们有几个库可以帮你完成这个任务: BeautifulSoup是python中一个非常流行 ...

随机推荐

【翻译】使用Sencha Ext JS创建美丽的图画（1）
原文:Creating Beautiful Drawings Using Sencha Ext JS – Part 1 许多人可能对Ext JS中的图表包相当熟悉了.通过它可以快速创建相当强悍的可视化 ...
android官方技术文档翻译——Case 标签中的常量字段
本文译自androd官方技术文档<Non-constant Fields in Case Labels>,原文地址:http://tools.android.com/tips/non-co ...
Java的字符串分割的不同实现
在java中实现字符串的分割相对而言是很简单的.我们一般会采取两中方式.一个是从jdk1.1就开始的StringTokenizer类,另一个是调用split方法进行分割.下面请看代码: import ...
OC:打僵尸问题(类的问题)
1.定义普通僵尸类: 实例变量:僵尸种类.僵尸总血量.僵尸每次失血量. 方法:初始化方法(设置僵尸种类,总血量).被打击失血.死亡. 2.定义路障僵尸类: 实例变量:僵尸种类.僵尸总血量.僵尸每次失血 ...
（五十九）iOS网络基础之UIWebView简易浏览器实现
[UIWebView网络浏览器] 通过webView的loadRequest方法可以发送请求显示相应的网站,例如: NSURL *url = [NSURL URLWithString:@"h ...
【翻译】EXTJS 编码风格指南与实例
原文:EXTJS Code Style Guide with examples Ext JS风格指南: 熟知的且易于学习快速开发,易于调试,轻松部署组织良好.可扩展和可维护 Ext JS应用程序的 ...
android 向webview传值
android中可以使用WebView加载网页,同时Android端的java代码可以与网页上的javascript代码之间相互调用. 效果图: (一)Android部分: 布局代码: <spa ...
OpenCV 矩形轮廓检测
转载请注明出处:http://blog.csdn.net/wangyaninglm/article/details/44151213, 来自:shiter编写程序的艺术基础介绍 OpenCV里提取目 ...
iOS开发讲解SDWebImage，你真的会用吗？
SDWebImage作为目前最受欢迎的图片下载第三方框架,使用率很高.但是你真的会用吗?本文接下来将通过例子分析如何合理使用SDWebImage. 使用场景:自定义的UITableViewCell上有 ...
根据isbn获得图书的所有信息
几点说明 1这个豆瓣的api https://api.douban.com/v2/book/isbn/:9787549208869 可以以json的形式返回书籍的所有信息 2最开始的时候是我自己写的用 ...

BeautifulSoup详解

BeautifulSoup

常用参数介绍

BeautifulSoup详解的更多相关文章

随机推荐

热门专题

常用参数介绍