BeautifulSoup

视频地址：
链接: https://pan.baidu.com/s/1rOwBJQ3DDbzT4id9hZe-VQ 密码: mynx

解析库

解析器	使用方法	优势	劣势
Python标准库	BeautifulSoup(markup, "html.parser")	Python的内置标准库、执行速度适中、文档容错能力强	Python 2.7.3 or 3.2.2)前的版本中文容错能力差
lxml HTML 解析器	BeautifulSoup(markup, "lxml")	速度快、文档容错能力强	需要安装C语言库
lxml XML 解析器	BeautifulSoup(markup, "xml")	速度快、唯一支持XML的解析器	需要安装C语言库
html5lib	BeautifulSoup(markup, "html5lib")	最好的容错性、以浏览器的方式解析文档、生成HTML5格式的文档	速度慢、不依赖外部扩展

基本使用

html = """
The Dormouse's story

The Dormouse's story
Once upon a time there were three little sisters; and their names were
,
Lacie and
Tillie;
and they lived at the bottom of a well.
...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.prettify())
print(soup.title.string)


 
  
   The Dormouse's story
  
 
 
  
   
    The Dormouse's story
   
  
  
   Once upon a time there were three little sisters; and their names were
   
    
   
   ,
   
    Lacie
   
   and
   
    Tillie
   
   ;
and they lived at the bottom of a well.
  
  
   ...
  
 

The Dormouse's story

标签选择器

选择元素

html = """
The Dormouse's story

The Dormouse's story
Once upon a time there were three little sisters; and their names were
,
Lacie and
Tillie;
and they lived at the bottom of a well.
...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.title)
print(type(soup.title))
print(soup.head)
print(soup.p)

The Dormouse's story

The Dormouse's story
The Dormouse's story

获取名称

html = """
The Dormouse's story

The Dormouse's story
Once upon a time there were three little sisters; and their names were
,
Lacie and
Tillie;
and they lived at the bottom of a well.
...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.title.name)

title

获取属性

html = """
The Dormouse's story

The Dormouse's story
Once upon a time there were three little sisters; and their names were
,
Lacie and
Tillie;
and they lived at the bottom of a well.
...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.p.attrs['name'])
print(soup.p['name'])

dromouse
dromouse

获取内容

html = """
The Dormouse's story

The Dormouse's story
Once upon a time there were three little sisters; and their names were
,
Lacie and
Tillie;
and they lived at the bottom of a well.
...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.p.string)

The Dormouse's story

嵌套选择

html = """
The Dormouse's story

The Dormouse's story
Once upon a time there were three little sisters; and their names were
,
Lacie and
Tillie;
and they lived at the bottom of a well.
...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.head.title.string)

The Dormouse's story

子节点和子孙节点

html = """

    
        The Dormouse's story
    
    
        
            Once upon a time there were three little sisters; and their names were
            
                Elsie
            
            Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
        ...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.p.contents)

['\n            Once upon a time there were three little sisters; and their names were\n            ', 
Elsie
, '\n', Lacie, ' \n            and\n            ', Tillie, '\n            and they lived at the bottom of a well.\n        ']

html = """

    
        The Dormouse's story
    
    
        
            Once upon a time there were three little sisters; and their names were
            
                Elsie
            
            Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
        ...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.p.children)
for i, child in enumerate(soup.p.children):
    print(i, child)


0 
            Once upon a time there were three little sisters; and their names were
            
1 
Elsie

2 

3 Lacie
4  
            and
            
5 Tillie
6 
            and they lived at the bottom of a well.

html = """

    
        The Dormouse's story
    
    
        
            Once upon a time there were three little sisters; and their names were
            
                Elsie
            
            Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
        ...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.p.descendants)
for i, child in enumerate(soup.p.descendants):
    print(i, child)


0 
            Once upon a time there were three little sisters; and their names were
            
1 
Elsie

2 

3 Elsie
4 Elsie
5 

6 

7 Lacie
8 Lacie
9  
            and
            
10 Tillie
11 Tillie
12 
            and they lived at the bottom of a well.

父节点和祖先节点

html = """

    
        The Dormouse's story
    
    
        
            Once upon a time there were three little sisters; and their names were
            
                Elsie
            
            Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
        ...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.a.parent)


            Once upon a time there were three little sisters; and their names were
            
Elsie

Lacie 
            and
            Tillie
            and they lived at the bottom of a well.

html = """

    
        The Dormouse's story
    
    
        
            Once upon a time there were three little sisters; and their names were
            
                Elsie
            
            Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
        ...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(list(enumerate(soup.a.parents)))

[(0, 
            Once upon a time there were three little sisters; and their names were
            
Elsie

Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        ), (1, 

            Once upon a time there were three little sisters; and their names were
            
Elsie

Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
...
), (2, 

The Dormouse's story



            Once upon a time there were three little sisters; and their names were
            
Elsie

Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
...
), (3, 

The Dormouse's story



            Once upon a time there were three little sisters; and their names were
            
Elsie

Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
...
)]

兄弟节点

html = """

    
        The Dormouse's story
    
    
        
            Once upon a time there were three little sisters; and their names were
            
                Elsie
            
            Lacie 
            and
            Tillie
            and they lived at the bottom of a well.
        
        ...
"""
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(list(enumerate(soup.a.next_siblings)))
print(list(enumerate(soup.a.previous_siblings)))

[(0, '\n'), (1, Lacie), (2, ' \n            and\n            '), (3, Tillie), (4, '\n            and they lived at the bottom of a well.\n        ')]
[(0, '\n            Once upon a time there were three little sisters; and their names were\n            ')]

标准选择器

find_all( name , attrs , recursive , text , **kwargs )

可根据标签名、属性、内容查找文档

name

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.find_all('ul'))
print(type(soup.find_all('ul')[0]))

[
Foo
Bar
Jay
, 
Foo
Bar
]

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
for ul in soup.find_all('ul'):
    print(ul.find_all('li'))

[Foo
, Bar
, Jay]
[Foo
, Bar]

attrs

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.find_all(attrs={'id': 'list-1'}))
print(soup.find_all(attrs={'name': 'elements'}))

[
Foo
Bar
Jay
]
[
Foo
Bar
Jay
]

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.find_all(id='list-1'))
print(soup.find_all(class_='element'))

[
Foo
Bar
Jay
]
[Foo
, Bar
, Jay
, Foo
, Bar]

text

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.find_all(text='Foo'))

['Foo', 'Foo']

find( name , attrs , recursive , text , **kwargs )

find返回单个元素，find_all返回所有元素

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.find('ul'))
print(type(soup.find('ul')))
print(soup.find('page'))


Foo
Bar
Jay


None

find_parents() find_parent()

find_parents()返回所有祖先节点，find_parent()返回直接父节点。

find_next_siblings() find_next_sibling()

find_next_siblings()返回后面所有兄弟节点，find_next_sibling()返回后面第一个兄弟节点。

find_previous_siblings() find_previous_sibling()

find_previous_siblings()返回前面所有兄弟节点，find_previous_sibling()返回前面第一个兄弟节点。

find_all_next() find_next()

find_all_next()返回节点后所有符合条件的节点, find_next()返回第一个符合条件的节点

find_all_previous() 和 find_previous()

find_all_previous()返回节点后所有符合条件的节点, find_previous()返回第一个符合条件的节点

CSS选择器

通过select()直接传入CSS选择器即可完成选择

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
print(soup.select('.panel .panel-heading'))
print(soup.select('ul li'))
print(soup.select('#list-2 .element'))
print(type(soup.select('ul')[0]))

[
Hello
]
[Foo
, Bar
, Jay
, Foo
, Bar]
[Foo
, Bar]

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
for ul in soup.select('ul'):
    print(ul.select('li'))

[Foo
, Bar
, Jay]
[Foo
, Bar]

获取属性

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
for ul in soup.select('ul'):
    print(ul['id'])
    print(ul.attrs['id'])

list-1
list-1
list-2
list-2

获取内容

html='''

    
        Hello
    
    
        
            Foo
            Bar
            Jay
        
        
            Foo
            Bar
        
    

'''
from bs4 import BeautifulSoup
soup = BeautifulSoup(html, 'lxml')
for li in soup.select('li'):
    print(li.get_text())

Foo
Bar
Jay
Foo
Bar

总结

推荐使用lxml解析库，必要时使用html.parser
标签选择筛选功能弱但是速度快
建议使用find()、find_all() 查询匹配单个结果或者多个结果
如果对CSS选择器熟悉建议使用select()
记住常用的获取属性和文本值的方法

BeautifulSoup库详解

BeautifulSoup

解析库

基本使用

标签选择器

选择元素

获取名称

获取属性

获取内容

嵌套选择

子节点和子孙节点

父节点和祖先节点

兄弟节点

标准选择器

find_all( name , attrs , recursive , text , **kwargs )

name

Hello

Hello

attrs

Hello

Hello

text

Hello

find( name , attrs , recursive , text , **kwargs )

Hello

find_parents() find_parent()

find_next_siblings() find_next_sibling()

find_previous_siblings() find_previous_sibling()

find_all_next() find_next()

find_all_previous() 和 find_previous()

CSS选择器

Hello

Hello

Hello

获取属性

Hello

获取内容

Hello

总结

你可能感兴趣的:(BeautifulSoup库详解)