博客园  :: 首页  :: 新随笔  :: 联系 :: 订阅 订阅  :: 管理

python3+ 简单爬虫笔记

Posted on 2017-11-15 11:52  遇见龙卷风  阅读(135)  评论(0)    收藏  举报
 1 import urllib.request
 2 import re
 3 
 4 def getHtml(url):
 5     html = urllib.request.urlopen(url).read()
 6     return html
 7 
 8 def getImg(html):
 9     reg = r'src="(.+?\.jpg)" pic_ext'
10     imgre = re.compile(reg)
11     html = html.decode('utf-8')
12     imglist = re.findall(imgre,html)
13 
14     x = 0
15 
16     for imgurl in imglist:
17         urllib.request.urlretrieve(imgurl,'%s.jpg' %x)
18         x += 1
19     return imglist
20 
21 html = getHtml("http://tieba.baidu.com/p/2460150866")
22 print(getImg(html))