Python网页爬虫

#!/usr/bin/python
import urllib2
import re
 
def downURL(url,filename):
    print url
    print filename
    try:
        fp = urllib2.urlopen(url)
    except:
        print 'download exception'
        return 0
    op = open(filename,"wb")
    while 1:
        s = fp.read()
        if not s:
            break
        op.write(s)
 
    fp.close()
    op.close()
    return 1
 
#downURL('http://www.sohu.com','http.log')
 
def getURL(url):
    try:
        fp = urllib2.urlopen(url)
    except:
        print 'get url exception'
        return []
     
    pattern = re.compile("http://sports.sina.com.cn/[^\>]+.shtml")
    while 1:
        s = fp.read()
        if not s:
            break
        urls = pattern.findall(s)
    fp.close()
    return urls
 
def spider(startURL,times):
    urls = []
    urls.append(startURL)
    i = 0
    while 1:
        if i > times:
            break;
        if len(urls)>0:
            url = urls.pop(0)
            print url,len(urls)
            downURL(url,str(i)+'.htm')
            i = i + 1
            if len(urls)<times:
                urllist = getURL(url)
                for url in urllist:
                    if urls.count(url) == 0:
                        urls.append(url)
        else:
            break
    return 1
spider('http://www.163.com',10)

查看全文

相关阅读:
nginx安装
 Mybatis使用generator自动生成映射配置文件信息
 Mysql报错Fatal error: Can't open and lock privilege tables: Table 'mysql.host' doesn't exist
JS获取浏览器窗口大小获取屏幕，浏览器，网页高度宽度
 js获取select标签选中的值
 linux下使用ffmpeg将amr转成mp3
String与InputStream互转的几种方法
 javascript，检测对象中是否存在某个属性
 SQL语句在查询分析器中可以执行，代码中不能执行
 shell实现SSH自动登陆

原文地址：https://www.cnblogs.com/xiaoCon/p/2942902.html