zoukankan      html  css  js  c++  java
  • 使用python获取博客园作者的文章列表的超链接以及标题

    # -*- coding: utf-8 -*-
    """
    Created on Thu Jun 12 09:37:48 2014
    
    @author: lifeix
    """
    
    import re
    import urllib2
    import cookielib
    
    url = 'http://www.cnblogs.com/wendingding/tag/IOS%E5%BC%80%E5%8F%91/default.html?page='
    #url = 'http://www.cnblogs.com/smileEvday/category/578973.html?page='
    reg = '<a id="w+" href="http://www.cnblogs.com/w+/p/w+.html">s*	*
    *s*	*s*.*?

    * * *s*</a>' def startParse(author,page=1): cj = cookielib.LWPCookieJar() cookie_support = urllib2.HTTPCookieProcessor(cj) opener = urllib2.build_opener(cookie_support,urllib2.HTTPHandler) urllib2.install_opener(opener) headers = {'User-Agent' : 'Mozilla/5.0 (Windows NT 6.1; WOW64; rv:14.0) Gecko/20100101 Firefox/14.0.1', 'Referer' : "http://www.cnblogs.com"} flag = True while flag == True: nurl = url + str(page) req = urllib2.Request(nurl,headers=headers) resp = urllib2.urlopen(req) data = resp.read() regex = re.compile(reg,flags=re.MULTILINE) result = regex.findall(data) for d in result: print d if len(result) < 20: flag = False else: page = page + 1 print 'finished----------------------page:%d'%page if __name__ == '__main__': startParse('',1)



    
       
    
  • 相关阅读:
    单点登录原理与简单实现
    关系型数据库中的关键字、主关键字和候选关键字
    无向图的顶点连通度
    memcmp()直接比较两个数组的大小
    静态字典树
    动态字典树
    poj 1149
    poj 2112 floyd+Dinic最大流+二分最小值
    POJ 1698 (二分图的多重匹配)
    网络流算法
  • 原文地址:https://www.cnblogs.com/jzssuanfa/p/6715723.html
Copyright © 2011-2022 走看看