zoukankan      html  css  js  c++  java
  • easyspider

    # -*- coding: utf-8 -*-
    """
    Created on Fri Aug 18 15:58:13 2017
    @author: JClian
    """
    import re
    import bs4
    import urllib.request  
    from bs4 import BeautifulSoup 
    import urllib.parse
    import sys
    
    search_item = input("Enter what you want(Enter 'out' to exit):")
    while search_item != 'out':
        if search_item == 'out':
            exit(0)
        print("please wait...")
        try:
            url = 'https://baike.baidu.com/item/'+urllib.parse.quote(search_item)
            html = urllib.request.urlopen(url)  
            content = html.read().decode('utf-8')
            html.close()
            soup = BeautifulSoup(content, "lxml")  
            text = soup.find('div', class_="lemma-summary").children
            print("search result:")
            for x in text:
                word = re.sub(re.compile(r"<(.+?)>"),'',str(x))
                words = re.sub(re.compile(r"[(.+?)]"),'',word)
                print(words,'
    ')
        except AttributeError:
            print("Failed!Please enter more in details!")
        search_item = input("Enter what you want(Enter 'out' to exit):")
    
  • 相关阅读:
    Linux系统服务
    Linux进程管理
    Linux压缩打包
    Linux输入输出
    Linux权限管理
    Linux用户管理
    Linux文件管理
    Linux-Shell
    Centos7 安装jdk1.8
    Python数据分析之路
  • 原文地址:https://www.cnblogs.com/sky-ai/p/9813126.html
Copyright © 2011-2022 走看看