增加了一些注释
This commit is contained in:
@@ -1,7 +1,6 @@
|
|||||||
import concurrent.futures
|
import concurrent.futures
|
||||||
import configparser
|
import configparser
|
||||||
import json
|
import json
|
||||||
import datetime
|
|
||||||
import logging
|
import logging
|
||||||
import traceback
|
import traceback
|
||||||
|
|
||||||
@@ -45,6 +44,7 @@ military_list_of_lists_url_list_jp = [
|
|||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
|
# wrapper函数是为了在concurrent中多线程调用
|
||||||
def get_web_content_wrapper(wiki_spider: WikiSpider, url: str, loggers: logging.Logger, is_from_file=False):
|
def get_web_content_wrapper(wiki_spider: WikiSpider, url: str, loggers: logging.Logger, is_from_file=False):
|
||||||
try:
|
try:
|
||||||
return wiki_spider.get_web_content(url, is_from_file=is_from_file)
|
return wiki_spider.get_web_content(url, is_from_file=is_from_file)
|
||||||
@@ -177,6 +177,10 @@ def get_web_list(configure, list_of_list_url_list, language, output_path):
|
|||||||
|
|
||||||
|
|
||||||
def test_baidu(config):
|
def test_baidu(config):
|
||||||
|
"""
|
||||||
|
一个单纯用来测试的函数
|
||||||
|
:param config:
|
||||||
|
"""
|
||||||
s = BaiduSpider(config)
|
s = BaiduSpider(config)
|
||||||
r = s.get_web_content("https://baike.baidu.com/item/%E6%AD%BC-20")
|
r = s.get_web_content("https://baike.baidu.com/item/%E6%AD%BC-20")
|
||||||
j = json.dumps(r, ensure_ascii=False)
|
j = json.dumps(r, ensure_ascii=False)
|
||||||
@@ -187,7 +191,7 @@ def test_baidu(config):
|
|||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
# url_list = []
|
# url_list = []
|
||||||
|
# 一些配置信息写在了config.ini中,主要是爬Wikipedia时用到的proxy信息
|
||||||
config = configparser.ConfigParser()
|
config = configparser.ConfigParser()
|
||||||
config.read("config.ini")
|
config.read("config.ini")
|
||||||
# get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt")
|
# get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt")
|
||||||
|
|||||||
@@ -16,15 +16,31 @@ class BaiduSpider:
|
|||||||
return
|
return
|
||||||
|
|
||||||
def get_web_content_by_keyword(self, key, is_from_file=False) -> dict:
|
def get_web_content_by_keyword(self, key, is_from_file=False) -> dict:
|
||||||
|
"""
|
||||||
|
根据词条名称,返回属性信息
|
||||||
|
:param key: 词条名称
|
||||||
|
:param is_from_file: 网页是否从现有文件中提取
|
||||||
|
:return:包含属性信息的dict
|
||||||
|
"""
|
||||||
return self.get_web_content(self.baidu_item_base_url + key, is_from_file)
|
return self.get_web_content(self.baidu_item_base_url + key, is_from_file)
|
||||||
|
|
||||||
def get_web_content(self, url, is_from_file=False) -> dict:
|
def get_web_content(self, url, is_from_file=False) -> dict:
|
||||||
|
"""
|
||||||
|
爬取url对应百科页面的属性信息
|
||||||
|
:param url: url地址
|
||||||
|
:param is_from_file: 是否从本地网页文件中爬取
|
||||||
|
:return: 属性dict
|
||||||
|
"""
|
||||||
body_text = self.get_web_body_text(url, is_from_file)
|
body_text = self.get_web_body_text(url, is_from_file)
|
||||||
if body_text is None:
|
if body_text is None:
|
||||||
return {}
|
return {}
|
||||||
return self.process_body(body_text)
|
return self.process_body(body_text)
|
||||||
|
|
||||||
def get_extra_links(self, url: str, is_from_file=False) -> List[str]:
|
def get_extra_links(self, url: str, is_from_file=False) -> List[str]:
|
||||||
|
"""
|
||||||
|
从url对应百科页面的属性表格中获取超链接
|
||||||
|
:rtype: 包含链接文本的list
|
||||||
|
"""
|
||||||
extra_links = []
|
extra_links = []
|
||||||
|
|
||||||
body_text = self.get_web_body_text(url, is_from_file)
|
body_text = self.get_web_body_text(url, is_from_file)
|
||||||
@@ -51,6 +67,12 @@ class BaiduSpider:
|
|||||||
return extra_links
|
return extra_links
|
||||||
|
|
||||||
def get_web_body_text(self, url, is_from_file=False) -> Union[str, None]:
|
def get_web_body_text(self, url, is_from_file=False) -> Union[str, None]:
|
||||||
|
"""
|
||||||
|
获取指定网页的html文本
|
||||||
|
:param url: url
|
||||||
|
:param is_from_file: 是否为本地网页文件,为True时 url为文件地址
|
||||||
|
:return:
|
||||||
|
"""
|
||||||
if not is_from_file:
|
if not is_from_file:
|
||||||
r = requests.get(url, headers=self.headers, timeout=5)
|
r = requests.get(url, headers=self.headers, timeout=5)
|
||||||
if not r.status_code == 200:
|
if not r.status_code == 200:
|
||||||
@@ -80,6 +102,10 @@ class BaiduSpider:
|
|||||||
return r
|
return r
|
||||||
|
|
||||||
def get_title(self, soup: BeautifulSoup):
|
def get_title(self, soup: BeautifulSoup):
|
||||||
|
"""
|
||||||
|
获取网页的title,即百科的词条名称
|
||||||
|
:rtype: object
|
||||||
|
"""
|
||||||
head = soup.find("dd", {"class": "lemmaWgt-lemmaTitle-title"}).findChild('h1')
|
head = soup.find("dd", {"class": "lemmaWgt-lemmaTitle-title"}).findChild('h1')
|
||||||
|
|
||||||
if head is None:
|
if head is None:
|
||||||
@@ -88,6 +114,7 @@ class BaiduSpider:
|
|||||||
|
|
||||||
def get_info(self, soup: BeautifulSoup) -> dict:
|
def get_info(self, soup: BeautifulSoup) -> dict:
|
||||||
"""
|
"""
|
||||||
|
提取百科页面属性表格的信息,以dict方式返回
|
||||||
:param body: html content
|
:param body: html content
|
||||||
:return: dictionary contains property retrieved from wikipedia infobox
|
:return: dictionary contains property retrieved from wikipedia infobox
|
||||||
"""
|
"""
|
||||||
@@ -108,11 +135,21 @@ class BaiduSpider:
|
|||||||
return info_dict
|
return info_dict
|
||||||
|
|
||||||
def get_summary(self, soup: BeautifulSoup):
|
def get_summary(self, soup: BeautifulSoup):
|
||||||
|
"""
|
||||||
|
获取百科词条页面的描述文本
|
||||||
|
:param soup:
|
||||||
|
:return:
|
||||||
|
"""
|
||||||
s = soup.find("div", {"class": "lemma-summary"}).text
|
s = soup.find("div", {"class": "lemma-summary"}).text
|
||||||
s = re.sub(r'\[(\d)*\]', "", s.replace("\n", ""))
|
s = re.sub(r'\[(\d)*\]', "", s.replace("\n", ""))
|
||||||
return s
|
return s
|
||||||
|
|
||||||
def get_image(self, soup: BeautifulSoup) -> List[str]:
|
def get_image(self, soup: BeautifulSoup) -> List[str]:
|
||||||
|
"""
|
||||||
|
获取百科页面的所有图片链接
|
||||||
|
:param soup:
|
||||||
|
:return:
|
||||||
|
"""
|
||||||
image_urls = []
|
image_urls = []
|
||||||
image_tags = soup.findAll("div", {"class": "lemma-picture"})
|
image_tags = soup.findAll("div", {"class": "lemma-picture"})
|
||||||
for tag in image_tags:
|
for tag in image_tags:
|
||||||
|
|||||||
Reference in New Issue
Block a user