增加了一些注释

This commit is contained in:
2021-06-16 11:03:20 +08:00
parent 87dac42d6d
commit 32de547b4d
2 changed files with 43 additions and 2 deletions
+6 -2
View File
@@ -1,7 +1,6 @@
import concurrent.futures import concurrent.futures
import configparser import configparser
import json import json
import datetime
import logging import logging
import traceback import traceback
@@ -45,6 +44,7 @@ military_list_of_lists_url_list_jp = [
] ]
# wrapper函数是为了在concurrent中多线程调用
def get_web_content_wrapper(wiki_spider: WikiSpider, url: str, loggers: logging.Logger, is_from_file=False): def get_web_content_wrapper(wiki_spider: WikiSpider, url: str, loggers: logging.Logger, is_from_file=False):
try: try:
return wiki_spider.get_web_content(url, is_from_file=is_from_file) return wiki_spider.get_web_content(url, is_from_file=is_from_file)
@@ -177,6 +177,10 @@ def get_web_list(configure, list_of_list_url_list, language, output_path):
def test_baidu(config): def test_baidu(config):
"""
一个单纯用来测试的函数
:param config:
"""
s = BaiduSpider(config) s = BaiduSpider(config)
r = s.get_web_content("https://baike.baidu.com/item/%E6%AD%BC-20") r = s.get_web_content("https://baike.baidu.com/item/%E6%AD%BC-20")
j = json.dumps(r, ensure_ascii=False) j = json.dumps(r, ensure_ascii=False)
@@ -187,7 +191,7 @@ def test_baidu(config):
if __name__ == '__main__': if __name__ == '__main__':
# url_list = [] # url_list = []
# 一些配置信息写在了config.ini中,主要是爬Wikipedia时用到的proxy信息
config = configparser.ConfigParser() config = configparser.ConfigParser()
config.read("config.ini") config.read("config.ini")
# get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt") # get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt")
+37
View File
@@ -16,15 +16,31 @@ class BaiduSpider:
return return
def get_web_content_by_keyword(self, key, is_from_file=False) -> dict: def get_web_content_by_keyword(self, key, is_from_file=False) -> dict:
"""
根据词条名称,返回属性信息
:param key: 词条名称
:param is_from_file: 网页是否从现有文件中提取
:return:包含属性信息的dict
"""
return self.get_web_content(self.baidu_item_base_url + key, is_from_file) return self.get_web_content(self.baidu_item_base_url + key, is_from_file)
def get_web_content(self, url, is_from_file=False) -> dict: def get_web_content(self, url, is_from_file=False) -> dict:
"""
爬取url对应百科页面的属性信息
:param url: url地址
:param is_from_file: 是否从本地网页文件中爬取
:return: 属性dict
"""
body_text = self.get_web_body_text(url, is_from_file) body_text = self.get_web_body_text(url, is_from_file)
if body_text is None: if body_text is None:
return {} return {}
return self.process_body(body_text) return self.process_body(body_text)
def get_extra_links(self, url: str, is_from_file=False) -> List[str]: def get_extra_links(self, url: str, is_from_file=False) -> List[str]:
"""
从url对应百科页面的属性表格中获取超链接
:rtype: 包含链接文本的list
"""
extra_links = [] extra_links = []
body_text = self.get_web_body_text(url, is_from_file) body_text = self.get_web_body_text(url, is_from_file)
@@ -51,6 +67,12 @@ class BaiduSpider:
return extra_links return extra_links
def get_web_body_text(self, url, is_from_file=False) -> Union[str, None]: def get_web_body_text(self, url, is_from_file=False) -> Union[str, None]:
"""
获取指定网页的html文本
:param url: url
:param is_from_file: 是否为本地网页文件,为True时 url为文件地址
:return:
"""
if not is_from_file: if not is_from_file:
r = requests.get(url, headers=self.headers, timeout=5) r = requests.get(url, headers=self.headers, timeout=5)
if not r.status_code == 200: if not r.status_code == 200:
@@ -80,6 +102,10 @@ class BaiduSpider:
return r return r
def get_title(self, soup: BeautifulSoup): def get_title(self, soup: BeautifulSoup):
"""
获取网页的title,即百科的词条名称
:rtype: object
"""
head = soup.find("dd", {"class": "lemmaWgt-lemmaTitle-title"}).findChild('h1') head = soup.find("dd", {"class": "lemmaWgt-lemmaTitle-title"}).findChild('h1')
if head is None: if head is None:
@@ -88,6 +114,7 @@ class BaiduSpider:
def get_info(self, soup: BeautifulSoup) -> dict: def get_info(self, soup: BeautifulSoup) -> dict:
""" """
提取百科页面属性表格的信息,以dict方式返回
:param body: html content :param body: html content
:return: dictionary contains property retrieved from wikipedia infobox :return: dictionary contains property retrieved from wikipedia infobox
""" """
@@ -108,11 +135,21 @@ class BaiduSpider:
return info_dict return info_dict
def get_summary(self, soup: BeautifulSoup): def get_summary(self, soup: BeautifulSoup):
"""
获取百科词条页面的描述文本
:param soup:
:return:
"""
s = soup.find("div", {"class": "lemma-summary"}).text s = soup.find("div", {"class": "lemma-summary"}).text
s = re.sub(r'\[(\d)*\]', "", s.replace("\n", "")) s = re.sub(r'\[(\d)*\]', "", s.replace("\n", ""))
return s return s
def get_image(self, soup: BeautifulSoup) -> List[str]: def get_image(self, soup: BeautifulSoup) -> List[str]:
"""
获取百科页面的所有图片链接
:param soup:
:return:
"""
image_urls = [] image_urls = []
image_tags = soup.findAll("div", {"class": "lemma-picture"}) image_tags = soup.findAll("div", {"class": "lemma-picture"})
for tag in image_tags: for tag in image_tags: