feat: 增加实体名称查询功能
This commit is contained in:
@@ -195,10 +195,12 @@ if __name__ == '__main__':
|
|||||||
config = configparser.ConfigParser()
|
config = configparser.ConfigParser()
|
||||||
config.read("config.ini")
|
config.read("config.ini")
|
||||||
# get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt")
|
# get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt")
|
||||||
get_web_content_json(config, 'zh', "data/baidu_baike_urls_extra.txt",
|
# get_web_content_json(config, 'zh', "data/baidu_baike_urls_extra.txt",
|
||||||
"data/baidu_baike_data_with_summary_extra.txt",
|
# "data/baidu_baike_data_with_summary_extra.txt",
|
||||||
is_from_file=False)
|
# is_from_file=False)
|
||||||
# s = BaiduSpider(config)
|
s = BaiduSpider(config)
|
||||||
|
entity_name = s.check_entity_name('mq-9')
|
||||||
|
print(entity_name)
|
||||||
# r = s.get_extra_links("https://baike.baidu.com/item/%E6%AD%BC-20")
|
# r = s.get_extra_links("https://baike.baidu.com/item/%E6%AD%BC-20")
|
||||||
# print(r)
|
# print(r)
|
||||||
# get_extra_links(config, "data/baidu_baike_urls.txt", "data/baidu_baike_urls_extra.txt")
|
# get_extra_links(config, "data/baidu_baike_urls.txt", "data/baidu_baike_urls_extra.txt")
|
||||||
|
|||||||
+29
-1
@@ -15,6 +15,34 @@ class BaiduSpider:
|
|||||||
}
|
}
|
||||||
return
|
return
|
||||||
|
|
||||||
|
def check_entity_name(self, key_word) -> Union[str, dict, None]:
|
||||||
|
"""
|
||||||
|
查询key_word在百科词条中的名称
|
||||||
|
:param key_word: 待查询的keyword
|
||||||
|
:return: 词条不存在则返回None,
|
||||||
|
词条存在多个释义是返回词典,key为词条名,value为str类型的数组
|
||||||
|
词条不存在歧义时返回百科中的名称
|
||||||
|
"""
|
||||||
|
url = self.baidu_item_base_url + key_word
|
||||||
|
r = requests.get(url, headers=self.headers, allow_redirects=False)
|
||||||
|
if r.status_code == 200: # 消歧义页面
|
||||||
|
body = r.text
|
||||||
|
soup = BeautifulSoup(body, features="lxml")
|
||||||
|
title = self.get_title(soup)
|
||||||
|
if title is not None:
|
||||||
|
return title
|
||||||
|
list_items = soup.findAll('li', {'class': 'list-dot list-dot-paddingleft'})
|
||||||
|
title_list = [item.text for item in list_items]
|
||||||
|
return {key_word: title_list}
|
||||||
|
location = r.headers['Location']
|
||||||
|
if 'error.html' in location:
|
||||||
|
return None
|
||||||
|
else:
|
||||||
|
body = self.get_web_body_text(url)
|
||||||
|
soup = BeautifulSoup(body, features="lxml")
|
||||||
|
title = self.get_title(soup)
|
||||||
|
return title
|
||||||
|
|
||||||
def get_web_content_by_keyword(self, key, is_from_file=False) -> dict:
|
def get_web_content_by_keyword(self, key, is_from_file=False) -> dict:
|
||||||
"""
|
"""
|
||||||
根据词条名称,返回属性信息
|
根据词条名称,返回属性信息
|
||||||
@@ -101,7 +129,7 @@ class BaiduSpider:
|
|||||||
r[title] = info
|
r[title] = info
|
||||||
return r
|
return r
|
||||||
|
|
||||||
def get_title(self, soup: BeautifulSoup):
|
def get_title(self, soup: BeautifulSoup) -> Union[str, None]:
|
||||||
"""
|
"""
|
||||||
获取网页的title,即百科的词条名称
|
获取网页的title,即百科的词条名称
|
||||||
:rtype: object
|
:rtype: object
|
||||||
|
|||||||
Reference in New Issue
Block a user