feat: add spider for baidubaike
This commit is contained in:
Generated
+1
-1
@@ -2,7 +2,7 @@
|
||||
<module type="PYTHON_MODULE" version="4">
|
||||
<component name="NewModuleRootManager">
|
||||
<content url="file://$MODULE_DIR$" />
|
||||
<orderEntry type="inheritedJdk" />
|
||||
<orderEntry type="jdk" jdkName="Python 3.9 (wiki_spider)" jdkType="Python SDK" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
</component>
|
||||
</module>
|
||||
+1
-1
@@ -1,2 +1,2 @@
|
||||
[PROXY]
|
||||
url = http://127.0.0.1:10809
|
||||
url = socks5://127.0.0.1:1089
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
import concurrent.futures
|
||||
import configparser
|
||||
import json
|
||||
import datetime
|
||||
import logging
|
||||
|
||||
from spider import WikiSpider
|
||||
from spider.baidu_spider import BaiduSpider
|
||||
from spider.wikipedia_spider import WikiSpider
|
||||
|
||||
Military_list_of_lists_url_list = [
|
||||
military_list_of_lists_url_list_en = [
|
||||
"https://en.wikipedia.org/wiki/Lists_of_accidents_and_incidents_involving_military_aircraft",
|
||||
"https://en.wikipedia.org/wiki/Lists_of_armoured_fighting_vehicles",
|
||||
"https://en.wikipedia.org/wiki/List_of_artillery", "https://en.wikipedia.org/wiki/Lists_of_gun_cartridges",
|
||||
@@ -17,27 +21,117 @@ Military_list_of_lists_url_list = [
|
||||
"https://en.wikipedia.org/wiki/Lists_of_World_War_II_military_equipment",
|
||||
]
|
||||
|
||||
military_list_of_lists_url_list_jp = [
|
||||
"https://ja.wikipedia.org/wiki/%E8%BB%8D%E4%BA%8B%E5%AD%A6%E8%80%85",
|
||||
"https://ja.wikipedia.org/wiki/%E8%BB%8D%E4%BA%8B%E7%95%A5%E8%AA%9E%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E8%BB%8D%E9%9A%8A%E3%81%AE%E9%9A%8E%E7%B4%9A",
|
||||
"https://ja.wikipedia.org/wiki/%E5%90%84%E5%9B%BD%E3%81%AE%E8%BB%8D%E9%9A%8A%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E9%99%B8%E8%BB%8D%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E6%B5%B7%E8%BB%8D%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E7%A9%BA%E8%BB%8D%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E6%AD%B4%E5%8F%B2%E4%B8%8A%E3%81%AE%E8%BB%8D%E9%9A%8A%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E7%A9%BA%E8%BB%8D%E5%9F%BA%E5%9C%B0%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E3%82%A2%E3%83%A1%E3%83%AA%E3%82%AB%E7%A9%BA%E8%BB%8D%E5%9F%BA%E5%9C%B0%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E7%89%B9%E6%AE%8A%E9%83%A8%E9%9A%8A%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E9%99%B8%E4%B8%8A%E8%87%AA%E8%A1%9B%E9%9A%8A%E3%81%AE%E9%A7%90%E5%B1%AF%E5%9C%B0%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E6%B5%B7%E4%B8%8A%E8%87%AA%E8%A1%9B%E9%9A%8A%E3%81%AE%E9%99%B8%E4%B8%8A%E6%96%BD%E8%A8%AD%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E8%88%AA%E7%A9%BA%E8%87%AA%E8%A1%9B%E9%9A%8A%E3%81%AE%E5%9F%BA%E5%9C%B0%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E8%BB%8D%E9%9A%8A%E3%82%92%E4%BF%9D%E6%9C%89%E3%81%97%E3%81%A6%E3%81%84%E3%81%AA%E3%81%84%E5%9B%BD%E5%AE%B6%E3%81%AE%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E9%98%B2%E8%A1%9B%E4%B8%8D%E7%A5%A5%E4%BA%8B#%E4%B8%BB%E3%81%AA%E4%B8%8D%E7%A5%A5%E4%BA%8B",
|
||||
"https://ja.wikipedia.org/wiki/%E8%BB%8D%E5%AD%A6%E8%80%85",
|
||||
"https://ja.wikipedia.org/wiki/%E6%88%A6%E4%BA%89%E4%B8%80%E8%A6%A7",
|
||||
"https://ja.wikipedia.org/wiki/%E6%88%A6%E9%97%98%E4%B8%80%E8%A6%A7",
|
||||
]
|
||||
|
||||
|
||||
def get_web_content_wrapper(wiki_spider: WikiSpider, url: str, loggers: logging.Logger, is_from_file=False):
|
||||
try:
|
||||
return wiki_spider.get_web_content(url, is_from_file=is_from_file)
|
||||
except Exception as e:
|
||||
loggers.warning("failed to get content: {}".format(url))
|
||||
return None
|
||||
|
||||
|
||||
def get_web_content_json(configure, language, url_list_file, output_file, is_from_file=False):
|
||||
logging.basicConfig(filename='spider.log', format="%(asctime)s %(filename)s : %(levelname)s %(message)s",
|
||||
datefmt='%Y-%m-%d: %H:%M:%S',
|
||||
level=logging.DEBUG)
|
||||
logger = logging.getLogger(__name__)
|
||||
logger.setLevel(logging.DEBUG)
|
||||
s = WikiSpider(configure, language)
|
||||
|
||||
with open(url_list_file, "r", encoding='utf-8') as f:
|
||||
for line in f:
|
||||
url_list.append(line[:-1])
|
||||
|
||||
result_list = []
|
||||
logger.info("spider start")
|
||||
|
||||
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||
futures = [executor.submit(get_web_content_wrapper, s, url, logger, is_from_file) for url in url_list]
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
r = future.result()
|
||||
if r is not None:
|
||||
result_list.append(r)
|
||||
logger.info("spider finished")
|
||||
logger.info("writing result to file: {}".format(output_file))
|
||||
with open(output_file, 'w', encoding='utf-8') as f:
|
||||
for result in result_list:
|
||||
j = json.dumps(result, ensure_ascii=False)
|
||||
f.write(j + '\n')
|
||||
logger.info("writing finished")
|
||||
logger.info("{} line writen".format(len(result_list)))
|
||||
|
||||
|
||||
def get_web_list(configure, list_of_list_url_list, language, output_path):
|
||||
logging.basicConfig(filename='spider.log', format="%(asctime)s %(filename)s : %(levelname)s %(message)s",
|
||||
datefmt='%Y-%m-%d: %H:%M:%S',
|
||||
level=logging.DEBUG)
|
||||
logger = logging.getLogger(__name__)
|
||||
logger.setLevel(logging.DEBUG)
|
||||
|
||||
s = WikiSpider(configure, language)
|
||||
|
||||
# for idx, url in enumerate(url_list):
|
||||
# s.get_web_content(url)
|
||||
# if idx % 10 == 0:
|
||||
# logger.debug("{} urls processed".format(idx))
|
||||
# # print(idx)
|
||||
|
||||
link_set = set()
|
||||
try:
|
||||
for list_url in list_of_list_url_list:
|
||||
list_list = s.get_lists(list_url)
|
||||
for list_page in list_list:
|
||||
new_url = s.get_links_from_list(list_page)
|
||||
|
||||
link_set |= new_url
|
||||
except Exception as e:
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
for u in link_set:
|
||||
f.write(u + "\n")
|
||||
print(e)
|
||||
else:
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
for u in link_set:
|
||||
f.write(u + "\n")
|
||||
print(len(link_set))
|
||||
|
||||
|
||||
def test_baidu(config):
|
||||
s = BaiduSpider(config)
|
||||
r = s.get_web_content("https://baike.baidu.com/item/%E6%AD%BC-20")
|
||||
j = json.dumps(r, ensure_ascii=False)
|
||||
with open("test.txt", "w", encoding="utf-8") as f:
|
||||
f.write(j + "\n")
|
||||
# print(s.get_web_content("https://baike.baidu.com/item/%E6%AD%BC-20"))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
url_list = []
|
||||
|
||||
config = configparser.ConfigParser()
|
||||
config.read("config.ini")
|
||||
s = WikiSpider(config)
|
||||
|
||||
print(s.get_web_content("https://en.wikipedia.org/wiki/Battle_of_Tskhinvali"))
|
||||
# link_set = set()
|
||||
# try:
|
||||
# for list_url in Military_list_of_lists_url_list:
|
||||
# list_list = s.get_lists(list_url)
|
||||
# for list_page in list_list:
|
||||
# new_url = s.get_links_from_list(list_page)
|
||||
#
|
||||
# link_set |= new_url
|
||||
# except Exception as e:
|
||||
# with open("urls.txt", "w", encoding="utf-8") as f:
|
||||
# for u in link_set:
|
||||
# f.write(u + "\n")
|
||||
# print(e)
|
||||
# else:
|
||||
# with open("urls.txt", "w", encoding="utf-8") as f:
|
||||
# for u in link_set:
|
||||
# f.write(u + "\n")
|
||||
# print(len(link_set))
|
||||
# get_web_list(config, military_list_of_lists_url_list_jp, "ja", "data/ja_wiki_urls.txt")
|
||||
# get_web_content_json(config, 'zh', "data/zh_wiki_urls.txt", "data/zh_data.txt", is_from_file=True)
|
||||
test_baidu(config)
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
|
||||
class BaiduSpider:
|
||||
def __init__(self, config):
|
||||
self.baidu_base_url = "https://baike.baidu.com/item/"
|
||||
return
|
||||
|
||||
def get_web_content_by_keyword(self, key, is_from_file=False) -> dict:
|
||||
return self.get_web_content(self.baidu_base_url + key, is_from_file)
|
||||
|
||||
def get_web_content(self, url, is_from_file=False) -> dict:
|
||||
headers = {
|
||||
"user-agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:88.0) Gecko/20100101 Firefox/88.0"
|
||||
}
|
||||
if not is_from_file:
|
||||
r = requests.get(url, headers=headers, timeout=5)
|
||||
# print(r.text)
|
||||
return self.process_body(r.text)
|
||||
else:
|
||||
text = ""
|
||||
with open(url, 'r', encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
return self.process_body(text)
|
||||
|
||||
def process_body(self, body: str) -> dict:
|
||||
r = {}
|
||||
soup = BeautifulSoup(body, features="lxml")
|
||||
title = self.get_title(soup)
|
||||
if title is None:
|
||||
return None
|
||||
info = self.get_info(soup)
|
||||
if info == {}:
|
||||
return None
|
||||
# img_urls = self.get_image(soup)
|
||||
# info["imgs"] = img_urls
|
||||
r[title] = info
|
||||
return r
|
||||
|
||||
def get_title(self, soup: BeautifulSoup):
|
||||
head = soup.find("dd", {"class": "lemmaWgt-lemmaTitle-title"}).findChild('h1')
|
||||
|
||||
if head is None:
|
||||
return None
|
||||
return head.text
|
||||
|
||||
def get_info(self, soup: BeautifulSoup) -> dict:
|
||||
"""
|
||||
:param body: html content
|
||||
:return: dictionary contains property retrieved from wikipedia infobox
|
||||
"""
|
||||
info_dict = {}
|
||||
left_form = soup.find("dl", {"class": "basicInfo-block basicInfo-left"})
|
||||
right_form = soup.find("dl", {"class": "basicInfo-block basicInfo-right"})
|
||||
left_keys = left_form.findChildren("dt")
|
||||
left_values = left_form.findChildren("dd")
|
||||
right_keys = right_form.findChildren("dt")
|
||||
right_values = right_form.findChildren("dd")
|
||||
|
||||
for i in range(len(left_keys)):
|
||||
info_dict[left_keys[i].text] = self.strip_info_value(left_values[i].text)
|
||||
for i in range(len(right_keys)):
|
||||
info_dict[right_keys[i].text] = self.strip_info_value(right_values[i].text)
|
||||
return info_dict
|
||||
|
||||
def get_proxy(self) -> dict:
|
||||
return {
|
||||
'http': self.proxy_config["url"],
|
||||
'https': self.proxy_config["url"],
|
||||
}
|
||||
|
||||
def strip_info_key(self, key: str):
|
||||
return key.strip("\n")
|
||||
|
||||
def strip_info_value(self, value: str):
|
||||
return value.strip("\n")
|
||||
@@ -4,27 +4,77 @@ import requests
|
||||
from bs4 import BeautifulSoup
|
||||
from pandas.io.html import read_html
|
||||
|
||||
_wiki_base_url_dict = {
|
||||
"en": "https://en.wikipedia.org",
|
||||
"ja": "https://ja.wikipedia.org",
|
||||
"ko": "https://ko.wikipedia.org",
|
||||
"zh": "https://zh.wikipedia.org"
|
||||
}
|
||||
|
||||
_wiki_list_regex_dict = {
|
||||
"en": r"^List of[\s\S]*$",
|
||||
"ja": r"^[\s\S]+一覧$",
|
||||
}
|
||||
|
||||
|
||||
def _is_en_content_page(url: str) -> bool:
|
||||
return not (":" in url or
|
||||
url.startswith("/wiki/List_of") or
|
||||
url.startswith("/wiki/Lists_of")
|
||||
)
|
||||
|
||||
|
||||
def _is_ja_content_page(url: str) -> bool:
|
||||
return not (":" in url or
|
||||
url.endswith("一覧")
|
||||
)
|
||||
|
||||
|
||||
def _is_ko_content_page(url: str) -> bool:
|
||||
raise NotImplementedError
|
||||
|
||||
|
||||
_is_content_page_func = {
|
||||
"en": _is_en_content_page,
|
||||
"ja": _is_ja_content_page,
|
||||
'ko': _is_ko_content_page,
|
||||
}
|
||||
|
||||
|
||||
class WikiSpider:
|
||||
def __init__(self, config):
|
||||
def __init__(self, config, language):
|
||||
self.proxy_config = config["PROXY"]
|
||||
self.wiki_base_url = "https://en.wikipedia.org"
|
||||
if language not in _wiki_base_url_dict:
|
||||
raise ValueError("language code [{}] is not supported".format(language))
|
||||
|
||||
self.language = language
|
||||
self.wiki_base_url = _wiki_base_url_dict[self.language]
|
||||
self.wiki_url = self.wiki_base_url + "/wiki/"
|
||||
|
||||
def search_by_chinese(self, keyword):
|
||||
url = self.get_wiki_url(keyword)
|
||||
return self.get_web_content(url)
|
||||
|
||||
def get_web_content(self, url) -> dict:
|
||||
r = requests.get(url, proxies=self.get_proxy())
|
||||
def get_web_content(self, url, is_from_file=False) -> dict:
|
||||
if not is_from_file:
|
||||
r = requests.get(url, proxies=self.get_proxy())
|
||||
|
||||
return self.process_body(r.text)
|
||||
return self.process_body(r.text)
|
||||
else:
|
||||
text = ""
|
||||
with open(url, 'r', encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
return self.process_body(text)
|
||||
|
||||
def process_body(self, body: str) -> dict:
|
||||
r = {}
|
||||
soup = BeautifulSoup(body, features="lxml")
|
||||
title = self.get_title(soup)
|
||||
if title is None:
|
||||
return None
|
||||
info = self.get_info(body)
|
||||
if info == {}:
|
||||
return None
|
||||
img_urls = self.get_image(soup)
|
||||
info["imgs"] = img_urls
|
||||
r[title] = info
|
||||
@@ -82,7 +132,7 @@ class WikiSpider:
|
||||
r = requests.get(lists_of_lists_url, proxies=self.get_proxy())
|
||||
body = r.text
|
||||
soup = BeautifulSoup(body, features="lxml")
|
||||
links = soup.findAll("a", {"title": re.compile(r"^List of[\s\S]*$")})
|
||||
links = soup.findAll("a", {"title": re.compile(_wiki_list_regex_dict[self.language])})
|
||||
list_link_set = {self.wiki_base_url + link["href"] for link in links}
|
||||
return list_link_set
|
||||
|
||||
@@ -108,18 +158,16 @@ class WikiSpider:
|
||||
return self.wiki_url + keyword
|
||||
|
||||
def is_content_page(self, url: str) -> bool:
|
||||
return not (":" in url or
|
||||
url.startswith("/wiki/List_of") or
|
||||
url.startswith("/wiki/Lists_of"))
|
||||
return _is_content_page_func[self.language](url)
|
||||
|
||||
def strip_info_key(self, info: str) -> str:
|
||||
r = re.sub(r'\[(\d)*\]', "", info.replace("\xa0", "")
|
||||
.replace("\ufeff", "")
|
||||
)
|
||||
return r
|
||||
# r = re.sub(r'\[(\d)*\]', "", info.replace("\xa0", "")
|
||||
# .replace("\ufeff", "")
|
||||
# )
|
||||
return info
|
||||
|
||||
def strip_info_value(self, info: str) -> str:
|
||||
r = re.sub(r'\[(\d)*\]', "", info.replace("\xa0", "").replace("\ufeff", "")
|
||||
r = re.sub(r'\[(\d)*\]', "", info
|
||||
.replace(".mw-parser-output", "")
|
||||
.replace(".geo-default", "")
|
||||
.replace(".geo-dms", "")
|
||||
@@ -133,7 +181,7 @@ class WikiSpider:
|
||||
.replace("{white-space:nowrap}", "")
|
||||
)
|
||||
|
||||
return r.strip().lstrip(",")
|
||||
return r.replace(" ,", "").strip().lstrip(",")
|
||||
|
||||
def get_proxy(self) -> dict:
|
||||
return {
|
||||
@@ -0,0 +1,14 @@
|
||||
url_file = "data/en_wiki_urls.txt"
|
||||
|
||||
if __name__ == '__main__':
|
||||
line_list = []
|
||||
with open(url_file, "r", encoding="utf-8") as f:
|
||||
line_list = f.readlines()
|
||||
for i, line in enumerate(line_list):
|
||||
idx = line.find("#")
|
||||
if idx != -1:
|
||||
line_list[i] = line[:idx] + "\n"
|
||||
result = list(set(line_list))
|
||||
with open("t.txt", "w", encoding="utf-8") as f:
|
||||
for r in result:
|
||||
f.write(r)
|
||||
Reference in New Issue
Block a user