backup commit
This commit is contained in:
+124
-3
@@ -1,14 +1,18 @@
|
||||
import concurrent
|
||||
import json
|
||||
import re
|
||||
from typing import List, Set, Union
|
||||
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
from pandas.io.html import read_html
|
||||
from tqdm import tqdm
|
||||
|
||||
_wiki_base_url_dict = {
|
||||
"en": "https://en.wikipedia.org",
|
||||
"ja": "https://ja.wikipedia.org",
|
||||
"ko": "https://ko.wikipedia.org",
|
||||
"ru": "https://ru.wikipedia.org",
|
||||
"zh": "https://zh.wikipedia.org"
|
||||
}
|
||||
|
||||
@@ -73,11 +77,13 @@ class WikiSpider:
|
||||
title = self.get_title(soup)
|
||||
if title is None:
|
||||
return None
|
||||
info = self.get_info(body)
|
||||
# info = self.get_info(body)
|
||||
info = self.get_info_plus(soup)
|
||||
if info == {}:
|
||||
return None
|
||||
img_urls = self.get_image(soup)
|
||||
info["imgs"] = img_urls
|
||||
info['paragraph_text'] = self.get_para_text(soup)
|
||||
# img_urls = self.get_image(soup)
|
||||
# info["imgs"] = img_urls
|
||||
r[title] = info
|
||||
return r
|
||||
|
||||
@@ -110,6 +116,29 @@ class WikiSpider:
|
||||
k != v and str(k).lower() != "nan" and str(v).lower() != "nan"}
|
||||
return info_dict
|
||||
|
||||
def get_info_plus(self, soup: BeautifulSoup):
|
||||
|
||||
infodict = {}
|
||||
infobox = soup.find("table", {"class": "infobox"})
|
||||
rows = infobox.findChildren("tr")
|
||||
for row in rows:
|
||||
if len(row.contents) != 2:
|
||||
continue
|
||||
# key = traverse_content('', row.contents[0])
|
||||
# value = traverse_content('', row.contents[1])
|
||||
key = row.contents[0].text
|
||||
value = row.contents[1].text
|
||||
infodict[key] = value
|
||||
return infodict
|
||||
|
||||
def get_para_text(self, soup: BeautifulSoup):
|
||||
result_list = []
|
||||
body = soup.find("div", {"class": "mw-parser-output"})
|
||||
paragraphs = body.findChildren('p')
|
||||
for p in paragraphs:
|
||||
result_list.append(p.text)
|
||||
return result_list
|
||||
|
||||
def get_image(self, soup: BeautifulSoup) -> List[str]:
|
||||
img_urls = []
|
||||
# soup = BeautifulSoup(body, features="lxml")
|
||||
@@ -155,6 +184,98 @@ class WikiSpider:
|
||||
link_set = {self.wiki_base_url + link["href"] for link in links if self.is_content_page(link["href"])}
|
||||
return link_set
|
||||
|
||||
def align_language_wrapper(self, url, lang_src, lang_tgt):
|
||||
r = requests.get(url, proxies=self.get_proxy())
|
||||
body = r.text
|
||||
soup = BeautifulSoup(body, features="lxml")
|
||||
lang_nav = soup.find("nav", {"id": "p-lang"})
|
||||
link_tag = lang_nav.find("a", string=lang_tgt)
|
||||
if link_tag is None:
|
||||
return None
|
||||
tgt_link = link_tag["href"]
|
||||
return {lang_src: url, lang_tgt: tgt_link}
|
||||
|
||||
def align_language(self, urls_file, lang_src, lang_tgt: str):
|
||||
with open(urls_file, 'r', encoding='utf-8') as f:
|
||||
urls = [line[:-1] for line in f]
|
||||
try:
|
||||
result_urls = []
|
||||
with tqdm(total=len(urls)) as pbar:
|
||||
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||
futures = [executor.submit(self.align_language_wrapper, url, lang_src, lang_tgt) for url in urls]
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
r = future.result()
|
||||
pbar.update(1)
|
||||
if r is not None:
|
||||
result_urls.append(r)
|
||||
except:
|
||||
file_name = f'data/{lang_src}-{lang_tgt}-align-urls.txt'
|
||||
with open(file_name, 'w', encoding='utf-8') as f:
|
||||
f.write(json.dumps(result_urls, ensure_ascii=False))
|
||||
else:
|
||||
file_name = f'data/{lang_src}-{lang_tgt}-align-urls.txt'
|
||||
with open(file_name, 'w', encoding='utf-8') as f:
|
||||
f.write(json.dumps(result_urls, ensure_ascii=False))
|
||||
|
||||
def align_chinese_wrapper(self, url):
|
||||
lang_ko = '한국어'
|
||||
lang_ru = 'Русский'
|
||||
r = requests.get(url, proxies=self.get_proxy())
|
||||
body = r.text
|
||||
soup = BeautifulSoup(body, features="lxml")
|
||||
lang_nav = soup.find("nav", {"id": "p-lang"})
|
||||
link_tag_ko = lang_nav.find("a", string=lang_ko)
|
||||
link_tag_ru = lang_nav.find("a", string=lang_ru)
|
||||
if link_tag_ko is None and link_tag_ru is None:
|
||||
return None
|
||||
|
||||
if link_tag_ko is not None:
|
||||
link_ko = link_tag_ko['href']
|
||||
res_ko = {'chinese': url, 'ko': link_ko}
|
||||
else:
|
||||
res_ko = None
|
||||
if link_tag_ru is not None:
|
||||
link_ru = link_tag_ru['href']
|
||||
res_ru = {'chinese': url, 'ru': link_ru}
|
||||
else:
|
||||
res_ru = None
|
||||
|
||||
return res_ko, res_ru
|
||||
|
||||
def align_chinese(self, urls_file: str):
|
||||
with open(urls_file, 'r', encoding='utf-8') as f:
|
||||
urls = ['https://zh.wikipedia.org/wiki/' + line[:-1] for line in f]
|
||||
|
||||
# urls = urls[:50]
|
||||
|
||||
try:
|
||||
result_urls = []
|
||||
with tqdm(total=len(urls)) as pbar:
|
||||
with concurrent.futures.ThreadPoolExecutor() as executor:
|
||||
futures = [executor.submit(self.align_chinese_wrapper, url) for url in urls]
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
r = future.result()
|
||||
pbar.update(1)
|
||||
result_urls.append(r)
|
||||
except:
|
||||
ko_list = [u[0] for u in result_urls if u is not None and u[0] is not None]
|
||||
ru_list = [u[1] for u in result_urls if u is not None and u[1] is not None]
|
||||
file_name_ko = 'data/chinese-ko-align-urls.txt'
|
||||
file_name_ru = 'data/chinese-ru-align-urls.txt'
|
||||
with open(file_name_ko, 'w', encoding='utf-8') as f:
|
||||
f.write(json.dumps(ko_list, ensure_ascii=False))
|
||||
with open(file_name_ru, 'w', encoding='utf-8') as f:
|
||||
f.write(json.dumps(ru_list, ensure_ascii=False))
|
||||
else:
|
||||
ko_list = [u[0] for u in result_urls if u is not None and u[0] is not None]
|
||||
ru_list = [u[1] for u in result_urls if u is not None and u[1] is not None]
|
||||
file_name_ko = 'data/chinese-ko-align-urls.txt'
|
||||
file_name_ru = 'data/chinese-ru-align-urls.txt'
|
||||
with open(file_name_ko, 'w', encoding='utf-8') as f:
|
||||
f.write(json.dumps(ko_list, ensure_ascii=False))
|
||||
with open(file_name_ru, 'w', encoding='utf-8') as f:
|
||||
f.write(json.dumps(ru_list, ensure_ascii=False))
|
||||
|
||||
def get_wiki_url(self, keyword: str) -> str:
|
||||
return self.wiki_url + keyword
|
||||
|
||||
|
||||
Reference in New Issue
Block a user