fix: error when no information box
This commit is contained in:
+59
-14
@@ -1,3 +1,7 @@
|
||||
import re
|
||||
from urllib.parse import unquote
|
||||
from typing import List, Union
|
||||
|
||||
import requests
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
@@ -15,18 +19,48 @@ class BaiduSpider:
|
||||
return self.get_web_content(self.baidu_item_base_url + key, is_from_file)
|
||||
|
||||
def get_web_content(self, url, is_from_file=False) -> dict:
|
||||
body_text = self.get_web_body_text(url, is_from_file)
|
||||
if body_text is None:
|
||||
return {}
|
||||
return self.process_body(body_text)
|
||||
|
||||
def get_extra_links(self, url: str, is_from_file=False) -> List[str]:
|
||||
extra_links = []
|
||||
|
||||
body_text = self.get_web_body_text(url, is_from_file)
|
||||
soup = BeautifulSoup(body_text, features="lxml")
|
||||
left_form = soup.find("dl", {"class": "basicInfo-block basicInfo-left"})
|
||||
right_form = soup.find("dl", {"class": "basicInfo-block basicInfo-right"})
|
||||
if left_form is not None:
|
||||
left_keys = left_form.findChildren("dt")
|
||||
left_values = left_form.findChildren("dd")
|
||||
for i in range(len(left_keys)):
|
||||
hyper_link_tag = left_values[i].find("a")
|
||||
if hyper_link_tag is not None:
|
||||
link = hyper_link_tag["href"]
|
||||
extra_links.append(unquote(self.baidu_base_url[:-1] + link))
|
||||
|
||||
if right_form is not None:
|
||||
right_keys = right_form.findChildren("dt")
|
||||
right_values = right_form.findChildren("dd")
|
||||
for i in range(len(right_keys)):
|
||||
hyper_link_tag = right_values[i].find("a")
|
||||
if hyper_link_tag is not None:
|
||||
link = hyper_link_tag["href"]
|
||||
extra_links.append(unquote(self.baidu_base_url[:-1] + link))
|
||||
return extra_links
|
||||
|
||||
def get_web_body_text(self, url, is_from_file=False) -> Union[str, None]:
|
||||
if not is_from_file:
|
||||
r = requests.get(url, headers=self.headers, timeout=5)
|
||||
if not r.status_code == 200:
|
||||
return {}
|
||||
# print(r.text)
|
||||
return self.process_body(r.text)
|
||||
return None
|
||||
return r.text
|
||||
else:
|
||||
text = ""
|
||||
with open(url, 'r', encoding='utf-8') as f:
|
||||
text = f.read()
|
||||
return self.process_body(text)
|
||||
return text
|
||||
|
||||
def process_body(self, body: str) -> dict:
|
||||
r = {}
|
||||
@@ -38,7 +72,10 @@ class BaiduSpider:
|
||||
if info == {}:
|
||||
return None
|
||||
img_urls = self.get_image(soup)
|
||||
summary = self.get_summary(soup)
|
||||
info["summary"] = summary
|
||||
info["imgs"] = img_urls
|
||||
|
||||
r[title] = info
|
||||
return r
|
||||
|
||||
@@ -57,25 +94,33 @@ class BaiduSpider:
|
||||
info_dict = {}
|
||||
left_form = soup.find("dl", {"class": "basicInfo-block basicInfo-left"})
|
||||
right_form = soup.find("dl", {"class": "basicInfo-block basicInfo-right"})
|
||||
left_keys = left_form.findChildren("dt")
|
||||
left_values = left_form.findChildren("dd")
|
||||
right_keys = right_form.findChildren("dt")
|
||||
right_values = right_form.findChildren("dd")
|
||||
if left_form is not None:
|
||||
left_keys = left_form.findChildren("dt")
|
||||
left_values = left_form.findChildren("dd")
|
||||
for i in range(len(left_keys)):
|
||||
info_dict[left_keys[i].text] = self.strip_info_value(left_values[i].text)
|
||||
if right_form is not None:
|
||||
right_keys = right_form.findChildren("dt")
|
||||
right_values = right_form.findChildren("dd")
|
||||
for i in range(len(right_keys)):
|
||||
info_dict[right_keys[i].text] = self.strip_info_value(right_values[i].text)
|
||||
|
||||
for i in range(len(left_keys)):
|
||||
info_dict[left_keys[i].text] = self.strip_info_value(left_values[i].text)
|
||||
for i in range(len(right_keys)):
|
||||
info_dict[right_keys[i].text] = self.strip_info_value(right_values[i].text)
|
||||
return info_dict
|
||||
|
||||
def get_image(self, soup: BeautifulSoup) -> list[str]:
|
||||
def get_summary(self, soup: BeautifulSoup):
|
||||
s = soup.find("div", {"class": "lemma-summary"}).text
|
||||
s = re.sub(r'\[(\d)*\]', "", s.replace("\n", ""))
|
||||
return s
|
||||
|
||||
def get_image(self, soup: BeautifulSoup) -> List[str]:
|
||||
image_urls = []
|
||||
image_tags = soup.findAll("div", {"class": "lemma-picture"})
|
||||
for tag in image_tags:
|
||||
image_href = tag.find("a", {"class": "image-link"})["href"]
|
||||
r = requests.get(self.baidu_base_url + image_href[1:], headers=self.headers)
|
||||
s = BeautifulSoup(r.text, features="lxml")
|
||||
print(s.find("img", {"id": "imgPicture"})["src"])
|
||||
if s.find("img", {"id": "imgPicture"}) is None:
|
||||
continue
|
||||
image_urls.append(s.find("img", {"id": "imgPicture"})["src"])
|
||||
return image_urls
|
||||
|
||||
|
||||
Reference in New Issue
Block a user