feat: get property and image links

This commit is contained in:
2021-04-15 23:50:25 +08:00
commit 5c3d92a353
11 changed files with 339 additions and 0 deletions
+213
View File
@@ -0,0 +1,213 @@
result.json
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
.pytest_cache/
cover/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
.pybuilder/
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
# For a library or package, you might want to ignore these files since the code is
# intended to run in multiple environments; otherwise, check them in:
# .python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
#Pipfile.lock
# PEP 582; used by e.g. github.com/David-OConnor/pyflow
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# pytype static type analyzer
.pytype/
# Cython debug symbols
cython_debug/
# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider
# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839
# User-specific stuff
.idea/**/workspace.xml
.idea/**/tasks.xml
.idea/**/usage.statistics.xml
.idea/**/dictionaries
.idea/**/shelf
# Generated files
.idea/**/contentModel.xml
# Sensitive or high-churn files
.idea/**/dataSources/
.idea/**/dataSources.ids
.idea/**/dataSources.local.xml
.idea/**/sqlDataSources.xml
.idea/**/dynamic.xml
.idea/**/uiDesigner.xml
.idea/**/dbnavigator.xml
# Gradle
.idea/**/gradle.xml
.idea/**/libraries
# Gradle and Maven with auto-import
# When using Gradle or Maven with auto-import, you should exclude module files,
# since they will be recreated, and may cause churn. Uncomment if using
# auto-import.
# .idea/artifacts
# .idea/compiler.xml
# .idea/jarRepositories.xml
# .idea/modules.xml
# .idea/*.iml
# .idea/modules
# *.iml
# *.ipr
# CMake
cmake-build-*/
# Mongo Explorer plugin
.idea/**/mongoSettings.xml
# File-based project format
*.iws
# IntelliJ
out/
# mpeltonen/sbt-idea plugin
.idea_modules/
# JIRA plugin
atlassian-ide-plugin.xml
# Cursive Clojure plugin
.idea/replstate.xml
# Crashlytics plugin (for Android Studio and IntelliJ)
com_crashlytics_export_strings.xml
crashlytics.properties
crashlytics-build.properties
fabric.properties
# Editor-based Rest Client
.idea/httpRequests
# Android studio 3.1+ serialized cache file
.idea/caches/build_file_checksums.ser
+3
View File
@@ -0,0 +1,3 @@
# Default ignored files
/shelf/
/workspace.xml
+6
View File
@@ -0,0 +1,6 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="Encoding">
<file url="file://$PROJECT_DIR$/result.json" charset="GBK" />
</component>
</project>
+6
View File
@@ -0,0 +1,6 @@
<component name="InspectionProjectProfileManager">
<settings>
<option name="USE_PROJECT_PROFILE" value="false" />
<version value="1.0" />
</settings>
</component>
+4
View File
@@ -0,0 +1,4 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="ProjectRootManager" version="2" project-jdk-name="Python 3.9 (wiki_spider)" project-jdk-type="Python SDK" />
</project>
+8
View File
@@ -0,0 +1,8 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="ProjectModuleManager">
<modules>
<module fileurl="file://$PROJECT_DIR$/.idea/wiki_spider.iml" filepath="$PROJECT_DIR$/.idea/wiki_spider.iml" />
</modules>
</component>
</project>
Generated
+6
View File
@@ -0,0 +1,6 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="VcsDirectoryMappings">
<mapping directory="$PROJECT_DIR$" vcs="Git" />
</component>
</project>
+8
View File
@@ -0,0 +1,8 @@
<?xml version="1.0" encoding="UTF-8"?>
<module type="PYTHON_MODULE" version="4">
<component name="NewModuleRootManager">
<content url="file://$MODULE_DIR$" />
<orderEntry type="inheritedJdk" />
<orderEntry type="sourceFolder" forTests="false" />
</component>
</module>
+2
View File
@@ -0,0 +1,2 @@
[PROXY]
url = http://127.0.0.1:10809
+15
View File
@@ -0,0 +1,15 @@
import configparser
import json
from spider import WikiSpider
keywords = ["AK-47突击步枪", "CAESAR自行火炮", "东北大学_(中国)", "北京理工大学"]
if __name__ == '__main__':
config = configparser.ConfigParser()
config.read("config.ini")
s = WikiSpider(config)
with open("result.json", 'w', encoding="utf-8") as f:
result_list = [s.search_by_chinese(k) for k in keywords]
f.write(json.dumps(result_list, ensure_ascii=False))
+68
View File
@@ -0,0 +1,68 @@
import re
import requests
from bs4 import BeautifulSoup
from pandas.io.html import read_html
class WikiSpider:
def __init__(self, config):
self.proxy_config = config["PROXY"]
self.wiki_base_url = "https://zh.wikipedia.org"
self.wiki_url = self.wiki_base_url + "/wiki/"
def search_by_chinese(self, keyword):
url = self.get_wiki_url(keyword)
return self.get_web_content(url)
def get_web_content(self, url) -> dict:
r = requests.get(url, proxies=self.get_proxy())
return self.process_body(r.text)
def process_body(self, body) -> dict:
info = self.get_info(body)
img_urls = self.get_image(body)
info["imgs"] = img_urls
return info
def get_info(self, body) -> dict:
"""
:param body: html content
:return: dictionary contains property retrieved from wikipedia infobox
"""
info_dict = {}
info_boxes = read_html(str(body), index_col=0, attrs={"class": "infobox vcard"})
tmp_dict = info_boxes[0].to_dict()
for k, v in tmp_dict.items():
info_dict = v
break # tmp_dict only has one key
# todo:translate to simplified Chinese
# remove cite chars and \xa0
info_dict = {k: re.sub(r'\[(\d)*\]', "", str(v).replace("\xa0", "")) for k, v in info_dict.items() if
k != v and str(k).lower() != "nan" and str(v).lower() != "nan"}
return info_dict
def get_image(self, body) -> list[str]:
img_urls = []
soup = BeautifulSoup(body, features="lxml")
thumbs = soup.findAll("img", {"class": "thumbimage"})
for thumb in thumbs:
img_page_url = self.wiki_base_url + thumb.parent["href"]
img_page = requests.get(img_page_url, proxies=self.get_proxy())
img_page_body = img_page.text
s = BeautifulSoup(img_page_body, features="lxml")
full_media_div = s.findAll("div", {"class": "fullMedia"})[0]
img_url = "https:" + full_media_div.findAll("a", {"class": "internal"})[0]["href"]
img_urls.append(img_url)
return img_urls
def get_wiki_url(self, keyword) -> str:
return self.wiki_url + keyword
def get_proxy(self) -> dict:
return {
'http': self.proxy_config["url"],
'https': self.proxy_config["url"],
}