commit 5c3d92a353c46268049e938cb1bd839a9bd8cee1 Author: c-my Date: Thu Apr 15 23:50:25 2021 +0800 feat: get property and image links diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..ec8b925 --- /dev/null +++ b/.gitignore @@ -0,0 +1,213 @@ +result.json + +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +share/python-wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.nox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +*.py,cover +.hypothesis/ +.pytest_cache/ +cover/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 +db.sqlite3-journal + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +.pybuilder/ +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# IPython +profile_default/ +ipython_config.py + +# pyenv +# For a library or package, you might want to ignore these files since the code is +# intended to run in multiple environments; otherwise, check them in: +# .python-version + +# pipenv +# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. +# However, in case of collaboration, if having platform-specific dependencies or dependencies +# having no cross-platform support, pipenv may install dependencies that don't work, or not +# install all needed dependencies. +#Pipfile.lock + +# PEP 582; used by e.g. github.com/David-OConnor/pyflow +__pypackages__/ + +# Celery stuff +celerybeat-schedule +celerybeat.pid + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ +.dmypy.json +dmypy.json + +# Pyre type checker +.pyre/ + +# pytype static type analyzer +.pytype/ + +# Cython debug symbols +cython_debug/ + +# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider +# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839 + +# User-specific stuff +.idea/**/workspace.xml +.idea/**/tasks.xml +.idea/**/usage.statistics.xml +.idea/**/dictionaries +.idea/**/shelf + +# Generated files +.idea/**/contentModel.xml + +# Sensitive or high-churn files +.idea/**/dataSources/ +.idea/**/dataSources.ids +.idea/**/dataSources.local.xml +.idea/**/sqlDataSources.xml +.idea/**/dynamic.xml +.idea/**/uiDesigner.xml +.idea/**/dbnavigator.xml + +# Gradle +.idea/**/gradle.xml +.idea/**/libraries + +# Gradle and Maven with auto-import +# When using Gradle or Maven with auto-import, you should exclude module files, +# since they will be recreated, and may cause churn. Uncomment if using +# auto-import. +# .idea/artifacts +# .idea/compiler.xml +# .idea/jarRepositories.xml +# .idea/modules.xml +# .idea/*.iml +# .idea/modules +# *.iml +# *.ipr + +# CMake +cmake-build-*/ + +# Mongo Explorer plugin +.idea/**/mongoSettings.xml + +# File-based project format +*.iws + +# IntelliJ +out/ + +# mpeltonen/sbt-idea plugin +.idea_modules/ + +# JIRA plugin +atlassian-ide-plugin.xml + +# Cursive Clojure plugin +.idea/replstate.xml + +# Crashlytics plugin (for Android Studio and IntelliJ) +com_crashlytics_export_strings.xml +crashlytics.properties +crashlytics-build.properties +fabric.properties + +# Editor-based Rest Client +.idea/httpRequests + +# Android studio 3.1+ serialized cache file +.idea/caches/build_file_checksums.ser + diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 0000000..26d3352 --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,3 @@ +# Default ignored files +/shelf/ +/workspace.xml diff --git a/.idea/encodings.xml b/.idea/encodings.xml new file mode 100644 index 0000000..7e1986f --- /dev/null +++ b/.idea/encodings.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 0000000..105ce2d --- /dev/null +++ b/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/.idea/misc.xml b/.idea/misc.xml new file mode 100644 index 0000000..ebff38b --- /dev/null +++ b/.idea/misc.xml @@ -0,0 +1,4 @@ + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 0000000..ddb1c03 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 0000000..94a25f7 --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/.idea/wiki_spider.iml b/.idea/wiki_spider.iml new file mode 100644 index 0000000..d0876a7 --- /dev/null +++ b/.idea/wiki_spider.iml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/config.ini b/config.ini new file mode 100644 index 0000000..a671870 --- /dev/null +++ b/config.ini @@ -0,0 +1,2 @@ +[PROXY] +url = http://127.0.0.1:10809 \ No newline at end of file diff --git a/main.py b/main.py new file mode 100644 index 0000000..f41e8d3 --- /dev/null +++ b/main.py @@ -0,0 +1,15 @@ +import configparser +import json + +from spider import WikiSpider + +keywords = ["AK-47突击步枪", "CAESAR自行火炮", "东北大学_(中国)", "北京理工大学"] + +if __name__ == '__main__': + config = configparser.ConfigParser() + config.read("config.ini") + s = WikiSpider(config) + + with open("result.json", 'w', encoding="utf-8") as f: + result_list = [s.search_by_chinese(k) for k in keywords] + f.write(json.dumps(result_list, ensure_ascii=False)) diff --git a/spider.py b/spider.py new file mode 100644 index 0000000..eb2d69c --- /dev/null +++ b/spider.py @@ -0,0 +1,68 @@ +import re + +import requests +from bs4 import BeautifulSoup +from pandas.io.html import read_html + + +class WikiSpider: + def __init__(self, config): + self.proxy_config = config["PROXY"] + self.wiki_base_url = "https://zh.wikipedia.org" + self.wiki_url = self.wiki_base_url + "/wiki/" + + def search_by_chinese(self, keyword): + url = self.get_wiki_url(keyword) + return self.get_web_content(url) + + def get_web_content(self, url) -> dict: + r = requests.get(url, proxies=self.get_proxy()) + + return self.process_body(r.text) + + def process_body(self, body) -> dict: + info = self.get_info(body) + img_urls = self.get_image(body) + info["imgs"] = img_urls + return info + + def get_info(self, body) -> dict: + """ + :param body: html content + :return: dictionary contains property retrieved from wikipedia infobox + """ + info_dict = {} + + info_boxes = read_html(str(body), index_col=0, attrs={"class": "infobox vcard"}) + tmp_dict = info_boxes[0].to_dict() + for k, v in tmp_dict.items(): + info_dict = v + break # tmp_dict only has one key + # todo:translate to simplified Chinese + # remove cite chars and \xa0 + info_dict = {k: re.sub(r'\[(\d)*\]', "", str(v).replace("\xa0", "")) for k, v in info_dict.items() if + k != v and str(k).lower() != "nan" and str(v).lower() != "nan"} + return info_dict + + def get_image(self, body) -> list[str]: + img_urls = [] + soup = BeautifulSoup(body, features="lxml") + thumbs = soup.findAll("img", {"class": "thumbimage"}) + for thumb in thumbs: + img_page_url = self.wiki_base_url + thumb.parent["href"] + img_page = requests.get(img_page_url, proxies=self.get_proxy()) + img_page_body = img_page.text + s = BeautifulSoup(img_page_body, features="lxml") + full_media_div = s.findAll("div", {"class": "fullMedia"})[0] + img_url = "https:" + full_media_div.findAll("a", {"class": "internal"})[0]["href"] + img_urls.append(img_url) + return img_urls + + def get_wiki_url(self, keyword) -> str: + return self.wiki_url + keyword + + def get_proxy(self) -> dict: + return { + 'http': self.proxy_config["url"], + 'https': self.proxy_config["url"], + }