import requests import asyncio import aiohttp from requests.exceptions import ConnectionError base_headers = { 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/54.0.2840.71 Safari/537.36', 'Accept-Encoding': 'gzip, deflate, sdch', 'Accept-Language': 'zh-CN,zh;q=0.8' } def get_page(url, options={}): headers = dict(base_headers, **options) print('Getting', url) try: r = requests.get(url, headers=headers) print('Getting result', url, r.status_code) if r.status_code == 200: return r.text except ConnectionError: print('Crawling Failed', url) return None class Downloader(object): """ 一个异步下载器,可以对代理源异步抓取,但是容易被BAN。 """ def __init__(self, urls): self.urls = urls self._htmls = [] async def download_single_page(self, url): async with aiohttp.ClientSession() as session: async with session.get(url) as resp: self._htmls.append(await resp.text()) def download(self): loop = asyncio.get_event_loop() tasks = [self.download_single_page(url) for url in self.urls] loop.run_until_complete(asyncio.wait(tasks)) @property def htmls(self): self.download() return self._htmls