百万级分布式爬虫

news/2024/10/30 19:32:56/
#要爬取的url
url_l = []
#请求头
MY_USER_AGENT = ["Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; AcooBrowser; .NET CLR 1.1.4322; .NET CLR 2.0.50727)","Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 6.0; Acoo Browser; SLCC1; .NET CLR 2.0.50727; Media Center PC 5.0; .NET CLR 3.0.04506)","Mozilla/4.0 (compatible; MSIE 7.0; AOL 9.5; AOLBuild 4337.35; Windows NT 5.1; .NET CLR 1.1.4322; .NET CLR 2.0.50727)","Mozilla/5.0 (Windows; U; MSIE 9.0; Windows NT 9.0; en-US)","Mozilla/5.0 (compatible; MSIE 9.0; Windows NT 6.1; Win64; x64; Trident/5.0; .NET CLR 3.5.30729; .NET CLR 3.0.30729; .NET CLR 2.0.50727; Media Center PC 6.0)","Mozilla/5.0 (compatible; MSIE 8.0; Windows NT 6.0; Trident/4.0; WOW64; Trident/4.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; .NET CLR 1.0.3705; .NET CLR 1.1.4322)","Mozilla/4.0 (compatible; MSIE 7.0b; Windows NT 5.2; .NET CLR 1.1.4322; .NET CLR 2.0.50727; InfoPath.2; .NET CLR 3.0.04506.30)","Mozilla/5.0 (Windows; U; Windows NT 5.1; zh-CN) AppleWebKit/523.15 (KHTML, like Gecko, Safari/419.3) Arora/0.3 (Change: 287 c9dfb30)","Mozilla/5.0 (X11; U; Linux; en-US) AppleWebKit/527+ (KHTML, like Gecko, Safari/419.3) Arora/0.6","Mozilla/5.0 (Windows; U; Windows NT 5.1; en-US; rv:1.8.1.2pre) Gecko/20070215 K-Ninja/2.1.1","Mozilla/5.0 (Windows; U; Windows NT 5.1; zh-CN; rv:1.9) Gecko/20080705 Firefox/3.0 Kapiko/3.0","Mozilla/5.0 (X11; Linux i686; U;) Gecko/20070322 Kazehakase/0.4.5","Mozilla/5.0 (X11; U; Linux i686; en-US; rv:1.9.0.8) Gecko Fedora/1.9.0.8-1.fc10 Kazehakase/0.5.6","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/535.11 (KHTML, like Gecko) Chrome/17.0.963.56 Safari/535.11","Mozilla/5.0 (Macintosh; Intel Mac OS X 10_7_3) AppleWebKit/535.20 (KHTML, like Gecko) Chrome/19.0.1036.7 Safari/535.20","Opera/9.80 (Macintosh; Intel Mac OS X 10.6.8; U; fr) Presto/2.9.168 Version/11.52","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/536.11 (KHTML, like Gecko) Chrome/20.0.1132.11 TaoBrowser/2.0 Safari/536.11","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.71 Safari/537.1 LBBROWSER","Mozilla/5.0 (compatible; MSIE 9.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E; LBBROWSER)","Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E; LBBROWSER)","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/535.11 (KHTML, like Gecko) Chrome/17.0.963.84 Safari/535.11 LBBROWSER","Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E)","Mozilla/5.0 (compatible; MSIE 9.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E; QQBrowser/7.0.3698.400)","Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E)","Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 5.1; Trident/4.0; SV1; QQDownload 732; .NET4.0C; .NET4.0E; 360SE)","Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E)","Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E)","Mozilla/5.0 (Windows NT 5.1) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.89 Safari/537.1","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.89 Safari/537.1","Mozilla/5.0 (iPad; U; CPU OS 4_2_1 like Mac OS X; zh-cn) AppleWebKit/533.17.9 (KHTML, like Gecko) Version/5.0.2 Mobile/8C148 Safari/6533.18.5","Mozilla/5.0 (Windows NT 6.1; Win64; x64; rv:2.0b13pre) Gecko/20110307 Firefox/4.0b13pre","Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:16.0) Gecko/20100101 Firefox/16.0","Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.11 (KHTML, like Gecko) Chrome/23.0.1271.64 Safari/537.11","Mozilla/5.0 (X11; U; Linux x86_64; zh-CN; rv:1.9.2.10) Gecko/20100922 Ubuntu/10.10 (maverick) Firefox/3.6.10","Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.36",
]Android_USER_AGENT = ['Mozilla/5.0 (Linux; Android 7.1.1; MI 6 Build/NMF26X; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/57.0.2987.132 MQQBrowser/6.2 TBS/043807 Mobile Safari/537.36 MicroMessenger/6.6.1.1220(0x26060135) NetType/WIFI Language/zh_CN','Mozilla/5.0 (Linux; Android 7.1.1; OD103 Build/NMF26F; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/53.0.2785.49 Mobile MQQBrowser/6.2 TBS/043632 Safari/537.36 MicroMessenger/6.6.1.1220(0x26060135) NetType/4G Language/zh_CN','Mozilla/5.0 (Linux; Android 6.0.1; SM919 Build/MXB48T; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/53.0.2785.49 Mobile MQQBrowser/6.2 TBS/043632 Safari/537.36 MicroMessenger/6.6.1.1220(0x26060135) NetType/WIFI Language/zh_CN','Mozilla/5.0 (Linux; Android 5.1.1; vivo X6S A Build/LMY47V; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/53.0.2785.49 Mobile MQQBrowser/6.2 TBS/043632 Safari/537.36 MicroMessenger/6.6.1.1220(0x26060135) NetType/WIFI Language/zh_CN','Mozilla/5.0 (Linux; Android 5.1; HUAWEI TAG-AL00 Build/HUAWEITAG-AL00; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/53.0.2785.49 Mobile MQQBrowser/6.2 TBS/043622 Safari/537.36 MicroMessenger/6.6.1.1220(0x26060135) NetType/4G Language/zh_CN','Mozilla/5.0 (Linux; Android 7.0; FRD-AL10 Build/HUAWEIFRD-AL10; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/62.0.3202.84 Mobile Safari/537.36 MicroMessenger/6.7.3.1360(0x26070336) NetType/WIFI Language/zh_CN Process/appbrand0'
]iPhone_USER_AGENT = ['Mozilla/5.0 (iPhone; CPU iPhone OS 9_3_2 like Mac OS X) AppleWebKit/601.1.46 (KHTML, like Gecko) Mobile/13F69 MicroMessenger/6.6.1 NetType/4G Language/zh_CN','Mozilla/5.0 (iPhone; CPU iPhone OS 11_2_2 like Mac OS X) AppleWebKit/604.4.7 (KHTML, like Gecko) Mobile/15C202 MicroMessenger/6.6.1 NetType/4G Language/zh_CN','Mozilla/5.0 (iPhone; CPU iPhone OS 11_1_1 like Mac OS X) AppleWebKit/604.3.5 (KHTML, like Gecko) Mobile/15B150 MicroMessenger/6.6.1 NetType/WIFI Language/zh_CN','Mozilla/5.0 (iphone x Build/MXB48T; wv) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/53.0.2785.49 Mobile MQQBrowser/6.2 TBS/043632 Safari/537.36 MicroMessenger/6.6.1.1220(0x26060135) NetType/WIFI Language/zh_CN'
]
import random
import multiprocessing.managers
from multiprocessing import Queuetask_queue = Queue()
result_queue = Queue()
import requests
from bs4 import BeautifulSoup
import multiprocessingsuccess_num = 0
CONSTANT = 0#获取代理
def getProxyIp():global CONSTANTproxy = []for i in range(1, 15):print(i)header = {'User-Agent': random.choice(random.choice([MY_USER_AGENT, iPhone_USER_AGENT, Android_USER_AGENT]))}r = requests.get('http://www.xicidaili.com/nt/{0}'.format(i), headers=header)html = r.textsoup = BeautifulSoup(html)table = soup.find('table', attrs={'id': 'ip_list'})tr = table.find_all('tr')[1:]# 解析得到代理ip的地址,端口,和类型for item in tr:tds = item.find_all('td')print(tds[1].get_text())temp_dict = {}kind = tds[5].get_text().lower()# exit()if 'http' in kind:temp_dict['http'] = "http://{0}:{1}".format(tds[1].get_text(), tds[2].get_text())if 'https' in kind:temp_dict['https'] = "https://{0}:{1}".format(tds[1].get_text(), tds[2].get_text())proxy.append(temp_dict)return proxydef return_task():return task_queuedef return_result():return result_queueclass QueueManager(multiprocessing.managers.BaseManager):pass#启动主程序派发代理和url和代理
if __name__ == "__main__":# 开启分布式支持multiprocessing.freeze_support()# 注册可以访问队列并得到结果的函数QueueManager.register('get_task', callable=return_task)QueueManager.register('get_result', callable=return_result)#这里设置为192开头的局域ip地址即可局域多台机器使用也可以同一台电脑使用#127开头只能在一台电脑上使用manager = QueueManager(address=('127.0.0.1', 8888), authkey='password'.encode('utf-8'))manager.start()task = manager.get_task()result = manager.get_result()proxies = getProxyIp()index_url = -1while True:for one_proxy in range(len(proxies)):if task.qsize() < len(proxies):print(task.qsize())index_url+=1if index_url>len(url_l)-1:index_url=-1task.put([url_l[index_url], proxies[one_proxy],random.choice(random.choice([MY_USER_AGENT, iPhone_USER_AGENT, Android_USER_AGENT]))])

#从机程序

import requests
from multiprocessing import Pool
import multiprocessing.managers
from multiprocessing import Queue
import timeclass QueueManager(multiprocessing.managers.BaseManager):passdef brash(url_data, proxy_dict, user_agent):header = {'User-Agent': user_agent}# header ={'Mozilla/5.0 (Linux; Android 4.4.2; 2014501 Build/KOT49H) AppleWebKit/537.36 (KHTML, like Gecko) Version/4.0 Chrome/30.0.0.0 Mobile Safari/537.36 Html5Plus/1.0 (Immersed/25.0)'}#提取网页内容程序放在这里try:r = requests.get(url_data, headers=header, proxies=proxy_dict, timeout=10)except Exception as e:pass  # CONSTANT +=1else:# print(url_data, "su")with open("log.txt","a") as f:f.write(url_data+"  successful \n")time.sleep(1)return Nonedef run(res):brash(res[0], res[1], res[2])if __name__ == '__main__':# 开启分布式支持multiprocessing.freeze_support()# 注册可以访问队列并得到结果的函数QueueManager.register('get_task')QueueManager.register('get_result')manager = QueueManager(address=('127.0.0.1', 8888), authkey='password'.encode('utf-8'))manager.connect()task = manager.get_task()result = manager.get_result()while True:start=time.time()pool = Pool(processes=32)results = []leth=task.qsize()for _ in range(leth//5):res = task.get()results.append(pool.apply_async(run,(res,)))for z in range(leth//5):results[z].get()pool.close()pool.join()

http://www.ppmy.cn/news/332843.html

相关文章

免费IP代理池定时维护,封装通用爬虫工具类每次随机更新IP代理池跟UserAgent池,并制作简易流量爬虫...

前言 我们之前的爬虫都是模拟成浏览器后直接爬取&#xff0c;并没有动态设置IP代理以及UserAgent标识&#xff0c;这样很容易被服务器封IP&#xff0c;因此需要设置IP代理&#xff0c;但又不想花钱买&#xff0c;网上有免费IP代理&#xff0c;但大多都数都是不可用&#xff0c;…

【网络协议】IPV4协议介绍

&#x1f4aa;本节内容&#xff1a;IPV4协议介绍、IPV4地址格式、IPV4数据格式及C项目结构体设计 &#x1f60f;【Qt6网络抓包工具项目实战】总导航目录&#xff08;建议收藏书签~~~&#xff09; ✌️ part1 &#x1f60f;【Qt6网络抓包工具项目实战】1.1Qt6.2.2环境搭建(免费…

二维码 | 如何实现一码多用

本人查阅了许多资料&#xff0c;网上大部分的描述都比较模棱两可&#xff0c;我这里就将我的想法分享出来提供大家学习。 在这里实现一码多用的功能指的是 同个二维码在不同端扫出的结果不一样 例如微信扫跳出 微信小程序&#xff0c;支付宝扫跳出 支付宝小程序&#xff0c;内…

mysql存储过程基本语法_MySql基本语法及存储过程学习

今日概要MySql 基本语法学习 MySql 存储过程 WebView 嵌入 JavaScript 代码 WebView 修改 UserAgent X5 访问外链的坑 Kotlin 正则表达式 大纲数据库的连接 数据库里的信息 查看数据库信息 调用存储过程 内容 数据库的连接 连接 MySQL 时需要提供 IP 地址、端口号、用户名、密码…

【Python小竞赛】ARIMA算法预测三日后招商银行收盘价

介绍 本文整理记录了参与的一次小型数据分析竞赛【数据游戏】&#xff0c;竞争目标是预测2019年5月15日A股闭市时招商银行600036的股价。 主要思路是利用ARIMA算法做时间序列预测。 使用的数据是公开的数据集 tushare。 拿到题目和数据之后&#xff0c;首先结合既往经历&am…

金融相关时间序列分析全指南

来源&#xff1a;yv.l1.pnn - kesci.com 原文链接&#xff1a;时间序列分析入门&#xff0c;看这一篇就够了 点击以上链接&#x1f446; 不用配置环境&#xff0c;直接在线运行 数据集下载&#xff1a;DJIA 30股票时间序列 本文包含时间序列的数据结构、可视化、统计学理论及模…

常用USER_AGENT

当前的UA复制 PC端 浏览器User-agentsafari 5.1 – MACMozilla/5.0 (Macintosh; U; Intel Mac OS X 10_6_8; en-us) AppleWebKit/534.50 (KHTML, like Gecko) Version/5.1 Safari/534.50复制safari 5.1 – WindowsMozilla/5.0 (Windows; U; Windows NT 6.1; en-us) AppleWebK…

JAVA程序设计:完全二叉树插入器(LeetCode:919)

完全二叉树是每一层&#xff08;除最后一层外&#xff09;都是完全填充&#xff08;即&#xff0c;结点数达到最大&#xff09;的&#xff0c;并且所有的结点都尽可能地集中在左侧。 设计一个用完全二叉树初始化的数据结构 CBTInserter&#xff0c;它支持以下几种操作&#xf…